diff --git a/Cargo.lock b/Cargo.lock index d8ffb6e91..d74bcdd5e 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3000,7 +3000,6 @@ dependencies = [ "ast-grep-core", "base64", "clap", - "dashmap", "fontdue", "globset", "grep-matcher", @@ -3022,6 +3021,7 @@ dependencies = [ "pi-iso", "pi-shell", "pi-uutils-ctx", + "pi-walker", "png", "portable-pty", "rayon", @@ -3057,6 +3057,7 @@ dependencies = [ "libc", "os_pipe", "pi-uutils-ctx", + "pi-walker", "pi_uu_grep", "regex", "serde", @@ -3088,6 +3089,18 @@ dependencies = [ "libc", ] +[[package]] +name = "pi-walker" +version = "16.2.9" +dependencies = [ + "dashmap", + "globset", + "ignore", + "libc", + "rayon", + "windows-sys 0.61.2", +] + [[package]] name = "pi_uu_grep" version = "0.8.0" @@ -3099,6 +3112,7 @@ dependencies = [ "grep-searcher", "ignore", "pi-uutils-ctx", + "pi-walker", ] [[package]] @@ -4873,10 +4887,10 @@ dependencies = [ "nix 0.29.0", "onig", "pi-uutils-ctx", + "pi-walker", "regex", "tempfile", "uucore 0.0.30", - "walkdir", ] [[package]] diff --git a/Cargo.toml b/Cargo.toml index 5097e0d56..f4bfbf11c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -171,6 +171,7 @@ too_many_lines = "allow" # Arbitrary limits don't account for necessary comp pi-ast = { path = "crates/pi-ast" } pi-iso = { path = "crates/pi-iso" } pi-shell = { path = "crates/pi-shell" } +pi-walker = { path = "crates/pi-walker" } brush-core = { path = "crates/vendor/brush-core" } brush-builtins = { path = "crates/vendor/brush-builtins" } diff --git a/crates/pi-natives/Cargo.toml b/crates/pi-natives/Cargo.toml index b3dfdf9a9..0928d3dc3 100644 --- a/crates/pi-natives/Cargo.toml +++ b/crates/pi-natives/Cargo.toml @@ -18,7 +18,6 @@ arboard.workspace = true ast-grep-core.workspace = true base64.workspace = true clap.workspace = true -dashmap.workspace = true globset.workspace = true fontdue.workspace = true grep-matcher.workspace = true @@ -37,6 +36,7 @@ phf.workspace = true pi-ast.workspace = true pi-iso.workspace = true pi-shell.workspace = true +pi-walker.workspace = true pi-uutils-ctx = { path = "../pi-uutils-ctx" } portable-pty.workspace = true png.workspace = true diff --git a/crates/pi-natives/src/ast.rs b/crates/pi-natives/src/ast.rs index c362a0f30..3a3ba054d 100644 --- a/crates/pi-natives/src/ast.rs +++ b/crates/pi-natives/src/ast.rs @@ -13,7 +13,7 @@ use pi_ast::{ ops::{self as shared_ops}, }; -use crate::{fs_cache, glob_util, task}; +use crate::{glob_util, iofs, task}; const DEFAULT_FIND_LIMIT: u32 = 50; @@ -341,33 +341,6 @@ fn normalize_search_path(path: Option) -> Result { Ok(std::fs::canonicalize(&absolute).unwrap_or(absolute)) } -fn collect_from_entries( - root: &Path, - entries: &[fs_cache::GlobMatch], - glob_set: Option<&globset::GlobSet>, - mentions_node_modules: bool, - ct: &task::CancelToken, -) -> Result> { - let mut files = Vec::new(); - for entry in entries { - ct.heartbeat()?; - if entry.file_type != fs_cache::FileType::File { - continue; - } - let relative = entry.path.replace('\\', "/"); - if fs_cache::should_skip_path(Path::new(&relative), mentions_node_modules) { - continue; - } - if let Some(glob_set) = glob_set - && !glob_set.is_match(&relative) - { - continue; - } - files.push(FileCandidate { absolute_path: root.join(&relative), display_path: relative }); - } - Ok(files) -} - fn collect_candidates( path: Option, glob: Option<&str>, @@ -393,44 +366,37 @@ fn collect_candidates( ))); } - let glob_set = glob_util::try_compile_glob(glob, false)?; let mentions_node_modules = glob.is_some_and(|value| value.contains("node_modules")); - let skip_node_modules = !mentions_node_modules; - let scan = fs_cache::get_or_scan( - &search_path, - fs_cache::ScanOptions { - include_hidden: true, - use_gitignore: true, - skip_node_modules, - follow_links: false, - detail: fs_cache::ScanDetail::Minimal, - }, - ct, - )?; - let mut files = collect_from_entries( - &search_path, - &scan.entries, - glob_set.as_ref(), - mentions_node_modules, - ct, - )?; - - if files.is_empty() && scan.cache_age_ms >= fs_cache::empty_recheck_ms() { - let fresh = fs_cache::force_rescan( - &search_path, - fs_cache::ScanOptions { - include_hidden: true, - use_gitignore: true, - skip_node_modules, - follow_links: false, - detail: fs_cache::ScanDetail::Minimal, - }, - true, - ct, - )?; - files = - collect_from_entries(&search_path, &fresh, glob_set.as_ref(), mentions_node_modules, ct)?; + let mut filter = + pi_walker::WalkFilter::files_only().node_modules_unless_mentioned(mentions_node_modules); + if let Some(glob) = glob.map(str::trim).filter(|value| !value.is_empty()) { + let pattern = glob_util::build_glob_pattern(glob, false); + let compiled = pi_walker::CompiledWalkGlob::new([pattern]) + .map_err(|err| Error::from_reason(format!("Invalid glob pattern: {err}")))?; + filter = filter.glob(compiled); } + let request = pi_walker::WalkRequest::new(&search_path) + .hidden(true) + .gitignore(true) + .skip_git(true) + .follow_links(pi_walker::FollowLinks::Never) + .detail(pi_walker::WalkDetail::Minimal) + .order(pi_walker::WalkOrder::Path) + .emit_root(false) + .depth(1, usize::MAX) + .directory_errors(pi_walker::DirectoryErrorMode::SkipSkippable) + .cache(true) + .empty_recheck(pi_walker::EmptyRecheck::Configured) + .filter(filter); + let mut files: Vec<_> = request + .collect_files_with_heartbeat(|| ct.heartbeat()) + .map_err(iofs::map_walker_error)? + .into_iter() + .map(|entry| FileCandidate { + absolute_path: entry.absolute_path(&search_path), + display_path: entry.path, + }) + .collect(); files.sort_by(|a, b| a.display_path.cmp(&b.display_path)); Ok(files) diff --git a/crates/pi-natives/src/fast_walk.rs b/crates/pi-natives/src/fast_walk.rs deleted file mode 100644 index 074e9b5ab..000000000 --- a/crates/pi-natives/src/fast_walk.rs +++ /dev/null @@ -1,899 +0,0 @@ -//! Platform directory traversal fast path for no-ignore filesystem scans. -//! -//! This module keeps ignore semantics out of the native path on purpose. The -//! `ignore` crate owns `.ignore`/`.gitignore`/global-exclude behavior; the fast -//! scanner is enabled only when callers explicitly disable those sources. - -use std::{ - ffi::{OsStr, OsString}, - io, - path::{Path, PathBuf}, -}; - -use napi::bindgen_prelude::*; - -use crate::{ - fs_cache::{FileType, GlobMatch, ScanDetail, ScanOptions}, - task, -}; - -const HEARTBEAT_INTERVAL: usize = 128; - -/// Result of attempting a platform-native directory scan. -pub enum FastEntryScan { - /// The current platform/configuration cannot use the native scanner safely. - Unsupported, - /// Native scanner produced entries with the same contract as `fs_cache` - /// scans. - Entries(Vec), -} - -/// Visitor decision for streaming native traversal. -pub enum FastWalkControl { - /// Continue traversing remaining entries. - Continue, - /// Stop traversal immediately after the current entry. - Stop, -} - -/// Status returned by streaming native traversal. -pub enum FastWalkStatus { - /// The current platform/configuration cannot use the native scanner safely. - Unsupported, - /// Traversal visited every reachable entry. - Complete, - /// The visitor stopped traversal early. - Stopped, -} - -#[derive(Clone)] -struct RawDirEntry { - name: OsString, - file_type: FileType, - mtime: Option, - size: Option, -} - -/// Scans entries using platform syscalls when the scan contract is equivalent. -pub fn collect_entries( - root: &Path, - options: ScanOptions, - ct: &task::CancelToken, -) -> Result { - let mut matches = Vec::new(); - let status = walk_entries(root, options, ct, |_path, entry| { - matches.push(entry); - Ok(FastWalkControl::Continue) - })?; - if matches!(status, FastWalkStatus::Unsupported) { - return Ok(FastEntryScan::Unsupported); - } - matches.sort_unstable_by(|a, b| a.path.cmp(&b.path)); - Ok(FastEntryScan::Entries(matches)) -} - -/// Streams entries using platform syscalls when the scan contract is -/// equivalent. -pub fn walk_entries( - root: &Path, - options: ScanOptions, - ct: &task::CancelToken, - visitor: F, -) -> Result -where - F: FnMut(&Path, GlobMatch) -> Result, -{ - if !can_use_fast_scan(options) { - return Ok(FastWalkStatus::Unsupported); - } - ct.heartbeat()?; - let mut visitor = visitor; - let mut visited = 0usize; - match walk_dir(root, "", options, ct, &mut visited, &mut visitor) { - Ok(true) => Ok(FastWalkStatus::Stopped), - Ok(false) => Ok(FastWalkStatus::Complete), - Err(FastScanError::Unsupported) => Ok(FastWalkStatus::Unsupported), - Err(FastScanError::Cancelled(err)) => Err(err), - Err(FastScanError::InvalidData { path, message }) => Err(Error::from_reason(format!( - "Native directory scan failed for {}: {message}", - path.display() - ))), - } -} - -const fn can_use_fast_scan(options: ScanOptions) -> bool { - platform::SUPPORTED && !options.use_gitignore && !options.follow_links -} - -enum FastScanError { - Unsupported, - Cancelled(Error), - InvalidData { path: PathBuf, message: String }, -} - -impl From for FastScanError { - fn from(err: Error) -> Self { - Self::Cancelled(err) - } -} - -fn walk_dir( - dir: &Path, - relative_dir: &str, - options: ScanOptions, - ct: &task::CancelToken, - visited: &mut usize, - visitor: &mut F, -) -> std::result::Result -where - F: FnMut(&Path, GlobMatch) -> Result, -{ - let mut raw_entries = match platform::read_dir_entries(dir, options.detail) { - Ok(entries) => entries, - Err(err) if err.kind() == io::ErrorKind::Unsupported => { - return Err(FastScanError::Unsupported); - }, - Err(err) if is_skippable_directory_error(&err) => return Ok(false), - Err(err) => { - return Err(FastScanError::InvalidData { - path: dir.to_path_buf(), - message: err.to_string(), - }); - }, - }; - raw_entries.sort_unstable_by(|a, b| a.name.cmp(&b.name)); - - for entry in raw_entries { - if *visited == 0 || *visited >= HEARTBEAT_INTERVAL { - *visited = 0; - ct.heartbeat()?; - } - *visited += 1; - - let name = entry_name(&entry.name); - if name.is_empty() || name == "." || name == ".." { - continue; - } - if !options.include_hidden && is_hidden_name(&name) { - continue; - } - if name == ".git" || (options.skip_node_modules && name == "node_modules") { - continue; - } - - let relative = join_relative_path(relative_dir, &name); - let is_dir = entry.file_type == FileType::Dir; - let absolute = dir.join(&entry.name); - let matched = GlobMatch { - path: relative.clone(), - file_type: entry.file_type, - mtime: entry.mtime, - size: entry.size, - }; - if matches!(visitor(&absolute, matched)?, FastWalkControl::Stop) { - return Ok(true); - } - if is_dir && walk_dir(&absolute, &relative, options, ct, visited, visitor)? { - return Ok(true); - } - } - - Ok(false) -} - -fn is_skippable_directory_error(err: &io::Error) -> bool { - matches!( - err.kind(), - io::ErrorKind::NotFound | io::ErrorKind::NotADirectory | io::ErrorKind::PermissionDenied - ) -} - -fn entry_name(name: &OsStr) -> String { - name.to_string_lossy().into_owned() -} - -fn is_hidden_name(name: &str) -> bool { - name.as_bytes().first() == Some(&b'.') -} - -fn join_relative_path(parent: &str, name: &str) -> String { - if parent.is_empty() { - name.to_string() - } else { - let mut path = String::with_capacity(parent.len() + 1 + name.len()); - path.push_str(parent); - path.push('/'); - path.push_str(name); - path - } -} - -fn mtime_millis(seconds: i64, nanos: i64) -> Option { - if seconds < 0 { - return None; - } - Some((seconds as f64).mul_add(1000.0, nanos.max(0) as f64 / 1_000_000.0)) -} - -#[cfg(target_os = "macos")] -mod platform { - use std::{ - ffi::CString, - io, - mem::size_of, - os::{ - fd::RawFd, - unix::ffi::{OsStrExt, OsStringExt}, - }, - path::Path, - }; - - use super::{FileType, RawDirEntry, ScanDetail, mtime_millis}; - - pub(super) const SUPPORTED: bool = true; - - const BUFFER_SIZE: usize = 256 * 1024; - const VREG: u32 = 1; - const VDIR: u32 = 2; - const VLNK: u32 = 5; - - struct FdGuard(RawFd); - - impl Drop for FdGuard { - fn drop(&mut self) { - // SAFETY: `FdGuard` owns this file descriptor and closes it exactly once. - unsafe { libc::close(self.0) }; - } - } - - pub(super) fn read_dir_entries(path: &Path, detail: ScanDetail) -> io::Result> { - let fd = open_dir(path)?; - let mut attrs = libc::attrlist { - bitmapcount: libc::ATTR_BIT_MAP_COUNT, - reserved: 0, - commonattr: libc::ATTR_CMN_NAME | libc::ATTR_CMN_OBJTYPE, - volattr: 0, - dirattr: 0, - fileattr: 0, - forkattr: 0, - }; - if detail == ScanDetail::Full { - attrs.commonattr |= libc::ATTR_CMN_MODTIME; - attrs.fileattr |= libc::ATTR_FILE_DATALENGTH; - } - - let mut buffer = vec![0u8; BUFFER_SIZE]; - let mut entries = Vec::new(); - loop { - // SAFETY: `fd` is an open directory descriptor, `attrs` points to a valid - // attrlist for the duration of the call, and `buffer` is writable. - let count = unsafe { - libc::getattrlistbulk( - fd.0, - std::ptr::addr_of_mut!(attrs).cast(), - buffer.as_mut_ptr().cast(), - buffer.len(), - libc::FSOPT_NOFOLLOW as u64, - ) - }; - if count == 0 { - break; - } - if count < 0 { - let err = io::Error::last_os_error(); - if err.kind() == io::ErrorKind::Interrupted { - continue; - } - return Err(map_unsupported(err)); - } - - let mut offset = 0usize; - for _ in 0..count { - if offset + size_of::() > buffer.len() { - return Err(invalid_data("truncated getattrlistbulk record length")); - } - let record_len = u32::from_ne_bytes( - buffer[offset..offset + size_of::()] - .try_into() - .expect("slice length checked"), - ) as usize; - if record_len < size_of::() || offset + record_len > buffer.len() { - return Err(invalid_data("invalid getattrlistbulk record length")); - } - let record = &buffer[offset..offset + record_len]; - if let Some(entry) = parse_record(record, detail)? { - entries.push(entry); - } - offset += record_len; - } - } - Ok(entries) - } - - fn open_dir(path: &Path) -> io::Result { - let path = CString::new(path.as_os_str().as_bytes()) - .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "path contains NUL"))?; - // SAFETY: `path` is a NUL-terminated C string; flags open the directory for - // metadata traversal only and do not transfer ownership of the string. - let fd = - unsafe { libc::open(path.as_ptr(), libc::O_RDONLY | libc::O_DIRECTORY | libc::O_CLOEXEC) }; - if fd < 0 { - Err(io::Error::last_os_error()) - } else { - Ok(FdGuard(fd)) - } - } - - fn parse_record(record: &[u8], detail: ScanDetail) -> io::Result> { - let mut cursor = size_of::(); - let name_ref_start = cursor; - let name_ref = read_value::(record, &mut cursor)?; - let obj_type = read_value::(record, &mut cursor)?; - let (mtime, data_length) = if detail == ScanDetail::Full { - let modified = read_value::(record, &mut cursor)?; - let data_length = read_value::(record, &mut cursor)?; - (mtime_millis(modified.tv_sec as i64, modified.tv_nsec as i64), Some(data_length)) - } else { - (None, None) - }; - - let name_start = checked_attr_offset(name_ref_start, name_ref.attr_dataoffset)?; - let name_len = name_ref.attr_length as usize; - if name_len == 0 || name_start + name_len > record.len() { - return Err(invalid_data("invalid getattrlistbulk name reference")); - } - let name_bytes = trim_nul(&record[name_start..name_start + name_len]); - if name_bytes.is_empty() { - return Ok(None); - } - - let Some(file_type) = file_type_from_vtype(obj_type) else { - return Ok(None); - }; - let size = if file_type == FileType::File { - data_length.map(|value| value as f64) - } else { - None - }; - Ok(Some(RawDirEntry { - name: std::ffi::OsString::from_vec(name_bytes.to_vec()), - file_type, - mtime, - size, - })) - } - - fn read_value(record: &[u8], cursor: &mut usize) -> io::Result { - let end = cursor.saturating_add(size_of::()); - if end > record.len() { - return Err(invalid_data("truncated getattrlistbulk attribute")); - } - let ptr = record[*cursor..end].as_ptr(); - *cursor = end; - // SAFETY: Bounds were checked above; `getattrlistbulk` records are byte - // packed, so unaligned reads are required and do not outlive `record`. - Ok(unsafe { std::ptr::read_unaligned(ptr.cast::()) }) - } - - fn checked_attr_offset(base: usize, offset: i32) -> io::Result { - if offset < 0 { - return Err(invalid_data("negative getattrlistbulk attribute offset")); - } - base - .checked_add(offset as usize) - .ok_or_else(|| invalid_data("overflowing getattrlistbulk attribute offset")) - } - - fn trim_nul(bytes: &[u8]) -> &[u8] { - let end = bytes.iter().position(|b| *b == 0).unwrap_or(bytes.len()); - &bytes[..end] - } - - const fn file_type_from_vtype(value: u32) -> Option { - match value { - VREG => Some(FileType::File), - VDIR => Some(FileType::Dir), - VLNK => Some(FileType::Symlink), - _ => None, - } - } - - fn map_unsupported(err: io::Error) -> io::Error { - if matches!(err.raw_os_error(), Some(libc::ENOTSUP | libc::EINVAL)) { - io::Error::new(io::ErrorKind::Unsupported, err) - } else { - err - } - } - - fn invalid_data(message: &'static str) -> io::Error { - io::Error::new(io::ErrorKind::InvalidData, message) - } -} - -#[cfg(target_os = "linux")] -mod platform { - use std::{ - ffi::{CString, OsString}, - io, - mem::{size_of, zeroed}, - os::unix::ffi::{OsStrExt, OsStringExt}, - path::Path, - }; - - use super::{FileType, RawDirEntry, ScanDetail, mtime_millis}; - - pub(super) const SUPPORTED: bool = true; - - const BUFFER_SIZE: usize = 256 * 1024; - const LINUX_DIRENT64_NAME_OFFSET: usize = 19; - const STATX_TYPE: u32 = 0x0001; - const STATX_SIZE: u32 = 0x0200; - const STATX_MTIME: u32 = 0x0040; - const STATX_BASIC_STATS: u32 = 0x07ff; - - #[repr(C)] - #[derive(Clone, Copy)] - struct StatxTimestamp { - tv_sec: i64, - tv_nsec: u32, - __reserved: i32, - } - - #[repr(C)] - #[derive(Clone, Copy)] - struct Statx { - stx_mask: u32, - stx_blksize: u32, - stx_attributes: u64, - stx_nlink: u32, - stx_uid: u32, - stx_gid: u32, - stx_mode: u16, - __spare0: [u16; 1], - stx_ino: u64, - stx_size: u64, - stx_blocks: u64, - stx_attributes_mask: u64, - stx_atime: StatxTimestamp, - stx_btime: StatxTimestamp, - stx_ctime: StatxTimestamp, - stx_mtime: StatxTimestamp, - stx_rdev_major: u32, - stx_rdev_minor: u32, - stx_dev_major: u32, - stx_dev_minor: u32, - stx_mnt_id: u64, - stx_dio_mem_align: u32, - stx_dio_offset_align: u32, - __spare3: [u64; 12], - } - - struct FdGuard(libc::c_int); - - impl Drop for FdGuard { - fn drop(&mut self) { - // SAFETY: `FdGuard` owns this file descriptor and closes it exactly once. - unsafe { libc::close(self.0) }; - } - } - - struct EntryStat { - file_type: FileType, - mtime: Option, - size: Option, - } - - pub(super) fn read_dir_entries(path: &Path, detail: ScanDetail) -> io::Result> { - let fd = open_dir(path)?; - let mut buffer = vec![0u8; BUFFER_SIZE]; - let mut entries = Vec::new(); - loop { - // SAFETY: `fd` is an open directory descriptor and `buffer` is writable. - let read = unsafe { - libc::syscall( - libc::SYS_getdents64, - fd.0, - buffer.as_mut_ptr().cast::(), - buffer.len(), - ) - }; - if read == 0 { - break; - } - if read < 0 { - let err = io::Error::last_os_error(); - if err.kind() == io::ErrorKind::Interrupted { - continue; - } - return Err(err); - } - - let mut offset = 0usize; - let read_len = read as usize; - while offset < read_len { - if offset + LINUX_DIRENT64_NAME_OFFSET > read_len { - return Err(invalid_data("truncated getdents64 record")); - } - let reclen = read_u16(&buffer[offset + 16..read_len])? as usize; - if reclen < LINUX_DIRENT64_NAME_OFFSET || offset + reclen > read_len { - return Err(invalid_data("invalid getdents64 record length")); - } - let d_type = buffer[offset + 18]; - let name_bytes = - trim_nul(&buffer[offset + LINUX_DIRENT64_NAME_OFFSET..offset + reclen]); - offset += reclen; - if name_bytes.is_empty() { - continue; - } - - let dtype_file_type = file_type_from_dtype(d_type); - let stat = if detail == ScanDetail::Full || dtype_file_type.is_none() { - match stat_entry(fd.0, name_bytes, detail) { - Ok(Some(stat)) => Some(stat), - Ok(None) => continue, - Err(err) if is_skippable_entry_error(&err) => continue, - Err(err) => return Err(err), - } - } else { - None - }; - let file_type = stat - .as_ref() - .map_or(dtype_file_type, |stat| Some(stat.file_type)); - let Some(file_type) = file_type else { - continue; - }; - entries.push(RawDirEntry { - name: OsString::from_vec(name_bytes.to_vec()), - file_type, - mtime: stat.as_ref().and_then(|stat| stat.mtime), - size: stat.as_ref().and_then(|stat| stat.size), - }); - } - } - Ok(entries) - } - - fn open_dir(path: &Path) -> io::Result { - let path = CString::new(path.as_os_str().as_bytes()) - .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "path contains NUL"))?; - // SAFETY: `path` is a NUL-terminated C string; flags request a directory - // descriptor used only with getdents/statx and do not retain the pointer. - let fd = - unsafe { libc::open(path.as_ptr(), libc::O_RDONLY | libc::O_DIRECTORY | libc::O_CLOEXEC) }; - if fd < 0 { - Err(io::Error::last_os_error()) - } else { - Ok(FdGuard(fd)) - } - } - - fn stat_entry( - dirfd: libc::c_int, - name: &[u8], - detail: ScanDetail, - ) -> io::Result> { - let name = CString::new(name) - .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "entry name contains NUL"))?; - match statx_entry(dirfd, &name, detail) { - Ok(value) => Ok(value), - Err(err) if matches!(err.raw_os_error(), Some(libc::ENOSYS | libc::EINVAL)) => { - fstatat_entry(dirfd, &name, detail) - }, - Err(err) => Err(err), - } - } - - fn statx_entry( - dirfd: libc::c_int, - name: &CString, - detail: ScanDetail, - ) -> io::Result> { - // SAFETY: `Statx` is a plain-old-data buffer whose all-zero value is a - // valid initialization before the kernel fills it. - let mut statx = unsafe { zeroed::() }; - let mask = if detail == ScanDetail::Full { - STATX_BASIC_STATS - } else { - STATX_TYPE - }; - // SAFETY: `name` is NUL-terminated, `statx` is writable, and `dirfd` is an - // open directory descriptor for an AT_* relative metadata query. - let rc = unsafe { - libc::syscall( - libc::SYS_statx, - dirfd, - name.as_ptr(), - libc::AT_SYMLINK_NOFOLLOW | libc::AT_NO_AUTOMOUNT, - mask, - std::ptr::addr_of_mut!(statx), - ) - }; - if rc != 0 { - return Err(io::Error::last_os_error()); - } - let Some(file_type) = file_type_from_mode(statx.stx_mode as libc::mode_t) else { - return Ok(None); - }; - let mtime = if detail == ScanDetail::Full && statx.stx_mask & STATX_MTIME != 0 { - mtime_millis(statx.stx_mtime.tv_sec, i64::from(statx.stx_mtime.tv_nsec)) - } else { - None - }; - let size = if detail == ScanDetail::Full - && file_type == FileType::File - && statx.stx_mask & STATX_SIZE != 0 - { - Some(statx.stx_size as f64) - } else { - None - }; - Ok(Some(EntryStat { file_type, mtime, size })) - } - - fn fstatat_entry( - dirfd: libc::c_int, - name: &CString, - detail: ScanDetail, - ) -> io::Result> { - // SAFETY: `libc::stat` is a POD buffer filled by fstatat. - let mut stat = unsafe { zeroed::() }; - // SAFETY: `name` is NUL-terminated, `stat` is writable, and `dirfd` is an - // open directory descriptor for an AT_* relative metadata query. - let rc = unsafe { - libc::fstatat( - dirfd, - name.as_ptr(), - std::ptr::addr_of_mut!(stat), - libc::AT_SYMLINK_NOFOLLOW, - ) - }; - if rc != 0 { - return Err(io::Error::last_os_error()); - } - let Some(file_type) = file_type_from_mode(stat.st_mode) else { - return Ok(None); - }; - let mtime = if detail == ScanDetail::Full { - mtime_millis(stat.st_mtime, stat.st_mtime_nsec as i64) - } else { - None - }; - let size = if detail == ScanDetail::Full && file_type == FileType::File { - Some(stat.st_size as f64) - } else { - None - }; - Ok(Some(EntryStat { file_type, mtime, size })) - } - - fn read_u16(bytes: &[u8]) -> io::Result { - if bytes.len() < size_of::() { - return Err(invalid_data("truncated u16")); - } - Ok(u16::from_ne_bytes( - bytes[..size_of::()] - .try_into() - .expect("slice length checked"), - )) - } - - fn trim_nul(bytes: &[u8]) -> &[u8] { - let end = bytes.iter().position(|b| *b == 0).unwrap_or(bytes.len()); - &bytes[..end] - } - - fn file_type_from_dtype(value: u8) -> Option { - match value { - libc::DT_REG => Some(FileType::File), - libc::DT_DIR => Some(FileType::Dir), - libc::DT_LNK => Some(FileType::Symlink), - _ => None, - } - } - - fn file_type_from_mode(mode: libc::mode_t) -> Option { - match mode & libc::S_IFMT { - libc::S_IFREG => Some(FileType::File), - libc::S_IFDIR => Some(FileType::Dir), - libc::S_IFLNK => Some(FileType::Symlink), - _ => None, - } - } - - fn is_skippable_entry_error(err: &io::Error) -> bool { - matches!( - err.kind(), - io::ErrorKind::NotFound | io::ErrorKind::PermissionDenied | io::ErrorKind::NotADirectory - ) - } - - fn invalid_data(message: &'static str) -> io::Error { - io::Error::new(io::ErrorKind::InvalidData, message) - } -} - -#[cfg(target_os = "windows")] -mod platform { - use std::{ - ffi::OsString, - io, - os::windows::ffi::{OsStrExt, OsStringExt}, - path::Path, - }; - - use windows_sys::{ - Wdk::Storage::FileSystem::{ - FILE_ID_FULL_DIR_INFORMATION, FileIdFullDirectoryInformation, NtQueryDirectoryFile, - }, - Win32::{ - Foundation::{CloseHandle, HANDLE, INVALID_HANDLE_VALUE, STATUS_NO_MORE_FILES}, - Storage::FileSystem::{ - CreateFileW, FILE_ATTRIBUTE_DIRECTORY, FILE_ATTRIBUTE_REPARSE_POINT, - FILE_FLAG_BACKUP_SEMANTICS, FILE_FLAG_OPEN_REPARSE_POINT, FILE_LIST_DIRECTORY, - FILE_SHARE_DELETE, FILE_SHARE_READ, FILE_SHARE_WRITE, OPEN_EXISTING, - }, - System::IO::IO_STATUS_BLOCK, - }, - }; - - use super::{FileType, RawDirEntry, ScanDetail, mtime_millis}; - - pub(super) const SUPPORTED: bool = true; - - const BUFFER_SIZE: usize = 256 * 1024; - const WINDOWS_TICK: i64 = 10_000_000; - const UNIX_EPOCH_AS_FILETIME: i64 = 116_444_736_000_000_000; - - struct HandleGuard(HANDLE); - - impl Drop for HandleGuard { - fn drop(&mut self) { - // SAFETY: `HandleGuard` owns this handle and closes it exactly once. - unsafe { CloseHandle(self.0) }; - } - } - - pub(super) fn read_dir_entries(path: &Path, detail: ScanDetail) -> io::Result> { - let handle = open_dir(path)?; - let mut buffer = vec![0u8; BUFFER_SIZE]; - let mut restart = true; - let mut entries = Vec::new(); - - loop { - let mut iosb = IO_STATUS_BLOCK::default(); - // SAFETY: `handle` is an open directory handle, `buffer` is writable, and - // the query class matches the record parser below. - let status = unsafe { - NtQueryDirectoryFile( - handle.0, - std::ptr::null_mut(), - None, - std::ptr::null(), - std::ptr::addr_of_mut!(iosb), - buffer.as_mut_ptr().cast(), - buffer.len() as u32, - FileIdFullDirectoryInformation, - false, - std::ptr::null(), - restart, - ) - }; - restart = false; - if status == STATUS_NO_MORE_FILES { - break; - } - if status < 0 { - return Err(io::Error::from_raw_os_error(status)); - } - - let mut offset = 0usize; - loop { - if offset + std::mem::size_of::() > buffer.len() { - return Err(invalid_data("truncated NtQueryDirectoryFile record")); - } - let info = unsafe { - // SAFETY: Bounds were checked above; records are byte-packed in the - // buffer and may not be aligned for Rust references. - std::ptr::read_unaligned( - buffer[offset..] - .as_ptr() - .cast::(), - ) - }; - let name_offset = offset + std::mem::offset_of!(FILE_ID_FULL_DIR_INFORMATION, FileName); - let name_len = info.FileNameLength as usize; - if name_len % 2 != 0 || name_offset + name_len > buffer.len() { - return Err(invalid_data("invalid NtQueryDirectoryFile name length")); - } - let name_units: Vec = buffer[name_offset..name_offset + name_len] - .chunks_exact(2) - .map(|chunk| u16::from_ne_bytes([chunk[0], chunk[1]])) - .collect(); - let name = OsString::from_wide(&name_units); - if let Some(file_type) = file_type_from_attributes(info.FileAttributes) { - let size = if detail == ScanDetail::Full && file_type == FileType::File { - Some(info.EndOfFile.max(0) as f64) - } else { - None - }; - let mtime = if detail == ScanDetail::Full { - mtime_from_filetime(info.LastWriteTime) - } else { - None - }; - entries.push(RawDirEntry { name, file_type, mtime, size }); - } - if info.NextEntryOffset == 0 { - break; - } - offset = offset.saturating_add(info.NextEntryOffset as usize); - } - } - Ok(entries) - } - - fn open_dir(path: &Path) -> io::Result { - let mut path: Vec = path.as_os_str().encode_wide().collect(); - path.push(0); - // SAFETY: `path` is NUL-terminated; the returned handle is owned by - // `HandleGuard` on success. - let handle = unsafe { - CreateFileW( - path.as_ptr(), - FILE_LIST_DIRECTORY, - FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, - std::ptr::null(), - OPEN_EXISTING, - FILE_FLAG_BACKUP_SEMANTICS | FILE_FLAG_OPEN_REPARSE_POINT, - std::ptr::null_mut(), - ) - }; - if handle == INVALID_HANDLE_VALUE { - Err(io::Error::last_os_error()) - } else { - Ok(HandleGuard(handle)) - } - } - - fn file_type_from_attributes(attributes: u32) -> Option { - if attributes & FILE_ATTRIBUTE_REPARSE_POINT != 0 { - Some(FileType::Symlink) - } else if attributes & FILE_ATTRIBUTE_DIRECTORY != 0 { - Some(FileType::Dir) - } else { - Some(FileType::File) - } - } - - fn mtime_from_filetime(filetime: i64) -> Option { - let ticks = filetime.checked_sub(UNIX_EPOCH_AS_FILETIME)?; - let seconds = ticks / WINDOWS_TICK; - let nanos = (ticks % WINDOWS_TICK) * 100; - mtime_millis(seconds, nanos) - } - - fn invalid_data(message: &'static str) -> io::Error { - io::Error::new(io::ErrorKind::InvalidData, message) - } -} - -#[cfg(not(any(target_os = "macos", target_os = "linux", target_os = "windows")))] -mod platform { - use std::{io, path::Path}; - - use super::{RawDirEntry, ScanDetail}; - - pub(super) const SUPPORTED: bool = false; - - pub(super) fn read_dir_entries( - _path: &Path, - _detail: ScanDetail, - ) -> io::Result> { - Err(io::Error::new( - io::ErrorKind::Unsupported, - "native directory scan unsupported on this platform", - )) - } -} diff --git a/crates/pi-natives/src/fd.rs b/crates/pi-natives/src/fd.rs index 832dc5a21..a9f58abae 100644 --- a/crates/pi-natives/src/fd.rs +++ b/crates/pi-natives/src/fd.rs @@ -1,14 +1,14 @@ //! Fuzzy file path discovery for autocomplete and @-mention resolution. //! //! Searches for files and directories whose paths match a query string via -//! subsequence scoring. Uses the shared [`fs_cache`] for directory scanning. +//! subsequence scoring. Uses `pi-walker` for directory traversal and caching. use std::path::Path; use napi::bindgen_prelude::*; use napi_derive::napi; -use crate::{fs_cache, task}; +use crate::{iofs, task}; /// Options for fuzzy file path search. #[napi(object)] @@ -21,7 +21,7 @@ pub struct FuzzyFindOptions<'env> { pub hidden: Option, /// Respect .gitignore (default: true). pub gitignore: Option, - /// Enable shared filesystem scan cache (default: false). + /// Enable walker scan caching (default: false). pub cache: Option, /// Maximum number of matches to return (default: 100). pub max_results: Option, @@ -161,7 +161,7 @@ struct FuzzyFindConfig { } fn score_entries( - entries: &[fs_cache::GlobMatch], + entries: &[iofs::GlobMatch], query_lower: &str, normalized_query: &str, query_chars: &[char], @@ -170,11 +170,11 @@ fn score_entries( let mut scored = Vec::with_capacity(entries.len().min(256)); for entry in entries { ct.heartbeat()?; - if entry.file_type == fs_cache::FileType::Symlink { + if entry.file_type == iofs::FileType::Symlink { continue; } - let is_directory = entry.file_type == fs_cache::FileType::Dir; + let is_directory = entry.file_type == iofs::FileType::Dir; let score = score_fuzzy_path(&entry.path, is_directory, query_lower, normalized_query, query_chars); if score == 0 { @@ -191,7 +191,7 @@ fn score_entries( } fn fuzzy_find_sync(config: FuzzyFindConfig, ct: task::CancelToken) -> Result { - let root = fs_cache::resolve_search_path(&config.path)?; + let root = pi_walker::resolve_search_path(&config.path).map_err(iofs::map_walker_error)?; let include_hidden = config.hidden.unwrap_or(false); let respect_gitignore = config.gitignore.unwrap_or(true); let max_results = config.max_results.unwrap_or(100) as usize; @@ -206,30 +206,27 @@ fn fuzzy_find_sync(config: FuzzyFindConfig, ct: task::CancelToken) -> Result= fs_cache::empty_recheck_ms() - { - let fresh = fs_cache::force_rescan(&root, scan_options, true, &ct)?; - scored = score_entries(&fresh, &query_lower, &normalized_query, &query_chars, &ct)?; - } - scored - } else { - let fresh = fs_cache::force_rescan(&root, scan_options, false, &ct)?; - score_entries(&fresh, &query_lower, &normalized_query, &query_chars, &ct)? - }; + let outcome = pi_walker::WalkRequest::new(root) + .hidden(include_hidden) + .gitignore(respect_gitignore) + .skip_git(true) + .skip_node_modules(true) + .follow_links(pi_walker::FollowLinks::Always) + .detail(pi_walker::WalkDetail::Minimal) + .order(pi_walker::WalkOrder::Path) + .emit_root(false) + .depth(1, usize::MAX) + .directory_errors(pi_walker::DirectoryErrorMode::SkipSkippable) + .cache(config.cache.unwrap_or(false)) + .empty_recheck(pi_walker::EmptyRecheck::Configured) + .collect_with_heartbeat(|| ct.heartbeat()) + .map_err(iofs::map_walker_error)?; + let entries: Vec = outcome + .entries + .into_iter() + .map(iofs::GlobMatch::from) + .collect(); + let mut scored = score_entries(&entries, &query_lower, &normalized_query, &query_chars, &ct)?; scored.sort_by(|a, b| b.score.cmp(&a.score).then_with(|| a.path.cmp(&b.path))); let total_matches = crate::utils::clamp_u32(scored.len() as u64); @@ -246,3 +243,92 @@ pub fn fuzzy_find(options: FuzzyFindOptions<'_>) -> task::Promise Self { + static COUNTER: AtomicU64 = AtomicU64::new(0); + let nanos = SystemTime::now() + .duration_since(UNIX_EPOCH) + .expect("system time is after UNIX_EPOCH") + .as_nanos(); + let seq = COUNTER.fetch_add(1, Ordering::Relaxed); + let pid = std::process::id(); + let path = std::env::temp_dir().join(format!("pi-fd-test-{pid}-{nanos}-{seq}")); + fs::create_dir_all(&path).expect("create temp test directory"); + Self(path) + } + + fn path(&self) -> &Path { + &self.0 + } + } + + #[cfg(unix)] + impl Drop for TempDirGuard { + fn drop(&mut self) { + let _ = fs::remove_dir_all(&self.0); + } + } + + #[cfg(unix)] + #[test] + fn fuzzy_find_without_cache_follows_symlinked_directories() { + let root = TempDirGuard::new(); + let real_dir = root.path().join("zz-real-dir"); + let link_dir_name = "aa-linked-dir"; + let link_dir = root.path().join(link_dir_name); + let file_name = "follow-links-fuzzy-needle.txt"; + + fs::create_dir_all(&real_dir).expect("create real directory"); + fs::write(real_dir.join(file_name), "needle\n").expect("write symlink target file"); + unix_fs::symlink(&real_dir, &link_dir).expect("create directory symlink"); + + let result = fuzzy_find_sync( + FuzzyFindConfig { + query: file_name.to_string(), + path: root.path().to_string_lossy().into_owned(), + hidden: Some(true), + gitignore: Some(false), + max_results: Some(4), + cache: Some(false), + }, + task::CancelToken::default(), + ) + .expect("fuzzy find succeeds"); + + assert!(!result.matches.is_empty(), "expected at least one fuzzy find match"); + let expected_path = format!("{link_dir_name}/{file_name}"); + assert!( + result + .matches + .iter() + .any(|entry| entry.path == expected_path), + "expected fuzzy find to include symlink traversal path {expected_path:?}, got {:?}", + result + .matches + .iter() + .map(|entry| entry.path.as_str()) + .collect::>() + ); + } +} diff --git a/crates/pi-natives/src/fs_cache.rs b/crates/pi-natives/src/fs_cache.rs deleted file mode 100644 index d7ee6fa50..000000000 --- a/crates/pi-natives/src/fs_cache.rs +++ /dev/null @@ -1,973 +0,0 @@ -//! Shared filesystem scan cache for discovery tools (glob, fd). -//! -//! Provides a TTL-based cache of scanned directory entries, with: -//! - Global policy (no per-call TTL tuning) -//! - Explicit invalidation for agent file mutations -//! - Empty-result fast recheck to avoid stale negatives -//! -//! # Policy Configuration (environment overrides) -//! - `FS_SCAN_CACHE_TTL_MS` – default `1000` -//! - `FS_SCAN_EMPTY_RECHECK_MS` – default `200` -//! - `FS_SCAN_CACHE_MAX_ENTRIES` – default `16` - -use std::{ - borrow::Cow, - path::{Path, PathBuf}, - sync::{Arc, LazyLock, Mutex}, - time::{Duration, Instant}, -}; - -use dashmap::DashMap; -use ignore::{ParallelVisitor, ParallelVisitorBuilder, WalkBuilder, WalkState}; -use napi::bindgen_prelude::*; -use napi_derive::napi; - -use crate::{env_uint, fast_walk, task}; - -// ═══════════════════════════════════════════════════════════════════════════ -// Public types (re-exported by glob for backward compatibility) -// ═══════════════════════════════════════════════════════════════════════════ - -/// Resolved filesystem entry kind for glob filters and match metadata. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -#[napi] -pub enum FileType { - /// Regular file. - File = 1, - /// Directory. - Dir = 2, - /// Symbolic link. - Symlink = 3, -} - -/// A single filesystem entry from a directory scan. -#[derive(Clone)] -#[napi(object)] -pub struct GlobMatch { - /// Relative path from the search root, using forward slashes. - pub path: String, - /// Resolved filesystem type for the match. - pub file_type: FileType, - /// Modification time in milliseconds since Unix epoch (from - /// `symlink_metadata`). - pub mtime: Option, - /// File size in bytes for regular files. - pub size: Option, -} - -// ═══════════════════════════════════════════════════════════════════════════ -// Cache policy -// ═══════════════════════════════════════════════════════════════════════════ - -env_uint! { - // Configured cache TTL in milliseconds. - static CACHE_TTL_MS: u64 = "FS_SCAN_CACHE_TTL_MS" or 1_000 => [0, u64::MAX]; - // Configured empty-result recheck threshold in milliseconds. - static EMPTY_RECHECK_MS: u64 = "FS_SCAN_EMPTY_RECHECK_MS" or 200 => [0, u64::MAX]; - // Configured maximum number of cache entries. - static MAX_CACHE_ENTRIES: usize = "FS_SCAN_CACHE_MAX_ENTRIES" or 16 => [0, usize::MAX]; -} - -env_uint! { - // Worker count for parallel filesystem walks. 0 lets ignore choose. - static GREP_WORKERS: usize = "PI_GREP_WORKERS" or 4 => [0, usize::MAX]; -} - -pub fn cache_ttl_ms() -> u64 { - *CACHE_TTL_MS -} - -pub fn empty_recheck_ms() -> u64 { - *EMPTY_RECHECK_MS -} - -pub fn max_cache_entries() -> usize { - *MAX_CACHE_ENTRIES -} - -pub fn grep_workers() -> usize { - *GREP_WORKERS -} - -// ═══════════════════════════════════════════════════════════════════════════ -// Cache internals -// ═══════════════════════════════════════════════════════════════════════════ - -#[derive(Clone, Debug, Eq, Hash, PartialEq)] -struct CacheKey { - root: PathBuf, - include_hidden: bool, - use_gitignore: bool, - skip_node_modules: bool, - detail: ScanDetail, -} - -#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] -pub enum ScanDetail { - Minimal, - Full, -} - -#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] -pub struct ScanOptions { - pub include_hidden: bool, - pub use_gitignore: bool, - pub skip_node_modules: bool, - pub follow_links: bool, - pub detail: ScanDetail, -} - -#[derive(Clone)] -struct CacheEntry { - created_at: Instant, - entries: Vec, -} - -static FS_CACHE: LazyLock> = LazyLock::new(DashMap::new); - -/// Result of a cache-aware scan, including the age of the cached data. -pub struct ScanResult { - /// Scanned filesystem entries. - pub entries: Vec, - /// How old the cached data is in milliseconds (0 = freshly scanned). - pub cache_age_ms: u64, -} - -fn evict_oldest() { - if FS_CACHE.len() > *MAX_CACHE_ENTRIES - && let Some(oldest_key) = FS_CACHE - .iter() - .min_by_key(|entry| entry.value().created_at) - .map(|entry| entry.key().clone()) - { - FS_CACHE.remove(&oldest_key); - } -} - -// ═══════════════════════════════════════════════════════════════════════════ -// Path utilities -// ═══════════════════════════════════════════════════════════════════════════ - -/// Resolve a search path string to a canonical `PathBuf` (must be a directory). -pub fn resolve_search_path(path: &str) -> Result { - let candidate = PathBuf::from(path); - let root = if candidate.is_absolute() { - candidate - } else { - let cwd = std::env::current_dir() - .map_err(|err| Error::from_reason(format!("Failed to resolve cwd: {err}")))?; - cwd.join(candidate) - }; - let metadata = std::fs::metadata(&root) - .map_err(|err| Error::from_reason(format!("Path not found: {err}")))?; - if !metadata.is_dir() { - return Err(Error::from_reason("Search path must be a directory".to_string())); - } - Ok(std::fs::canonicalize(&root).unwrap_or(root)) -} - -/// Normalize a filesystem path to a forward-slash relative string. -pub fn normalize_relative_path<'a>(root: &Path, path: &'a Path) -> Cow<'a, str> { - let relative = path.strip_prefix(root).unwrap_or(path); - if cfg!(windows) { - let relative = relative.to_string_lossy(); - if relative.contains('\\') { - Cow::Owned(relative.replace('\\', "/")) - } else { - relative - } - } else { - relative.to_string_lossy() - } -} - -pub fn contains_component(path: &Path, target: &str) -> bool { - path.components().any(|component| { - component - .as_os_str() - .to_str() - .is_some_and(|value| value == target) - }) -} - -pub fn should_skip_path(path: &Path, mentions_node_modules: bool) -> bool { - // Always skip VCS internals; they are noise for user-facing discovery. - if contains_component(path, ".git") { - return true; - } - if !mentions_node_modules && contains_component(path, "node_modules") { - // Skip node_modules by default unless explicitly requested/pattern-matched. - return true; - } - false -} - -fn file_type_from_std(file_type: std::fs::FileType) -> Option { - if file_type.is_symlink() { - Some(FileType::Symlink) - } else if file_type.is_dir() { - Some(FileType::Dir) - } else if file_type.is_file() { - Some(FileType::File) - } else { - None - } -} - -fn mtime_ms(metadata: &std::fs::Metadata) -> Option { - metadata - .modified() - .ok() - .and_then(|t| t.duration_since(std::time::UNIX_EPOCH).ok()) - .map(|d| d.as_millis() as f64) -} - -pub fn classify_file_type(path: &Path) -> Option<(FileType, Option, Option)> { - let metadata = std::fs::symlink_metadata(path).ok()?; - let file_type = file_type_from_std(metadata.file_type())?; - let size = if file_type == FileType::File { - Some(metadata.len()) - } else { - None - }; - Some((file_type, mtime_ms(&metadata), size)) -} - -// ═══════════════════════════════════════════════════════════════════════════ -// Walker + collection -// ═══════════════════════════════════════════════════════════════════════════ - -/// Builds a deterministic filesystem walker configured for visibility and -/// ignore rules. -/// -/// When `skip_node_modules` is true, `node_modules` directories are pruned at -/// traversal time (not just filtered post-scan). `.git` is always skipped. -#[allow(clippy::fn_params_excessive_bools, reason = "matches WalkBuilder option fields")] -pub fn build_walker( - root: &Path, - include_hidden: bool, - use_gitignore: bool, - skip_node_modules: bool, - follow_links: bool, -) -> WalkBuilder { - let mut builder = WalkBuilder::new(root); - builder - .hidden(!include_hidden) - .follow_links(follow_links) - .sort_by_file_path(|a, b| a.cmp(b)) - // filter_entry controls whether to yield an entry AND whether to descend - // into a directory. Returning false for a directory skips the entire subtree. - .filter_entry(move |entry| { - let name = entry.file_name().to_str().unwrap_or_default(); - // Always skip .git - if name == ".git" { - return false; - } - // Skip node_modules when skip_node_modules is true - if skip_node_modules && name == "node_modules" { - return false; - } - true - }); - - if use_gitignore { - // Honor repository and global ignore files for repo-like behavior. - builder - .git_ignore(true) - .git_exclude(true) - .git_global(true) - .ignore(true) - .parents(true); - } else { - // Disable all ignore sources for exhaustive filesystem traversal. - builder - .git_ignore(false) - .git_exclude(false) - .git_global(false) - .ignore(false) - .parents(false); - } - - builder -} - -struct EntryVisitor<'a> { - root: &'a Path, - detail: ScanDetail, - ct: &'a task::CancelToken, - entries: Vec, - shared_entries: Arc>>>, - error: Arc>>, - visited: usize, -} - -impl Drop for EntryVisitor<'_> { - fn drop(&mut self) { - if self.entries.is_empty() { - return; - } - let entries = std::mem::take(&mut self.entries); - self - .shared_entries - .lock() - .expect("entry collection lock poisoned") - .push(entries); - } -} - -impl ParallelVisitor for EntryVisitor<'_> { - fn visit(&mut self, entry: std::result::Result) -> WalkState { - if self.visited == 0 || self.visited >= 128 { - self.visited = 0; - if let Err(err) = self.ct.heartbeat() { - *self.error.lock().expect("error lock poisoned") = Some(err.to_string()); - return WalkState::Quit; - } - } - self.visited += 1; - - let Ok(entry) = entry else { - return WalkState::Continue; - }; - if let Some(entry) = collect_entry(self.root, &entry, self.detail) { - self.entries.push(entry); - } - WalkState::Continue - } -} - -struct EntryVisitorBuilder<'a> { - root: &'a Path, - detail: ScanDetail, - ct: &'a task::CancelToken, - shared_entries: Arc>>>, - error: Arc>>, -} - -impl<'a> ParallelVisitorBuilder<'a> for EntryVisitorBuilder<'a> { - fn build(&mut self) -> Box { - Box::new(EntryVisitor { - root: self.root, - detail: self.detail, - ct: self.ct, - entries: Vec::new(), - shared_entries: Arc::clone(&self.shared_entries), - error: Arc::clone(&self.error), - visited: 0, - }) - } -} - -/// Attempts the platform-native no-ignore scanner and reports unsupported -/// configs. -pub(crate) fn try_fast_collect_entries( - root: &Path, - options: ScanOptions, - ct: &task::CancelToken, -) -> Result>> { - match fast_walk::collect_entries(root, options, ct)? { - fast_walk::FastEntryScan::Entries(entries) => Ok(Some(entries)), - fast_walk::FastEntryScan::Unsupported => Ok(None), - } -} - -/// Scans filesystem entries and records normalized relative paths with file -/// metadata. -fn collect_entries( - root: &Path, - options: ScanOptions, - ct: &task::CancelToken, -) -> Result> { - if let Some(entries) = try_fast_collect_entries(root, options, ct)? { - return Ok(entries); - } - - let mut builder = build_walker( - root, - options.include_hidden, - options.use_gitignore, - options.skip_node_modules, - options.follow_links, - ); - let workers = grep_workers(); - if workers > 0 { - builder.threads(workers); - } - let shared_entries = Arc::new(Mutex::new(Vec::new())); - let error = Arc::new(Mutex::new(None)); - let mut visitor_builder = EntryVisitorBuilder { - root, - detail: options.detail, - ct, - shared_entries: Arc::clone(&shared_entries), - error: Arc::clone(&error), - }; - ct.heartbeat()?; - builder.build_parallel().visit(&mut visitor_builder); - - let walk_error = error.lock().expect("error lock poisoned").take(); - if let Some(error) = walk_error { - return Err(Error::from_reason(error)); - } - - let mut entries: Vec = shared_entries - .lock() - .expect("entry collection lock poisoned") - .drain(..) - .flatten() - .collect(); - entries.sort_unstable_by(|a, b| a.path.cmp(&b.path)); - Ok(entries) -} - -pub(crate) fn collect_entry( - root: &Path, - entry: &ignore::DirEntry, - detail: ScanDetail, -) -> Option { - let path = entry.path(); - let relative = normalize_relative_path(root, path); - if relative.is_empty() { - // Ignore the synthetic root entry ("" relative path). - return None; - } - - let (file_type, mtime, size) = match detail { - ScanDetail::Minimal => { - let file_type = file_type_from_std(entry.file_type()?)?; - (file_type, None, None) - }, - ScanDetail::Full => { - let metadata = entry - .metadata() - .or_else(|_| std::fs::symlink_metadata(path)) - .ok()?; - let file_type = file_type_from_std(metadata.file_type())?; - let size = if file_type == FileType::File { - Some(metadata.len() as f64) - } else { - None - }; - (file_type, mtime_ms(&metadata), size) - }, - }; - - Some(GlobMatch { path: relative.into_owned(), file_type, mtime, size }) -} - -// ═══════════════════════════════════════════════════════════════════════════ -// Cache API -// ═══════════════════════════════════════════════════════════════════════════ - -/// Returns scanned entries using the global TTL cache policy. -/// -/// The returned [`ScanResult::cache_age_ms`] lets callers implement -/// empty-result fast recheck: if a query produces zero matches and the cache is -/// older than [`empty_recheck_ms()`], call [`force_rescan`] before returning -/// empty. -pub fn get_or_scan( - root: &Path, - options: ScanOptions, - ct: &task::CancelToken, -) -> Result { - let ttl = *CACHE_TTL_MS; - if ttl == 0 { - // Caching disabled – always scan fresh. - let entries = collect_entries(root, options, ct)?; - return Ok(ScanResult { entries, cache_age_ms: 0 }); - } - - let key = CacheKey { - root: root.to_path_buf(), - include_hidden: options.include_hidden, - use_gitignore: options.use_gitignore, - skip_node_modules: options.skip_node_modules, - detail: options.detail, - }; - - let now = Instant::now(); - if let Some(entry) = FS_CACHE.get(&key) { - let age = now.duration_since(entry.created_at); - if age < Duration::from_millis(ttl) { - return Ok(ScanResult { - entries: entry.entries.clone(), - cache_age_ms: age.as_millis() as u64, - }); - } - drop(entry); - FS_CACHE.remove(&key); - } - - let entries = collect_entries(root, options, ct)?; - FS_CACHE.insert(key, CacheEntry { created_at: now, entries: entries.clone() }); - evict_oldest(); - Ok(ScanResult { entries, cache_age_ms: 0 }) -} - -/// Force a fresh scan, replacing any existing cache entry. -/// -/// Use when a cached query produced zero matches and the cache was old enough -/// to warrant a recheck. When `store` is false, the fresh scan result is -/// returned without repopulating the cache. -pub fn force_rescan( - root: &Path, - options: ScanOptions, - store: bool, - ct: &task::CancelToken, -) -> Result> { - let key = CacheKey { - root: root.to_path_buf(), - include_hidden: options.include_hidden, - use_gitignore: options.use_gitignore, - skip_node_modules: options.skip_node_modules, - detail: options.detail, - }; - FS_CACHE.remove(&key); - - let entries = collect_entries(root, options, ct)?; - if store { - let now = Instant::now(); - FS_CACHE.insert(key, CacheEntry { created_at: now, entries: entries.clone() }); - evict_oldest(); - } - Ok(entries) -} - -// ═══════════════════════════════════════════════════════════════════════════ -// Invalidation -// ═══════════════════════════════════════════════════════════════════════════ - -/// Invalidate cache entries whose root contains `target`. -/// -/// Removes any cache entry whose root is a prefix of (or equal to) `target`, -/// because a file mutation under that root makes the scan stale. -pub fn invalidate_path(target: &Path) { - let keys_to_remove: Vec = FS_CACHE - .iter() - .filter(|entry| target.starts_with(&entry.key().root)) - .map(|entry| entry.key().clone()) - .collect(); - for key in keys_to_remove { - FS_CACHE.remove(&key); - } -} - -/// Clear the entire scan cache. -pub fn invalidate_all() { - FS_CACHE.clear(); -} - -/// Invalidate the filesystem scan cache. -/// -/// When called with a path, removes entries for roots containing that path. -/// When called without a path, clears the entire cache. -/// -/// Intended to be called after agent file mutations (write, edit, rename, -/// delete). -#[napi] -pub fn invalidate_fs_scan_cache(path: Option) { - match path { - Some(p) => { - let candidate = PathBuf::from(&p); - let absolute = if candidate.is_absolute() { - candidate - } else if let Ok(cwd) = std::env::current_dir() { - cwd.join(candidate) - } else { - PathBuf::from(&p) - }; - let target = std::fs::canonicalize(&absolute) - .or_else(|_| { - absolute - .parent() - .and_then(|parent| std::fs::canonicalize(parent).ok()) - .and_then(|parent| absolute.file_name().map(|name| parent.join(name))) - .ok_or_else(|| std::io::Error::from(std::io::ErrorKind::NotFound)) - }) - .unwrap_or(absolute); - invalidate_path(&target); - }, - None => invalidate_all(), - } -} - -#[cfg(test)] -mod tests { - #[cfg(unix)] - use std::{ffi::CString, os::unix::ffi::OsStrExt}; - use std::{ - fs, - path::{Path, PathBuf}, - sync::atomic::{AtomicU64, Ordering}, - time::{Duration, SystemTime, UNIX_EPOCH}, - }; - - #[cfg(unix)] - use super::classify_file_type; - - static TEMP_COUNTER: AtomicU64 = AtomicU64::new(0); - - struct TempDirGuard(PathBuf); - - impl TempDirGuard { - fn new() -> Self { - let timestamp = SystemTime::now() - .duration_since(UNIX_EPOCH) - .expect("system time is after UNIX_EPOCH") - .as_nanos(); - let counter = TEMP_COUNTER.fetch_add(1, Ordering::Relaxed); - let path = std::env::temp_dir().join(format!("pi-fs-cache-test-{timestamp}-{counter}")); - fs::create_dir_all(&path).expect("create temp test directory"); - Self(path) - } - - fn path(&self) -> &Path { - &self.0 - } - } - - impl Drop for TempDirGuard { - fn drop(&mut self) { - let _ = fs::remove_dir_all(&self.0); - } - } - - #[cfg(unix)] - fn make_fifo(path: &Path) { - let fifo_path = - CString::new(path.as_os_str().as_bytes()).expect("fifo path has no NUL bytes"); - // SAFETY: `fifo_path` is a valid CString (NUL-terminated, no interior NULs), - // so `as_ptr()` yields a valid C string pointer. `0o600` is a valid mode. - // The CString is alive for the duration of the call. - let rc = unsafe { libc::mkfifo(fifo_path.as_ptr(), 0o600) }; - assert_eq!(rc, 0, "create fifo: {}", std::io::Error::last_os_error()); - } - - fn full_scan_options(include_hidden: bool, use_gitignore: bool) -> super::ScanOptions { - super::ScanOptions { - include_hidden, - use_gitignore, - skip_node_modules: true, - follow_links: false, - detail: super::ScanDetail::Full, - } - } - - fn assert_file_entry(entries: &[super::GlobMatch], path: &str, size: f64) { - let entry = entries - .iter() - .find(|entry| entry.path == path) - .unwrap_or_else(|| panic!("expected file entry {path}, got {}", entry_paths(entries))); - assert_eq!(entry.file_type, super::FileType::File); - assert!(entry.mtime.is_some(), "full scan should include mtime for {path}"); - assert_eq!(entry.size, Some(size)); - } - - fn assert_dir_entry(entries: &[super::GlobMatch], path: &str) { - let entry = entries - .iter() - .find(|entry| entry.path == path) - .unwrap_or_else(|| panic!("expected dir entry {path}, got {}", entry_paths(entries))); - assert_eq!(entry.file_type, super::FileType::Dir); - assert!(entry.mtime.is_some(), "full scan should include mtime for {path}"); - assert_eq!(entry.size, None); - } - - fn entry_paths(entries: &[super::GlobMatch]) -> String { - let paths: Vec<&str> = entries.iter().map(|entry| entry.path.as_str()).collect(); - format!("{paths:?}") - } - - #[cfg(unix)] - #[test] - fn classify_file_type_skips_fifo() { - let root = TempDirGuard::new(); - let fifo = root.path().join("skip-me.fifo"); - make_fifo(&fifo); - - assert_eq!(classify_file_type(&fifo), None); - } - - #[test] - fn build_walker_skips_git_and_node_modules() { - let root = TempDirGuard::new(); - fs::create_dir_all(root.path().join(".git/objects")).unwrap(); - fs::write(root.path().join(".git/objects/a.txt"), "git obj").unwrap(); - fs::create_dir_all(root.path().join("node_modules/pkg")).unwrap(); - fs::write(root.path().join("node_modules/pkg/index.js"), "nm").unwrap(); - fs::write(root.path().join("real.txt"), "ok").unwrap(); - - // skip_node_modules: true -> should only see real.txt - let walker = super::build_walker(root.path(), true, false, true, false); - let paths: Vec = walker - .build() - .filter_map(|e| e.ok()) - .filter(|e| e.path() != root.path()) - .map(|e| { - e.path() - .strip_prefix(root.path()) - .unwrap() - .to_string_lossy() - .into_owned() - }) - .collect(); - assert!( - !paths - .iter() - .any(|p| p.contains("node_modules") || p.contains(".git")), - "expected no .git or node_modules entries, got: {paths:?}" - ); - assert!(paths.iter().any(|p| p == "real.txt"), "expected real.txt, got: {paths:?}"); - - // skip_node_modules: false -> should see node_modules but not .git - let walker = super::build_walker(root.path(), true, false, false, false); - let paths: Vec = walker - .build() - .filter_map(|e| e.ok()) - .filter(|e| e.path() != root.path()) - .map(|e| { - e.path() - .strip_prefix(root.path()) - .unwrap() - .to_string_lossy() - .into_owned() - }) - .collect(); - assert!( - !paths.iter().any(|p| p.contains(".git")), - "expected no .git entries, got: {paths:?}" - ); - assert!( - paths.iter().any(|p| p.contains("node_modules")), - "expected node_modules entries, got: {paths:?}" - ); - } - - #[test] - fn collect_entries_skips_node_modules() { - let root = TempDirGuard::new(); - fs::create_dir_all(root.path().join("node_modules/pkg")).unwrap(); - fs::write(root.path().join("node_modules/pkg/index.js"), "nm").unwrap(); - fs::write(root.path().join("real.txt"), "ok").unwrap(); - - let ct = crate::task::CancelToken::default(); - let entries = super::collect_entries( - root.path(), - super::ScanOptions { - include_hidden: true, - use_gitignore: false, - skip_node_modules: true, - follow_links: false, - detail: super::ScanDetail::Full, - }, - &ct, - ) - .unwrap(); - let paths: Vec<&str> = entries.iter().map(|e| e.path.as_str()).collect(); - assert!( - !paths.iter().any(|p| p.contains("node_modules")), - "expected no node_modules entries, got: {paths:?}" - ); - assert!(paths.iter().any(|p| p == &"real.txt"), "expected real.txt, got: {paths:?}"); - } - - #[test] - fn traversal_gitignore_excludes_files_for_collect_entries_and_force_rescan() { - let root = TempDirGuard::new(); - fs::create_dir_all(root.path().join(".git")).unwrap(); - fs::write(root.path().join(".gitignore"), "ignored.txt\n").unwrap(); - fs::write(root.path().join("ignored.txt"), "ignored").unwrap(); - fs::write(root.path().join("kept.txt"), "keep").unwrap(); - - let ct = crate::task::CancelToken::default(); - let options = full_scan_options(true, true); - - let collected = super::collect_entries(root.path(), options, &ct).unwrap(); - assert!( - !collected.iter().any(|entry| entry.path == "ignored.txt"), - "collect_entries returned gitignored file: {}", - entry_paths(&collected) - ); - assert_file_entry(&collected, "kept.txt", 4.0); - - let rescanned = super::force_rescan(root.path(), options, false, &ct).unwrap(); - assert!( - !rescanned.iter().any(|entry| entry.path == "ignored.txt"), - "force_rescan returned gitignored file: {}", - entry_paths(&rescanned) - ); - assert_file_entry(&rescanned, "kept.txt", 4.0); - } - - #[test] - fn traversal_hidden_disabled_excludes_files_and_descendants() { - let root = TempDirGuard::new(); - fs::create_dir_all(root.path().join(".hidden-dir")).unwrap(); - fs::write(root.path().join(".hidden-dir/child.txt"), "child").unwrap(); - fs::write(root.path().join(".hidden-file"), "secret").unwrap(); - fs::write(root.path().join("visible.txt"), "visible").unwrap(); - - let ct = crate::task::CancelToken::default(); - let entries = - super::collect_entries(root.path(), full_scan_options(false, false), &ct).unwrap(); - - assert_eq!( - entries.len(), - 1, - "only visible.txt should be returned when hidden entries are disabled, got {}", - entry_paths(&entries) - ); - assert_file_entry(&entries, "visible.txt", 7.0); - assert!( - !entries - .iter() - .any(|entry| entry.path.starts_with(".hidden")), - "hidden entries should be pruned before yielding files or descendants, got {}", - entry_paths(&entries) - ); - } - - #[test] - fn traversal_hidden_enabled_includes_non_ignored_hidden_entries() { - let root = TempDirGuard::new(); - fs::create_dir_all(root.path().join(".git")).unwrap(); - fs::write(root.path().join(".gitignore"), ".ignored-hidden\n").unwrap(); - fs::create_dir_all(root.path().join(".hidden-dir")).unwrap(); - fs::write(root.path().join(".hidden-dir/child.txt"), "child").unwrap(); - fs::write(root.path().join(".hidden-file"), "secret").unwrap(); - fs::write(root.path().join(".ignored-hidden"), "ignored").unwrap(); - - let ct = crate::task::CancelToken::default(); - let entries = - super::collect_entries(root.path(), full_scan_options(true, true), &ct).unwrap(); - - assert_file_entry(&entries, ".hidden-file", 6.0); - assert_dir_entry(&entries, ".hidden-dir"); - assert_file_entry(&entries, ".hidden-dir/child.txt", 5.0); - assert!( - !entries.iter().any(|entry| entry.path == ".ignored-hidden"), - "gitignore should still exclude matching hidden files, got {}", - entry_paths(&entries) - ); - } - - #[test] - fn collect_entries_respects_pre_cancelled_token() { - let root = TempDirGuard::new(); - fs::write(root.path().join("real.txt"), "ok").unwrap(); - - let ct = crate::task::CancelToken::new(Some(0), None); - std::thread::sleep(Duration::from_millis(1)); - let result = super::collect_entries( - root.path(), - super::ScanOptions { - include_hidden: true, - use_gitignore: false, - skip_node_modules: true, - follow_links: false, - detail: super::ScanDetail::Minimal, - }, - &ct, - ); - - let Err(err) = result else { - panic!("pre-cancelled scans should fail before returning entries"); - }; - assert!( - err.to_string().contains("Timeout"), - "expected timeout cancellation error, got: {err}" - ); - } - - #[test] - fn force_rescan_respects_skip_node_modules() { - let root = TempDirGuard::new(); - // Create a nested node_modules with many files - for i in 0..100 { - let pkg_dir = root.path().join(format!("node_modules/pkg-{i}")); - fs::create_dir_all(&pkg_dir).unwrap(); - fs::write(pkg_dir.join("index.js"), "x").unwrap(); - } - fs::write(root.path().join("app.js"), "ok").unwrap(); - - let ct = crate::task::CancelToken::default(); - - // With skip: should only get app.js - let entries = super::force_rescan( - root.path(), - super::ScanOptions { - include_hidden: true, - use_gitignore: false, - skip_node_modules: true, - follow_links: false, - detail: super::ScanDetail::Full, - }, - false, - &ct, - ) - .unwrap(); - assert_eq!(entries.len(), 1, "skip=true got: {}", entries.len()); - assert_eq!(entries[0].path, "app.js"); - - // Without skip: should get app.js + 100 node_modules files + directories - let entries = super::force_rescan( - root.path(), - super::ScanOptions { - include_hidden: true, - use_gitignore: false, - skip_node_modules: false, - follow_links: false, - detail: super::ScanDetail::Full, - }, - false, - &ct, - ) - .unwrap(); - assert!(entries.len() > 100, "skip=false got: {}", entries.len()); - } - - #[test] - fn scan_detail_controls_metadata_collection() { - let root = TempDirGuard::new(); - fs::write(root.path().join("real.txt"), "ok").unwrap(); - - let ct = crate::task::CancelToken::default(); - let minimal = super::collect_entries( - root.path(), - super::ScanOptions { - include_hidden: true, - use_gitignore: false, - skip_node_modules: true, - follow_links: false, - detail: super::ScanDetail::Minimal, - }, - &ct, - ) - .unwrap(); - let minimal_file = minimal - .iter() - .find(|entry| entry.path == "real.txt") - .expect("minimal scan includes file"); - assert_eq!(minimal_file.mtime, None); - assert_eq!(minimal_file.size, None); - - let full = super::collect_entries( - root.path(), - super::ScanOptions { - include_hidden: true, - use_gitignore: false, - skip_node_modules: true, - follow_links: false, - detail: super::ScanDetail::Full, - }, - &ct, - ) - .unwrap(); - let full_file = full - .iter() - .find(|entry| entry.path == "real.txt") - .expect("full scan includes file"); - assert!(full_file.mtime.is_some(), "full scan should include mtime"); - assert_eq!(full_file.size, Some(2.0)); - } -} diff --git a/crates/pi-natives/src/glob.rs b/crates/pi-natives/src/glob.rs index b2cabfda3..e30d49ec4 100644 --- a/crates/pi-natives/src/glob.rs +++ b/crates/pi-natives/src/glob.rs @@ -2,9 +2,9 @@ //! caching. //! //! # Overview -//! Resolves a search root, obtains scanned entries via [`fs_cache`], applies -//! glob matching plus optional file-type filtering, and optionally streams each -//! accepted match through a callback. +//! Resolves a search root, scans entries via `pi-walker`, applies glob matching +//! plus optional file-type filtering, and optionally streams each accepted +//! match through a callback. //! //! The walker always skips `.git`, and skips `node_modules` unless explicitly //! requested. @@ -14,15 +14,8 @@ //! // JS: await native.glob({ pattern: "*.rs", path: "." }) //! ``` -use std::{ - cmp::Ordering, - collections::BinaryHeap, - path::Path, - sync::{Arc, Mutex}, -}; +use std::{cmp::Ordering, path::Path}; -use globset::GlobSet; -use ignore::{ParallelVisitor, ParallelVisitorBuilder, WalkState}; use napi::{ bindgen_prelude::*, threadsafe_function::{ThreadsafeFunction, ThreadsafeFunctionCallMode}, @@ -30,8 +23,8 @@ use napi::{ use napi_derive::napi; // Re-export entry types so existing `glob::FileType` / `glob::GlobMatch` paths still work. -pub use crate::fs_cache::{FileType, GlobMatch}; -use crate::{fs_cache, glob_util, task}; +pub use crate::iofs::{FileType, GlobMatch}; +use crate::{glob_util, iofs, task}; /// Input options for `glob`, including traversal, filtering, and cancellation. #[napi(object)] @@ -51,7 +44,7 @@ pub struct GlobOptions<'env> { pub max_results: Option, /// Respect .gitignore files (default: true). pub gitignore: Option, - /// Enable shared filesystem scan cache (default: false). + /// Enable walker scan caching (default: false). pub cache: Option, /// Sort results by mtime (most recent first) before applying limit. pub sort_by_mtime: Option, @@ -84,38 +77,7 @@ struct GlobConfig { use_gitignore: bool, mentions_node_modules: bool, sort_by_mtime: bool, - use_cache: bool, -} - -#[derive(Clone)] -struct RankedGlobMatch { - entry: GlobMatch, -} - -impl PartialEq for RankedGlobMatch { - fn eq(&self, other: &Self) -> bool { - compare_matches_by_rank(&self.entry, &other.entry) == Ordering::Equal - } -} - -impl Eq for RankedGlobMatch {} - -impl PartialOrd for RankedGlobMatch { - fn partial_cmp(&self, other: &Self) -> Option { - Some(self.cmp(other)) - } -} - -impl Ord for RankedGlobMatch { - fn cmp(&self, other: &Self) -> Ordering { - if match_is_worse(&self.entry, &other.entry) { - Ordering::Greater - } else if match_is_worse(&other.entry, &self.entry) { - Ordering::Less - } else { - Ordering::Equal - } - } + cache: bool, } fn match_mtime(entry: &GlobMatch) -> f64 { @@ -128,33 +90,6 @@ fn compare_matches_by_rank(a: &GlobMatch, b: &GlobMatch) -> Ordering { .then_with(|| a.path.cmp(&b.path)) } -fn match_is_worse(a: &GlobMatch, b: &GlobMatch) -> bool { - compare_matches_by_rank(a, b) == Ordering::Greater -} - -/// Returns `true` when `entry` was admitted into the bounded top-`limit` heap -/// (either filling free space or evicting a worse existing entry). -fn push_bounded_match( - heap: &mut BinaryHeap, - entry: GlobMatch, - limit: usize, -) -> bool { - if heap.len() < limit { - heap.push(RankedGlobMatch { entry }); - return true; - } - - let Some(worst) = heap.peek() else { - return false; - }; - if match_is_worse(&worst.entry, &entry) { - heap.pop(); - heap.push(RankedGlobMatch { entry }); - return true; - } - false -} - fn resolve_symlink_target_type(root: &Path, relative_path: &str) -> Option { let target_path = root.join(relative_path); let metadata = std::fs::metadata(target_path).ok()?; @@ -190,240 +125,111 @@ fn apply_file_type_filter(entry: &GlobMatch, config: &GlobConfig) -> Option>, ct: &task::CancelToken, ) -> Result> { - let mut matches = Vec::new(); - if config.max_results == 0 { - return Ok(matches); - } + let outcome = request + .collect_ranked_with_heartbeat( + pi_walker::WalkRank::MtimeDescPathAsc, + config.max_results, + || ct.heartbeat(), + ) + .map_err(iofs::map_walker_error)?; + Ok(outcome.entries.into_iter().map(GlobMatch::from).collect()) +} - for entry in entries { +fn collect_native_filtered_matches( + request: &pi_walker::WalkRequest, + config: &GlobConfig, + ct: &task::CancelToken, +) -> Result> { + let outcome = request + .collect_with_heartbeat(|| ct.heartbeat()) + .map_err(iofs::map_walker_error)?; + let mut collected = Vec::new(); + for entry in outcome.entries { ct.heartbeat()?; - if fs_cache::should_skip_path(Path::new(&entry.path), config.mentions_node_modules) { - // Apply post-scan node_modules policy before glob matching. - continue; - } - if !glob_set.is_match(&entry.path) { - continue; - } - let Some(effective_file_type) = apply_file_type_filter(entry, config) else { + let mut matched_entry = GlobMatch::from(entry); + let Some(effective_file_type) = apply_file_type_filter(&matched_entry, config) else { continue; }; - let mut matched_entry = entry.clone(); matched_entry.file_type = effective_file_type; - if !config.sort_by_mtime - && let Some(callback) = on_match - { - callback.call(Ok(matched_entry.clone()), ThreadsafeFunctionCallMode::NonBlocking); - } - - matches.push(matched_entry); - // Only early-break when not sorting; mtime sort requires full candidate set. - if !config.sort_by_mtime && matches.len() >= config.max_results { + collected.push(matched_entry); + if !config.sort_by_mtime && collected.len() >= config.max_results { break; } } - Ok(matches) + Ok(collected) } -struct SortedMatchVisitor<'a> { - glob_set: &'a GlobSet, - config: &'a GlobConfig, - on_match: Option<&'a ThreadsafeFunction>, - top_matches: BinaryHeap, - shared: Arc>>, - error: Arc>>, - ct: &'a task::CancelToken, - visited: usize, -} - -impl Drop for SortedMatchVisitor<'_> { - fn drop(&mut self) { - if self.top_matches.is_empty() { - return; - } - let drained = std::mem::take(&mut self.top_matches); - self - .shared - .lock() - .expect("glob match collection lock poisoned") - .extend(drained.into_iter().map(|ranked| ranked.entry)); - } -} - -impl ParallelVisitor for SortedMatchVisitor<'_> { - fn visit(&mut self, entry: std::result::Result) -> WalkState { - if self.visited == 0 || self.visited >= 128 { - self.visited = 0; - if let Err(err) = self.ct.heartbeat() { - *self.error.lock().expect("error lock poisoned") = Some(err.to_string()); - return WalkState::Quit; - } - } - self.visited += 1; - - let Ok(entry) = entry else { - return WalkState::Continue; - }; - let Some(mut matched_entry) = - fs_cache::collect_entry(&self.config.root, &entry, fs_cache::ScanDetail::Full) - else { - return WalkState::Continue; - }; - if fs_cache::should_skip_path( - Path::new(&matched_entry.path), - self.config.mentions_node_modules, - ) { - return WalkState::Continue; - } - if !self.glob_set.is_match(&matched_entry.path) { - return WalkState::Continue; - } - let Some(effective_file_type) = apply_file_type_filter(&matched_entry, self.config) else { - return WalkState::Continue; - }; - matched_entry.file_type = effective_file_type; - let streamable = self.on_match.map(|cb| (cb, matched_entry.clone())); - // Admission into the per-thread heap over-approximates the global top-N, - // so streamed partials are a superset; callers dedup and re-rank. - if push_bounded_match(&mut self.top_matches, matched_entry, self.config.max_results) - && let Some((callback, payload)) = streamable - { - callback.call(Ok(payload), ThreadsafeFunctionCallMode::NonBlocking); - } - WalkState::Continue - } -} - -struct SortedMatchVisitorBuilder<'a> { - glob_set: &'a GlobSet, - config: &'a GlobConfig, - on_match: Option<&'a ThreadsafeFunction>, - shared: Arc>>, - error: Arc>>, - ct: &'a task::CancelToken, -} - -impl<'a> ParallelVisitorBuilder<'a> for SortedMatchVisitorBuilder<'a> { - fn build(&mut self) -> Box { - Box::new(SortedMatchVisitor { - glob_set: self.glob_set, - config: self.config, - on_match: self.on_match, - top_matches: BinaryHeap::with_capacity(self.config.max_results.min(1024)), - shared: Arc::clone(&self.shared), - error: Arc::clone(&self.error), - ct: self.ct, - visited: 0, - }) - } -} - -/// Walk the tree in parallel, keeping a bounded top-`max_results` heap per -/// worker. The union of per-thread heaps always contains the global top-N; -/// `run_glob` re-sorts and truncates afterwards, so the final ranking is -/// deterministic (mtime desc, path tiebreak) regardless of walk order. -fn collect_sorted_matches_uncached( - glob_set: &GlobSet, - config: &GlobConfig, - on_match: Option<&ThreadsafeFunction>, - ct: &task::CancelToken, -) -> Result> { - let mut builder = fs_cache::build_walker( - &config.root, - config.include_hidden, - config.use_gitignore, - !config.mentions_node_modules, - false, - ); - let workers = fs_cache::grep_workers(); - if workers > 0 { - builder.threads(workers); - } - let shared = Arc::new(Mutex::new(Vec::new())); - let error = Arc::new(Mutex::new(None)); - let mut visitor_builder = SortedMatchVisitorBuilder { - glob_set, - config, - on_match, - shared: Arc::clone(&shared), - error: Arc::clone(&error), - ct, - }; - ct.heartbeat()?; - builder.build_parallel().visit(&mut visitor_builder); - - let walk_error = error.lock().expect("error lock poisoned").take(); - if let Some(error) = walk_error { - return Err(Error::from_reason(error)); - } - - let mut matches = - std::mem::take(&mut *shared.lock().expect("glob match collection lock poisoned")); - matches.sort_by(compare_matches_by_rank); - matches.truncate(config.max_results); - Ok(matches) -} - -/// Executes matching/filtering over scanned entries and optionally streams each -/// hit. +/// Executes walker-owned glob filtering plus optional native file-type +/// filtering, then optionally streams each returned match. fn run_glob( config: GlobConfig, on_match: Option<&ThreadsafeFunction>, ct: task::CancelToken, ) -> Result { - let glob_set = glob_util::compile_glob(&config.pattern, config.recursive)?; + let walk_glob_pattern = glob_util::build_glob_pattern(&config.pattern, config.recursive); + let walk_glob = pi_walker::CompiledWalkGlob::new([walk_glob_pattern]) + .map_err(|err| Error::from_reason(format!("Invalid glob pattern: {err}")))?; if config.max_results == 0 { return Ok(GlobResult { matches: Vec::new(), total_matches: 0 }); } - let skip_node_modules = !config.mentions_node_modules; - let scan_options = fs_cache::ScanOptions { - include_hidden: config.include_hidden, - use_gitignore: config.use_gitignore, - skip_node_modules, - follow_links: false, - detail: if config.sort_by_mtime { - fs_cache::ScanDetail::Full - } else { - fs_cache::ScanDetail::Minimal - }, - }; - let streams_bounded_sorted_partials = - config.sort_by_mtime && !config.use_cache && config.max_results != usize::MAX; - let mut matches = if streams_bounded_sorted_partials { - collect_sorted_matches_uncached(&glob_set, &config, on_match, &ct)? - } else if config.use_cache { - let scan = fs_cache::get_or_scan(&config.root, scan_options, &ct)?; - let mut matches = filter_entries(&scan.entries, &glob_set, &config, on_match, &ct)?; - // Empty-result recheck: if we got zero matches from a cached scan that's old - // enough, force a rescan and try once more before returning empty. - if matches.is_empty() && scan.cache_age_ms >= fs_cache::empty_recheck_ms() { - let fresh = fs_cache::force_rescan(&config.root, scan_options, true, &ct)?; - matches = filter_entries(&fresh, &glob_set, &config, on_match, &ct)?; - } - matches + let scan_detail = if config.sort_by_mtime { + pi_walker::WalkDetail::Full } else { - let fresh = fs_cache::force_rescan(&config.root, scan_options, false, &ct)?; - filter_entries(&fresh, &glob_set, &config, on_match, &ct)? + pi_walker::WalkDetail::Minimal + }; + let base_request = pi_walker::WalkRequest::new(config.root.clone()) + .hidden(config.include_hidden) + .gitignore(config.use_gitignore) + .skip_git(true) + .skip_node_modules(!config.mentions_node_modules) + .follow_links(pi_walker::FollowLinks::Never) + .detail(scan_detail) + .order(pi_walker::WalkOrder::Path) + .emit_root(false) + .depth(1, usize::MAX) + .directory_errors(pi_walker::DirectoryErrorMode::SkipSkippable) + .cache(config.cache) + .empty_recheck(pi_walker::EmptyRecheck::Configured) + .filter( + pi_walker::WalkFilter::all() + .glob(walk_glob) + .node_modules_unless_mentioned(config.mentions_node_modules), + ); + + let mut matches = if config.sort_by_mtime && config.file_type_filter.is_none() { + collect_ranked_matches(&base_request, &config, &ct)? + } else { + let request = if !config.sort_by_mtime && config.file_type_filter.is_none() { + base_request.limit(config.max_results) + } else { + base_request + }; + collect_native_filtered_matches(&request, &config, &ct)? }; if config.sort_by_mtime { // Sorting mode: rank by mtime descending, then apply max-results truncation. matches.sort_by(compare_matches_by_rank); matches.truncate(config.max_results); - if !streams_bounded_sorted_partials && let Some(callback) = on_match { + if let Some(callback) = on_match { for matched_entry in &matches { callback.call(Ok(matched_entry.clone()), ThreadsafeFunctionCallMode::NonBlocking); } } } + if !config.sort_by_mtime + && let Some(callback) = on_match + { + for matched_entry in &matches { + callback.call(Ok(matched_entry.clone()), ThreadsafeFunctionCallMode::NonBlocking); + } + } let total_matches = matches.len().min(u32::MAX as usize) as u32; Ok(GlobResult { matches, total_matches }) } @@ -433,9 +239,9 @@ fn run_glob( /// Resolves the search root, scans entries, applies glob and optional file-type /// filters, and optionally streams each accepted match through `on_match`. /// -/// If `sortByMtime` is enabled with a finite `maxResults`, uncached scans keep -/// only the current top results while traversing instead of collecting the full -/// tree. +/// When `sortByMtime` is enabled, the walker ranks matches by mtime before the +/// native layer applies final symlink-aware file-type filtering and callback +/// emission. /// /// # Errors /// Returns an error when the search path cannot be resolved, the path is not a @@ -471,7 +277,7 @@ pub fn glob( task::blocking("glob", ct, move |ct| { run_glob( GlobConfig { - root: fs_cache::resolve_search_path(&path)?, + root: pi_walker::resolve_search_path(&path).map_err(iofs::map_walker_error)?, include_hidden: hidden.unwrap_or(false), file_type_filter: file_type, recursive: recursive.unwrap_or(true), @@ -480,7 +286,7 @@ pub fn glob( mentions_node_modules: include_node_modules .unwrap_or_else(|| pattern.contains("node_modules")), sort_by_mtime: sort_by_mtime.unwrap_or(false), - use_cache: cache.unwrap_or(false), + cache: cache.unwrap_or(false), pattern, }, on_match.as_ref(), @@ -488,3 +294,88 @@ pub fn glob( ) }) } + +#[cfg(test)] +mod tests { + use std::{ + fs, + path::{Path, PathBuf}, + sync::atomic::{AtomicU64, Ordering}, + time::{SystemTime, UNIX_EPOCH}, + }; + + static TEMP_COUNTER: AtomicU64 = AtomicU64::new(0); + + struct TempDirGuard(PathBuf); + + impl TempDirGuard { + fn new() -> Self { + let timestamp = SystemTime::now() + .duration_since(UNIX_EPOCH) + .expect("system time is after UNIX_EPOCH") + .as_nanos(); + let counter = TEMP_COUNTER.fetch_add(1, Ordering::Relaxed); + let path = std::env::temp_dir().join(format!("pi-glob-test-{timestamp}-{counter}")); + fs::create_dir_all(&path).expect("create temp test directory"); + Self(path) + } + + fn path(&self) -> &Path { + &self.0 + } + } + + impl Drop for TempDirGuard { + fn drop(&mut self) { + let _ = fs::remove_dir_all(&self.0); + } + } + + fn match_paths(result: &super::GlobResult) -> Vec<&str> { + result + .matches + .iter() + .map(|entry| entry.path.as_str()) + .collect() + } + + #[test] + fn run_glob_with_gitignore_prunes_ignored_directory_but_keeps_matching_sibling() { + let root = TempDirGuard::new(); + fs::create_dir_all(root.path().join(".git")).expect("create repo marker"); + fs::write(root.path().join(".gitignore"), "ignored/\n").expect("write gitignore"); + fs::create_dir_all(root.path().join("ignored")).expect("create ignored directory"); + fs::write(root.path().join("ignored/drop.rs"), "fn ignored() {}\n") + .expect("write ignored rust file"); + fs::write(root.path().join("kept.rs"), "fn kept() {}\n").expect("write kept rust file"); + + let result = super::run_glob( + super::GlobConfig { + root: root.path().to_path_buf(), + pattern: "*.rs".to_string(), + recursive: true, + include_hidden: false, + file_type_filter: Some(super::FileType::File), + max_results: usize::MAX, + use_gitignore: true, + mentions_node_modules: false, + sort_by_mtime: false, + cache: false, + }, + None, + crate::task::CancelToken::default(), + ) + .expect("glob succeeds"); + + let paths = match_paths(&result); + assert_eq!(paths, ["kept.rs"]); + assert_eq!(result.total_matches, 1); + assert!( + !result + .matches + .iter() + .any(|entry| entry.path.starts_with("ignored/")), + "gitignored directory should be pruned before matching, got {paths:?}" + ); + } +} diff --git a/crates/pi-natives/src/glob_util.rs b/crates/pi-natives/src/glob_util.rs index 91ea12fe0..9ee36bb99 100644 --- a/crates/pi-natives/src/glob_util.rs +++ b/crates/pi-natives/src/glob_util.rs @@ -4,6 +4,50 @@ use globset::{GlobBuilder, GlobSet, GlobSetBuilder}; use napi::bindgen_prelude::*; +/// Compiled glob filter with cheap paths for common basename/extension queries. +pub struct CompiledGlob { + fast_path: GlobFastPath, + glob_set: GlobSet, +} + +enum GlobFastPath { + /// Matches any path regardless of depth or name (`**`, `**/*`). + All, + /// Matches only root-level paths (no `/`), regardless of name (`*`). + RootOnly, + /// Matches by extension at any depth (`**/*.ext`, `**/*.{a,b}`). + Extension(Vec), + /// Matches by extension only at the root level (`*.ext`, `*.{a,b}`). + RootExtension(Vec), + /// Matches a literal basename at any depth (`**/name`). + Basename(String), + /// Matches a literal basename only at the root level (bare `name`). + RootBasename(String), + /// Falls back to full glob matching. + GlobSet, +} + +impl CompiledGlob { + /// Returns true when the normalized relative path matches this glob. + pub fn is_match(&self, path: &str) -> bool { + match &self.fast_path { + GlobFastPath::All => true, + GlobFastPath::RootOnly => !path.contains('/'), + GlobFastPath::Extension(exts) => { + path_extension(path).is_some_and(|ext| exts.iter().any(|candidate| ext == candidate)) + }, + GlobFastPath::RootExtension(exts) => { + !path.contains('/') + && path_extension(path) + .is_some_and(|ext| exts.iter().any(|candidate| ext == candidate)) + }, + GlobFastPath::Basename(name) => path.rsplit('/').next() == Some(name.as_str()), + GlobFastPath::RootBasename(name) => path == name.as_str(), + GlobFastPath::GlobSet => self.glob_set.is_match(path), + } + } +} + /// Normalize a raw glob string: fix path separators, optionally prepend `**/` /// for recursive matching, and close any unclosed `{` alternation groups. pub fn build_glob_pattern(glob: &str, recursive: bool) -> String { @@ -20,32 +64,109 @@ pub fn build_glob_pattern(glob: &str, recursive: bool) -> String { fix_unclosed_braces(pattern) } -/// Compile a glob pattern string into a [`GlobSet`]. +/// Compile a glob pattern string into a [`CompiledGlob`]. /// /// When `recursive` is true, simple patterns (no path separators, no leading /// `**`) are automatically prefixed with `**/`. -pub fn compile_glob(glob: &str, recursive: bool) -> Result { +pub fn compile_glob(glob: &str, recursive: bool) -> Result { let mut builder = GlobSetBuilder::new(); let pattern = build_glob_pattern(glob, recursive); - let glob = GlobBuilder::new(&pattern) + let parsed = GlobBuilder::new(&pattern) .literal_separator(true) .build() .map_err(|err| Error::from_reason(format!("Invalid glob pattern: {err}")))?; - builder.add(glob); - builder + builder.add(parsed); + let glob_set = builder .build() - .map_err(|err| Error::from_reason(format!("Failed to build glob matcher: {err}"))) + .map_err(|err| Error::from_reason(format!("Failed to build glob matcher: {err}")))?; + Ok(CompiledGlob { fast_path: classify_fast_path(&pattern), glob_set }) } /// Like [`compile_glob`], but accepts an `Option<&str>` — returns `Ok(None)` /// when the input is `None`, empty, or whitespace-only. -pub fn try_compile_glob(glob: Option<&str>, recursive: bool) -> Result> { +pub fn try_compile_glob(glob: Option<&str>, recursive: bool) -> Result> { let Some(glob) = glob.map(str::trim).filter(|v| !v.is_empty()) else { return Ok(None); }; compile_glob(glob, recursive).map(Some) } +fn classify_fast_path(pattern: &str) -> GlobFastPath { + if matches!(pattern, "**" | "**/*") { + return GlobFastPath::All; + } + if pattern == "*" { + return GlobFastPath::RootOnly; + } + if let Some(ext) = pattern.strip_prefix("**/*.") { + if is_literal_component(ext) { + return GlobFastPath::Extension(vec![ext.to_string()]); + } + } else if let Some(ext) = pattern.strip_prefix("*.") + && is_literal_component(ext) + { + return GlobFastPath::RootExtension(vec![ext.to_string()]); + } + if let Some(inner) = pattern + .strip_prefix("**/*.{") + .and_then(|value| value.strip_suffix('}')) + { + if let Some(extensions) = literal_csv(inner) { + return GlobFastPath::Extension(extensions); + } + } else if let Some(inner) = pattern + .strip_prefix("*.{") + .and_then(|value| value.strip_suffix('}')) + && let Some(extensions) = literal_csv(inner) + { + return GlobFastPath::RootExtension(extensions); + } + if let Some(name) = pattern.strip_prefix("**/") { + if is_literal_path(name) { + return GlobFastPath::Basename(name.to_string()); + } + } else if is_literal_path(pattern) { + return GlobFastPath::RootBasename(pattern.to_string()); + } + GlobFastPath::GlobSet +} + +fn literal_csv(inner: &str) -> Option> { + let extensions: Vec = inner + .split(',') + .filter(|value| !value.is_empty() && is_literal_component(value)) + .map(ToOwned::to_owned) + .collect(); + if extensions.is_empty() || extensions.len() != inner.split(',').count() { + None + } else { + Some(extensions) + } +} + +fn path_extension(path: &str) -> Option<&str> { + let base = path.rsplit('/').next().unwrap_or(path); + let (_, ext) = base.rsplit_once('.')?; + if ext.is_empty() { None } else { Some(ext) } +} + +fn is_literal_component(value: &str) -> bool { + !value.is_empty() + && !value + .chars() + .any(|ch| matches!(ch, '*' | '?' | '[' | ']' | '{' | '}' | '/' | '\\')) +} + +/// True when `value` is a literal single path component: non-empty, no glob +/// metacharacters, and no path separator, so a "basename" fast path is safe +/// to apply regardless of how many directory levels precede it. +fn is_literal_path(value: &str) -> bool { + !value.is_empty() + && !value + .chars() + .any(|ch| matches!(ch, '*' | '?' | '[' | ']' | '{' | '}' | '\\' | '/')) +} + /// Close unclosed `{` alternation groups in a glob pattern. /// /// LLMs occasionally produce patterns like `*.{ts,js` without the closing `}`. @@ -99,6 +220,25 @@ mod tests { assert_eq!(build_glob_pattern("*.ts", false), "*.ts"); } + #[test] + fn compiled_non_recursive_extension_glob_matches_only_root_files() { + let glob = compile_glob("*.rs", false).expect("compile non-recursive extension glob"); + + assert!(glob.is_match("lib.rs")); + assert!(!glob.is_match("src/lib.rs")); + assert!(!glob.is_match("lib.ts")); + } + + #[test] + fn compiled_recursive_extension_glob_matches_nested_files_after_normalization() { + let glob = compile_glob("*.rs", true).expect("compile recursive extension glob"); + + assert!(glob.is_match("lib.rs")); + assert!(glob.is_match("src/lib.rs")); + assert!(glob.is_match("src/nested/lib.rs")); + assert!(!glob.is_match("src/lib.ts")); + } + #[test] fn backslashes_normalized() { assert_eq!(build_glob_pattern("src\\**\\*.ts", true), "src/**/*.ts"); diff --git a/crates/pi-natives/src/grep.rs b/crates/pi-natives/src/grep.rs index e7d9b8777..864215eda 100644 --- a/crates/pi-natives/src/grep.rs +++ b/crates/pi-natives/src/grep.rs @@ -18,13 +18,11 @@ use std::{ }, }; -use globset::GlobSet; use grep_matcher::Matcher; use grep_regex::RegexMatcherBuilder; use grep_searcher::{ BinaryDetection, Searcher, SearcherBuilder, Sink, SinkContext, SinkContextKind, SinkMatch, }; -use ignore::{ParallelVisitor, ParallelVisitorBuilder, WalkState}; use napi::{ JsString, bindgen_prelude::*, @@ -33,7 +31,7 @@ use napi::{ use napi_derive::napi; use smallvec::SmallVec; -use crate::{fast_walk, fs_cache, glob_util, task}; +use crate::{glob_util, iofs, task}; const MAX_FILE_BYTES: u64 = 4 * 1024 * 1024; const SMALL_FILE_READ_BYTES: u64 = 128 * 1024; @@ -146,7 +144,7 @@ pub struct GrepOptions<'env> { } /// A context line (before or after a match). -#[derive(Clone)] +#[derive(Clone, Debug)] #[napi(object)] pub struct ContextLine { /// 1-indexed line number in the source file. @@ -259,6 +257,7 @@ struct MatchCollector { context_before: SmallVec<[ContextLine; 8]>, } +#[derive(Debug)] struct CollectedMatch { line_number: u64, line: String, @@ -274,6 +273,7 @@ struct SearchResultInternal { limit_reached: bool, } +#[derive(Debug)] struct FileSearchResult { relative_path: String, matches: Vec, @@ -512,6 +512,18 @@ fn matches_type_filter(path: &Path, filter: &TypeFilter) -> bool { filter.match_ext(ext) } +fn matches_type_filter_str(path: &str, filter: &TypeFilter) -> bool { + let base = path.rsplit('/').next().unwrap_or(path); + if filter.match_name(base) { + return true; + } + let ext = base.rsplit_once('.').map_or("", |(_, ext)| ext); + if ext.is_empty() { + return false; + } + filter.match_ext(ext) +} + fn resolve_context( context: Option, context_before: Option, @@ -603,8 +615,17 @@ fn build_searcher( .build() } +fn file_len_exceeds_limit(len: usize) -> bool { + u64::try_from(len).map_or(true, |len| len > MAX_FILE_BYTES) +} + /// Read file bytes, distinguishing oversized files from other skips. fn read_file_bytes(path: &Path) -> io::Result { + read_file_bytes_with_size(path, None) +} + +/// Read file bytes with an optional size hint from directory traversal. +fn read_file_bytes_with_size(path: &Path, size_hint: Option) -> io::Result { let file = match File::open(path) { Ok(file) => file, Err(err) @@ -614,20 +635,28 @@ fn read_file_bytes(path: &Path) -> io::Result { }, Err(err) => return Err(err), }; - let metadata = file.metadata()?; - if !metadata.is_file() { - return Ok(ReadFile::Skipped); - } - let size = metadata.len(); + let size = if let Some(size) = size_hint { + size + } else { + let metadata = file.metadata()?; + if !metadata.is_file() { + return Ok(ReadFile::Skipped); + } + metadata.len() + }; if size > MAX_FILE_BYTES { return Ok(ReadFile::Oversized); } else if size == 0 { return Ok(ReadFile::Bytes(FileBytes::Owned(Vec::new()))); } if size <= SMALL_FILE_READ_BYTES { - let mut buffer = Vec::with_capacity(size as usize); + let mut buffer = + Vec::with_capacity(usize::try_from(size).expect("bounded small file size fits usize")); let mut handle = file; handle.read_to_end(&mut buffer)?; + if file_len_exceeds_limit(buffer.len()) { + return Ok(ReadFile::Oversized); + } return Ok(ReadFile::Bytes(FileBytes::Owned(buffer))); } @@ -639,11 +668,18 @@ fn read_file_bytes(path: &Path) -> io::Result { }; let bytes = if let Ok(mapped) = mapping { + if file_len_exceeds_limit(mapped.len()) { + return Ok(ReadFile::Oversized); + } FileBytes::Mapped(mapped) } else { - let mut buffer = Vec::with_capacity(size as usize); + let mut buffer = + Vec::with_capacity(usize::try_from(size).expect("bounded file size fits usize")); let mut handle = file; handle.read_to_end(&mut buffer)?; + if file_len_exceeds_limit(buffer.len()) { + return Ok(ReadFile::Oversized); + } FileBytes::Owned(buffer) }; @@ -1006,160 +1042,77 @@ fn search_file_bytes( run_search_slice(searcher, matcher, bytes, params).ok() } -struct StreamingGrepVisitor<'a> { - root: &'a Path, - matcher: &'a grep_regex::RegexMatcher, - glob_set: Option<&'a GlobSet>, - type_filter: Option<&'a TypeFilter>, - params: SearchParams, - searcher: Searcher, - results: Vec, - shared_results: Arc>>>, - error: Arc>>, - skipped_oversized: Arc, - files_searched: Arc, - stop_after_matches: Option, - emitted: Arc, - ct: &'a task::CancelToken, - visited: usize, -} - -impl Drop for StreamingGrepVisitor<'_> { - fn drop(&mut self) { - if self.results.is_empty() { - return; - } - let results = std::mem::take(&mut self.results); - self - .shared_results - .lock() - .expect("grep result collection lock poisoned") - .push(results); +fn build_grep_walk_request( + search_path: &Path, + glob: Option<&str>, + include_hidden: bool, + use_gitignore: bool, + skip_node_modules: bool, + order: pi_walker::WalkOrder, +) -> Result { + let mut filter = pi_walker::WalkFilter::files_only(); + if let Some(glob) = glob.map(str::trim).filter(|value| !value.is_empty()) { + let pattern = glob_util::build_glob_pattern(glob, true); + let compiled = pi_walker::CompiledWalkGlob::new([pattern]) + .map_err(|err| Error::from_reason(format!("Invalid glob pattern: {err}")))?; + filter = filter.glob(compiled); } + + Ok(pi_walker::WalkRequest::new(search_path) + .hidden(include_hidden) + .gitignore(use_gitignore) + .skip_git(true) + .skip_node_modules(skip_node_modules) + .follow_links(pi_walker::FollowLinks::Never) + .detail(pi_walker::WalkDetail::Minimal) + .size_hints(pi_walker::SizeHintPolicy::WhenCheap) + .order(order) + .emit_root(false) + .depth(1, usize::MAX) + .directory_errors(pi_walker::DirectoryErrorMode::SkipSkippable) + .cache(false) + .filter(filter)) } -impl ParallelVisitor for StreamingGrepVisitor<'_> { - fn visit(&mut self, entry: std::result::Result) -> WalkState { - if let Some(stop) = self.stop_after_matches - && self.emitted.load(Ordering::Relaxed) >= stop - { - return WalkState::Quit; - } - if self.visited == 0 || self.visited >= 128 { - self.visited = 0; - if let Err(err) = self.ct.heartbeat() { - *self.error.lock().expect("error lock poisoned") = Some(err.to_string()); - return WalkState::Quit; - } - } - self.visited += 1; - - let Ok(entry) = entry else { - return WalkState::Continue; - }; - if !entry - .file_type() - .is_some_and(|file_type| file_type.is_file()) - { - return WalkState::Continue; - } - - let relative = fs_cache::normalize_relative_path(self.root, entry.path()); - if relative.is_empty() { - return WalkState::Continue; - } - if let Some(glob_set) = self.glob_set - && !glob_set.is_match(Path::new(relative.as_ref())) - { - return WalkState::Continue; - } - if let Some(filter) = self.type_filter - && !matches_type_filter(entry.path(), filter) - { - return WalkState::Continue; - } - - let bytes = match read_file_bytes(entry.path()) { - Ok(ReadFile::Bytes(bytes)) => bytes, - Ok(ReadFile::Oversized) => { - self.skipped_oversized.fetch_add(1, Ordering::Relaxed); - return WalkState::Continue; - }, - Ok(ReadFile::Skipped) | Err(_) => return WalkState::Continue, - }; - self.files_searched.fetch_add(1, Ordering::Relaxed); - let Some(search) = - search_file_bytes(&mut self.searcher, self.matcher, bytes.as_slice(), self.params) - else { - return WalkState::Continue; - }; - - if search.match_count == 0 { - return WalkState::Continue; - } - let matched_count = search.match_count; - // Budget on matches we actually return (post per-file cap), mirroring the - // sequential path and the aggregator's page accounting. `match_count` can - // exceed `collected` by one when a file overflows its per-file cap, which - // would stop the walk short of the requested page. - let collected = search.collected; - self.results.push(FileSearchResult { - relative_path: relative.into_owned(), - matches: search.matches, - match_count: matched_count, - limit_reached: search.limit_reached, - }); - if let Some(stop) = self.stop_after_matches { - let previous = self.emitted.fetch_add(collected, Ordering::Relaxed); - if previous.saturating_add(collected) >= stop { - return WalkState::Quit; - } - } - WalkState::Continue +fn collect_grep_candidates( + search_path: &Path, + glob: Option<&str>, + type_filter: Option<&TypeFilter>, + include_hidden: bool, + use_gitignore: bool, + skip_node_modules: bool, + order: pi_walker::WalkOrder, + ct: &task::CancelToken, +) -> Result>> { + let request = build_grep_walk_request( + search_path, + glob, + include_hidden, + use_gitignore, + skip_node_modules, + order, + )?; + let mut candidates = match request.collect_file_candidates_with_heartbeat(|| ct.heartbeat()) { + Ok(candidates) => candidates, + Err(pi_walker::WalkError::Unsupported) => return Ok(None), + Err(err) => return Err(iofs::map_walker_error(err)), + }; + if let Some(filter) = type_filter { + candidates.retain(|candidate| matches_type_filter_str(&candidate.relative, filter)); } + Ok(Some(candidates)) } -struct StreamingGrepVisitorBuilder<'a> { - root: &'a Path, - matcher: &'a grep_regex::RegexMatcher, - glob_set: Option<&'a GlobSet>, - type_filter: Option<&'a TypeFilter>, - params: SearchParams, - shared_results: Arc>>>, - error: Arc>>, - skipped_oversized: Arc, - files_searched: Arc, - stop_after_matches: Option, - emitted: Arc, - ct: &'a task::CancelToken, -} - -impl<'a> ParallelVisitorBuilder<'a> for StreamingGrepVisitorBuilder<'a> { - fn build(&mut self) -> Box { - Box::new(StreamingGrepVisitor { - root: self.root, - matcher: self.matcher, - glob_set: self.glob_set, - type_filter: self.type_filter, - params: self.params, - searcher: build_searcher_for_params(self.params), - results: Vec::new(), - shared_results: Arc::clone(&self.shared_results), - error: Arc::clone(&self.error), - skipped_oversized: Arc::clone(&self.skipped_oversized), - files_searched: Arc::clone(&self.files_searched), - stop_after_matches: self.stop_after_matches, - emitted: Arc::clone(&self.emitted), - ct: self.ct, - visited: 0, - }) - } +fn file_size_hint(size: Option) -> Option { + size + .filter(|value| value.is_finite() && *value >= 0.0 && *value <= u64::MAX as f64) + .map(|value| value as u64) } fn run_sequential_grep( search_path: &Path, matcher: &grep_regex::RegexMatcher, - glob_set: Option<&GlobSet>, + glob: Option<&str>, type_filter: Option<&TypeFilter>, params: SearchParams, include_hidden: bool, @@ -1168,50 +1121,36 @@ fn run_sequential_grep( ct: &task::CancelToken, stop_after_matches: Option, ) -> Result<(Vec, u64, u64)> { - let builder = - fs_cache::build_walker(search_path, include_hidden, use_gitignore, skip_node_modules, false); + let Some(candidates) = collect_grep_candidates( + search_path, + glob, + type_filter, + include_hidden, + use_gitignore, + skip_node_modules, + pi_walker::WalkOrder::Path, + ct, + )? + else { + return Ok((Vec::new(), 0, 0)); + }; let file_params = per_file_params(params); let mut searcher = build_searcher_for_params(file_params); let mut results = Vec::new(); let mut skipped_oversized = 0u64; let mut files_searched = 0u64; - let mut visited = 0usize; let mut emitted = 0u64; ct.heartbeat()?; - for entry in builder.build() { - if visited == 0 || visited >= 128 { - visited = 0; - ct.heartbeat()?; - } - visited += 1; - - let Ok(entry) = entry else { - continue; - }; - if !entry - .file_type() - .is_some_and(|file_type| file_type.is_file()) + for file in candidates { + ct.heartbeat()?; + if let Some(stop_after_matches) = stop_after_matches + && emitted >= stop_after_matches { - continue; + break; } - let relative = fs_cache::normalize_relative_path(search_path, entry.path()); - if relative.is_empty() { - continue; - } - if let Some(glob_set) = glob_set - && !glob_set.is_match(Path::new(relative.as_ref())) - { - continue; - } - if let Some(filter) = type_filter - && !matches_type_filter(entry.path(), filter) - { - continue; - } - - let bytes = match read_file_bytes(entry.path()) { + let bytes = match read_file_bytes_with_size(&file.path, file_size_hint(file.size)) { Ok(ReadFile::Bytes(bytes)) => bytes, Ok(ReadFile::Oversized) => { skipped_oversized = skipped_oversized.saturating_add(1); @@ -1230,7 +1169,7 @@ fn run_sequential_grep( } let emitted_in_file = search.collected; results.push(FileSearchResult { - relative_path: relative.into_owned(), + relative_path: file.relative, matches: search.matches, match_count: search.match_count, limit_reached: search.limit_reached, @@ -1247,10 +1186,14 @@ fn run_sequential_grep( Ok((results, skipped_oversized, files_searched)) } -fn try_run_fast_walk_grep( +#[allow( + clippy::fn_params_excessive_bools, + reason = "matches options structure of underlying walk candidates collector" +)] +fn try_run_native_walk_grep( search_path: &Path, matcher: &grep_regex::RegexMatcher, - glob_set: Option<&GlobSet>, + glob: Option<&str>, type_filter: Option<&TypeFilter>, params: SearchParams, include_hidden: bool, @@ -1258,60 +1201,66 @@ fn try_run_fast_walk_grep( skip_node_modules: bool, ct: &task::CancelToken, stop_after_matches: Option, + parallel_search: bool, ) -> Result, u64, u64)>> { - let file_params = per_file_params(params); - let mut searcher = build_searcher_for_params(file_params); - let mut results = Vec::new(); - let mut skipped_oversized = 0u64; - let mut files_searched = 0u64; - let mut emitted = 0u64; - - let status = fast_walk::walk_entries( + let requires_path_order = stop_after_matches.is_some() || params.offset != 0; + let Some(candidates) = collect_grep_candidates( search_path, - fs_cache::ScanOptions { - include_hidden, - use_gitignore, - skip_node_modules, - follow_links: false, - detail: fs_cache::ScanDetail::Minimal, + glob, + type_filter, + include_hidden, + use_gitignore, + skip_node_modules, + if requires_path_order { + pi_walker::WalkOrder::Path + } else { + pi_walker::WalkOrder::Unordered }, ct, - |absolute_path, entry| { - if entry.file_type != fs_cache::FileType::File { - return Ok(fast_walk::FastWalkControl::Continue); - } - if let Some(glob_set) = glob_set - && !glob_set.is_match(Path::new(&entry.path)) + )? + else { + return Ok(None); + }; + + let file_params = per_file_params(params); + if !parallel_search { + let mut searcher = build_searcher_for_params(file_params); + let mut results = Vec::new(); + let mut skipped_oversized = 0u64; + let mut files_searched = 0u64; + let mut emitted = 0u64; + + ct.heartbeat()?; + for file in candidates { + ct.heartbeat()?; + if let Some(stop_after_matches) = stop_after_matches + && emitted >= stop_after_matches { - return Ok(fast_walk::FastWalkControl::Continue); - } - if let Some(filter) = type_filter - && !matches_type_filter(absolute_path, filter) - { - return Ok(fast_walk::FastWalkControl::Continue); + break; } - let bytes = match read_file_bytes(absolute_path) { + let bytes = match read_file_bytes_with_size(&file.path, file_size_hint(file.size)) { Ok(ReadFile::Bytes(bytes)) => bytes, Ok(ReadFile::Oversized) => { skipped_oversized = skipped_oversized.saturating_add(1); - return Ok(fast_walk::FastWalkControl::Continue); + continue; }, - Ok(ReadFile::Skipped) | Err(_) => return Ok(fast_walk::FastWalkControl::Continue), + Ok(ReadFile::Skipped) | Err(_) => continue, }; files_searched = files_searched.saturating_add(1); + let Some(search) = search_file_bytes(&mut searcher, matcher, bytes.as_slice(), file_params) else { - return Ok(fast_walk::FastWalkControl::Continue); + continue; }; if search.match_count == 0 { - return Ok(fast_walk::FastWalkControl::Continue); + continue; } let emitted_in_file = search.collected; results.push(FileSearchResult { - relative_path: entry.path, + relative_path: file.relative, matches: search.matches, match_count: search.match_count, limit_reached: search.limit_reached, @@ -1319,24 +1268,76 @@ fn try_run_fast_walk_grep( if let Some(stop_after_matches) = stop_after_matches { emitted = emitted.saturating_add(emitted_in_file); if emitted >= stop_after_matches { - return Ok(fast_walk::FastWalkControl::Stop); + break; } } - Ok(fast_walk::FastWalkControl::Continue) - }, - )?; + } - if matches!(status, fast_walk::FastWalkStatus::Unsupported) { - return Ok(None); + results.sort_unstable_by(|a, b| a.relative_path.cmp(&b.relative_path)); + return Ok(Some((results, skipped_oversized, files_searched))); } + + let results_mutex = Arc::new(Mutex::new(Vec::new())); + let skipped_oversized = Arc::new(AtomicU64::new(0)); + let files_searched = Arc::new(AtomicU64::new(0)); + let emitted = Arc::new(AtomicU64::new(0)); + + pi_walker::execute_candidates(&candidates, |file| -> Result<()> { + ct.heartbeat()?; + if let Some(stop_after_matches) = stop_after_matches + && emitted.load(Ordering::Relaxed) >= stop_after_matches + { + return Ok(()); + } + + let bytes = match read_file_bytes_with_size(&file.path, file_size_hint(file.size)) { + Ok(ReadFile::Bytes(bytes)) => bytes, + Ok(ReadFile::Oversized) => { + skipped_oversized.fetch_add(1, Ordering::Relaxed); + return Ok(()); + }, + Ok(ReadFile::Skipped) | Err(_) => return Ok(()), + }; + files_searched.fetch_add(1, Ordering::Relaxed); + + let mut searcher = build_searcher_for_params(file_params); + if let Some(search) = search_file_bytes(&mut searcher, matcher, bytes.as_slice(), file_params) + && search.match_count > 0 + { + let emitted_in_file = search.collected; + results_mutex + .lock() + .expect("results lock poisoned") + .push(FileSearchResult { + relative_path: file.relative.clone(), + matches: search.matches, + match_count: search.match_count, + limit_reached: search.limit_reached, + }); + if stop_after_matches.is_some() { + emitted.fetch_add(emitted_in_file, Ordering::Relaxed); + } + } + + Ok(()) + })?; + + let mut results = { + let mut locked = results_mutex.lock().expect("results lock poisoned"); + std::mem::take(&mut *locked) + }; results.sort_unstable_by(|a, b| a.relative_path.cmp(&b.relative_path)); - Ok(Some((results, skipped_oversized, files_searched))) + Ok(Some(( + results, + skipped_oversized.load(Ordering::Relaxed), + files_searched.load(Ordering::Relaxed), + ))) } fn run_streaming_grep( search_path: &Path, matcher: &grep_regex::RegexMatcher, - glob_set: Option<&GlobSet>, + glob: Option<&str>, type_filter: Option<&TypeFilter>, params: SearchParams, include_hidden: bool, @@ -1345,82 +1346,41 @@ fn run_streaming_grep( ct: &task::CancelToken, ) -> Result<(Vec, u64, u64)> { let (_active_guard, active_greps) = ActiveStreamingGrep::enter(); - let workers = fs_cache::grep_workers(); + let workers = pi_walker::walk_workers(); let stop_after_matches = streaming_stop_after(params); let small_budget = stop_after_matches.is_some_and(|max| max <= ORDERED_STREAMING_STOP_MAX_COUNT); // Sequential path: forced workers, contended pool, or a small first-page // budget where strict path-order matters. The parallel path below also // honors `stop_after_matches`, so larger budgets bound work without losing // parallelism. - if workers == 1 || small_budget || (workers > 1 && active_greps > 1) { - if let Some(result) = try_run_fast_walk_grep( - search_path, - matcher, - glob_set, - type_filter, - params, - include_hidden, - use_gitignore, - skip_node_modules, - ct, - stop_after_matches, - )? { - return Ok(result); - } - return run_sequential_grep( - search_path, - matcher, - glob_set, - type_filter, - params, - include_hidden, - use_gitignore, - skip_node_modules, - ct, - stop_after_matches, - ); - } - let file_params = per_file_params(params); - let mut builder = - fs_cache::build_walker(search_path, include_hidden, use_gitignore, skip_node_modules, false); - if workers > 0 { - builder.threads(workers); - } - let shared_results = Arc::new(Mutex::new(Vec::new())); - let error = Arc::new(Mutex::new(None)); - let skipped_oversized = Arc::new(AtomicU64::new(0)); - let files_searched = Arc::new(AtomicU64::new(0)); - let emitted = Arc::new(AtomicU64::new(0)); - let mut visitor_builder = StreamingGrepVisitorBuilder { - root: search_path, + let use_parallel_fast_path = workers > 1 && !small_budget && active_greps == 1; + if let Some(result) = try_run_native_walk_grep( + search_path, matcher, - glob_set, + glob, type_filter, - params: file_params, - shared_results: Arc::clone(&shared_results), - error: Arc::clone(&error), - skipped_oversized: Arc::clone(&skipped_oversized), - files_searched: Arc::clone(&files_searched), - stop_after_matches, - emitted: Arc::clone(&emitted), + params, + include_hidden, + use_gitignore, + skip_node_modules, ct, - }; - ct.heartbeat()?; - builder.build_parallel().visit(&mut visitor_builder); - - let walk_error = error.lock().expect("error lock poisoned").take(); - if let Some(error) = walk_error { - return Err(Error::from_reason(error)); + stop_after_matches, + use_parallel_fast_path, + )? { + return Ok(result); } - - let mut results: Vec = shared_results - .lock() - .expect("grep result collection lock poisoned") - .drain(..) - .flatten() - .collect(); - results.sort_unstable_by(|a, b| a.relative_path.cmp(&b.relative_path)); - Ok((results, skipped_oversized.load(Ordering::Relaxed), files_searched.load(Ordering::Relaxed))) + run_sequential_grep( + search_path, + matcher, + glob, + type_filter, + params, + include_hidden, + use_gitignore, + skip_node_modules, + ct, + stop_after_matches, + ) } fn push_count_match(matches: &mut Vec, path: String, match_count: u64) { @@ -1617,7 +1577,8 @@ fn grep_sync( let offset = options.offset.unwrap_or(0) as u64; let include_hidden = options.hidden.unwrap_or(true); let use_gitignore = options.gitignore.unwrap_or(true); - let glob_set = glob_util::try_compile_glob(options.glob.as_deref(), true)?; + let glob = options.glob.as_deref(); + let _ = glob_util::try_compile_glob(glob, true)?; let type_filter = resolve_type_filter(options.type_filter.as_deref()); let params = SearchParams { @@ -1771,14 +1732,11 @@ fn grep_sync( }); } - let mentions_node_modules = options - .glob - .as_deref() - .is_some_and(|g| g.contains("node_modules")); + let mentions_node_modules = glob.is_some_and(|g| g.contains("node_modules")); let results = run_streaming_grep( &search_path, &matcher, - glob_set.as_ref(), + glob, type_filter.as_ref(), params, include_hidden, @@ -2146,6 +2104,32 @@ mod tests { assert_eq!(result.matches[0].path, "a.txt"); } + #[cfg(unix)] + #[test] + fn grep_sync_with_gitignore_skips_ignored_rs_files() { + let root = TempDirGuard::new(); + fs::create_dir_all(root.path().join(".git")).expect("create repo marker"); + fs::write(root.path().join(".gitignore"), "ignored.rs\nignored-dir/\n") + .expect("write gitignore"); + write_file(&root.path().join("kept.rs"), "needle kept\n"); + write_file(&root.path().join("ignored.rs"), "needle ignored\n"); + write_file(&root.path().join("ignored-dir/nested.rs"), "needle nested\n"); + + let mut config = base_grep_config(root.path()); + config.gitignore = Some(true); + config.glob = Some("*.rs".to_string()); + + let result = grep_sync(config, None, task::CancelToken::default()) + .expect("gitignore-aware grep should succeed"); + + assert_eq!(result.total_matches, 1); + assert_eq!(result.files_with_matches, 1); + assert_eq!(result.files_searched, 1); + assert_eq!(result.matches.len(), 1); + assert_eq!(result.matches[0].path, "kept.rs"); + assert_eq!(result.matches[0].line, "needle kept"); + } + #[cfg(unix)] #[test] fn grep_files_with_matches_counts_all_searched_files_when_none_match() { diff --git a/crates/pi-natives/src/iofs.rs b/crates/pi-natives/src/iofs.rs new file mode 100644 index 000000000..392ed4881 --- /dev/null +++ b/crates/pi-natives/src/iofs.rs @@ -0,0 +1,86 @@ +//! N-API filesystem DTOs and conversion helpers. +//! +//! `pi-walker` owns traversal and cache policy. This module keeps only the +//! JavaScript-facing shapes plus conversions between walker entries and N-API +//! payloads. + +use napi::bindgen_prelude::*; +use napi_derive::napi; + +/// Resolved filesystem entry kind for glob filters and match metadata. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[napi] +pub enum FileType { + /// Regular file. + File = 1, + /// Directory. + Dir = 2, + /// Symbolic link. + Symlink = 3, +} + +/// A single filesystem entry from a directory scan. +#[derive(Clone)] +#[napi(object)] +pub struct GlobMatch { + /// Relative path from the search root, using forward slashes. + pub path: String, + /// Resolved filesystem type for the match. + pub file_type: FileType, + /// Modification time in milliseconds since Unix epoch. + pub mtime: Option, + /// File size in bytes for regular files. + pub size: Option, +} + +fn walker_error_to_napi(err: pi_walker::WalkError) -> Error { + match err { + pi_walker::WalkError::Unsupported => { + Error::from_reason("Native directory scan unsupported".to_string()) + }, + pi_walker::WalkError::Interrupted(err) => Error::from_reason(err.to_string()), + pi_walker::WalkError::InvalidData { path, message } => Error::from_reason(format!( + "Native directory scan failed for {}: {message}", + path.display() + )), + } +} + +pub(crate) const fn from_walker_file_type(file_type: pi_walker::FileType) -> FileType { + match file_type { + pi_walker::FileType::File => FileType::File, + pi_walker::FileType::Dir => FileType::Dir, + pi_walker::FileType::Symlink => FileType::Symlink, + } +} + +impl From for GlobMatch { + fn from(entry: pi_walker::CollectedEntry) -> Self { + Self { + path: entry.path, + file_type: from_walker_file_type(entry.file_type), + mtime: entry.mtime, + size: entry.size, + } + } +} + +/// Converts a native walker error into an N-API error. +pub(crate) fn map_walker_error(err: pi_walker::WalkError) -> Error { + walker_error_to_napi(err) +} + +/// Invalidate the walker scan cache. +/// +/// When called with a path, removes entries for roots containing that path. +/// When called without a path, clears the entire cache. +/// +/// Intended to be called after agent file mutations: write, edit, rename, or +/// delete. +#[napi] +pub fn invalidate_fs_scan_cache(path: Option) { + match path { + Some(path) => pi_walker::invalidate_path_string(&path), + None => pi_walker::invalidate_all(), + } +} diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index 78a8cc878..8784caaf3 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -27,14 +27,13 @@ pub mod ast; pub mod block; pub mod clipboard; pub mod crash_handler; -pub(crate) mod fast_walk; pub mod fd; -pub mod fs_cache; pub mod glob; pub mod glob_util; pub mod grep; pub mod highlight; pub mod html; +pub mod iofs; pub mod keys; pub mod sixel; pub mod snapcompact; diff --git a/crates/pi-natives/src/workspace.rs b/crates/pi-natives/src/workspace.rs index 174d2fb3d..feaaf40ba 100644 --- a/crates/pi-natives/src/workspace.rs +++ b/crates/pi-natives/src/workspace.rs @@ -9,16 +9,14 @@ use std::{ collections::HashSet, path::{Path, PathBuf}, - sync::{Arc, LazyLock}, + sync::LazyLock, }; -use ignore::{DirEntry, ParallelVisitor, ParallelVisitorBuilder, WalkBuilder, WalkState}; use napi::bindgen_prelude::*; use napi_derive::napi; -use parking_lot::Mutex; use crate::{ - fs_cache::{self, FileType, GlobMatch}, + iofs::{self, FileType, GlobMatch}, task, }; @@ -90,68 +88,35 @@ struct WorkspaceConfig { collect_agents_md: bool, } -fn build_workspace_walker(config: &WorkspaceConfig) -> WalkBuilder { - let mut builder = WalkBuilder::new(&config.root); - builder - .hidden(!config.include_hidden) - .follow_links(false) - .sort_by_file_path(|a, b| a.cmp(b)) - .max_depth(Some(config.walk_max_depth)) - .filter_entry(|entry| { - let name = entry.file_name().to_str().unwrap_or_default(); - if name == ".DS_Store" { - return false; - } - if entry - .file_type() - .is_some_and(|file_type| file_type.is_dir()) - && EXCLUDED_DIR_SET.contains(name) - { - return false; - } - true - }); - - if config.use_gitignore { - builder - .git_ignore(true) - .git_exclude(true) - .git_global(true) - .ignore(true) - .parents(true) - // Honor .gitignore even when the directory isn't a git repo, - // matching what users expect from a plain directory listing. - .require_git(false); - } else { - builder - .git_ignore(false) - .git_exclude(false) - .git_global(false) - .ignore(false) - .parents(false); - } - - builder +fn build_workspace_walk_request(config: &WorkspaceConfig) -> pi_walker::WalkRequest { + pi_walker::WalkRequest::new(config.root.clone()) + .hidden(config.include_hidden) + .gitignore(config.use_gitignore) + .skip_git(true) + .skip_node_modules(true) + .follow_links(pi_walker::FollowLinks::Never) + .detail(pi_walker::WalkDetail::Full) + .order(pi_walker::WalkOrder::Path) + .emit_root(false) + .depth(1, config.walk_max_depth) + .directory_errors(pi_walker::DirectoryErrorMode::SkipSkippable) + .cache(false) } fn glob_match_from_path(root: &Path, path: &Path) -> Option { - let relative = fs_cache::normalize_relative_path(root, path); + let relative = pi_walker::normalize_relative_path(root, path); if relative.is_empty() { return None; } - let (file_type, mtime, size) = fs_cache::classify_file_type(path)?; + let (file_type, mtime, size) = pi_walker::classify_file_type(path)?; Some(GlobMatch { path: relative.into_owned(), - file_type, + file_type: crate::iofs::from_walker_file_type(file_type), mtime, size: size.map(|value| value as f64), }) } -fn glob_match_from_entry(root: &Path, entry: &DirEntry) -> Option { - glob_match_from_path(root, entry.path()) -} - fn is_file_or_file_symlink(path: &Path, file_type: FileType) -> bool { match file_type { FileType::File => true, @@ -160,6 +125,24 @@ fn is_file_or_file_symlink(path: &Path, file_type: FileType) -> bool { } } +fn is_excluded_workspace_entry(relative: &str, file_type: FileType) -> bool { + let mut components = relative + .split('/') + .filter(|component| !component.is_empty()) + .peekable(); + while let Some(component) = components.next() { + if component == ".DS_Store" { + return true; + } + let is_final_component = components.peek().is_none(); + if EXCLUDED_DIR_SET.contains(component) && (!is_final_component || file_type == FileType::Dir) + { + return true; + } + } + false +} + fn collect_agents_md_in_directory( config: &WorkspaceConfig, directory: &Path, @@ -188,89 +171,6 @@ fn collect_agents_md_in_directory( } } -struct WorkspaceVisitor<'a> { - config: &'a WorkspaceConfig, - ct: &'a task::CancelToken, - entries: Vec, - agents_md_files: Vec, - shared_entries: Arc>>>, - shared_agents_md_files: Arc>>>, - error: Arc>>, - visited: usize, -} - -impl Drop for WorkspaceVisitor<'_> { - fn drop(&mut self) { - if !self.entries.is_empty() { - let entries = std::mem::take(&mut self.entries); - self.shared_entries.lock().push(entries); - } - if !self.agents_md_files.is_empty() { - let agents_md_files = std::mem::take(&mut self.agents_md_files); - self.shared_agents_md_files.lock().push(agents_md_files); - } - } -} - -impl ParallelVisitor for WorkspaceVisitor<'_> { - fn visit(&mut self, entry: std::result::Result) -> WalkState { - if self.visited == 0 || self.visited >= 128 { - self.visited = 0; - if let Err(err) = self.ct.heartbeat() { - *self.error.lock() = Some(err.to_string()); - return WalkState::Quit; - } - } - self.visited += 1; - - let Ok(entry) = entry else { - return WalkState::Continue; - }; - let entry_depth = entry.depth(); - if entry - .file_type() - .is_some_and(|file_type| file_type.is_dir()) - { - collect_agents_md_in_directory( - self.config, - entry.path(), - entry_depth, - &mut self.entries, - &mut self.agents_md_files, - ); - } - if entry_depth <= self.config.max_depth - && let Some(entry) = glob_match_from_entry(&self.config.root, &entry) - { - self.entries.push(entry); - } - WalkState::Continue - } -} - -struct WorkspaceVisitorBuilder<'a> { - config: &'a WorkspaceConfig, - ct: &'a task::CancelToken, - shared_entries: Arc>>>, - shared_agents_md_files: Arc>>>, - error: Arc>>, -} - -impl<'a> ParallelVisitorBuilder<'a> for WorkspaceVisitorBuilder<'a> { - fn build(&mut self) -> Box { - Box::new(WorkspaceVisitor { - config: self.config, - ct: self.ct, - entries: Vec::new(), - agents_md_files: Vec::new(), - shared_entries: Arc::clone(&self.shared_entries), - shared_agents_md_files: Arc::clone(&self.shared_agents_md_files), - error: Arc::clone(&self.error), - visited: 0, - }) - } -} - fn sort_dedup_entries(entries: &mut Vec) { entries.sort_unstable_by(|a, b| a.path.cmp(&b.path)); entries.dedup_by(|a, b| a.path == b.path); @@ -285,48 +185,38 @@ fn run_list_workspace( config: WorkspaceConfig, ct: task::CancelToken, ) -> Result { - let mut root_entries = Vec::new(); - let mut root_agents_md_files = Vec::new(); - collect_agents_md_in_directory( - &config, - &config.root, - 0, - &mut root_entries, - &mut root_agents_md_files, - ); + let mut entries = Vec::new(); + let mut agents_md_files = Vec::new(); + collect_agents_md_in_directory(&config, &config.root, 0, &mut entries, &mut agents_md_files); - let mut builder = build_workspace_walker(&config); - let workers = fs_cache::grep_workers(); - if workers > 0 { - builder.threads(workers); + let outcome = build_workspace_walk_request(&config) + .collect_with_heartbeat(|| ct.heartbeat()) + .map_err(iofs::map_walker_error)?; + + for entry in outcome.entries { + let file_type = iofs::from_walker_file_type(entry.file_type); + if is_excluded_workspace_entry(&entry.path, file_type) { + continue; + } + + let entry_depth = entry.depth(); + if file_type == FileType::Dir { + let directory = entry.absolute_path(&config.root); + collect_agents_md_in_directory( + &config, + &directory, + entry_depth, + &mut entries, + &mut agents_md_files, + ); + } + + if entry_depth <= config.max_depth { + entries.push(entry.into()); + } } - let shared_entries = Arc::new(Mutex::new(Vec::new())); - let shared_agents_md_files = Arc::new(Mutex::new(Vec::new())); - let error = Arc::new(Mutex::new(None)); - let mut visitor_builder = WorkspaceVisitorBuilder { - config: &config, - ct: &ct, - shared_entries: Arc::clone(&shared_entries), - shared_agents_md_files: Arc::clone(&shared_agents_md_files), - error: Arc::clone(&error), - }; - - ct.heartbeat()?; - builder.build_parallel().visit(&mut visitor_builder); - - let walk_error = error.lock().take(); - if let Some(error) = walk_error { - return Err(Error::from_reason(error)); - } - - let mut entries: Vec = shared_entries.lock().drain(..).flatten().collect(); - entries.extend(root_entries); sort_dedup_entries(&mut entries); - - let mut agents_md_files: Vec = - shared_agents_md_files.lock().drain(..).flatten().collect(); - agents_md_files.extend(root_agents_md_files); sort_dedup_paths(&mut agents_md_files); let entries_truncated = entries.len() > MAX_ENTRIES; @@ -373,7 +263,7 @@ pub fn list_workspace(options: ListWorkspaceOptions<'_>) -> task::Promise bool { - self.0.is_empty() - } - fn matches(&self, path: &Path, base_dir: &Path) -> bool { if self.0.is_empty() { return false; @@ -327,6 +324,151 @@ impl Excludes { } } +struct FdIgnoreMatcher { + enabled: bool, + root: PathBuf, + global: Vec, + states: HashMap>, +} + +struct FdIgnoreState { + parent: Option>, + matcher: Option, +} + +impl FdIgnoreMatcher { + fn new(base_dir: &Path, root: &Path, cli: &FdCli) -> io::Result { + let root = normalize_fdignore_path(root); + let enabled = !no_ignore(cli); + let mut matcher = + Self { enabled, root: root.clone(), global: Vec::new(), states: HashMap::new() }; + if !enabled { + return Ok(matcher); + } + + for ignore_file in &cli.ignore_files { + let path = if ignore_file.is_absolute() { + ignore_file.clone() + } else { + base_dir.join(ignore_file) + }; + let mut builder = ignore::gitignore::GitignoreBuilder::new(base_dir); + if let Some(err) = builder.add(&path) { + return Err(io::Error::other(err.to_string())); + } + let ignore = builder + .build() + .map_err(|err| io::Error::other(err.to_string()))?; + if !ignore.is_empty() { + matcher.global.push(ignore); + } + } + + let parent = if cli.no_ignore_parent { + None + } else { + build_fdignore_parent_states(root.parent()) + }; + let root_state = load_fdignore_state(&root, parent); + matcher.states.insert(root, root_state); + Ok(matcher) + } + + fn is_ignored(&mut self, path: &Path, is_dir: bool) -> bool { + if !self.enabled { + return false; + } + let path = normalize_fdignore_path(path); + let state_dir = path.parent().unwrap_or(&self.root).to_path_buf(); + let state = self.state_for_dir(&state_dir); + if let Some(ignored) = fdignore_state_match(&state, &path, is_dir) { + return ignored; + } + self + .global + .iter() + .find_map(|ignore| fdignore_match(ignore, &path, is_dir)) + .unwrap_or(false) + } + + fn state_for_dir(&mut self, dir: &Path) -> Arc { + if let Some(state) = self.states.get(dir) { + return Arc::clone(state); + } + if !dir.starts_with(&self.root) { + return self + .states + .get(&self.root) + .map_or_else(|| Arc::new(FdIgnoreState { parent: None, matcher: None }), Arc::clone); + } + let parent = if dir == self.root.as_path() { + self + .states + .get(&self.root) + .and_then(|state| state.parent.as_ref().map(Arc::clone)) + } else { + dir.parent().map(|parent| self.state_for_dir(parent)) + }; + let state = load_fdignore_state(dir, parent); + self.states.insert(dir.to_path_buf(), Arc::clone(&state)); + state + } +} + +fn build_fdignore_parent_states(mut dir: Option<&Path>) -> Option> { + let mut ancestors = Vec::new(); + while let Some(path) = dir { + ancestors.push(path); + dir = path.parent(); + } + let mut parent = None; + for ancestor in ancestors.into_iter().rev() { + parent = Some(load_fdignore_state(ancestor, parent)); + } + parent +} + +fn normalize_fdignore_path(path: &Path) -> PathBuf { + path.components().collect() +} + +fn load_fdignore_state(dir: &Path, parent: Option>) -> Arc { + let file = dir.join(".fdignore"); + let matcher = if file.is_file() { + let mut builder = ignore::gitignore::GitignoreBuilder::new(dir); + let _ = builder.add(&file); + builder.build().ok().filter(|ignore| !ignore.is_empty()) + } else { + None + }; + Arc::new(FdIgnoreState { parent, matcher }) +} + +fn fdignore_state_match(state: &Arc, path: &Path, is_dir: bool) -> Option { + let mut current = Some(state.as_ref()); + while let Some(frame) = current { + if let Some(matcher) = &frame.matcher + && let Some(ignored) = fdignore_match(matcher, path, is_dir) + { + return Some(ignored); + } + current = frame.parent.as_deref(); + } + None +} + +fn fdignore_match( + matcher: &ignore::gitignore::Gitignore, + path: &Path, + is_dir: bool, +) -> Option { + match matcher.matched(path, is_dir) { + ignore::Match::Ignore(_) => Some(true), + ignore::Match::Whitelist(_) => Some(false), + ignore::Match::None => None, + } +} + #[derive(Clone, Default)] struct TypeFilter { regular: bool, @@ -580,122 +722,196 @@ fn search( prune: cli.prune, }; - let mut builder = WalkBuilder::new(&search_paths[0].resolved); - for path in search_paths.iter().skip(1) { - builder.add(&path.resolved); - } - builder.current_dir(&config.base_dir); - if !no_ignore(&cli) { - builder.add_custom_ignore_filename(".fdignore"); - } - builder.hidden(!include_hidden(&cli)); - builder.ignore(!no_ignore(&cli)); - builder.git_ignore(!(no_ignore(&cli) || no_ignore_vcs(&cli))); - builder.git_global(!(no_ignore(&cli) || no_ignore_vcs(&cli))); - builder.git_exclude(!(no_ignore(&cli) || no_ignore_vcs(&cli))); - builder.parents(!(no_ignore(&cli) || cli.no_ignore_parent)); - builder.require_git(!cli.no_require_git); - builder.follow_links(cli.follow); - builder.same_file_system(cli.one_file_system); - if let Some(depth) = cli.exact_depth { - builder.min_depth(Some(depth)); - builder.max_depth(Some(depth)); - } else { - builder.min_depth(cli.min_depth); - builder.max_depth(cli.max_depth); - } - if let Some(threads) = cli.threads { - builder.threads(threads); - } - for ignore_file in &cli.ignore_files { - let path = if ignore_file.is_absolute() { - ignore_file.clone() - } else { - config.base_dir.join(ignore_file) - }; - if let Some(err) = builder.add_ignore(path) { - return Err(io::Error::other(err.to_string())); - } - } - if !config.excludes.is_empty() || !cli.ignore_contains.is_empty() || config.prune { - let excludes = config.excludes.clone(); - let base_dir = config.base_dir.clone(); - let ignore_contains = cli.ignore_contains; - let matcher = Arc::clone(&config.matcher); - let prune = config.prune; - let full_path = config.full_path; - builder.filter_entry(move |entry| { - if entry.depth() == 0 { - return true; - } - let path = entry.path(); - if excludes.matches(path, &base_dir) { - return false; - } - if entry - .file_type() - .is_some_and(|file_type| file_type.is_dir()) - { - if ignore_contains.iter().any(|name| path.join(name).exists()) { - return false; - } - if prune && matcher.matches(&match_target(path, &base_dir, full_path)) { - return false; - } - } - true - }); + if let Some(state) = + try_search_fast(&cli, &search_paths, &config, max_results, stdout, stderr, cancelled)? + { + return Ok(state); } + let use_gitignore = !(no_ignore(&cli) || no_ignore_vcs(&cli)); let mut out = BufWriter::new(stdout); let mut state = SearchState { matches: 0, had_error: false }; - for entry in builder.build() { + for search_path in &search_paths { if cancelled.load(Ordering::Relaxed) || max_results.is_some_and(|max| state.matches >= max) { break; } - match entry { - Ok(entry) => process_entry(&config, &entry, &mut out, &mut state)?, - Err(err) => { - if config.show_errors { - state.had_error = true; - let _ = writeln!(stderr, "fd: {err}"); - } - }, + let mut fd_ignores = FdIgnoreMatcher::new(&config.base_dir, &search_path.resolved, &cli)?; + let request = + fd_walk_request(&search_path.resolved, &cli, use_gitignore, cli.one_file_system); + let outcome = request + .collect_with_heartbeat(|| Ok::<(), io::Error>(())) + .map_err(walker_collect_error_to_io)?; + let mut pruned_dirs = Vec::new(); + for entry in &outcome.entries { + if cancelled.load(Ordering::Relaxed) || max_results.is_some_and(|max| state.matches >= max) + { + break; + } + process_collected_entry( + &config, + &cli.ignore_contains, + &mut fd_ignores, + &search_path.resolved, + entry, + &mut out, + &mut state, + &mut pruned_dirs, + )?; } } out.flush()?; Ok(state) } -fn process_entry( +fn fd_walk_request( + root: &Path, + cli: &FdCli, + use_gitignore: bool, + same_file_system: bool, +) -> pi_walker::WalkRequest { + let min_depth = cli.exact_depth.or(cli.min_depth).unwrap_or(0); + let max_depth = cli.exact_depth.or(cli.max_depth).unwrap_or(usize::MAX); + pi_walker::WalkRequest::new(root) + .hidden(include_hidden(cli)) + .gitignore(use_gitignore) + .skip_git(false) + .skip_node_modules(false) + .follow_links(cli.follow.into()) + .detail(pi_walker::WalkDetail::Minimal) + .order(pi_walker::WalkOrder::Path) + .emit_root(true) + .depth(min_depth, max_depth) + .directory_errors(pi_walker::DirectoryErrorMode::Visit) + .same_file_system(same_file_system) + .cache(false) + .visit_order(pi_walker::VisitOrder::PreOrder) +} + +fn try_search_fast( + cli: &FdCli, + search_paths: &[SearchPath], config: &SearchConfig, - entry: &DirEntry, + max_results: Option, + stdout: &mut OpenFile, + stderr: &mut OpenFile, + cancelled: &AtomicBool, +) -> io::Result> { + if !can_use_fast_search(cli, config) { + return Ok(None); + } + + let mut out = BufWriter::new(Vec::new()); + let mut err = Vec::new(); + let mut state = SearchState { matches: 0, had_error: false }; + for search_path in search_paths { + if cancelled.load(Ordering::Relaxed) || max_results.is_some_and(|max| state.matches >= max) { + break; + } + let mut matches = state.matches; + let mut had_error = state.had_error; + let request = fd_walk_request(&search_path.resolved, cli, false, false); + let status = request.for_each_entry_with_heartbeat( + || Ok::<(), io::Error>(()), + |entry| { + if cancelled.load(Ordering::Relaxed) || max_results.is_some_and(|max| matches >= max) { + return Ok(pi_walker::WalkDecision::Stop); + } + let decision = process_walker_entry( + config, + &cli.ignore_contains, + entry.absolute_path.as_ref(), + entry.depth, + entry.file_type, + &mut out, + &mut matches, + )?; + if cancelled.load(Ordering::Relaxed) || max_results.is_some_and(|max| matches >= max) { + Ok(pi_walker::WalkDecision::Stop) + } else { + Ok(decision) + } + }, + |error| { + if config.show_errors { + had_error = true; + let _ = writeln!(err, "fd: {}", error.error); + } + Ok(pi_walker::WalkDecision::Include) + }, + ); + state.matches = matches; + state.had_error = had_error; + match status { + Ok(pi_walker::WalkStatus::Unsupported) | Err(pi_walker::WalkError::Unsupported) => { + return Ok(None); + }, + Ok(pi_walker::WalkStatus::Complete | pi_walker::WalkStatus::Stopped) => {}, + Err(err) => return Err(walker_error_to_io(err)), + } + } + out.flush()?; + let output = out.into_inner().map_err(|err| err.into_error())?; + stdout.write_all(&output)?; + stderr.write_all(&err)?; + Ok(Some(state)) +} + +const fn can_use_fast_search(cli: &FdCli, config: &SearchConfig) -> bool { + no_ignore(cli) + && cli.ignore_files.is_empty() + && !cli.one_file_system + && fast_type_filter_supported(&config.types) +} + +const fn fast_type_filter_supported(filter: &TypeFilter) -> bool { + !filter.socket && !filter.pipe && !filter.block && !filter.character +} + +fn process_walker_entry( + config: &SearchConfig, + ignore_contains: &[OsString], + path: &Path, + depth: usize, + file_type: pi_walker::FileType, out: &mut W, - state: &mut SearchState, -) -> io::Result<()> { - if entry.depth() == 0 - && entry - .file_type() - .is_some_and(|file_type| file_type.is_dir()) - { - return Ok(()); + matches: &mut usize, +) -> io::Result { + let is_directory = file_type == pi_walker::FileType::Dir; + if depth == 0 && is_directory { + return Ok(pi_walker::WalkDecision::Skip); } - let path = entry.path(); if config.excludes.matches(path, &config.base_dir) { - return Ok(()); + return Ok(if is_directory { + pi_walker::WalkDecision::SkipDescend + } else { + pi_walker::WalkDecision::Skip + }); } - let metadata = entry.metadata().ok(); - if !matches_filters(config, entry, metadata.as_ref()) { - return Ok(()); + if is_directory { + if ignore_contains.iter().any(|name| path.join(name).exists()) { + return Ok(pi_walker::WalkDecision::SkipDescend); + } + if config.prune + && config + .matcher + .matches(&match_target(path, &config.base_dir, config.full_path)) + { + return Ok(pi_walker::WalkDecision::SkipDescend); + } + } + + let metadata = fs::symlink_metadata(path).ok(); + if !matches_walker_filters(config, path, file_type, metadata.as_ref()) { + return Ok(pi_walker::WalkDecision::Skip); } let target = match_target(path, &config.base_dir, config.full_path); if !config.matcher.matches(&target) { - return Ok(()); + return Ok(pi_walker::WalkDecision::Skip); } - state.matches = state.matches.saturating_add(1); + *matches = (*matches).saturating_add(1); if config.quiet { - return Ok(()); + return Ok(pi_walker::WalkDecision::Include); } let display = display_path(config, path); let text = if let Some(format) = config.format.as_deref() { @@ -709,14 +925,19 @@ fn process_entry( } else { out.write_all(b"\n")?; } - Ok(()) + Ok(pi_walker::WalkDecision::Include) } -fn matches_filters(config: &SearchConfig, entry: &DirEntry, metadata: Option<&Metadata>) -> bool { - if !matches_type_filter(&config.types, entry, metadata) { +fn matches_walker_filters( + config: &SearchConfig, + path: &Path, + file_type: pi_walker::FileType, + metadata: Option<&Metadata>, +) -> bool { + if !matches_walker_type_filter(&config.types, path, file_type, metadata) { return false; } - if !config.extensions.is_empty() && !matches_extension(entry.path(), &config.extensions) { + if !config.extensions.is_empty() && !matches_extension(path, &config.extensions) { return false; } if !config.sizes.is_empty() && !matches_size_filters(&config.sizes, metadata) { @@ -733,20 +954,19 @@ fn matches_filters(config: &SearchConfig, entry: &DirEntry, metadata: Option<&Me true } -fn matches_type_filter(filter: &TypeFilter, entry: &DirEntry, metadata: Option<&Metadata>) -> bool { +fn matches_walker_type_filter( + filter: &TypeFilter, + path: &Path, + file_type: pi_walker::FileType, + metadata: Option<&Metadata>, +) -> bool { if filter.is_empty() { return true; } - let path = entry.path(); - let is_symlink = fs::symlink_metadata(path).is_ok_and(|meta| meta.file_type().is_symlink()); - let file_type = entry.file_type(); let kind_matches = if filter.has_kind() { - file_type.is_some_and(|file_type| { - (filter.regular && file_type.is_file()) - || (filter.directory && file_type.is_dir()) - || (filter.symlink && is_symlink) - || matches_unix_file_type(filter, file_type) - }) + (filter.regular && file_type == pi_walker::FileType::File) + || (filter.directory && file_type == pi_walker::FileType::Dir) + || (filter.symlink && file_type == pi_walker::FileType::Symlink) } else { true }; @@ -762,18 +982,104 @@ fn matches_type_filter(filter: &TypeFilter, entry: &DirEntry, metadata: Option<& true } -#[cfg(unix)] -fn matches_unix_file_type(filter: &TypeFilter, file_type: fs::FileType) -> bool { - use std::os::unix::fs::FileTypeExt; - (filter.socket && file_type.is_socket()) - || (filter.pipe && file_type.is_fifo()) - || (filter.block && file_type.is_block_device()) - || (filter.character && file_type.is_char_device()) +fn walker_error_to_io(err: pi_walker::WalkError) -> io::Error { + match err { + pi_walker::WalkError::Unsupported => io::Error::new( + io::ErrorKind::Unsupported, + "native fd traversal is unsupported for this platform or option set", + ), + pi_walker::WalkError::Interrupted(err) => err, + pi_walker::WalkError::InvalidData { path, message } => { + io::Error::other(format!("{}: {message}", path.display())) + }, + } } -#[cfg(not(unix))] -fn matches_unix_file_type(_filter: &TypeFilter, _file_type: fs::FileType) -> bool { - false +fn walker_collect_error_to_io(err: pi_walker::WalkError) -> io::Error { + match err { + pi_walker::WalkError::Unsupported => io::Error::new( + io::ErrorKind::Unsupported, + "native fd traversal is unsupported for this platform or option set", + ), + pi_walker::WalkError::Interrupted(err) => io::Error::other(err), + pi_walker::WalkError::InvalidData { path, message } => { + io::Error::other(format!("{}: {message}", path.display())) + }, + } +} + +fn process_collected_entry( + config: &SearchConfig, + ignore_contains: &[OsString], + fd_ignores: &mut FdIgnoreMatcher, + root: &Path, + entry: &CollectedEntry, + out: &mut W, + state: &mut SearchState, + pruned_dirs: &mut Vec, +) -> io::Result<()> { + let path = entry.absolute_path(root); + if pruned_dirs.iter().any(|dir| path.starts_with(dir)) { + return Ok(()); + } + let depth = entry.depth(); + let is_directory = entry.file_type == pi_walker::FileType::Dir; + if depth == 0 && is_directory { + return Ok(()); + } + if fd_ignores.is_ignored(&path, is_directory) { + if is_directory { + pruned_dirs.push(path); + } + return Ok(()); + } + if config.excludes.matches(&path, &config.base_dir) { + if is_directory { + pruned_dirs.push(path); + } + return Ok(()); + } + if is_directory { + if ignore_contains.iter().any(|name| path.join(name).exists()) { + pruned_dirs.push(path); + return Ok(()); + } + if config.prune + && config + .matcher + .matches(&match_target(&path, &config.base_dir, config.full_path)) + { + pruned_dirs.push(path); + return Ok(()); + } + } + + let metadata = fs::symlink_metadata(&path).ok(); + if !matches_walker_filters(config, &path, entry.file_type, metadata.as_ref()) { + return Ok(()); + } + let target = match_target(&path, &config.base_dir, config.full_path); + if !config.matcher.matches(&target) { + return Ok(()); + } + + state.matches = state.matches.saturating_add(1); + if config.quiet { + return Ok(()); + } + let display = display_path(config, &path); + let text = if let Some(format) = config.format.as_deref() { + format_path(format, &path, &display) + } else { + display + }; + out.write_all(text.as_bytes())?; + if config.print0 { + out.write_all(b"\0")?; + } else { + out.write_all(b"\n")?; + } + Ok(()) } #[cfg(unix)] diff --git a/crates/pi-shell/src/process.rs b/crates/pi-shell/src/process.rs index 5664727d4..48a24a6f3 100644 --- a/crates/pi-shell/src/process.rs +++ b/crates/pi-shell/src/process.rs @@ -1511,8 +1511,8 @@ async fn wait_for_exit( pub fn kill_process_group(pgid: i32, signal: i32) -> bool { // Defense in depth: refuse to deliver a signal to the harness's own // process group. Doing so terminates the harness along with the targets. - // Higher layers (`add_new_descendants`) already filter pgids by descendant - // ownership; this catches any future caller that bypasses that filter. + // `SpawnRegistry` only ever records pgids brush created for this run (never + // the harness pgid); this catches any future caller that bypasses it. if pgid <= 0 || is_self_process_group(pgid) { return false; } @@ -1597,187 +1597,101 @@ impl TerminationTargets { } } -#[must_use] -pub fn current_descendant_pids() -> HashSet { - Process::from_pid(i32::try_from(std::process::id()).unwrap_or_default()).map_or_else( - HashSet::new, - |process| { - process - .live_descendants() - .into_iter() - .map(|child| child.pid()) - .collect() - }, - ) -} - -pub fn add_new_descendants( - targets: &mut TerminationTargets, - baseline: &HashSet, -) { - let self_pid = i32::try_from(std::process::id()).unwrap_or_default(); - let Some(process) = Process::from_pid(self_pid) else { - return; - }; - let descendants = process.live_descendants(); - let descendants_info: Vec = descendants - .iter() - .map(|child| DescendantInfo { pid: child.pid(), pgid: child.group_id() }) - .collect(); - - let selection = select_termination_targets(&descendants_info, baseline); - for pgid in selection.pgids { - targets.add_pgid(pgid); - } - for pid in selection.pids { - targets.add_pid(pid); - } -} - -/// Light view of a descendant for target classification — just enough to -/// decide which pgids/pids belong in the kill set without holding any -/// platform-specific process handles. +/// A single external child reported by the shell's spawn-observer hook. #[derive(Debug, Clone, Copy)] -struct DescendantInfo { +struct SpawnedProcess { pid: i32, pgid: Option, } -/// Classified termination targets returned by [`select_termination_targets`]. -#[derive(Debug, Default)] -struct TargetSelection { - pgids: Vec, - pids: Vec, +/// Per-run record of the OS processes a single shell command launched, +/// captured at spawn time via brush's `SpawnObserver` hook. +/// +/// Replaces the old process-global "new descendants since a baseline" diff, +/// which could not distinguish the children of concurrent runs sharing one +/// host process: a run that cancelled would signal *any* descendant spawned +/// after its baseline, including another run's children. Ownership is now +/// explicit — only processes this run actually spawned are ever signalled. +#[derive(Default)] +pub struct SpawnRegistry { + spawned: std::sync::Mutex>, } -/// Pure target-classifier separated from process discovery so it is testable -/// without depending on the platform's process-listing primitives (libproc on -/// macOS, `/proc` on Linux). -/// -/// **Critical**: a `pgid` is only adopted when its leader is itself one of the -/// new descendants. Without that check, a descendant that inherited the -/// harness's pgid — any subprocess started via APIs that do not call `setpgid`, -/// such as a sibling LSP/MCP helper spawned outside of brush — would drag -/// `harness.pgid` into the kill set, and the subsequent -/// `kill(-harness.pgid, SIGTERM)` would terminate the harness alongside the -/// intended targets. Pids of new descendants are still tracked individually so -/// the descendant tree can be reaped via `signal_tree`. -fn select_termination_targets( - descendants: &[DescendantInfo], - baseline: &HashSet, -) -> TargetSelection { - let new_descendant_pids: HashSet = descendants - .iter() - .map(|info| info.pid) - .filter(|pid| !baseline.contains(pid)) - .collect(); - - let mut selection = TargetSelection::default(); - let mut seen_pgids: HashSet = HashSet::new(); - for info in descendants { - if !new_descendant_pids.contains(&info.pid) { - continue; - } - if let Some(pgid) = info.pgid - && pgid > 0 - && new_descendant_pids.contains(&pgid) - && seen_pgids.insert(pgid) - { - selection.pgids.push(pgid); - } - selection.pids.push(info.pid); +impl SpawnRegistry { + /// Create an empty registry. + #[must_use] + pub fn new() -> Self { + Self::default() } - selection + + /// Record a freshly spawned child. Called from the spawn-observer hook. + pub fn record(&self, pid: i32, pgid: Option) { + self + .spawned + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .push(SpawnedProcess { pid, pgid }); + } + + /// Build the kill set from the processes recorded so far. Re-read on every + /// signal wave so a child spawned during a grace window — between the + /// cancel firing and the next wave — is still reaped. + /// + /// A recorded pid contributes only while alive (`add_pid` opens a stable + /// handle, skipping the dead); a recorded pgid contributes only while the + /// group still has members, so once the run's whole tree exits the targets + /// are empty and the wave loop can stop early. + #[must_use] + pub fn build_targets(&self) -> TerminationTargets { + let mut targets = TerminationTargets::new(); + let spawned = self + .spawned + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .clone(); + for entry in spawned { + targets.add_pid(entry.pid); + if let Some(pgid) = entry.pgid + && pgid > 0 + && process_group_alive(pgid) + { + targets.add_pgid(pgid); + } + } + targets + } +} + +/// True when process group `pgid` still has at least one member. `kill(2)` +/// with signal 0 performs permission/existence checks without delivering a +/// signal; `EPERM` means the group exists but is not ours to signal, which +/// still counts as alive. +#[must_use] +fn process_group_alive(pgid: i32) -> bool { + if pgid <= 0 { + return false; + } + platform_process_group_alive(pgid) +} + +#[cfg(unix)] +fn platform_process_group_alive(pgid: i32) -> bool { + // SAFETY: `kill` takes integer identifiers by value and does not access + // caller-owned memory. A negative pid targets the process group; signal 0 + // only runs the existence/permission checks. + let ret = unsafe { libc::kill(-pgid, 0) }; + ret == 0 || std::io::Error::last_os_error().raw_os_error() == Some(libc::EPERM) +} + +#[cfg(not(unix))] +const fn platform_process_group_alive(_pgid: i32) -> bool { + false } #[cfg(test)] mod tests { use super::*; - /// Regression test for the cancellation-kills-harness bug. - /// - /// When the descendant walk harvested each descendant's `pgid` and pushed - /// it onto the kill list, a descendant that inherited the harness's pgid - /// — any subprocess started via APIs that do not call `setpgid`, such as a - /// sibling LSP/MCP helper — dragged `harness.pgid` into the kill set, and - /// the subsequent `kill(-harness.pgid, SIGTERM)` killed the harness. - /// - /// Encode the dangerous shape directly: a new descendant whose `pgid` - /// resolves to something the harness owns (not in the new descendant set) - /// must contribute its pid for individual cleanup but **must not** drag its - /// pgid into the group-signal list. - #[test] - fn select_targets_drops_inherited_harness_pgid() { - const HARNESS_PGID: i32 = 1000; - const BASELINE_HELPER_PID: i32 = 1500; - - // Harness pgid is *not* a new descendant; a baseline helper happens to - // lead a group that a new descendant inherited. Neither pgid is safe to - // signal as a group. - let descendants = [DescendantInfo { pid: 2000, pgid: Some(HARNESS_PGID) }, DescendantInfo { - pid: 2001, - pgid: Some(BASELINE_HELPER_PID), - }]; - let baseline: HashSet = std::iter::once(BASELINE_HELPER_PID).collect(); - - let selection = select_termination_targets(&descendants, &baseline); - - assert!( - selection.pgids.is_empty(), - "no pgid should be added when leaders live outside the new descendant set; got {:?}", - selection.pgids, - ); - assert_eq!( - selection.pids, - vec![2000, 2001], - "new descendant pids must still be tracked individually for tree cleanup", - ); - } - - #[test] - fn select_targets_adopts_owned_process_group() { - // A new descendant that *is* the group leader — brush's `NewProcessGroup` - // path — contributes both its pid and its pgid, so grandchildren in the - // same group get reaped in one signal wave. - let leader = DescendantInfo { pid: 3000, pgid: Some(3000) }; - let grandchild = DescendantInfo { pid: 3001, pgid: Some(3000) }; - let baseline: HashSet = HashSet::new(); - - let selection = select_termination_targets(&[leader, grandchild], &baseline); - - assert_eq!(selection.pgids, vec![3000]); - assert_eq!(selection.pids, vec![3000, 3001]); - } - - #[test] - fn select_targets_skips_baseline_descendants() { - let old = DescendantInfo { pid: 4000, pgid: Some(4000) }; - let fresh = DescendantInfo { pid: 4100, pgid: Some(4100) }; - let baseline: HashSet = std::iter::once(4000).collect(); - - let selection = select_termination_targets(&[old, fresh], &baseline); - - assert_eq!(selection.pgids, vec![4100]); - assert_eq!(selection.pids, vec![4100]); - } - - #[test] - fn select_targets_dedupes_shared_process_group() { - let a = DescendantInfo { pid: 5000, pgid: Some(5000) }; - let b = DescendantInfo { pid: 5001, pgid: Some(5000) }; - let c = DescendantInfo { pid: 5002, pgid: Some(5000) }; - let baseline: HashSet = HashSet::new(); - - let selection = select_termination_targets(&[a, b, c], &baseline); - - assert_eq!( - selection.pgids, - vec![5000], - "each pgid should be recorded exactly once even when many descendants share it", - ); - assert_eq!(selection.pids, vec![5000, 5001, 5002]); - } - /// `kill_process_group` is the last line of defense: even if a future /// caller manages to feed the harness's own pgid into the signal path, /// this wrapper must refuse to deliver the signal. diff --git a/crates/pi-shell/src/shell.rs b/crates/pi-shell/src/shell.rs index d6ed6a59c..4b02f3dde 100644 --- a/crates/pi-shell/src/shell.rs +++ b/crates/pi-shell/src/shell.rs @@ -1,7 +1,9 @@ //! Runtime-agnostic brush shell execution. +#[cfg(windows)] +use std::collections::HashSet; use std::{ - collections::{HashMap, HashSet}, + collections::HashMap, fs, io::{self, Write}, str, @@ -14,7 +16,7 @@ use brush_builtins::{BuiltinSet, default_builtins}; use brush_core::{ ExecutionContext, ExecutionControlFlow, ExecutionExitCode, ExecutionParameters, ExecutionResult, ProcessGroupPolicy, ProfileLoadBehavior, RcLoadBehavior, Shell as BrushShell, ShellValue, - ShellVariable, SourceInfo, builtins, + ShellVariable, SourceInfo, SpawnObserver, builtins, env::EnvironmentScope, openfiles::{self, OpenFile, OpenFiles}, }; @@ -269,22 +271,38 @@ async fn run_shell_session( ct: &mut CancelToken, ) -> Result { let tokio_cancel = CancellationToken::new(); - let baseline_descendants = process::current_descendant_pids(); + let spawn_registry = Arc::new(process::SpawnRegistry::new()); + let process_cancel_bridge = tokio::spawn({ + let tokio_cancel = tokio_cancel.clone(); + let spawn_registry = spawn_registry.clone(); + async move { + tokio_cancel.cancelled().await; + terminate_run(&spawn_registry).await; + } + }); let mut run_task = tokio::spawn({ let session = session.clone(); let abort_state = abort_state.clone(); let tokio_cancel = tokio_cancel.clone(); let at = ct.emplace_abort_token(); + let spawn_registry = spawn_registry.clone(); async move { let mut session_guard = session.lock().await; let session = match &mut *session_guard { Some(session) => session, - None => session_guard.insert(create_session(&config).await?), + None => session_guard.insert( + create_session_for_run( + &config, + Some(spawn_registry.clone()), + Some(tokio_cancel.clone()), + ) + .await?, + ), }; abort_state.set(at).await; - run_shell_command(session, &run_config, on_chunk, tokio_cancel).await + run_shell_command(session, &run_config, on_chunk, tokio_cancel, spawn_registry).await } }); @@ -292,7 +310,6 @@ async fn run_shell_session( res = &mut run_task => res, reason = ct.wait() => { tokio_cancel.cancel(); - terminate_new_descendants(&baseline_descendants).await; let graceful = time::timeout(Duration::from_secs(2), &mut run_task).await; if graceful.is_err() { run_task.abort(); @@ -305,6 +322,7 @@ async fn run_shell_session( if let Ok(mut guard) = session.try_lock() { *guard = None; } + let _ = process_cancel_bridge.await; return Ok(ShellRunResult { exit_code: None, cancelled: matches!(reason, AbortReason::Signal), @@ -315,6 +333,8 @@ async fn run_shell_session( }; let res = res.unwrap_or_else(|err| Err(Error::msg(format!("Shell execution task failed: {err}")))); + process_cancel_bridge.abort(); + let _ = process_cancel_bridge.await; abort_state.clear().await; let keepalive = res.as_ref().is_ok_and(|pair| session_keepalive(&pair.0)); @@ -337,13 +357,27 @@ async fn run_shell_oneshot( ct: CancelToken, ) -> Result { let tokio_cancel = CancellationToken::new(); - let baseline_descendants = process::current_descendant_pids(); + let spawn_registry = Arc::new(process::SpawnRegistry::new()); + let process_cancel_bridge = tokio::spawn({ + let tokio_cancel = tokio_cancel.clone(); + let spawn_registry = spawn_registry.clone(); + async move { + tokio_cancel.cancelled().await; + terminate_run(&spawn_registry).await; + } + }); let mut task = tokio::spawn({ let tokio_cancel = tokio_cancel.clone(); + let spawn_registry = spawn_registry.clone(); async move { - let mut session = create_session(&config).await?; - run_shell_command(&mut session, &run_config, on_chunk, tokio_cancel).await + let mut session = create_session_for_run( + &config, + Some(spawn_registry.clone()), + Some(tokio_cancel.clone()), + ) + .await?; + run_shell_command(&mut session, &run_config, on_chunk, tokio_cancel, spawn_registry).await } }); @@ -351,12 +385,12 @@ async fn run_shell_oneshot( result = &mut task => result, reason = ct.wait() => { tokio_cancel.cancel(); - terminate_new_descendants(&baseline_descendants).await; let graceful = time::timeout(Duration::from_secs(2), &mut task).await; if graceful.is_err() { task.abort(); let _ = task.await; } + let _ = process_cancel_bridge.await; return Ok(ShellExecuteResult { exit_code: None, cancelled: matches!(reason, AbortReason::Signal), @@ -366,6 +400,8 @@ async fn run_shell_oneshot( }, }; + process_cancel_bridge.abort(); + let _ = process_cancel_bridge.await; let res = run_result .unwrap_or_else(|err| Err(Error::msg(format!("Shell execution task failed: {err}")))); let (exec, minimized) = res?; @@ -384,13 +420,28 @@ async fn run_shell_oneshot_streams( ct: CancelToken, ) -> Result { let tokio_cancel = CancellationToken::new(); - let baseline_descendants = process::current_descendant_pids(); + let spawn_registry = Arc::new(process::SpawnRegistry::new()); + let process_cancel_bridge = tokio::spawn({ + let tokio_cancel = tokio_cancel.clone(); + let spawn_registry = spawn_registry.clone(); + async move { + tokio_cancel.cancelled().await; + terminate_run(&spawn_registry).await; + } + }); let mut task = tokio::spawn({ let tokio_cancel = tokio_cancel.clone(); + let spawn_registry = spawn_registry.clone(); async move { - let mut session = create_session(&config).await?; - run_shell_command_streams(&mut session, &run_config, streams, tokio_cancel).await + let mut session = create_session_for_run( + &config, + Some(spawn_registry.clone()), + Some(tokio_cancel.clone()), + ) + .await?; + run_shell_command_streams(&mut session, &run_config, streams, tokio_cancel, spawn_registry) + .await } }); @@ -398,12 +449,12 @@ async fn run_shell_oneshot_streams( result = &mut task => result, reason = ct.wait() => { tokio_cancel.cancel(); - terminate_new_descendants(&baseline_descendants).await; let graceful = time::timeout(Duration::from_secs(2), &mut task).await; if graceful.is_err() { task.abort(); let _ = task.await; } + let _ = process_cancel_bridge.await; return Ok(ShellExecuteResult { exit_code: None, cancelled: matches!(reason, AbortReason::Signal), @@ -413,6 +464,8 @@ async fn run_shell_oneshot_streams( }, }; + process_cancel_bridge.abort(); + let _ = process_cancel_bridge.await; let res = run_result .unwrap_or_else(|err| Err(Error::msg(format!("Shell execution task failed: {err}")))); let exec = res?; @@ -501,7 +554,16 @@ fn merge_path_values(_existing: &str, incoming: &str) -> String { incoming.to_string() } +#[cfg(test)] async fn create_session(config: &ShellConfig) -> Result { + create_session_for_run(config, None, None).await +} + +async fn create_session_for_run( + config: &ShellConfig, + spawn_registry: Option>, + cancel_token: Option, +) -> Result { let mut shell = BrushShell::builder() .do_not_inherit_env(true) .profile(ProfileLoadBehavior::Skip) @@ -625,18 +687,29 @@ async fn create_session(config: &ShellConfig) -> Result { configure_windows_path(&mut shell)?; if let Some(snapshot_path) = config.snapshot_path.as_ref() { - source_snapshot(&mut shell, snapshot_path).await?; + source_snapshot(&mut shell, snapshot_path, spawn_registry, cancel_token).await?; } Ok(ShellSessionCore { shell }) } -async fn source_snapshot(shell: &mut BrushShell, snapshot_path: &str) -> Result<()> { +async fn source_snapshot( + shell: &mut BrushShell, + snapshot_path: &str, + spawn_registry: Option>, + cancel_token: Option, +) -> Result<()> { let mut params = shell.default_exec_params(); let source_info = SourceInfo::from("pi-natives:snapshot"); params.set_fd(OpenFiles::STDIN_FD, null_file()?); params.set_fd(OpenFiles::STDOUT_FD, null_file()?); params.set_fd(OpenFiles::STDERR_FD, null_file()?); + if let Some(cancel_token) = cancel_token { + params.set_cancel_token(cancel_token); + } + if let Some(spawn_registry) = spawn_registry { + params.set_spawn_observer(spawn_registry); + } let escaped = snapshot_path.replace('\'', "'\\''"); let command = format!("source '{escaped}'"); @@ -688,6 +761,7 @@ async fn run_shell_command( options: &ShellRunConfig, on_chunk: Option>, cancel_token: CancellationToken, + spawn_registry: Arc, ) -> Result<(ExecutionResult, Option)> { if let Some(cwd) = options.cwd.as_deref() { session @@ -706,10 +780,19 @@ async fn run_shell_command( let result = match minimizer_mode { minimizer::engine::MinimizerMode::SegmentedChain => { - run_shell_command_segmented_chain(session, options, on_chunk, cancel_token).await + run_shell_command_segmented_chain(session, options, on_chunk, cancel_token, spawn_registry) + .await }, minimizer::engine::MinimizerMode::WholeCommand | minimizer::engine::MinimizerMode::None => { - run_shell_command_single(session, options, on_chunk, cancel_token, minimizer_mode).await + run_shell_command_single( + session, + options, + on_chunk, + cancel_token, + spawn_registry, + minimizer_mode, + ) + .await }, }; @@ -729,6 +812,7 @@ async fn run_shell_command_single( options: &ShellRunConfig, on_chunk: Option>, cancel_token: CancellationToken, + spawn_registry: Arc, minimizer_mode: minimizer::engine::MinimizerMode, ) -> Result<(ExecutionResult, Option)> { debug_assert!(!matches!(minimizer_mode, minimizer::engine::MinimizerMode::SegmentedChain)); @@ -751,6 +835,7 @@ async fn run_shell_command_single( params, on_chunk, cancel_token, + spawn_registry, capture_mode, ) .await?; @@ -810,6 +895,7 @@ async fn run_shell_command_segmented_chain( options: &ShellRunConfig, on_chunk: Option>, cancel_token: CancellationToken, + spawn_registry: Arc, ) -> Result<(ExecutionResult, Option)> { let Some(config) = options.minimizer.as_ref() else { return run_shell_command_single( @@ -817,6 +903,7 @@ async fn run_shell_command_segmented_chain( options, on_chunk, cancel_token, + spawn_registry, minimizer::engine::MinimizerMode::None, ) .await; @@ -829,6 +916,7 @@ async fn run_shell_command_segmented_chain( options, on_chunk, cancel_token, + spawn_registry, minimizer::engine::MinimizerMode::None, ) .await; @@ -842,6 +930,7 @@ async fn run_shell_command_segmented_chain( options, on_chunk, cancel_token, + spawn_registry, minimizer::engine::MinimizerMode::None, ) .await; @@ -871,6 +960,7 @@ async fn run_shell_command_segmented_chain( segment_params, on_chunk.clone(), cancel_token.clone(), + spawn_registry.clone(), capture_mode, ) .await?; @@ -947,6 +1037,7 @@ async fn run_shell_command_once( mut params: ExecutionParameters, on_chunk: Option>, cancel_token: CancellationToken, + spawn_registry: Arc, capture_mode: CommandCaptureMode, ) -> Result { let (reader_file, writer_file) = pipe_to_files("output")?; @@ -963,7 +1054,7 @@ async fn run_shell_command_once( params.set_fd(OpenFiles::STDERR_FD, stderr_file); params.process_group_policy = ProcessGroupPolicy::NewProcessGroup; params.set_cancel_token(cancel_token.clone()); - let baseline_descendants = process::current_descendant_pids(); + params.set_spawn_observer(spawn_registry.clone()); let reader_cancel = CancellationToken::new(); let (activity_tx, mut activity_rx) = mpsc::channel::<()>(1); let reader_callback = on_chunk; @@ -998,14 +1089,6 @@ async fn run_shell_command_once( reader_cancel.cancel(); } }); - let process_cancel_bridge = tokio::spawn({ - let cancel_token = cancel_token.clone(); - let baseline_descendants = baseline_descendants.clone(); - async move { - cancel_token.cancelled().await; - terminate_new_descendants(&baseline_descendants).await; - } - }); ensure_trailing_newline_for_heredoc(&mut command); let source_info = SourceInfo::from("pi-natives:command"); let result = session @@ -1066,17 +1149,6 @@ async fn run_shell_command_once( } cancel_bridge.abort(); let _ = cancel_bridge.await; - if cancel_token.is_cancelled() { - // Cancel fired — the bridge is actively running its rescan-and-signal - // loop. Let it run to completion so all three waves get a chance to - // reach stragglers; aborting here would cut the kill loop short. - let _ = process_cancel_bridge.await; - } else { - // Happy path — the bridge is still parked on `cancel_token.cancelled()` - // and would never exit on its own. Tear it down. - process_cancel_bridge.abort(); - let _ = process_cancel_bridge.await; - } let result = result.map_err(|err| Error::msg(format!("Shell execution failed: {err}")))?; let buffered = match reader_output { @@ -1091,6 +1163,7 @@ async fn run_shell_command_streams( options: &ShellRunConfig, streams: StreamSinks, cancel_token: CancellationToken, + spawn_registry: Arc, ) -> Result { if let Some(cwd) = options.cwd.as_deref() { session @@ -1113,7 +1186,7 @@ async fn run_shell_command_streams( params.set_fd(OpenFiles::STDERR_FD, stderr_file); params.process_group_policy = ProcessGroupPolicy::NewProcessGroup; params.set_cancel_token(cancel_token.clone()); - let baseline_descendants = process::current_descendant_pids(); + params.set_spawn_observer(spawn_registry.clone()); let reader_cancel = CancellationToken::new(); let (activity_tx, mut activity_rx) = mpsc::channel::<()>(1); @@ -1139,35 +1212,6 @@ async fn run_shell_command_streams( reader_cancel.cancel(); } }); - let process_cancel_bridge = tokio::spawn({ - let cancel_token = cancel_token.clone(); - let baseline_descendants = baseline_descendants.clone(); - async move { - cancel_token.cancelled().await; - const WAVES: u32 = 3; - for wave in 0..WAVES { - let mut targets = process::TerminationTargets::new(); - process::add_new_descendants(&mut targets, &baseline_descendants); - if targets.is_empty() { - return; - } - let signal = if wave == 0 { - process::TERM_SIGNAL - } else { - process::KILL_SIGNAL - }; - targets.signal(signal); - if wave + 1 < WAVES { - let pause = if wave == 0 { - Duration::from_millis(75) - } else { - Duration::from_millis(150) - }; - time::sleep(pause).await; - } - } - } - }); let mut command = options.command.clone(); ensure_trailing_newline_for_heredoc(&mut command); let source_info = SourceInfo::from("pi-shell:streams"); @@ -1244,14 +1288,6 @@ async fn run_shell_command_streams( } cancel_bridge.abort(); let _ = cancel_bridge.await; - if cancel_token.is_cancelled() { - // Let the kill-wave bridge finish all three signal passes so stragglers - // have a chance to receive SIGKILL. - let _ = process_cancel_bridge.await; - } else { - process_cancel_bridge.abort(); - let _ = process_cancel_bridge.await; - } let result = result.map_err(|err| Error::msg(format!("Shell execution failed: {err}")))?; Ok(result) @@ -1315,23 +1351,37 @@ async fn read_output_bytes( } } -// Rescan-and-signal loop for cancellation. Each pass picks up descendants -// spawned during the previous wave's grace period, then exits as soon as no -// targets remain so unrelated later commands are not swept into old cancels. -async fn terminate_new_descendants(baseline: &HashSet) { +impl SpawnObserver for process::SpawnRegistry { + fn on_spawn(&self, pid: i32, pgid: Option) { + self.record(pid, pgid); + } +} + +// Escalating TERM -> KILL waves over the processes this run spawned, scoped via +// the per-run `SpawnRegistry`. The kill set is rebuilt each wave so a child +// spawned in a grace window — or a grandchild whose recorded parent already +// exited but whose process group is still live — is still reaped, and the loop +// stops as soon as the run's whole tree is gone. Scoping to the registry (vs a +// process-global descendant diff) is what keeps a cancel from reaping a +// concurrent run's children in a shared host process. +async fn terminate_run(registry: &process::SpawnRegistry) { const WAVES: u32 = 3; + let mut saw_targets = false; for wave in 0..WAVES { - let mut targets = process::TerminationTargets::new(); - process::add_new_descendants(&mut targets, baseline); + let targets = registry.build_targets(); if targets.is_empty() { - return; - } - let signal = if wave == 0 { - process::TERM_SIGNAL + if saw_targets || wave + 1 == WAVES { + return; + } } else { - process::KILL_SIGNAL - }; - targets.signal(signal); + saw_targets = true; + let signal = if wave == 0 { + process::TERM_SIGNAL + } else { + process::KILL_SIGNAL + }; + targets.signal(signal); + } if wave + 1 < WAVES { let pause = if wave == 0 { Duration::from_millis(75) @@ -3416,6 +3466,196 @@ replace = [{ pattern = "^.+$", replacement = "PWD" }] ); } + /// Cancelling one `Shell::run` must only signal processes spawned by that + /// run. Run B starts first so its old host-descendant baseline would not + /// include run A's later-spawned child; pre-fix, cancelling B classified A's + /// child as "new" and SIGTERM'd it, so run A returned 143 instead of 0. + #[cfg(unix)] + #[tokio::test(flavor = "multi_thread")] + async fn cancelling_one_run_spares_a_concurrent_runs_child() { + let _guard = shell_test_lock().lock().await; + + let shell_b = Shell::new(None); + let (tx_b, mut rx_b) = mpsc::unbounded_channel::(); + let mut ct_b = CancelToken::default(); + let abort_b = ct_b.emplace_abort_token(); + let handle_b = tokio::spawn(async move { + shell_b + .run( + ShellRunOptions { + command: "/bin/sh -c 'printf \"ready\\n\"; sleep 30'".into(), + ..Default::default() + }, + Some(tx_b), + ct_b, + ) + .await + }); + + let mut b_output = String::new(); + let b_ready = time::timeout(Duration::from_secs(5), async { + loop { + let chunk = rx_b + .recv() + .await + .expect("run B ended before printing readiness"); + b_output.push_str(&chunk); + if let Some(line_end) = b_output.find('\n') { + return b_output[..line_end].to_string(); + } + } + }) + .await + .expect("timed out waiting for run B readiness"); + assert_eq!(b_ready.trim(), "ready", "run B should reach its long sleep before run A starts"); + + let shell_a = Shell::new(None); + let (tx_a, mut rx_a) = mpsc::unbounded_channel::(); + let handle_a = tokio::spawn(async move { + shell_a + .run( + ShellRunOptions { + command: "/bin/sh -c 'printf \"%d\\n\" \"$$\"; sleep 2'".into(), + ..Default::default() + }, + Some(tx_a), + CancelToken::default(), + ) + .await + }); + + let mut a_output = String::new(); + let a_child_pid = time::timeout(Duration::from_secs(5), async { + loop { + let chunk = rx_a + .recv() + .await + .expect("run A ended before printing its child pid"); + a_output.push_str(&chunk); + if let Some(line_end) = a_output.find('\n') { + return a_output[..line_end] + .trim() + .parse::() + .expect("run A pid line should be an integer"); + } + } + }) + .await + .expect("timed out waiting for run A child pid"); + assert!(a_child_pid > 0, "got non-positive run A child pid: {a_child_pid}"); + + abort_b.abort(AbortReason::Signal); + + let result_a = time::timeout(Duration::from_secs(10), handle_a) + .await + .expect("run A timed out") + .expect("run A task panicked") + .expect("run A failed"); + let result_b = time::timeout(Duration::from_secs(10), handle_b) + .await + .expect("run B timed out") + .expect("run B task panicked") + .expect("run B failed"); + + assert_eq!(result_a.exit_code, Some(0), "cancelling run B must not SIGTERM run A's child"); + assert!(!result_a.cancelled, "run A was never cancelled"); + assert!(result_b.cancelled, "run B should report cancellation"); + } + + /// Cancelling while `Shell::run` is still sourcing a snapshot must terminate + /// the foreground process spawned by that snapshot. The snapshot runs before + /// the user command, so this specifically guards the shared cancel token and + /// spawn registry wiring passed into `source_snapshot`. + #[cfg(unix)] + #[tokio::test(flavor = "multi_thread")] + async fn cancelling_while_sourcing_snapshot_kills_snapshot_foreground_child() { + let _guard = shell_test_lock().lock().await; + let root = unique_temp_dir("snapshot-cancel"); + let snapshot_path = root.join("snapshot.sh"); + let pid_path = root.join("snapshot-child.pid"); + let escaped_pid_path = pid_path.to_string_lossy().replace('\'', "'\\''"); + std::fs::write( + &snapshot_path, + format!( + "/bin/sh -c 'printf \"%d\\n\" \"$$\" > \"$1\"; sleep 30' sh '{escaped_pid_path}'\n" + ), + ) + .expect("write snapshot file"); + + let shell = Shell::new(Some(ShellOptions { + snapshot_path: Some(snapshot_path.to_string_lossy().into_owned()), + ..Default::default() + })); + let mut cancel_token = CancelToken::default(); + let abort_token = cancel_token.emplace_abort_token(); + let run_handle = tokio::spawn(async move { + shell + .run( + ShellRunOptions { command: "printf done".into(), ..Default::default() }, + None, + cancel_token, + ) + .await + }); + + let child_pid = time::timeout(Duration::from_secs(5), async { + loop { + if let Ok(pid_text) = std::fs::read_to_string(&pid_path) + && let Ok(pid) = pid_text.trim().parse::() + && pid > 0 + { + return pid; + } + time::sleep(Duration::from_millis(20)).await; + } + }) + .await + .expect("timed out waiting for snapshot foreground child to write its positive PID"); + + abort_token.abort(AbortReason::Signal); + + let result = time::timeout(Duration::from_secs(10), run_handle) + .await + .expect("timed out waiting for cancellation while sourcing snapshot") + .expect("snapshot sourcing run task panicked") + .expect("shell run failed while cancelling snapshot sourcing"); + assert!(result.cancelled, "cancelling while sourcing a snapshot should report cancellation"); + assert_eq!( + result.exit_code, None, + "cancelled snapshot sourcing run should not report an exit code" + ); + assert!( + !result.timed_out, + "signal cancellation during snapshot sourcing must not report timeout" + ); + + let child_dead = time::timeout(Duration::from_secs(5), async { + loop { + // SAFETY: `child_pid` came from the foreground `/bin/sh` spawned by the + // snapshot; `kill(pid, 0)` only probes whether that process still exists. + let kill_result = unsafe { libc::kill(child_pid, 0) }; + if kill_result == -1 { + let err = std::io::Error::last_os_error(); + if err.raw_os_error() == Some(libc::ESRCH) { + return; + } + panic!( + "kill({child_pid}, 0) failed with unexpected error while checking snapshot \ + child cleanup: {err}" + ); + } + time::sleep(Duration::from_millis(20)).await; + } + }) + .await; + let _ = std::fs::remove_dir_all(&root); + assert!( + child_dead.is_ok(), + "snapshot foreground child PID {child_pid} was still alive after cancelling while \ + sourcing snapshot; cancel bridge did not terminate the snapshot-spawned process" + ); + } + /// Regression for the `suspended (tty input)` bug: an **interactive child /// inside a pipeline** (`zsh -i ... | awk`) used to stay in the host /// session, open `/dev/tty`, `tcsetpgrp` itself to the foreground, and diff --git a/crates/pi-uu-grep/Cargo.toml b/crates/pi-uu-grep/Cargo.toml index b63c67ce6..2e3f3aa48 100644 --- a/crates/pi-uu-grep/Cargo.toml +++ b/crates/pi-uu-grep/Cargo.toml @@ -14,6 +14,7 @@ path = "src/lib.rs" [dependencies] clap = { version = "4", features = ["derive"] } pi-uutils-ctx = { path = "../pi-uutils-ctx" } +pi-walker = { path = "../pi-walker" } grep-matcher = "0.1" grep-regex = "0.1" grep-searcher = "0.1" diff --git a/crates/pi-uu-grep/src/lib.rs b/crates/pi-uu-grep/src/lib.rs index 7902baab0..e2eb756f0 100644 --- a/crates/pi-uu-grep/src/lib.rs +++ b/crates/pi-uu-grep/src/lib.rs @@ -1,6 +1,6 @@ //! `grep` implemented as an in-process shell builtin on top of the ripgrep //! libraries (`grep-regex` for the matcher, `grep-searcher` for line scanning), -//! with directory recursion via `ignore` and `--include` filtering via +//! with directory recursion via `pi-walker` and `--include` filtering via //! `globset`. All I/O and path resolution is routed through `pi-uutils-ctx` so //! the builtin writes to the command's redirected file descriptors and resolves //! relative paths against the shell's working directory. @@ -24,7 +24,6 @@ use globset::{Glob, GlobSet, GlobSetBuilder}; use grep_matcher::Matcher; use grep_regex::{RegexMatcher, RegexMatcherBuilder}; use grep_searcher::{Searcher, SearcherBuilder, Sink, SinkContext, SinkFinish, SinkMatch}; -use ignore::WalkBuilder; pub use rg::run as run_rg; #[derive(Parser, Debug)] @@ -358,6 +357,83 @@ fn process_reader( Ok(sink.any_match) } +fn display_path_for_operand(operand: &OsStr, resolved: &Path, path: &Path) -> PathBuf { + let rel = path.strip_prefix(resolved).unwrap_or(path); + if rel.as_os_str().is_empty() { + PathBuf::from(operand) + } else { + Path::new(operand).join(rel) + } +} + +#[allow(clippy::too_many_arguments)] +fn search_file_path( + operand: &OsStr, + resolved: &Path, + path: &Path, + matcher: &RegexMatcher, + searcher: &mut Searcher, + opts: &Options, + include_set: Option<&GlobSet>, + show_names: bool, + out: &mut W, + had_error: &mut bool, +) -> bool { + if let Some(set) = include_set { + let name = path.file_name().unwrap_or_default(); + if !set.is_match(name) { + return false; + } + } + let display_path = display_path_for_operand(operand, resolved, path); + match File::open(path) { + Ok(file) => { + let bytes = display_path.as_os_str().as_encoded_bytes().to_vec(); + let name: Option<&[u8]> = if show_names { Some(&bytes) } else { None }; + match process_reader(matcher, searcher, file, name, opts, out) { + Ok(matched) => matched, + Err(err) => { + *had_error = true; + if !opts.no_messages { + let _ = writeln!( + pi_uutils_ctx::stderr(), + "grep: {}: {err}", + display_path.to_string_lossy() + ); + } + false + }, + } + }, + Err(err) => { + *had_error = true; + if !opts.no_messages { + let _ = + writeln!(pi_uutils_ctx::stderr(), "grep: {}: {err}", display_path.to_string_lossy()); + } + false + }, + } +} + +fn grep_walk_request(root: &Path, follow_links: bool) -> pi_walker::WalkRequest { + pi_walker::WalkRequest::new(root) + .hidden(true) + .gitignore(false) + .skip_git(false) + .skip_node_modules(false) + .follow_links(pi_walker::FollowLinks::from(follow_links)) + .detail(pi_walker::WalkDetail::Minimal) + .order(pi_walker::WalkOrder::Unordered) + .emit_root(true) + .depth(0, usize::MAX) + .visit_order(pi_walker::VisitOrder::PreOrder) + .directory_errors(pi_walker::DirectoryErrorMode::Visit) + .same_file_system(false) + .cache(false) + .filter(pi_walker::WalkFilter::files_only()) +} + /// Recursively search a directory operand. `operand` is the path as typed (used /// for display), `resolved` is the cwd-resolved root walked on the filesystem. #[allow(clippy::too_many_arguments)] @@ -373,78 +449,87 @@ fn search_dir( out: &mut W, had_error: &mut bool, ) -> bool { + let request = grep_walk_request(resolved, follow_links); let mut any = false; - let mut builder = WalkBuilder::new(resolved); - // GNU grep -r searches everything (hidden files, VCS-ignored files); disable - // ignore's standard filters so behaviour matches grep, not ripgrep. - builder.standard_filters(false); - builder.follow_links(follow_links); - - for result in builder.build() { - // -q: a single match anywhere in the tree is enough. - if opts.quiet && any { - break; - } - let entry = match result { - Ok(e) => e, - Err(err) => { - *had_error = true; - if !opts.no_messages { - let _ = writeln!(pi_uutils_ctx::stderr(), "grep: {err}"); - } - continue; - }, - }; - // Only search regular files (skip directories and other non-files). - if entry.file_type().is_none_or(|t| t.is_dir()) { - continue; - } - let path = entry.path(); - if let Some(set) = include_set { - let name = path.file_name().unwrap_or_default(); - if !set.is_match(name) { - continue; + let had_error_state = std::cell::Cell::new(*had_error); + let walk = request.for_each_entry_with_heartbeat( + || Ok::<(), io::Error>(()), + |entry: pi_walker::EntryMeta<'_>| { + if opts.quiet && any { + return Ok(pi_walker::WalkDecision::Stop); } - } - // Rebuild the display path as `/` so output uses the - // operand exactly as typed (GNU behaviour), not the resolved abs path. - let rel = path.strip_prefix(resolved).unwrap_or(path); - let display_path: PathBuf = if rel.as_os_str().is_empty() { - PathBuf::from(operand) - } else { - Path::new(operand).join(rel) - }; - match File::open(path) { - Ok(file) => { - let bytes = display_path.as_os_str().as_encoded_bytes().to_vec(); - let name: Option<&[u8]> = if show_names { Some(&bytes) } else { None }; - match process_reader(matcher, searcher, file, name, opts, out) { - Ok(m) => any |= m, - Err(e) => { - *had_error = true; - if !opts.no_messages { - let _ = writeln!( - pi_uutils_ctx::stderr(), - "grep: {}: {e}", - display_path.to_string_lossy() - ); - } - }, - } - }, - Err(e) => { - *had_error = true; - if !opts.no_messages { - let _ = writeln!( - pi_uutils_ctx::stderr(), - "grep: {}: {e}", - display_path.to_string_lossy() - ); - } - }, - } + if entry.file_type == pi_walker::FileType::Dir { + return Ok(pi_walker::WalkDecision::Skip); + } + let mut entry_had_error = had_error_state.get(); + let matched = search_file_path( + operand, + resolved, + entry.absolute_path.as_ref(), + matcher, + searcher, + opts, + include_set, + show_names, + out, + &mut entry_had_error, + ); + had_error_state.set(entry_had_error); + any |= matched; + if opts.quiet && any { + Ok(pi_walker::WalkDecision::Stop) + } else { + Ok(pi_walker::WalkDecision::Include) + } + }, + |error: pi_walker::DirectoryError<'_>| { + had_error_state.set(true); + if !opts.no_messages { + let display_path = display_path_for_operand(operand, resolved, error.path); + let _ = writeln!( + pi_uutils_ctx::stderr(), + "grep: {}: {}", + display_path.to_string_lossy(), + error.error + ); + } + Ok(pi_walker::WalkDecision::Include) + }, + ); + *had_error |= had_error_state.get(); + match walk { + Ok(pi_walker::WalkStatus::Complete | pi_walker::WalkStatus::Stopped) => any, + Ok(pi_walker::WalkStatus::Unsupported) | Err(pi_walker::WalkError::Unsupported) => { + *had_error = true; + if !opts.no_messages { + let _ = writeln!( + pi_uutils_ctx::stderr(), + "grep: {}: native directory scan unsupported", + operand.to_string_lossy() + ); + } + any + }, + Err(pi_walker::WalkError::Interrupted(err)) => { + *had_error = true; + if !opts.no_messages { + let _ = writeln!(pi_uutils_ctx::stderr(), "grep: {err}"); + } + any + }, + Err(pi_walker::WalkError::InvalidData { path, message }) => { + *had_error = true; + if !opts.no_messages { + let display_path = display_path_for_operand(operand, resolved, &path); + let _ = writeln!( + pi_uutils_ctx::stderr(), + "grep: {}: {message}", + display_path.to_string_lossy() + ); + } + any + }, } - any } /// In-process builtin entry point. The host installs a [`pi_uutils_ctx`] scope diff --git a/crates/pi-uu-grep/src/rg.rs b/crates/pi-uu-grep/src/rg.rs index a3018ea22..e3e696dd6 100644 --- a/crates/pi-uu-grep/src/rg.rs +++ b/crates/pi-uu-grep/src/rg.rs @@ -15,7 +15,11 @@ use grep_regex::{RegexMatcher, RegexMatcherBuilder}; use grep_searcher::{ BinaryDetection, Searcher, SearcherBuilder, Sink, SinkContext, SinkFinish, SinkMatch, }; -use ignore::{WalkBuilder, overrides::OverrideBuilder, types::TypesBuilder}; +use ignore::{ + Match, + overrides::{Override, OverrideBuilder}, + types::{Types, TypesBuilder}, +}; #[derive(Parser, Debug)] #[command( @@ -777,31 +781,57 @@ fn print_type_list(cli: &RgCli, out: &mut W) -> Result<(), String> { Ok(()) } -fn configure_walk(cli: &RgCli, root: &Path) -> Result { +struct RgWalk { + request: pi_walker::WalkRequest, + filters: PathFilters, +} + +struct PathFilters { + overrides: Option, + types: Option, + max_filesize: Option, +} + +impl PathFilters { + fn includes(&self, path: &Path, file_type: pi_walker::FileType, size: Option) -> bool { + let is_dir = file_type == pi_walker::FileType::Dir; + if self + .overrides + .as_ref() + .is_some_and(|overrides| matches!(overrides.matched(path, is_dir), Match::Ignore(_))) + { + return false; + } + if file_type != pi_walker::FileType::File { + return true; + } + if self + .types + .as_ref() + .is_some_and(|types| matches!(types.matched(path, false), Match::Ignore(_))) + { + return false; + } + if let Some(limit) = self.max_filesize { + let size = size.or_else(|| std::fs::metadata(path).ok().map(|meta| meta.len() as f64)); + if size.is_some_and(|size| size > limit as f64) { + return false; + } + } + true + } +} + +fn build_path_filters(cli: &RgCli) -> Result { let cwd = pi_uutils_ctx::cwd(); - let mut builder = WalkBuilder::new(root); - builder.current_dir(cwd.clone()); - builder.follow_links(cli.follow); - builder.max_depth(cli.max_depth); - if let Some(size) = &cli.max_filesize { - builder.max_filesize(Some(parse_size(size).map_err(|err| format!("rg: {err}"))?)); - } - - let unrestricted_no_ignore = cli.unrestricted >= 1; - let include_hidden = (cli.hidden || cli.unrestricted >= 2) && !cli.no_hidden; - let no_ignore = (cli.no_ignore || unrestricted_no_ignore) && !cli.ignore; - builder.hidden(!include_hidden); - builder.parents(!(no_ignore || (cli.no_ignore_parent && !cli.ignore_parent))); - builder.ignore(!(no_ignore || (cli.no_ignore_dot && !cli.ignore_dot))); - builder.git_ignore(!(no_ignore || (cli.no_ignore_vcs && !cli.ignore_vcs))); - builder.git_global(!(no_ignore || (cli.no_ignore_global && !cli.ignore_global))); - builder.git_exclude(!(no_ignore || (cli.no_ignore_exclude && !cli.ignore_exclude))); - builder.require_git(!cli.no_require_git || cli.require_git); - if !(no_ignore || (cli.no_ignore_dot && !cli.ignore_dot)) { - builder.add_custom_ignore_filename(".rgignore"); - } - - if !cli.globs.is_empty() || !cli.iglobs.is_empty() { + let max_filesize = cli + .max_filesize + .as_ref() + .map(|size| parse_size(size).map_err(|err| format!("rg: {err}"))) + .transpose()?; + let overrides = if cli.globs.is_empty() && cli.iglobs.is_empty() { + None + } else { let mut overrides = OverrideBuilder::new(&cwd); for glob in &cli.globs { overrides @@ -818,23 +848,49 @@ fn configure_walk(cli: &RgCli, root: &Path) -> Result { .map_err(|err| format!("rg: --iglob {glob:?}: {err}"))?; } } - builder.overrides(overrides.build().map_err(|err| format!("rg: {err}"))?); - } - - if !cli.types.is_empty() || !cli.type_nots.is_empty() { - builder.types( + Some(overrides.build().map_err(|err| format!("rg: {err}"))?) + }; + let types = if cli.types.is_empty() && cli.type_nots.is_empty() { + None + } else { + Some( type_builder(cli)? .build() .map_err(|err| format!("rg: {err}"))?, - ); - } + ) + }; + Ok(PathFilters { overrides, types, max_filesize }) +} - if cli.sort_files || cli.sort.as_deref() == Some("path") { - builder.sort_by_file_path(|a, b| a.cmp(b)); - } else if cli.sortr.as_deref() == Some("path") { - builder.sort_by_file_path(|a, b| b.cmp(a)); - } - Ok(builder) +fn build_walk(cli: &RgCli, root: &Path) -> Result { + let filters = build_path_filters(cli)?; + let unrestricted_no_ignore = cli.unrestricted >= 1; + let include_hidden = (cli.hidden || cli.unrestricted >= 2) && !cli.no_hidden; + let no_ignore = (cli.no_ignore || unrestricted_no_ignore) && !cli.ignore; + let order = if cli.sort_files || cli.sort.as_deref() == Some("path") { + pi_walker::WalkOrder::Path + } else { + pi_walker::WalkOrder::Unordered + }; + let request = pi_walker::WalkRequest::new(root) + .hidden(include_hidden) + .gitignore(!no_ignore) + .skip_git(!no_ignore) + .skip_node_modules(false) + .follow_links(pi_walker::FollowLinks::from(cli.follow)) + .detail(if filters.max_filesize.is_some() { + pi_walker::WalkDetail::Full + } else { + pi_walker::WalkDetail::Minimal + }) + .order(order) + .emit_root(false) + .depth(1, cli.max_depth.unwrap_or(usize::MAX)) + .visit_order(pi_walker::VisitOrder::PreOrder) + .directory_errors(pi_walker::DirectoryErrorMode::Visit) + .same_file_system(false) + .cache(false); + Ok(RgWalk { request, filters }) } fn display_path(operand: &OsStr, root: &Path, path: &Path) -> PathBuf { @@ -900,6 +956,46 @@ fn report_path_error( true } +#[allow( + clippy::too_many_arguments, + reason = "required by standard walk/configure interfaces and search parameters" +)] +fn search_collected_files( + cli: &RgCli, + matcher: &RegexMatcher, + searcher: &mut Searcher, + operand: &OsStr, + root: &Path, + show_names: bool, + opts: &SearchOptions, + out: &mut W, +) -> SearchOutcome { + let mut files = match collect_filtered_files(cli, root) { + Ok(files) => files, + Err(err) => { + if !opts.no_messages { + let _ = writeln!(pi_uutils_ctx::stderr(), "{err}"); + } + return SearchOutcome { any_match: false, had_error: true }; + }, + }; + files.sort_unstable_by(|a, b| b.cmp(a)); + let mut any_match = false; + let mut had_error = false; + for path in files { + if opts.quiet && any_match { + break; + } + let display_path = display_path(operand, root, &path); + let display_bytes = display_path.as_os_str().as_encoded_bytes().to_vec(); + let display = show_names.then_some(display_bytes.as_slice()); + let outcome = process_file(matcher, searcher, &path, display, opts, out); + any_match |= outcome.any_match; + had_error |= outcome.had_error; + } + SearchOutcome { any_match, had_error } +} + #[allow( clippy::too_many_arguments, reason = "required by standard walk/configure interfaces and search parameters" @@ -914,10 +1010,11 @@ fn search_dir( opts: &SearchOptions, out: &mut W, ) -> SearchOutcome { - let mut any_match = false; - let mut had_error = false; - let builder = match configure_walk(cli, root) { - Ok(builder) => builder, + if cli.sortr.as_deref() == Some("path") { + return search_collected_files(cli, matcher, searcher, operand, root, show_names, opts, out); + } + let walk = match build_walk(cli, root) { + Ok(walk) => walk, Err(err) => { if !opts.no_messages { let _ = writeln!(pi_uutils_ctx::stderr(), "{err}"); @@ -925,31 +1022,80 @@ fn search_dir( return SearchOutcome { any_match: false, had_error: true }; }, }; - for result in builder.build() { - if opts.quiet && any_match { - break; - } - let entry = match result { - Ok(entry) => entry, - Err(err) => { - had_error = true; - if !opts.no_messages { - let _ = writeln!(pi_uutils_ctx::stderr(), "rg: {err}"); - } - continue; - }, - }; - if entry.file_type().is_none_or(|kind| !kind.is_file()) { + let any_match = std::cell::Cell::new(false); + let had_error = std::cell::Cell::new(false); + let streamed = match walk.request.for_each_entry_with_heartbeat( + || Ok::<(), io::Error>(()), + |entry| { + if opts.quiet && any_match.get() { + return Ok(pi_walker::WalkDecision::Stop); + } + let path = entry.absolute_path.as_ref(); + if !walk.filters.includes(path, entry.file_type, entry.size) { + return Ok(if entry.file_type == pi_walker::FileType::Dir { + pi_walker::WalkDecision::SkipDescend + } else { + pi_walker::WalkDecision::Skip + }); + } + if entry.file_type != pi_walker::FileType::File { + return Ok(pi_walker::WalkDecision::Skip); + } + let display_path = display_path(operand, root, path); + let display_bytes = display_path.as_os_str().as_encoded_bytes().to_vec(); + let display = show_names.then_some(display_bytes.as_slice()); + let outcome = process_file(matcher, searcher, path, display, opts, out); + any_match.set(any_match.get() || outcome.any_match); + had_error.set(had_error.get() || outcome.had_error); + Ok(if opts.quiet && any_match.get() { + pi_walker::WalkDecision::Stop + } else { + pi_walker::WalkDecision::Include + }) + }, + |error| { + had_error.set(true); + if !opts.no_messages { + let _ = + writeln!(pi_uutils_ctx::stderr(), "rg: {}: {}", error.path.display(), error.error); + } + Ok(pi_walker::WalkDecision::Include) + }, + ) { + Ok(pi_walker::WalkStatus::Complete | pi_walker::WalkStatus::Stopped) => { + Some(SearchOutcome { any_match: any_match.get(), had_error: had_error.get() }) + }, + Ok(pi_walker::WalkStatus::Unsupported) => None, + Err(err) => { + had_error.set(true); + if !opts.no_messages { + let _ = writeln!(pi_uutils_ctx::stderr(), "rg: {err}"); + } + Some(SearchOutcome { any_match: any_match.get(), had_error: had_error.get() }) + }, + }; + streamed.unwrap_or_else(|| { + search_collected_files(cli, matcher, searcher, operand, root, show_names, opts, out) + }) +} + +fn collect_filtered_files(cli: &RgCli, root: &Path) -> Result, String> { + let walk = build_walk(cli, root)?; + let outcome = walk + .request + .collect_with_heartbeat(|| Ok::<(), io::Error>(())) + .map_err(|err| format!("rg: {err}"))?; + let mut files = Vec::new(); + for entry in outcome.entries { + if entry.file_type != pi_walker::FileType::File { continue; } - let display_path = display_path(operand, root, entry.path()); - let display_bytes = display_path.as_os_str().as_encoded_bytes().to_vec(); - let display = show_names.then_some(display_bytes.as_slice()); - let outcome = process_file(matcher, searcher, entry.path(), display, opts, out); - any_match |= outcome.any_match; - had_error |= outcome.had_error; + let path = entry.absolute_path(root); + if walk.filters.includes(&path, entry.file_type, entry.size) { + files.push(path); + } } - SearchOutcome { any_match, had_error } + Ok(files) } fn list_files(cli: &RgCli, paths: &[OsString], out: &mut W) -> SearchOutcome { @@ -959,27 +1105,19 @@ fn list_files(cli: &RgCli, paths: &[OsString], out: &mut W) -> SearchO let resolved = pi_uutils_ctx::resolve(operand); match std::fs::metadata(&resolved) { Ok(meta) if meta.is_dir() => { - let builder = match configure_walk(cli, &resolved) { - Ok(builder) => builder, + let mut files = match collect_filtered_files(cli, &resolved) { + Ok(files) => files, Err(err) => { let _ = writeln!(pi_uutils_ctx::stderr(), "{err}"); had_error = true; continue; }, }; - for result in builder.build() { - let entry = match result { - Ok(entry) => entry, - Err(err) => { - let _ = writeln!(pi_uutils_ctx::stderr(), "rg: {err}"); - had_error = true; - continue; - }, - }; - if entry.file_type().is_none_or(|kind| !kind.is_file()) { - continue; - } - let display = display_path(operand.as_os_str(), &resolved, entry.path()); + if cli.sortr.as_deref() == Some("path") { + files.sort_unstable_by(|a, b| b.cmp(a)); + } + for path in files { + let display = display_path(operand.as_os_str(), &resolved, &path); let _ = out.write_all(display.as_os_str().as_encoded_bytes()); let _ = out.write_all(if cli.null { b"\0" } else { b"\n" }); any = true; diff --git a/crates/pi-walker/Cargo.toml b/crates/pi-walker/Cargo.toml new file mode 100644 index 000000000..17e78d88f --- /dev/null +++ b/crates/pi-walker/Cargo.toml @@ -0,0 +1,22 @@ +[package] +name = "pi-walker" +version.workspace = true +edition.workspace = true +license.workspace = true +authors.workspace = true +repository.workspace = true + +[lints] +workspace = true + +[dependencies] +dashmap.workspace = true +globset.workspace = true +ignore.workspace = true +rayon.workspace = true + +[target.'cfg(unix)'.dependencies] +libc.workspace = true + +[target.'cfg(windows)'.dependencies] +windows-sys = { workspace = true, features = ["Wdk_Storage_FileSystem"] } diff --git a/crates/pi-walker/src/cache.rs b/crates/pi-walker/src/cache.rs new file mode 100644 index 000000000..37b93f53b --- /dev/null +++ b/crates/pi-walker/src/cache.rs @@ -0,0 +1,923 @@ +//! Shared walker scan cache used by owned-entry collection. + +use std::{ + borrow::Cow, + fmt, + path::{Path, PathBuf}, + sync::{Arc, LazyLock, Mutex}, + time::{Duration, Instant}, +}; + +use dashmap::DashMap; +use ignore::{ParallelVisitor, ParallelVisitorBuilder, WalkBuilder, WalkState}; +use rayon::{ThreadPool, prelude::*}; + +use crate::{ + CollectedEntries, CollectedEntry, EntryScan, FileType, FollowLinks, WalkDetail, WalkError, + WalkOptions, +}; + +#[derive(Clone, Debug, Eq, Hash, PartialEq)] +struct CacheKey { + root: PathBuf, + options: WalkOptions, +} + +#[derive(Clone)] +struct CacheEntry { + created_at: Instant, + entries: Vec, +} + +static CACHE_TTL_MS: LazyLock = + LazyLock::new(|| env_uint("FS_SCAN_CACHE_TTL_MS", 1_000, 0, u64::MAX)); +static EMPTY_RECHECK_MS: LazyLock = + LazyLock::new(|| env_uint("FS_SCAN_EMPTY_RECHECK_MS", 200, 0, u64::MAX)); +static MAX_CACHE_ENTRIES: LazyLock = + LazyLock::new(|| env_uint("FS_SCAN_CACHE_MAX_ENTRIES", 16, 0, usize::MAX)); +const DEFAULT_WALK_WORKERS: usize = 4; + +static WALK_WORKERS: LazyLock = LazyLock::new(|| { + normalize_worker_count(env_uint("PI_WALK_WORKERS", DEFAULT_WALK_WORKERS, 0, usize::MAX)) +}); +static WALK_POOL: LazyLock> = LazyLock::new(|| { + let workers = walk_workers(); + if workers <= 1 { + return None; + } + rayon::ThreadPoolBuilder::new() + .num_threads(workers) + .thread_name(|index| format!("pi-walker-{index}")) + .build() + .ok() +}); +static SCAN_CACHE: LazyLock> = LazyLock::new(DashMap::new); + +fn env_uint(name: &str, default: T, min: T, max: T) -> T +where + T: Copy + Ord + std::str::FromStr, +{ + std::env::var(name) + .ok() + .and_then(|value| value.parse().ok()) + .unwrap_or(default) + .clamp(min, max) +} + +fn normalize_worker_count_with_available(configured: usize, available: usize) -> usize { + if configured == 0 { + available.max(1) + } else { + configured.max(1) + } +} + +fn available_worker_count() -> usize { + std::thread::available_parallelism().map_or(DEFAULT_WALK_WORKERS, usize::from) +} + +fn normalize_worker_count(configured: usize) -> usize { + normalize_worker_count_with_available(configured, available_worker_count()) +} + +/// Configured cache TTL in milliseconds. +pub fn cache_ttl_ms() -> u64 { + *CACHE_TTL_MS +} + +/// Configured empty-result recheck threshold in milliseconds. +pub fn empty_recheck_ms() -> u64 { + *EMPTY_RECHECK_MS +} + +/// Configured maximum number of cache entries. +pub fn max_cache_entries() -> usize { + *MAX_CACHE_ENTRIES +} + +/// Effective worker count for filesystem traversal and related parallel work. +/// +/// `PI_WALK_WORKERS=0` means auto-detect; `PI_WALK_WORKERS=1` forces serial +/// work. +pub fn walk_workers() -> usize { + *WALK_WORKERS +} + +/// Apply the centralized walk worker count to an `ignore` walker. +fn configure_walk_builder(builder: &mut WalkBuilder) { + builder.threads(walk_workers()); +} + +/// Run parallel traversal-adjacent work on the centralized walker pool. +fn with_walk_pool(operation: impl FnOnce() -> R + Send) -> R +where + R: Send, +{ + if let Some(pool) = WALK_POOL.as_ref() { + pool.install(operation) + } else { + operation() + } +} + +const PARALLEL_MIN_FILES: usize = 256; + +/// Return whether traversal-adjacent work should run in parallel. +pub fn should_parallelize(item_count: usize) -> bool { + walk_workers() > 1 && item_count >= PARALLEL_MIN_FILES +} + +/// Run traversal-adjacent work serially or on the centralized walker pool. +pub fn parallel_for_each( + items: &[T], + operation: impl Fn(&T) -> std::result::Result<(), E> + Send + Sync, +) -> std::result::Result<(), E> +where + T: Sync, + E: Send, +{ + if !should_parallelize(items.len()) { + for item in items { + operation(item)?; + } + return Ok(()); + } + with_walk_pool(|| items.par_iter().try_for_each(operation)) +} + +fn evict_oldest() { + if SCAN_CACHE.len() > *MAX_CACHE_ENTRIES + && let Some(oldest_key) = SCAN_CACHE + .iter() + .min_by_key(|entry| entry.value().created_at) + .map(|entry| entry.key().clone()) + { + SCAN_CACHE.remove(&oldest_key); + } +} + +fn cache_key(root: &Path, mut options: WalkOptions) -> CacheKey { + options.cache = false; + CacheKey { root: root.to_path_buf(), options } +} + +/// Normalize a filesystem path to a forward-slash relative string. +pub fn normalize_relative_path<'a>(root: &Path, path: &'a Path) -> Cow<'a, str> { + let relative = path.strip_prefix(root).unwrap_or(path); + if cfg!(windows) { + let relative = relative.to_string_lossy(); + if relative.contains('\\') { + Cow::Owned(relative.replace('\\', "/")) + } else { + relative + } + } else { + relative.to_string_lossy() + } +} + +/// Return whether a path contains the exact component name. +pub fn contains_component(path: &Path, target: &str) -> bool { + path.components().any(|component| { + component + .as_os_str() + .to_str() + .is_some_and(|value| value == target) + }) +} + +/// Return whether user-facing discovery should skip a relative path. +pub fn should_skip_path(path: &Path, mentions_node_modules: bool) -> bool { + if contains_component(path, ".git") { + return true; + } + if !mentions_node_modules && contains_component(path, "node_modules") { + return true; + } + false +} + +fn file_type_from_std(file_type: std::fs::FileType) -> Option { + if file_type.is_symlink() { + Some(FileType::Symlink) + } else if file_type.is_dir() { + Some(FileType::Dir) + } else if file_type.is_file() { + Some(FileType::File) + } else { + None + } +} + +fn mtime_ms(metadata: &std::fs::Metadata) -> Option { + metadata + .modified() + .ok() + .and_then(|time| time.duration_since(std::time::UNIX_EPOCH).ok()) + .map(|duration| duration.as_millis() as f64) +} + +/// Classify an existing filesystem path, skipping unsupported special files. +pub fn classify_file_type(path: &Path) -> Option<(FileType, Option, Option)> { + let metadata = std::fs::symlink_metadata(path).ok()?; + let file_type = file_type_from_std(metadata.file_type())?; + let size = if file_type == FileType::File { + Some(metadata.len()) + } else { + None + }; + Some((file_type, mtime_ms(&metadata), size)) +} + +/// Resolve a search path string to a canonical directory path. +pub fn resolve_search_path(path: &str) -> Result> { + let candidate = PathBuf::from(path); + let root = if candidate.is_absolute() { + candidate + } else { + let cwd = std::env::current_dir().map_err(|err| WalkError::InvalidData { + path: PathBuf::from(path), + message: format!("Failed to resolve cwd: {err}"), + })?; + cwd.join(candidate) + }; + let metadata = std::fs::metadata(&root).map_err(|err| WalkError::InvalidData { + path: root.clone(), + message: format!("Path not found: {err}"), + })?; + if !metadata.is_dir() { + return Err(WalkError::InvalidData { + path: root, + message: "Search path must be a directory".to_string(), + }); + } + Ok(std::fs::canonicalize(&root).unwrap_or(root)) +} + +fn build_walker_for_options(root: &Path, options: WalkOptions) -> WalkBuilder { + build_walker_for_options_inner(root, options, None) +} + +pub fn build_walker_for_options_with_pruned_dirs( + root: &Path, + options: WalkOptions, + pruned_dirs: Arc>>, +) -> WalkBuilder { + build_walker_for_options_inner(root, options, Some(pruned_dirs)) +} + +fn build_walker_for_options_inner( + root: &Path, + options: WalkOptions, + pruned_dirs: Option>>>, +) -> WalkBuilder { + let mut builder = WalkBuilder::new(root); + builder + .hidden(!options.include_hidden) + .follow_links(matches!(options.follow_links, FollowLinks::Always)) + .sort_by_file_path(|a, b| a.cmp(b)) + .filter_entry(move |entry| { + if let Some(pruned_dirs) = &pruned_dirs + && pruned_dirs + .lock() + .expect("pruned directory lock poisoned") + .iter() + .any(|dir| entry.path().starts_with(dir)) + { + return false; + } + let name = entry.file_name().to_str().unwrap_or_default(); + if options.skip_git && name == ".git" { + return false; + } + if options.skip_node_modules && name == "node_modules" { + return false; + } + true + }); + if options.max_depth != usize::MAX { + builder.max_depth(Some(options.max_depth)); + } + + if options.use_gitignore { + builder + .git_ignore(true) + .git_exclude(true) + .git_global(true) + .ignore(true) + .parents(true); + } else { + builder + .git_ignore(false) + .git_exclude(false) + .git_global(false) + .ignore(false) + .parents(false); + } + + builder +} + +/// Converts one `ignore` walker entry into owned scan metadata. +pub fn collect_entry( + root: &Path, + entry: &ignore::DirEntry, + detail: WalkDetail, +) -> Option { + let path = entry.path(); + let relative = normalize_relative_path(root, path); + if relative.is_empty() { + return None; + } + + let (file_type, mtime, size) = match detail { + WalkDetail::Minimal => { + let file_type = file_type_from_std(entry.file_type()?)?; + (file_type, None, None) + }, + WalkDetail::Full => { + let metadata = entry + .metadata() + .or_else(|_| std::fs::symlink_metadata(path)) + .ok()?; + let file_type = file_type_from_std(metadata.file_type())?; + let size = if file_type == FileType::File { + Some(metadata.len() as f64) + } else { + None + }; + (file_type, mtime_ms(&metadata), size) + }, + }; + + Some(CollectedEntry { path: relative.into_owned(), file_type, mtime, size }) +} + +fn root_entry(root: &Path, detail: WalkDetail) -> Option { + let (file_type, mtime, size) = classify_file_type(root)?; + let size = if detail == WalkDetail::Full && file_type == FileType::File { + size.map(|value| value as f64) + } else { + None + }; + let mtime = if detail == WalkDetail::Full { + mtime + } else { + None + }; + Some(CollectedEntry { path: String::new(), file_type, mtime, size }) +} + +struct EntryVisitor<'a, H> { + root: &'a Path, + options: WalkOptions, + heartbeat: &'a H, + entries: Vec, + shared_entries: Arc>>>, + error: Arc>>, + visited: usize, +} + +impl Drop for EntryVisitor<'_, H> { + fn drop(&mut self) { + if self.entries.is_empty() { + return; + } + let entries = std::mem::take(&mut self.entries); + self + .shared_entries + .lock() + .expect("entry collection lock poisoned") + .push(entries); + } +} + +impl ParallelVisitor for EntryVisitor<'_, H> +where + H: Fn() -> std::result::Result<(), E> + Sync, + E: fmt::Display, +{ + fn visit(&mut self, entry: std::result::Result) -> WalkState { + if self.visited == 0 || self.visited >= 128 { + self.visited = 0; + if let Err(err) = (self.heartbeat)() { + *self.error.lock().expect("error lock poisoned") = Some(err.to_string()); + return WalkState::Quit; + } + } + self.visited += 1; + + let Ok(entry) = entry else { + return WalkState::Continue; + }; + if entry.depth() < self.options.min_depth { + return WalkState::Continue; + } + if let Some(entry) = collect_entry(self.root, &entry, self.options.detail) { + self.entries.push(entry); + } + WalkState::Continue + } +} + +struct EntryVisitorBuilder<'a, H> { + root: &'a Path, + options: WalkOptions, + heartbeat: &'a H, + shared_entries: Arc>>>, + error: Arc>>, +} + +impl<'a, H, E> ParallelVisitorBuilder<'a> for EntryVisitorBuilder<'a, H> +where + H: Fn() -> std::result::Result<(), E> + Sync + 'a, + E: fmt::Display + 'a, +{ + fn build(&mut self) -> Box { + Box::new(EntryVisitor { + root: self.root, + options: self.options, + heartbeat: self.heartbeat, + entries: Vec::new(), + shared_entries: Arc::clone(&self.shared_entries), + error: Arc::clone(&self.error), + visited: 0, + }) + } +} + +fn collect_entries_uncached( + root: &Path, + mut options: WalkOptions, + heartbeat: &H, +) -> Result> +where + H: Fn() -> std::result::Result<(), E> + Sync, + E: fmt::Display, +{ + options.cache = false; + match crate::collect_entries_native(root, options, || { + heartbeat().map_err(|err| err.to_string()) + })? { + EntryScan::Entries(scan) => return Ok(EntryScan::Entries(scan)), + EntryScan::Unsupported => {}, + } + + if options.contents_first || options.same_file_system || options.min_depth > options.max_depth { + return Ok(EntryScan::Unsupported); + } + + let mut entries = Vec::new(); + if options.emit_root + && options.min_depth == 0 + && let Some(entry) = root_entry(root, options.detail) + { + entries.push(entry); + } + + let mut builder = build_walker_for_options(root, options); + configure_walk_builder(&mut builder); + let shared_entries = Arc::new(Mutex::new(Vec::new())); + let error = Arc::new(Mutex::new(None)); + let mut visitor_builder = EntryVisitorBuilder { + root, + options, + heartbeat, + shared_entries: Arc::clone(&shared_entries), + error: Arc::clone(&error), + }; + heartbeat().map_err(|err| WalkError::Interrupted(err.to_string()))?; + builder.build_parallel().visit(&mut visitor_builder); + + let walk_error = error.lock().expect("error lock poisoned").take(); + if let Some(error) = walk_error { + return Err(WalkError::Interrupted(error)); + } + + entries.extend( + shared_entries + .lock() + .expect("entry collection lock poisoned") + .drain(..) + .flatten(), + ); + entries.sort_unstable_by(|a, b| a.path.cmp(&b.path)); + Ok(EntryScan::Entries(CollectedEntries { entries, cache_age_ms: 0 })) +} + +fn get_or_scan( + root: &Path, + options: WalkOptions, + heartbeat: &H, +) -> Result> +where + H: Fn() -> std::result::Result<(), E> + Sync, + E: fmt::Display, +{ + let ttl = *CACHE_TTL_MS; + if ttl == 0 { + let scan = collect_entries_uncached(root, options, heartbeat)?; + return Ok(scan); + } + + let key = cache_key(root, options); + let now = Instant::now(); + if let Some(entry) = SCAN_CACHE.get(&key) { + let age = now.duration_since(entry.created_at); + if age < Duration::from_millis(ttl) { + return Ok(EntryScan::Entries(CollectedEntries { + entries: entry.entries.clone(), + cache_age_ms: age.as_millis() as u64, + })); + } + drop(entry); + SCAN_CACHE.remove(&key); + } + + let scan = collect_entries_uncached(root, options, heartbeat)?; + let EntryScan::Entries(scan) = scan else { + return Ok(EntryScan::Unsupported); + }; + SCAN_CACHE.insert(key, CacheEntry { created_at: now, entries: scan.entries.clone() }); + evict_oldest(); + Ok(EntryScan::Entries(CollectedEntries { entries: scan.entries, cache_age_ms: 0 })) +} + +pub fn collect_entries( + root: &Path, + options: WalkOptions, + heartbeat: H, +) -> Result> +where + H: Fn() -> std::result::Result<(), E> + Sync, + E: fmt::Display, +{ + if options.cache { + get_or_scan(root, options, &heartbeat) + } else { + collect_entries_uncached(root, options, &heartbeat) + } +} + +/// Invalidate cache entries whose root contains `target`. +pub fn invalidate_path(target: &Path) { + let keys_to_remove: Vec = SCAN_CACHE + .iter() + .filter(|entry| target.starts_with(&entry.key().root)) + .map(|entry| entry.key().clone()) + .collect(); + for key in keys_to_remove { + SCAN_CACHE.remove(&key); + } +} + +/// Resolve a possibly relative path and invalidate matching cache roots. +pub fn invalidate_path_string(path: &str) { + let candidate = PathBuf::from(path); + let absolute = if candidate.is_absolute() { + candidate + } else if let Ok(cwd) = std::env::current_dir() { + cwd.join(candidate) + } else { + PathBuf::from(path) + }; + let target = std::fs::canonicalize(&absolute) + .or_else(|_| { + absolute + .parent() + .and_then(|parent| std::fs::canonicalize(parent).ok()) + .and_then(|parent| absolute.file_name().map(|name| parent.join(name))) + .ok_or_else(|| std::io::Error::from(std::io::ErrorKind::NotFound)) + }) + .unwrap_or(absolute); + invalidate_path(&target); +} + +/// Clear the entire scan cache. +pub fn invalidate_all() { + SCAN_CACHE.clear(); +} + +#[cfg(test)] +mod tests { + #[cfg(unix)] + use std::{ffi::CString, os::unix::ffi::OsStrExt}; + use std::{ + fs, + path::{Path, PathBuf}, + sync::atomic::{AtomicU64, Ordering}, + time::{Duration, SystemTime, UNIX_EPOCH}, + }; + + #[cfg(unix)] + use super::classify_file_type; + use crate::{CollectedEntry, FileType}; + + static TEMP_COUNTER: AtomicU64 = AtomicU64::new(0); + + struct TempDirGuard(PathBuf); + + impl TempDirGuard { + fn new() -> Self { + let timestamp = SystemTime::now() + .duration_since(UNIX_EPOCH) + .expect("system time is after UNIX_EPOCH") + .as_nanos(); + let counter = TEMP_COUNTER.fetch_add(1, Ordering::Relaxed); + let path = std::env::temp_dir().join(format!("pi-fs-cache-test-{timestamp}-{counter}")); + fs::create_dir_all(&path).expect("create temp test directory"); + Self(path) + } + + fn path(&self) -> &Path { + &self.0 + } + } + + impl Drop for TempDirGuard { + fn drop(&mut self) { + let _ = fs::remove_dir_all(&self.0); + } + } + + #[cfg(unix)] + fn make_fifo(path: &Path) { + let fifo_path = + CString::new(path.as_os_str().as_bytes()).expect("fifo path has no NUL bytes"); + // SAFETY: `fifo_path` is a valid CString (NUL-terminated, no interior NULs), + // so `as_ptr()` yields a valid C string pointer. `0o600` is a valid mode. + // The CString is alive for the duration of the call. + let rc = unsafe { libc::mkfifo(fifo_path.as_ptr(), 0o600) }; + assert_eq!(rc, 0, "create fifo: {}", std::io::Error::last_os_error()); + } + + #[allow( + clippy::unnecessary_wraps, + reason = "test heartbeat helper matches production callback signature" + )] + fn ok_heartbeat() -> std::result::Result<(), String> { + Ok(()) + } + + #[test] + fn worker_count_zero_uses_available_parallelism() { + assert_eq!(super::normalize_worker_count_with_available(0, 8), 8); + assert_eq!(super::normalize_worker_count_with_available(0, 0), 1); + assert_eq!(super::normalize_worker_count_with_available(1, 8), 1); + assert_eq!(super::normalize_worker_count_with_available(4, 8), 4); + } + + fn scan_options( + include_hidden: bool, + use_gitignore: bool, + detail: crate::WalkDetail, + ) -> crate::WalkOptions { + crate::WalkOptions { + include_hidden, + use_gitignore, + skip_git: true, + skip_node_modules: true, + follow_links: crate::FollowLinks::Never, + detail, + directory_errors: crate::DirectoryErrorMode::SkipSkippable, + ..crate::WalkOptions::default() + } + } + + fn assert_file_entry(entries: &[CollectedEntry], path: &str, size: f64) { + let entry = entries + .iter() + .find(|entry| entry.path == path) + .unwrap_or_else(|| panic!("expected file entry {path}, got {}", entry_paths(entries))); + assert_eq!(entry.file_type, FileType::File); + assert!(entry.mtime.is_some(), "full scan should include mtime for {path}"); + assert_eq!(entry.size, Some(size)); + } + + fn assert_dir_entry(entries: &[CollectedEntry], path: &str) { + let entry = entries + .iter() + .find(|entry| entry.path == path) + .unwrap_or_else(|| panic!("expected dir entry {path}, got {}", entry_paths(entries))); + assert_eq!(entry.file_type, FileType::Dir); + assert!(entry.mtime.is_some(), "full scan should include mtime for {path}"); + assert_eq!(entry.size, None); + } + + fn entry_paths(entries: &[CollectedEntry]) -> String { + let paths: Vec<&str> = entries.iter().map(|entry| entry.path.as_str()).collect(); + format!("{paths:?}") + } + + #[cfg(unix)] + #[test] + fn classify_file_type_skips_fifo() { + let root = TempDirGuard::new(); + let fifo = root.path().join("skip-me.fifo"); + make_fifo(&fifo); + + assert_eq!(classify_file_type(&fifo), None); + } + + #[test] + fn collect_entries_skips_node_modules() { + let root = TempDirGuard::new(); + fs::create_dir_all(root.path().join("node_modules/pkg")).unwrap(); + fs::write(root.path().join("node_modules/pkg/index.js"), "nm").unwrap(); + fs::write(root.path().join("real.txt"), "ok").unwrap(); + + let entries = super::collect_entries( + root.path(), + scan_options(true, false, crate::WalkDetail::Full), + ok_heartbeat, + ) + .unwrap(); + let crate::EntryScan::Entries(entries) = entries else { + panic!("fallback collection should return entries"); + }; + let entries = entries.entries; + let paths: Vec<&str> = entries.iter().map(|entry| entry.path.as_str()).collect(); + assert!( + !paths.iter().any(|path| path.contains("node_modules")), + "expected no node_modules entries, got: {paths:?}" + ); + assert!(paths.iter().any(|path| path == &"real.txt"), "expected real.txt, got: {paths:?}"); + } + + #[cfg(unix)] + #[test] + fn collect_entries_follow_links_uses_ignore_fallback() { + let root = TempDirGuard::new(); + fs::create_dir_all(root.path().join("target")).unwrap(); + fs::write(root.path().join("target/linked.txt"), "linked").unwrap(); + std::os::unix::fs::symlink(root.path().join("target"), root.path().join("link")).unwrap(); + + let mut options = scan_options(true, false, crate::WalkDetail::Minimal); + options.follow_links = crate::FollowLinks::Always; + + let entries = super::collect_entries(root.path(), options, ok_heartbeat).unwrap(); + let crate::EntryScan::Entries(entries) = entries else { + panic!("follow-links collection should fall back to ignore entries"); + }; + let paths: Vec<&str> = entries + .entries + .iter() + .map(|entry| entry.path.as_str()) + .collect(); + assert!( + paths.iter().any(|path| path == &"link/linked.txt"), + "follow-links fallback should yield symlink descendants, got: {paths:?}" + ); + } + + #[test] + fn traversal_gitignore_excludes_files() { + let root = TempDirGuard::new(); + fs::create_dir_all(root.path().join(".git")).unwrap(); + fs::write(root.path().join(".gitignore"), "ignored.txt\n").unwrap(); + fs::write(root.path().join("ignored.txt"), "ignored").unwrap(); + fs::write(root.path().join("kept.txt"), "keep").unwrap(); + + let collected = super::collect_entries( + root.path(), + scan_options(true, true, crate::WalkDetail::Full), + ok_heartbeat, + ) + .unwrap(); + let crate::EntryScan::Entries(collected) = collected else { + panic!("fallback collection should return entries"); + }; + let collected = collected.entries; + assert!( + !collected.iter().any(|entry| entry.path == "ignored.txt"), + "collect_entries returned gitignored file: {}", + entry_paths(&collected) + ); + assert_file_entry(&collected, "kept.txt", 4.0); + } + + #[test] + fn traversal_hidden_disabled_excludes_files_and_descendants() { + let root = TempDirGuard::new(); + fs::create_dir_all(root.path().join(".hidden-dir")).unwrap(); + fs::write(root.path().join(".hidden-dir/child.txt"), "child").unwrap(); + fs::write(root.path().join(".hidden-file"), "secret").unwrap(); + fs::write(root.path().join("visible.txt"), "visible").unwrap(); + + let entries = super::collect_entries( + root.path(), + scan_options(false, false, crate::WalkDetail::Full), + ok_heartbeat, + ) + .unwrap(); + let crate::EntryScan::Entries(entries) = entries else { + panic!("fallback collection should return entries"); + }; + let entries = entries.entries; + assert_eq!( + entries.len(), + 1, + "only visible.txt should be returned when hidden entries are disabled, got {}", + entry_paths(&entries) + ); + assert_file_entry(&entries, "visible.txt", 7.0); + assert!( + !entries + .iter() + .any(|entry| entry.path.starts_with(".hidden")), + "hidden entries should be pruned before yielding files or descendants, got {}", + entry_paths(&entries) + ); + } + + #[test] + fn traversal_hidden_enabled_includes_non_ignored_hidden_entries() { + let root = TempDirGuard::new(); + fs::create_dir_all(root.path().join(".git")).unwrap(); + fs::write(root.path().join(".gitignore"), ".ignored-hidden\n").unwrap(); + fs::create_dir_all(root.path().join(".hidden-dir")).unwrap(); + fs::write(root.path().join(".hidden-dir/child.txt"), "child").unwrap(); + fs::write(root.path().join(".hidden-file"), "secret").unwrap(); + fs::write(root.path().join(".ignored-hidden"), "ignored").unwrap(); + + let entries = super::collect_entries( + root.path(), + scan_options(true, true, crate::WalkDetail::Full), + ok_heartbeat, + ) + .unwrap(); + let crate::EntryScan::Entries(entries) = entries else { + panic!("fallback collection should return entries"); + }; + let entries = entries.entries; + assert_file_entry(&entries, ".hidden-file", 6.0); + assert_dir_entry(&entries, ".hidden-dir"); + assert_file_entry(&entries, ".hidden-dir/child.txt", 5.0); + assert!( + !entries.iter().any(|entry| entry.path == ".ignored-hidden"), + "gitignore should still exclude matching hidden files, got {}", + entry_paths(&entries) + ); + } + + #[test] + fn collect_entries_respects_pre_cancelled_token() { + let root = TempDirGuard::new(); + fs::write(root.path().join("real.txt"), "ok").unwrap(); + + std::thread::sleep(Duration::from_millis(1)); + let result = super::collect_entries( + root.path(), + scan_options(true, false, crate::WalkDetail::Minimal), + || Err("Timeout".to_string()), + ); + + let Err(err) = result else { + panic!("pre-cancelled scans should fail before returning entries"); + }; + assert!( + err.to_string().contains("Timeout"), + "expected timeout cancellation error, got: {err}" + ); + } + + #[test] + fn scan_detail_controls_metadata_collection() { + let root = TempDirGuard::new(); + fs::write(root.path().join("real.txt"), "ok").unwrap(); + + let minimal = super::collect_entries( + root.path(), + scan_options(true, false, crate::WalkDetail::Minimal), + ok_heartbeat, + ) + .unwrap(); + let crate::EntryScan::Entries(minimal) = minimal else { + panic!("fallback collection should return entries"); + }; + let minimal_file = minimal + .entries + .iter() + .find(|entry| entry.path == "real.txt") + .expect("minimal scan includes file"); + assert_eq!(minimal_file.mtime, None); + assert_eq!(minimal_file.size, None); + + let full = super::collect_entries( + root.path(), + scan_options(true, false, crate::WalkDetail::Full), + ok_heartbeat, + ) + .unwrap(); + let crate::EntryScan::Entries(full) = full else { + panic!("fallback collection should return entries"); + }; + let full_file = full + .entries + .iter() + .find(|entry| entry.path == "real.txt") + .expect("full scan includes file"); + assert!(full_file.mtime.is_some(), "full scan should include mtime"); + assert_eq!(full_file.size, Some(2.0)); + } +} diff --git a/crates/pi-walker/src/lib.rs b/crates/pi-walker/src/lib.rs new file mode 100644 index 000000000..fa9b910af --- /dev/null +++ b/crates/pi-walker/src/lib.rs @@ -0,0 +1,4037 @@ +//! Reusable platform directory traversal primitives. +//! +//! # Overview +//! `pi-walker` owns the native directory-read fast path that higher-level tools +//! use for globbing, grep candidate discovery, AST scans, and shell builtins. +//! The crate exposes plain Rust types, visitor interfaces, cache policy, and a +//! caller-supplied heartbeat so consumers do not inherit N-API dependencies. + +mod cache; + +use std::{ + borrow::Cow, + cmp::Ordering, + convert::Infallible, + ffi::{OsStr, OsString}, + fmt, + hash::{Hash, Hasher}, + io, + path::{Path, PathBuf}, + sync::{Arc, Mutex}, +}; + +pub use cache::{ + cache_ttl_ms, classify_file_type, contains_component, empty_recheck_ms, invalidate_all, + invalidate_path, invalidate_path_string, max_cache_entries, normalize_relative_path, + parallel_for_each, resolve_search_path, should_parallelize, should_skip_path, walk_workers, +}; +use globset::{GlobBuilder, GlobSet, GlobSetBuilder}; + +const HEARTBEAT_INTERVAL: usize = 128; + +/// Filesystem entry kind reported by the walker. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub enum FileType { + /// Regular file. + File, + /// Directory. + Dir, + /// Symbolic link. + Symlink, +} + +/// Amount of metadata to collect while reading directories. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub enum WalkDetail { + /// Collect only the entry name and file kind. + Minimal, + /// Also collect mtime and byte size for regular files. + Full, +} + +/// Traversal order for entries within each directory. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub enum WalkOrder { + /// Visit entries in the order returned by the platform API. + Unordered, + /// Sort entries by filename before visiting them. + Path, +} + +/// How directory-open errors are handled during traversal. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub enum DirectoryErrorMode { + /// Preserve the native glob fast-path contract: silently skip common race or + /// permission failures and fail on other directory errors. + SkipSkippable, + /// Deliver directory errors to [`EntryVisitor::visit_directory_error`] so + /// GNU-style consumers can report them and continue. + Visit, +} + +/// Symbolic-link traversal policy. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub enum FollowLinks { + /// Never follow symbolic links. + Never, + /// Follow root operands when they are symbolic links, but not descendants. + Roots, + /// Follow symbolic links at every depth. + Always, +} + +impl From for FollowLinks { + fn from(follow: bool) -> Self { + if follow { Self::Always } else { Self::Never } + } +} + +impl FollowLinks { + const fn follow_at_depth(self, depth: usize) -> bool { + match self { + Self::Never => false, + Self::Roots => depth == 0, + Self::Always => true, + } + } +} + +/// Shared cache use policy for high-level walk requests. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub enum CachePolicy { + /// Collect without using or updating the shared scan cache. + Disabled, + /// Use the shared scan cache for owned-entry collection. + Enabled, +} + +/// Empty cached-result revalidation policy for [`WalkRequest::collect`]. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub enum EmptyRecheck { + /// Never re-scan an empty cached result. + Never, + /// Re-scan empty cached results at or above the configured + /// [`empty_recheck_ms`] threshold. + Configured, + /// Re-scan empty cached results at or above this cache age. + AfterMillis(u64), +} + +/// Size metadata policy for high-level requests. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub enum SizeHintPolicy { + /// Preserve the request's [`WalkDetail`] setting. + FromDetail, + /// Request minimal metadata even on platforms with cheap size hints. + Never, + /// Request full metadata only when the platform exposes cheap file sizes. + WhenCheap, + /// Request full metadata for every yielded entry. + Always, +} + +/// Directory visit order for high-level requests. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub enum VisitOrder { + /// Yield a directory before its children. + PreOrder, + /// Yield a directory after its children when supported by the backend. + ContentsFirst, +} + +/// Concrete compiled glob filter for normalized walk-relative paths. +/// +/// Patterns are expected to already use the walker's normalized `/` separator +/// form. Each pattern is compiled with [`GlobBuilder::literal_separator`] so +/// wildcard matches never cross path separators. Equality and hashing use only +/// the normalized pattern list, not the compiled matcher internals, which keeps +/// [`WalkFilter`] suitable for static cacheable traversal policy. +#[derive(Clone)] +pub struct CompiledWalkGlob { + patterns: Arc<[String]>, + matcher: Arc, +} + +impl CompiledWalkGlob { + /// Compile normalized glob patterns for walk-relative paths. + pub fn new(patterns: I) -> Result + where + P: Into, + I: IntoIterator, + { + let mut normalized_patterns = Vec::new(); + let mut builder = GlobSetBuilder::new(); + for pattern in patterns { + let pattern = pattern.into(); + let glob = GlobBuilder::new(&pattern).literal_separator(true).build()?; + builder.add(glob); + normalized_patterns.push(pattern); + } + Ok(Self { patterns: normalized_patterns.into(), matcher: Arc::new(builder.build()?) }) + } + + /// Return whether `relative` matches any compiled pattern. + pub fn is_match(&self, relative: &str) -> bool { + self.matcher.is_match(relative) + } + + /// Return the normalized patterns backing this compiled filter. + pub fn patterns(&self) -> &[String] { + &self.patterns + } +} + +impl fmt::Debug for CompiledWalkGlob { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("CompiledWalkGlob") + .field("patterns", &self.patterns) + .finish() + } +} + +impl PartialEq for CompiledWalkGlob { + fn eq(&self, other: &Self) -> bool { + self.patterns == other.patterns + } +} + +impl Eq for CompiledWalkGlob {} + +impl Hash for CompiledWalkGlob { + fn hash(&self, state: &mut H) { + self.patterns.hash(state); + } +} + +/// High-level entry filter applied by collection and streaming APIs. +#[derive(Clone)] +pub struct WalkFilter { + kind: WalkFilterKind, + max_file_size: Option, + skip_node_modules_unless_seen: bool, + mentions_node_modules: bool, + glob: Option, +} + +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +enum WalkFilterKind { + All, + Files, + Dirs, +} + +impl Default for WalkFilter { + fn default() -> Self { + Self::all() + } +} + +impl fmt::Debug for WalkFilter { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("WalkFilter") + .field("kind", &self.kind) + .field("max_file_size", &self.max_file_size) + .field("skip_node_modules_unless_seen", &self.skip_node_modules_unless_seen) + .field("mentions_node_modules", &self.mentions_node_modules) + .field("glob", &self.glob) + .finish() + } +} + +impl PartialEq for WalkFilter { + fn eq(&self, other: &Self) -> bool { + self.kind == other.kind + && self.max_file_size == other.max_file_size + && self.skip_node_modules_unless_seen == other.skip_node_modules_unless_seen + && self.mentions_node_modules == other.mentions_node_modules + && self.glob == other.glob + } +} + +impl Eq for WalkFilter {} + +impl Hash for WalkFilter { + fn hash(&self, state: &mut H) { + self.kind.hash(state); + self.max_file_size.hash(state); + self.skip_node_modules_unless_seen.hash(state); + self.mentions_node_modules.hash(state); + self.glob.hash(state); + } +} + +impl WalkFilter { + /// Return a filter that accepts files and directories. + pub const fn all() -> Self { + Self { + kind: WalkFilterKind::All, + max_file_size: None, + skip_node_modules_unless_seen: false, + mentions_node_modules: false, + glob: None, + } + } + + /// Return a filter that emits only regular files. + pub const fn files_only() -> Self { + Self { + kind: WalkFilterKind::Files, + max_file_size: None, + skip_node_modules_unless_seen: false, + mentions_node_modules: false, + glob: None, + } + } + + /// Return a filter that emits only directories. + pub const fn dirs_only() -> Self { + Self { + kind: WalkFilterKind::Dirs, + max_file_size: None, + skip_node_modules_unless_seen: false, + mentions_node_modules: false, + glob: None, + } + } + + /// Limit emitted regular files to `max_file_size` bytes. + pub const fn max_file_size(mut self, max_file_size: u64) -> Self { + self.max_file_size = Some(max_file_size); + self + } + + /// Skip `node_modules` entries unless the caller's query mentioned them. + pub const fn node_modules_unless_mentioned(mut self, mentions_node_modules: bool) -> Self { + self.skip_node_modules_unless_seen = true; + self.mentions_node_modules = mentions_node_modules; + self + } + + /// Accept only entries whose normalized relative path matches `glob`. + pub fn glob(mut self, glob: CompiledWalkGlob) -> Self { + self.glob = Some(glob); + self + } + + fn accepts_path(&self, relative_path: &str) -> bool { + match &self.glob { + Some(glob) => glob.is_match(relative_path), + None => true, + } + } + + fn accepts_collected(&self, entry: &CollectedEntry) -> bool { + if self.skip_node_modules_unless_seen + && !self.mentions_node_modules + && entry + .path + .split('/') + .any(|component| component == "node_modules") + { + return false; + } + if self.max_file_size.is_some_and(|max| { + entry.file_type == FileType::File && entry.size.is_some_and(|size| size > max as f64) + }) { + return false; + } + let accepts_kind = match self.kind { + WalkFilterKind::All => true, + WalkFilterKind::Files => entry.file_type == FileType::File, + WalkFilterKind::Dirs => entry.file_type == FileType::Dir, + }; + accepts_kind && self.accepts_path(&entry.path) + } + + fn stream_decision(&self, meta: &EntryMeta<'_>) -> WalkDecision { + if self.skip_node_modules_unless_seen + && !self.mentions_node_modules + && meta + .relative_path + .split('/') + .any(|component| component == "node_modules") + { + return if meta.file_type == FileType::Dir { + WalkDecision::SkipDescend + } else { + WalkDecision::Skip + }; + } + if self.max_file_size.is_some_and(|max| { + meta.file_type == FileType::File && meta.size.is_some_and(|size| size > max as f64) + }) { + return WalkDecision::Skip; + } + match self.kind { + WalkFilterKind::All => {}, + WalkFilterKind::Files if meta.file_type == FileType::File => {}, + WalkFilterKind::Files => return WalkDecision::Skip, + WalkFilterKind::Dirs if meta.file_type == FileType::Dir => {}, + WalkFilterKind::Dirs => return WalkDecision::Skip, + } + if self.accepts_path(meta.relative_path) { + WalkDecision::Include + } else { + WalkDecision::Skip + } + } +} + +/// Traversal decision returned by [`WalkPredicate`] and closure streaming APIs. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub enum WalkDecision { + /// Emit this entry and continue traversal. + Include, + /// Do not emit this entry, but continue traversal. + Skip, + /// Do not emit this directory and do not descend into it. + SkipDescend, + /// Stop traversal immediately. + Stop, +} + +/// Predicate hook for dynamic walk consumers. +pub trait WalkPredicate { + /// Decide how the high-level walker should handle `entry`. + fn decide(&mut self, entry: &EntryMeta<'_>) -> WalkDecision; +} + +impl WalkPredicate for F +where + F: for<'entry, 'meta> FnMut(&'entry EntryMeta<'meta>) -> WalkDecision, +{ + fn decide(&mut self, entry: &EntryMeta<'_>) -> WalkDecision { + self(entry) + } +} + +#[derive(Clone, Copy, Debug, Default)] +struct IncludeAllPredicate; + +impl WalkPredicate for IncludeAllPredicate { + fn decide(&mut self, _entry: &EntryMeta<'_>) -> WalkDecision { + WalkDecision::Include + } +} + +/// Borrowed metadata view shared by owned and streaming entries. +pub struct EntryMeta<'a> { + /// Traversal root used to resolve relative paths. + pub root: &'a Path, + /// Absolute filesystem path. + pub absolute_path: Cow<'a, Path>, + /// Relative path from the root, using `/` separators. + pub relative_path: &'a str, + /// Filesystem entry kind. + pub file_type: FileType, + /// Modification time in milliseconds since the Unix epoch, when requested. + pub mtime: Option, + /// File size in bytes for regular files, when requested. + pub size: Option, + /// Depth below the traversal root. Root depth is 0. + pub depth: usize, +} + +impl<'a> EntryMeta<'a> { + /// Build metadata for an owned collected entry. + pub fn from_collected(root: &'a Path, entry: &'a CollectedEntry) -> Self { + Self { + root, + absolute_path: Cow::Owned(entry.absolute_path(root)), + relative_path: &entry.path, + file_type: entry.file_type, + mtime: entry.mtime, + size: entry.size, + depth: entry.depth(), + } + } + + /// Build metadata for a borrowed streaming entry. + pub const fn from_entry(root: &'a Path, entry: &Entry<'a>) -> Self { + Self { + root, + absolute_path: Cow::Borrowed(entry.path), + relative_path: entry.relative, + file_type: entry.file_type, + mtime: entry.mtime, + size: entry.size, + depth: entry.depth, + } + } +} + +/// Owned regular-file candidate returned by high-level file collection. +#[derive(Clone, Debug, PartialEq)] +pub struct FileCandidate { + /// Absolute filesystem path to the regular file. + pub path: PathBuf, + /// Relative path from the walk root, using `/` separators. + pub relative: String, + /// Modification time in milliseconds since the Unix epoch, when requested. + pub mtime: Option, + /// File size in bytes, when requested. + pub size: Option, +} + +impl FileCandidate { + fn from_entry(root: &Path, entry: CollectedEntry) -> Self { + Self { + path: entry.absolute_path(root), + relative: entry.path, + mtime: entry.mtime, + size: entry.size, + } + } +} + +/// Backend path used by a high-level collection. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub enum WalkBackend { + /// The request returned entries from a fresh backend scan. + Fresh, + /// The request returned entries from the shared cache. + Cached, +} + +/// Ranking applied to high-level collected entries after filtering. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub enum WalkRank { + /// Sort by normalized relative path in ascending byte order. + PathAsc, + /// Sort by modification time descending, then normalized relative path + /// ascending. + /// + /// Entries without modification times sort after entries with modification + /// times. [`WalkRequest::collect_ranked_with_heartbeat`] requests full + /// metadata for this rank so fresh scans can populate the mtime field. + MtimeDescPathAsc, +} + +/// Summary statistics for a high-level collection. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub struct WalkStats { + /// Age of the cache entry in milliseconds; zero means freshly scanned. + pub cache_age_ms: u64, + /// Entries before high-level filtering. + pub scanned_entries: usize, + /// Entries removed by high-level filtering. + pub filtered_entries: usize, + /// Entries removed by the high-level limit. + pub limited_entries: usize, +} + +/// Owned entries and metadata returned by [`WalkRequest::collect`]. +#[derive(Clone, Debug, PartialEq)] +pub struct WalkOutcome { + /// Entries after high-level filtering and limits. + pub entries: Vec, + /// Collection backend classification. + pub backend: WalkBackend, + /// Collection statistics. + pub stats: WalkStats, +} + +/// Options shared by native traversal consumers. +#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] +pub struct WalkOptions { + /// Include dot-prefixed entries. + pub include_hidden: bool, + /// Honor `.ignore`, `.gitignore`, repository excludes, and global gitignore. + pub use_gitignore: bool, + /// Prune `.git` directories during traversal. + pub skip_git: bool, + /// Prune `node_modules` directories during traversal. + pub skip_node_modules: bool, + /// Symbolic-link traversal policy. + pub follow_links: FollowLinks, + /// Metadata detail requested for each yielded entry. + pub detail: WalkDetail, + /// Per-directory visit order. + pub order: WalkOrder, + /// Yield the traversal root as a depth-0 entry before its children. + pub emit_root: bool, + /// Minimum depth yielded to the visitor. Root depth is 0. + pub min_depth: usize, + /// Maximum depth traversed and yielded. Root depth is 0. + pub max_depth: usize, + /// Yield directory entries after their children. Currently returns + /// unsupported so callers can use a post-order fallback. + pub contents_first: bool, + /// Directory-open error handling policy. + pub directory_errors: DirectoryErrorMode, + /// Stay on the root filesystem. Currently returns unsupported so callers can + /// use a device-aware fallback. + pub same_file_system: bool, + /// Use the shared scan cache when collecting owned entries. + pub cache: bool, +} + +impl Default for WalkOptions { + fn default() -> Self { + Self { + include_hidden: true, + use_gitignore: false, + skip_git: false, + skip_node_modules: false, + follow_links: FollowLinks::Never, + detail: WalkDetail::Minimal, + order: WalkOrder::Path, + emit_root: false, + min_depth: 1, + max_depth: usize::MAX, + contents_first: false, + directory_errors: DirectoryErrorMode::Visit, + same_file_system: false, + cache: false, + } + } +} + +/// High-level traversal request that owns a root and wraps [`WalkOptions`]. +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct WalkRequest { + root: PathBuf, + options: WalkOptions, + cache_policy: CachePolicy, + filter: WalkFilter, + limit: Option, + empty_recheck: EmptyRecheck, + visit_order: VisitOrder, + size_hint_policy: SizeHintPolicy, +} + +impl WalkRequest { + /// Create a request rooted at `root` with default [`WalkOptions`]. + pub fn new(root: impl Into) -> Self { + Self::from_options(root, WalkOptions::default()) + } + + /// Create a request from existing low-level options. + pub fn from_options(root: impl Into, options: WalkOptions) -> Self { + let cache_policy = if options.cache { + CachePolicy::Enabled + } else { + CachePolicy::Disabled + }; + let visit_order = if options.contents_first { + VisitOrder::ContentsFirst + } else { + VisitOrder::PreOrder + }; + Self { + root: root.into(), + options, + cache_policy, + filter: WalkFilter::default(), + limit: None, + empty_recheck: EmptyRecheck::Configured, + visit_order, + size_hint_policy: SizeHintPolicy::FromDetail, + } + } + + /// Return the traversal root. + pub fn root(&self) -> &Path { + &self.root + } + + /// Return low-level options after applying high-level policies. + pub const fn options(&self) -> WalkOptions { + self.effective_options() + } + + /// Include or exclude dot-prefixed entries. + pub const fn hidden(mut self, include_hidden: bool) -> Self { + self.options.include_hidden = include_hidden; + self + } + + /// Enable or disable `.ignore`/gitignore matching. + pub const fn gitignore(mut self, use_gitignore: bool) -> Self { + self.options.use_gitignore = use_gitignore; + self + } + + /// Enable or disable pruning `.git` directories. + pub const fn skip_git(mut self, skip_git: bool) -> Self { + self.options.skip_git = skip_git; + self + } + + /// Enable or disable pruning `node_modules` directories during traversal. + pub const fn skip_node_modules(mut self, skip_node_modules: bool) -> Self { + self.options.skip_node_modules = skip_node_modules; + self + } + + /// Set symbolic-link traversal policy. + pub const fn follow_links(mut self, follow_links: FollowLinks) -> Self { + self.options.follow_links = follow_links; + self + } + + /// Set metadata detail collected for entries. + pub const fn detail(mut self, detail: WalkDetail) -> Self { + self.options.detail = detail; + self + } + + /// Set per-directory entry order. + pub const fn order(mut self, order: WalkOrder) -> Self { + self.options.order = order; + self + } + + /// Enable or disable emitting the root entry. + pub const fn emit_root(mut self, emit_root: bool) -> Self { + self.options.emit_root = emit_root; + self + } + + /// Set minimum and maximum traversal depth. + pub const fn depth(mut self, min_depth: usize, max_depth: usize) -> Self { + self.options.min_depth = min_depth; + self.options.max_depth = max_depth; + self + } + + /// Set directory-open error handling. + pub const fn directory_errors(mut self, directory_errors: DirectoryErrorMode) -> Self { + self.options.directory_errors = directory_errors; + self + } + + /// Enable or disable staying on the root filesystem. + pub const fn same_file_system(mut self, same_file_system: bool) -> Self { + self.options.same_file_system = same_file_system; + self + } + + /// Enable or disable the shared scan cache for owned collection. + pub const fn cache(mut self, cache: bool) -> Self { + self.cache_policy = if cache { + CachePolicy::Enabled + } else { + CachePolicy::Disabled + }; + self.options.cache = cache; + self + } + + /// Set the static high-level filter. + pub fn filter(mut self, filter: WalkFilter) -> Self { + self.filter = filter; + self + } + + /// Limit the number of emitted entries after filtering. + pub const fn limit(mut self, limit: usize) -> Self { + self.limit = Some(limit); + self + } + + /// Remove any high-level entry limit. + pub const fn no_limit(mut self) -> Self { + self.limit = None; + self + } + + /// Set empty cached-result revalidation policy. + pub const fn empty_recheck(mut self, empty_recheck: EmptyRecheck) -> Self { + self.empty_recheck = empty_recheck; + self + } + + /// Set high-level directory visit order. + pub const fn visit_order(mut self, visit_order: VisitOrder) -> Self { + self.visit_order = visit_order; + self.options.contents_first = matches!(visit_order, VisitOrder::ContentsFirst); + self + } + + /// Set size metadata policy. + pub const fn size_hints(mut self, size_hint_policy: SizeHintPolicy) -> Self { + self.size_hint_policy = size_hint_policy; + self + } + + /// Collect owned entries, then apply high-level filters, limits, and + /// empty-cache rechecks. + pub fn collect(&self) -> std::result::Result> { + self.collect_with_heartbeat(|| Ok::<(), Infallible>(())) + } + + /// Collect owned entries with a caller-supplied heartbeat. + pub fn collect_with_heartbeat( + &self, + heartbeat: H, + ) -> std::result::Result> + where + H: Fn() -> std::result::Result<(), E> + Sync, + E: fmt::Display, + { + self.collect_with_rank_and_limit(None, self.limit, heartbeat) + } + + /// Collect owned entries, apply high-level filters, rank, then truncate to + /// `limit`. + /// + /// The request's stored [`WalkRequest::limit`] is intentionally not applied + /// before ranking; `limit` is the top-N bound for this ranked collection. + pub fn collect_ranked( + &self, + rank: WalkRank, + limit: usize, + ) -> std::result::Result> { + self.collect_ranked_with_heartbeat(rank, limit, || Ok::<(), Infallible>(())) + } + + /// Collect owned entries with a caller-supplied heartbeat, apply high-level + /// filters, rank, then truncate to `limit`. + /// + /// Ranking happens after all high-level filters and before truncation so + /// top-N callers observe the best matching entries, not the first traversed + /// entries. + pub fn collect_ranked_with_heartbeat( + &self, + rank: WalkRank, + limit: usize, + heartbeat: H, + ) -> std::result::Result> + where + H: Fn() -> std::result::Result<(), E> + Sync, + E: fmt::Display, + { + self.collect_with_rank_and_limit(Some(rank), Some(limit), heartbeat) + } + + /// Collect regular files accepted by this request. + pub fn collect_files(&self) -> std::result::Result, WalkError> { + self.collect_files_with_heartbeat(|| Ok::<(), Infallible>(())) + } + + /// Collect regular files accepted by this request with a caller-supplied + /// heartbeat. + pub fn collect_files_with_heartbeat( + &self, + heartbeat: H, + ) -> std::result::Result, WalkError> + where + H: Fn() -> std::result::Result<(), E> + Sync, + E: fmt::Display, + { + let outcome = self.collect_with_heartbeat(heartbeat)?; + Ok(outcome + .entries + .into_iter() + .filter(CollectedEntry::is_file) + .collect()) + } + + /// Collect directories accepted by this request. + pub fn collect_dirs(&self) -> std::result::Result, WalkError> { + self.collect_dirs_with_heartbeat(|| Ok::<(), Infallible>(())) + } + + /// Collect directories accepted by this request with a caller-supplied + /// heartbeat. + pub fn collect_dirs_with_heartbeat( + &self, + heartbeat: H, + ) -> std::result::Result, WalkError> + where + H: Fn() -> std::result::Result<(), E> + Sync, + E: fmt::Display, + { + let outcome = self.collect_with_heartbeat(heartbeat)?; + Ok(outcome + .entries + .into_iter() + .filter(CollectedEntry::is_dir) + .collect()) + } + + /// Collect regular-file candidates accepted by this request. + pub fn collect_file_candidates( + &self, + ) -> std::result::Result, WalkError> { + self.collect_file_candidates_with_heartbeat(|| Ok::<(), Infallible>(())) + } + + /// Collect regular-file candidates accepted by this request with a + /// caller-supplied heartbeat. + pub fn collect_file_candidates_with_heartbeat( + &self, + heartbeat: H, + ) -> std::result::Result, WalkError> + where + H: Fn() -> std::result::Result<(), E> + Sync, + E: fmt::Display, + { + Ok(self + .collect_file_candidates_with_stats_with_heartbeat(heartbeat)? + .0) + } + + /// Stream entries through `visitor` after applying high-level filters and + /// limits. + pub fn stream(&self, visitor: &mut V) -> std::result::Result> + where + V: EntryVisitor, + { + self.stream_with_heartbeat(visitor, || Ok::<(), V::Error>(())) + } + + /// Stream entries through `visitor` with a caller-supplied heartbeat. + pub fn stream_with_heartbeat( + &self, + visitor: &mut V, + heartbeat: H, + ) -> std::result::Result> + where + V: EntryVisitor, + H: FnMut() -> std::result::Result<(), V::Error>, + { + self.stream_with_predicate_and_heartbeat(visitor, IncludeAllPredicate, heartbeat) + } + + /// Stream accepted entries through a closure after applying high-level + /// filters and limits. + /// + /// This is the closure-based counterpart to [`WalkRequest::stream`]. The + /// closure receives borrowed [`EntryMeta`] for each entry that passes this + /// request's static [`WalkFilter`] and dynamic request limit, then returns a + /// [`WalkDecision`] to control traversal. [`WalkDecision::Include`] and + /// [`WalkDecision::Skip`] both continue because the entry has already been + /// delivered to the closure; [`WalkDecision::SkipDescend`] prunes the + /// current directory's descendants, and [`WalkDecision::Stop`] stops + /// traversal. + /// + /// Directory-open errors are ignored and traversal continues when + /// [`WalkOptions::directory_errors`] is [`DirectoryErrorMode::Visit`]. Use + /// [`WalkRequest::for_each_entry_with_heartbeat`] to observe those errors or + /// provide a heartbeat. + pub fn for_each_entry(&self, visit: V) -> std::result::Result> + where + V: for<'entry> FnMut(EntryMeta<'entry>) -> std::result::Result, + { + self.for_each_entry_with_heartbeat( + || Ok::<(), E>(()), + visit, + |_| Ok::(WalkDecision::Include), + ) + } + + /// Stream accepted entries through closures with a caller-supplied + /// heartbeat. + /// + /// `heartbeat` is invoked by the same traversal machinery used by + /// [`WalkRequest::stream_with_heartbeat`]. `visit` receives each accepted + /// [`EntryMeta`] and returns a [`WalkDecision`]: [`WalkDecision::Include`] and + /// [`WalkDecision::Skip`] continue, [`WalkDecision::SkipDescend`] skips + /// descendants of the current directory, and [`WalkDecision::Stop`] stops + /// the walk. `directory_error` receives [`DirectoryError`] values when the + /// request's [`WalkOptions::directory_errors`] is + /// [`DirectoryErrorMode::Visit`] and returns the same traversal decision, + /// letting GNU-style consumers report an error and continue or stop without + /// implementing [`EntryVisitor`]. + pub fn for_each_entry_with_heartbeat( + &self, + heartbeat: H, + visit: V, + directory_error: D, + ) -> std::result::Result> + where + H: FnMut() -> std::result::Result<(), E>, + V: for<'entry> FnMut(EntryMeta<'entry>) -> std::result::Result, + D: for<'error> FnMut(DirectoryError<'error>) -> std::result::Result, + { + let mut visitor = ClosureEntryVisitor { root: &self.root, visit, directory_error }; + self.stream_with_heartbeat(&mut visitor, heartbeat) + } + + /// Stream entries through `visitor` with an additional dynamic predicate. + pub fn stream_with_predicate( + &self, + visitor: &mut V, + predicate: P, + ) -> std::result::Result> + where + V: EntryVisitor, + P: WalkPredicate, + { + self.stream_with_predicate_and_heartbeat(visitor, predicate, || Ok::<(), V::Error>(())) + } + + /// Stream entries through `visitor` with a dynamic predicate and + /// caller-supplied heartbeat. + pub fn stream_with_predicate_and_heartbeat( + &self, + visitor: &mut V, + predicate: P, + mut heartbeat: H, + ) -> std::result::Result> + where + V: EntryVisitor, + P: WalkPredicate, + H: FnMut() -> std::result::Result<(), V::Error>, + { + let options = self.effective_options(); + let mut adapter = RequestVisitor { + root: &self.root, + filter: &self.filter, + limit: self.limit, + emitted: 0, + visitor, + predicate, + }; + match walk_entries(&self.root, options, &mut adapter, &mut heartbeat)? { + WalkStatus::Unsupported if Self::uses_collected_fallback(options) => { + let RequestVisitor { visitor, predicate, .. } = adapter; + self.stream_collected_fallback(visitor, predicate, heartbeat, options) + }, + status => Ok(status), + } + } + + /// Run `operation` for each accepted regular file. + pub fn for_each_file( + &self, + operation: impl Fn(&Path) -> std::result::Result<(), E> + Send + Sync, + ) -> std::result::Result> + where + E: fmt::Display + Send, + { + self.for_each_file_with_heartbeat(operation, || Ok::<(), Infallible>(())) + } + + /// Run `operation` for each accepted regular file with a caller-supplied + /// heartbeat. + pub fn for_each_file_with_heartbeat( + &self, + operation: impl Fn(&Path) -> std::result::Result<(), E> + Send + Sync, + heartbeat: H, + ) -> std::result::Result> + where + E: fmt::Display + Send, + H: Fn() -> std::result::Result<(), HE> + Sync, + HE: fmt::Display, + { + self.for_each_file_candidate_with_heartbeat(|candidate| operation(&candidate.path), heartbeat) + } + + /// Run `operation` for each accepted regular-file candidate. + pub fn for_each_file_candidate( + &self, + operation: impl Fn(&FileCandidate) -> std::result::Result<(), E> + Send + Sync, + ) -> std::result::Result> + where + E: fmt::Display + Send, + { + self.for_each_file_candidate_with_heartbeat(operation, || Ok::<(), Infallible>(())) + } + + /// Run `operation` for each accepted regular-file candidate with a + /// caller-supplied heartbeat. + pub fn for_each_file_candidate_with_heartbeat( + &self, + operation: impl Fn(&FileCandidate) -> std::result::Result<(), E> + Send + Sync, + heartbeat: H, + ) -> std::result::Result> + where + E: fmt::Display + Send, + H: Fn() -> std::result::Result<(), HE> + Sync, + HE: fmt::Display, + { + let (candidates, stats) = + self.collect_file_candidates_with_stats_with_heartbeat(heartbeat)?; + execute_candidates(&candidates, operation) + .map_err(|err| WalkError::Interrupted(err.to_string()))?; + Ok(stats) + } + + fn collect_with_rank_and_limit( + &self, + rank: Option, + limit: Option, + heartbeat: H, + ) -> std::result::Result> + where + H: Fn() -> std::result::Result<(), E> + Sync, + E: fmt::Display, + { + let mut options = self.effective_options(); + if matches!(rank, Some(WalkRank::MtimeDescPathAsc)) { + options.detail = WalkDetail::Full; + } + let mut scan = self.collect_entries_with_options(options, &heartbeat)?; + let mut backend = if scan.cache_age_ms == 0 { + WalkBackend::Fresh + } else { + WalkBackend::Cached + }; + let filter_entries = |entries: &mut Vec| { + let scanned_entries = entries.len(); + entries.retain(|entry| self.filter.accepts_collected(entry)); + (scanned_entries, scanned_entries - entries.len()) + }; + let (mut scanned_entries, mut filtered_entries) = filter_entries(&mut scan.entries); + if scan.entries.is_empty() && self.should_recheck_empty(scan.cache_age_ms) { + options.cache = false; + scan = self.collect_entries_with_options(options, &heartbeat)?; + backend = WalkBackend::Fresh; + (scanned_entries, filtered_entries) = filter_entries(&mut scan.entries); + } + if let Some(rank) = rank { + Self::rank_entries(&mut scan.entries, rank); + } + let limited_entries = if let Some(limit) = limit { + let limited_entries = scan.entries.len().saturating_sub(limit); + scan.entries.truncate(limit); + limited_entries + } else { + 0 + }; + let stats = WalkStats { + cache_age_ms: scan.cache_age_ms, + scanned_entries, + filtered_entries, + limited_entries, + }; + Ok(WalkOutcome { entries: scan.entries, backend, stats }) + } + + fn rank_entries(entries: &mut [CollectedEntry], rank: WalkRank) { + match rank { + WalkRank::PathAsc => entries.sort_by(|left, right| left.path.cmp(&right.path)), + WalkRank::MtimeDescPathAsc => entries.sort_by(Self::compare_mtime_desc_path_asc), + } + } + + fn compare_mtime_desc_path_asc(left: &CollectedEntry, right: &CollectedEntry) -> Ordering { + let mtime_order = match (left.mtime, right.mtime) { + (Some(left_mtime), Some(right_mtime)) => right_mtime.total_cmp(&left_mtime), + (Some(_), None) => Ordering::Less, + (None, Some(_)) => Ordering::Greater, + (None, None) => Ordering::Equal, + }; + mtime_order.then_with(|| left.path.cmp(&right.path)) + } + + const fn effective_options(&self) -> WalkOptions { + let mut options = self.options; + options.cache = matches!(self.cache_policy, CachePolicy::Enabled); + options.contents_first = matches!(self.visit_order, VisitOrder::ContentsFirst); + match self.size_hint_policy { + SizeHintPolicy::FromDetail => {}, + SizeHintPolicy::Never => options.detail = WalkDetail::Minimal, + SizeHintPolicy::WhenCheap => { + options.detail = if supports_cheap_size_hints() { + WalkDetail::Full + } else { + WalkDetail::Minimal + }; + }, + SizeHintPolicy::Always => options.detail = WalkDetail::Full, + } + if self.filter.max_file_size.is_some() { + options.detail = WalkDetail::Full; + } + options + } + + fn collect_entries_with_options( + &self, + options: WalkOptions, + heartbeat: &H, + ) -> std::result::Result> + where + H: Fn() -> std::result::Result<(), E> + Sync, + E: fmt::Display, + { + match collect_entries(&self.root, options, heartbeat) { + Ok(EntryScan::Entries(scan)) => Ok(scan), + Ok(EntryScan::Unsupported) | Err(WalkError::Unsupported) + if Self::uses_collected_fallback(options) => + { + self.collect_entries_fallback(options, || heartbeat().map_err(|err| err.to_string())) + }, + Ok(EntryScan::Unsupported) | Err(WalkError::Unsupported) => Err(WalkError::Unsupported), + Err(err) => Err(err), + } + } + + const fn uses_collected_fallback(options: WalkOptions) -> bool { + options.min_depth <= options.max_depth && (options.contents_first || options.same_file_system) + } + + fn collect_entries_fallback( + &self, + options: WalkOptions, + mut heartbeat: H, + ) -> std::result::Result> + where + H: FnMut() -> std::result::Result<(), String>, + { + let mut collector = FallbackCollector::new(&self.root, options); + let status = match walk_entries( + &self.root, + fallback_walk_options(options), + &mut collector, + &mut heartbeat, + ) { + Ok(status) => status, + Err(WalkError::Unsupported) => WalkStatus::Unsupported, + Err(err) => return Err(err), + }; + let mut entries = match status { + WalkStatus::Unsupported => { + let mut collector = FallbackCollector::new(&self.root, options); + let status = walk_entries_with_ignore( + &self.root, + fallback_walk_options(options), + &mut collector, + &mut heartbeat, + )?; + if matches!(status, WalkStatus::Unsupported) { + return Err(WalkError::Unsupported); + } + collector.into_entries() + }, + WalkStatus::Complete | WalkStatus::Stopped => collector.into_entries(), + }; + entries = filter_collected_same_file_system(&self.root, entries, options, &mut heartbeat)?; + if options.contents_first { + sort_collected_depth_first(&mut entries); + } + Ok(CollectedEntries { entries, cache_age_ms: 0 }) + } + + fn stream_collected_fallback( + &self, + visitor: &mut V, + mut predicate: P, + mut heartbeat: H, + options: WalkOptions, + ) -> std::result::Result> + where + V: EntryVisitor, + P: WalkPredicate, + H: FnMut() -> std::result::Result<(), V::Error>, + { + // The collected replay path cannot interleave content-first entry replay with + // directory-open errors, so directory errors are forwarded during collection + // before the accepted entries are replayed below. + let mut collector = FallbackStreamCollector::new(&self.root, options, visitor); + let status = match walk_entries( + &self.root, + fallback_walk_options(options), + &mut collector, + &mut heartbeat, + ) { + Ok(status) => status, + Err(WalkError::Unsupported) => WalkStatus::Unsupported, + Err(err) => return Err(err), + }; + let (mut entries, visitor) = match status { + WalkStatus::Unsupported => { + let (_entries, visitor) = collector.into_parts(); + let mut collector = FallbackStreamCollector::new(&self.root, options, visitor); + let status = walk_entries_with_ignore( + &self.root, + fallback_walk_options(options), + &mut collector, + &mut heartbeat, + )?; + if matches!(status, WalkStatus::Unsupported | WalkStatus::Stopped) { + return Ok(status); + } + collector.into_parts() + }, + WalkStatus::Stopped => return Ok(WalkStatus::Stopped), + WalkStatus::Complete => collector.into_parts(), + }; + entries = filter_collected_same_file_system(&self.root, entries, options, &mut heartbeat)?; + let mut accepted = Vec::new(); + let mut pruned_dirs = Vec::new(); + let mut stopped = false; + let mut visited = 0usize; + for entry in entries { + if visited == 0 || visited >= HEARTBEAT_INTERVAL { + visited = 0; + heartbeat().map_err(WalkError::Interrupted)?; + } + visited += 1; + if is_under_pruned_relative_dir(&entry.path, &pruned_dirs) { + continue; + } + let meta = EntryMeta { + root: &self.root, + absolute_path: Cow::Owned(entry.absolute_path(&self.root)), + relative_path: &entry.path, + file_type: entry.file_type, + mtime: entry.mtime, + size: entry.size, + depth: entry.depth(), + }; + match self.filter.stream_decision(&meta) { + WalkDecision::Include => {}, + WalkDecision::Skip => continue, + WalkDecision::SkipDescend => { + if entry.is_dir() { + pruned_dirs.push(entry.path.clone()); + } + continue; + }, + WalkDecision::Stop => { + stopped = true; + break; + }, + } + match predicate.decide(&meta) { + WalkDecision::Include => accepted.push(entry), + WalkDecision::Skip => {}, + WalkDecision::SkipDescend => { + if entry.is_dir() { + pruned_dirs.push(entry.path.clone()); + } + }, + WalkDecision::Stop => { + stopped = true; + + break; + }, + } + } + + if matches!(self.visit_order, VisitOrder::ContentsFirst) { + sort_collected_depth_first(&mut accepted); + } + let replay_status = self.replay_collected_entries(visitor, &accepted, &mut heartbeat)?; + if replay_status == WalkStatus::Stopped || stopped { + Ok(WalkStatus::Stopped) + } else { + Ok(WalkStatus::Complete) + } + } + + fn replay_collected_entries( + &self, + visitor: &mut V, + entries: &[CollectedEntry], + heartbeat: &mut H, + ) -> std::result::Result> + where + V: EntryVisitor, + H: FnMut() -> std::result::Result<(), V::Error>, + { + let mut emitted = 0usize; + let mut visited = 0usize; + let mut pruned_dirs = Vec::new(); + for entry in entries { + if self.limit.is_some_and(|limit| emitted >= limit) { + return Ok(WalkStatus::Stopped); + } + if visited == 0 || visited >= HEARTBEAT_INTERVAL { + visited = 0; + heartbeat().map_err(WalkError::Interrupted)?; + } + visited += 1; + if is_under_pruned_relative_dir(&entry.path, &pruned_dirs) { + continue; + } + match replay_collected_entry(&self.root, entry, visitor).map_err(WalkError::Interrupted)? { + WalkControl::Quit => return Ok(WalkStatus::Stopped), + WalkControl::SkipDescend => { + if entry.is_dir() { + pruned_dirs.push(entry.path.clone()); + } + }, + WalkControl::Continue => {}, + } + emitted += 1; + } + Ok(WalkStatus::Complete) + } + + fn should_recheck_empty(&self, cache_age_ms: u64) -> bool { + if cache_age_ms == 0 { + return false; + } + match self.empty_recheck { + EmptyRecheck::Never => false, + EmptyRecheck::Configured => { + let threshold = empty_recheck_ms(); + threshold > 0 && cache_age_ms >= threshold + }, + EmptyRecheck::AfterMillis(threshold) => cache_age_ms >= threshold, + } + } + + fn collect_file_candidates_with_stats_with_heartbeat( + &self, + heartbeat: H, + ) -> std::result::Result<(Vec, WalkStats), WalkError> + where + H: Fn() -> std::result::Result<(), E> + Sync, + E: fmt::Display, + { + let outcome = self.collect_with_heartbeat(heartbeat)?; + let candidates = outcome + .entries + .into_iter() + .filter(CollectedEntry::is_file) + .map(|entry| FileCandidate::from_entry(&self.root, entry)) + .collect(); + Ok((candidates, outcome.stats)) + } +} + +/// Execute work for regular-file candidates using the centralized walker pool. +pub fn execute_candidates( + candidates: &[FileCandidate], + operation: impl Fn(&FileCandidate) -> std::result::Result<(), E> + Send + Sync, +) -> std::result::Result<(), E> +where + E: Send, +{ + parallel_for_each(candidates, operation) +} + +/// Visitor decision for streaming traversal. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum WalkControl { + /// Continue traversing remaining entries. + Continue, + /// Skip descending into this directory entry. + SkipDescend, + /// Stop traversal immediately. + Quit, +} + +/// Status returned by native traversal. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum WalkStatus { + /// The current platform or option set cannot use the native scanner safely. + Unsupported, + /// Traversal visited every reachable entry. + Complete, + /// The visitor stopped traversal early. + Stopped, +} + +/// Owned entry returned by [`collect_entries`]. +#[derive(Clone, Debug, PartialEq)] +pub struct CollectedEntry { + /// Relative path from the root, using `/` separators. + pub path: String, + /// Filesystem entry kind. + pub file_type: FileType, + /// Modification time in milliseconds since the Unix epoch, when requested. + pub mtime: Option, + /// File size in bytes for regular files, when requested. + pub size: Option, +} + +impl CollectedEntry { + /// Return this entry's absolute path under `root`. + pub fn absolute_path(&self, root: &Path) -> PathBuf { + if self.path.is_empty() { + root.to_path_buf() + } else { + root.join(&self.path) + } + } + + /// Return this entry's depth below the traversal root. + pub fn depth(&self) -> usize { + if self.path.is_empty() { + 0 + } else { + self + .path + .split('/') + .filter(|component| !component.is_empty()) + .count() + } + } + + /// Return whether this entry is a regular file. + pub const fn is_file(&self) -> bool { + matches!(self.file_type, FileType::File) + } + + /// Return whether this entry is a directory. + pub const fn is_dir(&self) -> bool { + matches!(self.file_type, FileType::Dir) + } +} + +/// Return whether `parent` is a walk-relative ancestor of `child`. +/// +/// Both paths use the walker's normalized `/` separator. The root-relative +/// empty path is an ancestor of every non-root entry. +pub fn is_relative_ancestor(parent: &str, child: &str) -> bool { + if parent == child { + return false; + } + if parent.is_empty() { + return !child.is_empty(); + } + child + .strip_prefix(parent) + .is_some_and(|suffix| suffix.starts_with('/')) +} + +/// Compare normalized walk-relative paths in depth-first, +/// contents-before-parent order. +pub fn compare_depth_first_paths(left: &str, right: &str) -> Ordering { + if left == right { + return Ordering::Equal; + } + if is_relative_ancestor(left, right) { + return Ordering::Greater; + } + if is_relative_ancestor(right, left) { + return Ordering::Less; + } + left.cmp(right) +} + +/// Sort collected entries in depth-first, contents-before-parent path order. +pub fn sort_collected_depth_first(entries: &mut [CollectedEntry]) { + entries.sort_unstable_by(|left, right| compare_depth_first_paths(&left.path, &right.path)); +} + +/// Return whether `relative` is below any pruned normalized directory path. +pub fn is_under_pruned_relative_dir(relative: &str, pruned_dirs: &[String]) -> bool { + pruned_dirs + .iter() + .any(|dir| is_relative_ancestor(dir, relative)) +} + +/// Return the root device id used by same-filesystem traversal filters. +/// +/// Non-Unix platforms return `None`, making same-filesystem filtering a no-op. +#[cfg(unix)] +pub fn root_device_id(path: &Path, follow_links: FollowLinks) -> Option { + use std::os::unix::fs::MetadataExt; + + metadata_for_follow_policy(path, follow_links.follow_at_depth(0)) + .ok() + .map(|metadata| metadata.dev()) +} + +/// Return the root device id used by same-filesystem traversal filters. +/// +/// Non-Unix platforms return `None`, making same-filesystem filtering a no-op. +#[cfg(not(unix))] +pub fn root_device_id(_path: &Path, _follow_links: FollowLinks) -> Option { + None +} + +/// Return whether `path` is on the root filesystem represented by +/// `root_device`. +/// +/// When `root_device` is `None`, this returns true. On non-Unix platforms this +/// is always true, matching the existing no-op same-filesystem behavior there. +#[cfg(unix)] +pub fn is_path_on_root_file_system( + path: &Path, + depth: usize, + follow_links: FollowLinks, + root_device: Option, +) -> bool { + use std::os::unix::fs::MetadataExt; + + let Some(root_device) = root_device else { + return true; + }; + metadata_for_follow_policy(path, follow_links.follow_at_depth(depth)) + .is_ok_and(|metadata| metadata.dev() == root_device) +} + +/// Return whether `path` is on the root filesystem represented by +/// `root_device`. +/// +/// When `root_device` is `None`, this returns true. On non-Unix platforms this +/// is always true, matching the existing no-op same-filesystem behavior there. +#[cfg(not(unix))] +pub fn is_path_on_root_file_system( + _path: &Path, + _depth: usize, + _follow_links: FollowLinks, + _root_device: Option, +) -> bool { + true +} + +#[cfg(unix)] +fn metadata_for_follow_policy(path: &Path, follow: bool) -> io::Result { + if follow { + std::fs::metadata(path) + } else { + std::fs::symlink_metadata(path) + } +} + +/// Owned entries returned by [`collect_entries`] plus cache metadata. +#[derive(Clone, Debug, PartialEq)] +pub struct CollectedEntries { + /// Entries collected with the requested traversal contract. + pub entries: Vec, + /// Age of the cache entry in milliseconds; zero means freshly scanned. + pub cache_age_ms: u64, +} + +/// Result of attempting a native directory scan. +#[derive(Clone, Debug, PartialEq)] +pub enum EntryScan { + /// The native scanner is unavailable for this platform or option set. + Unsupported, + /// Entries collected with the requested traversal contract. + Entries(CollectedEntries), +} + +/// Borrowed entry passed to [`EntryVisitor`]. +pub struct Entry<'a> { + /// Absolute filesystem path for this entry. + pub path: &'a Path, + /// Relative path from the root, using `/` separators. + pub relative: &'a str, + /// Basename as returned by the platform directory API. + pub name: &'a OsStr, + /// Filesystem entry kind. + pub file_type: FileType, + /// Modification time in milliseconds since the Unix epoch, when requested. + pub mtime: Option, + /// File size in bytes for regular files, when requested. + pub size: Option, + /// Depth below the traversal root. Direct children have depth 1. + pub depth: usize, +} + +/// Directory-open error delivered to visitors when configured by +/// [`WalkOptions::directory_errors`]. +pub struct DirectoryError<'a> { + /// Directory path that could not be read. + pub path: &'a Path, + /// Underlying platform I/O error. + pub error: &'a io::Error, +} + +/// Consumer hook invoked for every accepted entry. +pub trait EntryVisitor { + /// Error type used by visitor and heartbeat callbacks. + type Error; + + /// Visit one filesystem entry and choose how traversal continues. + fn visit(&mut self, entry: Entry<'_>) -> std::result::Result; + + /// Handle a directory-open error and choose whether traversal continues. + fn visit_directory_error( + &mut self, + _error: DirectoryError<'_>, + ) -> std::result::Result { + Ok(WalkControl::Continue) + } +} + +struct RequestVisitor<'a, V, P> { + root: &'a Path, + filter: &'a WalkFilter, + limit: Option, + emitted: usize, + visitor: &'a mut V, + predicate: P, +} + +impl EntryVisitor for RequestVisitor<'_, V, P> +where + V: EntryVisitor, + P: WalkPredicate, +{ + type Error = V::Error; + + fn visit(&mut self, entry: Entry<'_>) -> std::result::Result { + if self.limit.is_some_and(|limit| self.emitted >= limit) { + return Ok(WalkControl::Quit); + } + let meta = EntryMeta { + root: self.root, + absolute_path: Cow::Borrowed(entry.path), + relative_path: entry.relative, + file_type: entry.file_type, + mtime: entry.mtime, + size: entry.size, + depth: entry.depth, + }; + match self.filter.stream_decision(&meta) { + WalkDecision::Include => {}, + WalkDecision::Skip => return Ok(WalkControl::Continue), + WalkDecision::SkipDescend => return Ok(WalkControl::SkipDescend), + WalkDecision::Stop => return Ok(WalkControl::Quit), + } + match self.predicate.decide(&meta) { + WalkDecision::Include => {}, + WalkDecision::Skip => return Ok(WalkControl::Continue), + WalkDecision::SkipDescend => return Ok(WalkControl::SkipDescend), + WalkDecision::Stop => return Ok(WalkControl::Quit), + } + self.emitted += 1; + self.visitor.visit(entry) + } + + fn visit_directory_error( + &mut self, + error: DirectoryError<'_>, + ) -> std::result::Result { + self.visitor.visit_directory_error(error) + } +} + +struct ClosureEntryVisitor<'a, V, D> { + root: &'a Path, + visit: V, + directory_error: D, +} + +impl EntryVisitor for ClosureEntryVisitor<'_, V, D> +where + V: for<'entry> FnMut(EntryMeta<'entry>) -> std::result::Result, + D: for<'error> FnMut(DirectoryError<'error>) -> std::result::Result, +{ + type Error = E; + + fn visit(&mut self, entry: Entry<'_>) -> std::result::Result { + let meta = EntryMeta { + root: self.root, + absolute_path: Cow::Borrowed(entry.path), + relative_path: entry.relative, + file_type: entry.file_type, + mtime: entry.mtime, + size: entry.size, + depth: entry.depth, + }; + (self.visit)(meta).map(walk_decision_to_control) + } + + fn visit_directory_error( + &mut self, + error: DirectoryError<'_>, + ) -> std::result::Result { + (self.directory_error)(error).map(walk_decision_to_control) + } +} + +const fn walk_decision_to_control(decision: WalkDecision) -> WalkControl { + match decision { + WalkDecision::Include | WalkDecision::Skip => WalkControl::Continue, + WalkDecision::SkipDescend => WalkControl::SkipDescend, + WalkDecision::Stop => WalkControl::Quit, + } +} + +/// Error returned by native traversal. +#[derive(Debug)] +pub enum WalkError { + /// The native scanner is unavailable for this platform or option set. + Unsupported, + /// A caller-supplied heartbeat or visitor returned an error. + Interrupted(E), + /// A platform directory API returned malformed data or an unskippable error. + InvalidData { + /// Directory whose scan failed. + path: PathBuf, + /// Human-readable failure detail. + message: String, + }, +} + +impl fmt::Display for WalkError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::Unsupported => f.write_str("native directory scan unsupported"), + Self::Interrupted(err) => write!(f, "native directory scan interrupted: {err}"), + Self::InvalidData { path, message } => { + write!(f, "native directory scan failed for {}: {message}", path.display()) + }, + } + } +} + +impl std::error::Error for WalkError +where + E: std::error::Error + 'static, +{ + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + match self { + Self::Interrupted(err) => Some(err), + Self::Unsupported | Self::InvalidData { .. } => None, + } + } +} + +struct CollectVisitor { + entries: Vec, + _error: std::marker::PhantomData E>, +} + +impl CollectVisitor { + const fn new() -> Self { + Self { entries: Vec::new(), _error: std::marker::PhantomData } + } +} + +impl EntryVisitor for CollectVisitor { + type Error = E; + + fn visit(&mut self, entry: Entry<'_>) -> std::result::Result { + self.entries.push(CollectedEntry { + path: entry.relative.to_string(), + file_type: entry.file_type, + mtime: entry.mtime, + size: entry.size, + }); + Ok(WalkControl::Continue) + } +} + +const fn fallback_walk_options(mut options: WalkOptions) -> WalkOptions { + options.contents_first = false; + options.same_file_system = false; + options.cache = false; + options +} + +fn collected_entry_from_entry(entry: Entry<'_>) -> CollectedEntry { + CollectedEntry { + path: entry.relative.to_string(), + file_type: entry.file_type, + mtime: entry.mtime, + size: entry.size, + } +} + +fn root_device_for_options(root: &Path, options: WalkOptions) -> Option { + if options.same_file_system { + root_device_id(root, options.follow_links) + } else { + None + } +} + +fn filter_collected_same_file_system( + root: &Path, + entries: Vec, + options: WalkOptions, + heartbeat: &mut H, +) -> std::result::Result, WalkError> +where + H: FnMut() -> std::result::Result<(), E>, +{ + if !options.same_file_system { + return Ok(entries); + } + let Some(root_device) = root_device_for_options(root, options) else { + return Ok(entries); + }; + let mut filtered = Vec::with_capacity(entries.len()); + let mut pruned_dirs = Vec::new(); + let mut visited = 0usize; + for entry in entries { + if visited == 0 || visited >= HEARTBEAT_INTERVAL { + visited = 0; + heartbeat().map_err(WalkError::Interrupted)?; + } + visited += 1; + if is_under_pruned_relative_dir(&entry.path, &pruned_dirs) { + continue; + } + if is_path_on_root_file_system( + &entry.absolute_path(root), + entry.depth(), + options.follow_links, + Some(root_device), + ) { + filtered.push(entry); + } else if entry.is_dir() { + pruned_dirs.push(entry.path); + } + } + Ok(filtered) +} + +struct FallbackCollector { + entries: Vec, + root_device: Option, + follow_links: FollowLinks, + _error: std::marker::PhantomData E>, +} + +impl FallbackCollector { + fn new(root: &Path, options: WalkOptions) -> Self { + Self { + entries: Vec::new(), + root_device: root_device_for_options(root, options), + follow_links: options.follow_links, + _error: std::marker::PhantomData, + } + } + + fn into_entries(self) -> Vec { + self.entries + } +} + +impl EntryVisitor for FallbackCollector { + type Error = E; + + fn visit(&mut self, entry: Entry<'_>) -> std::result::Result { + if !is_path_on_root_file_system(entry.path, entry.depth, self.follow_links, self.root_device) + { + return Ok(if entry.file_type == FileType::Dir { + WalkControl::SkipDescend + } else { + WalkControl::Continue + }); + } + self.entries.push(collected_entry_from_entry(entry)); + Ok(WalkControl::Continue) + } +} + +struct FallbackStreamCollector<'a, V> { + entries: Vec, + root_device: Option, + follow_links: FollowLinks, + visitor: &'a mut V, +} + +impl<'a, V> FallbackStreamCollector<'a, V> { + fn new(root: &Path, options: WalkOptions, visitor: &'a mut V) -> Self { + Self { + entries: Vec::new(), + root_device: root_device_for_options(root, options), + follow_links: options.follow_links, + visitor, + } + } + + fn into_parts(self) -> (Vec, &'a mut V) { + (self.entries, self.visitor) + } +} + +impl EntryVisitor for FallbackStreamCollector<'_, V> +where + V: EntryVisitor, +{ + type Error = V::Error; + + fn visit(&mut self, entry: Entry<'_>) -> std::result::Result { + if !is_path_on_root_file_system(entry.path, entry.depth, self.follow_links, self.root_device) + { + return Ok(if entry.file_type == FileType::Dir { + WalkControl::SkipDescend + } else { + WalkControl::Continue + }); + } + self.entries.push(collected_entry_from_entry(entry)); + Ok(WalkControl::Continue) + } + + fn visit_directory_error( + &mut self, + error: DirectoryError<'_>, + ) -> std::result::Result { + self.visitor.visit_directory_error(error) + } +} + +fn replay_collected_entry( + root: &Path, + entry: &CollectedEntry, + visitor: &mut V, +) -> std::result::Result +where + V: EntryVisitor, +{ + let absolute = entry.absolute_path(root); + let name = if entry.path.is_empty() { + root.file_name().unwrap_or(root.as_os_str()).to_os_string() + } else { + Path::new(&entry.path) + .file_name() + .unwrap_or_else(|| OsStr::new("")) + .to_os_string() + }; + visitor.visit(Entry { + path: &absolute, + relative: &entry.path, + name: &name, + file_type: entry.file_type, + mtime: entry.mtime, + size: entry.size, + depth: entry.depth(), + }) +} + +struct RawDirEntry<'a> { + name: Cow<'a, OsStr>, + file_type: FileType, + mtime: Option, + size: Option, +} + +struct OwnedDirEntry { + name: OsString, + file_type: FileType, + mtime: Option, + size: Option, +} + +impl RawDirEntry<'_> { + fn into_owned(self) -> OwnedDirEntry { + OwnedDirEntry { + name: self.name.into_owned(), + file_type: self.file_type, + mtime: self.mtime, + size: self.size, + } + } +} + +impl OwnedDirEntry { + fn as_raw(&self) -> RawDirEntry<'_> { + RawDirEntry { + name: Cow::Borrowed(self.name.as_os_str()), + file_type: self.file_type, + mtime: self.mtime, + size: self.size, + } + } +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum ReadDirControl { + Continue, + Stop, +} + +enum ReadDirError { + Io(io::Error), + Walk(WalkError), +} + +impl From for ReadDirError { + fn from(err: io::Error) -> Self { + Self::Io(err) + } +} + +fn file_type_from_metadata(metadata: &std::fs::Metadata) -> Option { + let file_type = metadata.file_type(); + if file_type.is_symlink() { + Some(FileType::Symlink) + } else if file_type.is_dir() { + Some(FileType::Dir) + } else if file_type.is_file() { + Some(FileType::File) + } else { + None + } +} + +struct RootEntry { + file_type: FileType, + mtime: Option, + size: Option, +} + +fn root_entry( + root: &Path, + detail: WalkDetail, + follow_links: FollowLinks, +) -> std::result::Result, WalkError> { + let metadata = if follow_links.follow_at_depth(0) { + match std::fs::metadata(root) { + Ok(metadata) => metadata, + Err(err) if is_missing_metadata_error(&err) => { + std::fs::symlink_metadata(root).map_err(|err| WalkError::InvalidData { + path: root.to_path_buf(), + message: err.to_string(), + })? + }, + Err(err) => { + return Err(WalkError::InvalidData { + path: root.to_path_buf(), + message: err.to_string(), + }); + }, + } + } else { + std::fs::symlink_metadata(root).map_err(|err| WalkError::InvalidData { + path: root.to_path_buf(), + message: err.to_string(), + })? + }; + Ok(entry_from_metadata(&metadata, detail)) +} + +fn entry_from_metadata(metadata: &std::fs::Metadata, detail: WalkDetail) -> Option { + let file_type = file_type_from_metadata(metadata)?; + let size = if detail == WalkDetail::Full && file_type == FileType::File { + Some(metadata.len() as f64) + } else { + None + }; + let mtime = if detail == WalkDetail::Full { + metadata + .modified() + .ok() + .and_then(|time| time.duration_since(std::time::UNIX_EPOCH).ok()) + .map(|duration| duration.as_millis() as f64) + } else { + None + }; + Some(RootEntry { file_type, mtime, size }) +} + +fn is_missing_metadata_error(err: &io::Error) -> bool { + matches!(err.kind(), io::ErrorKind::NotFound | io::ErrorKind::NotADirectory) +} + +fn visit_root( + root: &Path, + entry: &RootEntry, + visitor: &mut V, +) -> std::result::Result> +where + V: EntryVisitor, +{ + let name = root.file_name().unwrap_or(root.as_os_str()); + visitor + .visit(Entry { + path: root, + relative: "", + name, + file_type: entry.file_type, + mtime: entry.mtime, + size: entry.size, + depth: 0, + }) + .map_err(WalkError::Interrupted) +} + +/// Scans entries using the shared cache when [`WalkOptions::cache`] is true. +/// +/// Unsupported native scans fall back to the portable `ignore` walker, so this +/// is the owned-entry collection API most consumers should call. +pub fn collect_entries( + root: &Path, + options: WalkOptions, + heartbeat: H, +) -> std::result::Result> +where + H: Fn() -> std::result::Result<(), E> + Sync, + E: fmt::Display, +{ + cache::collect_entries(root, options, heartbeat) +} + +fn collect_entries_native( + root: &Path, + options: WalkOptions, + heartbeat: H, +) -> std::result::Result> +where + H: FnMut() -> std::result::Result<(), E>, +{ + let mut collector = CollectVisitor::new(); + let status = walk_entries(root, options, &mut collector, heartbeat)?; + if matches!(status, WalkStatus::Unsupported) { + return Ok(EntryScan::Unsupported); + } + collector + .entries + .sort_unstable_by(|a, b| a.path.cmp(&b.path)); + Ok(EntryScan::Entries(CollectedEntries { entries: collector.entries, cache_age_ms: 0 })) +} + +/// Scans entries without cancellation using platform syscalls when supported. +pub fn collect_entries_without_heartbeat( + root: &Path, + options: WalkOptions, +) -> std::result::Result> { + collect_entries(root, options, || Ok::<(), Infallible>(())) +} + +/// Streams entries using the native scanner when the scan contract is +/// equivalent, otherwise using the portable `ignore` walker when possible. +pub fn walk_entries( + root: &Path, + options: WalkOptions, + visitor: &mut V, + mut heartbeat: H, +) -> std::result::Result> +where + V: EntryVisitor, + H: FnMut() -> std::result::Result<(), V::Error>, +{ + if !can_use_fast_scan(options) { + return if can_use_ignore_walk(options) { + walk_entries_with_ignore(root, options, visitor, heartbeat) + } else { + Ok(WalkStatus::Unsupported) + }; + } + let root_entry = root_entry(root, options.detail, options.follow_links)?; + let Some(root_entry) = root_entry else { + return Ok(WalkStatus::Complete); + }; + if options.emit_root && options.min_depth == 0 { + match visit_root(root, &root_entry, visitor)? { + WalkControl::Quit => return Ok(WalkStatus::Stopped), + WalkControl::SkipDescend => return Ok(WalkStatus::Complete), + WalkControl::Continue => {}, + } + } + if root_entry.file_type != FileType::Dir || options.max_depth == 0 { + return Ok(WalkStatus::Complete); + } + let matcher = FastIgnore::new(options.use_gitignore); + let root_ignore = matcher.root_state(root); + let mut visited = 0usize; + match walk_dir( + root, + "", + 0, + &root_ignore, + &matcher, + options, + &mut heartbeat, + &mut visited, + visitor, + ) { + Ok(true) => Ok(WalkStatus::Stopped), + Ok(false) => Ok(WalkStatus::Complete), + Err(WalkError::Unsupported) if can_use_ignore_walk(options) => { + walk_entries_with_ignore(root, options, visitor, heartbeat) + }, + Err(WalkError::Unsupported) => Ok(WalkStatus::Unsupported), + Err(err) => Err(err), + } +} + +const fn can_use_fast_scan(options: WalkOptions) -> bool { + platform::SUPPORTED + && matches!(options.follow_links, FollowLinks::Never) + && !options.contents_first + && !options.same_file_system + && options.min_depth <= options.max_depth +} + +const fn can_use_ignore_walk(options: WalkOptions) -> bool { + !options.contents_first && !options.same_file_system && options.min_depth <= options.max_depth +} + +fn walk_entries_with_ignore( + root: &Path, + options: WalkOptions, + visitor: &mut V, + mut heartbeat: H, +) -> std::result::Result> +where + V: EntryVisitor, + H: FnMut() -> std::result::Result<(), V::Error>, +{ + let root_entry = root_entry(root, options.detail, options.follow_links)?; + let Some(root_entry) = root_entry else { + return Ok(WalkStatus::Complete); + }; + if options.emit_root && options.min_depth == 0 { + match visit_root(root, &root_entry, visitor)? { + WalkControl::Quit => return Ok(WalkStatus::Stopped), + WalkControl::SkipDescend => return Ok(WalkStatus::Complete), + WalkControl::Continue => {}, + } + } + if root_entry.file_type != FileType::Dir || options.max_depth == 0 { + return Ok(WalkStatus::Complete); + } + + heartbeat().map_err(WalkError::Interrupted)?; + let pruned_dirs = Arc::new(Mutex::new(Vec::new())); + let builder = + cache::build_walker_for_options_with_pruned_dirs(root, options, Arc::clone(&pruned_dirs)); + let mut visited = 0usize; + for entry in builder.build() { + if visited == 0 || visited >= HEARTBEAT_INTERVAL { + visited = 0; + heartbeat().map_err(WalkError::Interrupted)?; + } + visited += 1; + + let entry = match entry { + Ok(entry) => entry, + Err(err) => { + if handle_ignore_walk_error(root, &pruned_dirs, err, options, visitor)? { + return Ok(WalkStatus::Stopped); + } + continue; + }, + }; + if entry.depth() == 0 || is_pruned_path(entry.path(), &pruned_dirs) { + continue; + } + if entry.depth() < options.min_depth { + continue; + } + let Some(collected) = cache::collect_entry(root, &entry, options.detail) else { + continue; + }; + match visitor + .visit(Entry { + path: entry.path(), + relative: &collected.path, + name: entry.file_name(), + file_type: collected.file_type, + mtime: collected.mtime, + size: collected.size, + depth: entry.depth(), + }) + .map_err(WalkError::Interrupted)? + { + WalkControl::Quit => return Ok(WalkStatus::Stopped), + WalkControl::SkipDescend => { + if collected.file_type == FileType::Dir { + pruned_dirs + .lock() + .expect("pruned directory lock poisoned") + .push(entry.path().to_path_buf()); + } + }, + WalkControl::Continue => {}, + } + } + Ok(WalkStatus::Complete) +} + +fn handle_ignore_walk_error( + root: &Path, + pruned_dirs: &Arc>>, + error: ignore::Error, + options: WalkOptions, + visitor: &mut V, +) -> std::result::Result> +where + V: EntryVisitor, +{ + let path = ignore_error_path(&error).map_or_else(|| root.to_path_buf(), Path::to_path_buf); + if is_pruned_path(&path, pruned_dirs) { + return Ok(false); + } + let io_error = ignore_error_to_io(&error); + handle_read_dir_error(&path, ReadDirError::Io(io_error), options, visitor) +} + +fn ignore_error_path(error: &ignore::Error) -> Option<&Path> { + match error { + ignore::Error::Partial(errors) => errors.iter().find_map(ignore_error_path), + ignore::Error::WithLineNumber { err, .. } | ignore::Error::WithDepth { err, .. } => { + ignore_error_path(err) + }, + ignore::Error::WithPath { path, .. } => Some(path), + ignore::Error::Loop { child, .. } => Some(child), + ignore::Error::Io(_) + | ignore::Error::Glob { .. } + | ignore::Error::UnrecognizedFileType(_) + | ignore::Error::InvalidDefinition => None, + } +} + +fn ignore_error_to_io(error: &ignore::Error) -> io::Error { + if let Some(io_error) = error.io_error() { + io::Error::new(io_error.kind(), error.to_string()) + } else { + io::Error::other(error.to_string()) + } +} + +fn is_pruned_path(path: &Path, pruned_dirs: &Arc>>) -> bool { + pruned_dirs + .lock() + .expect("pruned directory lock poisoned") + .iter() + .any(|dir| path.starts_with(dir)) +} + +/// Return whether [`WalkDetail::Full`] provides file sizes without per-entry +/// metadata syscalls on this platform. +pub const fn supports_cheap_size_hints() -> bool { + platform::CHEAP_SIZE_HINTS +} + +struct IgnoreState { + parent: Option>, + ignore_matcher: Option, + gitignore_matcher: Option, + git_exclude_matcher: Option, + has_git: bool, +} + +struct FastIgnore { + global: Option, + use_gitignore: bool, +} + +fn has_repo_marker(dir: &Path) -> bool { + dir.join(".git").exists() || dir.join(".jj").exists() +} + +fn load_gitignore(root: &Path, file: &Path) -> Option { + if !file.is_file() { + return None; + } + let mut builder = ignore::gitignore::GitignoreBuilder::new(root); + let _ = builder.add(file); + builder.build().ok().filter(|matcher| !matcher.is_empty()) +} + +impl IgnoreState { + fn build(dir: &Path, parent: Option>) -> Arc { + let has_git = has_repo_marker(dir); + let git_exclude = dir.join(".git/info/exclude"); + Arc::new(Self { + parent, + ignore_matcher: load_gitignore(dir, &dir.join(".ignore")), + gitignore_matcher: load_gitignore(dir, &dir.join(".gitignore")), + git_exclude_matcher: if has_git { + load_gitignore(dir, &git_exclude) + } else { + None + }, + has_git, + }) + } + + fn build_parents(root: &Path, use_gitignore: bool) -> Option> { + if !use_gitignore { + return None; + } + let mut ancestors = Vec::new(); + let mut current = root.parent(); + while let Some(path) = current { + ancestors.push(path); + current = path.parent(); + } + + let mut parent = None; + for ancestor in ancestors.into_iter().rev() { + parent = Some(Self::build(ancestor, parent)); + } + parent + } +} + +impl FastIgnore { + fn new(use_gitignore: bool) -> Self { + let global = if use_gitignore { + let (matcher, _err) = ignore::gitignore::Gitignore::global(); + if matcher.is_empty() { + None + } else { + Some(matcher) + } + } else { + None + }; + Self { global, use_gitignore } + } + + fn root_state(&self, root: &Path) -> Arc { + IgnoreState::build(root, IgnoreState::build_parents(root, self.use_gitignore)) + } + + fn child_state(&self, parent: &Arc, abs_dir: &Path) -> Arc { + if self.use_gitignore { + IgnoreState::build(abs_dir, Some(Arc::clone(parent))) + } else { + Arc::clone(parent) + } + } + + fn is_ignored(&self, state: &Arc, path: &Path, is_dir: bool) -> bool { + if !self.use_gitignore { + return false; + } + + let any_git = Self::has_git_state(state); + let mut saw_git = false; + let mut ignore_match = ignore::Match::None; + let mut gitignore_match = ignore::Match::None; + let mut git_exclude_match = ignore::Match::None; + + let mut current = Some(state.as_ref()); + while let Some(frame) = current { + if ignore_match.is_none() + && let Some(matcher) = &frame.ignore_matcher + { + ignore_match = matcher.matched(path, is_dir); + } + if gitignore_match.is_none() + && let Some(matcher) = &frame.gitignore_matcher + { + gitignore_match = matcher.matched(path, is_dir); + } + if any_git + && !saw_git + && git_exclude_match.is_none() + && let Some(matcher) = &frame.git_exclude_matcher + { + git_exclude_match = matcher.matched(path, is_dir); + } + saw_git = saw_git || frame.has_git; + current = frame.parent.as_deref(); + } + + match ignore_match { + ignore::Match::Ignore(_) => return true, + ignore::Match::Whitelist(_) => return false, + ignore::Match::None => {}, + } + match gitignore_match { + ignore::Match::Ignore(_) => return true, + ignore::Match::Whitelist(_) => return false, + ignore::Match::None => {}, + } + match git_exclude_match { + ignore::Match::Ignore(_) => return true, + ignore::Match::Whitelist(_) => return false, + ignore::Match::None => {}, + } + if any_git && let Some(global) = &self.global { + match global.matched(path, is_dir) { + ignore::Match::Ignore(_) => return true, + ignore::Match::Whitelist(_) => return false, + ignore::Match::None => {}, + } + } + false + } + + fn has_git_state(state: &Arc) -> bool { + let mut current = Some(state.as_ref()); + while let Some(frame) = current { + if frame.has_git { + return true; + } + current = frame.parent.as_deref(); + } + false + } +} + +#[allow(clippy::too_many_arguments, reason = "hot traversal path keeps state explicit")] +fn walk_dir( + dir: &Path, + relative_dir: &str, + depth: usize, + ignore_state: &Arc, + matcher: &FastIgnore, + options: WalkOptions, + heartbeat: &mut H, + visited: &mut usize, + visitor: &mut V, +) -> std::result::Result> +where + V: EntryVisitor, + H: FnMut() -> std::result::Result<(), V::Error>, +{ + match options.order { + WalkOrder::Path => { + let mut entries = Vec::new(); + match platform::read_dir_entries(dir, options.detail, |entry| { + entries.push(entry.into_owned()); + Ok(ReadDirControl::Continue) + }) { + Ok(_) => {}, + Err(err) => return handle_read_dir_error(dir, err, options, visitor), + } + entries.sort_unstable_by(|a, b| a.name.cmp(&b.name)); + for entry in entries { + if process_entry( + dir, + relative_dir, + depth, + ignore_state, + matcher, + options, + heartbeat, + visited, + visitor, + entry.as_raw(), + )? == ReadDirControl::Stop + { + return Ok(true); + } + } + Ok(false) + }, + WalkOrder::Unordered => match platform::read_dir_entries(dir, options.detail, |entry| { + process_entry( + dir, + relative_dir, + depth, + ignore_state, + matcher, + options, + heartbeat, + visited, + visitor, + entry, + ) + }) { + Ok(ReadDirControl::Continue) => Ok(false), + Ok(ReadDirControl::Stop) => Ok(true), + Err(err) => handle_read_dir_error(dir, err, options, visitor), + }, + } +} + +#[allow(clippy::too_many_arguments, reason = "hot traversal path keeps state explicit")] +fn process_entry( + dir: &Path, + relative_dir: &str, + depth: usize, + ignore_state: &Arc, + matcher: &FastIgnore, + options: WalkOptions, + heartbeat: &mut H, + visited: &mut usize, + visitor: &mut V, + entry: RawDirEntry<'_>, +) -> std::result::Result> +where + V: EntryVisitor, + H: FnMut() -> std::result::Result<(), V::Error>, +{ + if *visited == 0 || *visited >= HEARTBEAT_INTERVAL { + *visited = 0; + heartbeat().map_err(WalkError::Interrupted)?; + } + *visited += 1; + + let name = entry.name.as_ref(); + if is_dot_entry(name) { + return Ok(ReadDirControl::Continue); + } + if !options.include_hidden && is_hidden_name(name) { + return Ok(ReadDirControl::Continue); + } + if (options.skip_git && is_git_name(name)) + || (options.skip_node_modules && is_node_modules_name(name)) + { + return Ok(ReadDirControl::Continue); + } + + let name = entry_name(name); + if name.is_empty() { + return Ok(ReadDirControl::Continue); + } + let relative = join_relative_path(relative_dir, &name); + let next_depth = depth + 1; + if next_depth > options.max_depth { + return Ok(ReadDirControl::Continue); + } + let is_dir = entry.file_type == FileType::Dir; + let entry_name: &OsStr = entry.name.as_ref(); + let absolute = dir.join(Path::new(entry_name)); + + if matcher.is_ignored(ignore_state, &absolute, is_dir) { + return Ok(ReadDirControl::Continue); + } + + if next_depth >= options.min_depth { + match visitor + .visit(Entry { + path: &absolute, + relative: &relative, + name: entry.name.as_ref(), + file_type: entry.file_type, + mtime: entry.mtime, + size: entry.size, + depth: next_depth, + }) + .map_err(WalkError::Interrupted)? + { + WalkControl::Quit => return Ok(ReadDirControl::Stop), + WalkControl::SkipDescend => return Ok(ReadDirControl::Continue), + WalkControl::Continue => {}, + } + } + + if is_dir && next_depth < options.max_depth { + let child_ignore = matcher.child_state(ignore_state, &absolute); + if walk_dir( + &absolute, + &relative, + next_depth, + &child_ignore, + matcher, + options, + heartbeat, + visited, + visitor, + )? { + return Ok(ReadDirControl::Stop); + } + } + + Ok(ReadDirControl::Continue) +} + +fn handle_read_dir_error( + dir: &Path, + err: ReadDirError, + options: WalkOptions, + visitor: &mut V, +) -> std::result::Result> +where + V: EntryVisitor, +{ + match err { + ReadDirError::Walk(err) => Err(err), + ReadDirError::Io(err) if err.kind() == io::ErrorKind::Unsupported => { + Err(WalkError::Unsupported) + }, + ReadDirError::Io(err) + if options.directory_errors == DirectoryErrorMode::SkipSkippable + && is_skippable_directory_error(&err) => + { + Ok(false) + }, + ReadDirError::Io(err) if options.directory_errors == DirectoryErrorMode::Visit => { + match visitor + .visit_directory_error(DirectoryError { path: dir, error: &err }) + .map_err(WalkError::Interrupted)? + { + WalkControl::Quit => Ok(true), + WalkControl::SkipDescend | WalkControl::Continue => Ok(false), + } + }, + ReadDirError::Io(err) => { + Err(WalkError::InvalidData { path: dir.to_path_buf(), message: err.to_string() }) + }, + } +} + +fn is_skippable_directory_error(err: &io::Error) -> bool { + matches!( + err.kind(), + io::ErrorKind::NotFound | io::ErrorKind::NotADirectory | io::ErrorKind::PermissionDenied + ) +} + +fn is_dot_entry(name: &OsStr) -> bool { + name == OsStr::new(".") || name == OsStr::new("..") +} + +fn is_git_name(name: &OsStr) -> bool { + name == OsStr::new(".git") +} + +fn is_node_modules_name(name: &OsStr) -> bool { + name == OsStr::new("node_modules") +} + +fn entry_name(name: &OsStr) -> String { + name.to_string_lossy().into_owned() +} + +#[cfg(unix)] +fn is_hidden_name(name: &OsStr) -> bool { + use std::os::unix::ffi::OsStrExt; + name.as_bytes().first() == Some(&b'.') +} + +#[cfg(windows)] +fn is_hidden_name(name: &OsStr) -> bool { + use std::os::windows::ffi::OsStrExt; + name.encode_wide().next() == Some(b'.' as u16) +} + +#[cfg(not(any(unix, windows)))] +fn is_hidden_name(name: &OsStr) -> bool { + name + .to_str() + .is_some_and(|value| value.as_bytes().first() == Some(&b'.')) +} + +fn join_relative_path(parent: &str, name: &str) -> String { + if parent.is_empty() { + name.to_string() + } else { + let mut path = String::with_capacity(parent.len() + 1 + name.len()); + path.push_str(parent); + path.push('/'); + path.push_str(name); + path + } +} + +fn mtime_millis(seconds: i64, nanos: i64) -> Option { + if seconds < 0 { + return None; + } + Some((seconds as f64).mul_add(1000.0, nanos.max(0) as f64 / 1_000_000.0)) +} + +#[cfg(target_os = "macos")] +mod platform { + use std::{ + ffi::{CString, OsStr}, + io, + mem::size_of, + os::{fd::RawFd, unix::ffi::OsStrExt}, + path::Path, + }; + + use super::{ + FileType, RawDirEntry, ReadDirControl, ReadDirError, WalkDetail, WalkError, mtime_millis, + }; + + pub const SUPPORTED: bool = true; + pub const CHEAP_SIZE_HINTS: bool = false; + + const BUFFER_SIZE: usize = 256 * 1024; + const VREG: u32 = 1; + const VDIR: u32 = 2; + const VLNK: u32 = 5; + + struct FdGuard(RawFd); + + impl Drop for FdGuard { + fn drop(&mut self) { + // SAFETY: `FdGuard` owns this file descriptor and closes it exactly once. + unsafe { libc::close(self.0) }; + } + } + + pub fn read_dir_entries( + path: &Path, + detail: WalkDetail, + mut emit: F, + ) -> std::result::Result> + where + F: FnMut(RawDirEntry<'_>) -> std::result::Result>, + { + let fd = open_dir(path)?; + let mut attrs = libc::attrlist { + bitmapcount: libc::ATTR_BIT_MAP_COUNT, + reserved: 0, + commonattr: libc::ATTR_CMN_NAME | libc::ATTR_CMN_OBJTYPE, + volattr: 0, + dirattr: 0, + fileattr: 0, + forkattr: 0, + }; + if detail == WalkDetail::Full { + attrs.commonattr |= libc::ATTR_CMN_MODTIME; + attrs.fileattr |= libc::ATTR_FILE_DATALENGTH; + } + + let mut buffer = vec![0u8; BUFFER_SIZE]; + loop { + // SAFETY: `fd` is an open directory descriptor, `attrs` points to a valid + // attrlist for the duration of the call, and `buffer` is writable. + let count = unsafe { + libc::getattrlistbulk( + fd.0, + std::ptr::addr_of_mut!(attrs).cast(), + buffer.as_mut_ptr().cast(), + buffer.len(), + libc::FSOPT_NOFOLLOW as u64, + ) + }; + if count == 0 { + break; + } + if count < 0 { + let err = io::Error::last_os_error(); + if err.kind() == io::ErrorKind::Interrupted { + continue; + } + return Err(ReadDirError::Io(map_unsupported(err))); + } + + let mut offset = 0usize; + for _ in 0..count { + if offset + size_of::() > buffer.len() { + return Err(invalid_data("truncated getattrlistbulk record length").into()); + } + let record_len = u32::from_ne_bytes( + buffer[offset..offset + size_of::()] + .try_into() + .expect("slice length checked"), + ) as usize; + if record_len < size_of::() || offset + record_len > buffer.len() { + return Err(invalid_data("invalid getattrlistbulk record length").into()); + } + let record = &buffer[offset..offset + record_len]; + if let Some(entry) = parse_record(record, detail)? + && emit(entry).map_err(ReadDirError::Walk)? == ReadDirControl::Stop + { + return Ok(ReadDirControl::Stop); + } + offset += record_len; + } + } + Ok(ReadDirControl::Continue) + } + + fn open_dir(path: &Path) -> io::Result { + let path = CString::new(path.as_os_str().as_bytes()) + .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "path contains NUL"))?; + // SAFETY: `path` is a NUL-terminated C string; flags open the directory for + // metadata traversal only and do not transfer ownership of the string. + let fd = + unsafe { libc::open(path.as_ptr(), libc::O_RDONLY | libc::O_DIRECTORY | libc::O_CLOEXEC) }; + if fd < 0 { + Err(io::Error::last_os_error()) + } else { + Ok(FdGuard(fd)) + } + } + + fn parse_record(record: &[u8], detail: WalkDetail) -> io::Result>> { + let mut cursor = size_of::(); + let name_ref_start = cursor; + let name_ref = read_value::(record, &mut cursor)?; + let obj_type = read_value::(record, &mut cursor)?; + let (mtime, data_length) = if detail == WalkDetail::Full { + let modified = read_value::(record, &mut cursor)?; + let data_length = read_value::(record, &mut cursor)?; + (mtime_millis(modified.tv_sec as i64, modified.tv_nsec as i64), Some(data_length)) + } else { + (None, None) + }; + + let name_start = checked_attr_offset(name_ref_start, name_ref.attr_dataoffset)?; + let name_len = name_ref.attr_length as usize; + if name_len == 0 || name_start + name_len > record.len() { + return Err(invalid_data("invalid getattrlistbulk name reference")); + } + let name_bytes = trim_nul(&record[name_start..name_start + name_len]); + if name_bytes.is_empty() { + return Ok(None); + } + + let Some(file_type) = file_type_from_vtype(obj_type) else { + return Ok(None); + }; + let size = if file_type == FileType::File { + data_length.map(|value| value as f64) + } else { + None + }; + Ok(Some(RawDirEntry { name: OsStr::from_bytes(name_bytes).into(), file_type, mtime, size })) + } + + fn read_value(record: &[u8], cursor: &mut usize) -> io::Result { + let end = cursor.saturating_add(size_of::()); + if end > record.len() { + return Err(invalid_data("truncated getattrlistbulk attribute")); + } + let ptr = record[*cursor..end].as_ptr(); + *cursor = end; + // SAFETY: Bounds were checked above; `getattrlistbulk` records are byte + // packed, so unaligned reads are required and do not outlive `record`. + Ok(unsafe { std::ptr::read_unaligned(ptr.cast::()) }) + } + + fn checked_attr_offset(base: usize, offset: i32) -> io::Result { + if offset < 0 { + return Err(invalid_data("negative getattrlistbulk attribute offset")); + } + base + .checked_add(offset as usize) + .ok_or_else(|| invalid_data("overflowing getattrlistbulk attribute offset")) + } + + fn trim_nul(bytes: &[u8]) -> &[u8] { + let end = bytes.iter().position(|b| *b == 0).unwrap_or(bytes.len()); + &bytes[..end] + } + + const fn file_type_from_vtype(value: u32) -> Option { + match value { + VREG => Some(FileType::File), + VDIR => Some(FileType::Dir), + VLNK => Some(FileType::Symlink), + _ => None, + } + } + + fn map_unsupported(err: io::Error) -> io::Error { + if matches!(err.raw_os_error(), Some(libc::ENOTSUP | libc::EINVAL)) { + io::Error::new(io::ErrorKind::Unsupported, err) + } else { + err + } + } + + fn invalid_data(message: &'static str) -> io::Error { + io::Error::new(io::ErrorKind::InvalidData, message) + } +} + +#[cfg(target_os = "linux")] +mod platform { + use std::{ + ffi::{CString, OsStr}, + io, + mem::{size_of, zeroed}, + os::unix::ffi::{OsStrExt, OsStringExt}, + path::Path, + }; + + use super::{ + FileType, RawDirEntry, ReadDirControl, ReadDirError, WalkDetail, WalkError, mtime_millis, + }; + + pub const SUPPORTED: bool = true; + pub const CHEAP_SIZE_HINTS: bool = false; + + const BUFFER_SIZE: usize = 256 * 1024; + const LINUX_DIRENT64_NAME_OFFSET: usize = 19; + const STATX_TYPE: u32 = 0x0001; + const STATX_SIZE: u32 = 0x0200; + const STATX_MTIME: u32 = 0x0040; + const STATX_BASIC_STATS: u32 = 0x07ff; + + #[repr(C)] + #[derive(Clone, Copy)] + struct StatxTimestamp { + tv_sec: i64, + tv_nsec: u32, + __reserved: i32, + } + + #[repr(C)] + #[derive(Clone, Copy)] + struct Statx { + stx_mask: u32, + stx_blksize: u32, + stx_attributes: u64, + stx_nlink: u32, + stx_uid: u32, + stx_gid: u32, + stx_mode: u16, + __spare0: [u16; 1], + stx_ino: u64, + stx_size: u64, + stx_blocks: u64, + stx_attributes_mask: u64, + stx_atime: StatxTimestamp, + stx_btime: StatxTimestamp, + stx_ctime: StatxTimestamp, + stx_mtime: StatxTimestamp, + stx_rdev_major: u32, + stx_rdev_minor: u32, + stx_dev_major: u32, + stx_dev_minor: u32, + stx_mnt_id: u64, + stx_dio_mem_align: u32, + stx_dio_offset_align: u32, + __spare3: [u64; 12], + } + + struct FdGuard(libc::c_int); + + impl Drop for FdGuard { + fn drop(&mut self) { + // SAFETY: `FdGuard` owns this file descriptor and closes it exactly once. + unsafe { libc::close(self.0) }; + } + } + + struct EntryStat { + file_type: FileType, + mtime: Option, + size: Option, + } + + pub fn read_dir_entries( + path: &Path, + detail: WalkDetail, + mut emit: F, + ) -> std::result::Result> + where + F: FnMut(RawDirEntry<'_>) -> std::result::Result>, + { + let fd = open_dir(path)?; + let mut buffer = vec![0u8; BUFFER_SIZE]; + loop { + // SAFETY: `fd` is an open directory descriptor and `buffer` is writable. + let read = unsafe { + libc::syscall( + libc::SYS_getdents64, + fd.0, + buffer.as_mut_ptr().cast::(), + buffer.len(), + ) + }; + if read == 0 { + break; + } + if read < 0 { + let err = io::Error::last_os_error(); + if err.kind() == io::ErrorKind::Interrupted { + continue; + } + return Err(err.into()); + } + + let mut offset = 0usize; + let read_len = read as usize; + while offset < read_len { + if offset + LINUX_DIRENT64_NAME_OFFSET > read_len { + return Err(invalid_data("truncated getdents64 record").into()); + } + let reclen = read_u16(&buffer[offset + 16..read_len])? as usize; + if reclen < LINUX_DIRENT64_NAME_OFFSET || offset + reclen > read_len { + return Err(invalid_data("invalid getdents64 record length").into()); + } + let d_type = buffer[offset + 18]; + let name_bytes = + trim_nul(&buffer[offset + LINUX_DIRENT64_NAME_OFFSET..offset + reclen]); + offset += reclen; + if name_bytes.is_empty() { + continue; + } + + let dtype_file_type = file_type_from_dtype(d_type); + let stat = if detail == WalkDetail::Full || dtype_file_type.is_none() { + match stat_entry(fd.0, name_bytes, detail) { + Ok(Some(stat)) => Some(stat), + Ok(None) => continue, + Err(err) if is_skippable_entry_error(&err) => continue, + Err(err) => return Err(err.into()), + } + } else { + None + }; + let file_type = stat + .as_ref() + .map_or(dtype_file_type, |stat| Some(stat.file_type)); + let Some(file_type) = file_type else { + continue; + }; + let entry = RawDirEntry { + name: OsStr::from_bytes(name_bytes).into(), + file_type, + mtime: stat.as_ref().and_then(|stat| stat.mtime), + size: stat.as_ref().and_then(|stat| stat.size), + }; + if emit(entry).map_err(ReadDirError::Walk)? == ReadDirControl::Stop { + return Ok(ReadDirControl::Stop); + } + } + } + Ok(ReadDirControl::Continue) + } + + fn open_dir(path: &Path) -> io::Result { + let path = CString::new(path.as_os_str().as_bytes()) + .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "path contains NUL"))?; + // SAFETY: `path` is a NUL-terminated C string; flags request a directory + // descriptor used only with getdents/statx and do not retain the pointer. + let fd = + unsafe { libc::open(path.as_ptr(), libc::O_RDONLY | libc::O_DIRECTORY | libc::O_CLOEXEC) }; + if fd < 0 { + Err(io::Error::last_os_error()) + } else { + Ok(FdGuard(fd)) + } + } + + fn stat_entry( + dirfd: libc::c_int, + name: &[u8], + detail: WalkDetail, + ) -> io::Result> { + let name = CString::new(name) + .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "entry name contains NUL"))?; + match statx_entry(dirfd, &name, detail) { + Ok(value) => Ok(value), + Err(err) if matches!(err.raw_os_error(), Some(libc::ENOSYS | libc::EINVAL)) => { + fstatat_entry(dirfd, &name, detail) + }, + Err(err) => Err(err), + } + } + + fn statx_entry( + dirfd: libc::c_int, + name: &CString, + detail: WalkDetail, + ) -> io::Result> { + // SAFETY: `Statx` is a plain-old-data buffer whose all-zero value is a + // valid initialization before the kernel fills it. + let mut statx = unsafe { zeroed::() }; + let mask = if detail == WalkDetail::Full { + STATX_BASIC_STATS + } else { + STATX_TYPE + }; + // SAFETY: `name` is NUL-terminated, `statx` is writable, and `dirfd` is an + // open directory descriptor for an AT_* relative metadata query. + let rc = unsafe { + libc::syscall( + libc::SYS_statx, + dirfd, + name.as_ptr(), + libc::AT_SYMLINK_NOFOLLOW | libc::AT_NO_AUTOMOUNT, + mask, + std::ptr::addr_of_mut!(statx), + ) + }; + if rc != 0 { + return Err(io::Error::last_os_error()); + } + let Some(file_type) = file_type_from_mode(statx.stx_mode as libc::mode_t) else { + return Ok(None); + }; + let mtime = if detail == WalkDetail::Full && statx.stx_mask & STATX_MTIME != 0 { + mtime_millis(statx.stx_mtime.tv_sec, i64::from(statx.stx_mtime.tv_nsec)) + } else { + None + }; + let size = if detail == WalkDetail::Full + && file_type == FileType::File + && statx.stx_mask & STATX_SIZE != 0 + { + Some(statx.stx_size as f64) + } else { + None + }; + Ok(Some(EntryStat { file_type, mtime, size })) + } + + fn fstatat_entry( + dirfd: libc::c_int, + name: &CString, + detail: WalkDetail, + ) -> io::Result> { + // SAFETY: `libc::stat` is a POD buffer filled by fstatat. + let mut stat = unsafe { zeroed::() }; + // SAFETY: `name` is NUL-terminated, `stat` is writable, and `dirfd` is an + // open directory descriptor for an AT_* relative metadata query. + let rc = unsafe { + libc::fstatat( + dirfd, + name.as_ptr(), + std::ptr::addr_of_mut!(stat), + libc::AT_SYMLINK_NOFOLLOW, + ) + }; + if rc != 0 { + return Err(io::Error::last_os_error()); + } + let Some(file_type) = file_type_from_mode(stat.st_mode) else { + return Ok(None); + }; + let mtime = if detail == WalkDetail::Full { + mtime_millis(stat.st_mtime, stat.st_mtime_nsec as i64) + } else { + None + }; + let size = if detail == WalkDetail::Full && file_type == FileType::File { + Some(stat.st_size as f64) + } else { + None + }; + Ok(Some(EntryStat { file_type, mtime, size })) + } + + fn read_u16(bytes: &[u8]) -> io::Result { + if bytes.len() < size_of::() { + return Err(invalid_data("truncated u16")); + } + Ok(u16::from_ne_bytes( + bytes[..size_of::()] + .try_into() + .expect("slice length checked"), + )) + } + + fn trim_nul(bytes: &[u8]) -> &[u8] { + let end = bytes.iter().position(|b| *b == 0).unwrap_or(bytes.len()); + &bytes[..end] + } + + fn file_type_from_dtype(value: u8) -> Option { + match value { + libc::DT_REG => Some(FileType::File), + libc::DT_DIR => Some(FileType::Dir), + libc::DT_LNK => Some(FileType::Symlink), + _ => None, + } + } + + fn file_type_from_mode(mode: libc::mode_t) -> Option { + match mode & libc::S_IFMT { + libc::S_IFREG => Some(FileType::File), + libc::S_IFDIR => Some(FileType::Dir), + libc::S_IFLNK => Some(FileType::Symlink), + _ => None, + } + } + + fn is_skippable_entry_error(err: &io::Error) -> bool { + matches!( + err.kind(), + io::ErrorKind::NotFound | io::ErrorKind::PermissionDenied | io::ErrorKind::NotADirectory + ) + } + + fn invalid_data(message: &'static str) -> io::Error { + io::Error::new(io::ErrorKind::InvalidData, message) + } +} + +#[cfg(target_os = "windows")] +mod platform { + use std::{ + ffi::OsString, + io, + os::windows::ffi::{OsStrExt, OsStringExt}, + path::Path, + }; + + use windows_sys::{ + Wdk::Storage::FileSystem::{ + FILE_ID_FULL_DIR_INFORMATION, FileIdFullDirectoryInformation, NtQueryDirectoryFile, + }, + Win32::{ + Foundation::{CloseHandle, HANDLE, INVALID_HANDLE_VALUE, STATUS_NO_MORE_FILES}, + Storage::FileSystem::{ + CreateFileW, FILE_ATTRIBUTE_DIRECTORY, FILE_ATTRIBUTE_REPARSE_POINT, + FILE_FLAG_BACKUP_SEMANTICS, FILE_FLAG_OPEN_REPARSE_POINT, FILE_LIST_DIRECTORY, + FILE_SHARE_DELETE, FILE_SHARE_READ, FILE_SHARE_WRITE, OPEN_EXISTING, + }, + System::IO::IO_STATUS_BLOCK, + }, + }; + + use super::{ + FileType, RawDirEntry, ReadDirControl, ReadDirError, WalkDetail, WalkError, mtime_millis, + }; + + pub const SUPPORTED: bool = true; + pub const CHEAP_SIZE_HINTS: bool = true; + + const BUFFER_SIZE: usize = 256 * 1024; + const WINDOWS_TICK: i64 = 10_000_000; + const UNIX_EPOCH_AS_FILETIME: i64 = 116_444_736_000_000_000; + + struct HandleGuard(HANDLE); + + impl Drop for HandleGuard { + fn drop(&mut self) { + // SAFETY: `HandleGuard` owns this handle and closes it exactly once. + unsafe { CloseHandle(self.0) }; + } + } + + pub fn read_dir_entries( + path: &Path, + detail: WalkDetail, + mut emit: F, + ) -> std::result::Result> + where + F: FnMut(RawDirEntry<'_>) -> std::result::Result>, + { + let handle = open_dir(path)?; + let mut buffer = vec![0u8; BUFFER_SIZE]; + let mut restart = true; + + loop { + let mut iosb = IO_STATUS_BLOCK::default(); + // SAFETY: `handle` is an open directory handle, `buffer` is writable, and + // the query class matches the record parser below. + let status = unsafe { + NtQueryDirectoryFile( + handle.0, + std::ptr::null_mut(), + None, + std::ptr::null(), + std::ptr::addr_of_mut!(iosb), + buffer.as_mut_ptr().cast(), + buffer.len() as u32, + FileIdFullDirectoryInformation, + false, + std::ptr::null(), + restart, + ) + }; + restart = false; + if status == STATUS_NO_MORE_FILES { + break; + } + if status < 0 { + return Err(io::Error::from_raw_os_error(status).into()); + } + + let mut offset = 0usize; + loop { + if offset + std::mem::size_of::() > buffer.len() { + return Err(invalid_data("truncated NtQueryDirectoryFile record").into()); + } + let info = unsafe { + // SAFETY: Bounds were checked above; records are byte-packed in the + // buffer and may not be aligned for Rust references. + std::ptr::read_unaligned( + buffer[offset..] + .as_ptr() + .cast::(), + ) + }; + let name_offset = offset + std::mem::offset_of!(FILE_ID_FULL_DIR_INFORMATION, FileName); + let name_len = info.FileNameLength as usize; + if name_len % 2 != 0 || name_offset + name_len > buffer.len() { + return Err(invalid_data("invalid NtQueryDirectoryFile name length").into()); + } + let name_units: Vec = buffer[name_offset..name_offset + name_len] + .chunks_exact(2) + .map(|chunk| u16::from_ne_bytes([chunk[0], chunk[1]])) + .collect(); + let name = OsString::from_wide(&name_units); + if let Some(file_type) = file_type_from_attributes(info.FileAttributes) { + let size = if detail == WalkDetail::Full && file_type == FileType::File { + Some(info.EndOfFile.max(0) as f64) + } else { + None + }; + let mtime = if detail == WalkDetail::Full { + mtime_from_filetime(info.LastWriteTime) + } else { + None + }; + let entry = RawDirEntry { name: name.into(), file_type, mtime, size }; + if emit(entry).map_err(ReadDirError::Walk)? == ReadDirControl::Stop { + return Ok(ReadDirControl::Stop); + } + } + if info.NextEntryOffset == 0 { + break; + } + offset = offset.saturating_add(info.NextEntryOffset as usize); + } + } + Ok(ReadDirControl::Continue) + } + + fn open_dir(path: &Path) -> io::Result { + let mut path: Vec = path.as_os_str().encode_wide().collect(); + path.push(0); + // SAFETY: `path` is NUL-terminated; the returned handle is owned by + // `HandleGuard` on success. + let handle = unsafe { + CreateFileW( + path.as_ptr(), + FILE_LIST_DIRECTORY, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + std::ptr::null(), + OPEN_EXISTING, + FILE_FLAG_BACKUP_SEMANTICS | FILE_FLAG_OPEN_REPARSE_POINT, + std::ptr::null_mut(), + ) + }; + if handle == INVALID_HANDLE_VALUE { + Err(io::Error::last_os_error()) + } else { + Ok(HandleGuard(handle)) + } + } + + fn file_type_from_attributes(attributes: u32) -> Option { + if attributes & FILE_ATTRIBUTE_REPARSE_POINT != 0 { + Some(FileType::Symlink) + } else if attributes & FILE_ATTRIBUTE_DIRECTORY != 0 { + Some(FileType::Dir) + } else { + Some(FileType::File) + } + } + + fn mtime_from_filetime(filetime: i64) -> Option { + let ticks = filetime.checked_sub(UNIX_EPOCH_AS_FILETIME)?; + let seconds = ticks / WINDOWS_TICK; + let nanos = (ticks % WINDOWS_TICK) * 100; + mtime_millis(seconds, nanos) + } + + fn invalid_data(message: &'static str) -> io::Error { + io::Error::new(io::ErrorKind::InvalidData, message) + } +} + +#[cfg(not(any(target_os = "macos", target_os = "linux", target_os = "windows")))] +mod platform { + use std::{io, path::Path}; + + use super::{RawDirEntry, ReadDirControl, ReadDirError, WalkDetail, WalkError}; + + pub const SUPPORTED: bool = false; + pub const CHEAP_SIZE_HINTS: bool = false; + + pub fn read_dir_entries( + _path: &Path, + _detail: WalkDetail, + _emit: F, + ) -> std::result::Result> + where + F: FnMut(RawDirEntry<'_>) -> std::result::Result>, + { + Err( + io::Error::new( + io::ErrorKind::Unsupported, + "native directory scan unsupported on this platform", + ) + .into(), + ) + } +} + +#[cfg(test)] +mod tests { + use std::{ + fs, + path::PathBuf, + time::{Duration, SystemTime, UNIX_EPOCH}, + }; + + use super::*; + + struct TempTree { + root: PathBuf, + } + + impl TempTree { + fn path(&self) -> &Path { + &self.root + } + } + + impl Drop for TempTree { + fn drop(&mut self) { + let _ = fs::remove_dir_all(&self.root); + } + } + + fn temp_tree(name: &str) -> TempTree { + let unique = SystemTime::now() + .duration_since(UNIX_EPOCH) + .expect("system time should be after UNIX_EPOCH") + .as_nanos(); + let root = std::env::temp_dir().join(format!("pi-walker-{name}-{unique}")); + fs::create_dir_all(&root).expect("temp root should be created"); + TempTree { root } + } + + struct CachePathGuard { + root: PathBuf, + } + + impl CachePathGuard { + fn new(root: &Path) -> Self { + invalidate_path(root); + Self { root: root.to_path_buf() } + } + } + + impl Drop for CachePathGuard { + fn drop(&mut self) { + invalidate_path(&self.root); + } + } + + fn wait_for_nonzero_cache_age() { + let started = std::time::Instant::now(); + while started.elapsed() < Duration::from_millis(1) { + std::thread::yield_now(); + } + } + + fn test_options() -> WalkOptions { + WalkOptions { + include_hidden: true, + use_gitignore: false, + skip_git: true, + skip_node_modules: true, + follow_links: FollowLinks::Never, + detail: WalkDetail::Minimal, + order: WalkOrder::Path, + emit_root: false, + min_depth: 1, + max_depth: usize::MAX, + contents_first: false, + directory_errors: DirectoryErrorMode::SkipSkippable, + same_file_system: false, + cache: false, + } + } + + #[test] + fn walk_request_files_only_returns_relative_files_and_excludes_directories() { + let tree = temp_tree("request-files-only"); + fs::write(tree.path().join("alpha.txt"), "alpha").expect("top-level file should be written"); + fs::create_dir_all(tree.path().join("nested")).expect("nested dir should be created"); + fs::write(tree.path().join("nested").join("beta.txt"), "beta") + .expect("nested file should be written"); + + let outcome = WalkRequest::from_options(tree.path(), test_options()) + .filter(WalkFilter::files_only()) + .collect() + .expect("files-only request should collect successfully"); + let paths = outcome + .entries + .iter() + .map(|entry| entry.path.as_str()) + .collect::>(); + + assert_eq!( + paths, + vec!["alpha.txt", "nested/beta.txt"], + "files-only requests should preserve relative file paths and exclude directory entries" + ); + assert!( + outcome.entries.iter().all(CollectedEntry::is_file), + "files-only requests should not return directories or other entry kinds: {:?}", + outcome.entries + ); + } + + #[test] + fn compare_depth_first_paths_orders_children_before_parent() { + let mut paths = vec!["dir", "alpha", "dir/child", "dir/child/grandchild"]; + + paths.sort_unstable_by(|left, right| compare_depth_first_paths(left, right)); + + assert_eq!( + paths, + vec!["alpha", "dir/child/grandchild", "dir/child", "dir"], + "depth-first ordering should keep lexical siblings stable while placing descendants \ + before ancestors" + ); + } + + #[test] + fn walk_request_collect_contents_first_uses_collected_fallback() { + let tree = temp_tree("request-contents-first"); + fs::create_dir_all(tree.path().join("nested")).expect("nested dir should be created"); + fs::write(tree.path().join("nested").join("leaf.txt"), "leaf") + .expect("nested file should be written"); + + let outcome = WalkRequest::from_options(tree.path(), test_options()) + .visit_order(VisitOrder::ContentsFirst) + .collect() + .expect("contents-first request should collect through high-level fallback"); + let paths = outcome + .entries + .iter() + .map(|entry| entry.path.as_str()) + .collect::>(); + + assert_eq!( + paths, + vec!["nested/leaf.txt", "nested"], + "contents-first collection should replay children before their directory" + ); + } + + #[test] + fn walk_request_stream_fallback_preserves_predicate_skip_descend() { + struct RecordingVisitor { + paths: Vec, + } + + impl EntryVisitor for RecordingVisitor { + type Error = Infallible; + + fn visit(&mut self, entry: Entry<'_>) -> std::result::Result { + self.paths.push(entry.relative.to_string()); + Ok(WalkControl::Continue) + } + } + + let tree = temp_tree("request-stream-skip-descend"); + fs::create_dir_all(tree.path().join("keep")).expect("keep dir should be created"); + fs::write(tree.path().join("keep").join("leaf.txt"), "leaf") + .expect("kept leaf should be written"); + fs::create_dir_all(tree.path().join("skip")).expect("skip dir should be created"); + fs::write(tree.path().join("skip").join("hidden.txt"), "hidden") + .expect("skipped leaf should be written"); + + let mut visitor = RecordingVisitor { paths: Vec::new() }; + let status = WalkRequest::from_options(tree.path(), test_options()) + .visit_order(VisitOrder::ContentsFirst) + .stream_with_predicate(&mut visitor, |meta: &EntryMeta<'_>| { + if meta.relative_path == "skip" { + WalkDecision::SkipDescend + } else { + WalkDecision::Include + } + }) + .expect("contents-first stream fallback should complete"); + + assert_eq!(status, WalkStatus::Complete); + assert_eq!( + visitor.paths, + vec!["keep/leaf.txt", "keep"], + "predicate SkipDescend should prune the skipped directory and its descendants before \ + contents-first replay" + ); + } + + #[test] + fn walk_request_collect_with_heartbeat_propagates_interruptions() { + let tree = temp_tree("request-heartbeat"); + fs::write(tree.path().join("alpha.txt"), "alpha").expect("file should be written"); + + let result = WalkRequest::from_options(tree.path(), test_options()) + .collect_with_heartbeat(|| Err("stop requested")); + + let Err(WalkError::Interrupted(message)) = result else { + panic!("heartbeat interruption should be surfaced as WalkError::Interrupted"); + }; + assert_eq!(message, "stop requested"); + } + + #[test] + fn walk_request_rechecks_stale_empty_cached_files_only_result() { + let tree = temp_tree("request-empty-recheck"); + let _cache_guard = CachePathGuard::new(tree.path()); + let request = WalkRequest::from_options(tree.path(), test_options()) + .cache(true) + .filter(WalkFilter::files_only()) + .empty_recheck(EmptyRecheck::Never); + + let primed = request + .collect() + .expect("empty request should collect successfully"); + assert_eq!(primed.backend, WalkBackend::Fresh); + assert!( + primed.entries.is_empty(), + "empty root should prime an empty files-only cache entry, got {:?}", + primed.entries + ); + + fs::write(tree.path().join("created.txt"), "created") + .expect("file created after cache prime should be written"); + wait_for_nonzero_cache_age(); + + let stale = request + .collect() + .expect("empty recheck disabled request should read the cached empty result"); + assert_eq!(stale.backend, WalkBackend::Cached); + assert!( + stale.stats.cache_age_ms > 0, + "cached empty result should be old enough to exercise empty recheck" + ); + assert!( + stale.entries.is_empty(), + "disabled empty recheck should leave the cached empty files-only result untouched" + ); + + let refreshed = request + .empty_recheck(EmptyRecheck::AfterMillis(0)) + .collect() + .expect("stale empty cache should be rechecked without manual invalidation"); + let paths = refreshed + .entries + .iter() + .map(|entry| entry.path.as_str()) + .collect::>(); + + assert_eq!( + refreshed.backend, + WalkBackend::Fresh, + "stale empty cache recheck should force a fresh scan" + ); + assert_eq!( + paths, + vec!["created.txt"], + "empty cached files-only requests should observe files created after the cache was primed" + ); + assert!( + refreshed.entries.iter().all(CollectedEntry::is_file), + "files-only empty recheck should still return only files: {:?}", + refreshed.entries + ); + } + + #[test] + fn walk_request_rechecks_stale_cache_empty_after_glob_filter() { + let tree = temp_tree("request-filtered-empty-recheck"); + let _cache_guard = CachePathGuard::new(tree.path()); + fs::write(tree.path().join("old.txt"), "old") + .expect("nonmatching file should be written before cache prime"); + let request = WalkRequest::from_options(tree.path(), test_options()) + .cache(true) + .filter( + WalkFilter::files_only() + .glob(CompiledWalkGlob::new(["*.rs"]).expect("test glob should compile")), + ) + .empty_recheck(EmptyRecheck::Never); + + let primed = request + .collect() + .expect("filtered request should prime the raw nonempty cache successfully"); + assert_eq!(primed.backend, WalkBackend::Fresh); + assert_eq!( + primed.stats.scanned_entries, 1, + "cache prime should scan the nonmatching file before filtering" + ); + assert_eq!( + primed.stats.filtered_entries, 1, + "glob filter should remove the nonmatching cached file" + ); + assert!( + primed.entries.is_empty(), + "nonmatching file should leave the filtered prime result empty: {:?}", + primed.entries + ); + + fs::write(tree.path().join("new.rs"), "new") + .expect("matching file should be written after cache prime"); + wait_for_nonzero_cache_age(); + + let refreshed = request + .empty_recheck(EmptyRecheck::AfterMillis(0)) + .collect() + .expect("filtered empty cached result should be rechecked without manual invalidation"); + let paths = refreshed + .entries + .iter() + .map(|entry| entry.path.as_str()) + .collect::>(); + + assert_eq!( + refreshed.backend, + WalkBackend::Fresh, + "stale cache that is empty only after filtering should force a fresh scan" + ); + assert_eq!( + paths, + vec!["new.rs"], + "filtered empty-cache recheck should return the file that matches the glob" + ); + assert!( + refreshed.entries.iter().all(CollectedEntry::is_file), + "glob-filtered files-only recheck should still return only files: {:?}", + refreshed.entries + ); + } + + #[test] + fn collect_entries_honors_gitignore_without_repo_marker() { + let tree = temp_tree("plain-gitignore"); + fs::write(tree.path().join(".gitignore"), "ignored.txt\n") + .expect(".gitignore should be written"); + fs::write(tree.path().join("ignored.txt"), "ignored") + .expect("ignored file should be written"); + fs::write(tree.path().join("kept.txt"), "keep").expect("kept file should be written"); + + let scan = collect_entries( + tree.path(), + WalkOptions { use_gitignore: true, cache: false, ..test_options() }, + || Ok::<(), Infallible>(()), + ) + .expect("collection should not fail"); + let EntryScan::Entries(scan) = scan else { + panic!("collection should return entries"); + }; + let paths = scan + .entries + .into_iter() + .map(|entry| entry.path) + .collect::>(); + + assert!( + paths.iter().any(|path| path == "kept.txt"), + "collect_entries should include kept.txt from a plain directory, got: {paths:?}" + ); + assert!( + !paths.iter().any(|path| path == "ignored.txt"), + "collect_entries should exclude .gitignore matches without a .git marker, got: {paths:?}" + ); + } + + #[test] + fn collect_entries_prunes_hidden_git_and_node_modules() { + let tree = temp_tree("filters"); + fs::write(tree.path().join("visible.txt"), "ok").expect("visible file should be written"); + fs::write(tree.path().join(".hidden"), "hidden").expect("hidden file should be written"); + fs::create_dir_all(tree.path().join(".git")).expect(".git should be created"); + fs::write(tree.path().join(".git").join("config"), "git") + .expect("git file should be written"); + fs::create_dir_all(tree.path().join("node_modules")).expect("node_modules should be created"); + fs::write(tree.path().join("node_modules").join("pkg.js"), "pkg") + .expect("node module file should be written"); + + let scan = collect_entries( + tree.path(), + WalkOptions { include_hidden: false, ..test_options() }, + || Ok::<(), Infallible>(()), + ) + .expect("collection should not fail"); + let EntryScan::Entries(scan) = scan else { + return; + }; + let paths = scan + .entries + .into_iter() + .map(|entry| entry.path) + .collect::>(); + assert_eq!(paths, vec!["visible.txt"]); + } + + struct PruneVisitor { + seen: Vec, + } + + impl EntryVisitor for PruneVisitor { + type Error = Infallible; + + fn visit(&mut self, entry: Entry<'_>) -> std::result::Result { + self.seen.push(entry.relative.to_string()); + if entry.relative == "skip" { + Ok(WalkControl::SkipDescend) + } else { + Ok(WalkControl::Continue) + } + } + } + + #[cfg(unix)] + struct PathsVisitor { + seen: Vec, + } + + #[cfg(unix)] + impl EntryVisitor for PathsVisitor { + type Error = Infallible; + + fn visit(&mut self, entry: Entry<'_>) -> std::result::Result { + self.seen.push(entry.relative.to_string()); + Ok(WalkControl::Continue) + } + } + + #[cfg(unix)] + fn walk_paths(root: &Path, follow_links: FollowLinks) -> Vec { + let mut visitor = PathsVisitor { seen: Vec::new() }; + let status = + walk_entries(root, WalkOptions { follow_links, ..test_options() }, &mut visitor, || { + Ok::<(), Infallible>(()) + }) + .expect("walk should not fail"); + assert_eq!(status, WalkStatus::Complete); + visitor.seen + } + + #[test] + fn skip_descend_prunes_directory_children() { + let tree = temp_tree("skip-descend"); + fs::create_dir_all(tree.path().join("keep")).expect("keep dir should be created"); + fs::write(tree.path().join("keep").join("file.txt"), "ok") + .expect("keep file should be written"); + fs::create_dir_all(tree.path().join("skip")).expect("skip dir should be created"); + fs::write(tree.path().join("skip").join("file.txt"), "no") + .expect("skip file should be written"); + + let mut visitor = PruneVisitor { seen: Vec::new() }; + let status = + walk_entries(tree.path(), test_options(), &mut visitor, || Ok::<(), Infallible>(())) + .expect("walk should not fail"); + if matches!(status, WalkStatus::Unsupported) { + return; + } + assert!(visitor.seen.iter().any(|path| path == "skip")); + assert!(visitor.seen.iter().any(|path| path == "keep/file.txt")); + assert!(!visitor.seen.iter().any(|path| path == "skip/file.txt")); + } + + #[cfg(unix)] + #[test] + fn walk_entries_always_follows_descendant_symlink_directories() { + let tree = temp_tree("follow-always"); + fs::create_dir_all(tree.path().join("target")).expect("target dir should be created"); + fs::write(tree.path().join("target").join("child.txt"), "ok") + .expect("target child should be written"); + std::os::unix::fs::symlink(tree.path().join("target"), tree.path().join("link")) + .expect("directory symlink should be created"); + + let paths = walk_paths(tree.path(), FollowLinks::Always); + + assert!( + paths.iter().any(|path| path == "link/child.txt"), + "FollowLinks::Always should yield descendants through symlink paths, got: {paths:?}" + ); + } + + #[cfg(unix)] + #[test] + fn walk_entries_roots_follows_root_symlink_but_not_descendant_symlinks() { + let target = temp_tree("follow-roots-target"); + fs::write(target.path().join("child.txt"), "ok").expect("root child should be written"); + + let linked_target = temp_tree("follow-roots-linked-target"); + fs::write(linked_target.path().join("linked-child.txt"), "linked") + .expect("linked child should be written"); + std::os::unix::fs::symlink(linked_target.path(), target.path().join("descendant-link")) + .expect("descendant directory symlink should be created"); + + let link_parent = temp_tree("follow-roots-link-parent"); + let root_link = link_parent.path().join("root-link"); + std::os::unix::fs::symlink(target.path(), &root_link) + .expect("root directory symlink should be created"); + + let paths = walk_paths(&root_link, FollowLinks::Roots); + + assert!( + paths.iter().any(|path| path == "child.txt"), + "FollowLinks::Roots should traverse children of a symlink root, got: {paths:?}" + ); + assert!( + !paths + .iter() + .any(|path| path == "descendant-link/linked-child.txt"), + "FollowLinks::Roots should not traverse descendant symlink directories, got: {paths:?}" + ); + } + + #[cfg(unix)] + #[test] + fn walk_entries_never_does_not_follow_root_symlink_directory() { + let target = temp_tree("follow-never-target"); + fs::write(target.path().join("child.txt"), "ok").expect("root child should be written"); + + let link_parent = temp_tree("follow-never-link-parent"); + let root_link = link_parent.path().join("root-link"); + std::os::unix::fs::symlink(target.path(), &root_link) + .expect("root directory symlink should be created"); + + let paths = walk_paths(&root_link, FollowLinks::Never); + + assert!( + !paths.iter().any(|path| path == "child.txt"), + "FollowLinks::Never should not traverse a symlink root, got: {paths:?}" + ); + } +} diff --git a/crates/vendor/uu-find/Cargo.toml b/crates/vendor/uu-find/Cargo.toml index d9d6996db..fe00f82be 100644 --- a/crates/vendor/uu-find/Cargo.toml +++ b/crates/vendor/uu-find/Cargo.toml @@ -15,13 +15,13 @@ test = false chrono = "0.4.40" clap = "4.5" faccess = "0.2.4" -walkdir = "2.5" regex = "1.11" onig = { version = "6.4", default-features = false } uucore = { version = "0.0.30", features = ["entries", "fs", "fsext", "mode"] } nix = { version = "0.29", features = ["fs", "user"] } argmax = "0.3.1" pi-uutils-ctx = { path = "../../pi-uutils-ctx" } +pi-walker = { path = "../../pi-walker" } [dev-dependencies] tempfile = "3" diff --git a/crates/vendor/uu-find/src/find/matchers/entry.rs b/crates/vendor/uu-find/src/find/matchers/entry.rs index 8fb1f28f4..bba033e37 100644 --- a/crates/vendor/uu-find/src/find/matchers/entry.rs +++ b/crates/vendor/uu-find/src/find/matchers/entry.rs @@ -12,19 +12,8 @@ use std::{ path::{Path, PathBuf}, }; -use walkdir::DirEntry; - use super::Follow; -/// Wrapper for a directory entry. -#[derive(Debug)] -enum Entry { - /// Wraps an explicit path and depth. - Explicit(PathBuf, usize), - /// Wraps a WalkDir entry. - WalkDir(DirEntry), -} - /// File types. #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub enum FileType { @@ -163,22 +152,6 @@ impl From<&io::Error> for WalkError { } } -impl From for WalkError { - fn from(e: walkdir::Error) -> Self { - Self::from(&e) - } -} - -impl From<&walkdir::Error> for WalkError { - fn from(e: &walkdir::Error) -> Self { - Self { - path: e.path().map(|p| p.to_owned()), - depth: Some(e.depth()), - raw: e.io_error().and_then(|e| e.raw_os_error()), - } - } -} - impl From for io::Error { fn from(e: WalkError) -> Self { Self::from(&e) @@ -196,8 +169,10 @@ impl From<&WalkError> for io::Error { /// A path encountered while walking a file system. #[derive(Debug)] pub struct WalkEntry { - /// The wrapped path/dirent. - inner: Entry, + /// Filesystem path for this entry. + path: PathBuf, + /// Depth below the traversal root. + depth: usize, /// Whether to follow symlinks. follow: Follow, /// Cached metadata. @@ -214,67 +189,17 @@ pub struct WalkEntry { impl WalkEntry { /// Create a new WalkEntry for a specific file. pub fn new(path: impl Into, depth: usize, follow: Follow) -> Self { - Self { - inner: Entry::Explicit(path.into(), depth), - follow, - meta: OnceCell::new(), - display: None, - } - } - - /// Convert a [walkdir::DirEntry] to a [WalkEntry]. Errors due to broken - /// symbolic links will be converted to valid entries, but other errors will - /// be propagated. - pub fn from_walkdir( - result: walkdir::Result, - follow: Follow, - ) -> Result { - let result = result.map_err(WalkError::from); - - match result { - Ok(entry) => { - let ret = if entry.depth() == 0 && follow != Follow::Never { - // DirEntry::file_type() is wrong for root symlinks when follow_root_links is - // set - Self::new(entry.path(), 0, follow) - } else { - Self { inner: Entry::WalkDir(entry), follow, meta: OnceCell::new(), display: None } - }; - Ok(ret) - }, - Err(e) if e.is_not_found() => { - // Detect broken symlinks and replace them with explicit entries - if let (Some(path), Some(depth)) = (e.path(), e.depth()) - && let Ok(meta) = path.symlink_metadata() - { - return Ok(Self { - inner: Entry::Explicit(path.into(), depth), - follow: Follow::Never, - meta: Ok(meta).into(), - display: None, - }); - } - - Err(e) - }, - Err(e) => Err(e), - } + Self { path: path.into(), depth, follow, meta: OnceCell::new(), display: None } } /// Get the path to this entry. pub fn path(&self) -> &Path { - match &self.inner { - Entry::Explicit(path, _) => path.as_path(), - Entry::WalkDir(ent) => ent.path(), - } + self.path.as_path() } /// Get the path to this entry. pub fn into_path(self) -> PathBuf { - match self.inner { - Entry::Explicit(path, _) => path, - Entry::WalkDir(ent) => ent.into_path(), - } + self.path } /// Path used for display (`-print`, `-ls`) and path-based matching @@ -300,25 +225,18 @@ impl WalkEntry { /// Get the name of this entry. pub fn file_name(&self) -> &OsStr { - match &self.inner { - Entry::Explicit(path, _) => { - // Path::file_name() only works if the last component is normal - path - .components() - .next_back() - .map(|c| c.as_os_str()) - .unwrap_or_else(|| path.as_os_str()) - }, - Entry::WalkDir(ent) => ent.file_name(), - } + // Path::file_name() only works if the last component is normal. + self + .path + .components() + .next_back() + .map(|c| c.as_os_str()) + .unwrap_or_else(|| self.path.as_os_str()) } /// Get the depth of this entry below the root. pub fn depth(&self) -> usize { - match &self.inner { - Entry::Explicit(_, depth) => *depth, - Entry::WalkDir(ent) => ent.depth(), - } + self.depth } /// Get whether symbolic links are followed for this entry. @@ -328,45 +246,35 @@ impl WalkEntry { /// Get the metadata on a cache miss. fn get_metadata(&self) -> Result { - self.follow.metadata_at_depth(self.path(), self.depth()) + self.follow.metadata_at_depth(&self.path, self.depth) } /// Get the [Metadata] for this entry, following symbolic links if /// appropriate. Multiple calls to this function will cache and re-use the /// same [Metadata]. pub fn metadata(&self) -> Result<&Metadata, WalkError> { - let result = self.meta.get_or_init(|| match &self.inner { - Entry::Explicit(..) => Ok(self.get_metadata()?), - Entry::WalkDir(ent) => Ok(ent.metadata()?), - }); + let result = self.meta.get_or_init(|| self.get_metadata()); result.as_ref().map_err(|e| e.clone()) } /// Get the file type of this entry. pub fn file_type(&self) -> FileType { - match &self.inner { - Entry::Explicit(..) => self - .metadata() - .map(|m| m.file_type().into()) - .unwrap_or(FileType::Unknown), - Entry::WalkDir(ent) => ent.file_type().into(), - } + self + .metadata() + .map(|m| m.file_type().into()) + .unwrap_or(FileType::Unknown) } /// Check whether this entry is a symbolic link, regardless of whether links /// are being followed. pub fn path_is_symlink(&self) -> bool { - match &self.inner { - Entry::Explicit(path, _) => { - if self.follow() { - path - .symlink_metadata() - .is_ok_and(|m| m.file_type().is_symlink()) - } else { - self.file_type().is_symlink() - } - }, - Entry::WalkDir(ent) => ent.path_is_symlink(), + if self.follow() { + self + .path + .symlink_metadata() + .is_ok_and(|m| m.file_type().is_symlink()) + } else { + self.file_type().is_symlink() } } } diff --git a/crates/vendor/uu-find/src/find/mod.rs b/crates/vendor/uu-find/src/find/mod.rs index 713f6ac7a..cc9f79386 100644 --- a/crates/vendor/uu-find/src/find/mod.rs +++ b/crates/vendor/uu-find/src/find/mod.rs @@ -7,9 +7,9 @@ pub mod matchers; use std::{ - cell::RefCell, + cell::{Cell, RefCell}, error::Error, - io::Write, + io::{self, Write}, path::{Path, PathBuf}, rc::Rc, time::SystemTime, @@ -17,7 +17,6 @@ use std::{ use matchers::{Follow, WalkEntry}; use pi_uutils_ctx::{stderr, stdout}; -use walkdir::WalkDir; pub struct Config { same_file_system: bool, @@ -45,9 +44,9 @@ impl Default for Config { help_requested: false, version_requested: false, today_start: false, - // Directory information and traversal are done by walkdir, - // and this configuration field will exist as - // a compatibility item for GNU findutils. + // Directory information and traversal are handled by pi_walker, + // and this configuration field exists as a compatibility item for + // GNU findutils. no_leaf_dirs: false, follow: Follow::Never, new_paths: None, // This option exclusively for -files0-from argument. @@ -151,6 +150,157 @@ fn parse_args(args: &[&str]) -> Result> { Ok(ParsedInfo { matcher, paths, config }) } +fn apply_find_entry( + mut entry: WalkEntry, + operand: &Path, + resolved_root: &Path, + deps: &dyn Dependencies, + matcher: &dyn matchers::Matcher, + current_dir: &mut Option, + ret: &mut i32, +) -> (bool, bool) { + entry.set_display_root(operand, resolved_root); + let mut matcher_io = matchers::MatcherIO::new(deps); + + let new_dir = entry.path().parent().map(|x| x.to_path_buf()); + if new_dir != *current_dir { + if let Some(dir) = current_dir.take() { + matcher.finished_dir(dir.as_path(), &mut matcher_io); + } + *current_dir = new_dir; + } + + matcher.matches(&entry, &mut matcher_io); + match matcher_io.exit_code() { + 0 => {}, + code => *ret = code, + } + (matcher_io.should_quit(), matcher_io.should_skip_current_dir()) +} + +fn finish_find_walk( + deps: &dyn Dependencies, + matcher: &dyn matchers::Matcher, + current_dir: &mut Option, + ret: &mut i32, +) { + let mut matcher_io = matchers::MatcherIO::new(deps); + if let Some(dir) = current_dir.take() { + matcher.finished_dir(dir.as_path(), &mut matcher_io); + } + matcher.finished(&mut matcher_io); + // This is implemented for exec +. + match matcher_io.exit_code() { + 0 => {}, + code => *ret = code, + } +} + +fn walker_follow_links(follow: Follow) -> pi_walker::FollowLinks { + match follow { + Follow::Never => pi_walker::FollowLinks::Never, + Follow::Roots => pi_walker::FollowLinks::Roots, + Follow::Always => pi_walker::FollowLinks::Always, + } +} + +fn build_find_walk_request(config: &Config, root: &Path) -> pi_walker::WalkRequest { + pi_walker::WalkRequest::new(root) + .hidden(true) + .gitignore(false) + .skip_git(false) + .skip_node_modules(false) + .follow_links(walker_follow_links(config.follow)) + .detail(pi_walker::WalkDetail::Minimal) + .order(if config.sorted_output { + pi_walker::WalkOrder::Path + } else { + pi_walker::WalkOrder::Unordered + }) + .emit_root(true) + .depth(config.min_depth, config.max_depth) + .visit_order(if config.depth_first { + pi_walker::VisitOrder::ContentsFirst + } else { + pi_walker::VisitOrder::PreOrder + }) + .directory_errors(pi_walker::DirectoryErrorMode::Visit) + .same_file_system(config.same_file_system) + .cache(false) +} +fn process_dir_walk_request( + config: &Config, + deps: &dyn Dependencies, + matcher: &dyn matchers::Matcher, + quit: &mut bool, + resolved_root: &Path, + operand: &Path, +) -> i32 { + let request = build_find_walk_request(config, resolved_root); + let current_dir = RefCell::new(None); + let ret = Cell::new(0); + let local_quit = Cell::new(false); + let status = request.for_each_entry_with_heartbeat( + || Ok::<(), io::Error>(()), + |entry: pi_walker::EntryMeta<'_>| { + let walk_entry = + WalkEntry::new(entry.absolute_path.as_ref().to_path_buf(), entry.depth, config.follow); + let mut current_dir = current_dir.borrow_mut(); + let mut ret_value = ret.get(); + let (should_quit, should_skip_current_dir) = apply_find_entry( + walk_entry, + operand, + resolved_root, + deps, + matcher, + &mut current_dir, + &mut ret_value, + ); + ret.set(ret_value); + if should_quit { + local_quit.set(true); + Ok(pi_walker::WalkDecision::Stop) + } else if should_skip_current_dir { + Ok(pi_walker::WalkDecision::SkipDescend) + } else { + Ok(pi_walker::WalkDecision::Include) + } + }, + |error| { + ret.set(1); + writeln!(&mut stderr(), "Error: {}: {}", error.path.display(), error.error).unwrap(); + Ok(pi_walker::WalkDecision::Include) + }, + ); + let mut current_dir = current_dir.into_inner(); + let mut ret_value = ret.get(); + match status { + Ok(pi_walker::WalkStatus::Complete | pi_walker::WalkStatus::Stopped) => { + finish_find_walk(deps, matcher, &mut current_dir, &mut ret_value); + if local_quit.get() { + *quit = true; + } + ret_value + }, + Ok(pi_walker::WalkStatus::Unsupported) | Err(pi_walker::WalkError::Unsupported) => { + writeln!(&mut stderr(), "Error: directory scan unsupported").unwrap(); + 1 + }, + Err(pi_walker::WalkError::Interrupted(err)) => { + ret_value = 1; + writeln!(&mut stderr(), "Error: {err}").unwrap(); + finish_find_walk(deps, matcher, &mut current_dir, &mut ret_value); + ret_value + }, + Err(pi_walker::WalkError::InvalidData { path, message }) => { + ret_value = 1; + writeln!(&mut stderr(), "Error: {}: {message}", path.display()).unwrap(); + finish_find_walk(deps, matcher, &mut current_dir, &mut ret_value); + ret_value + }, + } +} + fn process_dir( dir: &str, config: &Config, @@ -160,71 +310,13 @@ fn process_dir( ) -> i32 { let resolved_root = pi_uutils_ctx::resolve(dir); let operand = Path::new(dir); - let mut walkdir = WalkDir::new(&resolved_root) - .contents_first(config.depth_first) - .max_depth(config.max_depth) - .min_depth(config.min_depth) - .same_file_system(config.same_file_system) - .follow_links(config.follow == Follow::Always) - .follow_root_links(config.follow != Follow::Never); - if config.sorted_output { - walkdir = walkdir.sort_by(|a, b| a.file_name().cmp(b.file_name())); + if config.min_depth > config.max_depth { + let mut current_dir = None; + let mut ret = 0; + finish_find_walk(deps, matcher, &mut current_dir, &mut ret); + return ret; } - - let mut ret = 0; - - // Slightly yucky loop handling here :-(. See docs for - // WalkDirIterator::skip_current_dir for explanation. - let mut it = walkdir.into_iter(); - // As WalkDir seems not providing a function to check its stack, - // using current_dir is a workaround to check leaving directory. - let mut current_dir: Option = None; - while let Some(result) = it.next() { - match WalkEntry::from_walkdir(result, config.follow) { - Err(err) => { - ret = 1; - writeln!(&mut stderr(), "Error: {err}").unwrap(); - }, - Ok(mut entry) => { - entry.set_display_root(operand, &resolved_root); - let mut matcher_io = matchers::MatcherIO::new(deps); - - let new_dir = entry.path().parent().map(|x| x.to_path_buf()); - if new_dir != current_dir { - if let Some(dir) = current_dir.take() { - matcher.finished_dir(dir.as_path(), &mut matcher_io); - } - current_dir = new_dir; - } - - matcher.matches(&entry, &mut matcher_io); - match matcher_io.exit_code() { - 0 => {}, - code => ret = code, - } - if matcher_io.should_quit() { - *quit = true; - break; - } - if matcher_io.should_skip_current_dir() { - it.skip_current_dir(); - } - }, - } - } - - let mut matcher_io = matchers::MatcherIO::new(deps); - if let Some(dir) = current_dir.take() { - matcher.finished_dir(dir.as_path(), &mut matcher_io); - } - matcher.finished(&mut matcher_io); - // This is implemented for exec +. - match matcher_io.exit_code() { - 0 => {}, - code => ret = code, - } - - ret + process_dir_walk_request(config, deps, matcher, quit, &resolved_root, operand) } fn do_find(args: &[&str], deps: &dyn Dependencies) -> Result> { diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 7db5d97a1..489ffc389 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,14 +4,15 @@ ### Changed +- Renamed the filesystem walker worker count environment variable from PI_GREP_WORKERS to PI_WALK_WORKERS +- Centralized filesystem traversal policy in `pi-walker`. + - Changed the in-session `/resume` session picker to open as a fullscreen window on the terminal's alternate screen, matching the startup `--resume` picker and `/settings`. It borrows the alt buffer for its lifetime (the transcript is untouched underneath) and enables mouse tracking — the wheel scrolls the list and a left click resumes the row under the pointer — with the keybinding hint and bottom border pinned to the screen bottom. Previously it mounted inline in the editor slot and rendered compactly without mouse support. ### Fixed - Fixed Git subcommands ignoring repository paths by stripping ambient Git environment variables - - Fixed concurrent bash commands cross-killing each other on cancel/timeout. Cancellation cleanup previously walked the whole host process tree and signalled every descendant spawned since a per-run baseline, so cancelling or timing out one command could SIGTERM an unrelated command's child still running in parallel (it looked "new" relative to the canceller's baseline). Each run now tracks only the processes it actually spawned (via a brush-core spawn-observer hook) and scopes its TERM/KILL waves to that set, leaving concurrent runs untouched. - - Fixed `/skill:` invocation losing the user's prompt context when the slash token was reached mid-prompt via the autocomplete. The slash-command parser now recognizes a `/skill:` token surrounded by whitespace in non-slash, non-local-execution drafts (in addition to the leading form) and threads the surrounding prose through to the skill as `args`, so the typed prompt survives both in the editor (see the TUI changelog) and in the dispatched skill message. Drafts that already begin with another slash command (`/compact /skill:foo`), a bash sigil (`!echo /skill:foo`, `!!echo /skill:foo`), or a python sigil (`$ run.py /skill:foo`, `$$ run.py /skill:foo`) keep their existing dispatcher precedence and are not hijacked by the mid-prompt skill parser. Applies to the interactive TUI, ACP, and RPC dispatch paths via the shared `parseSkillInvocation` helper in `extensibility/skills` ([#3913](https://github.com/can1357/oh-my-pi/issues/3913)). ## [16.2.9] - 2026-06-30 diff --git a/packages/coding-agent/src/cli/grep-cli.ts b/packages/coding-agent/src/cli/grep-cli.ts index a80a2959f..a89934b12 100644 --- a/packages/coding-agent/src/cli/grep-cli.ts +++ b/packages/coding-agent/src/cli/grep-cli.ts @@ -150,7 +150,7 @@ ${chalk.bold("Options:")} --no-gitignore Include files excluded by .gitignore ${chalk.bold("Environment:")} - PI_GREP_WORKERS=N Set filesystem walker workers (default 4, 0 = auto) + PI_WALK_WORKERS=N Set filesystem walker workers (default 4, 0 = auto) ${chalk.bold("Examples:")} ${APP_NAME} grep "import" src/ diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index 2dad0d444..eefd3a0ef 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -610,7 +610,7 @@ export interface FuzzyFindOptions { hidden?: boolean /** Respect .gitignore (default: true). */ gitignore?: boolean - /** Enable shared filesystem scan cache (default: false). */ + /** Enable walker scan caching (default: false). */ cache?: boolean /** Maximum number of matches to return (default: 100). */ maxResults?: number @@ -645,9 +645,9 @@ export declare function getWorkProfile(lastSeconds: number): WorkProfile * Resolves the search root, scans entries, applies glob and optional file-type * filters, and optionally streams each accepted match through `on_match`. * - * If `sortByMtime` is enabled with a finite `maxResults`, uncached scans keep - * only the current top results while traversing instead of collecting the full - * tree. + * When `sortByMtime` is enabled, the walker ranks matches by mtime before the + * native layer applies final symlink-aware file-type filtering and callback + * emission. * * # Errors * Returns an error when the search path cannot be resolved, the path is not a @@ -662,10 +662,7 @@ export interface GlobMatch { path: string /** Resolved filesystem type for the match. */ fileType: FileType - /** - * Modification time in milliseconds since Unix epoch (from - * `symlink_metadata`). - */ + /** Modification time in milliseconds since Unix epoch. */ mtime?: number /** File size in bytes for regular files. */ size?: number @@ -690,7 +687,7 @@ export interface GlobOptions { maxResults?: number /** Respect .gitignore files (default: true). */ gitignore?: boolean - /** Enable shared filesystem scan cache (default: false). */ + /** Enable walker scan caching (default: false). */ cache?: boolean /** Sort results by mtime (most recent first) before applying limit. */ sortByMtime?: boolean @@ -890,13 +887,13 @@ export interface HtmlToMarkdownOptions { } /** - * Invalidate the filesystem scan cache. + * Invalidate the walker scan cache. * * When called with a path, removes entries for roots containing that path. * When called without a path, clears the entire cache. * - * Intended to be called after agent file mutations (write, edit, rename, - * delete). + * Intended to be called after agent file mutations: write, edit, rename, or + * delete. */ export declare function invalidateFsScanCache(path?: string | undefined | null): void