diff --git a/Cargo.lock b/Cargo.lock index 5cd6d3ef9..cc40d1526 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1115,6 +1115,18 @@ dependencies = [ "syn", ] +[[package]] +name = "dns-lookup" +version = "3.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6e39034cee21a2f5bbb66ba0e3689819c4bb5d00382a282006e802a7ffa6c41d" +dependencies = [ + "cfg-if", + "libc", + "socket2", + "windows-sys 0.60.2", +] + [[package]] name = "downcast-rs" version = "1.2.1" @@ -3022,6 +3034,17 @@ dependencies = [ "windows-link", ] +[[package]] +name = "parse_datetime" +version = "0.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "413775a7eac2261d2211a79d10ef275e5b6f7b527eec42ad09adce2ffa92b6e5" +dependencies = [ + "jiff", + "num-traits", + "winnow 0.7.15", +] + [[package]] name = "pcre2" version = "0.2.11" @@ -3392,6 +3415,7 @@ dependencies = [ "parking_lot", "pi-uutils-ctx", "pi-walker", + "pi_uu_diff", "pi_uu_grep", "regex", "serde", @@ -3406,28 +3430,44 @@ dependencies = [ "uu_cat", "uu_comm", "uu_cut", + "uu_date", "uu_dirname", "uu_find", "uu_head", + "uu_hostname", + "uu_ln", "uu_ls", "uu_md5sum", "uu_mkdir", + "uu_mktemp", "uu_mv", + "uu_nproc", "uu_paste", + "uu_printenv", + "uu_readlink", + "uu_realpath", "uu_rm", "uu_sed", + "uu_seq", "uu_sha1sum", "uu_sha224sum", "uu_sha256sum", "uu_sha384sum", "uu_sha512sum", "uu_sort", + "uu_stat", + "uu_tac", "uu_tail", "uu_tee", + "uu_touch", "uu_tr", + "uu_truncate", + "uu_uname", "uu_uniq", "uu_wc", + "uu_whoami", "uu_xargs", + "uu_yes", "windows-sys 0.61.2", "winreg 0.56.0", "xxhash-rust", @@ -3453,6 +3493,17 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "pi_uu_diff" +version = "0.8.0" +dependencies = [ + "clap", + "parking_lot", + "pi-uutils-ctx", + "similar 3.1.1", + "tempfile", +] + [[package]] name = "pi_uu_grep" version = "0.8.0" @@ -3484,6 +3535,16 @@ version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e" +[[package]] +name = "platform-info" +version = "2.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9368d62437c8cbb7c31ee37fd8c08a7d390e09a3ff75698a674953f46705ffcb" +dependencies = [ + "libc", + "windows-sys 0.59.0", +] + [[package]] name = "png" version = "0.18.1" @@ -4474,7 +4535,7 @@ dependencies = [ "toml_datetime", "toml_parser", "toml_writer", - "winnow", + "winnow 1.0.3", ] [[package]] @@ -4495,7 +4556,7 @@ dependencies = [ "indexmap", "toml_datetime", "toml_parser", - "winnow", + "winnow 1.0.3", ] [[package]] @@ -4504,7 +4565,7 @@ version = "1.1.2+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a2abe9b86193656635d2411dc43050282ca48aa31c2451210f4202550afb7526" dependencies = [ - "winnow", + "winnow 1.0.3", ] [[package]] @@ -5338,6 +5399,21 @@ dependencies = [ "uucore 0.8.0", ] +[[package]] +name = "uu_date" +version = "0.8.0" +dependencies = [ + "clap", + "jiff", + "parking_lot", + "parse_datetime", + "pi-uutils-ctx", + "regex", + "rustix", + "tempfile", + "uucore 0.8.0", +] + [[package]] name = "uu_dirname" version = "0.8.0" @@ -5377,6 +5453,31 @@ dependencies = [ "uucore 0.8.0", ] +[[package]] +name = "uu_hostname" +version = "0.8.0" +dependencies = [ + "clap", + "dns-lookup", + "hostname", + "parking_lot", + "pi-uutils-ctx", + "uucore 0.8.0", + "windows-sys 0.61.2", +] + +[[package]] +name = "uu_ln" +version = "0.8.0" +dependencies = [ + "clap", + "parking_lot", + "pi-uutils-ctx", + "tempfile", + "thiserror 2.0.18", + "uucore 0.8.0", +] + [[package]] name = "uu_ls" version = "0.8.0" @@ -5412,6 +5513,19 @@ dependencies = [ "uucore 0.8.0", ] +[[package]] +name = "uu_mktemp" +version = "0.8.0" +dependencies = [ + "clap", + "parking_lot", + "pi-uutils-ctx", + "rand 0.10.2", + "tempfile", + "thiserror 2.0.18", + "uucore 0.8.0", +] + [[package]] name = "uu_mv" version = "0.8.0" @@ -5427,6 +5541,17 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "uu_nproc" +version = "0.8.0" +dependencies = [ + "clap", + "libc", + "parking_lot", + "pi-uutils-ctx", + "uucore 0.8.0", +] + [[package]] name = "uu_paste" version = "0.8.0" @@ -5436,6 +5561,38 @@ dependencies = [ "uucore 0.8.0", ] +[[package]] +name = "uu_printenv" +version = "0.8.0" +dependencies = [ + "clap", + "parking_lot", + "pi-uutils-ctx", + "uucore 0.8.0", +] + +[[package]] +name = "uu_readlink" +version = "0.8.0" +dependencies = [ + "clap", + "parking_lot", + "pi-uutils-ctx", + "tempfile", + "uucore 0.8.0", +] + +[[package]] +name = "uu_realpath" +version = "0.8.0" +dependencies = [ + "clap", + "parking_lot", + "pi-uutils-ctx", + "tempfile", + "uucore 0.8.0", +] + [[package]] name = "uu_rm" version = "0.8.0" @@ -5464,6 +5621,20 @@ dependencies = [ "uucore 0.9.0", ] +[[package]] +name = "uu_seq" +version = "0.8.0" +dependencies = [ + "bigdecimal", + "clap", + "num-bigint", + "num-traits", + "parking_lot", + "pi-uutils-ctx", + "thiserror 2.0.18", + "uucore 0.8.0", +] + [[package]] name = "uu_sha1sum" version = "0.8.0" @@ -5536,6 +5707,33 @@ dependencies = [ "uucore 0.8.0", ] +[[package]] +name = "uu_stat" +version = "0.8.0" +dependencies = [ + "clap", + "parking_lot", + "pi-uutils-ctx", + "tempfile", + "thiserror 2.0.18", + "uucore 0.8.0", +] + +[[package]] +name = "uu_tac" +version = "0.8.0" +dependencies = [ + "clap", + "memchr", + "memmap2", + "parking_lot", + "pi-uutils-ctx", + "regex", + "tempfile", + "thiserror 2.0.18", + "uucore 0.8.0", +] + [[package]] name = "uu_tail" version = "0.8.0" @@ -5560,6 +5758,24 @@ dependencies = [ "uucore 0.8.0", ] +[[package]] +name = "uu_touch" +version = "0.8.0" +dependencies = [ + "clap", + "filetime", + "jiff", + "libc", + "parking_lot", + "parse_datetime", + "pi-uutils-ctx", + "rustix", + "tempfile", + "thiserror 2.0.18", + "uucore 0.8.0", + "windows-sys 0.61.2", +] + [[package]] name = "uu_tr" version = "0.8.0" @@ -5571,6 +5787,28 @@ dependencies = [ "uucore 0.8.0", ] +[[package]] +name = "uu_truncate" +version = "0.8.0" +dependencies = [ + "clap", + "parking_lot", + "pi-uutils-ctx", + "tempfile", + "uucore 0.8.0", +] + +[[package]] +name = "uu_uname" +version = "0.8.0" +dependencies = [ + "clap", + "parking_lot", + "pi-uutils-ctx", + "platform-info", + "uucore 0.8.0", +] + [[package]] name = "uu_uniq" version = "0.8.0" @@ -5594,6 +5832,17 @@ dependencies = [ "uucore 0.8.0", ] +[[package]] +name = "uu_whoami" +version = "0.8.0" +dependencies = [ + "clap", + "parking_lot", + "pi-uutils-ctx", + "uucore 0.8.0", + "windows-sys 0.61.2", +] + [[package]] name = "uu_xargs" version = "0.8.0" @@ -5605,6 +5854,17 @@ dependencies = [ "tempfile", ] +[[package]] +name = "uu_yes" +version = "0.8.0" +dependencies = [ + "clap", + "itertools", + "parking_lot", + "pi-uutils-ctx", + "uucore 0.8.0", +] + [[package]] name = "uucore" version = "0.0.30" @@ -6330,6 +6590,15 @@ version = "0.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d6bbff5f0aada427a1e5a6da5f1f98158182f26556f345ac9e04d36d0ebed650" +[[package]] +name = "winnow" +version = "0.7.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "df79d97927682d2fd8adb29682d1140b343be4ac0f08fd68b7765d9c059d3945" +dependencies = [ + "memchr", +] + [[package]] name = "winnow" version = "1.0.3" diff --git a/crates/pi-shell/Cargo.toml b/crates/pi-shell/Cargo.toml index adad30dd9..f62d2d3b3 100644 --- a/crates/pi-shell/Cargo.toml +++ b/crates/pi-shell/Cargo.toml @@ -54,6 +54,23 @@ uu_sha512sum = { path = "../vendor/uu-sha512sum" } uu_b2sum = { path = "../vendor/uu-b2sum" } uu_basename = { path = "../vendor/uu-basename" } uu_dirname = { path = "../vendor/uu-dirname" } +uu_readlink = { path = "../vendor/uu-readlink" } +uu_realpath = { path = "../vendor/uu-realpath" } +uu_touch = { path = "../vendor/uu-touch" } +uu_stat = { path = "../vendor/uu-stat" } +uu_date = { path = "../vendor/uu-date" } +uu_mktemp = { path = "../vendor/uu-mktemp" } +uu_seq = { path = "../vendor/uu-seq" } +uu_yes = { path = "../vendor/uu-yes" } +uu_printenv = { path = "../vendor/uu-printenv" } +uu_ln = { path = "../vendor/uu-ln" } +uu_truncate = { path = "../vendor/uu-truncate" } +uu_tac = { path = "../vendor/uu-tac" } +uu_nproc = { path = "../vendor/uu-nproc" } +uu_uname = { path = "../vendor/uu-uname" } +uu_whoami = { path = "../vendor/uu-whoami" } +uu_hostname = { path = "../vendor/uu-hostname" } +pi_uu_diff = { path = "../pi-uu-diff" } uu_cut = { path = "../vendor/uu-cut" } uu_tee = { path = "../vendor/uu-tee" } uu_tr = { path = "../vendor/uu-tr" } diff --git a/crates/pi-shell/src/coreutils.rs b/crates/pi-shell/src/coreutils.rs index a1372f404..caf7c064c 100644 --- a/crates/pi-shell/src/coreutils.rs +++ b/crates/pi-shell/src/coreutils.rs @@ -219,6 +219,23 @@ uutil_builtin!(pub fn sha512sum_builtin => uu_sha512sum::run); uutil_builtin!(pub fn b2sum_builtin => uu_b2sum::run); uutil_builtin!(pub fn basename_builtin => uu_basename::run); uutil_builtin!(pub fn dirname_builtin => uu_dirname::run); +uutil_builtin!(pub fn readlink_builtin => uu_readlink::run); +uutil_builtin!(pub fn realpath_builtin => uu_realpath::run); +uutil_builtin!(pub fn touch_builtin => uu_touch::run); +uutil_builtin!(pub fn stat_builtin => uu_stat::run); +uutil_builtin!(pub fn date_builtin => uu_date::run); +uutil_builtin!(pub fn mktemp_builtin => uu_mktemp::run); +uutil_builtin!(pub fn seq_builtin => uu_seq::run); +uutil_builtin!(pub fn yes_builtin => uu_yes::run); +uutil_builtin!(pub fn printenv_builtin => uu_printenv::run); +uutil_builtin!(pub fn ln_builtin => uu_ln::run); +uutil_builtin!(pub fn truncate_builtin => uu_truncate::run); +uutil_builtin!(pub fn tac_builtin => uu_tac::run); +uutil_builtin!(pub fn nproc_builtin => uu_nproc::run); +uutil_builtin!(pub fn uname_builtin => uu_uname::run); +uutil_builtin!(pub fn whoami_builtin => uu_whoami::run); +uutil_builtin!(pub fn hostname_builtin => uu_hostname::run); +uutil_builtin!(pub fn diff_builtin => pi_uu_diff::run); uutil_builtin!(pub fn cut_builtin => uu_cut::run); uutil_builtin!(pub fn tee_builtin => uu_tee::run); uutil_builtin!(pub fn tr_builtin => uu_tr::run); diff --git a/crates/pi-shell/src/lib.rs b/crates/pi-shell/src/lib.rs index 2f3f5929f..204724703 100644 --- a/crates/pi-shell/src/lib.rs +++ b/crates/pi-shell/src/lib.rs @@ -4,6 +4,7 @@ mod fd; pub mod minimizer; pub mod process; pub mod shell; +mod which; #[cfg(windows)] pub mod windows; diff --git a/crates/pi-shell/src/shell.rs b/crates/pi-shell/src/shell.rs index 46e454e8f..f2254a258 100644 --- a/crates/pi-shell/src/shell.rs +++ b/crates/pi-shell/src/shell.rs @@ -632,6 +632,23 @@ async fn create_session_for_run( shell.register_builtin("b2sum", crate::coreutils::b2sum_builtin()); shell.register_builtin("basename", crate::coreutils::basename_builtin()); shell.register_builtin("dirname", crate::coreutils::dirname_builtin()); + shell.register_builtin("readlink", crate::coreutils::readlink_builtin()); + shell.register_builtin("realpath", crate::coreutils::realpath_builtin()); + shell.register_builtin("touch", crate::coreutils::touch_builtin()); + shell.register_builtin("stat", crate::coreutils::stat_builtin()); + shell.register_builtin("date", crate::coreutils::date_builtin()); + shell.register_builtin("mktemp", crate::coreutils::mktemp_builtin()); + shell.register_builtin("seq", crate::coreutils::seq_builtin()); + shell.register_builtin("yes", crate::coreutils::yes_builtin()); + shell.register_builtin("printenv", crate::coreutils::printenv_builtin()); + shell.register_builtin("truncate", crate::coreutils::truncate_builtin()); + shell.register_builtin("tac", crate::coreutils::tac_builtin()); + shell.register_builtin("nproc", crate::coreutils::nproc_builtin()); + shell.register_builtin("uname", crate::coreutils::uname_builtin()); + shell.register_builtin("whoami", crate::coreutils::whoami_builtin()); + shell.register_builtin("hostname", crate::coreutils::hostname_builtin()); + shell.register_builtin("which", crate::which::which_builtin()); + shell.register_builtin("diff", crate::coreutils::diff_builtin()); shell.register_builtin("cut", crate::coreutils::cut_builtin()); shell.register_builtin("tee", crate::coreutils::tee_builtin()); shell.register_builtin("tr", crate::coreutils::tr_builtin()); @@ -647,6 +664,8 @@ async fn create_session_for_run( if !uutils_env_disabled(config, "PI_DISABLE_MV_BUILTIN") { shell.register_builtin("mv", crate::coreutils::mv_builtin()); } + // ln can clobber existing files via -f; gate it with the destructive set. + shell.register_builtin("ln", crate::coreutils::ln_builtin()); } } diff --git a/crates/pi-shell/src/which.rs b/crates/pi-shell/src/which.rs new file mode 100644 index 000000000..253a414e7 --- /dev/null +++ b/crates/pi-shell/src/which.rs @@ -0,0 +1,252 @@ +//! In-process `which` builtin backed by brush's PATH-search helpers. +//! +//! Follows which(1) (GNU/debianutils) semantics: each name operand is looked +//! up in the shell's `PATH`; the first match is printed (all matches with +//! `-a`). Lookup failures are silent; the exit status is 0 when every name +//! was found and 1 when any name was missing. + +use std::{ + ffi::OsString, + io::{self, Write}, + path::{Path, PathBuf}, +}; + +use brush_core::{ + Error, + builtins::{BoxFuture, ContentOptions, ContentType, Registration}, + commands::{CommandArg, ExecutionContext}, + extensions::ShellExtensions, + openfiles::{OpenFile, OpenFiles, null}, + pathsearch, + results::ExecutionResult, + sys, +}; +use clap::{Parser, error::ErrorKind}; + +#[derive(Parser, Debug)] +#[command(name = "which", about = "Locate a command's executable in the shell's PATH")] +struct WhichCli { + /// Print all matching executables in PATH, not just the first. + #[arg(short = 'a', long = "all")] + all: bool, + + /// Command names to locate. + #[arg(value_name = "name")] + names: Vec, +} + +/// Creates the `which` shell builtin registration. +pub fn which_builtin() -> Registration { + fn execute( + context: ExecutionContext<'_, SE>, + args: Vec, + ) -> BoxFuture<'_, Result> { + Box::pin(std::future::ready(Ok(run_which(context, args)))) + } + + Registration { + execute_func: execute::, + content_func: which_content, + disabled: false, + special_builtin: false, + declaration_builtin: false, + transparent_background_wrapper: false, + } +} + +fn run_which( + context: ExecutionContext<'_, SE>, + args: Vec, +) -> ExecutionResult { + let mut stdout = context + .try_fd(OpenFiles::STDOUT_FD) + .unwrap_or_else(null_sink); + let mut stderr = context + .try_fd(OpenFiles::STDERR_FD) + .unwrap_or_else(null_sink); + let cwd = context.shell.working_dir().to_path_buf(); + let path_var = context + .shell + .env_str("PATH") + .map(std::borrow::Cow::into_owned) + .unwrap_or_default(); + let argv: Vec = args + .iter() + .map(|arg| OsString::from(arg.to_string())) + .collect(); + + let cli = match WhichCli::try_parse_from(argv) { + Ok(cli) => cli, + Err(err) => { + let rendered = err.to_string(); + let code = match err.kind() { + ErrorKind::DisplayHelp | ErrorKind::DisplayVersion => { + let _ = write!(stdout, "{rendered}"); + 0 + }, + _ => { + let _ = write!(stderr, "{rendered}"); + 2 + }, + }; + return ExecutionResult::new(code); + }, + }; + + let mut all_found = true; + for name in &cli.names { + let matches = find_matches(name, &path_var, &cwd, cli.all); + if matches.is_empty() { + // which(1) reports missing names via the exit status only. + all_found = false; + } + for path in matches { + let _ = writeln!(stdout, "{}", path.display()); + } + } + + ExecutionResult::new(u8::from(!all_found)) +} + +/// Collects the executable matches for a single `which` name operand. +/// +/// A name containing a path separator is checked directly against `cwd` +/// (yielding at most one match); otherwise each `PATH` entry — with relative +/// and empty entries resolved against `cwd` — is probed in `PATH` order. +/// Returns only the first match unless `all` is set. Windows `PATHEXT` +/// resolution is handled by [`brush_core::sys::fs::resolve_executable`]. +fn find_matches(name: &str, path_var: &str, cwd: &Path, all: bool) -> Vec { + if sys::fs::contains_path_separator(name) { + let candidate = cwd.join(name); + if candidate.is_dir() { + return Vec::new(); + } + return sys::fs::resolve_executable(candidate).into_iter().collect(); + } + + let dirs = sys::fs::split_paths(path_var).map(|dir| { + if dir.as_os_str().is_empty() { + // POSIX: an empty PATH entry names the current directory. + cwd.to_path_buf() + } else if dir.is_relative() { + cwd.join(dir) + } else { + dir + } + }); + + let mut found = pathsearch::search_for_executable(dirs, name); + if all { + found.collect() + } else { + found.next().into_iter().collect() + } +} + +fn null_sink() -> OpenFile { + null().unwrap_or_else(|_| OpenFile::from(io::stdout())) +} + +#[allow( + clippy::unnecessary_wraps, + reason = "signature must match brush's CommandContentFunc fn pointer" +)] +fn which_content( + _name: &str, + _content_type: ContentType, + _options: &ContentOptions, +) -> Result { + Ok("which: which [-a] name [name ...]\n".to_string()) +} + +#[cfg(test)] +#[cfg(unix)] +mod tests { + use std::{ + env, fs, + os::unix::fs::PermissionsExt, + path::PathBuf, + sync::atomic::{AtomicUsize, Ordering}, + time::{SystemTime, UNIX_EPOCH}, + }; + + use super::find_matches; + + static COUNTER: AtomicUsize = AtomicUsize::new(0); + + /// Creates a fresh, canonicalized temp directory (macOS `/var` is a + /// symlink; canonicalizing keeps constructed and probed paths identical). + fn temp_root(tag: &str) -> PathBuf { + let nanos = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_or(0, |d| d.as_nanos()); + let root = env::temp_dir().join(format!( + "pi-shell-which-{tag}-{}-{}-{}", + std::process::id(), + nanos, + COUNTER.fetch_add(1, Ordering::Relaxed), + )); + fs::create_dir_all(&root).expect("temp dir should be created"); + fs::canonicalize(&root).expect("temp dir should canonicalize") + } + + fn place_file(dir: &std::path::Path, name: &str, executable: bool) -> PathBuf { + let path = dir.join(name); + fs::write(&path, b"#!/bin/sh\n").expect("file should be written"); + let mode = if executable { 0o755 } else { 0o644 }; + fs::set_permissions(&path, fs::Permissions::from_mode(mode)) + .expect("permissions should be set"); + path + } + + #[test] + fn finds_only_executable_files() { + let dir = temp_root("exec-only"); + let tool = place_file(&dir, "tool", true); + place_file(&dir, "blob", false); + let path_var = dir.display().to_string(); + + assert_eq!(find_matches("tool", &path_var, &dir, false), vec![tool]); + assert!(find_matches("blob", &path_var, &dir, false).is_empty()); + assert!(find_matches("missing", &path_var, &dir, false).is_empty()); + } + + #[test] + fn all_flag_returns_matches_in_path_order() { + let dir_a = temp_root("all-a"); + let dir_b = temp_root("all-b"); + let tool_a = place_file(&dir_a, "tool", true); + let tool_b = place_file(&dir_b, "tool", true); + let path_var = format!("{}:{}", dir_a.display(), dir_b.display()); + let cwd = temp_root("all-cwd"); + + assert_eq!(find_matches("tool", &path_var, &cwd, true), vec![tool_a.clone(), tool_b]); + // Without -a only the first PATH entry's match is returned. + assert_eq!(find_matches("tool", &path_var, &cwd, false), vec![tool_a]); + } + + #[test] + fn name_with_separator_resolves_against_cwd() { + let cwd = temp_root("slash"); + let bin = cwd.join("bin"); + fs::create_dir_all(&bin).expect("bin dir should be created"); + let tool = place_file(&bin, "tool", true); + place_file(&bin, "blob", false); + + // PATH is irrelevant for names containing a separator. + assert_eq!(find_matches("bin/tool", "", &cwd, false), vec![tool]); + assert!(find_matches("bin/blob", "", &cwd, false).is_empty()); + // A directory is never a match, even with execute bits set. + assert!(find_matches("./bin", "", &cwd, false).is_empty()); + } + + #[test] + fn relative_path_entries_resolve_against_cwd() { + let cwd = temp_root("rel-entry"); + let bin = cwd.join("bin"); + fs::create_dir_all(&bin).expect("bin dir should be created"); + let tool = place_file(&bin, "tool", true); + + assert_eq!(find_matches("tool", "bin", &cwd, false), vec![tool]); + } +} diff --git a/crates/pi-uu-diff/Cargo.toml b/crates/pi-uu-diff/Cargo.toml new file mode 100644 index 000000000..4d9bed570 --- /dev/null +++ b/crates/pi-uu-diff/Cargo.toml @@ -0,0 +1,21 @@ +# diff implemented from scratch on top of the `similar` diffing library, with +# I/O and path resolution routed through pi-uutils-ctx so it can run in-process +# as a shell builtin. Entry point: `pi_uu_diff::run`. +[package] +name = "pi_uu_diff" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "diff ~ similar-backed file comparison (in-process shell builtin)" + +[lib] +path = "src/lib.rs" + +[dependencies] +clap = { version = "4", features = ["wrap_help"] } +pi-uutils-ctx = { path = "../pi-uutils-ctx" } +similar = "3.1.0" + +[dev-dependencies] +parking_lot.workspace = true +tempfile = "3" diff --git a/crates/pi-uu-diff/src/lib.rs b/crates/pi-uu-diff/src/lib.rs new file mode 100644 index 000000000..330e8833e --- /dev/null +++ b/crates/pi-uu-diff/src/lib.rs @@ -0,0 +1,608 @@ +//! `diff` implemented as an in-process shell builtin on top of the `similar` +//! diffing library. All I/O and path resolution is routed through +//! `pi-uutils-ctx` so the builtin writes to the command's redirected file +//! descriptors and resolves relative paths against the shell's working +//! directory, while operands are printed as typed. +//! +//! Scope: unified output only (`-u` is accepted and implied, `-U N` controls +//! the context size), `-q/--brief`, `-N/--new-file` (absent files compare as +//! empty), binary detection, `-` for the context stdin, and unconditional +//! recursive directory comparison (`Only in : ` lines plus +//! `diff -r A/x B/x`-headed per-pair diffs). +//! +//! Entry point: [`run`]. It never calls `std::process::exit`; clap +//! help/usage/error output is rendered to the context streams and an exit code +//! is returned following the GNU convention (0 = identical, 1 = differences +//! found, 2 = trouble). + +use std::{ + collections::BTreeSet, + ffi::{OsStr, OsString}, + fs, + io::{Read, Write}, + path::{Path, PathBuf}, +}; + +use clap::{Arg, ArgAction, ArgMatches, Command}; +use pi_uutils_ctx::format_usage; +use similar::TextDiff; + +const OPT_UNIFIED_FLAG: &str = "unified-flag"; +const OPT_UNIFIED: &str = "unified"; +const OPT_BRIEF: &str = "brief"; +const OPT_RECURSIVE: &str = "recursive"; +const OPT_NEW_FILE: &str = "new-file"; +const OPT_COLOR: &str = "color"; +const ARG_FILES: &str = "files"; + +/// In-process builtin entry point. Parses the arguments directly, renders clap +/// help/usage/version to the context streams, and maps errors to the GNU diff +/// exit-code convention, so it is safe to run inside the host shell process. +pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 2; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match diff_main(&matches) { + Ok(code) => code, + Err(msg) => { + let _ = writeln!(pi_uutils_ctx::stderr(), "diff: {msg}"); + 2 + }, + } +} + +pub fn uu_app() -> Command { + Command::new("diff") + .version(concat!("diff (pi-uu-diff) ", env!("CARGO_PKG_VERSION"))) + .about("Compare files line by line.") + .override_usage(format_usage("diff [OPTION]... FILE1 FILE2")) + .infer_long_args(true) + .arg( + Arg::new(OPT_UNIFIED_FLAG) + .short('u') + .help("output 3 lines of unified context (the default output format)") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_UNIFIED) + .short('U') + .long(OPT_UNIFIED) + .value_name("NUM") + .help("output NUM lines of unified context") + .value_parser(clap::value_parser!(usize)), + ) + .arg( + Arg::new(OPT_BRIEF) + .short('q') + .long(OPT_BRIEF) + .help("report only when files differ") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_RECURSIVE) + .short('r') + .long(OPT_RECURSIVE) + .help("recursively compare subdirectories (always on for directories)") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_NEW_FILE) + .short('N') + .long(OPT_NEW_FILE) + .help("treat absent files as empty") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_COLOR) + .long(OPT_COLOR) + .value_name("WHEN") + .num_args(0..=1) + .require_equals(true) + .default_missing_value("auto") + .help("accepted for compatibility; output is never colorized"), + ) + .arg( + Arg::new(ARG_FILES) + .required(true) + .num_args(2) + .value_parser(clap::value_parser!(OsString)) + .value_hint(clap::ValueHint::AnyPath), + ) +} + +#[derive(Clone, Copy)] +struct Options { + context: usize, + brief: bool, + new_file: bool, +} + +/// A classified operand: what the name as typed refers to on disk after +/// resolution against the scope working directory. +enum Operand { + /// The context stdin (`-`). + Stdin, + /// A regular (or other non-directory) file at the resolved path. + File(PathBuf), + /// A directory at the resolved path. + Dir(PathBuf), + /// A missing file tolerated by `-N` and compared as empty. + Absent, +} + +fn diff_main(matches: &ArgMatches) -> Result { + let files: Vec<&OsString> = matches.get_many::(ARG_FILES).unwrap().collect(); + let opts = Options { + context: matches.get_one::(OPT_UNIFIED).copied().unwrap_or(3), + brief: matches.get_flag(OPT_BRIEF), + new_file: matches.get_flag(OPT_NEW_FILE), + }; + + let (mut name_a, mut name_b) = (PathBuf::from(files[0]), PathBuf::from(files[1])); + let mut op_a = classify(&name_a, opts.new_file)?; + let mut op_b = classify(&name_b, opts.new_file)?; + + // GNU: comparing a directory with a non-directory compares + // / with the other operand. + let a_is_dir = matches!(op_a, Operand::Dir(_)); + let b_is_dir = matches!(op_b, Operand::Dir(_)); + if a_is_dir != b_is_dir { + if matches!(op_a, Operand::Stdin) || matches!(op_b, Operand::Stdin) { + return Err("cannot compare '-' to a directory".to_string()); + } + if a_is_dir { + name_a = descend(&name_a, &name_b)?; + op_a = classify(&name_a, opts.new_file)?; + } else { + name_b = descend(&name_b, &name_a)?; + op_b = classify(&name_b, opts.new_file)?; + } + } + + let differed = if let (Operand::Dir(res_a), Operand::Dir(res_b)) = (&op_a, &op_b) { + diff_dirs(&name_a, res_a, &name_b, res_b, opts)? + } else { + let bytes_a = read_operand(&op_a, &name_a)?; + let bytes_b = read_operand(&op_b, &name_b)?; + diff_pair(&name_a, &bytes_a, &name_b, &bytes_b, opts, None)? + }; + Ok(i32::from(differed)) +} + +/// Replaces a directory operand with `/` for the GNU +/// dir-vs-file comparison form. +fn descend(dir: &Path, other: &Path) -> Result { + let base = other + .file_name() + .ok_or_else(|| format!("cannot compare {} to a directory", other.display()))?; + Ok(dir.join(base)) +} + +fn classify(name: &Path, new_file: bool) -> Result { + if name.as_os_str() == OsStr::new("-") { + return Ok(Operand::Stdin); + } + // Resolve the operand against the shell working directory; `name` is kept + // for display (GNU prints operands as typed). + let resolved = pi_uutils_ctx::resolve(name); + match fs::metadata(&resolved) { + Ok(meta) if meta.is_dir() => Ok(Operand::Dir(resolved)), + Ok(_) => Ok(Operand::File(resolved)), + Err(err) if err.kind() == std::io::ErrorKind::NotFound && new_file => Ok(Operand::Absent), + Err(err) => Err(format!("{}: {}", name.display(), io_msg(&err))), + } +} + +fn read_operand(op: &Operand, name: &Path) -> Result, String> { + match op { + Operand::Stdin => { + let mut buf = Vec::new(); + pi_uutils_ctx::stdin() + .read_to_end(&mut buf) + .map_err(|err| format!("-: {}", io_msg(&err)))?; + Ok(buf) + }, + Operand::File(resolved) => { + fs::read(resolved).map_err(|err| format!("{}: {}", name.display(), io_msg(&err))) + }, + Operand::Dir(_) => unreachable!("directories are handled by diff_dirs"), + Operand::Absent => Ok(Vec::new()), + } +} + +/// Diffs one pair of already-read inputs, writing to the context stdout. +/// `prefix` is the `diff -r A/x B/x` line emitted before per-pair output in +/// directory mode. Returns whether the inputs differed. +fn diff_pair( + name_a: &Path, + bytes_a: &[u8], + name_b: &Path, + bytes_b: &[u8], + opts: Options, + prefix: Option<&str>, +) -> Result { + if bytes_a == bytes_b { + return Ok(false); + } + let mut out = pi_uutils_ctx::stdout(); + let (label_a, label_b) = (name_a.display().to_string(), name_b.display().to_string()); + if opts.brief { + writeln!(out, "Files {label_a} and {label_b} differ").map_err(|e| io_msg(&e))?; + return Ok(true); + } + if is_binary(bytes_a) || is_binary(bytes_b) { + writeln!(out, "Binary files {label_a} and {label_b} differ").map_err(|e| io_msg(&e))?; + return Ok(true); + } + if let Some(line) = prefix { + writeln!(out, "{line}").map_err(|e| io_msg(&e))?; + } + let old = String::from_utf8_lossy(bytes_a); + let new = String::from_utf8_lossy(bytes_b); + let diff = TextDiff::from_lines(old.as_ref(), new.as_ref()); + write!( + out, + "{}", + diff + .unified_diff() + .context_radius(opts.context) + .header(&label_a, &label_b) + ) + .map_err(|e| io_msg(&e))?; + Ok(true) +} + +/// Recursively compares two directories over the sorted union of their +/// entries, GNU `diff -r` style. Returns whether any difference was found. +fn diff_dirs( + name_a: &Path, + res_a: &Path, + name_b: &Path, + res_b: &Path, + opts: Options, +) -> Result { + let mut names: BTreeSet = BTreeSet::new(); + for (dir_name, dir_res) in [(name_a, res_a), (name_b, res_b)] { + let entries = fs::read_dir(dir_res) + .map_err(|err| format!("{}: {}", dir_name.display(), io_msg(&err)))?; + for entry in entries { + let entry = entry.map_err(|err| format!("{}: {}", dir_name.display(), io_msg(&err)))?; + names.insert(entry.file_name()); + } + } + + let mut differed = false; + for name in names { + if pi_uutils_ctx::is_cancelled() { + return Err("interrupted".to_string()); + } + let (child_name_a, child_res_a) = (name_a.join(&name), res_a.join(&name)); + let (child_name_b, child_res_b) = (name_b.join(&name), res_b.join(&name)); + let meta_a = fs::metadata(&child_res_a).ok(); + let meta_b = fs::metadata(&child_res_b).ok(); + match (meta_a.as_ref(), meta_b.as_ref()) { + (Some(ma), Some(mb)) if ma.is_dir() && mb.is_dir() => { + differed |= diff_dirs(&child_name_a, &child_res_a, &child_name_b, &child_res_b, opts)?; + }, + (Some(ma), Some(mb)) if ma.is_dir() != mb.is_dir() => { + let (dir, file) = if ma.is_dir() { + (&child_name_a, &child_name_b) + } else { + (&child_name_b, &child_name_a) + }; + writeln!( + pi_uutils_ctx::stdout(), + "File {} is a directory while file {} is a regular file", + dir.display(), + file.display() + ) + .map_err(|e| io_msg(&e))?; + differed = true; + }, + (Some(_), Some(_)) => { + let bytes_a = fs::read(&child_res_a) + .map_err(|err| format!("{}: {}", child_name_a.display(), io_msg(&err)))?; + let bytes_b = fs::read(&child_res_b) + .map_err(|err| format!("{}: {}", child_name_b.display(), io_msg(&err)))?; + let prefix = format!("diff -r {} {}", child_name_a.display(), child_name_b.display()); + differed |= + diff_pair(&child_name_a, &bytes_a, &child_name_b, &bytes_b, opts, Some(&prefix))?; + }, + (Some(meta), None) | (None, Some(meta)) => { + let in_a = meta_b.is_none(); + if opts.new_file && meta.is_file() { + // -N: compare the present file against an empty absent one. + let (present_name, present_res) = if in_a { + (&child_name_a, &child_res_a) + } else { + (&child_name_b, &child_res_b) + }; + let bytes = fs::read(present_res) + .map_err(|err| format!("{}: {}", present_name.display(), io_msg(&err)))?; + let prefix = + format!("diff -r {} {}", child_name_a.display(), child_name_b.display()); + let (ba, bb): (&[u8], &[u8]) = if in_a { (&bytes, &[]) } else { (&[], &bytes) }; + differed |= diff_pair(&child_name_a, ba, &child_name_b, bb, opts, Some(&prefix))?; + } else { + let present_dir = if in_a { name_a } else { name_b }; + writeln!( + pi_uutils_ctx::stdout(), + "Only in {}: {}", + present_dir.display(), + Path::new(&name).display() + ) + .map_err(|e| io_msg(&e))?; + differed = true; + } + }, + (None, None) => {}, + } + } + Ok(differed) +} + +/// NUL byte within the first 8 KiB marks the input as binary, matching the +/// heuristic GNU diff applies to decide between text and binary output. +fn is_binary(bytes: &[u8]) -> bool { + bytes.iter().take(8192).any(|&b| b == 0) +} + +/// Renders an I/O error without the Rust-specific ` (os error N)` suffix so +/// messages read like GNU diff's (`diff: x: No such file or directory`). +fn io_msg(err: &std::io::Error) -> String { + let msg = err.to_string(); + match msg.find(" (os error") { + Some(idx) => msg[..idx].to_string(), + None => msg, + } +} + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + fn run_with(cwd: PathBuf, stdin: &[u8], args: Vec<&str>) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::Cursor::new(stdin.to_vec())), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd, + env: HashMap::new(), + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; + + let argv: Vec = std::iter::once("diff") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + fn run_in(cwd: PathBuf, args: Vec<&str>) -> (i32, String, String) { + run_with(cwd, b"", args) + } + + /// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var). + fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) { + let dir = tempfile::tempdir().unwrap(); + let canon = fs::canonicalize(dir.path()).unwrap(); + (dir, canon) + } + + #[test] + fn identical_files_print_nothing_and_exit_zero() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("a.txt"), "one\ntwo\n").unwrap(); + fs::write(root.join("b.txt"), "one\ntwo\n").unwrap(); + + let (code, stdout, stderr) = run_in(root, vec!["a.txt", "b.txt"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", "")); + } + + /// Relative operands must resolve against the scope cwd (a tempdir), not + /// the process cwd — the pi-specific contract. + #[test] + fn differing_files_emit_unified_diff_with_typed_headers() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("a.txt"), "one\ntwo\nthree\n").unwrap(); + fs::write(root.join("b.txt"), "one\nTWO\nthree\n").unwrap(); + + let (code, stdout, stderr) = run_in(root, vec!["a.txt", "b.txt"]); + assert_eq!(code, 1); + assert_eq!(stderr, ""); + assert!(stdout.starts_with("--- a.txt\n+++ b.txt\n@@ "), "got: {stdout}"); + assert!(stdout.contains("\n-two\n"), "got: {stdout}"); + assert!(stdout.contains("\n+TWO\n"), "got: {stdout}"); + // Context lines around the change (default -U 3). + assert!(stdout.contains("\n one\n"), "got: {stdout}"); + assert!(stdout.contains("\n three\n"), "got: {stdout}"); + } + + #[test] + fn unified_zero_drops_context_lines() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("a.txt"), "one\ntwo\nthree\n").unwrap(); + fs::write(root.join("b.txt"), "one\nTWO\nthree\n").unwrap(); + + let (code, stdout, _) = run_in(root, vec!["-U", "0", "a.txt", "b.txt"]); + assert_eq!(code, 1); + assert!(!stdout.contains("\n one\n"), "got: {stdout}"); + assert!(!stdout.contains("\n three\n"), "got: {stdout}"); + assert!(stdout.contains("\n-two\n"), "got: {stdout}"); + assert!(stdout.contains("\n+TWO\n"), "got: {stdout}"); + } + + #[test] + fn brief_reports_one_line_per_differing_pair() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("a.txt"), "x\n").unwrap(); + fs::write(root.join("b.txt"), "y\n").unwrap(); + + let (code, stdout, stderr) = run_in(root, vec!["-q", "a.txt", "b.txt"]); + assert_eq!(code, 1); + assert_eq!(stdout, "Files a.txt and b.txt differ\n"); + assert_eq!(stderr, ""); + } + + #[test] + fn compat_flags_are_accepted_and_ignored() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("a.txt"), "x\n").unwrap(); + fs::write(root.join("b.txt"), "y\n").unwrap(); + + let (code, stdout, stderr) = + run_in(root, vec!["-u", "-r", "--color=always", "a.txt", "b.txt"]); + assert_eq!(code, 1); + assert_eq!(stderr, ""); + // Plain unified output, no ANSI escapes. + assert!(stdout.starts_with("--- a.txt\n+++ b.txt\n"), "got: {stdout}"); + assert!(!stdout.contains('\u{1b}'), "got: {stdout}"); + } + + #[test] + fn binary_inputs_report_binary_difference() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("a.bin"), b"aa\x00bb").unwrap(); + fs::write(root.join("b.bin"), b"aa\x00cc").unwrap(); + + let (code, stdout, stderr) = run_in(root, vec!["a.bin", "b.bin"]); + assert_eq!(code, 1); + assert_eq!(stdout, "Binary files a.bin and b.bin differ\n"); + assert_eq!(stderr, ""); + } + + #[test] + fn missing_operand_file_is_trouble() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("a.txt"), "x\n").unwrap(); + + let (code, stdout, stderr) = run_in(root, vec!["a.txt", "nope.txt"]); + assert_eq!(code, 2); + assert_eq!(stdout, ""); + assert_eq!(stderr, "diff: nope.txt: No such file or directory\n"); + } + + #[test] + fn missing_second_operand_is_usage_error() { + let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["only-one"]); + assert_eq!(code, 2); + assert_eq!(stdout, ""); + assert!(stderr.contains("required"), "got: {stderr}"); + } + + #[test] + fn new_file_treats_missing_operand_as_empty() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("a.txt"), "one\n").unwrap(); + + let (code, stdout, stderr) = run_in(root, vec!["-N", "nope.txt", "a.txt"]); + assert_eq!(code, 1); + assert_eq!(stderr, ""); + assert!(stdout.starts_with("--- nope.txt\n+++ a.txt\n"), "got: {stdout}"); + assert!(stdout.contains("\n+one\n"), "got: {stdout}"); + } + + #[test] + fn dash_reads_context_stdin() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("a.txt"), "one\ntwo\n").unwrap(); + + let (code, stdout, stderr) = run_with(root.clone(), b"one\ntwo\n", vec!["a.txt", "-"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", "")); + + let (code, stdout, _) = run_with(root, b"one\nTWO\n", vec!["a.txt", "-"]); + assert_eq!(code, 1); + assert!(stdout.starts_with("--- a.txt\n+++ -\n"), "got: {stdout}"); + } + + #[test] + fn directories_diff_recursively_with_only_in_lines() { + let (_dir, root) = canonical_tempdir(); + let (a, b) = (root.join("a"), root.join("b")); + fs::create_dir_all(a.join("sub")).unwrap(); + fs::create_dir_all(b.join("sub")).unwrap(); + fs::write(a.join("common.txt"), "same\n").unwrap(); + fs::write(b.join("common.txt"), "same\n").unwrap(); + fs::write(a.join("only.txt"), "left\n").unwrap(); + fs::write(b.join("other.txt"), "right\n").unwrap(); + fs::write(a.join("sub/inner.txt"), "old\n").unwrap(); + fs::write(b.join("sub/inner.txt"), "new\n").unwrap(); + + let (code, stdout, stderr) = run_in(root, vec!["a", "b"]); + assert_eq!(code, 1); + assert_eq!(stderr, ""); + assert!(stdout.contains("Only in a: only.txt\n"), "got: {stdout}"); + assert!(stdout.contains("Only in b: other.txt\n"), "got: {stdout}"); + assert!( + stdout.contains( + "diff -r a/sub/inner.txt b/sub/inner.txt\n--- a/sub/inner.txt\n+++ b/sub/inner.txt\n" + ), + "got: {stdout}" + ); + assert!(stdout.contains("\n-old\n"), "got: {stdout}"); + assert!(stdout.contains("\n+new\n"), "got: {stdout}"); + // Identical common.txt must not appear at all. + assert!(!stdout.contains("common.txt"), "got: {stdout}"); + } + + #[test] + fn identical_directories_exit_zero() { + let (_dir, root) = canonical_tempdir(); + let (a, b) = (root.join("a"), root.join("b")); + fs::create_dir_all(&a).unwrap(); + fs::create_dir_all(&b).unwrap(); + fs::write(a.join("f.txt"), "same\n").unwrap(); + fs::write(b.join("f.txt"), "same\n").unwrap(); + + let (code, stdout, stderr) = run_in(root, vec!["-r", "a", "b"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", "")); + } + + #[test] + fn help_renders_to_scope_stdout() { + let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["--help"]); + assert_eq!(code, 0); + assert!(stdout.contains("Usage:")); + assert!(stdout.contains("Compare files line by line")); + assert_eq!(stderr, ""); + } +} diff --git a/crates/vendor/uu-date/Cargo.toml b/crates/vendor/uu-date/Cargo.toml new file mode 100644 index 000000000..0d23dea93 --- /dev/null +++ b/crates/vendor/uu-date/Cargo.toml @@ -0,0 +1,36 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/date), patched to resolve +# path arguments against the shell working directory and route I/O through +# pi-uutils-ctx so it can run in-process as a shell builtin. The set-date +# capability (and its libc/windows-sys settime dependencies) is removed +# entirely, along with the fluent/icu localization stack. See src/date.rs for +# the patch markers (`pi-uutils:` comments). +[package] +name = "uu_date" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "date ~ (uutils) display the current time (vendored + patched for in-process embedding)" + +[lib] +path = "src/date.rs" + +[dependencies] +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +jiff = { version = "0.2", features = [ + "tzdb-bundle-platform", + "tzdb-zoneinfo", + "tzdb-concatenated", +] } +parse_datetime = "0.14" +regex = "1.11" +uucore = { version = "0.8.0", features = ["parser"] } +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +# pi-uutils: rustix is kept only for `clock_getres` (--resolution); the +# `clock_settime` user is gone with the set-date capability. +[target.'cfg(unix)'.dependencies] +rustix = { version = "1.1", features = ["time"] } + +[dev-dependencies] +parking_lot = "0.12" +tempfile = "3" diff --git a/crates/vendor/uu-date/LICENSE b/crates/vendor/uu-date/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-date/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-date/src/date.rs b/crates/vendor/uu-date/src/date.rs new file mode 100644 index 000000000..e4db89445 --- /dev/null +++ b/crates/vendor/uu-date/src/date.rs @@ -0,0 +1,1336 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +// spell-checker:ignore strtime ; (format) DATEFILE MMDDhhmm ; (vars) datetime +// datetimes getres AWST ACST AEST foobarbaz + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. The set-date capability (`--set`, clock_settime / +// SetSystemTime) is removed entirely — a builtin must never mutate the host +// system clock — and `--set` now reports "setting the date is not supported by +// this builtin". The fluent/icu localization stack is dropped (`translate!` +// strings are literalized with the en-US locale text, the i18n-datetime +// feature is not vendored, and the locale.rs default-format probe — which +// calls the process-global setlocale(3) — is replaced by upstream's 24-hour +// fallback format). File operands (`--file`, `--reference`) resolve against +// the shell working directory via `pi_uutils_ctx::resolve` AT THE CALL SITE +// while the original operands are kept for display/error messages, stdio is +// routed through `pi_uutils_ctx`, and the entry point no longer calls +// `std::process::exit`. Time-zone handling stays process-global: jiff reads +// the host TZ environment variable and tzdb (same behavior as upstream). + +mod format_modifiers; + +use std::{ + borrow::Cow, + collections::HashMap, + ffi::OsString, + fs::File, + io::{BufRead, BufReader, BufWriter, Read, Write}, + path::PathBuf, + sync::LazyLock, +}; + +use clap::{Arg, ArgAction, ArgMatches, Command}; +use jiff::{ + Timestamp, Zoned, + fmt::strtime::{self, BrokenDownTime, Config, PosixCustom}, + tz::{Offset, TimeZone, TimeZoneDatabase}, +}; +use pi_uutils_ctx::format_usage; +use uucore::{ + display::Quotable, + error::{FromIo, UResult, USimpleError}, + parser::shortcut_value_parser::ShortcutValueParser, +}; + +// Options +const DATE: &str = "date"; +const HOURS: &str = "hours"; +const MINUTES: &str = "minutes"; +const SECONDS: &str = "seconds"; +const NS: &str = "ns"; + +const OPT_DATE: &str = "date"; +const OPT_FORMAT: &str = "format"; +const OPT_FILE: &str = "file"; +const OPT_DEBUG: &str = "debug"; +const OPT_ISO_8601: &str = "iso-8601"; +const OPT_RESOLUTION: &str = "resolution"; +const OPT_RFC_EMAIL: &str = "rfc-email"; +const OPT_RFC_822: &str = "rfc-822"; +const OPT_RFC_2822: &str = "rfc-2822"; +const OPT_RFC_3339: &str = "rfc-3339"; +const OPT_SET: &str = "set"; +const OPT_REFERENCE: &str = "reference"; +const OPT_UNIVERSAL: &str = "universal"; +const OPT_UNIVERSAL_2: &str = "utc"; + +/// Settings for this program, parsed from the command line +// pi-uutils: the upstream `set_to` field is gone with the set-date capability. +struct Settings { + utc: bool, + format: Format, + date_source: DateSource, + debug: bool, +} + +/// Options for parsing dates +#[derive(Clone, Copy)] +struct DebugOptions { + /// Enable debug output + debug: bool, + /// Warn when midnight is used without explicit time specification + warn_midnight: bool, +} + +impl DebugOptions { + fn new(debug: bool, warn_midnight: bool) -> Self { + Self { debug, warn_midnight } + } +} + +/// Various ways of displaying the date +enum Format { + Iso8601(Iso8601Format), + Rfc5322, + Rfc3339(Rfc3339Format), + Resolution, + Custom(String), + Default, +} + +/// Various places that dates can come from +enum DateSource { + Now, + File(PathBuf), + FileMtime(PathBuf), + Stdin, + Human(String), + Resolution, +} + +enum Iso8601Format { + Date, + Hours, + Minutes, + Seconds, + Ns, +} + +impl From<&str> for Iso8601Format { + fn from(s: &str) -> Self { + match s { + HOURS => Self::Hours, + MINUTES => Self::Minutes, + SECONDS => Self::Seconds, + NS => Self::Ns, + DATE => Self::Date, + // Note: This is caught by clap via `possible_values` + _ => unreachable!(), + } + } +} + +enum Rfc3339Format { + Date, + Seconds, + Ns, +} + +impl From<&str> for Rfc3339Format { + fn from(s: &str) -> Self { + match s { + DATE => Self::Date, + SECONDS => Self::Seconds, + NS => Self::Ns, + // Should be caught by clap + _ => panic!("Invalid format: {s}"), + } + } +} + +/// Indicates whether parsing a military timezone causes the date to remain the +/// same, roll back to the previous day, or advance to the next day. +/// This can occur when applying a military timezone with an optional hour +/// offset crosses midnight in either direction. +#[derive(PartialEq, Debug)] +enum DayDelta { + /// The date does not change + Same, + /// The date rolls back to the previous day. + Previous, + /// The date advances to the next day. + Next, +} + +/// Escape invalid UTF-8 bytes in GNU-compatible octal notation. +/// +/// Converts bytes to a string with printable ASCII characters preserved +/// and non-printable/invalid UTF-8 bytes escaped as `\NNN` octal sequences. +/// +/// This matches GNU date's behavior for invalid input. +/// +/// # Arguments +/// * `bytes` - The byte sequence to escape +/// +/// # Returns +/// A string with invalid bytes escaped in octal notation +/// +/// # Example +/// ```ignore +/// let invalid = b"\xb0"; +/// assert_eq!(escape_invalid_bytes(invalid), "\\260"); +/// ``` +fn escape_invalid_bytes(bytes: &[u8]) -> String { + let escaped = bytes + .iter() + .flat_map(|&b| { + // Preserve printable ASCII except backslash + if (0x20..0x7f).contains(&b) && b != b'\\' { + vec![b] + } else { + // Escape as octal: \NNN + format!("\\{b:03o}").into_bytes() + } + }) + .collect::>(); + String::from_utf8_lossy(&escaped).into_owned() +} + +/// Strip parenthesized comments from a date string. +/// +/// GNU date removes balanced parentheses and their content, treating them as +/// comments. If parentheses are unbalanced, everything from the unmatched '(' +/// onwards is ignored. +/// +/// Examples: +/// - "2026(comment)-01-05" -> "2026-01-05" +/// - "1(ignore comment to eol" -> "1" +/// - "(" -> "" +/// - "((foo)2026-01-05)" -> "" +fn strip_parenthesized_comments(input: &str) -> Cow<'_, str> { + if !input.contains('(') { + return Cow::Borrowed(input); + } + + let mut result = String::with_capacity(input.len()); + let mut depth = 0; + + for c in input.chars() { + match c { + '(' => { + depth += 1; + }, + ')' if depth > 0 => { + depth -= 1; + }, + _ if depth == 0 => { + result.push(c); + }, + _ => {}, + } + } + + Cow::Owned(result) +} + +/// Parse military timezone with optional hour offset. +/// Pattern: single letter (a-z except j) optionally followed by 1-2 digits. +/// Returns Some(total_hours_in_utc) or None if pattern doesn't match. +/// +/// Military timezone mappings: +/// - A-I: UTC+1 to UTC+9 (J is skipped for local time) +/// - K-M: UTC+10 to UTC+12 +/// - N-Y: UTC-1 to UTC-12 +/// - Z: UTC+0 +/// +/// The hour offset from digits is added to the base military timezone offset. +/// Examples: "m" -> 12 (noon UTC), "m9" -> 21 (9pm UTC), "a5" -> 4 (4am UTC +/// next day) +fn parse_military_timezone_with_offset(s: &str) -> Option<(i32, DayDelta)> { + if s.is_empty() || s.len() > 3 { + return None; + } + + let mut chars = s.chars(); + let letter = chars.next()?.to_ascii_lowercase(); + + // Check if first character is a letter (a-z, except j which is handled + // separately) + if !letter.is_ascii_lowercase() || letter == 'j' { + return None; + } + + // Parse optional digits (1-2 digits for hour offset) + let additional_hours: i32 = if let Some(rest) = chars.as_str().chars().next() { + if !rest.is_ascii_digit() { + return None; + } + chars.as_str().parse().ok()? + } else { + 0 + }; + + // Map military timezone letter to UTC offset + let tz_offset = match letter { + 'a'..='i' => (letter as i32 - 'a' as i32) + 1, // A=+1, B=+2, ..., I=+9 + 'k'..='m' => (letter as i32 - 'k' as i32) + 10, // K=+10, L=+11, M=+12 + 'n'..='y' => -((letter as i32 - 'n' as i32) + 1), // N=-1, O=-2, ..., Y=-12 + 'z' => 0, // Z=+0 + _ => return None, + }; + + let day_delta = match additional_hours - tz_offset { + h if h < 0 => DayDelta::Previous, + h if h >= 24 => DayDelta::Next, + _ => DayDelta::Same, + }; + + // Calculate total hours: midnight (0) + tz_offset + additional_hours + // Midnight in timezone X converted to UTC + let hours_from_midnight = (0 - tz_offset + additional_hours).rem_euclid(24); + + Some((hours_from_midnight, day_delta)) +} + +/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the +/// arguments directly (without the uucore clap-localization helper that would +/// terminate the process), renders clap help/usage/version to the context +/// streams, and maps the `UResult` to an exit code, so it is safe to run inside +/// the host shell process. +pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match date_main(&matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "date: {msg}"); + } + if code == 0 { 1 } else { code } + }, + } +} + +#[allow(clippy::cognitive_complexity)] +fn date_main(matches: &ArgMatches) -> UResult<()> { + // pi-uutils: the set-date capability is removed — a shell builtin must + // never mutate the host system clock, so `--set` fails up front instead of + // parsing the operand and calling clock_settime(2)/SetSystemTime. + if matches.get_one::(OPT_SET).is_some() { + return Err(USimpleError::new(1, "setting the date is not supported by this builtin")); + } + + let date_source = if let Some(date_os) = matches.get_one::(OPT_DATE) { + // Convert OsString to String, handling invalid UTF-8 with GNU-compatible error + let date = date_os.to_str().ok_or_else(|| { + let bytes = date_os.as_encoded_bytes(); + let escaped_str = escape_invalid_bytes(bytes); + USimpleError::new(1, format!("invalid date '{escaped_str}'")) + })?; + DateSource::Human(date.into()) + } else if let Some(file) = matches.get_one::(OPT_FILE) { + match file.as_ref() { + "-" => DateSource::Stdin, + _ => DateSource::File(file.into()), + } + } else if let Some(file) = matches.get_one::(OPT_REFERENCE) { + DateSource::FileMtime(file.into()) + } else if matches.get_flag(OPT_RESOLUTION) { + DateSource::Resolution + } else { + DateSource::Now + }; + + // Check for extra operands (multiple positional arguments) + if let Some(formats) = matches.get_many::(OPT_FORMAT) { + let format_args: Vec<&String> = formats.collect(); + if format_args.len() > 1 { + return Err(USimpleError::new(1, format!("extra operand '{}'", format_args[1]))); + } + } + + let format = if let Some(form) = matches.get_one::(OPT_FORMAT) { + if !form.starts_with('+') { + // if an optional Format String was found but the user has not provided an input + // date GNU prints an invalid date Error + if !matches!(date_source, DateSource::Human(_)) { + return Err(USimpleError::new(1, format!("invalid date '{form}'"))); + } + // If the user did provide an input date with the --date flag and the Format + // String is not starting with '+' GNU prints the missing '+' error message + return Err(USimpleError::new( + 1, + format!( + "the argument {form} lacks a leading '+';\nwhen using an option to specify \ + date(s), any non-option\nargument must be a format string beginning with '+'" + ), + )); + } + let form = form[1..].to_string(); + Format::Custom(form) + } else if let Some(fmt) = matches + .get_many::(OPT_ISO_8601) + .map(|mut iter| iter.next().unwrap_or(&DATE.to_string()).as_str().into()) + { + Format::Iso8601(fmt) + } else if matches.get_flag(OPT_RFC_EMAIL) { + Format::Rfc5322 + } else if let Some(fmt) = matches + .get_one::(OPT_RFC_3339) + .map(|s| s.as_str().into()) + { + Format::Rfc3339(fmt) + } else if matches.get_flag(OPT_RESOLUTION) { + Format::Resolution + } else { + Format::Default + }; + + let utc = matches.get_flag(OPT_UNIVERSAL); + let debug_mode = matches.get_flag(OPT_DEBUG); + + // Get the current time, either in the local time zone or UTC. + // pi-uutils: time-zone handling stays process-global — jiff reads the host + // TZ environment variable and system tzdb here, as upstream does. + let now = if utc { + Timestamp::now().to_zoned(TimeZone::UTC) + } else { + Zoned::now() + }; + + let settings = Settings { utc, format, date_source, debug: debug_mode }; + + // Iterate over all dates - whether it's a single date or a file. + let dates: Box> = match &settings.date_source { + DateSource::Human(input) => { + // GNU compatibility (Comments in parentheses) + let input = strip_parenthesized_comments(input); + let input = input.trim(); + + // GNU compatibility (Empty string): + // An empty string (or whitespace-only) should be treated as midnight today. + let is_empty_or_whitespace = input.is_empty(); + + // GNU compatibility (Military timezone 'J'): + // 'J' is reserved for local time in military timezones. + // GNU date accepts it and treats it as midnight today (00:00:00). + let is_military_j = input.eq_ignore_ascii_case("j"); + + // GNU compatibility (Military timezone with optional hour offset): + // Single letter (a-z except j) optionally followed by 1-2 digits. + // Letter represents midnight in that military timezone (UTC offset). + // Digits represent additional hours to add. + // Examples: "m" -> noon UTC (12:00); "m9" -> 21:00 UTC; "a5" -> 04:00 UTC + let military_tz_with_offset = parse_military_timezone_with_offset(input); + + // GNU compatibility (Pure numbers in date strings): + // - Manual: https://www.gnu.org/software/coreutils/manual/html_node/Pure-numbers-in-date-strings.html + // - Semantics: a pure decimal number denotes today's time-of-day (HH or HHMM). + // Examples: "0"/"00" => 00:00 today; "7"/"07" => 07:00 today; "0700" => 07:00 + // today. + // For all other forms, fall back to the general parser. + let is_pure_digits = + !input.is_empty() && input.len() <= 4 && input.chars().all(|c| c.is_ascii_digit()); + + let date = if is_empty_or_whitespace || is_military_j { + // Treat empty string or 'J' as midnight today (00:00:00) in local time + let date_part = + strtime::format("%F", &now).unwrap_or_else(|_| String::from("1970-01-01")); + let offset = if settings.utc { + String::from("+00:00") + } else { + strtime::format("%:z", &now).unwrap_or_default() + }; + let composed = if offset.is_empty() { + format!("{date_part} 00:00") + } else { + format!("{date_part} 00:00 {offset}") + }; + if settings.debug { + let _ = writeln!( + pi_uutils_ctx::stderr(), + "date: warning: using midnight as starting time: 00:00:00" + ); + } + parse_date(composed, &now, DebugOptions::new(settings.debug, false)) + } else if let Some((total_hours, day_delta)) = military_tz_with_offset { + // Military timezone with optional hour offset + // Convert to UTC time: midnight + military_tz_offset + additional_hours + + // When calculating a military timezone with an optional hour offset, midnight + // may be crossed in either direction. `day_delta` indicates whether the + // date remains the same, moves to the previous day, or advances to the next + // day. Changing day can result in error, this closure will help handle + // these errors gracefully. + let format_date_with_epoch_fallback = |date: Result| -> String { + date + .and_then(|d| strtime::format("%F", &d)) + .unwrap_or_else(|_| String::from("1970-01-01")) + }; + let date_part = match day_delta { + DayDelta::Same => format_date_with_epoch_fallback(Ok(now.clone())), + DayDelta::Next => format_date_with_epoch_fallback(now.tomorrow()), + DayDelta::Previous => format_date_with_epoch_fallback(now.yesterday()), + }; + let composed = format!("{date_part} {total_hours:02}:00:00 +00:00"); + parse_date(composed, &now, DebugOptions::new(settings.debug, false)) + } else if is_pure_digits { + // Derive HH and MM from the input + let (hh_opt, mm_opt) = if input.len() <= 2 { + (input.parse::().ok(), Some(0u32)) + } else { + let (h, m) = input.split_at(input.len() - 2); + (h.parse::().ok(), m.parse::().ok()) + }; + + if let (Some(hh), Some(mm)) = (hh_opt, mm_opt) { + // Compose a concrete datetime string for today with zone offset. + // Use the already-determined 'now' and settings.utc to select offset. + let date_part = + strtime::format("%F", &now).unwrap_or_else(|_| String::from("1970-01-01")); + // If -u, force +00:00; otherwise use the local offset of 'now'. + let offset = if settings.utc { + String::from("+00:00") + } else { + strtime::format("%:z", &now).unwrap_or_default() + }; + let composed = if offset.is_empty() { + format!("{date_part} {hh:02}:{mm:02}") + } else { + format!("{date_part} {hh:02}:{mm:02} {offset}") + }; + parse_date(composed, &now, DebugOptions::new(settings.debug, false)) + } else { + // Fallback on parse failure of digits + parse_date(input, &now, DebugOptions::new(settings.debug, true)) + } + } else { + parse_date(input, &now, DebugOptions::new(settings.debug, true)) + }; + + let iter = std::iter::once(date); + Box::new(iter) + }, + // pi-uutils: `-f -` reads the context stdin, not the process stdin. + DateSource::Stdin => parse_dates_from_reader( + pi_uutils_ctx::stdin(), + &now, + DebugOptions::new(settings.debug, true), + ), + DateSource::File(path) => { + // pi-uutils: resolve the DATEFILE operand against the shell working + // directory; `path` is kept for display. + let resolved = pi_uutils_ctx::resolve(path); + if resolved.is_dir() { + return Err(USimpleError::new( + 2, + format!("expected file, got directory {}", path.quote()), + )); + } + let file = + File::open(&resolved).map_err_context(|| path.as_os_str().maybe_quote().to_string())?; + parse_dates_from_reader(file, &now, DebugOptions::new(settings.debug, true)) + }, + DateSource::FileMtime(path) => { + // pi-uutils: resolve the --reference FILE against the shell working + // directory; `path` is kept for display. + let metadata = std::fs::metadata(pi_uutils_ctx::resolve(path)) + .map_err_context(|| path.as_os_str().maybe_quote().to_string())?; + let mtime = metadata.modified()?; + let ts = Timestamp::try_from(mtime) + .map_err(|_| USimpleError::new(1, "cannot set date".to_string()))?; + // pi-uutils: process-global TZ lookup, as upstream. + let date = ts.to_zoned(TimeZone::try_system().unwrap_or(TimeZone::UTC)); + let iter = std::iter::once(Ok(date)); + Box::new(iter) + }, + DateSource::Resolution => { + let resolution = get_clock_resolution(); + // pi-uutils: process-global TZ lookup, as upstream. + let date = resolution.to_zoned(TimeZone::system()); + let iter = std::iter::once(Ok(date)); + Box::new(iter) + }, + DateSource::Now => { + let iter = std::iter::once(Ok(now.clone())); + Box::new(iter) + }, + }; + + let format_string = make_format_string(&settings); + // pi-uutils: buffered context stdout instead of the process stdout. + let mut stdout = BufWriter::new(pi_uutils_ctx::stdout()); + + // Format all the dates + let config = Config::new().custom(PosixCustom::new()).lenient(true); + for date in dates { + // pi-uutils: a DATEFILE/stdin stream can be arbitrarily long; observe + // host cancellation between lines. + if pi_uutils_ctx::is_cancelled() { + break; + } + match date { + Ok(date) => { + let date = if settings.utc { + date.with_time_zone(TimeZone::UTC) + } else { + date + }; + match format_date(&date, format_string, &config) { + Ok(s) => writeln!(stdout, "{s}") + .map_err(|e| USimpleError::new(1, format!("write error: {e}")))?, + Err(e) => { + let _ = stdout.flush(); + return Err(USimpleError::new( + 1, + format!("invalid format '{format_string}' ({e})"), + )); + }, + } + }, + Err((input, _err)) => { + let _ = stdout.flush(); + // pi-uutils: upstream `show!` — report the bad line to the + // context stderr, record the failure exit code, and keep + // processing the remaining lines. + let _ = writeln!(pi_uutils_ctx::stderr(), "date: invalid date '{input}'"); + pi_uutils_ctx::set_exit_code(1); + }, + } + } + + stdout + .flush() + .map_err(|e| USimpleError::new(1, format!("write error: {e}")))?; + Ok(()) +} + +pub fn uu_app() -> Command { + Command::new("date") + .version(uucore::crate_version!()) + .about("Print or set the system date and time") + // pi-uutils: the localized usage blob's FORMAT reference table moved to + // `after_help` below; the usage proper is just the two command lines. + .override_usage(format_usage( + "date [OPTION]... [+FORMAT]...\ndate [OPTION]... [MMDDhhmm[[CC]YY][.ss]]", + )) + .after_help(FORMAT_HELP) + .infer_long_args(true) + .arg( + Arg::new(OPT_DATE) + .short('d') + .long(OPT_DATE) + .value_name("STRING") + .allow_hyphen_values(true) + .overrides_with(OPT_DATE) + .value_parser(clap::value_parser!(OsString)) + .help("display time described by STRING, not 'now'"), + ) + .arg( + Arg::new(OPT_FILE) + .short('f') + .long(OPT_FILE) + .value_name("DATEFILE") + .value_hint(clap::ValueHint::FilePath) + .conflicts_with(OPT_DATE) + .help("like --date; once for each line of DATEFILE"), + ) + .arg( + Arg::new(OPT_ISO_8601) + .short('I') + .long(OPT_ISO_8601) + .value_name("FMT") + .value_parser(ShortcutValueParser::new([DATE, HOURS, MINUTES, SECONDS, NS])) + .num_args(0..=1) + .default_missing_value(OPT_DATE) + .help( + "output date/time in ISO 8601 format.\nFMT='date' for date only (the \ + default),\n'hours', 'minutes', 'seconds', or 'ns'\nfor date and time to the \ + indicated precision.\nExample: 2006-08-14T02:34:56-06:00", + ), + ) + .arg( + Arg::new(OPT_RESOLUTION) + .long(OPT_RESOLUTION) + .conflicts_with_all([OPT_DATE, OPT_FILE]) + .overrides_with(OPT_RESOLUTION) + .help("output the available resolution of timestamps\nExample: 0.000000001") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_RFC_EMAIL) + .short('R') + .long(OPT_RFC_EMAIL) + .alias(OPT_RFC_2822) + .alias(OPT_RFC_822) + .overrides_with(OPT_RFC_EMAIL) + .help( + "output date and time in RFC 5322 format.\nExample: Mon, 14 Aug 2006 02:34:56 -0600", + ) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_RFC_3339) + .long(OPT_RFC_3339) + .value_name("FMT") + .value_parser(ShortcutValueParser::new([DATE, SECONDS, NS])) + .help( + "output date/time in RFC 3339 format.\nFMT='date', 'seconds', or 'ns'\nfor date \ + and time to the indicated precision.\nExample: 2006-08-14 02:34:56-06:00", + ), + ) + .arg( + Arg::new(OPT_DEBUG) + .long(OPT_DEBUG) + .help("annotate the parsed date, and warn about questionable usage to stderr") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_REFERENCE) + .short('r') + .long(OPT_REFERENCE) + .value_name("FILE") + .value_hint(clap::ValueHint::AnyPath) + .conflicts_with_all([OPT_DATE, OPT_FILE, OPT_RESOLUTION]) + .help("display the last modification time of FILE"), + ) + .arg( + Arg::new(OPT_SET) + .short('s') + .long(OPT_SET) + .value_name("STRING") + .allow_hyphen_values(true) + // pi-uutils: the set-date capability is removed; the option is + // still parsed so it fails with a clear message instead of a + // clap "unexpected argument" error. + .help("set time described by STRING (not supported by this builtin)"), + ) + .arg( + Arg::new(OPT_UNIVERSAL) + .short('u') + .long(OPT_UNIVERSAL) + .visible_alias(OPT_UNIVERSAL_2) + .alias("uct") + .overrides_with(OPT_UNIVERSAL) + .help("print or set Coordinated Universal Time (UTC)") + .action(ArgAction::SetTrue), + ) + .arg(Arg::new(OPT_FORMAT).num_args(0..)) +} + +// pi-uutils: literalized en-US FORMAT reference from the `date-usage` locale +// blob, rendered as plain text for clap's after_help. +const FORMAT_HELP: &str = "\ +FORMAT controls the output. Interpreted sequences are: + %% a literal % + %a locale's abbreviated weekday name (e.g., Sun) + %A locale's full weekday name (e.g., Sunday) + %b locale's abbreviated month name (e.g., Jan) + %B locale's full month name (e.g., January) + %c locale's date and time (e.g., Thu Mar 3 23:05:25 2005) + %C century; like %Y, except omit last two digits (e.g., 20) + %d day of month (e.g., 01) + %D date; same as %m/%d/%y + %e day of month, space padded; same as %_d + %F full date; same as %Y-%m-%d + %g last two digits of year of ISO week number (see %G) + %G year of ISO week number (see %V); normally useful only with %V + %h same as %b + %H hour (00..23) + %I hour (01..12) + %j day of year (001..366) + %k hour, space padded ( 0..23); same as %_H + %l hour, space padded ( 1..12); same as %_I + %m month (01..12) + %M minute (00..59) + %n a newline + %N nanoseconds (000000000..999999999) + %p locale's equivalent of either AM or PM; blank if not known + %P like %p, but lower case + %q quarter of year (1..4) + %r locale's 12-hour clock time (e.g., 11:11:04 PM) + %R 24-hour hour and minute; same as %H:%M + %s seconds since 1970-01-01 00:00:00 UTC + %S second (00..60) + %t a tab + %T time; same as %H:%M:%S + %u day of week (1..7); 1 is Monday + %U week number of year, with Sunday as first day of week (00..53) + %V ISO week number, with Monday as first day of week (01..53) + %w day of week (0..6); 0 is Sunday + %W week number of year, with Monday as first day of week (00..53) + %x locale's date representation (e.g., 03/03/2005) + %X locale's time representation (e.g., 23:30:30) + %y last two digits of year (00..99) + %Y year + %z +hhmm numeric time zone (e.g., -0400) + %:z +hh:mm numeric time zone (e.g., -04:00) + %::z +hh:mm:ss numeric time zone (e.g., -04:00:00) + %:::z numeric time zone with : to necessary precision (e.g., -04, +05:30) + %Z alphabetic time zone abbreviation (e.g., EDT) + +By default, date pads numeric fields with zeroes. +The following optional flags may follow '%': + - (hyphen) do not pad the field + _ (underscore) pad with spaces + 0 (zero) pad with zeros + ^ use upper case if possible + # use opposite case if possible +After any flags comes an optional field width, as a decimal number; +then an optional modifier, which is either + E to use the locale's alternate representations if available, or + O to use the locale's alternate numeric symbols if available. + +Examples: + Convert seconds since the epoch (1970-01-01 UTC) to a date + date --date='@2147483647' + Show the time on the west coast of the US (use tzselect(1) to find TZ) + TZ='America/Los_Angeles' date"; + +/// pi-uutils: upstream's `format_date_with_locale_aware_months` minus the +/// optional icu locale-aware month/day name substitution (the i18n-datetime +/// feature is not vendored, so no localization ever applies). +fn format_date( + date: &Zoned, + format_string: &str, + config: &Config, +) -> Result { + // Check if format string has GNU modifiers (width/flags) and format if present + if let Some(result) = + format_modifiers::format_with_modifiers_if_present(date, format_string, config) + { + return result.map_err(|e| e.to_string()); + } + + let broken_down = BrokenDownTime::from(date); + broken_down + .to_string_with_config(config, format_string) + .map_err(|e| e.to_string()) +} + +/// Return the appropriate format string for the given settings. +fn make_format_string(settings: &Settings) -> &str { + match &settings.format { + Format::Iso8601(fmt) => match fmt { + Iso8601Format::Date => "%F", + Iso8601Format::Hours => "%FT%H%:z", + Iso8601Format::Minutes => "%FT%H:%M%:z", + Iso8601Format::Seconds => "%FT%T%:z", + Iso8601Format::Ns => "%FT%T,%N%:z", + }, + Format::Rfc5322 => "%a, %d %h %Y %T %z", + Format::Rfc3339(fmt) => match fmt { + Rfc3339Format::Date => "%F", + Rfc3339Format::Seconds => "%F %T%:z", + Rfc3339Format::Ns => "%F %T.%N%:z", + }, + Format::Resolution => "%s.%N", + Format::Custom(fmt) => fmt, + // pi-uutils: upstream derives the default format from the process + // locale via setlocale(3)/nl_langinfo(3) (src/uu/date/src/locale.rs). + // setlocale mutates process-global state, which a builtin must not do, + // so upstream's 24-hour fallback format is used unconditionally. + Format::Default => "%a %b %e %X %Z %Y", + } +} + +/// Timezone abbreviations with known fixed UTC offsets. +/// Checked first because the abbreviation encodes the exact offset +/// (e.g., EDT always means UTC-4, even in winter when New York observes EST). +/// Offset is in seconds to support half-hour zones like IST (UTC+5:30). +/// All other timezones (JST, CET, etc.) are dynamically resolved from IANA +/// database. +/* spell-checker: disable */ +static FIXED_OFFSET_ABBREVIATIONS: &[(&str, i32)] = &[ + ("UTC", 0), + ("GMT", 0), + ("MEST", 7200), // UTC+2 Middle European Summer Time + // US timezones (GNU compatible) + ("PST", -28800), // UTC-8 + ("PDT", -25200), // UTC-7 + ("MST", -25200), // UTC-7 + ("MDT", -21600), // UTC-6 + ("CST", -21600), // UTC-6 (Ambiguous: US Central, not China/Cuba) + ("CDT", -18000), // UTC-5 + ("EST", -18000), // UTC-5 + ("EDT", -14400), // UTC-4 + // Indian Standard Time (Ambiguous: India vs Israel vs Ireland) + ("IST", 19800), // UTC+5:30 + // Australian timezones + ("AWST", 28800), // UTC+8 + ("ACST", 34200), // UTC+9:30 + ("ACDT", 37800), // UTC+10:30 + ("AEST", 36000), // UTC+10 + ("AEDT", 39600), // UTC+11 + // German timezones + ("MEZ", 3600), // UTC+1 + ("MESZ", 7200), // UTC+2 + // Asian timezones + ("KST", 32400), // UTC+9 Korean Standard Time +]; +/* spell-checker: enable */ + +/// Lazy-loaded timezone abbreviation lookup map built from IANA database. +// pi-uutils: `LazyLock` instead of upstream's `OnceLock` + `get_or_init`. +static TZ_ABBREV_CACHE: LazyLock> = LazyLock::new(build_tz_abbrev_map); + +/// Build timezone abbreviation lookup map from IANA database. +/// This is a fallback for abbreviations not covered by +/// FIXED_OFFSET_ABBREVIATIONS. +fn build_tz_abbrev_map() -> HashMap { + let mut map = HashMap::new(); + + let tzdb = TimeZoneDatabase::from_env(); // spell-checker:disable-line + // spell-checker:disable-next-line + for tz_name in tzdb.available() { + let tz_str = tz_name.as_str(); + // Use last component as potential abbreviation + // e.g., "Pacific/Fiji" could map to "FIJI" + if let Some(last_part) = tz_str.split('/').next_back() { + let potential_abbrev = last_part.to_uppercase(); + // Only add if it looks like an abbreviation (2-5 uppercase chars) + if potential_abbrev.len() >= 2 + && potential_abbrev.len() <= 5 + && potential_abbrev.chars().all(|c| c.is_ascii_uppercase()) + { + map.entry(potential_abbrev) + .or_insert_with(|| tz_str.to_string()); + } + } + } + + map +} + +/// Get IANA timezone name for a given abbreviation. +/// Uses lazy-loaded cache with preferred mappings for disambiguation. +fn tz_abbrev_to_iana(abbrev: &str) -> Option<&str> { + TZ_ABBREV_CACHE.get(abbrev).map(String::as_str) +} + +/// Attempts to parse a date string that contains a timezone abbreviation (e.g. +/// "EST"). +/// +/// If an abbreviation is found and the date is parsable, returns `Some(Zoned)`. +/// Returns `None` if no abbreviation is detected or if parsing fails, +/// indicating that standard parsing should be attempted. +fn try_parse_with_abbreviation>(date_str: S, now: &Zoned) -> Option { + let s = date_str.as_ref(); + + // Look for timezone abbreviation at the end of the string + // Pattern: ends with uppercase letters (2-5 chars) + if let Some(last_word) = s.split_whitespace().last() { + // Check if it's a potential timezone abbreviation (all uppercase, 2-5 chars) + if last_word.len() >= 2 + && last_word.len() <= 5 + && last_word.chars().all(|c| c.is_ascii_uppercase()) + { + let tz = if let Some(&(_, offset_secs)) = FIXED_OFFSET_ABBREVIATIONS + .iter() + .find(|(abbr, _)| *abbr == last_word) + { + Offset::from_seconds(offset_secs).ok().map(TimeZone::fixed) + } else { + tz_abbrev_to_iana(last_word).and_then(|name| TimeZone::get(name).ok()) + }; + + if let Some(tz) = tz { + let date_part = s.trim_end_matches(last_word).trim(); + // Parse in the target timezone so "10:30 EDT" means 10:30 in EDT + if let Ok(parsed) = parse_datetime::parse_datetime_at_date(now.clone(), date_part) { + let dt = parsed.datetime(); + if let Ok(zoned) = dt.to_zoned(tz) { + return Some(zoned); + } + } + } + } + } + + // No abbreviation found or couldn't resolve, return original + None +} + +/// Helper function to parse dates from a line-based reader (stdin or file) +/// +/// Takes any `Read` source, reads it line by line, and parses each line as a +/// date. Returns a boxed iterator over the parse results. +fn parse_dates_from_reader( + reader: R, + now: &Zoned, + dbg_opts: DebugOptions, +) -> Box> + '_> { + let lines = BufReader::new(reader).lines(); + Box::new( + lines + .map_while(Result::ok) + .map(move |s| parse_date(s, now, dbg_opts)), + ) +} + +/// Parse a `String` into a `DateTime`. +/// If it fails, return a tuple of the `String` along with its `ParseError`. +fn parse_date + Clone>( + s: S, + now: &Zoned, + dbg_opts: DebugOptions, +) -> Result { + let input_str = s.as_ref(); + + if dbg_opts.debug { + let _ = writeln!(pi_uutils_ctx::stderr(), "date: input string: {input_str}"); + } + + // First, try to parse any timezone abbreviations + if let Some(zoned) = try_parse_with_abbreviation(input_str, now) { + if dbg_opts.debug { + // pi-uutils: context stderr instead of `stderr().lock()`. + let mut err = pi_uutils_ctx::stderr(); + let _ = writeln!( + err, + "date: parsed date part: (Y-M-D) {}", + strtime::format("%Y-%m-%d", &zoned).unwrap_or_default() + ); + let _ = writeln!( + err, + "date: parsed time part: {}", + strtime::format("%H:%M:%S", &zoned).unwrap_or_default() + ); + let tz_display = zoned.time_zone().iana_name().unwrap_or("system default"); + let _ = writeln!(err, "date: input timezone: {tz_display}"); + } + return Ok(zoned); + } + + match parse_datetime::parse_datetime_at_date(now.clone(), input_str) { + // Convert to system timezone for display + // (parse_datetime returns Zoned in the input's timezone) + Ok(date) => { + let result = date.timestamp().to_zoned(now.time_zone().clone()); + if dbg_opts.debug { + // Show final parsed date and time + // pi-uutils: context stderr instead of `stderr().lock()`. + let mut err = pi_uutils_ctx::stderr(); + let _ = writeln!( + err, + "date: parsed date part: (Y-M-D) {}", + strtime::format("%Y-%m-%d", &result).unwrap_or_default() + ); + let _ = writeln!( + err, + "date: parsed time part: {}", + strtime::format("%H:%M:%S", &result).unwrap_or_default() + ); + + // Show timezone information + let _ = writeln!(err, "date: input timezone: system default"); + + // Check if time component was specified, if not warn about midnight usage + // Only warn for date-only inputs (no time specified), but not for epoch formats + // (@N) or inputs that explicitly specify a time (containing ':') + if dbg_opts.warn_midnight && !input_str.contains(':') && !input_str.contains('@') { + // Input likely didn't specify a time, so midnight was assumed + let time_str = strtime::format("%H:%M:%S", &result).unwrap_or_default(); + if time_str == "00:00:00" { + let _ = writeln!(err, "date: warning: using midnight as starting time: 00:00:00"); + } + } + } + Ok(result) + }, + Err(e) => Err((input_str.into(), e)), + } +} + +#[cfg(not(any(unix, windows)))] +fn get_clock_resolution() -> Timestamp { + unimplemented!("getting clock resolution not implemented (unsupported target)"); +} + +#[cfg(all(unix, not(target_os = "redox")))] +/// Returns the resolution of the system’s realtime clock. +/// +/// # Panics +/// +/// Panics if `clock_getres` fails. On a POSIX-compliant system this should not +/// occur, as `CLOCK_REALTIME` is required to be supported. +/// Failure would indicate a non-conforming or otherwise broken implementation. +fn get_clock_resolution() -> Timestamp { + use rustix::time::{ClockId, clock_getres}; + + let timespec = clock_getres(ClockId::Realtime); + + #[allow(clippy::unnecessary_cast, reason = "needed for 32 bit target")] + Timestamp::constant(timespec.tv_sec as _, timespec.tv_nsec as _) +} + +#[cfg(all(unix, target_os = "redox"))] +fn get_clock_resolution() -> Timestamp { + // Redox OS does not support the posix clock_getres function, however + // internally it uses a resolution of 1ns to represent timestamps. + // https://gitlab.redox-os.org/redox-os/kernel/-/blob/master/src/time.rs + Timestamp::constant(0, 1) +} + +#[cfg(windows)] +fn get_clock_resolution() -> Timestamp { + // Windows does not expose a system call for getting the resolution of the + // clock, however the FILETIME struct returned by GetSystemTimeAsFileTime, + // and GetSystemTimePreciseAsFileTime has a resolution of 100ns. + // https://learn.microsoft.com/en-us/windows/win32/api/minwinbase/ns-minwinbase-filetime + Timestamp::constant(0, 100) +} + +// pi-uutils: upstream's `convert_for_set` and the `set_system_datetime` +// variants (clock_settime / SetSystemTime) are removed with the set-date +// capability. + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + fn run_in(cwd: PathBuf, args: Vec<&str>) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd, + env: HashMap::new(), + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; + + let argv: Vec = std::iter::once("date") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + /// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var). + fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) { + let dir = tempfile::tempdir().unwrap(); + let canon = std::fs::canonicalize(dir.path()).unwrap(); + (dir, canon) + } + + // --- upstream unit tests (0.8.0) --- + + #[test] + fn test_parse_military_timezone_with_offset() { + // Valid cases: letter only, letter + digit, uppercase + assert_eq!(parse_military_timezone_with_offset("m"), Some((12, DayDelta::Previous))); // UTC+12 -> 12:00 UTC + assert_eq!(parse_military_timezone_with_offset("m9"), Some((21, DayDelta::Previous))); // 12 + 9 = 21 + assert_eq!(parse_military_timezone_with_offset("a5"), Some((4, DayDelta::Same))); // 23 + 5 = 28 % 24 = 4 + assert_eq!(parse_military_timezone_with_offset("z"), Some((0, DayDelta::Same))); // UTC+0 -> 00:00 UTC + assert_eq!(parse_military_timezone_with_offset("M9"), Some((21, DayDelta::Previous))); // Uppercase works + + // Invalid cases: 'j' reserved, empty, too long, starts with digit + assert_eq!(parse_military_timezone_with_offset("j"), None); // Reserved for local time + assert_eq!(parse_military_timezone_with_offset(""), None); // Empty + assert_eq!(parse_military_timezone_with_offset("m999"), None); // Too long + assert_eq!(parse_military_timezone_with_offset("9m"), None); // Starts with digit + } + + #[test] + fn test_abbreviation_resolves_relative_date_against_now() { + let now = "2025-03-15T20:00:00+00:00[UTC]".parse::().unwrap(); + let result = + parse_date("yesterday 10:00 GMT", &now, DebugOptions::new(false, false)).unwrap(); + assert_eq!(result.date(), jiff::civil::date(2025, 3, 14)); + } + + #[test] + fn test_strip_parenthesized_comments() { + assert_eq!(strip_parenthesized_comments("hello"), "hello"); + assert_eq!(strip_parenthesized_comments("2026-01-05"), "2026-01-05"); + assert_eq!(strip_parenthesized_comments("("), ""); + assert_eq!(strip_parenthesized_comments("1(comment"), "1"); + assert_eq!(strip_parenthesized_comments("2026-01-05(this is a comment"), "2026-01-05"); + assert_eq!(strip_parenthesized_comments("2026(comment)-01-05"), "2026-01-05"); + assert_eq!(strip_parenthesized_comments("()"), ""); + assert_eq!(strip_parenthesized_comments("((foo)2026-01-05)"), ""); + + // These cases test the balanced parentheses removal feature + // which extends beyond what GNU date strictly supports + assert_eq!(strip_parenthesized_comments("a(b)c"), "ac"); + assert_eq!(strip_parenthesized_comments("a(b)c(d)e"), "ace"); + assert_eq!(strip_parenthesized_comments("(a)(b)"), ""); + + // When parentheses are unmatched, processing stops at the unmatched opening + // paren + assert_eq!(strip_parenthesized_comments("a(b)c(d"), "ac"); + + // Additional edge cases for nested and complex parentheses + assert_eq!(strip_parenthesized_comments("a(b(c)d)e"), "ae"); // Nested balanced + assert_eq!(strip_parenthesized_comments("a(b(c)d"), "a"); // Nested unbalanced + assert_eq!(strip_parenthesized_comments("a(b)c(d)e(f"), "ace"); // Multiple groups, last unmatched + } + + // --- pi-uutils behavior contracts --- + + #[test] + fn utc_date_string_formats_exactly() { + let (code, stdout, stderr) = + run_in(PathBuf::from("."), vec!["-u", "-d", "2026-01-02 03:04:05", "+%Y-%m-%dT%H:%M:%S"]); + assert_eq!(code, 0); + assert_eq!(stdout, "2026-01-02T03:04:05\n"); + assert_eq!(stderr, ""); + } + + #[test] + fn epoch_input_round_trips_through_seconds_format() { + let (code, stdout, stderr) = + run_in(PathBuf::from("."), vec!["-u", "-d", "@1767323045", "+%s"]); + assert_eq!(code, 0); + assert_eq!(stdout, "1767323045\n"); + assert_eq!(stderr, ""); + } + + #[test] + fn set_is_unsupported() { + let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["--set", "2026-01-02 03:04:05"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert_eq!(stderr, "date: setting the date is not supported by this builtin\n"); + + let (code, _, stderr) = run_in(PathBuf::from("."), vec!["-s", "now"]); + assert_eq!(code, 1); + assert!(stderr.contains("not supported by this builtin")); + } + + #[test] + fn datefile_relative_path_resolves_against_scope_cwd() { + let (_dir, root) = canonical_tempdir(); + std::fs::write(root.join("dates.txt"), "2026-01-02 03:04:05\n@0\n").unwrap(); + + // Relative operand + scope cwd differing from the process cwd: only the + // call-site `pi_uutils_ctx::resolve` patch makes this find the file. + let (code, stdout, stderr) = run_in(root, vec!["-u", "-f", "dates.txt", "+%F"]); + assert_eq!(code, 0); + assert_eq!(stdout, "2026-01-02\n1970-01-01\n"); + assert_eq!(stderr, ""); + } + + #[test] + fn datefile_bad_line_reports_but_keeps_processing() { + let (_dir, root) = canonical_tempdir(); + std::fs::write(root.join("dates.txt"), "foobarbaz\n@86400\n").unwrap(); + + let (code, stdout, stderr) = run_in(root, vec!["-u", "-f", "dates.txt", "+%F"]); + assert_eq!(code, 1, "a bad line must fail the invocation"); + assert_eq!(stdout, "1970-01-02\n", "good lines after a bad one still print"); + assert!(stderr.contains("invalid date 'foobarbaz'"), "stderr: {stderr}"); + } + + #[test] + fn invalid_date_string_reports_and_fails() { + let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["-d", "foobarbaz"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert_eq!(stderr, "date: invalid date 'foobarbaz'\n"); + } + + #[test] + fn rfc_email_iso_and_rfc3339_formats() { + let fixed = "2026-01-02 03:04:05"; + + let (code, stdout, _) = run_in(PathBuf::from("."), vec!["-u", "-d", fixed, "-R"]); + assert_eq!((code, stdout.as_str()), (0, "Fri, 02 Jan 2026 03:04:05 +0000\n")); + + let (code, stdout, _) = run_in(PathBuf::from("."), vec!["-u", "-d", fixed, "-Iseconds"]); + assert_eq!((code, stdout.as_str()), (0, "2026-01-02T03:04:05+00:00\n")); + + let (code, stdout, _) = + run_in(PathBuf::from("."), vec!["-u", "-d", fixed, "--rfc-3339=seconds"]); + assert_eq!((code, stdout.as_str()), (0, "2026-01-02 03:04:05+00:00\n")); + } + + #[test] + fn reference_relative_path_resolves_against_scope_cwd() { + let (_dir, root) = canonical_tempdir(); + std::fs::write(root.join("ref-file"), b"x").unwrap(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["-u", "-r", "ref-file", "+%s"]); + assert_eq!(code, 0); + assert_eq!(stderr, ""); + let secs: i64 = stdout.trim().parse().unwrap(); + let now = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_secs() as i64; + assert!((now - secs).abs() < 60, "mtime epoch {secs} should be close to now {now}"); + + // Missing file fails, naming the operand as typed. + let (code, _, stderr) = run_in(root, vec!["-r", "missing-file"]); + assert_eq!(code, 1); + assert!(stderr.contains("missing-file"), "stderr: {stderr}"); + } + + #[test] + fn format_operand_without_plus_is_rejected() { + // With -d: GNU's "lacks a leading '+'" message. + let (code, _, stderr) = run_in(PathBuf::from("."), vec!["-d", "2026-01-02", "%F"]); + assert_eq!(code, 1); + assert!(stderr.contains("lacks a leading '+'"), "stderr: {stderr}"); + + // Without a date source: GNU treats the operand as an invalid date. + let (code, _, stderr) = run_in(PathBuf::from("."), vec!["%F"]); + assert_eq!(code, 1); + assert!(stderr.contains("invalid date '%F'"), "stderr: {stderr}"); + } + + #[test] + fn help_renders_to_scope_stdout() { + let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["--help"]); + assert_eq!(code, 0); + assert!(stdout.contains("Usage:")); + assert!(stdout.contains("FORMAT controls the output")); + assert_eq!(stderr, ""); + } +} diff --git a/crates/vendor/uu-date/src/format_modifiers.rs b/crates/vendor/uu-date/src/format_modifiers.rs new file mode 100644 index 000000000..20f3f987c --- /dev/null +++ b/crates/vendor/uu-date/src/format_modifiers.rs @@ -0,0 +1,760 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. +// spell-checker:ignore strtime + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin: the `translate!` error string is literalized with the +// en-US locale text and the regex cache uses `LazyLock` instead of `OnceLock`. +// No behavior changes. + +//! GNU date format modifier support +//! +//! This module implements GNU-compatible format modifiers for date formatting. +//! These modifiers extend standard strftime format specifiers with optional +//! width and flag modifiers. +//! +//! ## Syntax +//! +//! Format: `%[flags][width]specifier` +//! +//! ### Flags +//! - `-`: Do not pad the field +//! - `_`: Pad with spaces instead of zeros +//! - `0`: Pad with zeros (default for numeric fields) +//! - `^`: Convert to uppercase +//! - `#`: Use opposite case (uppercase becomes lowercase and vice versa) +//! - `+`: Force display of sign (+ for positive, - for negative) +//! +//! ### Width +//! - One or more digits specifying minimum field width +//! - Field will be padded to this width using the padding character +//! +//! ### Examples +//! - `%10Y`: Year padded to 10 digits with zeros (0000001999) +//! - `%_10m`: Month padded to 10 digits with spaces ( 06) +//! - `%-d`: Day without padding (1 instead of 01) +//! - `%^B`: Month name in uppercase (JUNE) +//! - `%+4C`: Century with sign, padded to 4 characters (+019) + +use std::{fmt, sync::LazyLock}; + +use jiff::{ + Zoned, + fmt::strtime::{BrokenDownTime, Config, PosixCustom}, +}; +use regex::Regex; + +/// Error type for format modifier operations +#[derive(Debug)] +pub enum FormatError { + /// Error from the underlying jiff library + JiffError(jiff::Error), + /// Field width calculation overflowed or required allocation failed + FieldWidthTooLarge { width: usize, specifier: String }, +} + +impl fmt::Display for FormatError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::JiffError(e) => write!(f, "{e}"), + // pi-uutils: literalized en-US translation of + // `date-error-format-modifier-width-too-large`. + Self::FieldWidthTooLarge { width, specifier } => { + write!(f, "format modifier width '{width}' is too large for specifier '%{specifier}'") + }, + } + } +} + +impl From for FormatError { + fn from(e: jiff::Error) -> Self { + Self::JiffError(e) + } +} + +/// Regex to match format specifiers with optional modifiers +/// Pattern: % \[flags\] \[width\] specifier +/// Flags: -, _, 0, ^, #, + +/// Width: one or more digits +/// Specifier: any letter or special sequence like :z, ::z, :::z +// pi-uutils: `LazyLock` instead of upstream's function-local `OnceLock`. +static FORMAT_SPEC_REGEX: LazyLock = + LazyLock::new(|| Regex::new(r"%([_0^#+-]*)(\d*)(:*[a-zA-Z])").unwrap()); + +/// Check if format string contains any GNU modifiers and format if present. +/// +/// This function combines modifier detection and formatting in a single pass +/// for better performance. If no modifiers are found, returns None and the +/// caller should use standard formatting. If modifiers are found, returns +/// the formatted string. +pub fn format_with_modifiers_if_present( + date: &Zoned, + format_string: &str, + config: &Config, +) -> Option> { + let re = &*FORMAT_SPEC_REGEX; + + // Quick check: does the string contain any modifiers? + let has_modifiers = re.captures_iter(format_string).any(|cap| { + let flags = cap.get(1).map_or("", |m| m.as_str()); + let width_str = cap.get(2).map_or("", |m| m.as_str()); + !flags.is_empty() || !width_str.is_empty() + }); + + if !has_modifiers { + return None; + } + + // If we have modifiers, format the string + Some(format_with_modifiers(date, format_string, config)) +} + +/// Process a format string with GNU modifiers. +/// +/// # Arguments +/// * `date` - The date to format +/// * `format_string` - Format string with GNU modifiers +/// * `config` - Strftime configuration +/// +/// # Returns +/// Formatted string with modifiers applied +/// +/// # Errors +/// Returns `FormatError` if formatting fails +fn format_with_modifiers( + date: &Zoned, + format_string: &str, + config: &Config, +) -> Result { + // First, replace %% with a placeholder to avoid matching it + let placeholder = "\x00PERCENT\x00"; + let temp_format = format_string.replace("%%", placeholder); + + let re = &*FORMAT_SPEC_REGEX; + let mut result = String::new(); + let mut last_end = 0; + + let broken_down = BrokenDownTime::from(date); + + for cap in re.captures_iter(&temp_format) { + let whole_match = cap.get(0).unwrap(); + let flags = cap.get(1).map_or("", |m| m.as_str()); + let width_str = cap.get(2).map_or("", |m| m.as_str()); + let spec = cap.get(3).unwrap().as_str(); + + // Add text before this match + result.push_str(&temp_format[last_end..whole_match.start()]); + + // Format the base specifier first + let base_format = format!("%{spec}"); + let formatted = broken_down.to_string_with_config(config, &base_format)?; + + // Check if this specifier has modifiers + if !flags.is_empty() || !width_str.is_empty() { + // Apply modifiers to the formatted value + let width: usize = width_str.parse().unwrap_or(0); + let explicit_width = !width_str.is_empty(); + let modified = apply_modifiers(&formatted, flags, width, spec, explicit_width)?; + result.push_str(&modified); + } else { + // No modifiers, use formatted value as-is + result.push_str(&formatted); + } + + last_end = whole_match.end(); + } + + // Add remaining text + result.push_str(&temp_format[last_end..]); + + // Restore %% by converting placeholder to % + let result = result.replace(placeholder, "%"); + + Ok(result) +} + +/// Returns true if the specifier produces text output (default pad is space) +/// rather than numeric output (default pad is zero). +fn is_text_specifier(specifier: &str) -> bool { + matches!(specifier.chars().last(), Some('A' | 'a' | 'B' | 'b' | 'h' | 'Z' | 'p' | 'P')) +} + +/// Returns true if the specifier defaults to space padding. +/// This includes text specifiers and numeric specifiers like %e and %k +/// that use blank-padding by default in GNU date. +fn is_space_padded_specifier(specifier: &str) -> bool { + matches!( + specifier.chars().last(), + Some('A' | 'a' | 'B' | 'b' | 'h' | 'Z' | 'p' | 'P' | 'e' | 'k' | 'l') + ) +} + +/// Returns the default width for a specifier. +/// This is used when a flag like `_` is used without an explicit width. +fn get_default_width(specifier: &str) -> usize { + match specifier.chars().last() { + // Day of month: 2 digits (01-31) + Some('d') | Some('e') => 2, + // Month: 2 digits (01-12) + Some('m') => 2, + // Hour: 2 digits (00-23) + Some('H') | Some('k') => 2, + // Hour (12-hour): 2 digits (01-12) + Some('I') | Some('l') => 2, + // Minute: 2 digits (00-59) + Some('M') => 2, + // Second: 2 digits (00-60) + Some('S') => 2, + // Year (2-digit): 2 digits + Some('y') => 2, + // Day of year: 3 digits (001-366) + Some('j') => 3, + // Week number: 2 digits (00-53) + Some('U') | Some('W') | Some('V') => 2, + // Day of week: 1 digit (0-6 or 1-7) + Some('w') | Some('u') => 1, + // Century: 2 digits (00-99) + Some('C') => 2, + // Full year: 4 digits + Some('Y') | Some('G') => 4, + // ISO week year (2-digit): 2 digits + Some('g') => 2, + // Epoch seconds: typically 10 digits (but variable) + Some('s') => 0, + // Nanoseconds: 9 digits + Some('N') => 9, + // Quarter: 1 digit + Some('q') => 1, + // Timezone offset: varies + Some('z') => 0, + // Text specifiers have no default width + _ => 0, + } +} + +/// Strip default padding (leading zeros or leading spaces) from a value, +/// preserving at least one character. +fn strip_default_padding(value: &str) -> String { + if value.starts_with('0') && value.len() >= 2 { + let stripped = value.trim_start_matches('0'); + if stripped.is_empty() { + return "0".to_string(); + } + if let Some(first_char) = stripped.chars().next() + && first_char.is_ascii_digit() + { + return stripped.to_string(); + } + } + if value.starts_with(' ') { + let stripped = value.trim_start(); + if !stripped.is_empty() { + return stripped.to_string(); + } + } + value.to_string() +} + +/// Apply width and flag modifiers to a formatted value. +/// +/// The `specifier` parameter is the format specifier (e.g., "d", "B", "Y") +/// which determines the default padding character (space for text, zero for +/// numeric). Flags are processed in order so that when conflicting flags +/// appear, the last one takes precedence (e.g., `_+` means `+` wins for +/// padding). +/// +/// The `explicit_width` parameter indicates whether a width was explicitly +/// specified in the format string (true) or if width is 0 (false). +fn apply_modifiers( + value: &str, + flags: &str, + width: usize, + specifier: &str, + explicit_width: bool, +) -> Result { + let mut result = value.to_string(); + + // Determine default pad character based on specifier type + // Determine default pad character based on specifier type. + // Text specifiers (month names, etc.) and numeric specifiers like %e, %k, %l + // default to space padding; other numeric specifiers default to zero padding. + let default_pad = if is_space_padded_specifier(specifier) { + ' ' + } else { + '0' + }; + + // Process flags in order - last conflicting flag wins + let mut pad_char = default_pad; + let mut no_pad = false; + let mut uppercase = false; + let mut swap_case = false; + let mut force_sign = false; + let mut underscore_flag = false; + + for flag in flags.chars() { + match flag { + '-' => { + no_pad = true; + }, + '_' => { + no_pad = false; + pad_char = ' '; + underscore_flag = true; + }, + '0' => { + no_pad = false; + pad_char = '0'; + }, + '^' => { + uppercase = true; + swap_case = false; // ^ overrides # + }, + '#' if !uppercase => { + // Only apply # if ^ hasn't been set + swap_case = true; + }, + '+' => { + force_sign = true; + no_pad = false; + pad_char = '0'; + }, + _ => {}, + } + } + + // Apply case modifications (uppercase takes precedence over swap_case) + if uppercase { + result = result.to_uppercase(); + } else if swap_case { + if result + .chars() + .all(|c| !c.is_alphabetic() || c.is_uppercase()) + { + result = result.to_lowercase(); + } else if !result + .chars() + .all(|c| !c.is_alphabetic() || c.is_lowercase()) + { + result = result.to_uppercase(); + } + } + + // If no_pad flag is active, suppress all padding and return + if no_pad { + return Ok(strip_default_padding(&result)); + } + + // Handle padding flag without explicit width: use default width + // This applies when _ or 0 flag overrides the default padding character + // and no explicit width is specified (e.g., %_m, %0e) + let effective_width = if !explicit_width && (underscore_flag || pad_char != default_pad) { + get_default_width(specifier) + } else { + width + }; + + // When the requested width is narrower than the default formatted width, GNU + // first removes default padding and then reapplies the requested width. + if effective_width > 0 && effective_width < result.len() { + result = strip_default_padding(&result); + } + + // Strip default padding when switching pad characters on numeric fields + if !is_text_specifier(specifier) && result.len() >= 2 { + if pad_char == ' ' && result.starts_with('0') { + // Switching to space padding: strip leading zeros + result = strip_default_padding(&result); + } else if pad_char == '0' && result.starts_with(' ') { + // Switching to zero padding: strip leading spaces + result = strip_default_padding(&result); + } + } + + // Apply force sign for numeric values + // GNU behavior: + only adds sign if: + // 1. An explicit width is provided, OR + // 2. The value exceeds the default width for that specifier (e.g., year > 4 + // digits) + if force_sign + && !result.starts_with('+') + && !result.starts_with('-') + && result.chars().next().is_some_and(|c| c.is_ascii_digit()) + { + let default_w = get_default_width(specifier); + // Add sign only if explicit width provided OR result exceeds default width + if explicit_width || (default_w > 0 && result.len() > default_w) { + result.insert(0, '+'); + } + } + + // Apply width padding + if effective_width > result.len() { + let padding = effective_width - result.len(); + let has_sign = result.starts_with('+') || result.starts_with('-'); + + if pad_char == '0' && has_sign { + // Zero padding: sign first, then zeros (e.g., "-0022") + let sign = result.chars().next().unwrap(); + let rest = &result[1..]; + let mut padded = try_alloc_padded(result.len(), padding, effective_width, specifier)?; + padded.push(sign); + padded.extend(std::iter::repeat_n('0', padding)); + padded.push_str(rest); + result = padded; + } else { + // Default: pad on the left (e.g., " -22" or " 1999") + let mut padded = try_alloc_padded(result.len(), padding, effective_width, specifier)?; + padded.extend(std::iter::repeat_n(pad_char, padding)); + padded.push_str(&result); + result = padded; + } + } + + Ok(result) +} + +/// Allocate a `String` with enough capacity for `current_len + padding`, +/// returning `FieldWidthTooLarge` on arithmetic overflow or allocation failure. +fn try_alloc_padded( + current_len: usize, + padding: usize, + width: usize, + specifier: &str, +) -> Result { + let target_len = current_len + .checked_add(padding) + .ok_or_else(|| FormatError::FieldWidthTooLarge { width, specifier: specifier.to_string() })?; + let mut s = String::new(); + s.try_reserve(target_len) + .map_err(|_| FormatError::FieldWidthTooLarge { width, specifier: specifier.to_string() })?; + Ok(s) +} + +#[cfg(test)] +mod tests { + use jiff::{civil, tz::TimeZone}; + + use super::*; + + fn make_test_date(year: i16, month: i8, day: i8, hour: i8) -> Zoned { + civil::date(year, month, day) + .at(hour, 0, 0, 0) + .to_zoned(TimeZone::UTC) + .unwrap() + } + + fn get_config() -> Config { + Config::new().custom(PosixCustom::new()).lenient(true) + } + + #[test] + fn test_width_and_padding_modifiers() { + let date = make_test_date(1999, 6, 1, 0); + let config = get_config(); + + // Test basic width with zero padding + let result = format_with_modifiers(&date, "%10Y", &config).unwrap(); + assert_eq!(result, "0000001999"); + + // Test large width + let result = format_with_modifiers(&date, "%20Y", &config).unwrap(); + assert_eq!(result, "00000000000000001999"); + assert_eq!(result.len(), 20); + + // Test underscore (space) padding with month + let result = format_with_modifiers(&date, "%_10m", &config).unwrap(); + assert_eq!(result, " 6"); + assert_eq!(result.len(), 10); + + // Test underscore padding with day + let date_day5 = make_test_date(1999, 6, 5, 0); + let result = format_with_modifiers(&date_day5, "%_10d", &config).unwrap(); + assert_eq!(result, " 5"); + } + + #[test] + fn test_no_pad_and_case_flags() { + let date = make_test_date(1999, 6, 1, 0); + let config = get_config(); + + // Test no-pad: %-10Y suppresses all padding (width ignored) + let result = format_with_modifiers(&date, "%-10Y", &config).unwrap(); + assert_eq!(result, "1999"); + + // Test no-pad: %-d strips default zero padding + let result = format_with_modifiers(&date, "%-d", &config).unwrap(); + assert_eq!(result, "1"); + + // Test uppercase: %^B should uppercase month name + let result = format_with_modifiers(&date, "%^B", &config).unwrap(); + assert_eq!(result, "JUNE"); + + // Test uppercase with width: %^10B should uppercase and space-pad (text + // specifier) + let result = format_with_modifiers(&date, "%^10B", &config).unwrap(); + assert_eq!(result, " JUNE"); + assert_eq!(result.len(), 10); + } + + #[test] + fn test_sign_flags() { + let date = make_test_date(1970, 1, 1, 0); + let config = get_config(); + + // Test force sign with century: %+4C + let result = format_with_modifiers(&date, "%+4C", &config).unwrap(); + assert!(result.starts_with('+')); + assert_eq!(result.len(), 4); + + // Test force sign with zero padding: %+6Y + let result = format_with_modifiers(&date, "%+6Y", &config).unwrap(); + assert_eq!(result, "+01970"); + } + + #[test] + fn test_combined_flags_underscore_and_sign() { + let date = make_test_date(1970, 1, 1, 0); + let config = get_config(); + // %_+6Y: _ sets space pad, then + overrides to zero pad with sign (last wins) + let result = format_with_modifiers(&date, "%_+6Y", &config).unwrap(); + assert_eq!(result, "+01970"); + } + + #[test] + fn test_combined_flags_no_pad_and_uppercase() { + let date = make_test_date(1999, 6, 1, 0); + let config = get_config(); + // %-^10B: uppercase + no-pad (- suppresses all padding, width ignored) + let result = format_with_modifiers(&date, "%-^10B", &config).unwrap(); + assert_eq!(result, "JUNE"); + } + + #[test] + fn test_swap_case_flag() { + let date = make_test_date(1999, 6, 1, 0); + let config = get_config(); + // %#B: swap case on "June" (mixed case) → uppercase + let result = format_with_modifiers(&date, "%#B", &config).unwrap(); + assert_eq!(result, "JUNE"); + } + + #[test] + fn test_width_smaller_than_result() { + let date = make_test_date(1999, 6, 1, 0); + let config = get_config(); + // %1d: width 1 < "01".len() → strip zero padding → "1" + let result = format_with_modifiers(&date, "%1d", &config).unwrap(); + assert_eq!(result, "1"); + } + + #[test] + fn test_edge_cases_and_special_formats() { + let date = make_test_date(1999, 6, 1, 0); + let config = get_config(); + + // Test width zero (no effect) + let result = format_with_modifiers(&date, "%Y", &config).unwrap(); + assert_eq!(result, "1999"); + + // Test no modifiers (standard format) + let result = format_with_modifiers(&date, "%Y-%m-%d", &config).unwrap(); + assert_eq!(result, "1999-06-01"); + + // Test %% escape sequence + let result = format_with_modifiers(&date, "%%Y=%Y", &config).unwrap(); + assert_eq!(result, "%Y=1999"); + + // Test multiple modifiers in one format string + // %-5d: no-pad suppresses all padding → "1" (width ignored) + let result = format_with_modifiers(&date, "%10Y-%_5m-%-5d", &config).unwrap(); + assert_eq!(result, "0000001999- 6-1"); + } + + #[test] + fn test_modifier_detection() { + let date = make_test_date(1999, 6, 1, 0); + let config = get_config(); + + // Should detect modifiers + let result = format_with_modifiers_if_present(&date, "%10Y", &config); + assert!(result.is_some()); + + // Should not detect modifiers + let result = format_with_modifiers_if_present(&date, "%Y-%m-%d", &config); + assert!(result.is_none()); + + // Should detect flag without width + let result = format_with_modifiers_if_present(&date, "%^B", &config); + assert!(result.is_some()); + } + + #[test] + fn test_negative_values_with_space_padding() { + // Test case from GNU test: neg-secs2 + // Format: %_5s with value -22 should produce " -22" (space-padded) + use jiff::Timestamp; + + let ts = Timestamp::from_second(-22).unwrap(); + let date = ts.to_zoned(TimeZone::UTC); + let config = get_config(); + + let result = format_with_modifiers(&date, "%_5s", &config).unwrap(); + assert_eq!(result, " -22", "Space padding should pad before the sign for negative numbers"); + } + + // Unit tests for apply_modifiers function + #[test] + fn test_apply_modifiers_basic() { + // No modifiers (numeric specifier) + assert_eq!(apply_modifiers("1999", "", 0, "Y", false).unwrap(), "1999"); + // Zero padding + assert_eq!(apply_modifiers("1999", "0", 10, "Y", true).unwrap(), "0000001999"); + // Space padding (strips leading zeros) + assert_eq!(apply_modifiers("06", "_", 5, "m", true).unwrap(), " 6"); + // No-pad (strips leading zeros, width ignored) + assert_eq!(apply_modifiers("01", "-", 5, "d", true).unwrap(), "1"); + // Uppercase + assert_eq!(apply_modifiers("june", "^", 0, "B", false).unwrap(), "JUNE"); + // Swap case: all uppercase → lowercase + assert_eq!(apply_modifiers("UTC", "#", 0, "Z", false).unwrap(), "utc"); + // Swap case: mixed case → uppercase + assert_eq!(apply_modifiers("June", "#", 0, "B", false).unwrap(), "JUNE"); + } + + #[test] + fn test_apply_modifiers_signs() { + // Force sign with explicit width + assert_eq!(apply_modifiers("1970", "+", 6, "Y", true).unwrap(), "+01970"); + // Force sign without explicit width: should NOT add sign for 4-digit year + assert_eq!(apply_modifiers("1999", "+", 0, "Y", false).unwrap(), "1999"); + // Force sign without explicit width: SHOULD add sign for year > 4 digits + assert_eq!(apply_modifiers("12345", "+", 0, "Y", false).unwrap(), "+12345"); + // Negative with zero padding: sign first, then zeros + assert_eq!(apply_modifiers("-22", "0", 5, "s", true).unwrap(), "-0022"); + // Negative with space padding: spaces first, then sign + assert_eq!(apply_modifiers("-22", "_", 5, "s", true).unwrap(), " -22"); + // Force sign (_+): + is last, overrides _ → zero pad with sign + assert_eq!(apply_modifiers("5", "_+", 5, "s", true).unwrap(), "+0005"); + // No-pad + uppercase: no padding applied + assert_eq!(apply_modifiers("june", "-^", 10, "B", true).unwrap(), "JUNE"); + } + + #[test] + fn test_case_flag_precedence() { + // Test that ^ (uppercase) overrides # (swap case) + assert_eq!(apply_modifiers("June", "^#", 0, "B", false).unwrap(), "JUNE"); + assert_eq!(apply_modifiers("June", "#^", 0, "B", false).unwrap(), "JUNE"); + // Test # alone (swap case) + assert_eq!(apply_modifiers("June", "#", 0, "B", false).unwrap(), "JUNE"); + assert_eq!(apply_modifiers("JUNE", "#", 0, "B", false).unwrap(), "june"); + } + + #[test] + fn test_apply_modifiers_text_specifiers() { + // Text specifiers default to space padding + assert_eq!(apply_modifiers("June", "", 10, "B", true).unwrap(), " June"); + assert_eq!(apply_modifiers("Mon", "", 10, "a", true).unwrap(), " Mon"); + // Numeric specifiers default to zero padding + assert_eq!(apply_modifiers("6", "", 10, "m", true).unwrap(), "0000000006"); + } + + #[test] + fn test_apply_modifiers_width_smaller_than_result() { + // Width smaller than result strips default padding + assert_eq!(apply_modifiers("01", "", 1, "d", true).unwrap(), "1"); + assert_eq!(apply_modifiers("06", "", 1, "m", true).unwrap(), "6"); + } + + #[test] + fn test_apply_modifiers_parametrized() { + let test_cases = vec![ + ("1", "0", 3, "Y", true, "001"), + ("1", "_", 3, "d", true, " 1"), + ("1", "-", 3, "d", true, "1"), // no-pad: width ignored + ("abc", "^", 5, "B", true, " ABC"), // text specifier: space pad + ("5", "+", 4, "s", true, "+005"), + ("5", "_+", 4, "s", true, "+005"), // + is last: zero pad with sign + ("-3", "0", 5, "s", true, "-0003"), + ("05", "_", 3, "d", true, " 5"), + ("09", "-", 4, "d", true, "9"), // no-pad: width ignored + ("1970", "_+", 6, "Y", true, "+01970"), // + is last: zero pad with sign + ]; + + for (value, flags, width, spec, explicit_width, expected) in test_cases { + assert_eq!( + apply_modifiers(value, flags, width, spec, explicit_width).unwrap(), + expected, + "value='{value}', flags='{flags}', width={width}, spec='{spec}', \ + explicit_width={explicit_width}", + ); + } + } + + #[test] + fn test_apply_modifiers_width_too_large() { + let err = apply_modifiers("x", "", usize::MAX, "c", true).unwrap_err(); + assert!(matches!( + err, + FormatError::FieldWidthTooLarge { width, specifier } + if width == usize::MAX && specifier == "c" + )); + } + + #[test] + fn test_underscore_flag_without_width() { + // %_m should pad month to default width 2 with spaces + assert_eq!(apply_modifiers("6", "_", 0, "m", false).unwrap(), " 6"); + // %_d should pad day to default width 2 with spaces + assert_eq!(apply_modifiers("1", "_", 0, "d", false).unwrap(), " 1"); + // %_H should pad hour to default width 2 with spaces + assert_eq!(apply_modifiers("5", "_", 0, "H", false).unwrap(), " 5"); + // %_Y should pad year to default width 4 with spaces + assert_eq!(apply_modifiers("1999", "_", 0, "Y", false).unwrap(), "1999"); + // already at default width + } + + #[test] + fn test_plus_flag_without_width() { + // %+Y without width should NOT add sign for 4-digit year + assert_eq!(apply_modifiers("1999", "+", 0, "Y", false).unwrap(), "1999"); + // %+Y without width SHOULD add sign for year > 4 digits + assert_eq!(apply_modifiers("12345", "+", 0, "Y", false).unwrap(), "+12345"); + // %+Y with explicit width should add sign + assert_eq!(apply_modifiers("1999", "+", 6, "Y", true).unwrap(), "+01999"); + } + + #[test] + fn test_zero_flag_on_space_padded_specifiers() { + // GNU date: %0e should override space-padding with zero-padding + // Verified: `date -d "2024-06-05" "+%0e"` → "05" + let date = make_test_date(1999, 6, 5, 5); + let config = get_config(); + + // %0e: day-of-month (normally space-padded) with 0 flag → zero-padded + let result = format_with_modifiers(&date, "%0e", &config).unwrap(); + assert_eq!(result, "05", "GNU: %0e should produce '05', not ' 5'"); + + // %0k: hour (normally space-padded) with 0 flag → zero-padded + let result = format_with_modifiers(&date, "%0k", &config).unwrap(); + assert_eq!(result, "05", "GNU: %0k should produce '05', not ' 5'"); + } + + #[test] + fn test_underscore_century_default_width() { + // GNU date: %C default width is 2, not 4 + // Verified: `date -d "2024-06-15" "+%_C"` → "20" (no extra padding) + let date = make_test_date(1999, 6, 1, 0); + let config = get_config(); + + // %_C: century with underscore flag, no explicit width + // Default width for %C should be 2 (century is 00-99) + let result = format_with_modifiers(&date, "%_C", &config).unwrap(); + assert_eq!( + result, "19", + "GNU: %_C should produce '19', not ' 19' (default width is 2, not 4)" + ); + } +} diff --git a/crates/vendor/uu-hostname/Cargo.toml b/crates/vendor/uu-hostname/Cargo.toml new file mode 100644 index 000000000..94660b7d7 --- /dev/null +++ b/crates/vendor/uu-hostname/Cargo.toml @@ -0,0 +1,31 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/hostname), patched to route +# output through pi-uutils-ctx and to reject the set-hostname path so it can run +# in-process as a shell builtin. See src/hostname.rs for the patch markers +# (`pi-uutils:` comments). +[package] +name = "uu_hostname" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "hostname ~ (uutils) display the host name of the current host (vendored + patched for in-process embedding)" + +[lib] +path = "src/hostname.rs" + +[dependencies] +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +hostname = "0.4" +uucore = { version = "0.8.0", features = ["wide"] } +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[target.'cfg(any(target_os = "freebsd", target_os = "openbsd"))'.dependencies] +dns-lookup = "3.0.0" + +[target.'cfg(target_os = "windows")'.dependencies] +windows-sys = { version = "0.61.0", features = [ + "Win32_Networking_WinSock", + "Win32_Foundation", +] } + +[dev-dependencies] +parking_lot = "0.12" diff --git a/crates/vendor/uu-hostname/LICENSE b/crates/vendor/uu-hostname/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-hostname/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-hostname/src/hostname.rs b/crates/vendor/uu-hostname/src/hostname.rs new file mode 100644 index 000000000..24e79b3e9 --- /dev/null +++ b/crates/vendor/uu-hostname/src/hostname.rs @@ -0,0 +1,308 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +// spell-checker:ignore hashset Addrs addrs + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. The set-hostname path is removed entirely (a NAME operand +// is rejected with an "unsupported" error instead of calling `hostname::set`, +// so the `hostname` crate's "set" feature is dropped), all output is routed +// through the context streams, `translate!` strings are literalized, and the +// entry point no longer calls `std::process::exit`. + +#[cfg(not(any(target_os = "freebsd", target_os = "openbsd")))] +use std::net::ToSocketAddrs; +use std::{collections::hash_set::HashSet, ffi::OsString, io::Write, str}; + +use clap::{Arg, ArgAction, ArgMatches, Command, builder::ValueParser}; +#[cfg(any(target_os = "freebsd", target_os = "openbsd"))] +use dns_lookup::lookup_host; +use pi_uutils_ctx::format_usage; +use uucore::error::{FromIo, UResult, USimpleError}; + +static OPT_DOMAIN: &str = "domain"; +static OPT_IP_ADDRESS: &str = "ip-address"; +static OPT_FQDN: &str = "fqdn"; +static OPT_SHORT: &str = "short"; +static OPT_HOST: &str = "host"; + +#[cfg(windows)] +mod wsa { + use std::io; + + use windows_sys::Win32::Networking::WinSock::{WSACleanup, WSADATA, WSAStartup}; + + pub(super) struct WsaHandle(()); + + pub(super) fn start() -> io::Result { + let mut data = std::mem::MaybeUninit::::uninit(); + let err = unsafe { WSAStartup(0x0202, data.as_mut_ptr()) }; + if err == 0 { + Ok(WsaHandle(())) + } else { + Err(io::Error::from_raw_os_error(err)) + } + } + + impl Drop for WsaHandle { + fn drop(&mut self) { + // This possibly returns an error but we can't handle it + let _ = unsafe { WSACleanup() }; + } + } +} + +/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the +/// arguments directly (without the uucore clap-localization helper that would +/// terminate the process), renders clap help/usage/version to the context +/// streams, and maps the `UResult` to an exit code, so it is safe to run inside +/// the host shell process. +pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match hostname_main(&matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "hostname: {msg}"); + } + if code == 0 { 1 } else { code } + }, + } +} + +fn hostname_main(matches: &ArgMatches) -> UResult<()> { + #[cfg(windows)] + let _handle = wsa::start().map_err_context(|| "failed to start Winsock".to_string())?; + + match matches.get_one::(OPT_HOST) { + None => display_hostname(matches), + // pi-uutils: setting the process-global hostname from inside a shell + // builtin is refused (upstream calls `hostname::set` here). + Some(_host) => Err(USimpleError::new( + 1, + "setting the hostname is not supported by this builtin".to_string(), + )), + } +} + +pub fn uu_app() -> Command { + Command::new("hostname") + .version(uucore::crate_version!()) + .about("Display or set the system's host name.") + .override_usage(format_usage("hostname [OPTION]... [HOSTNAME]")) + .infer_long_args(true) + .arg( + Arg::new(OPT_DOMAIN) + .short('d') + .long("domain") + .overrides_with_all([OPT_DOMAIN, OPT_IP_ADDRESS, OPT_FQDN, OPT_SHORT]) + .help("Display the name of the DNS domain if possible") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_IP_ADDRESS) + .short('i') + .long("ip-address") + .overrides_with_all([OPT_DOMAIN, OPT_IP_ADDRESS, OPT_FQDN, OPT_SHORT]) + .help("Display the network address(es) of the host") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_FQDN) + .short('f') + .long("fqdn") + .overrides_with_all([OPT_DOMAIN, OPT_IP_ADDRESS, OPT_FQDN, OPT_SHORT]) + .help("Display the FQDN (Fully Qualified Domain Name) (default)") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_SHORT) + .short('s') + .long("short") + .overrides_with_all([OPT_DOMAIN, OPT_IP_ADDRESS, OPT_FQDN, OPT_SHORT]) + .help("Display the short hostname (the portion before the first dot) if possible") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_HOST) + .value_parser(ValueParser::os_string()) + .value_hint(clap::ValueHint::Hostname), + ) +} + +fn display_hostname(matches: &ArgMatches) -> UResult<()> { + let hostname = hostname::get() + .map_err_context(|| "failed to get hostname".to_owned())? + .to_string_lossy() + .into_owned(); + + // pi-uutils: all output below goes to the context stdout instead of the + // process stdout. + let mut out = pi_uutils_ctx::stdout(); + + if matches.get_flag(OPT_IP_ADDRESS) { + let addresses; + + #[cfg(not(any(target_os = "freebsd", target_os = "openbsd")))] + { + let hostname = hostname + ":1"; + let addrs = hostname + .to_socket_addrs() + .map_err_context(|| "failed to resolve socket addresses".to_owned())?; + addresses = addrs; + } + + // DNS reverse lookup via "hostname:1" does not work on FreeBSD and OpenBSD + // use dns-lookup crate instead + #[cfg(any(target_os = "freebsd", target_os = "openbsd"))] + { + let addrs: Vec = lookup_host(hostname.as_str()) + .map_err_context(|| "failed to lookup hostname".to_owned())? + .collect(); + addresses = addrs; + } + + let mut hashset = HashSet::new(); + let mut output = String::new(); + for addr in addresses { + // XXX: not sure why this is necessary... + if !hashset.contains(&addr) { + let mut ip = addr.to_string(); + if ip.ends_with(":1") { + let len = ip.len(); + ip.truncate(len - 2); + } + output.push_str(&ip); + output.push(' '); + hashset.insert(addr); + } + } + let len = output.len(); + if len > 0 { + writeln!(out, "{}", &output[0..len - 1])?; + } + + Ok(()) + } else { + if matches.get_flag(OPT_SHORT) || matches.get_flag(OPT_DOMAIN) { + let mut it = hostname.char_indices().filter(|&ci| ci.1 == '.'); + if let Some(ci) = it.next() { + if matches.get_flag(OPT_SHORT) { + writeln!(out, "{}", &hostname[0..ci.0])?; + } else { + writeln!(out, "{}", &hostname[ci.0 + 1..])?; + } + } else if matches.get_flag(OPT_SHORT) { + writeln!(out, "{hostname}")?; + } + return Ok(()); + } + + writeln!(out, "{hostname}")?; + + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + fn run_in(args: Vec<&str>) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd: PathBuf::from("."), + env: HashMap::new(), + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; + + let argv: Vec = std::iter::once("hostname") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + #[test] + fn bare_invocation_prints_hostname() { + let (code, stdout, stderr) = run_in(vec![]); + assert_eq!((code, stderr.as_str()), (0, "")); + assert!(stdout.ends_with('\n')); + assert!(!stdout.trim_end().is_empty()); + } + + #[test] + fn set_attempt_is_rejected() { + let (code, stdout, stderr) = run_in(vec!["new-name.example.com"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert_eq!(stderr, "hostname: setting the hostname is not supported by this builtin\n"); + } + + #[test] + fn short_is_dotless_prefix_of_full_hostname() { + let (code, short, stderr) = run_in(vec!["-s"]); + let (_, full, _) = run_in(vec![]); + assert_eq!((code, stderr.as_str()), (0, "")); + let short = short.trim_end(); + assert!(!short.contains('.'), "-s must strip everything after the first dot"); + assert!(full.trim_end().starts_with(short)); + } + + #[test] + fn fqdn_flag_matches_default_display() { + // -f is the default display mode; it must print the same name as the + // bare invocation, not attempt any set path. + let (code, fqdn, stderr) = run_in(vec!["-f"]); + let (_, bare, _) = run_in(vec![]); + assert_eq!((code, stderr.as_str()), (0, "")); + assert_eq!(fqdn, bare); + } +} diff --git a/crates/vendor/uu-ln/Cargo.toml b/crates/vendor/uu-ln/Cargo.toml new file mode 100644 index 000000000..2cc4020a4 --- /dev/null +++ b/crates/vendor/uu-ln/Cargo.toml @@ -0,0 +1,23 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/ln), patched to resolve path +# arguments against the shell working directory and route I/O + prompts through +# pi-uutils-ctx so it can run in-process as a shell builtin. See src/ln.rs for +# the patch markers (`pi-uutils:` comments). +[package] +name = "uu_ln" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "ln ~ (uutils) create a (file system) link to TARGET (vendored + patched for in-process embedding)" + +[lib] +path = "src/ln.rs" + +[dependencies] +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +thiserror = "2.0.3" +uucore = { version = "0.8.0", features = ["backup-control", "fs"] } +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[dev-dependencies] +parking_lot = "0.12" +tempfile = "3" diff --git a/crates/vendor/uu-ln/LICENSE b/crates/vendor/uu-ln/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-ln/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-ln/src/ln.rs b/crates/vendor/uu-ln/src/ln.rs new file mode 100644 index 000000000..d193c8b26 --- /dev/null +++ b/crates/vendor/uu-ln/src/ln.rs @@ -0,0 +1,831 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +// spell-checker:ignore (ToDO) srcpath targetpath EEXIST + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. Every filesystem syscall resolves its path operand +// against the shell working directory via `pi_uutils_ctx::resolve` AT THE CALL +// SITE, while the original operands are kept for display/error messages (GNU +// prints operands as typed) — and, crucially, for the CONTENT of symbolic +// links, which stays exactly as typed like GNU ln (only the location where the +// link is created gets resolved). All process-global stdio and the `-i` prompt +// are routed through `pi_uutils_ctx`, `translate!` strings are literalized, and +// the entry point no longer calls `std::process::exit`. + +#[cfg(any(unix, target_os = "redox"))] +use std::os::unix::fs::symlink; +#[cfg(windows)] +use std::os::windows::fs::{symlink_dir, symlink_file}; +use std::{ + borrow::Cow, + collections::HashSet, + ffi::OsString, + fs, + io::Write, + path::{Path, PathBuf}, +}; + +use clap::{Arg, ArgAction, ArgMatches, Command}; +use pi_uutils_ctx::format_usage; +use thiserror::Error; +use uucore::{ + backup_control::{self, BackupMode}, + display::Quotable, + error::{FromIo, UError, UResult, USimpleError, strip_errno}, + fs::{ + MissingHandling, ResolveMode, canonicalize, make_path_relative_to, paths_refer_to_same_file, + }, +}; + +pub struct Settings { + overwrite: OverwriteMode, + backup: BackupMode, + suffix: OsString, + symbolic: bool, + relative: bool, + logical: bool, + target_dir: Option, + no_target_dir: bool, + no_dereference: bool, + verbose: bool, +} + +#[derive(Clone, Debug, Eq, PartialEq)] +pub enum OverwriteMode { + NoClobber, + Interactive, + Force, +} + +// pi-uutils: the `translate!` message templates are literalized with the +// en-US strings from upstream's locales/en-US.ftl. +#[derive(Error, Debug)] +enum LnError { + #[error("target {} is not a directory", _0.quote())] + TargetIsNotADirectory(PathBuf), + + #[error("")] + SomeLinksFailed, + + #[error("{} and {} are the same file", _0.quote(), _1.quote())] + SameFile(PathBuf, PathBuf), + + #[error("missing destination file operand after {}", _0.quote())] + MissingDestination(PathBuf), + + #[error("extra operand {}\nTry '{} --help' for more information.", _0.quote(), _1)] + ExtraOperand(OsString, String), + + #[error("{}: hard link not allowed for directory", _0.to_string_lossy())] + FailedToCreateHardLinkDir(PathBuf), +} + +impl UError for LnError { + fn code(&self) -> i32 { + 1 + } +} + +mod options { + pub const FORCE: &str = "force"; + //pub const DIRECTORY: &str = "directory"; + pub const INTERACTIVE: &str = "interactive"; + pub const NO_DEREFERENCE: &str = "no-dereference"; + pub const SYMBOLIC: &str = "symbolic"; + pub const LOGICAL: &str = "logical"; + pub const PHYSICAL: &str = "physical"; + pub const TARGET_DIRECTORY: &str = "target-directory"; + pub const NO_TARGET_DIRECTORY: &str = "no-target-directory"; + pub const RELATIVE: &str = "relative"; + pub const VERBOSE: &str = "verbose"; +} + +static ARG_FILES: &str = "files"; + +/// pi-uutils: replacement for uucore's `show_error!` — writes the diagnostic +/// to the context stderr instead of the process-global one. Errors that render +/// to an empty message (e.g. [`LnError::SomeLinksFailed`]) print nothing +/// rather than a dangling "ln: " prefix. +fn show_error(msg: impl std::fmt::Display) { + let rendered = msg.to_string(); + if !rendered.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "ln: {rendered}"); + } +} + +/// pi-uutils: replacement for uucore's `read_yes`, reading from the context +/// stdin one byte at a time (no buffering) so consecutive prompts don't +/// over-read into a later prompt's input. Returns true when the first character +/// of the line is `y`/`Y`. +fn read_yes() -> bool { + use std::io::Read as _; + let mut stdin = pi_uutils_ctx::stdin(); + let mut buf = [0u8; 1]; + let mut first = None; + loop { + match stdin.read(&mut buf) { + Ok(0) => break, // EOF + Ok(_) => { + if buf[0] == b'\n' { + break; + } + if first.is_none() { + first = Some(buf[0]); + } + }, + Err(_) => return false, + } + } + matches!(first, Some(b'y' | b'Y')) +} + +/// pi-uutils: replacement for uucore's `prompt_yes!` — writes +/// "ln: \ " to the context stderr, then reads the answer from the +/// context stdin. +fn prompt_yes(prompt: impl std::fmt::Display) -> bool { + let mut err = pi_uutils_ctx::stderr(); + let _ = write!(err, "ln: {prompt} "); + let _ = err.flush(); + read_yes() +} + +/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the +/// arguments directly (without the uucore clap-localization helper that would +/// terminate the process), renders clap help/usage/version to the context +/// streams, and maps the `UResult` to an exit code, so it is safe to run inside +/// the host shell process. +pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match ln_main(&matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + // pi-uutils: `SomeLinksFailed` renders to an empty message + // (upstream prints the per-file diagnostics as it goes); don't + // emit a dangling "ln: " prefix for it. + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "ln: {msg}"); + } + if code == 0 { 1 } else { code } + }, + } +} + +fn ln_main(matches: &ArgMatches) -> UResult<()> { + /* the list of files */ + + let paths: Vec = matches + .get_many::(ARG_FILES) + .unwrap() + .map(PathBuf::from) + .collect(); + + let symbolic = matches.get_flag(options::SYMBOLIC); + + let overwrite_mode = if matches.get_flag(options::FORCE) { + OverwriteMode::Force + } else if matches.get_flag(options::INTERACTIVE) { + OverwriteMode::Interactive + } else { + OverwriteMode::NoClobber + }; + + let backup_mode = backup_control::determine_backup_mode(matches)?; + let backup_suffix = backup_control::determine_backup_suffix(matches); + + // When we have "-L" or "-L -P", false otherwise + let logical = matches.get_flag(options::LOGICAL); + + let settings = Settings { + overwrite: overwrite_mode, + backup: backup_mode, + suffix: OsString::from(backup_suffix), + symbolic, + logical, + relative: matches.get_flag(options::RELATIVE), + target_dir: matches + .get_one::(options::TARGET_DIRECTORY) + .map(PathBuf::from), + no_target_dir: matches.get_flag(options::NO_TARGET_DIRECTORY), + no_dereference: matches.get_flag(options::NO_DEREFERENCE), + verbose: matches.get_flag(options::VERBOSE), + }; + + exec(&paths[..], &settings) +} + +pub fn uu_app() -> Command { + let after_help = format!( + "In the 1st form, create a link to TARGET with the name LINK_NAME.\nIn the 2nd form, create \ + a link to TARGET in the current directory.\nIn the 3rd and 4th forms, create links to each \ + TARGET in DIRECTORY.\nCreate hard links by default, symbolic links with --symbolic.\nBy \ + default, each destination (name of new link) should not already exist.\nWhen creating hard \ + links, each TARGET must exist. Symbolic links\ncan hold arbitrary text; if later resolved, \ + a relative link is\ninterpreted in relation to its parent directory.\n\n{}", + backup_control::BACKUP_CONTROL_LONG_HELP + ); + + Command::new("ln") + .version(uucore::crate_version!()) + .about("Make links between files.") + .override_usage(format_usage( + "ln [OPTION]... [-T] TARGET LINK_NAME\nln [OPTION]... TARGET\nln [OPTION]... TARGET... \ + DIRECTORY\nln [OPTION]... -t DIRECTORY TARGET...", + )) + .infer_long_args(true) + .after_help(after_help) + .arg(backup_control::arguments::backup()) + .arg(backup_control::arguments::backup_no_args()) + /*.arg( + Arg::new(options::DIRECTORY) + .short('d') + .long(options::DIRECTORY) + .help("allow users with appropriate privileges to attempt to make hard links to directories") + )*/ + .arg( + Arg::new(options::FORCE) + .short('f') + .long(options::FORCE) + .help("remove existing destination files") + .overrides_with(options::INTERACTIVE) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::INTERACTIVE) + .short('i') + .long(options::INTERACTIVE) + .help("prompt whether to remove existing destination files") + .overrides_with(options::FORCE) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::NO_DEREFERENCE) + .short('n') + .long(options::NO_DEREFERENCE) + .help("treat LINK_NAME as a normal file if it is a\nsymbolic link to a directory") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::LOGICAL) + .short('L') + .long(options::LOGICAL) + .help("follow TARGETs that are symbolic links") + .overrides_with(options::PHYSICAL) + .action(ArgAction::SetTrue), + ) + .arg( + // Not implemented yet + Arg::new(options::PHYSICAL) + .short('P') + .long(options::PHYSICAL) + .help("make hard links directly to symbolic links") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::SYMBOLIC) + .short('s') + .long(options::SYMBOLIC) + .help("make symbolic links instead of hard links") + // override added for https://github.com/uutils/coreutils/issues/2359 + .overrides_with(options::SYMBOLIC) + .action(ArgAction::SetTrue), + ) + .arg(backup_control::arguments::suffix()) + .arg( + Arg::new(options::TARGET_DIRECTORY) + .short('t') + .long(options::TARGET_DIRECTORY) + .help("specify the DIRECTORY in which to create the links") + .value_name("DIRECTORY") + .value_hint(clap::ValueHint::DirPath) + .value_parser(clap::value_parser!(OsString)) + .conflicts_with(options::NO_TARGET_DIRECTORY), + ) + .arg( + Arg::new(options::NO_TARGET_DIRECTORY) + .short('T') + .long(options::NO_TARGET_DIRECTORY) + .help("treat LINK_NAME as a normal file always") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::RELATIVE) + .short('r') + .long(options::RELATIVE) + .help("create symbolic links relative to link location") + .requires(options::SYMBOLIC) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::VERBOSE) + .short('v') + .long(options::VERBOSE) + .help("print name of each linked file") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(ARG_FILES) + .action(ArgAction::Append) + .value_hint(clap::ValueHint::AnyPath) + .value_parser(clap::value_parser!(OsString)) + .required(true) + .num_args(1..), + ) +} + +fn exec(files: &[PathBuf], settings: &Settings) -> UResult<()> { + // Handle cases where we create links in a directory first. + if let Some(target_path) = &settings.target_dir { + // 4th form: a directory is specified by -t. + return link_files_in_dir(files, target_path, settings); + } + if !settings.no_target_dir { + if files.len() == 1 { + // 2nd form: the target directory is the current directory. + return link_files_in_dir(files, &PathBuf::from("."), settings); + } + let last_file = &PathBuf::from(files.last().unwrap()); + // pi-uutils: probe the destination via the resolved path. + if files.len() > 2 || pi_uutils_ctx::resolve(last_file).is_dir() { + // 3rd form: create links in the last argument. + return link_files_in_dir(&files[0..files.len() - 1], last_file, settings); + } + } + + // 1st form. Now there should be only two operands, but if -T is + // specified we may have a wrong number of operands. + if files.len() == 1 { + return Err(LnError::MissingDestination(files[0].clone()).into()); + } + if files.len() > 2 { + // pi-uutils: `uucore::execution_phrase()` reads the process argv, + // which is the host shell's; the builtin is always invoked as "ln". + return Err(LnError::ExtraOperand(files[2].clone().into(), "ln".to_string()).into()); + } + assert!(!files.is_empty()); + + link(&files[0], &files[1], settings) +} + +#[allow(clippy::cognitive_complexity)] +fn link_files_in_dir(files: &[PathBuf], target_dir: &Path, settings: &Settings) -> UResult<()> { + // pi-uutils: resolved target directory for every syscall below; the + // operand keeps its as-typed spelling for display and link-name building. + let target_dir_fs = pi_uutils_ctx::resolve(target_dir); + if !target_dir_fs.is_dir() { + return Err(LnError::TargetIsNotADirectory(target_dir.to_owned()).into()); + } + // remember the linked destinations for further usage + let mut linked_destinations: HashSet = HashSet::with_capacity(files.len()); + + let mut all_successful = true; + for srcpath in files { + let targetpath = if settings.no_dereference && target_dir_fs.is_symlink() { + let remove_target = || { + // In that case, we don't want to do link resolution + // We need to clean the target + if target_dir_fs.is_file() + && let Err(e) = fs::remove_file(&target_dir_fs) + { + show_error(format_args!("Could not update {}: {e}", target_dir.quote())); + } + #[cfg(windows)] + if target_dir_fs.is_dir() { + // Not sure why but on Windows, the symlink can be + // considered as a dir + // See test_ln::test_symlink_no_deref_dir + if let Err(e) = fs::remove_dir(&target_dir_fs) { + show_error(format_args!("Could not update {}: {e}", target_dir.quote())); + } + } + }; + match settings.overwrite { + OverwriteMode::NoClobber => {}, + OverwriteMode::Interactive => { + if prompt_yes(format_args!("replace {}?", target_dir.quote())) { + remove_target(); + } + }, + OverwriteMode::Force => { + remove_target(); + }, + } + target_dir.to_path_buf() + } else if let Some(name) = srcpath.as_os_str().to_str() { + match Path::new(name).file_name() { + Some(basename) => target_dir.join(basename), + // This can be None only for "." or "..". Trying + // to create a link with such name will fail with + // EEXIST, which agrees with the behavior of GNU + // coreutils. + None => target_dir.join(name), + } + } else { + show_error(format_args!("cannot stat {}: No such file or directory", srcpath.quote())); + all_successful = false; + continue; + }; + + if linked_destinations.contains(&targetpath) { + // If the target file was already created in this ln call, do not overwrite + show_error(format_args!( + "will not overwrite just-created {} with {}", + targetpath.quote(), + srcpath.quote() + )); + all_successful = false; + } else if let Err(e) = link(srcpath, &targetpath, settings) { + show_error(format_args!("{e}")); + all_successful = false; + } + + linked_destinations.insert(targetpath.clone()); + } + if all_successful { + Ok(()) + } else { + Err(LnError::SomeLinksFailed.into()) + } +} + +fn relative_path<'a>(src: &'a Path, dst: &Path) -> Cow<'a, Path> { + // pi-uutils: canonicalize from the resolved operands so `-r` computes the + // link text against the shell working directory (uucore's canonicalize + // would otherwise fall back to the process cwd for relative paths). + if let Ok(src_abs) = + canonicalize(pi_uutils_ctx::resolve(src), MissingHandling::Missing, ResolveMode::Physical) + && let Ok(dst_abs) = canonicalize( + pi_uutils_ctx::resolve(dst.parent().unwrap()), + MissingHandling::Missing, + ResolveMode::Physical, + ) { + return make_path_relative_to(src_abs, dst_abs).into(); + } + src.into() +} + +#[allow(clippy::cognitive_complexity)] +fn link(src: &Path, dst: &Path, settings: &Settings) -> UResult<()> { + let mut backup_path = None; + let source: Cow<'_, Path> = if settings.relative { + relative_path(src, dst) + } else { + src.into() + }; + + // pi-uutils: resolved counterparts of both operands for every filesystem + // syscall below. `src`/`dst`/`source` keep the as-typed spelling for + // display — and `source` is what gets stored as the symlink CONTENT, so it + // must never be resolved. + let src_fs = pi_uutils_ctx::resolve(src); + let dst_fs = pi_uutils_ctx::resolve(dst); + + if dst_fs.is_symlink() || dst_fs.exists() { + // pi-uutils: probe numbered backups from the resolved destination so + // the directory scan hits the shell's working directory. + backup_path = backup_control::get_backup_path(settings.backup, &dst_fs, &settings.suffix); + if settings.backup == BackupMode::Existing && !settings.symbolic { + // when ln --backup f f, it should detect that it is the same file + if paths_refer_to_same_file(&src_fs, &dst_fs, true) { + return Err(LnError::SameFile(src.to_owned(), dst.to_owned()).into()); + } + } + if let Some(p) = &backup_path { + fs::rename(&dst_fs, p).map_err_context(|| format!("cannot backup {}", dst.quote()))?; + } + match settings.overwrite { + OverwriteMode::NoClobber => {}, + OverwriteMode::Interactive => { + if !prompt_yes(format_args!("replace {}?", dst.quote())) { + return Err(LnError::SomeLinksFailed.into()); + } + + let _ = fs::remove_file(&dst_fs); + // In case of error, don't do anything + }, + OverwriteMode::Force => { + if !dst_fs.is_symlink() && paths_refer_to_same_file(&src_fs, &dst_fs, true) { + // Even in force overwrite mode, verify we are not targeting the same entry and + // return a SameFile error if so + let same_entry = match ( + canonicalize(&src_fs, MissingHandling::Missing, ResolveMode::Physical), + canonicalize(&dst_fs, MissingHandling::Missing, ResolveMode::Physical), + ) { + (Ok(src), Ok(dst)) => src == dst, + _ => true, + }; + if same_entry { + return Err(LnError::SameFile(src.to_owned(), dst.to_owned()).into()); + } + } + let _ = fs::remove_file(&dst_fs); + // In case of error, don't do anything + }, + } + } + + let res: UResult<()> = if settings.symbolic { + // pi-uutils: the link is created at the resolved location, but its + // content (`source`) stays exactly as typed, like GNU ln. uucore's + // io-error conversion renders EEXIST as "Already exists"; format the + // GNU-style diagnostic ("failed to create symbolic link 'x': File + // exists") from the raw OS error instead. + symlink(&source, &dst_fs).map_err(|e| { + USimpleError::new( + 1, + format!("failed to create symbolic link {}: {}", dst.quote(), strip_errno(&e)), + ) + }) + } else { + // pi-uutils: hard links dereference their target, so the resolved + // source is what the syscalls get. + let source_fs = pi_uutils_ctx::resolve(&source); + let p = if settings.logical && source_fs.is_symlink() { + fs::canonicalize(&source_fs) + .map_err_context(|| format!("failed to access {}", source.quote()))? + } else { + source_fs + }; + match fs::hard_link(&p, &dst_fs) { + Ok(()) => Ok(()), + Err(_) if p.is_dir() => { + Err(LnError::FailedToCreateHardLinkDir(source.to_path_buf()).into()) + }, + // pi-uutils: same GNU-style rendering as the symlink arm (uucore + // would print "Already exists" for EEXIST). + Err(e) => Err(USimpleError::new( + 1, + format!( + "failed to create hard link {} => {}: {}", + source.quote(), + dst.quote(), + strip_errno(&e) + ), + )), + } + }; + + if let Err(e) = res { + if let Some(p) = &backup_path { + fs::rename(p, &dst_fs).map_err_context(|| format!("cannot backup {}", dst.quote()))?; + } + return Err(e); + } + + if settings.verbose { + // pi-uutils: verbose output goes to the context stdout. + let mut out = pi_uutils_ctx::stdout(); + write!(out, "{} -> {}", dst.quote(), source.quote())?; + match backup_path { + Some(path) => { + // pi-uutils: `path` derives from the resolved (absolute) + // destination; rebuild a display path from the operand for + // the verbose message. + let backup_display = match (dst.parent(), path.file_name()) { + (Some(parent), Some(name)) if !parent.as_os_str().is_empty() => parent.join(name), + (_, Some(name)) => PathBuf::from(name), + _ => path.clone(), + }; + writeln!(out, " (backup: {})", backup_display.quote())?; + }, + None => writeln!(out)?, + } + } + Ok(()) +} + +#[cfg(windows)] +pub fn symlink, P2: AsRef>(src: P1, dst: P2) -> std::io::Result<()> { + // pi-uutils: the dir/file probe resolves the target against the shell + // working directory (upstream consults the process cwd); the stored link + // content is still the caller's as-typed `src`. + if pi_uutils_ctx::resolve(src.as_ref()).is_dir() { + symlink_dir(src, dst) + } else { + symlink_file(src, dst) + } +} + +#[cfg(target_os = "wasi")] +fn symlink, P2: AsRef>(_src: P1, _dst: P2) -> std::io::Result<()> { + Err(std::io::Error::new( + std::io::ErrorKind::Unsupported, + "symlinks not supported on this platform", + )) +} + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + fn run_with_stdin(cwd: PathBuf, args: Vec<&str>, stdin: &[u8]) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::Cursor::new(stdin.to_vec())), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd, + env: HashMap::new(), + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; + + let argv: Vec = std::iter::once("ln") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + fn run_in(cwd: PathBuf, args: Vec<&str>) -> (i32, String, String) { + run_with_stdin(cwd, args, b"") + } + + /// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var, + /// which canonicalizing code paths would otherwise expand mid-assertion). + fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) { + let dir = tempfile::tempdir().unwrap(); + let canon = fs::canonicalize(dir.path()).unwrap(); + (dir, canon) + } + + #[cfg(unix)] + #[test] + fn symlink_relative_operands_create_in_scope_cwd_with_literal_content() { + let (_dir, root) = canonical_tempdir(); + + // Relative operands + scope cwd differing from the process cwd: only + // the call-site `pi_uutils_ctx::resolve` patch places the link in the + // tempdir — while the CONTENT must stay exactly as typed. + let (code, stdout, stderr) = run_in(root.clone(), vec!["-s", "target", "link"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", "")); + + let link = root.join("link"); + assert!(link.is_symlink(), "link must be created inside the scope cwd"); + assert_eq!(fs::read_link(&link).unwrap(), PathBuf::from("target")); + } + + #[cfg(unix)] + #[test] + fn hard_link_shares_inode() { + use std::os::unix::fs::MetadataExt; + + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("a"), b"payload").unwrap(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["a", "b"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", "")); + + assert_eq!(fs::read(root.join("b")).unwrap(), b"payload"); + assert_eq!(fs::metadata(root.join("a")).unwrap().nlink(), 2); + assert_eq!( + fs::metadata(root.join("a")).unwrap().ino(), + fs::metadata(root.join("b")).unwrap().ino() + ); + } + + #[cfg(unix)] + #[test] + fn existing_destination_without_force_fails_with_file_exists() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("link"), b"old").unwrap(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["-s", "target", "link"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert_eq!(stderr, "ln: failed to create symbolic link 'link': File exists\n"); + assert_eq!(fs::read(root.join("link")).unwrap(), b"old", "destination must be untouched"); + } + + #[cfg(unix)] + #[test] + fn force_overwrites_existing_destination() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("link"), b"old").unwrap(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["-sf", "target", "link"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", "")); + assert_eq!(fs::read_link(root.join("link")).unwrap(), PathBuf::from("target")); + } + + #[cfg(unix)] + #[test] + fn verbose_symlink_prints_mapping_to_stdout() { + let (_dir, root) = canonical_tempdir(); + + let (code, stdout, stderr) = run_in(root, vec!["-sv", "target", "link"]); + assert_eq!(code, 0); + assert_eq!(stdout, "'link' -> 'target'\n"); + assert_eq!(stderr, ""); + } + + #[cfg(unix)] + #[test] + fn interactive_prompt_reads_ctx_stdin() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("link"), b"old").unwrap(); + + // Decline: destination untouched, some-links-failed exit code, no + // dangling "ln: " diagnostic beyond the prompt itself. + let (code, stdout, stderr) = + run_with_stdin(root.clone(), vec!["-si", "target", "link"], b"n\n"); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert_eq!(stderr, "ln: replace 'link'? "); + assert!(!root.join("link").is_symlink()); + + // Accept: existing file is replaced by the symlink. + let (code, _, stderr) = run_with_stdin(root.clone(), vec!["-si", "target", "link"], b"y\n"); + assert_eq!(code, 0); + assert_eq!(stderr, "ln: replace 'link'? "); + assert_eq!(fs::read_link(root.join("link")).unwrap(), PathBuf::from("target")); + } + + #[cfg(unix)] + #[test] + fn relative_flag_computes_link_text_against_scope_cwd() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("target"), b"x").unwrap(); + fs::create_dir(root.join("sub")).unwrap(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["-sr", "target", "sub/link"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", "")); + assert_eq!(fs::read_link(root.join("sub").join("link")).unwrap(), PathBuf::from("../target")); + } + + #[cfg(unix)] + #[test] + fn target_directory_flag_places_links_in_directory() { + let (_dir, root) = canonical_tempdir(); + fs::create_dir(root.join("d")).unwrap(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["-s", "-t", "d", "x"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", "")); + assert_eq!(fs::read_link(root.join("d").join("x")).unwrap(), PathBuf::from("x")); + } + + #[test] + fn missing_destination_is_an_error() { + let (_dir, root) = canonical_tempdir(); + + let (code, stdout, stderr) = run_in(root, vec!["-T", "only"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert!( + stderr.contains("missing destination file operand after 'only'"), + "stderr was: {stderr:?}" + ); + } + + #[test] + fn help_renders_to_scope_stdout() { + let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["--help"]); + assert_eq!(code, 0); + assert!(stdout.contains("Usage:")); + assert!(stdout.contains("Make links between files.")); + assert_eq!(stderr, ""); + } +} diff --git a/crates/vendor/uu-mktemp/Cargo.toml b/crates/vendor/uu-mktemp/Cargo.toml new file mode 100644 index 000000000..dae4ba329 --- /dev/null +++ b/crates/vendor/uu-mktemp/Cargo.toml @@ -0,0 +1,24 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/mktemp), patched to resolve +# path arguments against the shell working directory and route I/O through +# pi-uutils-ctx so it can run in-process as a shell builtin. See src/mktemp.rs +# for the patch markers (`pi-uutils:` comments). +[package] +name = "uu_mktemp" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "mktemp ~ (uutils) create and display a temporary file or directory from TEMPLATE (vendored + patched for in-process embedding)" + +[lib] +path = "src/mktemp.rs" + +[dependencies] +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +rand = { version = "0.10.0", features = ["std_rng"] } +tempfile = "3.15.0" +thiserror = "2.0.3" +uucore = "0.8.0" +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[dev-dependencies] +parking_lot = "0.12" diff --git a/crates/vendor/uu-mktemp/LICENSE b/crates/vendor/uu-mktemp/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-mktemp/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-mktemp/src/mktemp.rs b/crates/vendor/uu-mktemp/src/mktemp.rs new file mode 100644 index 000000000..7e1769b53 --- /dev/null +++ b/crates/vendor/uu-mktemp/src/mktemp.rs @@ -0,0 +1,877 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +// spell-checker:ignore (paths) GPGHome findxs + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. The temp-file parent directory (from `-p`/`--tmpdir` or a +// relative TEMPLATE prefix) is resolved against the shell working directory via +// `pi_uutils_ctx::resolve` at the creation sites, so the printed path is the +// path actually created. TMPDIR/POSIXLY_CORRECT are read from the scope +// environment, all stdio goes through `pi_uutils_ctx`, `translate!` strings are +// literalized, and the entry point no longer calls `std::process::exit`. + +#[cfg(unix)] +use std::fs; +#[cfg(unix)] +use std::os::unix::prelude::PermissionsExt; +use std::{ + env, + ffi::{OsStr, OsString}, + io::{ErrorKind, Write}, + iter, + path::{MAIN_SEPARATOR, Path, PathBuf}, +}; + +use clap::{ + Arg, ArgAction, ArgMatches, Command, + builder::{TypedValueParser, ValueParserFactory}, +}; +use pi_uutils_ctx::format_usage; +use rand::{ + RngExt as _, SeedableRng as _, + rngs::{self, SmallRng}, +}; +use tempfile::Builder; +use thiserror::Error; +use uucore::{ + display::Quotable, + error::{FromIo, UError, UResult}, +}; + +static DEFAULT_TEMPLATE: &str = "tmp.XXXXXXXXXX"; + +static OPT_DIRECTORY: &str = "directory"; +static OPT_DRY_RUN: &str = "dry-run"; +static OPT_QUIET: &str = "quiet"; +static OPT_SUFFIX: &str = "suffix"; +static OPT_TMPDIR: &str = "tmpdir"; +static OPT_P: &str = "p"; +static OPT_T: &str = "t"; + +static ARG_TEMPLATE: &str = "template"; + +#[cfg(not(windows))] +const TMPDIR_ENV_VAR: &str = "TMPDIR"; +#[cfg(windows)] +const TMPDIR_ENV_VAR: &str = "TMP"; + +const FALLBACK_TMPDIR: &str = "/tmp"; + +// pi-uutils: `translate!` error strings literalized from locales/en-US.ftl. +#[derive(Error, Debug)] +enum MkTempError { + #[error("could not persist file {}", .0.quote())] + PersistError(PathBuf), + + #[error("with --suffix, template {} must end in X", .0.quote())] + MustEndInX(String), + + #[error("too few X's in template {}", .0.quote())] + TooFewXs(String), + + #[error("invalid template, {}, contains directory separator", .0.quote())] + PrefixContainsDirSeparator(String), + + #[error("invalid suffix {}, contains directory separator", .0.quote())] + SuffixContainsDirSeparator(String), + + #[error("invalid template, {}; with --tmpdir, it may not be absolute", .0.quote())] + InvalidTemplate(OsString), + + #[error("too many templates")] + TooManyTemplates, + + #[error("failed to create {} via template {}: No such file or directory", .0, .1.quote())] + NotFound(String, PathBuf), +} + +impl UError for MkTempError { + fn usage(&self) -> bool { + matches!(self, Self::TooManyTemplates) + } +} + +/// Options parsed from the command-line. +/// +/// This provides a layer of indirection between the application logic +/// and the argument parsing library `clap`, allowing each to vary +/// independently. +#[derive(Clone)] +pub struct Options { + /// Whether to create a temporary directory instead of a file. + pub directory: bool, + + /// Whether to just print the name of a file that would have been created. + pub dry_run: bool, + + /// Whether to suppress file creation error messages. + pub quiet: bool, + + /// The directory in which to create the temporary file. + /// + /// If `None`, the file will be created in the current directory. + pub tmpdir: Option, + + /// The suffix to append to the temporary file, if any. + pub suffix: Option, + + /// Whether to treat the template argument as a single file path component. + pub treat_as_template: bool, + + /// The template to use for the name of the temporary file. + pub template: OsString, +} + +impl Options { + fn from(matches: &ArgMatches) -> Self { + let tmpdir = matches + .get_one::>(OPT_TMPDIR) + .or_else(|| matches.get_one::>(OPT_P)) + .map(|dir| match dir { + // If the argument of -p/--tmpdir is non-empty, use it as the + // tmpdir. + Some(d) => d.clone(), + // Otherwise use $TMPDIR if set, else use the system's default + // temporary directory. + None => get_tmpdir_env_or_default(), + }); + let (tmpdir, template) = match matches.get_one::(ARG_TEMPLATE) { + // If no template argument is given, `--tmpdir` is implied. + None => { + let tmpdir = Some(tmpdir.unwrap_or_else(get_tmpdir_env_or_default)); + let template = DEFAULT_TEMPLATE; + (tmpdir, OsString::from(template)) + }, + Some(template) => { + // pi-uutils: TMPDIR comes from the scope environment, not the + // host process environment. + let tmpdir = if let Some(tmpdir) = pi_uutils_ctx::var(TMPDIR_ENV_VAR) + && matches.get_flag(OPT_T) + { + Some(PathBuf::from(tmpdir)) + } else if tmpdir.is_some() { + tmpdir + } else if matches.get_flag(OPT_T) || matches.contains_id(OPT_TMPDIR) { + // If --tmpdir is given without an argument, or -t is given + // export in TMPDIR + Some(env::temp_dir()) + } else { + None + }; + (tmpdir, template.clone()) + }, + }; + Self { + directory: matches.get_flag(OPT_DIRECTORY), + dry_run: matches.get_flag(OPT_DRY_RUN), + quiet: matches.get_flag(OPT_QUIET), + tmpdir, + suffix: matches.get_one::(OPT_SUFFIX).cloned(), + treat_as_template: matches.get_flag(OPT_T), + template, + } + } +} + +/// Parameters that control the path to and name of the temporary file. +/// +/// The temporary file will be created at +/// +/// ```text +/// {directory}/{prefix}{XXX}{suffix} +/// ``` +/// +/// where `{XXX}` is a sequence of random characters whose length is +/// `num_rand_chars`. +struct Params { + /// The directory that will contain the temporary file. + directory: PathBuf, + + /// The (non-random) prefix of the temporary file. + prefix: String, + + /// The number of random characters in the name of the temporary file. + num_rand_chars: usize, + + /// The (non-random) suffix of the temporary file. + suffix: String, +} + +/// Find the start and end indices of the last contiguous block of Xs. +/// +/// If no contiguous block of at least three Xs could be found, this +/// function returns `None`. +/// +/// # Examples +/// +/// ```rust,ignore +/// assert_eq!(find_last_contiguous_block_of_xs("XXX_XXX"), Some((4, 7))); +/// assert_eq!(find_last_contiguous_block_of_xs("aXbXcX"), None); +/// ``` +fn find_last_contiguous_block_of_xs(s: &str) -> Option<(usize, usize)> { + let bytes = s.as_bytes(); + + // Find the index of the last 'X'. + let end = bytes.iter().rposition(|&b| b == b'X')?; + + // Walk left to find the start of the run of Xs that ends at `end`. + let mut start = end; + while start > 0 && bytes[start - 1] == b'X' { + start -= 1; + } + + if end + 1 - start >= 3 { + Some((start, end + 1)) + } else { + None + } +} + +impl Params { + fn from(options: Options) -> Result { + // Convert OsString template to string for processing + // When using -t flag, be permissive with invalid UTF-8 like GNU mktemp + // Otherwise, maintain strict UTF-8 validation (existing behavior) + let template_str = if options.treat_as_template { + // For -t templates, use lossy conversion for GNU compatibility + options.template.to_string_lossy().into_owned() + } else { + // For regular templates, maintain strict validation + match options.template.to_str() { + Some(s) => s.to_string(), + None => { + return Err(MkTempError::InvalidTemplate("template contains invalid UTF-8".into())); + }, + } + }; + + // The template argument must end in 'X' if a suffix option is given. + if options.suffix.is_some() && !template_str.ends_with('X') { + return Err(MkTempError::MustEndInX(template_str.clone())); + } + + // Get the start and end indices of the randomized part of the template. + // + // For example, if the template is "abcXXXXyz", then `i` is 3 and `j` is 7. + let Some((i, j)) = find_last_contiguous_block_of_xs(&template_str) else { + let s = match options.suffix { + // If a suffix is specified, the error message includes the template without the suffix. + Some(_) => template_str + .chars() + .take(template_str.len()) + .collect::(), + None => template_str.clone(), + }; + return Err(MkTempError::TooFewXs(s)); + }; + + // Combine the directory given as an option and the prefix of the template. + // + // For example, if `tmpdir` is "a/b" and the template is "c/dXXX", + // then `prefix` is "a/b/c/d". + let tmpdir = options.tmpdir; + let prefix_from_option = tmpdir.clone().unwrap_or_default(); + let prefix_from_template = &template_str[..i]; + let prefix_path = Path::new(&prefix_from_option).join(prefix_from_template); + if options.treat_as_template && prefix_from_template.contains(MAIN_SEPARATOR) { + return Err(MkTempError::PrefixContainsDirSeparator(template_str.clone())); + } + if tmpdir.is_some() && Path::new(prefix_from_template).is_absolute() { + return Err(MkTempError::InvalidTemplate(template_str.clone().into())); + } + + // Split the parent directory from the file part of the prefix. + // + // For example, if `prefix_path` is "a/b/c/d", then `directory` is + // "a/b/c" and `prefix` gets reassigned to "d". + let (directory, prefix) = { + let prefix_str = prefix_path.to_string_lossy(); + if prefix_str.ends_with(MAIN_SEPARATOR) { + (prefix_path, String::new()) + } else { + let directory = match prefix_path.parent() { + None => PathBuf::new(), + Some(d) => d.to_path_buf(), + }; + let prefix = match prefix_path.file_name() { + None => String::new(), + Some(f) => f.to_string_lossy().to_string(), + }; + (directory, prefix) + } + }; + + // Combine the suffix from the template with the suffix given as an option. + // + // For example, if the suffix command-line argument is ".txt" and + // the template is "XXXabc", then `suffix` is "abc.txt". + let suffix_from_option = options + .suffix + .map(|s| s.to_string_lossy().to_string()) + .unwrap_or_default(); + let suffix_from_template = &template_str[j..]; + let suffix = format!("{suffix_from_template}{suffix_from_option}"); + if suffix.contains(MAIN_SEPARATOR) { + return Err(MkTempError::SuffixContainsDirSeparator(suffix)); + } + + // The number of random characters in the template. + // + // For example, if the template is "abcXXXXyz", then the number of + // random characters is four. + let num_rand_chars = j - i; + + Ok(Self { directory, prefix, num_rand_chars, suffix }) + } +} + +/// Custom parser that converts empty string to `None`, and non-empty string to +/// `Some(PathBuf)`. +/// +/// This parser is used for the `-p` and `--tmpdir` options where an empty +/// string argument should be treated as "not provided", causing mktemp to fall +/// back to using the `$TMPDIR` environment variable or the system's default +/// temporary directory. +/// +/// # Examples +/// +/// - Empty string `""` -> `None` +/// - Non-empty string `"/tmp"` -> `Some(PathBuf::from("/tmp"))` +/// +/// This handles the special case where users can pass an empty directory name +/// to explicitly request fallback behavior. +#[derive(Clone, Debug)] +struct OptionalPathBufParser; + +impl TypedValueParser for OptionalPathBufParser { + type Value = Option; + + fn parse_ref( + &self, + _cmd: &Command, + _arg: Option<&Arg>, + value: &OsStr, + ) -> Result { + if value.is_empty() { + Ok(None) + } else { + Ok(Some(PathBuf::from(value))) + } + } +} + +impl ValueParserFactory for OptionalPathBufParser { + type Parser = Self; + + fn value_parser() -> Self::Parser { + Self + } +} + +/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the +/// arguments directly (without the uucore clap-localization helper that would +/// terminate the process), renders clap help/usage/version to the context +/// streams, and maps the `UResult` to an exit code, so it is safe to run inside +/// the host shell process. +pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(&argv) { + Ok(matches) => matches, + Err(err) => { + // pi-uutils: upstream maps a too-many-values clap error on the + // TEMPLATE argument to the GNU "too many templates" usage error. + if err.kind() == clap::error::ErrorKind::TooManyValues + && err.context().any(|(kind, val)| { + kind == clap::error::ContextKind::InvalidArg + && val == &clap::error::ContextValue::String("[template]".into()) + }) { + let _ = writeln!(pi_uutils_ctx::stderr(), "mktemp: too many templates"); + return 1; + } + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match mktemp_main(&argv, &matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + // pi-uutils: --quiet failures surface as bare exit-code errors + // that render to an empty message; don't emit a dangling + // "mktemp: " prefix. + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "mktemp: {msg}"); + } + if code == 0 { 1 } else { code } + }, + } +} + +fn mktemp_main(args: &[OsString], matches: &ArgMatches) -> UResult<()> { + // Parse command-line options into a format suitable for the + // application logic. + let options = Options::from(matches); + + // pi-uutils: POSIXLY_CORRECT comes from the scope environment. + if pi_uutils_ctx::var("POSIXLY_CORRECT").is_some() { + // If POSIXLY_CORRECT was set, template MUST be the last argument. + if matches.contains_id(ARG_TEMPLATE) { + // Template argument was provided, check if was the last one. + if args.last().unwrap() != &options.template { + return Err(Box::new(MkTempError::TooManyTemplates)); + } + } + } + + let dry_run = options.dry_run; + let suppress_file_err = options.quiet; + let make_dir = options.directory; + + // Parse file path parameters from the command-line options. + let Params { directory: tmpdir, prefix, num_rand_chars: rand, suffix } = Params::from(options)?; + + // Create the temporary file or directory, or simulate creating it. + let res = if dry_run { + Ok(dry_exec(&tmpdir, &prefix, rand, &suffix)) + } else { + exec(&tmpdir, &prefix, rand, &suffix, make_dir) + }; + + let res = if suppress_file_err { + // Mapping all UErrors to ExitCodes prevents the errors from being printed + res.map_err(|e| e.code().into()) + } else { + res + }; + + // pi-uutils: replacement for upstream's `println_verbatim` — writes the + // created path's bytes verbatim to the context stdout instead of the + // process stdout. + let path = res?; + let print = || -> std::io::Result<()> { + let mut out = pi_uutils_ctx::stdout(); + out.write_all(uucore::os_str_as_bytes(path.as_os_str()).map_err(std::io::Error::other)?)?; + out.write_all(b"\n")?; + out.flush() + }; + print().map_err_context(|| "failed to print directory name".to_string())?; + Ok(()) +} + +pub fn uu_app() -> Command { + Command::new("mktemp") + .version(uucore::crate_version!()) + .about("Create a temporary file or directory.") + .override_usage(format_usage("mktemp [OPTION]... [TEMPLATE]")) + .infer_long_args(true) + .arg( + Arg::new(OPT_DIRECTORY) + .short('d') + .long(OPT_DIRECTORY) + .help("Make a directory instead of a file") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_DRY_RUN) + .short('u') + .long(OPT_DRY_RUN) + .help("do not create anything; merely print a name (unsafe)") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_QUIET) + .short('q') + .long("quiet") + .help("Fail silently if an error occurs.") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_SUFFIX) + .long(OPT_SUFFIX) + .help( + "append SUFFIX to TEMPLATE; SUFFIX must not contain a path separator. This option \ + is implied if TEMPLATE does not end with X.", + ) + .value_name("SUFFIX") + .value_parser(clap::value_parser!(OsString)), + ) + .arg( + Arg::new(OPT_P) + .short('p') + .help("short form of --tmpdir") + .value_name("DIR") + .num_args(1) + .value_parser(OptionalPathBufParser) + .value_hint(clap::ValueHint::DirPath), + ) + .arg( + Arg::new(OPT_TMPDIR) + .long(OPT_TMPDIR) + .help( + "interpret TEMPLATE relative to DIR; if DIR is not specified, use $TMPDIR ($TMP on \ + windows) if set, else /tmp. With this option, TEMPLATE must not be an absolute \ + name; unlike with -t, TEMPLATE may contain slashes, but mktemp creates only the \ + final component", + ) + .value_name("DIR") + // Allows use of default argument just by setting --tmpdir. Else, + // use provided input to generate tmpdir + .num_args(0..=1) + // Require an equals to avoid ambiguity if no tmpdir is supplied + .require_equals(true) + .overrides_with(OPT_P) + .value_parser(OptionalPathBufParser) + .value_hint(clap::ValueHint::DirPath), + ) + .arg( + Arg::new(OPT_T) + .short('t') + .help( + "Generate a template (using the supplied prefix and TMPDIR (TMP on windows) if \ + set) to create a filename template [deprecated]", + ) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(ARG_TEMPLATE) + .num_args(..=1) + .value_parser(clap::value_parser!(OsString)), + ) +} + +fn dry_exec(tmpdir: &Path, prefix: &str, rand: usize, suffix: &str) -> PathBuf { + // pi-uutils: resolve the parent directory against the shell working + // directory so the printed candidate matches where creation would occur. + let tmpdir = pi_uutils_ctx::resolve(tmpdir); + let len = prefix.len() + suffix.len() + rand; + let mut buf = Vec::with_capacity(len); + buf.extend(prefix.as_bytes()); + buf.extend(iter::repeat_n(b'X', rand)); + buf.extend(suffix.as_bytes()); + + // Randomize. + let bytes = &mut buf[prefix.len()..prefix.len() + rand]; + SmallRng::try_from_rng(&mut rngs::SysRng) + .unwrap_or_else(|_| { + //rand::rng panics if getrandom failed + SmallRng::seed_from_u64(bytes.as_ptr() as usize as u64) + }) + .fill(bytes); + for byte in bytes { + *byte = match *byte % 62 { + v @ 0..=9 => v + b'0', + v @ 10..=35 => v - 10 + b'a', + v @ 36..=61 => v - 36 + b'A', + _ => unreachable!(), + } + } + // We guarantee utf8. + let buf = String::from_utf8(buf).unwrap(); + tmpdir.join(buf) +} + +/// Create a temporary directory with the given parameters. +/// +/// This function creates a temporary directory as a subdirectory of +/// `dir`. The name of the directory is the concatenation of `prefix`, +/// a string of `rand` random characters, and `suffix`. The +/// permissions of the directory are set to `u+rwx` +/// +/// # Errors +/// +/// If the temporary directory could not be written to disk or if the +/// given directory `dir` does not exist. +fn make_temp_dir(dir: &Path, prefix: &str, rand: usize, suffix: &str) -> UResult { + let mut builder = Builder::new(); + builder.prefix(prefix).rand_bytes(rand).suffix(suffix); + + // On *nix platforms grant read-write-execute for owner only. + // The directory is created with these permission at creation time, using + // mkdir(3) syscall. This is not relevant on Windows systems. See: https://docs.rs/tempfile/latest/tempfile/#security + // `fs` is not imported on Windows anyways. + #[cfg(not(windows))] + builder.permissions(fs::Permissions::from_mode(0o700)); + + match builder.tempdir_in(dir) { + Ok(d) => { + // `keep` consumes the TempDir without removing it + let path = d.keep(); + Ok(path) + }, + Err(e) if e.kind() == ErrorKind::NotFound => { + let filename = format!("{prefix}{}{suffix}", "X".repeat(rand)); + let path = Path::new(dir).join(filename); + Err(MkTempError::NotFound("directory".to_string(), path).into()) + }, + Err(e) => Err(e.into()), + } +} + +/// Create a temporary file with the given parameters. +/// +/// This function creates a temporary file in the directory `dir`. The +/// name of the file is the concatenation of `prefix`, a string of +/// `rand` random characters, and `suffix`. The permissions of the +/// file are set to `u+rw`. +/// +/// # Errors +/// +/// If the file could not be written to disk or if the directory does +/// not exist. +fn make_temp_file(dir: &Path, prefix: &str, rand: usize, suffix: &str) -> UResult { + let mut builder = Builder::new(); + builder.prefix(prefix).rand_bytes(rand).suffix(suffix); + match builder.tempfile_in(dir) { + // `keep` ensures that the file is not deleted + Ok(named_tempfile) => match named_tempfile.keep() { + Ok((_, pathbuf)) => Ok(pathbuf), + Err(e) => Err(MkTempError::PersistError(e.file.path().to_path_buf()).into()), + }, + Err(e) if e.kind() == ErrorKind::NotFound => { + let filename = format!("{prefix}{}{suffix}", "X".repeat(rand)); + let path = Path::new(dir).join(filename); + Err(MkTempError::NotFound("file".to_string(), path).into()) + }, + Err(e) => Err(e.into()), + } +} + +fn exec(dir: &Path, prefix: &str, rand: usize, suffix: &str, make_dir: bool) -> UResult { + // pi-uutils: resolve the parent directory against the shell working + // directory at the creation site; the resolved form is also what gets + // printed, so the printed path is the path actually created. + let dir = pi_uutils_ctx::resolve(dir); + let path = if make_dir { + make_temp_dir(&dir, prefix, rand, suffix)? + } else { + make_temp_file(&dir, prefix, rand, suffix)? + }; + + // Get just the last component of the path to the created + // temporary file or directory. + let filename = path.file_name(); + let filename = filename.unwrap().to_str().unwrap(); + + // Join the directory to the path to get the path to print. + // pi-uutils: unlike upstream (which re-joins the operand as typed), join + // the resolved directory so the printed path names the created entry even + // when the shell cwd differs from the process cwd. + let path = dir.join(filename); + + Ok(path) +} + +/// Reads from `TMPDIR_ENV_VAR` but defaults to /tmp if value is set to empty +/// string. +fn get_tmpdir_env_or_default() -> PathBuf { + // pi-uutils: read TMPDIR from the scope environment; when it is unset + // there, fall back to the host default temp dir as upstream does. + match pi_uutils_ctx::var(TMPDIR_ENV_VAR) { + Some(val) if val.is_empty() => PathBuf::from(FALLBACK_TMPDIR), + Some(val) => PathBuf::from(val), + None => env::temp_dir(), + } +} + +/// Create a temporary file or directory +/// +/// Behavior is determined by the `options` parameter, see [`Options`] for +/// details. +pub fn mktemp(options: &Options) -> UResult { + // Parse file path parameters from the command-line options. + let Params { directory: tmpdir, prefix, num_rand_chars: rand, suffix } = + Params::from(options.clone())?; + + // Create the temporary file or directory, or simulate creating it. + if options.dry_run { + Ok(dry_exec(&tmpdir, &prefix, rand, &suffix)) + } else { + exec(&tmpdir, &prefix, rand, &suffix, options.directory) + } +} + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + fn run_in(cwd: PathBuf, env: HashMap, args: Vec<&str>) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd, + env, + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; + + let argv: Vec = std::iter::once("mktemp") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + /// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var, + /// which would otherwise break printed-path assertions). + fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) { + let dir = tempfile::tempdir().unwrap(); + let canon = std::fs::canonicalize(dir.path()).unwrap(); + (dir, canon) + } + + fn tmpdir_env(dir: &Path) -> HashMap { + HashMap::from([("TMPDIR".to_string(), dir.display().to_string())]) + } + + #[test] + fn default_invocation_creates_file_at_printed_path() { + let (_dir, root) = canonical_tempdir(); + + let (code, stdout, stderr) = run_in(root.clone(), tmpdir_env(&root), vec![]); + assert_eq!(code, 0); + assert_eq!(stderr, ""); + let printed = PathBuf::from(stdout.trim_end_matches('\n')); + assert!(printed.is_file(), "printed path {printed:?} must be a regular file"); + // Scope TMPDIR is honored for the default template. + assert_eq!(printed.parent(), Some(root.as_path())); + assert!( + printed + .file_name() + .unwrap() + .to_str() + .unwrap() + .starts_with("tmp.") + ); + } + + #[test] + fn directory_flag_creates_directory() { + let (_dir, root) = canonical_tempdir(); + + let (code, stdout, stderr) = run_in(root.clone(), tmpdir_env(&root), vec!["-d"]); + assert_eq!(code, 0); + assert_eq!(stderr, ""); + let printed = PathBuf::from(stdout.trim_end_matches('\n')); + assert!(printed.is_dir(), "printed path {printed:?} must be a directory"); + assert_eq!(printed.parent(), Some(root.as_path())); + } + + #[test] + fn relative_tmpdir_resolves_against_scope_cwd() { + let (_dir, root) = canonical_tempdir(); + std::fs::create_dir(root.join("sub")).unwrap(); + + // Relative -p operand + scope cwd differing from the process cwd: only + // the creation-site `pi_uutils_ctx::resolve` patch makes this land in + // the scope cwd's subdir. + let (code, stdout, stderr) = + run_in(root.clone(), HashMap::new(), vec!["-p", "sub", "foo.XXXX"]); + assert_eq!(code, 0); + assert_eq!(stderr, ""); + let printed = PathBuf::from(stdout.trim_end_matches('\n')); + assert!(printed.is_file(), "printed path {printed:?} must exist"); + assert_eq!(printed.parent(), Some(root.join("sub").as_path())); + assert!( + printed + .file_name() + .unwrap() + .to_str() + .unwrap() + .starts_with("foo.") + ); + } + + #[test] + fn too_few_xs_is_an_error() { + let (_dir, root) = canonical_tempdir(); + + let (code, stdout, stderr) = run_in(root, HashMap::new(), vec!["foo.XX"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert_eq!(stderr, "mktemp: too few X's in template 'foo.XX'\n"); + } + + #[test] + fn dry_run_prints_nonexistent_path() { + let (_dir, root) = canonical_tempdir(); + + let (code, stdout, stderr) = run_in(root.clone(), tmpdir_env(&root), vec!["-u"]); + assert_eq!(code, 0); + assert_eq!(stderr, ""); + let printed = PathBuf::from(stdout.trim_end_matches('\n')); + assert_eq!(printed.parent(), Some(root.as_path())); + assert!(!printed.exists(), "dry-run path {printed:?} must not be created"); + } + + #[test] + fn suffix_is_appended_after_random_block() { + let (_dir, root) = canonical_tempdir(); + + let (code, stdout, stderr) = + run_in(root.clone(), HashMap::new(), vec!["--suffix=.txt", "-p", ".", "fooXXXX"]); + assert_eq!(code, 0); + assert_eq!(stderr, ""); + let printed = PathBuf::from(stdout.trim_end_matches('\n')); + assert!(printed.is_file()); + let name = printed.file_name().unwrap().to_str().unwrap().to_string(); + assert!(name.starts_with("foo") && name.ends_with(".txt"), "unexpected name {name}"); + } + + #[test] + fn quiet_suppresses_creation_error_message_but_not_exit_code() { + let (_dir, root) = canonical_tempdir(); + + let (code, stdout, stderr) = + run_in(root, HashMap::new(), vec!["-q", "-p", "missing-dir", "foo.XXXX"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert_eq!(stderr, "", "--quiet must suppress the creation error message"); + } + + #[test] + fn help_renders_to_scope_stdout() { + let (code, stdout, stderr) = run_in(PathBuf::from("."), HashMap::new(), vec!["--help"]); + assert_eq!(code, 0); + assert!(stdout.contains("Usage:")); + assert!(stdout.contains("temporary file or directory")); + assert_eq!(stderr, ""); + } +} diff --git a/crates/vendor/uu-nproc/Cargo.toml b/crates/vendor/uu-nproc/Cargo.toml new file mode 100644 index 000000000..a075046ad --- /dev/null +++ b/crates/vendor/uu-nproc/Cargo.toml @@ -0,0 +1,22 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/nproc), patched to read the +# OMP_NUM_THREADS/OMP_THREAD_LIMIT environment variables from the shell scope +# environment and route I/O through pi-uutils-ctx so it can run in-process as a +# shell builtin. See src/nproc.rs for the patch markers (`pi-uutils:` comments). +[package] +name = "uu_nproc" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "nproc ~ (uutils) display the number of processing units available (vendored + patched for in-process embedding)" + +[lib] +path = "src/nproc.rs" + +[dependencies] +libc = "0.2.172" +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +uucore = { version = "0.8.0", features = ["fs"] } +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[dev-dependencies] +parking_lot = "0.12" diff --git a/crates/vendor/uu-nproc/LICENSE b/crates/vendor/uu-nproc/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-nproc/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-nproc/src/nproc.rs b/crates/vendor/uu-nproc/src/nproc.rs new file mode 100644 index 000000000..5b6cfed2b --- /dev/null +++ b/crates/vendor/uu-nproc/src/nproc.rs @@ -0,0 +1,287 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +// spell-checker:ignore (ToDO) NPROCESSORS nprocs numstr sysconf + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. The OMP_NUM_THREADS and OMP_THREAD_LIMIT environment +// variables are read from the scope environment via `pi_uutils_ctx::var` (the +// shell's exported variables), not the host process environment. All output is +// routed through the context stdout, `translate!` strings are literalized, and +// the entry point no longer calls `std::process::exit`. + +use std::{ffi::OsString, io::Write, thread}; + +use clap::{Arg, ArgAction, ArgMatches, Command}; +use pi_uutils_ctx::format_usage; +use uucore::{ + display::Quotable, + error::{UResult, USimpleError}, +}; + +static OPT_ALL: &str = "all"; +static OPT_IGNORE: &str = "ignore"; + +/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the +/// arguments directly (without the uucore clap-localization helper that would +/// terminate the process), renders clap help/usage/version to the context +/// streams, and maps the `UResult` to an exit code, so it is safe to run inside +/// the host shell process. +pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match nproc_main(&matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "nproc: {msg}"); + } + if code == 0 { 1 } else { code } + }, + } +} + +fn nproc_main(matches: &ArgMatches) -> UResult<()> { + let ignore = match matches.get_one::(OPT_IGNORE) { + Some(numstr) => match numstr.trim().parse::() { + Ok(num) => num, + Err(e) => { + return Err(USimpleError::new( + 1, + // pi-uutils: literalized translate!("nproc-error-invalid-number") + format!("{} is not a valid number: {e}", numstr.quote()), + )); + }, + }, + None => 0, + }; + + // pi-uutils: OMP_THREAD_LIMIT comes from the scope environment (the + // shell's exported variables), not the host process environment. + let limit = match pi_uutils_ctx::var("OMP_THREAD_LIMIT") { + // Uses the OpenMP variable to limit the number of threads + // If the parsing fails, returns the max size (so, no impact) + // If OMP_THREAD_LIMIT=0, rejects the value + Some(threads) => match threads.parse() { + Ok(0) | Err(_) => usize::MAX, + Ok(n) => n, + }, + // the variable 'OMP_THREAD_LIMIT' doesn't exist + // fallback to the max + None => usize::MAX, + }; + + let mut cores = if matches.get_flag(OPT_ALL) { + num_cpus_all() + } else { + // OMP_NUM_THREADS doesn't have an impact on --all + // pi-uutils: OMP_NUM_THREADS comes from the scope environment. + match pi_uutils_ctx::var("OMP_NUM_THREADS") { + // Uses the OpenMP variable to force the number of threads + // If the parsing fails, returns the number of CPU + Some(threads) => { + // In some cases, OMP_NUM_THREADS can be "x,y,z" + // In this case, only take the first one (like GNU) + // If OMP_NUM_THREADS=0, rejects the value + match threads.split_terminator(',').next() { + None => available_parallelism(), + Some(s) => match s.trim().parse() { + Ok(0) | Err(_) => available_parallelism(), + Ok(n) => n, + }, + } + }, + // the variable 'OMP_NUM_THREADS' doesn't exist + // fallback to the regular CPU detection + None => available_parallelism(), + } + }; + + cores = std::cmp::min(limit, cores); + if cores <= ignore { + cores = 1; + } else { + cores -= ignore; + } + // pi-uutils: write to the context stdout instead of the process stdout. + pi_uutils_ctx::stdout() + .write_all(format!("{cores}\n").as_bytes()) + .map_err(|e| USimpleError::new(1, e.to_string()))?; + Ok(()) +} + +pub fn uu_app() -> Command { + Command::new("nproc") + .version(uucore::crate_version!()) + .about( + "Print the number of cores available to the current process.\nIf the OMP_NUM_THREADS or \ + OMP_THREAD_LIMIT environment variables are set, then\nthey will determine the minimum \ + and maximum returned value respectively.", + ) + .override_usage(format_usage("nproc [OPTIONS]...")) + .infer_long_args(true) + .arg( + Arg::new(OPT_ALL) + .long(OPT_ALL) + .help("print the number of cores available to the system") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_IGNORE) + .long(OPT_IGNORE) + .value_name("N") + .help("ignore up to N cores"), + ) +} + +#[cfg(unix)] +fn num_cpus_all() -> usize { + // In some situation, /proc and /sys are not mounted, and sysconf returns 1. + // However, we want to guarantee that `nproc --all` >= `nproc`. + unsafe { libc::sysconf(libc::_SC_NPROCESSORS_CONF) } + .try_into() + .ok() + .filter(|&n: &isize| n > 1) + .map_or_else(available_parallelism, |n| n as usize) +} + +// Other platforms (e.g., windows), available_parallelism() directly. +#[cfg(not(unix))] +fn num_cpus_all() -> usize { + available_parallelism() +} + +/// In some cases, [`thread::available_parallelism`]() may return an Err +/// In this case, we will return 1 (like GNU) +fn available_parallelism() -> usize { + thread::available_parallelism().map_or(1, std::num::NonZeroUsize::get) +} + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + fn run_in(env: HashMap, args: Vec<&str>) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd: PathBuf::from("."), + env, + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; + + let argv: Vec = std::iter::once("nproc") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + #[test] + fn scope_env_omp_num_threads_forces_count() { + let env = HashMap::from([("OMP_NUM_THREADS".to_string(), "3".to_string())]); + let (code, stdout, stderr) = run_in(env, vec![]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "3\n", "")); + } + + #[test] + fn omp_thread_limit_caps_omp_num_threads() { + let env = HashMap::from([ + ("OMP_NUM_THREADS".to_string(), "64".to_string()), + ("OMP_THREAD_LIMIT".to_string(), "2".to_string()), + ]); + let (code, stdout, stderr) = run_in(env, vec![]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "2\n", "")); + } + + #[test] + fn all_prints_positive_integer_and_ignores_omp_num_threads() { + // --all reports hardware CPUs; OMP_NUM_THREADS must not force it. + let env = HashMap::from([("OMP_NUM_THREADS".to_string(), "0".to_string())]); + let (code, stdout, stderr) = run_in(env, vec!["--all"]); + assert_eq!(code, 0); + assert_eq!(stderr, ""); + let n: usize = stdout + .trim_end() + .parse() + .expect("--all output is an integer"); + assert!(n >= 1); + } + + #[test] + fn process_environment_is_not_consulted() { + // The variable exists only in the host process environment, not the + // scope map: only the un-patched `std::env::var` path would see it. + unsafe { std::env::set_var("OMP_NUM_THREADS", "1234") }; + let (code, stdout, stderr) = run_in(HashMap::new(), vec![]); + unsafe { std::env::remove_var("OMP_NUM_THREADS") }; + assert_eq!((code, stderr.as_str()), (0, "")); + assert_ne!(stdout, "1234\n"); + let n: usize = stdout.trim_end().parse().expect("output is an integer"); + assert!(n >= 1); + } + + #[test] + fn ignore_subtracts_and_floors_at_one() { + let env = HashMap::from([("OMP_NUM_THREADS".to_string(), "8".to_string())]); + let (code, stdout, _) = run_in(env, vec!["--ignore=3"]); + assert_eq!((code, stdout.as_str()), (0, "5\n")); + + let env = HashMap::from([("OMP_NUM_THREADS".to_string(), "2".to_string())]); + let (code, stdout, _) = run_in(env, vec!["--ignore=5"]); + assert_eq!((code, stdout.as_str()), (0, "1\n")); + } + + #[test] + fn invalid_ignore_value_is_an_error() { + let (code, stdout, stderr) = run_in(HashMap::new(), vec!["--ignore=bogus"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert!(stderr.contains("is not a valid number"), "stderr: {stderr}"); + } +} diff --git a/crates/vendor/uu-printenv/Cargo.toml b/crates/vendor/uu-printenv/Cargo.toml new file mode 100644 index 000000000..db7e8c291 --- /dev/null +++ b/crates/vendor/uu-printenv/Cargo.toml @@ -0,0 +1,21 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/printenv), patched to read +# the environment from the shell scope and route I/O through pi-uutils-ctx so +# it can run in-process as a shell builtin. See src/printenv.rs for the patch +# markers (`pi-uutils:` comments). +[package] +name = "uu_printenv" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "printenv ~ (uutils) display value of environment VAR (vendored + patched for in-process embedding)" + +[lib] +path = "src/printenv.rs" + +[dependencies] +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +uucore = { version = "0.8.0" } +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[dev-dependencies] +parking_lot = "0.12" diff --git a/crates/vendor/uu-printenv/LICENSE b/crates/vendor/uu-printenv/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-printenv/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-printenv/src/printenv.rs b/crates/vendor/uu-printenv/src/printenv.rs new file mode 100644 index 000000000..5e0eac263 --- /dev/null +++ b/crates/vendor/uu-printenv/src/printenv.rs @@ -0,0 +1,243 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. The environment comes from the SCOPE, not the process: +// the no-argument dump iterates `pi_uutils_ctx::env_snapshot()` and named +// lookups go through `pi_uutils_ctx::var`, because the embedding shell's +// exported variables are not present in the host process environment. All +// output is routed through the context stdout, `translate!` strings are +// literalized, and the entry point no longer calls `std::process::exit`. + +use std::{ffi::OsString, io::Write}; + +use clap::{Arg, ArgAction, ArgMatches, Command}; +use pi_uutils_ctx::format_usage; +use uucore::{error::UResult, line_ending::LineEnding}; + +static OPT_NULL: &str = "null"; + +static ARG_VARIABLES: &str = "variables"; + +/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the +/// arguments directly (without the uucore clap-localization helper that would +/// terminate the process), renders clap help/usage/version to the context +/// streams, and maps the `UResult` to an exit code, so it is safe to run inside +/// the host shell process. +pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match printenv_main(&matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + // pi-uutils: unset-variable failures surface as bare exit-code + // errors that render to an empty message (upstream prints nothing + // for them); don't emit a dangling "printenv: " prefix. + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "printenv: {msg}"); + } + if code == 0 { 1 } else { code } + }, + } +} + +fn printenv_main(matches: &ArgMatches) -> UResult<()> { + let variables: Vec = matches + .get_many::(ARG_VARIABLES) + .map(|v| v.map(ToString::to_string).collect()) + .unwrap_or_default(); + + let separator = LineEnding::from_zero_flag(matches.get_flag(OPT_NULL)); + + if variables.is_empty() { + // pi-uutils: replacement for `uucore::display::print_all_env_vars` — + // dumps the scope environment map to the context stdout. + let mut stdout = pi_uutils_ctx::stdout(); + for (key, value) in pi_uutils_ctx::env_snapshot() { + write!(stdout, "{key}={value}{separator}")?; + } + stdout.flush()?; + return Ok(()); + } + + let mut error_found = false; + for env_var in variables { + // we silently ignore a=b as variable but we trigger an error + if env_var.contains('=') { + error_found = true; + continue; + } + // pi-uutils: look the variable up in the scope environment (upstream + // uses `std::env::var_os`) and write it to the context stdout. + if let Some(var) = pi_uutils_ctx::var(&env_var) { + let mut stdout = pi_uutils_ctx::stdout(); + write!(stdout, "{var}{separator}")?; + stdout.flush()?; + } else { + error_found = true; + } + } + + if error_found { Err(1.into()) } else { Ok(()) } +} + +pub fn uu_app() -> Command { + Command::new("printenv") + .version(uucore::crate_version!()) + .about( + "Display the values of the specified environment VARIABLE(s), or (with no VARIABLE) \ + display name and value pairs for them all.", + ) + .override_usage(format_usage("printenv [OPTION]... [VARIABLE]...")) + .infer_long_args(true) + .arg( + Arg::new(OPT_NULL) + .short('0') + .long(OPT_NULL) + .help("end each output line with 0 byte rather than newline") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(ARG_VARIABLES) + .action(ArgAction::Append) + .num_args(1..), + ) +} + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, io::Write, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + fn run_with_env(env: HashMap, args: Vec<&str>) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd: std::path::PathBuf::from("."), + env, + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; + + let argv: Vec = std::iter::once("printenv") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + fn scope_env() -> HashMap { + HashMap::from([ + ("FOO".to_string(), "bar".to_string()), + ("BAZ".to_string(), "qux".to_string()), + ]) + } + + #[test] + fn named_variable_prints_scope_value() { + let (code, stdout, stderr) = run_with_env(scope_env(), vec!["FOO"]); + assert_eq!(code, 0); + assert_eq!(stdout, "bar\n"); + assert_eq!(stderr, ""); + } + + #[test] + fn unset_variable_is_silent_failure() { + let (code, stdout, stderr) = run_with_env(scope_env(), vec!["NOPE"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert_eq!(stderr, "", "unset variables fail without a message"); + } + + #[test] + fn mixed_set_and_unset_prints_set_ones_and_fails() { + let (code, stdout, stderr) = run_with_env(scope_env(), vec!["FOO", "NOPE", "BAZ"]); + assert_eq!(code, 1); + assert_eq!(stdout, "bar\nqux\n"); + assert_eq!(stderr, ""); + } + + #[test] + fn no_args_dumps_scope_env_not_process_env() { + // The host process certainly has PATH set; the scope env deliberately + // does not, so its absence proves the dump reads the scope map. + assert!(std::env::var_os("PATH").is_some()); + let (code, stdout, stderr) = run_with_env(scope_env(), vec![]); + assert_eq!(code, 0); + assert_eq!(stderr, ""); + let lines: Vec<&str> = stdout.lines().collect(); + assert_eq!(lines.len(), 2); + assert!(lines.contains(&"FOO=bar")); + assert!(lines.contains(&"BAZ=qux")); + assert!(!lines.iter().any(|l| l.starts_with("PATH="))); + } + + #[test] + fn null_flag_terminates_with_nul() { + let (code, stdout, _) = run_with_env(scope_env(), vec!["-0", "FOO"]); + assert_eq!((code, stdout.as_str()), (0, "bar\0")); + + let (code, stdout, _) = run_with_env(scope_env(), vec!["--null", "FOO", "BAZ"]); + assert_eq!((code, stdout.as_str()), (0, "bar\0qux\0")); + } + + #[test] + fn name_containing_equals_is_ignored_but_fails() { + let (code, stdout, stderr) = run_with_env(scope_env(), vec!["FOO=bar", "BAZ"]); + assert_eq!(code, 1); + assert_eq!(stdout, "qux\n"); + assert_eq!(stderr, ""); + } + + #[test] + fn help_renders_to_scope_stdout() { + let (code, stdout, stderr) = run_with_env(HashMap::new(), vec!["--help"]); + assert_eq!(code, 0); + assert!(stdout.contains("Usage:")); + assert!(stdout.contains("environment VARIABLE")); + assert_eq!(stderr, ""); + } +} diff --git a/crates/vendor/uu-readlink/Cargo.toml b/crates/vendor/uu-readlink/Cargo.toml new file mode 100644 index 000000000..e54321c8b --- /dev/null +++ b/crates/vendor/uu-readlink/Cargo.toml @@ -0,0 +1,22 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/readlink), patched to resolve +# path arguments against the shell working directory and route I/O through +# pi-uutils-ctx so it can run in-process as a shell builtin. See src/readlink.rs +# for the patch markers (`pi-uutils:` comments). +[package] +name = "uu_readlink" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "readlink ~ (uutils) display resolved path of PATHNAME (vendored + patched for in-process embedding)" + +[lib] +path = "src/readlink.rs" + +[dependencies] +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +uucore = { version = "0.8.0", features = ["fs"] } +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[dev-dependencies] +parking_lot = "0.12" +tempfile = "3" diff --git a/crates/vendor/uu-readlink/LICENSE b/crates/vendor/uu-readlink/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-readlink/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-readlink/src/readlink.rs b/crates/vendor/uu-readlink/src/readlink.rs new file mode 100644 index 000000000..813d4a458 --- /dev/null +++ b/crates/vendor/uu-readlink/src/readlink.rs @@ -0,0 +1,422 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +// spell-checker:ignore (ToDO) errno + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. Every filesystem syscall resolves its path operand +// against the shell working directory via `pi_uutils_ctx::resolve` AT THE CALL +// SITE, while the original operands are kept for display/error messages (GNU +// prints operands as typed). All process-global stdio is routed through +// `pi_uutils_ctx`, `translate!` strings are literalized, POSIXLY_CORRECT is +// read from the scope environment, and the entry point no longer calls +// `std::process::exit`. + +use std::{ + ffi::OsString, + fs, + io::Write, + path::{Path, PathBuf}, +}; + +use clap::{Arg, ArgAction, ArgMatches, Command}; +use pi_uutils_ctx::format_usage; +use uucore::{ + display::Quotable, + error::{FromIo, UResult, UUsageError}, + fs::{MissingHandling, ResolveMode, canonicalize}, + libc::EINVAL, + line_ending::LineEnding, +}; + +const OPT_CANONICALIZE: &str = "canonicalize"; +const OPT_CANONICALIZE_MISSING: &str = "canonicalize-missing"; +const OPT_CANONICALIZE_EXISTING: &str = "canonicalize-existing"; +const OPT_NO_NEWLINE: &str = "no-newline"; +const OPT_QUIET: &str = "quiet"; +const OPT_SILENT: &str = "silent"; +const OPT_VERBOSE: &str = "verbose"; +const OPT_ZERO: &str = "zero"; + +const ARG_FILES: &str = "files"; + +/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the +/// arguments directly (without the uucore clap-localization helper that would +/// terminate the process), renders clap help/usage/version to the context +/// streams, and maps the `UResult` to an exit code, so it is safe to run inside +/// the host shell process. +pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match readlink_main(&matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + // pi-uutils: silent failures surface as bare exit-code errors that + // render to an empty message (upstream prints nothing for them); + // don't emit a dangling "readlink: " prefix. + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "readlink: {msg}"); + } + if code == 0 { 1 } else { code } + }, + } +} + +fn readlink_main(matches: &ArgMatches) -> UResult<()> { + let mut no_trailing_delimiter = matches.get_flag(OPT_NO_NEWLINE); + let use_zero = matches.get_flag(OPT_ZERO); + // pi-uutils: POSIXLY_CORRECT comes from the scope environment (the shell's + // exported variables), not the host process environment. + let verbose = matches.get_flag(OPT_VERBOSE) || pi_uutils_ctx::var("POSIXLY_CORRECT").is_some(); + + // GNU readlink -f/-e/-m follows symlinks first and then applies `..` (physical + // resolution). ResolveMode::Logical collapses `..` before following links, + // which yields the opposite order, so we choose Physical here for GNU + // compatibility. + let res_mode = if matches.get_flag(OPT_CANONICALIZE) + || matches.get_flag(OPT_CANONICALIZE_EXISTING) + || matches.get_flag(OPT_CANONICALIZE_MISSING) + { + ResolveMode::Physical + } else { + ResolveMode::None + }; + + let can_mode = if matches.get_flag(OPT_CANONICALIZE_EXISTING) { + MissingHandling::Existing + } else if matches.get_flag(OPT_CANONICALIZE_MISSING) { + MissingHandling::Missing + } else { + MissingHandling::Normal + }; + + let files: Vec = matches + .get_many::(ARG_FILES) + .map(|v| v.map(PathBuf::from).collect()) + .unwrap_or_default(); + + if files.is_empty() { + return Err(UUsageError::new(1, "missing operand".to_string())); + } + + if no_trailing_delimiter && files.len() > 1 { + let _ = writeln!( + pi_uutils_ctx::stderr(), + "readlink: ignoring --no-newline with multiple arguments" + ); + no_trailing_delimiter = false; + } + + let line_ending = if no_trailing_delimiter { + None + } else { + Some(LineEnding::from_zero_flag(use_zero)) + }; + + for p in &files { + // pi-uutils: resolve the operand against the shell working directory; + // `p` is kept for display. Resolving before `canonicalize` also keeps + // uucore's internal `env::current_dir()` fallback from being consulted. + let resolved = pi_uutils_ctx::resolve(p); + let path_result = if res_mode == ResolveMode::None { + fs::read_link(&resolved) + } else { + canonicalize(&resolved, can_mode, res_mode) + }; + + match path_result { + Ok(path) => { + show(&path, line_ending)?; + }, + Err(err) => { + if !verbose { + return Err(1.into()); + } + + let message = if err.raw_os_error() == Some(EINVAL) { + format!("{}: Invalid argument", p.maybe_quote()) + } else { + err.map_err_context(|| p.maybe_quote().to_string()) + .to_string() + }; + let _ = writeln!(pi_uutils_ctx::stderr(), "readlink: {message}"); + return Err(1.into()); + }, + } + } + Ok(()) +} + +pub fn uu_app() -> Command { + Command::new("readlink") + .version(uucore::crate_version!()) + .about("Print value of a symbolic link or canonical file name.") + .override_usage(format_usage("readlink [OPTION]... [FILE]...")) + .infer_long_args(true) + .arg( + Arg::new(OPT_CANONICALIZE) + .short('f') + .long(OPT_CANONICALIZE) + .help( + "canonicalize by following every symlink in every component of the given name \ + recursively; all but the last component must exist", + ) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_CANONICALIZE_EXISTING) + .short('e') + .long("canonicalize-existing") + .help( + "canonicalize by following every symlink in every component of the given name \ + recursively, all components must exist", + ) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_CANONICALIZE_MISSING) + .short('m') + .long(OPT_CANONICALIZE_MISSING) + .help( + "canonicalize by following every symlink in every component of the given name \ + recursively, without requirements on components existence", + ) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_NO_NEWLINE) + .short('n') + .long(OPT_NO_NEWLINE) + .help("do not output the trailing delimiter") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_QUIET) + .short('q') + .long(OPT_QUIET) + .help("suppress most error messages") + .overrides_with_all([OPT_QUIET, OPT_SILENT, OPT_VERBOSE]) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_SILENT) + .short('s') + .long(OPT_SILENT) + .help("suppress most error messages") + .overrides_with_all([OPT_QUIET, OPT_SILENT, OPT_VERBOSE]) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_VERBOSE) + .short('v') + .long(OPT_VERBOSE) + .help("report error message") + .overrides_with_all([OPT_QUIET, OPT_SILENT, OPT_VERBOSE]) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_ZERO) + .short('z') + .long(OPT_ZERO) + .help("separate output with NUL rather than newline") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(ARG_FILES) + .action(ArgAction::Append) + .value_parser(clap::value_parser!(OsString)) + .value_hint(clap::ValueHint::AnyPath), + ) +} + +/// pi-uutils: replacement for upstream's `show` — writes the resolved path +/// bytes verbatim to the context stdout instead of the process stdout. +fn show(path: &Path, line_ending: Option) -> UResult<()> { + let mut out = pi_uutils_ctx::stdout(); + out.write_all(uucore::os_str_as_bytes(path.as_os_str())?)?; + if let Some(line_ending) = line_ending { + write!(out, "{line_ending}")?; + } + out.flush()?; + Ok(()) +} + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + fn run_in(cwd: PathBuf, args: Vec<&str>) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd, + env: HashMap::new(), + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; + + let argv: Vec = std::iter::once("readlink") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + /// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var, + /// which -f/-e/-m resolution would otherwise expand mid-assertion). + fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) { + let dir = tempfile::tempdir().unwrap(); + let canon = fs::canonicalize(dir.path()).unwrap(); + (dir, canon) + } + + #[cfg(unix)] + #[test] + fn resolves_relative_operand_against_scope_cwd() { + let (_dir, root) = canonical_tempdir(); + std::os::unix::fs::symlink("target-file", root.join("link")).unwrap(); + + // Relative operand + scope cwd differing from the process cwd: only the + // call-site `pi_uutils_ctx::resolve` patch makes this find the link. + let (code, stdout, stderr) = run_in(root, vec!["link"]); + assert_eq!(code, 0); + assert_eq!(stdout, "target-file\n"); + assert_eq!(stderr, ""); + } + + #[cfg(unix)] + #[test] + fn canonicalize_follows_symlink_to_absolute_path() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("target"), b"x").unwrap(); + std::os::unix::fs::symlink("target", root.join("link")).unwrap(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["-f", "link"]); + assert_eq!(code, 0); + assert_eq!(stdout, format!("{}\n", root.join("target").display())); + assert_eq!(stderr, ""); + } + + #[test] + fn canonicalize_missing_builds_path_from_scope_cwd() { + let (_dir, root) = canonical_tempdir(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["-m", "missing/sub"]); + assert_eq!(code, 0); + assert_eq!(stdout, format!("{}\n", root.join("missing").join("sub").display())); + assert_eq!(stderr, ""); + } + + #[cfg(unix)] + #[test] + fn canonicalize_existing_fails_silently_on_missing_final_component() { + let (_dir, root) = canonical_tempdir(); + + let (code, stdout, stderr) = run_in(root, vec!["-e", "missing"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert_eq!(stderr, "", "non-verbose failures print nothing"); + } + + #[cfg(unix)] + #[test] + fn non_symlink_is_silent_failure_by_default_and_einval_with_verbose() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("plain"), b"x").unwrap(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["plain"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (1, "", "")); + + let (code, stdout, stderr) = run_in(root, vec!["-v", "plain"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert_eq!(stderr, "readlink: plain: Invalid argument\n"); + } + + #[cfg(unix)] + #[test] + fn no_newline_with_multiple_args_warns_and_keeps_delimiter() { + let (_dir, root) = canonical_tempdir(); + std::os::unix::fs::symlink("a", root.join("l1")).unwrap(); + std::os::unix::fs::symlink("b", root.join("l2")).unwrap(); + + let (code, stdout, stderr) = run_in(root, vec!["-n", "l1", "l2"]); + assert_eq!(code, 0); + assert_eq!(stdout, "a\nb\n"); + assert_eq!(stderr, "readlink: ignoring --no-newline with multiple arguments\n"); + } + + #[cfg(unix)] + #[test] + fn zero_terminates_with_nul_and_no_newline_drops_delimiter() { + let (_dir, root) = canonical_tempdir(); + std::os::unix::fs::symlink("a", root.join("l1")).unwrap(); + + let (code, stdout, _) = run_in(root.clone(), vec!["-z", "l1"]); + assert_eq!((code, stdout.as_str()), (0, "a\0")); + + let (code, stdout, _) = run_in(root, vec!["-n", "l1"]); + assert_eq!((code, stdout.as_str()), (0, "a")); + } + + #[test] + fn missing_operand_is_usage_error() { + let (code, stdout, stderr) = run_in(PathBuf::from("."), vec![]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert!(stderr.contains("missing operand")); + } + + #[test] + fn help_renders_to_scope_stdout() { + let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["--help"]); + assert_eq!(code, 0); + assert!(stdout.contains("Usage:")); + assert!(stdout.contains("canonical file name")); + assert_eq!(stderr, ""); + } +} diff --git a/crates/vendor/uu-realpath/Cargo.toml b/crates/vendor/uu-realpath/Cargo.toml new file mode 100644 index 000000000..0a3954649 --- /dev/null +++ b/crates/vendor/uu-realpath/Cargo.toml @@ -0,0 +1,22 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/realpath), patched to resolve +# path arguments against the shell working directory and route I/O through +# pi-uutils-ctx so it can run in-process as a shell builtin. See src/realpath.rs +# for the patch markers (`pi-uutils:` comments). +[package] +name = "uu_realpath" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "realpath ~ (uutils) display resolved absolute path of PATHNAME (vendored + patched for in-process embedding)" + +[lib] +path = "src/realpath.rs" + +[dependencies] +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +uucore = { version = "0.8.0", features = ["fs"] } +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[dev-dependencies] +parking_lot = "0.12" +tempfile = "3" diff --git a/crates/vendor/uu-realpath/LICENSE b/crates/vendor/uu-realpath/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-realpath/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-realpath/src/realpath.rs b/crates/vendor/uu-realpath/src/realpath.rs new file mode 100644 index 000000000..061f67dfe --- /dev/null +++ b/crates/vendor/uu-realpath/src/realpath.rs @@ -0,0 +1,568 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +// spell-checker:ignore (ToDO) retcode + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. Every filesystem syscall resolves its path operand +// against the shell working directory via `pi_uutils_ctx::resolve` AT THE CALL +// SITE (FILE operands and the --relative-to/--relative-base option paths), +// while the original operands are kept for display/error messages (GNU prints +// operands as typed). All process-global stdio is routed through +// `pi_uutils_ctx`, `translate!` strings are literalized, `show_if_err!` is +// replaced by a context-stderr write plus `pi_uutils_ctx::set_exit_code`, and +// the entry point no longer calls `std::process::exit`. + +use std::{ + ffi::{OsStr, OsString}, + io::Write, + path::{Path, PathBuf}, +}; + +use clap::{ + Arg, ArgAction, ArgMatches, Command, + builder::{TypedValueParser, ValueParserFactory}, +}; +use pi_uutils_ctx::format_usage; +use uucore::{ + display::Quotable, + error::{FromIo, UResult}, + fs::{MissingHandling, ResolveMode, canonicalize, make_path_relative_to}, + line_ending::LineEnding, +}; + +const OPT_QUIET: &str = "quiet"; +const OPT_STRIP: &str = "strip"; +const OPT_ZERO: &str = "zero"; +const OPT_PHYSICAL: &str = "physical"; +const OPT_LOGICAL: &str = "logical"; +const OPT_CANONICALIZE_MISSING: &str = "canonicalize-missing"; +const OPT_CANONICALIZE: &str = "canonicalize"; +const OPT_CANONICALIZE_EXISTING: &str = "canonicalize-existing"; +const OPT_RELATIVE_TO: &str = "relative-to"; +const OPT_RELATIVE_BASE: &str = "relative-base"; + +const ARG_FILES: &str = "files"; + +/// Custom parser that validates `OsString` is not empty +#[derive(Clone, Debug)] +struct NonEmptyOsStringParser; + +impl TypedValueParser for NonEmptyOsStringParser { + type Value = OsString; + + fn parse_ref( + &self, + _cmd: &Command, + _arg: Option<&Arg>, + value: &OsStr, + ) -> Result { + if value.is_empty() { + let mut err = clap::Error::new(clap::error::ErrorKind::ValueValidation); + err.insert( + clap::error::ContextKind::Custom, + // pi-uutils: literalized `translate!("realpath-invalid-empty-operand")` + clap::error::ContextValue::String("invalid operand: empty string".to_string()), + ); + return Err(err); + } + Ok(value.to_os_string()) + } +} + +impl ValueParserFactory for NonEmptyOsStringParser { + type Parser = Self; + + fn value_parser() -> Self::Parser { + Self + } +} + +/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the +/// arguments directly (without the uucore clap-localization helper that would +/// terminate the process), renders clap help/usage/version to the context +/// streams, and maps the `UResult` to an exit code, so it is safe to run inside +/// the host shell process. +pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match realpath_main(&matches) { + // pi-uutils: per-file failures accumulate their exit code via + // `pi_uutils_ctx::set_exit_code` (upstream's `show!` machinery). + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "realpath: {msg}"); + } + if code == 0 { 1 } else { code } + }, + } +} + +fn realpath_main(matches: &ArgMatches) -> UResult<()> { + /* the list of files */ + + let paths: Vec = matches + .get_many::(ARG_FILES) + .unwrap() + .map(PathBuf::from) + .collect(); + + let strip = matches.get_flag(OPT_STRIP); + let line_ending = LineEnding::from_zero_flag(matches.get_flag(OPT_ZERO)); + let quiet = matches.get_flag(OPT_QUIET); + let logical = matches.get_flag(OPT_LOGICAL); + let can_mode = if matches.get_flag(OPT_CANONICALIZE_MISSING) { + MissingHandling::Missing + } else if matches.get_flag(OPT_CANONICALIZE_EXISTING) { + // -e: all components must exist + // Despite the name, MissingHandling::Existing requires all components to exist + MissingHandling::Existing + } else { + // Default behavior (same as -E): all but last component must exist + // MissingHandling::Normal allows the final component to not exist + MissingHandling::Normal + }; + let resolve_mode = if strip { + ResolveMode::None + } else if logical { + ResolveMode::Logical + } else { + ResolveMode::Physical + }; + let (relative_to, relative_base) = prepare_relative_options(matches, can_mode, resolve_mode)?; + for path in &paths { + let result = resolve_path( + path, + line_ending, + resolve_mode, + can_mode, + relative_to.as_deref(), + relative_base.as_deref(), + ); + if !quiet { + // pi-uutils: replacement for `show_if_err!` — report the error on + // the context stderr and record the exit code, then keep + // processing the remaining operands (upstream continue semantics). + if let Err(err) = result.map_err_context(|| path.maybe_quote().to_string()) { + let _ = writeln!(pi_uutils_ctx::stderr(), "realpath: {err}"); + pi_uutils_ctx::set_exit_code(err.code()); + } + } + } + // Although we return `Ok`, it is possible that a call to + // `show!()` above has set the exit code for the program to a + // non-zero integer. + Ok(()) +} + +pub fn uu_app() -> Command { + Command::new("realpath") + .version(uucore::crate_version!()) + .about("Print the resolved path") + .override_usage(format_usage("realpath [OPTION]... FILE...")) + .infer_long_args(true) + .arg( + Arg::new(OPT_QUIET) + .short('q') + .long(OPT_QUIET) + .help("Do not print warnings for invalid paths") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_STRIP) + .short('s') + .long(OPT_STRIP) + .visible_alias("no-symlinks") + .help("Only strip '.' and '..' components, but don't resolve symbolic links") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_ZERO) + .short('z') + .long(OPT_ZERO) + .help("Separate output filenames with \\0 rather than newline") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_LOGICAL) + .short('L') + .long(OPT_LOGICAL) + .help("resolve '..' components before symlinks") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_PHYSICAL) + .short('P') + .long(OPT_PHYSICAL) + .overrides_with_all([OPT_STRIP, OPT_LOGICAL]) + .help("resolve symlinks as encountered (default)") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_CANONICALIZE) + .short('E') + .long(OPT_CANONICALIZE) + .overrides_with_all([OPT_CANONICALIZE_EXISTING, OPT_CANONICALIZE_MISSING]) + .help("all but the last component must exist (default)") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_CANONICALIZE_EXISTING) + .short('e') + .long(OPT_CANONICALIZE_EXISTING) + .overrides_with_all([OPT_CANONICALIZE, OPT_CANONICALIZE_MISSING]) + .help( + "canonicalize by following every symlink in every component of the given name \ + recursively, all components must exist", + ) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_CANONICALIZE_MISSING) + .short('m') + .long(OPT_CANONICALIZE_MISSING) + .overrides_with_all([OPT_CANONICALIZE, OPT_CANONICALIZE_EXISTING]) + .help( + "canonicalize by following every symlink in every component of the given name \ + recursively, without requirements on components existence", + ) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_RELATIVE_TO) + .long(OPT_RELATIVE_TO) + .value_name("DIR") + .value_parser(NonEmptyOsStringParser) + .help("print the resolved path relative to DIR"), + ) + .arg( + Arg::new(OPT_RELATIVE_BASE) + .long(OPT_RELATIVE_BASE) + .value_name("DIR") + .value_parser(NonEmptyOsStringParser) + .help("print absolute paths unless paths below DIR"), + ) + .arg( + Arg::new(ARG_FILES) + .action(ArgAction::Append) + .required(true) + .value_parser(NonEmptyOsStringParser) + .value_hint(clap::ValueHint::AnyPath), + ) +} + +/// Prepare `--relative-to` and `--relative-base` options. +/// Convert them to their absolute values. +/// Check if `--relative-to` is a descendant of `--relative-base`, +/// otherwise nullify their value. +fn prepare_relative_options( + matches: &ArgMatches, + can_mode: MissingHandling, + resolve_mode: ResolveMode, +) -> UResult<(Option, Option)> { + let relative_to = matches + .get_one::(OPT_RELATIVE_TO) + .map(PathBuf::from); + let relative_base = matches + .get_one::(OPT_RELATIVE_BASE) + .map(PathBuf::from); + let relative_to = canonicalize_relative_option(relative_to, can_mode, resolve_mode)?; + let relative_base = canonicalize_relative_option(relative_base, can_mode, resolve_mode)?; + if let (Some(base), Some(to)) = (relative_base.as_deref(), relative_to.as_deref()) + && !to.starts_with(base) + { + return Ok((None, None)); + } + Ok((relative_to, relative_base)) +} + +/// Prepare single `relative-*` option. +fn canonicalize_relative_option( + relative: Option, + can_mode: MissingHandling, + resolve_mode: ResolveMode, +) -> UResult> { + Ok(match relative { + None => None, + Some(p) => Some( + canonicalize_relative(&p, can_mode, resolve_mode) + .map_err_context(|| p.maybe_quote().to_string())?, + ), + }) +} + +/// Make `relative-to` or `relative-base` path values absolute. +/// +/// # Errors +/// +/// If the given path is not a directory the function returns an error. +/// If some parts of the file don't exist, or symlinks make loops, or +/// some other IO error happens, the function returns error, too. +fn canonicalize_relative( + r: &Path, + can_mode: MissingHandling, + resolve: ResolveMode, +) -> std::io::Result { + // pi-uutils: resolve the option path against the shell working directory; + // `r` is kept by the caller for display. Resolving before `canonicalize` + // also keeps uucore's internal `env::current_dir()` fallback from being + // consulted. + let abs = canonicalize(pi_uutils_ctx::resolve(r), can_mode, resolve)?; + if can_mode == MissingHandling::Existing && !abs.is_dir() { + abs.read_dir()?; // raise not a directory error + } + Ok(abs) +} + +/// Resolve a path to an absolute form and print it. +/// +/// If `relative_to` and/or `relative_base` is given +/// the path is printed in a relative form to one of this options. +/// See the details in `process_relative` function. +/// If `zero` is `true`, then this function +/// prints the path followed by the null byte (`'\0'`) instead of a +/// newline character (`'\n'`). +/// +/// # Errors +/// +/// This function returns an error if there is a problem resolving +/// symbolic links. +fn resolve_path( + p: &Path, + line_ending: LineEnding, + resolve: ResolveMode, + can_mode: MissingHandling, + relative_to: Option<&Path>, + relative_base: Option<&Path>, +) -> std::io::Result<()> { + // pi-uutils: resolve the operand against the shell working directory; `p` + // is kept by the caller for display. Resolving before `canonicalize` also + // keeps uucore's internal `env::current_dir()` fallback from being + // consulted. + let abs = canonicalize(pi_uutils_ctx::resolve(p), can_mode, resolve)?; + + let abs = process_relative(abs, relative_base, relative_to); + + // pi-uutils: replacement for `print_verbatim` + process stdout — writes + // the resolved path bytes verbatim to the context stdout. + let mut out = pi_uutils_ctx::stdout(); + out.write_all( + uucore::os_str_as_bytes(abs.as_os_str()).map_err(|e| std::io::Error::other(e.to_string()))?, + )?; + out.write_all(&[line_ending.into()])?; + out.flush()?; + Ok(()) +} + +/// Conditionally converts an absolute path to a relative form, +/// according to the rules: +/// 1. if only `relative_to` is given, the result is relative to `relative_to` +/// 2. if only `relative_base` is given, it checks whether given `path` is a +/// descendant of `relative_base`, on success the result is relative to +/// `relative_base`, otherwise the result is the given `path` +/// 3. if both `relative_to` and `relative_base` are given, the result is +/// relative to `relative_to` if `path` is a descendant of `relative_base`, +/// otherwise the result is `path` +/// +/// For more information see +/// +fn process_relative( + path: PathBuf, + relative_base: Option<&Path>, + relative_to: Option<&Path>, +) -> PathBuf { + if let Some(base) = relative_base { + if path.starts_with(base) { + make_path_relative_to(path, relative_to.unwrap_or(base)) + } else { + path + } + } else if let Some(to) = relative_to { + make_path_relative_to(path, to) + } else { + path + } +} + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, fs, io::Write, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + fn run_in(cwd: PathBuf, args: Vec<&str>) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd, + env: HashMap::new(), + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; + + let argv: Vec = std::iter::once("realpath") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + /// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var, + /// which canonicalization would otherwise expand mid-assertion). + fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) { + let dir = tempfile::tempdir().unwrap(); + let canon = fs::canonicalize(dir.path()).unwrap(); + (dir, canon) + } + + #[cfg(unix)] + #[test] + fn resolves_relative_operand_against_scope_cwd() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("target"), b"x").unwrap(); + std::os::unix::fs::symlink("target", root.join("link")).unwrap(); + + // Relative operand + scope cwd differing from the process cwd: only + // the call-site `pi_uutils_ctx::resolve` patch makes this find the + // symlink and print its canonical target. + let (code, stdout, stderr) = run_in(root.clone(), vec!["link"]); + assert_eq!(code, 0); + assert_eq!(stdout, format!("{}\n", root.join("target").display())); + assert_eq!(stderr, ""); + } + + #[test] + fn canonicalize_missing_builds_path_from_scope_cwd() { + let (_dir, root) = canonical_tempdir(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["-m", "missing/sub"]); + assert_eq!(code, 0); + assert_eq!(stdout, format!("{}\n", root.join("missing").join("sub").display())); + assert_eq!(stderr, ""); + } + + #[test] + fn relative_to_option_resolves_against_scope_cwd_and_relativizes_output() { + let (_dir, root) = canonical_tempdir(); + fs::create_dir(root.join("sub")).unwrap(); + fs::write(root.join("sub").join("file"), b"x").unwrap(); + + // Both the operand and the (relative) --relative-to directory resolve + // against the scope cwd. + let (code, stdout, stderr) = run_in(root, vec!["--relative-to", "sub", "sub/file"]); + assert_eq!(code, 0); + assert_eq!(stdout, "file\n"); + assert_eq!(stderr, ""); + } + + #[test] + fn zero_flag_terminates_with_nul() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("f"), b"x").unwrap(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["-z", "f"]); + assert_eq!(code, 0); + assert_eq!(stdout, format!("{}\0", root.join("f").display())); + assert_eq!(stderr, ""); + } + + #[test] + fn nonexistent_operand_errors_but_later_operands_still_process() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("f"), b"x").unwrap(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["missing/x", "f"]); + assert_eq!(code, 1); + assert_eq!(stdout, format!("{}\n", root.join("f").display())); + assert!(stderr.contains("realpath: missing/x"), "stderr: {stderr}"); + assert!(stderr.contains("No such file"), "stderr: {stderr}"); + } + + #[test] + fn quiet_suppresses_error_messages() { + let (_dir, root) = canonical_tempdir(); + + // Upstream drops the per-file result entirely under -q (the error is + // neither printed nor accumulated into the exit code). + let (code, stdout, stderr) = run_in(root, vec!["-q", "missing/x"]); + assert_eq!(code, 0); + assert_eq!(stdout, ""); + assert_eq!(stderr, ""); + } + + #[cfg(unix)] + #[test] + fn strip_keeps_symlinks_unresolved() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("target"), b"x").unwrap(); + std::os::unix::fs::symlink("target", root.join("link")).unwrap(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["-s", "link"]); + assert_eq!(code, 0); + assert_eq!(stdout, format!("{}\n", root.join("link").display())); + assert_eq!(stderr, ""); + } + + #[test] + fn empty_operand_is_rejected() { + // The NonEmptyOsStringParser turns "" into a clap parse error (rendered + // by clap's default renderer since the uucore localization layer is + // patched out) instead of a filesystem lookup. + let (code, stdout, stderr) = run_in(PathBuf::from("."), vec![""]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert!(stderr.contains("invalid value"), "stderr: {stderr}"); + } + + #[test] + fn help_renders_to_scope_stdout() { + let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["--help"]); + assert_eq!(code, 0); + assert!(stdout.contains("Usage:")); + assert!(stdout.contains("Print the resolved path")); + assert_eq!(stderr, ""); + } +} diff --git a/crates/vendor/uu-seq/Cargo.toml b/crates/vendor/uu-seq/Cargo.toml new file mode 100644 index 000000000..f301f621d --- /dev/null +++ b/crates/vendor/uu-seq/Cargo.toml @@ -0,0 +1,32 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/seq), patched to route I/O +# through pi-uutils-ctx so it can run in-process as a shell builtin. See +# src/seq.rs for the patch markers (`pi-uutils:` comments). +[package] +name = "uu_seq" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "seq ~ (uutils) display a sequence of numbers (vendored + patched for in-process embedding)" + +[lib] +path = "src/seq.rs" + +[dependencies] +bigdecimal = "0.4" +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +num-bigint = "0.4" +num-traits = "0.2" +thiserror = "2.0.3" +# pi-uutils: upstream also enables "signals" (SIGPIPE probing) — dropped, the +# in-process builtin has no process-global signal handling. +uucore = { version = "0.8.0", features = [ + "extendedbigdecimal", + "fast-inc", + "format", + "parser", + "quoting-style", +] } +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[dev-dependencies] +parking_lot = "0.12" diff --git a/crates/vendor/uu-seq/LICENSE b/crates/vendor/uu-seq/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-seq/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-seq/src/error.rs b/crates/vendor/uu-seq/src/error.rs new file mode 100644 index 000000000..417dc0960 --- /dev/null +++ b/crates/vendor/uu-seq/src/error.rs @@ -0,0 +1,57 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. +// spell-checker:ignore numberparse +//! Errors returned by seq. + +// pi-uutils: `translate!` message lookups are literalized with the en-US +// strings from upstream's locales/en-US.ftl. + +use thiserror::Error; +use uucore::{display::Quotable, error::UError}; + +use crate::numberparse::ParseNumberError; + +#[derive(Debug, Error)] +pub enum SeqError { + /// An error parsing the input arguments. + /// + /// The parameters are the [`String`] argument as read from the + /// command line and the underlying parsing error itself. + #[error("invalid {} argument: {}", parse_error_type(.1), .0.quote())] + ParseError(String, ParseNumberError), + + /// The increment argument was zero, which is not allowed. + /// + /// The parameter is the increment argument as a [`String`] as read + /// from the command line. + #[error("invalid Zero increment value: {}", .0.quote())] + ZeroIncrement(String), + + /// No arguments were passed to this function, 1 or more is required + #[error("missing operand")] + NoArguments, + + /// Both a format and equal width where passed to seq + #[error("format string may not be specified when printing equal width strings")] + FormatAndEqualWidth, +} + +fn parse_error_type(e: &ParseNumberError) -> &'static str { + match e { + ParseNumberError::Float => "floating point", + ParseNumberError::Nan => "'not-a-number'", + } +} + +impl UError for SeqError { + /// Always return 1. + fn code(&self) -> i32 { + 1 + } + + fn usage(&self) -> bool { + true + } +} diff --git a/crates/vendor/uu-seq/src/number.rs b/crates/vendor/uu-seq/src/number.rs new file mode 100644 index 000000000..caa530b55 --- /dev/null +++ b/crates/vendor/uu-seq/src/number.rs @@ -0,0 +1,52 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. +// spell-checker:ignore extendedbigdecimal +use num_traits::Zero; +use uucore::extendedbigdecimal::ExtendedBigDecimal; + +/// A number with a specified number of integer and fractional digits. +/// +/// This struct can be used to represent a number along with information +/// on how many significant digits to use when displaying the number. +/// The [`PreciseNumber::num_integral_digits`] field also includes the width +/// needed to display the "-" character for a negative number. +/// [`PreciseNumber::num_fractional_digits`] provides the number of decimal +/// digits after the decimal point (a.k.a. precision), or None if that number +/// cannot intuitively be obtained (i.e. hexadecimal floats). +/// Note: Those 2 fields should not necessarily be interpreted literally, but as +/// matching GNU `seq` behavior: the exact way of guessing desired precision +/// from user input is a matter of interpretation. +/// +/// You can get an instance of this struct by calling [`str::parse`]. +#[derive(Debug)] +pub struct PreciseNumber { + pub number: ExtendedBigDecimal, + pub num_integral_digits: usize, + pub num_fractional_digits: Option, +} + +impl PreciseNumber { + // pi-uutils: upstream's unused `new` constructor (only reachable from the + // fuzzing harness) is dropped to keep the vendored crate warning-free. + + pub fn one() -> Self { + // We would like to implement `num_traits::One`, but it requires + // a multiplication implementation, and we don't want to + // implement that here. + Self { + number: ExtendedBigDecimal::one(), + num_integral_digits: 1, + num_fractional_digits: Some(0), + } + } + + /// Decide whether this number is zero (either positive or negative). + pub fn is_zero(&self) -> bool { + // We would like to implement `num_traits::Zero`, but it + // requires an addition implementation, and we don't want to + // implement that here. + self.number.is_zero() + } +} diff --git a/crates/vendor/uu-seq/src/numberparse.rs b/crates/vendor/uu-seq/src/numberparse.rs new file mode 100644 index 000000000..74b832d4d --- /dev/null +++ b/crates/vendor/uu-seq/src/numberparse.rs @@ -0,0 +1,351 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. +// spell-checker:ignore extendedbigdecimal bigdecimal numberparse +// hexadecimalfloat +//! Parsing numbers for use in `seq`. +//! +//! This module provides an implementation of [`FromStr`] for the +//! [`PreciseNumber`] struct. +use std::str::FromStr; + +use uucore::{ + extendedbigdecimal::ExtendedBigDecimal, + parser::num_parser::{ExtendedParser, ExtendedParserError}, +}; + +use crate::number::PreciseNumber; + +/// An error returned when parsing a number fails. +#[derive(Debug, PartialEq, Eq)] +pub enum ParseNumberError { + Float, + Nan, +} + +/// Compute the number of integral and fractional digits in input string, +/// and wrap the result in a PreciseNumber. +/// We know that the string has already been parsed correctly, so we don't +/// need to be too careful. +fn compute_num_digits(input: &str, ebd: ExtendedBigDecimal) -> PreciseNumber { + let input = input.to_lowercase(); + let input = input.trim_start(); + + // Leading + is ignored for this. + let input = input.strip_prefix('+').unwrap_or(input); + + // Integral digits for any hex number is ill-defined (0 is fine as an output) + // Fractional digits for an floating hex number is ill-defined, return None + // as we'll totally ignore that number for precision computations. + // Still return 0 for hex integers though. + if input.starts_with("0x") || input.starts_with("-0x") { + return PreciseNumber { + number: ebd, + num_integral_digits: 0, + num_fractional_digits: if input.contains('.') || input.contains('p') { + None + } else { + Some(0) + }, + }; + } + + // Split the exponent part, if any + let parts: Vec<&str> = input.split('e').collect(); + debug_assert!(parts.len() <= 2); + + // Count all the digits up to `.`, `-` sign is included. + let (mut int_digits, mut frac_digits) = match parts[0].find('.') { + Some(i) => { + // Cover special case .X and -.X where we behave as if there was a leading 0: + // 0.X, -0.X. + let int_digits = match i { + 0 => 1, + 1 if parts[0].starts_with('-') => 2, + _ => i, + }; + + (int_digits, parts[0].len() - i - 1) + }, + None => (parts[0].len(), 0), + }; + + // If there is an exponent, reparse that (yes this is not optimal, + // but we can't necessarily exactly recover that from the parsed number). + if parts.len() == 2 { + let exp = parts[1].parse::().unwrap_or(0); + // For positive exponents, effectively expand the number. Ignore negative + // exponents. Also ignore overflowed exponents (unwrap_or(0)). + if exp > 0 { + int_digits += exp.try_into().unwrap_or(0); + } + frac_digits = if exp < frac_digits as i64 { + // Subtract from i128 to avoid any overflow + (frac_digits as i128 - exp as i128).try_into().unwrap_or(0) + } else { + 0 + } + } + + PreciseNumber { + number: ebd, + num_integral_digits: int_digits, + num_fractional_digits: Some(frac_digits), + } +} + +// Note: We could also have provided an `ExtendedParser` implementation for +// PreciseNumber, but we want a simpler custom error. +impl FromStr for PreciseNumber { + type Err = ParseNumberError; + + fn from_str(input: &str) -> Result { + let ebd = match ExtendedBigDecimal::extended_parse(input) { + Ok(ebd) => match ebd { + // Handle special values + ExtendedBigDecimal::BigDecimal(_) | ExtendedBigDecimal::MinusZero => { + // TODO: GNU `seq` treats small numbers < 1e-4950 as 0, we could do the same + // to avoid printing senselessly small numbers. + ebd + }, + ExtendedBigDecimal::Infinity | ExtendedBigDecimal::MinusInfinity => { + return Ok(Self { + number: ebd, + num_integral_digits: 0, + num_fractional_digits: Some(0), + }); + }, + ExtendedBigDecimal::Nan | ExtendedBigDecimal::MinusNan => { + return Err(ParseNumberError::Nan); + }, + }, + Err(ExtendedParserError::Underflow(ebd)) => ebd, // Treat underflow as 0 + Err(_) => return Err(ParseNumberError::Float), + }; + + Ok(compute_num_digits(input, ebd)) + } +} + +#[cfg(test)] +mod tests { + use bigdecimal::BigDecimal; + use uucore::extendedbigdecimal::ExtendedBigDecimal; + + use crate::{number::PreciseNumber, numberparse::ParseNumberError}; + + /// Convenience function for parsing a [`Number`] and unwrapping. + fn parse(s: &str) -> ExtendedBigDecimal { + s.parse::().unwrap().number + } + + /// Convenience function for getting the number of integral digits. + fn num_integral_digits(s: &str) -> usize { + s.parse::().unwrap().num_integral_digits + } + + /// Convenience function for getting the number of fractional digits. + fn num_fractional_digits(s: &str) -> usize { + s.parse::() + .unwrap() + .num_fractional_digits + .unwrap() + } + + /// Convenience function for making sure the number of fractional digits is + /// "None" + fn num_fractional_digits_is_none(s: &str) -> bool { + s.parse::() + .unwrap() + .num_fractional_digits + .is_none() + } + + #[test] + fn test_parse_minus_zero_int() { + assert_eq!(parse("-0e0"), ExtendedBigDecimal::MinusZero); + assert_eq!(parse("-0e-0"), ExtendedBigDecimal::MinusZero); + assert_eq!(parse("-0e1"), ExtendedBigDecimal::MinusZero); + assert_eq!(parse("-0e+1"), ExtendedBigDecimal::MinusZero); + assert_eq!(parse("-0.0e1"), ExtendedBigDecimal::MinusZero); + assert_eq!(parse("-0x0"), ExtendedBigDecimal::MinusZero); + } + + #[test] + fn test_parse_minus_zero_float() { + assert_eq!(parse("-0.0"), ExtendedBigDecimal::MinusZero); + assert_eq!(parse("-0e-1"), ExtendedBigDecimal::MinusZero); + assert_eq!(parse("-0.0e-1"), ExtendedBigDecimal::MinusZero); + } + + #[test] + fn test_parse_big_int() { + assert_eq!(parse("0"), ExtendedBigDecimal::zero()); + assert_eq!(parse("0.1e1"), ExtendedBigDecimal::one()); + assert_eq!(parse("0.1E1"), ExtendedBigDecimal::one()); + assert_eq!( + parse("1.0e1"), + ExtendedBigDecimal::BigDecimal("10".parse::().unwrap()) + ); + } + + #[test] + fn test_parse_hexadecimal_big_int() { + assert_eq!(parse("0x0"), ExtendedBigDecimal::zero()); + assert_eq!( + parse("0x10"), + ExtendedBigDecimal::BigDecimal("16".parse::().unwrap()) + ); + } + + #[test] + fn test_parse_big_decimal() { + assert_eq!( + parse("0.0"), + ExtendedBigDecimal::BigDecimal("0.0".parse::().unwrap()) + ); + assert_eq!(parse(".0"), ExtendedBigDecimal::BigDecimal("0.0".parse::().unwrap())); + assert_eq!( + parse("1.0"), + ExtendedBigDecimal::BigDecimal("1.0".parse::().unwrap()) + ); + assert_eq!( + parse("10e-1"), + ExtendedBigDecimal::BigDecimal("1.0".parse::().unwrap()) + ); + assert_eq!( + parse("-1e-3"), + ExtendedBigDecimal::BigDecimal("-0.001".parse::().unwrap()) + ); + } + + #[test] + fn test_parse_inf() { + assert_eq!(parse("inf"), ExtendedBigDecimal::Infinity); + assert_eq!(parse("infinity"), ExtendedBigDecimal::Infinity); + assert_eq!(parse("+inf"), ExtendedBigDecimal::Infinity); + assert_eq!(parse("+infinity"), ExtendedBigDecimal::Infinity); + assert_eq!(parse("-inf"), ExtendedBigDecimal::MinusInfinity); + assert_eq!(parse("-infinity"), ExtendedBigDecimal::MinusInfinity); + } + + #[test] + fn test_parse_invalid_float() { + assert_eq!("1.2.3".parse::().unwrap_err(), ParseNumberError::Float); + assert_eq!("1e2e3".parse::().unwrap_err(), ParseNumberError::Float); + assert_eq!("1e2.3".parse::().unwrap_err(), ParseNumberError::Float); + assert_eq!("-+-1".parse::().unwrap_err(), ParseNumberError::Float); + } + + #[test] + fn test_parse_invalid_hex() { + assert_eq!("0xg".parse::().unwrap_err(), ParseNumberError::Float); + } + + #[test] + fn test_parse_invalid_nan() { + assert_eq!("nan".parse::().unwrap_err(), ParseNumberError::Nan); + assert_eq!("NAN".parse::().unwrap_err(), ParseNumberError::Nan); + assert_eq!("NaN".parse::().unwrap_err(), ParseNumberError::Nan); + assert_eq!("nAn".parse::().unwrap_err(), ParseNumberError::Nan); + assert_eq!("-nan".parse::().unwrap_err(), ParseNumberError::Nan); + } + + #[test] + #[allow(clippy::cognitive_complexity)] + fn test_num_integral_digits() { + // no decimal, no exponent + assert_eq!(num_integral_digits("123"), 3); + // decimal, no exponent + assert_eq!(num_integral_digits("123.45"), 3); + assert_eq!(num_integral_digits("-0.1"), 2); + assert_eq!(num_integral_digits("-.1"), 2); + // exponent, no decimal + assert_eq!(num_integral_digits("123e4"), 3 + 4); + assert_eq!(num_integral_digits("123e-4"), 3); + assert_eq!(num_integral_digits("-1e-3"), 2); + // decimal and exponent + assert_eq!(num_integral_digits("123.45e6"), 3 + 6); + assert_eq!(num_integral_digits("123.45e-6"), 3); + assert_eq!(num_integral_digits("123.45e-1"), 3); + assert_eq!(num_integral_digits("-0.1e0"), 2); + assert_eq!(num_integral_digits("-0.1e2"), 4); + assert_eq!(num_integral_digits("-.1e0"), 2); + assert_eq!(num_integral_digits("-.1e2"), 4); + assert_eq!(num_integral_digits("-1.e-3"), 2); + assert_eq!(num_integral_digits("-1.0e-4"), 2); + // minus zero int + assert_eq!(num_integral_digits("-0e0"), 2); + assert_eq!(num_integral_digits("-0e-0"), 2); + assert_eq!(num_integral_digits("-0e1"), 3); + assert_eq!(num_integral_digits("-0e+1"), 3); + assert_eq!(num_integral_digits("-0.0e1"), 3); + // minus zero float + assert_eq!(num_integral_digits("-0.0"), 2); + assert_eq!(num_integral_digits("-0e-1"), 2); + assert_eq!(num_integral_digits("-0.0e-1"), 2); + + // TODO In GNU `seq`, the `-w` option does not seem to work with + // hexadecimal arguments. In order to match that behavior, we + // report the number of integral digits as zero for hexadecimal + // inputs. + assert_eq!(num_integral_digits("0xff"), 0); + } + + #[test] + #[allow(clippy::cognitive_complexity)] + fn test_num_fractional_digits() { + // no decimal, no exponent + assert_eq!(num_fractional_digits("123"), 0); + assert_eq!(num_fractional_digits("0xff"), 0); + // decimal, no exponent + assert_eq!(num_fractional_digits("123.45"), 2); + assert_eq!(num_fractional_digits("-0.1"), 1); + assert_eq!(num_fractional_digits("-.1"), 1); + // exponent, no decimal + assert_eq!(num_fractional_digits("123e4"), 0); + assert_eq!(num_fractional_digits("123e-4"), 4); + assert_eq!(num_fractional_digits("123e-1"), 1); + assert_eq!(num_fractional_digits("-1e-3"), 3); + // decimal and exponent + assert_eq!(num_fractional_digits("123.45e6"), 0); + assert_eq!(num_fractional_digits("123.45e1"), 1); + assert_eq!(num_fractional_digits("123.45e-6"), 8); + assert_eq!(num_fractional_digits("123.45e-1"), 3); + assert_eq!(num_fractional_digits("-0.1e0"), 1); + assert_eq!(num_fractional_digits("-0.1e2"), 0); + assert_eq!(num_fractional_digits("-.1e0"), 1); + assert_eq!(num_fractional_digits("-.1e2"), 0); + assert_eq!(num_fractional_digits("-1.e-3"), 3); + assert_eq!(num_fractional_digits("-1.0e-4"), 5); + // minus zero int + assert_eq!(num_fractional_digits("-0e0"), 0); + assert_eq!(num_fractional_digits("-0e-0"), 0); + assert_eq!(num_fractional_digits("-0e1"), 0); + assert_eq!(num_fractional_digits("-0e+1"), 0); + assert_eq!(num_fractional_digits("-0.0e1"), 0); + // minus zero float + assert_eq!(num_fractional_digits("-0.0"), 1); + assert_eq!(num_fractional_digits("-0e-1"), 1); + assert_eq!(num_fractional_digits("-0.0e-1"), 2); + // Hexadecimal numbers + assert_eq!(num_fractional_digits("0xff"), 0); + assert!(num_fractional_digits_is_none("0xff.1")); + } + + #[test] + fn test_parse_min_exponents() { + // Make sure exponents < i64::MIN do not cause errors + assert!("1e-9223372036854775807".parse::().is_ok()); + assert!("1e-9223372036854775808".parse::().is_ok()); + assert!("1e-92233720368547758080".parse::().is_ok()); + } + + #[test] + fn test_parse_max_exponents() { + // Make sure exponents much bigger than i64::MAX cause errors + assert!("1e9223372036854775807".parse::().is_ok()); + assert!("1e92233720368547758070".parse::().is_err()); + } +} diff --git a/crates/vendor/uu-seq/src/seq.rs b/crates/vendor/uu-seq/src/seq.rs new file mode 100644 index 000000000..82cdfd46c --- /dev/null +++ b/crates/vendor/uu-seq/src/seq.rs @@ -0,0 +1,561 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +// spell-checker:ignore (ToDO) bigdecimal extendedbigdecimal numberparse +// hexadecimalfloat biguint + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. seq is pure computation + stdout: all process-global +// stdio is routed through `pi_uutils_ctx` (the emission loops write to a +// `BufWriter` around the context stdout handle and poll +// `pi_uutils_ctx::is_cancelled()` periodically, since seq can generate +// unbounded output), `translate!` strings are literalized, SIGPIPE probing is +// dropped, and the entry point no longer calls `std::process::exit`. + +use std::{ + ffi::{OsStr, OsString}, + io::{BufWriter, Write}, +}; + +use clap::{Arg, ArgAction, ArgMatches, Command}; +use num_bigint::BigUint; +use num_traits::{ToPrimitive, Zero}; +use pi_uutils_ctx::format_usage; +use uucore::{ + error::{FromIo, UResult}, + extendedbigdecimal::ExtendedBigDecimal, + fast_inc::fast_inc, + format::{Format, num_format, num_format::FloatVariant}, +}; + +mod error; + +mod number; +mod numberparse; +use crate::{error::SeqError, number::PreciseNumber}; + +const OPT_SEPARATOR: &str = "separator"; +const OPT_TERMINATOR: &str = "terminator"; +const OPT_EQUAL_WIDTH: &str = "equal-width"; +const OPT_FORMAT: &str = "format"; + +const ARG_NUMBERS: &str = "numbers"; + +/// pi-uutils: how many emitted numbers to print between cancellation polls in +/// the (potentially unbounded) emission loops. +const CANCEL_POLL_INTERVAL: u64 = 4096; + +#[derive(Clone)] +struct SeqOptions<'a> { + separator: OsString, + terminator: OsString, + equal_width: bool, + format: Option<&'a str>, +} + +/// A range of floats. +/// +/// The elements are (first, increment, last). +type RangeFloat = (ExtendedBigDecimal, ExtendedBigDecimal, ExtendedBigDecimal); + +/// Turn short args with attached value, for example "-s,", into two args "-s" +/// and "," to make them work with clap. +fn split_short_args_with_value(args: impl uucore::Args) -> impl uucore::Args { + let mut v: Vec = Vec::new(); + + for arg in args { + let bytes = arg.as_encoded_bytes(); + + if bytes.len() > 2 + && (bytes.starts_with(b"-f") || bytes.starts_with(b"-s") || bytes.starts_with(b"-t")) + { + let (short_arg, value) = bytes.split_at(2); + // SAFETY: + // Both `short_arg` and `value` only contain content that originated from + // `OsStr::as_encoded_bytes` + v.push(unsafe { OsString::from_encoded_bytes_unchecked(short_arg.to_vec()) }); + v.push(unsafe { OsString::from_encoded_bytes_unchecked(value.to_vec()) }); + } else { + v.push(arg); + } + } + + v.into_iter() +} + +fn select_precision( + first: &PreciseNumber, + increment: &PreciseNumber, + last: &PreciseNumber, +) -> Option { + match (first.num_fractional_digits, increment.num_fractional_digits, last.num_fractional_digits) + { + (Some(0), Some(0), Some(0)) => Some(0), + (Some(f), Some(i), Some(_)) => Some(f.max(i)), + _ => None, + } +} + +/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the +/// arguments directly (without the uucore clap-localization helper that would +/// terminate the process), renders clap help/usage/version to the context +/// streams, and maps the `UResult` to an exit code, so it is safe to run inside +/// the host shell process. +pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(split_short_args_with_value(argv.into_iter())) + { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match seq_main(&matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "seq: {msg}"); + } + if code == 0 { 1 } else { code } + }, + } +} + +fn seq_main(matches: &ArgMatches) -> UResult<()> { + let numbers_option = matches.get_many::(ARG_NUMBERS); + + if numbers_option.is_none() { + return Err(SeqError::NoArguments.into()); + } + + let numbers = numbers_option.unwrap().collect::>(); + + let options = SeqOptions { + separator: matches + .get_one::(OPT_SEPARATOR) + .cloned() + .unwrap_or_else(|| OsString::from("\n")), + terminator: matches + .get_one::(OPT_TERMINATOR) + .cloned() + .unwrap_or_else(|| OsString::from("\n")), + equal_width: matches.get_flag(OPT_EQUAL_WIDTH), + format: matches.get_one::(OPT_FORMAT).map(String::as_str), + }; + + if options.equal_width && options.format.is_some() { + return Err(SeqError::FormatAndEqualWidth.into()); + } + + let first = if numbers.len() > 1 { + match numbers[0].parse() { + Ok(num) => num, + Err(e) => return Err(SeqError::ParseError(numbers[0].to_owned(), e).into()), + } + } else { + PreciseNumber::one() + }; + let increment = if numbers.len() > 2 { + match numbers[1].parse() { + Ok(num) => num, + Err(e) => return Err(SeqError::ParseError(numbers[1].to_owned(), e).into()), + } + } else { + PreciseNumber::one() + }; + if increment.is_zero() { + return Err(SeqError::ZeroIncrement(numbers[1].to_owned()).into()); + } + let last: PreciseNumber = { + // We are guaranteed that `numbers.len()` is greater than zero + // and at most three because of the argument specification in + // `uu_app()`. + let n: usize = numbers.len(); + match numbers[n - 1].parse() { + Ok(num) => num, + Err(e) => return Err(SeqError::ParseError(numbers[n - 1].to_owned(), e).into()), + } + }; + + // If a format was passed on the command line, use that. + // If not, use some default format based on parameters precision. + let (format, padding, fast_allowed) = if let Some(str) = options.format { + (Format::::parse(str)?, 0, false) + } else { + let precision = select_precision(&first, &increment, &last); + + let padding = if options.equal_width { + let precision_value = precision.unwrap_or(0); + first + .num_integral_digits + .max(increment.num_integral_digits) + .max(last.num_integral_digits) + + if precision_value > 0 { + precision_value + 1 + } else { + 0 + } + } else { + 0 + }; + + let formatter = match precision { + // format with precision: decimal floats and integers + Some(precision) => num_format::Float { + variant: FloatVariant::Decimal, + width: padding, + alignment: num_format::NumberAlignment::RightZero, + precision: Some(precision), + ..Default::default() + }, + // format without precision: hexadecimal floats + None => num_format::Float { variant: FloatVariant::Shortest, ..Default::default() }, + }; + // Allow fast printing if precision is 0 (integer inputs), `print_seq` will do + // further checks. + (Format::from_formatter(formatter), padding, precision == Some(0)) + }; + + let result = print_seq( + (first.number, increment.number, last.number), + &options.separator, + &options.terminator, + &format, + fast_allowed, + padding, + ); + + match result { + Ok(()) => Ok(()), + Err(err) if err.kind() == std::io::ErrorKind::BrokenPipe => { + // GNU seq prints the Broken pipe message but still exits with status 0 + // unless SIGPIPE was explicitly ignored, in which case it should fail. + // pi-uutils: the in-process builtin does not manipulate process + // signal dispositions, so the upstream `sigpipe_was_ignored` probe + // is dropped and the message goes to the context stderr. + let err = err.map_err_context(|| "write error".into()); + let _ = writeln!(pi_uutils_ctx::stderr(), "seq: {err}"); + Ok(()) + }, + Err(err) => Err(err.map_err_context(|| "write error".into())), + } +} + +pub fn uu_app() -> Command { + Command::new("seq") + .trailing_var_arg(true) + .infer_long_args(true) + .version(uucore::crate_version!()) + .about("Display numbers from FIRST to LAST, in steps of INCREMENT.") + .override_usage(format_usage( + "seq [OPTION]... LAST\nseq [OPTION]... FIRST LAST\nseq [OPTION]... FIRST INCREMENT LAST", + )) + .arg( + Arg::new(OPT_SEPARATOR) + .short('s') + .long("separator") + .help("Separator character (defaults to \\n)") + .value_parser(clap::value_parser!(OsString)), + ) + .arg( + Arg::new(OPT_TERMINATOR) + .short('t') + .long("terminator") + .help("Terminator character (defaults to \\n)") + .value_parser(clap::value_parser!(OsString)), + ) + .arg( + Arg::new(OPT_EQUAL_WIDTH) + .short('w') + .long("equal-width") + .help("Equalize widths of all numbers by padding with zeros") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(OPT_FORMAT) + .short('f') + .long(OPT_FORMAT) + .help("use printf style floating-point FORMAT"), + ) + .arg( + // we use allow_hyphen_values instead of allow_negative_numbers because clap removed + // the support for "exotic" negative numbers like -.1 (see https://github.com/clap-rs/clap/discussions/5837) + Arg::new(ARG_NUMBERS) + .allow_hyphen_values(true) + .action(ArgAction::Append) + .num_args(1..=3), + ) +} + +/// Integer print, default format, positive increment: fast code path +/// that avoids reformatting digit at all iterations. +fn fast_print_seq( + mut stdout: impl Write, + first: &BigUint, + increment: u64, + last: &BigUint, + separator: &OsStr, + terminator: &OsStr, + padding: usize, +) -> std::io::Result<()> { + // Nothing to do, just return. + if last < first { + return Ok(()); + } + + // Do at most u64::MAX loops. We can print in the order of 1e8 digits per + // second, u64::MAX is 1e19, so it'd take hundreds of years for this to + // complete anyway. TODO: we can move this test to `print_seq` if we care about + // this case. + let loop_cnt = ((last - first) / increment).to_u64().unwrap_or(u64::MAX); + + // Format the first number. + let first_str = first.to_string(); + + // Makeshift log10.ceil + let last_length = last.to_string().len(); + + // Allocate a large u8 buffer, that contains a preformatted string + // of the number followed by the `separator`. + // + // | ... head space ... | number | separator | + // ^0 ^ start ^ num_end ^ size (==buf.len()) + // + // We keep track of start in this buffer, as the number grows. + // When printing, we take a slice between start and end. + let size = last_length.max(padding) + separator.len(); + // Fill with '0', this is needed for equal_width, and harmless otherwise. + let mut buf = vec![b'0'; size]; + let buf = buf.as_mut_slice(); + + let num_end = buf.len() - separator.len(); + let mut start = num_end - first_str.len(); + + // Initialize buf with first and separator. + buf[start..num_end].copy_from_slice(first_str.as_bytes()); + buf[num_end..].copy_from_slice(separator.as_encoded_bytes()); + + // Normally, if padding is > 0, it should be equal to last_length, + // so start would be == 0, but there are corner cases. + start = start.min(num_end - padding); + + // Prepare the number to increment with as a string + let inc_str = increment.to_string(); + let inc_str = inc_str.as_bytes(); + + for i in 0..loop_cnt { + // pi-uutils: seq can generate effectively unbounded output; poll the + // host cancel flag periodically so shell abort/timeout is observed. + if i % CANCEL_POLL_INTERVAL == 0 && pi_uutils_ctx::is_cancelled() { + return Ok(()); + } + stdout.write_all(&buf[start..])?; + fast_inc(buf, &mut start, num_end, inc_str); + } + // Write the last number without separator, but with terminator. + stdout.write_all(&buf[start..num_end])?; + stdout.write_all(terminator.as_encoded_bytes())?; + stdout.flush()?; + Ok(()) +} + +fn done_printing(next: &T, increment: &T, last: &T) -> bool { + if increment >= &T::zero() { + next > last + } else { + next < last + } +} + +/// Arbitrary precision decimal number code path ("slow" path) +fn print_seq( + range: RangeFloat, + separator: &OsStr, + terminator: &OsStr, + format: &Format, + fast_allowed: bool, + padding: usize, // Used by fast path only +) -> std::io::Result<()> { + // pi-uutils: buffer the context stdout handle instead of the (locked) + // process stdout. + let mut stdout = BufWriter::new(pi_uutils_ctx::stdout()); + let (first, increment, last) = range; + + if fast_allowed { + // Test if we can use fast code path. + // First try to convert the range to BigUint (u64 for the increment). + let (first_bui, increment_u64, last_bui) = + (first.to_biguint(), increment.to_biguint().and_then(|x| x.to_u64()), last.to_biguint()); + if let (Some(first_bui), Some(increment_u64), Some(last_bui)) = + (first_bui, increment_u64, last_bui) + { + return fast_print_seq( + stdout, + &first_bui, + increment_u64, + &last_bui, + separator, + terminator, + padding, + ); + } + } + + let mut value = first; + + let mut is_first_iteration = true; + // pi-uutils: iteration counter for periodic cancellation polling. + let mut iterations: u64 = 0; + while !done_printing(&value, &increment, &last) { + // pi-uutils: seq can generate effectively unbounded output; poll the + // host cancel flag periodically so shell abort/timeout is observed. + if iterations.is_multiple_of(CANCEL_POLL_INTERVAL) && pi_uutils_ctx::is_cancelled() { + return Ok(()); + } + iterations += 1; + if !is_first_iteration { + stdout.write_all(separator.as_encoded_bytes())?; + } + format.fmt(&mut stdout, &value)?; + // TODO Implement augmenting addition. + value = value + increment.clone(); + is_first_iteration = false; + } + if !is_first_iteration { + stdout.write_all(terminator.as_encoded_bytes())?; + } + stdout.flush()?; + Ok(()) +} + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + fn run_scoped(args: Vec<&str>, cancelled: bool) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd: PathBuf::from("."), + env: HashMap::new(), + cancel: Arc::new(std::sync::atomic::AtomicBool::new(cancelled)), + }; + + let argv: Vec = std::iter::once("seq") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + fn run_in(args: Vec<&str>) -> (i32, String, String) { + run_scoped(args, false) + } + + #[test] + fn single_operand_counts_from_one() { + let (code, stdout, stderr) = run_in(vec!["3"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "1\n2\n3\n", "")); + } + + #[test] + fn first_increment_last_arithmetic() { + let (code, stdout, stderr) = run_in(vec!["2", "2", "10"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "2\n4\n6\n8\n10\n", "")); + } + + #[test] + fn separator_joins_values_terminator_ends_them() { + let (code, stdout, stderr) = run_in(vec!["-s", ",", "1", "3"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "1,2,3\n", "")); + + // Attached short-arg value goes through `split_short_args_with_value`. + let (code, stdout, _) = run_in(vec!["-s,", "1", "3"]); + assert_eq!((code, stdout.as_str()), (0, "1,2,3\n")); + } + + #[test] + fn equal_width_pads_with_zeros() { + let (code, stdout, stderr) = run_in(vec!["-w", "8", "10"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "08\n09\n10\n", "")); + } + + #[test] + fn float_increment_selects_widest_precision() { + let (code, stdout, stderr) = run_in(vec!["1", "0.5", "2"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "1.0\n1.5\n2.0\n", "")); + } + + #[test] + fn invalid_operand_reports_error_and_fails() { + let (code, stdout, stderr) = run_in(vec!["foo"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert_eq!(stderr, "seq: invalid floating point argument: 'foo'\n"); + } + + #[test] + fn zero_increment_is_rejected() { + let (code, stdout, stderr) = run_in(vec!["1", "0", "5"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert_eq!(stderr, "seq: invalid Zero increment value: '0'\n"); + } + + #[test] + fn cancelled_scope_stops_emission() { + // pi-specific contract: a pre-cancelled scope aborts the (potentially + // unbounded) emission loop instead of printing the full range. + let (code, stdout, stderr) = run_scoped(vec!["1", "1000000"], true); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", "")); + } + + #[test] + fn help_renders_to_scope_stdout() { + let (code, stdout, stderr) = run_in(vec!["--help"]); + assert_eq!(code, 0); + assert!(stdout.contains("Usage:")); + assert!(stdout.contains("steps of INCREMENT")); + assert_eq!(stderr, ""); + } +} diff --git a/crates/vendor/uu-stat/Cargo.toml b/crates/vendor/uu-stat/Cargo.toml new file mode 100644 index 000000000..5c7f30de8 --- /dev/null +++ b/crates/vendor/uu-stat/Cargo.toml @@ -0,0 +1,23 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/stat), patched to resolve +# path arguments against the shell working directory and route I/O through +# pi-uutils-ctx so it can run in-process as a shell builtin. See src/stat.rs +# for the patch markers (`pi-uutils:` comments). +[package] +name = "uu_stat" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "stat ~ (uutils) display FILE status (vendored + patched for in-process embedding)" + +[lib] +path = "src/stat.rs" + +[dependencies] +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +thiserror = "2.0.3" +uucore = { version = "0.8.0", features = ["entries", "libc", "fs", "fsext", "time"] } +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[dev-dependencies] +parking_lot = "0.12" +tempfile = "3" diff --git a/crates/vendor/uu-stat/LICENSE b/crates/vendor/uu-stat/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-stat/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-stat/src/stat.rs b/crates/vendor/uu-stat/src/stat.rs new file mode 100644 index 000000000..6cca57259 --- /dev/null +++ b/crates/vendor/uu-stat/src/stat.rs @@ -0,0 +1,1785 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. +// spell-checker:ignore datetime + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. Every filesystem syscall (stat/lstat/statfs/readlink) +// resolves its path operand against the shell working directory via +// `pi_uutils_ctx::resolve` AT THE CALL SITE, while the original operands are +// kept for display/error messages and `%n` output (GNU prints operands as +// typed). All process-global stdio is routed through `pi_uutils_ctx`, +// `translate!` strings are literalized from locales/en-US.ftl, QUOTING_STYLE is +// read from the scope environment, SELinux support is dropped, and the entry +// point no longer calls `std::process::exit`. The upstream implementation is +// unix-only (it relies on `std::os::unix`), so it lives behind `#[cfg(unix)]`; +// non-unix targets get a stub that reports the builtin as unsupported. + +#[cfg(unix)] +pub use imp::{run, uu_app}; + +/// pi-uutils: non-unix stub — upstream stat cannot be built off unix. +#[cfg(not(unix))] +pub fn run(_argv: Vec) -> i32 { + use std::io::Write; + let _ = writeln!(pi_uutils_ctx::stderr(), "stat: unsupported on this platform"); + 1 +} + +/// pi-uutils: minimal non-unix counterpart of the real `uu_app`. +#[cfg(not(unix))] +pub fn uu_app() -> clap::Command { + clap::Command::new("stat") + .version(uucore::crate_version!()) + .about("Display file or file system status.") + .override_usage(pi_uutils_ctx::format_usage("stat [OPTION]... FILE...")) +} + +#[cfg(unix)] +mod imp { + use std::{ + borrow::Cow, + cell::OnceCell, + ffi::{OsStr, OsString}, + fs::{self, FileType, Metadata}, + io::Write, + os::unix::fs::{FileTypeExt, MetadataExt}, + path::Path, + }; + + use clap::{Arg, ArgAction, ArgMatches, Command, builder::ValueParser}; + use pi_uutils_ctx::format_usage; + use thiserror::Error; + use uucore::{ + display::Quotable, + entries, + error::{UError, UResult, USimpleError}, + fs::{display_permissions, major, minor}, + fsext::{ + FsMeta, MetadataTimeField, StatFs, metadata_get_time, pretty_filetype, pretty_fstype, + read_fs_list, statfs, + }, + libc::mode_t, + time::{FormatSystemTimeFallback, format_system_time, system_time_to_sec}, + }; + + const ABOUT: &str = "Display file or file system status."; + const USAGE: &str = "stat [OPTION]... FILE..."; + // pi-uutils: literalized from locales/en-US.ftl (`stat-after-help`). + const AFTER_HELP: &str = "Valid format sequences for files (without `--file-system`): + +-`%a`: access rights in octal (note '#' and '0' printf flags) +-`%A`: access rights in human readable form +-`%b`: number of blocks allocated (see %B) +-`%B`: the size in bytes of each block reported by %b +-`%C`: SELinux security context string +-`%d`: device number in decimal +-`%D`: device number in hex +-`%f`: raw mode in hex +-`%F`: file type +-`%g`: group ID of owner +-`%G`: group name of owner +-`%h`: number of hard links +-`%i`: inode number +-`%m`: mount point +-`%n`: file name +-`%N`: quoted file name with dereference (follow) if symbolic link +-`%o`: optimal I/O transfer size hint +-`%s`: total size, in bytes +-`%t`: major device type in hex, for character/block device special files +-`%T`: minor device type in hex, for character/block device special files +-`%u`: user ID of owner +-`%U`: user name of owner +-`%w`: time of file birth, human-readable; - if unknown +-`%W`: time of file birth, seconds since Epoch; 0 if unknown +-`%x`: time of last access, human-readable +-`%X`: time of last access, seconds since Epoch +-`%y`: time of last data modification, human-readable + +-`%Y`: time of last data modification, seconds since Epoch +-`%z`: time of last status change, human-readable +-`%Z`: time of last status change, seconds since Epoch + +Valid format sequences for file systems: + +-`%a`: free blocks available to non-superuser +-`%b`: total data blocks in file system +-`%c`: total file nodes in file system +-`%d`: free file nodes in file system +-`%f`: free blocks in file system +-`%i`: file system ID in hex +-`%l`: maximum length of filenames +-`%n`: file name +-`%s`: block size (for faster transfers) +-`%S`: fundamental block size (for block counts) +-`%t`: file system type in hex +-`%T`: file system type in human readable form + +NOTE: your shell may have its own version of stat, which usually supersedes +the version described here. Please refer to your shell's documentation +for details about the options it supports."; + + // pi-uutils: `translate!` error strings literalized from locales/en-US.ftl. + #[derive(Debug, Error)] + enum StatError { + #[error("Invalid quoting style: {style}")] + InvalidQuotingStyle { style: String }, + #[error("missing operand\nTry 'stat --help' for more information.")] + MissingOperand, + #[error("{directive}: invalid directive")] + InvalidDirective { directive: String }, + #[error("cannot read table of mounted file systems: {error}")] + CannotReadFilesystem { error: String }, + #[error("using '-' to denote standard input does not work in file system mode")] + StdinFilesystemMode, + #[error("cannot read file system information for {file}: {error}")] + CannotReadFilesystemInfo { file: String, error: String }, + #[error("cannot stat {file}: {error}")] + CannotStat { file: String, error: String }, + } + + impl UError for StatError { + fn code(&self) -> i32 { + 1 + } + } + + mod options { + pub const DEREFERENCE: &str = "dereference"; + pub const FILE_SYSTEM: &str = "file-system"; + pub const FORMAT: &str = "format"; + pub const PRINTF: &str = "printf"; + pub const TERSE: &str = "terse"; + pub const FILES: &str = "files"; + } + + #[derive(Default, Debug, PartialEq, Eq, Clone, Copy)] + struct Flags { + alter: bool, + zero: bool, + left: bool, + space: bool, + sign: bool, + group: bool, + major: bool, + minor: bool, + } + + /// checks if the string is within the specified bound, + /// if it gets out of bound, error out by printing sub-string from index + /// `beg` to`end`, where `beg` & `end` is the beginning and end index of + /// sub-string, respectively + fn check_bound(slice: &str, bound: usize, beg: usize, end: usize) -> UResult<()> { + if end >= bound { + return Err(USimpleError::new( + 1, + StatError::InvalidDirective { directive: slice[beg..end].quote().to_string() } + .to_string(), + )); + } + Ok(()) + } + + enum Padding { + Zero, + Space, + } + + /// pads the string with zeroes or spaces and prints it + /// + /// # Example + /// ```ignore + /// uu_stat::pad_and_print("1", false, 5, Padding::Zero) == "00001"; + /// ``` + /// currently only supports '0' & ' ' as the padding character + /// because the format specification of print! does not support general + /// fill characters. + fn pad_and_print(result: &str, left: bool, width: usize, padding: Padding) { + // pi-uutils: write to the context stdout instead of `print!`. + let mut out = pi_uutils_ctx::stdout(); + let _ = match (left, padding) { + (false, Padding::Zero) => write!(out, "{result:0>width$}"), + (false, Padding::Space) => write!(out, "{result:>width$}"), + (true, Padding::Zero) => write!(out, "{result:0 write!(out, "{result:( + mut writer: W, + bytes: &[u8], + left: bool, + width: usize, + precision: Precision, + ) -> Result<(), std::io::Error> { + let display_bytes = match precision { + Precision::Number(p) if p < bytes.len() => &bytes[..p], + _ => bytes, + }; + + let display_len = display_bytes.len(); + let padding_needed = width.saturating_sub(display_len); + + let (left_pad, right_pad) = if left { + (0, padding_needed) + } else { + (padding_needed, 0) + }; + + if left_pad > 0 { + write_padding(&mut writer, left_pad)?; + } + writer.write_all(display_bytes)?; + if right_pad > 0 { + write_padding(&mut writer, right_pad)?; + } + + Ok(()) + } + + /// write padding based on a writer W and n size + /// writer is genric to be any buffer like: `std::io::stdout` + /// n is the calculated padding size + fn write_padding(writer: &mut W, n: usize) -> Result<(), std::io::Error> { + for _ in 0..n { + writer.write_all(b" ")?; + } + Ok(()) + } + + #[derive(Debug)] + pub enum OutputType<'a> { + Str(String), + OsStr(&'a OsString), + Integer(i64), + Unsigned(u64), + UnsignedHex(u64), + UnsignedOct(u32), + Float(f64), + Unknown, + } + + #[derive(Default)] + enum QuotingStyle { + Locale, + Shell, + #[default] + ShellEscapeAlways, + Quote, + } + + impl std::str::FromStr for QuotingStyle { + type Err = StatError; + + fn from_str(s: &str) -> Result { + match s { + "locale" => Ok(Self::Locale), + "shell" => Ok(Self::Shell), + "shell-escape-always" => Ok(Self::ShellEscapeAlways), + // The others aren't exposed to the user + _ => Err(StatError::InvalidQuotingStyle { style: s.to_string() }), + } + } + } + + #[derive(Debug, PartialEq, Eq, Clone, Copy)] + enum Precision { + NotSpecified, + NoNumber, + Number(usize), + } + + #[derive(Debug, PartialEq, Eq)] + enum Token { + Char(char), + Byte(u8), + Directive { flag: Flags, width: usize, precision: Precision, format: char }, + } + + trait ScanUtil { + fn scan_num(&self) -> Option<(F, usize)> + where + F: std::str::FromStr; + fn scan_char(&self, radix: u32) -> Option<(char, usize)>; + } + + impl ScanUtil for str { + /// Scans for a number at the beginning of the string + /// Returns the parsed number and the character count + /// Since we only deal with ASCII characters (+, -, 0-9), character count + /// equals byte count + fn scan_num(&self) -> Option<(F, usize)> + where + F: std::str::FromStr, + { + let mut chars = self.chars(); + let count = chars + .next() + .filter(|&c| c.is_ascii_digit() || c == '-' || c == '+') + .map_or(0, |_| 1 + chars.take_while(char::is_ascii_digit).count()); + + if count > 0 { + F::from_str(&self[..count]).ok().map(|x| (x, count)) + } else { + None + } + } + + fn scan_char(&self, radix: u32) -> Option<(char, usize)> { + let count = match radix { + 8 => 3, + 16 => 2, + _ => return None, + }; + let chars = self.chars().enumerate(); + let mut res = 0; + let mut offset = 0; + for (i, c) in chars { + if i >= count { + break; + } + match c.to_digit(radix) { + Some(digit) => { + let tmp = res * radix + digit; + if tmp < 256 { + res = tmp; + } else { + break; + } + }, + None => break, + } + offset = i + 1; + } + if offset > 0 { + Some((res as u8 as char, offset)) + } else { + None + } + } + } + + fn group_num(s: &str) -> Cow<'_, str> { + let is_negative = s.starts_with('-'); + assert!(is_negative || s.chars().take(1).all(|c| c.is_ascii_digit())); + assert!(s.chars().skip(1).all(|c| c.is_ascii_digit())); + if s.len() < 4 { + return s.into(); + } + let mut res = String::with_capacity((s.len() - 1) / 3); + let s = if is_negative { + res.push('-'); + &s[1..] + } else { + s + }; + let mut alone = (s.len() - 1) % 3 + 1; + res.push_str(&s[..alone]); + while alone != s.len() { + res.push(','); + res.push_str(&s[alone..alone + 3]); + alone += 3; + } + res.into() + } + + struct Stater { + follow: bool, + show_fs: bool, + from_user: bool, + files: Vec, + mount_list: OnceCell>>, + mount_list_needed: bool, + default_tokens: Vec, + default_dev_tokens: Vec, + } + + /// Prints a formatted output based on the provided output type, flags, + /// width, and precision. + /// + /// # Arguments + /// + /// * `output` - A reference to the [`OutputType`] enum containing the value + /// to be printed. + /// * `flags` - A Flags struct containing formatting flags. + /// * `width` - The width of the field for the printed output. + /// * `precision` - How many digits of precision, if any. + /// + /// This function delegates the printing process to more specialized + /// functions depending on the output type. + fn print_it(output: &OutputType, flags: Flags, width: usize, precision: Precision) { + // If the precision is given as just '.', the precision is taken to be zero. + // A negative precision is taken as if the precision were omitted. + // This gives the minimum number of digits to appear for d, i, o, u, x, and X + // conversions, the maximum number of characters to be printed from a string + // for s and S conversions. + + // # + // The value should be converted to an "alternate form". + // For o conversions, the first character of the output string is made zero + // (by prefixing a 0 if it was not zero already). For x and X conversions, a + // nonzero result has the string "0x" (or "0X" for X conversions) prepended to + // it. + + // 0 + // The value should be zero padded. + // For d, i, o, u, x, X, a, A, e, E, f, F, g, and G conversions, the converted + // value is padded on the left with zeros rather than blanks. If the 0 and - + // flags both appear, the 0 flag is ignored. If a precision is given with a + // numeric conversion (d, i, o, u, x, and X), the 0 flag is ignored. For other + // conversions, the behavior is undefined. + + // - + // The converted value is to be left adjusted on the field boundary. (The + // default is right justification.) The converted value is padded on the + // right with blanks, rather than on the left with blanks or zeros. + // A - overrides a 0 if both are given. + + // ' ' (a space) + // A blank should be left before a positive number (or empty string) produced by + // a signed conversion. + + // + + // A sign (+ or -) should always be placed before a number produced by a signed + // conversion. By default, a sign is used only for negative numbers. + // A + overrides a space if both are used. + let padding_char = determine_padding_char(flags); + + match output { + OutputType::Str(s) => print_str(s, flags, width, precision), + OutputType::OsStr(s) => print_os_str(s, flags, width, precision), + OutputType::Integer(num) => print_integer(*num, flags, width, precision, padding_char), + OutputType::Unsigned(num) => { + print_unsigned(*num, flags, width, precision, padding_char); + }, + OutputType::UnsignedOct(num) => { + print_unsigned_oct(*num, flags, width, precision, padding_char); + }, + OutputType::UnsignedHex(num) => { + print_unsigned_hex(*num, flags, width, precision, padding_char); + }, + OutputType::Float(num) => { + print_float(*num, flags, width, precision, padding_char); + }, + // pi-uutils: context stdout instead of `print!`. + OutputType::Unknown => { + let _ = write!(pi_uutils_ctx::stdout(), "?"); + }, + } + } + + /// Determines the padding character based on the provided flags and + /// precision. + /// + /// # Arguments + /// + /// * `flags` - A reference to the Flags struct containing formatting flags. + /// + /// # Returns + /// + /// * Padding - An instance of the Padding enum representing the padding + /// character. + fn determine_padding_char(flags: Flags) -> Padding { + if flags.zero && !flags.left { + Padding::Zero + } else { + Padding::Space + } + } + + /// Prints a string value based on the provided flags, width, and precision. + /// + /// # Arguments + /// + /// * `s` - The string to be printed. + /// * `flags` - A reference to the Flags struct containing formatting flags. + /// * `width` - The width of the field for the printed string. + /// * `precision` - How many digits of precision, if any. + fn print_str(s: &str, flags: Flags, width: usize, precision: Precision) { + let s = match precision { + Precision::Number(p) if p < s.len() => &s[..p], + _ => s, + }; + pad_and_print(s, flags.left, width, Padding::Space); + } + + /// Prints a `OsString` value based on the provided flags, width, and + /// precision. It converts the value to bytes and prints them; if that + /// fails, it prints the lossy string version. + /// + /// # Arguments + /// + /// * `s` - The `OsString` to be printed. + /// * `flags` - A reference to the Flags struct containing formatting flags. + /// * `width` - The width of the field for the printed string. + /// * `precision` - How many digits of precision, if any. + fn print_os_str(s: &OsString, flags: Flags, width: usize, precision: Precision) { + // pi-uutils: this module is unix-only, so upstream's `cfg(not(unix))` + // lossy fallback branch is dropped; bytes go to the context stdout. + use std::os::unix::ffi::OsStrExt; + + let bytes = s.as_bytes(); + + if pad_and_print_bytes(pi_uutils_ctx::stdout(), bytes, flags.left, width, precision).is_err() + { + // if an error occurred while trying to print bytes fall back to normal lossy + // string so it can be printed + let fallback_string = s.to_string_lossy(); + print_str(&fallback_string, flags, width, precision); + } + } + + fn quote_file_name(file_name: &str, quoting_style: &QuotingStyle) -> String { + match quoting_style { + QuotingStyle::Locale | QuotingStyle::Shell => { + let escaped = file_name.replace('\'', r"\'"); + format!("'{escaped}'") + }, + QuotingStyle::ShellEscapeAlways => { + let quote = if file_name.contains('\'') { '"' } else { '\'' }; + format!("{quote}{file_name}{quote}") + }, + QuotingStyle::Quote => file_name.to_string(), + } + } + + fn get_quoted_file_name( + display_name: &str, + // pi-uutils: takes the operand resolved against the shell working + // directory for the `readlink` syscall; `display_name` stays as typed. + resolved: &Path, + file_type: FileType, + from_user: bool, + ) -> Result { + // pi-uutils: QUOTING_STYLE comes from the scope environment (the + // shell's exported variables), not the host process environment. + let quoting_style = pi_uutils_ctx::var("QUOTING_STYLE") + .and_then(|style| style.parse().ok()) + .unwrap_or_default(); + + if file_type.is_symlink() { + let quoted_display_name = quote_file_name(display_name, "ing_style); + match fs::read_link(resolved) { + Ok(dst) => { + let quoted_dst = quote_file_name(&dst.to_string_lossy(), "ing_style); + Ok(format!("{quoted_display_name} -> {quoted_dst}")) + }, + Err(e) => { + // pi-uutils: `show_error!` replaced with a context-stderr write. + let _ = writeln!(pi_uutils_ctx::stderr(), "stat: {e}"); + Err(1) + }, + } + } else { + let style = if from_user { + quoting_style + } else { + QuotingStyle::Quote + }; + Ok(quote_file_name(display_name, &style)) + } + } + + fn process_token_filesystem(t: &Token, meta: &StatFs, display_name: &str) { + match *t { + Token::Byte(byte) => write_raw_byte(byte), + // pi-uutils: context stdout instead of `print!`. + Token::Char(c) => { + let _ = write!(pi_uutils_ctx::stdout(), "{c}"); + }, + Token::Directive { flag, width, precision, format } => { + let output = match format { + // free blocks available to non-superuser + 'a' => OutputType::Unsigned(meta.avail_blocks()), + // total data blocks in file system + 'b' => OutputType::Unsigned(meta.total_blocks()), + // total file nodes in file system + 'c' => OutputType::Unsigned(meta.total_file_nodes()), + // free file nodes in file system + 'd' => OutputType::Unsigned(meta.free_file_nodes()), + // free blocks in file system + 'f' => OutputType::Unsigned(meta.free_blocks()), + // file system ID in hex + 'i' => OutputType::UnsignedHex(meta.fsid()), + // maximum length of filenames + 'l' => OutputType::Unsigned(meta.namelen()), + // file name + 'n' => OutputType::Str(display_name.to_string()), + // block size (for faster transfers) + 's' => OutputType::Unsigned(meta.io_size()), + // fundamental block size (for block counts) + 'S' => OutputType::Integer(meta.block_size()), + // file system type in hex + 't' => OutputType::UnsignedHex(meta.fs_type() as u64), + // file system type in human readable form + 'T' => OutputType::Str(pretty_fstype(meta.fs_type()).into()), + _ => OutputType::Unknown, + }; + + print_it(&output, flag, width, precision); + }, + } + } + + /// Prints an integer value based on the provided flags, width, and + /// precision. + /// + /// # Arguments + /// + /// * `num` - The integer value to be printed. + /// * `flags` - A reference to the Flags struct containing formatting flags. + /// * `width` - The width of the field for the printed integer. + /// * `precision` - How many digits of precision, if any. + /// * `padding_char` - The padding character as determined by + /// `determine_padding_char`. + fn print_integer( + num: i64, + flags: Flags, + width: usize, + precision: Precision, + padding_char: Padding, + ) { + let num = num.to_string(); + let arg = if flags.group { + group_num(&num) + } else { + Cow::Borrowed(num.as_str()) + }; + let prefix = if flags.sign { + "+" + } else if flags.space { + " " + } else { + "" + }; + let extended = match precision { + Precision::NotSpecified => format!("{prefix}{arg}"), + Precision::NoNumber => format!("{prefix}{arg}"), + Precision::Number(p) => format!("{prefix}{arg:0>p$}"), + }; + pad_and_print(&extended, flags.left, width, padding_char); + } + + /// Truncate a float to the given number of digits after the decimal point. + fn precision_trunc(num: f64, precision: Precision) -> String { + // GNU `stat` doesn't round, it just seems to truncate to the + // given precision: + // + // $ stat -c "%.5Y" /dev/pts/ptmx + // 1736344012.76399 + // $ stat -c "%.4Y" /dev/pts/ptmx + // 1736344012.7639 + // $ stat -c "%.3Y" /dev/pts/ptmx + // 1736344012.763 + // + // Contrast this with `printf`, which seems to round the + // numbers: + // + // $ printf "%.5f\n" 1736344012.76399 + // 1736344012.76399 + // $ printf "%.4f\n" 1736344012.76399 + // 1736344012.7640 + // $ printf "%.3f\n" 1736344012.76399 + // 1736344012.764 + // + let num_str = num.to_string(); + let n = num_str.len(); + match (num_str.find('.'), precision) { + (None, Precision::NotSpecified) => num_str, + (None, Precision::NoNumber) => num_str, + (None, Precision::Number(0)) => num_str, + (None, Precision::Number(p)) => format!("{num_str}.{zeros}", zeros = "0".repeat(p)), + (Some(i), Precision::NotSpecified) => num_str[..i].to_string(), + (Some(_), Precision::NoNumber) => num_str, + (Some(i), Precision::Number(0)) => num_str[..i].to_string(), + (Some(i), Precision::Number(p)) if p < n - i => num_str[..i + 1 + p].to_string(), + (Some(i), Precision::Number(p)) => { + format!("{num_str}{zeros}", zeros = "0".repeat(p - (n - i - 1))) + }, + } + } + + fn print_float( + num: f64, + flags: Flags, + width: usize, + precision: Precision, + padding_char: Padding, + ) { + let prefix = if flags.sign { + "+" + } else if flags.space { + " " + } else { + "" + }; + let num_str = precision_trunc(num, precision); + let extended = format!("{prefix}{num_str}"); + pad_and_print(&extended, flags.left, width, padding_char); + } + + /// Prints an unsigned integer value based on the provided flags, width, and + /// precision. + /// + /// # Arguments + /// + /// * `num` - The unsigned integer value to be printed. + /// * `flags` - A reference to the Flags struct containing formatting flags. + /// * `width` - The width of the field for the printed unsigned integer. + /// * `precision` - How many digits of precision, if any. + /// * `padding_char` - The padding character as determined by + /// `determine_padding_char`. + fn print_unsigned( + num: u64, + flags: Flags, + width: usize, + precision: Precision, + padding_char: Padding, + ) { + let num = num.to_string(); + let s = if flags.group { + group_num(&num) + } else { + Cow::Borrowed(num.as_str()) + }; + let s = match precision { + Precision::NotSpecified => s, + Precision::NoNumber => s, + Precision::Number(p) => format!("{s:0>p$}").into(), + }; + pad_and_print(&s, flags.left, width, padding_char); + } + + /// Prints an unsigned octal integer value based on the provided flags, + /// width, and precision. + /// + /// # Arguments + /// + /// * `num` - The unsigned octal integer value to be printed. + /// * `flags` - A reference to the Flags struct containing formatting flags. + /// * `width` - The width of the field for the printed unsigned octal + /// integer. + /// * `precision` - How many digits of precision, if any. + /// * `padding_char` - The padding character as determined by + /// `determine_padding_char`. + fn print_unsigned_oct( + num: u32, + flags: Flags, + width: usize, + precision: Precision, + padding_char: Padding, + ) { + let prefix = if flags.alter { "0" } else { "" }; + let s = match precision { + Precision::NotSpecified => format!("{prefix}{num:o}"), + Precision::NoNumber => format!("{prefix}{num:o}"), + Precision::Number(p) => format!("{prefix}{num:0>p$o}"), + }; + pad_and_print(&s, flags.left, width, padding_char); + } + + /// Prints an unsigned hexadecimal integer value based on the provided flags, + /// width, and precision. + /// + /// # Arguments + /// + /// * `num` - The unsigned hexadecimal integer value to be printed. + /// * `flags` - A reference to the Flags struct containing formatting flags. + /// * `width` - The width of the field for the printed unsigned hexadecimal + /// integer. + /// * `precision` - How many digits of precision, if any. + /// * `padding_char` - The padding character as determined by + /// `determine_padding_char`. + fn print_unsigned_hex( + num: u64, + flags: Flags, + width: usize, + precision: Precision, + padding_char: Padding, + ) { + let prefix = if flags.alter { "0x" } else { "" }; + let s = match precision { + Precision::NotSpecified => format!("{prefix}{num:x}"), + Precision::NoNumber => format!("{prefix}{num:x}"), + Precision::Number(p) => format!("{prefix}{num:0>p$x}"), + }; + pad_and_print(&s, flags.left, width, padding_char); + } + + fn write_raw_byte(byte: u8) { + // pi-uutils: context stdout instead of the process stdout, and no + // `unwrap` — an in-process builtin must not panic on a broken pipe. + let _ = pi_uutils_ctx::stdout().write_all(&[byte]); + } + + impl Stater { + fn process_flags(chars: &[char], i: &mut usize, bound: usize, flag: &mut Flags) { + while *i < bound { + match chars[*i] { + '#' => flag.alter = true, + '0' => flag.zero = true, + '-' => flag.left = true, + ' ' => flag.space = true, + // This is not documented but the behavior seems to be + // the same as a space. For example `stat -c "%I5s" f` + // prints " 0". + 'I' => flag.space = true, + '+' => flag.sign = true, + '\'' => flag.group = true, + _ => break, + } + *i += 1; + } + } + + /// Converts a character index to a byte index in a UTF-8 string + /// This is necessary because Rust strings are UTF-8 encoded, so character + /// positions don't always align with byte positions for multi-byte + /// characters + fn char_index_to_byte_index(format_str: &str, char_index: usize) -> usize { + format_str + .char_indices() + .nth(char_index) + .map_or(format_str.len(), |(byte_idx, _)| byte_idx) + } + + fn handle_percent_case( + chars: &[char], + i: &mut usize, + bound: usize, + format_str: &str, + ) -> UResult { + let old = *i; + + *i += 1; + if *i >= bound { + return Ok(Token::Char('%')); + } + if chars[*i] == '%' { + return Ok(Token::Char('%')); + } + + let mut flag = Flags::default(); + + Self::process_flags(chars, i, bound, &mut flag); + + let mut width = 0; + let mut precision = Precision::NotSpecified; + let mut j = *i; + + let j_byte = Self::char_index_to_byte_index(format_str, j); + if let Some((field_width, offset)) = format_str[j_byte..].scan_num::() { + width = field_width; + j += offset; + + // Reject directives like `%` by checking if width has been parsed. + if j >= bound || chars[j] == '%' { + let invalid_directive: String = chars[old..=j.min(bound - 1)].iter().collect(); + return Err(USimpleError::new( + 1, + StatError::InvalidDirective { directive: invalid_directive.quote().to_string() } + .to_string(), + )); + } + } + check_bound(format_str, bound, old, j)?; + + if chars[j] == '.' { + j += 1; + check_bound(format_str, bound, old, j)?; + + let j_byte = Self::char_index_to_byte_index(format_str, j); + match format_str[j_byte..].scan_num::() { + Some((value, offset)) => { + if value >= 0 { + precision = Precision::Number(value as usize); + } + j += offset; + }, + None => precision = Precision::NoNumber, + } + check_bound(format_str, bound, old, j)?; + } + + *i = j; + + // Check for multi-character specifiers (e.g., `%Hd`, `%Lr`) + if *i + 1 < bound + && let Some(&next_char) = chars.get(*i + 1) + && (chars[*i] == 'H' || chars[*i] == 'L') + && (next_char == 'd' || next_char == 'r') + { + flag.major = chars[*i] == 'H'; + flag.minor = chars[*i] == 'L'; + *i += 1; + return Ok(Token::Directive { flag, width, precision, format: next_char }); + } + + Ok(Token::Directive { flag, width, precision, format: chars[*i] }) + } + + fn handle_escape_sequences( + chars: &[char], + i: &mut usize, + bound: usize, + format_str: &str, + ) -> Token { + *i += 1; + if *i >= bound { + // pi-uutils: `show_warning!` replaced with a context-stderr + // write; message literalized from locales/en-US.ftl. + let _ = writeln!(pi_uutils_ctx::stderr(), "stat: warning: backslash at end of format"); + return Token::Char('\\'); + } + match chars[*i] { + 'a' => Token::Byte(0x07), // BEL + 'b' => Token::Byte(0x08), // Backspace + 'f' => Token::Byte(0x0c), // Form feed + 'n' => Token::Byte(0x0a), // Line feed + 'r' => Token::Byte(0x0d), // Carriage return + 't' => Token::Byte(0x09), // Horizontal tab + '\\' => Token::Byte(b'\\'), // Backslash + '\'' => Token::Byte(b'\''), // Single quote + '"' => Token::Byte(b'"'), // Double quote + '0'..='7' => { + // Parse octal escape sequence (up to 3 digits) + let mut value = 0u8; + let mut count = 0; + while *i < bound && count < 3 { + if let Some(digit) = chars[*i].to_digit(8) { + value = value * 8 + digit as u8; + *i += 1; + count += 1; + } else { + break; + } + } + *i -= 1; // Adjust index to account for the outer loop increment + Token::Byte(value) + }, + 'x' => { + // Parse hexadecimal escape sequence (\xNN format) + // Uses UTF-8 safe byte indexing to handle multi-byte characters properly + if *i + 1 < bound { + let byte_index = Self::char_index_to_byte_index(format_str, *i + 1); + if let Some((c, offset)) = format_str[byte_index..].scan_char(16) { + *i += offset; + Token::Byte(c as u8) + } else { + // pi-uutils: `show_warning!` replaced with a + // context-stderr write. + let _ = writeln!( + pi_uutils_ctx::stderr(), + "stat: warning: unrecognized escape '\\x'" + ); + Token::Byte(b'x') + } + } else { + // pi-uutils: `show_warning!` replaced with a + // context-stderr write. + let _ = writeln!( + pi_uutils_ctx::stderr(), + "stat: warning: incomplete hex escape '\\x'" + ); + Token::Byte(b'x') + } + }, + other => { + // pi-uutils: `show_warning!` replaced with a context-stderr + // write. + let _ = writeln!( + pi_uutils_ctx::stderr(), + "stat: warning: unrecognized escape '\\{other}'" + ); + Token::Byte(other as u8) + }, + } + } + + fn generate_tokens(format_str: &str, use_printf: bool) -> UResult> { + let mut tokens = Vec::new(); + let chars = format_str.chars().collect::>(); + let bound = chars.len(); + let mut i = 0; + while i < bound { + match chars.get(i) { + Some('%') => { + tokens.push(Self::handle_percent_case(&chars, &mut i, bound, format_str)?); + }, + Some('\\') => { + if use_printf { + tokens.push(Self::handle_escape_sequences(&chars, &mut i, bound, format_str)); + } else { + tokens.push(Token::Char('\\')); + } + }, + Some(c) => tokens.push(Token::Char(*c)), + None => break, + } + i += 1; + } + if !use_printf && !format_str.ends_with('\n') { + tokens.push(Token::Char('\n')); + } + Ok(tokens) + } + + fn populate_mount_list() -> UResult> { + let mut mount_list = read_fs_list() + .map_err(|e| { + USimpleError::new( + e.code(), + StatError::CannotReadFilesystem { error: e.to_string() }.to_string(), + ) + })? + .iter() + .map(|mi| mi.mount_dir.clone()) + .collect::>(); + + // Reverse sort. The longer comes first. + mount_list.sort(); + mount_list.reverse(); + + Ok(mount_list) + } + + fn new(matches: &ArgMatches) -> UResult { + let files: Vec = matches + .get_many::(options::FILES) + .map(|v| v.map(OsString::from).collect()) + .unwrap_or_default(); + if files.is_empty() { + return Err(Box::new(StatError::MissingOperand) as Box); + } + let format_str = if matches.contains_id(options::PRINTF) { + matches + .get_one::(options::PRINTF) + .expect("Invalid format string") + } else { + matches + .get_one::(options::FORMAT) + .map_or("", |s| s.as_str()) + }; + + let use_printf = matches.contains_id(options::PRINTF); + let terse = matches.get_flag(options::TERSE); + let show_fs = matches.get_flag(options::FILE_SYSTEM); + + let default_tokens = if format_str.is_empty() { + Self::generate_tokens(&Self::default_format(show_fs, terse, false), use_printf)? + } else { + Self::generate_tokens(format_str, use_printf)? + }; + let default_dev_tokens = + Self::generate_tokens(&Self::default_format(show_fs, terse, true), use_printf)?; + + // mount points aren't displayed when showing filesystem information, or + // whenever the format string does not request the mount point. + let mount_list_needed = !show_fs + && default_tokens + .iter() + .any(|tok| matches!(tok, Token::Directive { format: 'm', .. })); + + Ok(Self { + follow: matches.get_flag(options::DEREFERENCE), + show_fs, + from_user: !format_str.is_empty(), + files, + mount_list: OnceCell::new(), + mount_list_needed, + default_tokens, + default_dev_tokens, + }) + } + + fn find_mount_point>(&self, p: P) -> Option<&OsString> { + if !self.mount_list_needed { + return None; + } + + let mount_list = self.mount_list.get_or_init(|| { + match Self::populate_mount_list() { + Ok(list) => Some(list), + Err(e) => { + // Show warning like GNU does when mount information cannot be read + // pi-uutils: `show_warning!` replaced with a + // context-stderr write. + let _ = writeln!( + pi_uutils_ctx::stderr(), + "stat: warning: cannot read table of mounted file systems: {e}" + ); + None + }, + } + }); + + let path = p.as_ref().canonicalize().ok()?; + mount_list + .as_ref()? + .iter() + .find(|root| path.starts_with(root)) + } + + fn exec(&self) -> i32 { + let mut stdin_is_fifo = false; + if let Ok(md) = fs::metadata("/dev/stdin") { + stdin_is_fifo = md.file_type().is_fifo(); + } + + let mut ret = 0; + for f in &self.files { + ret |= self.do_stat(f, stdin_is_fifo); + } + ret + } + + fn process_token_files( + &self, + t: &Token, + meta: &Metadata, + display_name: &str, + // pi-uutils: takes the operand resolved against the shell working + // directory for the `%m`/`%N` syscalls (upstream passed the raw + // operand); display output keeps `display_name` as typed. The + // SELinux `follow_symbolic_links` parameter is dropped along with + // SELinux support. + resolved: &Path, + file_type: FileType, + from_user: bool, + ) -> Result<(), i32> { + match *t { + Token::Byte(byte) => write_raw_byte(byte), + // pi-uutils: context stdout instead of `print!`. + Token::Char(c) => { + let _ = write!(pi_uutils_ctx::stdout(), "{c}"); + }, + + Token::Directive { flag, width, precision, format } => { + let output = match format { + // access rights in octal + 'a' => OutputType::UnsignedOct(0o7777 & meta.mode()), + // access rights in human readable form + 'A' => OutputType::Str(display_permissions(meta, true)), + // number of blocks allocated (see %B) + 'b' => OutputType::Unsigned(meta.blocks()), + + // the size in bytes of each block reported by %b + // FIXME: blocksize differs on various platform + // See coreutils/gnulib/lib/stat-size.h ST_NBLOCKSIZE // + // spell-checker:disable-line + 'B' => OutputType::Unsigned(512), + // SELinux security context string + // pi-uutils: SELinux support is dropped; this is + // upstream's non-SELinux fallback string. + 'C' => OutputType::Str("unsupported for this operating system".to_string()), + // device number in decimal + 'd' if flag.major => OutputType::Unsigned(major(meta.dev() as _) as u64), + 'd' if flag.minor => OutputType::Unsigned(minor(meta.dev() as _) as u64), + 'd' => OutputType::Unsigned(meta.dev()), + // device number in hex + 'D' => OutputType::UnsignedHex(meta.dev()), + // raw mode in hex + 'f' => OutputType::UnsignedHex(meta.mode() as u64), + // file type + 'F' => OutputType::Str(pretty_filetype(meta.mode() as mode_t, meta.len())), + // group ID of owner + 'g' => OutputType::Unsigned(meta.gid() as u64), + // group name of owner + 'G' => { + let group_name = + entries::gid2grp(meta.gid()).unwrap_or_else(|_| "UNKNOWN".to_owned()); + OutputType::Str(group_name) + }, + // number of hard links + 'h' => OutputType::Unsigned(meta.nlink()), + // inode number + 'i' => OutputType::Unsigned(meta.ino()), + // mount point + 'm' => match self.find_mount_point(resolved) { + Some(s) => OutputType::OsStr(s), + None => OutputType::Str(String::new()), + }, + // file name + 'n' => OutputType::Str(display_name.to_string()), + // quoted file name with dereference if symbolic link + 'N' => { + let file_name = + get_quoted_file_name(display_name, resolved, file_type, from_user)?; + OutputType::Str(file_name) + }, + // optimal I/O transfer size hint + 'o' => OutputType::Unsigned(meta.blksize()), + // total size, in bytes + 's' => OutputType::Integer(meta.len() as i64), + // major device type in hex, for character/block device special + // files + 't' => OutputType::UnsignedHex(major(meta.rdev() as _) as u64), + // minor device type in hex, for character/block device special + // files + 'T' => OutputType::UnsignedHex(minor(meta.rdev() as _) as u64), + // user ID of owner + 'u' => OutputType::Unsigned(meta.uid() as u64), + // user name of owner + 'U' => { + let user_name = + entries::uid2usr(meta.uid()).unwrap_or_else(|_| "UNKNOWN".to_owned()); + OutputType::Str(user_name) + }, + + // time of file birth, human-readable; - if unknown + 'w' => OutputType::Str(pretty_time(meta, MetadataTimeField::Birth)), + + // time of file birth, seconds since Epoch; 0 if unknown + 'W' => OutputType::Integer( + metadata_get_time(meta, MetadataTimeField::Birth) + .map_or(0, |x| system_time_to_sec(x).0), + ), + + // time of last access, human-readable + 'x' => OutputType::Str(pretty_time(meta, MetadataTimeField::Access)), + // time of last access, seconds since Epoch + 'X' => { + let (sec, nsec) = metadata_get_time(meta, MetadataTimeField::Access) + .map_or((0, 0), system_time_to_sec); + OutputType::Float(sec as f64 + nsec as f64 / 1_000_000_000.0) + }, + // time of last data modification, human-readable + 'y' => OutputType::Str(pretty_time(meta, MetadataTimeField::Modification)), + // time of last data modification, seconds since Epoch + 'Y' => { + let (sec, nsec) = metadata_get_time(meta, MetadataTimeField::Modification) + .map_or((0, 0), system_time_to_sec); + OutputType::Float(sec as f64 + nsec as f64 / 1_000_000_000.0) + }, + // time of last status change, human-readable + 'z' => OutputType::Str(pretty_time(meta, MetadataTimeField::Change)), + // time of last status change, seconds since Epoch + 'Z' => { + let (sec, nsec) = metadata_get_time(meta, MetadataTimeField::Change) + .map_or((0, 0), system_time_to_sec); + OutputType::Float(sec as f64 + nsec as f64 / 1_000_000_000.0) + }, + 'R' => OutputType::UnsignedHex(meta.rdev()), + 'r' if flag.major => OutputType::Unsigned(major(meta.rdev() as _) as u64), + 'r' if flag.minor => OutputType::Unsigned(minor(meta.rdev() as _) as u64), + 'r' => OutputType::Unsigned(meta.rdev()), + _ => OutputType::Unknown, + }; + print_it(&output, flag, width, precision); + }, + } + Ok(()) + } + + fn do_stat(&self, file: &OsStr, stdin_is_fifo: bool) -> i32 { + let display_name = file.to_string_lossy(); + let file = if display_name == "-" { + if self.show_fs { + // pi-uutils: `show_error!` replaced with a context-stderr + // write. + let _ = + writeln!(pi_uutils_ctx::stderr(), "stat: {}", StatError::StdinFilesystemMode); + return 1; + } + if let Ok(p) = Path::new("/dev/stdin").canonicalize() { + p.into_os_string() + } else { + OsString::from("/dev/stdin") + } + } else { + OsString::from(file) + }; + // pi-uutils: resolve the operand against the shell working + // directory for every syscall below; `display_name` keeps the + // operand as typed for `%n` and error messages. + let resolved = pi_uutils_ctx::resolve(&file); + if self.show_fs { + match statfs(resolved.as_os_str()) { + Ok(meta) => { + let tokens = &self.default_tokens; + + // Usage + for t in tokens { + process_token_filesystem(t, &meta, &display_name); + } + }, + Err(error) => { + // pi-uutils: `show_error!` replaced with a + // context-stderr write. + let _ = writeln!( + pi_uutils_ctx::stderr(), + "stat: {}", + StatError::CannotReadFilesystemInfo { + file: display_name.quote().to_string(), + error, + } + ); + return 1; + }, + } + } else { + let follow_symbolic_links = self.follow || stdin_is_fifo && display_name == "-"; + let result = if follow_symbolic_links { + fs::metadata(&resolved) + } else { + fs::symlink_metadata(&resolved) + }; + match result { + Ok(meta) => { + let file_type = meta.file_type(); + let tokens = if self.from_user + || !(file_type.is_char_device() || file_type.is_block_device()) + { + &self.default_tokens + } else { + &self.default_dev_tokens + }; + + for t in tokens { + if let Err(code) = self.process_token_files( + t, + &meta, + &display_name, + &resolved, + file_type, + self.from_user, + ) { + return code; + } + } + }, + Err(e) => { + // pi-uutils: `show_error!` replaced with a + // context-stderr write. + let _ = writeln!(pi_uutils_ctx::stderr(), "stat: {}", StatError::CannotStat { + file: display_name.quote().to_string(), + error: e.to_string(), + }); + return 1; + }, + } + } + 0 + } + + fn default_format(show_fs: bool, terse: bool, show_dev_type: bool) -> String { + // SELinux related format is *ignored* + // pi-uutils: `translate!` word lookups literalized from + // locales/en-US.ftl. + + if show_fs { + if terse { + "%n %i %l %t %s %S %b %f %a %c %d\n".into() + } else { + " File: \"%n\"\n ID: %-8i Namelen: %-7l Type: %T\nBlock size: %-10s Fundamental \ + block size: %S\nBlocks: Total: %-10b Free: %-10f Available: %a\nInodes: Total: \ + %-10c Free: %d\n" + .into() + } + } else if terse { + "%n %s %b %f %u %g %D %i %h %t %T %X %Y %Z %W %o\n".into() + } else { + let device_line = if show_dev_type { + "Device: %Hd,%Ld\tInode: %-10i Links: %-5h Device type: %t,%T\n" + } else { + "Device: %Hd,%Ld\tInode: %-10i Links: %h\n" + }; + + format!( + " File: %N\n size: %-10s\tBlocks: %-10b IO Block: %-6o %F\n{device_line}Access: \ + (%04a/%10.10A) Uid: (%5u/%8U) Gid: (%5g/%8G)\nAccess: %x\nModify: %y\nChange: \ + %z\n Birth: %w\n" + ) + } + } + } + + /// In-process builtin entry point. Unlike upstream's `uumain`, this parses + /// the arguments directly (without the uucore clap-localization helper that + /// would terminate the process), renders clap help/usage/version to the + /// context streams, and maps the `UResult` to an exit code, so it is safe + /// to run inside the host shell process. + pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match stat_main(&matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + // pi-uutils: `do_stat` already reports its errors to the + // context stderr and surfaces a bare exit-code error that + // renders to an empty message; don't emit a dangling + // "stat: " prefix for it. + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "stat: {msg}"); + } + if code == 0 { 1 } else { code } + }, + } + } + + fn stat_main(matches: &ArgMatches) -> UResult<()> { + let stater = Stater::new(matches)?; + let exit_status = stater.exec(); + if exit_status == 0 { + Ok(()) + } else { + Err(exit_status.into()) + } + } + + pub fn uu_app() -> Command { + Command::new("stat") + .version(uucore::crate_version!()) + .about(ABOUT) + .after_help(AFTER_HELP) + .override_usage(format_usage(USAGE)) + .infer_long_args(true) + .arg( + Arg::new(options::DEREFERENCE) + .short('L') + .long(options::DEREFERENCE) + .help("follow links") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::FILE_SYSTEM) + .short('f') + .long(options::FILE_SYSTEM) + .help("display file system status instead of file status") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::TERSE) + .short('t') + .long(options::TERSE) + .help("print the information in terse form") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::FORMAT) + .short('c') + .long(options::FORMAT) + .help( + "use the specified FORMAT instead of the default;\noutput a newline after each \ + use of FORMAT", + ) + .value_name("FORMAT"), + ) + .arg( + Arg::new(options::PRINTF) + .long(options::PRINTF) + .value_name("FORMAT") + .help( + "like --format, but interpret backslash escapes,\nand do not output a mandatory \ + trailing newline;\nif you want a newline, include \\n in FORMAT", + ), + ) + .arg( + Arg::new(options::FILES) + .action(ArgAction::Append) + .value_parser(ValueParser::os_string()) + .value_hint(clap::ValueHint::FilePath), + ) + } + + const PRETTY_DATETIME_FORMAT: &str = "%Y-%m-%d %H:%M:%S.%N %z"; + + fn pretty_time(meta: &Metadata, md_time_field: MetadataTimeField) -> String { + if let Some(time) = metadata_get_time(meta, md_time_field) { + let mut tmp = Vec::new(); + if format_system_time( + &mut tmp, + time, + PRETTY_DATETIME_FORMAT, + FormatSystemTimeFallback::Float, + ) + .is_ok() + { + return String::from_utf8(tmp).unwrap(); + } + } + "-".to_string() + } + + /// Upstream format-parser unit tests, kept because the token parser is the + /// most intricate part of the utility and the print paths were repatched. + #[cfg(test)] + mod unit_tests { + use super::{Flags, Precision, ScanUtil, Stater, Token, group_num, precision_trunc}; + + #[test] + fn test_scanners() { + assert_eq!(Some((-5, 2)), "-5zxc".scan_num::()); + assert_eq!(Some((51, 2)), "51zxc".scan_num::()); + assert_eq!(Some((192, 4)), "+192zxc".scan_num::()); + assert_eq!(None, "z192zxc".scan_num::()); + + assert_eq!(Some(('a', 3)), "141zxc".scan_char(8)); + assert_eq!(Some(('\n', 2)), "12qzxc".scan_char(8)); // spell-checker:disable-line + assert_eq!(Some(('\r', 1)), "dqzxc".scan_char(16)); // spell-checker:disable-line + assert_eq!(None, "z2qzxc".scan_char(8)); // spell-checker:disable-line + } + + #[test] + fn test_group_num() { + assert_eq!("12,379,821,234", group_num("12379821234")); + assert_eq!("821,234", group_num("821234")); + assert_eq!("1,234", group_num("1234")); + assert_eq!("234", group_num("234")); + assert_eq!("", group_num("")); + assert_eq!("-5", group_num("-5")); + assert_eq!("-1,234", group_num("-1234")); + } + + #[test] + fn normal_format() { + let s = "%'010.2ac%-#5.w\n"; + let expected = vec![ + Token::Directive { + flag: Flags { group: true, zero: true, ..Default::default() }, + width: 10, + precision: Precision::Number(2), + format: 'a', + }, + Token::Char('c'), + Token::Directive { + flag: Flags { left: true, alter: true, ..Default::default() }, + width: 5, + precision: Precision::NoNumber, + format: 'w', + }, + Token::Char('\n'), + ]; + assert_eq!(&expected, &Stater::generate_tokens(s, false).unwrap()); + } + + #[test] + fn printf_format() { + let s = r#"%-# 15a\t\r\"\\\a\b\x1B\f\x0B%+020.-23w\x12\167\132\112\n"#; + let expected = vec![ + Token::Directive { + flag: Flags { left: true, alter: true, space: true, ..Default::default() }, + width: 15, + precision: Precision::NotSpecified, + format: 'a', + }, + Token::Byte(b'\t'), + Token::Byte(b'\r'), + Token::Byte(b'"'), + Token::Byte(b'\\'), + Token::Byte(b'\x07'), + Token::Byte(b'\x08'), + Token::Byte(b'\x1B'), + Token::Byte(b'\x0C'), + Token::Byte(b'\x0B'), + Token::Directive { + flag: Flags { sign: true, zero: true, ..Default::default() }, + width: 20, + precision: Precision::NotSpecified, + format: 'w', + }, + Token::Byte(b'\x12'), + Token::Byte(b'w'), + Token::Byte(b'Z'), + Token::Byte(b'J'), + Token::Byte(b'\n'), + ]; + assert_eq!(&expected, &Stater::generate_tokens(s, true).unwrap()); + } + + #[test] + fn test_precision_trunc() { + assert_eq!(precision_trunc(123.456, Precision::NotSpecified), "123"); + assert_eq!(precision_trunc(123.456, Precision::NoNumber), "123.456"); + assert_eq!(precision_trunc(123.456, Precision::Number(0)), "123"); + assert_eq!(precision_trunc(123.456, Precision::Number(1)), "123.4"); + assert_eq!(precision_trunc(123.456, Precision::Number(5)), "123.45600"); + } + } +} + +#[cfg(all(test, unix))] +mod tests { + use std::{collections::HashMap, ffi::OsString, fs, io::Write, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::run; + + fn run_in(cwd: PathBuf, args: Vec<&str>) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd, + env: HashMap::new(), + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; + + let argv: Vec = std::iter::once("stat") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + /// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var, + /// which mount-point/canonicalize logic would otherwise expand + /// mid-assertion). + fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) { + let dir = tempfile::tempdir().unwrap(); + let canon = fs::canonicalize(dir.path()).unwrap(); + (dir, canon) + } + + #[test] + fn resolves_relative_operand_against_scope_cwd() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("data.bin"), b"hello world!").unwrap(); + + // Relative operand + scope cwd differing from the process cwd: only the + // call-site `pi_uutils_ctx::resolve` patch makes this find the file. + let (code, stdout, stderr) = run_in(root, vec!["-c", "%s", "data.bin"]); + assert_eq!(code, 0); + assert_eq!(stdout, "12\n"); + assert_eq!(stderr, ""); + } + + #[test] + fn percent_n_prints_operand_as_typed() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("data.bin"), b"x").unwrap(); + + // GNU prints the file name exactly as typed, not the resolved path. + let (code, stdout, stderr) = run_in(root, vec!["-c", "%n", "data.bin"]); + assert_eq!(code, 0); + assert_eq!(stdout, "data.bin\n"); + assert_eq!(stderr, ""); + } + + #[test] + fn dereference_switches_between_link_and_target() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("target"), b"abc").unwrap(); + std::os::unix::fs::symlink("target", root.join("link")).unwrap(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["-c", "%F", "link"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "symbolic link\n", "")); + + let (code, stdout, stderr) = run_in(root, vec!["-L", "-c", "%F", "link"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "regular file\n", "")); + } + + #[test] + fn nonexistent_file_reports_cannot_stat() { + let (_dir, root) = canonical_tempdir(); + + let (code, stdout, stderr) = run_in(root, vec!["missing"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert!(stderr.starts_with("stat: cannot stat 'missing':"), "unexpected stderr: {stderr:?}"); + } + + #[test] + fn file_system_mode_succeeds() { + let (_dir, root) = canonical_tempdir(); + + let (code, stdout, stderr) = run_in(root, vec!["-f", "-c", "%S", "."]); + assert_eq!(code, 0); + assert_eq!(stderr, ""); + assert!( + stdout.trim_end().parse::().is_ok(), + "fundamental block size should be numeric: {stdout:?}" + ); + } + + #[test] + fn printf_controls_trailing_newline_and_escapes() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("data.bin"), b"hello world!").unwrap(); + + // --printf emits no mandatory trailing newline... + let (code, stdout, _) = run_in(root.clone(), vec!["--printf", "%s", "data.bin"]); + assert_eq!((code, stdout.as_str()), (0, "12")); + + // ...but interprets backslash escapes. + let (code, stdout, _) = run_in(root, vec!["--printf", r"%s\t%n\n", "data.bin"]); + assert_eq!((code, stdout.as_str()), (0, "12\tdata.bin\n")); + } + + #[test] + fn terse_prints_name_as_typed_and_size() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("data.bin"), b"hello world!").unwrap(); + + let (code, stdout, stderr) = run_in(root, vec!["-t", "data.bin"]); + assert_eq!(code, 0); + assert_eq!(stderr, ""); + let fields: Vec<&str> = stdout.split_whitespace().collect(); + assert_eq!(fields[0], "data.bin"); + assert_eq!(fields[1], "12"); + assert_eq!(fields.len(), 16, "terse format has 16 fields: {stdout:?}"); + } + + #[test] + fn missing_operand_is_error() { + let (code, stdout, stderr) = run_in(PathBuf::from("."), vec![]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert!(stderr.contains("stat: missing operand"), "unexpected stderr: {stderr:?}"); + assert!(stderr.contains("Try 'stat --help'"), "unexpected stderr: {stderr:?}"); + } + + #[test] + fn help_renders_to_scope_stdout() { + let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["--help"]); + assert_eq!(code, 0); + assert!(stdout.contains("Usage:")); + assert!(stdout.contains("file system status")); + assert_eq!(stderr, ""); + } +} diff --git a/crates/vendor/uu-tac/Cargo.toml b/crates/vendor/uu-tac/Cargo.toml new file mode 100644 index 000000000..c7bd7c526 --- /dev/null +++ b/crates/vendor/uu-tac/Cargo.toml @@ -0,0 +1,26 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/tac), patched to resolve +# path arguments against the shell working directory and route I/O through +# pi-uutils-ctx so it can run in-process as a shell builtin. See src/tac.rs +# for the patch markers (`pi-uutils:` comments). +[package] +name = "uu_tac" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "tac ~ (uutils) concatenate and display input lines in reverse order (vendored + patched for in-process embedding)" + +[lib] +path = "src/tac.rs" + +[dependencies] +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +memchr = "2.7.4" +memmap2 = "0.9" +regex = "1.11" +thiserror = "2.0.3" +uucore = "0.8.0" +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[dev-dependencies] +parking_lot = "0.12" +tempfile = "3" diff --git a/crates/vendor/uu-tac/LICENSE b/crates/vendor/uu-tac/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-tac/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-tac/src/error.rs b/crates/vendor/uu-tac/src/error.rs new file mode 100644 index 000000000..cc58fd224 --- /dev/null +++ b/crates/vendor/uu-tac/src/error.rs @@ -0,0 +1,47 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. +//! Errors returned by tac during processing of a file. + +// pi-uutils: vendored from uutils/coreutils 0.8.0; `translate!` strings are +// literalized with the en-US locale text. + +use std::ffi::OsString; + +use thiserror::Error; +use uucore::{ + display::Quotable, + error::{UError, strip_errno}, +}; + +#[derive(Debug, Error)] +pub enum TacError { + /// A regular expression given by the user is invalid. + #[error("invalid regular expression: {0}")] + InvalidRegex(regex::Error), + /// An error opening a file for reading. + /// + /// The parameters are the name of the file and the underlying + /// [`std::io::Error`] that caused this error. + #[error("failed to open {} for reading: {}", .0.quote(), strip_errno(.1))] + OpenError(OsString, std::io::Error), + /// An error reading the contents of a file or stdin. + /// + /// The parameters are the name of the file and the underlying + /// [`std::io::Error`] that caused this error. + #[error("{}: read error: {}", .0.maybe_quote(), strip_errno(.1))] + ReadError(OsString, std::io::Error), + /// An error writing the (reversed) contents of a file or stdin. + /// + /// The parameter is the underlying [`std::io::Error`] that caused + /// this error. + #[error("failed to write to stdout: {}", strip_errno(.0))] + WriteError(std::io::Error), +} + +impl UError for TacError { + fn code(&self) -> i32 { + 1 + } +} diff --git a/crates/vendor/uu-tac/src/tac.rs b/crates/vendor/uu-tac/src/tac.rs new file mode 100644 index 000000000..90c124e61 --- /dev/null +++ b/crates/vendor/uu-tac/src/tac.rs @@ -0,0 +1,634 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +// spell-checker:ignore (ToDO) sbytes slen dlen memmem memmap Mmap mmap SIGBUS + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. FILE operands resolve against the shell working directory +// via `pi_uutils_ctx::resolve` at the open/mmap call site (the original operand +// is kept for error messages), `-`/no-operand read the context stdin, output is +// written through the context stdout, recoverable per-file errors go to the +// context stderr with `pi_uutils_ctx::set_exit_code` (upstream `show!`), the +// `translate!` strings are literalized, and the process-global signal handling +// plus the stdin mmap/tempfile buffering (which target the process stdin fd) +// are removed. + +mod error; + +use std::{ + ffi::{OsStr, OsString}, + fs::File, + io::{BufWriter, Read, Write}, +}; + +use clap::{Arg, ArgAction, ArgMatches, Command}; +use memchr::memmem; +use memmap2::Mmap; +use pi_uutils_ctx::format_usage; +use uucore::error::UResult; + +use crate::error::TacError; + +mod options { + pub static BEFORE: &str = "before"; + pub static REGEX: &str = "regex"; + pub static SEPARATOR: &str = "separator"; + pub static FILE: &str = "file"; +} + +/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the +/// arguments directly (without the uucore clap-localization helper that would +/// terminate the process), renders clap help/usage/version to the context +/// streams, and maps the `UResult` to an exit code, so it is safe to run inside +/// the host shell process. +pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match tac_main(&matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "tac: {msg}"); + } + if code == 0 { 1 } else { code } + }, + } +} + +fn tac_main(matches: &ArgMatches) -> UResult<()> { + let before = matches.get_flag(options::BEFORE); + let regex = matches.get_flag(options::REGEX); + let raw_separator = matches + .get_one::(options::SEPARATOR) + .map_or(OsStr::new("\n"), |s| s.as_os_str()); + + let separator = if raw_separator.is_empty() { + OsStr::new("\0") + } else { + raw_separator + }; + + let files: Vec = match matches.get_many::(options::FILE) { + Some(v) => v.cloned().collect(), + None => vec![OsString::from("-")], + }; + + tac(&files, before, regex, separator) +} + +pub fn uu_app() -> Command { + Command::new("tac") + .version(uucore::crate_version!()) + .override_usage(format_usage("tac [OPTION]... [FILE]...")) + .about("Write each file to standard output, last line first.") + .infer_long_args(true) + .arg( + Arg::new(options::BEFORE) + .short('b') + .long(options::BEFORE) + .help("attach the separator before instead of after") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::REGEX) + .short('r') + .long(options::REGEX) + .help("interpret the sequence as a regular expression") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::SEPARATOR) + .short('s') + .long(options::SEPARATOR) + .help("use STRING as the separator instead of newline") + .value_parser(clap::value_parser!(OsString)) + .value_name("STRING"), + ) + .arg( + Arg::new(options::FILE) + .hide(true) + .action(ArgAction::Append) + .value_parser(clap::value_parser!(OsString)) + .value_hint(clap::ValueHint::FilePath), + ) +} + +/// pi-uutils: replacement for upstream's `show!` — reports a recoverable +/// per-file error to the context stderr and accumulates a non-zero exit code +/// while processing continues with the next operand. +fn show(err: &TacError) { + let _ = writeln!(pi_uutils_ctx::stderr(), "tac: {err}"); + pi_uutils_ctx::set_exit_code(1); +} + +/// Print lines of a buffer in reverse, with line separator given as a regex. +/// +/// `data` contains the bytes of the file. +/// +/// `pattern` is the regular expression given as a +/// [`regex::bytes::Regex`] (not a [`regex::Regex`], since the input is +/// given as a slice of bytes). If `before` is `true`, then each match +/// of this pattern in `data` is interpreted as the start of a line. If +/// `before` is `false`, then each match of this pattern is interpreted +/// as the end of a line. +/// +/// This function writes each line in `data` to the context stdout in +/// reverse. +/// +/// # Errors +/// +/// If there is a problem writing to stdout, then this function +/// returns [`std::io::Error`]. +fn buffer_tac_regex( + data: &[u8], + pattern: ®ex::bytes::Regex, + before: bool, +) -> std::io::Result<()> { + // pi-uutils: write through the context stdout instead of the process stdout. + let mut out = BufWriter::new(pi_uutils_ctx::stdout()); + + // The index of the line separator for the current line. + // + // As we scan through the `data` from right to left, we update this + // variable each time we find a new line separator. We restrict our + // regular expression search to only those bytes up to the line + // separator. + let mut this_line_end = data.len(); + + // The index of the start of the next line in the `data`. + // + // As we scan through the `data` from right to left, we update this + // variable each time we find a new line. + // + // If `before` is `true`, then each line starts immediately before + // the line separator. Otherwise, each line starts immediately after + // the line separator. + let mut following_line_start = data.len(); + + // Iterate over each byte in the buffer in reverse. When we find a + // line separator, write the line to stdout. + // + // The `before` flag controls whether the line separator appears at + // the end of the line (as in "abc\ndef\n") or at the beginning of + // the line (as in "/abc/def"). + for i in (0..data.len()).rev() { + // Determine if there is a match for `pattern` starting at index + // `i` in `data`. Only search up to the line ending that was + // found previously. + if let Some(match_) = pattern.find_at(&data[..this_line_end], i) + && match_.start() == i + { + // Record this index as the ending of the current line. + this_line_end = i; + + // The length of the match (that is, the line separator), in bytes. + let slen = match_.end() - match_.start(); + + if before { + out.write_all(&data[i..following_line_start])?; + following_line_start = i; + } else { + out.write_all(&data[i + slen..following_line_start])?; + following_line_start = i + slen; + } + } + } + + // After the loop terminates, write whatever bytes are remaining at + // the beginning of the buffer. + out.write_all(&data[0..following_line_start])?; + out.flush()?; + Ok(()) +} + +/// Write lines from `data` to stdout in reverse. +/// +/// This function writes to the context stdout each line appearing in `data`, +/// starting with the last line and ending with the first line. The +/// `separator` parameter defines what characters to use as a line +/// separator. +/// +/// If `before` is `false`, then this function assumes that the +/// `separator` appears at the end of each line, as in `"abc\ndef\n"`. +/// If `before` is `true`, then this function assumes that the +/// `separator` appears at the beginning of each line, as in +/// `"/abc/def"`. +fn buffer_tac(data: &[u8], before: bool, separator: &OsStr) -> std::io::Result<()> { + // pi-uutils: write through the context stdout instead of the process stdout. + let mut out = BufWriter::new(pi_uutils_ctx::stdout()); + + // The number of bytes in the line separator. + let slen = separator.len(); + + // The index of the start of the next line in the `data`. + // + // As we scan through the `data` from right to left, we update this + // variable each time we find a new line. + // + // If `before` is `true`, then each line starts immediately before + // the line separator. Otherwise, each line starts immediately after + // the line separator. + let mut following_line_start = data.len(); + + // Iterate over each byte in the buffer in reverse. When we find a + // line separator, write the line to stdout. + // + // The `before` flag controls whether the line separator appears at + // the end of the line (as in "abc\ndef\n") or at the beginning of + // the line (as in "/abc/def"). + for i in memmem::rfind_iter(data, separator.as_encoded_bytes()) { + if before { + out.write_all(&data[i..following_line_start])?; + following_line_start = i; + } else { + out.write_all(&data[i + slen..following_line_start])?; + following_line_start = i + slen; + } + } + + // After the loop terminates, write whatever bytes are remaining at + // the beginning of the buffer. + out.write_all(&data[0..following_line_start])?; + out.flush()?; + Ok(()) +} + +/// Make the regex flavor compatible with `regex` crate +/// +/// Concretely: +/// - Toggle escaping of (), |, {} +/// - Escape ^ and $ when not at edges +/// - Leave only ASCII bytes inside [] +/// - Escape non-ASCII bytes as `(?-u:\xFF)` outside [] +fn translate_regex_flavor(bytes: &[u8]) -> String { + let mut result = Vec::new(); + let mut i = 0; + let mut inside_brackets = false; + let mut prev_was_backslash = false; + let mut last_byte: Option = None; + + while let Some(b) = bytes.get(i) { + let is_escaped = prev_was_backslash; + prev_was_backslash = false; + + match b { + _ if inside_brackets && !b.is_ascii() => { + i += 1; + continue; + }, + // Unescape escaped (), |, {} when not inside brackets + b'\\' if !inside_brackets && !is_escaped => { + if let Some(next) = bytes.get(i + 1) + && matches!(next, b'(' | b')' | b'|' | b'{' | b'}') + { + result.push(*next); + last_byte = Some(*next); + i += 2; + continue; + } + + result.push(b'\\'); + last_byte = Some(b'\\'); + prev_was_backslash = true; + }, + // Bracket tracking + b'[' => { + inside_brackets = true; + result.push(*b); + last_byte = Some(*b); + }, + b']' => { + inside_brackets = false; + result.push(*b); + last_byte = Some(*b); + }, + // Escape (), |, {} when not escaped and outside brackets + b'(' | b')' | b'|' | b'{' | b'}' if !inside_brackets && !is_escaped => { + result.push(b'\\'); + result.push(*b); + last_byte = Some(*b); + }, + b'^' if !inside_brackets && !is_escaped => { + let is_anchor_position = result.is_empty() || matches!(last_byte, Some(b'(' | b'|')); + if !is_anchor_position { + result.push(b'\\'); + } + result.push(*b); + last_byte = Some(*b); + }, + b'$' if !inside_brackets && !is_escaped => { + let next_is_anchor_position = match bytes.get(i + 1) { + None => true, + Some(b')' | b'|') => true, + Some(b'\\') => { + // Peek two ahead to see if it's \) or \| + matches!(bytes.get(i + 2), Some(b')' | b'|')) + }, + _ => false, + }; + if !next_is_anchor_position { + result.push(b'\\'); + } + result.push(*b); + last_byte = Some(*b); + }, + _ if !b.is_ascii() => { + let _ = write!(result, r"(?-u:\x{b:02x})"); + last_byte = None; + }, + _ => { + result.push(*b); + last_byte = Some(*b); + }, + } + + i += 1; + } + + String::from_utf8(result).expect("produces ASCII bytes") +} + +#[allow(clippy::cognitive_complexity)] +fn tac(filenames: &[OsString], before: bool, regex: bool, separator: &OsStr) -> UResult<()> { + // Compile the regular expression pattern if it is provided. + let maybe_pattern = if regex { + match regex::bytes::RegexBuilder::new(&translate_regex_flavor(separator.as_encoded_bytes())) + .multi_line(true) + .build() + { + Ok(p) => Some(p), + Err(e) => return Err(TacError::InvalidRegex(e).into()), + } + } else { + None + }; + + for filename in filenames { + let mmap; + let buf; + + let data: &[u8] = if filename == "-" { + // pi-uutils: in-process stdin is a context stream, not the process + // stdin fd; upstream's stdin mmap / tempfile buffering and the + // `stdin_was_closed` signal check do not apply. Read it fully. + let mut contents = Vec::new(); + match pi_uutils_ctx::stdin().read_to_end(&mut contents) { + Ok(_) => { + buf = contents; + &buf + }, + Err(e) => { + show(&TacError::ReadError(OsString::from("stdin"), e)); + continue; + }, + } + } else { + // pi-uutils: resolve the operand against the shell working + // directory at the open site; `filename` is kept for errors. + let path = pi_uutils_ctx::resolve(filename); + let mut file = match File::open(&path) { + Ok(f) => f, + Err(e) => { + show(&TacError::OpenError(filename.clone(), e)); + continue; + }, + }; + + if let Some(mmap1) = try_mmap_file(&file) { + mmap = mmap1; + &mmap + } else { + let mut contents = Vec::new(); + match file.read_to_end(&mut contents) { + Ok(_) => { + buf = contents; + &buf + }, + Err(e) => { + show(&TacError::ReadError(filename.clone(), e)); + continue; + }, + } + } + }; + + // Select the appropriate `tac` algorithm based on whether the + // separator is given as a regular expression or a fixed string. + // pi-uutils: match ergonomics instead of upstream's `Some(ref pattern)`. + let result = match &maybe_pattern { + Some(pattern) => buffer_tac_regex(data, pattern, before), + None => buffer_tac(data, before, separator), + }; + + // If there is any error in writing the output, terminate immediately. + if let Err(e) = result { + return Err(TacError::WriteError(e).into()); + } + } + Ok(()) +} + +fn try_mmap_file(file: &File) -> Option { + // SAFETY: If the file is truncated while we map it, SIGBUS will be raised + // and our process will be terminated, thus preventing access of invalid memory. + unsafe { Mmap::map(file).ok() } +} + +#[cfg(test)] +mod tests_hybrid_flavor { + use super::translate_regex_flavor; + + #[test] + fn test_grouping_and_alternation() { + assert_eq!(translate_regex_flavor(br"\(abc\)"), r"(abc)"); + + assert_eq!(translate_regex_flavor(br"(abc)"), r"\(abc\)"); + + assert_eq!(translate_regex_flavor(br"a\|b"), r"a|b"); + + assert_eq!(translate_regex_flavor(br"a|b"), r"a\|b"); + } + + #[test] + fn test_anchors_context() { + assert_eq!(translate_regex_flavor(br"^abc$"), r"^abc$"); + + assert_eq!(translate_regex_flavor(br"a^b"), r"a\^b"); + assert_eq!(translate_regex_flavor(br"a$b"), r"a\$b"); + + // Anchors inside groups (reset by \(...\) regardless of position) + assert_eq!(translate_regex_flavor(br"\(^abc\)"), r"(^abc)"); + assert_eq!(translate_regex_flavor(br"\(abc$\)"), r"(abc$)"); + + // Anchors inside alternation (reset by \| regardless of position) + assert_eq!(translate_regex_flavor(br"^a\|^b"), r"^a|^b"); + assert_eq!(translate_regex_flavor(br"a$\|b$"), r"a$|b$"); + } + + #[test] + fn test_character_classes() { + assert_eq!(translate_regex_flavor(br"[a-z]"), r"[a-z]"); + + assert_eq!(translate_regex_flavor(br"[.]"), r"[.]"); + + assert_eq!(translate_regex_flavor(br"[]abc]"), r"[]abc]"); + + assert_eq!(translate_regex_flavor(br"[^]abc]"), r"[^]abc]"); + } +} + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, fs, io::Write, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + fn run_with(cwd: PathBuf, stdin: &[u8], args: Vec<&str>) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::Cursor::new(stdin.to_vec())), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd, + env: HashMap::new(), + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; + + let argv: Vec = std::iter::once("tac") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + /// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var). + fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) { + let dir = tempfile::tempdir().unwrap(); + let canon = fs::canonicalize(dir.path()).unwrap(); + (dir, canon) + } + + #[test] + fn resolves_relative_operand_against_scope_cwd() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("input.txt"), b"a\nb\nc\n").unwrap(); + + // Relative operand + scope cwd differing from the process cwd: only the + // call-site `pi_uutils_ctx::resolve` patch makes this find the file. + let (code, stdout, stderr) = run_with(root, b"", vec!["input.txt"]); + assert_eq!(code, 0); + assert_eq!(stdout, "c\nb\na\n"); + assert_eq!(stderr, ""); + } + + #[test] + fn no_operand_reads_context_stdin() { + let (code, stdout, stderr) = run_with(PathBuf::from("."), b"one\ntwo\nthree\n", vec![]); + assert_eq!(code, 0); + assert_eq!(stdout, "three\ntwo\none\n"); + assert_eq!(stderr, ""); + } + + #[test] + fn dash_operand_reads_context_stdin() { + let (code, stdout, stderr) = run_with(PathBuf::from("."), b"x\ny\n", vec!["-"]); + assert_eq!(code, 0); + assert_eq!(stdout, "y\nx\n"); + assert_eq!(stderr, ""); + } + + #[test] + fn custom_separator_reverses_fields() { + let (code, stdout, stderr) = run_with(PathBuf::from("."), b"a,b,c,", vec!["-s", ","]); + assert_eq!(code, 0); + assert_eq!(stdout, "c,b,a,"); + assert_eq!(stderr, ""); + } + + #[test] + fn before_flag_attaches_separator_before_each_line() { + let (code, stdout, stderr) = run_with(PathBuf::from("."), b"/abc/def", vec!["-b", "-s", "/"]); + assert_eq!(code, 0); + assert_eq!(stdout, "/def/abc"); + assert_eq!(stderr, ""); + } + + #[test] + fn regex_separator_splits_on_character_class() { + // `[,;]` treats either byte as a separator; records are emitted in + // reverse with each separator kept attached to its preceding record. + let (code, stdout, stderr) = run_with(PathBuf::from("."), b"a,b;c", vec!["-r", "-s", "[,;]"]); + assert_eq!(code, 0); + assert_eq!(stdout, "cb;a,"); + assert_eq!(stderr, ""); + } + + #[test] + fn invalid_regex_is_fatal_error() { + let (code, stdout, stderr) = run_with(PathBuf::from("."), b"abc", vec!["-r", "-s", "["]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert!(stderr.starts_with("tac: invalid regular expression:"), "stderr: {stderr}"); + } + + #[test] + fn missing_file_continues_with_next_operand_and_exits_nonzero() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("good.txt"), b"1\n2\n").unwrap(); + + let (code, stdout, stderr) = run_with(root, b"", vec!["nope.txt", "good.txt"]); + assert_eq!(code, 1); + assert_eq!(stdout, "2\n1\n", "valid operand still printed after the failure"); + assert!(stderr.contains("tac: failed to open 'nope.txt' for reading:"), "stderr: {stderr}"); + } + + #[test] + fn help_renders_to_scope_stdout() { + let (code, stdout, stderr) = run_with(PathBuf::from("."), b"", vec!["--help"]); + assert_eq!(code, 0); + assert!(stdout.contains("Usage:")); + assert!(stdout.contains("last line first")); + assert_eq!(stderr, ""); + } +} diff --git a/crates/vendor/uu-touch/Cargo.toml b/crates/vendor/uu-touch/Cargo.toml new file mode 100644 index 000000000..679bfcb87 --- /dev/null +++ b/crates/vendor/uu-touch/Cargo.toml @@ -0,0 +1,36 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/touch), patched to resolve +# path arguments against the shell working directory and route I/O through +# pi-uutils-ctx so it can run in-process as a shell builtin. See src/touch.rs +# for the patch markers (`pi-uutils:` comments). +[package] +name = "uu_touch" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "touch ~ (uutils) change FILE timestamps (vendored + patched for in-process embedding)" + +[lib] +path = "src/touch.rs" + +[dependencies] +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +filetime = "0.2.23" +jiff = "0.2.18" +parse_datetime = "0.14.0" +thiserror = "2.0.3" +uucore = { version = "0.8.0", features = ["libc", "parser"] } +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[target.'cfg(unix)'.dependencies] +libc = "0.2.172" +rustix = { version = "1.1.4", features = ["fs"] } + +[target.'cfg(windows)'.dependencies] +windows-sys = { version = "0.61.0", default-features = false, features = [ + "Win32_Storage_FileSystem", + "Win32_Foundation", +] } + +[dev-dependencies] +parking_lot = "0.12" +tempfile = "3" diff --git a/crates/vendor/uu-touch/LICENSE b/crates/vendor/uu-touch/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-touch/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-touch/src/touch.rs b/crates/vendor/uu-touch/src/touch.rs new file mode 100644 index 000000000..4b874d06e --- /dev/null +++ b/crates/vendor/uu-touch/src/touch.rs @@ -0,0 +1,1170 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +// spell-checker:ignore (ToDO) datelike datetime filetime lpszfilepath mktime +// strtime timelike utime DATETIME UTIME futimens spell-checker:ignore (FORMATS) +// MMDDhhmm YYYYMMDDHHMM YYMMDDHHMM YYYYMMDDHHMMS + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. Every filesystem syscall resolves its path operand +// against the shell working directory via `pi_uutils_ctx::resolve` AT THE CALL +// SITE, while the original operands are kept for display/error messages (GNU +// prints operands as typed). All process-global stdio is routed through +// `pi_uutils_ctx`, `translate!` strings are literalized, `_POSIX2_VERSION` is +// read from the scope environment, `show!` accumulation goes through +// `pi_uutils_ctx::set_exit_code`, and the entry point no longer calls +// `std::process::exit`. Upstream's `src/error.rs` is inlined below as +// `pub mod error`. jiff's `TimeZone::system()` (and thus `TZ`) intentionally +// stays process-global. + +#[cfg(unix)] +use std::fs::OpenOptions; +#[cfg(unix)] +use std::os::unix::fs::OpenOptionsExt; +use std::{ + borrow::Cow, + ffi::{OsStr, OsString}, + fs::{self, File}, + io::{Error, ErrorKind, Write}, + path::{Path, PathBuf}, + time::SystemTime, +}; + +use clap::{ + Arg, ArgAction, ArgGroup, ArgMatches, Command, + builder::{PossibleValue, ValueParser}, +}; +use filetime::{FileTime, set_file_times, set_symlink_file_times}; +use jiff::{Timestamp, ToSpan, Zoned, civil::Time, fmt::strtime, tz::TimeZone}; +#[cfg(unix)] +use libc::O_NONBLOCK; +use pi_uutils_ctx::format_usage; +#[cfg(unix)] +use rustix::fs::Timestamps; +#[cfg(unix)] +use rustix::fs::futimens; +#[cfg(target_os = "linux")] +use uucore::libc; +use uucore::{ + display::Quotable, + error::{FromIo, UError, UResult, USimpleError}, + parser::shortcut_value_parser::ShortcutValueParser, +}; + +use crate::error::TouchError; + +// pi-uutils: upstream `src/error.rs`, inlined so the vendored crate is a +// single source file. `translate!` message templates are literalized with the +// en-US strings. +pub mod error { + use std::path::PathBuf; + + use filetime::FileTime; + use thiserror::Error; + use uucore::{ + display::Quotable, + error::{UError, UIoError}, + }; + + #[derive(Debug, Error)] + pub enum TouchError { + #[error("Unable to parse date: {0}")] + InvalidDateFormat(String), + + /// The source time couldn't be converted to a [`jiff::Zoned`] + #[error("Source has invalid access or modification time: {0}")] + InvalidFiletime(FileTime), + + /// The reference file's attributes could not be found or read + #[error("failed to get attributes of {}: {}", .0.quote(), to_uioerror(.1))] + ReferenceFileInaccessible(PathBuf, std::io::Error), + + /// An error getting a path to stdout on Windows + #[error("GetFinalPathNameByHandleW failed with code {0}")] + WindowsStdoutPathError(String), + + /// An error encountered on a specific file + #[error("{error}")] + TouchFileError { path: PathBuf, index: usize, error: Box }, + } + + fn to_uioerror(err: &std::io::Error) -> UIoError { + let copy = if let Some(code) = err.raw_os_error() { + std::io::Error::from_raw_os_error(code) + } else { + std::io::Error::from(err.kind()) + }; + UIoError::from(copy) + } + + impl UError for TouchError {} +} + +/// Options contains all the possible behaviors and flags for touch. +/// +/// All options are public so that the options can be programmatically +/// constructed by other crates, such as nushell. That means that this struct is +/// part of our public API. It should therefore not be changed without good +/// reason. +/// +/// The fields are documented with the arguments that determine their value. +#[derive(Debug, Clone, Eq, PartialEq)] +pub struct Options { + /// Do not create any files. Set by `-c`/`--no-create`. + pub no_create: bool, + + /// Affect each symbolic link instead of any referenced file. Set by + /// `-h`/`--no-dereference`. + pub no_deref: bool, + + /// Where to get access and modification times from + pub source: Source, + + /// If given, uses time from `source` but on given date + pub date: Option, + + /// Whether to change access time only, modification time only, or both + pub change_times: ChangeTimes, + + /// When true, error when file doesn't exist and either `--no-dereference` + /// was passed or the file couldn't be created + pub strict: bool, +} + +pub enum InputFile { + /// A regular file + Path(PathBuf), + /// Touch stdout. `--no-dereference` will be ignored in this case. + Stdout, +} + +/// Whether to set access time only, modification time only, or both +#[derive(Debug, Clone, Eq, PartialEq)] +pub enum ChangeTimes { + /// Change only access time + AtimeOnly, + /// Change only modification time + MtimeOnly, + /// Change both access and modification times + Both, +} + +#[derive(Debug, Clone, Eq, PartialEq)] +pub enum Source { + /// Use access/modification times of given file + Reference(PathBuf), + Timestamp(FileTime), + /// Use current time + Now, +} + +pub mod options { + // Both SOURCES and sources are needed as we need to be able to refer to the + // ArgGroup. + pub static SOURCES: &str = "sources"; + pub mod sources { + pub static DATE: &str = "date"; + pub static REFERENCE: &str = "reference"; + pub static TIMESTAMP: &str = "timestamp"; + } + pub static HELP: &str = "help"; + pub static ACCESS: &str = "access"; + pub static MODIFICATION: &str = "modification"; + pub static NO_CREATE: &str = "no-create"; + pub static NO_DEREF: &str = "no-dereference"; + pub static TIME: &str = "time"; + pub static FORCE: &str = "force"; +} + +static ARG_FILES: &str = "files"; + +mod format { + pub(crate) const POSIX_LOCALE: &str = "%a %b %e %H:%M:%S %Y"; + pub(crate) const ISO_8601: &str = "%Y-%m-%d"; + // "%Y%m%d%H%M.%S" 15 chars + pub(crate) const YYYYMMDDHHMM_DOT_SS: &str = "%Y%m%d%H%M.%S"; + // "%Y-%m-%d %H:%M:%S.%SS" 12 chars + pub(crate) const YYYYMMDDHHMMSS: &str = "%Y-%m-%d %H:%M:%S.%f"; + // "%Y-%m-%d %H:%M:%S" 12 chars + pub(crate) const YYYYMMDDHHMMS: &str = "%Y-%m-%d %H:%M:%S"; + // "%Y-%m-%d %H:%M" 12 chars + // Used for example in tests/touch/no-rights.sh + pub(crate) const YYYY_MM_DD_HH_MM: &str = "%Y-%m-%d %H:%M"; + // "%Y%m%d%H%M" 12 chars + pub(crate) const YYYYMMDDHHMM: &str = "%Y%m%d%H%M"; + // "%Y-%m-%d %H:%M +offset" + // Used for example in tests/touch/relative.sh + pub(crate) const YYYYMMDDHHMM_OFFSET: &str = "%Y-%m-%d %H:%M %z"; +} + +fn timestamp_to_filetime(ts: Timestamp) -> FileTime { + FileTime::from_system_time(SystemTime::from(ts)) +} + +fn filetime_to_zoned(ft: &FileTime) -> Option { + let ts = Timestamp::new(ft.unix_seconds(), ft.nanoseconds() as i32).ok()?; + Some(Zoned::new(ts, TimeZone::system())) +} + +/// Whether all characters in the string are digits. +fn all_digits(s: &str) -> bool { + s.as_bytes().iter().all(u8::is_ascii_digit) +} + +/// Convert a two-digit year string to the corresponding number. +/// +/// `s` must be of length two or more. The last two bytes of `s` are +/// assumed to be the two digits of the year. +fn get_year(s: &str) -> u8 { + let bytes = s.as_bytes(); + let n = bytes.len(); + let y1 = bytes[n - 2] - b'0'; + let y2 = bytes[n - 1] - b'0'; + 10 * y1 + y2 +} + +/// Whether the first filename should be interpreted as a timestamp. +fn is_first_filename_timestamp( + reference: Option<&OsString>, + date: Option<&str>, + timestamp: Option<&str>, + files: &[&OsString], +) -> bool { + timestamp.is_none() + && reference.is_none() + && date.is_none() + && files.len() >= 2 + // pi-uutils: `_POSIX2_VERSION` comes from the scope environment (the + // shell's exported variables), not the host process environment. + // env check is last as the slowest op + && pi_uutils_ctx::var("_POSIX2_VERSION").as_deref() == Some("199209") + && files[0].to_str().is_some_and(is_timestamp) +} + +// Check if string is a valid POSIX timestamp (8 digits or 10 digits with valid +// year range) +fn is_timestamp(s: &str) -> bool { + all_digits(s) && (s.len() == 8 || (s.len() == 10 && (69..=99).contains(&get_year(s)))) +} + +/// Cycle the last two characters to the beginning of the string. +/// +/// `s` must have length at least two. +fn shr2(s: &str) -> String { + let n = s.len(); + let (a, b) = s.split_at(n - 2); + let mut result = String::with_capacity(n); + result.push_str(b); + result.push_str(a); + result +} + +/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the +/// arguments directly (without the uucore clap-localization helper that would +/// terminate the process), renders clap help/usage/version to the context +/// streams, and maps the `UResult` to an exit code, so it is safe to run inside +/// the host shell process. +pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match touch_main(&matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "touch: {msg}"); + } + if code == 0 { 1 } else { code } + }, + } +} + +fn touch_main(matches: &ArgMatches) -> UResult<()> { + let mut filenames: Vec<&OsString> = matches + .get_many::(ARG_FILES) + .ok_or_else(|| { + // pi-uutils: literalized; `uucore::execution_phrase()` is "touch" + // when running as a builtin. + USimpleError::new(1, "missing file operand\nTry 'touch --help' for more information.") + })? + .collect(); + + let no_deref = matches.get_flag(options::NO_DEREF); + + let reference = matches.get_one::(options::sources::REFERENCE); + let date = matches + .get_one::(options::sources::DATE) + .map(ToOwned::to_owned); + + let mut timestamp = matches + .get_one::(options::sources::TIMESTAMP) + .map(ToOwned::to_owned); + + if is_first_filename_timestamp(reference, date.as_deref(), timestamp.as_deref(), &filenames) { + let first_file = filenames[0].to_str().unwrap(); + timestamp = if first_file.len() == 10 { + Some(shr2(first_file)) + } else { + Some(first_file.to_string()) + }; + filenames = filenames[1..].to_vec(); + } + + let source = if let Some(reference) = reference { + Source::Reference(PathBuf::from(reference)) + } else if let Some(ts) = timestamp { + Source::Timestamp(parse_timestamp(&ts)?) + } else { + Source::Now + }; + + let files: Vec = filenames + .into_iter() + .map(|filename| { + if filename == "-" { + InputFile::Stdout + } else { + InputFile::Path(PathBuf::from(filename)) + } + }) + .collect(); + + let opts = Options { + no_create: matches.get_flag(options::NO_CREATE), + no_deref, + source, + date, + change_times: determine_atime_mtime_change(matches), + strict: false, + }; + + touch(&files, &opts)?; + + Ok(()) +} + +pub fn uu_app() -> Command { + Command::new("touch") + .version(uucore::crate_version!()) + .about("Update the access and modification times of each FILE to the current time.") + .override_usage(format_usage("touch [OPTION]... [FILE]...")) + .infer_long_args(true) + .disable_help_flag(true) + .arg( + Arg::new(options::HELP) + .long(options::HELP) + .help("Print help information.") + .action(ArgAction::Help), + ) + .arg( + Arg::new(options::ACCESS) + .short('a') + .help("change only the access time") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::sources::TIMESTAMP) + .short('t') + .help("use [[CC]YY]MMDDhhmm[.ss] instead of the current time") + .value_name("STAMP"), + ) + .arg( + Arg::new(options::sources::DATE) + .short('d') + .long(options::sources::DATE) + .allow_hyphen_values(true) + .help("parse argument and use it instead of current time") + .value_name("STRING") + .conflicts_with(options::sources::TIMESTAMP), + ) + .arg( + Arg::new(options::FORCE) + .short('f') + .help("(ignored)") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::MODIFICATION) + .short('m') + .help("change only the modification time") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::NO_CREATE) + .short('c') + .long(options::NO_CREATE) + .help("do not create any files") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::NO_DEREF) + .short('h') + .long(options::NO_DEREF) + .help( + "affect each symbolic link instead of any referenced file (only for systems that \ + can change the timestamps of a symlink)", + ) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::sources::REFERENCE) + .short('r') + .long(options::sources::REFERENCE) + .help("use this file's times instead of the current time") + .value_name("FILE") + .value_parser(ValueParser::os_string()) + .value_hint(clap::ValueHint::AnyPath) + .conflicts_with(options::sources::TIMESTAMP), + ) + .arg( + Arg::new(options::TIME) + .long(options::TIME) + .help( + "change only the specified time: \"access\", \"atime\", or \"use\" are equivalent \ + to -a; \"modify\" or \"mtime\" are equivalent to -m", + ) + .value_name("WORD") + .value_parser(ShortcutValueParser::new([ + PossibleValue::new("atime").alias("access").alias("use"), + PossibleValue::new("mtime").alias("modify"), + ])), + ) + .arg( + Arg::new(ARG_FILES) + .action(ArgAction::Append) + .num_args(1..) + .value_parser(clap::value_parser!(OsString)) + .value_hint(clap::ValueHint::AnyPath), + ) + .group( + ArgGroup::new(options::SOURCES) + .args([ + options::sources::TIMESTAMP, + options::sources::DATE, + options::sources::REFERENCE, + ]) + .multiple(true), + ) +} + +/// Execute the touch command. +/// +/// # Errors +/// +/// Possible causes: +/// - The user doesn't have permission to access the file +/// - One of the directory components of the file path doesn't exist. +/// - Dangling symlink is given and -r/--reference is used. +/// +/// It will return an `Err` on the first error. However, for any of the files, +/// if all of the following are true, it will print the error and continue +/// touching the rest of the files. +/// - `opts.strict` is `false` +/// - The file doesn't already exist +/// - `-c`/`--no-create` was passed (`opts.no_create`) +/// - Either `-h`/`--no-dereference` was passed (`opts.no_deref`) or the file +/// couldn't be created +pub fn touch(files: &[InputFile], opts: &Options) -> Result<(), TouchError> { + let (atime, mtime) = match &opts.source { + Source::Reference(reference) => { + // pi-uutils: resolve the reference operand against the shell + // working directory at the syscall site; the original operand is + // kept for the error message. + let (atime, mtime) = stat(&pi_uutils_ctx::resolve(reference), !opts.no_deref) + .map_err(|e| TouchError::ReferenceFileInaccessible(reference.to_owned(), e))?; + + (atime, mtime) + }, + Source::Now => { + let now: FileTime; + #[cfg(target_os = "linux")] + { + if opts.date.is_none() { + now = FileTime::from_unix_time(0, libc::UTIME_NOW as u32); + } else { + now = timestamp_to_filetime(Timestamp::now()); + } + } + #[cfg(not(target_os = "linux"))] + { + now = timestamp_to_filetime(Timestamp::now()); + } + (now, now) + }, + &Source::Timestamp(ts) => (ts, ts), + }; + + let (atime, mtime) = if let Some(date) = &opts.date { + ( + parse_date( + filetime_to_zoned(&atime).ok_or_else(|| TouchError::InvalidFiletime(atime))?, + date, + )?, + parse_date( + filetime_to_zoned(&mtime).ok_or_else(|| TouchError::InvalidFiletime(mtime))?, + date, + )?, + ) + } else { + (atime, mtime) + }; + + for (ind, file) in files.iter().enumerate() { + let (path, is_stdout) = match file { + InputFile::Stdout => (Cow::Owned(pathbuf_from_stdout()?), true), + InputFile::Path(path) => (Cow::Borrowed(path), false), + }; + touch_file(&path, is_stdout, opts, atime, mtime).map_err(|e| TouchError::TouchFileError { + path: path.into_owned(), + index: ind, + error: e, + })?; + } + + Ok(()) +} + +/// Create or update the timestamp for a single file. +/// +/// # Arguments +/// +/// - `path` - The path to the file to create/update timestamp for +/// - `is_stdout` - Stdout is handled specially, see [`update_times`] for more +/// info +/// - `atime` - Access time to set for the file +/// - `mtime` - Modification time to set for the file +fn touch_file( + path: &Path, + is_stdout: bool, + opts: &Options, + atime: FileTime, + mtime: FileTime, +) -> UResult<()> { + let filename = if is_stdout { + OsStr::new("-") + } else { + path.as_os_str() + }; + + // pi-uutils: resolve the operand against the shell working directory for + // every syscall below; `path`/`filename` keep the operand as typed for + // error messages. + let resolved = pi_uutils_ctx::resolve(path); + + let metadata_result = if opts.no_deref { + resolved.symlink_metadata() + } else { + resolved.metadata() + }; + + if let Err(e) = metadata_result { + if e.kind() != ErrorKind::NotFound { + return Err(e.map_err_context(|| format!("setting times of {}", filename.quote()))); + } + + if opts.no_create { + return Ok(()); + } + + if opts.no_deref { + let e = USimpleError::new( + 1, + format!("setting times of {}: No such file or directory", filename.quote()), + ); + if opts.strict { + return Err(e); + } + // pi-uutils: upstream `show!` — print the error and accumulate the + // exit code in the scope instead of process-global state. + let _ = writeln!(pi_uutils_ctx::stderr(), "touch: {e}"); + pi_uutils_ctx::set_exit_code(e.code()); + return Ok(()); + } + + if let Err(e) = File::create(&resolved) { + // we need to check if the path is the path to a directory (ends with a + // separator) we can't use File::create to create a directory + // we cannot use path.is_dir() because it calls fs::metadata which we already + // called when stable, we can change to use e.kind() == + // std::io::ErrorKind::IsADirectory + let is_directory = if let Some(last_char) = path.to_string_lossy().chars().last() { + last_char == std::path::MAIN_SEPARATOR + } else { + false + }; + if is_directory { + let custom_err = Error::other("No such file or directory"); + return Err( + custom_err.map_err_context(|| format!("cannot touch {}", filename.quote())), + ); + } + let e = e.map_err_context(|| format!("cannot touch {}", path.quote())); + if opts.strict { + return Err(e); + } + // pi-uutils: upstream `show!` — see above. + let _ = writeln!(pi_uutils_ctx::stderr(), "touch: {e}"); + pi_uutils_ctx::set_exit_code(e.code()); + return Ok(()); + } + + // Minor optimization: if no reference time, timestamp, or date was specified, + // we're done. + if opts.source == Source::Now && opts.date.is_none() { + return Ok(()); + } + } + + update_times(path, is_stdout, opts, atime, mtime) +} + +/// Returns which of the times (access, modification) are to be changed. +/// +/// Note that "-a" and "-m" may be passed together; this is not an xor. +/// - If `-a` is passed but not `-m`, only access time is changed +/// - If `-m` is passed but not `-a`, only modification time is changed +/// - If neither or both are passed, both times are changed +fn determine_atime_mtime_change(matches: &ArgMatches) -> ChangeTimes { + // If `--time` is given, Some(true) if equivalent to `-a`, Some(false) if + // equivalent to `-m` If `--time` not given, None + let time_access_only = if matches.contains_id(options::TIME) { + matches + .get_one::(options::TIME) + .map(|time| time.contains("access") || time.contains("atime") || time.contains("use")) + } else { + None + }; + + let atime_only = matches.get_flag(options::ACCESS) || time_access_only.unwrap_or_default(); + let mtime_only = matches.get_flag(options::MODIFICATION) || !time_access_only.unwrap_or(true); + + if atime_only && !mtime_only { + ChangeTimes::AtimeOnly + } else if mtime_only && !atime_only { + ChangeTimes::MtimeOnly + } else { + ChangeTimes::Both + } +} + +/// Updating file access and modification times based on user-specified options +/// +/// If the file is not stdout (`!is_stdout`) and `-h`/`--no-dereference` was +/// passed, then, if the given file is a symlink, its own times will be updated, +/// rather than the file it points to. +fn update_times( + path: &Path, + is_stdout: bool, + opts: &Options, + atime: FileTime, + mtime: FileTime, +) -> UResult<()> { + // pi-uutils: resolve the operand against the shell working directory for + // every syscall below; `path` keeps the operand as typed for error + // messages. + let resolved = pi_uutils_ctx::resolve(path); + + // If changing "only" atime or mtime, grab the existing value of the other. + let (atime, mtime) = match opts.change_times { + ChangeTimes::AtimeOnly => ( + atime, + stat(&resolved, !opts.no_deref) + .map_err_context(|| format!("failed to get attributes of {}", path.quote()))? + .1, + ), + ChangeTimes::MtimeOnly => ( + stat(&resolved, !opts.no_deref) + .map_err_context(|| format!("failed to get attributes of {}", path.quote()))? + .0, + mtime, + ), + ChangeTimes::Both => (atime, mtime), + }; + + // sets the file access and modification times for a file or a symbolic link. + // The filename, access time (atime), and modification time (mtime) are provided + // as inputs. + + if opts.no_deref && !is_stdout { + return set_symlink_file_times(&resolved, atime, mtime) + .map_err_context(|| format!("setting times of {}", path.quote())); + } + + #[cfg(unix)] + { + // Open write-only and use futimens to trigger IN_CLOSE_WRITE on Linux. + if !is_stdout && try_futimens_via_write_fd(&resolved, atime, mtime).is_ok() { + return Ok(()); + } + } + + set_file_times(&resolved, atime, mtime) + .map_err_context(|| format!("setting times of {}", path.quote())) +} + +#[cfg(unix)] +/// Set file times via file descriptor using `futimens`. +/// +/// This opens the file write-only and uses the POSIX `futimens` call to set +/// access and modification times on the open FD (not by path), which also +/// triggers `IN_CLOSE_WRITE` on Linux when the FD is closed. +fn try_futimens_via_write_fd(path: &Path, atime: FileTime, mtime: FileTime) -> std::io::Result<()> { + let file = OpenOptions::new() + .write(true) + // Avoid blocking on special files (e.g. FIFOs) before we can inspect metadata. + .custom_flags(O_NONBLOCK) + .open(path)?; + + let timestamps = Timestamps { + last_access: rustix::fs::Timespec { + tv_sec: atime.unix_seconds(), + tv_nsec: atime.nanoseconds() as _, + }, + last_modification: rustix::fs::Timespec { + tv_sec: mtime.unix_seconds(), + tv_nsec: mtime.nanoseconds() as _, + }, + }; + + futimens(&file, ×tamps).map_err(|e| Error::from_raw_os_error(e.raw_os_error())) +} + +/// Get metadata of the provided path +/// If `follow` is `true`, the function will try to follow symlinks. Errors if +/// the symlink is dangling, otherwise defaults to symlink metadata. If `follow` +/// is `false`, the function will return metadata of the symlink itself +fn stat(path: &Path, follow: bool) -> std::io::Result<(FileTime, FileTime)> { + let metadata = if follow { + match fs::metadata(path) { + // Successfully followed symlink + Ok(meta) => meta, + // Dangling symlink + Err(e) if e.kind() == ErrorKind::NotFound => return Err(e), + // Other error (?), try to get the symlink metadata + Err(_) => fs::symlink_metadata(path)?, + } + } else { + fs::symlink_metadata(path)? + }; + + Ok(( + FileTime::from_last_access_time(&metadata), + FileTime::from_last_modification_time(&metadata), + )) +} + +fn parse_date(ref_zoned: Zoned, s: &str) -> Result { + // This isn't actually compatible with GNU touch, but there doesn't seem to + // be any simple specification for what format this parameter allows and I'm + // not about to implement GNU parse_datetime. + // http://git.savannah.gnu.org/gitweb/?p=gnulib.git;a=blob_plain;f=lib/parse-datetime.y + + // TODO: match on char count? + + // "The preferred date and time representation for the current locale." + // "(In the POSIX locale this is equivalent to %a %b %e %H:%M:%S %Y.)" + // time 0.1.43 parsed this as 'a b e T Y' + // which is equivalent to the POSIX locale: %a %b %e %H:%M:%S %Y + // Tue Dec 3 ... + // ("%c", POSIX_LOCALE_FORMAT), + // + if let Ok(parsed) = strtime::parse(format::POSIX_LOCALE, s) + .and_then(|tm| tm.to_datetime()) + .and_then(|dt| TimeZone::UTC.to_zoned(dt)) + { + return Ok(timestamp_to_filetime(parsed.timestamp())); + } + + // Also support other formats found in the GNU tests like + // in tests/misc/stat-nanoseconds.sh + // or tests/touch/no-rights.sh + for fmt in [ + format::YYYYMMDDHHMMS, + format::YYYYMMDDHHMMSS, + format::YYYY_MM_DD_HH_MM, + format::YYYYMMDDHHMM_OFFSET, + ] { + if let Ok(parsed) = strtime::parse(fmt, s) + .and_then(|tm| tm.to_datetime()) + .and_then(|dt| TimeZone::UTC.to_zoned(dt)) + { + return Ok(timestamp_to_filetime(parsed.timestamp())); + } + } + + // "Equivalent to %Y-%m-%d (the ISO 8601 date format). (C99)" + // ("%F", ISO_8601_FORMAT), + // pi-uutils: `TimeZone::system()` (and the `TZ` variable it consults) + // intentionally stays process-global; jiff reads it internally. + if let Ok(filetime) = strtime::parse(format::ISO_8601, s) + .and_then(|tm| tm.to_date()) + .and_then(|date| { + TimeZone::system() + .to_ambiguous_zoned(date.to_datetime(Time::midnight())) + .unambiguous() + }) + .map(|zdt| timestamp_to_filetime(zdt.timestamp())) + { + return Ok(filetime); + } + + // "@%s" is "The number of seconds since the Epoch, 1970-01-01 00:00:00 +0000 + // (UTC). (TZ) (Calculated from mktime(tm).)" + if s.bytes().next() == Some(b'@') + && let Ok(ts) = &s[1..].parse::() + { + return Ok(FileTime::from_unix_time(*ts, 0)); + } + + if let Ok(zoned) = parse_datetime::parse_datetime_at_date(ref_zoned, s) { + return Ok(timestamp_to_filetime(zoned.timestamp())); + } + + Err(TouchError::InvalidDateFormat(s.to_owned())) +} + +/// Prepends 19 or 20 to the year if it is a 2 digit year +/// +/// GNU `touch` behavior: +/// +/// - 68 and before is interpreted as 20xx +/// - 69 and after is interpreted as 19xx +fn prepend_century(s: &str) -> UResult { + let first_two_digits = s[..2] + .parse::() + .map_err(|_| USimpleError::new(1, format!("invalid date ts format {}", s.quote())))?; + Ok(format!("{}{s}", if first_two_digits > 68 { 19 } else { 20 })) +} + +/// Parses a timestamp string into a [`FileTime`]. +/// +/// This function attempts to parse a string into a [`FileTime`] +/// As expected by gnu touch -t : `[[cc]yy]mmddhhmm[.ss]` +/// +/// Note that If the year is specified with only two digits, +/// then cc is 20 for years in the range 0 … 68, and 19 for years in 69 … 99. +/// in order to be compatible with GNU `touch`. +fn parse_timestamp(s: &str) -> UResult { + use format::{YYYYMMDDHHMM, YYYYMMDDHHMM_DOT_SS}; + + // pi-uutils: `TimeZone::system()` intentionally stays process-global. + let current_year = || Timestamp::now().to_zoned(TimeZone::system()).year(); + + let (format, ts) = match s.chars().count() { + 15 => (YYYYMMDDHHMM_DOT_SS, s.to_owned()), + 12 => (YYYYMMDDHHMM, s.to_owned()), + // If we don't add "19" or "20", we have insufficient information to parse + 13 => (YYYYMMDDHHMM_DOT_SS, prepend_century(s)?), + 10 => (YYYYMMDDHHMM, prepend_century(s)?), + 11 => (YYYYMMDDHHMM_DOT_SS, format!("{}{s}", current_year())), + 8 => (YYYYMMDDHHMM, format!("{}{s}", current_year())), + _ => { + return Err(USimpleError::new(1, format!("invalid date format {}", s.quote()))); + }, + }; + + let mut dt = strtime::parse(format, &ts) + .and_then(|parsed| parsed.to_datetime()) + .map_err(|_| USimpleError::new(1, format!("invalid date ts format {}", ts.quote())))?; + + // Jiff caps seconds at 59, but 60 is valid. It might be a leap second + // or wrap to the next minute. But that doesn't really matter, because we + // only care about the timestamp anyway. + // Tested in gnu/tests/touch/60-seconds + if dt.second() == 59 && ts.ends_with(".60") { + dt += 1.second(); + } + + // Due to daylight saving time switch, local time can jump from 1:59 AM to + // 3:00 AM, in which case any time between 2:00 AM and 2:59 AM is not valid. + // Jiff's `to_ambiguous_zoned(...).unambiguous()` handles this case. + let local = TimeZone::system() + .to_ambiguous_zoned(dt) + .unambiguous() + .map_err(|_| USimpleError::new(1, format!("invalid date ts format {}", ts.quote())))?; + + Ok(timestamp_to_filetime(local.timestamp())) +} + +// TODO: this may be a good candidate to put in fsext.rs +/// Returns a [`PathBuf`] to stdout. +/// +/// On Windows, uses `GetFinalPathNameByHandleW` to attempt to get the path +/// from the stdout handle. +#[cfg_attr(not(windows), expect(clippy::unnecessary_wraps))] +fn pathbuf_from_stdout() -> Result { + #[cfg(all(unix, not(target_os = "android")))] + { + Ok(PathBuf::from("/dev/stdout")) + } + #[cfg(target_os = "android")] + { + Ok(PathBuf::from("/proc/self/fd/1")) + } + #[cfg(windows)] + { + use std::os::windows::prelude::AsRawHandle; + + use windows_sys::Win32::{ + Foundation::{ + ERROR_INVALID_PARAMETER, ERROR_NOT_ENOUGH_MEMORY, ERROR_PATH_NOT_FOUND, GetLastError, + HANDLE, MAX_PATH, + }, + Storage::FileSystem::{FILE_NAME_OPENED, GetFinalPathNameByHandleW}, + }; + + let handle = std::io::stdout().lock().as_raw_handle() as HANDLE; + let mut file_path_buffer: [u16; MAX_PATH as usize] = [0; MAX_PATH as usize]; + + // https://docs.microsoft.com/en-us/windows/win32/api/fileapi/nf-fileapi-getfinalpathnamebyhandlea#examples + // SAFETY: We transmute the handle to be able to cast *mut c_void into a + // HANDLE (i32) so rustc will let us call GetFinalPathNameByHandleW. The + // reference example code for GetFinalPathNameByHandleW implies that + // it is safe for us to leave lpszfilepath uninitialized, so long as + // the buffer size is correct. We know the buffer size (MAX_PATH) at + // compile time. MAX_PATH is a small number (260) so we can cast it + // to a u32. + let ret = unsafe { + GetFinalPathNameByHandleW( + handle, + file_path_buffer.as_mut_ptr(), + file_path_buffer.len() as u32, + FILE_NAME_OPENED, + ) + }; + + // pi-uutils: literalized error strings; the variant's Display supplies + // the "GetFinalPathNameByHandleW failed with code" prefix, so only the + // code payload is stored. + let buffer_size = match ret { + ERROR_PATH_NOT_FOUND | ERROR_NOT_ENOUGH_MEMORY | ERROR_INVALID_PARAMETER => { + return Err(TouchError::WindowsStdoutPathError(ret.to_string())); + }, + 0 => { + return Err(TouchError::WindowsStdoutPathError(format!( + "{}", + // SAFETY: GetLastError is thread-safe and has no documented memory unsafety. + unsafe { GetLastError() } + ))); + }, + e => e as usize, + }; + + // Don't include the null terminator + Ok(String::from_utf16(&file_path_buffer[0..buffer_size]) + .map_err(|e| TouchError::WindowsStdoutPathError(e.to_string()))? + .into()) + } + #[cfg(target_os = "wasi")] + { + Ok(PathBuf::from("/dev/stdout")) + } +} + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + fn run_in(cwd: PathBuf, args: Vec<&str>) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd, + env: HashMap::new(), + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; + + let argv: Vec = std::iter::once("touch") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + /// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var). + fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) { + let dir = tempfile::tempdir().unwrap(); + let canon = fs::canonicalize(dir.path()).unwrap(); + (dir, canon) + } + + fn times_of(path: &Path) -> (FileTime, FileTime) { + let metadata = fs::metadata(path).unwrap(); + (FileTime::from_last_access_time(&metadata), FileTime::from_last_modification_time(&metadata)) + } + + #[test] + fn relative_operand_creates_file_in_scope_cwd() { + let (_dir, root) = canonical_tempdir(); + + // Relative operand + scope cwd differing from the process cwd: only + // the call-site `pi_uutils_ctx::resolve` patch makes the file land in + // the scope cwd instead of the process cwd. + let (code, stdout, stderr) = run_in(root.clone(), vec!["created.txt"]); + assert_eq!(code, 0); + assert_eq!(stdout, ""); + assert_eq!(stderr, ""); + assert!(root.join("created.txt").is_file()); + assert!( + !std::env::current_dir() + .unwrap() + .join("created.txt") + .exists() + ); + } + + #[test] + fn no_create_on_missing_file_is_silent_success() { + let (_dir, root) = canonical_tempdir(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["-c", "missing.txt"]); + assert_eq!(code, 0); + assert_eq!(stdout, ""); + assert_eq!(stderr, ""); + assert!(!root.join("missing.txt").exists()); + } + + #[test] + fn reference_copies_times_from_relative_reference() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("ref"), b"x").unwrap(); + let ref_atime = FileTime::from_unix_time(1_000_000, 0); + let ref_mtime = FileTime::from_unix_time(2_000_000, 0); + set_file_times(root.join("ref"), ref_atime, ref_mtime).unwrap(); + + // Both the `-r` reference and the FILE operand are relative: each is + // resolved against the scope cwd at its own syscall site. + let (code, stdout, stderr) = run_in(root.clone(), vec!["-r", "ref", "new"]); + assert_eq!(code, 0); + assert_eq!(stdout, ""); + assert_eq!(stderr, ""); + let (atime, mtime) = times_of(&root.join("new")); + assert_eq!(atime, ref_atime); + assert_eq!(mtime, ref_mtime); + } + + #[test] + fn date_sets_mtime_to_fixed_utc_instant() { + let (_dir, root) = canonical_tempdir(); + + // "%Y-%m-%d %H:%M:%S" dates are interpreted in UTC, so the expected + // epoch is timezone-independent: 2001-02-03T04:05:06Z. + let (code, stdout, stderr) = run_in(root.clone(), vec!["-d", "2001-02-03 04:05:06", "f"]); + assert_eq!(code, 0); + assert_eq!(stdout, ""); + assert_eq!(stderr, ""); + let (atime, mtime) = times_of(&root.join("f")); + assert_eq!(mtime.unix_seconds(), 981_173_106); + assert_eq!(atime.unix_seconds(), 981_173_106); + } + + #[test] + fn modification_only_preserves_existing_atime() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("f"), b"x").unwrap(); + let old_atime = FileTime::from_unix_time(1_111, 0); + let old_mtime = FileTime::from_unix_time(2_222, 0); + set_file_times(root.join("f"), old_atime, old_mtime).unwrap(); + + let (code, _, stderr) = run_in(root.clone(), vec!["-m", "-d", "@981173106", "f"]); + assert_eq!(code, 0); + assert_eq!(stderr, ""); + let (atime, mtime) = times_of(&root.join("f")); + assert_eq!(atime, old_atime, "-m must not change atime"); + assert_eq!(mtime, FileTime::from_unix_time(981_173_106, 0)); + } + + #[test] + fn missing_operand_is_usage_error() { + let (code, stdout, stderr) = run_in(PathBuf::from("."), vec![]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert!(stderr.contains("missing file operand"), "stderr: {stderr}"); + assert!(stderr.contains("Try 'touch --help'"), "stderr: {stderr}"); + } + + #[test] + fn help_renders_to_scope_stdout() { + let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["--help"]); + assert_eq!(code, 0); + assert!(stdout.contains("Usage:")); + assert!(stdout.contains("access and modification times")); + assert_eq!(stderr, ""); + } + + #[test] + fn time_word_and_flags_select_change_times() { + assert_eq!( + ChangeTimes::Both, + determine_atime_mtime_change(&uu_app().try_get_matches_from(vec!["touch", "f"]).unwrap()) + ); + assert_eq!( + ChangeTimes::Both, + determine_atime_mtime_change( + &uu_app() + .try_get_matches_from(vec!["touch", "-a", "-m", "--time", "modify", "f"]) + .unwrap() + ) + ); + assert_eq!( + ChangeTimes::AtimeOnly, + determine_atime_mtime_change( + &uu_app() + .try_get_matches_from(vec!["touch", "--time", "access", "f"]) + .unwrap() + ) + ); + assert_eq!( + ChangeTimes::MtimeOnly, + determine_atime_mtime_change( + &uu_app() + .try_get_matches_from(vec!["touch", "-m", "f"]) + .unwrap() + ) + ); + } +} diff --git a/crates/vendor/uu-truncate/Cargo.toml b/crates/vendor/uu-truncate/Cargo.toml new file mode 100644 index 000000000..4c05a9e67 --- /dev/null +++ b/crates/vendor/uu-truncate/Cargo.toml @@ -0,0 +1,22 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/truncate), patched to resolve +# path arguments against the shell working directory and route I/O through +# pi-uutils-ctx so it can run in-process as a shell builtin. See src/truncate.rs +# for the patch markers (`pi-uutils:` comments). +[package] +name = "uu_truncate" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "truncate ~ (uutils) truncate (or extend) FILE to SIZE (vendored + patched for in-process embedding)" + +[lib] +path = "src/truncate.rs" + +[dependencies] +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +uucore = { version = "0.8.0", features = ["parser-size"] } +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[dev-dependencies] +parking_lot = "0.12" +tempfile = "3" diff --git a/crates/vendor/uu-truncate/LICENSE b/crates/vendor/uu-truncate/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-truncate/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-truncate/src/truncate.rs b/crates/vendor/uu-truncate/src/truncate.rs new file mode 100644 index 000000000..0b6a35ff7 --- /dev/null +++ b/crates/vendor/uu-truncate/src/truncate.rs @@ -0,0 +1,557 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +// spell-checker:ignore (ToDO) RFILE refsize rfilename fsize tsize + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. Every filesystem syscall resolves its path operand +// against the shell working directory via `pi_uutils_ctx::resolve` AT THE CALL +// SITE, while the original operands are kept for display/error messages (GNU +// prints operands as typed). All process-global stdio is routed through +// `pi_uutils_ctx`, `translate!` strings are literalized, per-file errors are +// reported through the context stderr with `set_exit_code` (continue-on-error +// like GNU truncate), and the entry point no longer calls `std::process::exit`. + +#[cfg(unix)] +use std::os::unix::fs::FileTypeExt; +use std::{ + ffi::OsString, + fs::{OpenOptions, metadata}, + io::{ErrorKind, Write}, +}; + +use clap::{Arg, ArgAction, ArgMatches, Command}; +use pi_uutils_ctx::format_usage; +use uucore::{ + display::Quotable, + error::{FromIo, UResult, USimpleError, UUsageError}, + parser::parse_size::{ParseSizeError, Parser, allow_list_with_all_suffixes}, +}; + +#[derive(Debug, Eq, PartialEq)] +enum TruncateMode { + Absolute(u64), + Extend(u64), + Reduce(u64), + AtMost(u64), + AtLeast(u64), + RoundDown(u64), + RoundUp(u64), +} + +impl TruncateMode { + /// Compute a target size in bytes for this truncate mode. + /// + /// `fsize` is the size of the reference file, in bytes. + /// + /// If the mode is [`TruncateMode::Reduce`] and the value to + /// reduce by is greater than `fsize`, then this function returns + /// 0 (since it cannot return a negative number). + /// + /// # Returns + /// + /// `None` if rounding by 0, else the target size. + fn to_size(&self, fsize: u64) -> Option { + match self { + Self::Absolute(size) => Some(*size), + Self::Extend(size) => Some(fsize + size), + Self::Reduce(size) => Some(fsize.saturating_sub(*size)), + Self::AtMost(size) => Some(fsize.min(*size)), + Self::AtLeast(size) => Some(fsize.max(*size)), + Self::RoundDown(size) => fsize.checked_rem(*size).map(|remainder| fsize - remainder), + Self::RoundUp(size) => fsize.checked_next_multiple_of(*size), + } + } + + /// Determine if mode is absolute + /// + /// # Returns + /// + /// `true` is self matches Self::Absolute(_), `false` otherwise. + fn is_absolute(&self) -> bool { + matches!(self, Self::Absolute(_)) + } +} + +pub mod options { + pub static IO_BLOCKS: &str = "io-blocks"; + pub static NO_CREATE: &str = "no-create"; + pub static REFERENCE: &str = "reference"; + pub static SIZE: &str = "size"; + pub static ARG_FILES: &str = "files"; +} + +/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the +/// arguments directly (without the uucore clap-localization helper that would +/// terminate the process), renders clap help/usage/version to the context +/// streams, and maps the `UResult` to an exit code, so it is safe to run inside +/// the host shell process. +pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match truncate_main(&matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + // pi-uutils: don't emit a dangling "truncate: " prefix when the + // error renders to an empty message. + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "truncate: {msg}"); + } + if code == 0 { 1 } else { code } + }, + } +} + +fn truncate_main(matches: &ArgMatches) -> UResult<()> { + let files: Vec = matches + .get_many::(options::ARG_FILES) + .map(|v| v.cloned().collect()) + .unwrap_or_default(); + + if files.is_empty() { + Err(UUsageError::new(1, "missing file operand".to_string())) + } else { + let io_blocks = matches.get_flag(options::IO_BLOCKS); + let no_create = matches.get_flag(options::NO_CREATE); + let reference = matches + .get_one::(options::REFERENCE) + .map(String::from); + let size = matches.get_one::(options::SIZE).map(String::from); + truncate(no_create, io_blocks, reference, size, &files) + } +} + +pub fn uu_app() -> Command { + Command::new("truncate") + .version(uucore::crate_version!()) + .about("Shrink or extend the size of each file to the specified size.") + .override_usage(format_usage("truncate [OPTION]... [FILE]...")) + .after_help( + "SIZE is an integer with an optional prefix and optional unit.\nThe available units (K, \ + M, G, T, P, E, Z, and Y) use the following format:\n 'KB' => 1000 (kilobytes)\n \ + 'K' => 1024 (kibibytes)\n 'MB' => 1000*1000 (megabytes)\n 'M' => 1024*1024 \ + (mebibytes)\n 'GB' => 1000*1000*1000 (gigabytes)\n 'G' => 1024*1024*1024 \ + (gibibytes)\nSIZE may also be prefixed by one of the following to adjust the size of \ + each\nfile based on its current size:\n '+' => extend by\n '-' => reduce by\n \ + '<' => at most\n '>' => at least\n '/' => round down to multiple of\n '%' => \ + round up to multiple of", + ) + .infer_long_args(true) + .arg( + Arg::new(options::IO_BLOCKS) + .short('o') + .long(options::IO_BLOCKS) + .help( + "treat SIZE as the number of I/O blocks of the file rather than bytes (NOT \ + IMPLEMENTED)", + ) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::NO_CREATE) + .short('c') + .long(options::NO_CREATE) + .help("do not create files that do not exist") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::REFERENCE) + .short('r') + .long(options::REFERENCE) + .required_unless_present(options::SIZE) + .help("base the size of each file on the size of RFILE") + .value_name("RFILE") + .value_hint(clap::ValueHint::FilePath), + ) + .arg( + Arg::new(options::SIZE) + .short('s') + .long(options::SIZE) + .required_unless_present(options::REFERENCE) + .help( + "set or adjust the size of each file according to SIZE, which is in bytes unless \ + --io-blocks is specified", + ) + .allow_hyphen_values(true) + .value_name("SIZE"), + ) + .arg( + Arg::new(options::ARG_FILES) + .value_name("FILE") + .action(ArgAction::Append) + .required(true) + .value_hint(clap::ValueHint::FilePath) + .value_parser(clap::value_parser!(OsString)), + ) +} + +/// Truncate the named file to the specified size. +/// +/// If `create` is true, then the file will be created if it does not +/// already exist. If `size` is larger than the number of bytes in the +/// file, then the file will be padded with zeros. If `size` is smaller +/// than the number of bytes in the file, then the file will be +/// truncated and any bytes beyond `size` will be lost. +/// +/// # Errors +/// +/// If the file could not be opened, or there was a problem setting the +/// size of the file. +fn do_file_truncate(filename: &OsString, create: bool, size: u64) -> UResult<()> { + // pi-uutils: resolve the operand against the shell working directory at + // the open site; `filename` is kept for the error message. + let resolved = pi_uutils_ctx::resolve(filename); + + match OpenOptions::new() + .write(true) + .create(create) + .open(&resolved) + { + Ok(file) => file.set_len(size), + Err(e) if e.kind() == ErrorKind::NotFound && !create => Ok(()), + Err(e) => Err(e), + } + .map_err_context(|| format!("cannot open {} for writing", filename.quote())) +} + +fn file_truncate( + no_create: bool, + reference_size: Option, + mode: &TruncateMode, + filename: &OsString, +) -> UResult<()> { + // pi-uutils: resolve the operand against the shell working directory at + // the metadata site; `filename` is kept for the error message. + let resolved = pi_uutils_ctx::resolve(filename); + + // Get the length of the file. + let file_size = match metadata(&resolved) { + Ok(metadata) => { + // A pipe has no length. Do this check here to avoid duplicate `stat()` syscall. + #[cfg(unix)] + if metadata.file_type().is_fifo() { + return Err(USimpleError::new( + 1, + format!( + "cannot open {} for writing: No such device or address", + filename.to_string_lossy().quote() + ), + )); + } + metadata.len() + }, + Err(_) => 0, + }; + + // The reference size can be either: + // + // 1. The size of a given file + // 2. The size of the file to be truncated if no reference has been provided. + let actual_reference_size = reference_size.unwrap_or(file_size); + + let Some(truncate_size) = mode.to_size(actual_reference_size) else { + return Err(USimpleError::new(1, "division by zero".to_string())); + }; + + do_file_truncate(filename, !no_create, truncate_size) +} + +fn truncate( + no_create: bool, + _: bool, + reference: Option, + size: Option, + filenames: &[OsString], +) -> UResult<()> { + let reference_size = match reference { + Some(reference_path) => { + // pi-uutils: resolve the reference operand against the shell + // working directory; `reference_path` is kept for the message. + let reference_metadata = + metadata(pi_uutils_ctx::resolve(&reference_path)).map_err(|error| { + match error.kind() { + ErrorKind::NotFound => USimpleError::new( + 1, + format!("cannot stat {}: No such file or directory", reference_path.quote()), + ), + _ => error.map_err_context(String::new), + } + })?; + + Some(reference_metadata.len()) + }, + None => None, + }; + + let size_string = size.as_deref(); + + // Omitting the mode is equivalent to extending a file by 0 bytes. + let mode = match size_string { + Some(string) => match parse_mode_and_size(string) { + Err(error) => { + return Err(USimpleError::new(1, format!("Invalid number: {error}"))); + }, + Ok(mode) => mode, + }, + None => TruncateMode::Extend(0), + }; + + // If a reference file has been given, the truncate mode cannot be absolute. + if reference_size.is_some() && mode.is_absolute() { + return Err(USimpleError::new( + 1, + "you must specify a relative '--size' with '--reference'".to_string(), + )); + } + + for filename in filenames { + // pi-uutils: upstream aborts on the first failing file; report the + // error through the context stderr and continue with the remaining + // operands (GNU behavior), accumulating the exit code. + if let Err(err) = file_truncate(no_create, reference_size, &mode, filename) { + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "truncate: {msg}"); + } + pi_uutils_ctx::set_exit_code(if err.code() == 0 { 1 } else { err.code() }); + } + } + + Ok(()) +} + +/// Decide whether a character is one of the size modifiers, like '+' or '<'. +fn is_modifier(c: char) -> bool { + c == '+' || c == '-' || c == '<' || c == '>' || c == '/' || c == '%' +} + +/// Parse a size string with optional modifier symbol as its first character. +/// +/// A size string is as described in [`Parser::parse_u64`]. The first character +/// of `size_string` might be a modifier symbol, like `'+'` or +/// `'<'`. The first element of the pair returned by this function +/// indicates which modifier symbol was present, or +/// [`TruncateMode::Absolute`] if none. +fn parse_mode_and_size(size_string: &str) -> Result { + // Trim any whitespace. + let mut size_string = size_string.trim(); + + // Get the modifier character from the size string, if any. For + // example, if the argument is "+123", then the modifier is '+'. + if let Some(c) = size_string.chars().next() { + if is_modifier(c) { + size_string = &size_string[1..]; + } + let allow_list = allow_list_with_all_suffixes("EgGkKmMPQRtTYZ"); + let allow_list_ref = allow_list.iter().map(AsRef::as_ref).collect::>(); + Parser::default() + .with_allow_list(&allow_list_ref) + .parse_u64(size_string) + .map(match c { + '+' => TruncateMode::Extend, + '-' => TruncateMode::Reduce, + '<' => TruncateMode::AtMost, + '>' => TruncateMode::AtLeast, + '/' => TruncateMode::RoundDown, + '%' => TruncateMode::RoundUp, + _ => TruncateMode::Absolute, + }) + } else { + Err(ParseSizeError::ParseFailure(size_string.to_string())) + } +} + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, fs, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + fn run_in(cwd: PathBuf, args: Vec<&str>) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd, + env: HashMap::new(), + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; + + let argv: Vec = std::iter::once("truncate") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + /// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var). + fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) { + let dir = tempfile::tempdir().unwrap(); + let canon = fs::canonicalize(dir.path()).unwrap(); + (dir, canon) + } + + fn len(path: &PathBuf) -> u64 { + fs::metadata(path).unwrap().len() + } + + #[test] + fn resolves_relative_operand_against_scope_cwd() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("f"), b"12345678").unwrap(); + + // Relative operand + scope cwd differing from the process cwd: only the + // call-site `pi_uutils_ctx::resolve` patch makes this find the file. + let (code, stdout, stderr) = run_in(root.clone(), vec!["-s", "5", "f"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", "")); + assert_eq!(len(&root.join("f")), 5); + } + + #[test] + fn extend_grows_by_relative_amount() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("f"), b"1234").unwrap(); + + let (code, _, stderr) = run_in(root.clone(), vec!["-s", "+3", "f"]); + assert_eq!((code, stderr.as_str()), (0, "")); + assert_eq!(len(&root.join("f")), 7); + } + + #[test] + fn at_most_caps_only_larger_files() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("big"), vec![0u8; 20]).unwrap(); + fs::write(root.join("small"), b"abc").unwrap(); + + let (code, _, stderr) = run_in(root.clone(), vec!["-s", "<10", "big", "small"]); + assert_eq!((code, stderr.as_str()), (0, "")); + assert_eq!(len(&root.join("big")), 10); + assert_eq!(len(&root.join("small")), 3); + } + + #[test] + fn no_create_skips_missing_file() { + let (_dir, root) = canonical_tempdir(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["-c", "-s", "5", "missing"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", "")); + assert!(!root.join("missing").exists()); + } + + #[test] + fn missing_file_without_no_create_is_created_at_size() { + let (_dir, root) = canonical_tempdir(); + + let (code, _, stderr) = run_in(root.clone(), vec!["-s", "9", "fresh"]); + assert_eq!((code, stderr.as_str()), (0, "")); + assert_eq!(len(&root.join("fresh")), 9); + } + + #[test] + fn reference_copies_size_of_rfile() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("ref"), b"123456").unwrap(); + fs::write(root.join("f"), b"x").unwrap(); + + let (code, _, stderr) = run_in(root.clone(), vec!["-r", "ref", "f"]); + assert_eq!((code, stderr.as_str()), (0, "")); + assert_eq!(len(&root.join("f")), 6); + } + + #[test] + fn missing_reference_file_fails_with_stat_error() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("f"), b"x").unwrap(); + + let (code, _, stderr) = run_in(root.clone(), vec!["-r", "nope", "f"]); + assert_eq!(code, 1); + assert!(stderr.contains("cannot stat 'nope': No such file or directory")); + assert_eq!(len(&root.join("f")), 1, "operand must be untouched"); + } + + #[test] + fn invalid_size_reports_error_and_exit_1() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("f"), b"x").unwrap(); + + let (code, stdout, stderr) = run_in(root.clone(), vec!["-s", "bogus", "f"]); + assert_eq!((code, stdout.as_str()), (1, "")); + assert!(stderr.contains("truncate: Invalid number:")); + assert_eq!(len(&root.join("f")), 1, "operand must be untouched"); + } + + #[test] + fn reference_with_absolute_size_is_rejected() { + let (_dir, root) = canonical_tempdir(); + fs::write(root.join("ref"), b"123").unwrap(); + fs::write(root.join("f"), b"x").unwrap(); + + let (code, _, stderr) = run_in(root.clone(), vec!["-r", "ref", "-s", "5", "f"]); + assert_eq!(code, 1); + assert!(stderr.contains("you must specify a relative '--size' with '--reference'")); + } + + #[test] + fn parse_mode_and_size_prefixes() { + assert_eq!(parse_mode_and_size("10"), Ok(TruncateMode::Absolute(10))); + assert_eq!(parse_mode_and_size("+10"), Ok(TruncateMode::Extend(10))); + assert_eq!(parse_mode_and_size("-10"), Ok(TruncateMode::Reduce(10))); + assert_eq!(parse_mode_and_size("<10"), Ok(TruncateMode::AtMost(10))); + assert_eq!(parse_mode_and_size(">10"), Ok(TruncateMode::AtLeast(10))); + assert_eq!(parse_mode_and_size("/10"), Ok(TruncateMode::RoundDown(10))); + assert_eq!(parse_mode_and_size("%10"), Ok(TruncateMode::RoundUp(10))); + assert_eq!(parse_mode_and_size("1kB"), Ok(TruncateMode::Absolute(1000))); + assert!(parse_mode_and_size("1b").is_err()); + } + + #[test] + fn help_renders_to_scope_stdout() { + let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["--help"]); + assert_eq!(code, 0); + assert!(stdout.contains("Usage:")); + assert!(stdout.contains("round up to multiple of")); + assert_eq!(stderr, ""); + } +} diff --git a/crates/vendor/uu-uname/Cargo.toml b/crates/vendor/uu-uname/Cargo.toml new file mode 100644 index 000000000..f5884aeee --- /dev/null +++ b/crates/vendor/uu-uname/Cargo.toml @@ -0,0 +1,21 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/uname), patched to route +# output through pi-uutils-ctx so it can run in-process as a shell builtin. See +# src/uname.rs for the patch markers (`pi-uutils:` comments). +[package] +name = "uu_uname" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "uname ~ (uutils) display system information (vendored + patched for in-process embedding)" + +[lib] +path = "src/uname.rs" + +[dependencies] +platform-info = "2.0.3" +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +uucore = "0.8.0" +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[dev-dependencies] +parking_lot = "0.12" diff --git a/crates/vendor/uu-uname/LICENSE b/crates/vendor/uu-uname/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-uname/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-uname/src/uname.rs b/crates/vendor/uu-uname/src/uname.rs new file mode 100644 index 000000000..58bafde0a --- /dev/null +++ b/crates/vendor/uu-uname/src/uname.rs @@ -0,0 +1,342 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +// spell-checker:ignore (API) nodename osname sysname (options) mnrsv mnrsvo + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. Output goes to the context stdout (upstream's +// `println_verbatim` writes to the process stdout), `translate!` strings are +// literalized, and the entry point no longer calls `std::process::exit`. + +use std::{ + ffi::{OsStr, OsString}, + io::Write, +}; + +use clap::{Arg, ArgAction, ArgMatches, Command}; +use pi_uutils_ctx::format_usage; +use platform_info::{PlatformInfo, PlatformInfoAPI, UNameAPI}; +use uucore::error::{UResult, USimpleError}; + +pub mod options { + pub static ALL: &str = "all"; + pub static KERNEL_NAME: &str = "kernel-name"; + pub static NODENAME: &str = "nodename"; + pub static KERNEL_VERSION: &str = "kernel-version"; + pub static KERNEL_RELEASE: &str = "kernel-release"; + pub static MACHINE: &str = "machine"; + pub static PROCESSOR: &str = "processor"; + pub static HARDWARE_PLATFORM: &str = "hardware-platform"; + pub static OS: &str = "operating-system"; +} + +pub struct UNameOutput { + pub kernel_name: Option, + pub nodename: Option, + pub kernel_release: Option, + pub kernel_version: Option, + pub machine: Option, + pub os: Option, + pub processor: Option, + pub hardware_platform: Option, +} + +impl UNameOutput { + fn display(&self) -> OsString { + [ + self.kernel_name.as_ref(), + self.nodename.as_ref(), + self.kernel_release.as_ref(), + self.kernel_version.as_ref(), + self.machine.as_ref(), + self.processor.as_ref(), + self.hardware_platform.as_ref(), + self.os.as_ref(), + ] + .into_iter() + .flatten() + .map(OsString::as_os_str) + .collect::>() + .join(OsStr::new(" ")) + } + + pub fn new(opts: &Options) -> UResult { + let uname = PlatformInfo::new() + .map_err(|_e| USimpleError::new(1, "cannot get system name".to_string()))?; + let none = !(opts.all + || opts.kernel_name + || opts.nodename + || opts.kernel_release + || opts.kernel_version + || opts.machine + || opts.os + || opts.processor + || opts.hardware_platform); + + let kernel_name = (opts.kernel_name || opts.all || none).then(|| uname.sysname().to_owned()); + + let nodename = (opts.nodename || opts.all).then(|| uname.nodename().to_owned()); + + let kernel_release = (opts.kernel_release || opts.all).then(|| uname.release().to_owned()); + + let kernel_version = (opts.kernel_version || opts.all).then(|| uname.version().to_owned()); + + let machine = (opts.machine || opts.all).then(|| uname.machine().to_owned()); + + let os = (opts.os || opts.all).then(|| uname.osname().to_owned()); + + // This option is unsupported on modern Linux systems + // See: https://lists.gnu.org/archive/html/bug-coreutils/2005-09/msg00063.html + let processor = opts.processor.then(|| "unknown".into()); + + // This option is unsupported on modern Linux systems + // See: https://lists.gnu.org/archive/html/bug-coreutils/2005-09/msg00063.html + let hardware_platform = opts.hardware_platform.then(|| "unknown".into()); + + Ok(Self { + kernel_name, + nodename, + kernel_release, + kernel_version, + machine, + os, + processor, + hardware_platform, + }) + } +} + +pub struct Options { + pub all: bool, + pub kernel_name: bool, + pub nodename: bool, + pub kernel_version: bool, + pub kernel_release: bool, + pub machine: bool, + pub processor: bool, + pub hardware_platform: bool, + pub os: bool, +} + +/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the +/// arguments directly (without the uucore clap-localization helper that would +/// terminate the process), renders clap help/usage/version to the context +/// streams, and maps the `UResult` to an exit code, so it is safe to run inside +/// the host shell process. +pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match uname_main(&matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "uname: {msg}"); + } + if code == 0 { 1 } else { code } + }, + } +} + +fn uname_main(matches: &ArgMatches) -> UResult<()> { + let options = Options { + all: matches.get_flag(options::ALL), + kernel_name: matches.get_flag(options::KERNEL_NAME), + nodename: matches.get_flag(options::NODENAME), + kernel_release: matches.get_flag(options::KERNEL_RELEASE), + kernel_version: matches.get_flag(options::KERNEL_VERSION), + machine: matches.get_flag(options::MACHINE), + processor: matches.get_flag(options::PROCESSOR), + hardware_platform: matches.get_flag(options::HARDWARE_PLATFORM), + os: matches.get_flag(options::OS), + }; + let output = UNameOutput::new(&options)?; + // pi-uutils: replacement for upstream's `println_verbatim` — writes the + // output bytes verbatim to the context stdout instead of the process + // stdout. + let mut out = pi_uutils_ctx::stdout(); + out.write_all(uucore::os_str_as_bytes(output.display().as_os_str())?) + .and_then(|()| out.write_all(b"\n")) + .and_then(|()| out.flush()) + .map_err(|e| USimpleError::new(1, e.to_string()))?; + Ok(()) +} + +pub fn uu_app() -> Command { + Command::new("uname") + .version(uucore::crate_version!()) + .about("Print certain system information.\nWith no OPTION, same as -s.") + .override_usage(format_usage("uname [OPTION]...")) + .infer_long_args(true) + .arg( + Arg::new(options::ALL) + .short('a') + .long(options::ALL) + .help("Behave as though all of the options -mnrsvo were specified.") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::KERNEL_NAME) + .short('s') + .long(options::KERNEL_NAME) + .alias("sysname") // Obsolescent option in GNU uname + .help("print the kernel name.") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::NODENAME) + .short('n') + .long(options::NODENAME) + .help( + "print the nodename (the nodename may be a name that the system is known by to a \ + communications network).", + ) + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::KERNEL_RELEASE) + .short('r') + .long(options::KERNEL_RELEASE) + .alias("release") // Obsolescent option in GNU uname + .help("print the operating system release.") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::KERNEL_VERSION) + .short('v') + .long(options::KERNEL_VERSION) + .help("print the operating system version.") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::MACHINE) + .short('m') + .long(options::MACHINE) + .help("print the machine hardware name.") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::OS) + .short('o') + .long(options::OS) + .help("print the operating system name.") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::PROCESSOR) + .short('p') + .long(options::PROCESSOR) + .help("print the processor type (non-portable)") + .action(ArgAction::SetTrue) + .hide(true), + ) + .arg( + Arg::new(options::HARDWARE_PLATFORM) + .short('i') + .long(options::HARDWARE_PLATFORM) + .help("print the hardware platform (non-portable)") + .action(ArgAction::SetTrue) + .hide(true), + ) +} + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + fn run_in(args: Vec<&str>) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd: PathBuf::from("."), + env: HashMap::new(), + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; + + let argv: Vec = std::iter::once("uname") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + #[test] + fn kernel_name_matches_platform() { + let (code, stdout, stderr) = run_in(vec!["-s"]); + assert_eq!((code, stderr.as_str()), (0, "")); + #[cfg(target_os = "macos")] + assert_eq!(stdout, "Darwin\n"); + #[cfg(target_os = "linux")] + assert_eq!(stdout, "Linux\n"); + #[cfg(not(any(target_os = "macos", target_os = "linux")))] + assert!(stdout.trim_end().len() > 0); + } + + #[test] + fn no_options_defaults_to_kernel_name() { + let (code, bare, _) = run_in(vec![]); + let (_, with_s, _) = run_in(vec!["-s"]); + assert_eq!(code, 0); + assert_eq!(bare, with_s); + } + + #[test] + fn all_contains_kernel_name_and_more() { + let (code, all, stderr) = run_in(vec!["-a"]); + let (_, kernel, _) = run_in(vec!["-s"]); + assert_eq!((code, stderr.as_str()), (0, "")); + let kernel = kernel.trim_end(); + assert!(all.starts_with(kernel), "-a output {all:?} must start with {kernel:?}"); + assert!(all.trim_end().len() > kernel.len(), "-a must print more fields than -s"); + } + + #[test] + fn processor_prints_unknown() { + let (code, stdout, stderr) = run_in(vec!["-p"]); + assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "unknown\n", "")); + } +} diff --git a/crates/vendor/uu-whoami/Cargo.toml b/crates/vendor/uu-whoami/Cargo.toml new file mode 100644 index 000000000..d227d61ae --- /dev/null +++ b/crates/vendor/uu-whoami/Cargo.toml @@ -0,0 +1,27 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/whoami), patched to route +# output through pi-uutils-ctx so it can run in-process as a shell builtin. See +# src/whoami.rs for the patch markers (`pi-uutils:` comments). +[package] +name = "uu_whoami" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "whoami ~ (uutils) display user name of current effective user ID (vendored + patched for in-process embedding)" + +[lib] +path = "src/whoami.rs" + +[dependencies] +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +uucore = { version = "0.8.0", features = ["entries", "process"] } +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[target.'cfg(target_os = "windows")'.dependencies] +windows-sys = { version = "0.61.0", features = [ + "Win32_NetworkManagement_NetManagement", + "Win32_System_WindowsProgramming", + "Win32_Foundation", +] } + +[dev-dependencies] +parking_lot = "0.12" diff --git a/crates/vendor/uu-whoami/LICENSE b/crates/vendor/uu-whoami/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-whoami/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-whoami/src/whoami.rs b/crates/vendor/uu-whoami/src/whoami.rs new file mode 100644 index 000000000..564f6c7ab --- /dev/null +++ b/crates/vendor/uu-whoami/src/whoami.rs @@ -0,0 +1,198 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +// spell-checker:ignore (ToDO) getusername + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. Output goes to the context stdout (upstream's +// `println_verbatim` writes to the process stdout), `translate!` strings are +// literalized, the `platform` module (upstream +// src/platform/{mod,unix,windows}.rs) is inlined, and the entry point no longer +// calls `std::process::exit`. + +use std::{ffi::OsString, io::Write}; + +use clap::Command; +use uucore::error::{FromIo, UResult, USimpleError}; + +// pi-uutils: inlined from upstream src/platform/{mod,unix,windows}.rs (verbatim +// bodies); the platform user lookup itself is process-global state and needs no +// scope patching. +mod platform { + #[cfg(unix)] + pub use self::unix::get_username; + #[cfg(windows)] + pub use self::windows::get_username; + + #[cfg(unix)] + mod unix { + use std::{ffi::OsString, io}; + + use uucore::{entries::uid2usr, process::geteuid}; + + pub fn get_username() -> io::Result { + // uid2usr should arguably return an OsString but currently doesn't + uid2usr(geteuid()).map(Into::into) + } + } + + #[cfg(windows)] + mod windows { + use std::{ffi::OsString, io, os::windows::ffi::OsStringExt}; + + use windows_sys::Win32::{ + NetworkManagement::NetManagement::UNLEN, System::WindowsProgramming::GetUserNameW, + }; + + pub fn get_username() -> io::Result { + const BUF_LEN: u32 = UNLEN + 1; + let mut buffer = [0_u16; BUF_LEN as usize]; + let mut len = BUF_LEN; + // SAFETY: buffer.len() == len + if unsafe { GetUserNameW(buffer.as_mut_ptr(), &raw mut len) } == 0 { + return Err(io::Error::last_os_error()); + } + Ok(OsString::from_wide(&buffer[..len as usize - 1])) + } + } +} + +/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the +/// arguments directly (without the uucore clap-localization helper that would +/// terminate the process), renders clap help/usage/version to the context +/// streams, and maps the `UResult` to an exit code, so it is safe to run inside +/// the host shell process. +pub fn run(argv: Vec) -> i32 { + match uu_app().try_get_matches_from(argv) { + Ok(_matches) => {}, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + } + match whoami_main() { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + let msg = err.to_string(); + if !msg.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "whoami: {msg}"); + } + if code == 0 { 1 } else { code } + }, + } +} + +fn whoami_main() -> UResult<()> { + let username = whoami()?; + // pi-uutils: replacement for upstream's `println_verbatim` — writes the + // username bytes verbatim to the context stdout instead of the process + // stdout. + let mut out = pi_uutils_ctx::stdout(); + out.write_all(uucore::os_str_as_bytes(&username)?) + .and_then(|()| out.write_all(b"\n")) + .and_then(|()| out.flush()) + .map_err(|e| USimpleError::new(1, format!("failed to print username: {e}")))?; + Ok(()) +} + +/// Get the current username +pub fn whoami() -> UResult { + platform::get_username().map_err_context(|| "failed to get username".to_string()) +} + +pub fn uu_app() -> Command { + Command::new("whoami") + .version(uucore::crate_version!()) + .about("Print the current username.") + .override_usage("whoami") + .infer_long_args(true) +} + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + fn run_in(args: Vec<&str>) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd: PathBuf::from("."), + env: HashMap::new(), + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; + + let argv: Vec = std::iter::once("whoami") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + #[test] + fn prints_process_user_with_trailing_newline() { + let (code, stdout, stderr) = run_in(vec![]); + assert_eq!((code, stderr.as_str()), (0, "")); + assert!(stdout.ends_with('\n')); + let name = stdout.trim_end(); + assert!(!name.is_empty()); + // When the host exports USER it names the same effective user the + // platform lookup resolves. + if let Ok(user) = std::env::var("USER") { + assert_eq!(name, user); + } + } + + #[test] + fn rejects_operands() { + let (code, stdout, stderr) = run_in(vec!["extra"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert!(!stderr.is_empty(), "clap usage error must go to scope stderr"); + } + + #[test] + fn help_renders_to_scope_stdout() { + let (code, stdout, stderr) = run_in(vec!["--help"]); + assert_eq!((code, stderr.as_str()), (0, "")); + assert!(stdout.contains("Print the current username.")); + } +} diff --git a/crates/vendor/uu-yes/Cargo.toml b/crates/vendor/uu-yes/Cargo.toml new file mode 100644 index 000000000..ff19aa1d0 --- /dev/null +++ b/crates/vendor/uu-yes/Cargo.toml @@ -0,0 +1,23 @@ +# Vendored from uutils/coreutils tag 0.8.0 (src/uu/yes), patched to route +# output through pi-uutils-ctx, handle a closed consumer (broken pipe) as a +# clean in-process exit, and poll the scope cancel flag so it can run +# in-process as a shell builtin. See src/yes.rs for the patch markers +# (`pi-uutils:` comments). +[package] +name = "uu_yes" +version = "0.8.0" +edition = "2024" +license = "MIT" +description = "yes ~ (uutils) repeatedly display a line with STRING (or 'y') (vendored + patched for in-process embedding)" + +[lib] +path = "src/yes.rs" + +[dependencies] +clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] } +itertools = "0.14.0" +uucore = "0.8.0" +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[dev-dependencies] +parking_lot = "0.12" diff --git a/crates/vendor/uu-yes/LICENSE b/crates/vendor/uu-yes/LICENSE new file mode 100644 index 000000000..21bd44404 --- /dev/null +++ b/crates/vendor/uu-yes/LICENSE @@ -0,0 +1,18 @@ +Copyright (c) uutils developers + +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software is furnished to do so, +subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/uu-yes/src/yes.rs b/crates/vendor/uu-yes/src/yes.rs new file mode 100644 index 000000000..26dea7f91 --- /dev/null +++ b/crates/vendor/uu-yes/src/yes.rs @@ -0,0 +1,351 @@ +// This file is part of the uutils coreutils package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +// cSpell:ignore strs + +// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process +// as a shell builtin. All process-global stdio is routed through +// `pi_uutils_ctx`, `translate!` strings are literalized, and the entry point no +// longer calls `std::process::exit`. Because the utility runs inside the shell +// process there is no SIGPIPE to terminate it when the consumer closes, so a +// broken-pipe write error exits cleanly with code 0 (GNU behaviour) on every +// platform, and the output loop polls the scope cancel flag so shell +// abort/timeout stops it promptly. + +use std::{ + error::Error, + ffi::OsString, + io::{self, Write}, +}; + +use clap::{Arg, ArgAction, Command, builder::ValueParser}; +use pi_uutils_ctx::format_usage; +use uucore::error::strip_errno; + +// it's possible that using a smaller or larger buffer might provide better +// performance on some systems, but honestly this is good enough +const BUF_SIZE: usize = 16 * 1024; + +/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the +/// arguments directly (without the uucore clap-localization helper that would +/// terminate the process), renders clap help/usage/version to the context +/// streams, and maps the outcome to an exit code, so it is safe to run inside +/// the host shell process. +pub fn run(argv: Vec) -> i32 { + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + + let mut buffer = Vec::with_capacity(BUF_SIZE); + #[allow(clippy::unwrap_used, reason = "clap provides 'y' by default")] + let _ = args_into_buffer(&mut buffer, matches.get_many::("STRING").unwrap()); + prepare_buffer(&mut buffer); + + match exec(&buffer) { + // pi-uutils: a broken pipe means the consumer closed its end; a + // process `yes` would die from SIGPIPE (or handle EPIPE on Windows), + // so the in-process builtin exits cleanly with 0 on every platform. + ExecStop::Io(err) if err.kind() == io::ErrorKind::BrokenPipe => 0, + ExecStop::Io(err) => { + let _ = writeln!(pi_uutils_ctx::stderr(), "yes: standard output: {}", strip_errno(&err)); + 1 + }, + // pi-uutils: the shell asked the scope to cancel (abort/timeout); + // there is no signal-style exit status in-process, so return 1. + ExecStop::Cancelled => 1, + } +} + +pub fn uu_app() -> Command { + Command::new("yes") + .version(uucore::crate_version!()) + .about("Repeatedly display a line with STRING (or 'y')") + .override_usage(format_usage("yes [STRING]...")) + .arg( + Arg::new("STRING") + .default_value("y") + .value_parser(ValueParser::os_string()) + .action(ArgAction::Append), + ) + .infer_long_args(true) +} + +/// Copies words from `i` into `buf`, separated by spaces. +#[allow(clippy::unnecessary_wraps, reason = "needed on some platforms")] +fn args_into_buffer<'a>( + buf: &mut Vec, + i: impl Iterator, +) -> Result<(), Box> { + // On Unix (and wasi), OsStrs are just &[u8]'s underneath... + #[cfg(any(unix, target_os = "wasi"))] + { + #[cfg(unix)] + use std::os::unix::ffi::OsStrExt; + #[cfg(target_os = "wasi")] + use std::os::wasi::ffi::OsStrExt; + + for part in itertools::intersperse(i.map(|a| a.as_bytes()), b" ") { + buf.extend_from_slice(part); + } + } + + // But, on Windows, we must hop through a String. + #[cfg(not(any(unix, target_os = "wasi")))] + { + for part in itertools::intersperse(i.map(|a| a.to_str()), Some(" ")) { + let bytes = match part { + Some(part) => part.as_bytes(), + // pi-uutils: literalized `translate!("yes-error-invalid-utf8")`. + None => return Err("arguments contain invalid UTF-8".into()), + }; + buf.extend_from_slice(bytes); + } + } + + buf.push(b'\n'); + + Ok(()) +} + +/// Assumes buf holds a single output line forged from the command line +/// arguments, copies it repeatedly until the buffer holds as many copies as it +/// can under [`BUF_SIZE`]. +fn prepare_buffer(buf: &mut Vec) { + let line_len = buf.len(); + debug_assert!(line_len > 0, "buffer is not empty since we have newline"); + let target_size = line_len * (BUF_SIZE / line_len); // 0 if line_len is already large enough + + while buf.len() < target_size { + let to_copy = std::cmp::min(target_size - buf.len(), buf.len()); + debug_assert_eq!(to_copy % line_len, 0); + buf.extend_from_within(..to_copy); + } +} + +/// pi-uutils: why the output loop stopped. Upstream's `exec` only ever returns +/// an I/O error (the loop is infinite); in-process we also stop on scope +/// cancellation. +enum ExecStop { + Io(io::Error), + Cancelled, +} + +/// pi-uutils: replacement for upstream's `exec` — writes to the context stdout +/// instead of the process stdout and polls the scope cancel flag every +/// iteration (each iteration writes a full [`BUF_SIZE`]-ish batch, so polling +/// per iteration is cheap) so shell abort/timeout stops the loop promptly. +fn exec(bytes: &[u8]) -> ExecStop { + let mut stdout = pi_uutils_ctx::stdout(); + + loop { + if pi_uutils_ctx::is_cancelled() { + return ExecStop::Cancelled; + } + if let Err(err) = stdout.write_all(bytes) { + return ExecStop::Io(err); + } + } +} + +#[cfg(test)] +mod tests { + use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc}; + + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; + + use super::*; + + /// Writer that accepts up to `budget` bytes into a shared buffer, then + /// fails every further write with `fail_kind` — models a consumer that + /// closes the pipe after reading some output. + struct FailingWriter { + buf: Arc>>, + budget: usize, + fail_kind: io::ErrorKind, + } + impl Write for FailingWriter { + fn write(&mut self, buf: &[u8]) -> io::Result { + if self.budget == 0 { + return Err(io::Error::new(self.fail_kind, "consumer gone")); + } + let n = buf.len().min(self.budget); + self.budget -= n; + self.buf.lock().extend_from_slice(&buf[..n]); + Ok(n) + } + + fn flush(&mut self) -> io::Result<()> { + Ok(()) + } + } + + fn run_with( + args: Vec<&str>, + budget: usize, + fail_kind: io::ErrorKind, + cancelled: bool, + ) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); + + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> io::Result { + self.buf.lock().write(buf) + } + + fn flush(&mut self) -> io::Result<()> { + self.buf.lock().flush() + } + } + + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(FailingWriter { + buf: stdout_buf.clone(), + budget, + fail_kind, + }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd: PathBuf::from("."), + env: HashMap::new(), + cancel: Arc::new(std::sync::atomic::AtomicBool::new(cancelled)), + }; + + let argv: Vec = std::iter::once("yes") + .chain(args) + .map(OsString::from) + .collect(); + + let code = pi_uutils_ctx::scope(io, || run(argv)); + + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + + (code, out_str, err_str) + } + + #[test] + fn broken_pipe_is_clean_exit() { + // Consumer takes 100 bytes then closes: exit 0, like GNU yes dying to + // SIGPIPE without an error status visible to the shell. + let (code, stdout, stderr) = run_with(vec![], 100, io::ErrorKind::BrokenPipe, false); + assert_eq!(code, 0); + assert!(stdout.starts_with("y\ny\n"), "expected default 'y' lines, got {stdout:?}"); + assert_eq!(stdout.len(), 100); + assert_eq!(stderr, ""); + } + + #[test] + fn custom_operands_join_with_spaces_and_repeat() { + // Budget is a multiple of the line length ("hello world\n" = 12 bytes) + // so the captured output is whole lines. + let (code, stdout, stderr) = + run_with(vec!["hello", "world"], 12 * 100, io::ErrorKind::BrokenPipe, false); + assert_eq!(code, 0); + assert_eq!(stdout.lines().count(), 100); + for line in stdout.lines() { + assert_eq!(line, "hello world"); + } + assert_eq!(stderr, ""); + } + + #[test] + fn cancellation_stops_loop_promptly() { + // Pre-set cancel flag: the loop must observe it and return 1 before + // writing anything. The finite write budget is a backstop so a broken + // cancel path fails the test (as exit 0) instead of hanging forever. + let (code, stdout, stderr) = run_with(vec![], 1 << 20, io::ErrorKind::BrokenPipe, true); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert_eq!(stderr, ""); + } + + #[test] + fn non_pipe_write_error_reports_and_fails() { + let (code, stdout, stderr) = run_with(vec![], 2, io::ErrorKind::Other, false); + assert_eq!(code, 1); + assert_eq!(stdout, "y\n"); + assert_eq!(stderr, "yes: standard output: consumer gone\n"); + } + + #[test] + fn help_renders_to_scope_stdout() { + let (code, stdout, stderr) = run_with(vec!["--help"], 1 << 20, io::ErrorKind::Other, false); + assert_eq!(code, 0); + assert!(stdout.contains("Usage:")); + assert!(stdout.contains("Repeatedly display a line")); + assert_eq!(stderr, ""); + } + + // Upstream unit tests (uutils/coreutils 0.8.0), kept verbatim apart from + // indentation. + + #[test] + fn test_prepare_buffer() { + let tests = [ + (150, 16350), + (1000, 16000), + (4093, 16372), + (4099, 12297), + (4111, 12333), + (2, 16384), + (3, 16383), + (4, 16384), + (5, 16380), + (8192, 16384), + (8191, 16382), + (8193, 8193), + (10000, 10000), + (15000, 15000), + (25000, 25000), + ]; + + for (line, final_len) in tests { + let mut v = std::iter::repeat_n(b'a', line).collect::>(); + prepare_buffer(&mut v); + assert_eq!(v.len(), final_len); + } + } + + #[test] + fn test_args_into_buf() { + { + let mut v = Vec::with_capacity(BUF_SIZE); + let default_args = ["y".into()]; + args_into_buffer(&mut v, default_args.iter()).unwrap(); + assert_eq!(String::from_utf8(v).unwrap(), "y\n"); + } + + { + let mut v = Vec::with_capacity(BUF_SIZE); + let args = ["foo".into()]; + args_into_buffer(&mut v, args.iter()).unwrap(); + assert_eq!(String::from_utf8(v).unwrap(), "foo\n"); + } + + { + let mut v = Vec::with_capacity(BUF_SIZE); + let args = ["foo".into(), "bar baz".into(), "qux".into()]; + args_into_buffer(&mut v, args.iter()).unwrap(); + assert_eq!(String::from_utf8(v).unwrap(), "foo bar baz qux\n"); + } + } +} diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index f45d4778f..2852d393e 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -2,6 +2,11 @@ ## [Unreleased] +### Added + +- Added an in-process `readlink` shell builtin (vendored from uutils coreutils 0.8.0), supporting `-f`/`-e`/`-m` canonicalization, `-n`/`-z` delimiters, and `-v`/`-q`/`-s` verbosity, with path operands resolved against the shell working directory. +- Added in-process shell builtins for `realpath`, `touch`, `stat`, `date`, `mktemp`, `seq`, `yes`, `printenv`, `ln`, `truncate`, `tac`, `nproc`, `uname`, `whoami`, and `hostname` (vendored from uutils coreutils 0.8.0), plus native `which` (shell PATH lookup) and `diff` (unified output, `-U`/`-q`/`-N`, binary detection, recursive directory compare) builtins. All resolve path operands against the shell working directory, read the shell's exported environment, and honor abort/timeout cancellation; `ln` is gated with the destructive set (`PI_DISABLE_UUTILS_DESTRUCTIVE`), and system-mutating modes (`date --set`, hostname setting) are disabled. + ## [16.4.5] - 2026-07-11 ### Added