feat: integrated coreutils as in-process shell builtins

- Integrated 17 new coreutils-based shell builtins including `diff`, `date`, `ln`, `stat`, `seq`, `touch`, and others.
- Refactored vendored utilities to execute as in-process shell builtins by routing I/O, environment access, and path resolution through `pi_uutils_ctx`.
- Disabled process-level modifications (e.g., clock setting, hostname modification) to ensure safety and scope adherence within the shell environment.
- Implemented shell-specific features such as cancellation polling, custom exit code management, and efficient output streaming for all new builtins.
This commit is contained in:
can1357
2026-07-11 20:35:56 +02:00
parent 6c292b97c3
commit 0ae8efd649
62 changed files with 13648 additions and 3 deletions
Generated
+272 -3
View File
@@ -1115,6 +1115,18 @@ dependencies = [
"syn",
]
[[package]]
name = "dns-lookup"
version = "3.0.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6e39034cee21a2f5bbb66ba0e3689819c4bb5d00382a282006e802a7ffa6c41d"
dependencies = [
"cfg-if",
"libc",
"socket2",
"windows-sys 0.60.2",
]
[[package]]
name = "downcast-rs"
version = "1.2.1"
@@ -3022,6 +3034,17 @@ dependencies = [
"windows-link",
]
[[package]]
name = "parse_datetime"
version = "0.14.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "413775a7eac2261d2211a79d10ef275e5b6f7b527eec42ad09adce2ffa92b6e5"
dependencies = [
"jiff",
"num-traits",
"winnow 0.7.15",
]
[[package]]
name = "pcre2"
version = "0.2.11"
@@ -3392,6 +3415,7 @@ dependencies = [
"parking_lot",
"pi-uutils-ctx",
"pi-walker",
"pi_uu_diff",
"pi_uu_grep",
"regex",
"serde",
@@ -3406,28 +3430,44 @@ dependencies = [
"uu_cat",
"uu_comm",
"uu_cut",
"uu_date",
"uu_dirname",
"uu_find",
"uu_head",
"uu_hostname",
"uu_ln",
"uu_ls",
"uu_md5sum",
"uu_mkdir",
"uu_mktemp",
"uu_mv",
"uu_nproc",
"uu_paste",
"uu_printenv",
"uu_readlink",
"uu_realpath",
"uu_rm",
"uu_sed",
"uu_seq",
"uu_sha1sum",
"uu_sha224sum",
"uu_sha256sum",
"uu_sha384sum",
"uu_sha512sum",
"uu_sort",
"uu_stat",
"uu_tac",
"uu_tail",
"uu_tee",
"uu_touch",
"uu_tr",
"uu_truncate",
"uu_uname",
"uu_uniq",
"uu_wc",
"uu_whoami",
"uu_xargs",
"uu_yes",
"windows-sys 0.61.2",
"winreg 0.56.0",
"xxhash-rust",
@@ -3453,6 +3493,17 @@ dependencies = [
"windows-sys 0.61.2",
]
[[package]]
name = "pi_uu_diff"
version = "0.8.0"
dependencies = [
"clap",
"parking_lot",
"pi-uutils-ctx",
"similar 3.1.1",
"tempfile",
]
[[package]]
name = "pi_uu_grep"
version = "0.8.0"
@@ -3484,6 +3535,16 @@ version = "0.3.33"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e"
[[package]]
name = "platform-info"
version = "2.1.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9368d62437c8cbb7c31ee37fd8c08a7d390e09a3ff75698a674953f46705ffcb"
dependencies = [
"libc",
"windows-sys 0.59.0",
]
[[package]]
name = "png"
version = "0.18.1"
@@ -4474,7 +4535,7 @@ dependencies = [
"toml_datetime",
"toml_parser",
"toml_writer",
"winnow",
"winnow 1.0.3",
]
[[package]]
@@ -4495,7 +4556,7 @@ dependencies = [
"indexmap",
"toml_datetime",
"toml_parser",
"winnow",
"winnow 1.0.3",
]
[[package]]
@@ -4504,7 +4565,7 @@ version = "1.1.2+spec-1.1.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a2abe9b86193656635d2411dc43050282ca48aa31c2451210f4202550afb7526"
dependencies = [
"winnow",
"winnow 1.0.3",
]
[[package]]
@@ -5338,6 +5399,21 @@ dependencies = [
"uucore 0.8.0",
]
[[package]]
name = "uu_date"
version = "0.8.0"
dependencies = [
"clap",
"jiff",
"parking_lot",
"parse_datetime",
"pi-uutils-ctx",
"regex",
"rustix",
"tempfile",
"uucore 0.8.0",
]
[[package]]
name = "uu_dirname"
version = "0.8.0"
@@ -5377,6 +5453,31 @@ dependencies = [
"uucore 0.8.0",
]
[[package]]
name = "uu_hostname"
version = "0.8.0"
dependencies = [
"clap",
"dns-lookup",
"hostname",
"parking_lot",
"pi-uutils-ctx",
"uucore 0.8.0",
"windows-sys 0.61.2",
]
[[package]]
name = "uu_ln"
version = "0.8.0"
dependencies = [
"clap",
"parking_lot",
"pi-uutils-ctx",
"tempfile",
"thiserror 2.0.18",
"uucore 0.8.0",
]
[[package]]
name = "uu_ls"
version = "0.8.0"
@@ -5412,6 +5513,19 @@ dependencies = [
"uucore 0.8.0",
]
[[package]]
name = "uu_mktemp"
version = "0.8.0"
dependencies = [
"clap",
"parking_lot",
"pi-uutils-ctx",
"rand 0.10.2",
"tempfile",
"thiserror 2.0.18",
"uucore 0.8.0",
]
[[package]]
name = "uu_mv"
version = "0.8.0"
@@ -5427,6 +5541,17 @@ dependencies = [
"windows-sys 0.61.2",
]
[[package]]
name = "uu_nproc"
version = "0.8.0"
dependencies = [
"clap",
"libc",
"parking_lot",
"pi-uutils-ctx",
"uucore 0.8.0",
]
[[package]]
name = "uu_paste"
version = "0.8.0"
@@ -5436,6 +5561,38 @@ dependencies = [
"uucore 0.8.0",
]
[[package]]
name = "uu_printenv"
version = "0.8.0"
dependencies = [
"clap",
"parking_lot",
"pi-uutils-ctx",
"uucore 0.8.0",
]
[[package]]
name = "uu_readlink"
version = "0.8.0"
dependencies = [
"clap",
"parking_lot",
"pi-uutils-ctx",
"tempfile",
"uucore 0.8.0",
]
[[package]]
name = "uu_realpath"
version = "0.8.0"
dependencies = [
"clap",
"parking_lot",
"pi-uutils-ctx",
"tempfile",
"uucore 0.8.0",
]
[[package]]
name = "uu_rm"
version = "0.8.0"
@@ -5464,6 +5621,20 @@ dependencies = [
"uucore 0.9.0",
]
[[package]]
name = "uu_seq"
version = "0.8.0"
dependencies = [
"bigdecimal",
"clap",
"num-bigint",
"num-traits",
"parking_lot",
"pi-uutils-ctx",
"thiserror 2.0.18",
"uucore 0.8.0",
]
[[package]]
name = "uu_sha1sum"
version = "0.8.0"
@@ -5536,6 +5707,33 @@ dependencies = [
"uucore 0.8.0",
]
[[package]]
name = "uu_stat"
version = "0.8.0"
dependencies = [
"clap",
"parking_lot",
"pi-uutils-ctx",
"tempfile",
"thiserror 2.0.18",
"uucore 0.8.0",
]
[[package]]
name = "uu_tac"
version = "0.8.0"
dependencies = [
"clap",
"memchr",
"memmap2",
"parking_lot",
"pi-uutils-ctx",
"regex",
"tempfile",
"thiserror 2.0.18",
"uucore 0.8.0",
]
[[package]]
name = "uu_tail"
version = "0.8.0"
@@ -5560,6 +5758,24 @@ dependencies = [
"uucore 0.8.0",
]
[[package]]
name = "uu_touch"
version = "0.8.0"
dependencies = [
"clap",
"filetime",
"jiff",
"libc",
"parking_lot",
"parse_datetime",
"pi-uutils-ctx",
"rustix",
"tempfile",
"thiserror 2.0.18",
"uucore 0.8.0",
"windows-sys 0.61.2",
]
[[package]]
name = "uu_tr"
version = "0.8.0"
@@ -5571,6 +5787,28 @@ dependencies = [
"uucore 0.8.0",
]
[[package]]
name = "uu_truncate"
version = "0.8.0"
dependencies = [
"clap",
"parking_lot",
"pi-uutils-ctx",
"tempfile",
"uucore 0.8.0",
]
[[package]]
name = "uu_uname"
version = "0.8.0"
dependencies = [
"clap",
"parking_lot",
"pi-uutils-ctx",
"platform-info",
"uucore 0.8.0",
]
[[package]]
name = "uu_uniq"
version = "0.8.0"
@@ -5594,6 +5832,17 @@ dependencies = [
"uucore 0.8.0",
]
[[package]]
name = "uu_whoami"
version = "0.8.0"
dependencies = [
"clap",
"parking_lot",
"pi-uutils-ctx",
"uucore 0.8.0",
"windows-sys 0.61.2",
]
[[package]]
name = "uu_xargs"
version = "0.8.0"
@@ -5605,6 +5854,17 @@ dependencies = [
"tempfile",
]
[[package]]
name = "uu_yes"
version = "0.8.0"
dependencies = [
"clap",
"itertools",
"parking_lot",
"pi-uutils-ctx",
"uucore 0.8.0",
]
[[package]]
name = "uucore"
version = "0.0.30"
@@ -6330,6 +6590,15 @@ version = "0.53.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d6bbff5f0aada427a1e5a6da5f1f98158182f26556f345ac9e04d36d0ebed650"
[[package]]
name = "winnow"
version = "0.7.15"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "df79d97927682d2fd8adb29682d1140b343be4ac0f08fd68b7765d9c059d3945"
dependencies = [
"memchr",
]
[[package]]
name = "winnow"
version = "1.0.3"
+17
View File
@@ -54,6 +54,23 @@ uu_sha512sum = { path = "../vendor/uu-sha512sum" }
uu_b2sum = { path = "../vendor/uu-b2sum" }
uu_basename = { path = "../vendor/uu-basename" }
uu_dirname = { path = "../vendor/uu-dirname" }
uu_readlink = { path = "../vendor/uu-readlink" }
uu_realpath = { path = "../vendor/uu-realpath" }
uu_touch = { path = "../vendor/uu-touch" }
uu_stat = { path = "../vendor/uu-stat" }
uu_date = { path = "../vendor/uu-date" }
uu_mktemp = { path = "../vendor/uu-mktemp" }
uu_seq = { path = "../vendor/uu-seq" }
uu_yes = { path = "../vendor/uu-yes" }
uu_printenv = { path = "../vendor/uu-printenv" }
uu_ln = { path = "../vendor/uu-ln" }
uu_truncate = { path = "../vendor/uu-truncate" }
uu_tac = { path = "../vendor/uu-tac" }
uu_nproc = { path = "../vendor/uu-nproc" }
uu_uname = { path = "../vendor/uu-uname" }
uu_whoami = { path = "../vendor/uu-whoami" }
uu_hostname = { path = "../vendor/uu-hostname" }
pi_uu_diff = { path = "../pi-uu-diff" }
uu_cut = { path = "../vendor/uu-cut" }
uu_tee = { path = "../vendor/uu-tee" }
uu_tr = { path = "../vendor/uu-tr" }
+17
View File
@@ -219,6 +219,23 @@ uutil_builtin!(pub fn sha512sum_builtin => uu_sha512sum::run);
uutil_builtin!(pub fn b2sum_builtin => uu_b2sum::run);
uutil_builtin!(pub fn basename_builtin => uu_basename::run);
uutil_builtin!(pub fn dirname_builtin => uu_dirname::run);
uutil_builtin!(pub fn readlink_builtin => uu_readlink::run);
uutil_builtin!(pub fn realpath_builtin => uu_realpath::run);
uutil_builtin!(pub fn touch_builtin => uu_touch::run);
uutil_builtin!(pub fn stat_builtin => uu_stat::run);
uutil_builtin!(pub fn date_builtin => uu_date::run);
uutil_builtin!(pub fn mktemp_builtin => uu_mktemp::run);
uutil_builtin!(pub fn seq_builtin => uu_seq::run);
uutil_builtin!(pub fn yes_builtin => uu_yes::run);
uutil_builtin!(pub fn printenv_builtin => uu_printenv::run);
uutil_builtin!(pub fn ln_builtin => uu_ln::run);
uutil_builtin!(pub fn truncate_builtin => uu_truncate::run);
uutil_builtin!(pub fn tac_builtin => uu_tac::run);
uutil_builtin!(pub fn nproc_builtin => uu_nproc::run);
uutil_builtin!(pub fn uname_builtin => uu_uname::run);
uutil_builtin!(pub fn whoami_builtin => uu_whoami::run);
uutil_builtin!(pub fn hostname_builtin => uu_hostname::run);
uutil_builtin!(pub fn diff_builtin => pi_uu_diff::run);
uutil_builtin!(pub fn cut_builtin => uu_cut::run);
uutil_builtin!(pub fn tee_builtin => uu_tee::run);
uutil_builtin!(pub fn tr_builtin => uu_tr::run);
+1
View File
@@ -4,6 +4,7 @@ mod fd;
pub mod minimizer;
pub mod process;
pub mod shell;
mod which;
#[cfg(windows)]
pub mod windows;
+19
View File
@@ -632,6 +632,23 @@ async fn create_session_for_run(
shell.register_builtin("b2sum", crate::coreutils::b2sum_builtin());
shell.register_builtin("basename", crate::coreutils::basename_builtin());
shell.register_builtin("dirname", crate::coreutils::dirname_builtin());
shell.register_builtin("readlink", crate::coreutils::readlink_builtin());
shell.register_builtin("realpath", crate::coreutils::realpath_builtin());
shell.register_builtin("touch", crate::coreutils::touch_builtin());
shell.register_builtin("stat", crate::coreutils::stat_builtin());
shell.register_builtin("date", crate::coreutils::date_builtin());
shell.register_builtin("mktemp", crate::coreutils::mktemp_builtin());
shell.register_builtin("seq", crate::coreutils::seq_builtin());
shell.register_builtin("yes", crate::coreutils::yes_builtin());
shell.register_builtin("printenv", crate::coreutils::printenv_builtin());
shell.register_builtin("truncate", crate::coreutils::truncate_builtin());
shell.register_builtin("tac", crate::coreutils::tac_builtin());
shell.register_builtin("nproc", crate::coreutils::nproc_builtin());
shell.register_builtin("uname", crate::coreutils::uname_builtin());
shell.register_builtin("whoami", crate::coreutils::whoami_builtin());
shell.register_builtin("hostname", crate::coreutils::hostname_builtin());
shell.register_builtin("which", crate::which::which_builtin());
shell.register_builtin("diff", crate::coreutils::diff_builtin());
shell.register_builtin("cut", crate::coreutils::cut_builtin());
shell.register_builtin("tee", crate::coreutils::tee_builtin());
shell.register_builtin("tr", crate::coreutils::tr_builtin());
@@ -647,6 +664,8 @@ async fn create_session_for_run(
if !uutils_env_disabled(config, "PI_DISABLE_MV_BUILTIN") {
shell.register_builtin("mv", crate::coreutils::mv_builtin());
}
// ln can clobber existing files via -f; gate it with the destructive set.
shell.register_builtin("ln", crate::coreutils::ln_builtin());
}
}
+252
View File
@@ -0,0 +1,252 @@
//! In-process `which` builtin backed by brush's PATH-search helpers.
//!
//! Follows which(1) (GNU/debianutils) semantics: each name operand is looked
//! up in the shell's `PATH`; the first match is printed (all matches with
//! `-a`). Lookup failures are silent; the exit status is 0 when every name
//! was found and 1 when any name was missing.
use std::{
ffi::OsString,
io::{self, Write},
path::{Path, PathBuf},
};
use brush_core::{
Error,
builtins::{BoxFuture, ContentOptions, ContentType, Registration},
commands::{CommandArg, ExecutionContext},
extensions::ShellExtensions,
openfiles::{OpenFile, OpenFiles, null},
pathsearch,
results::ExecutionResult,
sys,
};
use clap::{Parser, error::ErrorKind};
#[derive(Parser, Debug)]
#[command(name = "which", about = "Locate a command's executable in the shell's PATH")]
struct WhichCli {
/// Print all matching executables in PATH, not just the first.
#[arg(short = 'a', long = "all")]
all: bool,
/// Command names to locate.
#[arg(value_name = "name")]
names: Vec<String>,
}
/// Creates the `which` shell builtin registration.
pub fn which_builtin<SE: ShellExtensions>() -> Registration<SE> {
fn execute<SE: ShellExtensions>(
context: ExecutionContext<'_, SE>,
args: Vec<CommandArg>,
) -> BoxFuture<'_, Result<ExecutionResult, Error>> {
Box::pin(std::future::ready(Ok(run_which(context, args))))
}
Registration {
execute_func: execute::<SE>,
content_func: which_content,
disabled: false,
special_builtin: false,
declaration_builtin: false,
transparent_background_wrapper: false,
}
}
fn run_which<SE: ShellExtensions>(
context: ExecutionContext<'_, SE>,
args: Vec<CommandArg>,
) -> ExecutionResult {
let mut stdout = context
.try_fd(OpenFiles::STDOUT_FD)
.unwrap_or_else(null_sink);
let mut stderr = context
.try_fd(OpenFiles::STDERR_FD)
.unwrap_or_else(null_sink);
let cwd = context.shell.working_dir().to_path_buf();
let path_var = context
.shell
.env_str("PATH")
.map(std::borrow::Cow::into_owned)
.unwrap_or_default();
let argv: Vec<OsString> = args
.iter()
.map(|arg| OsString::from(arg.to_string()))
.collect();
let cli = match WhichCli::try_parse_from(argv) {
Ok(cli) => cli,
Err(err) => {
let rendered = err.to_string();
let code = match err.kind() {
ErrorKind::DisplayHelp | ErrorKind::DisplayVersion => {
let _ = write!(stdout, "{rendered}");
0
},
_ => {
let _ = write!(stderr, "{rendered}");
2
},
};
return ExecutionResult::new(code);
},
};
let mut all_found = true;
for name in &cli.names {
let matches = find_matches(name, &path_var, &cwd, cli.all);
if matches.is_empty() {
// which(1) reports missing names via the exit status only.
all_found = false;
}
for path in matches {
let _ = writeln!(stdout, "{}", path.display());
}
}
ExecutionResult::new(u8::from(!all_found))
}
/// Collects the executable matches for a single `which` name operand.
///
/// A name containing a path separator is checked directly against `cwd`
/// (yielding at most one match); otherwise each `PATH` entry — with relative
/// and empty entries resolved against `cwd` — is probed in `PATH` order.
/// Returns only the first match unless `all` is set. Windows `PATHEXT`
/// resolution is handled by [`brush_core::sys::fs::resolve_executable`].
fn find_matches(name: &str, path_var: &str, cwd: &Path, all: bool) -> Vec<PathBuf> {
if sys::fs::contains_path_separator(name) {
let candidate = cwd.join(name);
if candidate.is_dir() {
return Vec::new();
}
return sys::fs::resolve_executable(candidate).into_iter().collect();
}
let dirs = sys::fs::split_paths(path_var).map(|dir| {
if dir.as_os_str().is_empty() {
// POSIX: an empty PATH entry names the current directory.
cwd.to_path_buf()
} else if dir.is_relative() {
cwd.join(dir)
} else {
dir
}
});
let mut found = pathsearch::search_for_executable(dirs, name);
if all {
found.collect()
} else {
found.next().into_iter().collect()
}
}
fn null_sink() -> OpenFile {
null().unwrap_or_else(|_| OpenFile::from(io::stdout()))
}
#[allow(
clippy::unnecessary_wraps,
reason = "signature must match brush's CommandContentFunc fn pointer"
)]
fn which_content(
_name: &str,
_content_type: ContentType,
_options: &ContentOptions,
) -> Result<String, Error> {
Ok("which: which [-a] name [name ...]\n".to_string())
}
#[cfg(test)]
#[cfg(unix)]
mod tests {
use std::{
env, fs,
os::unix::fs::PermissionsExt,
path::PathBuf,
sync::atomic::{AtomicUsize, Ordering},
time::{SystemTime, UNIX_EPOCH},
};
use super::find_matches;
static COUNTER: AtomicUsize = AtomicUsize::new(0);
/// Creates a fresh, canonicalized temp directory (macOS `/var` is a
/// symlink; canonicalizing keeps constructed and probed paths identical).
fn temp_root(tag: &str) -> PathBuf {
let nanos = SystemTime::now()
.duration_since(UNIX_EPOCH)
.map_or(0, |d| d.as_nanos());
let root = env::temp_dir().join(format!(
"pi-shell-which-{tag}-{}-{}-{}",
std::process::id(),
nanos,
COUNTER.fetch_add(1, Ordering::Relaxed),
));
fs::create_dir_all(&root).expect("temp dir should be created");
fs::canonicalize(&root).expect("temp dir should canonicalize")
}
fn place_file(dir: &std::path::Path, name: &str, executable: bool) -> PathBuf {
let path = dir.join(name);
fs::write(&path, b"#!/bin/sh\n").expect("file should be written");
let mode = if executable { 0o755 } else { 0o644 };
fs::set_permissions(&path, fs::Permissions::from_mode(mode))
.expect("permissions should be set");
path
}
#[test]
fn finds_only_executable_files() {
let dir = temp_root("exec-only");
let tool = place_file(&dir, "tool", true);
place_file(&dir, "blob", false);
let path_var = dir.display().to_string();
assert_eq!(find_matches("tool", &path_var, &dir, false), vec![tool]);
assert!(find_matches("blob", &path_var, &dir, false).is_empty());
assert!(find_matches("missing", &path_var, &dir, false).is_empty());
}
#[test]
fn all_flag_returns_matches_in_path_order() {
let dir_a = temp_root("all-a");
let dir_b = temp_root("all-b");
let tool_a = place_file(&dir_a, "tool", true);
let tool_b = place_file(&dir_b, "tool", true);
let path_var = format!("{}:{}", dir_a.display(), dir_b.display());
let cwd = temp_root("all-cwd");
assert_eq!(find_matches("tool", &path_var, &cwd, true), vec![tool_a.clone(), tool_b]);
// Without -a only the first PATH entry's match is returned.
assert_eq!(find_matches("tool", &path_var, &cwd, false), vec![tool_a]);
}
#[test]
fn name_with_separator_resolves_against_cwd() {
let cwd = temp_root("slash");
let bin = cwd.join("bin");
fs::create_dir_all(&bin).expect("bin dir should be created");
let tool = place_file(&bin, "tool", true);
place_file(&bin, "blob", false);
// PATH is irrelevant for names containing a separator.
assert_eq!(find_matches("bin/tool", "", &cwd, false), vec![tool]);
assert!(find_matches("bin/blob", "", &cwd, false).is_empty());
// A directory is never a match, even with execute bits set.
assert!(find_matches("./bin", "", &cwd, false).is_empty());
}
#[test]
fn relative_path_entries_resolve_against_cwd() {
let cwd = temp_root("rel-entry");
let bin = cwd.join("bin");
fs::create_dir_all(&bin).expect("bin dir should be created");
let tool = place_file(&bin, "tool", true);
assert_eq!(find_matches("tool", "bin", &cwd, false), vec![tool]);
}
}
+21
View File
@@ -0,0 +1,21 @@
# diff implemented from scratch on top of the `similar` diffing library, with
# I/O and path resolution routed through pi-uutils-ctx so it can run in-process
# as a shell builtin. Entry point: `pi_uu_diff::run`.
[package]
name = "pi_uu_diff"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "diff ~ similar-backed file comparison (in-process shell builtin)"
[lib]
path = "src/lib.rs"
[dependencies]
clap = { version = "4", features = ["wrap_help"] }
pi-uutils-ctx = { path = "../pi-uutils-ctx" }
similar = "3.1.0"
[dev-dependencies]
parking_lot.workspace = true
tempfile = "3"
+608
View File
@@ -0,0 +1,608 @@
//! `diff` implemented as an in-process shell builtin on top of the `similar`
//! diffing library. All I/O and path resolution is routed through
//! `pi-uutils-ctx` so the builtin writes to the command's redirected file
//! descriptors and resolves relative paths against the shell's working
//! directory, while operands are printed as typed.
//!
//! Scope: unified output only (`-u` is accepted and implied, `-U N` controls
//! the context size), `-q/--brief`, `-N/--new-file` (absent files compare as
//! empty), binary detection, `-` for the context stdin, and unconditional
//! recursive directory comparison (`Only in <dir>: <name>` lines plus
//! `diff -r A/x B/x`-headed per-pair diffs).
//!
//! Entry point: [`run`]. It never calls `std::process::exit`; clap
//! help/usage/error output is rendered to the context streams and an exit code
//! is returned following the GNU convention (0 = identical, 1 = differences
//! found, 2 = trouble).
use std::{
collections::BTreeSet,
ffi::{OsStr, OsString},
fs,
io::{Read, Write},
path::{Path, PathBuf},
};
use clap::{Arg, ArgAction, ArgMatches, Command};
use pi_uutils_ctx::format_usage;
use similar::TextDiff;
const OPT_UNIFIED_FLAG: &str = "unified-flag";
const OPT_UNIFIED: &str = "unified";
const OPT_BRIEF: &str = "brief";
const OPT_RECURSIVE: &str = "recursive";
const OPT_NEW_FILE: &str = "new-file";
const OPT_COLOR: &str = "color";
const ARG_FILES: &str = "files";
/// In-process builtin entry point. Parses the arguments directly, renders clap
/// help/usage/version to the context streams, and maps errors to the GNU diff
/// exit-code convention, so it is safe to run inside the host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 2;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
match diff_main(&matches) {
Ok(code) => code,
Err(msg) => {
let _ = writeln!(pi_uutils_ctx::stderr(), "diff: {msg}");
2
},
}
}
pub fn uu_app() -> Command {
Command::new("diff")
.version(concat!("diff (pi-uu-diff) ", env!("CARGO_PKG_VERSION")))
.about("Compare files line by line.")
.override_usage(format_usage("diff [OPTION]... FILE1 FILE2"))
.infer_long_args(true)
.arg(
Arg::new(OPT_UNIFIED_FLAG)
.short('u')
.help("output 3 lines of unified context (the default output format)")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_UNIFIED)
.short('U')
.long(OPT_UNIFIED)
.value_name("NUM")
.help("output NUM lines of unified context")
.value_parser(clap::value_parser!(usize)),
)
.arg(
Arg::new(OPT_BRIEF)
.short('q')
.long(OPT_BRIEF)
.help("report only when files differ")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_RECURSIVE)
.short('r')
.long(OPT_RECURSIVE)
.help("recursively compare subdirectories (always on for directories)")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_NEW_FILE)
.short('N')
.long(OPT_NEW_FILE)
.help("treat absent files as empty")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_COLOR)
.long(OPT_COLOR)
.value_name("WHEN")
.num_args(0..=1)
.require_equals(true)
.default_missing_value("auto")
.help("accepted for compatibility; output is never colorized"),
)
.arg(
Arg::new(ARG_FILES)
.required(true)
.num_args(2)
.value_parser(clap::value_parser!(OsString))
.value_hint(clap::ValueHint::AnyPath),
)
}
#[derive(Clone, Copy)]
struct Options {
context: usize,
brief: bool,
new_file: bool,
}
/// A classified operand: what the name as typed refers to on disk after
/// resolution against the scope working directory.
enum Operand {
/// The context stdin (`-`).
Stdin,
/// A regular (or other non-directory) file at the resolved path.
File(PathBuf),
/// A directory at the resolved path.
Dir(PathBuf),
/// A missing file tolerated by `-N` and compared as empty.
Absent,
}
fn diff_main(matches: &ArgMatches) -> Result<i32, String> {
let files: Vec<&OsString> = matches.get_many::<OsString>(ARG_FILES).unwrap().collect();
let opts = Options {
context: matches.get_one::<usize>(OPT_UNIFIED).copied().unwrap_or(3),
brief: matches.get_flag(OPT_BRIEF),
new_file: matches.get_flag(OPT_NEW_FILE),
};
let (mut name_a, mut name_b) = (PathBuf::from(files[0]), PathBuf::from(files[1]));
let mut op_a = classify(&name_a, opts.new_file)?;
let mut op_b = classify(&name_b, opts.new_file)?;
// GNU: comparing a directory with a non-directory compares
// <dir>/<basename-of-other> with the other operand.
let a_is_dir = matches!(op_a, Operand::Dir(_));
let b_is_dir = matches!(op_b, Operand::Dir(_));
if a_is_dir != b_is_dir {
if matches!(op_a, Operand::Stdin) || matches!(op_b, Operand::Stdin) {
return Err("cannot compare '-' to a directory".to_string());
}
if a_is_dir {
name_a = descend(&name_a, &name_b)?;
op_a = classify(&name_a, opts.new_file)?;
} else {
name_b = descend(&name_b, &name_a)?;
op_b = classify(&name_b, opts.new_file)?;
}
}
let differed = if let (Operand::Dir(res_a), Operand::Dir(res_b)) = (&op_a, &op_b) {
diff_dirs(&name_a, res_a, &name_b, res_b, opts)?
} else {
let bytes_a = read_operand(&op_a, &name_a)?;
let bytes_b = read_operand(&op_b, &name_b)?;
diff_pair(&name_a, &bytes_a, &name_b, &bytes_b, opts, None)?
};
Ok(i32::from(differed))
}
/// Replaces a directory operand with `<dir>/<basename of other>` for the GNU
/// dir-vs-file comparison form.
fn descend(dir: &Path, other: &Path) -> Result<PathBuf, String> {
let base = other
.file_name()
.ok_or_else(|| format!("cannot compare {} to a directory", other.display()))?;
Ok(dir.join(base))
}
fn classify(name: &Path, new_file: bool) -> Result<Operand, String> {
if name.as_os_str() == OsStr::new("-") {
return Ok(Operand::Stdin);
}
// Resolve the operand against the shell working directory; `name` is kept
// for display (GNU prints operands as typed).
let resolved = pi_uutils_ctx::resolve(name);
match fs::metadata(&resolved) {
Ok(meta) if meta.is_dir() => Ok(Operand::Dir(resolved)),
Ok(_) => Ok(Operand::File(resolved)),
Err(err) if err.kind() == std::io::ErrorKind::NotFound && new_file => Ok(Operand::Absent),
Err(err) => Err(format!("{}: {}", name.display(), io_msg(&err))),
}
}
fn read_operand(op: &Operand, name: &Path) -> Result<Vec<u8>, String> {
match op {
Operand::Stdin => {
let mut buf = Vec::new();
pi_uutils_ctx::stdin()
.read_to_end(&mut buf)
.map_err(|err| format!("-: {}", io_msg(&err)))?;
Ok(buf)
},
Operand::File(resolved) => {
fs::read(resolved).map_err(|err| format!("{}: {}", name.display(), io_msg(&err)))
},
Operand::Dir(_) => unreachable!("directories are handled by diff_dirs"),
Operand::Absent => Ok(Vec::new()),
}
}
/// Diffs one pair of already-read inputs, writing to the context stdout.
/// `prefix` is the `diff -r A/x B/x` line emitted before per-pair output in
/// directory mode. Returns whether the inputs differed.
fn diff_pair(
name_a: &Path,
bytes_a: &[u8],
name_b: &Path,
bytes_b: &[u8],
opts: Options,
prefix: Option<&str>,
) -> Result<bool, String> {
if bytes_a == bytes_b {
return Ok(false);
}
let mut out = pi_uutils_ctx::stdout();
let (label_a, label_b) = (name_a.display().to_string(), name_b.display().to_string());
if opts.brief {
writeln!(out, "Files {label_a} and {label_b} differ").map_err(|e| io_msg(&e))?;
return Ok(true);
}
if is_binary(bytes_a) || is_binary(bytes_b) {
writeln!(out, "Binary files {label_a} and {label_b} differ").map_err(|e| io_msg(&e))?;
return Ok(true);
}
if let Some(line) = prefix {
writeln!(out, "{line}").map_err(|e| io_msg(&e))?;
}
let old = String::from_utf8_lossy(bytes_a);
let new = String::from_utf8_lossy(bytes_b);
let diff = TextDiff::from_lines(old.as_ref(), new.as_ref());
write!(
out,
"{}",
diff
.unified_diff()
.context_radius(opts.context)
.header(&label_a, &label_b)
)
.map_err(|e| io_msg(&e))?;
Ok(true)
}
/// Recursively compares two directories over the sorted union of their
/// entries, GNU `diff -r` style. Returns whether any difference was found.
fn diff_dirs(
name_a: &Path,
res_a: &Path,
name_b: &Path,
res_b: &Path,
opts: Options,
) -> Result<bool, String> {
let mut names: BTreeSet<OsString> = BTreeSet::new();
for (dir_name, dir_res) in [(name_a, res_a), (name_b, res_b)] {
let entries = fs::read_dir(dir_res)
.map_err(|err| format!("{}: {}", dir_name.display(), io_msg(&err)))?;
for entry in entries {
let entry = entry.map_err(|err| format!("{}: {}", dir_name.display(), io_msg(&err)))?;
names.insert(entry.file_name());
}
}
let mut differed = false;
for name in names {
if pi_uutils_ctx::is_cancelled() {
return Err("interrupted".to_string());
}
let (child_name_a, child_res_a) = (name_a.join(&name), res_a.join(&name));
let (child_name_b, child_res_b) = (name_b.join(&name), res_b.join(&name));
let meta_a = fs::metadata(&child_res_a).ok();
let meta_b = fs::metadata(&child_res_b).ok();
match (meta_a.as_ref(), meta_b.as_ref()) {
(Some(ma), Some(mb)) if ma.is_dir() && mb.is_dir() => {
differed |= diff_dirs(&child_name_a, &child_res_a, &child_name_b, &child_res_b, opts)?;
},
(Some(ma), Some(mb)) if ma.is_dir() != mb.is_dir() => {
let (dir, file) = if ma.is_dir() {
(&child_name_a, &child_name_b)
} else {
(&child_name_b, &child_name_a)
};
writeln!(
pi_uutils_ctx::stdout(),
"File {} is a directory while file {} is a regular file",
dir.display(),
file.display()
)
.map_err(|e| io_msg(&e))?;
differed = true;
},
(Some(_), Some(_)) => {
let bytes_a = fs::read(&child_res_a)
.map_err(|err| format!("{}: {}", child_name_a.display(), io_msg(&err)))?;
let bytes_b = fs::read(&child_res_b)
.map_err(|err| format!("{}: {}", child_name_b.display(), io_msg(&err)))?;
let prefix = format!("diff -r {} {}", child_name_a.display(), child_name_b.display());
differed |=
diff_pair(&child_name_a, &bytes_a, &child_name_b, &bytes_b, opts, Some(&prefix))?;
},
(Some(meta), None) | (None, Some(meta)) => {
let in_a = meta_b.is_none();
if opts.new_file && meta.is_file() {
// -N: compare the present file against an empty absent one.
let (present_name, present_res) = if in_a {
(&child_name_a, &child_res_a)
} else {
(&child_name_b, &child_res_b)
};
let bytes = fs::read(present_res)
.map_err(|err| format!("{}: {}", present_name.display(), io_msg(&err)))?;
let prefix =
format!("diff -r {} {}", child_name_a.display(), child_name_b.display());
let (ba, bb): (&[u8], &[u8]) = if in_a { (&bytes, &[]) } else { (&[], &bytes) };
differed |= diff_pair(&child_name_a, ba, &child_name_b, bb, opts, Some(&prefix))?;
} else {
let present_dir = if in_a { name_a } else { name_b };
writeln!(
pi_uutils_ctx::stdout(),
"Only in {}: {}",
present_dir.display(),
Path::new(&name).display()
)
.map_err(|e| io_msg(&e))?;
differed = true;
}
},
(None, None) => {},
}
}
Ok(differed)
}
/// NUL byte within the first 8 KiB marks the input as binary, matching the
/// heuristic GNU diff applies to decide between text and binary output.
fn is_binary(bytes: &[u8]) -> bool {
bytes.iter().take(8192).any(|&b| b == 0)
}
/// Renders an I/O error without the Rust-specific ` (os error N)` suffix so
/// messages read like GNU diff's (`diff: x: No such file or directory`).
fn io_msg(err: &std::io::Error) -> String {
let msg = err.to_string();
match msg.find(" (os error") {
Some(idx) => msg[..idx].to_string(),
None => msg,
}
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc};
use parking_lot::Mutex;
use pi_uutils_ctx::ScopeIo;
use super::*;
fn run_with(cwd: PathBuf, stdin: &[u8], args: Vec<&str>) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
self.buf.lock().write(buf)
}
fn flush(&mut self) -> std::io::Result<()> {
self.buf.lock().flush()
}
}
let io = ScopeIo {
stdin: Box::new(std::io::Cursor::new(stdin.to_vec())),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd,
env: HashMap::new(),
cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)),
};
let argv: Vec<OsString> = std::iter::once("diff")
.chain(args)
.map(OsString::from)
.collect();
let code = pi_uutils_ctx::scope(io, || run(argv));
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
(code, out_str, err_str)
}
fn run_in(cwd: PathBuf, args: Vec<&str>) -> (i32, String, String) {
run_with(cwd, b"", args)
}
/// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var).
fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) {
let dir = tempfile::tempdir().unwrap();
let canon = fs::canonicalize(dir.path()).unwrap();
(dir, canon)
}
#[test]
fn identical_files_print_nothing_and_exit_zero() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("a.txt"), "one\ntwo\n").unwrap();
fs::write(root.join("b.txt"), "one\ntwo\n").unwrap();
let (code, stdout, stderr) = run_in(root, vec!["a.txt", "b.txt"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", ""));
}
/// Relative operands must resolve against the scope cwd (a tempdir), not
/// the process cwd — the pi-specific contract.
#[test]
fn differing_files_emit_unified_diff_with_typed_headers() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("a.txt"), "one\ntwo\nthree\n").unwrap();
fs::write(root.join("b.txt"), "one\nTWO\nthree\n").unwrap();
let (code, stdout, stderr) = run_in(root, vec!["a.txt", "b.txt"]);
assert_eq!(code, 1);
assert_eq!(stderr, "");
assert!(stdout.starts_with("--- a.txt\n+++ b.txt\n@@ "), "got: {stdout}");
assert!(stdout.contains("\n-two\n"), "got: {stdout}");
assert!(stdout.contains("\n+TWO\n"), "got: {stdout}");
// Context lines around the change (default -U 3).
assert!(stdout.contains("\n one\n"), "got: {stdout}");
assert!(stdout.contains("\n three\n"), "got: {stdout}");
}
#[test]
fn unified_zero_drops_context_lines() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("a.txt"), "one\ntwo\nthree\n").unwrap();
fs::write(root.join("b.txt"), "one\nTWO\nthree\n").unwrap();
let (code, stdout, _) = run_in(root, vec!["-U", "0", "a.txt", "b.txt"]);
assert_eq!(code, 1);
assert!(!stdout.contains("\n one\n"), "got: {stdout}");
assert!(!stdout.contains("\n three\n"), "got: {stdout}");
assert!(stdout.contains("\n-two\n"), "got: {stdout}");
assert!(stdout.contains("\n+TWO\n"), "got: {stdout}");
}
#[test]
fn brief_reports_one_line_per_differing_pair() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("a.txt"), "x\n").unwrap();
fs::write(root.join("b.txt"), "y\n").unwrap();
let (code, stdout, stderr) = run_in(root, vec!["-q", "a.txt", "b.txt"]);
assert_eq!(code, 1);
assert_eq!(stdout, "Files a.txt and b.txt differ\n");
assert_eq!(stderr, "");
}
#[test]
fn compat_flags_are_accepted_and_ignored() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("a.txt"), "x\n").unwrap();
fs::write(root.join("b.txt"), "y\n").unwrap();
let (code, stdout, stderr) =
run_in(root, vec!["-u", "-r", "--color=always", "a.txt", "b.txt"]);
assert_eq!(code, 1);
assert_eq!(stderr, "");
// Plain unified output, no ANSI escapes.
assert!(stdout.starts_with("--- a.txt\n+++ b.txt\n"), "got: {stdout}");
assert!(!stdout.contains('\u{1b}'), "got: {stdout}");
}
#[test]
fn binary_inputs_report_binary_difference() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("a.bin"), b"aa\x00bb").unwrap();
fs::write(root.join("b.bin"), b"aa\x00cc").unwrap();
let (code, stdout, stderr) = run_in(root, vec!["a.bin", "b.bin"]);
assert_eq!(code, 1);
assert_eq!(stdout, "Binary files a.bin and b.bin differ\n");
assert_eq!(stderr, "");
}
#[test]
fn missing_operand_file_is_trouble() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("a.txt"), "x\n").unwrap();
let (code, stdout, stderr) = run_in(root, vec!["a.txt", "nope.txt"]);
assert_eq!(code, 2);
assert_eq!(stdout, "");
assert_eq!(stderr, "diff: nope.txt: No such file or directory\n");
}
#[test]
fn missing_second_operand_is_usage_error() {
let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["only-one"]);
assert_eq!(code, 2);
assert_eq!(stdout, "");
assert!(stderr.contains("required"), "got: {stderr}");
}
#[test]
fn new_file_treats_missing_operand_as_empty() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("a.txt"), "one\n").unwrap();
let (code, stdout, stderr) = run_in(root, vec!["-N", "nope.txt", "a.txt"]);
assert_eq!(code, 1);
assert_eq!(stderr, "");
assert!(stdout.starts_with("--- nope.txt\n+++ a.txt\n"), "got: {stdout}");
assert!(stdout.contains("\n+one\n"), "got: {stdout}");
}
#[test]
fn dash_reads_context_stdin() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("a.txt"), "one\ntwo\n").unwrap();
let (code, stdout, stderr) = run_with(root.clone(), b"one\ntwo\n", vec!["a.txt", "-"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", ""));
let (code, stdout, _) = run_with(root, b"one\nTWO\n", vec!["a.txt", "-"]);
assert_eq!(code, 1);
assert!(stdout.starts_with("--- a.txt\n+++ -\n"), "got: {stdout}");
}
#[test]
fn directories_diff_recursively_with_only_in_lines() {
let (_dir, root) = canonical_tempdir();
let (a, b) = (root.join("a"), root.join("b"));
fs::create_dir_all(a.join("sub")).unwrap();
fs::create_dir_all(b.join("sub")).unwrap();
fs::write(a.join("common.txt"), "same\n").unwrap();
fs::write(b.join("common.txt"), "same\n").unwrap();
fs::write(a.join("only.txt"), "left\n").unwrap();
fs::write(b.join("other.txt"), "right\n").unwrap();
fs::write(a.join("sub/inner.txt"), "old\n").unwrap();
fs::write(b.join("sub/inner.txt"), "new\n").unwrap();
let (code, stdout, stderr) = run_in(root, vec!["a", "b"]);
assert_eq!(code, 1);
assert_eq!(stderr, "");
assert!(stdout.contains("Only in a: only.txt\n"), "got: {stdout}");
assert!(stdout.contains("Only in b: other.txt\n"), "got: {stdout}");
assert!(
stdout.contains(
"diff -r a/sub/inner.txt b/sub/inner.txt\n--- a/sub/inner.txt\n+++ b/sub/inner.txt\n"
),
"got: {stdout}"
);
assert!(stdout.contains("\n-old\n"), "got: {stdout}");
assert!(stdout.contains("\n+new\n"), "got: {stdout}");
// Identical common.txt must not appear at all.
assert!(!stdout.contains("common.txt"), "got: {stdout}");
}
#[test]
fn identical_directories_exit_zero() {
let (_dir, root) = canonical_tempdir();
let (a, b) = (root.join("a"), root.join("b"));
fs::create_dir_all(&a).unwrap();
fs::create_dir_all(&b).unwrap();
fs::write(a.join("f.txt"), "same\n").unwrap();
fs::write(b.join("f.txt"), "same\n").unwrap();
let (code, stdout, stderr) = run_in(root, vec!["-r", "a", "b"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", ""));
}
#[test]
fn help_renders_to_scope_stdout() {
let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["--help"]);
assert_eq!(code, 0);
assert!(stdout.contains("Usage:"));
assert!(stdout.contains("Compare files line by line"));
assert_eq!(stderr, "");
}
}
+36
View File
@@ -0,0 +1,36 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/date), patched to resolve
# path arguments against the shell working directory and route I/O through
# pi-uutils-ctx so it can run in-process as a shell builtin. The set-date
# capability (and its libc/windows-sys settime dependencies) is removed
# entirely, along with the fluent/icu localization stack. See src/date.rs for
# the patch markers (`pi-uutils:` comments).
[package]
name = "uu_date"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "date ~ (uutils) display the current time (vendored + patched for in-process embedding)"
[lib]
path = "src/date.rs"
[dependencies]
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
jiff = { version = "0.2", features = [
"tzdb-bundle-platform",
"tzdb-zoneinfo",
"tzdb-concatenated",
] }
parse_datetime = "0.14"
regex = "1.11"
uucore = { version = "0.8.0", features = ["parser"] }
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
# pi-uutils: rustix is kept only for `clock_getres` (--resolution); the
# `clock_settime` user is gone with the set-date capability.
[target.'cfg(unix)'.dependencies]
rustix = { version = "1.1", features = ["time"] }
[dev-dependencies]
parking_lot = "0.12"
tempfile = "3"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+1336
View File
File diff suppressed because it is too large Load Diff
+760
View File
@@ -0,0 +1,760 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore strtime
// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process
// as a shell builtin: the `translate!` error string is literalized with the
// en-US locale text and the regex cache uses `LazyLock` instead of `OnceLock`.
// No behavior changes.
//! GNU date format modifier support
//!
//! This module implements GNU-compatible format modifiers for date formatting.
//! These modifiers extend standard strftime format specifiers with optional
//! width and flag modifiers.
//!
//! ## Syntax
//!
//! Format: `%[flags][width]specifier`
//!
//! ### Flags
//! - `-`: Do not pad the field
//! - `_`: Pad with spaces instead of zeros
//! - `0`: Pad with zeros (default for numeric fields)
//! - `^`: Convert to uppercase
//! - `#`: Use opposite case (uppercase becomes lowercase and vice versa)
//! - `+`: Force display of sign (+ for positive, - for negative)
//!
//! ### Width
//! - One or more digits specifying minimum field width
//! - Field will be padded to this width using the padding character
//!
//! ### Examples
//! - `%10Y`: Year padded to 10 digits with zeros (0000001999)
//! - `%_10m`: Month padded to 10 digits with spaces ( 06)
//! - `%-d`: Day without padding (1 instead of 01)
//! - `%^B`: Month name in uppercase (JUNE)
//! - `%+4C`: Century with sign, padded to 4 characters (+019)
use std::{fmt, sync::LazyLock};
use jiff::{
Zoned,
fmt::strtime::{BrokenDownTime, Config, PosixCustom},
};
use regex::Regex;
/// Error type for format modifier operations
#[derive(Debug)]
pub enum FormatError {
/// Error from the underlying jiff library
JiffError(jiff::Error),
/// Field width calculation overflowed or required allocation failed
FieldWidthTooLarge { width: usize, specifier: String },
}
impl fmt::Display for FormatError {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
match self {
Self::JiffError(e) => write!(f, "{e}"),
// pi-uutils: literalized en-US translation of
// `date-error-format-modifier-width-too-large`.
Self::FieldWidthTooLarge { width, specifier } => {
write!(f, "format modifier width '{width}' is too large for specifier '%{specifier}'")
},
}
}
}
impl From<jiff::Error> for FormatError {
fn from(e: jiff::Error) -> Self {
Self::JiffError(e)
}
}
/// Regex to match format specifiers with optional modifiers
/// Pattern: % \[flags\] \[width\] specifier
/// Flags: -, _, 0, ^, #, +
/// Width: one or more digits
/// Specifier: any letter or special sequence like :z, ::z, :::z
// pi-uutils: `LazyLock` instead of upstream's function-local `OnceLock`.
static FORMAT_SPEC_REGEX: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"%([_0^#+-]*)(\d*)(:*[a-zA-Z])").unwrap());
/// Check if format string contains any GNU modifiers and format if present.
///
/// This function combines modifier detection and formatting in a single pass
/// for better performance. If no modifiers are found, returns None and the
/// caller should use standard formatting. If modifiers are found, returns
/// the formatted string.
pub fn format_with_modifiers_if_present(
date: &Zoned,
format_string: &str,
config: &Config<PosixCustom>,
) -> Option<Result<String, FormatError>> {
let re = &*FORMAT_SPEC_REGEX;
// Quick check: does the string contain any modifiers?
let has_modifiers = re.captures_iter(format_string).any(|cap| {
let flags = cap.get(1).map_or("", |m| m.as_str());
let width_str = cap.get(2).map_or("", |m| m.as_str());
!flags.is_empty() || !width_str.is_empty()
});
if !has_modifiers {
return None;
}
// If we have modifiers, format the string
Some(format_with_modifiers(date, format_string, config))
}
/// Process a format string with GNU modifiers.
///
/// # Arguments
/// * `date` - The date to format
/// * `format_string` - Format string with GNU modifiers
/// * `config` - Strftime configuration
///
/// # Returns
/// Formatted string with modifiers applied
///
/// # Errors
/// Returns `FormatError` if formatting fails
fn format_with_modifiers(
date: &Zoned,
format_string: &str,
config: &Config<PosixCustom>,
) -> Result<String, FormatError> {
// First, replace %% with a placeholder to avoid matching it
let placeholder = "\x00PERCENT\x00";
let temp_format = format_string.replace("%%", placeholder);
let re = &*FORMAT_SPEC_REGEX;
let mut result = String::new();
let mut last_end = 0;
let broken_down = BrokenDownTime::from(date);
for cap in re.captures_iter(&temp_format) {
let whole_match = cap.get(0).unwrap();
let flags = cap.get(1).map_or("", |m| m.as_str());
let width_str = cap.get(2).map_or("", |m| m.as_str());
let spec = cap.get(3).unwrap().as_str();
// Add text before this match
result.push_str(&temp_format[last_end..whole_match.start()]);
// Format the base specifier first
let base_format = format!("%{spec}");
let formatted = broken_down.to_string_with_config(config, &base_format)?;
// Check if this specifier has modifiers
if !flags.is_empty() || !width_str.is_empty() {
// Apply modifiers to the formatted value
let width: usize = width_str.parse().unwrap_or(0);
let explicit_width = !width_str.is_empty();
let modified = apply_modifiers(&formatted, flags, width, spec, explicit_width)?;
result.push_str(&modified);
} else {
// No modifiers, use formatted value as-is
result.push_str(&formatted);
}
last_end = whole_match.end();
}
// Add remaining text
result.push_str(&temp_format[last_end..]);
// Restore %% by converting placeholder to %
let result = result.replace(placeholder, "%");
Ok(result)
}
/// Returns true if the specifier produces text output (default pad is space)
/// rather than numeric output (default pad is zero).
fn is_text_specifier(specifier: &str) -> bool {
matches!(specifier.chars().last(), Some('A' | 'a' | 'B' | 'b' | 'h' | 'Z' | 'p' | 'P'))
}
/// Returns true if the specifier defaults to space padding.
/// This includes text specifiers and numeric specifiers like %e and %k
/// that use blank-padding by default in GNU date.
fn is_space_padded_specifier(specifier: &str) -> bool {
matches!(
specifier.chars().last(),
Some('A' | 'a' | 'B' | 'b' | 'h' | 'Z' | 'p' | 'P' | 'e' | 'k' | 'l')
)
}
/// Returns the default width for a specifier.
/// This is used when a flag like `_` is used without an explicit width.
fn get_default_width(specifier: &str) -> usize {
match specifier.chars().last() {
// Day of month: 2 digits (01-31)
Some('d') | Some('e') => 2,
// Month: 2 digits (01-12)
Some('m') => 2,
// Hour: 2 digits (00-23)
Some('H') | Some('k') => 2,
// Hour (12-hour): 2 digits (01-12)
Some('I') | Some('l') => 2,
// Minute: 2 digits (00-59)
Some('M') => 2,
// Second: 2 digits (00-60)
Some('S') => 2,
// Year (2-digit): 2 digits
Some('y') => 2,
// Day of year: 3 digits (001-366)
Some('j') => 3,
// Week number: 2 digits (00-53)
Some('U') | Some('W') | Some('V') => 2,
// Day of week: 1 digit (0-6 or 1-7)
Some('w') | Some('u') => 1,
// Century: 2 digits (00-99)
Some('C') => 2,
// Full year: 4 digits
Some('Y') | Some('G') => 4,
// ISO week year (2-digit): 2 digits
Some('g') => 2,
// Epoch seconds: typically 10 digits (but variable)
Some('s') => 0,
// Nanoseconds: 9 digits
Some('N') => 9,
// Quarter: 1 digit
Some('q') => 1,
// Timezone offset: varies
Some('z') => 0,
// Text specifiers have no default width
_ => 0,
}
}
/// Strip default padding (leading zeros or leading spaces) from a value,
/// preserving at least one character.
fn strip_default_padding(value: &str) -> String {
if value.starts_with('0') && value.len() >= 2 {
let stripped = value.trim_start_matches('0');
if stripped.is_empty() {
return "0".to_string();
}
if let Some(first_char) = stripped.chars().next()
&& first_char.is_ascii_digit()
{
return stripped.to_string();
}
}
if value.starts_with(' ') {
let stripped = value.trim_start();
if !stripped.is_empty() {
return stripped.to_string();
}
}
value.to_string()
}
/// Apply width and flag modifiers to a formatted value.
///
/// The `specifier` parameter is the format specifier (e.g., "d", "B", "Y")
/// which determines the default padding character (space for text, zero for
/// numeric). Flags are processed in order so that when conflicting flags
/// appear, the last one takes precedence (e.g., `_+` means `+` wins for
/// padding).
///
/// The `explicit_width` parameter indicates whether a width was explicitly
/// specified in the format string (true) or if width is 0 (false).
fn apply_modifiers(
value: &str,
flags: &str,
width: usize,
specifier: &str,
explicit_width: bool,
) -> Result<String, FormatError> {
let mut result = value.to_string();
// Determine default pad character based on specifier type
// Determine default pad character based on specifier type.
// Text specifiers (month names, etc.) and numeric specifiers like %e, %k, %l
// default to space padding; other numeric specifiers default to zero padding.
let default_pad = if is_space_padded_specifier(specifier) {
' '
} else {
'0'
};
// Process flags in order - last conflicting flag wins
let mut pad_char = default_pad;
let mut no_pad = false;
let mut uppercase = false;
let mut swap_case = false;
let mut force_sign = false;
let mut underscore_flag = false;
for flag in flags.chars() {
match flag {
'-' => {
no_pad = true;
},
'_' => {
no_pad = false;
pad_char = ' ';
underscore_flag = true;
},
'0' => {
no_pad = false;
pad_char = '0';
},
'^' => {
uppercase = true;
swap_case = false; // ^ overrides #
},
'#' if !uppercase => {
// Only apply # if ^ hasn't been set
swap_case = true;
},
'+' => {
force_sign = true;
no_pad = false;
pad_char = '0';
},
_ => {},
}
}
// Apply case modifications (uppercase takes precedence over swap_case)
if uppercase {
result = result.to_uppercase();
} else if swap_case {
if result
.chars()
.all(|c| !c.is_alphabetic() || c.is_uppercase())
{
result = result.to_lowercase();
} else if !result
.chars()
.all(|c| !c.is_alphabetic() || c.is_lowercase())
{
result = result.to_uppercase();
}
}
// If no_pad flag is active, suppress all padding and return
if no_pad {
return Ok(strip_default_padding(&result));
}
// Handle padding flag without explicit width: use default width
// This applies when _ or 0 flag overrides the default padding character
// and no explicit width is specified (e.g., %_m, %0e)
let effective_width = if !explicit_width && (underscore_flag || pad_char != default_pad) {
get_default_width(specifier)
} else {
width
};
// When the requested width is narrower than the default formatted width, GNU
// first removes default padding and then reapplies the requested width.
if effective_width > 0 && effective_width < result.len() {
result = strip_default_padding(&result);
}
// Strip default padding when switching pad characters on numeric fields
if !is_text_specifier(specifier) && result.len() >= 2 {
if pad_char == ' ' && result.starts_with('0') {
// Switching to space padding: strip leading zeros
result = strip_default_padding(&result);
} else if pad_char == '0' && result.starts_with(' ') {
// Switching to zero padding: strip leading spaces
result = strip_default_padding(&result);
}
}
// Apply force sign for numeric values
// GNU behavior: + only adds sign if:
// 1. An explicit width is provided, OR
// 2. The value exceeds the default width for that specifier (e.g., year > 4
// digits)
if force_sign
&& !result.starts_with('+')
&& !result.starts_with('-')
&& result.chars().next().is_some_and(|c| c.is_ascii_digit())
{
let default_w = get_default_width(specifier);
// Add sign only if explicit width provided OR result exceeds default width
if explicit_width || (default_w > 0 && result.len() > default_w) {
result.insert(0, '+');
}
}
// Apply width padding
if effective_width > result.len() {
let padding = effective_width - result.len();
let has_sign = result.starts_with('+') || result.starts_with('-');
if pad_char == '0' && has_sign {
// Zero padding: sign first, then zeros (e.g., "-0022")
let sign = result.chars().next().unwrap();
let rest = &result[1..];
let mut padded = try_alloc_padded(result.len(), padding, effective_width, specifier)?;
padded.push(sign);
padded.extend(std::iter::repeat_n('0', padding));
padded.push_str(rest);
result = padded;
} else {
// Default: pad on the left (e.g., " -22" or " 1999")
let mut padded = try_alloc_padded(result.len(), padding, effective_width, specifier)?;
padded.extend(std::iter::repeat_n(pad_char, padding));
padded.push_str(&result);
result = padded;
}
}
Ok(result)
}
/// Allocate a `String` with enough capacity for `current_len + padding`,
/// returning `FieldWidthTooLarge` on arithmetic overflow or allocation failure.
fn try_alloc_padded(
current_len: usize,
padding: usize,
width: usize,
specifier: &str,
) -> Result<String, FormatError> {
let target_len = current_len
.checked_add(padding)
.ok_or_else(|| FormatError::FieldWidthTooLarge { width, specifier: specifier.to_string() })?;
let mut s = String::new();
s.try_reserve(target_len)
.map_err(|_| FormatError::FieldWidthTooLarge { width, specifier: specifier.to_string() })?;
Ok(s)
}
#[cfg(test)]
mod tests {
use jiff::{civil, tz::TimeZone};
use super::*;
fn make_test_date(year: i16, month: i8, day: i8, hour: i8) -> Zoned {
civil::date(year, month, day)
.at(hour, 0, 0, 0)
.to_zoned(TimeZone::UTC)
.unwrap()
}
fn get_config() -> Config<PosixCustom> {
Config::new().custom(PosixCustom::new()).lenient(true)
}
#[test]
fn test_width_and_padding_modifiers() {
let date = make_test_date(1999, 6, 1, 0);
let config = get_config();
// Test basic width with zero padding
let result = format_with_modifiers(&date, "%10Y", &config).unwrap();
assert_eq!(result, "0000001999");
// Test large width
let result = format_with_modifiers(&date, "%20Y", &config).unwrap();
assert_eq!(result, "00000000000000001999");
assert_eq!(result.len(), 20);
// Test underscore (space) padding with month
let result = format_with_modifiers(&date, "%_10m", &config).unwrap();
assert_eq!(result, " 6");
assert_eq!(result.len(), 10);
// Test underscore padding with day
let date_day5 = make_test_date(1999, 6, 5, 0);
let result = format_with_modifiers(&date_day5, "%_10d", &config).unwrap();
assert_eq!(result, " 5");
}
#[test]
fn test_no_pad_and_case_flags() {
let date = make_test_date(1999, 6, 1, 0);
let config = get_config();
// Test no-pad: %-10Y suppresses all padding (width ignored)
let result = format_with_modifiers(&date, "%-10Y", &config).unwrap();
assert_eq!(result, "1999");
// Test no-pad: %-d strips default zero padding
let result = format_with_modifiers(&date, "%-d", &config).unwrap();
assert_eq!(result, "1");
// Test uppercase: %^B should uppercase month name
let result = format_with_modifiers(&date, "%^B", &config).unwrap();
assert_eq!(result, "JUNE");
// Test uppercase with width: %^10B should uppercase and space-pad (text
// specifier)
let result = format_with_modifiers(&date, "%^10B", &config).unwrap();
assert_eq!(result, " JUNE");
assert_eq!(result.len(), 10);
}
#[test]
fn test_sign_flags() {
let date = make_test_date(1970, 1, 1, 0);
let config = get_config();
// Test force sign with century: %+4C
let result = format_with_modifiers(&date, "%+4C", &config).unwrap();
assert!(result.starts_with('+'));
assert_eq!(result.len(), 4);
// Test force sign with zero padding: %+6Y
let result = format_with_modifiers(&date, "%+6Y", &config).unwrap();
assert_eq!(result, "+01970");
}
#[test]
fn test_combined_flags_underscore_and_sign() {
let date = make_test_date(1970, 1, 1, 0);
let config = get_config();
// %_+6Y: _ sets space pad, then + overrides to zero pad with sign (last wins)
let result = format_with_modifiers(&date, "%_+6Y", &config).unwrap();
assert_eq!(result, "+01970");
}
#[test]
fn test_combined_flags_no_pad_and_uppercase() {
let date = make_test_date(1999, 6, 1, 0);
let config = get_config();
// %-^10B: uppercase + no-pad (- suppresses all padding, width ignored)
let result = format_with_modifiers(&date, "%-^10B", &config).unwrap();
assert_eq!(result, "JUNE");
}
#[test]
fn test_swap_case_flag() {
let date = make_test_date(1999, 6, 1, 0);
let config = get_config();
// %#B: swap case on "June" (mixed case) → uppercase
let result = format_with_modifiers(&date, "%#B", &config).unwrap();
assert_eq!(result, "JUNE");
}
#[test]
fn test_width_smaller_than_result() {
let date = make_test_date(1999, 6, 1, 0);
let config = get_config();
// %1d: width 1 < "01".len() → strip zero padding → "1"
let result = format_with_modifiers(&date, "%1d", &config).unwrap();
assert_eq!(result, "1");
}
#[test]
fn test_edge_cases_and_special_formats() {
let date = make_test_date(1999, 6, 1, 0);
let config = get_config();
// Test width zero (no effect)
let result = format_with_modifiers(&date, "%Y", &config).unwrap();
assert_eq!(result, "1999");
// Test no modifiers (standard format)
let result = format_with_modifiers(&date, "%Y-%m-%d", &config).unwrap();
assert_eq!(result, "1999-06-01");
// Test %% escape sequence
let result = format_with_modifiers(&date, "%%Y=%Y", &config).unwrap();
assert_eq!(result, "%Y=1999");
// Test multiple modifiers in one format string
// %-5d: no-pad suppresses all padding → "1" (width ignored)
let result = format_with_modifiers(&date, "%10Y-%_5m-%-5d", &config).unwrap();
assert_eq!(result, "0000001999- 6-1");
}
#[test]
fn test_modifier_detection() {
let date = make_test_date(1999, 6, 1, 0);
let config = get_config();
// Should detect modifiers
let result = format_with_modifiers_if_present(&date, "%10Y", &config);
assert!(result.is_some());
// Should not detect modifiers
let result = format_with_modifiers_if_present(&date, "%Y-%m-%d", &config);
assert!(result.is_none());
// Should detect flag without width
let result = format_with_modifiers_if_present(&date, "%^B", &config);
assert!(result.is_some());
}
#[test]
fn test_negative_values_with_space_padding() {
// Test case from GNU test: neg-secs2
// Format: %_5s with value -22 should produce " -22" (space-padded)
use jiff::Timestamp;
let ts = Timestamp::from_second(-22).unwrap();
let date = ts.to_zoned(TimeZone::UTC);
let config = get_config();
let result = format_with_modifiers(&date, "%_5s", &config).unwrap();
assert_eq!(result, " -22", "Space padding should pad before the sign for negative numbers");
}
// Unit tests for apply_modifiers function
#[test]
fn test_apply_modifiers_basic() {
// No modifiers (numeric specifier)
assert_eq!(apply_modifiers("1999", "", 0, "Y", false).unwrap(), "1999");
// Zero padding
assert_eq!(apply_modifiers("1999", "0", 10, "Y", true).unwrap(), "0000001999");
// Space padding (strips leading zeros)
assert_eq!(apply_modifiers("06", "_", 5, "m", true).unwrap(), " 6");
// No-pad (strips leading zeros, width ignored)
assert_eq!(apply_modifiers("01", "-", 5, "d", true).unwrap(), "1");
// Uppercase
assert_eq!(apply_modifiers("june", "^", 0, "B", false).unwrap(), "JUNE");
// Swap case: all uppercase → lowercase
assert_eq!(apply_modifiers("UTC", "#", 0, "Z", false).unwrap(), "utc");
// Swap case: mixed case → uppercase
assert_eq!(apply_modifiers("June", "#", 0, "B", false).unwrap(), "JUNE");
}
#[test]
fn test_apply_modifiers_signs() {
// Force sign with explicit width
assert_eq!(apply_modifiers("1970", "+", 6, "Y", true).unwrap(), "+01970");
// Force sign without explicit width: should NOT add sign for 4-digit year
assert_eq!(apply_modifiers("1999", "+", 0, "Y", false).unwrap(), "1999");
// Force sign without explicit width: SHOULD add sign for year > 4 digits
assert_eq!(apply_modifiers("12345", "+", 0, "Y", false).unwrap(), "+12345");
// Negative with zero padding: sign first, then zeros
assert_eq!(apply_modifiers("-22", "0", 5, "s", true).unwrap(), "-0022");
// Negative with space padding: spaces first, then sign
assert_eq!(apply_modifiers("-22", "_", 5, "s", true).unwrap(), " -22");
// Force sign (_+): + is last, overrides _ → zero pad with sign
assert_eq!(apply_modifiers("5", "_+", 5, "s", true).unwrap(), "+0005");
// No-pad + uppercase: no padding applied
assert_eq!(apply_modifiers("june", "-^", 10, "B", true).unwrap(), "JUNE");
}
#[test]
fn test_case_flag_precedence() {
// Test that ^ (uppercase) overrides # (swap case)
assert_eq!(apply_modifiers("June", "^#", 0, "B", false).unwrap(), "JUNE");
assert_eq!(apply_modifiers("June", "#^", 0, "B", false).unwrap(), "JUNE");
// Test # alone (swap case)
assert_eq!(apply_modifiers("June", "#", 0, "B", false).unwrap(), "JUNE");
assert_eq!(apply_modifiers("JUNE", "#", 0, "B", false).unwrap(), "june");
}
#[test]
fn test_apply_modifiers_text_specifiers() {
// Text specifiers default to space padding
assert_eq!(apply_modifiers("June", "", 10, "B", true).unwrap(), " June");
assert_eq!(apply_modifiers("Mon", "", 10, "a", true).unwrap(), " Mon");
// Numeric specifiers default to zero padding
assert_eq!(apply_modifiers("6", "", 10, "m", true).unwrap(), "0000000006");
}
#[test]
fn test_apply_modifiers_width_smaller_than_result() {
// Width smaller than result strips default padding
assert_eq!(apply_modifiers("01", "", 1, "d", true).unwrap(), "1");
assert_eq!(apply_modifiers("06", "", 1, "m", true).unwrap(), "6");
}
#[test]
fn test_apply_modifiers_parametrized() {
let test_cases = vec![
("1", "0", 3, "Y", true, "001"),
("1", "_", 3, "d", true, " 1"),
("1", "-", 3, "d", true, "1"), // no-pad: width ignored
("abc", "^", 5, "B", true, " ABC"), // text specifier: space pad
("5", "+", 4, "s", true, "+005"),
("5", "_+", 4, "s", true, "+005"), // + is last: zero pad with sign
("-3", "0", 5, "s", true, "-0003"),
("05", "_", 3, "d", true, " 5"),
("09", "-", 4, "d", true, "9"), // no-pad: width ignored
("1970", "_+", 6, "Y", true, "+01970"), // + is last: zero pad with sign
];
for (value, flags, width, spec, explicit_width, expected) in test_cases {
assert_eq!(
apply_modifiers(value, flags, width, spec, explicit_width).unwrap(),
expected,
"value='{value}', flags='{flags}', width={width}, spec='{spec}', \
explicit_width={explicit_width}",
);
}
}
#[test]
fn test_apply_modifiers_width_too_large() {
let err = apply_modifiers("x", "", usize::MAX, "c", true).unwrap_err();
assert!(matches!(
err,
FormatError::FieldWidthTooLarge { width, specifier }
if width == usize::MAX && specifier == "c"
));
}
#[test]
fn test_underscore_flag_without_width() {
// %_m should pad month to default width 2 with spaces
assert_eq!(apply_modifiers("6", "_", 0, "m", false).unwrap(), " 6");
// %_d should pad day to default width 2 with spaces
assert_eq!(apply_modifiers("1", "_", 0, "d", false).unwrap(), " 1");
// %_H should pad hour to default width 2 with spaces
assert_eq!(apply_modifiers("5", "_", 0, "H", false).unwrap(), " 5");
// %_Y should pad year to default width 4 with spaces
assert_eq!(apply_modifiers("1999", "_", 0, "Y", false).unwrap(), "1999");
// already at default width
}
#[test]
fn test_plus_flag_without_width() {
// %+Y without width should NOT add sign for 4-digit year
assert_eq!(apply_modifiers("1999", "+", 0, "Y", false).unwrap(), "1999");
// %+Y without width SHOULD add sign for year > 4 digits
assert_eq!(apply_modifiers("12345", "+", 0, "Y", false).unwrap(), "+12345");
// %+Y with explicit width should add sign
assert_eq!(apply_modifiers("1999", "+", 6, "Y", true).unwrap(), "+01999");
}
#[test]
fn test_zero_flag_on_space_padded_specifiers() {
// GNU date: %0e should override space-padding with zero-padding
// Verified: `date -d "2024-06-05" "+%0e"` → "05"
let date = make_test_date(1999, 6, 5, 5);
let config = get_config();
// %0e: day-of-month (normally space-padded) with 0 flag → zero-padded
let result = format_with_modifiers(&date, "%0e", &config).unwrap();
assert_eq!(result, "05", "GNU: %0e should produce '05', not ' 5'");
// %0k: hour (normally space-padded) with 0 flag → zero-padded
let result = format_with_modifiers(&date, "%0k", &config).unwrap();
assert_eq!(result, "05", "GNU: %0k should produce '05', not ' 5'");
}
#[test]
fn test_underscore_century_default_width() {
// GNU date: %C default width is 2, not 4
// Verified: `date -d "2024-06-15" "+%_C"` → "20" (no extra padding)
let date = make_test_date(1999, 6, 1, 0);
let config = get_config();
// %_C: century with underscore flag, no explicit width
// Default width for %C should be 2 (century is 00-99)
let result = format_with_modifiers(&date, "%_C", &config).unwrap();
assert_eq!(
result, "19",
"GNU: %_C should produce '19', not ' 19' (default width is 2, not 4)"
);
}
}
+31
View File
@@ -0,0 +1,31 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/hostname), patched to route
# output through pi-uutils-ctx and to reject the set-hostname path so it can run
# in-process as a shell builtin. See src/hostname.rs for the patch markers
# (`pi-uutils:` comments).
[package]
name = "uu_hostname"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "hostname ~ (uutils) display the host name of the current host (vendored + patched for in-process embedding)"
[lib]
path = "src/hostname.rs"
[dependencies]
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
hostname = "0.4"
uucore = { version = "0.8.0", features = ["wide"] }
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[target.'cfg(any(target_os = "freebsd", target_os = "openbsd"))'.dependencies]
dns-lookup = "3.0.0"
[target.'cfg(target_os = "windows")'.dependencies]
windows-sys = { version = "0.61.0", features = [
"Win32_Networking_WinSock",
"Win32_Foundation",
] }
[dev-dependencies]
parking_lot = "0.12"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+308
View File
@@ -0,0 +1,308 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore hashset Addrs addrs
// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process
// as a shell builtin. The set-hostname path is removed entirely (a NAME operand
// is rejected with an "unsupported" error instead of calling `hostname::set`,
// so the `hostname` crate's "set" feature is dropped), all output is routed
// through the context streams, `translate!` strings are literalized, and the
// entry point no longer calls `std::process::exit`.
#[cfg(not(any(target_os = "freebsd", target_os = "openbsd")))]
use std::net::ToSocketAddrs;
use std::{collections::hash_set::HashSet, ffi::OsString, io::Write, str};
use clap::{Arg, ArgAction, ArgMatches, Command, builder::ValueParser};
#[cfg(any(target_os = "freebsd", target_os = "openbsd"))]
use dns_lookup::lookup_host;
use pi_uutils_ctx::format_usage;
use uucore::error::{FromIo, UResult, USimpleError};
static OPT_DOMAIN: &str = "domain";
static OPT_IP_ADDRESS: &str = "ip-address";
static OPT_FQDN: &str = "fqdn";
static OPT_SHORT: &str = "short";
static OPT_HOST: &str = "host";
#[cfg(windows)]
mod wsa {
use std::io;
use windows_sys::Win32::Networking::WinSock::{WSACleanup, WSADATA, WSAStartup};
pub(super) struct WsaHandle(());
pub(super) fn start() -> io::Result<WsaHandle> {
let mut data = std::mem::MaybeUninit::<WSADATA>::uninit();
let err = unsafe { WSAStartup(0x0202, data.as_mut_ptr()) };
if err == 0 {
Ok(WsaHandle(()))
} else {
Err(io::Error::from_raw_os_error(err))
}
}
impl Drop for WsaHandle {
fn drop(&mut self) {
// This possibly returns an error but we can't handle it
let _ = unsafe { WSACleanup() };
}
}
}
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
/// arguments directly (without the uucore clap-localization helper that would
/// terminate the process), renders clap help/usage/version to the context
/// streams, and maps the `UResult` to an exit code, so it is safe to run inside
/// the host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
match hostname_main(&matches) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
let msg = err.to_string();
if !msg.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "hostname: {msg}");
}
if code == 0 { 1 } else { code }
},
}
}
fn hostname_main(matches: &ArgMatches) -> UResult<()> {
#[cfg(windows)]
let _handle = wsa::start().map_err_context(|| "failed to start Winsock".to_string())?;
match matches.get_one::<OsString>(OPT_HOST) {
None => display_hostname(matches),
// pi-uutils: setting the process-global hostname from inside a shell
// builtin is refused (upstream calls `hostname::set` here).
Some(_host) => Err(USimpleError::new(
1,
"setting the hostname is not supported by this builtin".to_string(),
)),
}
}
pub fn uu_app() -> Command {
Command::new("hostname")
.version(uucore::crate_version!())
.about("Display or set the system's host name.")
.override_usage(format_usage("hostname [OPTION]... [HOSTNAME]"))
.infer_long_args(true)
.arg(
Arg::new(OPT_DOMAIN)
.short('d')
.long("domain")
.overrides_with_all([OPT_DOMAIN, OPT_IP_ADDRESS, OPT_FQDN, OPT_SHORT])
.help("Display the name of the DNS domain if possible")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_IP_ADDRESS)
.short('i')
.long("ip-address")
.overrides_with_all([OPT_DOMAIN, OPT_IP_ADDRESS, OPT_FQDN, OPT_SHORT])
.help("Display the network address(es) of the host")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_FQDN)
.short('f')
.long("fqdn")
.overrides_with_all([OPT_DOMAIN, OPT_IP_ADDRESS, OPT_FQDN, OPT_SHORT])
.help("Display the FQDN (Fully Qualified Domain Name) (default)")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_SHORT)
.short('s')
.long("short")
.overrides_with_all([OPT_DOMAIN, OPT_IP_ADDRESS, OPT_FQDN, OPT_SHORT])
.help("Display the short hostname (the portion before the first dot) if possible")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_HOST)
.value_parser(ValueParser::os_string())
.value_hint(clap::ValueHint::Hostname),
)
}
fn display_hostname(matches: &ArgMatches) -> UResult<()> {
let hostname = hostname::get()
.map_err_context(|| "failed to get hostname".to_owned())?
.to_string_lossy()
.into_owned();
// pi-uutils: all output below goes to the context stdout instead of the
// process stdout.
let mut out = pi_uutils_ctx::stdout();
if matches.get_flag(OPT_IP_ADDRESS) {
let addresses;
#[cfg(not(any(target_os = "freebsd", target_os = "openbsd")))]
{
let hostname = hostname + ":1";
let addrs = hostname
.to_socket_addrs()
.map_err_context(|| "failed to resolve socket addresses".to_owned())?;
addresses = addrs;
}
// DNS reverse lookup via "hostname:1" does not work on FreeBSD and OpenBSD
// use dns-lookup crate instead
#[cfg(any(target_os = "freebsd", target_os = "openbsd"))]
{
let addrs: Vec<std::net::IpAddr> = lookup_host(hostname.as_str())
.map_err_context(|| "failed to lookup hostname".to_owned())?
.collect();
addresses = addrs;
}
let mut hashset = HashSet::new();
let mut output = String::new();
for addr in addresses {
// XXX: not sure why this is necessary...
if !hashset.contains(&addr) {
let mut ip = addr.to_string();
if ip.ends_with(":1") {
let len = ip.len();
ip.truncate(len - 2);
}
output.push_str(&ip);
output.push(' ');
hashset.insert(addr);
}
}
let len = output.len();
if len > 0 {
writeln!(out, "{}", &output[0..len - 1])?;
}
Ok(())
} else {
if matches.get_flag(OPT_SHORT) || matches.get_flag(OPT_DOMAIN) {
let mut it = hostname.char_indices().filter(|&ci| ci.1 == '.');
if let Some(ci) = it.next() {
if matches.get_flag(OPT_SHORT) {
writeln!(out, "{}", &hostname[0..ci.0])?;
} else {
writeln!(out, "{}", &hostname[ci.0 + 1..])?;
}
} else if matches.get_flag(OPT_SHORT) {
writeln!(out, "{hostname}")?;
}
return Ok(());
}
writeln!(out, "{hostname}")?;
Ok(())
}
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc};
use parking_lot::Mutex;
use pi_uutils_ctx::ScopeIo;
use super::*;
fn run_in(args: Vec<&str>) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
self.buf.lock().write(buf)
}
fn flush(&mut self) -> std::io::Result<()> {
self.buf.lock().flush()
}
}
let io = ScopeIo {
stdin: Box::new(std::io::empty()),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd: PathBuf::from("."),
env: HashMap::new(),
cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)),
};
let argv: Vec<OsString> = std::iter::once("hostname")
.chain(args)
.map(OsString::from)
.collect();
let code = pi_uutils_ctx::scope(io, || run(argv));
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
(code, out_str, err_str)
}
#[test]
fn bare_invocation_prints_hostname() {
let (code, stdout, stderr) = run_in(vec![]);
assert_eq!((code, stderr.as_str()), (0, ""));
assert!(stdout.ends_with('\n'));
assert!(!stdout.trim_end().is_empty());
}
#[test]
fn set_attempt_is_rejected() {
let (code, stdout, stderr) = run_in(vec!["new-name.example.com"]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert_eq!(stderr, "hostname: setting the hostname is not supported by this builtin\n");
}
#[test]
fn short_is_dotless_prefix_of_full_hostname() {
let (code, short, stderr) = run_in(vec!["-s"]);
let (_, full, _) = run_in(vec![]);
assert_eq!((code, stderr.as_str()), (0, ""));
let short = short.trim_end();
assert!(!short.contains('.'), "-s must strip everything after the first dot");
assert!(full.trim_end().starts_with(short));
}
#[test]
fn fqdn_flag_matches_default_display() {
// -f is the default display mode; it must print the same name as the
// bare invocation, not attempt any set path.
let (code, fqdn, stderr) = run_in(vec!["-f"]);
let (_, bare, _) = run_in(vec![]);
assert_eq!((code, stderr.as_str()), (0, ""));
assert_eq!(fqdn, bare);
}
}
+23
View File
@@ -0,0 +1,23 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/ln), patched to resolve path
# arguments against the shell working directory and route I/O + prompts through
# pi-uutils-ctx so it can run in-process as a shell builtin. See src/ln.rs for
# the patch markers (`pi-uutils:` comments).
[package]
name = "uu_ln"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "ln ~ (uutils) create a (file system) link to TARGET (vendored + patched for in-process embedding)"
[lib]
path = "src/ln.rs"
[dependencies]
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
thiserror = "2.0.3"
uucore = { version = "0.8.0", features = ["backup-control", "fs"] }
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[dev-dependencies]
parking_lot = "0.12"
tempfile = "3"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+831
View File
@@ -0,0 +1,831 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore (ToDO) srcpath targetpath EEXIST
// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process
// as a shell builtin. Every filesystem syscall resolves its path operand
// against the shell working directory via `pi_uutils_ctx::resolve` AT THE CALL
// SITE, while the original operands are kept for display/error messages (GNU
// prints operands as typed) — and, crucially, for the CONTENT of symbolic
// links, which stays exactly as typed like GNU ln (only the location where the
// link is created gets resolved). All process-global stdio and the `-i` prompt
// are routed through `pi_uutils_ctx`, `translate!` strings are literalized, and
// the entry point no longer calls `std::process::exit`.
#[cfg(any(unix, target_os = "redox"))]
use std::os::unix::fs::symlink;
#[cfg(windows)]
use std::os::windows::fs::{symlink_dir, symlink_file};
use std::{
borrow::Cow,
collections::HashSet,
ffi::OsString,
fs,
io::Write,
path::{Path, PathBuf},
};
use clap::{Arg, ArgAction, ArgMatches, Command};
use pi_uutils_ctx::format_usage;
use thiserror::Error;
use uucore::{
backup_control::{self, BackupMode},
display::Quotable,
error::{FromIo, UError, UResult, USimpleError, strip_errno},
fs::{
MissingHandling, ResolveMode, canonicalize, make_path_relative_to, paths_refer_to_same_file,
},
};
pub struct Settings {
overwrite: OverwriteMode,
backup: BackupMode,
suffix: OsString,
symbolic: bool,
relative: bool,
logical: bool,
target_dir: Option<PathBuf>,
no_target_dir: bool,
no_dereference: bool,
verbose: bool,
}
#[derive(Clone, Debug, Eq, PartialEq)]
pub enum OverwriteMode {
NoClobber,
Interactive,
Force,
}
// pi-uutils: the `translate!` message templates are literalized with the
// en-US strings from upstream's locales/en-US.ftl.
#[derive(Error, Debug)]
enum LnError {
#[error("target {} is not a directory", _0.quote())]
TargetIsNotADirectory(PathBuf),
#[error("")]
SomeLinksFailed,
#[error("{} and {} are the same file", _0.quote(), _1.quote())]
SameFile(PathBuf, PathBuf),
#[error("missing destination file operand after {}", _0.quote())]
MissingDestination(PathBuf),
#[error("extra operand {}\nTry '{} --help' for more information.", _0.quote(), _1)]
ExtraOperand(OsString, String),
#[error("{}: hard link not allowed for directory", _0.to_string_lossy())]
FailedToCreateHardLinkDir(PathBuf),
}
impl UError for LnError {
fn code(&self) -> i32 {
1
}
}
mod options {
pub const FORCE: &str = "force";
//pub const DIRECTORY: &str = "directory";
pub const INTERACTIVE: &str = "interactive";
pub const NO_DEREFERENCE: &str = "no-dereference";
pub const SYMBOLIC: &str = "symbolic";
pub const LOGICAL: &str = "logical";
pub const PHYSICAL: &str = "physical";
pub const TARGET_DIRECTORY: &str = "target-directory";
pub const NO_TARGET_DIRECTORY: &str = "no-target-directory";
pub const RELATIVE: &str = "relative";
pub const VERBOSE: &str = "verbose";
}
static ARG_FILES: &str = "files";
/// pi-uutils: replacement for uucore's `show_error!` — writes the diagnostic
/// to the context stderr instead of the process-global one. Errors that render
/// to an empty message (e.g. [`LnError::SomeLinksFailed`]) print nothing
/// rather than a dangling "ln: " prefix.
fn show_error(msg: impl std::fmt::Display) {
let rendered = msg.to_string();
if !rendered.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "ln: {rendered}");
}
}
/// pi-uutils: replacement for uucore's `read_yes`, reading from the context
/// stdin one byte at a time (no buffering) so consecutive prompts don't
/// over-read into a later prompt's input. Returns true when the first character
/// of the line is `y`/`Y`.
fn read_yes() -> bool {
use std::io::Read as _;
let mut stdin = pi_uutils_ctx::stdin();
let mut buf = [0u8; 1];
let mut first = None;
loop {
match stdin.read(&mut buf) {
Ok(0) => break, // EOF
Ok(_) => {
if buf[0] == b'\n' {
break;
}
if first.is_none() {
first = Some(buf[0]);
}
},
Err(_) => return false,
}
}
matches!(first, Some(b'y' | b'Y'))
}
/// pi-uutils: replacement for uucore's `prompt_yes!` — writes
/// "ln: \<prompt\> " to the context stderr, then reads the answer from the
/// context stdin.
fn prompt_yes(prompt: impl std::fmt::Display) -> bool {
let mut err = pi_uutils_ctx::stderr();
let _ = write!(err, "ln: {prompt} ");
let _ = err.flush();
read_yes()
}
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
/// arguments directly (without the uucore clap-localization helper that would
/// terminate the process), renders clap help/usage/version to the context
/// streams, and maps the `UResult` to an exit code, so it is safe to run inside
/// the host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
match ln_main(&matches) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
// pi-uutils: `SomeLinksFailed` renders to an empty message
// (upstream prints the per-file diagnostics as it goes); don't
// emit a dangling "ln: " prefix for it.
let msg = err.to_string();
if !msg.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "ln: {msg}");
}
if code == 0 { 1 } else { code }
},
}
}
fn ln_main(matches: &ArgMatches) -> UResult<()> {
/* the list of files */
let paths: Vec<PathBuf> = matches
.get_many::<OsString>(ARG_FILES)
.unwrap()
.map(PathBuf::from)
.collect();
let symbolic = matches.get_flag(options::SYMBOLIC);
let overwrite_mode = if matches.get_flag(options::FORCE) {
OverwriteMode::Force
} else if matches.get_flag(options::INTERACTIVE) {
OverwriteMode::Interactive
} else {
OverwriteMode::NoClobber
};
let backup_mode = backup_control::determine_backup_mode(matches)?;
let backup_suffix = backup_control::determine_backup_suffix(matches);
// When we have "-L" or "-L -P", false otherwise
let logical = matches.get_flag(options::LOGICAL);
let settings = Settings {
overwrite: overwrite_mode,
backup: backup_mode,
suffix: OsString::from(backup_suffix),
symbolic,
logical,
relative: matches.get_flag(options::RELATIVE),
target_dir: matches
.get_one::<OsString>(options::TARGET_DIRECTORY)
.map(PathBuf::from),
no_target_dir: matches.get_flag(options::NO_TARGET_DIRECTORY),
no_dereference: matches.get_flag(options::NO_DEREFERENCE),
verbose: matches.get_flag(options::VERBOSE),
};
exec(&paths[..], &settings)
}
pub fn uu_app() -> Command {
let after_help = format!(
"In the 1st form, create a link to TARGET with the name LINK_NAME.\nIn the 2nd form, create \
a link to TARGET in the current directory.\nIn the 3rd and 4th forms, create links to each \
TARGET in DIRECTORY.\nCreate hard links by default, symbolic links with --symbolic.\nBy \
default, each destination (name of new link) should not already exist.\nWhen creating hard \
links, each TARGET must exist. Symbolic links\ncan hold arbitrary text; if later resolved, \
a relative link is\ninterpreted in relation to its parent directory.\n\n{}",
backup_control::BACKUP_CONTROL_LONG_HELP
);
Command::new("ln")
.version(uucore::crate_version!())
.about("Make links between files.")
.override_usage(format_usage(
"ln [OPTION]... [-T] TARGET LINK_NAME\nln [OPTION]... TARGET\nln [OPTION]... TARGET... \
DIRECTORY\nln [OPTION]... -t DIRECTORY TARGET...",
))
.infer_long_args(true)
.after_help(after_help)
.arg(backup_control::arguments::backup())
.arg(backup_control::arguments::backup_no_args())
/*.arg(
Arg::new(options::DIRECTORY)
.short('d')
.long(options::DIRECTORY)
.help("allow users with appropriate privileges to attempt to make hard links to directories")
)*/
.arg(
Arg::new(options::FORCE)
.short('f')
.long(options::FORCE)
.help("remove existing destination files")
.overrides_with(options::INTERACTIVE)
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::INTERACTIVE)
.short('i')
.long(options::INTERACTIVE)
.help("prompt whether to remove existing destination files")
.overrides_with(options::FORCE)
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::NO_DEREFERENCE)
.short('n')
.long(options::NO_DEREFERENCE)
.help("treat LINK_NAME as a normal file if it is a\nsymbolic link to a directory")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::LOGICAL)
.short('L')
.long(options::LOGICAL)
.help("follow TARGETs that are symbolic links")
.overrides_with(options::PHYSICAL)
.action(ArgAction::SetTrue),
)
.arg(
// Not implemented yet
Arg::new(options::PHYSICAL)
.short('P')
.long(options::PHYSICAL)
.help("make hard links directly to symbolic links")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::SYMBOLIC)
.short('s')
.long(options::SYMBOLIC)
.help("make symbolic links instead of hard links")
// override added for https://github.com/uutils/coreutils/issues/2359
.overrides_with(options::SYMBOLIC)
.action(ArgAction::SetTrue),
)
.arg(backup_control::arguments::suffix())
.arg(
Arg::new(options::TARGET_DIRECTORY)
.short('t')
.long(options::TARGET_DIRECTORY)
.help("specify the DIRECTORY in which to create the links")
.value_name("DIRECTORY")
.value_hint(clap::ValueHint::DirPath)
.value_parser(clap::value_parser!(OsString))
.conflicts_with(options::NO_TARGET_DIRECTORY),
)
.arg(
Arg::new(options::NO_TARGET_DIRECTORY)
.short('T')
.long(options::NO_TARGET_DIRECTORY)
.help("treat LINK_NAME as a normal file always")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::RELATIVE)
.short('r')
.long(options::RELATIVE)
.help("create symbolic links relative to link location")
.requires(options::SYMBOLIC)
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::VERBOSE)
.short('v')
.long(options::VERBOSE)
.help("print name of each linked file")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(ARG_FILES)
.action(ArgAction::Append)
.value_hint(clap::ValueHint::AnyPath)
.value_parser(clap::value_parser!(OsString))
.required(true)
.num_args(1..),
)
}
fn exec(files: &[PathBuf], settings: &Settings) -> UResult<()> {
// Handle cases where we create links in a directory first.
if let Some(target_path) = &settings.target_dir {
// 4th form: a directory is specified by -t.
return link_files_in_dir(files, target_path, settings);
}
if !settings.no_target_dir {
if files.len() == 1 {
// 2nd form: the target directory is the current directory.
return link_files_in_dir(files, &PathBuf::from("."), settings);
}
let last_file = &PathBuf::from(files.last().unwrap());
// pi-uutils: probe the destination via the resolved path.
if files.len() > 2 || pi_uutils_ctx::resolve(last_file).is_dir() {
// 3rd form: create links in the last argument.
return link_files_in_dir(&files[0..files.len() - 1], last_file, settings);
}
}
// 1st form. Now there should be only two operands, but if -T is
// specified we may have a wrong number of operands.
if files.len() == 1 {
return Err(LnError::MissingDestination(files[0].clone()).into());
}
if files.len() > 2 {
// pi-uutils: `uucore::execution_phrase()` reads the process argv,
// which is the host shell's; the builtin is always invoked as "ln".
return Err(LnError::ExtraOperand(files[2].clone().into(), "ln".to_string()).into());
}
assert!(!files.is_empty());
link(&files[0], &files[1], settings)
}
#[allow(clippy::cognitive_complexity)]
fn link_files_in_dir(files: &[PathBuf], target_dir: &Path, settings: &Settings) -> UResult<()> {
// pi-uutils: resolved target directory for every syscall below; the
// operand keeps its as-typed spelling for display and link-name building.
let target_dir_fs = pi_uutils_ctx::resolve(target_dir);
if !target_dir_fs.is_dir() {
return Err(LnError::TargetIsNotADirectory(target_dir.to_owned()).into());
}
// remember the linked destinations for further usage
let mut linked_destinations: HashSet<PathBuf> = HashSet::with_capacity(files.len());
let mut all_successful = true;
for srcpath in files {
let targetpath = if settings.no_dereference && target_dir_fs.is_symlink() {
let remove_target = || {
// In that case, we don't want to do link resolution
// We need to clean the target
if target_dir_fs.is_file()
&& let Err(e) = fs::remove_file(&target_dir_fs)
{
show_error(format_args!("Could not update {}: {e}", target_dir.quote()));
}
#[cfg(windows)]
if target_dir_fs.is_dir() {
// Not sure why but on Windows, the symlink can be
// considered as a dir
// See test_ln::test_symlink_no_deref_dir
if let Err(e) = fs::remove_dir(&target_dir_fs) {
show_error(format_args!("Could not update {}: {e}", target_dir.quote()));
}
}
};
match settings.overwrite {
OverwriteMode::NoClobber => {},
OverwriteMode::Interactive => {
if prompt_yes(format_args!("replace {}?", target_dir.quote())) {
remove_target();
}
},
OverwriteMode::Force => {
remove_target();
},
}
target_dir.to_path_buf()
} else if let Some(name) = srcpath.as_os_str().to_str() {
match Path::new(name).file_name() {
Some(basename) => target_dir.join(basename),
// This can be None only for "." or "..". Trying
// to create a link with such name will fail with
// EEXIST, which agrees with the behavior of GNU
// coreutils.
None => target_dir.join(name),
}
} else {
show_error(format_args!("cannot stat {}: No such file or directory", srcpath.quote()));
all_successful = false;
continue;
};
if linked_destinations.contains(&targetpath) {
// If the target file was already created in this ln call, do not overwrite
show_error(format_args!(
"will not overwrite just-created {} with {}",
targetpath.quote(),
srcpath.quote()
));
all_successful = false;
} else if let Err(e) = link(srcpath, &targetpath, settings) {
show_error(format_args!("{e}"));
all_successful = false;
}
linked_destinations.insert(targetpath.clone());
}
if all_successful {
Ok(())
} else {
Err(LnError::SomeLinksFailed.into())
}
}
fn relative_path<'a>(src: &'a Path, dst: &Path) -> Cow<'a, Path> {
// pi-uutils: canonicalize from the resolved operands so `-r` computes the
// link text against the shell working directory (uucore's canonicalize
// would otherwise fall back to the process cwd for relative paths).
if let Ok(src_abs) =
canonicalize(pi_uutils_ctx::resolve(src), MissingHandling::Missing, ResolveMode::Physical)
&& let Ok(dst_abs) = canonicalize(
pi_uutils_ctx::resolve(dst.parent().unwrap()),
MissingHandling::Missing,
ResolveMode::Physical,
) {
return make_path_relative_to(src_abs, dst_abs).into();
}
src.into()
}
#[allow(clippy::cognitive_complexity)]
fn link(src: &Path, dst: &Path, settings: &Settings) -> UResult<()> {
let mut backup_path = None;
let source: Cow<'_, Path> = if settings.relative {
relative_path(src, dst)
} else {
src.into()
};
// pi-uutils: resolved counterparts of both operands for every filesystem
// syscall below. `src`/`dst`/`source` keep the as-typed spelling for
// display — and `source` is what gets stored as the symlink CONTENT, so it
// must never be resolved.
let src_fs = pi_uutils_ctx::resolve(src);
let dst_fs = pi_uutils_ctx::resolve(dst);
if dst_fs.is_symlink() || dst_fs.exists() {
// pi-uutils: probe numbered backups from the resolved destination so
// the directory scan hits the shell's working directory.
backup_path = backup_control::get_backup_path(settings.backup, &dst_fs, &settings.suffix);
if settings.backup == BackupMode::Existing && !settings.symbolic {
// when ln --backup f f, it should detect that it is the same file
if paths_refer_to_same_file(&src_fs, &dst_fs, true) {
return Err(LnError::SameFile(src.to_owned(), dst.to_owned()).into());
}
}
if let Some(p) = &backup_path {
fs::rename(&dst_fs, p).map_err_context(|| format!("cannot backup {}", dst.quote()))?;
}
match settings.overwrite {
OverwriteMode::NoClobber => {},
OverwriteMode::Interactive => {
if !prompt_yes(format_args!("replace {}?", dst.quote())) {
return Err(LnError::SomeLinksFailed.into());
}
let _ = fs::remove_file(&dst_fs);
// In case of error, don't do anything
},
OverwriteMode::Force => {
if !dst_fs.is_symlink() && paths_refer_to_same_file(&src_fs, &dst_fs, true) {
// Even in force overwrite mode, verify we are not targeting the same entry and
// return a SameFile error if so
let same_entry = match (
canonicalize(&src_fs, MissingHandling::Missing, ResolveMode::Physical),
canonicalize(&dst_fs, MissingHandling::Missing, ResolveMode::Physical),
) {
(Ok(src), Ok(dst)) => src == dst,
_ => true,
};
if same_entry {
return Err(LnError::SameFile(src.to_owned(), dst.to_owned()).into());
}
}
let _ = fs::remove_file(&dst_fs);
// In case of error, don't do anything
},
}
}
let res: UResult<()> = if settings.symbolic {
// pi-uutils: the link is created at the resolved location, but its
// content (`source`) stays exactly as typed, like GNU ln. uucore's
// io-error conversion renders EEXIST as "Already exists"; format the
// GNU-style diagnostic ("failed to create symbolic link 'x': File
// exists") from the raw OS error instead.
symlink(&source, &dst_fs).map_err(|e| {
USimpleError::new(
1,
format!("failed to create symbolic link {}: {}", dst.quote(), strip_errno(&e)),
)
})
} else {
// pi-uutils: hard links dereference their target, so the resolved
// source is what the syscalls get.
let source_fs = pi_uutils_ctx::resolve(&source);
let p = if settings.logical && source_fs.is_symlink() {
fs::canonicalize(&source_fs)
.map_err_context(|| format!("failed to access {}", source.quote()))?
} else {
source_fs
};
match fs::hard_link(&p, &dst_fs) {
Ok(()) => Ok(()),
Err(_) if p.is_dir() => {
Err(LnError::FailedToCreateHardLinkDir(source.to_path_buf()).into())
},
// pi-uutils: same GNU-style rendering as the symlink arm (uucore
// would print "Already exists" for EEXIST).
Err(e) => Err(USimpleError::new(
1,
format!(
"failed to create hard link {} => {}: {}",
source.quote(),
dst.quote(),
strip_errno(&e)
),
)),
}
};
if let Err(e) = res {
if let Some(p) = &backup_path {
fs::rename(p, &dst_fs).map_err_context(|| format!("cannot backup {}", dst.quote()))?;
}
return Err(e);
}
if settings.verbose {
// pi-uutils: verbose output goes to the context stdout.
let mut out = pi_uutils_ctx::stdout();
write!(out, "{} -> {}", dst.quote(), source.quote())?;
match backup_path {
Some(path) => {
// pi-uutils: `path` derives from the resolved (absolute)
// destination; rebuild a display path from the operand for
// the verbose message.
let backup_display = match (dst.parent(), path.file_name()) {
(Some(parent), Some(name)) if !parent.as_os_str().is_empty() => parent.join(name),
(_, Some(name)) => PathBuf::from(name),
_ => path.clone(),
};
writeln!(out, " (backup: {})", backup_display.quote())?;
},
None => writeln!(out)?,
}
}
Ok(())
}
#[cfg(windows)]
pub fn symlink<P1: AsRef<Path>, P2: AsRef<Path>>(src: P1, dst: P2) -> std::io::Result<()> {
// pi-uutils: the dir/file probe resolves the target against the shell
// working directory (upstream consults the process cwd); the stored link
// content is still the caller's as-typed `src`.
if pi_uutils_ctx::resolve(src.as_ref()).is_dir() {
symlink_dir(src, dst)
} else {
symlink_file(src, dst)
}
}
#[cfg(target_os = "wasi")]
fn symlink<P1: AsRef<Path>, P2: AsRef<Path>>(_src: P1, _dst: P2) -> std::io::Result<()> {
Err(std::io::Error::new(
std::io::ErrorKind::Unsupported,
"symlinks not supported on this platform",
))
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc};
use parking_lot::Mutex;
use pi_uutils_ctx::ScopeIo;
use super::*;
fn run_with_stdin(cwd: PathBuf, args: Vec<&str>, stdin: &[u8]) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
self.buf.lock().write(buf)
}
fn flush(&mut self) -> std::io::Result<()> {
self.buf.lock().flush()
}
}
let io = ScopeIo {
stdin: Box::new(std::io::Cursor::new(stdin.to_vec())),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd,
env: HashMap::new(),
cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)),
};
let argv: Vec<OsString> = std::iter::once("ln")
.chain(args)
.map(OsString::from)
.collect();
let code = pi_uutils_ctx::scope(io, || run(argv));
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
(code, out_str, err_str)
}
fn run_in(cwd: PathBuf, args: Vec<&str>) -> (i32, String, String) {
run_with_stdin(cwd, args, b"")
}
/// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var,
/// which canonicalizing code paths would otherwise expand mid-assertion).
fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) {
let dir = tempfile::tempdir().unwrap();
let canon = fs::canonicalize(dir.path()).unwrap();
(dir, canon)
}
#[cfg(unix)]
#[test]
fn symlink_relative_operands_create_in_scope_cwd_with_literal_content() {
let (_dir, root) = canonical_tempdir();
// Relative operands + scope cwd differing from the process cwd: only
// the call-site `pi_uutils_ctx::resolve` patch places the link in the
// tempdir — while the CONTENT must stay exactly as typed.
let (code, stdout, stderr) = run_in(root.clone(), vec!["-s", "target", "link"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", ""));
let link = root.join("link");
assert!(link.is_symlink(), "link must be created inside the scope cwd");
assert_eq!(fs::read_link(&link).unwrap(), PathBuf::from("target"));
}
#[cfg(unix)]
#[test]
fn hard_link_shares_inode() {
use std::os::unix::fs::MetadataExt;
let (_dir, root) = canonical_tempdir();
fs::write(root.join("a"), b"payload").unwrap();
let (code, stdout, stderr) = run_in(root.clone(), vec!["a", "b"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", ""));
assert_eq!(fs::read(root.join("b")).unwrap(), b"payload");
assert_eq!(fs::metadata(root.join("a")).unwrap().nlink(), 2);
assert_eq!(
fs::metadata(root.join("a")).unwrap().ino(),
fs::metadata(root.join("b")).unwrap().ino()
);
}
#[cfg(unix)]
#[test]
fn existing_destination_without_force_fails_with_file_exists() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("link"), b"old").unwrap();
let (code, stdout, stderr) = run_in(root.clone(), vec!["-s", "target", "link"]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert_eq!(stderr, "ln: failed to create symbolic link 'link': File exists\n");
assert_eq!(fs::read(root.join("link")).unwrap(), b"old", "destination must be untouched");
}
#[cfg(unix)]
#[test]
fn force_overwrites_existing_destination() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("link"), b"old").unwrap();
let (code, stdout, stderr) = run_in(root.clone(), vec!["-sf", "target", "link"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", ""));
assert_eq!(fs::read_link(root.join("link")).unwrap(), PathBuf::from("target"));
}
#[cfg(unix)]
#[test]
fn verbose_symlink_prints_mapping_to_stdout() {
let (_dir, root) = canonical_tempdir();
let (code, stdout, stderr) = run_in(root, vec!["-sv", "target", "link"]);
assert_eq!(code, 0);
assert_eq!(stdout, "'link' -> 'target'\n");
assert_eq!(stderr, "");
}
#[cfg(unix)]
#[test]
fn interactive_prompt_reads_ctx_stdin() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("link"), b"old").unwrap();
// Decline: destination untouched, some-links-failed exit code, no
// dangling "ln: " diagnostic beyond the prompt itself.
let (code, stdout, stderr) =
run_with_stdin(root.clone(), vec!["-si", "target", "link"], b"n\n");
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert_eq!(stderr, "ln: replace 'link'? ");
assert!(!root.join("link").is_symlink());
// Accept: existing file is replaced by the symlink.
let (code, _, stderr) = run_with_stdin(root.clone(), vec!["-si", "target", "link"], b"y\n");
assert_eq!(code, 0);
assert_eq!(stderr, "ln: replace 'link'? ");
assert_eq!(fs::read_link(root.join("link")).unwrap(), PathBuf::from("target"));
}
#[cfg(unix)]
#[test]
fn relative_flag_computes_link_text_against_scope_cwd() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("target"), b"x").unwrap();
fs::create_dir(root.join("sub")).unwrap();
let (code, stdout, stderr) = run_in(root.clone(), vec!["-sr", "target", "sub/link"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", ""));
assert_eq!(fs::read_link(root.join("sub").join("link")).unwrap(), PathBuf::from("../target"));
}
#[cfg(unix)]
#[test]
fn target_directory_flag_places_links_in_directory() {
let (_dir, root) = canonical_tempdir();
fs::create_dir(root.join("d")).unwrap();
let (code, stdout, stderr) = run_in(root.clone(), vec!["-s", "-t", "d", "x"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", ""));
assert_eq!(fs::read_link(root.join("d").join("x")).unwrap(), PathBuf::from("x"));
}
#[test]
fn missing_destination_is_an_error() {
let (_dir, root) = canonical_tempdir();
let (code, stdout, stderr) = run_in(root, vec!["-T", "only"]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert!(
stderr.contains("missing destination file operand after 'only'"),
"stderr was: {stderr:?}"
);
}
#[test]
fn help_renders_to_scope_stdout() {
let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["--help"]);
assert_eq!(code, 0);
assert!(stdout.contains("Usage:"));
assert!(stdout.contains("Make links between files."));
assert_eq!(stderr, "");
}
}
+24
View File
@@ -0,0 +1,24 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/mktemp), patched to resolve
# path arguments against the shell working directory and route I/O through
# pi-uutils-ctx so it can run in-process as a shell builtin. See src/mktemp.rs
# for the patch markers (`pi-uutils:` comments).
[package]
name = "uu_mktemp"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "mktemp ~ (uutils) create and display a temporary file or directory from TEMPLATE (vendored + patched for in-process embedding)"
[lib]
path = "src/mktemp.rs"
[dependencies]
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
rand = { version = "0.10.0", features = ["std_rng"] }
tempfile = "3.15.0"
thiserror = "2.0.3"
uucore = "0.8.0"
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[dev-dependencies]
parking_lot = "0.12"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+877
View File
@@ -0,0 +1,877 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore (paths) GPGHome findxs
// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process
// as a shell builtin. The temp-file parent directory (from `-p`/`--tmpdir` or a
// relative TEMPLATE prefix) is resolved against the shell working directory via
// `pi_uutils_ctx::resolve` at the creation sites, so the printed path is the
// path actually created. TMPDIR/POSIXLY_CORRECT are read from the scope
// environment, all stdio goes through `pi_uutils_ctx`, `translate!` strings are
// literalized, and the entry point no longer calls `std::process::exit`.
#[cfg(unix)]
use std::fs;
#[cfg(unix)]
use std::os::unix::prelude::PermissionsExt;
use std::{
env,
ffi::{OsStr, OsString},
io::{ErrorKind, Write},
iter,
path::{MAIN_SEPARATOR, Path, PathBuf},
};
use clap::{
Arg, ArgAction, ArgMatches, Command,
builder::{TypedValueParser, ValueParserFactory},
};
use pi_uutils_ctx::format_usage;
use rand::{
RngExt as _, SeedableRng as _,
rngs::{self, SmallRng},
};
use tempfile::Builder;
use thiserror::Error;
use uucore::{
display::Quotable,
error::{FromIo, UError, UResult},
};
static DEFAULT_TEMPLATE: &str = "tmp.XXXXXXXXXX";
static OPT_DIRECTORY: &str = "directory";
static OPT_DRY_RUN: &str = "dry-run";
static OPT_QUIET: &str = "quiet";
static OPT_SUFFIX: &str = "suffix";
static OPT_TMPDIR: &str = "tmpdir";
static OPT_P: &str = "p";
static OPT_T: &str = "t";
static ARG_TEMPLATE: &str = "template";
#[cfg(not(windows))]
const TMPDIR_ENV_VAR: &str = "TMPDIR";
#[cfg(windows)]
const TMPDIR_ENV_VAR: &str = "TMP";
const FALLBACK_TMPDIR: &str = "/tmp";
// pi-uutils: `translate!` error strings literalized from locales/en-US.ftl.
#[derive(Error, Debug)]
enum MkTempError {
#[error("could not persist file {}", .0.quote())]
PersistError(PathBuf),
#[error("with --suffix, template {} must end in X", .0.quote())]
MustEndInX(String),
#[error("too few X's in template {}", .0.quote())]
TooFewXs(String),
#[error("invalid template, {}, contains directory separator", .0.quote())]
PrefixContainsDirSeparator(String),
#[error("invalid suffix {}, contains directory separator", .0.quote())]
SuffixContainsDirSeparator(String),
#[error("invalid template, {}; with --tmpdir, it may not be absolute", .0.quote())]
InvalidTemplate(OsString),
#[error("too many templates")]
TooManyTemplates,
#[error("failed to create {} via template {}: No such file or directory", .0, .1.quote())]
NotFound(String, PathBuf),
}
impl UError for MkTempError {
fn usage(&self) -> bool {
matches!(self, Self::TooManyTemplates)
}
}
/// Options parsed from the command-line.
///
/// This provides a layer of indirection between the application logic
/// and the argument parsing library `clap`, allowing each to vary
/// independently.
#[derive(Clone)]
pub struct Options {
/// Whether to create a temporary directory instead of a file.
pub directory: bool,
/// Whether to just print the name of a file that would have been created.
pub dry_run: bool,
/// Whether to suppress file creation error messages.
pub quiet: bool,
/// The directory in which to create the temporary file.
///
/// If `None`, the file will be created in the current directory.
pub tmpdir: Option<PathBuf>,
/// The suffix to append to the temporary file, if any.
pub suffix: Option<OsString>,
/// Whether to treat the template argument as a single file path component.
pub treat_as_template: bool,
/// The template to use for the name of the temporary file.
pub template: OsString,
}
impl Options {
fn from(matches: &ArgMatches) -> Self {
let tmpdir = matches
.get_one::<Option<PathBuf>>(OPT_TMPDIR)
.or_else(|| matches.get_one::<Option<PathBuf>>(OPT_P))
.map(|dir| match dir {
// If the argument of -p/--tmpdir is non-empty, use it as the
// tmpdir.
Some(d) => d.clone(),
// Otherwise use $TMPDIR if set, else use the system's default
// temporary directory.
None => get_tmpdir_env_or_default(),
});
let (tmpdir, template) = match matches.get_one::<OsString>(ARG_TEMPLATE) {
// If no template argument is given, `--tmpdir` is implied.
None => {
let tmpdir = Some(tmpdir.unwrap_or_else(get_tmpdir_env_or_default));
let template = DEFAULT_TEMPLATE;
(tmpdir, OsString::from(template))
},
Some(template) => {
// pi-uutils: TMPDIR comes from the scope environment, not the
// host process environment.
let tmpdir = if let Some(tmpdir) = pi_uutils_ctx::var(TMPDIR_ENV_VAR)
&& matches.get_flag(OPT_T)
{
Some(PathBuf::from(tmpdir))
} else if tmpdir.is_some() {
tmpdir
} else if matches.get_flag(OPT_T) || matches.contains_id(OPT_TMPDIR) {
// If --tmpdir is given without an argument, or -t is given
// export in TMPDIR
Some(env::temp_dir())
} else {
None
};
(tmpdir, template.clone())
},
};
Self {
directory: matches.get_flag(OPT_DIRECTORY),
dry_run: matches.get_flag(OPT_DRY_RUN),
quiet: matches.get_flag(OPT_QUIET),
tmpdir,
suffix: matches.get_one::<OsString>(OPT_SUFFIX).cloned(),
treat_as_template: matches.get_flag(OPT_T),
template,
}
}
}
/// Parameters that control the path to and name of the temporary file.
///
/// The temporary file will be created at
///
/// ```text
/// {directory}/{prefix}{XXX}{suffix}
/// ```
///
/// where `{XXX}` is a sequence of random characters whose length is
/// `num_rand_chars`.
struct Params {
/// The directory that will contain the temporary file.
directory: PathBuf,
/// The (non-random) prefix of the temporary file.
prefix: String,
/// The number of random characters in the name of the temporary file.
num_rand_chars: usize,
/// The (non-random) suffix of the temporary file.
suffix: String,
}
/// Find the start and end indices of the last contiguous block of Xs.
///
/// If no contiguous block of at least three Xs could be found, this
/// function returns `None`.
///
/// # Examples
///
/// ```rust,ignore
/// assert_eq!(find_last_contiguous_block_of_xs("XXX_XXX"), Some((4, 7)));
/// assert_eq!(find_last_contiguous_block_of_xs("aXbXcX"), None);
/// ```
fn find_last_contiguous_block_of_xs(s: &str) -> Option<(usize, usize)> {
let bytes = s.as_bytes();
// Find the index of the last 'X'.
let end = bytes.iter().rposition(|&b| b == b'X')?;
// Walk left to find the start of the run of Xs that ends at `end`.
let mut start = end;
while start > 0 && bytes[start - 1] == b'X' {
start -= 1;
}
if end + 1 - start >= 3 {
Some((start, end + 1))
} else {
None
}
}
impl Params {
fn from(options: Options) -> Result<Self, MkTempError> {
// Convert OsString template to string for processing
// When using -t flag, be permissive with invalid UTF-8 like GNU mktemp
// Otherwise, maintain strict UTF-8 validation (existing behavior)
let template_str = if options.treat_as_template {
// For -t templates, use lossy conversion for GNU compatibility
options.template.to_string_lossy().into_owned()
} else {
// For regular templates, maintain strict validation
match options.template.to_str() {
Some(s) => s.to_string(),
None => {
return Err(MkTempError::InvalidTemplate("template contains invalid UTF-8".into()));
},
}
};
// The template argument must end in 'X' if a suffix option is given.
if options.suffix.is_some() && !template_str.ends_with('X') {
return Err(MkTempError::MustEndInX(template_str.clone()));
}
// Get the start and end indices of the randomized part of the template.
//
// For example, if the template is "abcXXXXyz", then `i` is 3 and `j` is 7.
let Some((i, j)) = find_last_contiguous_block_of_xs(&template_str) else {
let s = match options.suffix {
// If a suffix is specified, the error message includes the template without the suffix.
Some(_) => template_str
.chars()
.take(template_str.len())
.collect::<String>(),
None => template_str.clone(),
};
return Err(MkTempError::TooFewXs(s));
};
// Combine the directory given as an option and the prefix of the template.
//
// For example, if `tmpdir` is "a/b" and the template is "c/dXXX",
// then `prefix` is "a/b/c/d".
let tmpdir = options.tmpdir;
let prefix_from_option = tmpdir.clone().unwrap_or_default();
let prefix_from_template = &template_str[..i];
let prefix_path = Path::new(&prefix_from_option).join(prefix_from_template);
if options.treat_as_template && prefix_from_template.contains(MAIN_SEPARATOR) {
return Err(MkTempError::PrefixContainsDirSeparator(template_str.clone()));
}
if tmpdir.is_some() && Path::new(prefix_from_template).is_absolute() {
return Err(MkTempError::InvalidTemplate(template_str.clone().into()));
}
// Split the parent directory from the file part of the prefix.
//
// For example, if `prefix_path` is "a/b/c/d", then `directory` is
// "a/b/c" and `prefix` gets reassigned to "d".
let (directory, prefix) = {
let prefix_str = prefix_path.to_string_lossy();
if prefix_str.ends_with(MAIN_SEPARATOR) {
(prefix_path, String::new())
} else {
let directory = match prefix_path.parent() {
None => PathBuf::new(),
Some(d) => d.to_path_buf(),
};
let prefix = match prefix_path.file_name() {
None => String::new(),
Some(f) => f.to_string_lossy().to_string(),
};
(directory, prefix)
}
};
// Combine the suffix from the template with the suffix given as an option.
//
// For example, if the suffix command-line argument is ".txt" and
// the template is "XXXabc", then `suffix` is "abc.txt".
let suffix_from_option = options
.suffix
.map(|s| s.to_string_lossy().to_string())
.unwrap_or_default();
let suffix_from_template = &template_str[j..];
let suffix = format!("{suffix_from_template}{suffix_from_option}");
if suffix.contains(MAIN_SEPARATOR) {
return Err(MkTempError::SuffixContainsDirSeparator(suffix));
}
// The number of random characters in the template.
//
// For example, if the template is "abcXXXXyz", then the number of
// random characters is four.
let num_rand_chars = j - i;
Ok(Self { directory, prefix, num_rand_chars, suffix })
}
}
/// Custom parser that converts empty string to `None`, and non-empty string to
/// `Some(PathBuf)`.
///
/// This parser is used for the `-p` and `--tmpdir` options where an empty
/// string argument should be treated as "not provided", causing mktemp to fall
/// back to using the `$TMPDIR` environment variable or the system's default
/// temporary directory.
///
/// # Examples
///
/// - Empty string `""` -> `None`
/// - Non-empty string `"/tmp"` -> `Some(PathBuf::from("/tmp"))`
///
/// This handles the special case where users can pass an empty directory name
/// to explicitly request fallback behavior.
#[derive(Clone, Debug)]
struct OptionalPathBufParser;
impl TypedValueParser for OptionalPathBufParser {
type Value = Option<PathBuf>;
fn parse_ref(
&self,
_cmd: &Command,
_arg: Option<&Arg>,
value: &OsStr,
) -> Result<Self::Value, clap::Error> {
if value.is_empty() {
Ok(None)
} else {
Ok(Some(PathBuf::from(value)))
}
}
}
impl ValueParserFactory for OptionalPathBufParser {
type Parser = Self;
fn value_parser() -> Self::Parser {
Self
}
}
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
/// arguments directly (without the uucore clap-localization helper that would
/// terminate the process), renders clap help/usage/version to the context
/// streams, and maps the `UResult` to an exit code, so it is safe to run inside
/// the host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(&argv) {
Ok(matches) => matches,
Err(err) => {
// pi-uutils: upstream maps a too-many-values clap error on the
// TEMPLATE argument to the GNU "too many templates" usage error.
if err.kind() == clap::error::ErrorKind::TooManyValues
&& err.context().any(|(kind, val)| {
kind == clap::error::ContextKind::InvalidArg
&& val == &clap::error::ContextValue::String("[template]".into())
}) {
let _ = writeln!(pi_uutils_ctx::stderr(), "mktemp: too many templates");
return 1;
}
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
match mktemp_main(&argv, &matches) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
// pi-uutils: --quiet failures surface as bare exit-code errors
// that render to an empty message; don't emit a dangling
// "mktemp: " prefix.
let msg = err.to_string();
if !msg.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "mktemp: {msg}");
}
if code == 0 { 1 } else { code }
},
}
}
fn mktemp_main(args: &[OsString], matches: &ArgMatches) -> UResult<()> {
// Parse command-line options into a format suitable for the
// application logic.
let options = Options::from(matches);
// pi-uutils: POSIXLY_CORRECT comes from the scope environment.
if pi_uutils_ctx::var("POSIXLY_CORRECT").is_some() {
// If POSIXLY_CORRECT was set, template MUST be the last argument.
if matches.contains_id(ARG_TEMPLATE) {
// Template argument was provided, check if was the last one.
if args.last().unwrap() != &options.template {
return Err(Box::new(MkTempError::TooManyTemplates));
}
}
}
let dry_run = options.dry_run;
let suppress_file_err = options.quiet;
let make_dir = options.directory;
// Parse file path parameters from the command-line options.
let Params { directory: tmpdir, prefix, num_rand_chars: rand, suffix } = Params::from(options)?;
// Create the temporary file or directory, or simulate creating it.
let res = if dry_run {
Ok(dry_exec(&tmpdir, &prefix, rand, &suffix))
} else {
exec(&tmpdir, &prefix, rand, &suffix, make_dir)
};
let res = if suppress_file_err {
// Mapping all UErrors to ExitCodes prevents the errors from being printed
res.map_err(|e| e.code().into())
} else {
res
};
// pi-uutils: replacement for upstream's `println_verbatim` — writes the
// created path's bytes verbatim to the context stdout instead of the
// process stdout.
let path = res?;
let print = || -> std::io::Result<()> {
let mut out = pi_uutils_ctx::stdout();
out.write_all(uucore::os_str_as_bytes(path.as_os_str()).map_err(std::io::Error::other)?)?;
out.write_all(b"\n")?;
out.flush()
};
print().map_err_context(|| "failed to print directory name".to_string())?;
Ok(())
}
pub fn uu_app() -> Command {
Command::new("mktemp")
.version(uucore::crate_version!())
.about("Create a temporary file or directory.")
.override_usage(format_usage("mktemp [OPTION]... [TEMPLATE]"))
.infer_long_args(true)
.arg(
Arg::new(OPT_DIRECTORY)
.short('d')
.long(OPT_DIRECTORY)
.help("Make a directory instead of a file")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_DRY_RUN)
.short('u')
.long(OPT_DRY_RUN)
.help("do not create anything; merely print a name (unsafe)")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_QUIET)
.short('q')
.long("quiet")
.help("Fail silently if an error occurs.")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_SUFFIX)
.long(OPT_SUFFIX)
.help(
"append SUFFIX to TEMPLATE; SUFFIX must not contain a path separator. This option \
is implied if TEMPLATE does not end with X.",
)
.value_name("SUFFIX")
.value_parser(clap::value_parser!(OsString)),
)
.arg(
Arg::new(OPT_P)
.short('p')
.help("short form of --tmpdir")
.value_name("DIR")
.num_args(1)
.value_parser(OptionalPathBufParser)
.value_hint(clap::ValueHint::DirPath),
)
.arg(
Arg::new(OPT_TMPDIR)
.long(OPT_TMPDIR)
.help(
"interpret TEMPLATE relative to DIR; if DIR is not specified, use $TMPDIR ($TMP on \
windows) if set, else /tmp. With this option, TEMPLATE must not be an absolute \
name; unlike with -t, TEMPLATE may contain slashes, but mktemp creates only the \
final component",
)
.value_name("DIR")
// Allows use of default argument just by setting --tmpdir. Else,
// use provided input to generate tmpdir
.num_args(0..=1)
// Require an equals to avoid ambiguity if no tmpdir is supplied
.require_equals(true)
.overrides_with(OPT_P)
.value_parser(OptionalPathBufParser)
.value_hint(clap::ValueHint::DirPath),
)
.arg(
Arg::new(OPT_T)
.short('t')
.help(
"Generate a template (using the supplied prefix and TMPDIR (TMP on windows) if \
set) to create a filename template [deprecated]",
)
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(ARG_TEMPLATE)
.num_args(..=1)
.value_parser(clap::value_parser!(OsString)),
)
}
fn dry_exec(tmpdir: &Path, prefix: &str, rand: usize, suffix: &str) -> PathBuf {
// pi-uutils: resolve the parent directory against the shell working
// directory so the printed candidate matches where creation would occur.
let tmpdir = pi_uutils_ctx::resolve(tmpdir);
let len = prefix.len() + suffix.len() + rand;
let mut buf = Vec::with_capacity(len);
buf.extend(prefix.as_bytes());
buf.extend(iter::repeat_n(b'X', rand));
buf.extend(suffix.as_bytes());
// Randomize.
let bytes = &mut buf[prefix.len()..prefix.len() + rand];
SmallRng::try_from_rng(&mut rngs::SysRng)
.unwrap_or_else(|_| {
//rand::rng panics if getrandom failed
SmallRng::seed_from_u64(bytes.as_ptr() as usize as u64)
})
.fill(bytes);
for byte in bytes {
*byte = match *byte % 62 {
v @ 0..=9 => v + b'0',
v @ 10..=35 => v - 10 + b'a',
v @ 36..=61 => v - 36 + b'A',
_ => unreachable!(),
}
}
// We guarantee utf8.
let buf = String::from_utf8(buf).unwrap();
tmpdir.join(buf)
}
/// Create a temporary directory with the given parameters.
///
/// This function creates a temporary directory as a subdirectory of
/// `dir`. The name of the directory is the concatenation of `prefix`,
/// a string of `rand` random characters, and `suffix`. The
/// permissions of the directory are set to `u+rwx`
///
/// # Errors
///
/// If the temporary directory could not be written to disk or if the
/// given directory `dir` does not exist.
fn make_temp_dir(dir: &Path, prefix: &str, rand: usize, suffix: &str) -> UResult<PathBuf> {
let mut builder = Builder::new();
builder.prefix(prefix).rand_bytes(rand).suffix(suffix);
// On *nix platforms grant read-write-execute for owner only.
// The directory is created with these permission at creation time, using
// mkdir(3) syscall. This is not relevant on Windows systems. See: https://docs.rs/tempfile/latest/tempfile/#security
// `fs` is not imported on Windows anyways.
#[cfg(not(windows))]
builder.permissions(fs::Permissions::from_mode(0o700));
match builder.tempdir_in(dir) {
Ok(d) => {
// `keep` consumes the TempDir without removing it
let path = d.keep();
Ok(path)
},
Err(e) if e.kind() == ErrorKind::NotFound => {
let filename = format!("{prefix}{}{suffix}", "X".repeat(rand));
let path = Path::new(dir).join(filename);
Err(MkTempError::NotFound("directory".to_string(), path).into())
},
Err(e) => Err(e.into()),
}
}
/// Create a temporary file with the given parameters.
///
/// This function creates a temporary file in the directory `dir`. The
/// name of the file is the concatenation of `prefix`, a string of
/// `rand` random characters, and `suffix`. The permissions of the
/// file are set to `u+rw`.
///
/// # Errors
///
/// If the file could not be written to disk or if the directory does
/// not exist.
fn make_temp_file(dir: &Path, prefix: &str, rand: usize, suffix: &str) -> UResult<PathBuf> {
let mut builder = Builder::new();
builder.prefix(prefix).rand_bytes(rand).suffix(suffix);
match builder.tempfile_in(dir) {
// `keep` ensures that the file is not deleted
Ok(named_tempfile) => match named_tempfile.keep() {
Ok((_, pathbuf)) => Ok(pathbuf),
Err(e) => Err(MkTempError::PersistError(e.file.path().to_path_buf()).into()),
},
Err(e) if e.kind() == ErrorKind::NotFound => {
let filename = format!("{prefix}{}{suffix}", "X".repeat(rand));
let path = Path::new(dir).join(filename);
Err(MkTempError::NotFound("file".to_string(), path).into())
},
Err(e) => Err(e.into()),
}
}
fn exec(dir: &Path, prefix: &str, rand: usize, suffix: &str, make_dir: bool) -> UResult<PathBuf> {
// pi-uutils: resolve the parent directory against the shell working
// directory at the creation site; the resolved form is also what gets
// printed, so the printed path is the path actually created.
let dir = pi_uutils_ctx::resolve(dir);
let path = if make_dir {
make_temp_dir(&dir, prefix, rand, suffix)?
} else {
make_temp_file(&dir, prefix, rand, suffix)?
};
// Get just the last component of the path to the created
// temporary file or directory.
let filename = path.file_name();
let filename = filename.unwrap().to_str().unwrap();
// Join the directory to the path to get the path to print.
// pi-uutils: unlike upstream (which re-joins the operand as typed), join
// the resolved directory so the printed path names the created entry even
// when the shell cwd differs from the process cwd.
let path = dir.join(filename);
Ok(path)
}
/// Reads from `TMPDIR_ENV_VAR` but defaults to /tmp if value is set to empty
/// string.
fn get_tmpdir_env_or_default() -> PathBuf {
// pi-uutils: read TMPDIR from the scope environment; when it is unset
// there, fall back to the host default temp dir as upstream does.
match pi_uutils_ctx::var(TMPDIR_ENV_VAR) {
Some(val) if val.is_empty() => PathBuf::from(FALLBACK_TMPDIR),
Some(val) => PathBuf::from(val),
None => env::temp_dir(),
}
}
/// Create a temporary file or directory
///
/// Behavior is determined by the `options` parameter, see [`Options`] for
/// details.
pub fn mktemp(options: &Options) -> UResult<PathBuf> {
// Parse file path parameters from the command-line options.
let Params { directory: tmpdir, prefix, num_rand_chars: rand, suffix } =
Params::from(options.clone())?;
// Create the temporary file or directory, or simulate creating it.
if options.dry_run {
Ok(dry_exec(&tmpdir, &prefix, rand, &suffix))
} else {
exec(&tmpdir, &prefix, rand, &suffix, options.directory)
}
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc};
use parking_lot::Mutex;
use pi_uutils_ctx::ScopeIo;
use super::*;
fn run_in(cwd: PathBuf, env: HashMap<String, String>, args: Vec<&str>) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
self.buf.lock().write(buf)
}
fn flush(&mut self) -> std::io::Result<()> {
self.buf.lock().flush()
}
}
let io = ScopeIo {
stdin: Box::new(std::io::empty()),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd,
env,
cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)),
};
let argv: Vec<OsString> = std::iter::once("mktemp")
.chain(args)
.map(OsString::from)
.collect();
let code = pi_uutils_ctx::scope(io, || run(argv));
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
(code, out_str, err_str)
}
/// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var,
/// which would otherwise break printed-path assertions).
fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) {
let dir = tempfile::tempdir().unwrap();
let canon = std::fs::canonicalize(dir.path()).unwrap();
(dir, canon)
}
fn tmpdir_env(dir: &Path) -> HashMap<String, String> {
HashMap::from([("TMPDIR".to_string(), dir.display().to_string())])
}
#[test]
fn default_invocation_creates_file_at_printed_path() {
let (_dir, root) = canonical_tempdir();
let (code, stdout, stderr) = run_in(root.clone(), tmpdir_env(&root), vec![]);
assert_eq!(code, 0);
assert_eq!(stderr, "");
let printed = PathBuf::from(stdout.trim_end_matches('\n'));
assert!(printed.is_file(), "printed path {printed:?} must be a regular file");
// Scope TMPDIR is honored for the default template.
assert_eq!(printed.parent(), Some(root.as_path()));
assert!(
printed
.file_name()
.unwrap()
.to_str()
.unwrap()
.starts_with("tmp.")
);
}
#[test]
fn directory_flag_creates_directory() {
let (_dir, root) = canonical_tempdir();
let (code, stdout, stderr) = run_in(root.clone(), tmpdir_env(&root), vec!["-d"]);
assert_eq!(code, 0);
assert_eq!(stderr, "");
let printed = PathBuf::from(stdout.trim_end_matches('\n'));
assert!(printed.is_dir(), "printed path {printed:?} must be a directory");
assert_eq!(printed.parent(), Some(root.as_path()));
}
#[test]
fn relative_tmpdir_resolves_against_scope_cwd() {
let (_dir, root) = canonical_tempdir();
std::fs::create_dir(root.join("sub")).unwrap();
// Relative -p operand + scope cwd differing from the process cwd: only
// the creation-site `pi_uutils_ctx::resolve` patch makes this land in
// the scope cwd's subdir.
let (code, stdout, stderr) =
run_in(root.clone(), HashMap::new(), vec!["-p", "sub", "foo.XXXX"]);
assert_eq!(code, 0);
assert_eq!(stderr, "");
let printed = PathBuf::from(stdout.trim_end_matches('\n'));
assert!(printed.is_file(), "printed path {printed:?} must exist");
assert_eq!(printed.parent(), Some(root.join("sub").as_path()));
assert!(
printed
.file_name()
.unwrap()
.to_str()
.unwrap()
.starts_with("foo.")
);
}
#[test]
fn too_few_xs_is_an_error() {
let (_dir, root) = canonical_tempdir();
let (code, stdout, stderr) = run_in(root, HashMap::new(), vec!["foo.XX"]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert_eq!(stderr, "mktemp: too few X's in template 'foo.XX'\n");
}
#[test]
fn dry_run_prints_nonexistent_path() {
let (_dir, root) = canonical_tempdir();
let (code, stdout, stderr) = run_in(root.clone(), tmpdir_env(&root), vec!["-u"]);
assert_eq!(code, 0);
assert_eq!(stderr, "");
let printed = PathBuf::from(stdout.trim_end_matches('\n'));
assert_eq!(printed.parent(), Some(root.as_path()));
assert!(!printed.exists(), "dry-run path {printed:?} must not be created");
}
#[test]
fn suffix_is_appended_after_random_block() {
let (_dir, root) = canonical_tempdir();
let (code, stdout, stderr) =
run_in(root.clone(), HashMap::new(), vec!["--suffix=.txt", "-p", ".", "fooXXXX"]);
assert_eq!(code, 0);
assert_eq!(stderr, "");
let printed = PathBuf::from(stdout.trim_end_matches('\n'));
assert!(printed.is_file());
let name = printed.file_name().unwrap().to_str().unwrap().to_string();
assert!(name.starts_with("foo") && name.ends_with(".txt"), "unexpected name {name}");
}
#[test]
fn quiet_suppresses_creation_error_message_but_not_exit_code() {
let (_dir, root) = canonical_tempdir();
let (code, stdout, stderr) =
run_in(root, HashMap::new(), vec!["-q", "-p", "missing-dir", "foo.XXXX"]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert_eq!(stderr, "", "--quiet must suppress the creation error message");
}
#[test]
fn help_renders_to_scope_stdout() {
let (code, stdout, stderr) = run_in(PathBuf::from("."), HashMap::new(), vec!["--help"]);
assert_eq!(code, 0);
assert!(stdout.contains("Usage:"));
assert!(stdout.contains("temporary file or directory"));
assert_eq!(stderr, "");
}
}
+22
View File
@@ -0,0 +1,22 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/nproc), patched to read the
# OMP_NUM_THREADS/OMP_THREAD_LIMIT environment variables from the shell scope
# environment and route I/O through pi-uutils-ctx so it can run in-process as a
# shell builtin. See src/nproc.rs for the patch markers (`pi-uutils:` comments).
[package]
name = "uu_nproc"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "nproc ~ (uutils) display the number of processing units available (vendored + patched for in-process embedding)"
[lib]
path = "src/nproc.rs"
[dependencies]
libc = "0.2.172"
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
uucore = { version = "0.8.0", features = ["fs"] }
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[dev-dependencies]
parking_lot = "0.12"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+287
View File
@@ -0,0 +1,287 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore (ToDO) NPROCESSORS nprocs numstr sysconf
// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process
// as a shell builtin. The OMP_NUM_THREADS and OMP_THREAD_LIMIT environment
// variables are read from the scope environment via `pi_uutils_ctx::var` (the
// shell's exported variables), not the host process environment. All output is
// routed through the context stdout, `translate!` strings are literalized, and
// the entry point no longer calls `std::process::exit`.
use std::{ffi::OsString, io::Write, thread};
use clap::{Arg, ArgAction, ArgMatches, Command};
use pi_uutils_ctx::format_usage;
use uucore::{
display::Quotable,
error::{UResult, USimpleError},
};
static OPT_ALL: &str = "all";
static OPT_IGNORE: &str = "ignore";
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
/// arguments directly (without the uucore clap-localization helper that would
/// terminate the process), renders clap help/usage/version to the context
/// streams, and maps the `UResult` to an exit code, so it is safe to run inside
/// the host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
match nproc_main(&matches) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
let msg = err.to_string();
if !msg.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "nproc: {msg}");
}
if code == 0 { 1 } else { code }
},
}
}
fn nproc_main(matches: &ArgMatches) -> UResult<()> {
let ignore = match matches.get_one::<String>(OPT_IGNORE) {
Some(numstr) => match numstr.trim().parse::<usize>() {
Ok(num) => num,
Err(e) => {
return Err(USimpleError::new(
1,
// pi-uutils: literalized translate!("nproc-error-invalid-number")
format!("{} is not a valid number: {e}", numstr.quote()),
));
},
},
None => 0,
};
// pi-uutils: OMP_THREAD_LIMIT comes from the scope environment (the
// shell's exported variables), not the host process environment.
let limit = match pi_uutils_ctx::var("OMP_THREAD_LIMIT") {
// Uses the OpenMP variable to limit the number of threads
// If the parsing fails, returns the max size (so, no impact)
// If OMP_THREAD_LIMIT=0, rejects the value
Some(threads) => match threads.parse() {
Ok(0) | Err(_) => usize::MAX,
Ok(n) => n,
},
// the variable 'OMP_THREAD_LIMIT' doesn't exist
// fallback to the max
None => usize::MAX,
};
let mut cores = if matches.get_flag(OPT_ALL) {
num_cpus_all()
} else {
// OMP_NUM_THREADS doesn't have an impact on --all
// pi-uutils: OMP_NUM_THREADS comes from the scope environment.
match pi_uutils_ctx::var("OMP_NUM_THREADS") {
// Uses the OpenMP variable to force the number of threads
// If the parsing fails, returns the number of CPU
Some(threads) => {
// In some cases, OMP_NUM_THREADS can be "x,y,z"
// In this case, only take the first one (like GNU)
// If OMP_NUM_THREADS=0, rejects the value
match threads.split_terminator(',').next() {
None => available_parallelism(),
Some(s) => match s.trim().parse() {
Ok(0) | Err(_) => available_parallelism(),
Ok(n) => n,
},
}
},
// the variable 'OMP_NUM_THREADS' doesn't exist
// fallback to the regular CPU detection
None => available_parallelism(),
}
};
cores = std::cmp::min(limit, cores);
if cores <= ignore {
cores = 1;
} else {
cores -= ignore;
}
// pi-uutils: write to the context stdout instead of the process stdout.
pi_uutils_ctx::stdout()
.write_all(format!("{cores}\n").as_bytes())
.map_err(|e| USimpleError::new(1, e.to_string()))?;
Ok(())
}
pub fn uu_app() -> Command {
Command::new("nproc")
.version(uucore::crate_version!())
.about(
"Print the number of cores available to the current process.\nIf the OMP_NUM_THREADS or \
OMP_THREAD_LIMIT environment variables are set, then\nthey will determine the minimum \
and maximum returned value respectively.",
)
.override_usage(format_usage("nproc [OPTIONS]..."))
.infer_long_args(true)
.arg(
Arg::new(OPT_ALL)
.long(OPT_ALL)
.help("print the number of cores available to the system")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_IGNORE)
.long(OPT_IGNORE)
.value_name("N")
.help("ignore up to N cores"),
)
}
#[cfg(unix)]
fn num_cpus_all() -> usize {
// In some situation, /proc and /sys are not mounted, and sysconf returns 1.
// However, we want to guarantee that `nproc --all` >= `nproc`.
unsafe { libc::sysconf(libc::_SC_NPROCESSORS_CONF) }
.try_into()
.ok()
.filter(|&n: &isize| n > 1)
.map_or_else(available_parallelism, |n| n as usize)
}
// Other platforms (e.g., windows), available_parallelism() directly.
#[cfg(not(unix))]
fn num_cpus_all() -> usize {
available_parallelism()
}
/// In some cases, [`thread::available_parallelism`]() may return an Err
/// In this case, we will return 1 (like GNU)
fn available_parallelism() -> usize {
thread::available_parallelism().map_or(1, std::num::NonZeroUsize::get)
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc};
use parking_lot::Mutex;
use pi_uutils_ctx::ScopeIo;
use super::*;
fn run_in(env: HashMap<String, String>, args: Vec<&str>) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
self.buf.lock().write(buf)
}
fn flush(&mut self) -> std::io::Result<()> {
self.buf.lock().flush()
}
}
let io = ScopeIo {
stdin: Box::new(std::io::empty()),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd: PathBuf::from("."),
env,
cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)),
};
let argv: Vec<OsString> = std::iter::once("nproc")
.chain(args)
.map(OsString::from)
.collect();
let code = pi_uutils_ctx::scope(io, || run(argv));
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
(code, out_str, err_str)
}
#[test]
fn scope_env_omp_num_threads_forces_count() {
let env = HashMap::from([("OMP_NUM_THREADS".to_string(), "3".to_string())]);
let (code, stdout, stderr) = run_in(env, vec![]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "3\n", ""));
}
#[test]
fn omp_thread_limit_caps_omp_num_threads() {
let env = HashMap::from([
("OMP_NUM_THREADS".to_string(), "64".to_string()),
("OMP_THREAD_LIMIT".to_string(), "2".to_string()),
]);
let (code, stdout, stderr) = run_in(env, vec![]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "2\n", ""));
}
#[test]
fn all_prints_positive_integer_and_ignores_omp_num_threads() {
// --all reports hardware CPUs; OMP_NUM_THREADS must not force it.
let env = HashMap::from([("OMP_NUM_THREADS".to_string(), "0".to_string())]);
let (code, stdout, stderr) = run_in(env, vec!["--all"]);
assert_eq!(code, 0);
assert_eq!(stderr, "");
let n: usize = stdout
.trim_end()
.parse()
.expect("--all output is an integer");
assert!(n >= 1);
}
#[test]
fn process_environment_is_not_consulted() {
// The variable exists only in the host process environment, not the
// scope map: only the un-patched `std::env::var` path would see it.
unsafe { std::env::set_var("OMP_NUM_THREADS", "1234") };
let (code, stdout, stderr) = run_in(HashMap::new(), vec![]);
unsafe { std::env::remove_var("OMP_NUM_THREADS") };
assert_eq!((code, stderr.as_str()), (0, ""));
assert_ne!(stdout, "1234\n");
let n: usize = stdout.trim_end().parse().expect("output is an integer");
assert!(n >= 1);
}
#[test]
fn ignore_subtracts_and_floors_at_one() {
let env = HashMap::from([("OMP_NUM_THREADS".to_string(), "8".to_string())]);
let (code, stdout, _) = run_in(env, vec!["--ignore=3"]);
assert_eq!((code, stdout.as_str()), (0, "5\n"));
let env = HashMap::from([("OMP_NUM_THREADS".to_string(), "2".to_string())]);
let (code, stdout, _) = run_in(env, vec!["--ignore=5"]);
assert_eq!((code, stdout.as_str()), (0, "1\n"));
}
#[test]
fn invalid_ignore_value_is_an_error() {
let (code, stdout, stderr) = run_in(HashMap::new(), vec!["--ignore=bogus"]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert!(stderr.contains("is not a valid number"), "stderr: {stderr}");
}
}
+21
View File
@@ -0,0 +1,21 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/printenv), patched to read
# the environment from the shell scope and route I/O through pi-uutils-ctx so
# it can run in-process as a shell builtin. See src/printenv.rs for the patch
# markers (`pi-uutils:` comments).
[package]
name = "uu_printenv"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "printenv ~ (uutils) display value of environment VAR (vendored + patched for in-process embedding)"
[lib]
path = "src/printenv.rs"
[dependencies]
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
uucore = { version = "0.8.0" }
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[dev-dependencies]
parking_lot = "0.12"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+243
View File
@@ -0,0 +1,243 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process
// as a shell builtin. The environment comes from the SCOPE, not the process:
// the no-argument dump iterates `pi_uutils_ctx::env_snapshot()` and named
// lookups go through `pi_uutils_ctx::var`, because the embedding shell's
// exported variables are not present in the host process environment. All
// output is routed through the context stdout, `translate!` strings are
// literalized, and the entry point no longer calls `std::process::exit`.
use std::{ffi::OsString, io::Write};
use clap::{Arg, ArgAction, ArgMatches, Command};
use pi_uutils_ctx::format_usage;
use uucore::{error::UResult, line_ending::LineEnding};
static OPT_NULL: &str = "null";
static ARG_VARIABLES: &str = "variables";
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
/// arguments directly (without the uucore clap-localization helper that would
/// terminate the process), renders clap help/usage/version to the context
/// streams, and maps the `UResult` to an exit code, so it is safe to run inside
/// the host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
match printenv_main(&matches) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
// pi-uutils: unset-variable failures surface as bare exit-code
// errors that render to an empty message (upstream prints nothing
// for them); don't emit a dangling "printenv: " prefix.
let msg = err.to_string();
if !msg.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "printenv: {msg}");
}
if code == 0 { 1 } else { code }
},
}
}
fn printenv_main(matches: &ArgMatches) -> UResult<()> {
let variables: Vec<String> = matches
.get_many::<String>(ARG_VARIABLES)
.map(|v| v.map(ToString::to_string).collect())
.unwrap_or_default();
let separator = LineEnding::from_zero_flag(matches.get_flag(OPT_NULL));
if variables.is_empty() {
// pi-uutils: replacement for `uucore::display::print_all_env_vars` —
// dumps the scope environment map to the context stdout.
let mut stdout = pi_uutils_ctx::stdout();
for (key, value) in pi_uutils_ctx::env_snapshot() {
write!(stdout, "{key}={value}{separator}")?;
}
stdout.flush()?;
return Ok(());
}
let mut error_found = false;
for env_var in variables {
// we silently ignore a=b as variable but we trigger an error
if env_var.contains('=') {
error_found = true;
continue;
}
// pi-uutils: look the variable up in the scope environment (upstream
// uses `std::env::var_os`) and write it to the context stdout.
if let Some(var) = pi_uutils_ctx::var(&env_var) {
let mut stdout = pi_uutils_ctx::stdout();
write!(stdout, "{var}{separator}")?;
stdout.flush()?;
} else {
error_found = true;
}
}
if error_found { Err(1.into()) } else { Ok(()) }
}
pub fn uu_app() -> Command {
Command::new("printenv")
.version(uucore::crate_version!())
.about(
"Display the values of the specified environment VARIABLE(s), or (with no VARIABLE) \
display name and value pairs for them all.",
)
.override_usage(format_usage("printenv [OPTION]... [VARIABLE]..."))
.infer_long_args(true)
.arg(
Arg::new(OPT_NULL)
.short('0')
.long(OPT_NULL)
.help("end each output line with 0 byte rather than newline")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(ARG_VARIABLES)
.action(ArgAction::Append)
.num_args(1..),
)
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, io::Write, sync::Arc};
use parking_lot::Mutex;
use pi_uutils_ctx::ScopeIo;
use super::*;
fn run_with_env(env: HashMap<String, String>, args: Vec<&str>) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
self.buf.lock().write(buf)
}
fn flush(&mut self) -> std::io::Result<()> {
self.buf.lock().flush()
}
}
let io = ScopeIo {
stdin: Box::new(std::io::empty()),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd: std::path::PathBuf::from("."),
env,
cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)),
};
let argv: Vec<OsString> = std::iter::once("printenv")
.chain(args)
.map(OsString::from)
.collect();
let code = pi_uutils_ctx::scope(io, || run(argv));
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
(code, out_str, err_str)
}
fn scope_env() -> HashMap<String, String> {
HashMap::from([
("FOO".to_string(), "bar".to_string()),
("BAZ".to_string(), "qux".to_string()),
])
}
#[test]
fn named_variable_prints_scope_value() {
let (code, stdout, stderr) = run_with_env(scope_env(), vec!["FOO"]);
assert_eq!(code, 0);
assert_eq!(stdout, "bar\n");
assert_eq!(stderr, "");
}
#[test]
fn unset_variable_is_silent_failure() {
let (code, stdout, stderr) = run_with_env(scope_env(), vec!["NOPE"]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert_eq!(stderr, "", "unset variables fail without a message");
}
#[test]
fn mixed_set_and_unset_prints_set_ones_and_fails() {
let (code, stdout, stderr) = run_with_env(scope_env(), vec!["FOO", "NOPE", "BAZ"]);
assert_eq!(code, 1);
assert_eq!(stdout, "bar\nqux\n");
assert_eq!(stderr, "");
}
#[test]
fn no_args_dumps_scope_env_not_process_env() {
// The host process certainly has PATH set; the scope env deliberately
// does not, so its absence proves the dump reads the scope map.
assert!(std::env::var_os("PATH").is_some());
let (code, stdout, stderr) = run_with_env(scope_env(), vec![]);
assert_eq!(code, 0);
assert_eq!(stderr, "");
let lines: Vec<&str> = stdout.lines().collect();
assert_eq!(lines.len(), 2);
assert!(lines.contains(&"FOO=bar"));
assert!(lines.contains(&"BAZ=qux"));
assert!(!lines.iter().any(|l| l.starts_with("PATH=")));
}
#[test]
fn null_flag_terminates_with_nul() {
let (code, stdout, _) = run_with_env(scope_env(), vec!["-0", "FOO"]);
assert_eq!((code, stdout.as_str()), (0, "bar\0"));
let (code, stdout, _) = run_with_env(scope_env(), vec!["--null", "FOO", "BAZ"]);
assert_eq!((code, stdout.as_str()), (0, "bar\0qux\0"));
}
#[test]
fn name_containing_equals_is_ignored_but_fails() {
let (code, stdout, stderr) = run_with_env(scope_env(), vec!["FOO=bar", "BAZ"]);
assert_eq!(code, 1);
assert_eq!(stdout, "qux\n");
assert_eq!(stderr, "");
}
#[test]
fn help_renders_to_scope_stdout() {
let (code, stdout, stderr) = run_with_env(HashMap::new(), vec!["--help"]);
assert_eq!(code, 0);
assert!(stdout.contains("Usage:"));
assert!(stdout.contains("environment VARIABLE"));
assert_eq!(stderr, "");
}
}
+22
View File
@@ -0,0 +1,22 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/readlink), patched to resolve
# path arguments against the shell working directory and route I/O through
# pi-uutils-ctx so it can run in-process as a shell builtin. See src/readlink.rs
# for the patch markers (`pi-uutils:` comments).
[package]
name = "uu_readlink"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "readlink ~ (uutils) display resolved path of PATHNAME (vendored + patched for in-process embedding)"
[lib]
path = "src/readlink.rs"
[dependencies]
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
uucore = { version = "0.8.0", features = ["fs"] }
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[dev-dependencies]
parking_lot = "0.12"
tempfile = "3"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+422
View File
@@ -0,0 +1,422 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore (ToDO) errno
// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process
// as a shell builtin. Every filesystem syscall resolves its path operand
// against the shell working directory via `pi_uutils_ctx::resolve` AT THE CALL
// SITE, while the original operands are kept for display/error messages (GNU
// prints operands as typed). All process-global stdio is routed through
// `pi_uutils_ctx`, `translate!` strings are literalized, POSIXLY_CORRECT is
// read from the scope environment, and the entry point no longer calls
// `std::process::exit`.
use std::{
ffi::OsString,
fs,
io::Write,
path::{Path, PathBuf},
};
use clap::{Arg, ArgAction, ArgMatches, Command};
use pi_uutils_ctx::format_usage;
use uucore::{
display::Quotable,
error::{FromIo, UResult, UUsageError},
fs::{MissingHandling, ResolveMode, canonicalize},
libc::EINVAL,
line_ending::LineEnding,
};
const OPT_CANONICALIZE: &str = "canonicalize";
const OPT_CANONICALIZE_MISSING: &str = "canonicalize-missing";
const OPT_CANONICALIZE_EXISTING: &str = "canonicalize-existing";
const OPT_NO_NEWLINE: &str = "no-newline";
const OPT_QUIET: &str = "quiet";
const OPT_SILENT: &str = "silent";
const OPT_VERBOSE: &str = "verbose";
const OPT_ZERO: &str = "zero";
const ARG_FILES: &str = "files";
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
/// arguments directly (without the uucore clap-localization helper that would
/// terminate the process), renders clap help/usage/version to the context
/// streams, and maps the `UResult` to an exit code, so it is safe to run inside
/// the host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
match readlink_main(&matches) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
// pi-uutils: silent failures surface as bare exit-code errors that
// render to an empty message (upstream prints nothing for them);
// don't emit a dangling "readlink: " prefix.
let msg = err.to_string();
if !msg.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "readlink: {msg}");
}
if code == 0 { 1 } else { code }
},
}
}
fn readlink_main(matches: &ArgMatches) -> UResult<()> {
let mut no_trailing_delimiter = matches.get_flag(OPT_NO_NEWLINE);
let use_zero = matches.get_flag(OPT_ZERO);
// pi-uutils: POSIXLY_CORRECT comes from the scope environment (the shell's
// exported variables), not the host process environment.
let verbose = matches.get_flag(OPT_VERBOSE) || pi_uutils_ctx::var("POSIXLY_CORRECT").is_some();
// GNU readlink -f/-e/-m follows symlinks first and then applies `..` (physical
// resolution). ResolveMode::Logical collapses `..` before following links,
// which yields the opposite order, so we choose Physical here for GNU
// compatibility.
let res_mode = if matches.get_flag(OPT_CANONICALIZE)
|| matches.get_flag(OPT_CANONICALIZE_EXISTING)
|| matches.get_flag(OPT_CANONICALIZE_MISSING)
{
ResolveMode::Physical
} else {
ResolveMode::None
};
let can_mode = if matches.get_flag(OPT_CANONICALIZE_EXISTING) {
MissingHandling::Existing
} else if matches.get_flag(OPT_CANONICALIZE_MISSING) {
MissingHandling::Missing
} else {
MissingHandling::Normal
};
let files: Vec<PathBuf> = matches
.get_many::<OsString>(ARG_FILES)
.map(|v| v.map(PathBuf::from).collect())
.unwrap_or_default();
if files.is_empty() {
return Err(UUsageError::new(1, "missing operand".to_string()));
}
if no_trailing_delimiter && files.len() > 1 {
let _ = writeln!(
pi_uutils_ctx::stderr(),
"readlink: ignoring --no-newline with multiple arguments"
);
no_trailing_delimiter = false;
}
let line_ending = if no_trailing_delimiter {
None
} else {
Some(LineEnding::from_zero_flag(use_zero))
};
for p in &files {
// pi-uutils: resolve the operand against the shell working directory;
// `p` is kept for display. Resolving before `canonicalize` also keeps
// uucore's internal `env::current_dir()` fallback from being consulted.
let resolved = pi_uutils_ctx::resolve(p);
let path_result = if res_mode == ResolveMode::None {
fs::read_link(&resolved)
} else {
canonicalize(&resolved, can_mode, res_mode)
};
match path_result {
Ok(path) => {
show(&path, line_ending)?;
},
Err(err) => {
if !verbose {
return Err(1.into());
}
let message = if err.raw_os_error() == Some(EINVAL) {
format!("{}: Invalid argument", p.maybe_quote())
} else {
err.map_err_context(|| p.maybe_quote().to_string())
.to_string()
};
let _ = writeln!(pi_uutils_ctx::stderr(), "readlink: {message}");
return Err(1.into());
},
}
}
Ok(())
}
pub fn uu_app() -> Command {
Command::new("readlink")
.version(uucore::crate_version!())
.about("Print value of a symbolic link or canonical file name.")
.override_usage(format_usage("readlink [OPTION]... [FILE]..."))
.infer_long_args(true)
.arg(
Arg::new(OPT_CANONICALIZE)
.short('f')
.long(OPT_CANONICALIZE)
.help(
"canonicalize by following every symlink in every component of the given name \
recursively; all but the last component must exist",
)
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_CANONICALIZE_EXISTING)
.short('e')
.long("canonicalize-existing")
.help(
"canonicalize by following every symlink in every component of the given name \
recursively, all components must exist",
)
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_CANONICALIZE_MISSING)
.short('m')
.long(OPT_CANONICALIZE_MISSING)
.help(
"canonicalize by following every symlink in every component of the given name \
recursively, without requirements on components existence",
)
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_NO_NEWLINE)
.short('n')
.long(OPT_NO_NEWLINE)
.help("do not output the trailing delimiter")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_QUIET)
.short('q')
.long(OPT_QUIET)
.help("suppress most error messages")
.overrides_with_all([OPT_QUIET, OPT_SILENT, OPT_VERBOSE])
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_SILENT)
.short('s')
.long(OPT_SILENT)
.help("suppress most error messages")
.overrides_with_all([OPT_QUIET, OPT_SILENT, OPT_VERBOSE])
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_VERBOSE)
.short('v')
.long(OPT_VERBOSE)
.help("report error message")
.overrides_with_all([OPT_QUIET, OPT_SILENT, OPT_VERBOSE])
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_ZERO)
.short('z')
.long(OPT_ZERO)
.help("separate output with NUL rather than newline")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(ARG_FILES)
.action(ArgAction::Append)
.value_parser(clap::value_parser!(OsString))
.value_hint(clap::ValueHint::AnyPath),
)
}
/// pi-uutils: replacement for upstream's `show` — writes the resolved path
/// bytes verbatim to the context stdout instead of the process stdout.
fn show(path: &Path, line_ending: Option<LineEnding>) -> UResult<()> {
let mut out = pi_uutils_ctx::stdout();
out.write_all(uucore::os_str_as_bytes(path.as_os_str())?)?;
if let Some(line_ending) = line_ending {
write!(out, "{line_ending}")?;
}
out.flush()?;
Ok(())
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc};
use parking_lot::Mutex;
use pi_uutils_ctx::ScopeIo;
use super::*;
fn run_in(cwd: PathBuf, args: Vec<&str>) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
self.buf.lock().write(buf)
}
fn flush(&mut self) -> std::io::Result<()> {
self.buf.lock().flush()
}
}
let io = ScopeIo {
stdin: Box::new(std::io::empty()),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd,
env: HashMap::new(),
cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)),
};
let argv: Vec<OsString> = std::iter::once("readlink")
.chain(args)
.map(OsString::from)
.collect();
let code = pi_uutils_ctx::scope(io, || run(argv));
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
(code, out_str, err_str)
}
/// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var,
/// which -f/-e/-m resolution would otherwise expand mid-assertion).
fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) {
let dir = tempfile::tempdir().unwrap();
let canon = fs::canonicalize(dir.path()).unwrap();
(dir, canon)
}
#[cfg(unix)]
#[test]
fn resolves_relative_operand_against_scope_cwd() {
let (_dir, root) = canonical_tempdir();
std::os::unix::fs::symlink("target-file", root.join("link")).unwrap();
// Relative operand + scope cwd differing from the process cwd: only the
// call-site `pi_uutils_ctx::resolve` patch makes this find the link.
let (code, stdout, stderr) = run_in(root, vec!["link"]);
assert_eq!(code, 0);
assert_eq!(stdout, "target-file\n");
assert_eq!(stderr, "");
}
#[cfg(unix)]
#[test]
fn canonicalize_follows_symlink_to_absolute_path() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("target"), b"x").unwrap();
std::os::unix::fs::symlink("target", root.join("link")).unwrap();
let (code, stdout, stderr) = run_in(root.clone(), vec!["-f", "link"]);
assert_eq!(code, 0);
assert_eq!(stdout, format!("{}\n", root.join("target").display()));
assert_eq!(stderr, "");
}
#[test]
fn canonicalize_missing_builds_path_from_scope_cwd() {
let (_dir, root) = canonical_tempdir();
let (code, stdout, stderr) = run_in(root.clone(), vec!["-m", "missing/sub"]);
assert_eq!(code, 0);
assert_eq!(stdout, format!("{}\n", root.join("missing").join("sub").display()));
assert_eq!(stderr, "");
}
#[cfg(unix)]
#[test]
fn canonicalize_existing_fails_silently_on_missing_final_component() {
let (_dir, root) = canonical_tempdir();
let (code, stdout, stderr) = run_in(root, vec!["-e", "missing"]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert_eq!(stderr, "", "non-verbose failures print nothing");
}
#[cfg(unix)]
#[test]
fn non_symlink_is_silent_failure_by_default_and_einval_with_verbose() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("plain"), b"x").unwrap();
let (code, stdout, stderr) = run_in(root.clone(), vec!["plain"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (1, "", ""));
let (code, stdout, stderr) = run_in(root, vec!["-v", "plain"]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert_eq!(stderr, "readlink: plain: Invalid argument\n");
}
#[cfg(unix)]
#[test]
fn no_newline_with_multiple_args_warns_and_keeps_delimiter() {
let (_dir, root) = canonical_tempdir();
std::os::unix::fs::symlink("a", root.join("l1")).unwrap();
std::os::unix::fs::symlink("b", root.join("l2")).unwrap();
let (code, stdout, stderr) = run_in(root, vec!["-n", "l1", "l2"]);
assert_eq!(code, 0);
assert_eq!(stdout, "a\nb\n");
assert_eq!(stderr, "readlink: ignoring --no-newline with multiple arguments\n");
}
#[cfg(unix)]
#[test]
fn zero_terminates_with_nul_and_no_newline_drops_delimiter() {
let (_dir, root) = canonical_tempdir();
std::os::unix::fs::symlink("a", root.join("l1")).unwrap();
let (code, stdout, _) = run_in(root.clone(), vec!["-z", "l1"]);
assert_eq!((code, stdout.as_str()), (0, "a\0"));
let (code, stdout, _) = run_in(root, vec!["-n", "l1"]);
assert_eq!((code, stdout.as_str()), (0, "a"));
}
#[test]
fn missing_operand_is_usage_error() {
let (code, stdout, stderr) = run_in(PathBuf::from("."), vec![]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert!(stderr.contains("missing operand"));
}
#[test]
fn help_renders_to_scope_stdout() {
let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["--help"]);
assert_eq!(code, 0);
assert!(stdout.contains("Usage:"));
assert!(stdout.contains("canonical file name"));
assert_eq!(stderr, "");
}
}
+22
View File
@@ -0,0 +1,22 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/realpath), patched to resolve
# path arguments against the shell working directory and route I/O through
# pi-uutils-ctx so it can run in-process as a shell builtin. See src/realpath.rs
# for the patch markers (`pi-uutils:` comments).
[package]
name = "uu_realpath"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "realpath ~ (uutils) display resolved absolute path of PATHNAME (vendored + patched for in-process embedding)"
[lib]
path = "src/realpath.rs"
[dependencies]
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
uucore = { version = "0.8.0", features = ["fs"] }
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[dev-dependencies]
parking_lot = "0.12"
tempfile = "3"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+568
View File
@@ -0,0 +1,568 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore (ToDO) retcode
// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process
// as a shell builtin. Every filesystem syscall resolves its path operand
// against the shell working directory via `pi_uutils_ctx::resolve` AT THE CALL
// SITE (FILE operands and the --relative-to/--relative-base option paths),
// while the original operands are kept for display/error messages (GNU prints
// operands as typed). All process-global stdio is routed through
// `pi_uutils_ctx`, `translate!` strings are literalized, `show_if_err!` is
// replaced by a context-stderr write plus `pi_uutils_ctx::set_exit_code`, and
// the entry point no longer calls `std::process::exit`.
use std::{
ffi::{OsStr, OsString},
io::Write,
path::{Path, PathBuf},
};
use clap::{
Arg, ArgAction, ArgMatches, Command,
builder::{TypedValueParser, ValueParserFactory},
};
use pi_uutils_ctx::format_usage;
use uucore::{
display::Quotable,
error::{FromIo, UResult},
fs::{MissingHandling, ResolveMode, canonicalize, make_path_relative_to},
line_ending::LineEnding,
};
const OPT_QUIET: &str = "quiet";
const OPT_STRIP: &str = "strip";
const OPT_ZERO: &str = "zero";
const OPT_PHYSICAL: &str = "physical";
const OPT_LOGICAL: &str = "logical";
const OPT_CANONICALIZE_MISSING: &str = "canonicalize-missing";
const OPT_CANONICALIZE: &str = "canonicalize";
const OPT_CANONICALIZE_EXISTING: &str = "canonicalize-existing";
const OPT_RELATIVE_TO: &str = "relative-to";
const OPT_RELATIVE_BASE: &str = "relative-base";
const ARG_FILES: &str = "files";
/// Custom parser that validates `OsString` is not empty
#[derive(Clone, Debug)]
struct NonEmptyOsStringParser;
impl TypedValueParser for NonEmptyOsStringParser {
type Value = OsString;
fn parse_ref(
&self,
_cmd: &Command,
_arg: Option<&Arg>,
value: &OsStr,
) -> Result<Self::Value, clap::Error> {
if value.is_empty() {
let mut err = clap::Error::new(clap::error::ErrorKind::ValueValidation);
err.insert(
clap::error::ContextKind::Custom,
// pi-uutils: literalized `translate!("realpath-invalid-empty-operand")`
clap::error::ContextValue::String("invalid operand: empty string".to_string()),
);
return Err(err);
}
Ok(value.to_os_string())
}
}
impl ValueParserFactory for NonEmptyOsStringParser {
type Parser = Self;
fn value_parser() -> Self::Parser {
Self
}
}
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
/// arguments directly (without the uucore clap-localization helper that would
/// terminate the process), renders clap help/usage/version to the context
/// streams, and maps the `UResult` to an exit code, so it is safe to run inside
/// the host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
match realpath_main(&matches) {
// pi-uutils: per-file failures accumulate their exit code via
// `pi_uutils_ctx::set_exit_code` (upstream's `show!` machinery).
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
let msg = err.to_string();
if !msg.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "realpath: {msg}");
}
if code == 0 { 1 } else { code }
},
}
}
fn realpath_main(matches: &ArgMatches) -> UResult<()> {
/* the list of files */
let paths: Vec<PathBuf> = matches
.get_many::<OsString>(ARG_FILES)
.unwrap()
.map(PathBuf::from)
.collect();
let strip = matches.get_flag(OPT_STRIP);
let line_ending = LineEnding::from_zero_flag(matches.get_flag(OPT_ZERO));
let quiet = matches.get_flag(OPT_QUIET);
let logical = matches.get_flag(OPT_LOGICAL);
let can_mode = if matches.get_flag(OPT_CANONICALIZE_MISSING) {
MissingHandling::Missing
} else if matches.get_flag(OPT_CANONICALIZE_EXISTING) {
// -e: all components must exist
// Despite the name, MissingHandling::Existing requires all components to exist
MissingHandling::Existing
} else {
// Default behavior (same as -E): all but last component must exist
// MissingHandling::Normal allows the final component to not exist
MissingHandling::Normal
};
let resolve_mode = if strip {
ResolveMode::None
} else if logical {
ResolveMode::Logical
} else {
ResolveMode::Physical
};
let (relative_to, relative_base) = prepare_relative_options(matches, can_mode, resolve_mode)?;
for path in &paths {
let result = resolve_path(
path,
line_ending,
resolve_mode,
can_mode,
relative_to.as_deref(),
relative_base.as_deref(),
);
if !quiet {
// pi-uutils: replacement for `show_if_err!` — report the error on
// the context stderr and record the exit code, then keep
// processing the remaining operands (upstream continue semantics).
if let Err(err) = result.map_err_context(|| path.maybe_quote().to_string()) {
let _ = writeln!(pi_uutils_ctx::stderr(), "realpath: {err}");
pi_uutils_ctx::set_exit_code(err.code());
}
}
}
// Although we return `Ok`, it is possible that a call to
// `show!()` above has set the exit code for the program to a
// non-zero integer.
Ok(())
}
pub fn uu_app() -> Command {
Command::new("realpath")
.version(uucore::crate_version!())
.about("Print the resolved path")
.override_usage(format_usage("realpath [OPTION]... FILE..."))
.infer_long_args(true)
.arg(
Arg::new(OPT_QUIET)
.short('q')
.long(OPT_QUIET)
.help("Do not print warnings for invalid paths")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_STRIP)
.short('s')
.long(OPT_STRIP)
.visible_alias("no-symlinks")
.help("Only strip '.' and '..' components, but don't resolve symbolic links")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_ZERO)
.short('z')
.long(OPT_ZERO)
.help("Separate output filenames with \\0 rather than newline")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_LOGICAL)
.short('L')
.long(OPT_LOGICAL)
.help("resolve '..' components before symlinks")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_PHYSICAL)
.short('P')
.long(OPT_PHYSICAL)
.overrides_with_all([OPT_STRIP, OPT_LOGICAL])
.help("resolve symlinks as encountered (default)")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_CANONICALIZE)
.short('E')
.long(OPT_CANONICALIZE)
.overrides_with_all([OPT_CANONICALIZE_EXISTING, OPT_CANONICALIZE_MISSING])
.help("all but the last component must exist (default)")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_CANONICALIZE_EXISTING)
.short('e')
.long(OPT_CANONICALIZE_EXISTING)
.overrides_with_all([OPT_CANONICALIZE, OPT_CANONICALIZE_MISSING])
.help(
"canonicalize by following every symlink in every component of the given name \
recursively, all components must exist",
)
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_CANONICALIZE_MISSING)
.short('m')
.long(OPT_CANONICALIZE_MISSING)
.overrides_with_all([OPT_CANONICALIZE, OPT_CANONICALIZE_EXISTING])
.help(
"canonicalize by following every symlink in every component of the given name \
recursively, without requirements on components existence",
)
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_RELATIVE_TO)
.long(OPT_RELATIVE_TO)
.value_name("DIR")
.value_parser(NonEmptyOsStringParser)
.help("print the resolved path relative to DIR"),
)
.arg(
Arg::new(OPT_RELATIVE_BASE)
.long(OPT_RELATIVE_BASE)
.value_name("DIR")
.value_parser(NonEmptyOsStringParser)
.help("print absolute paths unless paths below DIR"),
)
.arg(
Arg::new(ARG_FILES)
.action(ArgAction::Append)
.required(true)
.value_parser(NonEmptyOsStringParser)
.value_hint(clap::ValueHint::AnyPath),
)
}
/// Prepare `--relative-to` and `--relative-base` options.
/// Convert them to their absolute values.
/// Check if `--relative-to` is a descendant of `--relative-base`,
/// otherwise nullify their value.
fn prepare_relative_options(
matches: &ArgMatches,
can_mode: MissingHandling,
resolve_mode: ResolveMode,
) -> UResult<(Option<PathBuf>, Option<PathBuf>)> {
let relative_to = matches
.get_one::<OsString>(OPT_RELATIVE_TO)
.map(PathBuf::from);
let relative_base = matches
.get_one::<OsString>(OPT_RELATIVE_BASE)
.map(PathBuf::from);
let relative_to = canonicalize_relative_option(relative_to, can_mode, resolve_mode)?;
let relative_base = canonicalize_relative_option(relative_base, can_mode, resolve_mode)?;
if let (Some(base), Some(to)) = (relative_base.as_deref(), relative_to.as_deref())
&& !to.starts_with(base)
{
return Ok((None, None));
}
Ok((relative_to, relative_base))
}
/// Prepare single `relative-*` option.
fn canonicalize_relative_option(
relative: Option<PathBuf>,
can_mode: MissingHandling,
resolve_mode: ResolveMode,
) -> UResult<Option<PathBuf>> {
Ok(match relative {
None => None,
Some(p) => Some(
canonicalize_relative(&p, can_mode, resolve_mode)
.map_err_context(|| p.maybe_quote().to_string())?,
),
})
}
/// Make `relative-to` or `relative-base` path values absolute.
///
/// # Errors
///
/// If the given path is not a directory the function returns an error.
/// If some parts of the file don't exist, or symlinks make loops, or
/// some other IO error happens, the function returns error, too.
fn canonicalize_relative(
r: &Path,
can_mode: MissingHandling,
resolve: ResolveMode,
) -> std::io::Result<PathBuf> {
// pi-uutils: resolve the option path against the shell working directory;
// `r` is kept by the caller for display. Resolving before `canonicalize`
// also keeps uucore's internal `env::current_dir()` fallback from being
// consulted.
let abs = canonicalize(pi_uutils_ctx::resolve(r), can_mode, resolve)?;
if can_mode == MissingHandling::Existing && !abs.is_dir() {
abs.read_dir()?; // raise not a directory error
}
Ok(abs)
}
/// Resolve a path to an absolute form and print it.
///
/// If `relative_to` and/or `relative_base` is given
/// the path is printed in a relative form to one of this options.
/// See the details in `process_relative` function.
/// If `zero` is `true`, then this function
/// prints the path followed by the null byte (`'\0'`) instead of a
/// newline character (`'\n'`).
///
/// # Errors
///
/// This function returns an error if there is a problem resolving
/// symbolic links.
fn resolve_path(
p: &Path,
line_ending: LineEnding,
resolve: ResolveMode,
can_mode: MissingHandling,
relative_to: Option<&Path>,
relative_base: Option<&Path>,
) -> std::io::Result<()> {
// pi-uutils: resolve the operand against the shell working directory; `p`
// is kept by the caller for display. Resolving before `canonicalize` also
// keeps uucore's internal `env::current_dir()` fallback from being
// consulted.
let abs = canonicalize(pi_uutils_ctx::resolve(p), can_mode, resolve)?;
let abs = process_relative(abs, relative_base, relative_to);
// pi-uutils: replacement for `print_verbatim` + process stdout — writes
// the resolved path bytes verbatim to the context stdout.
let mut out = pi_uutils_ctx::stdout();
out.write_all(
uucore::os_str_as_bytes(abs.as_os_str()).map_err(|e| std::io::Error::other(e.to_string()))?,
)?;
out.write_all(&[line_ending.into()])?;
out.flush()?;
Ok(())
}
/// Conditionally converts an absolute path to a relative form,
/// according to the rules:
/// 1. if only `relative_to` is given, the result is relative to `relative_to`
/// 2. if only `relative_base` is given, it checks whether given `path` is a
/// descendant of `relative_base`, on success the result is relative to
/// `relative_base`, otherwise the result is the given `path`
/// 3. if both `relative_to` and `relative_base` are given, the result is
/// relative to `relative_to` if `path` is a descendant of `relative_base`,
/// otherwise the result is `path`
///
/// For more information see
/// <https://www.gnu.org/software/coreutils/manual/html_node/Realpath-usage-examples.html>
fn process_relative(
path: PathBuf,
relative_base: Option<&Path>,
relative_to: Option<&Path>,
) -> PathBuf {
if let Some(base) = relative_base {
if path.starts_with(base) {
make_path_relative_to(path, relative_to.unwrap_or(base))
} else {
path
}
} else if let Some(to) = relative_to {
make_path_relative_to(path, to)
} else {
path
}
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, fs, io::Write, path::PathBuf, sync::Arc};
use parking_lot::Mutex;
use pi_uutils_ctx::ScopeIo;
use super::*;
fn run_in(cwd: PathBuf, args: Vec<&str>) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
self.buf.lock().write(buf)
}
fn flush(&mut self) -> std::io::Result<()> {
self.buf.lock().flush()
}
}
let io = ScopeIo {
stdin: Box::new(std::io::empty()),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd,
env: HashMap::new(),
cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)),
};
let argv: Vec<OsString> = std::iter::once("realpath")
.chain(args)
.map(OsString::from)
.collect();
let code = pi_uutils_ctx::scope(io, || run(argv));
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
(code, out_str, err_str)
}
/// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var,
/// which canonicalization would otherwise expand mid-assertion).
fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) {
let dir = tempfile::tempdir().unwrap();
let canon = fs::canonicalize(dir.path()).unwrap();
(dir, canon)
}
#[cfg(unix)]
#[test]
fn resolves_relative_operand_against_scope_cwd() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("target"), b"x").unwrap();
std::os::unix::fs::symlink("target", root.join("link")).unwrap();
// Relative operand + scope cwd differing from the process cwd: only
// the call-site `pi_uutils_ctx::resolve` patch makes this find the
// symlink and print its canonical target.
let (code, stdout, stderr) = run_in(root.clone(), vec!["link"]);
assert_eq!(code, 0);
assert_eq!(stdout, format!("{}\n", root.join("target").display()));
assert_eq!(stderr, "");
}
#[test]
fn canonicalize_missing_builds_path_from_scope_cwd() {
let (_dir, root) = canonical_tempdir();
let (code, stdout, stderr) = run_in(root.clone(), vec!["-m", "missing/sub"]);
assert_eq!(code, 0);
assert_eq!(stdout, format!("{}\n", root.join("missing").join("sub").display()));
assert_eq!(stderr, "");
}
#[test]
fn relative_to_option_resolves_against_scope_cwd_and_relativizes_output() {
let (_dir, root) = canonical_tempdir();
fs::create_dir(root.join("sub")).unwrap();
fs::write(root.join("sub").join("file"), b"x").unwrap();
// Both the operand and the (relative) --relative-to directory resolve
// against the scope cwd.
let (code, stdout, stderr) = run_in(root, vec!["--relative-to", "sub", "sub/file"]);
assert_eq!(code, 0);
assert_eq!(stdout, "file\n");
assert_eq!(stderr, "");
}
#[test]
fn zero_flag_terminates_with_nul() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("f"), b"x").unwrap();
let (code, stdout, stderr) = run_in(root.clone(), vec!["-z", "f"]);
assert_eq!(code, 0);
assert_eq!(stdout, format!("{}\0", root.join("f").display()));
assert_eq!(stderr, "");
}
#[test]
fn nonexistent_operand_errors_but_later_operands_still_process() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("f"), b"x").unwrap();
let (code, stdout, stderr) = run_in(root.clone(), vec!["missing/x", "f"]);
assert_eq!(code, 1);
assert_eq!(stdout, format!("{}\n", root.join("f").display()));
assert!(stderr.contains("realpath: missing/x"), "stderr: {stderr}");
assert!(stderr.contains("No such file"), "stderr: {stderr}");
}
#[test]
fn quiet_suppresses_error_messages() {
let (_dir, root) = canonical_tempdir();
// Upstream drops the per-file result entirely under -q (the error is
// neither printed nor accumulated into the exit code).
let (code, stdout, stderr) = run_in(root, vec!["-q", "missing/x"]);
assert_eq!(code, 0);
assert_eq!(stdout, "");
assert_eq!(stderr, "");
}
#[cfg(unix)]
#[test]
fn strip_keeps_symlinks_unresolved() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("target"), b"x").unwrap();
std::os::unix::fs::symlink("target", root.join("link")).unwrap();
let (code, stdout, stderr) = run_in(root.clone(), vec!["-s", "link"]);
assert_eq!(code, 0);
assert_eq!(stdout, format!("{}\n", root.join("link").display()));
assert_eq!(stderr, "");
}
#[test]
fn empty_operand_is_rejected() {
// The NonEmptyOsStringParser turns "" into a clap parse error (rendered
// by clap's default renderer since the uucore localization layer is
// patched out) instead of a filesystem lookup.
let (code, stdout, stderr) = run_in(PathBuf::from("."), vec![""]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert!(stderr.contains("invalid value"), "stderr: {stderr}");
}
#[test]
fn help_renders_to_scope_stdout() {
let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["--help"]);
assert_eq!(code, 0);
assert!(stdout.contains("Usage:"));
assert!(stdout.contains("Print the resolved path"));
assert_eq!(stderr, "");
}
}
+32
View File
@@ -0,0 +1,32 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/seq), patched to route I/O
# through pi-uutils-ctx so it can run in-process as a shell builtin. See
# src/seq.rs for the patch markers (`pi-uutils:` comments).
[package]
name = "uu_seq"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "seq ~ (uutils) display a sequence of numbers (vendored + patched for in-process embedding)"
[lib]
path = "src/seq.rs"
[dependencies]
bigdecimal = "0.4"
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
num-bigint = "0.4"
num-traits = "0.2"
thiserror = "2.0.3"
# pi-uutils: upstream also enables "signals" (SIGPIPE probing) — dropped, the
# in-process builtin has no process-global signal handling.
uucore = { version = "0.8.0", features = [
"extendedbigdecimal",
"fast-inc",
"format",
"parser",
"quoting-style",
] }
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[dev-dependencies]
parking_lot = "0.12"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+57
View File
@@ -0,0 +1,57 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore numberparse
//! Errors returned by seq.
// pi-uutils: `translate!` message lookups are literalized with the en-US
// strings from upstream's locales/en-US.ftl.
use thiserror::Error;
use uucore::{display::Quotable, error::UError};
use crate::numberparse::ParseNumberError;
#[derive(Debug, Error)]
pub enum SeqError {
/// An error parsing the input arguments.
///
/// The parameters are the [`String`] argument as read from the
/// command line and the underlying parsing error itself.
#[error("invalid {} argument: {}", parse_error_type(.1), .0.quote())]
ParseError(String, ParseNumberError),
/// The increment argument was zero, which is not allowed.
///
/// The parameter is the increment argument as a [`String`] as read
/// from the command line.
#[error("invalid Zero increment value: {}", .0.quote())]
ZeroIncrement(String),
/// No arguments were passed to this function, 1 or more is required
#[error("missing operand")]
NoArguments,
/// Both a format and equal width where passed to seq
#[error("format string may not be specified when printing equal width strings")]
FormatAndEqualWidth,
}
fn parse_error_type(e: &ParseNumberError) -> &'static str {
match e {
ParseNumberError::Float => "floating point",
ParseNumberError::Nan => "'not-a-number'",
}
}
impl UError for SeqError {
/// Always return 1.
fn code(&self) -> i32 {
1
}
fn usage(&self) -> bool {
true
}
}
+52
View File
@@ -0,0 +1,52 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore extendedbigdecimal
use num_traits::Zero;
use uucore::extendedbigdecimal::ExtendedBigDecimal;
/// A number with a specified number of integer and fractional digits.
///
/// This struct can be used to represent a number along with information
/// on how many significant digits to use when displaying the number.
/// The [`PreciseNumber::num_integral_digits`] field also includes the width
/// needed to display the "-" character for a negative number.
/// [`PreciseNumber::num_fractional_digits`] provides the number of decimal
/// digits after the decimal point (a.k.a. precision), or None if that number
/// cannot intuitively be obtained (i.e. hexadecimal floats).
/// Note: Those 2 fields should not necessarily be interpreted literally, but as
/// matching GNU `seq` behavior: the exact way of guessing desired precision
/// from user input is a matter of interpretation.
///
/// You can get an instance of this struct by calling [`str::parse`].
#[derive(Debug)]
pub struct PreciseNumber {
pub number: ExtendedBigDecimal,
pub num_integral_digits: usize,
pub num_fractional_digits: Option<usize>,
}
impl PreciseNumber {
// pi-uutils: upstream's unused `new` constructor (only reachable from the
// fuzzing harness) is dropped to keep the vendored crate warning-free.
pub fn one() -> Self {
// We would like to implement `num_traits::One`, but it requires
// a multiplication implementation, and we don't want to
// implement that here.
Self {
number: ExtendedBigDecimal::one(),
num_integral_digits: 1,
num_fractional_digits: Some(0),
}
}
/// Decide whether this number is zero (either positive or negative).
pub fn is_zero(&self) -> bool {
// We would like to implement `num_traits::Zero`, but it
// requires an addition implementation, and we don't want to
// implement that here.
self.number.is_zero()
}
}
+351
View File
@@ -0,0 +1,351 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore extendedbigdecimal bigdecimal numberparse
// hexadecimalfloat
//! Parsing numbers for use in `seq`.
//!
//! This module provides an implementation of [`FromStr`] for the
//! [`PreciseNumber`] struct.
use std::str::FromStr;
use uucore::{
extendedbigdecimal::ExtendedBigDecimal,
parser::num_parser::{ExtendedParser, ExtendedParserError},
};
use crate::number::PreciseNumber;
/// An error returned when parsing a number fails.
#[derive(Debug, PartialEq, Eq)]
pub enum ParseNumberError {
Float,
Nan,
}
/// Compute the number of integral and fractional digits in input string,
/// and wrap the result in a PreciseNumber.
/// We know that the string has already been parsed correctly, so we don't
/// need to be too careful.
fn compute_num_digits(input: &str, ebd: ExtendedBigDecimal) -> PreciseNumber {
let input = input.to_lowercase();
let input = input.trim_start();
// Leading + is ignored for this.
let input = input.strip_prefix('+').unwrap_or(input);
// Integral digits for any hex number is ill-defined (0 is fine as an output)
// Fractional digits for an floating hex number is ill-defined, return None
// as we'll totally ignore that number for precision computations.
// Still return 0 for hex integers though.
if input.starts_with("0x") || input.starts_with("-0x") {
return PreciseNumber {
number: ebd,
num_integral_digits: 0,
num_fractional_digits: if input.contains('.') || input.contains('p') {
None
} else {
Some(0)
},
};
}
// Split the exponent part, if any
let parts: Vec<&str> = input.split('e').collect();
debug_assert!(parts.len() <= 2);
// Count all the digits up to `.`, `-` sign is included.
let (mut int_digits, mut frac_digits) = match parts[0].find('.') {
Some(i) => {
// Cover special case .X and -.X where we behave as if there was a leading 0:
// 0.X, -0.X.
let int_digits = match i {
0 => 1,
1 if parts[0].starts_with('-') => 2,
_ => i,
};
(int_digits, parts[0].len() - i - 1)
},
None => (parts[0].len(), 0),
};
// If there is an exponent, reparse that (yes this is not optimal,
// but we can't necessarily exactly recover that from the parsed number).
if parts.len() == 2 {
let exp = parts[1].parse::<i64>().unwrap_or(0);
// For positive exponents, effectively expand the number. Ignore negative
// exponents. Also ignore overflowed exponents (unwrap_or(0)).
if exp > 0 {
int_digits += exp.try_into().unwrap_or(0);
}
frac_digits = if exp < frac_digits as i64 {
// Subtract from i128 to avoid any overflow
(frac_digits as i128 - exp as i128).try_into().unwrap_or(0)
} else {
0
}
}
PreciseNumber {
number: ebd,
num_integral_digits: int_digits,
num_fractional_digits: Some(frac_digits),
}
}
// Note: We could also have provided an `ExtendedParser` implementation for
// PreciseNumber, but we want a simpler custom error.
impl FromStr for PreciseNumber {
type Err = ParseNumberError;
fn from_str(input: &str) -> Result<Self, Self::Err> {
let ebd = match ExtendedBigDecimal::extended_parse(input) {
Ok(ebd) => match ebd {
// Handle special values
ExtendedBigDecimal::BigDecimal(_) | ExtendedBigDecimal::MinusZero => {
// TODO: GNU `seq` treats small numbers < 1e-4950 as 0, we could do the same
// to avoid printing senselessly small numbers.
ebd
},
ExtendedBigDecimal::Infinity | ExtendedBigDecimal::MinusInfinity => {
return Ok(Self {
number: ebd,
num_integral_digits: 0,
num_fractional_digits: Some(0),
});
},
ExtendedBigDecimal::Nan | ExtendedBigDecimal::MinusNan => {
return Err(ParseNumberError::Nan);
},
},
Err(ExtendedParserError::Underflow(ebd)) => ebd, // Treat underflow as 0
Err(_) => return Err(ParseNumberError::Float),
};
Ok(compute_num_digits(input, ebd))
}
}
#[cfg(test)]
mod tests {
use bigdecimal::BigDecimal;
use uucore::extendedbigdecimal::ExtendedBigDecimal;
use crate::{number::PreciseNumber, numberparse::ParseNumberError};
/// Convenience function for parsing a [`Number`] and unwrapping.
fn parse(s: &str) -> ExtendedBigDecimal {
s.parse::<PreciseNumber>().unwrap().number
}
/// Convenience function for getting the number of integral digits.
fn num_integral_digits(s: &str) -> usize {
s.parse::<PreciseNumber>().unwrap().num_integral_digits
}
/// Convenience function for getting the number of fractional digits.
fn num_fractional_digits(s: &str) -> usize {
s.parse::<PreciseNumber>()
.unwrap()
.num_fractional_digits
.unwrap()
}
/// Convenience function for making sure the number of fractional digits is
/// "None"
fn num_fractional_digits_is_none(s: &str) -> bool {
s.parse::<PreciseNumber>()
.unwrap()
.num_fractional_digits
.is_none()
}
#[test]
fn test_parse_minus_zero_int() {
assert_eq!(parse("-0e0"), ExtendedBigDecimal::MinusZero);
assert_eq!(parse("-0e-0"), ExtendedBigDecimal::MinusZero);
assert_eq!(parse("-0e1"), ExtendedBigDecimal::MinusZero);
assert_eq!(parse("-0e+1"), ExtendedBigDecimal::MinusZero);
assert_eq!(parse("-0.0e1"), ExtendedBigDecimal::MinusZero);
assert_eq!(parse("-0x0"), ExtendedBigDecimal::MinusZero);
}
#[test]
fn test_parse_minus_zero_float() {
assert_eq!(parse("-0.0"), ExtendedBigDecimal::MinusZero);
assert_eq!(parse("-0e-1"), ExtendedBigDecimal::MinusZero);
assert_eq!(parse("-0.0e-1"), ExtendedBigDecimal::MinusZero);
}
#[test]
fn test_parse_big_int() {
assert_eq!(parse("0"), ExtendedBigDecimal::zero());
assert_eq!(parse("0.1e1"), ExtendedBigDecimal::one());
assert_eq!(parse("0.1E1"), ExtendedBigDecimal::one());
assert_eq!(
parse("1.0e1"),
ExtendedBigDecimal::BigDecimal("10".parse::<BigDecimal>().unwrap())
);
}
#[test]
fn test_parse_hexadecimal_big_int() {
assert_eq!(parse("0x0"), ExtendedBigDecimal::zero());
assert_eq!(
parse("0x10"),
ExtendedBigDecimal::BigDecimal("16".parse::<BigDecimal>().unwrap())
);
}
#[test]
fn test_parse_big_decimal() {
assert_eq!(
parse("0.0"),
ExtendedBigDecimal::BigDecimal("0.0".parse::<BigDecimal>().unwrap())
);
assert_eq!(parse(".0"), ExtendedBigDecimal::BigDecimal("0.0".parse::<BigDecimal>().unwrap()));
assert_eq!(
parse("1.0"),
ExtendedBigDecimal::BigDecimal("1.0".parse::<BigDecimal>().unwrap())
);
assert_eq!(
parse("10e-1"),
ExtendedBigDecimal::BigDecimal("1.0".parse::<BigDecimal>().unwrap())
);
assert_eq!(
parse("-1e-3"),
ExtendedBigDecimal::BigDecimal("-0.001".parse::<BigDecimal>().unwrap())
);
}
#[test]
fn test_parse_inf() {
assert_eq!(parse("inf"), ExtendedBigDecimal::Infinity);
assert_eq!(parse("infinity"), ExtendedBigDecimal::Infinity);
assert_eq!(parse("+inf"), ExtendedBigDecimal::Infinity);
assert_eq!(parse("+infinity"), ExtendedBigDecimal::Infinity);
assert_eq!(parse("-inf"), ExtendedBigDecimal::MinusInfinity);
assert_eq!(parse("-infinity"), ExtendedBigDecimal::MinusInfinity);
}
#[test]
fn test_parse_invalid_float() {
assert_eq!("1.2.3".parse::<PreciseNumber>().unwrap_err(), ParseNumberError::Float);
assert_eq!("1e2e3".parse::<PreciseNumber>().unwrap_err(), ParseNumberError::Float);
assert_eq!("1e2.3".parse::<PreciseNumber>().unwrap_err(), ParseNumberError::Float);
assert_eq!("-+-1".parse::<PreciseNumber>().unwrap_err(), ParseNumberError::Float);
}
#[test]
fn test_parse_invalid_hex() {
assert_eq!("0xg".parse::<PreciseNumber>().unwrap_err(), ParseNumberError::Float);
}
#[test]
fn test_parse_invalid_nan() {
assert_eq!("nan".parse::<PreciseNumber>().unwrap_err(), ParseNumberError::Nan);
assert_eq!("NAN".parse::<PreciseNumber>().unwrap_err(), ParseNumberError::Nan);
assert_eq!("NaN".parse::<PreciseNumber>().unwrap_err(), ParseNumberError::Nan);
assert_eq!("nAn".parse::<PreciseNumber>().unwrap_err(), ParseNumberError::Nan);
assert_eq!("-nan".parse::<PreciseNumber>().unwrap_err(), ParseNumberError::Nan);
}
#[test]
#[allow(clippy::cognitive_complexity)]
fn test_num_integral_digits() {
// no decimal, no exponent
assert_eq!(num_integral_digits("123"), 3);
// decimal, no exponent
assert_eq!(num_integral_digits("123.45"), 3);
assert_eq!(num_integral_digits("-0.1"), 2);
assert_eq!(num_integral_digits("-.1"), 2);
// exponent, no decimal
assert_eq!(num_integral_digits("123e4"), 3 + 4);
assert_eq!(num_integral_digits("123e-4"), 3);
assert_eq!(num_integral_digits("-1e-3"), 2);
// decimal and exponent
assert_eq!(num_integral_digits("123.45e6"), 3 + 6);
assert_eq!(num_integral_digits("123.45e-6"), 3);
assert_eq!(num_integral_digits("123.45e-1"), 3);
assert_eq!(num_integral_digits("-0.1e0"), 2);
assert_eq!(num_integral_digits("-0.1e2"), 4);
assert_eq!(num_integral_digits("-.1e0"), 2);
assert_eq!(num_integral_digits("-.1e2"), 4);
assert_eq!(num_integral_digits("-1.e-3"), 2);
assert_eq!(num_integral_digits("-1.0e-4"), 2);
// minus zero int
assert_eq!(num_integral_digits("-0e0"), 2);
assert_eq!(num_integral_digits("-0e-0"), 2);
assert_eq!(num_integral_digits("-0e1"), 3);
assert_eq!(num_integral_digits("-0e+1"), 3);
assert_eq!(num_integral_digits("-0.0e1"), 3);
// minus zero float
assert_eq!(num_integral_digits("-0.0"), 2);
assert_eq!(num_integral_digits("-0e-1"), 2);
assert_eq!(num_integral_digits("-0.0e-1"), 2);
// TODO In GNU `seq`, the `-w` option does not seem to work with
// hexadecimal arguments. In order to match that behavior, we
// report the number of integral digits as zero for hexadecimal
// inputs.
assert_eq!(num_integral_digits("0xff"), 0);
}
#[test]
#[allow(clippy::cognitive_complexity)]
fn test_num_fractional_digits() {
// no decimal, no exponent
assert_eq!(num_fractional_digits("123"), 0);
assert_eq!(num_fractional_digits("0xff"), 0);
// decimal, no exponent
assert_eq!(num_fractional_digits("123.45"), 2);
assert_eq!(num_fractional_digits("-0.1"), 1);
assert_eq!(num_fractional_digits("-.1"), 1);
// exponent, no decimal
assert_eq!(num_fractional_digits("123e4"), 0);
assert_eq!(num_fractional_digits("123e-4"), 4);
assert_eq!(num_fractional_digits("123e-1"), 1);
assert_eq!(num_fractional_digits("-1e-3"), 3);
// decimal and exponent
assert_eq!(num_fractional_digits("123.45e6"), 0);
assert_eq!(num_fractional_digits("123.45e1"), 1);
assert_eq!(num_fractional_digits("123.45e-6"), 8);
assert_eq!(num_fractional_digits("123.45e-1"), 3);
assert_eq!(num_fractional_digits("-0.1e0"), 1);
assert_eq!(num_fractional_digits("-0.1e2"), 0);
assert_eq!(num_fractional_digits("-.1e0"), 1);
assert_eq!(num_fractional_digits("-.1e2"), 0);
assert_eq!(num_fractional_digits("-1.e-3"), 3);
assert_eq!(num_fractional_digits("-1.0e-4"), 5);
// minus zero int
assert_eq!(num_fractional_digits("-0e0"), 0);
assert_eq!(num_fractional_digits("-0e-0"), 0);
assert_eq!(num_fractional_digits("-0e1"), 0);
assert_eq!(num_fractional_digits("-0e+1"), 0);
assert_eq!(num_fractional_digits("-0.0e1"), 0);
// minus zero float
assert_eq!(num_fractional_digits("-0.0"), 1);
assert_eq!(num_fractional_digits("-0e-1"), 1);
assert_eq!(num_fractional_digits("-0.0e-1"), 2);
// Hexadecimal numbers
assert_eq!(num_fractional_digits("0xff"), 0);
assert!(num_fractional_digits_is_none("0xff.1"));
}
#[test]
fn test_parse_min_exponents() {
// Make sure exponents < i64::MIN do not cause errors
assert!("1e-9223372036854775807".parse::<PreciseNumber>().is_ok());
assert!("1e-9223372036854775808".parse::<PreciseNumber>().is_ok());
assert!("1e-92233720368547758080".parse::<PreciseNumber>().is_ok());
}
#[test]
fn test_parse_max_exponents() {
// Make sure exponents much bigger than i64::MAX cause errors
assert!("1e9223372036854775807".parse::<PreciseNumber>().is_ok());
assert!("1e92233720368547758070".parse::<PreciseNumber>().is_err());
}
}
+561
View File
@@ -0,0 +1,561 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore (ToDO) bigdecimal extendedbigdecimal numberparse
// hexadecimalfloat biguint
// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process
// as a shell builtin. seq is pure computation + stdout: all process-global
// stdio is routed through `pi_uutils_ctx` (the emission loops write to a
// `BufWriter` around the context stdout handle and poll
// `pi_uutils_ctx::is_cancelled()` periodically, since seq can generate
// unbounded output), `translate!` strings are literalized, SIGPIPE probing is
// dropped, and the entry point no longer calls `std::process::exit`.
use std::{
ffi::{OsStr, OsString},
io::{BufWriter, Write},
};
use clap::{Arg, ArgAction, ArgMatches, Command};
use num_bigint::BigUint;
use num_traits::{ToPrimitive, Zero};
use pi_uutils_ctx::format_usage;
use uucore::{
error::{FromIo, UResult},
extendedbigdecimal::ExtendedBigDecimal,
fast_inc::fast_inc,
format::{Format, num_format, num_format::FloatVariant},
};
mod error;
mod number;
mod numberparse;
use crate::{error::SeqError, number::PreciseNumber};
const OPT_SEPARATOR: &str = "separator";
const OPT_TERMINATOR: &str = "terminator";
const OPT_EQUAL_WIDTH: &str = "equal-width";
const OPT_FORMAT: &str = "format";
const ARG_NUMBERS: &str = "numbers";
/// pi-uutils: how many emitted numbers to print between cancellation polls in
/// the (potentially unbounded) emission loops.
const CANCEL_POLL_INTERVAL: u64 = 4096;
#[derive(Clone)]
struct SeqOptions<'a> {
separator: OsString,
terminator: OsString,
equal_width: bool,
format: Option<&'a str>,
}
/// A range of floats.
///
/// The elements are (first, increment, last).
type RangeFloat = (ExtendedBigDecimal, ExtendedBigDecimal, ExtendedBigDecimal);
/// Turn short args with attached value, for example "-s,", into two args "-s"
/// and "," to make them work with clap.
fn split_short_args_with_value(args: impl uucore::Args) -> impl uucore::Args {
let mut v: Vec<OsString> = Vec::new();
for arg in args {
let bytes = arg.as_encoded_bytes();
if bytes.len() > 2
&& (bytes.starts_with(b"-f") || bytes.starts_with(b"-s") || bytes.starts_with(b"-t"))
{
let (short_arg, value) = bytes.split_at(2);
// SAFETY:
// Both `short_arg` and `value` only contain content that originated from
// `OsStr::as_encoded_bytes`
v.push(unsafe { OsString::from_encoded_bytes_unchecked(short_arg.to_vec()) });
v.push(unsafe { OsString::from_encoded_bytes_unchecked(value.to_vec()) });
} else {
v.push(arg);
}
}
v.into_iter()
}
fn select_precision(
first: &PreciseNumber,
increment: &PreciseNumber,
last: &PreciseNumber,
) -> Option<usize> {
match (first.num_fractional_digits, increment.num_fractional_digits, last.num_fractional_digits)
{
(Some(0), Some(0), Some(0)) => Some(0),
(Some(f), Some(i), Some(_)) => Some(f.max(i)),
_ => None,
}
}
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
/// arguments directly (without the uucore clap-localization helper that would
/// terminate the process), renders clap help/usage/version to the context
/// streams, and maps the `UResult` to an exit code, so it is safe to run inside
/// the host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(split_short_args_with_value(argv.into_iter()))
{
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
match seq_main(&matches) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
let msg = err.to_string();
if !msg.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "seq: {msg}");
}
if code == 0 { 1 } else { code }
},
}
}
fn seq_main(matches: &ArgMatches) -> UResult<()> {
let numbers_option = matches.get_many::<String>(ARG_NUMBERS);
if numbers_option.is_none() {
return Err(SeqError::NoArguments.into());
}
let numbers = numbers_option.unwrap().collect::<Vec<_>>();
let options = SeqOptions {
separator: matches
.get_one::<OsString>(OPT_SEPARATOR)
.cloned()
.unwrap_or_else(|| OsString::from("\n")),
terminator: matches
.get_one::<OsString>(OPT_TERMINATOR)
.cloned()
.unwrap_or_else(|| OsString::from("\n")),
equal_width: matches.get_flag(OPT_EQUAL_WIDTH),
format: matches.get_one::<String>(OPT_FORMAT).map(String::as_str),
};
if options.equal_width && options.format.is_some() {
return Err(SeqError::FormatAndEqualWidth.into());
}
let first = if numbers.len() > 1 {
match numbers[0].parse() {
Ok(num) => num,
Err(e) => return Err(SeqError::ParseError(numbers[0].to_owned(), e).into()),
}
} else {
PreciseNumber::one()
};
let increment = if numbers.len() > 2 {
match numbers[1].parse() {
Ok(num) => num,
Err(e) => return Err(SeqError::ParseError(numbers[1].to_owned(), e).into()),
}
} else {
PreciseNumber::one()
};
if increment.is_zero() {
return Err(SeqError::ZeroIncrement(numbers[1].to_owned()).into());
}
let last: PreciseNumber = {
// We are guaranteed that `numbers.len()` is greater than zero
// and at most three because of the argument specification in
// `uu_app()`.
let n: usize = numbers.len();
match numbers[n - 1].parse() {
Ok(num) => num,
Err(e) => return Err(SeqError::ParseError(numbers[n - 1].to_owned(), e).into()),
}
};
// If a format was passed on the command line, use that.
// If not, use some default format based on parameters precision.
let (format, padding, fast_allowed) = if let Some(str) = options.format {
(Format::<num_format::Float, &ExtendedBigDecimal>::parse(str)?, 0, false)
} else {
let precision = select_precision(&first, &increment, &last);
let padding = if options.equal_width {
let precision_value = precision.unwrap_or(0);
first
.num_integral_digits
.max(increment.num_integral_digits)
.max(last.num_integral_digits)
+ if precision_value > 0 {
precision_value + 1
} else {
0
}
} else {
0
};
let formatter = match precision {
// format with precision: decimal floats and integers
Some(precision) => num_format::Float {
variant: FloatVariant::Decimal,
width: padding,
alignment: num_format::NumberAlignment::RightZero,
precision: Some(precision),
..Default::default()
},
// format without precision: hexadecimal floats
None => num_format::Float { variant: FloatVariant::Shortest, ..Default::default() },
};
// Allow fast printing if precision is 0 (integer inputs), `print_seq` will do
// further checks.
(Format::from_formatter(formatter), padding, precision == Some(0))
};
let result = print_seq(
(first.number, increment.number, last.number),
&options.separator,
&options.terminator,
&format,
fast_allowed,
padding,
);
match result {
Ok(()) => Ok(()),
Err(err) if err.kind() == std::io::ErrorKind::BrokenPipe => {
// GNU seq prints the Broken pipe message but still exits with status 0
// unless SIGPIPE was explicitly ignored, in which case it should fail.
// pi-uutils: the in-process builtin does not manipulate process
// signal dispositions, so the upstream `sigpipe_was_ignored` probe
// is dropped and the message goes to the context stderr.
let err = err.map_err_context(|| "write error".into());
let _ = writeln!(pi_uutils_ctx::stderr(), "seq: {err}");
Ok(())
},
Err(err) => Err(err.map_err_context(|| "write error".into())),
}
}
pub fn uu_app() -> Command {
Command::new("seq")
.trailing_var_arg(true)
.infer_long_args(true)
.version(uucore::crate_version!())
.about("Display numbers from FIRST to LAST, in steps of INCREMENT.")
.override_usage(format_usage(
"seq [OPTION]... LAST\nseq [OPTION]... FIRST LAST\nseq [OPTION]... FIRST INCREMENT LAST",
))
.arg(
Arg::new(OPT_SEPARATOR)
.short('s')
.long("separator")
.help("Separator character (defaults to \\n)")
.value_parser(clap::value_parser!(OsString)),
)
.arg(
Arg::new(OPT_TERMINATOR)
.short('t')
.long("terminator")
.help("Terminator character (defaults to \\n)")
.value_parser(clap::value_parser!(OsString)),
)
.arg(
Arg::new(OPT_EQUAL_WIDTH)
.short('w')
.long("equal-width")
.help("Equalize widths of all numbers by padding with zeros")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_FORMAT)
.short('f')
.long(OPT_FORMAT)
.help("use printf style floating-point FORMAT"),
)
.arg(
// we use allow_hyphen_values instead of allow_negative_numbers because clap removed
// the support for "exotic" negative numbers like -.1 (see https://github.com/clap-rs/clap/discussions/5837)
Arg::new(ARG_NUMBERS)
.allow_hyphen_values(true)
.action(ArgAction::Append)
.num_args(1..=3),
)
}
/// Integer print, default format, positive increment: fast code path
/// that avoids reformatting digit at all iterations.
fn fast_print_seq(
mut stdout: impl Write,
first: &BigUint,
increment: u64,
last: &BigUint,
separator: &OsStr,
terminator: &OsStr,
padding: usize,
) -> std::io::Result<()> {
// Nothing to do, just return.
if last < first {
return Ok(());
}
// Do at most u64::MAX loops. We can print in the order of 1e8 digits per
// second, u64::MAX is 1e19, so it'd take hundreds of years for this to
// complete anyway. TODO: we can move this test to `print_seq` if we care about
// this case.
let loop_cnt = ((last - first) / increment).to_u64().unwrap_or(u64::MAX);
// Format the first number.
let first_str = first.to_string();
// Makeshift log10.ceil
let last_length = last.to_string().len();
// Allocate a large u8 buffer, that contains a preformatted string
// of the number followed by the `separator`.
//
// | ... head space ... | number | separator |
// ^0 ^ start ^ num_end ^ size (==buf.len())
//
// We keep track of start in this buffer, as the number grows.
// When printing, we take a slice between start and end.
let size = last_length.max(padding) + separator.len();
// Fill with '0', this is needed for equal_width, and harmless otherwise.
let mut buf = vec![b'0'; size];
let buf = buf.as_mut_slice();
let num_end = buf.len() - separator.len();
let mut start = num_end - first_str.len();
// Initialize buf with first and separator.
buf[start..num_end].copy_from_slice(first_str.as_bytes());
buf[num_end..].copy_from_slice(separator.as_encoded_bytes());
// Normally, if padding is > 0, it should be equal to last_length,
// so start would be == 0, but there are corner cases.
start = start.min(num_end - padding);
// Prepare the number to increment with as a string
let inc_str = increment.to_string();
let inc_str = inc_str.as_bytes();
for i in 0..loop_cnt {
// pi-uutils: seq can generate effectively unbounded output; poll the
// host cancel flag periodically so shell abort/timeout is observed.
if i % CANCEL_POLL_INTERVAL == 0 && pi_uutils_ctx::is_cancelled() {
return Ok(());
}
stdout.write_all(&buf[start..])?;
fast_inc(buf, &mut start, num_end, inc_str);
}
// Write the last number without separator, but with terminator.
stdout.write_all(&buf[start..num_end])?;
stdout.write_all(terminator.as_encoded_bytes())?;
stdout.flush()?;
Ok(())
}
fn done_printing<T: Zero + PartialOrd>(next: &T, increment: &T, last: &T) -> bool {
if increment >= &T::zero() {
next > last
} else {
next < last
}
}
/// Arbitrary precision decimal number code path ("slow" path)
fn print_seq(
range: RangeFloat,
separator: &OsStr,
terminator: &OsStr,
format: &Format<num_format::Float, &ExtendedBigDecimal>,
fast_allowed: bool,
padding: usize, // Used by fast path only
) -> std::io::Result<()> {
// pi-uutils: buffer the context stdout handle instead of the (locked)
// process stdout.
let mut stdout = BufWriter::new(pi_uutils_ctx::stdout());
let (first, increment, last) = range;
if fast_allowed {
// Test if we can use fast code path.
// First try to convert the range to BigUint (u64 for the increment).
let (first_bui, increment_u64, last_bui) =
(first.to_biguint(), increment.to_biguint().and_then(|x| x.to_u64()), last.to_biguint());
if let (Some(first_bui), Some(increment_u64), Some(last_bui)) =
(first_bui, increment_u64, last_bui)
{
return fast_print_seq(
stdout,
&first_bui,
increment_u64,
&last_bui,
separator,
terminator,
padding,
);
}
}
let mut value = first;
let mut is_first_iteration = true;
// pi-uutils: iteration counter for periodic cancellation polling.
let mut iterations: u64 = 0;
while !done_printing(&value, &increment, &last) {
// pi-uutils: seq can generate effectively unbounded output; poll the
// host cancel flag periodically so shell abort/timeout is observed.
if iterations.is_multiple_of(CANCEL_POLL_INTERVAL) && pi_uutils_ctx::is_cancelled() {
return Ok(());
}
iterations += 1;
if !is_first_iteration {
stdout.write_all(separator.as_encoded_bytes())?;
}
format.fmt(&mut stdout, &value)?;
// TODO Implement augmenting addition.
value = value + increment.clone();
is_first_iteration = false;
}
if !is_first_iteration {
stdout.write_all(terminator.as_encoded_bytes())?;
}
stdout.flush()?;
Ok(())
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc};
use parking_lot::Mutex;
use pi_uutils_ctx::ScopeIo;
use super::*;
fn run_scoped(args: Vec<&str>, cancelled: bool) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
self.buf.lock().write(buf)
}
fn flush(&mut self) -> std::io::Result<()> {
self.buf.lock().flush()
}
}
let io = ScopeIo {
stdin: Box::new(std::io::empty()),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd: PathBuf::from("."),
env: HashMap::new(),
cancel: Arc::new(std::sync::atomic::AtomicBool::new(cancelled)),
};
let argv: Vec<OsString> = std::iter::once("seq")
.chain(args)
.map(OsString::from)
.collect();
let code = pi_uutils_ctx::scope(io, || run(argv));
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
(code, out_str, err_str)
}
fn run_in(args: Vec<&str>) -> (i32, String, String) {
run_scoped(args, false)
}
#[test]
fn single_operand_counts_from_one() {
let (code, stdout, stderr) = run_in(vec!["3"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "1\n2\n3\n", ""));
}
#[test]
fn first_increment_last_arithmetic() {
let (code, stdout, stderr) = run_in(vec!["2", "2", "10"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "2\n4\n6\n8\n10\n", ""));
}
#[test]
fn separator_joins_values_terminator_ends_them() {
let (code, stdout, stderr) = run_in(vec!["-s", ",", "1", "3"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "1,2,3\n", ""));
// Attached short-arg value goes through `split_short_args_with_value`.
let (code, stdout, _) = run_in(vec!["-s,", "1", "3"]);
assert_eq!((code, stdout.as_str()), (0, "1,2,3\n"));
}
#[test]
fn equal_width_pads_with_zeros() {
let (code, stdout, stderr) = run_in(vec!["-w", "8", "10"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "08\n09\n10\n", ""));
}
#[test]
fn float_increment_selects_widest_precision() {
let (code, stdout, stderr) = run_in(vec!["1", "0.5", "2"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "1.0\n1.5\n2.0\n", ""));
}
#[test]
fn invalid_operand_reports_error_and_fails() {
let (code, stdout, stderr) = run_in(vec!["foo"]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert_eq!(stderr, "seq: invalid floating point argument: 'foo'\n");
}
#[test]
fn zero_increment_is_rejected() {
let (code, stdout, stderr) = run_in(vec!["1", "0", "5"]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert_eq!(stderr, "seq: invalid Zero increment value: '0'\n");
}
#[test]
fn cancelled_scope_stops_emission() {
// pi-specific contract: a pre-cancelled scope aborts the (potentially
// unbounded) emission loop instead of printing the full range.
let (code, stdout, stderr) = run_scoped(vec!["1", "1000000"], true);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", ""));
}
#[test]
fn help_renders_to_scope_stdout() {
let (code, stdout, stderr) = run_in(vec!["--help"]);
assert_eq!(code, 0);
assert!(stdout.contains("Usage:"));
assert!(stdout.contains("steps of INCREMENT"));
assert_eq!(stderr, "");
}
}
+23
View File
@@ -0,0 +1,23 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/stat), patched to resolve
# path arguments against the shell working directory and route I/O through
# pi-uutils-ctx so it can run in-process as a shell builtin. See src/stat.rs
# for the patch markers (`pi-uutils:` comments).
[package]
name = "uu_stat"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "stat ~ (uutils) display FILE status (vendored + patched for in-process embedding)"
[lib]
path = "src/stat.rs"
[dependencies]
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
thiserror = "2.0.3"
uucore = { version = "0.8.0", features = ["entries", "libc", "fs", "fsext", "time"] }
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[dev-dependencies]
parking_lot = "0.12"
tempfile = "3"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+1785
View File
File diff suppressed because it is too large Load Diff
+26
View File
@@ -0,0 +1,26 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/tac), patched to resolve
# path arguments against the shell working directory and route I/O through
# pi-uutils-ctx so it can run in-process as a shell builtin. See src/tac.rs
# for the patch markers (`pi-uutils:` comments).
[package]
name = "uu_tac"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "tac ~ (uutils) concatenate and display input lines in reverse order (vendored + patched for in-process embedding)"
[lib]
path = "src/tac.rs"
[dependencies]
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
memchr = "2.7.4"
memmap2 = "0.9"
regex = "1.11"
thiserror = "2.0.3"
uucore = "0.8.0"
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[dev-dependencies]
parking_lot = "0.12"
tempfile = "3"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+47
View File
@@ -0,0 +1,47 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
//! Errors returned by tac during processing of a file.
// pi-uutils: vendored from uutils/coreutils 0.8.0; `translate!` strings are
// literalized with the en-US locale text.
use std::ffi::OsString;
use thiserror::Error;
use uucore::{
display::Quotable,
error::{UError, strip_errno},
};
#[derive(Debug, Error)]
pub enum TacError {
/// A regular expression given by the user is invalid.
#[error("invalid regular expression: {0}")]
InvalidRegex(regex::Error),
/// An error opening a file for reading.
///
/// The parameters are the name of the file and the underlying
/// [`std::io::Error`] that caused this error.
#[error("failed to open {} for reading: {}", .0.quote(), strip_errno(.1))]
OpenError(OsString, std::io::Error),
/// An error reading the contents of a file or stdin.
///
/// The parameters are the name of the file and the underlying
/// [`std::io::Error`] that caused this error.
#[error("{}: read error: {}", .0.maybe_quote(), strip_errno(.1))]
ReadError(OsString, std::io::Error),
/// An error writing the (reversed) contents of a file or stdin.
///
/// The parameter is the underlying [`std::io::Error`] that caused
/// this error.
#[error("failed to write to stdout: {}", strip_errno(.0))]
WriteError(std::io::Error),
}
impl UError for TacError {
fn code(&self) -> i32 {
1
}
}
+634
View File
@@ -0,0 +1,634 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore (ToDO) sbytes slen dlen memmem memmap Mmap mmap SIGBUS
// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process
// as a shell builtin. FILE operands resolve against the shell working directory
// via `pi_uutils_ctx::resolve` at the open/mmap call site (the original operand
// is kept for error messages), `-`/no-operand read the context stdin, output is
// written through the context stdout, recoverable per-file errors go to the
// context stderr with `pi_uutils_ctx::set_exit_code` (upstream `show!`), the
// `translate!` strings are literalized, and the process-global signal handling
// plus the stdin mmap/tempfile buffering (which target the process stdin fd)
// are removed.
mod error;
use std::{
ffi::{OsStr, OsString},
fs::File,
io::{BufWriter, Read, Write},
};
use clap::{Arg, ArgAction, ArgMatches, Command};
use memchr::memmem;
use memmap2::Mmap;
use pi_uutils_ctx::format_usage;
use uucore::error::UResult;
use crate::error::TacError;
mod options {
pub static BEFORE: &str = "before";
pub static REGEX: &str = "regex";
pub static SEPARATOR: &str = "separator";
pub static FILE: &str = "file";
}
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
/// arguments directly (without the uucore clap-localization helper that would
/// terminate the process), renders clap help/usage/version to the context
/// streams, and maps the `UResult` to an exit code, so it is safe to run inside
/// the host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
match tac_main(&matches) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
let msg = err.to_string();
if !msg.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "tac: {msg}");
}
if code == 0 { 1 } else { code }
},
}
}
fn tac_main(matches: &ArgMatches) -> UResult<()> {
let before = matches.get_flag(options::BEFORE);
let regex = matches.get_flag(options::REGEX);
let raw_separator = matches
.get_one::<OsString>(options::SEPARATOR)
.map_or(OsStr::new("\n"), |s| s.as_os_str());
let separator = if raw_separator.is_empty() {
OsStr::new("\0")
} else {
raw_separator
};
let files: Vec<OsString> = match matches.get_many::<OsString>(options::FILE) {
Some(v) => v.cloned().collect(),
None => vec![OsString::from("-")],
};
tac(&files, before, regex, separator)
}
pub fn uu_app() -> Command {
Command::new("tac")
.version(uucore::crate_version!())
.override_usage(format_usage("tac [OPTION]... [FILE]..."))
.about("Write each file to standard output, last line first.")
.infer_long_args(true)
.arg(
Arg::new(options::BEFORE)
.short('b')
.long(options::BEFORE)
.help("attach the separator before instead of after")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::REGEX)
.short('r')
.long(options::REGEX)
.help("interpret the sequence as a regular expression")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::SEPARATOR)
.short('s')
.long(options::SEPARATOR)
.help("use STRING as the separator instead of newline")
.value_parser(clap::value_parser!(OsString))
.value_name("STRING"),
)
.arg(
Arg::new(options::FILE)
.hide(true)
.action(ArgAction::Append)
.value_parser(clap::value_parser!(OsString))
.value_hint(clap::ValueHint::FilePath),
)
}
/// pi-uutils: replacement for upstream's `show!` — reports a recoverable
/// per-file error to the context stderr and accumulates a non-zero exit code
/// while processing continues with the next operand.
fn show(err: &TacError) {
let _ = writeln!(pi_uutils_ctx::stderr(), "tac: {err}");
pi_uutils_ctx::set_exit_code(1);
}
/// Print lines of a buffer in reverse, with line separator given as a regex.
///
/// `data` contains the bytes of the file.
///
/// `pattern` is the regular expression given as a
/// [`regex::bytes::Regex`] (not a [`regex::Regex`], since the input is
/// given as a slice of bytes). If `before` is `true`, then each match
/// of this pattern in `data` is interpreted as the start of a line. If
/// `before` is `false`, then each match of this pattern is interpreted
/// as the end of a line.
///
/// This function writes each line in `data` to the context stdout in
/// reverse.
///
/// # Errors
///
/// If there is a problem writing to stdout, then this function
/// returns [`std::io::Error`].
fn buffer_tac_regex(
data: &[u8],
pattern: &regex::bytes::Regex,
before: bool,
) -> std::io::Result<()> {
// pi-uutils: write through the context stdout instead of the process stdout.
let mut out = BufWriter::new(pi_uutils_ctx::stdout());
// The index of the line separator for the current line.
//
// As we scan through the `data` from right to left, we update this
// variable each time we find a new line separator. We restrict our
// regular expression search to only those bytes up to the line
// separator.
let mut this_line_end = data.len();
// The index of the start of the next line in the `data`.
//
// As we scan through the `data` from right to left, we update this
// variable each time we find a new line.
//
// If `before` is `true`, then each line starts immediately before
// the line separator. Otherwise, each line starts immediately after
// the line separator.
let mut following_line_start = data.len();
// Iterate over each byte in the buffer in reverse. When we find a
// line separator, write the line to stdout.
//
// The `before` flag controls whether the line separator appears at
// the end of the line (as in "abc\ndef\n") or at the beginning of
// the line (as in "/abc/def").
for i in (0..data.len()).rev() {
// Determine if there is a match for `pattern` starting at index
// `i` in `data`. Only search up to the line ending that was
// found previously.
if let Some(match_) = pattern.find_at(&data[..this_line_end], i)
&& match_.start() == i
{
// Record this index as the ending of the current line.
this_line_end = i;
// The length of the match (that is, the line separator), in bytes.
let slen = match_.end() - match_.start();
if before {
out.write_all(&data[i..following_line_start])?;
following_line_start = i;
} else {
out.write_all(&data[i + slen..following_line_start])?;
following_line_start = i + slen;
}
}
}
// After the loop terminates, write whatever bytes are remaining at
// the beginning of the buffer.
out.write_all(&data[0..following_line_start])?;
out.flush()?;
Ok(())
}
/// Write lines from `data` to stdout in reverse.
///
/// This function writes to the context stdout each line appearing in `data`,
/// starting with the last line and ending with the first line. The
/// `separator` parameter defines what characters to use as a line
/// separator.
///
/// If `before` is `false`, then this function assumes that the
/// `separator` appears at the end of each line, as in `"abc\ndef\n"`.
/// If `before` is `true`, then this function assumes that the
/// `separator` appears at the beginning of each line, as in
/// `"/abc/def"`.
fn buffer_tac(data: &[u8], before: bool, separator: &OsStr) -> std::io::Result<()> {
// pi-uutils: write through the context stdout instead of the process stdout.
let mut out = BufWriter::new(pi_uutils_ctx::stdout());
// The number of bytes in the line separator.
let slen = separator.len();
// The index of the start of the next line in the `data`.
//
// As we scan through the `data` from right to left, we update this
// variable each time we find a new line.
//
// If `before` is `true`, then each line starts immediately before
// the line separator. Otherwise, each line starts immediately after
// the line separator.
let mut following_line_start = data.len();
// Iterate over each byte in the buffer in reverse. When we find a
// line separator, write the line to stdout.
//
// The `before` flag controls whether the line separator appears at
// the end of the line (as in "abc\ndef\n") or at the beginning of
// the line (as in "/abc/def").
for i in memmem::rfind_iter(data, separator.as_encoded_bytes()) {
if before {
out.write_all(&data[i..following_line_start])?;
following_line_start = i;
} else {
out.write_all(&data[i + slen..following_line_start])?;
following_line_start = i + slen;
}
}
// After the loop terminates, write whatever bytes are remaining at
// the beginning of the buffer.
out.write_all(&data[0..following_line_start])?;
out.flush()?;
Ok(())
}
/// Make the regex flavor compatible with `regex` crate
///
/// Concretely:
/// - Toggle escaping of (), |, {}
/// - Escape ^ and $ when not at edges
/// - Leave only ASCII bytes inside []
/// - Escape non-ASCII bytes as `(?-u:\xFF)` outside []
fn translate_regex_flavor(bytes: &[u8]) -> String {
let mut result = Vec::new();
let mut i = 0;
let mut inside_brackets = false;
let mut prev_was_backslash = false;
let mut last_byte: Option<u8> = None;
while let Some(b) = bytes.get(i) {
let is_escaped = prev_was_backslash;
prev_was_backslash = false;
match b {
_ if inside_brackets && !b.is_ascii() => {
i += 1;
continue;
},
// Unescape escaped (), |, {} when not inside brackets
b'\\' if !inside_brackets && !is_escaped => {
if let Some(next) = bytes.get(i + 1)
&& matches!(next, b'(' | b')' | b'|' | b'{' | b'}')
{
result.push(*next);
last_byte = Some(*next);
i += 2;
continue;
}
result.push(b'\\');
last_byte = Some(b'\\');
prev_was_backslash = true;
},
// Bracket tracking
b'[' => {
inside_brackets = true;
result.push(*b);
last_byte = Some(*b);
},
b']' => {
inside_brackets = false;
result.push(*b);
last_byte = Some(*b);
},
// Escape (), |, {} when not escaped and outside brackets
b'(' | b')' | b'|' | b'{' | b'}' if !inside_brackets && !is_escaped => {
result.push(b'\\');
result.push(*b);
last_byte = Some(*b);
},
b'^' if !inside_brackets && !is_escaped => {
let is_anchor_position = result.is_empty() || matches!(last_byte, Some(b'(' | b'|'));
if !is_anchor_position {
result.push(b'\\');
}
result.push(*b);
last_byte = Some(*b);
},
b'$' if !inside_brackets && !is_escaped => {
let next_is_anchor_position = match bytes.get(i + 1) {
None => true,
Some(b')' | b'|') => true,
Some(b'\\') => {
// Peek two ahead to see if it's \) or \|
matches!(bytes.get(i + 2), Some(b')' | b'|'))
},
_ => false,
};
if !next_is_anchor_position {
result.push(b'\\');
}
result.push(*b);
last_byte = Some(*b);
},
_ if !b.is_ascii() => {
let _ = write!(result, r"(?-u:\x{b:02x})");
last_byte = None;
},
_ => {
result.push(*b);
last_byte = Some(*b);
},
}
i += 1;
}
String::from_utf8(result).expect("produces ASCII bytes")
}
#[allow(clippy::cognitive_complexity)]
fn tac(filenames: &[OsString], before: bool, regex: bool, separator: &OsStr) -> UResult<()> {
// Compile the regular expression pattern if it is provided.
let maybe_pattern = if regex {
match regex::bytes::RegexBuilder::new(&translate_regex_flavor(separator.as_encoded_bytes()))
.multi_line(true)
.build()
{
Ok(p) => Some(p),
Err(e) => return Err(TacError::InvalidRegex(e).into()),
}
} else {
None
};
for filename in filenames {
let mmap;
let buf;
let data: &[u8] = if filename == "-" {
// pi-uutils: in-process stdin is a context stream, not the process
// stdin fd; upstream's stdin mmap / tempfile buffering and the
// `stdin_was_closed` signal check do not apply. Read it fully.
let mut contents = Vec::new();
match pi_uutils_ctx::stdin().read_to_end(&mut contents) {
Ok(_) => {
buf = contents;
&buf
},
Err(e) => {
show(&TacError::ReadError(OsString::from("stdin"), e));
continue;
},
}
} else {
// pi-uutils: resolve the operand against the shell working
// directory at the open site; `filename` is kept for errors.
let path = pi_uutils_ctx::resolve(filename);
let mut file = match File::open(&path) {
Ok(f) => f,
Err(e) => {
show(&TacError::OpenError(filename.clone(), e));
continue;
},
};
if let Some(mmap1) = try_mmap_file(&file) {
mmap = mmap1;
&mmap
} else {
let mut contents = Vec::new();
match file.read_to_end(&mut contents) {
Ok(_) => {
buf = contents;
&buf
},
Err(e) => {
show(&TacError::ReadError(filename.clone(), e));
continue;
},
}
}
};
// Select the appropriate `tac` algorithm based on whether the
// separator is given as a regular expression or a fixed string.
// pi-uutils: match ergonomics instead of upstream's `Some(ref pattern)`.
let result = match &maybe_pattern {
Some(pattern) => buffer_tac_regex(data, pattern, before),
None => buffer_tac(data, before, separator),
};
// If there is any error in writing the output, terminate immediately.
if let Err(e) = result {
return Err(TacError::WriteError(e).into());
}
}
Ok(())
}
fn try_mmap_file(file: &File) -> Option<Mmap> {
// SAFETY: If the file is truncated while we map it, SIGBUS will be raised
// and our process will be terminated, thus preventing access of invalid memory.
unsafe { Mmap::map(file).ok() }
}
#[cfg(test)]
mod tests_hybrid_flavor {
use super::translate_regex_flavor;
#[test]
fn test_grouping_and_alternation() {
assert_eq!(translate_regex_flavor(br"\(abc\)"), r"(abc)");
assert_eq!(translate_regex_flavor(br"(abc)"), r"\(abc\)");
assert_eq!(translate_regex_flavor(br"a\|b"), r"a|b");
assert_eq!(translate_regex_flavor(br"a|b"), r"a\|b");
}
#[test]
fn test_anchors_context() {
assert_eq!(translate_regex_flavor(br"^abc$"), r"^abc$");
assert_eq!(translate_regex_flavor(br"a^b"), r"a\^b");
assert_eq!(translate_regex_flavor(br"a$b"), r"a\$b");
// Anchors inside groups (reset by \(...\) regardless of position)
assert_eq!(translate_regex_flavor(br"\(^abc\)"), r"(^abc)");
assert_eq!(translate_regex_flavor(br"\(abc$\)"), r"(abc$)");
// Anchors inside alternation (reset by \| regardless of position)
assert_eq!(translate_regex_flavor(br"^a\|^b"), r"^a|^b");
assert_eq!(translate_regex_flavor(br"a$\|b$"), r"a$|b$");
}
#[test]
fn test_character_classes() {
assert_eq!(translate_regex_flavor(br"[a-z]"), r"[a-z]");
assert_eq!(translate_regex_flavor(br"[.]"), r"[.]");
assert_eq!(translate_regex_flavor(br"[]abc]"), r"[]abc]");
assert_eq!(translate_regex_flavor(br"[^]abc]"), r"[^]abc]");
}
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, fs, io::Write, path::PathBuf, sync::Arc};
use parking_lot::Mutex;
use pi_uutils_ctx::ScopeIo;
use super::*;
fn run_with(cwd: PathBuf, stdin: &[u8], args: Vec<&str>) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
self.buf.lock().write(buf)
}
fn flush(&mut self) -> std::io::Result<()> {
self.buf.lock().flush()
}
}
let io = ScopeIo {
stdin: Box::new(std::io::Cursor::new(stdin.to_vec())),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd,
env: HashMap::new(),
cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)),
};
let argv: Vec<OsString> = std::iter::once("tac")
.chain(args)
.map(OsString::from)
.collect();
let code = pi_uutils_ctx::scope(io, || run(argv));
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
(code, out_str, err_str)
}
/// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var).
fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) {
let dir = tempfile::tempdir().unwrap();
let canon = fs::canonicalize(dir.path()).unwrap();
(dir, canon)
}
#[test]
fn resolves_relative_operand_against_scope_cwd() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("input.txt"), b"a\nb\nc\n").unwrap();
// Relative operand + scope cwd differing from the process cwd: only the
// call-site `pi_uutils_ctx::resolve` patch makes this find the file.
let (code, stdout, stderr) = run_with(root, b"", vec!["input.txt"]);
assert_eq!(code, 0);
assert_eq!(stdout, "c\nb\na\n");
assert_eq!(stderr, "");
}
#[test]
fn no_operand_reads_context_stdin() {
let (code, stdout, stderr) = run_with(PathBuf::from("."), b"one\ntwo\nthree\n", vec![]);
assert_eq!(code, 0);
assert_eq!(stdout, "three\ntwo\none\n");
assert_eq!(stderr, "");
}
#[test]
fn dash_operand_reads_context_stdin() {
let (code, stdout, stderr) = run_with(PathBuf::from("."), b"x\ny\n", vec!["-"]);
assert_eq!(code, 0);
assert_eq!(stdout, "y\nx\n");
assert_eq!(stderr, "");
}
#[test]
fn custom_separator_reverses_fields() {
let (code, stdout, stderr) = run_with(PathBuf::from("."), b"a,b,c,", vec!["-s", ","]);
assert_eq!(code, 0);
assert_eq!(stdout, "c,b,a,");
assert_eq!(stderr, "");
}
#[test]
fn before_flag_attaches_separator_before_each_line() {
let (code, stdout, stderr) = run_with(PathBuf::from("."), b"/abc/def", vec!["-b", "-s", "/"]);
assert_eq!(code, 0);
assert_eq!(stdout, "/def/abc");
assert_eq!(stderr, "");
}
#[test]
fn regex_separator_splits_on_character_class() {
// `[,;]` treats either byte as a separator; records are emitted in
// reverse with each separator kept attached to its preceding record.
let (code, stdout, stderr) = run_with(PathBuf::from("."), b"a,b;c", vec!["-r", "-s", "[,;]"]);
assert_eq!(code, 0);
assert_eq!(stdout, "cb;a,");
assert_eq!(stderr, "");
}
#[test]
fn invalid_regex_is_fatal_error() {
let (code, stdout, stderr) = run_with(PathBuf::from("."), b"abc", vec!["-r", "-s", "["]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert!(stderr.starts_with("tac: invalid regular expression:"), "stderr: {stderr}");
}
#[test]
fn missing_file_continues_with_next_operand_and_exits_nonzero() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("good.txt"), b"1\n2\n").unwrap();
let (code, stdout, stderr) = run_with(root, b"", vec!["nope.txt", "good.txt"]);
assert_eq!(code, 1);
assert_eq!(stdout, "2\n1\n", "valid operand still printed after the failure");
assert!(stderr.contains("tac: failed to open 'nope.txt' for reading:"), "stderr: {stderr}");
}
#[test]
fn help_renders_to_scope_stdout() {
let (code, stdout, stderr) = run_with(PathBuf::from("."), b"", vec!["--help"]);
assert_eq!(code, 0);
assert!(stdout.contains("Usage:"));
assert!(stdout.contains("last line first"));
assert_eq!(stderr, "");
}
}
+36
View File
@@ -0,0 +1,36 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/touch), patched to resolve
# path arguments against the shell working directory and route I/O through
# pi-uutils-ctx so it can run in-process as a shell builtin. See src/touch.rs
# for the patch markers (`pi-uutils:` comments).
[package]
name = "uu_touch"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "touch ~ (uutils) change FILE timestamps (vendored + patched for in-process embedding)"
[lib]
path = "src/touch.rs"
[dependencies]
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
filetime = "0.2.23"
jiff = "0.2.18"
parse_datetime = "0.14.0"
thiserror = "2.0.3"
uucore = { version = "0.8.0", features = ["libc", "parser"] }
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[target.'cfg(unix)'.dependencies]
libc = "0.2.172"
rustix = { version = "1.1.4", features = ["fs"] }
[target.'cfg(windows)'.dependencies]
windows-sys = { version = "0.61.0", default-features = false, features = [
"Win32_Storage_FileSystem",
"Win32_Foundation",
] }
[dev-dependencies]
parking_lot = "0.12"
tempfile = "3"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
File diff suppressed because it is too large Load Diff
+22
View File
@@ -0,0 +1,22 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/truncate), patched to resolve
# path arguments against the shell working directory and route I/O through
# pi-uutils-ctx so it can run in-process as a shell builtin. See src/truncate.rs
# for the patch markers (`pi-uutils:` comments).
[package]
name = "uu_truncate"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "truncate ~ (uutils) truncate (or extend) FILE to SIZE (vendored + patched for in-process embedding)"
[lib]
path = "src/truncate.rs"
[dependencies]
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
uucore = { version = "0.8.0", features = ["parser-size"] }
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[dev-dependencies]
parking_lot = "0.12"
tempfile = "3"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+557
View File
@@ -0,0 +1,557 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore (ToDO) RFILE refsize rfilename fsize tsize
// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process
// as a shell builtin. Every filesystem syscall resolves its path operand
// against the shell working directory via `pi_uutils_ctx::resolve` AT THE CALL
// SITE, while the original operands are kept for display/error messages (GNU
// prints operands as typed). All process-global stdio is routed through
// `pi_uutils_ctx`, `translate!` strings are literalized, per-file errors are
// reported through the context stderr with `set_exit_code` (continue-on-error
// like GNU truncate), and the entry point no longer calls `std::process::exit`.
#[cfg(unix)]
use std::os::unix::fs::FileTypeExt;
use std::{
ffi::OsString,
fs::{OpenOptions, metadata},
io::{ErrorKind, Write},
};
use clap::{Arg, ArgAction, ArgMatches, Command};
use pi_uutils_ctx::format_usage;
use uucore::{
display::Quotable,
error::{FromIo, UResult, USimpleError, UUsageError},
parser::parse_size::{ParseSizeError, Parser, allow_list_with_all_suffixes},
};
#[derive(Debug, Eq, PartialEq)]
enum TruncateMode {
Absolute(u64),
Extend(u64),
Reduce(u64),
AtMost(u64),
AtLeast(u64),
RoundDown(u64),
RoundUp(u64),
}
impl TruncateMode {
/// Compute a target size in bytes for this truncate mode.
///
/// `fsize` is the size of the reference file, in bytes.
///
/// If the mode is [`TruncateMode::Reduce`] and the value to
/// reduce by is greater than `fsize`, then this function returns
/// 0 (since it cannot return a negative number).
///
/// # Returns
///
/// `None` if rounding by 0, else the target size.
fn to_size(&self, fsize: u64) -> Option<u64> {
match self {
Self::Absolute(size) => Some(*size),
Self::Extend(size) => Some(fsize + size),
Self::Reduce(size) => Some(fsize.saturating_sub(*size)),
Self::AtMost(size) => Some(fsize.min(*size)),
Self::AtLeast(size) => Some(fsize.max(*size)),
Self::RoundDown(size) => fsize.checked_rem(*size).map(|remainder| fsize - remainder),
Self::RoundUp(size) => fsize.checked_next_multiple_of(*size),
}
}
/// Determine if mode is absolute
///
/// # Returns
///
/// `true` is self matches Self::Absolute(_), `false` otherwise.
fn is_absolute(&self) -> bool {
matches!(self, Self::Absolute(_))
}
}
pub mod options {
pub static IO_BLOCKS: &str = "io-blocks";
pub static NO_CREATE: &str = "no-create";
pub static REFERENCE: &str = "reference";
pub static SIZE: &str = "size";
pub static ARG_FILES: &str = "files";
}
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
/// arguments directly (without the uucore clap-localization helper that would
/// terminate the process), renders clap help/usage/version to the context
/// streams, and maps the `UResult` to an exit code, so it is safe to run inside
/// the host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
match truncate_main(&matches) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
// pi-uutils: don't emit a dangling "truncate: " prefix when the
// error renders to an empty message.
let msg = err.to_string();
if !msg.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "truncate: {msg}");
}
if code == 0 { 1 } else { code }
},
}
}
fn truncate_main(matches: &ArgMatches) -> UResult<()> {
let files: Vec<OsString> = matches
.get_many::<OsString>(options::ARG_FILES)
.map(|v| v.cloned().collect())
.unwrap_or_default();
if files.is_empty() {
Err(UUsageError::new(1, "missing file operand".to_string()))
} else {
let io_blocks = matches.get_flag(options::IO_BLOCKS);
let no_create = matches.get_flag(options::NO_CREATE);
let reference = matches
.get_one::<String>(options::REFERENCE)
.map(String::from);
let size = matches.get_one::<String>(options::SIZE).map(String::from);
truncate(no_create, io_blocks, reference, size, &files)
}
}
pub fn uu_app() -> Command {
Command::new("truncate")
.version(uucore::crate_version!())
.about("Shrink or extend the size of each file to the specified size.")
.override_usage(format_usage("truncate [OPTION]... [FILE]..."))
.after_help(
"SIZE is an integer with an optional prefix and optional unit.\nThe available units (K, \
M, G, T, P, E, Z, and Y) use the following format:\n 'KB' => 1000 (kilobytes)\n \
'K' => 1024 (kibibytes)\n 'MB' => 1000*1000 (megabytes)\n 'M' => 1024*1024 \
(mebibytes)\n 'GB' => 1000*1000*1000 (gigabytes)\n 'G' => 1024*1024*1024 \
(gibibytes)\nSIZE may also be prefixed by one of the following to adjust the size of \
each\nfile based on its current size:\n '+' => extend by\n '-' => reduce by\n \
'<' => at most\n '>' => at least\n '/' => round down to multiple of\n '%' => \
round up to multiple of",
)
.infer_long_args(true)
.arg(
Arg::new(options::IO_BLOCKS)
.short('o')
.long(options::IO_BLOCKS)
.help(
"treat SIZE as the number of I/O blocks of the file rather than bytes (NOT \
IMPLEMENTED)",
)
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::NO_CREATE)
.short('c')
.long(options::NO_CREATE)
.help("do not create files that do not exist")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::REFERENCE)
.short('r')
.long(options::REFERENCE)
.required_unless_present(options::SIZE)
.help("base the size of each file on the size of RFILE")
.value_name("RFILE")
.value_hint(clap::ValueHint::FilePath),
)
.arg(
Arg::new(options::SIZE)
.short('s')
.long(options::SIZE)
.required_unless_present(options::REFERENCE)
.help(
"set or adjust the size of each file according to SIZE, which is in bytes unless \
--io-blocks is specified",
)
.allow_hyphen_values(true)
.value_name("SIZE"),
)
.arg(
Arg::new(options::ARG_FILES)
.value_name("FILE")
.action(ArgAction::Append)
.required(true)
.value_hint(clap::ValueHint::FilePath)
.value_parser(clap::value_parser!(OsString)),
)
}
/// Truncate the named file to the specified size.
///
/// If `create` is true, then the file will be created if it does not
/// already exist. If `size` is larger than the number of bytes in the
/// file, then the file will be padded with zeros. If `size` is smaller
/// than the number of bytes in the file, then the file will be
/// truncated and any bytes beyond `size` will be lost.
///
/// # Errors
///
/// If the file could not be opened, or there was a problem setting the
/// size of the file.
fn do_file_truncate(filename: &OsString, create: bool, size: u64) -> UResult<()> {
// pi-uutils: resolve the operand against the shell working directory at
// the open site; `filename` is kept for the error message.
let resolved = pi_uutils_ctx::resolve(filename);
match OpenOptions::new()
.write(true)
.create(create)
.open(&resolved)
{
Ok(file) => file.set_len(size),
Err(e) if e.kind() == ErrorKind::NotFound && !create => Ok(()),
Err(e) => Err(e),
}
.map_err_context(|| format!("cannot open {} for writing", filename.quote()))
}
fn file_truncate(
no_create: bool,
reference_size: Option<u64>,
mode: &TruncateMode,
filename: &OsString,
) -> UResult<()> {
// pi-uutils: resolve the operand against the shell working directory at
// the metadata site; `filename` is kept for the error message.
let resolved = pi_uutils_ctx::resolve(filename);
// Get the length of the file.
let file_size = match metadata(&resolved) {
Ok(metadata) => {
// A pipe has no length. Do this check here to avoid duplicate `stat()` syscall.
#[cfg(unix)]
if metadata.file_type().is_fifo() {
return Err(USimpleError::new(
1,
format!(
"cannot open {} for writing: No such device or address",
filename.to_string_lossy().quote()
),
));
}
metadata.len()
},
Err(_) => 0,
};
// The reference size can be either:
//
// 1. The size of a given file
// 2. The size of the file to be truncated if no reference has been provided.
let actual_reference_size = reference_size.unwrap_or(file_size);
let Some(truncate_size) = mode.to_size(actual_reference_size) else {
return Err(USimpleError::new(1, "division by zero".to_string()));
};
do_file_truncate(filename, !no_create, truncate_size)
}
fn truncate(
no_create: bool,
_: bool,
reference: Option<String>,
size: Option<String>,
filenames: &[OsString],
) -> UResult<()> {
let reference_size = match reference {
Some(reference_path) => {
// pi-uutils: resolve the reference operand against the shell
// working directory; `reference_path` is kept for the message.
let reference_metadata =
metadata(pi_uutils_ctx::resolve(&reference_path)).map_err(|error| {
match error.kind() {
ErrorKind::NotFound => USimpleError::new(
1,
format!("cannot stat {}: No such file or directory", reference_path.quote()),
),
_ => error.map_err_context(String::new),
}
})?;
Some(reference_metadata.len())
},
None => None,
};
let size_string = size.as_deref();
// Omitting the mode is equivalent to extending a file by 0 bytes.
let mode = match size_string {
Some(string) => match parse_mode_and_size(string) {
Err(error) => {
return Err(USimpleError::new(1, format!("Invalid number: {error}")));
},
Ok(mode) => mode,
},
None => TruncateMode::Extend(0),
};
// If a reference file has been given, the truncate mode cannot be absolute.
if reference_size.is_some() && mode.is_absolute() {
return Err(USimpleError::new(
1,
"you must specify a relative '--size' with '--reference'".to_string(),
));
}
for filename in filenames {
// pi-uutils: upstream aborts on the first failing file; report the
// error through the context stderr and continue with the remaining
// operands (GNU behavior), accumulating the exit code.
if let Err(err) = file_truncate(no_create, reference_size, &mode, filename) {
let msg = err.to_string();
if !msg.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "truncate: {msg}");
}
pi_uutils_ctx::set_exit_code(if err.code() == 0 { 1 } else { err.code() });
}
}
Ok(())
}
/// Decide whether a character is one of the size modifiers, like '+' or '<'.
fn is_modifier(c: char) -> bool {
c == '+' || c == '-' || c == '<' || c == '>' || c == '/' || c == '%'
}
/// Parse a size string with optional modifier symbol as its first character.
///
/// A size string is as described in [`Parser::parse_u64`]. The first character
/// of `size_string` might be a modifier symbol, like `'+'` or
/// `'<'`. The first element of the pair returned by this function
/// indicates which modifier symbol was present, or
/// [`TruncateMode::Absolute`] if none.
fn parse_mode_and_size(size_string: &str) -> Result<TruncateMode, ParseSizeError> {
// Trim any whitespace.
let mut size_string = size_string.trim();
// Get the modifier character from the size string, if any. For
// example, if the argument is "+123", then the modifier is '+'.
if let Some(c) = size_string.chars().next() {
if is_modifier(c) {
size_string = &size_string[1..];
}
let allow_list = allow_list_with_all_suffixes("EgGkKmMPQRtTYZ");
let allow_list_ref = allow_list.iter().map(AsRef::as_ref).collect::<Vec<&str>>();
Parser::default()
.with_allow_list(&allow_list_ref)
.parse_u64(size_string)
.map(match c {
'+' => TruncateMode::Extend,
'-' => TruncateMode::Reduce,
'<' => TruncateMode::AtMost,
'>' => TruncateMode::AtLeast,
'/' => TruncateMode::RoundDown,
'%' => TruncateMode::RoundUp,
_ => TruncateMode::Absolute,
})
} else {
Err(ParseSizeError::ParseFailure(size_string.to_string()))
}
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, fs, path::PathBuf, sync::Arc};
use parking_lot::Mutex;
use pi_uutils_ctx::ScopeIo;
use super::*;
fn run_in(cwd: PathBuf, args: Vec<&str>) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
self.buf.lock().write(buf)
}
fn flush(&mut self) -> std::io::Result<()> {
self.buf.lock().flush()
}
}
let io = ScopeIo {
stdin: Box::new(std::io::empty()),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd,
env: HashMap::new(),
cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)),
};
let argv: Vec<OsString> = std::iter::once("truncate")
.chain(args)
.map(OsString::from)
.collect();
let code = pi_uutils_ctx::scope(io, || run(argv));
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
(code, out_str, err_str)
}
/// Canonicalized temp dir (macOS tempdirs live behind /var -> /private/var).
fn canonical_tempdir() -> (tempfile::TempDir, PathBuf) {
let dir = tempfile::tempdir().unwrap();
let canon = fs::canonicalize(dir.path()).unwrap();
(dir, canon)
}
fn len(path: &PathBuf) -> u64 {
fs::metadata(path).unwrap().len()
}
#[test]
fn resolves_relative_operand_against_scope_cwd() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("f"), b"12345678").unwrap();
// Relative operand + scope cwd differing from the process cwd: only the
// call-site `pi_uutils_ctx::resolve` patch makes this find the file.
let (code, stdout, stderr) = run_in(root.clone(), vec!["-s", "5", "f"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", ""));
assert_eq!(len(&root.join("f")), 5);
}
#[test]
fn extend_grows_by_relative_amount() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("f"), b"1234").unwrap();
let (code, _, stderr) = run_in(root.clone(), vec!["-s", "+3", "f"]);
assert_eq!((code, stderr.as_str()), (0, ""));
assert_eq!(len(&root.join("f")), 7);
}
#[test]
fn at_most_caps_only_larger_files() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("big"), vec![0u8; 20]).unwrap();
fs::write(root.join("small"), b"abc").unwrap();
let (code, _, stderr) = run_in(root.clone(), vec!["-s", "<10", "big", "small"]);
assert_eq!((code, stderr.as_str()), (0, ""));
assert_eq!(len(&root.join("big")), 10);
assert_eq!(len(&root.join("small")), 3);
}
#[test]
fn no_create_skips_missing_file() {
let (_dir, root) = canonical_tempdir();
let (code, stdout, stderr) = run_in(root.clone(), vec!["-c", "-s", "5", "missing"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "", ""));
assert!(!root.join("missing").exists());
}
#[test]
fn missing_file_without_no_create_is_created_at_size() {
let (_dir, root) = canonical_tempdir();
let (code, _, stderr) = run_in(root.clone(), vec!["-s", "9", "fresh"]);
assert_eq!((code, stderr.as_str()), (0, ""));
assert_eq!(len(&root.join("fresh")), 9);
}
#[test]
fn reference_copies_size_of_rfile() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("ref"), b"123456").unwrap();
fs::write(root.join("f"), b"x").unwrap();
let (code, _, stderr) = run_in(root.clone(), vec!["-r", "ref", "f"]);
assert_eq!((code, stderr.as_str()), (0, ""));
assert_eq!(len(&root.join("f")), 6);
}
#[test]
fn missing_reference_file_fails_with_stat_error() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("f"), b"x").unwrap();
let (code, _, stderr) = run_in(root.clone(), vec!["-r", "nope", "f"]);
assert_eq!(code, 1);
assert!(stderr.contains("cannot stat 'nope': No such file or directory"));
assert_eq!(len(&root.join("f")), 1, "operand must be untouched");
}
#[test]
fn invalid_size_reports_error_and_exit_1() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("f"), b"x").unwrap();
let (code, stdout, stderr) = run_in(root.clone(), vec!["-s", "bogus", "f"]);
assert_eq!((code, stdout.as_str()), (1, ""));
assert!(stderr.contains("truncate: Invalid number:"));
assert_eq!(len(&root.join("f")), 1, "operand must be untouched");
}
#[test]
fn reference_with_absolute_size_is_rejected() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("ref"), b"123").unwrap();
fs::write(root.join("f"), b"x").unwrap();
let (code, _, stderr) = run_in(root.clone(), vec!["-r", "ref", "-s", "5", "f"]);
assert_eq!(code, 1);
assert!(stderr.contains("you must specify a relative '--size' with '--reference'"));
}
#[test]
fn parse_mode_and_size_prefixes() {
assert_eq!(parse_mode_and_size("10"), Ok(TruncateMode::Absolute(10)));
assert_eq!(parse_mode_and_size("+10"), Ok(TruncateMode::Extend(10)));
assert_eq!(parse_mode_and_size("-10"), Ok(TruncateMode::Reduce(10)));
assert_eq!(parse_mode_and_size("<10"), Ok(TruncateMode::AtMost(10)));
assert_eq!(parse_mode_and_size(">10"), Ok(TruncateMode::AtLeast(10)));
assert_eq!(parse_mode_and_size("/10"), Ok(TruncateMode::RoundDown(10)));
assert_eq!(parse_mode_and_size("%10"), Ok(TruncateMode::RoundUp(10)));
assert_eq!(parse_mode_and_size("1kB"), Ok(TruncateMode::Absolute(1000)));
assert!(parse_mode_and_size("1b").is_err());
}
#[test]
fn help_renders_to_scope_stdout() {
let (code, stdout, stderr) = run_in(PathBuf::from("."), vec!["--help"]);
assert_eq!(code, 0);
assert!(stdout.contains("Usage:"));
assert!(stdout.contains("round up to multiple of"));
assert_eq!(stderr, "");
}
}
+21
View File
@@ -0,0 +1,21 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/uname), patched to route
# output through pi-uutils-ctx so it can run in-process as a shell builtin. See
# src/uname.rs for the patch markers (`pi-uutils:` comments).
[package]
name = "uu_uname"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "uname ~ (uutils) display system information (vendored + patched for in-process embedding)"
[lib]
path = "src/uname.rs"
[dependencies]
platform-info = "2.0.3"
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
uucore = "0.8.0"
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[dev-dependencies]
parking_lot = "0.12"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+342
View File
@@ -0,0 +1,342 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore (API) nodename osname sysname (options) mnrsv mnrsvo
// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process
// as a shell builtin. Output goes to the context stdout (upstream's
// `println_verbatim` writes to the process stdout), `translate!` strings are
// literalized, and the entry point no longer calls `std::process::exit`.
use std::{
ffi::{OsStr, OsString},
io::Write,
};
use clap::{Arg, ArgAction, ArgMatches, Command};
use pi_uutils_ctx::format_usage;
use platform_info::{PlatformInfo, PlatformInfoAPI, UNameAPI};
use uucore::error::{UResult, USimpleError};
pub mod options {
pub static ALL: &str = "all";
pub static KERNEL_NAME: &str = "kernel-name";
pub static NODENAME: &str = "nodename";
pub static KERNEL_VERSION: &str = "kernel-version";
pub static KERNEL_RELEASE: &str = "kernel-release";
pub static MACHINE: &str = "machine";
pub static PROCESSOR: &str = "processor";
pub static HARDWARE_PLATFORM: &str = "hardware-platform";
pub static OS: &str = "operating-system";
}
pub struct UNameOutput {
pub kernel_name: Option<OsString>,
pub nodename: Option<OsString>,
pub kernel_release: Option<OsString>,
pub kernel_version: Option<OsString>,
pub machine: Option<OsString>,
pub os: Option<OsString>,
pub processor: Option<OsString>,
pub hardware_platform: Option<OsString>,
}
impl UNameOutput {
fn display(&self) -> OsString {
[
self.kernel_name.as_ref(),
self.nodename.as_ref(),
self.kernel_release.as_ref(),
self.kernel_version.as_ref(),
self.machine.as_ref(),
self.processor.as_ref(),
self.hardware_platform.as_ref(),
self.os.as_ref(),
]
.into_iter()
.flatten()
.map(OsString::as_os_str)
.collect::<Vec<_>>()
.join(OsStr::new(" "))
}
pub fn new(opts: &Options) -> UResult<Self> {
let uname = PlatformInfo::new()
.map_err(|_e| USimpleError::new(1, "cannot get system name".to_string()))?;
let none = !(opts.all
|| opts.kernel_name
|| opts.nodename
|| opts.kernel_release
|| opts.kernel_version
|| opts.machine
|| opts.os
|| opts.processor
|| opts.hardware_platform);
let kernel_name = (opts.kernel_name || opts.all || none).then(|| uname.sysname().to_owned());
let nodename = (opts.nodename || opts.all).then(|| uname.nodename().to_owned());
let kernel_release = (opts.kernel_release || opts.all).then(|| uname.release().to_owned());
let kernel_version = (opts.kernel_version || opts.all).then(|| uname.version().to_owned());
let machine = (opts.machine || opts.all).then(|| uname.machine().to_owned());
let os = (opts.os || opts.all).then(|| uname.osname().to_owned());
// This option is unsupported on modern Linux systems
// See: https://lists.gnu.org/archive/html/bug-coreutils/2005-09/msg00063.html
let processor = opts.processor.then(|| "unknown".into());
// This option is unsupported on modern Linux systems
// See: https://lists.gnu.org/archive/html/bug-coreutils/2005-09/msg00063.html
let hardware_platform = opts.hardware_platform.then(|| "unknown".into());
Ok(Self {
kernel_name,
nodename,
kernel_release,
kernel_version,
machine,
os,
processor,
hardware_platform,
})
}
}
pub struct Options {
pub all: bool,
pub kernel_name: bool,
pub nodename: bool,
pub kernel_version: bool,
pub kernel_release: bool,
pub machine: bool,
pub processor: bool,
pub hardware_platform: bool,
pub os: bool,
}
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
/// arguments directly (without the uucore clap-localization helper that would
/// terminate the process), renders clap help/usage/version to the context
/// streams, and maps the `UResult` to an exit code, so it is safe to run inside
/// the host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
match uname_main(&matches) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
let msg = err.to_string();
if !msg.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "uname: {msg}");
}
if code == 0 { 1 } else { code }
},
}
}
fn uname_main(matches: &ArgMatches) -> UResult<()> {
let options = Options {
all: matches.get_flag(options::ALL),
kernel_name: matches.get_flag(options::KERNEL_NAME),
nodename: matches.get_flag(options::NODENAME),
kernel_release: matches.get_flag(options::KERNEL_RELEASE),
kernel_version: matches.get_flag(options::KERNEL_VERSION),
machine: matches.get_flag(options::MACHINE),
processor: matches.get_flag(options::PROCESSOR),
hardware_platform: matches.get_flag(options::HARDWARE_PLATFORM),
os: matches.get_flag(options::OS),
};
let output = UNameOutput::new(&options)?;
// pi-uutils: replacement for upstream's `println_verbatim` — writes the
// output bytes verbatim to the context stdout instead of the process
// stdout.
let mut out = pi_uutils_ctx::stdout();
out.write_all(uucore::os_str_as_bytes(output.display().as_os_str())?)
.and_then(|()| out.write_all(b"\n"))
.and_then(|()| out.flush())
.map_err(|e| USimpleError::new(1, e.to_string()))?;
Ok(())
}
pub fn uu_app() -> Command {
Command::new("uname")
.version(uucore::crate_version!())
.about("Print certain system information.\nWith no OPTION, same as -s.")
.override_usage(format_usage("uname [OPTION]..."))
.infer_long_args(true)
.arg(
Arg::new(options::ALL)
.short('a')
.long(options::ALL)
.help("Behave as though all of the options -mnrsvo were specified.")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::KERNEL_NAME)
.short('s')
.long(options::KERNEL_NAME)
.alias("sysname") // Obsolescent option in GNU uname
.help("print the kernel name.")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::NODENAME)
.short('n')
.long(options::NODENAME)
.help(
"print the nodename (the nodename may be a name that the system is known by to a \
communications network).",
)
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::KERNEL_RELEASE)
.short('r')
.long(options::KERNEL_RELEASE)
.alias("release") // Obsolescent option in GNU uname
.help("print the operating system release.")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::KERNEL_VERSION)
.short('v')
.long(options::KERNEL_VERSION)
.help("print the operating system version.")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::MACHINE)
.short('m')
.long(options::MACHINE)
.help("print the machine hardware name.")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::OS)
.short('o')
.long(options::OS)
.help("print the operating system name.")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::PROCESSOR)
.short('p')
.long(options::PROCESSOR)
.help("print the processor type (non-portable)")
.action(ArgAction::SetTrue)
.hide(true),
)
.arg(
Arg::new(options::HARDWARE_PLATFORM)
.short('i')
.long(options::HARDWARE_PLATFORM)
.help("print the hardware platform (non-portable)")
.action(ArgAction::SetTrue)
.hide(true),
)
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc};
use parking_lot::Mutex;
use pi_uutils_ctx::ScopeIo;
use super::*;
fn run_in(args: Vec<&str>) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
self.buf.lock().write(buf)
}
fn flush(&mut self) -> std::io::Result<()> {
self.buf.lock().flush()
}
}
let io = ScopeIo {
stdin: Box::new(std::io::empty()),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd: PathBuf::from("."),
env: HashMap::new(),
cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)),
};
let argv: Vec<OsString> = std::iter::once("uname")
.chain(args)
.map(OsString::from)
.collect();
let code = pi_uutils_ctx::scope(io, || run(argv));
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
(code, out_str, err_str)
}
#[test]
fn kernel_name_matches_platform() {
let (code, stdout, stderr) = run_in(vec!["-s"]);
assert_eq!((code, stderr.as_str()), (0, ""));
#[cfg(target_os = "macos")]
assert_eq!(stdout, "Darwin\n");
#[cfg(target_os = "linux")]
assert_eq!(stdout, "Linux\n");
#[cfg(not(any(target_os = "macos", target_os = "linux")))]
assert!(stdout.trim_end().len() > 0);
}
#[test]
fn no_options_defaults_to_kernel_name() {
let (code, bare, _) = run_in(vec![]);
let (_, with_s, _) = run_in(vec!["-s"]);
assert_eq!(code, 0);
assert_eq!(bare, with_s);
}
#[test]
fn all_contains_kernel_name_and_more() {
let (code, all, stderr) = run_in(vec!["-a"]);
let (_, kernel, _) = run_in(vec!["-s"]);
assert_eq!((code, stderr.as_str()), (0, ""));
let kernel = kernel.trim_end();
assert!(all.starts_with(kernel), "-a output {all:?} must start with {kernel:?}");
assert!(all.trim_end().len() > kernel.len(), "-a must print more fields than -s");
}
#[test]
fn processor_prints_unknown() {
let (code, stdout, stderr) = run_in(vec!["-p"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "unknown\n", ""));
}
}
+27
View File
@@ -0,0 +1,27 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/whoami), patched to route
# output through pi-uutils-ctx so it can run in-process as a shell builtin. See
# src/whoami.rs for the patch markers (`pi-uutils:` comments).
[package]
name = "uu_whoami"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "whoami ~ (uutils) display user name of current effective user ID (vendored + patched for in-process embedding)"
[lib]
path = "src/whoami.rs"
[dependencies]
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
uucore = { version = "0.8.0", features = ["entries", "process"] }
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[target.'cfg(target_os = "windows")'.dependencies]
windows-sys = { version = "0.61.0", features = [
"Win32_NetworkManagement_NetManagement",
"Win32_System_WindowsProgramming",
"Win32_Foundation",
] }
[dev-dependencies]
parking_lot = "0.12"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+198
View File
@@ -0,0 +1,198 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore (ToDO) getusername
// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process
// as a shell builtin. Output goes to the context stdout (upstream's
// `println_verbatim` writes to the process stdout), `translate!` strings are
// literalized, the `platform` module (upstream
// src/platform/{mod,unix,windows}.rs) is inlined, and the entry point no longer
// calls `std::process::exit`.
use std::{ffi::OsString, io::Write};
use clap::Command;
use uucore::error::{FromIo, UResult, USimpleError};
// pi-uutils: inlined from upstream src/platform/{mod,unix,windows}.rs (verbatim
// bodies); the platform user lookup itself is process-global state and needs no
// scope patching.
mod platform {
#[cfg(unix)]
pub use self::unix::get_username;
#[cfg(windows)]
pub use self::windows::get_username;
#[cfg(unix)]
mod unix {
use std::{ffi::OsString, io};
use uucore::{entries::uid2usr, process::geteuid};
pub fn get_username() -> io::Result<OsString> {
// uid2usr should arguably return an OsString but currently doesn't
uid2usr(geteuid()).map(Into::into)
}
}
#[cfg(windows)]
mod windows {
use std::{ffi::OsString, io, os::windows::ffi::OsStringExt};
use windows_sys::Win32::{
NetworkManagement::NetManagement::UNLEN, System::WindowsProgramming::GetUserNameW,
};
pub fn get_username() -> io::Result<OsString> {
const BUF_LEN: u32 = UNLEN + 1;
let mut buffer = [0_u16; BUF_LEN as usize];
let mut len = BUF_LEN;
// SAFETY: buffer.len() == len
if unsafe { GetUserNameW(buffer.as_mut_ptr(), &raw mut len) } == 0 {
return Err(io::Error::last_os_error());
}
Ok(OsString::from_wide(&buffer[..len as usize - 1]))
}
}
}
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
/// arguments directly (without the uucore clap-localization helper that would
/// terminate the process), renders clap help/usage/version to the context
/// streams, and maps the `UResult` to an exit code, so it is safe to run inside
/// the host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
match uu_app().try_get_matches_from(argv) {
Ok(_matches) => {},
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
}
match whoami_main() {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
let msg = err.to_string();
if !msg.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "whoami: {msg}");
}
if code == 0 { 1 } else { code }
},
}
}
fn whoami_main() -> UResult<()> {
let username = whoami()?;
// pi-uutils: replacement for upstream's `println_verbatim` — writes the
// username bytes verbatim to the context stdout instead of the process
// stdout.
let mut out = pi_uutils_ctx::stdout();
out.write_all(uucore::os_str_as_bytes(&username)?)
.and_then(|()| out.write_all(b"\n"))
.and_then(|()| out.flush())
.map_err(|e| USimpleError::new(1, format!("failed to print username: {e}")))?;
Ok(())
}
/// Get the current username
pub fn whoami() -> UResult<OsString> {
platform::get_username().map_err_context(|| "failed to get username".to_string())
}
pub fn uu_app() -> Command {
Command::new("whoami")
.version(uucore::crate_version!())
.about("Print the current username.")
.override_usage("whoami")
.infer_long_args(true)
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc};
use parking_lot::Mutex;
use pi_uutils_ctx::ScopeIo;
use super::*;
fn run_in(args: Vec<&str>) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
self.buf.lock().write(buf)
}
fn flush(&mut self) -> std::io::Result<()> {
self.buf.lock().flush()
}
}
let io = ScopeIo {
stdin: Box::new(std::io::empty()),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd: PathBuf::from("."),
env: HashMap::new(),
cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)),
};
let argv: Vec<OsString> = std::iter::once("whoami")
.chain(args)
.map(OsString::from)
.collect();
let code = pi_uutils_ctx::scope(io, || run(argv));
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
(code, out_str, err_str)
}
#[test]
fn prints_process_user_with_trailing_newline() {
let (code, stdout, stderr) = run_in(vec![]);
assert_eq!((code, stderr.as_str()), (0, ""));
assert!(stdout.ends_with('\n'));
let name = stdout.trim_end();
assert!(!name.is_empty());
// When the host exports USER it names the same effective user the
// platform lookup resolves.
if let Ok(user) = std::env::var("USER") {
assert_eq!(name, user);
}
}
#[test]
fn rejects_operands() {
let (code, stdout, stderr) = run_in(vec!["extra"]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert!(!stderr.is_empty(), "clap usage error must go to scope stderr");
}
#[test]
fn help_renders_to_scope_stdout() {
let (code, stdout, stderr) = run_in(vec!["--help"]);
assert_eq!((code, stderr.as_str()), (0, ""));
assert!(stdout.contains("Print the current username."));
}
}
+23
View File
@@ -0,0 +1,23 @@
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/yes), patched to route
# output through pi-uutils-ctx, handle a closed consumer (broken pipe) as a
# clean in-process exit, and poll the scope cancel flag so it can run
# in-process as a shell builtin. See src/yes.rs for the patch markers
# (`pi-uutils:` comments).
[package]
name = "uu_yes"
version = "0.8.0"
edition = "2024"
license = "MIT"
description = "yes ~ (uutils) repeatedly display a line with STRING (or 'y') (vendored + patched for in-process embedding)"
[lib]
path = "src/yes.rs"
[dependencies]
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
itertools = "0.14.0"
uucore = "0.8.0"
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[dev-dependencies]
parking_lot = "0.12"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) uutils developers
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+351
View File
@@ -0,0 +1,351 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// cSpell:ignore strs
// pi-uutils: vendored from uutils/coreutils 0.8.0 and patched to run in-process
// as a shell builtin. All process-global stdio is routed through
// `pi_uutils_ctx`, `translate!` strings are literalized, and the entry point no
// longer calls `std::process::exit`. Because the utility runs inside the shell
// process there is no SIGPIPE to terminate it when the consumer closes, so a
// broken-pipe write error exits cleanly with code 0 (GNU behaviour) on every
// platform, and the output loop polls the scope cancel flag so shell
// abort/timeout stops it promptly.
use std::{
error::Error,
ffi::OsString,
io::{self, Write},
};
use clap::{Arg, ArgAction, Command, builder::ValueParser};
use pi_uutils_ctx::format_usage;
use uucore::error::strip_errno;
// it's possible that using a smaller or larger buffer might provide better
// performance on some systems, but honestly this is good enough
const BUF_SIZE: usize = 16 * 1024;
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
/// arguments directly (without the uucore clap-localization helper that would
/// terminate the process), renders clap help/usage/version to the context
/// streams, and maps the outcome to an exit code, so it is safe to run inside
/// the host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
let mut buffer = Vec::with_capacity(BUF_SIZE);
#[allow(clippy::unwrap_used, reason = "clap provides 'y' by default")]
let _ = args_into_buffer(&mut buffer, matches.get_many::<OsString>("STRING").unwrap());
prepare_buffer(&mut buffer);
match exec(&buffer) {
// pi-uutils: a broken pipe means the consumer closed its end; a
// process `yes` would die from SIGPIPE (or handle EPIPE on Windows),
// so the in-process builtin exits cleanly with 0 on every platform.
ExecStop::Io(err) if err.kind() == io::ErrorKind::BrokenPipe => 0,
ExecStop::Io(err) => {
let _ = writeln!(pi_uutils_ctx::stderr(), "yes: standard output: {}", strip_errno(&err));
1
},
// pi-uutils: the shell asked the scope to cancel (abort/timeout);
// there is no signal-style exit status in-process, so return 1.
ExecStop::Cancelled => 1,
}
}
pub fn uu_app() -> Command {
Command::new("yes")
.version(uucore::crate_version!())
.about("Repeatedly display a line with STRING (or 'y')")
.override_usage(format_usage("yes [STRING]..."))
.arg(
Arg::new("STRING")
.default_value("y")
.value_parser(ValueParser::os_string())
.action(ArgAction::Append),
)
.infer_long_args(true)
}
/// Copies words from `i` into `buf`, separated by spaces.
#[allow(clippy::unnecessary_wraps, reason = "needed on some platforms")]
fn args_into_buffer<'a>(
buf: &mut Vec<u8>,
i: impl Iterator<Item = &'a OsString>,
) -> Result<(), Box<dyn Error>> {
// On Unix (and wasi), OsStrs are just &[u8]'s underneath...
#[cfg(any(unix, target_os = "wasi"))]
{
#[cfg(unix)]
use std::os::unix::ffi::OsStrExt;
#[cfg(target_os = "wasi")]
use std::os::wasi::ffi::OsStrExt;
for part in itertools::intersperse(i.map(|a| a.as_bytes()), b" ") {
buf.extend_from_slice(part);
}
}
// But, on Windows, we must hop through a String.
#[cfg(not(any(unix, target_os = "wasi")))]
{
for part in itertools::intersperse(i.map(|a| a.to_str()), Some(" ")) {
let bytes = match part {
Some(part) => part.as_bytes(),
// pi-uutils: literalized `translate!("yes-error-invalid-utf8")`.
None => return Err("arguments contain invalid UTF-8".into()),
};
buf.extend_from_slice(bytes);
}
}
buf.push(b'\n');
Ok(())
}
/// Assumes buf holds a single output line forged from the command line
/// arguments, copies it repeatedly until the buffer holds as many copies as it
/// can under [`BUF_SIZE`].
fn prepare_buffer(buf: &mut Vec<u8>) {
let line_len = buf.len();
debug_assert!(line_len > 0, "buffer is not empty since we have newline");
let target_size = line_len * (BUF_SIZE / line_len); // 0 if line_len is already large enough
while buf.len() < target_size {
let to_copy = std::cmp::min(target_size - buf.len(), buf.len());
debug_assert_eq!(to_copy % line_len, 0);
buf.extend_from_within(..to_copy);
}
}
/// pi-uutils: why the output loop stopped. Upstream's `exec` only ever returns
/// an I/O error (the loop is infinite); in-process we also stop on scope
/// cancellation.
enum ExecStop {
Io(io::Error),
Cancelled,
}
/// pi-uutils: replacement for upstream's `exec` — writes to the context stdout
/// instead of the process stdout and polls the scope cancel flag every
/// iteration (each iteration writes a full [`BUF_SIZE`]-ish batch, so polling
/// per iteration is cheap) so shell abort/timeout stops the loop promptly.
fn exec(bytes: &[u8]) -> ExecStop {
let mut stdout = pi_uutils_ctx::stdout();
loop {
if pi_uutils_ctx::is_cancelled() {
return ExecStop::Cancelled;
}
if let Err(err) = stdout.write_all(bytes) {
return ExecStop::Io(err);
}
}
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, io::Write, path::PathBuf, sync::Arc};
use parking_lot::Mutex;
use pi_uutils_ctx::ScopeIo;
use super::*;
/// Writer that accepts up to `budget` bytes into a shared buffer, then
/// fails every further write with `fail_kind` — models a consumer that
/// closes the pipe after reading some output.
struct FailingWriter {
buf: Arc<Mutex<Vec<u8>>>,
budget: usize,
fail_kind: io::ErrorKind,
}
impl Write for FailingWriter {
fn write(&mut self, buf: &[u8]) -> io::Result<usize> {
if self.budget == 0 {
return Err(io::Error::new(self.fail_kind, "consumer gone"));
}
let n = buf.len().min(self.budget);
self.budget -= n;
self.buf.lock().extend_from_slice(&buf[..n]);
Ok(n)
}
fn flush(&mut self) -> io::Result<()> {
Ok(())
}
}
fn run_with(
args: Vec<&str>,
budget: usize,
fail_kind: io::ErrorKind,
cancelled: bool,
) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> io::Result<usize> {
self.buf.lock().write(buf)
}
fn flush(&mut self) -> io::Result<()> {
self.buf.lock().flush()
}
}
let io = ScopeIo {
stdin: Box::new(std::io::empty()),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(FailingWriter {
buf: stdout_buf.clone(),
budget,
fail_kind,
}),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd: PathBuf::from("."),
env: HashMap::new(),
cancel: Arc::new(std::sync::atomic::AtomicBool::new(cancelled)),
};
let argv: Vec<OsString> = std::iter::once("yes")
.chain(args)
.map(OsString::from)
.collect();
let code = pi_uutils_ctx::scope(io, || run(argv));
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
(code, out_str, err_str)
}
#[test]
fn broken_pipe_is_clean_exit() {
// Consumer takes 100 bytes then closes: exit 0, like GNU yes dying to
// SIGPIPE without an error status visible to the shell.
let (code, stdout, stderr) = run_with(vec![], 100, io::ErrorKind::BrokenPipe, false);
assert_eq!(code, 0);
assert!(stdout.starts_with("y\ny\n"), "expected default 'y' lines, got {stdout:?}");
assert_eq!(stdout.len(), 100);
assert_eq!(stderr, "");
}
#[test]
fn custom_operands_join_with_spaces_and_repeat() {
// Budget is a multiple of the line length ("hello world\n" = 12 bytes)
// so the captured output is whole lines.
let (code, stdout, stderr) =
run_with(vec!["hello", "world"], 12 * 100, io::ErrorKind::BrokenPipe, false);
assert_eq!(code, 0);
assert_eq!(stdout.lines().count(), 100);
for line in stdout.lines() {
assert_eq!(line, "hello world");
}
assert_eq!(stderr, "");
}
#[test]
fn cancellation_stops_loop_promptly() {
// Pre-set cancel flag: the loop must observe it and return 1 before
// writing anything. The finite write budget is a backstop so a broken
// cancel path fails the test (as exit 0) instead of hanging forever.
let (code, stdout, stderr) = run_with(vec![], 1 << 20, io::ErrorKind::BrokenPipe, true);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert_eq!(stderr, "");
}
#[test]
fn non_pipe_write_error_reports_and_fails() {
let (code, stdout, stderr) = run_with(vec![], 2, io::ErrorKind::Other, false);
assert_eq!(code, 1);
assert_eq!(stdout, "y\n");
assert_eq!(stderr, "yes: standard output: consumer gone\n");
}
#[test]
fn help_renders_to_scope_stdout() {
let (code, stdout, stderr) = run_with(vec!["--help"], 1 << 20, io::ErrorKind::Other, false);
assert_eq!(code, 0);
assert!(stdout.contains("Usage:"));
assert!(stdout.contains("Repeatedly display a line"));
assert_eq!(stderr, "");
}
// Upstream unit tests (uutils/coreutils 0.8.0), kept verbatim apart from
// indentation.
#[test]
fn test_prepare_buffer() {
let tests = [
(150, 16350),
(1000, 16000),
(4093, 16372),
(4099, 12297),
(4111, 12333),
(2, 16384),
(3, 16383),
(4, 16384),
(5, 16380),
(8192, 16384),
(8191, 16382),
(8193, 8193),
(10000, 10000),
(15000, 15000),
(25000, 25000),
];
for (line, final_len) in tests {
let mut v = std::iter::repeat_n(b'a', line).collect::<Vec<_>>();
prepare_buffer(&mut v);
assert_eq!(v.len(), final_len);
}
}
#[test]
fn test_args_into_buf() {
{
let mut v = Vec::with_capacity(BUF_SIZE);
let default_args = ["y".into()];
args_into_buffer(&mut v, default_args.iter()).unwrap();
assert_eq!(String::from_utf8(v).unwrap(), "y\n");
}
{
let mut v = Vec::with_capacity(BUF_SIZE);
let args = ["foo".into()];
args_into_buffer(&mut v, args.iter()).unwrap();
assert_eq!(String::from_utf8(v).unwrap(), "foo\n");
}
{
let mut v = Vec::with_capacity(BUF_SIZE);
let args = ["foo".into(), "bar baz".into(), "qux".into()];
args_into_buffer(&mut v, args.iter()).unwrap();
assert_eq!(String::from_utf8(v).unwrap(), "foo bar baz qux\n");
}
}
}
+5
View File
@@ -2,6 +2,11 @@
## [Unreleased]
### Added
- Added an in-process `readlink` shell builtin (vendored from uutils coreutils 0.8.0), supporting `-f`/`-e`/`-m` canonicalization, `-n`/`-z` delimiters, and `-v`/`-q`/`-s` verbosity, with path operands resolved against the shell working directory.
- Added in-process shell builtins for `realpath`, `touch`, `stat`, `date`, `mktemp`, `seq`, `yes`, `printenv`, `ln`, `truncate`, `tac`, `nproc`, `uname`, `whoami`, and `hostname` (vendored from uutils coreutils 0.8.0), plus native `which` (shell PATH lookup) and `diff` (unified output, `-U`/`-q`/`-N`, binary detection, recursive directory compare) builtins. All resolve path operands against the shell working directory, read the shell's exported environment, and honor abort/timeout cancellation; `ln` is gated with the destructive set (`PI_DISABLE_UUTILS_DESTRUCTIVE`), and system-mutating modes (`date --set`, hostname setting) are disabled.
## [16.4.5] - 2026-07-11
### Added