feat(shell): integrated additional coreutils into shell builtins

- Added base64, checksum utilities (md5, sha1, sha224, sha256, sha384, sha512, b2sum), path utilities (basename, dirname), and text processing tools (cut, tee, tr, paste, comm) to the shell.
- Registered new utility builtins in crates/pi-shell/src/coreutils.rs and crates/pi-shell/src/shell.rs.
- Updated Cargo.toml to include the necessary dependencies for the new core utilities.
This commit is contained in:
can1357
2026-07-11 18:36:56 +02:00
parent 8a510052af
commit 5a73d65b7a
66 changed files with 18167 additions and 4369 deletions
Generated
+634 -2
View File
@@ -152,6 +152,12 @@ dependencies = [
"nix 0.24.3",
]
[[package]]
name = "arrayref"
version = "0.3.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "76a2e8124351fda1ef8aaaa3bbd7ebbcb486bbcd4225aca0aa0d84bb2db8fecb"
[[package]]
name = "arrayvec"
version = "0.7.8"
@@ -213,6 +219,16 @@ version = "0.22.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6"
[[package]]
name = "base64-simd"
version = "0.8.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "339abbe78e73178762e23bea9dfd08e697eb3f3301cd4be981c0f78ba5859195"
dependencies = [
"outref",
"vsimd",
]
[[package]]
name = "bigdecimal"
version = "0.4.10"
@@ -283,6 +299,49 @@ dependencies = [
"wyz",
]
[[package]]
name = "blake2b_simd"
version = "1.0.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b79834656f71332577234b50bfc009996f7449e0c056884e6a02492ded0ca2f3"
dependencies = [
"arrayref",
"arrayvec",
"constant_time_eq",
]
[[package]]
name = "blake3"
version = "1.8.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0aa83c34e62843d924f905e0f5c866eb1dd6545fc4d719e803d9ba6030371fce"
dependencies = [
"arrayref",
"arrayvec",
"cc",
"cfg-if",
"constant_time_eq",
"cpufeatures 0.3.0",
]
[[package]]
name = "block-buffer"
version = "0.10.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71"
dependencies = [
"generic-array",
]
[[package]]
name = "block-buffer"
version = "0.12.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d2f6c7dbe95a6ed67ad9f18e57daf93a2f034c524b99fd2b76d18fdfeb6660aa"
dependencies = [
"hybrid-array",
]
[[package]]
name = "bon"
version = "3.9.3"
@@ -572,7 +631,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d524456ba66e72eb8b115ff89e01e497f8e6d11d78b70b1aa13c0fbd97540a81"
dependencies = [
"cfg-if",
"cpufeatures",
"cpufeatures 0.3.0",
"rand_core 0.10.1",
]
@@ -658,6 +717,12 @@ dependencies = [
"error-code",
]
[[package]]
name = "codesnake"
version = "0.2.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2205f7f6d3de68ecf4c291c789b3edf07b6569268abd0188819086f71ae42225"
[[package]]
name = "color-print"
version = "0.3.7"
@@ -719,6 +784,12 @@ dependencies = [
"windows-sys 0.61.2",
]
[[package]]
name = "const-oid"
version = "0.10.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a6ef517f0926dd24a1582492c791b6a4818a4d94e789a334894aa15b0d12f55c"
[[package]]
name = "const-random"
version = "0.1.18"
@@ -739,6 +810,12 @@ dependencies = [
"tiny-keccak",
]
[[package]]
name = "constant_time_eq"
version = "0.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3d52eff69cd5e647efe296129160853a42795992097e8af39800e1060caeea9b"
[[package]]
name = "convert_case"
version = "0.11.0"
@@ -763,6 +840,15 @@ dependencies = [
"libm",
]
[[package]]
name = "cpufeatures"
version = "0.2.17"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280"
dependencies = [
"libc",
]
[[package]]
name = "cpufeatures"
version = "0.3.0"
@@ -772,6 +858,16 @@ dependencies = [
"libc",
]
[[package]]
name = "crc-fast"
version = "1.10.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e75b2483e97a5a7da73ac68a05b629f9c53cff58d8ed1c77866079e18b00dba5"
dependencies = [
"digest 0.10.7",
"spin 0.10.0",
]
[[package]]
name = "crc32fast"
version = "1.5.0"
@@ -812,6 +908,25 @@ version = "0.2.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5"
[[package]]
name = "crypto-common"
version = "0.1.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a"
dependencies = [
"generic-array",
"typenum",
]
[[package]]
name = "crypto-common"
version = "0.2.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ce6e4c961d6cd6c9a86db418387425e8bdeaf05b3c8bc1411e6dca4c252f1453"
dependencies = [
"hybrid-array",
]
[[package]]
name = "ctor"
version = "1.0.8"
@@ -901,6 +1016,32 @@ dependencies = [
"parking_lot_core",
]
[[package]]
name = "data-encoding"
version = "2.11.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a4ae5f15dda3c708c0ade84bfee31ccab44a3da4f88015ed22f63732abe300c8"
[[package]]
name = "data-encoding-macro"
version = "0.1.20"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3259c913752a86488b501ed8680446a5ed2d5aeac6e596cb23ba3800768ea32c"
dependencies = [
"data-encoding",
"data-encoding-macro-internal",
]
[[package]]
name = "data-encoding-macro-internal"
version = "0.1.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ccc2776f0c61eca1ca32528f85548abd1a4be8fb53d1b21c013e4f18da1e7090"
dependencies = [
"data-encoding",
"syn",
]
[[package]]
name = "defmt"
version = "1.1.1"
@@ -932,6 +1073,27 @@ dependencies = [
"thiserror 2.0.18",
]
[[package]]
name = "digest"
version = "0.10.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292"
dependencies = [
"block-buffer 0.10.4",
"crypto-common 0.1.7",
]
[[package]]
name = "digest"
version = "0.11.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f1dd6dbb5841937940781866fa1281a1ff7bd3bf827091440879f9994983d5c2"
dependencies = [
"block-buffer 0.12.1",
"const-oid",
"crypto-common 0.2.2",
]
[[package]]
name = "dispatch2"
version = "0.3.1"
@@ -965,6 +1127,12 @@ version = "1.0.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "92773504d58c093f6de2459af4af33faa518c13451eb8f2b5698ed3d36e7c813"
[[package]]
name = "dyn-clone"
version = "1.0.20"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555"
[[package]]
name = "either"
version = "1.16.0"
@@ -1050,6 +1218,17 @@ dependencies = [
"regex-syntax",
]
[[package]]
name = "fancy-regex"
version = "0.18.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e1e1dacd0d2082dfcf1351c4bdd566bbe89a2b263235a2b50058f1e130a47277"
dependencies = [
"bit-set",
"regex-automata",
"regex-syntax",
]
[[package]]
name = "fast-srgb8"
version = "1.0.0"
@@ -1185,7 +1364,7 @@ dependencies = [
"futures-core",
"futures-sink",
"nanorand",
"spin",
"spin 0.9.8",
]
[[package]]
@@ -1330,6 +1509,16 @@ dependencies = [
"slab",
]
[[package]]
name = "generic-array"
version = "0.14.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a"
dependencies = [
"typenum",
"version_check",
]
[[package]]
name = "gethostname"
version = "1.1.0"
@@ -1542,6 +1731,12 @@ version = "0.4.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7f24254aa9a54b5c858eaee2f5bccdb46aaf0e486a595ed5fd8f86ba55232a70"
[[package]]
name = "hifijson"
version = "0.2.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0a7763b98ba8a24f59e698bf9ab197e7676c640d6455d1580b4ce7dc560f0f0d"
[[package]]
name = "hostname"
version = "0.4.2"
@@ -1586,6 +1781,15 @@ dependencies = [
"markup5ever",
]
[[package]]
name = "hybrid-array"
version = "0.4.13"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "818356c5132c1fede50f837ca96afbe78ff42413047f4abb886217845e1b6c8c"
dependencies = [
"typenum",
]
[[package]]
name = "iana-time-zone"
version = "0.1.65"
@@ -2066,6 +2270,63 @@ version = "1.0.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
[[package]]
name = "jaq"
version = "2.3.0"
dependencies = [
"codesnake",
"hifijson",
"jaq-core",
"jaq-json",
"jaq-std",
"memmap2",
"parking_lot",
"pi-uutils-ctx",
"tempfile",
"unicode-width 0.1.14",
"yansi",
]
[[package]]
name = "jaq-core"
version = "2.2.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "77526a72eb79412c29fd141767a6549bbfcb1cb40e00556fe16532d5e878e098"
dependencies = [
"dyn-clone",
"once_cell",
"typed-arena",
]
[[package]]
name = "jaq-json"
version = "1.1.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "01dbdbd07b076e8403abac68ce7744d93e2ecd953bbc44bf77bf00e1e81172bc"
dependencies = [
"foldhash 0.1.5",
"hifijson",
"indexmap",
"jaq-core",
"jaq-std",
]
[[package]]
name = "jaq-std"
version = "2.1.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2c264fe397c981705976c71f1bfe020382b9eda52ae950e57fe885e147bdd67d"
dependencies = [
"aho-corasick",
"base64",
"chrono",
"jaq-core",
"libm",
"log",
"regex-lite",
"urlencoding",
]
[[package]]
name = "jiff"
version = "0.2.32"
@@ -2140,6 +2401,15 @@ dependencies = [
"wasm-bindgen",
]
[[package]]
name = "keccak"
version = "0.1.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "cb26cec98cce3a3d96cbb7bced3c4b16e3d13f27ec56dbd62cbc8f39cfb9d653"
dependencies = [
"cpufeatures 0.2.17",
]
[[package]]
name = "kqueue"
version = "1.2.0"
@@ -2260,6 +2530,16 @@ dependencies = [
"web_atoms",
]
[[package]]
name = "md-5"
version = "0.10.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d89e7ee0cfbedfc4da3340218492196241d89eefb6dab27de5df917a6d2e78cf"
dependencies = [
"cfg-if",
"digest 0.10.7",
]
[[package]]
name = "memchr"
version = "2.8.3"
@@ -2689,6 +2969,12 @@ dependencies = [
"windows-sys 0.61.2",
]
[[package]]
name = "outref"
version = "0.5.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1a80800c0488c3a21695ea981a54918fbb37abf04f4d0720c453632255e2ff0e"
[[package]]
name = "palette"
version = "0.7.6"
@@ -3101,6 +3387,7 @@ dependencies = [
"flume",
"globset",
"ignore",
"jaq",
"libc",
"os_pipe",
"parking_lot",
@@ -3114,17 +3401,34 @@ dependencies = [
"tokio",
"tokio-util",
"toml",
"uu_b2sum",
"uu_base64",
"uu_basename",
"uu_cat",
"uu_comm",
"uu_cut",
"uu_dirname",
"uu_find",
"uu_head",
"uu_ls",
"uu_md5sum",
"uu_mkdir",
"uu_mv",
"uu_paste",
"uu_rm",
"uu_sed",
"uu_sha1sum",
"uu_sha224sum",
"uu_sha256sum",
"uu_sha384sum",
"uu_sha512sum",
"uu_sort",
"uu_tail",
"uu_tee",
"uu_tr",
"uu_uniq",
"uu_wc",
"uu_xargs",
"windows-sys 0.61.2",
"winreg 0.56.0",
"xxhash-rust",
@@ -3504,6 +3808,12 @@ dependencies = [
"regex-syntax",
]
[[package]]
name = "regex-lite"
version = "0.1.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "cab834c73d247e67f4fae452806d17d3c7501756d98c8808d7c9c7aa7d18f973"
[[package]]
name = "regex-syntax"
version = "0.8.11"
@@ -3713,6 +4023,38 @@ dependencies = [
"windows-sys 0.61.2",
]
[[package]]
name = "sha1"
version = "0.10.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a978451301f4db1d02937a4ab3ccce137717b81826e79b7d49ffe3244a13c3b8"
dependencies = [
"cfg-if",
"cpufeatures 0.2.17",
"digest 0.10.7",
]
[[package]]
name = "sha2"
version = "0.10.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283"
dependencies = [
"cfg-if",
"cpufeatures 0.2.17",
"digest 0.10.7",
]
[[package]]
name = "sha3"
version = "0.10.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "77fd7028345d415a4034cf8777cd4f8ab1851274233b45f84e3d955502d93874"
dependencies = [
"digest 0.10.7",
"keccak",
]
[[package]]
name = "shared_library"
version = "0.1.9"
@@ -3778,6 +4120,15 @@ version = "0.4.12"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5"
[[package]]
name = "sm3"
version = "0.5.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "da6a89ba31723d185fd7413b98c576a575f356d9b84729d8ecb6ead60000a5b6"
dependencies = [
"digest 0.11.3",
]
[[package]]
name = "smallvec"
version = "1.15.2"
@@ -3806,6 +4157,12 @@ dependencies = [
"lock_api",
]
[[package]]
name = "spin"
version = "0.10.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d5fe4ccb98d9c292d56fec89a5e07da7fc4cf0dc11e156b41793132775d3e591"
[[package]]
name = "stable_deref_trait"
version = "1.2.1"
@@ -4802,6 +5159,18 @@ dependencies = [
"rustc-hash 2.1.3",
]
[[package]]
name = "typed-arena"
version = "2.0.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6af6ae20167a9ece4bcb41af5b80f8a1f1df981f6391189ce00fd257af04126a"
[[package]]
name = "typenum"
version = "1.20.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20"
[[package]]
name = "ucd-trie"
version = "0.1.7"
@@ -4856,6 +5225,12 @@ version = "0.5.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "81e544489bf3d8ef66c953931f56617f423cd4b5494be343d9b9d3dda037b9a3"
[[package]]
name = "urlencoding"
version = "2.1.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "daf8dba3b7eb870caf1ddeed7bc9d2a049f3cfdfae7cb521b087cc33ae4c49da"
[[package]]
name = "utf16_iter"
version = "1.0.5"
@@ -4883,6 +5258,44 @@ version = "0.2.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821"
[[package]]
name = "uu_b2sum"
version = "0.8.0"
dependencies = [
"clap",
"pi-uutils-ctx",
"uu_checksum_common",
"uucore 0.8.0",
]
[[package]]
name = "uu_base32"
version = "0.8.0"
dependencies = [
"clap",
"pi-uutils-ctx",
"uucore 0.8.0",
]
[[package]]
name = "uu_base64"
version = "0.8.0"
dependencies = [
"clap",
"pi-uutils-ctx",
"uu_base32",
"uucore 0.8.0",
]
[[package]]
name = "uu_basename"
version = "0.8.0"
dependencies = [
"clap",
"pi-uutils-ctx",
"uucore 0.8.0",
]
[[package]]
name = "uu_cat"
version = "0.8.0"
@@ -4894,6 +5307,48 @@ dependencies = [
"uucore 0.8.0",
]
[[package]]
name = "uu_checksum_common"
version = "0.8.0"
dependencies = [
"base64-simd",
"clap",
"hex",
"os_display",
"pi-uutils-ctx",
"uucore 0.8.0",
]
[[package]]
name = "uu_comm"
version = "0.8.0"
dependencies = [
"clap",
"pi-uutils-ctx",
"uucore 0.8.0",
]
[[package]]
name = "uu_cut"
version = "0.8.0"
dependencies = [
"bstr",
"clap",
"memchr",
"pi-uutils-ctx",
"uucore 0.8.0",
]
[[package]]
name = "uu_dirname"
version = "0.8.0"
dependencies = [
"clap",
"parking_lot",
"pi-uutils-ctx",
"uucore 0.8.0",
]
[[package]]
name = "uu_find"
version = "0.8.0"
@@ -4939,6 +5394,15 @@ dependencies = [
"uutils_term_grid",
]
[[package]]
name = "uu_md5sum"
version = "0.8.0"
dependencies = [
"clap",
"uu_checksum_common",
"uucore 0.8.0",
]
[[package]]
name = "uu_mkdir"
version = "0.8.0"
@@ -4964,6 +5428,15 @@ dependencies = [
"windows-sys 0.61.2",
]
[[package]]
name = "uu_paste"
version = "0.8.0"
dependencies = [
"clap",
"pi-uutils-ctx",
"uucore 0.8.0",
]
[[package]]
name = "uu_rm"
version = "0.8.0"
@@ -4977,6 +5450,71 @@ dependencies = [
"windows-sys 0.61.2",
]
[[package]]
name = "uu_sed"
version = "0.1.1"
dependencies = [
"clap",
"fancy-regex 0.18.0",
"memchr",
"memmap2",
"parking_lot",
"pi-uutils-ctx",
"regex",
"tempfile",
"uucore 0.9.0",
]
[[package]]
name = "uu_sha1sum"
version = "0.8.0"
dependencies = [
"clap",
"pi-uutils-ctx",
"uu_checksum_common",
"uucore 0.8.0",
]
[[package]]
name = "uu_sha224sum"
version = "0.8.0"
dependencies = [
"clap",
"pi-uutils-ctx",
"uu_checksum_common",
"uucore 0.8.0",
]
[[package]]
name = "uu_sha256sum"
version = "0.8.0"
dependencies = [
"clap",
"pi-uutils-ctx",
"uu_checksum_common",
"uucore 0.8.0",
]
[[package]]
name = "uu_sha384sum"
version = "0.8.0"
dependencies = [
"clap",
"pi-uutils-ctx",
"uu_checksum_common",
"uucore 0.8.0",
]
[[package]]
name = "uu_sha512sum"
version = "0.8.0"
dependencies = [
"clap",
"pi-uutils-ctx",
"uu_checksum_common",
"uucore 0.8.0",
]
[[package]]
name = "uu_sort"
version = "0.8.0"
@@ -5014,6 +5552,26 @@ dependencies = [
"windows-sys 0.61.2",
]
[[package]]
name = "uu_tee"
version = "0.8.0"
dependencies = [
"clap",
"pi-uutils-ctx",
"uucore 0.8.0",
]
[[package]]
name = "uu_tr"
version = "0.8.0"
dependencies = [
"bytecount",
"clap",
"nom 8.0.0",
"pi-uutils-ctx",
"uucore 0.8.0",
]
[[package]]
name = "uu_uniq"
version = "0.8.0"
@@ -5037,6 +5595,17 @@ dependencies = [
"uucore 0.8.0",
]
[[package]]
name = "uu_xargs"
version = "0.8.0"
dependencies = [
"clap",
"libc",
"parking_lot",
"pi-uutils-ctx",
"tempfile",
]
[[package]]
name = "uucore"
version = "0.0.30"
@@ -5065,14 +5634,22 @@ version = "0.8.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "07d779636d827cde4100f0e65ff3fd23b0b1f1195055475c6e6813d425f30c8e"
dependencies = [
"base64-simd",
"bigdecimal",
"blake2b_simd",
"blake3",
"bstr",
"clap",
"crc-fast",
"data-encoding",
"data-encoding-macro",
"digest 0.10.7",
"dunce",
"fluent",
"fluent-bundle",
"fluent-syntax",
"glob",
"hex",
"icu_calendar",
"icu_collator",
"icu_datetime",
@@ -5083,12 +5660,18 @@ dependencies = [
"jiff",
"jiff-icu",
"libc",
"md-5",
"memchr",
"nix 0.31.3",
"num-traits",
"os_display",
"procfs",
"rustc-hash 2.1.3",
"rustix",
"sha1",
"sha2",
"sha3",
"sm3",
"thiserror 2.0.18",
"unic-langid",
"unit-prefix",
@@ -5097,6 +5680,27 @@ dependencies = [
"winapi-util",
"windows-sys 0.61.2",
"xattr",
"z85",
]
[[package]]
name = "uucore"
version = "0.9.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "069b34217c27f611e1589f540f58118dbf226a9e407d38ab472052ff075a1dc2"
dependencies = [
"clap",
"fluent",
"fluent-syntax",
"libc",
"nix 0.31.3",
"os_display",
"rustc-hash 2.1.3",
"rustix",
"thiserror 2.0.18",
"unic-langid",
"uucore_procs 0.9.0",
"wild",
]
[[package]]
@@ -5120,6 +5724,16 @@ dependencies = [
"quote",
]
[[package]]
name = "uucore_procs"
version = "0.9.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "34337da211e7abfff7189b794afb3b5018fe356fb36474b73645458fc1201350"
dependencies = [
"proc-macro2",
"quote",
]
[[package]]
name = "uuhelp_parser"
version = "0.0.30"
@@ -5161,6 +5775,12 @@ version = "0.9.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
[[package]]
name = "vsimd"
version = "0.8.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5c3082ca00d5a5ef149bb8b555a72ae84c9c59f7250f013ac822ac2e49b19c64"
[[package]]
name = "walkdir"
version = "2.5.0"
@@ -5829,6 +6449,12 @@ dependencies = [
"linked-hash-map",
]
[[package]]
name = "yansi"
version = "1.0.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "cfe53a6657fd280eaa890a3bc59152892ffa3e30101319d168b781ed6529b049"
[[package]]
name = "yoke"
version = "0.8.3"
@@ -5852,6 +6478,12 @@ dependencies = [
"synstructure",
]
[[package]]
name = "z85"
version = "3.0.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c6e61e59a957b7ccee15d2049f86e8bfd6f66968fcd88f018950662d9b86e675"
[[package]]
name = "zerocopy"
version = "0.8.54"
+1 -1
View File
@@ -273,7 +273,7 @@ mod tests {
d
}
/// `CF_DIBV5` as PixPin (Qt) places it, after arboard's
/// `CF_DIBV5` as `PixPin` (Qt) places it, after arboard's
/// `maybe_tweak_header` rewrite: a 124-byte `BITMAPV5HEADER` carrying
/// `BI_BITFIELDS` compression with the BGRA masks embedded in the header
/// and pixels immediately after it. This is the exact buffer shape that
+2 -3
View File
@@ -12,8 +12,7 @@ use pi_shell::{
MinimizerResult as CoreMinimizerResult, Shell as CoreShell,
ShellExecuteOptions as CoreShellExecuteOptions, ShellOptions as CoreShellOptions,
ShellRunOptions as CoreShellRunOptions, ShellRunResult as CoreShellRunResult,
execute_shell as core_execute_shell,
minimizer,
execute_shell as core_execute_shell, minimizer,
};
use crate::task;
@@ -372,7 +371,7 @@ mod tests {
/// the pre-fix bridge (`flume::unbounded` + fire-and-forget
/// `ThreadsafeFunctionCallMode::NonBlocking`) the same harness accumulates
/// the producer's entire surplus in the queue (measured: a 32 MiB stream
/// queued all 33_554_432 bytes while the consumer stalled).
/// queued all `33_554_432` bytes while the consumer stalled).
#[tokio::test(flavor = "multi_thread")]
async fn bridge_pump_bounds_queue_and_delivers_all_bytes() {
const CHUNKS: usize = 512;
+18
View File
@@ -44,6 +44,24 @@ uu_find = { path = "../vendor/uu-find" }
pi_uu_grep = { path = "../pi-uu-grep" }
uu_cat = { path = "../vendor/uu-cat" }
uu_uniq = { path = "../vendor/uu-uniq" }
uu_base64 = { path = "../vendor/uu-base64" }
uu_md5sum = { path = "../vendor/uu-md5sum" }
uu_sha1sum = { path = "../vendor/uu-sha1sum" }
uu_sha224sum = { path = "../vendor/uu-sha224sum" }
uu_sha256sum = { path = "../vendor/uu-sha256sum" }
uu_sha384sum = { path = "../vendor/uu-sha384sum" }
uu_sha512sum = { path = "../vendor/uu-sha512sum" }
uu_b2sum = { path = "../vendor/uu-b2sum" }
uu_basename = { path = "../vendor/uu-basename" }
uu_dirname = { path = "../vendor/uu-dirname" }
uu_cut = { path = "../vendor/uu-cut" }
uu_tee = { path = "../vendor/uu-tee" }
uu_tr = { path = "../vendor/uu-tr" }
uu_paste = { path = "../vendor/uu-paste" }
uu_comm = { path = "../vendor/uu-comm" }
uu_sed = { path = "../vendor/uu-sed" }
uu_xargs = { path = "../vendor/uu-xargs" }
jaq = { path = "../vendor/jaq" }
[target.'cfg(unix)'.dependencies]
libc.workspace = true
+18
View File
@@ -209,6 +209,24 @@ uutil_builtin!(pub fn rm_builtin => uu_rm::run);
uutil_builtin!(pub fn mv_builtin => uu_mv::run);
uutil_builtin!(pub fn cat_builtin => uu_cat::run);
uutil_builtin!(pub fn uniq_builtin => uu_uniq::run);
uutil_builtin!(pub fn base64_builtin => uu_base64::run);
uutil_builtin!(pub fn md5sum_builtin => uu_md5sum::run);
uutil_builtin!(pub fn sha1sum_builtin => uu_sha1sum::run);
uutil_builtin!(pub fn sha224sum_builtin => uu_sha224sum::run);
uutil_builtin!(pub fn sha256sum_builtin => uu_sha256sum::run);
uutil_builtin!(pub fn sha384sum_builtin => uu_sha384sum::run);
uutil_builtin!(pub fn sha512sum_builtin => uu_sha512sum::run);
uutil_builtin!(pub fn b2sum_builtin => uu_b2sum::run);
uutil_builtin!(pub fn basename_builtin => uu_basename::run);
uutil_builtin!(pub fn dirname_builtin => uu_dirname::run);
uutil_builtin!(pub fn cut_builtin => uu_cut::run);
uutil_builtin!(pub fn tee_builtin => uu_tee::run);
uutil_builtin!(pub fn tr_builtin => uu_tr::run);
uutil_builtin!(pub fn paste_builtin => uu_paste::run);
uutil_builtin!(pub fn comm_builtin => uu_comm::run);
uutil_builtin!(pub fn sed_builtin => uu_sed::run);
uutil_builtin!(pub fn xargs_builtin => uu_xargs::run);
uutil_builtin!(pub fn jq_builtin => jaq::run);
#[cfg(test)]
mod tests {
+1 -1
View File
@@ -523,7 +523,7 @@ mod tests {
};
static CONFIG_COUNTER: AtomicUsize = AtomicUsize::new(0);
pub(crate) static TEST_LOCK: parking_lot::Mutex<()> = parking_lot::Mutex::new(());
pub static TEST_LOCK: parking_lot::Mutex<()> = parking_lot::Mutex::new(());
use super::*;
use crate::minimizer::MinimizerOptions;
+18
View File
@@ -622,6 +622,24 @@ async fn create_session_for_run(
shell.register_builtin("fd", crate::fd::fd_builtin());
shell.register_builtin("cat", crate::coreutils::cat_builtin());
shell.register_builtin("uniq", crate::coreutils::uniq_builtin());
shell.register_builtin("base64", crate::coreutils::base64_builtin());
shell.register_builtin("md5sum", crate::coreutils::md5sum_builtin());
shell.register_builtin("sha1sum", crate::coreutils::sha1sum_builtin());
shell.register_builtin("sha224sum", crate::coreutils::sha224sum_builtin());
shell.register_builtin("sha256sum", crate::coreutils::sha256sum_builtin());
shell.register_builtin("sha384sum", crate::coreutils::sha384sum_builtin());
shell.register_builtin("sha512sum", crate::coreutils::sha512sum_builtin());
shell.register_builtin("b2sum", crate::coreutils::b2sum_builtin());
shell.register_builtin("basename", crate::coreutils::basename_builtin());
shell.register_builtin("dirname", crate::coreutils::dirname_builtin());
shell.register_builtin("cut", crate::coreutils::cut_builtin());
shell.register_builtin("tee", crate::coreutils::tee_builtin());
shell.register_builtin("tr", crate::coreutils::tr_builtin());
shell.register_builtin("paste", crate::coreutils::paste_builtin());
shell.register_builtin("comm", crate::coreutils::comm_builtin());
shell.register_builtin("sed", crate::coreutils::sed_builtin());
shell.register_builtin("xargs", crate::coreutils::xargs_builtin());
shell.register_builtin("jq", crate::coreutils::jq_builtin());
if !uutils_env_disabled(config, "PI_DISABLE_UUTILS_DESTRUCTIVE") {
if !uutils_env_disabled(config, "PI_DISABLE_RM_BUILTIN") {
shell.register_builtin("rm", crate::coreutils::rm_builtin());
+17
View File
@@ -188,6 +188,23 @@ pub fn var(key: &str) -> Option<String> {
.and_then(|ctx| ctx.env.get(key).cloned())
})
}
/// Returns a snapshot of the scope's entire environment map (the shell's
/// exported variables), or an empty vector when no scope is installed.
/// Utilities that spawn child processes use this to build the child
/// environment (`env_clear().envs(..)`), because the shell's exported
/// variables are not present in the host process environment.
#[must_use]
pub fn env_snapshot() -> Vec<(String, String)> {
CTX.with(|c| {
c.borrow().as_ref().map_or_else(Vec::new, |ctx| {
ctx.env
.iter()
.map(|(k, v)| (k.clone(), v.clone()))
.collect()
})
})
}
/// Returns true when scoped stdin is a shell pipe or custom stream that should
/// be treated as `rg PATTERN`'s implicit input instead of searching `.`.
#[must_use]
+26
View File
@@ -0,0 +1,26 @@
[package]
name = "jaq"
version = "2.3.0"
edition = "2024"
license = "MIT"
[lib]
path = "src/lib.rs"
[dependencies]
# Interpreter libraries from crates.io, pinned as upstream jaq v2.3.0 pins them.
jaq-core = "2.1.1"
jaq-std = "2.1.0"
jaq-json = "1.1.1"
codesnake = "0.2"
hifijson = "0.2.0"
memmap2 = "0.9"
tempfile = "3.3.0"
unicode-width = "0.1.13"
yansi = "1.0.1"
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[dev-dependencies]
tempfile = "3"
parking_lot = "0.12"
+23
View File
@@ -0,0 +1,23 @@
Permission is hereby granted, free of charge, to any
person obtaining a copy of this software and associated
documentation files (the "Software"), to deal in the
Software without restriction, including without
limitation the rights to use, copy, modify, merge,
publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software
is furnished to do so, subject to the following
conditions:
The above copyright notice and this permission notice
shall be included in all copies or substantial portions
of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF
ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED
TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A
PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT
SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY
CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION
OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR
IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
DEALINGS IN THE SOFTWARE.
+236
View File
@@ -0,0 +1,236 @@
//! Command-line argument parsing
use core::fmt;
use std::{ffi::OsString, path::PathBuf};
/// Remaining arguments; upstream used `std::env::ArgsOs`, but as an in-process
/// builtin the argv comes from the host, not the process.
type Args = std::vec::IntoIter<OsString>;
#[derive(Debug, Default)]
pub struct Cli {
// Input options
pub null_input: bool,
/// When the option `--slurp` is used additionally,
/// then the whole input is read into a single string.
pub raw_input: bool,
/// When input is read from files,
/// jaq yields an array for each file, whereas
/// jq produces only a single array.
pub slurp: bool,
// Output options
pub compact_output: bool,
pub raw_output: bool,
/// This flag enables `--raw-output`.
pub join_output: bool,
pub in_place: bool,
pub sort_keys: bool,
pub color_output: bool,
pub monochrome_output: bool,
pub tab: bool,
pub indent: usize,
// Compilation options
pub from_file: bool,
/// If this option is given multiple times, all given directories are
/// searched.
pub library_path: Vec<PathBuf>,
// Key-value options
pub arg: Vec<(String, String)>,
pub argjson: Vec<(String, String)>,
pub slurpfile: Vec<(String, OsString)>,
pub rawfile: Vec<(String, OsString)>,
// Positional arguments
/// If this argument is not given, it is assumed to be `.`, the identity
/// filter.
pub filter: Option<Filter>,
pub files: Vec<PathBuf>,
pub args: Vec<String>,
//pub jsonargs: Vec<String>,
pub run_tests: Option<Vec<PathBuf>>,
/// If there is some last output value `v`,
/// then the exit status code is
/// 1 if `v < true` (that is, if `v` is `false` or `null`) and
/// 0 otherwise.
/// If there is no output value, then the exit status code is 4.
///
/// If any error occurs, then this option has no effect.
pub exit_status: bool,
pub version: bool,
pub help: bool,
}
#[derive(Debug)]
pub enum Filter {
Inline(String),
FromFile(PathBuf),
}
impl Cli {
fn positional(&mut self, mode: &Mode, arg: OsString) -> Result<(), Error> {
if self.filter.is_none() {
self.filter = Some(if self.from_file {
Filter::FromFile(arg.into())
} else {
Filter::Inline(arg.into_string()?)
})
} else {
match mode {
Mode::Files => self.files.push(arg.into()),
Mode::Args => self.args.push(arg.into_string()?),
//Mode::JsonArgs => self.jsonargs.push(arg.into_string()?),
}
}
Ok(())
}
fn long(&mut self, mode: &mut Mode, arg: &str, args: &mut Args) -> Result<(), Error> {
let int = |s: OsString| s.into_string().ok()?.parse().ok();
match arg {
// handle all arguments after "--"
"" => args.try_for_each(|arg| self.positional(mode, arg))?,
"null-input" => self.short('n', args)?,
"raw-input" => self.short('R', args)?,
"slurp" => self.short('s', args)?,
"compact-output" => self.short('c', args)?,
"raw-output" => self.short('r', args)?,
"join-output" => self.short('j', args)?,
"in-place" => self.short('i', args)?,
"sort-keys" => self.short('S', args)?,
"color-output" => self.short('C', args)?,
"monochrome-output" => self.short('M', args)?,
"tab" => self.tab = true,
"indent" => self.indent = args.next().and_then(int).ok_or(Error::Int("--indent"))?,
"from-file" => self.short('f', args)?,
"library-path" => self.short('L', args)?,
"arg" => {
let (name, value) = parse_key_val("--arg", args)?;
self.arg.push((name, value.into_string()?));
},
"argjson" => {
let (name, value) = parse_key_val("--argjson", args)?;
self.argjson.push((name, value.into_string()?));
},
"slurpfile" => self.slurpfile.push(parse_key_val("--slurpfile", args)?),
"rawfile" => self.rawfile.push(parse_key_val("--rawfile", args)?),
"args" => *mode = Mode::Args,
//"jsonargs" => *mode = Mode::JsonArgs,
"run-tests" => self.run_tests = Some(args.map(PathBuf::from).collect()),
"exit-status" => self.short('e', args)?,
"version" => self.short('V', args)?,
"help" => self.short('h', args)?,
arg => Err(Error::Flag(format!("--{arg}")))?,
}
Ok(())
}
fn short(&mut self, arg: char, args: &mut Args) -> Result<(), Error> {
match arg {
'n' => self.null_input = true,
'R' => self.raw_input = true,
's' => self.slurp = true,
'c' => self.compact_output = true,
'r' => self.raw_output = true,
'j' => self.join_output = true,
'i' => self.in_place = true,
'S' => self.sort_keys = true,
'C' => self.color_output = true,
'M' => self.monochrome_output = true,
'f' => self.from_file = true,
// resolve -L directories against the shell's cwd here; module
// loading happens inside unpatched jaq-core
'L' => self
.library_path
.push(pi_uutils_ctx::resolve(args.next().ok_or(Error::Path("-L"))?)),
'e' => self.exit_status = true,
'V' => self.version = true,
'h' => self.help = true,
arg => Err(Error::Flag(format!("-{arg}")))?,
}
Ok(())
}
pub fn parse(argv: Vec<OsString>) -> Result<Self, Error> {
let mut cli = Self { indent: 2, ..Self::default() };
let mut mode = Mode::Files;
let mut args = argv.into_iter();
args.next(); // skip the command name (argv[0])
while let Some(arg) = args.next() {
match arg.to_str() {
// we've got a valid UTF-8 argument here
Some(s) => match s.strip_prefix("--") {
Some(rest) => cli.long(&mut mode, rest, &mut args)?,
None => match s.strip_prefix("-") {
Some(rest) => rest.chars().try_for_each(|c| cli.short(c, &mut args))?,
None => cli.positional(&mode, arg)?,
},
},
// we've got invalid UTF-8, so it is no valid flag
// note that we do not check here whether arg starts with `-`,
// because this seems to be quite difficult to do in a portable way
None => cli.positional(&mode, arg)?,
}
}
Ok(cli)
}
pub fn color_if(&self, f: impl Fn() -> bool) -> bool {
if self.monochrome_output {
false
} else if self.color_output {
true
} else {
f()
}
}
}
#[derive(Debug)]
pub enum Error {
Flag(String),
Utf8(OsString),
KeyValue(&'static str),
Int(&'static str),
Path(&'static str),
}
impl fmt::Display for Error {
fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
match self {
Self::Flag(s) => write!(f, "unknown flag: {s}"),
Self::Utf8(s) => write!(f, "invalid UTF-8: {s:?}"),
Self::KeyValue(o) => write!(f, "{o} expects a key and a value"),
Self::Int(o) => write!(f, "{o} expects an integer"),
Self::Path(o) => write!(f, "{o} expects a path"),
}
}
}
/// Conversion of errors from [`OsString::into_string`].
impl From<OsString> for Error {
fn from(e: OsString) -> Self {
Self::Utf8(e)
}
}
fn parse_key_val(arg: &'static str, args: &mut Args) -> Result<(String, OsString), Error> {
let err = || Error::KeyValue(arg);
let key = args.next().ok_or_else(err)?.into_string()?;
let val = args.next().ok_or_else(err)?;
Ok((key, val))
}
/// Interpretation of positional arguments.
enum Mode {
Args,
//JsonArgs,
Files,
}
+358
View File
@@ -0,0 +1,358 @@
//! Filter parsing, compilation, and execution.
use core::{
cell::Cell,
fmt::{self, Display, Formatter},
};
use std::{
io::{self, Write},
path::PathBuf,
};
use jaq_core::{
Ctx, Error as CoreError, Exn, Native, RcIter, RunPtr, UpdatePtr, ValT, compile, load,
};
use crate::{Cli, Error, Val, read};
pub type Filter = jaq_core::Filter<Native<Val>>;
thread_local! {
/// Exit code requested by `halt`/`halt_error` in the current invocation.
/// The overridden natives set this instead of `std::process::exit` and
/// abort the run with a sentinel error; the entry point checks it first.
static HALT: Cell<Option<i32>> = const { Cell::new(None) };
}
/// Takes (and clears) the exit code requested by `halt`/`halt_error`.
pub fn take_halt() -> Option<i32> {
HALT.with(Cell::take)
}
/// Replacements for jaq-std natives that are unsound inside a long-lived host
/// process. Prepended before `jaq_std::funs()`: the compiler resolves native
/// calls by first match, so these shadow the crates.io implementations.
///
/// - `env`: reads the shell's exported environment, not the host process's.
/// - `halt`/`halt_error`: record the exit code and abort the run with a
/// sentinel error instead of `std::process::exit`, which would kill the
/// shell.
/// - `debug`/`stderr`: write to the ctx stderr stream directly instead of going
/// through the process-global `log` facade (whose single global logger may
/// belong to the host).
fn overrides() -> impl Iterator<Item = jaq_std::Filter<Native<Val>>> {
use jaq_core::box_iter::box_once;
use jaq_std::ValT as _;
fn halt_with<'a>(code: i32, sentinel: &'static str) -> jaq_core::ValXs<'a, Val> {
HALT.with(|h| h.set(Some(code)));
box_once(Err(Exn::from(CoreError::str(sentinel))))
}
fn debug_msg(v: &Val) {
// upstream format: env_logger renders `["DEBUG:", <args>]\n`
let _ = writeln!(pi_uutils_ctx::stderr(), "[\"DEBUG:\", {v}]");
}
fn stderr_msg(v: &Val) {
// like jq, print strings raw and everything else as JSON, no newline
if let Some(s) = v.as_str() {
let _ = write!(pi_uutils_ctx::stderr(), "{s}");
} else {
let _ = write!(pi_uutils_ctx::stderr(), "{v}");
}
}
let run_funs: [jaq_std::Filter<RunPtr<Val>>; 3] = [
("env", jaq_std::v(0), |_, _| {
let env = pi_uutils_ctx::env_snapshot()
.into_iter()
.map(|(k, v)| (k.into(), Val::from(v)));
box_once(Ok(Val::obj(env.collect())))
}),
("halt", jaq_std::v(0), |_, _| halt_with(0, "halt")),
("halt_error", jaq_std::v(1), |_, mut cv| {
match cv.0.pop_var().as_isize() {
Some(code) => {
// upstream prints the input to stdout: raw for strings
// (no trailing newline), JSON + newline otherwise
if let Some(s) = cv.1.as_str() {
let _ = write!(pi_uutils_ctx::stdout(), "{s}");
} else {
let _ = writeln!(pi_uutils_ctx::stdout(), "{}", cv.1);
}
halt_with(code as i32, "halt_error")
},
None => box_once(Err(Exn::from(CoreError::typ(cv.1, "integer")))),
}
}),
];
// `debug` and `stderr` are identity filters with an output effect; they
// need an update pointer so `debug |= f` keeps working.
let upd_funs: [jaq_std::Filter<(RunPtr<Val>, UpdatePtr<Val>)>; 2] = [
(
"debug",
jaq_std::v(0),
(
|_, cv| {
debug_msg(&cv.1);
box_once(Ok(cv.1))
},
|_, cv, f| {
debug_msg(&cv.1);
f(cv.1)
},
),
),
(
"stderr",
jaq_std::v(0),
(
|_, cv| {
stderr_msg(&cv.1);
box_once(Ok(cv.1))
},
|_, cv, f| {
stderr_msg(&cv.1);
f(cv.1)
},
),
),
];
let upd = |(name, arity, (run, update)): jaq_std::Filter<(RunPtr<Val>, UpdatePtr<Val>)>| {
(name, arity, Native::new(run).with_update(update))
};
let run_funs = run_funs.into_iter().map(jaq_std::run);
run_funs.chain(upd_funs.into_iter().map(upd))
}
pub fn parse_compile(
path: &PathBuf,
code: &str,
vars: &[String],
paths: &[PathBuf],
) -> Result<(Vec<Val>, Filter), Vec<FileReports>> {
use compile::Compiler;
use load::{Arena, File, Loader, import};
let default = ["~/.jq", "$ORIGIN/../lib/jq", "$ORIGIN/../lib"].map(|x| x.into());
let paths = if paths.is_empty() { &default } else { paths };
let vars: Vec<_> = vars.iter().map(|v| format!("${v}")).collect();
let arena = Arena::default();
let defs = jaq_std::defs().chain(jaq_json::defs());
let loader = Loader::new(defs).with_std_read(paths);
let path = path.into();
let modules = loader
.load(&arena, File { path, code })
.map_err(load_errors)?;
let mut vals = Vec::new();
import(&modules, |p| {
let path = p.find(paths, "json")?;
vals.push(read::json_array(path).map_err(|e| e.to_string())?);
Ok(())
})
.map_err(load_errors)?;
// overrides first: native lookup is first-match-wins
let funs = overrides().chain(jaq_std::funs()).chain(jaq_json::funs());
let compiler = Compiler::default()
.with_funs(funs)
.with_global_vars(vars.iter().map(|v| &**v));
let filter = compiler.compile(modules).map_err(compile_errors)?;
Ok((vals, filter))
}
/// Run a filter with given input values and run `f` for every value output.
///
/// This function cannot return an `Iterator` because it creates an `RcIter`.
/// This is most unfortunate. We should think about how to simplify this ...
pub(crate) fn run(
cli: &Cli,
filter: &Filter,
vars: Vec<Val>,
iter: impl Iterator<Item = io::Result<Val>>,
mut f: impl FnMut(Val) -> io::Result<()>,
) -> Result<Option<bool>, Error> {
let mut last = None;
let iter = iter.map(|r| r.map_err(|e| e.to_string()));
let iter = Box::new(iter) as Box<dyn Iterator<Item = _>>;
let null = Box::new(core::iter::once(Ok(Val::Null))) as Box<dyn Iterator<Item = _>>;
let iter = RcIter::new(iter);
let null = RcIter::new(null);
let ctx = Ctx::new(vars, &iter);
for item in if cli.null_input { &null } else { &iter } {
// host abort/timeout: stdin reads observe the cancel flag themselves,
// but file/slurped inputs and long-running filters do not
if pi_uutils_ctx::is_cancelled() {
break;
}
let input = item.map_err(Error::Parse)?;
for output in filter.run((ctx.clone(), input)) {
if pi_uutils_ctx::is_cancelled() {
return Ok(last);
}
let output = output.map_err(Error::Jaq)?;
last = Some(output.as_bool());
f(output)?;
}
}
Ok(last)
}
#[derive(Debug)]
pub struct FileReports(load::File<String, PathBuf>, Vec<Report>);
impl Display for FileReports {
fn fmt(&self, f: &mut Formatter) -> fmt::Result {
let Self(file, reports) = self;
let idx = codesnake::LineIndex::new(&file.code);
reports.iter().try_for_each(|e| {
writeln!(f, "Error: {}", e.message)?;
let block = e.to_block(&idx);
writeln!(f, "{}[{}]", block.prologue(), file.path.display())?;
writeln!(f, "{}{}", block, block.epilogue())
})
}
}
fn load_errors(errs: load::Errors<&str, PathBuf>) -> Vec<FileReports> {
use load::Error;
let errs = errs.into_iter().map(|(file, err)| {
let code = file.code;
let err = match err {
Error::Io(errs) => errs.into_iter().map(|e| report_io(code, e)).collect(),
Error::Lex(errs) => errs.into_iter().map(|e| report_lex(code, e)).collect(),
Error::Parse(errs) => errs.into_iter().map(|e| report_parse(code, e)).collect(),
};
FileReports(file.map_code(|s| s.into()), err)
});
errs.collect()
}
fn compile_errors(errs: compile::Errors<&str, PathBuf>) -> Vec<FileReports> {
let errs = errs.into_iter().map(|(file, errs)| {
let code = file.code;
let errs = errs.into_iter().map(|e| report_compile(code, e)).collect();
FileReports(file.map_code(|s| s.into()), errs)
});
errs.collect()
}
type StringColors = Vec<(String, Option<Color>)>;
#[derive(Debug)]
struct Report {
message: String,
labels: Vec<(core::ops::Range<usize>, StringColors, Color)>,
}
#[derive(Clone, Debug)]
enum Color {
Yellow,
Red,
}
impl Color {
fn apply(&self, d: impl Display) -> String {
use yansi::{Color, Paint};
let color = match self {
Self::Yellow => Color::Yellow,
Self::Red => Color::Red,
};
d.fg(color).to_string()
}
}
fn report_io(code: &str, (path, error): (&str, String)) -> Report {
let path_range = load::span(code, path);
Report {
message: format!("could not load file {}: {}", path, error),
labels: [(path_range, [(error, None)].into(), Color::Red)].into(),
}
}
fn report_lex(code: &str, (expected, found): load::lex::Error<&str>) -> Report {
// truncate found string to its first character
let found = &found[..found.char_indices().nth(1).map_or(found.len(), |(i, _)| i)];
let found_range = load::span(code, found);
let found = match found {
"" => [("unexpected end of input".to_string(), None)].into(),
c => [("unexpected character ", None), (c, Some(Color::Red))]
.map(|(s, c)| (s.into(), c))
.into(),
};
let label = (found_range, found, Color::Red);
let labels = match expected {
load::lex::Expect::Delim(open) => {
let text = [("unclosed delimiter ", None), (open, Some(Color::Yellow))]
.map(|(s, c)| (s.into(), c));
Vec::from([(load::span(code, open), text.into(), Color::Yellow), label])
},
_ => Vec::from([label]),
};
Report { message: format!("expected {}", expected.as_str()), labels }
}
fn report_parse(code: &str, (expected, found): load::parse::Error<&str>) -> Report {
let found_range = load::span(code, found);
let found = if found.is_empty() {
"unexpected end of input"
} else {
"unexpected token"
};
let found = [(found.to_string(), None)].into();
Report {
message: format!("expected {}", expected.as_str()),
labels: Vec::from([(found_range, found, Color::Red)]),
}
}
fn report_compile(code: &str, (found, undefined): compile::Error<&str>) -> Report {
use compile::Undefined::Filter;
let found_range = load::span(code, found);
let wnoa = |exp, got| format!("wrong number of arguments (expected {exp}, found {got})");
let message = match (found, undefined) {
("reduce", Filter(arity)) => wnoa("2", arity),
("foreach", Filter(arity)) => wnoa("2 or 3", arity),
(_, undefined) => format!("undefined {}", undefined.as_str()),
};
let found = [(message.clone(), None)].into();
Report { message, labels: Vec::from([(found_range, found, Color::Red)]) }
}
type CodeBlock = codesnake::Block<codesnake::CodeWidth<String>, String>;
impl Report {
fn to_block(&self, idx: &codesnake::LineIndex) -> CodeBlock {
use codesnake::{Block, CodeWidth, Label};
let color_maybe = |(text, color): (_, Option<Color>)| match color {
None => text,
Some(color) => color.apply(text).to_string(),
};
let labels = self.labels.iter().cloned().map(|(range, text, color)| {
let text = text.into_iter().map(color_maybe).collect::<Vec<_>>();
Label::new(range)
.with_text(text.join(""))
.with_style(move |s| color.apply(s).to_string())
});
Block::new(idx, labels).unwrap().map_code(|c| {
let c = c.replace('\t', " ");
let w = unicode_width::UnicodeWidthStr::width(&*c);
CodeWidth::new(c, core::cmp::max(w, 1))
})
}
}
+40
View File
@@ -0,0 +1,40 @@
Just Another Query Tool
Usage: jaq [OPTION]... [FILTER] [ARG]...
Arguments:
[FILTER] Filter to execute
[ARG]... Positional arguments, by default used as input files
Input options:
-n, --null-input Use null as single input value
-R, --raw-input Read lines of the input as sequence of strings
-s, --slurp Read (slurp) all input values into one array
Output options:
-c, --compact-output Print JSON compactly, omitting whitespace
-r, --raw-output Write strings without escaping them with quotes
-j, --join-output Do not print a newline after each value
-i, --in-place Overwrite input file with its output
-S, --sort-keys Print objects sorted by their keys
-C, --color-output Always color output
-M, --monochrome-output Do not color output
--tab Use tabs for indentation rather than spaces
--indent <N> Use N spaces for indentation [default: 2]
Compilation options:
-f, --from-file Read filter from a file given by filter argument
-L, --library-path <DIR> Search for modules and data in given directory
Variable options:
--arg <A> <V> Set variable `$A` to string `V`
--argjson <A> <V> Set variable `$A` to JSON value `V`
--slurpfile <A> <F> Set variable `$A` to array containing the JSON values in file `F`
--rawfile <A> <F> Set variable `$A` to string containing the contents of file `F`
--args Collect remaining positional arguments into `$ARGS.positional`
Remaining options:
--run-tests <FILE> Run tests from a file
-e, --exit-status Use the last output value as exit status code
-V, --version Print version
-h, --help Print help
+333
View File
@@ -0,0 +1,333 @@
//! Vendored, patched `jaq` CLI (jq-compatible JSON processor), wired to run
//! in-process as a shell builtin via [`pi_uutils_ctx`].
//!
//! Upstream: <https://github.com/01mf02/jaq>, tag `v2.3.0`,
//! commit `0ce6e86a5e038a623dc894ad5cc70aaa9142daf2` (MIT).
//!
//! Only the CLI crate (`jaq/`) is vendored; the interpreter libraries
//! (`jaq-core`, `jaq-std`, `jaq-json`) come from crates.io. Patches vs
//! upstream:
//! - `main()` is restructured as [`run`], returning the exit code instead of
//! `ExitCode`/`Termination`; no `std::process::exit` anywhere.
//! - stdio goes through the [`pi_uutils_ctx`] streams; every file path operand
//! resolves through `pi_uutils_ctx::resolve` against the shell's cwd.
//! - The ctx streams are never a tty, so `--color` auto mode always resolves to
//! plain output; `-C/--color-output` still forces ANSI. Color state is
//! thread-local (see `color` in this module) instead of yansi's global
//! enable/disable, so concurrent invocations don't race.
//! - The mimalloc global allocator, env_logger, and the rustyline `repl` filter
//! are stripped (binary-only / interactive-only).
//! - jaq-std's `env`, `halt`, `halt_error`, `debug`, and `stderr` natives are
//! shadowed (first-match-wins in the compiler's native table) because the
//! crates.io implementations call `std::process::exit`, read the host process
//! environment, or log through the global `log` facade. See
//! `filter::overrides`.
mod cli;
mod filter;
mod read;
mod write;
use core::fmt::{self, Display, Formatter};
use std::{
io::{self, BufRead, Write},
path::PathBuf,
};
use cli::Cli;
use filter::{FileReports, Filter};
use jaq_core::{Ctx, RcIter, load};
use jaq_json::Val;
use write::{print, with_stdout};
/// In-process builtin entry point. The host installs a [`pi_uutils_ctx`] scope
/// (stdio + working directory + environment) on a dedicated blocking thread,
/// then calls this with `argv[0]` = command name (`jq`).
pub fn run(argv: Vec<std::ffi::OsString>) -> i32 {
color::init();
color::set(false);
filter::take_halt(); // clear leftover state from a prior scope on this thread
let cli = match Cli::parse(argv) {
Ok(cli) => cli,
Err(e) => {
let _ = writeln!(pi_uutils_ctx::stderr(), "Error: {e}");
return 2;
},
};
if cli.version {
let _ = writeln!(
pi_uutils_ctx::stdout(),
"{} {}",
env!("CARGO_PKG_NAME"),
env!("CARGO_PKG_VERSION")
);
return 0;
} else if cli.help {
let _ = writeln!(pi_uutils_ctx::stdout(), "{}", include_str!("help.txt"));
return 0;
}
// Upstream enables color when stdout is a terminal and NO_COLOR is unset.
// The ctx streams are never a terminal, so auto mode is always plain;
// only -C/--color-output (minus -M) forces ANSI.
color::set(!cli.in_place && cli.color_if(|| false));
let res = real_main(&cli);
// `halt`/`halt_error` abort the filter run with a sentinel error; the
// requested exit code wins over the error path below.
if let Some(code) = filter::take_halt() {
return code;
}
match res {
Ok(exit) => exit,
Err(e) => {
color::set(cli.color_if(|| false));
let _ = write!(pi_uutils_ctx::stderr(), "{e}");
e.report()
},
}
}
/// Thread-local color toggle backing yansi's process-global condition.
///
/// `yansi::enable`/`disable` flip process-global state, which races when
/// several shell pipeline stages run jaq concurrently on different threads.
/// Instead, a process-global yansi condition (installed once) reads this
/// thread-local flag, giving each invocation its own color state.
mod color {
use std::{cell::Cell, sync::Once};
thread_local! {
static COLOR: Cell<bool> = const { Cell::new(false) };
}
pub fn init() {
static ONCE: Once = Once::new();
ONCE.call_once(|| yansi::whenever(yansi::Condition(|| COLOR.with(Cell::get))));
}
pub fn set(on: bool) {
COLOR.with(|c| c.set(on));
}
}
fn real_main(cli: &Cli) -> Result<i32, Error> {
if let Some(test_files) = &cli.run_tests {
return Ok(match test_files.last() {
Some(file) => {
run_tests(io::BufReader::new(std::fs::File::open(pi_uutils_ctx::resolve(file))?))
},
None => run_tests(io::BufReader::new(pi_uutils_ctx::stdin())),
});
}
let (vars, mut ctx): (Vec<String>, Vec<Val>) = binds(cli)?.into_iter().unzip();
let (vals, filter) = match &cli.filter {
None => (Vec::new(), Filter::default()),
Some(filter) => {
let (path, code) = match filter {
cli::Filter::FromFile(path) => {
(path.into(), std::fs::read_to_string(pi_uutils_ctx::resolve(path))?)
},
cli::Filter::Inline(filter) => ("<inline>".into(), filter.clone()),
};
filter::parse_compile(&path, &code, &vars, &cli.library_path).map_err(Error::Report)?
},
};
ctx.extend(vals);
let last = if cli.files.is_empty() {
let inputs = read::buffered(cli, io::BufReader::new(pi_uutils_ctx::stdin()));
with_stdout(|out| filter::run(cli, &filter, ctx, inputs, |v| print(out, cli, &v)))?
} else {
let mut last = None;
for file in &cli.files {
// Resolve the operand against the shell's cwd; all later path
// operations (open, metadata, in-place temp+rename) use the
// resolved path so nothing touches the host process cwd.
let resolved = pi_uutils_ctx::resolve(file);
let path = resolved.as_path();
let file =
read::load_file(path).map_err(|e| Error::Io(Some(path.display().to_string()), e))?;
let inputs = read::slice(cli, &file);
if cli.in_place {
// create a temporary file where output is written to,
// in the resolved target's directory so the final rename
// stays on the same filesystem
let location = path.parent().unwrap();
let mut tmp = tempfile::Builder::new()
.prefix("jaq")
.tempfile_in(location)?;
last = filter::run(cli, &filter, ctx.clone(), inputs, |output| {
print(tmp.as_file_mut(), cli, &output)
})?;
// replace the input file with the temporary file
std::mem::drop(file);
let perms = std::fs::metadata(path)?.permissions();
tmp.persist(path).map_err(Error::Persist)?;
std::fs::set_permissions(path, perms)?;
} else {
last = with_stdout(|out| {
filter::run(cli, &filter, ctx.clone(), inputs, |v| print(out, cli, &v))
})?;
}
}
last
};
if cli.exit_status {
last.map_or_else(|| Err(Error::NoOutput), |b| if b { Ok(0) } else { Err(Error::FalseOrNull) })
} else {
Ok(0)
}
}
fn binds(cli: &Cli) -> Result<Vec<(String, Val)>, Error> {
let arg = cli.arg.iter().map(|(k, s)| {
let s = s.to_owned();
Ok((k.to_owned(), Val::Str(s.into())))
});
let argjson = cli.argjson.iter().map(|(k, s)| {
use hifijson::token::Lex;
let mut lexer = hifijson::SliceLexer::new(s.as_bytes());
let err = |e| Error::Parse(format!("{e} (for value passed to `--argjson {k}`)"));
Ok((k.to_owned(), lexer.exactly_one(Val::parse).map_err(err)?))
});
let rawfile = cli.rawfile.iter().map(|(k, path)| {
let s = std::fs::read_to_string(pi_uutils_ctx::resolve(path))
.map_err(|e| Error::Io(Some(format!("{path:?}")), e));
Ok((k.to_owned(), Val::Str(s?.into())))
});
let slurpfile = cli.slurpfile.iter().map(|(k, path)| {
let a = read::json_array(path).map_err(|e| Error::Io(Some(format!("{path:?}")), e));
Ok((k.to_owned(), a?))
});
let positional = cli.args.iter().cloned().map(|s| Ok(Val::from(s)));
let positional = positional.collect::<Result<Vec<_>, Error>>()?;
let var_val = arg.chain(rawfile).chain(slurpfile).chain(argjson);
let mut var_val = var_val.collect::<Result<Vec<_>, Error>>()?;
var_val.push(("ARGS".to_string(), args(&positional, &var_val)));
// the shell's exported environment, not the host process environment
let env = pi_uutils_ctx::env_snapshot()
.into_iter()
.map(|(k, v)| (k.into(), Val::from(v)));
var_val.push(("ENV".to_string(), Val::obj(env.collect())));
Ok(var_val)
}
fn args(positional: &[Val], named: &[(String, Val)]) -> Val {
let key = |k: &str| k.to_string().into();
let positional = positional.iter().cloned();
let named = named.iter().map(|(var, val)| (key(var), val.clone()));
let obj = [(key("positional"), positional.collect()), (key("named"), Val::obj(named.collect()))];
Val::obj(obj.into_iter().collect())
}
#[derive(Debug)]
enum Error {
Io(Option<String>, io::Error),
Report(Vec<FileReports>),
Parse(String),
Jaq(jaq_core::Error<Val>),
Persist(tempfile::PersistError),
FalseOrNull,
NoOutput,
}
impl Display for Error {
fn fmt(&self, f: &mut Formatter) -> fmt::Result {
match self {
Self::FalseOrNull | Self::NoOutput => Ok(()),
Self::Io(prefix, e) => {
write!(f, "Error: ")?;
if let Some(p) = prefix {
write!(f, "{p}: ")?;
}
writeln!(f, "{e}")
},
Self::Persist(e) => {
writeln!(f, "Error: {e}")
},
Self::Report(reports) => reports.iter().try_for_each(|fr| write!(f, "{fr}")),
Self::Parse(e) => writeln!(f, "Error: failed to parse: {e}"),
Self::Jaq(e) => writeln!(f, "Error: {e}"),
}
}
}
impl Error {
/// Upstream's `Termination` exit-code mapping, kept verbatim.
fn report(&self) -> i32 {
match self {
Self::FalseOrNull => 1,
Self::Io(..) | Self::Persist(_) => 2,
Self::Report(_) => 3,
Self::NoOutput => 4,
Self::Parse(_) | Self::Jaq(_) => 5,
}
}
}
impl From<io::Error> for Error {
fn from(e: io::Error) -> Self {
Self::Io(None, e)
}
}
fn run_test(test: load::test::Test<String>) -> Result<(Val, Val), Error> {
let (ctx, filter) =
filter::parse_compile(&PathBuf::new(), &test.filter, &[], &[]).map_err(Error::Report)?;
let inputs = RcIter::new(Box::new(core::iter::empty()));
let ctx = Ctx::new(ctx, &inputs);
let json = |s: String| {
use hifijson::token::Lex;
hifijson::SliceLexer::new(s.as_bytes())
.exactly_one(Val::parse)
.map_err(read::invalid_data)
};
let input = json(test.input)?;
let expect: Result<Val, _> = test.output.into_iter().map(json).collect();
let obtain: Result<Val, _> = filter.run((ctx, input)).collect();
Ok((expect?, obtain.map_err(Error::Jaq)?))
}
fn run_tests(read: impl BufRead) -> i32 {
let lines = read.lines().map(Result::unwrap);
let tests = load::test::Parser::new(lines);
let (mut passed, mut total) = (0, 0);
for test in tests {
if pi_uutils_ctx::is_cancelled() {
break;
}
let _ = writeln!(pi_uutils_ctx::stdout(), "Testing {}", test.filter);
match run_test(test) {
Err(e) => {
let _ = writeln!(pi_uutils_ctx::stderr(), "{e:?}");
},
Ok((expect, obtain)) if expect != obtain => {
let _ = writeln!(pi_uutils_ctx::stderr(), "expected {expect}, obtained {obtain}",);
},
Ok(_) => passed += 1,
}
total += 1;
}
let _ = writeln!(pi_uutils_ctx::stdout(), "{passed} out of {total} tests passed");
i32::from(total > passed)
}
#[cfg(test)]
mod tests;
+89
View File
@@ -0,0 +1,89 @@
use std::{
io::{self, BufRead},
path::Path,
};
use crate::{Cli, Val};
/// Try to load file by memory mapping and fall back to regular loading if it
/// fails.
///
/// The path is resolved against the shell's working directory: as an
/// in-process builtin, the host process cwd is unrelated to the shell's.
pub fn load_file(path: impl AsRef<Path>) -> io::Result<Box<dyn core::ops::Deref<Target = [u8]>>> {
let path = pi_uutils_ctx::resolve(path.as_ref());
let file = std::fs::File::open(&path)?;
match unsafe { memmap2::Mmap::map(&file) } {
Ok(mmap) => Ok(Box::new(mmap)),
Err(_) => Ok(Box::new(std::fs::read(&path)?)),
}
}
pub fn invalid_data(e: impl std::error::Error + Send + Sync + 'static) -> std::io::Error {
io::Error::new(io::ErrorKind::InvalidData, e)
}
fn json_slice(slice: &[u8]) -> impl Iterator<Item = io::Result<Val>> + '_ {
let mut lexer = hifijson::SliceLexer::new(slice);
core::iter::from_fn(move || {
use hifijson::token::Lex;
Some(Val::parse(lexer.ws_token()?, &mut lexer).map_err(invalid_data))
})
}
fn json_read<'a>(read: impl BufRead + 'a) -> impl Iterator<Item = io::Result<Val>> + 'a {
let mut lexer = hifijson::IterLexer::new(read.bytes());
core::iter::from_fn(move || {
use hifijson::token::Lex;
let v = Val::parse(lexer.ws_token()?, &mut lexer);
Some(v.map_err(|e| core::mem::take(&mut lexer.error).unwrap_or_else(|| invalid_data(e))))
})
}
pub fn json_array(path: impl AsRef<Path>) -> io::Result<Val> {
json_slice(&load_file(path.as_ref())?).collect()
}
pub fn buffered<'a, R>(cli: &Cli, read: R) -> Box<dyn Iterator<Item = io::Result<Val>> + 'a>
where
R: BufRead + 'a,
{
if cli.raw_input {
Box::new(raw_input(cli.slurp, read).map(|r| r.map(Val::from)))
} else {
Box::new(collect_if(cli.slurp, json_read(read)))
}
}
pub fn slice<'a>(cli: &Cli, slice: &'a [u8]) -> Box<dyn Iterator<Item = io::Result<Val>> + 'a> {
if cli.raw_input {
let read = io::BufReader::new(slice);
Box::new(raw_input(cli.slurp, read).map(|r| r.map(Val::from)))
} else {
Box::new(collect_if(cli.slurp, json_slice(slice)))
}
}
fn raw_input<'a, R>(slurp: bool, mut read: R) -> impl Iterator<Item = io::Result<String>> + 'a
where
R: BufRead + 'a,
{
if slurp {
let mut buf = String::new();
let s = read.read_to_string(&mut buf).map(|_| buf);
Box::new(std::iter::once(s))
} else {
Box::new(read.lines()) as Box<dyn Iterator<Item = _>>
}
}
fn collect_if<'a, T: FromIterator<T> + 'a, E: 'a>(
slurp: bool,
iter: impl Iterator<Item = Result<T, E>> + 'a,
) -> Box<dyn Iterator<Item = Result<T, E>> + 'a> {
if slurp {
Box::new(core::iter::once(iter.collect()))
} else {
Box::new(iter)
}
}
+299
View File
@@ -0,0 +1,299 @@
//! Behavioral contract tests driving [`crate::run`] under a
//! [`pi_uutils_ctx::scope`], the way the shell host invokes the builtin.
use std::{
collections::HashMap,
ffi::OsString,
io::{self, Write},
path::PathBuf,
sync::{Arc, atomic::AtomicBool},
};
use parking_lot::Mutex;
/// `Send + Write` capture buffer for the scope's stdout/stderr.
#[derive(Clone, Default)]
struct Buf(Arc<Mutex<Vec<u8>>>);
impl Buf {
fn take_string(&self) -> String {
String::from_utf8(std::mem::take(&mut *self.0.lock())).expect("utf8 output")
}
}
impl Write for Buf {
fn write(&mut self, buf: &[u8]) -> io::Result<usize> {
self.0.lock().extend_from_slice(buf);
Ok(buf.len())
}
fn flush(&mut self) -> io::Result<()> {
Ok(())
}
}
/// Runs `jq <args>` with `stdin` under a fresh scope; returns
/// `(exit code, stdout, stderr)`.
fn run_jq_in(
cwd: PathBuf,
env: HashMap<String, String>,
args: &[&str],
stdin: &str,
) -> (i32, String, String) {
let out = Buf::default();
let err = Buf::default();
let io_ = pi_uutils_ctx::ScopeIo {
stdin: Box::new(io::Cursor::new(stdin.as_bytes().to_vec())),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(out.clone()),
stderr: Box::new(err.clone()),
cwd,
env,
cancel: Arc::new(AtomicBool::new(false)),
};
let mut argv = vec![OsString::from("jq")];
argv.extend(args.iter().map(OsString::from));
let code = pi_uutils_ctx::scope(io_, || crate::run(argv));
(code, out.take_string(), err.take_string())
}
fn run_jq(args: &[&str], stdin: &str) -> (i32, String, String) {
run_jq_in(PathBuf::from("."), HashMap::new(), args, stdin)
}
#[test]
fn identity_pretty_prints() {
let (code, out, err) = run_jq(&["."], "{\"a\":1}");
assert_eq!(code, 0);
assert_eq!(out, "{\n \"a\": 1\n}\n");
assert_eq!(err, "");
}
#[test]
fn compact_output() {
let (code, out, _) = run_jq(&["-c", ".a"], "{\"a\":[1,2]}");
assert_eq!(code, 0);
assert_eq!(out, "[1,2]\n");
}
#[test]
fn raw_output_strips_quotes() {
let (code, out, _) = run_jq(&["-r", ".s"], "{\"s\":\"x y\"}");
assert_eq!(code, 0);
assert_eq!(out, "x y\n");
let (code, out, _) = run_jq(&[".s"], "{\"s\":\"x y\"}");
assert_eq!(code, 0);
assert_eq!(out, "\"x y\"\n");
}
#[test]
fn null_input_evaluates_filter() {
let (code, out, _) = run_jq(&["-n", "1+2"], "");
assert_eq!(code, 0);
assert_eq!(out, "3\n");
}
#[test]
fn slurp_collects_documents() {
let (code, out, _) = run_jq(&["-s", "length"], "{\"a\":1}\n{\"b\":2}\n");
assert_eq!(code, 0);
assert_eq!(out, "2\n");
}
#[test]
fn named_arg_binds_variable() {
let (code, out, _) = run_jq(&["-n", "--arg", "k", "v", "$k"], "");
assert_eq!(code, 0);
assert_eq!(out, "\"v\"\n");
}
#[test]
fn argjson_binds_json_value() {
let (code, out, _) = run_jq(&["-nc", "--argjson", "k", "[1,2]", "$k"], "");
assert_eq!(code, 0);
assert_eq!(out, "[1,2]\n");
}
#[test]
fn exit_status_flag() {
// false -> 1
let (code, out, _) = run_jq(&["-n", "-e", "false"], "");
assert_eq!(code, 1);
assert_eq!(out, "false\n");
// null (missing key) -> 1
let (code, out, _) = run_jq(&["-e", ".missing"], "{}");
assert_eq!(code, 1);
assert_eq!(out, "null\n");
// truthy -> 0
let (code, ..) = run_jq(&["-e", "."], "true");
assert_eq!(code, 0);
// no output at all -> 4 (jaq-specific; jq also uses 4 here)
let (code, ..) = run_jq(&["-n", "-e", "empty"], "");
assert_eq!(code, 4);
}
#[test]
fn compile_error_exits_3_with_diagnostic() {
let (code, out, err) = run_jq(&["("], "null");
assert_eq!(code, 3);
assert_eq!(out, "", "compile error must not produce output");
assert!(err.contains("Error:"), "diagnostic on stderr: {err:?}");
assert!(err.contains("<inline>"), "names the filter source: {err:?}");
}
#[test]
fn runtime_error_exits_5_with_diagnostic() {
// indexing a number is a runtime (Jaq) error
let (code, out, err) = run_jq(&[".[0]"], "1");
assert_eq!(code, 5);
assert_eq!(out, "");
assert!(err.starts_with("Error:"), "diagnostic on stderr: {err:?}");
}
#[test]
fn usage_error_exits_2() {
let (code, _, err) = run_jq(&["--bogus", "."], "");
assert_eq!(code, 2);
assert!(err.contains("unknown flag: --bogus"), "stderr: {err:?}");
}
#[test]
fn relative_file_operand_resolves_against_scope_cwd() {
let dir = tempfile::TempDir::new().expect("tempdir");
std::fs::write(dir.path().join("in.json"), "{\"a\":[1,2]}").expect("write input");
// relative operand: must resolve against ScopeIo.cwd, not the process cwd
let (code, out, err) =
run_jq_in(dir.path().to_path_buf(), HashMap::new(), &["-c", ".a", "in.json"], "");
assert_eq!(code, 0, "stderr: {err:?}");
assert_eq!(out, "[1,2]\n");
}
#[test]
fn missing_file_operand_exits_2() {
let dir = tempfile::TempDir::new().expect("tempdir");
let (code, out, err) =
run_jq_in(dir.path().to_path_buf(), HashMap::new(), &[".", "nope.json"], "");
assert_eq!(code, 2);
assert_eq!(out, "");
assert!(err.contains("nope.json"), "stderr names the operand: {err:?}");
}
#[test]
fn in_place_edit_rewrites_relative_file() {
let dir = tempfile::TempDir::new().expect("tempdir");
std::fs::write(dir.path().join("in.json"), "{\"a\":1}").expect("write input");
let (code, _, err) =
run_jq_in(dir.path().to_path_buf(), HashMap::new(), &["-c", "-i", ".a", "in.json"], "");
assert_eq!(code, 0, "stderr: {err:?}");
let rewritten = std::fs::read_to_string(dir.path().join("in.json")).expect("read back");
assert_eq!(rewritten, "1\n");
}
#[test]
fn invalid_trailing_json_on_stdin_fails() {
let (code, out, err) = run_jq(&["-c", "."], "{\"a\":1} xyz");
assert_eq!(code, 5);
assert_eq!(out, "{\"a\":1}\n", "valid leading document is still emitted");
assert!(err.contains("Error:"), "stderr diagnostic: {err:?}");
}
#[test]
fn env_var_and_dollar_env_read_scope_environment() {
let env = HashMap::from([("FOO".to_string(), "bar".to_string())]);
let (code, out, _) = run_jq_in(PathBuf::from("."), env, &["-n", "$ENV.FOO, env.FOO"], "");
assert_eq!(code, 0);
assert_eq!(out, "\"bar\"\n\"bar\"\n", "$ENV and env read the shell env");
}
#[test]
fn halt_returns_instead_of_killing_process() {
let (code, out, err) = run_jq(&["-n", "1, halt, 2"], "");
assert_eq!(code, 0, "halt exits 0");
assert_eq!(out, "1\n", "outputs before halt are emitted, none after");
assert_eq!(err, "");
}
#[test]
fn halt_error_prints_message_and_exit_code() {
let (code, out, _) = run_jq(&["-n", "\"bye\\n\" | halt_error(3)"], "");
assert_eq!(code, 3);
assert_eq!(out, "bye\n", "string message printed raw");
}
#[test]
fn stderr_filter_writes_to_scope_stderr() {
let (code, out, err) = run_jq(&["-n", "\"msg\" | stderr | length"], "");
assert_eq!(code, 0);
assert_eq!(out, "3\n", "stderr is an identity filter");
assert_eq!(err, "msg", "raw string on stderr, no newline");
}
#[test]
fn debug_filter_writes_to_scope_stderr() {
let (code, out, err) = run_jq(&["-nc", "[1,2] | debug"], "");
assert_eq!(code, 0);
assert_eq!(out, "[1,2]\n");
assert_eq!(err, "[\"DEBUG:\", [1,2]]\n");
}
#[test]
fn rawfile_and_slurpfile_resolve_against_scope_cwd() {
let dir = tempfile::TempDir::new().expect("tempdir");
std::fs::write(dir.path().join("raw.txt"), "hi").expect("write raw");
std::fs::write(dir.path().join("vals.json"), "1 2").expect("write vals");
let (code, out, err) = run_jq_in(
dir.path().to_path_buf(),
HashMap::new(),
&["-nc", "--rawfile", "r", "raw.txt", "--slurpfile", "v", "vals.json", "$r, $v"],
"",
);
assert_eq!(code, 0, "stderr: {err:?}");
assert_eq!(out, "\"hi\"\n[1,2]\n");
}
#[test]
fn version_flag_prints_and_exits_0() {
let (code, out, _) = run_jq(&["--version"], "");
assert_eq!(code, 0);
assert_eq!(out, format!("jaq {}\n", env!("CARGO_PKG_VERSION")));
}
#[test]
fn tab_and_indent_control_pretty_printing() {
let (code, out, _) = run_jq(&["--tab", "."], "{\"a\":1}");
assert_eq!(code, 0);
assert_eq!(out, "{\n\t\"a\": 1\n}\n");
let (code, out, _) = run_jq(&["--indent", "4", "."], "{\"a\":1}");
assert_eq!(code, 0);
assert_eq!(out, "{\n \"a\": 1\n}\n");
}
#[test]
fn from_file_reads_filter_relative_to_scope_cwd() {
let dir = tempfile::TempDir::new().expect("tempdir");
std::fs::write(dir.path().join("f.jq"), ".a + 1").expect("write filter");
let (code, out, err) =
run_jq_in(dir.path().to_path_buf(), HashMap::new(), &["-f", "f.jq"], "{\"a\":1}");
assert_eq!(code, 0, "stderr: {err:?}");
assert_eq!(out, "2\n");
}
#[test]
fn join_output_omits_newlines() {
let (code, out, _) = run_jq(&["-j", ".[]"], "[\"a\",\"b\"]");
assert_eq!(code, 0);
assert_eq!(out, "ab");
}
#[test]
fn positional_args_after_double_dash_args() {
let (code, out, _) = run_jq(&["-nc", "$ARGS.positional", "--args", "x", "y"], "");
assert_eq!(code, 0);
assert_eq!(out, "[\"x\",\"y\"]\n");
}
+127
View File
@@ -0,0 +1,127 @@
use core::fmt::{self, Display, Formatter};
use std::io::{self, Write};
use crate::{Cli, Val};
struct FormatterFn<F>(F);
impl<F: Fn(&mut Formatter) -> fmt::Result> Display for FormatterFn<F> {
fn fmt(&self, f: &mut Formatter) -> fmt::Result {
self.0(f)
}
}
struct PpOpts {
compact: bool,
indent: String,
sort_keys: bool,
}
impl PpOpts {
fn indent(&self, f: &mut Formatter, level: usize) -> fmt::Result {
if !self.compact {
write!(f, "{}", self.indent.repeat(level))?;
}
Ok(())
}
fn newline(&self, f: &mut Formatter) -> fmt::Result {
if !self.compact {
writeln!(f)?;
}
Ok(())
}
}
fn fmt_seq<T, I, F>(fmt: &mut Formatter, opts: &PpOpts, level: usize, xs: I, f: F) -> fmt::Result
where
I: IntoIterator<Item = T>,
F: Fn(&mut Formatter, T) -> fmt::Result,
{
opts.newline(fmt)?;
let mut iter = xs.into_iter().peekable();
while let Some(x) = iter.next() {
opts.indent(fmt, level + 1)?;
f(fmt, x)?;
if iter.peek().is_some() {
write!(fmt, ",")?;
}
opts.newline(fmt)?;
}
opts.indent(fmt, level)
}
fn fmt_val(f: &mut Formatter, opts: &PpOpts, level: usize, v: &Val) -> fmt::Result {
use yansi::Paint;
match v {
Val::Null | Val::Bool(_) | Val::Int(_) | Val::Float(_) | Val::Num(_) => v.fmt(f),
Val::Str(_) => write!(f, "{}", v.green()),
Val::Arr(a) => {
'['.bold().fmt(f)?;
if !a.is_empty() {
fmt_seq(f, opts, level, &**a, |f, x| fmt_val(f, opts, level + 1, x))?;
}
']'.bold().fmt(f)
},
Val::Obj(o) => {
'{'.bold().fmt(f)?;
let kv = |f: &mut Formatter, (k, val): (&std::rc::Rc<String>, &Val)| {
write!(f, "{}:", Val::Str(k.clone()).bold())?;
if !opts.compact {
write!(f, " ")?;
}
fmt_val(f, opts, level + 1, val)
};
if !o.is_empty() {
if opts.sort_keys {
let mut o: Vec<_> = o.iter().collect();
o.sort_by_key(|(k, _v)| *k);
fmt_seq(f, opts, level, o, kv)
} else {
fmt_seq(f, opts, level, &**o, kv)
}?
}
'}'.bold().fmt(f)
},
}
}
pub fn print(w: &mut (impl Write + ?Sized), cli: &Cli, val: &Val) -> io::Result<()> {
let f = |f: &mut Formatter| {
let opts = PpOpts {
compact: cli.compact_output,
indent: if cli.tab {
String::from("\t")
} else {
" ".repeat(cli.indent)
},
sort_keys: cli.sort_keys,
};
fmt_val(f, &opts, 0, val)
};
match val {
Val::Str(s) if cli.raw_output || cli.join_output => write!(w, "{s}")?,
_ => write!(w, "{}", FormatterFn(f))?,
};
if cli.join_output {
// when running `jaq -jn '"prompt> " | (., input)'`,
// this flush is necessary to make "prompt> " appear first
w.flush()
} else {
writeln!(w)
}
}
/// Runs `f` with a buffered writer over the ctx stdout stream.
///
/// Upstream used an unbuffered lock when stdout was a terminal; the ctx
/// stream never is, so output is always buffered and flushed at the end
/// (flush errors are dropped, matching upstream's `BufWriter` drop).
pub fn with_stdout<T>(f: impl FnOnce(&mut dyn Write) -> T) -> T {
let mut out = io::BufWriter::new(pi_uutils_ctx::stdout());
let res = f(&mut out);
let _ = out.flush();
res
}
+18 -17
View File
@@ -5,31 +5,32 @@
// spell-checker:ignore (ToDO) algo
// pi-uutils: Patched for in-process embedding via the shared `uu-checksum-common` crate,
// which redirects all standard stream I/O and file resolution through `pi-uutils-ctx`.
// pi-uutils: Patched for in-process embedding via the shared
// `uu-checksum-common` crate, which redirects all standard stream I/O and file
// resolution through `pi-uutils-ctx`.
use clap::Command;
use std::ffi::OsString;
use clap::Command;
use uucore::checksum::{AlgoKind, BlakeLength, parse_blake_length};
pub fn run(argv: Vec<OsString>) -> i32 {
let calculate_blake2b_length =
|s: &str| parse_blake_length(AlgoKind::Blake2b, BlakeLength::String(s));
uu_checksum_common::run_standalone_with_length(
"b2sum",
AlgoKind::Blake2b,
uu_app(),
argv,
calculate_blake2b_length,
)
let calculate_blake2b_length =
|s: &str| parse_blake_length(AlgoKind::Blake2b, BlakeLength::String(s));
uu_checksum_common::run_standalone_with_length(
"b2sum",
AlgoKind::Blake2b,
uu_app(),
argv,
calculate_blake2b_length,
)
}
#[inline]
pub fn uu_app() -> Command {
uu_checksum_common::standalone_checksum_app_with_length(
"Print or check BLAKE2b (512-bit) checksums.",
"b2sum [OPTION]... [FILE]...",
)
.name("b2sum")
uu_checksum_common::standalone_checksum_app_with_length(
"Print or check BLAKE2b (512-bit) checksums.",
"b2sum [OPTION]... [FILE]...",
)
.name("b2sum")
}
+36 -31
View File
@@ -5,43 +5,48 @@
pub mod base_common;
use std::{ffi::OsString, io::Write};
use clap::Command;
use std::ffi::OsString;
use std::io::Write;
use uucore::encoding::Format;
/// pi-uutils: safe in-process entry point using invocation-scoped streams.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
}
};
let result = base_common::Config::from(&matches).and_then(|config| {
let mut input = base_common::get_input(&config)?;
base_common::handle_input(&mut input, Format::Base32, config)
});
match result {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
let _ = writeln!(pi_uutils_ctx::stderr(), "base32: {err}");
if code == 0 { 1 } else { code }
}
}
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
let result = base_common::Config::from(&matches).and_then(|config| {
let mut input = base_common::get_input(&config)?;
base_common::handle_input(&mut input, Format::Base32, config)
});
match result {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
let _ = writeln!(pi_uutils_ctx::stderr(), "base32: {err}");
if code == 0 { 1 } else { code }
},
}
}
pub fn uu_app() -> Command {
base_common::base_app(
"encode/decode data and print to standard output\nWith no FILE, or when FILE is -, read standard input.\n\nThe data are encoded as described for the base32 alphabet in RFC 4648.\nWhen decoding, the input may contain newlines in addition to the bytes of the formal base32 alphabet. Use --ignore-garbage to attempt to recover from any other non-alphabet bytes in the encoded stream.".into(),
"base32 [OPTION]... [FILE]".into(),
)
.name("base32")
base_common::base_app(
"encode/decode data and print to standard output\nWith no FILE, or when FILE is -, read \
standard input.\n\nThe data are encoded as described for the base32 alphabet in RFC \
4648.\nWhen decoding, the input may contain newlines in addition to the bytes of the \
formal base32 alphabet. Use --ignore-garbage to attempt to recover from any other \
non-alphabet bytes in the encoded stream."
.into(),
"base32 [OPTION]... [FILE]".into(),
)
.name("base32")
}
File diff suppressed because it is too large Load Diff
+36 -31
View File
@@ -3,44 +3,49 @@
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use std::{ffi::OsString, io::Write};
use clap::Command;
use std::ffi::OsString;
use std::io::Write;
use uu_base32::base_common;
use uucore::encoding::Format;
/// pi-uutils: safe in-process entry point using invocation-scoped streams.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
}
};
let result = base_common::Config::from(&matches).and_then(|config| {
let mut input = base_common::get_input(&config)?;
base_common::handle_input(&mut input, Format::Base64, config)
});
match result {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
let _ = writeln!(pi_uutils_ctx::stderr(), "base64: {err}");
if code == 0 { 1 } else { code }
}
}
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
let result = base_common::Config::from(&matches).and_then(|config| {
let mut input = base_common::get_input(&config)?;
base_common::handle_input(&mut input, Format::Base64, config)
});
match result {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
let _ = writeln!(pi_uutils_ctx::stderr(), "base64: {err}");
if code == 0 { 1 } else { code }
},
}
}
pub fn uu_app() -> Command {
base_common::base_app(
"encode/decode data and print to standard output\nWith no FILE, or when FILE is -, read standard input.\n\nThe data are encoded as described for the base64 alphabet in RFC 3548.\nWhen decoding, the input may contain newlines in addition to the bytes of the formal base64 alphabet. Use --ignore-garbage to attempt to recover from any other non-alphabet bytes in the encoded stream.".into(),
"base64 [OPTION]... [FILE]".into(),
)
.name("base64")
base_common::base_app(
"encode/decode data and print to standard output\nWith no FILE, or when FILE is -, read \
standard input.\n\nThe data are encoded as described for the base64 alphabet in RFC \
3548.\nWhen decoding, the input may contain newlines in addition to the bytes of the \
formal base64 alphabet. Use --ignore-garbage to attempt to recover from any other \
non-alphabet bytes in the encoded stream."
.into(),
"base64 [OPTION]... [FILE]".into(),
)
.name("base64")
}
+111 -114
View File
@@ -6,25 +6,25 @@
// spell-checker:ignore (ToDO) fullname
// pi-uutils: Patched for in-process embedding in the shell.
// All I/O is routed through thread-local stream buffers provided by `pi-uutils-ctx`.
// Command-line arguments are parsed and errors are mapped without process-global
// termination or stdout/stderr pollution.
// All I/O is routed through thread-local stream buffers provided by
// `pi-uutils-ctx`. Command-line arguments are parsed and errors are mapped
// without process-global termination or stdout/stderr pollution.
use clap::builder::ValueParser;
use clap::{Arg, ArgAction, ArgMatches, Command};
use std::ffi::OsString;
use std::io::Write;
use std::path::PathBuf;
use uucore::display::Quotable;
use uucore::error::{UResult, UUsageError};
use std::{ffi::OsString, io::Write, path::PathBuf};
use clap::{Arg, ArgAction, ArgMatches, Command, builder::ValueParser};
use pi_uutils_ctx::format_usage;
use uucore::line_ending::LineEnding;
use uucore::{
display::Quotable,
error::{UResult, UUsageError},
line_ending::LineEnding,
};
pub mod options {
pub static MULTIPLE: &str = "multiple";
pub static NAME: &str = "name";
pub static SUFFIX: &str = "suffix";
pub static ZERO: &str = "zero";
pub static MULTIPLE: &str = "multiple";
pub static NAME: &str = "name";
pub static SUFFIX: &str = "suffix";
pub static ZERO: &str = "zero";
}
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
@@ -56,117 +56,114 @@ pub fn run(argv: Vec<OsString>) -> i32 {
}
fn basename_main(matches: &ArgMatches) -> UResult<()> {
let line_ending = LineEnding::from_zero_flag(matches.get_flag(options::ZERO));
let line_ending = LineEnding::from_zero_flag(matches.get_flag(options::ZERO));
let mut name_args = matches
.get_many::<OsString>(options::NAME)
.unwrap_or_default()
.collect::<Vec<_>>();
if name_args.is_empty() {
return Err(UUsageError::new(
1,
"missing operand".to_string(),
));
}
let multiple_paths = matches.get_one::<OsString>(options::SUFFIX).is_some()
|| matches.get_flag(options::MULTIPLE);
let suffix = if multiple_paths {
matches
.get_one::<OsString>(options::SUFFIX)
.cloned()
.unwrap_or_default()
} else {
// "simple format"
match name_args.len() {
0 => panic!("already checked"),
1 => OsString::default(),
2 => name_args.pop().unwrap().clone(),
_ => {
return Err(UUsageError::new(
1,
format!("extra operand {}", name_args[2].quote()),
));
}
}
};
let mut name_args = matches
.get_many::<OsString>(options::NAME)
.unwrap_or_default()
.collect::<Vec<_>>();
if name_args.is_empty() {
return Err(UUsageError::new(1, "missing operand".to_string()));
}
let multiple_paths =
matches.get_one::<OsString>(options::SUFFIX).is_some() || matches.get_flag(options::MULTIPLE);
let suffix = if multiple_paths {
matches
.get_one::<OsString>(options::SUFFIX)
.cloned()
.unwrap_or_default()
} else {
// "simple format"
match name_args.len() {
0 => panic!("already checked"),
1 => OsString::default(),
2 => name_args.pop().unwrap().clone(),
_ => {
return Err(UUsageError::new(1, format!("extra operand {}", name_args[2].quote())));
},
}
};
//
// Main Program Processing
//
let mut out = pi_uutils_ctx::stdout();
for path in name_args {
out.write_all(&basename(path, &suffix)?)?;
write!(out, "{line_ending}")?;
}
//
// Main Program Processing
//
let mut out = pi_uutils_ctx::stdout();
for path in name_args {
out.write_all(&basename(path, &suffix)?)?;
write!(out, "{line_ending}")?;
}
Ok(())
Ok(())
}
pub fn uu_app() -> Command {
Command::new("basename")
.version(uucore::crate_version!())
.about("Print NAME with any leading directory components removed\nIf specified, also remove a trailing SUFFIX")
.override_usage(format_usage("basename [-z] NAME [SUFFIX]\n basename OPTION... NAME..."))
.infer_long_args(true)
.arg(
Arg::new(options::MULTIPLE)
.short('a')
.long(options::MULTIPLE)
.help("support multiple arguments and treat each as a NAME")
.action(ArgAction::SetTrue)
.overrides_with(options::MULTIPLE),
)
.arg(
Arg::new(options::NAME)
.action(ArgAction::Append)
.value_parser(ValueParser::os_string())
.value_hint(clap::ValueHint::AnyPath)
.hide(true)
.trailing_var_arg(true),
)
.arg(
Arg::new(options::SUFFIX)
.short('s')
.long(options::SUFFIX)
.value_name("SUFFIX")
.value_parser(ValueParser::os_string())
.help("remove a trailing SUFFIX; implies -a")
.overrides_with(options::SUFFIX),
)
.arg(
Arg::new(options::ZERO)
.short('z')
.long(options::ZERO)
.help("end each output line with NUL, not newline")
.action(ArgAction::SetTrue)
.overrides_with(options::ZERO),
)
Command::new("basename")
.version(uucore::crate_version!())
.about(
"Print NAME with any leading directory components removed\nIf specified, also remove a \
trailing SUFFIX",
)
.override_usage(format_usage("basename [-z] NAME [SUFFIX]\n basename OPTION... NAME..."))
.infer_long_args(true)
.arg(
Arg::new(options::MULTIPLE)
.short('a')
.long(options::MULTIPLE)
.help("support multiple arguments and treat each as a NAME")
.action(ArgAction::SetTrue)
.overrides_with(options::MULTIPLE),
)
.arg(
Arg::new(options::NAME)
.action(ArgAction::Append)
.value_parser(ValueParser::os_string())
.value_hint(clap::ValueHint::AnyPath)
.hide(true)
.trailing_var_arg(true),
)
.arg(
Arg::new(options::SUFFIX)
.short('s')
.long(options::SUFFIX)
.value_name("SUFFIX")
.value_parser(ValueParser::os_string())
.help("remove a trailing SUFFIX; implies -a")
.overrides_with(options::SUFFIX),
)
.arg(
Arg::new(options::ZERO)
.short('z')
.long(options::ZERO)
.help("end each output line with NUL, not newline")
.action(ArgAction::SetTrue)
.overrides_with(options::ZERO),
)
}
// We return a Vec<u8>. Returning a seemingly more proper `OsString` would
// require back and forth conversions as we need a &[u8] for printing anyway.
fn basename(fullname: &OsString, suffix: &OsString) -> UResult<Vec<u8>> {
let fullname_bytes = uucore::os_str_as_bytes(fullname)?;
let fullname_bytes = uucore::os_str_as_bytes(fullname)?;
// Handle special case where path ends with /.
if fullname_bytes.ends_with(b"/.") {
return Ok(b".".into());
}
// Handle special case where path ends with /.
if fullname_bytes.ends_with(b"/.") {
return Ok(b".".into());
}
// Convert to path buffer and get last path component
let pb = PathBuf::from(fullname);
// Convert to path buffer and get last path component
let pb = PathBuf::from(fullname);
pb.components().next_back().map_or(Ok([].into()), |c| {
let name = c.as_os_str();
let name_bytes = uucore::os_str_as_bytes(name)?;
if name == suffix {
Ok(name_bytes.into())
} else {
let suffix_bytes = uucore::os_str_as_bytes(suffix)?;
Ok(name_bytes
.strip_suffix(suffix_bytes)
.unwrap_or(name_bytes)
.into())
}
})
pb.components().next_back().map_or(Ok([].into()), |c| {
let name = c.as_os_str();
let name_bytes = uucore::os_str_as_bytes(name)?;
if name == suffix {
Ok(name_bytes.into())
} else {
let suffix_bytes = uucore::os_str_as_bytes(suffix)?;
Ok(name_bytes
.strip_suffix(suffix_bytes)
.unwrap_or(name_bytes)
.into())
}
})
}
+175 -171
View File
@@ -8,210 +8,214 @@ use uucore::checksum::SUPPORTED_ALGORITHMS;
/// List of all options that can be encountered in checksum utils
pub mod options {
// cksum-specific
pub const ALGORITHM: &str = "algorithm";
pub const DEBUG: &str = "debug";
// cksum-specific
pub const ALGORITHM: &str = "algorithm";
pub const DEBUG: &str = "debug";
// positional arg
pub const FILE: &str = "file";
// positional arg
pub const FILE: &str = "file";
pub const UNTAGGED: &str = "untagged";
pub const TAG: &str = "tag";
pub const LENGTH: &str = "length";
pub const RAW: &str = "raw";
pub const BASE64: &str = "base64";
pub const CHECK: &str = "check";
pub const TEXT: &str = "text";
pub const BINARY: &str = "binary";
pub const ZERO: &str = "zero";
pub const UNTAGGED: &str = "untagged";
pub const TAG: &str = "tag";
pub const LENGTH: &str = "length";
pub const RAW: &str = "raw";
pub const BASE64: &str = "base64";
pub const CHECK: &str = "check";
pub const TEXT: &str = "text";
pub const BINARY: &str = "binary";
pub const ZERO: &str = "zero";
// check-specific
pub const STRICT: &str = "strict";
pub const STATUS: &str = "status";
pub const WARN: &str = "warn";
pub const IGNORE_MISSING: &str = "ignore-missing";
pub const QUIET: &str = "quiet";
// check-specific
pub const STRICT: &str = "strict";
pub const STATUS: &str = "status";
pub const WARN: &str = "warn";
pub const IGNORE_MISSING: &str = "ignore-missing";
pub const QUIET: &str = "quiet";
}
/// `ChecksumCommand` is a convenience trait to more easily declare checksum
/// CLI interfaces with
pub trait ChecksumCommand {
fn with_algo(self) -> Self;
fn with_algo(self) -> Self;
fn with_length(self) -> Self;
fn with_length(self) -> Self;
fn with_check_and_opts(self) -> Self;
fn with_check_and_opts(self) -> Self;
fn with_binary(self) -> Self;
fn with_binary(self) -> Self;
fn with_text(self, is_default: bool) -> Self;
fn with_text(self, is_default: bool) -> Self;
fn with_tag(self, is_default: bool) -> Self;
fn with_tag(self, is_default: bool) -> Self;
fn with_untagged(self) -> Self;
fn with_untagged(self) -> Self;
fn with_raw(self) -> Self;
fn with_raw(self) -> Self;
fn with_base64(self) -> Self;
fn with_base64(self) -> Self;
fn with_zero(self) -> Self;
fn with_zero(self) -> Self;
fn with_debug(self) -> Self;
fn with_debug(self) -> Self;
}
impl ChecksumCommand for Command {
fn with_algo(self) -> Self {
self.arg(
Arg::new(options::ALGORITHM)
.long(options::ALGORITHM)
.short('a')
.help("select the digest type to use. See DIGEST below")
.value_name("ALGORITHM")
.value_parser(SUPPORTED_ALGORITHMS),
)
}
fn with_algo(self) -> Self {
self.arg(
Arg::new(options::ALGORITHM)
.long(options::ALGORITHM)
.short('a')
.help("select the digest type to use. See DIGEST below")
.value_name("ALGORITHM")
.value_parser(SUPPORTED_ALGORITHMS),
)
}
fn with_length(self) -> Self {
self.arg(
Arg::new(options::LENGTH)
.long(options::LENGTH)
.short('l')
.help("digest length in bits; must not exceed the maximum and must be a multiple of 8 for BLAKE2b")
.action(ArgAction::Set),
)
}
fn with_length(self) -> Self {
self.arg(
Arg::new(options::LENGTH)
.long(options::LENGTH)
.short('l')
.help(
"digest length in bits; must not exceed the maximum and must be a multiple of 8 \
for BLAKE2b",
)
.action(ArgAction::Set),
)
}
fn with_check_and_opts(self) -> Self {
self.arg(
Arg::new(options::CHECK)
.short('c')
.long(options::CHECK)
.help("read checksums from the FILEs and check them")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::WARN)
.short('w')
.long("warn")
.help("warn about improperly formatted checksum lines")
.action(ArgAction::SetTrue)
.overrides_with_all([options::STATUS, options::QUIET]),
)
.arg(
Arg::new(options::STATUS)
.long("status")
.help("don't output anything, status code shows success")
.action(ArgAction::SetTrue)
.overrides_with_all([options::WARN, options::QUIET]),
)
.arg(
Arg::new(options::QUIET)
.long(options::QUIET)
.help("don't print OK for each successfully verified file")
.action(ArgAction::SetTrue)
.overrides_with_all([options::STATUS, options::WARN]),
)
.arg(
Arg::new(options::IGNORE_MISSING)
.long(options::IGNORE_MISSING)
.help("don't fail or report status for missing files")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::STRICT)
.long(options::STRICT)
.help("exit non-zero for improperly formatted checksum lines")
.action(ArgAction::SetTrue),
)
}
fn with_check_and_opts(self) -> Self {
self
.arg(
Arg::new(options::CHECK)
.short('c')
.long(options::CHECK)
.help("read checksums from the FILEs and check them")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::WARN)
.short('w')
.long("warn")
.help("warn about improperly formatted checksum lines")
.action(ArgAction::SetTrue)
.overrides_with_all([options::STATUS, options::QUIET]),
)
.arg(
Arg::new(options::STATUS)
.long("status")
.help("don't output anything, status code shows success")
.action(ArgAction::SetTrue)
.overrides_with_all([options::WARN, options::QUIET]),
)
.arg(
Arg::new(options::QUIET)
.long(options::QUIET)
.help("don't print OK for each successfully verified file")
.action(ArgAction::SetTrue)
.overrides_with_all([options::STATUS, options::WARN]),
)
.arg(
Arg::new(options::IGNORE_MISSING)
.long(options::IGNORE_MISSING)
.help("don't fail or report status for missing files")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::STRICT)
.long(options::STRICT)
.help("exit non-zero for improperly formatted checksum lines")
.action(ArgAction::SetTrue),
)
}
fn with_binary(self) -> Self {
self.arg(
Arg::new(options::BINARY)
.long(options::BINARY)
.short('b')
.hide(true)
.overrides_with(options::TEXT)
.action(ArgAction::SetTrue),
)
}
fn with_binary(self) -> Self {
self.arg(
Arg::new(options::BINARY)
.long(options::BINARY)
.short('b')
.hide(true)
.overrides_with(options::TEXT)
.action(ArgAction::SetTrue),
)
}
fn with_text(self, is_default: bool) -> Self {
let mut arg = Arg::new(options::TEXT)
.long(options::TEXT)
.short('t')
.action(ArgAction::SetTrue);
fn with_text(self, is_default: bool) -> Self {
let mut arg = Arg::new(options::TEXT)
.long(options::TEXT)
.short('t')
.action(ArgAction::SetTrue);
arg = if is_default {
arg.help("read in text mode (default)")
} else {
arg.hide(true)
};
arg = if is_default {
arg.help("read in text mode (default)")
} else {
arg.hide(true)
};
self.arg(arg)
}
self.arg(arg)
}
fn with_tag(self, default: bool) -> Self {
let mut arg = Arg::new(options::TAG)
.long(options::TAG)
.action(ArgAction::SetTrue);
fn with_tag(self, default: bool) -> Self {
let mut arg = Arg::new(options::TAG)
.long(options::TAG)
.action(ArgAction::SetTrue);
arg = if default {
arg.help("create a BSD style checksum (default)")
} else {
arg.help("create a BSD style checksum")
};
arg = if default {
arg.help("create a BSD style checksum (default)")
} else {
arg.help("create a BSD style checksum")
};
self.arg(arg)
}
self.arg(arg)
}
fn with_untagged(self) -> Self {
self.arg(
Arg::new(options::UNTAGGED)
.long(options::UNTAGGED)
.help("create a reversed style checksum, without digest type")
.overrides_with(options::TAG)
.action(ArgAction::SetTrue),
)
}
fn with_untagged(self) -> Self {
self.arg(
Arg::new(options::UNTAGGED)
.long(options::UNTAGGED)
.help("create a reversed style checksum, without digest type")
.overrides_with(options::TAG)
.action(ArgAction::SetTrue),
)
}
fn with_raw(self) -> Self {
self.arg(
Arg::new(options::RAW)
.long(options::RAW)
.help("emit a raw binary digest, not hexadecimal")
.action(ArgAction::SetTrue),
)
}
fn with_raw(self) -> Self {
self.arg(
Arg::new(options::RAW)
.long(options::RAW)
.help("emit a raw binary digest, not hexadecimal")
.action(ArgAction::SetTrue),
)
}
fn with_base64(self) -> Self {
self.arg(
Arg::new(options::BASE64)
.long(options::BASE64)
.help("emit base64-encoded digests, not hexadecimal")
.action(ArgAction::SetTrue)
// Even though this could easily just override an earlier '--raw',
// GNU cksum does not permit these flags to be combined:
.conflicts_with(options::RAW),
)
}
fn with_base64(self) -> Self {
self.arg(
Arg::new(options::BASE64)
.long(options::BASE64)
.help("emit base64-encoded digests, not hexadecimal")
.action(ArgAction::SetTrue)
// Even though this could easily just override an earlier '--raw',
// GNU cksum does not permit these flags to be combined:
.conflicts_with(options::RAW),
)
}
fn with_zero(self) -> Self {
self.arg(
Arg::new(options::ZERO)
.long(options::ZERO)
.short('z')
.help("end each output line with NUL, not newline, and disable file name escaping")
.action(ArgAction::SetTrue),
)
}
fn with_zero(self) -> Self {
self.arg(
Arg::new(options::ZERO)
.long(options::ZERO)
.short('z')
.help("end each output line with NUL, not newline, and disable file name escaping")
.action(ArgAction::SetTrue),
)
}
fn with_debug(self) -> Self {
self.arg(
Arg::new(options::DEBUG)
.long(options::DEBUG)
.help("print CPU hardware capability detection info used by cksum")
.action(ArgAction::SetTrue),
)
}
fn with_debug(self) -> Self {
self.arg(
Arg::new(options::DEBUG)
.long(options::DEBUG)
.help("print CPU hardware capability detection info used by cksum")
.action(ArgAction::SetTrue),
)
}
}
+227 -233
View File
@@ -5,17 +5,22 @@
// spell-checker:ignore bitlen
use std::ffi::OsStr;
use std::fs::File;
use std::io::{BufReader, Read, Write};
use std::path::Path;
use uucore::checksum::{
AlgoKind, ChecksumError, ReadingMode, SizedAlgoKind, digest_reader, escape_filename,
use std::{
ffi::OsStr,
fs::File,
io::{BufReader, Read, Write},
path::Path,
};
use uucore::error::{FromIo, UResult, USimpleError};
use uucore::line_ending::LineEnding;
use uucore::sum::DigestOutput;
use uucore::{
checksum::{
AlgoKind, ChecksumError, ReadingMode, SizedAlgoKind, digest_reader, escape_filename,
},
error::{FromIo, UResult, USimpleError},
line_ending::LineEnding,
sum::DigestOutput,
};
use crate::report_error;
/// Use the same buffer size as GNU when reading a file to create a checksum
@@ -28,200 +33,192 @@ const READ_BUFFER_SIZE: usize = 32 * 1024;
/// deprecated anyway, it was decided in #9168 to ignore the difference when
/// computing the checksum.
pub struct ChecksumComputeOptions {
/// Which algorithm to use to compute the digest.
pub algo_kind: SizedAlgoKind,
/// Which algorithm to use to compute the digest.
pub algo_kind: SizedAlgoKind,
/// Printing format to use for each checksum.
pub output_format: OutputFormat,
/// Printing format to use for each checksum.
pub output_format: OutputFormat,
/// Whether to finish lines with '\n' or '\0'.
pub line_ending: LineEnding,
/// Whether to finish lines with '\n' or '\0'.
pub line_ending: LineEnding,
}
/// Whether to write the digest as hexadecimal or encoded in base64.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum DigestFormat {
Hexadecimal,
Base64,
Hexadecimal,
Base64,
}
impl DigestFormat {
#[inline]
fn is_base64(self) -> bool {
self == Self::Base64
}
#[inline]
fn is_base64(self) -> bool {
self == Self::Base64
}
}
/// Holds the representation that shall be used for printing a checksum line
#[derive(Debug, PartialEq, Eq)]
pub enum OutputFormat {
/// Raw digest
Raw,
/// Raw digest
Raw,
/// Selected for older algorithms which had their custom formatting
///
/// Default for crc, sysv, bsd
Legacy,
/// Selected for older algorithms which had their custom formatting
///
/// Default for crc, sysv, bsd
Legacy,
/// `$ALGO_NAME ($FILENAME) = $DIGEST`
Tagged(DigestFormat),
/// `$ALGO_NAME ($FILENAME) = $DIGEST`
Tagged(DigestFormat),
/// '$DIGEST $FLAG$FILENAME'
/// where 'flag' depends on the reading mode
///
/// Default for standalone checksum utilities
Untagged(DigestFormat, ReadingMode),
/// '$DIGEST $FLAG$FILENAME'
/// where 'flag' depends on the reading mode
///
/// Default for standalone checksum utilities
Untagged(DigestFormat, ReadingMode),
}
impl OutputFormat {
#[inline]
fn is_raw(&self) -> bool {
*self == Self::Raw
}
#[inline]
fn is_raw(&self) -> bool {
*self == Self::Raw
}
/// Find the correct output format for cksum.
pub fn from_cksum(algo: AlgoKind, tag: bool, binary: bool, raw: bool, base64: bool) -> Self {
// Raw output format takes precedence over anything else.
if raw {
return Self::Raw;
}
/// Find the correct output format for cksum.
pub fn from_cksum(algo: AlgoKind, tag: bool, binary: bool, raw: bool, base64: bool) -> Self {
// Raw output format takes precedence over anything else.
if raw {
return Self::Raw;
}
// Then, if the algo is legacy, takes precedence over the rest
if algo.is_legacy() {
return Self::Legacy;
}
// Then, if the algo is legacy, takes precedence over the rest
if algo.is_legacy() {
return Self::Legacy;
}
let digest_format = if base64 {
DigestFormat::Base64
} else {
DigestFormat::Hexadecimal
};
let digest_format = if base64 {
DigestFormat::Base64
} else {
DigestFormat::Hexadecimal
};
// After that, decide between tagged and untagged output
if tag {
Self::Tagged(digest_format)
} else {
let reading_mode = if binary {
ReadingMode::Binary
} else {
ReadingMode::Text
};
Self::Untagged(digest_format, reading_mode)
}
}
// After that, decide between tagged and untagged output
if tag {
Self::Tagged(digest_format)
} else {
let reading_mode = if binary {
ReadingMode::Binary
} else {
ReadingMode::Text
};
Self::Untagged(digest_format, reading_mode)
}
}
/// Find the correct output format for a standalone checksum util (b2sum,
/// md5sum, etc)
///
/// Since standalone utils can't use the Raw or Legacy output format, it is
/// decided only using the --tag, --binary and --text arguments.
pub fn from_standalone(text: bool, tag: bool) -> Self {
if tag {
Self::Tagged(DigestFormat::Hexadecimal)
} else {
Self::Untagged(
DigestFormat::Hexadecimal,
if text {
ReadingMode::Text
} else {
ReadingMode::Binary
},
)
}
}
/// Find the correct output format for a standalone checksum util (b2sum,
/// md5sum, etc)
///
/// Since standalone utils can't use the Raw or Legacy output format, it is
/// decided only using the --tag, --binary and --text arguments.
pub fn from_standalone(text: bool, tag: bool) -> Self {
if tag {
Self::Tagged(DigestFormat::Hexadecimal)
} else {
Self::Untagged(
DigestFormat::Hexadecimal,
if text {
ReadingMode::Text
} else {
ReadingMode::Binary
},
)
}
}
}
fn print_legacy_checksum(
options: &ChecksumComputeOptions,
filename: &OsStr,
sum: &DigestOutput,
size: usize,
options: &ChecksumComputeOptions,
filename: &OsStr,
sum: &DigestOutput,
size: usize,
) {
debug_assert!(options.algo_kind.is_legacy());
debug_assert!(matches!(sum, DigestOutput::U16(_) | DigestOutput::Crc(_)));
debug_assert!(options.algo_kind.is_legacy());
debug_assert!(matches!(sum, DigestOutput::U16(_) | DigestOutput::Crc(_)));
let (escaped_filename, prefix) = if options.line_ending == LineEnding::Nul {
(filename.to_string_lossy().to_string(), "")
} else {
escape_filename(filename)
};
let (escaped_filename, prefix) = if options.line_ending == LineEnding::Nul {
(filename.to_string_lossy().to_string(), "")
} else {
escape_filename(filename)
};
// Print the sum
match (options.algo_kind, sum) {
(SizedAlgoKind::Sysv, DigestOutput::U16(sum)) => {
let _ = write!(
pi_uutils_ctx::stdout(),
"{prefix}{sum} {}",
size.div_ceil(options.algo_kind.bitlen()),
);
}
(SizedAlgoKind::Bsd, DigestOutput::U16(sum)) => {
// The BSD checksum output is 5 digit integer
let bsd_width = 5;
let _ = write!(
pi_uutils_ctx::stdout(),
"{prefix}{sum:0bsd_width$} {:bsd_width$}",
size.div_ceil(options.algo_kind.bitlen()),
);
}
(SizedAlgoKind::Crc | SizedAlgoKind::Crc32b, DigestOutput::Crc(sum)) => {
let _ = write!(pi_uutils_ctx::stdout(), "{prefix}{sum} {size}");
}
(algo, output) => unreachable!("Bug: Invalid legacy checksum ({algo:?}, {output:?})"),
}
// Print the sum
match (options.algo_kind, sum) {
(SizedAlgoKind::Sysv, DigestOutput::U16(sum)) => {
let _ = write!(
pi_uutils_ctx::stdout(),
"{prefix}{sum} {}",
size.div_ceil(options.algo_kind.bitlen()),
);
},
(SizedAlgoKind::Bsd, DigestOutput::U16(sum)) => {
// The BSD checksum output is 5 digit integer
let bsd_width = 5;
let _ = write!(
pi_uutils_ctx::stdout(),
"{prefix}{sum:0bsd_width$} {:bsd_width$}",
size.div_ceil(options.algo_kind.bitlen()),
);
},
(SizedAlgoKind::Crc | SizedAlgoKind::Crc32b, DigestOutput::Crc(sum)) => {
let _ = write!(pi_uutils_ctx::stdout(), "{prefix}{sum} {size}");
},
(algo, output) => unreachable!("Bug: Invalid legacy checksum ({algo:?}, {output:?})"),
}
// Print the filename after a space if not stdin
if escaped_filename != "-" {
let _ = write!(pi_uutils_ctx::stdout(), " ");
let _dropped_result = pi_uutils_ctx::stdout().write_all(escaped_filename.as_bytes());
}
// Print the filename after a space if not stdin
if escaped_filename != "-" {
let _ = write!(pi_uutils_ctx::stdout(), " ");
let _dropped_result = pi_uutils_ctx::stdout().write_all(escaped_filename.as_bytes());
}
}
fn print_tagged_checksum(options: &ChecksumComputeOptions, filename: &OsStr, sum: &String) {
let (escaped_filename, prefix) = if options.line_ending == LineEnding::Nul {
(filename.to_string_lossy().to_string(), "")
} else {
escape_filename(filename)
};
let (escaped_filename, prefix) = if options.line_ending == LineEnding::Nul {
(filename.to_string_lossy().to_string(), "")
} else {
escape_filename(filename)
};
// Print algo name and opening parenthesis.
let _ = write!(
pi_uutils_ctx::stdout(),
"{prefix}{} (",
options.algo_kind.to_tag()
);
// Print algo name and opening parenthesis.
let _ = write!(pi_uutils_ctx::stdout(), "{prefix}{} (", options.algo_kind.to_tag());
// Print filename
let _dropped_result = pi_uutils_ctx::stdout().write_all(escaped_filename.as_bytes());
// Print filename
let _dropped_result = pi_uutils_ctx::stdout().write_all(escaped_filename.as_bytes());
// Print closing parenthesis and sum
let _ = write!(pi_uutils_ctx::stdout(), ") = {sum}");
// Print closing parenthesis and sum
let _ = write!(pi_uutils_ctx::stdout(), ") = {sum}");
}
fn print_untagged_checksum(
options: &ChecksumComputeOptions,
filename: &OsStr,
sum: &String,
reading_mode: ReadingMode,
options: &ChecksumComputeOptions,
filename: &OsStr,
sum: &String,
reading_mode: ReadingMode,
) {
let (escaped_filename, prefix) = if options.line_ending == LineEnding::Nul {
(filename.to_string_lossy().to_string(), "")
} else {
escape_filename(filename)
};
let (escaped_filename, prefix) = if options.line_ending == LineEnding::Nul {
(filename.to_string_lossy().to_string(), "")
} else {
escape_filename(filename)
};
// Print checksum and reading mode flag
let _ = write!(
pi_uutils_ctx::stdout(),
"{prefix}{sum} {}",
match reading_mode {
ReadingMode::Binary => '*',
ReadingMode::Text => ' ',
}
);
// Print checksum and reading mode flag
let _ = write!(pi_uutils_ctx::stdout(), "{prefix}{sum} {}", match reading_mode {
ReadingMode::Binary => '*',
ReadingMode::Text => ' ',
});
// Print filename
let _dropped_result = pi_uutils_ctx::stdout().write_all(escaped_filename.as_bytes());
// Print filename
let _dropped_result = pi_uutils_ctx::stdout().write_all(escaped_filename.as_bytes());
}
/// Calculate checksum
@@ -229,89 +226,86 @@ fn print_untagged_checksum(
/// # Arguments
///
/// * `options` - CLI options for the assigning checksum algorithm
/// * `files` - A iterator of [`OsStr`] which is a bunch of files that are using for calculating checksum
/// * `files` - A iterator of [`OsStr`] which is a bunch of files that are using
/// for calculating checksum
pub fn perform_checksum_computation<'a, I>(options: ChecksumComputeOptions, files: I) -> UResult<()>
where
I: Iterator<Item = &'a OsStr>,
I: Iterator<Item = &'a OsStr>,
{
let mut files = files.peekable();
let mut files = files.peekable();
while let Some(filename) = files.next() {
// Check that in raw mode, we are not provided with several files.
if options.output_format.is_raw() && files.peek().is_some() {
return Err(Box::new(ChecksumError::RawMultipleFiles));
}
while let Some(filename) = files.next() {
// Check that in raw mode, we are not provided with several files.
if options.output_format.is_raw() && files.peek().is_some() {
return Err(Box::new(ChecksumError::RawMultipleFiles));
}
let filepath = Path::new(filename);
let resolved_filepath = pi_uutils_ctx::resolve(filepath);
let stdin_buf;
let file_buf;
if resolved_filepath.is_dir() {
report_error(&USimpleError::new(1, format!("{}: Is a directory", filepath.display())));
continue;
}
let filepath = Path::new(filename);
let resolved_filepath = pi_uutils_ctx::resolve(filepath);
let stdin_buf;
let file_buf;
if resolved_filepath.is_dir() {
report_error(&USimpleError::new(1, format!("{}: Is a directory", filepath.display())));
continue;
}
// Handle the file input
let mut file = BufReader::with_capacity(
READ_BUFFER_SIZE,
if filename == "-" {
stdin_buf = pi_uutils_ctx::stdin();
Box::new(stdin_buf) as Box<dyn Read>
} else {
file_buf = match File::open(&resolved_filepath) {
Ok(file) => file,
Err(err) => {
report_error(&err.map_err_context(|| filepath.to_string_lossy().into()));
continue;
}
};
Box::new(file_buf) as Box<dyn Read>
},
);
// Handle the file input
let mut file = BufReader::with_capacity(
READ_BUFFER_SIZE,
if filename == "-" {
stdin_buf = pi_uutils_ctx::stdin();
Box::new(stdin_buf) as Box<dyn Read>
} else {
file_buf = match File::open(&resolved_filepath) {
Ok(file) => file,
Err(err) => {
report_error(&err.map_err_context(|| filepath.to_string_lossy().into()));
continue;
},
};
Box::new(file_buf) as Box<dyn Read>
},
);
let mut digest = options.algo_kind.create_digest();
let mut digest = options.algo_kind.create_digest();
// Always compute the "binary" version of the digest, i.e. on Windows,
// never handle CRLFs specifically.
let (digest_output, sz) = digest_reader(&mut digest, &mut file, ReadingMode::Binary)
.map_err_context(|| "failed to read input".to_string())?;
// Always compute the "binary" version of the digest, i.e. on Windows,
// never handle CRLFs specifically.
let (digest_output, sz) = digest_reader(&mut digest, &mut file, ReadingMode::Binary)
.map_err_context(|| "failed to read input".to_string())?;
// Encodes the sum if df is Base64, leaves as-is otherwise.
let encode_sum = |sum: DigestOutput, df: DigestFormat| {
if df.is_base64() {
sum.to_base64()
} else {
sum.to_hex()
}
};
// Encodes the sum if df is Base64, leaves as-is otherwise.
let encode_sum = |sum: DigestOutput, df: DigestFormat| {
if df.is_base64() {
sum.to_base64()
} else {
sum.to_hex()
}
};
match options.output_format {
OutputFormat::Raw => {
// Cannot handle multiple files anyway, output immediately.
digest_output.write_raw(pi_uutils_ctx::stdout())?;
return Ok(());
}
OutputFormat::Legacy => {
print_legacy_checksum(&options, filename, &digest_output, sz);
}
OutputFormat::Tagged(digest_format) => {
print_tagged_checksum(
&options,
filename,
&encode_sum(digest_output, digest_format)?,
);
}
OutputFormat::Untagged(digest_format, reading_mode) => {
print_untagged_checksum(
&options,
filename,
&encode_sum(digest_output, digest_format)?,
reading_mode,
);
}
}
match options.output_format {
OutputFormat::Raw => {
// Cannot handle multiple files anyway, output immediately.
digest_output.write_raw(pi_uutils_ctx::stdout())?;
return Ok(());
},
OutputFormat::Legacy => {
print_legacy_checksum(&options, filename, &digest_output, sz);
},
OutputFormat::Tagged(digest_format) => {
print_tagged_checksum(&options, filename, &encode_sum(digest_output, digest_format)?);
},
OutputFormat::Untagged(digest_format, reading_mode) => {
print_untagged_checksum(
&options,
filename,
&encode_sum(digest_output, digest_format)?,
reading_mode,
);
},
}
let _ = write!(pi_uutils_ctx::stdout(), "{}", options.line_ending);
}
Ok(())
let _ = write!(pi_uutils_ctx::stdout(), "{}", options.line_ending);
}
Ok(())
}
+173 -155
View File
@@ -6,218 +6,236 @@
// pi-uutils: vendored from uutils/coreutils 0.8.0 checksum_common and patched
// to use invocation-scoped I/O and cwd resolution for in-process builtins.
use std::borrow::Borrow;
use std::cell::RefCell;
use std::ffi::OsString;
use std::io::Write;
use std::{borrow::Borrow, cell::RefCell, ffi::OsString, io::Write};
use clap::builder::ValueParser;
use clap::{Arg, ArgAction, ArgMatches, Command, ValueHint};
use uucore::checksum::{AlgoKind, ChecksumError, SizedAlgoKind};
use uucore::error::{UError, UResult};
use uucore::line_ending::LineEnding;
use clap::{Arg, ArgAction, ArgMatches, Command, ValueHint, builder::ValueParser};
use uucore::{
checksum::{AlgoKind, ChecksumError, SizedAlgoKind},
error::{UError, UResult},
line_ending::LineEnding,
};
mod cli;
mod compute;
mod validate;
pub use cli::{options, ChecksumCommand};
pub use cli::{ChecksumCommand, options};
pub use compute::{ChecksumComputeOptions, DigestFormat, OutputFormat};
pub use validate::{ChecksumValidateOptions, ChecksumVerbose};
thread_local! {
static COMMAND_NAME: RefCell<&'static str> = const { RefCell::new("checksum") };
static COMMAND_NAME: RefCell<&'static str> = const { RefCell::new("checksum") };
}
pub(crate) fn command_name() -> &'static str {
COMMAND_NAME.with(|name| *name.borrow())
COMMAND_NAME.with(|name| *name.borrow())
}
pub(crate) fn report_error(error: &dyn std::fmt::Display) {
let _ = writeln!(pi_uutils_ctx::stderr(), "{}: {error}", command_name());
pi_uutils_ctx::set_exit_code(1);
let _ = writeln!(pi_uutils_ctx::stderr(), "{}: {error}", command_name());
pi_uutils_ctx::set_exit_code(1);
}
pub(crate) fn report_warning(message: &str) {
let _ = writeln!(pi_uutils_ctx::stderr(), "{}: {message}", command_name());
let _ = writeln!(pi_uutils_ctx::stderr(), "{}: {message}", command_name());
}
/// Generate a context-safe standalone checksum wrapper.
#[macro_export]
macro_rules! declare_standalone {
($bin:literal, $kind:expr) => {
pub fn run(argv: Vec<::std::ffi::OsString>) -> i32 {
::uu_checksum_common::run_standalone($bin, $kind, uu_app(), argv)
}
($bin:literal, $kind:expr) => {
pub fn run(argv: Vec<::std::ffi::OsString>) -> i32 {
::uu_checksum_common::run_standalone($bin, $kind, uu_app(), argv)
}
#[inline]
pub fn uu_app() -> ::clap::Command {
let (about, usage) = ::uu_checksum_common::standalone_strings($bin);
::uu_checksum_common::standalone_checksum_app(about, usage).name($bin)
}
};
#[inline]
pub fn uu_app() -> ::clap::Command {
let (about, usage) = ::uu_checksum_common::standalone_strings($bin);
::uu_checksum_common::standalone_checksum_app(about, usage).name($bin)
}
};
}
/// English descriptions used by standalone wrappers (localization is
/// intentionally literalized because embedded commands have no global locale).
pub fn standalone_strings(bin: &str) -> (&'static str, &'static str) {
match bin {
"md5sum" => ("Print or check the MD5 checksums", "md5sum [OPTIONS] [FILE]..."),
"sha1sum" => ("Print or check SHA1 (160-bit) checksums", "sha1sum [OPTION]... [FILE]..."),
"sha224sum" => ("Print or check SHA224 (224-bit) checksums", "sha224sum [OPTION]... [FILE]..."),
"sha256sum" => ("Print or check SHA256 (256-bit) checksums", "sha256sum [OPTION]... [FILE]..."),
"sha384sum" => ("Print or check SHA384 (384-bit) checksums", "sha384sum [OPTION]... [FILE]..."),
"sha512sum" => ("Print or check SHA512 (512-bit) checksums", "sha512sum [OPTION]... [FILE]..."),
"b2sum" => ("Print or check BLAKE2b (512-bit) checksums", "b2sum [OPTION]... [FILE]..."),
_ => ("Print or check checksums", "checksum [OPTION]... [FILE]..."),
}
match bin {
"md5sum" => ("Print or check the MD5 checksums", "md5sum [OPTIONS] [FILE]..."),
"sha1sum" => ("Print or check SHA1 (160-bit) checksums", "sha1sum [OPTION]... [FILE]..."),
"sha224sum" => {
("Print or check SHA224 (224-bit) checksums", "sha224sum [OPTION]... [FILE]...")
},
"sha256sum" => {
("Print or check SHA256 (256-bit) checksums", "sha256sum [OPTION]... [FILE]...")
},
"sha384sum" => {
("Print or check SHA384 (384-bit) checksums", "sha384sum [OPTION]... [FILE]...")
},
"sha512sum" => {
("Print or check SHA512 (512-bit) checksums", "sha512sum [OPTION]... [FILE]...")
},
"b2sum" => ("Print or check BLAKE2b (512-bit) checksums", "b2sum [OPTION]... [FILE]..."),
_ => ("Print or check checksums", "checksum [OPTION]... [FILE]..."),
}
}
pub fn run_standalone(
bin: &'static str,
algo: AlgoKind,
cmd: Command,
argv: Vec<OsString>,
) -> i32 {
run_with_optional_length(bin, algo, cmd, argv, None)
pub fn run_standalone(bin: &'static str, algo: AlgoKind, cmd: Command, argv: Vec<OsString>) -> i32 {
run_with_optional_length(bin, algo, cmd, argv, None)
}
/// Context-safe entrypoint for b2sum and other standalone hashes supporting
/// `--length`. The validator is applied only when that option is present.
pub fn run_standalone_with_length(
bin: &'static str,
algo: AlgoKind,
cmd: Command,
argv: Vec<OsString>,
validate_len: fn(&str) -> UResult<usize>,
bin: &'static str,
algo: AlgoKind,
cmd: Command,
argv: Vec<OsString>,
validate_len: fn(&str) -> UResult<usize>,
) -> i32 {
run_with_optional_length(bin, algo, cmd, argv, Some(validate_len))
run_with_optional_length(bin, algo, cmd, argv, Some(validate_len))
}
fn run_with_optional_length(
bin: &'static str,
algo: AlgoKind,
cmd: Command,
argv: Vec<OsString>,
validate_len: Option<fn(&str) -> UResult<usize>>,
bin: &'static str,
algo: AlgoKind,
cmd: Command,
argv: Vec<OsString>,
validate_len: Option<fn(&str) -> UResult<usize>>,
) -> i32 {
COMMAND_NAME.with(|name| *name.borrow_mut() = bin);
let matches = match cmd.try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 2;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
}
};
let length = match validate_len {
Some(validate_len) => match matches
.get_one::<String>(options::LENGTH)
.map(String::as_str)
.map(validate_len)
.transpose()
{
Ok(length) => length,
Err(err) => return finish_error(bin, err),
},
None => None,
};
let text = !matches.get_flag(options::BINARY);
let tag = matches.get_flag(options::TAG);
let format = OutputFormat::from_standalone(text, tag);
match checksum_main(Some(algo), length, matches, format) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => finish_error(bin, err),
}
COMMAND_NAME.with(|name| *name.borrow_mut() = bin);
let matches = match cmd.try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 2;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
let length = match validate_len {
Some(validate_len) => match matches
.get_one::<String>(options::LENGTH)
.map(String::as_str)
.map(validate_len)
.transpose()
{
Ok(length) => length,
Err(err) => return finish_error(bin, err),
},
None => None,
};
let text = !matches.get_flag(options::BINARY);
let tag = matches.get_flag(options::TAG);
let format = OutputFormat::from_standalone(text, tag);
match checksum_main(Some(algo), length, matches, format) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => finish_error(bin, err),
}
}
fn finish_error(bin: &str, err: Box<dyn UError>) -> i32 {
let code = err.code();
let message = err.to_string();
if !message.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "{bin}: {message}");
}
if code == 0 { 1 } else { code }
let code = err.code();
let message = err.to_string();
if !message.is_empty() {
let _ = writeln!(pi_uutils_ctx::stderr(), "{bin}: {message}");
}
if code == 0 { 1 } else { code }
}
pub fn default_checksum_app(about: impl Into<String>, usage: impl Into<String>) -> Command {
Command::new("")
.version("0.8.0")
.about(about.into())
.override_usage(usage.into())
.infer_long_args(true)
.args_override_self(true)
.after_help("With no FILE or when FILE is -, read standard input")
.arg(
Arg::new(options::FILE)
.hide(true)
.action(ArgAction::Append)
.value_parser(ValueParser::os_string())
.default_value("-")
.hide_default_value(true)
.value_hint(ValueHint::FilePath),
)
Command::new("")
.version("0.8.0")
.about(about.into())
.override_usage(usage.into())
.infer_long_args(true)
.args_override_self(true)
.after_help("With no FILE or when FILE is -, read standard input")
.arg(
Arg::new(options::FILE)
.hide(true)
.action(ArgAction::Append)
.value_parser(ValueParser::os_string())
.default_value("-")
.hide_default_value(true)
.value_hint(ValueHint::FilePath),
)
}
pub fn standalone_checksum_app_with_length(
about: impl Into<String>,
usage: impl Into<String>,
about: impl Into<String>,
usage: impl Into<String>,
) -> Command {
default_checksum_app(about, usage)
.with_binary().with_check_and_opts().with_length().with_tag(false).with_text(true).with_zero()
default_checksum_app(about, usage)
.with_binary()
.with_check_and_opts()
.with_length()
.with_tag(false)
.with_text(true)
.with_zero()
}
pub fn standalone_checksum_app(
about: impl Into<String>,
usage: impl Into<String>,
) -> Command {
default_checksum_app(about, usage)
.with_binary().with_check_and_opts().with_tag(false).with_text(true).with_zero()
pub fn standalone_checksum_app(about: impl Into<String>, usage: impl Into<String>) -> Command {
default_checksum_app(about, usage)
.with_binary()
.with_check_and_opts()
.with_tag(false)
.with_text(true)
.with_zero()
}
pub fn checksum_main(
algo: Option<AlgoKind>,
length: Option<usize>,
matches: ArgMatches,
output_format: OutputFormat,
algo: Option<AlgoKind>,
length: Option<usize>,
matches: ArgMatches,
output_format: OutputFormat,
) -> UResult<()> {
let check = matches.get_flag(options::CHECK);
let check_flag = |flag| match (check, matches.get_flag(flag)) {
(_, false) => Ok(false),
(true, true) => Ok(true),
(false, true) => Err(ChecksumError::CheckOnlyFlag(flag.into())),
};
let ignore_missing = check_flag(options::IGNORE_MISSING)?;
let warn = check_flag(options::WARN)?;
let quiet = check_flag(options::QUIET)?;
let strict = check_flag(options::STRICT)?;
let status = check_flag(options::STATUS)?;
let text_flag = matches.get_flag(options::TEXT);
let binary_flag = matches.get_flag(options::BINARY);
let tag = matches.get_flag(options::TAG);
let files = matches.get_many::<OsString>(options::FILE).unwrap().map(Borrow::borrow);
let check = matches.get_flag(options::CHECK);
let check_flag = |flag| match (check, matches.get_flag(flag)) {
(_, false) => Ok(false),
(true, true) => Ok(true),
(false, true) => Err(ChecksumError::CheckOnlyFlag(flag.into())),
};
let ignore_missing = check_flag(options::IGNORE_MISSING)?;
let warn = check_flag(options::WARN)?;
let quiet = check_flag(options::QUIET)?;
let strict = check_flag(options::STRICT)?;
let status = check_flag(options::STATUS)?;
let text_flag = matches.get_flag(options::TEXT);
let binary_flag = matches.get_flag(options::BINARY);
let tag = matches.get_flag(options::TAG);
let files = matches
.get_many::<OsString>(options::FILE)
.unwrap()
.map(Borrow::borrow);
if text_flag && tag { return Err(ChecksumError::TextAfterTag.into()); }
if check {
if algo.is_some_and(AlgoKind::is_legacy) { return Err(ChecksumError::AlgorithmNotSupportedWithCheck.into()); }
if tag { return Err(ChecksumError::TagCheck.into()); }
if binary_flag || text_flag { return Err(ChecksumError::BinaryTextConflict.into()); }
let opts = ChecksumValidateOptions {
ignore_missing,
strict,
verbose: ChecksumVerbose::new(status, quiet, warn),
};
return validate::perform_checksum_validation(files, algo, length, opts);
}
if text_flag && tag {
return Err(ChecksumError::TextAfterTag.into());
}
if check {
if algo.is_some_and(AlgoKind::is_legacy) {
return Err(ChecksumError::AlgorithmNotSupportedWithCheck.into());
}
if tag {
return Err(ChecksumError::TagCheck.into());
}
if binary_flag || text_flag {
return Err(ChecksumError::BinaryTextConflict.into());
}
let opts = ChecksumValidateOptions {
ignore_missing,
strict,
verbose: ChecksumVerbose::new(status, quiet, warn),
};
return validate::perform_checksum_validation(files, algo, length, opts);
}
let algo = SizedAlgoKind::from_unsized(algo.unwrap_or(AlgoKind::Crc), length)?;
let opts = ChecksumComputeOptions {
algo_kind: algo,
output_format,
line_ending: LineEnding::from_zero_flag(matches.get_flag(options::ZERO)),
};
compute::perform_checksum_computation(opts, files)
let algo = SizedAlgoKind::from_unsized(algo.unwrap_or(AlgoKind::Crc), length)?;
let opts = ChecksumComputeOptions {
algo_kind: algo,
output_format,
line_ending: LineEnding::from_zero_flag(matches.get_flag(options::ZERO)),
};
compute::perform_checksum_computation(opts, files)
}
File diff suppressed because it is too large Load Diff
+245 -66
View File
@@ -4,17 +4,21 @@
// file that was distributed with this source code.
// Vendored from uutils/coreutils 0.8.0 and patched for pi-uutils context I/O.
use std::cmp::Ordering;
use std::ffi::{OsStr, OsString};
use std::fs::{self, File};
use std::io::{self, BufRead, BufReader, BufWriter, Read, Write};
use std::path::Path;
use std::{
cmp::Ordering,
ffi::{OsStr, OsString},
fs::{self, File},
io::{self, BufRead, BufReader, BufWriter, Read, Write},
path::Path,
};
use clap::{Arg, ArgAction, ArgMatches, Command};
use pi_uutils_ctx::format_usage;
use uucore::display::Quotable;
use uucore::error::{FromIo, UResult, USimpleError};
use uucore::line_ending::LineEnding;
use uucore::{
display::Quotable,
error::{FromIo, UResult, USimpleError},
line_ending::LineEnding,
};
mod options {
pub const COLUMN_1: &str = "1";
@@ -30,21 +34,30 @@ mod options {
}
#[derive(Clone, Copy)]
enum FileNumber { One, Two }
enum FileNumber {
One,
Two,
}
impl FileNumber {
fn as_str(self) -> &'static str { match self { Self::One => "1", Self::Two => "2" } }
fn as_str(self) -> &'static str {
match self {
Self::One => "1",
Self::Two => "2",
}
}
}
struct OrderChecker {
last_line: Vec<u8>,
file_num: FileNumber,
last_line: Vec<u8>,
file_num: FileNumber,
check_order: bool,
has_error: bool,
has_error: bool,
}
impl OrderChecker {
fn new(file_num: FileNumber, check_order: bool) -> Self {
Self { last_line: Vec::new(), file_num, check_order, has_error: false }
}
fn verify_order(&mut self, line: &[u8]) -> bool {
if self.last_line.is_empty() {
self.last_line = line.to_vec();
@@ -52,7 +65,11 @@ impl OrderChecker {
}
let ordered = line >= self.last_line.as_slice();
if !ordered && !self.has_error {
let _ = writeln!(pi_uutils_ctx::stderr(), "comm: file {} is not in sorted order", self.file_num.as_str());
let _ = writeln!(
pi_uutils_ctx::stderr(),
"comm: file {} is not in sorted order",
self.file_num.as_str()
);
self.has_error = true;
}
self.last_line.clear();
@@ -63,15 +80,18 @@ impl OrderChecker {
struct LineReader {
line_ending: u8,
input: Box<dyn BufRead>,
input: Box<dyn BufRead>,
}
impl LineReader {
fn new(input: Box<dyn BufRead>, line_ending: LineEnding) -> Self {
Self { input, line_ending: line_ending.into() }
}
fn read_line(&mut self, buf: &mut Vec<u8>) -> io::Result<usize> {
let result = self.input.read_until(self.line_ending, buf)?;
if result != 0 && !buf.ends_with(&[self.line_ending]) { buf.push(self.line_ending); }
if result != 0 && !buf.ends_with(&[self.line_ending]) {
buf.push(self.line_ending);
}
Ok(result)
}
}
@@ -79,91 +99,184 @@ impl LineReader {
fn files_identical(path1: &Path, path2: &Path) -> io::Result<bool> {
let m1 = fs::metadata(path1)?;
let m2 = fs::metadata(path2)?;
if !m1.is_file() || !m2.is_file() || m1.len() != m2.len() { return Ok(false); }
if !m1.is_file() || !m2.is_file() || m1.len() != m2.len() {
return Ok(false);
}
let mut a = BufReader::new(File::open(path1)?);
let mut b = BufReader::new(File::open(path2)?);
let mut ba = [0; 8192];
let mut bb = [0; 8192];
loop {
let na = loop { match a.read(&mut ba) { Err(e) if e.kind() == io::ErrorKind::Interrupted => {}, r => break r? } };
let nb = loop { match b.read(&mut bb) { Err(e) if e.kind() == io::ErrorKind::Interrupted => {}, r => break r? } };
if na != nb || ba[..na] != bb[..nb] { return Ok(false); }
if na == 0 { return Ok(true); }
let na = loop {
match a.read(&mut ba) {
Err(e) if e.kind() == io::ErrorKind::Interrupted => {},
r => break r?,
}
};
let nb = loop {
match b.read(&mut bb) {
Err(e) if e.kind() == io::ErrorKind::Interrupted => {},
r => break r?,
}
};
if na != nb || ba[..na] != bb[..nb] {
return Ok(false);
}
if na == 0 {
return Ok(true);
}
}
}
fn write_delimited(writer: &mut impl Write, delim: &[u8], line: &[u8]) -> UResult<()> {
writer.write_all(delim).map_err_context(|| "write error".to_string())?;
writer.write_all(line).map_err_context(|| "write error".to_string())
writer
.write_all(delim)
.map_err_context(|| "write error".to_string())?;
writer
.write_all(line)
.map_err_context(|| "write error".to_string())
}
fn compare(a: &mut LineReader, b: &mut LineReader, name1: &OsStr, name2: &OsStr, delim: &str, opts: &ArgMatches, identical: bool) -> UResult<bool> {
fn compare(
a: &mut LineReader,
b: &mut LineReader,
name1: &OsStr,
name2: &OsStr,
delim: &str,
opts: &ArgMatches,
identical: bool,
) -> UResult<bool> {
let col2 = delim.repeat(usize::from(!opts.get_flag(options::COLUMN_1)));
let col3 = delim.repeat(usize::from(!opts.get_flag(options::COLUMN_1)) + usize::from(!opts.get_flag(options::COLUMN_2)));
let col3 = delim.repeat(
usize::from(!opts.get_flag(options::COLUMN_1))
+ usize::from(!opts.get_flag(options::COLUMN_2)),
);
let mut writer = BufWriter::new(pi_uutils_ctx::stdout());
let (mut ra, mut rb) = (Vec::new(), Vec::new());
let mut na = a.read_line(&mut ra).map_err_context(|| name1.maybe_quote().to_string())?;
let mut nb = b.read_line(&mut rb).map_err_context(|| name2.maybe_quote().to_string())?;
let mut na = a
.read_line(&mut ra)
.map_err_context(|| name1.maybe_quote().to_string())?;
let mut nb = b
.read_line(&mut rb)
.map_err_context(|| name2.maybe_quote().to_string())?;
let (mut n1, mut n2, mut n3) = (0usize, 0usize, 0usize);
let explicit = opts.get_flag(options::CHECK_ORDER);
let should_check = !opts.get_flag(options::NO_CHECK_ORDER) && (explicit || !identical);
let (mut c1, mut c2) = (OrderChecker::new(FileNumber::One, explicit), OrderChecker::new(FileNumber::Two, explicit));
let (mut c1, mut c2) =
(OrderChecker::new(FileNumber::One, explicit), OrderChecker::new(FileNumber::Two, explicit));
let mut delayed_error = false;
while na != 0 || nb != 0 {
let ord = match (na, nb) { (0, _) => Ordering::Greater, (_, 0) => Ordering::Less, _ => ra.cmp(&rb) };
let ord = match (na, nb) {
(0, _) => Ordering::Greater,
(_, 0) => Ordering::Less,
_ => ra.cmp(&rb),
};
match ord {
Ordering::Less => {
if should_check && !c1.verify_order(&ra) { break; }
if !opts.get_flag(options::COLUMN_1) { writer.write_all(&ra).map_err_context(|| "write error".to_string())?; }
ra.clear(); na = a.read_line(&mut ra).map_err_context(|| name1.maybe_quote().to_string())?; n1 += 1;
if should_check && !c1.verify_order(&ra) {
break;
}
if !opts.get_flag(options::COLUMN_1) {
writer
.write_all(&ra)
.map_err_context(|| "write error".to_string())?;
}
ra.clear();
na = a
.read_line(&mut ra)
.map_err_context(|| name1.maybe_quote().to_string())?;
n1 += 1;
},
Ordering::Greater => {
if should_check && !c2.verify_order(&rb) { break; }
if !opts.get_flag(options::COLUMN_2) { write_delimited(&mut writer, col2.as_bytes(), &rb)?; }
rb.clear(); nb = b.read_line(&mut rb).map_err_context(|| name2.maybe_quote().to_string())?; n2 += 1;
if should_check && !c2.verify_order(&rb) {
break;
}
if !opts.get_flag(options::COLUMN_2) {
write_delimited(&mut writer, col2.as_bytes(), &rb)?;
}
rb.clear();
nb = b
.read_line(&mut rb)
.map_err_context(|| name2.maybe_quote().to_string())?;
n2 += 1;
},
Ordering::Equal => {
if should_check && (!c1.verify_order(&ra) || !c2.verify_order(&rb)) { break; }
if !opts.get_flag(options::COLUMN_3) { write_delimited(&mut writer, col3.as_bytes(), &ra)?; }
ra.clear(); rb.clear();
na = a.read_line(&mut ra).map_err_context(|| name1.maybe_quote().to_string())?;
nb = b.read_line(&mut rb).map_err_context(|| name2.maybe_quote().to_string())?; n3 += 1;
if should_check && (!c1.verify_order(&ra) || !c2.verify_order(&rb)) {
break;
}
if !opts.get_flag(options::COLUMN_3) {
write_delimited(&mut writer, col3.as_bytes(), &ra)?;
}
ra.clear();
rb.clear();
na = a
.read_line(&mut ra)
.map_err_context(|| name1.maybe_quote().to_string())?;
nb = b
.read_line(&mut rb)
.map_err_context(|| name2.maybe_quote().to_string())?;
n3 += 1;
},
}
if (c1.has_error || c2.has_error) && !explicit { delayed_error = true; }
if (c1.has_error || c2.has_error) && !explicit {
delayed_error = true;
}
}
if opts.get_flag(options::TOTAL) {
let ending = LineEnding::from_zero_flag(opts.get_flag(options::ZERO_TERMINATED));
write!(writer, "{n1}{delim}{n2}{delim}{n3}{delim}total{ending}").map_err_context(|| "write error".to_string())?;
write!(writer, "{n1}{delim}{n2}{delim}{n3}{delim}total{ending}")
.map_err_context(|| "write error".to_string())?;
}
writer.flush().map_err_context(|| "write error".to_string())?;
writer
.flush()
.map_err_context(|| "write error".to_string())?;
if should_check && (c1.has_error || c2.has_error) {
if delayed_error { let _ = writeln!(pi_uutils_ctx::stderr(), "comm: input is not in sorted order"); }
if delayed_error {
let _ = writeln!(pi_uutils_ctx::stderr(), "comm: input is not in sorted order");
}
Ok(false)
} else { Ok(true) }
} else {
Ok(true)
}
}
fn open_file(name: &OsStr, ending: LineEnding) -> io::Result<LineReader> {
if name == "-" { return Ok(LineReader::new(Box::new(BufReader::new(pi_uutils_ctx::stdin())), ending)); }
if name == "-" {
return Ok(LineReader::new(Box::new(BufReader::new(pi_uutils_ctx::stdin())), ending));
}
let resolved = pi_uutils_ctx::resolve(name);
if fs::metadata(&resolved)?.is_dir() { return Err(io::Error::other("is a directory")); }
if fs::metadata(&resolved)?.is_dir() {
return Err(io::Error::other("is a directory"));
}
Ok(LineReader::new(Box::new(BufReader::new(File::open(resolved)?)), ending))
}
fn comm_main(matches: &ArgMatches) -> UResult<bool> {
let name1 = matches.get_one::<OsString>(options::FILE_1).unwrap();
let name2 = matches.get_one::<OsString>(options::FILE_2).unwrap();
if name1 == "-" && name2 == "-" { return Err(USimpleError::new(1, "standard input is specified twice")); }
if name1 == "-" && name2 == "-" {
return Err(USimpleError::new(1, "standard input is specified twice"));
}
let ending = LineEnding::from_zero_flag(matches.get_flag(options::ZERO_TERMINATED));
let mut f1 = open_file(name1, ending).map_err_context(|| name1.maybe_quote().to_string())?;
let mut f2 = open_file(name2, ending).map_err_context(|| name2.maybe_quote().to_string())?;
let delimiters: Vec<_> = matches.get_many::<String>(options::DELIMITER).unwrap().collect();
let delimiters: Vec<_> = matches
.get_many::<String>(options::DELIMITER)
.unwrap()
.collect();
if delimiters[1..].iter().any(|d| *d != delimiters[0]) {
return Err(USimpleError::new(1, "multiple conflicting output delimiters specified"));
}
let delim = if delimiters[0].is_empty() { "\0" } else { delimiters[0] };
let identical = if name1 == "-" || name2 == "-" { false } else {
files_identical(&pi_uutils_ctx::resolve(name1), &pi_uutils_ctx::resolve(name2)).unwrap_or(false)
let delim = if delimiters[0].is_empty() {
"\0"
} else {
delimiters[0]
};
let identical = if name1 == "-" || name2 == "-" {
false
} else {
files_identical(&pi_uutils_ctx::resolve(name1), &pi_uutils_ctx::resolve(name2))
.unwrap_or(false)
};
compare(&mut f1, &mut f2, name1, name2, delim, matches, identical)
}
@@ -174,14 +287,22 @@ pub fn run(argv: Vec<OsString>) -> i32 {
Ok(m) => m,
Err(e) => {
let rendered = e.to_string();
if e.use_stderr() { let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); return 1; }
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); return 0;
if e.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
match comm_main(&matches) {
Ok(true) => pi_uutils_ctx::exit_code(),
Ok(false) => 1,
Err(e) => { let code = e.code(); let _ = writeln!(pi_uutils_ctx::stderr(), "comm: {e}"); if code == 0 { 1 } else { code } },
Err(e) => {
let code = e.code();
let _ = writeln!(pi_uutils_ctx::stderr(), "comm: {e}");
if code == 0 { 1 } else { code }
},
}
}
@@ -190,15 +311,73 @@ pub fn uu_app() -> Command {
.version(uucore::crate_version!())
.about("Compare sorted files FILE1 and FILE2 line by line.")
.override_usage(format_usage("comm [OPTION]... FILE1 FILE2"))
.infer_long_args(true).args_override_self(true)
.arg(Arg::new(options::COLUMN_1).short('1').help("suppress column 1 (lines unique to FILE1)").action(ArgAction::SetTrue))
.arg(Arg::new(options::COLUMN_2).short('2').help("suppress column 2 (lines unique to FILE2)").action(ArgAction::SetTrue))
.arg(Arg::new(options::COLUMN_3).short('3').help("suppress column 3 (lines that appear in both files)").action(ArgAction::SetTrue))
.arg(Arg::new(options::DELIMITER).long(options::DELIMITER).help("separate columns with STR").value_name("STR").default_value("\t").allow_hyphen_values(true).action(ArgAction::Append).hide_default_value(true))
.arg(Arg::new(options::ZERO_TERMINATED).long(options::ZERO_TERMINATED).short('z').overrides_with(options::ZERO_TERMINATED).help("line delimiter is NUL, not newline").action(ArgAction::SetTrue))
.arg(Arg::new(options::FILE_1).required(true).value_hint(clap::ValueHint::FilePath).value_parser(clap::value_parser!(OsString)))
.arg(Arg::new(options::FILE_2).required(true).value_hint(clap::ValueHint::FilePath).value_parser(clap::value_parser!(OsString)))
.arg(Arg::new(options::TOTAL).long(options::TOTAL).help("output a summary").action(ArgAction::SetTrue))
.arg(Arg::new(options::CHECK_ORDER).long(options::CHECK_ORDER).help("check that input is correctly sorted, even if all input lines are pairable").action(ArgAction::SetTrue))
.arg(Arg::new(options::NO_CHECK_ORDER).long(options::NO_CHECK_ORDER).help("do not check that input is correctly sorted").action(ArgAction::SetTrue).conflicts_with(options::CHECK_ORDER))
.infer_long_args(true)
.args_override_self(true)
.arg(
Arg::new(options::COLUMN_1)
.short('1')
.help("suppress column 1 (lines unique to FILE1)")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::COLUMN_2)
.short('2')
.help("suppress column 2 (lines unique to FILE2)")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::COLUMN_3)
.short('3')
.help("suppress column 3 (lines that appear in both files)")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::DELIMITER)
.long(options::DELIMITER)
.help("separate columns with STR")
.value_name("STR")
.default_value("\t")
.allow_hyphen_values(true)
.action(ArgAction::Append)
.hide_default_value(true),
)
.arg(
Arg::new(options::ZERO_TERMINATED)
.long(options::ZERO_TERMINATED)
.short('z')
.overrides_with(options::ZERO_TERMINATED)
.help("line delimiter is NUL, not newline")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::FILE_1)
.required(true)
.value_hint(clap::ValueHint::FilePath)
.value_parser(clap::value_parser!(OsString)),
)
.arg(
Arg::new(options::FILE_2)
.required(true)
.value_hint(clap::ValueHint::FilePath)
.value_parser(clap::value_parser!(OsString)),
)
.arg(
Arg::new(options::TOTAL)
.long(options::TOTAL)
.help("output a summary")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::CHECK_ORDER)
.long(options::CHECK_ORDER)
.help("check that input is correctly sorted, even if all input lines are pairable")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::NO_CHECK_ORDER)
.long(options::NO_CHECK_ORDER)
.help("do not check that input is correctly sorted")
.action(ArgAction::SetTrue)
.conflicts_with(options::CHECK_ORDER),
)
}
+646 -638
View File
File diff suppressed because it is too large Load Diff
+79 -78
View File
@@ -6,112 +6,113 @@
use memchr::{memchr, memchr2};
// Find the next matching byte sequence positions
// Return (first, last) where haystack[first..last] corresponds to the matched pattern
// Return (first, last) where haystack[first..last] corresponds to the matched
// pattern
pub trait Matcher {
fn next_match(&self, haystack: &[u8]) -> Option<(usize, usize)>;
fn next_match(&self, haystack: &[u8]) -> Option<(usize, usize)>;
}
// Matches for the exact byte sequence pattern
pub struct ExactMatcher<'a> {
needle: &'a [u8],
needle: &'a [u8],
}
impl<'a> ExactMatcher<'a> {
pub fn new(needle: &'a [u8]) -> Self {
assert!(!needle.is_empty());
Self { needle }
}
pub fn new(needle: &'a [u8]) -> Self {
assert!(!needle.is_empty());
Self { needle }
}
}
impl Matcher for ExactMatcher<'_> {
fn next_match(&self, haystack: &[u8]) -> Option<(usize, usize)> {
let mut pos = 0usize;
loop {
let match_idx = memchr(self.needle[0], &haystack[pos..])?;
let match_idx = match_idx + pos; // account for starting from pos
fn next_match(&self, haystack: &[u8]) -> Option<(usize, usize)> {
let mut pos = 0usize;
loop {
let match_idx = memchr(self.needle[0], &haystack[pos..])?;
let match_idx = match_idx + pos; // account for starting from pos
if self.needle.len() == 1 || haystack[match_idx + 1..].starts_with(&self.needle[1..]) {
return Some((match_idx, match_idx + self.needle.len()));
}
if self.needle.len() == 1 || haystack[match_idx + 1..].starts_with(&self.needle[1..]) {
return Some((match_idx, match_idx + self.needle.len()));
}
pos = match_idx + 1;
}
}
pos = match_idx + 1;
}
}
}
// Matches for any number of SPACE or TAB
pub struct WhitespaceMatcher {}
impl Matcher for WhitespaceMatcher {
fn next_match(&self, haystack: &[u8]) -> Option<(usize, usize)> {
let match_idx = memchr2(b' ', b'\t', haystack)?;
let mut skip = match_idx + 1;
fn next_match(&self, haystack: &[u8]) -> Option<(usize, usize)> {
let match_idx = memchr2(b' ', b'\t', haystack)?;
let mut skip = match_idx + 1;
while skip < haystack.len() {
match haystack[skip] {
b' ' | b'\t' => skip += 1,
_ => break,
}
}
while skip < haystack.len() {
match haystack[skip] {
b' ' | b'\t' => skip += 1,
_ => break,
}
}
Some((match_idx, skip))
}
Some((match_idx, skip))
}
}
#[cfg(test)]
mod matcher_tests {
use super::*;
use super::*;
#[test]
fn test_exact_matcher_single_byte() {
let matcher = ExactMatcher::new(":".as_bytes());
// spell-checker:disable
assert_eq!(matcher.next_match("".as_bytes()), None);
assert_eq!(matcher.next_match(":".as_bytes()), Some((0, 1)));
assert_eq!(matcher.next_match(":abcxyz".as_bytes()), Some((0, 1)));
assert_eq!(matcher.next_match("abc:xyz".as_bytes()), Some((3, 4)));
assert_eq!(matcher.next_match("abcxyz:".as_bytes()), Some((6, 7)));
assert_eq!(matcher.next_match("abcxyz".as_bytes()), None);
// spell-checker:enable
}
#[test]
fn test_exact_matcher_single_byte() {
let matcher = ExactMatcher::new(":".as_bytes());
// spell-checker:disable
assert_eq!(matcher.next_match("".as_bytes()), None);
assert_eq!(matcher.next_match(":".as_bytes()), Some((0, 1)));
assert_eq!(matcher.next_match(":abcxyz".as_bytes()), Some((0, 1)));
assert_eq!(matcher.next_match("abc:xyz".as_bytes()), Some((3, 4)));
assert_eq!(matcher.next_match("abcxyz:".as_bytes()), Some((6, 7)));
assert_eq!(matcher.next_match("abcxyz".as_bytes()), None);
// spell-checker:enable
}
#[test]
fn test_exact_matcher_multi_bytes() {
let matcher = ExactMatcher::new("<>".as_bytes());
// spell-checker:disable
assert_eq!(matcher.next_match("".as_bytes()), None);
assert_eq!(matcher.next_match("<>".as_bytes()), Some((0, 2)));
assert_eq!(matcher.next_match("<>abcxyz".as_bytes()), Some((0, 2)));
assert_eq!(matcher.next_match("abc<>xyz".as_bytes()), Some((3, 5)));
assert_eq!(matcher.next_match("abcxyz<>".as_bytes()), Some((6, 8)));
assert_eq!(matcher.next_match("abcxyz".as_bytes()), None);
// spell-checker:enable
}
#[test]
fn test_exact_matcher_multi_bytes() {
let matcher = ExactMatcher::new("<>".as_bytes());
// spell-checker:disable
assert_eq!(matcher.next_match("".as_bytes()), None);
assert_eq!(matcher.next_match("<>".as_bytes()), Some((0, 2)));
assert_eq!(matcher.next_match("<>abcxyz".as_bytes()), Some((0, 2)));
assert_eq!(matcher.next_match("abc<>xyz".as_bytes()), Some((3, 5)));
assert_eq!(matcher.next_match("abcxyz<>".as_bytes()), Some((6, 8)));
assert_eq!(matcher.next_match("abcxyz".as_bytes()), None);
// spell-checker:enable
}
#[test]
fn test_whitespace_matcher_single_space() {
let matcher = WhitespaceMatcher {};
// spell-checker:disable
assert_eq!(matcher.next_match("".as_bytes()), None);
assert_eq!(matcher.next_match(" ".as_bytes()), Some((0, 1)));
assert_eq!(matcher.next_match("\tabcxyz".as_bytes()), Some((0, 1)));
assert_eq!(matcher.next_match("abc\txyz".as_bytes()), Some((3, 4)));
assert_eq!(matcher.next_match("abcxyz ".as_bytes()), Some((6, 7)));
assert_eq!(matcher.next_match("abcxyz".as_bytes()), None);
// spell-checker:enable
}
#[test]
fn test_whitespace_matcher_single_space() {
let matcher = WhitespaceMatcher {};
// spell-checker:disable
assert_eq!(matcher.next_match("".as_bytes()), None);
assert_eq!(matcher.next_match(" ".as_bytes()), Some((0, 1)));
assert_eq!(matcher.next_match("\tabcxyz".as_bytes()), Some((0, 1)));
assert_eq!(matcher.next_match("abc\txyz".as_bytes()), Some((3, 4)));
assert_eq!(matcher.next_match("abcxyz ".as_bytes()), Some((6, 7)));
assert_eq!(matcher.next_match("abcxyz".as_bytes()), None);
// spell-checker:enable
}
#[test]
fn test_whitespace_matcher_multi_spaces() {
let matcher = WhitespaceMatcher {};
// spell-checker:disable
assert_eq!(matcher.next_match("".as_bytes()), None);
assert_eq!(matcher.next_match(" \t ".as_bytes()), Some((0, 3)));
assert_eq!(matcher.next_match("\t\tabcxyz".as_bytes()), Some((0, 2)));
assert_eq!(matcher.next_match("abc \txyz".as_bytes()), Some((3, 5)));
assert_eq!(matcher.next_match("abcxyz ".as_bytes()), Some((6, 8)));
assert_eq!(matcher.next_match("abcxyz".as_bytes()), None);
// spell-checker:enable
}
#[test]
fn test_whitespace_matcher_multi_spaces() {
let matcher = WhitespaceMatcher {};
// spell-checker:disable
assert_eq!(matcher.next_match("".as_bytes()), None);
assert_eq!(matcher.next_match(" \t ".as_bytes()), Some((0, 3)));
assert_eq!(matcher.next_match("\t\tabcxyz".as_bytes()), Some((0, 2)));
assert_eq!(matcher.next_match("abc \txyz".as_bytes()), Some((3, 5)));
assert_eq!(matcher.next_match("abcxyz ".as_bytes()), Some((6, 8)));
assert_eq!(matcher.next_match("abcxyz".as_bytes()), None);
// spell-checker:enable
}
}
+128 -132
View File
@@ -9,172 +9,168 @@ use super::matcher::Matcher;
// Generic searcher that relies on a specific matcher
pub struct Searcher<'a, 'b, M: Matcher> {
matcher: &'a M,
haystack: &'b [u8],
position: usize,
matcher: &'a M,
haystack: &'b [u8],
position: usize,
}
impl<'a, 'b, M: Matcher> Searcher<'a, 'b, M> {
pub fn new(matcher: &'a M, haystack: &'b [u8]) -> Self {
Self {
matcher,
haystack,
position: 0,
}
}
pub fn new(matcher: &'a M, haystack: &'b [u8]) -> Self {
Self { matcher, haystack, position: 0 }
}
}
// Iterate over field delimiters
// Returns (first, last) positions of each sequence, where `haystack[first..last]`
// corresponds to the delimiter.
// Returns (first, last) positions of each sequence, where
// `haystack[first..last]` corresponds to the delimiter.
impl<M: Matcher> Iterator for Searcher<'_, '_, M> {
type Item = (usize, usize);
type Item = (usize, usize);
fn next(&mut self) -> Option<Self::Item> {
let (first, last) = self.matcher.next_match(&self.haystack[self.position..])?;
let result = (first + self.position, last + self.position);
self.position += last;
fn next(&mut self) -> Option<Self::Item> {
let (first, last) = self.matcher.next_match(&self.haystack[self.position..])?;
let result = (first + self.position, last + self.position);
self.position += last;
Some(result)
}
Some(result)
}
}
#[cfg(test)]
mod exact_searcher_tests {
use super::super::matcher::ExactMatcher;
use super::*;
use super::{super::matcher::ExactMatcher, *};
#[test]
fn test_normal() {
let matcher = ExactMatcher::new("a".as_bytes());
let iter = Searcher::new(&matcher, "a.a.a".as_bytes());
let items: Vec<(usize, usize)> = iter.collect();
assert_eq!(vec![(0, 1), (2, 3), (4, 5)], items);
}
#[test]
fn test_normal() {
let matcher = ExactMatcher::new("a".as_bytes());
let iter = Searcher::new(&matcher, "a.a.a".as_bytes());
let items: Vec<(usize, usize)> = iter.collect();
assert_eq!(vec![(0, 1), (2, 3), (4, 5)], items);
}
#[test]
fn test_empty() {
let matcher = ExactMatcher::new("a".as_bytes());
let iter = Searcher::new(&matcher, "".as_bytes());
let items: Vec<(usize, usize)> = iter.collect();
assert!(items.is_empty());
}
#[test]
fn test_empty() {
let matcher = ExactMatcher::new("a".as_bytes());
let iter = Searcher::new(&matcher, "".as_bytes());
let items: Vec<(usize, usize)> = iter.collect();
assert!(items.is_empty());
}
fn test_multibyte(line: &[u8], expected: &[(usize, usize)]) {
let matcher = ExactMatcher::new("ab".as_bytes());
let iter = Searcher::new(&matcher, line);
let items: Vec<(usize, usize)> = iter.collect();
assert_eq!(expected, items);
}
fn test_multibyte(line: &[u8], expected: &[(usize, usize)]) {
let matcher = ExactMatcher::new("ab".as_bytes());
let iter = Searcher::new(&matcher, line);
let items: Vec<(usize, usize)> = iter.collect();
assert_eq!(expected, items);
}
#[test]
fn test_multibyte_normal() {
test_multibyte("...ab...ab...".as_bytes(), &[(3, 5), (8, 10)]);
}
#[test]
fn test_multibyte_normal() {
test_multibyte("...ab...ab...".as_bytes(), &[(3, 5), (8, 10)]);
}
#[test]
fn test_multibyte_needle_head_at_end() {
test_multibyte("a".as_bytes(), &[]);
}
#[test]
fn test_multibyte_needle_head_at_end() {
test_multibyte("a".as_bytes(), &[]);
}
#[test]
fn test_multibyte_starting_needle() {
test_multibyte("ab...ab...".as_bytes(), &[(0, 2), (5, 7)]);
}
#[test]
fn test_multibyte_starting_needle() {
test_multibyte("ab...ab...".as_bytes(), &[(0, 2), (5, 7)]);
}
#[test]
fn test_multibyte_trailing_needle() {
test_multibyte("...ab...ab".as_bytes(), &[(3, 5), (8, 10)]);
}
#[test]
fn test_multibyte_trailing_needle() {
test_multibyte("...ab...ab".as_bytes(), &[(3, 5), (8, 10)]);
}
#[test]
fn test_multibyte_first_byte_false_match() {
test_multibyte("aA..aCaC..ab..aD".as_bytes(), &[(10, 12)]);
}
#[test]
fn test_multibyte_first_byte_false_match() {
test_multibyte("aA..aCaC..ab..aD".as_bytes(), &[(10, 12)]);
}
#[test]
fn test_searcher_with_exact_matcher() {
let matcher = ExactMatcher::new("<>".as_bytes());
let haystack = "<><>a<>b<><>cd<><>".as_bytes();
let mut searcher = Searcher::new(&matcher, haystack);
assert_eq!(searcher.next(), Some((0, 2)));
assert_eq!(searcher.next(), Some((2, 4)));
assert_eq!(searcher.next(), Some((5, 7)));
assert_eq!(searcher.next(), Some((8, 10)));
assert_eq!(searcher.next(), Some((10, 12)));
assert_eq!(searcher.next(), Some((14, 16)));
assert_eq!(searcher.next(), Some((16, 18)));
assert_eq!(searcher.next(), None);
assert_eq!(searcher.next(), None);
}
#[test]
fn test_searcher_with_exact_matcher() {
let matcher = ExactMatcher::new("<>".as_bytes());
let haystack = "<><>a<>b<><>cd<><>".as_bytes();
let mut searcher = Searcher::new(&matcher, haystack);
assert_eq!(searcher.next(), Some((0, 2)));
assert_eq!(searcher.next(), Some((2, 4)));
assert_eq!(searcher.next(), Some((5, 7)));
assert_eq!(searcher.next(), Some((8, 10)));
assert_eq!(searcher.next(), Some((10, 12)));
assert_eq!(searcher.next(), Some((14, 16)));
assert_eq!(searcher.next(), Some((16, 18)));
assert_eq!(searcher.next(), None);
assert_eq!(searcher.next(), None);
}
}
#[cfg(test)]
mod whitespace_searcher_tests {
use super::super::matcher::WhitespaceMatcher;
use super::*;
use super::{super::matcher::WhitespaceMatcher, *};
#[test]
fn test_space() {
let matcher = WhitespaceMatcher {};
let iter = Searcher::new(&matcher, " . . ".as_bytes());
let items: Vec<(usize, usize)> = iter.collect();
assert_eq!(vec![(0, 1), (2, 3), (4, 5)], items);
}
#[test]
fn test_space() {
let matcher = WhitespaceMatcher {};
let iter = Searcher::new(&matcher, " . . ".as_bytes());
let items: Vec<(usize, usize)> = iter.collect();
assert_eq!(vec![(0, 1), (2, 3), (4, 5)], items);
}
#[test]
fn test_tab() {
let matcher = WhitespaceMatcher {};
let iter = Searcher::new(&matcher, "\t.\t.\t".as_bytes());
let items: Vec<(usize, usize)> = iter.collect();
assert_eq!(vec![(0, 1), (2, 3), (4, 5)], items);
}
#[test]
fn test_tab() {
let matcher = WhitespaceMatcher {};
let iter = Searcher::new(&matcher, "\t.\t.\t".as_bytes());
let items: Vec<(usize, usize)> = iter.collect();
assert_eq!(vec![(0, 1), (2, 3), (4, 5)], items);
}
#[test]
fn test_empty() {
let matcher = WhitespaceMatcher {};
let iter = Searcher::new(&matcher, "".as_bytes());
let items: Vec<(usize, usize)> = iter.collect();
assert!(items.is_empty());
}
#[test]
fn test_empty() {
let matcher = WhitespaceMatcher {};
let iter = Searcher::new(&matcher, "".as_bytes());
let items: Vec<(usize, usize)> = iter.collect();
assert!(items.is_empty());
}
fn test_multispace(line: &[u8], expected: &[(usize, usize)]) {
let matcher = WhitespaceMatcher {};
let iter = Searcher::new(&matcher, line);
let items: Vec<(usize, usize)> = iter.collect();
assert_eq!(expected, items);
}
fn test_multispace(line: &[u8], expected: &[(usize, usize)]) {
let matcher = WhitespaceMatcher {};
let iter = Searcher::new(&matcher, line);
let items: Vec<(usize, usize)> = iter.collect();
assert_eq!(expected, items);
}
#[test]
fn test_multispace_normal() {
test_multispace(
"... ... \t...\t ... \t ...".as_bytes(),
&[(3, 5), (8, 10), (13, 15), (18, 21)],
);
}
#[test]
fn test_multispace_normal() {
test_multispace("... ... \t...\t ... \t ...".as_bytes(), &[
(3, 5),
(8, 10),
(13, 15),
(18, 21),
]);
}
#[test]
fn test_multispace_begin() {
test_multispace(" \t\t...".as_bytes(), &[(0, 3)]);
}
#[test]
fn test_multispace_begin() {
test_multispace(" \t\t...".as_bytes(), &[(0, 3)]);
}
#[test]
fn test_multispace_end() {
test_multispace("...\t ".as_bytes(), &[(3, 6)]);
}
#[test]
fn test_multispace_end() {
test_multispace("...\t ".as_bytes(), &[(3, 6)]);
}
#[test]
fn test_searcher_with_whitespace_matcher() {
let matcher = WhitespaceMatcher {};
let haystack = "\t a b \t cd\t\t".as_bytes();
let mut searcher = Searcher::new(&matcher, haystack);
assert_eq!(searcher.next(), Some((0, 2)));
assert_eq!(searcher.next(), Some((3, 4)));
assert_eq!(searcher.next(), Some((5, 8)));
assert_eq!(searcher.next(), Some((10, 12)));
assert_eq!(searcher.next(), None);
assert_eq!(searcher.next(), None);
}
#[test]
fn test_searcher_with_whitespace_matcher() {
let matcher = WhitespaceMatcher {};
let haystack = "\t a b \t cd\t\t".as_bytes();
let mut searcher = Searcher::new(&matcher, haystack);
assert_eq!(searcher.next(), Some((0, 2)));
assert_eq!(searcher.next(), Some((3, 4)));
assert_eq!(searcher.next(), Some((5, 8)));
assert_eq!(searcher.next(), Some((10, 12)));
assert_eq!(searcher.next(), None);
assert_eq!(searcher.next(), None);
}
}
+217 -216
View File
@@ -5,16 +5,15 @@
// pi-uutils: modified for in-process embedding using pi-uutils-ctx streams.
use clap::{Arg, ArgAction, Command, ArgMatches};
use std::borrow::Cow;
use std::ffi::OsString;
use std::io::Write;
use uucore::error::{UResult, UUsageError};
use std::{borrow::Cow, ffi::OsString, io::Write};
use clap::{Arg, ArgAction, ArgMatches, Command};
use pi_uutils_ctx::format_usage;
use uucore::error::{UResult, UUsageError};
mod options {
pub const ZERO: &str = "zero";
pub const DIR: &str = "dir";
pub const ZERO: &str = "zero";
pub const DIR: &str = "dir";
}
/// Perform dirname as pure string manipulation per POSIX/GNU behavior.
@@ -38,63 +37,63 @@ mod options {
///
/// See issue #8910 and similar fix in basename (#8373, commit c5268a897).
fn dirname_string_manipulation(path_bytes: &[u8]) -> Cow<'_, [u8]> {
if path_bytes.is_empty() {
return Cow::Borrowed(b".");
}
if path_bytes.is_empty() {
return Cow::Borrowed(b".");
}
let mut bytes = path_bytes;
let mut bytes = path_bytes;
// Step 1: Strip trailing slashes (but not if the entire path is slashes)
let all_slashes = bytes.iter().all(|&b| b == b'/');
if all_slashes {
return Cow::Borrowed(b"/");
}
// Step 1: Strip trailing slashes (but not if the entire path is slashes)
let all_slashes = bytes.iter().all(|&b| b == b'/');
if all_slashes {
return Cow::Borrowed(b"/");
}
while bytes.len() > 1 && bytes.ends_with(b"/") {
bytes = &bytes[..bytes.len() - 1];
}
while bytes.len() > 1 && bytes.ends_with(b"/") {
bytes = &bytes[..bytes.len() - 1];
}
// Step 2: Check if it ends with `/.` and strip the `/+.` pattern
if bytes.ends_with(b".") && bytes.len() >= 2 {
let dot_pos = bytes.len() - 1;
if bytes[dot_pos - 1] == b'/' {
// Find where the slashes before the dot start
let mut slash_start = dot_pos - 1;
while slash_start > 0 && bytes[slash_start - 1] == b'/' {
slash_start -= 1;
}
// Return the stripped result
if slash_start == 0 {
// Result would be empty
return if path_bytes.starts_with(b"/") {
Cow::Borrowed(b"/")
} else {
Cow::Borrowed(b".")
};
}
return Cow::Borrowed(&bytes[..slash_start]);
}
}
// Step 2: Check if it ends with `/.` and strip the `/+.` pattern
if bytes.ends_with(b".") && bytes.len() >= 2 {
let dot_pos = bytes.len() - 1;
if bytes[dot_pos - 1] == b'/' {
// Find where the slashes before the dot start
let mut slash_start = dot_pos - 1;
while slash_start > 0 && bytes[slash_start - 1] == b'/' {
slash_start -= 1;
}
// Return the stripped result
if slash_start == 0 {
// Result would be empty
return if path_bytes.starts_with(b"/") {
Cow::Borrowed(b"/")
} else {
Cow::Borrowed(b".")
};
}
return Cow::Borrowed(&bytes[..slash_start]);
}
}
// Step 3: Normal dirname - find last / and remove everything after it
if let Some(last_slash_pos) = bytes.iter().rposition(|&b| b == b'/') {
// Found a slash, remove everything after it
let mut result = &bytes[..last_slash_pos];
// Step 3: Normal dirname - find last / and remove everything after it
if let Some(last_slash_pos) = bytes.iter().rposition(|&b| b == b'/') {
// Found a slash, remove everything after it
let mut result = &bytes[..last_slash_pos];
// Strip trailing slashes from result (but keep at least one if at the start)
while result.len() > 1 && result.ends_with(b"/") {
result = &result[..result.len() - 1];
}
// Strip trailing slashes from result (but keep at least one if at the start)
while result.len() > 1 && result.ends_with(b"/") {
result = &result[..result.len() - 1];
}
if result.is_empty() {
return Cow::Borrowed(b"/");
}
if result.is_empty() {
return Cow::Borrowed(b"/");
}
return Cow::Borrowed(result);
}
return Cow::Borrowed(result);
}
// No slash found, return "."
Cow::Borrowed(b".")
// No slash found, return "."
Cow::Borrowed(b".")
}
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
@@ -102,196 +101,198 @@ fn dirname_string_manipulation(path_bytes: &[u8]) -> Cow<'_, [u8]> {
/// streams, and maps the `UResult` to an exit code, so it is safe to run inside
/// the host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
}
};
match dirname_main(&matches) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
let _ = writeln!(pi_uutils_ctx::stderr(), "dirname: {err}");
if code == 0 { 1 } else { code }
}
}
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
match dirname_main(&matches) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
let _ = writeln!(pi_uutils_ctx::stderr(), "dirname: {err}");
if code == 0 { 1 } else { code }
},
}
}
fn dirname_main(matches: &ArgMatches) -> UResult<()> {
let dirnames: Vec<OsString> = matches
.get_many::<OsString>(options::DIR)
.unwrap_or_default()
.cloned()
.collect();
let dirnames: Vec<OsString> = matches
.get_many::<OsString>(options::DIR)
.unwrap_or_default()
.cloned()
.collect();
if dirnames.is_empty() {
return Err(UUsageError::new(1, "missing operand".to_string()));
}
if dirnames.is_empty() {
return Err(UUsageError::new(1, "missing operand".to_string()));
}
let line_ending = if matches.get_flag(options::ZERO) {
b"\0" as &[u8]
} else {
b"\n" as &[u8]
};
let line_ending = if matches.get_flag(options::ZERO) {
b"\0" as &[u8]
} else {
b"\n" as &[u8]
};
let mut stdout = pi_uutils_ctx::stdout();
let mut stdout = pi_uutils_ctx::stdout();
for path in &dirnames {
let path_bytes = uucore::os_str_as_bytes(path.as_os_str())?;
let result = dirname_string_manipulation(path_bytes);
for path in &dirnames {
let path_bytes = uucore::os_str_as_bytes(path.as_os_str())?;
let result = dirname_string_manipulation(path_bytes);
stdout.write_all(&result)?;
stdout.write_all(line_ending)?;
}
stdout.write_all(&result)?;
stdout.write_all(line_ending)?;
}
Ok(())
Ok(())
}
pub fn uu_app() -> Command {
Command::new("dirname")
.about("Strip last component from file name")
.version(uucore::crate_version!())
.override_usage(format_usage("dirname [OPTION] NAME..."))
.args_override_self(true)
.infer_long_args(true)
.after_help("Output each NAME with its last non-slash component and trailing slashes\n removed; if NAME contains no /'s, output '.' (meaning the current directory).")
.arg(
Arg::new(options::ZERO)
.long(options::ZERO)
.short('z')
.help("separate output with NUL rather than newline")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::DIR)
.hide(true)
.action(ArgAction::Append)
.value_hint(clap::ValueHint::AnyPath)
.value_parser(clap::value_parser!(OsString)),
)
Command::new("dirname")
.about("Strip last component from file name")
.version(uucore::crate_version!())
.override_usage(format_usage("dirname [OPTION] NAME..."))
.args_override_self(true)
.infer_long_args(true)
.after_help(
"Output each NAME with its last non-slash component and trailing slashes\n removed; if \
NAME contains no /'s, output '.' (meaning the current directory).",
)
.arg(
Arg::new(options::ZERO)
.long(options::ZERO)
.short('z')
.help("separate output with NUL rather than newline")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::DIR)
.hide(true)
.action(ArgAction::Append)
.value_hint(clap::ValueHint::AnyPath)
.value_parser(clap::value_parser!(OsString)),
)
}
#[cfg(test)]
mod tests {
use super::*;
use std::sync::Arc;
use std::collections::HashMap;
use std::path::PathBuf;
use pi_uutils_ctx::ScopeIo;
use parking_lot::Mutex;
use std::{collections::HashMap, path::PathBuf, sync::Arc};
fn run_test(args: Vec<&str>) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
use parking_lot::Mutex;
use pi_uutils_ctx::ScopeIo;
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
self.buf.lock().write(buf)
}
fn flush(&mut self) -> std::io::Result<()> {
self.buf.lock().flush()
}
}
use super::*;
let io = ScopeIo {
stdin: Box::new(std::io::empty()),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd: PathBuf::from("."),
env: HashMap::new(),
cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)),
};
fn run_test(args: Vec<&str>) -> (i32, String, String) {
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
let argv: Vec<OsString> = std::iter::once("dirname")
.chain(args)
.map(OsString::from)
.collect();
#[derive(Clone)]
struct SharedWriter {
buf: Arc<Mutex<Vec<u8>>>,
}
impl Write for SharedWriter {
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
self.buf.lock().write(buf)
}
let code = pi_uutils_ctx::scope(io, || {
run(argv)
});
fn flush(&mut self) -> std::io::Result<()> {
self.buf.lock().flush()
}
}
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
let io = ScopeIo {
stdin: Box::new(std::io::empty()),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
cwd: PathBuf::from("."),
env: HashMap::new(),
cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)),
};
(code, out_str, err_str)
}
let argv: Vec<OsString> = std::iter::once("dirname")
.chain(args)
.map(OsString::from)
.collect();
#[test]
fn test_normal() {
let (code, stdout, stderr) = run_test(vec!["foo/bar"]);
assert_eq!(code, 0);
assert_eq!(stdout, "foo\n");
assert_eq!(stderr, "");
}
let code = pi_uutils_ctx::scope(io, || run(argv));
#[test]
fn test_trailing_slash() {
let (code, stdout, stderr) = run_test(vec!["foo/bar/"]);
assert_eq!(code, 0);
assert_eq!(stdout, "foo\n");
assert_eq!(stderr, "");
}
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
#[test]
fn test_root() {
let (code, stdout, stderr) = run_test(vec!["/"]);
assert_eq!(code, 0);
assert_eq!(stdout, "/\n");
assert_eq!(stderr, "");
}
(code, out_str, err_str)
}
#[test]
fn test_multiple() {
let (code, stdout, stderr) = run_test(vec!["a/b", "c/d/e"]);
assert_eq!(code, 0);
assert_eq!(stdout, "a\nc/d\n");
assert_eq!(stderr, "");
}
#[test]
fn test_normal() {
let (code, stdout, stderr) = run_test(vec!["foo/bar"]);
assert_eq!(code, 0);
assert_eq!(stdout, "foo\n");
assert_eq!(stderr, "");
}
#[test]
fn test_zero_delimited() {
let (code, stdout, stderr) = run_test(vec!["-z", "a/b", "c/d/e"]);
assert_eq!(code, 0);
assert_eq!(stdout, "a\0c/d\0");
assert_eq!(stderr, "");
}
#[test]
fn test_trailing_slash() {
let (code, stdout, stderr) = run_test(vec!["foo/bar/"]);
assert_eq!(code, 0);
assert_eq!(stdout, "foo\n");
assert_eq!(stderr, "");
}
#[test]
fn test_help() {
let (code, stdout, stderr) = run_test(vec!["--help"]);
assert_eq!(code, 0);
assert!(stdout.contains("Usage:"));
assert!(stdout.contains("Strip last component"));
assert_eq!(stderr, "");
}
#[test]
fn test_root() {
let (code, stdout, stderr) = run_test(vec!["/"]);
assert_eq!(code, 0);
assert_eq!(stdout, "/\n");
assert_eq!(stderr, "");
}
#[test]
fn test_invalid_arg() {
let (code, stdout, stderr) = run_test(vec!["--invalid-flag"]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert!(stderr.contains("unexpected argument"));
}
#[test]
fn test_multiple() {
let (code, stdout, stderr) = run_test(vec!["a/b", "c/d/e"]);
assert_eq!(code, 0);
assert_eq!(stdout, "a\nc/d\n");
assert_eq!(stderr, "");
}
#[test]
fn test_missing_operand() {
let (code, stdout, stderr) = run_test(vec![]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert!(stderr.contains("missing operand"));
}
#[test]
fn test_zero_delimited() {
let (code, stdout, stderr) = run_test(vec!["-z", "a/b", "c/d/e"]);
assert_eq!(code, 0);
assert_eq!(stdout, "a\0c/d\0");
assert_eq!(stderr, "");
}
#[test]
fn test_help() {
let (code, stdout, stderr) = run_test(vec!["--help"]);
assert_eq!(code, 0);
assert!(stdout.contains("Usage:"));
assert!(stdout.contains("Strip last component"));
assert_eq!(stderr, "");
}
#[test]
fn test_invalid_arg() {
let (code, stdout, stderr) = run_test(vec!["--invalid-flag"]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert!(stderr.contains("unexpected argument"));
}
#[test]
fn test_missing_operand() {
let (code, stdout, stderr) = run_test(vec![]);
assert_eq!(code, 1);
assert_eq!(stdout, "");
assert!(stderr.contains("missing operand"));
}
}
+145 -36
View File
@@ -3,16 +3,21 @@
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use std::{
cell::RefCell,
ffi::OsString,
fs::File,
io::{BufRead, BufReader, Read, Write},
iter::Cycle,
rc::Rc,
slice::Iter,
};
use clap::{Arg, ArgAction, Command};
use std::cell::RefCell;
use std::ffi::OsString;
use std::fs::File;
use std::io::{BufRead, BufReader, Read, Write};
use std::iter::Cycle;
use std::rc::Rc;
use std::slice::Iter;
use uucore::error::{UResult, USimpleError, strip_errno};
use uucore::i18n::charmap::mb_char_len;
use uucore::{
error::{UResult, USimpleError, strip_errno},
i18n::charmap::mb_char_len,
};
mod options {
pub const DELIMITER: &str = "delimiters";
@@ -39,8 +44,16 @@ pub fn run(argv: Vec<OsString>) -> i32 {
let serial = matches.get_flag(options::SERIAL);
let delimiters = matches.get_one::<OsString>(options::DELIMITER).unwrap();
let files = matches.get_many::<OsString>(options::FILE).unwrap().cloned().collect();
let line_ending = if matches.get_flag(options::ZERO_TERMINATED) { b'\0' } else { b'\n' };
let files = matches
.get_many::<OsString>(options::FILE)
.unwrap()
.cloned()
.collect();
let line_ending = if matches.get_flag(options::ZERO_TERMINATED) {
b'\0'
} else {
b'\n'
};
match paste(files, serial, delimiters, line_ending) {
Ok(()) => pi_uutils_ctx::exit_code(),
@@ -58,13 +71,46 @@ pub fn uu_app() -> Command {
.about("Merge lines of files")
.override_usage(pi_uutils_ctx::format_usage("paste [OPTION]... [FILE]..."))
.infer_long_args(true)
.arg(Arg::new(options::SERIAL).long(options::SERIAL).short('s').help("paste one file at a time instead of in parallel").action(ArgAction::SetTrue))
.arg(Arg::new(options::DELIMITER).long(options::DELIMITER).short('d').help("reuse characters from LIST instead of TABs").value_name("LIST").default_value("\t").hide_default_value(true).value_parser(clap::value_parser!(OsString)))
.arg(Arg::new(options::FILE).value_name("FILE").action(ArgAction::Append).default_value("-").value_hint(clap::ValueHint::FilePath).value_parser(clap::value_parser!(OsString)))
.arg(Arg::new(options::ZERO_TERMINATED).long(options::ZERO_TERMINATED).short('z').help("line delimiter is NUL, not newline").action(ArgAction::SetTrue))
.arg(
Arg::new(options::SERIAL)
.long(options::SERIAL)
.short('s')
.help("paste one file at a time instead of in parallel")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::DELIMITER)
.long(options::DELIMITER)
.short('d')
.help("reuse characters from LIST instead of TABs")
.value_name("LIST")
.default_value("\t")
.hide_default_value(true)
.value_parser(clap::value_parser!(OsString)),
)
.arg(
Arg::new(options::FILE)
.value_name("FILE")
.action(ArgAction::Append)
.default_value("-")
.value_hint(clap::ValueHint::FilePath)
.value_parser(clap::value_parser!(OsString)),
)
.arg(
Arg::new(options::ZERO_TERMINATED)
.long(options::ZERO_TERMINATED)
.short('z')
.help("line delimiter is NUL, not newline")
.action(ArgAction::SetTrue),
)
}
fn paste(filenames: Vec<OsString>, serial: bool, delimiters: &OsString, line_ending: u8) -> UResult<()> {
fn paste(
filenames: Vec<OsString>,
serial: bool,
delimiters: &OsString,
line_ending: u8,
) -> UResult<()> {
let delimiters = parse_delimiters(delimiters)?;
// pi-uutils: all `-` operands share the scoped stdin and consume it in order.
let stdin = Rc::new(RefCell::new(BufReader::new(pi_uutils_ctx::stdin())));
@@ -94,7 +140,9 @@ fn paste(filenames: Vec<OsString>, serial: bool, delimiters: &OsString, line_end
for source in &mut sources {
output.clear();
loop {
if source.read_until(line_ending, &mut output)? == 0 { break; }
if source.read_until(line_ending, &mut output)? == 0 {
break;
}
remove_trailing_line_ending(line_ending, &mut output);
delimiter_state.write_delimiter(&mut output);
}
@@ -118,7 +166,9 @@ fn paste(filenames: Vec<OsString>, serial: bool, delimiters: &OsString, line_end
}
delimiter_state.write_delimiter(&mut output);
}
if eof_count == source_count { break; }
if eof_count == source_count {
break;
}
delimiter_state.remove_trailing_delimiter(&mut output);
stdout.write_all(&output)?;
stdout.write_all(&[line_ending])?;
@@ -128,18 +178,26 @@ fn paste(filenames: Vec<OsString>, serial: bool, delimiters: &OsString, line_end
Ok(())
}
fn write_single_input_source(writer: &mut impl Write, mut source: InputSource, line_ending: u8) -> UResult<()> {
fn write_single_input_source(
writer: &mut impl Write,
mut source: InputSource,
line_ending: u8,
) -> UResult<()> {
let mut buffer = [0_u8; 8192];
let mut has_data = false;
let mut last_byte = line_ending;
loop {
let count = source.read(&mut buffer)?;
if count == 0 { break; }
if count == 0 {
break;
}
has_data = true;
last_byte = buffer[count - 1];
writer.write_all(&buffer[..count])?;
}
if has_data && last_byte != line_ending { writer.write_all(&[line_ending])?; }
if has_data && last_byte != line_ending {
writer.write_all(&[line_ending])?;
}
Ok(())
}
@@ -151,14 +209,29 @@ fn parse_delimiters(delimiters: &OsString) -> UResult<Box<[Box<[u8]>]>> {
if bytes[i] == b'\\' {
i += 1;
if i >= bytes.len() {
return Err(USimpleError::new(1, format!("delimiter list ends with an unescaped backslash: {}", delimiters.to_string_lossy())));
return Err(USimpleError::new(
1,
format!(
"delimiter list ends with an unescaped backslash: {}",
delimiters.to_string_lossy()
),
));
}
match bytes[i] {
b'0' => result.push(Box::new([])), b'\\' => result.push(Box::new([b'\\'])),
b'n' => result.push(Box::new([b'\n'])), b't' => result.push(Box::new([b'\t'])),
b'b' => result.push(Box::new([b'\x08'])), b'f' => result.push(Box::new([b'\x0c'])),
b'r' => result.push(Box::new([b'\r'])), b'v' => result.push(Box::new([b'\x0b'])),
_ => { let len = mb_char_len(&bytes[i..]).min(bytes.len() - i); result.push(Box::from(&bytes[i..i + len])); i += len; continue; }
b'0' => result.push(Box::new([])),
b'\\' => result.push(Box::new(*b"\\")),
b'n' => result.push(Box::new(*b"\n")),
b't' => result.push(Box::new(*b"\t")),
b'b' => result.push(Box::new(*b"\x08")),
b'f' => result.push(Box::new(*b"\x0c")),
b'r' => result.push(Box::new(*b"\r")),
b'v' => result.push(Box::new(*b"\x0b")),
_ => {
let len = mb_char_len(&bytes[i..]).min(bytes.len() - i);
result.push(Box::from(&bytes[i..i + len]));
i += len;
continue;
},
}
i += 1;
} else {
@@ -171,13 +244,19 @@ fn parse_delimiters(delimiters: &OsString) -> UResult<Box<[Box<[u8]>]>> {
}
fn remove_trailing_line_ending(line_ending: u8, output: &mut Vec<u8>) {
if output.last() == Some(&line_ending) { output.pop(); }
if output.last() == Some(&line_ending) {
output.pop();
}
}
enum DelimiterState<'a> {
NoDelimiters,
OneDelimiter(&'a [u8]),
MultipleDelimiters { current: &'a [u8], delimiters: &'a [Box<[u8]>], iterator: Cycle<Iter<'a, Box<[u8]>>> },
MultipleDelimiters {
current: &'a [u8],
delimiters: &'a [Box<[u8]>],
iterator: Cycle<Iter<'a, Box<[u8]>>>,
},
}
impl<'a> DelimiterState<'a> {
@@ -186,21 +265,40 @@ impl<'a> DelimiterState<'a> {
[] => Self::NoDelimiters,
[only] if only.is_empty() => Self::NoDelimiters,
[only] => Self::OneDelimiter(only),
[first, ..] => Self::MultipleDelimiters { current: first, delimiters, iterator: delimiters.iter().cycle() },
[first, ..] => Self::MultipleDelimiters {
current: first,
delimiters,
iterator: delimiters.iter().cycle(),
},
}
}
fn reset_to_first_delimiter(&mut self) {
if let Self::MultipleDelimiters { delimiters, iterator, .. } = self { *iterator = delimiters.iter().cycle(); }
if let Self::MultipleDelimiters { delimiters, iterator, .. } = self {
*iterator = delimiters.iter().cycle();
}
}
fn remove_trailing_delimiter(&self, output: &mut Vec<u8>) {
let len = match self { Self::NoDelimiters => return, Self::OneDelimiter(d) => d.len(), Self::MultipleDelimiters { current, .. } => current.len() };
if len > 0 { output.truncate(output.len().saturating_sub(len)); }
let len = match self {
Self::NoDelimiters => return,
Self::OneDelimiter(d) => d.len(),
Self::MultipleDelimiters { current, .. } => current.len(),
};
if len > 0 {
output.truncate(output.len().saturating_sub(len));
}
}
fn write_delimiter(&mut self, output: &mut Vec<u8>) {
match self {
Self::NoDelimiters => {},
Self::OneDelimiter(d) => output.extend_from_slice(d),
Self::MultipleDelimiters { current, iterator, .. } => { let d = iterator.next().unwrap(); output.extend_from_slice(d); *current = d; },
Self::MultipleDelimiters { current, iterator, .. } => {
let d = iterator.next().unwrap();
output.extend_from_slice(d);
*current = d;
},
}
}
}
@@ -214,13 +312,24 @@ impl InputSource {
fn read(&mut self, buf: &mut [u8]) -> UResult<usize> {
Ok(match self {
Self::File(reader) => reader.read(buf)?,
Self::StandardInput(stdin) => stdin.try_borrow_mut().map_err(|err| USimpleError::new(1, format!("standard input is already borrowed: {err}")))?.read(buf)?,
Self::StandardInput(stdin) => stdin
.try_borrow_mut()
.map_err(|err| {
USimpleError::new(1, format!("standard input is already borrowed: {err}"))
})?
.read(buf)?,
})
}
fn read_until(&mut self, byte: u8, buf: &mut Vec<u8>) -> UResult<usize> {
Ok(match self {
Self::File(reader) => reader.read_until(byte, buf)?,
Self::StandardInput(stdin) => stdin.try_borrow_mut().map_err(|err| USimpleError::new(1, format!("standard input is already borrowed: {err}")))?.read_until(byte, buf)?,
Self::StandardInput(stdin) => stdin
.try_borrow_mut()
.map_err(|err| {
USimpleError::new(1, format!("standard input is already borrowed: {err}"))
})?
.read_until(byte, buf)?,
})
}
}
+26
View File
@@ -0,0 +1,26 @@
# Vendored from uutils/sed commit b37e23fa987888572e02e4e9b6906b3ede749bc6,
# patched for in-process embedding through pi-uutils-ctx.
[package]
name = "uu_sed"
version = "0.1.1"
edition = "2024"
license = "MIT"
description = "sed ~ (uutils) stream editor for filtering and transforming text (vendored + patched for in-process embedding)"
[lib]
path = "src/lib.rs"
[dependencies]
clap = { version = "4.5", features = ["wrap_help", "cargo"] }
fancy-regex = "0.18"
memchr = "2.7"
regex = "1.11"
tempfile = "3"
uucore = { version = "0.9.0", features = ["libc"] }
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[target.'cfg(unix)'.dependencies]
memmap2 = "0.9"
[dev-dependencies]
parking_lot = "0.12"
+21
View File
@@ -0,0 +1,21 @@
MIT License
Copyright (c) 2025 Diomidis Spinellis
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.
+246
View File
@@ -0,0 +1,246 @@
// This file is part of the uutils sed package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
//! Vendored, patched `sed` from uutils/sed, wired to run in-process as a
//! shell builtin via [`pi_uutils_ctx`].
//!
//! Upstream: <https://github.com/uutils/sed>
//! Pinned commit: `b37e23fa987888572e02e4e9b6906b3ede749bc6` (default-branch
//! HEAD, 2026-07-10, version 0.1.1).
//!
//! Patches applied for in-process embedding:
//! - all stdio goes through the `pi_uutils_ctx` streams,
//! - every path operand resolves against the shell working directory via
//! `pi_uutils_ctx::resolve`,
//! - no `std::process::exit`: `q`/`Q` exit codes flow through
//! `pi_uutils_ctx::set_exit_code`, clap errors are rendered manually,
//! - the `s///e` shell escape spawns with the shell's cwd and piped stdio,
//! - output is never assumed to be a terminal (no `-l` width auto-detect, no
//! tty-triggered unbuffered mode).
pub mod sed;
use std::{ffi::OsString, io::Write};
/// In-process builtin entry point. The host installs a [`pi_uutils_ctx`]
/// scope (stdio + working directory + environment) on a dedicated blocking
/// thread, then calls this.
///
/// Unlike upstream's `main` (which `std::process::exit`s on the result of
/// `uumain`), this returns the exit code so it is safe to run inside the
/// long-lived host shell process.
pub fn run(argv: Vec<OsString>) -> i32 {
// A reused blocking thread may still hold `w`/`s///w` writers registered
// by a previous invocation that failed before flushing; drop them.
sed::named_writer::reset();
let matches = match sed::uu_app().try_get_matches_from(sed::normalize_args(argv)) {
Ok(m) => m,
Err(e) => {
let rendered = e.to_string();
if e.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
// Upstream prints help and exits 1 when invoked without any argument.
if !matches.args_present() {
let _ = write!(pi_uutils_ctx::stdout(), "{}", sed::uu_app().render_help());
return 1;
}
match sed::sed_main(&matches) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(e) => {
let code = e.code();
let _ = writeln!(pi_uutils_ctx::stderr(), "sed: {e}");
if code == 0 { 1 } else { code }
},
}
}
#[cfg(test)]
mod tests {
use std::{
collections::HashMap,
ffi::OsString,
io::{self, Write},
path::PathBuf,
sync::{Arc, atomic::AtomicBool},
};
use parking_lot::Mutex;
use super::run;
/// `Send` writer capturing everything a run writes to a scope stream.
#[derive(Clone, Default)]
struct SharedBuf(Arc<Mutex<Vec<u8>>>);
impl Write for SharedBuf {
fn write(&mut self, buf: &[u8]) -> io::Result<usize> {
self.0.lock().extend_from_slice(buf);
Ok(buf.len())
}
fn flush(&mut self) -> io::Result<()> {
Ok(())
}
}
impl SharedBuf {
fn take(&self) -> String {
String::from_utf8(self.0.lock().clone()).expect("utf8 stream")
}
}
/// Drive `run()` under a pi-uutils-ctx scope with `cwd` as the shell
/// working directory; returns (exit code, stdout, stderr).
fn run_sed_in(cwd: PathBuf, stdin: &[u8], args: &[&str]) -> (i32, String, String) {
let stdout = SharedBuf::default();
let stderr = SharedBuf::default();
let argv: Vec<OsString> = std::iter::once("sed")
.chain(args.iter().copied())
.map(OsString::from)
.collect();
let code = pi_uutils_ctx::scope(
pi_uutils_ctx::ScopeIo {
stdin: Box::new(io::Cursor::new(stdin.to_vec())),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(stdout.clone()),
stderr: Box::new(stderr.clone()),
cwd,
env: HashMap::new(),
cancel: Arc::new(AtomicBool::new(false)),
},
|| run(argv),
);
(code, stdout.take(), stderr.take())
}
fn run_sed(stdin: &[u8], args: &[&str]) -> (i32, String, String) {
run_sed_in(PathBuf::from("."), stdin, args)
}
#[test]
fn substitutes_basic_from_stdin() {
let (code, out, err) = run_sed(b"hello\n", &["s/hello/world/"]);
assert_eq!(code, 0);
assert_eq!(out, "world\n");
assert!(err.is_empty(), "unexpected stderr: {err}");
}
#[test]
fn quiet_prints_address_range() {
let (code, out, _) = run_sed(b"a\nb\nc\nd\n", &["-n", "2,3p"]);
assert_eq!(code, 0);
assert_eq!(out, "b\nc\n");
}
#[test]
fn substitution_global_flag() {
let (code, out, _) = run_sed(b"aaa\n", &["s/a/b/g"]);
assert_eq!(code, 0);
assert_eq!(out, "bbb\n");
}
#[test]
fn substitution_numbered_occurrence() {
let (code, out, _) = run_sed(b"aaa\n", &["s/a/b/2"]);
assert_eq!(code, 0);
assert_eq!(out, "aba\n");
}
#[test]
fn ere_capture_groups_swap() {
let (code, out, _) = run_sed(b"john smith\n", &["-E", r"s/([a-z]+) ([a-z]+)/\2 \1/"]);
assert_eq!(code, 0);
assert_eq!(out, "smith john\n");
}
#[test]
fn bre_backreference_in_pattern() {
let (code, out, _) = run_sed(b"abab\nabcd\n", &["-n", r"/\(ab\)\1/p"]);
assert_eq!(code, 0);
assert_eq!(out, "abab\n");
}
#[test]
fn hold_space_tac() {
let (code, out, _) = run_sed(b"1\n2\n3\n", &["1!G;h;$!d"]);
assert_eq!(code, 0);
assert_eq!(out, "3\n2\n1\n");
}
#[test]
fn transliterates() {
let (code, out, _) = run_sed(b"abcabc\n", &["y/abc/xyz/"]);
assert_eq!(code, 0);
assert_eq!(out, "xyzxyz\n");
}
#[test]
fn multiple_expressions_compose_in_order() {
let (code, out, _) = run_sed(b"a\n", &["-e", "s/a/b/", "-e", "s/b/c/"]);
assert_eq!(code, 0);
assert_eq!(out, "c\n");
}
#[test]
fn q_with_operand_propagates_exit_code() {
let (code, out, _) = run_sed(b"one\ntwo\nthree\n", &["2q42"]);
assert_eq!(code, 42);
assert_eq!(out, "one\ntwo\n");
}
#[test]
fn q_stops_before_later_lines() {
let (code, out, _) = run_sed(b"one\ntwo\n", &["1q"]);
assert_eq!(code, 0);
assert_eq!(out, "one\n");
}
#[test]
fn in_place_edits_relative_path_against_scope_cwd() {
let dir = tempfile::tempdir().unwrap();
std::fs::write(dir.path().join("file.txt"), "x marks\n").unwrap();
let (code, out, err) =
run_sed_in(dir.path().to_path_buf(), b"", &["-i", "s/x/y/", "file.txt"]);
assert_eq!(code, 0, "stderr: {err}");
assert!(out.is_empty(), "in-place edit must not print: {out}");
assert_eq!(std::fs::read_to_string(dir.path().join("file.txt")).unwrap(), "y marks\n");
}
#[test]
fn in_place_backup_suffix_keeps_original() {
let dir = tempfile::tempdir().unwrap();
std::fs::write(dir.path().join("file.txt"), "x marks\n").unwrap();
let (code, _, err) =
run_sed_in(dir.path().to_path_buf(), b"", &["-i.bak", "s/x/y/", "file.txt"]);
assert_eq!(code, 0, "stderr: {err}");
assert_eq!(std::fs::read_to_string(dir.path().join("file.txt")).unwrap(), "y marks\n");
assert_eq!(std::fs::read_to_string(dir.path().join("file.txt.bak")).unwrap(), "x marks\n");
}
#[test]
fn null_data_mode_substitutes_per_record() {
let (code, out, _) = run_sed(b"a\0b\0", &["-z", "s/a/X/"]);
assert_eq!(code, 0);
assert_eq!(out, "X\0b\0");
}
#[test]
fn unknown_option_diagnoses_on_stderr() {
let (code, out, err) = run_sed(b"", &["--definitely-not-an-option", "p"]);
assert_ne!(code, 0);
assert!(out.is_empty(), "usage errors must not write stdout: {out}");
assert!(!err.is_empty(), "expected a diagnostic on stderr");
}
}
+550
View File
@@ -0,0 +1,550 @@
// Definitions for the compiled code data structures
//
// SPDX-License-Identifier: MIT
// Copyright (c) 2025 Diomidis Spinellis
//
// This file is part of the uutils sed package.
// It is licensed under the MIT License.
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use std::path::PathBuf; // For file descriptors and equivalent
use std::{cell::RefCell, collections::HashMap, rc::Rc};
use uucore::error::UResult;
use crate::sed::{
error_handling::{ScriptLocation, runtime_error},
fast_regex::{Captures, Match, Regex},
named_writer::NamedWriter,
script_char_provider::ScriptCharProvider,
script_line_provider::ScriptLineProvider,
};
#[derive(Debug, Default, Clone)]
/// Compilation and processing options provided mostly through the
/// command-line interface
pub struct ProcessingContext {
// Command-line flags with corresponding names
pub all_output_files: bool,
pub debug: bool,
pub regex_extended: bool,
pub follow_symlinks: bool,
pub in_place: bool,
pub in_place_suffix: Option<String>,
pub length: usize,
pub quiet: bool,
pub posix: bool,
pub separate: bool,
pub sandbox: bool,
pub unbuffered: bool,
pub null_data: bool,
// Other context
/// Currently processed input file name (not script) in quoted form
pub input_name: String,
/// Current input line number
pub line_number: usize,
/// True if this is the last address of a range
pub last_address: bool,
/// True if the line read is the last line
pub last_line: bool,
/// True if the file is the last file of the ones specified
pub last_file: bool,
/// Stop processing further input.
pub stop_processing: bool,
/// Previously compiled RE, saved for reuse when specifying an empty RE
pub saved_regex: Option<Regex>,
/// Modification of input processing action
// This is required to avoid doubly borrowing the reader in the 'N'
// command.
pub input_action: Option<InputAction>,
/// Hold space
pub hold: StringSpace,
/// Nesting of { } at compile time
pub parsed_block_nesting: usize,
/// Command associated with each label
pub label_to_command_map: HashMap<String, Rc<RefCell<Command>>>,
/// Commands with a (latchable and resetable) address range
pub range_commands: Vec<Rc<RefCell<Command>>>,
/// True if a substitution was made as specified in the t command
pub substitution_made: bool,
/// Elements to append at the end of each command processing cycle
pub append_elements: Vec<AppendElement>,
}
#[derive(Clone, Debug)]
/// Elements that shall be appended at the end of each command processing cycle
pub enum AppendElement {
Text(Rc<str>), // The specified text string
Path(PathBuf), // The contents of the specified file path
}
#[derive(Clone, Debug, Default, PartialEq)]
/// A space mirroring IOChunk, but only with a String
pub struct StringSpace {
pub content: String, // Line content without newline
pub has_newline: bool, // True if \n-terminated
}
#[derive(Debug)]
/// Types of address specifications that precede commands
pub enum Address {
Re(Option<Regex>), // Line that matches (optional) regex
Line(usize), // Specific line
RelLine(usize), // Relative line
Last, // Last line
StepMatch(usize), // Lines matching specified step from first
StepEnd(usize), // Range ending at specified step from first
}
#[derive(Debug)]
/// A single part of an RE replacement
pub enum ReplacementPart {
Literal(String), // Normal text
WholeMatch, // &
Group(u32), // \1 to \9
}
// The maximum value allowed in regex quantifier
pub const RE_DUP_MAX: usize = 32767;
/// Regex modes (BRE or ERE)
#[derive(Copy, Clone, Debug)]
pub enum RegexMode {
Basic,
Extended,
}
#[derive(Debug)]
/// All specified replacements for an RE
pub struct ReplacementTemplate {
pub parts: Vec<ReplacementPart>,
pub max_group_number: usize, // Highest used group number (e.g. 8 for \8)
}
impl Default for ReplacementTemplate {
/// Create an empty template.
fn default() -> Self {
ReplacementTemplate::new(Vec::new())
}
}
impl ReplacementTemplate {
/// Construct from the parts
pub fn new(parts: Vec<ReplacementPart>) -> Self {
let max_group_number = parts
.iter()
.filter_map(|part| match part {
ReplacementPart::Group(n) => Some(*n),
_ => None,
})
.max()
.unwrap_or(0);
Self { parts, max_group_number: max_group_number.try_into().unwrap() }
}
/// Apply the template to the given RE captures.
/// Example:
/// let result = regex.replace_all(input, |caps: &Captures| {
/// template.apply_captures(&command, caps) });
/// Returns an error if a backreference in the template was not matched by
/// the RE.
pub fn apply_captures(&self, command: &Command, caps: &Captures) -> UResult<String> {
let mut result = String::new();
// Invalid group numbers may end here through (unkown at compile time)
// reused REs.
if self.max_group_number > caps.len() - 1 {
return runtime_error(
&command.location,
format!("invalid reference \\{} on command's RHS", self.max_group_number),
);
}
for part in &self.parts {
match part {
ReplacementPart::Literal(s) => result.push_str(s),
ReplacementPart::WholeMatch => {
result.push_str(caps.get(0)?.map(|m| m.as_str()).unwrap_or_default());
},
ReplacementPart::Group(n) => {
let i: usize = (*n).try_into().unwrap();
result.push_str(caps.get(i)?.map(|m| m.as_str()).unwrap_or_default());
},
}
}
Ok(result)
}
/// Apply the template to the given RE single match.
pub fn apply_match(&self, m: &Match) -> String {
let mut result = String::new();
for part in &self.parts {
match part {
ReplacementPart::Literal(s) => result.push_str(s),
ReplacementPart::WholeMatch => result.push_str(m.as_str()),
ReplacementPart::Group(_) => {
panic!("unexpected Regex group replacement")
},
}
}
result
}
}
#[derive(Debug, Default)]
/// Substitution command
pub struct Substitution {
pub regex: Option<Regex>, // Regular expression
pub replacement: ReplacementTemplate, // Specified broken-down replacement
pub occurrence: usize, // Which occurrence to substitute
pub print_flag: bool, // True if 'p' flag
pub ignore_case: bool, // True if 'I' flag
pub execute: bool, // True if 'e' flag (GNU extension)
pub multiline: bool, // True if 'm' or 'M' flag (GNU extension)
pub write_file: Option<Rc<RefCell<NamedWriter>>>, // Writer to file if 'w' flag is used
}
/// The block of the first and most common Unicode characters:
/// ASCII, Latin Extended, Greek, Curillic, Coptic, Arabic, etc.
/// It comprises all UCS-2 characters. We use a fast lookup array for these.
const COMMON_UNICODE: usize = 2048;
#[derive(Debug)]
/// Transliteration command (y)
pub struct Transliteration {
fast: [char; COMMON_UNICODE],
slow: HashMap<char, char>,
}
impl Default for Transliteration {
/// Create a new Transliteration with identity mapping for the fast-path.
fn default() -> Self {
let mut fast = ['\0'; COMMON_UNICODE];
for (i, slot) in fast.iter_mut().enumerate() {
*slot = char::from_u32(i as u32).unwrap_or('\0');
}
Self { fast, slow: HashMap::new() }
}
}
impl Transliteration {
/// Create through character mappings from `source` to `target`.
pub fn from_strings(source: &str, target: &str) -> Self {
let mut result = Self::default();
for (from, to) in source.chars().zip(target.chars()) {
result.insert(from, to);
}
result
}
/// Set a transliteration mapping from one character to another.
fn insert(&mut self, from: char, to: char) {
let cp = from as usize;
if cp < COMMON_UNICODE {
self.fast[cp] = to;
} else {
self.slow.insert(from, to);
}
}
/// Look up a character transliteration.
pub fn lookup(&self, ch: char) -> char {
let cp = ch as usize;
if cp < COMMON_UNICODE {
self.fast[cp]
} else {
self.slow.get(&ch).copied().unwrap_or(ch)
}
}
}
#[derive(Debug)]
/// An internally compiled command.
pub struct Command {
pub code: char, // Command code
pub addr1: Option<Address>, // Start address
pub addr2: Option<Address>, // End address
pub non_select: bool, // True if '!'
pub start_line: Option<usize>, // Start line number (or None if unlatched)
pub data: CommandData, // Command-specific data
pub next: Option<Rc<RefCell<Command>>>, // Pointer to next command
pub location: ScriptLocation, // Command's definition location
}
impl Default for Command {
fn default() -> Self {
Command {
code: '_',
addr1: None,
addr2: None,
non_select: false,
start_line: None,
data: CommandData::None,
next: None,
location: ScriptLocation::default(),
}
}
}
impl Command {
/// Construct with position information from the given providers.
pub fn at_position(lines: &ScriptLineProvider, line: &ScriptCharProvider) -> Self {
Command { location: ScriptLocation::at_position(lines, line), ..Default::default() }
}
}
#[derive(Debug)]
/// Command-specific data
/// After parsing, t, b Label elements are converted into BranchTarget ones.
pub enum CommandData {
None,
BranchTarget(Option<Rc<RefCell<Command>>>), // Commands for 'b', 't', '{'
Label(Option<String>), // Label name for 'b', 't', ':'
Path(PathBuf), // File path for 'r'
NamedWriter(Rc<RefCell<NamedWriter>>), // File output for 'w'
Number(usize), // Number for 'l', 'q', 'Q' (GNU)
Substitution(Box<Substitution>), // Substitute command 's'
Text(Rc<str>), // Text for 'a', 'c', 'i'
Transliteration(Box<Transliteration>), // Transliteration command 'y'
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
/// Flag for space modifications
pub enum SpaceFlag {
Append, // Append to contents
Replace, // Replace contents
}
#[derive(Debug, Clone)]
/// Action to execute after reading a new input line
pub struct InputAction {
/// Next command to execute (rather than commands from start)
pub next_command: Option<Rc<RefCell<Command>>>,
/// Data to prepend to the read contents
pub prepend: String,
}
#[cfg(test)]
mod tests {
use super::*;
use crate::sed::fast_io::IOChunk;
// Return the captures for the RE applied to the specified string
fn caps_for<'a>(re: &str, chunk: &'a mut IOChunk) -> Captures<'a> {
Regex::new(re)
.unwrap()
.captures(chunk)
.unwrap()
.expect("captures")
}
#[test]
// s/foo//
fn test_empty_template() {
let template = ReplacementTemplate::default();
let input = &mut IOChunk::new_from_str("foo");
let caps = caps_for("foo", input);
let cmd = Command::default();
let result = template.apply_captures(&cmd, &caps).unwrap();
assert_eq!(result, "");
}
#[test]
// s/abc/hello/
fn test_literal_only() {
let template = ReplacementTemplate::new(vec![ReplacementPart::Literal("hello".into())]);
let input = &mut IOChunk::new_from_str("abc");
let caps = caps_for("abc", input);
let cmd = Command::default();
let result = template.apply_captures(&cmd, &caps).unwrap();
assert_eq!(result, "hello");
}
#[test]
// s/foo\d+/got: &/
fn test_whole_match() {
let template = ReplacementTemplate::new(vec![
ReplacementPart::Literal("got: ".into()),
ReplacementPart::WholeMatch,
]);
let input = &mut IOChunk::new_from_str("foo42");
let caps = caps_for(r"foo\d+", input);
let cmd = Command::default();
let result = template.apply_captures(&cmd, &caps).unwrap();
assert_eq!(result, "got: foo42");
}
#[test]
// s/foo(\d+)/number: \1/
fn test_backreference() {
let template = ReplacementTemplate::new(vec![
ReplacementPart::Literal("number: ".into()),
ReplacementPart::Group(1),
]);
let input = &mut IOChunk::new_from_str("foo42");
let caps = caps_for(r"foo(\d+)", input);
let cmd = Command::default();
let result = template.apply_captures(&cmd, &caps).unwrap();
assert_eq!(result, "number: 42");
}
#[test]
// s/(\w+):(\d+)/key: \1, value: \2/
fn test_multiple_parts() {
let template = ReplacementTemplate::new(vec![
ReplacementPart::Literal("key: ".into()),
ReplacementPart::Group(1),
ReplacementPart::Literal(", value: ".into()),
ReplacementPart::Group(2),
]);
let input = &mut IOChunk::new_from_str("x:123");
let caps = caps_for(r"(\w+):(\d+)", input);
let cmd = Command::default();
let result = template.apply_captures(&cmd, &caps).unwrap();
assert_eq!(result, "key: x, value: 123");
}
#[test]
// s/(\w+):(\d+)/key: \1, value: \3/
fn test_invalid_group() {
let template = ReplacementTemplate::new(vec![
ReplacementPart::Literal("key: ".into()),
ReplacementPart::Group(1),
ReplacementPart::Literal(", value: ".into()),
ReplacementPart::Group(3),
]);
let input = &mut IOChunk::new_from_str("x:123");
let caps = caps_for(r"(\w+):(\d+)", input);
let cmd = Command::default();
let result = template.apply_captures(&cmd, &caps);
assert!(result.is_err());
let msg = result.unwrap_err().to_string();
assert!(msg.contains("invalid reference \\3"));
}
// max_group_number
#[test]
fn test_max_group_number_with_groups() {
let template = ReplacementTemplate::new(vec![
ReplacementPart::Literal("a".into()),
ReplacementPart::Group(2),
ReplacementPart::WholeMatch,
ReplacementPart::Group(5),
ReplacementPart::Literal("z".into()),
]);
assert_eq!(template.max_group_number, 5);
}
#[test]
fn test_max_group_number_without_groups() {
let template = ReplacementTemplate::new(vec![
ReplacementPart::Literal("no".into()),
ReplacementPart::WholeMatch,
ReplacementPart::Literal("groups".into()),
]);
assert_eq!(template.max_group_number, 0);
}
// Transliteration
// Creation and internal functions
#[test]
fn test_identity_lookup_fast_path() {
let t = Transliteration::default();
assert_eq!(t.lookup('A'), 'A');
assert_eq!(t.lookup('z'), 'z');
assert_eq!(t.lookup('\u{07FF}'), '\u{07FF}'); // highest 2-byte UTF-8 char
}
#[test]
fn test_identity_lookup_slow_path() {
let t = Transliteration::default();
assert_eq!(t.lookup('\u{0800}'), '\u{0800}'); // just outside fast path
assert_eq!(t.lookup('\u{1F600}'), '\u{1F600}'); // 😀
}
#[test]
fn test_insert_and_lookup_fast_path() {
let mut t = Transliteration::default();
t.insert('a', 'α');
t.insert('b', 'β');
assert_eq!(t.lookup('a'), 'α');
assert_eq!(t.lookup('b'), 'β');
assert_eq!(t.lookup('c'), 'c'); // unchanged
}
#[test]
fn test_insert_and_lookup_slow_path() {
let mut t = Transliteration::default();
t.insert('🦀', 'c'); // U+1F980 Crab emoji -> 'c'
assert_eq!(t.lookup('🦀'), 'c');
assert_eq!(t.lookup('🦁'), '🦁'); // unchanged
}
#[test]
fn test_overwrite_mapping() {
let mut t = Transliteration::default();
t.insert('x', '1');
assert_eq!(t.lookup('x'), '1');
t.insert('x', '2');
assert_eq!(t.lookup('x'), '2');
}
#[test]
fn test_all_fast_path_mapped_to_space() {
let mut t = Transliteration::default();
for cp in 0..COMMON_UNICODE {
if let Some(ch) = char::from_u32(cp as u32) {
t.insert(ch, ' ');
}
}
assert_eq!(t.lookup('A'), ' ');
assert_eq!(t.lookup('\u{07FF}'), ' ');
}
// from_strings
#[test]
fn test_basic_transliteration() {
let t = Transliteration::from_strings("abcδ", "1234");
assert_eq!(t.lookup('a'), '1');
assert_eq!(t.lookup('b'), '2');
assert_eq!(t.lookup('c'), '3');
assert_eq!(t.lookup('δ'), '4');
assert_eq!(t.lookup('e'), 'e'); // not mapped, fallback
}
#[test]
fn test_unicode_slow_path() {
let source = "é漢🦀";
let target = "e文c";
let t = Transliteration::from_strings(source, target);
assert_eq!(t.lookup('é'), 'e');
assert_eq!(t.lookup('漢'), '文');
assert_eq!(t.lookup('🦀'), 'c');
assert_eq!(t.lookup('x'), 'x'); // fast fallback
assert_eq!(t.lookup('文'), '文'); // slow fallback
}
#[test]
fn test_overwrite_fast_path() {
let t = Transliteration::from_strings("aa", "12");
assert_eq!(t.lookup('a'), '2'); // last mapping wins
}
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+113
View File
@@ -0,0 +1,113 @@
// Parse delimited character sequences
//
// SPDX-License-Identifier: MIT
// Copyright (c) 2025 Diomidis Spinellis
//
// This file is part of the uutils sed package.
// It is licensed under the MIT License.
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use std::rc::Rc;
use uucore::error::{UResult, USimpleError};
use crate::sed::{
command::ProcessingContext, script_char_provider::ScriptCharProvider,
script_line_provider::ScriptLineProvider,
};
#[derive(Clone, Debug)]
/// The location in a script where a command is defined
pub struct ScriptLocation {
pub input_name: Rc<str>, // Shared input name
pub line_number: usize, // 1-based line number
pub column_number: usize, // 1-based column number
}
impl Default for ScriptLocation {
fn default() -> Self {
ScriptLocation { input_name: Rc::from("<unknown>"), line_number: 1, column_number: 1 }
}
}
impl ScriptLocation {
/// Construct with position information from the given providers.
pub fn at_position(lines: &ScriptLineProvider, line: &ScriptCharProvider) -> Self {
ScriptLocation {
line_number: lines.get_line_number(),
column_number: line.get_pos() + 1,
input_name: Rc::from(lines.get_input_name()),
}
}
}
/// Fail with msg as a compile error at the provider location.
/// The error's exit code is 1 (compilation phase).
pub fn compilation_error<T>(
lines: &ScriptLineProvider,
line: &ScriptCharProvider,
msg: impl ToString,
) -> UResult<T> {
Err(USimpleError::new(
1,
format!(
"{}:{}:{}: error: {}",
lines.get_input_name(),
lines.get_line_number(),
line.get_pos() + 1,
msg.to_string()
),
))
}
/// Fail with msg as a compilation error at the command's location.
/// The error's exit code is as specified.
fn location_error<T>(location: &ScriptLocation, msg: impl ToString, exit_code: i32) -> UResult<T> {
Err(USimpleError::new(
exit_code,
format!(
"{}:{}:{}: error: {}",
location.input_name,
location.line_number,
location.column_number,
msg.to_string()
),
))
}
/// Fail with msg as a compilation error at the command's location.
/// The error's exit code is 1 (compilation phase).
pub fn semantic_error<T>(location: &ScriptLocation, msg: impl ToString) -> UResult<T> {
location_error(location, msg, 1)
}
/// Fail with msg as a runtime error at the command's location.
/// The error's exit code is 2 (processing phase).
pub fn runtime_error<T>(location: &ScriptLocation, msg: impl ToString) -> UResult<T> {
location_error(location, msg, 2)
}
/// Fail with msg as a runtime error at the command's and input's location.
/// This is to be used in cases where the error depends on both, for example,
/// a fancy regular expression applied on invalid UTF-8 input.
/// (A fixed string match will not err in this case.)
/// The error's exit code is 2 (processing phase).
pub fn input_runtime_error<T>(
location: &ScriptLocation,
context: &ProcessingContext,
msg: impl ToString,
) -> UResult<T> {
Err(USimpleError::new(
2,
format!(
"{}:{}:{}: {}:{} error: {}",
location.input_name,
location.line_number,
location.column_number,
context.input_name,
context.line_number,
msg.to_string()
),
))
}
File diff suppressed because it is too large Load Diff
+703
View File
@@ -0,0 +1,703 @@
// A unified interface to byte and fancy Regex
//
// This allows using byte Regex when possible, resorting to the
// slower fancy_regex crate when needed.
//
// SPDX-License-Identifier: MIT
// Copyright (c) 2025 Diomidis Spinellis
//
// This file is part of the uutils sed package.
// It is licensed under the MIT License.
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use std::{error::Error, sync::LazyLock};
use fancy_regex::{
CaptureMatches as FancyCaptureMatches, Captures as FancyCaptures, Regex as FancyRegex,
};
use memchr::memmem;
use regex::{
Regex as RustRegex,
bytes::{CaptureMatches as ByteCaptureMatches, Captures as ByteCaptures, Regex as ByteRegex},
};
use uucore::error::{UResult, USimpleError};
use crate::sed::fast_io::IOChunk;
/// REs requiring the fancy_regex capabilities rather than the
/// faster regex::bytes engine
// False positives only result in a small performance pessimization,
// so this is just a maximally sensitive, good-enough approximation.
// For example, r"\\1" and r"[\1]" will match, whereas only a number
// after an odd number of backslashes and outside a character class
// should match.
static NEEDS_FANCY_RE: LazyLock<RustRegex> =
LazyLock::new(|| regex::Regex::new(r"\\[1-9]").unwrap());
/// All characters signifying that the match must be handled by an RE
/// rather than by plain string pattern matching.
// These do not include the ^$ metacharacters, which we can easily handle.
// Plain string fixed-string matching is currently faster than Regex
// matching, because Regex always constructs an automaton and needs
// to handle state transitions, whereas plain string matching can
// use tailored CPU string or vectored instructions.
static NEEDS_RE: LazyLock<RustRegex> = LazyLock::new(|| {
regex::Regex::new(
r"(?x) # Turn on verbose mode
( ^ # Non-escaped: i.e. at BOL
| ^[^\\] # or after a BOL non \
| [^\\] {2} # or after two non \ characters
| \\. # or after a consumed or escaped \
)
( # A potentially incompatible match
[.?|+(\[{*] # Any magic RE character
# Some are operators so illegal at
# BOL but they should error there,
# not use them as literals.
| \\[WwDdSsPp] # Unicode classes
| \\[AzBb] # Empty matches
| \\[0-9] # Back-references
)
",
)
.unwrap()
});
#[derive(Clone, Debug)]
/// Types of literal string anchored matches
enum AnchoredMatch {
Begin, // ^...
End, // ...$
Both, // ^...$
Free, // ...
}
#[derive(Clone, Debug)]
/// A fast Regex-like matcher for literal strings using memchr:memmem
pub struct LiteralMatcher {
needle: Vec<u8>, // Bytes without any anchors
match_type: AnchoredMatch, // Type of anchoring specified
}
impl LiteralMatcher {
/// Construct a new matcher based on a needle possible with anchors.
pub fn new(needle: &str) -> Self {
let needle_bytes = needle.as_bytes();
if needle_bytes[0] == b'^' && needle_bytes[needle_bytes.len() - 1] == b'$' {
LiteralMatcher {
match_type: AnchoredMatch::Both,
needle: needle_bytes[1..needle_bytes.len() - 1].to_vec(),
}
} else if needle_bytes[0] == b'^' {
LiteralMatcher {
match_type: AnchoredMatch::Begin,
needle: needle_bytes[1..needle_bytes.len()].to_vec(),
}
} else if needle_bytes[needle_bytes.len() - 1] == b'$' {
LiteralMatcher {
match_type: AnchoredMatch::End,
needle: needle_bytes[0..needle_bytes.len() - 1].to_vec(),
}
} else {
LiteralMatcher { match_type: AnchoredMatch::Free, needle: needle_bytes.to_vec() }
}
}
/// Returns the start index of a match, if any
fn anchored_find(&self, haystack: &[u8]) -> Option<usize> {
let nlen = self.needle.len();
let hlen = haystack.len();
match self.match_type {
AnchoredMatch::Both => {
if hlen == nlen && haystack == self.needle.as_slice() {
Some(0)
} else {
None
}
},
AnchoredMatch::Begin => {
if hlen >= nlen && &haystack[..nlen] == self.needle.as_slice() {
Some(0)
} else {
None
}
},
AnchoredMatch::End => {
if hlen >= nlen && &haystack[hlen - nlen..] == self.needle.as_slice() {
Some(hlen - nlen)
} else {
None
}
},
AnchoredMatch::Free => memmem::find(haystack, &self.needle),
}
}
/// Return true if the needle occurs in the haystack.
pub fn is_match(&self, haystack: &[u8]) -> bool {
self.anchored_find(haystack).is_some()
}
/// Return the position and contents of the matched needle.
pub fn find<'t>(&self, haystack: &'t [u8]) -> Option<(usize, usize, &'t str)> {
self.anchored_find(haystack).and_then(|start| {
let end = start + self.needle.len();
std::str::from_utf8(&haystack[start..end])
.ok()
.map(|s| (start, end, s))
})
}
/// Return all positions and contents of the matched needle.
pub fn iter<'t>(
&'t self,
haystack: &'t [u8],
) -> Box<dyn Iterator<Item = (usize, usize, &'t str)> + 't> {
let needle = &self.needle;
let nlen = needle.len();
match self.match_type {
AnchoredMatch::Both | AnchoredMatch::Begin | AnchoredMatch::End => {
// At most one match; yield it if present
Box::new(self.find(haystack).into_iter())
},
AnchoredMatch::Free => {
// Multiple potential matches
Box::new(memmem::find_iter(haystack, needle).filter_map(move |start| {
let end = start + nlen;
std::str::from_utf8(&haystack[start..end])
.ok()
.map(|s| (start, end, s))
}))
},
}
}
}
/// Return the passed pattern without any backslash escapes.
pub fn remove_escapes(pattern: &str) -> String {
let mut chars = pattern.chars().peekable();
let mut result = String::with_capacity(pattern.len());
while let Some(c) = chars.next() {
if c == '\\' {
// Look ahead and consume the next character if present
if let Some(&next) = chars.peek() {
result.push(next);
chars.next(); // consume the peeked char
}
} else {
result.push(c);
}
}
result
}
#[derive(Clone, Debug)]
/// A regular expression that can be implemented in diverse efficient ways
pub enum Regex {
Literal(LiteralMatcher), // Fastest: literal bytes
Byte(ByteRegex), // Slower: byte-based RE
Fancy(FancyRegex), // Slowest: RE supporting UTF-8 and back-references
}
/// Ensure that a regex matches GNU sed's default semantics for `.`
/// through the appropriate use of the s flag.
pub fn ensure_dotall(pattern: &str) -> String {
// Add (?s) if no flags present.
if !pattern.starts_with("(?") {
return format!("(?s){pattern}");
}
let Some(close) = pattern.find(')') else {
// Malformed inline flag group.
return pattern.to_owned();
};
// Add s flag to ?(...) unless 's' or its complement 'm' is there.
let flags = &pattern[2..close];
if flags.contains('m') || flags.contains('s') {
pattern.to_owned()
} else {
format!("(?{flags}s){}", &pattern[close + 1..])
}
}
impl Regex {
/// Construct the most efficient RE-like matching engine possible.
pub fn new(pattern: &str) -> Result<Self, Box<dyn Error>> {
if NEEDS_FANCY_RE.is_match(pattern) {
Ok(Self::Fancy(FancyRegex::new(&ensure_dotall(pattern))?))
} else if NEEDS_RE.is_match(pattern) {
Ok(Self::Byte(ByteRegex::new(&ensure_dotall(pattern))?))
} else {
Ok(Self::Literal(LiteralMatcher::new(&remove_escapes(pattern))))
}
}
/// Check if the regex matches the content of the IOChunk.
pub fn is_match(&self, chunk: &mut IOChunk) -> UResult<bool> {
match self {
Regex::Literal(m) => Ok(m.is_match(chunk.as_bytes())),
Regex::Byte(re) => Ok(re.is_match(chunk.as_bytes())),
Regex::Fancy(re) => {
let text = chunk.as_str()?;
re.is_match(text)
.map_err(|e| USimpleError::new(2, e.to_string()))
},
}
}
/// Return an iterator over capture groups.
pub fn captures_iter<'t>(&'t self, chunk: &'t IOChunk) -> UResult<CaptureMatches<'t>> {
match self {
Regex::Literal(m) => {
let haystack = chunk.as_bytes();
Ok(CaptureMatches::Literal(Box::new(
m.iter(haystack)
.map(|(start, end, text)| Ok(Captures::Literal(Match { start, end, text }))),
)))
},
Regex::Byte(re) => Ok(CaptureMatches::Byte(re.captures_iter(chunk.as_bytes()))),
Regex::Fancy(re) => {
let text = chunk.as_str()?;
Ok(CaptureMatches::Fancy(re.captures_iter(text)))
},
}
}
/// Return the number of capture groups, including group 0.
pub fn captures_len(&self) -> usize {
match self {
Regex::Literal(_) => 1, // Only group 0
Regex::Byte(re) => re.captures_len(),
Regex::Fancy(re) => re.captures_len(),
}
}
/// Return the elements of the first capture.
pub fn captures<'t>(&self, chunk: &'t IOChunk) -> UResult<Option<Captures<'t>>> {
match self {
Regex::Literal(m) => {
let haystack = chunk.as_bytes();
match m.find(haystack) {
Some((start, end, text)) => Ok(Some(Captures::Literal(Match { start, end, text }))),
None => Ok(None),
}
},
Regex::Byte(re) => {
let bytes = chunk.as_bytes();
Ok(re.captures(bytes).map(Captures::Byte))
},
Regex::Fancy(re) => {
let text = chunk.as_str()?;
match re.captures(text) {
Ok(Some(caps)) => Ok(Some(Captures::Fancy(caps))),
Ok(None) => Ok(None),
Err(e) => Err(USimpleError::new(2, e.to_string())),
}
},
}
}
/// Return a non-capturing result for a single match.
pub fn find<'t>(&self, chunk: &'t IOChunk) -> UResult<Option<Match<'t>>> {
match self {
Regex::Literal(m) => {
let haystack = chunk.as_bytes();
match m.find(haystack) {
Some((start, end, text)) => Ok(Some(Match { start, end, text })),
None => Ok(None),
}
},
Regex::Byte(re) => {
let haystack = chunk.as_bytes();
if let Some(m) = re.find(haystack) {
// Attempt UTF-8 decode for the match region only
let text = std::str::from_utf8(&haystack[m.start()..m.end()])
.map_err(|e| USimpleError::new(2, e.to_string()))?;
Ok(Some(Match { start: m.start(), end: m.end(), text }))
} else {
Ok(None)
}
},
Regex::Fancy(re) => {
let text = chunk.as_str()?;
match re.find(text) {
Ok(Some(m)) => {
Ok(Some(Match { start: m.start(), end: m.end(), text: m.as_str() }))
},
Ok(None) => Ok(None),
Err(e) => Err(USimpleError::new(2, e.to_string())),
}
},
}
}
}
/// Unified enum for holding either byte or fancy capture iterators.
pub enum CaptureMatches<'t> {
Literal(Box<dyn Iterator<Item = UResult<Captures<'t>>> + 't>),
Byte(ByteCaptureMatches<'t, 't>),
Fancy(FancyCaptureMatches<'t, 't>),
}
impl<'t> Iterator for CaptureMatches<'t> {
type Item = UResult<Captures<'t>>;
fn next(&mut self) -> Option<Self::Item> {
match self {
CaptureMatches::Literal(iter) => iter.next(),
CaptureMatches::Byte(iter) => iter.next().map(|caps| Ok(Captures::Byte(caps))),
CaptureMatches::Fancy(iter) => match iter.next() {
Some(Ok(caps)) => Some(Ok(Captures::Fancy(caps))),
Some(Err(e)) => {
Some(Err(USimpleError::new(2, format!("error retrieving RE captures: {e}"))))
},
None => None,
},
}
}
}
#[derive(Clone, Debug)]
/// Result type for RE capture get(n)
pub struct Match<'t> {
start: usize, // Match start
end: usize, // Match end
text: &'t str, // Actual match
}
/// Provide interface compatible with Regex::Match.
impl<'t> Match<'t> {
pub fn start(&self) -> usize {
self.start
}
pub fn end(&self) -> usize {
self.end
}
pub fn as_str(&self) -> &'t str {
self.text
}
}
/// Provide interface compatible with Regex::Captures.
pub enum Captures<'t> {
Literal(Match<'t>), // only group 0
Byte(ByteCaptures<'t>),
Fancy(FancyCaptures<'t>),
}
impl<'t> Captures<'t> {
/// Get capture group at index `i`
/// Returns Ok(None) if the group didn't match.
/// Returns Err if UTF-8 conversion fails (in Byte variant).
pub fn get(&self, i: usize) -> UResult<Option<Match<'t>>> {
match self {
Captures::Literal(m) => Ok(if i == 0 { Some(m.clone()) } else { None }),
Captures::Byte(caps) => match caps.get(i) {
Some(m) => Ok(Some(Match {
start: m.start(),
end: m.end(),
text: std::str::from_utf8(m.as_bytes())
.map_err(|e| USimpleError::new(1, e.to_string()))?,
})),
None => Ok(None),
},
Captures::Fancy(caps) => match caps.get(i) {
Some(m) => Ok(Some(Match { start: m.start(), end: m.end(), text: m.as_str() })),
None => Ok(None),
},
}
}
/// Return the number of capture groups (including group 0).
pub fn len(&self) -> usize {
match self {
Captures::Literal(_) => 1,
Captures::Byte(caps) => caps.len(),
Captures::Fancy(caps) => caps.len(),
}
}
/// Return true if there are no captures.
// Unused, but provided for completeness.
pub fn is_empty(&self) -> bool {
match self {
Captures::Literal(_) => false, // A literal match always has group 0
Captures::Byte(caps) => caps.len() == 0,
Captures::Fancy(caps) => caps.len() == 0,
}
}
}
#[cfg(test)]
mod tests {
use super::*;
// FANCY_RE
#[test]
fn test_needs_fancy_re_matches() {
let should_match = [
r"(\w+):\1", // back-reference \1
];
for pat in &should_match {
assert!(NEEDS_FANCY_RE.is_match(pat), "Expected NEEDS_FANCY_RE to match: {pat:?}");
}
}
#[test]
fn test_needs_fancy_re_does_not_match() {
let should_not_match = [
r"\ 1", // Non-adjacent
r"\0", // Only \[1-9]
// Simple ASCII
r"foo",
r"foo|bar",
r"^foo[0-9]+bar$",
];
for pat in &should_not_match {
assert!(!NEEDS_FANCY_RE.is_match(pat), "Expected NEEDS_FANCY_RE to NOT match: {pat:?}");
}
}
// NEEDS_RE
#[test]
fn test_needs_re_matches() {
let should_match = [
r".", // Single regex wildcard
r"a+b", // Regex +
r"foo|bar", // Regex alternation
r"abc?", // Regex optional
r"a*b", // Regex star
r"[abc]", // Character class
r"(abc)", // Group
r"{1,2}", // Repetition
r"\d", // Class shorthand
r"\S", // Class shorthand
r"\1", // Backreference
r"a\Pb", // Unicode property
];
for pat in &should_match {
assert!(NEEDS_RE.is_match(pat), "Expected NEEDS_RE to match: {pat:?}");
}
}
#[test]
fn test_needs_re_does_not_match() {
let should_not_match = [
r"abc",
r"a\.b", // Escaped dot
r"hello world",
r"^abc$", // Anchors alone
r"file\.", // Escaped dot
r"literal123",
r"\\", // Escaped backslash
];
for pat in &should_not_match {
assert!(!NEEDS_RE.is_match(pat), "Expected NEEDS_RE to NOT match: {pat:?}");
}
}
// Regex::new
#[test]
fn assert_byte_selection() {
let re = Regex::new(r"x*").unwrap();
assert!(matches!(re, Regex::Byte(_)));
}
#[test]
fn assert_fancy() {
let re = Regex::new(r"(.)\1").unwrap();
assert!(matches!(re, Regex::Fancy(_)));
}
#[test]
fn assert_literal() {
let re = Regex::new(r"x\.").unwrap();
assert!(matches!(re, Regex::Literal(_)));
}
#[test]
fn handles_invalid_regex_gracefully() {
let err = Regex::new("(").unwrap_err().to_string();
assert!(
err.contains("unclosed group") || err.contains("error parsing"),
"Unexpected error: {err:?}"
);
}
// remove_escapes
#[test]
fn test_remove_escapes() {
use super::remove_escapes;
assert_eq!(remove_escapes("abc"), "abc");
assert_eq!(remove_escapes(r"a\.c"), "a.c");
assert_eq!(remove_escapes(r"\\d"), r"\d");
assert_eq!(remove_escapes(r"\.\*\+\?"), ".*+?");
assert_eq!(remove_escapes(r"escaped\\backslash"), r"escaped\backslash");
assert_eq!(remove_escapes(r"trailing\\"), r"trailing\");
}
// LiteralMatcher
#[test]
fn test_literal_matcher_basic_match() {
let matcher = LiteralMatcher::new("needle");
assert!(matcher.is_match(b"this is a needle in a haystack"));
assert!(!matcher.is_match(b"no match here"));
}
#[test]
fn test_literal_matcher_anchor_start_match() {
let matcher = LiteralMatcher::new("^needle");
assert!(matcher.is_match(b"needle in a haystack"));
assert!(!matcher.is_match(b"no needle match here"));
assert!(!matcher.is_match(b"no"));
}
#[test]
fn test_literal_matcher_anchor_end_match() {
let matcher = LiteralMatcher::new("needle$");
assert!(matcher.is_match(b"In a haystack there's a needle"));
assert!(!matcher.is_match(b"no needle match here"));
assert!(!matcher.is_match(b"no"));
}
#[test]
fn test_literal_matcher_anchor_begin_end_match() {
let matcher = LiteralMatcher::new("^needle$");
assert!(matcher.is_match(b"needle"));
assert!(!matcher.is_match(b"no needle match"));
assert!(!matcher.is_match(b"needle no match"));
assert!(!matcher.is_match(b"no match needle"));
assert!(!matcher.is_match(b"nada"));
}
#[test]
fn test_literal_matcher_utf8_match() {
let matcher = LiteralMatcher::new("✓"); // U+2713 CHECK MARK (3 bytes)
let haystack = "contains ✓ unicode".as_bytes();
assert!(matcher.is_match(haystack));
let found = matcher.find(haystack).unwrap();
assert_eq!(found.2, "✓");
}
#[test]
fn test_literal_matcher_find_location() {
let matcher = LiteralMatcher::new("abc");
let haystack = b"___abc___";
let result = matcher.find(haystack);
assert!(result.is_some());
let (start, end, text) = result.unwrap();
assert_eq!((start, end), (3, 6));
assert_eq!(text, "abc");
}
#[test]
fn test_literal_matcher_find_location_end() {
let matcher = LiteralMatcher::new("abc$");
let haystack = b"012abc";
let result = matcher.find(haystack);
assert!(result.is_some());
let (start, end, text) = result.unwrap();
assert_eq!((start, end), (3, 6));
assert_eq!(text, "abc");
}
#[test]
fn test_literal_matcher_iter_multiple() {
let matcher = LiteralMatcher::new("test");
let haystack = b"this test is a test of test matching";
let matches: Vec<_> = matcher.iter(haystack).collect();
assert_eq!(matches.len(), 3);
let strings: Vec<_> = matches.iter().map(|(_, _, s)| *s).collect();
assert_eq!(strings, ["test", "test", "test"]);
}
#[test]
fn test_literal_matcher_iter_begin() {
let matcher = LiteralMatcher::new("^test");
let haystack = b"test is a test of test matching";
let matches: Vec<_> = matcher.iter(haystack).collect();
assert_eq!(matches.len(), 1);
let strings: Vec<_> = matches.iter().map(|(_, _, s)| *s).collect();
assert_eq!(strings, ["test"]);
}
#[test]
fn test_literal_matcher_iter_end() {
let matcher = LiteralMatcher::new("test$");
let haystack = b"this test is a test of test";
let matches: Vec<_> = matcher.iter(haystack).collect();
assert_eq!(matches.len(), 1);
let strings: Vec<_> = matches.iter().map(|(_, _, s)| *s).collect();
assert_eq!(strings, ["test"]);
}
#[test]
fn test_literal_matcher_no_match() {
let matcher = LiteralMatcher::new("missing");
let haystack = b"nothing to see here";
assert!(!matcher.is_match(haystack));
assert!(matcher.find(haystack).is_none());
assert_eq!(matcher.iter(haystack).count(), 0);
}
#[test]
fn test_literal_matcher_anchored_no_match() {
let matcher = LiteralMatcher::new("^see$");
let haystack = b"nothing to see here";
assert!(!matcher.is_match(haystack));
assert!(matcher.find(haystack).is_none());
assert_eq!(matcher.iter(haystack).count(), 0);
}
#[test]
fn prepends_s_when_no_flag_group() {
assert_eq!(ensure_dotall("abc"), "(?s)abc");
}
#[test]
fn adds_s_when_no_m_or_s() {
assert_eq!(ensure_dotall("(?i)abc"), "(?is)abc");
assert_eq!(ensure_dotall("(?)abc"), "(?s)abc");
}
#[test]
fn leaves_m_unchanged() {
assert_eq!(ensure_dotall("(?m)abc"), "(?m)abc");
assert_eq!(ensure_dotall("(?im)abc"), "(?im)abc");
assert_eq!(ensure_dotall("(?mi)abc"), "(?mi)abc");
}
#[test]
fn leaves_existing_s_unchanged() {
assert_eq!(ensure_dotall("(?s)abc"), "(?s)abc");
assert_eq!(ensure_dotall("(?is)abc"), "(?is)abc");
}
#[test]
fn leaves_malformed_flag_group_unchanged() {
assert_eq!(ensure_dotall("(?iabc"), "(?iabc");
}
}
+320
View File
@@ -0,0 +1,320 @@
// Support for in-place editing
//
// SPDX-License-Identifier: MIT
// Copyright (c) 2025 Diomidis Spinellis
//
// This file is part of the uutils sed package.
// It is licensed under the MIT License.
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
#[cfg(unix)]
use std::os::unix::fs::MetadataExt;
#[cfg(unix)]
use std::os::unix::fs::PermissionsExt;
use std::{
fs,
path::{Path, PathBuf},
};
use tempfile::NamedTempFile;
use uucore::{
display::Quotable,
error::{FromIo, UIoError, UResult, USimpleError},
};
use crate::sed::{command::ProcessingContext, fast_io::OutputBuffer};
/// Context for in-place editing
pub struct InPlace {
pub output: OutputBuffer,
pub in_place: bool,
pub in_place_suffix: Option<String>,
pub follow_symlinks: bool,
pub temp_file: Option<NamedTempFile>,
pub original_path: Option<PathBuf>,
}
impl InPlace {
/// Create an in-place editing engine based on ProcessingContext.
/// Depending on its settings it may or may not perform in-place
/// editing, backup the original file, or follow symlinks.
pub fn new(context: ProcessingContext) -> Self {
Self {
output: OutputBuffer::new(Box::new(pi_uutils_ctx::stdout())),
in_place: context.in_place,
in_place_suffix: context.in_place_suffix,
follow_symlinks: context.follow_symlinks,
temp_file: None,
original_path: None,
}
}
/// Return an OutputBuffer for outputting the edits to the specified file.
/// The file may be a symbolic link, which will be processed according
/// to the context specification.
pub fn begin(&mut self, file_name: &Path) -> UResult<&mut OutputBuffer> {
// Patched for pi-uutils-ctx embedding: resolve the operand against
// the shell working directory so the in-place temp file lands in the
// real target's parent directory, never the host process cwd.
let file_name = pi_uutils_ctx::resolve(file_name);
let resolved = if self.follow_symlinks {
fs::canonicalize(&file_name)
.map_err_context(|| format!("resolving symlink {}", file_name.quote()))?
} else {
file_name
};
self.begin_resolved(&resolved)
}
/// Return an OutputBuffer for outputting the edits to the specified file.
/// The passed file name should have resolved symbolic links according
/// to the context settings.
fn begin_resolved(&mut self, file_name: &Path) -> UResult<&mut OutputBuffer> {
if !self.in_place {
self.output = OutputBuffer::new(Box::new(pi_uutils_ctx::stdout()));
return Ok(&mut self.output);
}
let metadata = fs::metadata(file_name).map_err_context(|| {
format!("error Reading metadata of {} for in-place edit", file_name.quote())
})?;
if !metadata.is_file() {
return Err(USimpleError::new(
2,
format!("cannot in-place edit non-regular file {}", file_name.quote()),
));
}
let dir = file_name.parent().unwrap_or_else(|| Path::new("."));
let temp_file = NamedTempFile::new_in(dir)
.map_err_context(|| format!("error creating temporary file in {}", dir.quote()))?;
// TODO: On Unix use fchown(metadata.{uid,dig}) and fchmod(mode)
// on let fd = temp_file.as_file().as_raw_fd() when uucore::libc
// support them.
#[cfg(unix)]
{
let mode = metadata.mode() & 0o7777;
let perms = fs::Permissions::from_mode(mode);
fs::set_permissions(temp_file.path(), perms)?;
}
let output =
OutputBuffer::new(Box::new(temp_file.reopen().expect("reopening NamedTempFile")));
self.output = output;
self.temp_file = Some(temp_file);
self.original_path = Some(file_name.to_path_buf());
Ok(&mut self.output)
}
/// Finish (potentially in-place) editing.
pub fn end(&mut self) -> UResult<()> {
self.output.flush()?;
if !self.in_place {
return Ok(());
}
let orig = self.original_path.take().expect("original_path unset");
let temp = self.temp_file.take().expect("temp_file unset");
// Backup original if suffix is provided
if let Some(ref suffix) = self.in_place_suffix {
let mut backup_path = orig.clone();
let file_name = backup_path
.file_name()
.expect("Missing file name for backup")
.to_os_string();
let mut backup_name = file_name;
backup_name.push(suffix);
backup_path.set_file_name(backup_name);
#[cfg(windows)]
// Try to remove to ensure the rename won't fail on Windows.
let _ = fs::remove_file(&backup_path);
fs::rename(&orig, &backup_path).map_err_context(|| {
format!("error backing up {} to {}", orig.quote(), backup_path.quote())
})?;
} else {
#[cfg(windows)]
// On Windows delete the original file for temp.persist to work
if orig.exists() {
fs::remove_file(&orig).map_err_context(|| {
format!("error removing original input file {}", orig.quote())
})?;
}
}
// Atomically replace the original
match temp.persist(&orig) {
Ok(_) => {},
Err(e) => {
return Err(UIoError::new(
e.error.kind(),
format!(
"error persisting temporary file {} to {}",
e.file.path().quote(),
orig.quote()
),
));
},
}
Ok(())
}
}
#[cfg(test)]
mod tests {
use std::path::PathBuf;
use tempfile::TempDir;
use super::*;
// Minimal stand-in for the assert_fs fixture API used by these
// upstream tests, so tempfile (already a dependency) suffices.
struct ChildPath(PathBuf);
impl ChildPath {
fn path(&self) -> &Path {
&self.0
}
}
trait PathChild {
fn child(&self, name: &str) -> ChildPath;
}
impl PathChild for TempDir {
fn child(&self, name: &str) -> ChildPath {
ChildPath(self.path().join(name))
}
}
use std::{
fs,
io::{Read, Write},
path::Path,
};
fn minimal_context() -> ProcessingContext {
ProcessingContext {
in_place: false,
in_place_suffix: None,
follow_symlinks: false,
// fill in default values for the rest as needed
..Default::default()
}
}
fn write_original(file: &Path, content: &str) {
fs::write(file, content).unwrap();
}
fn read_file(file: &Path) -> String {
let mut contents = String::new();
fs::File::open(file)
.unwrap()
.read_to_string(&mut contents)
.unwrap();
contents
}
#[test]
fn test_in_place_editing() {
let temp = TempDir::new().unwrap();
let file = temp.child("file.txt");
write_original(file.path(), "original\n");
let mut ctx = minimal_context();
ctx.in_place = true;
let mut inplace = InPlace::new(ctx);
let buf = inplace.begin(file.path()).unwrap();
writeln!(buf, "updated").unwrap();
inplace.end().unwrap();
assert_eq!(read_file(file.path()), "updated\n");
}
#[test]
fn test_in_place_backup() {
let temp = TempDir::new().unwrap();
let file = temp.child("file.txt");
let backup = temp.child("file.txt.bak");
write_original(file.path(), "original\n");
let mut ctx = minimal_context();
ctx.in_place = true;
ctx.in_place_suffix = Some(".bak".to_string());
let mut inplace = InPlace::new(ctx);
let buf = inplace.begin(file.path()).unwrap();
writeln!(buf, "new content").unwrap();
inplace.end().unwrap();
assert_eq!(read_file(file.path()), "new content\n");
assert_eq!(read_file(backup.path()), "original\n");
}
#[cfg(unix)]
#[test]
fn test_symlink_follow_true() {
let temp = TempDir::new().unwrap();
let real = temp.child("target.txt");
let link = temp.child("link.txt");
write_original(real.path(), "real\n");
std::os::unix::fs::symlink(real.path(), link.path()).unwrap();
let mut ctx = minimal_context();
ctx.in_place = true;
ctx.follow_symlinks = true;
let mut inplace = InPlace::new(ctx);
let buf = inplace.begin(link.path()).unwrap();
writeln!(buf, "changed").unwrap();
inplace.end().unwrap();
assert_eq!(read_file(real.path()), "changed\n");
assert!(link.path().exists()); // Symlink still exists
}
#[cfg(unix)]
#[test]
fn test_symlink_follow_false() {
let temp = TempDir::new().unwrap();
let real = temp.child("target.txt");
let link = temp.child("link.txt");
write_original(real.path(), "real\n");
std::os::unix::fs::symlink(real.path(), link.path()).unwrap();
let mut ctx = minimal_context();
ctx.in_place = true;
ctx.follow_symlinks = false;
let mut inplace = InPlace::new(ctx);
let buf = inplace.begin(link.path()).unwrap();
writeln!(buf, "linked").unwrap();
inplace.end().unwrap();
// real file should remain untouched
assert_eq!(read_file(real.path()), "real\n");
// link (symlink path) now contains the new content
let contents = read_file(link.path());
assert_eq!(contents, "linked\n");
}
#[test]
fn test_no_in_place_outputs_to_stdout() {
let mut ctx = minimal_context();
ctx.in_place = false;
let mut inplace = InPlace::new(ctx);
let _buf = inplace.begin(Path::new("fake.txt")).unwrap();
assert!(inplace.end().is_ok());
}
}
+431
View File
@@ -0,0 +1,431 @@
// Program entry point and CLI processing
//
// SPDX-License-Identifier: MIT
// Copyright (c) 2025 Diomidis Spinellis
//
// This file is part of the uutils sed package.
// It is licensed under the MIT License.
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
pub mod command;
pub mod compiler;
pub mod delimited_parser;
pub mod error_handling;
pub mod fast_io;
pub mod fast_regex;
pub mod in_place;
pub mod named_writer;
pub mod processor;
pub mod script_char_provider;
pub mod script_line_provider;
use std::{collections::HashMap, path::PathBuf};
use clap::{Arg, ArgMatches, Command, arg, crate_version};
use pi_uutils_ctx::format_usage;
use uucore::error::{UResult, UUsageError};
use crate::sed::{
command::{ProcessingContext, StringSpace},
compiler::compile,
processor::process_all_files,
script_line_provider::ScriptValue,
};
const ABOUT: &str = "Stream editor for filtering and transforming text";
const USAGE: &str = "sed [OPTION]... [script] [file]...";
// Patched for pi-uutils-ctx embedding: upstream's `#[uucore::main] uumain`
// (which printed help to the process stdout and called `std::process::exit`)
// is replaced by this plain function; argument parsing, the no-args help
// path, and exit-code mapping live in the crate-level `run` wrapper.
pub fn sed_main(matches: &ArgMatches) -> UResult<()> {
let (scripts, files) = get_scripts_files(matches)?;
let mut context = build_context(matches);
let executable = compile(scripts, &mut context)?;
process_all_files(executable, files, &mut context)?;
Ok(())
}
/// Rewrite GNU-style attached `-i` backup suffixes (`-i.bak`, `-ibak`) into
/// the `-i=.bak` form clap needs with `require_equals`. GNU sed's `-i`
/// takes its optional suffix only when directly attached, so a separate
/// following token must stay a script/file operand; scanning stops at `--`.
pub fn normalize_args(argv: Vec<std::ffi::OsString>) -> Vec<std::ffi::OsString> {
let mut out = Vec::with_capacity(argv.len());
let mut iter = argv.into_iter();
// argv[0] is the command name; never rewritten.
if let Some(first) = iter.next() {
out.push(first);
}
let mut past_separator = false;
for arg in iter {
if !past_separator {
if arg == "--" {
past_separator = true;
} else if let Some(s) = arg.to_str()
&& let Some(suffix) = s.strip_prefix("-i")
&& !suffix.is_empty()
&& !suffix.starts_with('=')
{
out.push(format!("-i={suffix}").into());
continue;
}
}
out.push(arg);
}
out
}
#[allow(clippy::cognitive_complexity)]
pub fn uu_app() -> Command {
let util_name = "sed";
Command::new(util_name)
.version(crate_version!())
.about(ABOUT)
.override_usage(format_usage(USAGE))
.args_override_self(true)
.infer_long_args(true)
.args([
arg!([script] "Script to execute if not otherwise provided."),
Arg::new("file")
.help("Input files")
.value_parser(clap::value_parser!(PathBuf))
.num_args(0..),
Arg::new("all-output-files")
.long("all-output-files")
.short('a')
.help("Create or truncate all output files before processing.")
.action(clap::ArgAction::SetTrue),
arg!(--debug "Annotate program execution."),
Arg::new("regexp-extended")
.short('E')
.long("regexp-extended")
.short_alias('r')
.help("Use extended regular expressions.")
.action(clap::ArgAction::SetTrue),
arg!(-e --expression <SCRIPT> "Add script to executed commands.")
.action(clap::ArgAction::Append),
// Access with .get_many::<PathBuf>("file")
Arg::new("script-file")
.short('f')
.long("script-file")
.help("Specify script file.")
.value_parser(clap::value_parser!(PathBuf))
.action(clap::ArgAction::Append),
Arg::new("follow-symlinks")
.long("follow-symlinks")
.help("Follow symlinks when processing in place.")
.action(clap::ArgAction::SetTrue),
// Access with .get_one::<String>("in-place")
Arg::new("in-place")
.short('i')
.long("in-place")
.help("Edit files in place, making a backup if SUFFIX is supplied.")
.num_args(0..=1)
// Patched: GNU sed only accepts the backup suffix attached
// (`-i.bak`, `--in-place=.bak`); without this clap would eat
// the following script/file operand as the suffix.
.require_equals(true)
.default_missing_value(""),
// Access with .get_one::<u32>("line-length")
arg!(-l --length <NUM> "Specify the 'l' command line-wrap length.")
.value_parser(clap::value_parser!(u32)),
arg!(-n --quiet "Suppress automatic printing of pattern space.").aliases(["silent"]),
arg!(--posix "Disable non-POSIX extensions."),
arg!(-s --separate "Consider files as separate rather than as a long stream."),
arg!(--sandbox "Operate in a sandbox by disabling e/r/w commands."),
arg!(-u --unbuffered "Load minimal input data and flush output buffers regularly."),
Arg::new("null-data")
.short('z')
.long("null-data")
.help("Separate lines by NUL characters.")
.action(clap::ArgAction::SetTrue),
])
}
// Iterate through script and file arguments specified in matches and
// return vectors of all scripts and input files in the specified order.
// If no script is specified fail with "missing script" error.
fn get_scripts_files(matches: &ArgMatches) -> UResult<(Vec<ScriptValue>, Vec<PathBuf>)> {
let mut indexed_scripts: Vec<(usize, ScriptValue)> = Vec::new();
let mut files: Vec<PathBuf> = Vec::new();
let script_through_options =
// The specification of a script: through a string or a file.
matches.contains_id("expression") || matches.contains_id("script-file");
if script_through_options {
// Second and third POSIX usage cases; clap script arg is actually an input file
// sed [-En] -e script [-e script]... [-f script_file]... [file...]
// sed [-En] [-e script]... -f script_file [-f script_file]... [file...]
if let Some(val) = matches.get_one::<String>("script") {
files.push(PathBuf::from(val.to_owned()));
}
} else {
// First POSIX spec usage case; script is the first arg.
// sed [-En] script [file...]
if let Some(val) = matches.get_one::<String>("script") {
indexed_scripts.push((0, ScriptValue::StringVal(val.to_owned())));
} else {
return Err(UUsageError::new(1, "missing script"));
}
}
// Capture -e occurrences (STRING)
if let Some(indices) = matches.indices_of("expression") {
for (idx, val) in indices.zip(matches.get_many::<String>("expression").unwrap_or_default()) {
indexed_scripts.push((idx, ScriptValue::StringVal(val.to_owned())));
}
}
// Capture -f occurrences (FILE)
if let Some(indices) = matches.indices_of("script-file") {
for (idx, val) in indices.zip(
matches
.get_many::<PathBuf>("script-file")
.unwrap_or_default(),
) {
indexed_scripts.push((idx, ScriptValue::PathVal(val.to_owned())));
}
}
// Sort by index to preserve argument order.
indexed_scripts.sort_by_key(|k| k.0);
// Keep only the values.
let scripts = indexed_scripts
.into_iter()
.map(|(_, value)| value)
.collect();
let rest_files: Vec<PathBuf> = matches
.get_many::<PathBuf>("file")
.unwrap_or_default()
.cloned()
.collect();
if !rest_files.is_empty() {
files.extend(rest_files);
}
// Read from stdin if no file has been specified.
if files.is_empty() {
files.push(PathBuf::from("-"));
}
Ok((scripts, files))
}
// Parse CLI flag arguments and return a ProcessingContext struct based on them
fn build_context(matches: &ArgMatches) -> ProcessingContext {
ProcessingContext {
all_output_files: matches.get_flag("all-output-files"),
debug: matches.get_flag("debug"),
regex_extended: matches.get_flag("regexp-extended"),
follow_symlinks: matches.get_flag("follow-symlinks"),
in_place: matches.contains_id("in-place"),
in_place_suffix: matches
.get_one::<String>("in-place")
.and_then(|s| if s.is_empty() { None } else { Some(s.clone()) }),
length: matches.get_one::<u32>("length").map_or(70, |v| *v as usize),
quiet: matches.get_flag("quiet"),
posix: matches.get_flag("posix"),
separate: matches.get_flag("separate"),
sandbox: matches.get_flag("sandbox"),
unbuffered: matches.get_flag("unbuffered"),
null_data: matches.get_flag("null-data"),
// Other context
input_name: "<stdin>".to_string(),
line_number: 0,
last_address: false,
last_line: false,
last_file: false,
stop_processing: false,
saved_regex: None,
input_action: None,
hold: StringSpace { content: String::new(), has_newline: true },
parsed_block_nesting: 0,
label_to_command_map: HashMap::new(),
range_commands: Vec::new(),
substitution_made: false,
append_elements: Vec::new(),
}
}
#[cfg(test)]
mod tests {
use super::*; // Allows access to private functions/items in this module
// get_scripts_files
// Helper function for supplying arguments
fn get_test_matches(args: &[&str]) -> ArgMatches {
uu_app()
.try_get_matches_from(["myapp"].iter().chain(args.iter()))
.expect("test args parse")
}
#[test]
fn test_script_as_first_argument() {
let matches = get_test_matches(&["1d", "file1.txt"]);
let (scripts, files) = get_scripts_files(&matches).expect("Should succeed");
assert_eq!(scripts, vec![ScriptValue::StringVal("1d".to_string())]);
assert_eq!(files, vec![PathBuf::from("file1.txt")]);
}
#[test]
fn test_expression_argument() {
let matches = get_test_matches(&["-e", "s/foo/bar/", "file1.txt"]);
let (scripts, files) = get_scripts_files(&matches).expect("Should succeed");
assert_eq!(scripts, vec![ScriptValue::StringVal("s/foo/bar/".to_string())]);
assert_eq!(files, vec![PathBuf::from("file1.txt")]);
}
#[test]
fn test_script_file_argument() {
let matches = get_test_matches(&["-f", "script.sed", "file1.txt"]);
let (scripts, files) = get_scripts_files(&matches).expect("Should succeed");
assert_eq!(scripts, vec![ScriptValue::PathVal(PathBuf::from("script.sed"))]);
assert_eq!(files, vec![PathBuf::from("file1.txt")]);
}
#[test]
fn test_multiple_files() {
let matches = get_test_matches(&["-e", "s/foo/bar/", "file1.txt", "file2.txt"]);
let (scripts, files) = get_scripts_files(&matches).expect("Should succeed");
assert_eq!(scripts, vec![ScriptValue::StringVal("s/foo/bar/".to_string())]);
assert_eq!(files, vec![PathBuf::from("file1.txt"), PathBuf::from("file2.txt")]);
}
#[test]
fn test_multiple_files_script() {
let matches = get_test_matches(&["s/foo/bar/", "file1.txt", "file2.txt"]);
let (scripts, files) = get_scripts_files(&matches).expect("Should succeed");
assert_eq!(scripts, vec![ScriptValue::StringVal("s/foo/bar/".to_string())]);
assert_eq!(files, vec![PathBuf::from("file1.txt"), PathBuf::from("file2.txt")]);
}
#[test]
fn test_stdin_when_no_files() {
let matches = get_test_matches(&["-e", "s/foo/bar/"]);
let (scripts, files) = get_scripts_files(&matches).expect("Should succeed");
assert_eq!(scripts, vec![ScriptValue::StringVal("s/foo/bar/".to_string())]);
assert_eq!(files, vec![PathBuf::from("-")]); // Stdin should be used
}
#[test]
fn test_stdin_when_no_files_script() {
let matches = get_test_matches(&["s/foo/bar/"]);
let (scripts, files) = get_scripts_files(&matches).expect("Should succeed");
assert_eq!(scripts, vec![ScriptValue::StringVal("s/foo/bar/".to_string())]);
assert_eq!(files, vec![PathBuf::from("-")]); // Stdin should be used
}
// build_context
fn test_matches(args: &[&str]) -> ArgMatches {
let argv = normalize_args(
["sed"]
.into_iter()
.chain(args.iter().copied())
.map(std::ffi::OsString::from)
.collect(),
);
uu_app()
.try_get_matches_from(argv)
.expect("test args parse")
}
#[test]
fn test_defaults() {
let matches = test_matches(&[]);
let ctx = build_context(&matches);
assert!(!ctx.all_output_files);
assert!(!ctx.debug);
assert!(!ctx.regex_extended);
assert!(!ctx.follow_symlinks);
assert!(!ctx.in_place);
assert_eq!(ctx.in_place_suffix, None);
assert_eq!(ctx.length, 70);
assert!(!ctx.quiet);
assert!(!ctx.posix);
assert!(!ctx.separate);
assert!(!ctx.sandbox);
assert!(!ctx.unbuffered);
assert!(!ctx.null_data);
}
#[test]
fn test_all_flags() {
let matches = test_matches(&[
"--all-output-files",
"--debug",
"-E",
"--follow-symlinks",
"-i",
"-l",
"80",
"-n",
"--posix",
"-s",
"--sandbox",
"-u",
"-z",
]);
let ctx = build_context(&matches);
assert!(ctx.all_output_files);
assert!(ctx.debug);
assert!(ctx.regex_extended);
assert!(ctx.follow_symlinks);
assert!(ctx.in_place);
assert!(ctx.in_place_suffix.is_none());
assert_eq!(ctx.length, 80);
assert!(ctx.quiet);
assert!(ctx.posix);
assert!(ctx.separate);
assert!(ctx.sandbox);
assert!(ctx.unbuffered);
assert!(ctx.null_data);
}
#[test]
fn test_multiple_same_arguments() {
let matches = test_matches(&["-E", "-r"]);
let ctx = build_context(&matches);
assert!(ctx.regex_extended);
}
#[test]
fn test_in_place_with_suffix() {
let matches = test_matches(&["-i.bak"]);
let ctx = build_context(&matches);
assert!(ctx.in_place);
assert_eq!(ctx.in_place_suffix, Some(".bak".to_string()));
}
#[test]
fn test_length_default_and_custom() {
let matches_default = test_matches(&[]);
let matches_custom = test_matches(&["-l", "120"]);
let ctx_default = build_context(&matches_default);
let ctx_custom = build_context(&matches_custom);
assert_eq!(ctx_default.length, 70);
assert_eq!(ctx_custom.length, 120);
}
}
+95
View File
@@ -0,0 +1,95 @@
// An abstraction for output files created on entry and flushed on exit
//
// SPDX-License-Identifier: MIT
// Copyright (c) 2025 Diomidis Spinellis
//
// This file is part of the uutils sed package.
// It is licensed under the MIT License.
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use std::{
cell::RefCell,
fs::{File, OpenOptions},
io::{BufWriter, Write},
path::PathBuf,
rc::Rc,
};
use uucore::{display::Quotable, error::UResult};
use crate::sed::error_handling::{ScriptLocation, runtime_error};
thread_local! {
/// Global list of all writers that should be flushed at shutdown
static FLUSH_LIST: RefCell<Vec<Rc<RefCell<NamedWriter>>>> = const { RefCell::new(Vec::new()) };
}
#[derive(Debug)]
/// Writer that tracks its file name for better error messages
pub struct NamedWriter {
pub path: PathBuf,
writer: BufWriter<File>,
location: ScriptLocation,
}
impl NamedWriter {
/// Create a new writer, truncate the file, and register it for flushing.
pub fn new(path: PathBuf, location: ScriptLocation) -> UResult<Rc<RefCell<Self>>> {
let file = OpenOptions::new()
.create(true)
.write(true)
.truncate(true)
.open(&path)
.map_err(|e| {
runtime_error::<()>(&location, format!("creating file {}: {}", path.quote(), e))
.unwrap_err()
})?;
let writer =
Rc::new(RefCell::new(NamedWriter { path, writer: BufWriter::new(file), location }));
FLUSH_LIST.with(|list| list.borrow_mut().push(Rc::clone(&writer)));
Ok(writer)
}
/// Write a line to the file with a newline, returning descriptive errors.
pub fn write_line(&mut self, line: &str) -> UResult<()> {
writeln!(self.writer, "{line}").map_err(|e| {
runtime_error::<()>(&self.location, format!("writing to file {}: {e}", self.path.quote()))
.unwrap_err()
})
}
/// Flush the writer, returning a descriptive error.
pub fn flush(&mut self) -> UResult<()> {
self.writer.flush().map_err(|e| {
runtime_error::<()>(
&self.location,
format!("writing to file {}: {}", self.path.quote(), e),
)
.unwrap_err()
})
}
}
/// Flush buffered content to the files and drop the writers, returning
/// descriptive errors.
// Patched for pi-uutils-ctx embedding: the registry is drained (not just
// iterated) so open files do not outlive the invocation on a reused thread.
pub fn flush_all() -> UResult<()> {
FLUSH_LIST.with(|cell| {
for handle in cell.borrow_mut().drain(..) {
handle.borrow_mut().flush()?;
}
Ok(())
})
}
/// Clear the thread-local writer registry. Called at builtin entry so
/// writers registered by a previous invocation on the same thread (one that
/// failed before reaching `flush_all`) cannot leak into this run.
pub fn reset() {
FLUSH_LIST.with(|cell| cell.borrow_mut().clear());
}
+818
View File
@@ -0,0 +1,818 @@
// Process the files with the compiled scripts
//
// SPDX-License-Identifier: MIT
// Copyright (c) 2025 Diomidis Spinellis
//
// This file is part of the uutils sed package.
// It is licensed under the MIT License.
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use std::{borrow::Cow, cell::RefCell, path::PathBuf, rc::Rc};
use uucore::{
display::Quotable,
error::{FromIo, UResult},
};
use crate::sed::{
command::{
Address, AppendElement, Command, CommandData, InputAction, ProcessingContext, Transliteration,
},
error_handling::{ScriptLocation, input_runtime_error},
fast_io::{IOChunk, LineReader, OutputBuffer},
fast_regex::Regex,
in_place::InPlace,
named_writer,
};
/// Return the specified command variant or panic.
// Example: let path = extract_variant!(command, Path);
macro_rules! extract_variant {
($cmd:expr, $variant:ident) => {
match &$cmd.data {
CommandData::$variant(inner) => inner,
_ => panic!(concat!("Expected ", stringify!($variant), " command data")),
}
};
}
/// Return true if the passed address matches the current I/O context.
fn match_address(
addr: &Address,
reader: &mut LineReader,
pattern: &mut IOChunk,
context: &mut ProcessingContext,
location: &ScriptLocation,
) -> UResult<bool> {
match addr {
Address::Re(re) => {
let regex = re_or_saved_re(re.as_ref(), context, location)?;
match regex.is_match(pattern) {
Ok(result) => Ok(result),
Err(e) => input_runtime_error(location, context, e.to_string()),
}
},
Address::Line(lineno) => Ok(context.line_number == *lineno),
// Recognize "$" as the last line of last file. This is consistent
// with the original 7th Research Edition implementation:
// https://github.com/dspinellis/unix-history-repo/blob/Research-V7/usr/src/cmd/sed/sed1.c#L665
// The FreeBSD version checked for subsequent empty files, but this
// can lead to destructive reads (e.g. from named pipes),
// and is probably an overkill.
Address::Last => Ok(reader.last_line()? && (context.last_file || context.separate)),
_ => panic!("invalid address type in match_address"),
}
}
#[allow(dead_code)]
/// Return true if the command applies to the given pattern.
fn applies(
command: &mut Command,
reader: &mut LineReader,
pattern: &mut IOChunk,
context: &mut ProcessingContext,
) -> UResult<bool> {
let linenum = context.line_number;
let result = if command.addr1.is_none() && command.addr2.is_none() {
// No address
Ok(true)
} else if let Some(addr2) = &command.addr2 {
// Two addresses
if let Some(start) = command.start_line {
// Range is already latched active.
match addr2 {
Address::RelLine(n) => {
if linenum - start > *n {
command.start_line = None;
Ok(false)
} else {
Ok(true)
}
},
Address::Line(n) => {
// Special case: already ended
if linenum > *n {
command.start_line = None;
Ok(false)
} else {
Ok(true)
}
},
Address::StepMatch(step) => Ok((linenum - start).is_multiple_of(*step)),
Address::StepEnd(step) => {
// Inclusive end on multiple of step
if linenum.is_multiple_of(*step) {
command.start_line = None;
}
Ok(true)
},
_ => {
if match_address(addr2, reader, pattern, context, &command.location)? {
command.start_line = None;
context.last_address = true;
}
Ok(true)
},
}
} else if let Some(addr1) = &command.addr1 {
// See if latch must start.
if match_address(addr1, reader, pattern, context, &command.location)? {
match addr2 {
Address::Line(n) if linenum >= *n => {
context.last_address = true;
},
Address::RelLine(n) if *n == 0 => {
context.last_address = true;
},
_ => {
command.start_line = Some(linenum);
},
}
Ok(true)
} else {
Ok(false)
}
} else {
Ok(false)
}
} else if let Some(addr1) = &command.addr1 {
// Single address
Ok(match_address(addr1, reader, pattern, context, &command.location)?)
} else {
// All allowed cases have been covered by the above logic.
panic!("impossible address combination");
};
if command.non_select {
result.map(|v| !v)
} else {
result
}
}
/// Write the specified chunk to the output for a given processing context.
fn write_chunk(
output: &mut OutputBuffer,
context: &ProcessingContext,
chunk: &IOChunk,
) -> std::io::Result<()> {
output.write_chunk(chunk)?;
if context.unbuffered {
output.flush()?;
}
Ok(())
}
/// Return a reference to the current or the saved RE if the RE is None.
/// Update the saved RE to RE.
fn re_or_saved_re<'a>(
regex: Option<&Regex>,
context: &'a mut ProcessingContext,
location: &ScriptLocation,
) -> UResult<&'a Regex> {
if let Some(re) = regex {
// First time we see this regex: clone it *once* into the context.
context.saved_regex = Some(re.clone());
// Return a reference into context.saved_regex.
Ok(context.saved_regex.as_ref().unwrap())
} else if let Some(ref saved_re) = context.saved_regex {
// We already have one: just borrow it.
Ok(saved_re)
} else {
input_runtime_error(location, context, "no previous regular expression")
}
}
#[cfg(unix)]
fn shell_command(cmd: &str) -> std::process::Command {
let mut c = std::process::Command::new("/bin/sh");
c.arg("-c").arg(cmd);
// Patched for pi-uutils-ctx embedding: run relative to the shell's cwd,
// not the host process cwd. `output()` already keeps the child's stdio
// away from the host's (stdin closed, stdout/stderr captured).
c.current_dir(pi_uutils_ctx::cwd());
c
}
#[cfg(windows)]
fn shell_command(cmd: &str) -> std::process::Command {
let mut c = std::process::Command::new("cmd.exe");
c.arg("/C").arg(cmd);
// Patched for pi-uutils-ctx embedding: see the unix variant above.
c.current_dir(pi_uutils_ctx::cwd());
c
}
// Fallback if the target OS is neither Windows nor UNIX-like
#[cfg(not(any(unix, windows)))]
fn shell_command(_cmd: &str) -> std::process::Command {
unimplemented!("the 'e' substitute flag requires a platform shell (/bin/sh or cmd.exe)");
}
/// Perform the specified RE replacement in the provided pattern space.
fn substitute(
pattern: &mut IOChunk,
command: &Command,
context: &mut ProcessingContext,
output: &mut OutputBuffer,
) -> UResult<()> {
let sub = extract_variant!(command, Substitution);
let mut count = 0;
let mut last_end = 0;
let mut result = String::new();
let mut replaced = false;
let mut text: Option<&str> = None;
let regex = re_or_saved_re(sub.regex.as_ref(), context, &command.location)?;
// The following let block allows a common input_runtime_error to be
// called once in all cases, and most importantly, to finish the regex
// mutable borrowing of context, so as to reuse context in the error call.
let subst_result = match (sub.occurrence, sub.replacement.max_group_number) {
(1, 0) => {
// Example: s/foo/bar/: find() is enough.
match regex.find(pattern) {
Err(e) => Err(e),
Ok(Some(m)) => {
text = Some(pattern.as_str()?);
result.push_str(&text.unwrap()[last_end..m.start()]);
let replacement = sub.replacement.apply_match(&m);
result.push_str(&replacement);
replaced = true;
last_end = m.end();
Ok(())
},
Ok(None) => Ok(()), // No match
}
},
(1, _) => {
// Example: s/\(.\)\(.\)/\2\1/: captures() is enough.
match regex.captures(pattern) {
Err(e) => Err(e),
Ok(Some(caps)) => {
let m = caps.get(0)?.unwrap();
text = Some(pattern.as_str()?);
result.push_str(&text.unwrap()[last_end..m.start()]);
let replacement = sub.replacement.apply_captures(command, &caps)?;
result.push_str(&replacement);
replaced = true;
last_end = m.end();
Ok(())
},
Ok(None) => Ok(()), // No match
}
},
(..) => {
// Example: s/(.)(.)/\2\1/3: captures_iter() is needed.
// Iterate over multiple captures of the RE in the pattern.
'captures: {
for caps_result in regex.captures_iter(pattern)? {
let caps = match caps_result {
Ok(caps) => caps,
Err(e) => break 'captures Err(e),
};
count += 1;
let m = caps.get(0)?.unwrap();
// Always write the unmatched text before this match.
if text.is_none() {
text = Some(pattern.as_str()?);
}
result.push_str(&text.unwrap()[last_end..m.start()]);
if sub.occurrence == 0 || count == sub.occurrence {
let replacement = sub.replacement.apply_captures(command, &caps)?;
result.push_str(&replacement);
replaced = true;
} else {
// Not the target match — leave the match unchanged.
result.push_str(m.as_str());
}
last_end = m.end();
// Early exit if only a specific occurrence,
// (likely 1) needed replacing.
if count == sub.occurrence {
break 'captures Ok(());
}
}
break 'captures Ok(());
}
},
};
// Handle errors.
if let Err(e) = subst_result {
return input_runtime_error(&command.location, context, e.to_string());
}
// Handle substitution success.
if replaced {
result.push_str(&text.unwrap()[last_end..]);
pattern.set_to_string(result, pattern.is_newline_terminated());
// Execute the pattern space as a shell command if the 'e' flag is set
if sub.execute {
let cmd_str = pattern.as_str()?.to_string();
let output_bytes = shell_command(&cmd_str).output().map_err(|e| {
input_runtime_error::<()>(
&command.location,
context,
format!("failed to execute shell command: {e}"),
)
.unwrap_err()
})?;
let mut shell_out = String::from_utf8_lossy(&output_bytes.stdout).into_owned();
if shell_out.ends_with("\r\n") {
// On windows, both return carriage and newline characters are used
shell_out.truncate(shell_out.len() - 2);
} else if shell_out.ends_with('\n') {
// Strip the trailing newline, as GNU sed does
shell_out.pop();
}
pattern.set_to_string(shell_out, pattern.is_newline_terminated());
}
if sub.print_flag {
write_chunk(output, context, pattern)?;
}
// Write to file if needed.
if let Some(ref writer) = sub.write_file {
writer.borrow_mut().write_line(pattern.as_str()?)?;
}
context.substitution_made = true;
}
Ok(())
}
/// Apply the specified transliteration in the provided pattern space.
fn transliterate(pattern: &mut IOChunk, trans: &Transliteration) -> UResult<()> {
let text = pattern.as_str()?;
let mut result = String::with_capacity(text.len());
let mut replaced = false;
// Perform the transliteration.
for ch in text.chars() {
let mapped = trans.lookup(ch);
if mapped != ch {
replaced = true;
}
result.push(mapped);
}
// Lazy replace.
if replaced {
pattern.set_to_string(result, pattern.is_newline_terminated());
}
Ok(())
}
/// Output any data queued for output at the end of the cycle.
fn flush_appends(output: &mut OutputBuffer, context: &mut ProcessingContext) -> UResult<()> {
for elem in &context.append_elements {
match elem {
AppendElement::Text(text) => {
output.write_str(&**text)?;
},
AppendElement::Path(path) => {
output.copy_file(path)?;
},
}
}
context.append_elements.clear();
Ok(())
}
/// List the passed pattern space in unambiguous form.
fn list(output: &mut OutputBuffer, line: &IOChunk, max_width: usize) -> UResult<()> {
// Special case for an empty pattern space
if line.is_empty() {
if line.is_newline_terminated() {
output.write_str("$\n")?;
}
return Ok(());
}
let line = line.as_str()?;
let mut buff = String::new();
let mut line_width = 0;
for ch in line.chars() {
if ch == '\n' {
buff.push_str("$\n");
output.write_str(&buff)?;
line_width = 0;
continue;
}
let mut char_buff = [0u8; 1];
let out_str: Cow<str> = match ch {
'\x07' => Cow::Borrowed(r"\a"),
'\x08' => Cow::Borrowed(r"\b"),
'\x0b' => Cow::Borrowed(r"\v"),
'\x0c' => Cow::Borrowed(r"\f"),
'\\' => Cow::Borrowed(r"\\"),
'\r' => Cow::Borrowed(r"\r"),
'\t' => Cow::Borrowed(r"\t"),
c if c.is_ascii_control() => Cow::Owned(format!("\\{:03o}", ch as u8)),
c if c == ' ' || c.is_ascii_graphic() => Cow::Borrowed(ch.encode_utf8(&mut char_buff)),
c if (c as u32) <= 0xffff => Cow::Owned(format!("\\u{:04X}", c as u32)),
_ => Cow::Owned(format!("\\U{:08X}", ch as u32)),
};
// See if folding is required before adding out_str and terminator.
let out_len = out_str.len();
if line_width + out_len + 1 > max_width {
buff.push_str("\\\n");
output.write_str(&buff)?;
line_width = 0;
buff.clear();
}
buff.push_str(out_str.as_ref());
line_width += out_len;
}
if !buff.is_empty() {
buff.push_str("$\n");
output.write_str(buff)?;
}
Ok(())
}
/// Handle address 0 read at the beginning of each file.
fn process_address_0(
commands: Option<Rc<RefCell<Command>>>,
output: &mut OutputBuffer,
) -> UResult<()> {
// Prescan for zero-address which must produce output
// before any input line is read.
{
let mut current = commands;
while let Some(cmd_rc) = current {
let next = {
let cmd = cmd_rc.borrow();
if cmd.code == 'r' && matches!(cmd.addr1, Some(Address::Line(0))) && cmd.addr2.is_none()
{
let path = extract_variant!(cmd, Path);
output.copy_file(path)?;
}
cmd.next.clone()
};
current = next;
}
}
Ok(())
}
#[allow(clippy::cognitive_complexity)]
/// Process a single input file
fn process_file(
commands: Option<Rc<RefCell<Command>>>,
reader: &mut LineReader,
output: &mut OutputBuffer,
context: &mut ProcessingContext,
) -> UResult<()> {
process_address_0(commands.clone(), output)?;
// Loop over the input lines as pattern space.
'lines: while let Some(mut pattern) = reader.get_line()? {
// Patched for pi-uutils-ctx embedding: mmap-backed input never
// touches the (cancel-aware) stdin reader, so poll the host cancel
// flag here to keep long file runs abortable.
if pi_uutils_ctx::is_cancelled() {
break;
}
context.line_number += 1;
context.substitution_made = false;
// Set the script command from which to start.
let mut current: Option<Rc<RefCell<Command>>> =
if let Some(action) = context.input_action.take() {
// Continue processing the `N` command.
let current_line = pattern.as_str()?;
let mut combined_lines = action.prepend;
combined_lines.push('\n');
combined_lines.push_str(current_line);
pattern.set_to_string(combined_lines, pattern.is_newline_terminated());
action.next_command
} else {
// Start from the script top.
commands.clone()
};
// Loop over script commands.
while let Some(command_rc) = current.take() {
let mut command = command_rc.borrow_mut();
if !applies(&mut command, reader, &mut pattern, context)? {
// Advance to next command
current.clone_from(&command.next);
continue;
}
match command.code {
'{' => {
// Block begin; start processing the enclosed ones.
let body = extract_variant!(command, BranchTarget);
current.clone_from(body);
continue;
},
'}' => {
// Block end: continue with the block's patched next.
},
'a' => {
// Write the text to standard output at a later point.
let text = extract_variant!(command, Text);
context
.append_elements
.push(AppendElement::Text(text.clone()));
},
'b' => {
// Branch to the specified label or end if none is given.
let target = extract_variant!(command, BranchTarget);
if target.is_some() {
// New command to execute
current.clone_from(target);
continue;
}
// Branch to the end of the script.
break;
},
'c' => {
// At range end replace pattern space with text and
// start the next cycle.
pattern.clear();
if command.addr2.is_none() || context.last_address || reader.last_line()? {
let text = extract_variant!(command, Text);
output.write_str(text.as_ref())?;
}
break;
},
'd' => {
// Delete the pattern space and start the next cycle.
pattern.clear();
break;
},
'D' => {
// Delete up to \n and start a new cycle without new input.
if let Some(pos) = pattern.as_str()?.find('\n') {
let (s, _) = pattern.fields_mut()?;
s.drain(..=pos);
current.clone_from(&commands);
continue;
}
// Same as d
pattern.clear();
break;
},
'g' => {
// Replace pattern with the contents of the hold space.
pattern.set_to_string(context.hold.content.clone(), context.hold.has_newline);
},
'G' => {
// Append to pattern \n followed by hold space contents.
let (pat_content, pat_has_newline) = pattern.fields_mut()?;
pat_content.push('\n');
pat_content.push_str(&context.hold.content);
*pat_has_newline = context.hold.has_newline;
},
'h' => {
// Replace hold with the contents of the pattern space.
context.hold.content = pattern.as_str()?.to_string();
context.hold.has_newline = pattern.is_newline_terminated();
},
'H' => {
// Append to hold \n followed by pattern space contents.
context.hold.content.push('\n');
context.hold.content.push_str(pattern.as_str()?);
context.hold.has_newline = pattern.is_newline_terminated();
},
'i' => {
// Write text to standard output.
let text = extract_variant!(command, Text);
output.write_str(text.as_ref())?;
},
'l' => {
let width = *extract_variant!(command, Number);
list(output, &pattern, width)?;
},
'n' => {
break;
},
'N' => {
flush_appends(output, context)?;
// Append to pattern `\n` and the next line
// Rather than reading input here, which would result
// in a double borrow on reader, modify the action
// to perform when the next line is read.
context.input_action = Some(InputAction {
next_command: command.next.clone(),
prepend: pattern.as_str()?.to_string(),
});
continue 'lines;
},
'p' => {
write_chunk(output, context, &pattern)?;
},
'P' => {
let line = pattern.as_str()?;
if let Some(pos) = line.find('\n') {
output.write_str(&line[..=pos])?;
} else {
write_chunk(output, context, &pattern)?;
}
},
'q' => {
// Quit after printing the pattern space.
pi_uutils_ctx::set_exit_code(*extract_variant!(command, Number) as i32);
context.stop_processing = true;
break;
},
'Q' => {
// Quit immediatelly.
pi_uutils_ctx::set_exit_code(*extract_variant!(command, Number) as i32);
context.stop_processing = true;
context.quiet = true;
break;
},
'r' => {
// Copy the file to standard output at a later point.
let path = extract_variant!(command, Path);
context
.append_elements
.push(AppendElement::Path(path.clone()));
},
's' => {
substitute(&mut pattern, &command, context, output)?;
},
't' if !context.substitution_made => { /* Do nothing. */ },
't' => {
// Branch to the specified label or end if none is given
// if a substitution was made since last cycle or t.
let target = extract_variant!(command, BranchTarget);
context.substitution_made = false;
if target.is_some() {
// New command to execute
current.clone_from(target);
continue;
}
// Branch to the end of the script.
break;
},
'w' => {
// Append the pattern space to the specified file.
let writer = extract_variant!(command, NamedWriter);
writer.borrow_mut().write_line(pattern.as_str()?)?;
},
'x' => {
// Exchange the contents of the pattern and hold spaces.
let (pat_content, pat_has_newline) = pattern.fields_mut()?;
// Swap newline if hold space is logically non-empty.
if !context.hold.content.is_empty() || context.hold.has_newline {
std::mem::swap(pat_has_newline, &mut context.hold.has_newline);
}
std::mem::swap(pat_content, &mut context.hold.content);
},
'y' => {
let trans = extract_variant!(command, Transliteration);
transliterate(&mut pattern, trans)?;
},
'z' => {
// Clear the pattern contents, but preserve newline state
// so automatic printing still emits an empty record.
let (pat_content, _) = pattern.fields_mut()?;
pat_content.clear();
},
':' => {
// Branch target; do nothing.
},
'=' => {
// Output current line number.
output.write_str(format!("{}\n", context.line_number))?;
},
// The compilation should supply only valid codes.
_ => panic!("invalid command code"),
} // match
// Advance to next command.
current.clone_from(&command.next);
}
if !context.quiet {
write_chunk(output, context, &pattern)?;
}
flush_appends(output, context)?;
if context.stop_processing {
output.flush_pending_newline()?;
break;
}
}
// Handle any N command remains.
if context.separate
&& !context.quiet
&& let Some(action) = context.input_action.take()
{
let mut pending = action.prepend;
pending.push('\n');
output.write_str(pending)?;
if context.unbuffered {
output.flush()?;
}
}
Ok(())
}
/// Mark all address ranges non-active (and 0-starting ones as active).
fn reset_latched_address_ranges(range_commands: &mut [Rc<RefCell<Command>>]) {
for cmd_rc in range_commands.iter() {
let mut cmd = cmd_rc.borrow_mut();
cmd.start_line =
// Check for address-spec line 0 pre-latch extension.
if let Some(addr1) = &cmd.addr1 && matches!(addr1, Address::Line(0)) {
Some(0)
} else {
None
};
}
}
/// Process all input files
pub fn process_all_files(
commands: Option<Rc<RefCell<Command>>>,
files: Vec<PathBuf>,
context: &mut ProcessingContext,
) -> UResult<()> {
// Patched for pi-uutils-ctx embedding: the context streams are never a
// terminal, so upstream's stdout-tty check for auto-unbuffered output is
// dropped; `-u` alone controls flushing.
let mut in_place = InPlace::new(context.clone());
let last_file_index = files.len() - 1;
for (index, path) in files.iter().enumerate() {
context.last_file = index == last_file_index;
let mut reader = LineReader::open(path)
.map_err_context(|| format!("error opening input file {}", path.quote()))?;
let output = in_place.begin(path)?;
if context.separate || index == 0 {
context.line_number = 0;
reset_latched_address_ranges(&mut context.range_commands);
// Reset hold space for separate file processing
context.hold.content.clear();
context.hold.has_newline = true;
}
context.input_name = path.quote().to_string();
process_file(commands.clone(), &mut reader, output, context)?;
// Handle any N command remains.
if context.last_file
&& !context.separate
&& !context.quiet
&& let Some(action) = context.input_action.take()
{
let mut pending = action.prepend;
pending.push('\n');
output.write_str(pending)?;
}
in_place.end()?;
if context.stop_processing {
break;
}
}
// Flush all output files
named_writer::flush_all()?;
Ok(())
}
+139
View File
@@ -0,0 +1,139 @@
// Provide the script contents character by character
//
// SPDX-License-Identifier: MIT
// Copyright (c) 2025 Diomidis Spinellis
//
// This file is part of the uutils sed package.
// It is licensed under the MIT License.
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
#[derive(Debug)]
pub struct ScriptCharProvider {
line: Vec<char>,
pos: usize,
}
impl ScriptCharProvider {
pub fn new(line_string: &str) -> Self {
Self { line: line_string.chars().collect(), pos: 0 }
}
/// Advances to the next character, if not at end of line.
pub fn advance(&mut self) {
if self.pos < self.line.len() {
self.pos += 1;
}
}
/// Retreats current position by specified number or to beginning.
pub fn retreat(&mut self, n: usize) {
self.pos = self.pos.saturating_sub(n);
}
/// Sets new current position.
pub fn set_position(&mut self, pos: usize) {
self.pos = pos;
}
/// Returns the current character. Panics if out of bounds.
pub fn current(&self) -> char {
self.line[self.pos]
}
/// Returns true if at the end of the line.
pub fn eol(&self) -> bool {
self.pos >= self.line.len()
}
/// Advances the position past any whitespace characters.
pub fn eat_spaces(&mut self) {
while self.pos < self.line.len() && self.line[self.pos].is_whitespace() {
self.pos += 1;
}
}
/// Return current position
pub fn get_pos(&self) -> usize {
self.pos
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_basic_navigation() {
let mut provider = ScriptCharProvider::new("abc");
assert_eq!(provider.get_pos(), 0);
assert_eq!(provider.current(), 'a');
provider.advance();
assert_eq!(provider.get_pos(), 1);
assert_eq!(provider.current(), 'b');
provider.advance();
assert_eq!(provider.get_pos(), 2);
assert_eq!(provider.current(), 'c');
provider.advance();
assert_eq!(provider.get_pos(), 3);
assert!(provider.eol());
}
#[test]
#[should_panic]
fn test_current_panics_out_of_bounds() {
let mut provider = ScriptCharProvider::new("x");
provider.advance(); // now at end
provider.current(); // should panic
}
#[test]
fn test_eat_spaces() {
let mut provider = ScriptCharProvider::new(" xyz");
provider.eat_spaces();
assert_eq!(provider.current(), 'x');
}
#[test]
fn test_eol_on_empty() {
let provider = ScriptCharProvider::new("");
assert!(provider.eol());
}
#[test]
fn test_eat_spaces_mixed() {
let mut provider = ScriptCharProvider::new(" \t\nabc");
provider.eat_spaces();
assert_eq!(provider.current(), 'a');
}
#[test]
fn test_retreat_normal() {
let mut chars = ScriptCharProvider::new("abcdef");
chars.pos = 4; // simulate position at 'e'
chars.retreat(2);
assert_eq!(chars.get_pos(), 2);
assert_eq!(chars.current(), 'c');
}
#[test]
fn test_retreat_to_start() {
let mut chars = ScriptCharProvider::new("abcdef");
chars.pos = 3; // simulate position at 'd'
chars.retreat(5); // retreat more than current pos
assert_eq!(chars.get_pos(), 0);
assert_eq!(chars.current(), 'a');
}
#[test]
fn test_retreat_zero() {
let mut chars = ScriptCharProvider::new("abcdef");
chars.pos = 2; // at 'c'
chars.retreat(0); // retreat by 0
assert_eq!(chars.get_pos(), 2);
assert_eq!(chars.current(), 'c');
}
}
+284
View File
@@ -0,0 +1,284 @@
//! Provide the script contents line by line
//
// SPDX-License-Identifier: MIT
// Copyright (c) 2025 Diomidis Spinellis
//
// This file is part of the uutils sed package.
// It is licensed under the MIT License.
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use std::{
fmt,
fs::File,
io::{BufRead, BufReader},
path::PathBuf,
};
use uucore::{
display::Quotable,
error::{FromIo, UResult},
};
#[derive(Debug, PartialEq)]
/// The specification of a script: through a string or a file
pub enum ScriptValue {
StringVal(String),
PathVal(PathBuf),
}
#[derive(Debug)]
/// The provider of script lines across all specified scripts
/// Scripts can be specified to sed as files or as strings.
pub struct ScriptLineProvider {
sources: Vec<ScriptValue>,
state: State,
}
/// Encapsulation of the script line provider's state
enum State {
NotStarted, // Processing has not yet started
Active {
index: usize,
reader: Box<dyn BufRead>, // Object on which read_line is called
input_name: String, // Input description (path or script string)
line_number: usize, // Current line number
},
Done, // All scripts have been processed
}
impl ScriptLineProvider {
/// Construct the script provider from the specified script sources
pub fn new(sources: Vec<ScriptValue>) -> Self {
Self { sources, state: State::NotStarted }
}
/// Return the currently processed script line number.
pub fn get_line_number(&self) -> usize {
match &self.state {
State::Active { line_number, .. } => *line_number,
_ => 0,
}
}
/// Return the currently processed script descriptive name.
pub fn get_input_name(&self) -> &str {
match &self.state {
State::Active { input_name, .. } => input_name.as_str(),
_ => "",
}
}
/// Return the next script line to process across all scripts.
pub fn next_line(&mut self) -> UResult<Option<String>> {
let mut line = String::new();
loop {
let advance = match &mut self.state {
State::NotStarted => Some(0),
State::Active { index, reader, line_number, .. } => {
line.clear();
let bytes = reader.read_line(&mut line)?;
if bytes == 0 {
Some(*index + 1) // finished reading this source
} else {
*line_number += 1;
// Remove trailing newline
if line.ends_with('\n') {
line.pop();
}
return Ok(Some(line));
}
},
State::Done => {
return Ok(None);
},
};
if let Some(next_index) = advance {
self.advance_source(next_index)?;
}
}
}
// Move to the next available script source.
fn advance_source(&mut self, next_index: usize) -> UResult<()> {
if next_index >= self.sources.len() {
self.state = State::Done;
return Ok(());
}
match &self.sources[next_index] {
ScriptValue::StringVal(s) => {
let cursor = std::io::Cursor::new(s.clone());
self.state = State::Active {
index: next_index,
reader: Box::new(BufReader::new(cursor)),
input_name: format!("<script argument {}>", next_index + 1),
line_number: 0,
};
},
ScriptValue::PathVal(p) => {
if p.to_string_lossy() == "-" {
self.state = State::Active {
index: next_index,
reader: Box::new(BufReader::new(pi_uutils_ctx::stdin())),
input_name: "<stdin>".to_string(),
line_number: 0,
};
} else {
// Patched for pi-uutils-ctx embedding: resolve `-f`
// script files against the shell working directory.
let file = File::open(pi_uutils_ctx::resolve(p))
.map_err_context(|| format!("error opening script file {}", p.quote()))?;
self.state = State::Active {
index: next_index,
reader: Box::new(BufReader::new(file)),
input_name: p.to_string_lossy().to_string(),
line_number: 0,
};
}
},
}
Ok(())
}
}
impl fmt::Debug for State {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
match self {
State::NotStarted => f.debug_struct("NotStarted").finish(),
State::Done => f.debug_struct("Done").finish(),
State::Active { index, input_name, line_number, .. } => f
.debug_struct("Active")
.field("index", index)
.field("input_name", input_name)
.field("line_number", line_number)
.field("reader", &"<BufRead>")
.finish(),
}
}
}
#[cfg(test)]
impl ScriptLineProvider {
pub fn with_active_state(input_name: &str, line_number: usize) -> Self {
Self {
sources: vec![],
state: State::Active {
input_name: input_name.to_string(),
line_number,
index: 0,
reader: Box::new(BufReader::new(pi_uutils_ctx::stdin())),
},
}
}
}
#[cfg(test)]
mod tests {
use std::io::Write;
use tempfile::NamedTempFile;
use super::*;
#[test]
fn test_string_source() {
let input = vec![
ScriptValue::StringVal("line one\nline two\n".to_string()),
ScriptValue::StringVal("line three".to_string()),
];
let mut provider = ScriptLineProvider::new(input);
let mut lines = Vec::new();
while let Some(line) = provider.next_line().unwrap() {
lines.push(line.trim_end().to_string());
}
assert_eq!(lines, vec!["line one", "line two", "line three"]);
}
#[test]
fn test_file_source() {
let mut temp_file = NamedTempFile::new().unwrap();
writeln!(temp_file, "file line 1").unwrap();
writeln!(temp_file, "file line 2").unwrap();
let input = vec![ScriptValue::PathVal(temp_file.path().to_path_buf())];
let mut provider = ScriptLineProvider::new(input);
let mut lines = Vec::new();
while let Some(line) = provider.next_line().unwrap() {
lines.push(line.trim_end().to_string());
}
assert_eq!(lines, vec!["file line 1", "file line 2"]);
}
#[test]
fn test_mixed_source() {
let mut temp_file = NamedTempFile::new().unwrap();
writeln!(temp_file, "file line 1").unwrap();
writeln!(temp_file, "file line 2").unwrap();
let temp_file2 = NamedTempFile::new().unwrap();
let input = vec![
ScriptValue::PathVal(temp_file.path().to_path_buf()),
ScriptValue::StringVal("script line 1".to_string()),
ScriptValue::PathVal(temp_file.path().to_path_buf()),
ScriptValue::StringVal(String::new()),
ScriptValue::PathVal(temp_file2.path().to_path_buf()),
ScriptValue::StringVal("other script line 1".to_string()),
];
let mut provider = ScriptLineProvider::new(input);
let mut lines = Vec::new();
while let Some(line) = provider.next_line().unwrap() {
lines.push(line.trim_end().to_string());
}
assert_eq!(lines, vec![
"file line 1",
"file line 2",
"script line 1",
"file line 1",
"file line 2",
"other script line 1",
]);
}
#[test]
fn test_getters() {
let input = vec![
ScriptValue::StringVal("l1\nl2\n".to_string()),
ScriptValue::StringVal("l3".to_string()),
];
let mut provider = ScriptLineProvider::new(input);
if let Some(line) = provider.next_line().unwrap() {
assert_eq!(line.trim(), "l1");
assert_eq!(provider.get_line_number(), 1);
assert_eq!(provider.get_input_name(), "<script argument 1>");
} else {
panic!("Expected a line");
}
if let Some(line) = provider.next_line().unwrap() {
assert_eq!(line.trim(), "l2");
assert_eq!(provider.get_line_number(), 2);
assert_eq!(provider.get_input_name(), "<script argument 1>");
} else {
panic!("Expected a line");
}
if let Some(line) = provider.next_line().unwrap() {
assert_eq!(line.trim(), "l3");
assert_eq!(provider.get_line_number(), 1);
assert_eq!(provider.get_input_name(), "<script argument 2>");
} else {
panic!("Expected a line");
}
}
}
+3 -2
View File
@@ -3,7 +3,8 @@
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// pi-uutils: Patched for in-process embedding via the shared `uu-checksum-common` crate,
// which redirects all standard stream I/O and file resolution through `pi-uutils-ctx`.
// pi-uutils: Patched for in-process embedding via the shared
// `uu-checksum-common` crate, which redirects all standard stream I/O and file
// resolution through `pi-uutils-ctx`.
uu_checksum_common::declare_standalone!("sha1sum", uucore::checksum::AlgoKind::Sha1);
+3 -2
View File
@@ -3,7 +3,8 @@
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// pi-uutils: Patched for in-process embedding via the shared `uu-checksum-common` crate,
// which redirects all standard stream I/O and file resolution through `pi-uutils-ctx`.
// pi-uutils: Patched for in-process embedding via the shared
// `uu-checksum-common` crate, which redirects all standard stream I/O and file
// resolution through `pi-uutils-ctx`.
uu_checksum_common::declare_standalone!("sha224sum", uucore::checksum::AlgoKind::Sha224);
+3 -2
View File
@@ -3,7 +3,8 @@
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// pi-uutils: Patched for in-process embedding via the shared `uu-checksum-common` crate,
// which redirects all standard stream I/O and file resolution through `pi-uutils-ctx`.
// pi-uutils: Patched for in-process embedding via the shared
// `uu-checksum-common` crate, which redirects all standard stream I/O and file
// resolution through `pi-uutils-ctx`.
uu_checksum_common::declare_standalone!("sha256sum", uucore::checksum::AlgoKind::Sha256);
+3 -2
View File
@@ -3,7 +3,8 @@
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// pi-uutils: Patched for in-process embedding via the shared `uu-checksum-common` crate,
// which redirects all standard stream I/O and file resolution through `pi-uutils-ctx`.
// pi-uutils: Patched for in-process embedding via the shared
// `uu-checksum-common` crate, which redirects all standard stream I/O and file
// resolution through `pi-uutils-ctx`.
uu_checksum_common::declare_standalone!("sha384sum", uucore::checksum::AlgoKind::Sha384);
+3 -2
View File
@@ -3,7 +3,8 @@
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// pi-uutils: Patched for in-process embedding via the shared `uu-checksum-common` crate,
// which redirects all standard stream I/O and file resolution through `pi-uutils-ctx`.
// pi-uutils: Patched for in-process embedding via the shared
// `uu-checksum-common` crate, which redirects all standard stream I/O and file
// resolution through `pi-uutils-ctx`.
uu_checksum_common::declare_standalone!("sha512sum", uucore::checksum::AlgoKind::Sha512);
+41 -10
View File
@@ -3,9 +3,10 @@
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use clap::{Arg, ArgAction, Command, builder::PossibleValue};
use std::ffi::OsString;
use clap::{Arg, ArgAction, Command, builder::PossibleValue};
pub mod options {
pub const APPEND: &str = "append";
pub const IGNORE_INTERRUPTS: &str = "ignore-interrupts";
@@ -23,8 +24,8 @@ pub enum OutputErrorMode {
}
pub struct Options {
pub append: bool,
pub files: Vec<OsString>,
pub append: bool,
pub files: Vec<OsString>,
pub output_error: Option<OutputErrorMode>,
}
@@ -36,11 +37,39 @@ pub fn uu_app() -> Command {
.after_help("If a FILE is -, copy again to standard output.")
.infer_long_args(true)
.disable_help_flag(true)
.arg(Arg::new("--help").short('h').long("help").help("Print help").action(ArgAction::HelpLong))
.arg(Arg::new(options::APPEND).long(options::APPEND).short('a').help("append to the given FILEs, do not overwrite").action(ArgAction::SetTrue))
.arg(Arg::new(options::IGNORE_INTERRUPTS).long(options::IGNORE_INTERRUPTS).short('i').help("ignore interrupt signals (accepted without installing a process-global handler)").action(ArgAction::SetTrue))
.arg(Arg::new(options::FILE).action(ArgAction::Append).value_hint(clap::ValueHint::FilePath).value_parser(clap::value_parser!(OsString)))
.arg(Arg::new(options::IGNORE_PIPE_ERRORS).short('p').help("diagnose errors writing to non pipes").action(ArgAction::SetTrue))
.arg(
Arg::new("--help")
.short('h')
.long("help")
.help("Print help")
.action(ArgAction::HelpLong),
)
.arg(
Arg::new(options::APPEND)
.long(options::APPEND)
.short('a')
.help("append to the given FILEs, do not overwrite")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::IGNORE_INTERRUPTS)
.long(options::IGNORE_INTERRUPTS)
.short('i')
.help("ignore interrupt signals (accepted without installing a process-global handler)")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::FILE)
.action(ArgAction::Append)
.value_hint(clap::ValueHint::FilePath)
.value_parser(clap::value_parser!(OsString)),
)
.arg(
Arg::new(options::IGNORE_PIPE_ERRORS)
.short('p')
.help("diagnose errors writing to non pipes")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::OUTPUT_ERROR)
.long(options::OUTPUT_ERROR)
@@ -49,9 +78,11 @@ pub fn uu_app() -> Command {
.default_missing_value("warn-nopipe")
.value_parser([
PossibleValue::new("warn").help("diagnose errors writing to any output"),
PossibleValue::new("warn-nopipe").help("diagnose errors writing to any output not a pipe"),
PossibleValue::new("warn-nopipe")
.help("diagnose errors writing to any output not a pipe"),
PossibleValue::new("exit").help("exit on error writing to any output"),
PossibleValue::new("exit-nopipe").help("exit on error writing to any output not a pipe"),
PossibleValue::new("exit-nopipe")
.help("exit on error writing to any output not a pipe"),
])
.help("set behavior on write error"),
)
+33 -11
View File
@@ -3,9 +3,11 @@
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use std::ffi::OsString;
use std::fs::{File, OpenOptions};
use std::io::{Error, ErrorKind, Read, Result, Write};
use std::{
ffi::OsString,
fs::{File, OpenOptions},
io::{Error, ErrorKind, Read, Result, Write},
};
use uucore::display::Quotable;
@@ -37,7 +39,11 @@ pub fn run(argv: Vec<OsString>) -> i32 {
"exit-nopipe" => OutputErrorMode::ExitNoPipe,
_ => unreachable!("clap validates output-error"),
})
.or_else(|| matches.get_flag(options::IGNORE_PIPE_ERRORS).then_some(OutputErrorMode::WarnNoPipe));
.or_else(|| {
matches
.get_flag(options::IGNORE_PIPE_ERRORS)
.then_some(OutputErrorMode::WarnNoPipe)
});
let files = matches
.get_many::<OsString>(options::FILE)
.map(|values| values.cloned().collect())
@@ -61,7 +67,8 @@ fn tee(options: &Options) -> Result<()> {
let mut had_open_errors = false;
for name in &options.files {
if name == "-" {
writers.push(NamedWriter { name: OsString::from("standard output"), inner: Writer::Stdout });
writers
.push(NamedWriter { name: OsString::from("standard output"), inner: Writer::Stdout });
continue;
}
match open(name, options.append) {
@@ -83,7 +90,12 @@ fn tee(options: &Options) -> Result<()> {
let copy_result = copy(pi_uutils_ctx::stdin(), &mut output);
let flush_result = output.flush();
if had_open_errors || copy_result.is_err() || flush_result.is_err() || output.error_occurred() {
Err(copy_result.err().or_else(|| flush_result.err()).unwrap_or_else(|| Error::other("output error")))
Err(
copy_result
.err()
.or_else(|| flush_result.err())
.unwrap_or_else(|| Error::other("output error")),
)
} else {
Ok(())
}
@@ -123,24 +135,30 @@ fn open(name: &OsString, append: bool) -> Result<NamedWriter> {
}
struct MultiWriter {
writers: Vec<NamedWriter>,
writers: Vec<NamedWriter>,
output_error_mode: Option<OutputErrorMode>,
ignored_errors: usize,
ignored_errors: usize,
}
impl MultiWriter {
fn new(writers: Vec<NamedWriter>, output_error_mode: Option<OutputErrorMode>) -> Self {
Self { writers, output_error_mode, ignored_errors: 0 }
}
fn error_occurred(&self) -> bool {
self.ignored_errors != 0
}
fn process(&mut self, flush: bool, buf: &[u8]) -> Result<()> {
let mode = self.output_error_mode.clone();
let mut aborted = None;
let mut errors = 0;
self.writers.retain_mut(|writer| {
let result = if flush { writer.flush() } else { writer.write_all(buf) };
let result = if flush {
writer.flush()
} else {
writer.write_all(buf)
};
match result {
Ok(()) => true,
Err(err) => {
@@ -149,7 +167,8 @@ impl MultiWriter {
matches!(mode.as_ref(), Some(OutputErrorMode::Warn | OutputErrorMode::Exit))
|| !is_pipe;
if report {
let _ = writeln!(pi_uutils_ctx::stderr(), "tee: {}: {err}", writer.name.maybe_quote());
let _ =
writeln!(pi_uutils_ctx::stderr(), "tee: {}: {err}", writer.name.maybe_quote());
errors += 1;
}
let exit = matches!(mode.as_ref(), Some(OutputErrorMode::Exit))
@@ -177,6 +196,7 @@ impl Write for MultiWriter {
self.process(false, buf)?;
Ok(buf.len())
}
fn flush(&mut self) -> Result<()> {
self.process(true, &[])
}
@@ -194,6 +214,7 @@ impl Write for Writer {
Self::Stdout => pi_uutils_ctx::stdout().write(buf),
}
}
fn flush(&mut self) -> Result<()> {
match self {
Self::File(file) => file.flush(),
@@ -204,13 +225,14 @@ impl Write for Writer {
struct NamedWriter {
inner: Writer,
name: OsString,
name: OsString,
}
impl Write for NamedWriter {
fn write(&mut self, buf: &[u8]) -> Result<usize> {
self.inner.write(buf)
}
fn flush(&mut self) -> Result<()> {
self.inner.flush()
}
+573 -602
View File
File diff suppressed because it is too large Load Diff
+58 -54
View File
@@ -5,91 +5,95 @@
//! I/O processing infrastructure for tr operations with SIMD optimizations
use crate::operation::ChunkProcessor;
use std::io::{BufRead, Write};
use uucore::error::{FromIo, UResult};
use crate::operation::ChunkProcessor;
/// Helper to detect single-character operations for optimization
pub fn find_single_change<T, F>(table: &[T; 256], check: F) -> Option<(u8, T)>
where
F: Fn(usize, &T) -> bool,
T: Copy,
F: Fn(usize, &T) -> bool,
T: Copy,
{
let matches: Vec<_> = table
.iter()
.enumerate()
.filter_map(|(i, val)| check(i, val).then_some((i as u8, *val)))
.take(2)
.collect();
let matches: Vec<_> = table
.iter()
.enumerate()
.filter_map(|(i, val)| check(i, val).then_some((i as u8, *val)))
.take(2)
.collect();
(matches.len() == 1).then(|| matches[0])
(matches.len() == 1).then(|| matches[0])
}
/// SIMD-optimized single character replacement
#[inline]
pub fn process_single_char_replace(
input: &[u8],
output: &mut Vec<u8>,
source_char: u8,
target_char: u8,
input: &[u8],
output: &mut Vec<u8>,
source_char: u8,
target_char: u8,
) {
let count = bytecount::count(input, source_char);
if count == 0 {
output.extend_from_slice(input);
} else if count == input.len() {
output.resize(output.len() + input.len(), target_char);
} else {
output.extend(
input
.iter()
.map(|&b| if b == source_char { target_char } else { b }),
);
}
let count = bytecount::count(input, source_char);
if count == 0 {
output.extend_from_slice(input);
} else if count == input.len() {
output.resize(output.len() + input.len(), target_char);
} else {
output.extend(
input
.iter()
.map(|&b| if b == source_char { target_char } else { b }),
);
}
}
/// SIMD-optimized delete operation for single character
pub fn process_single_delete(input: &[u8], output: &mut Vec<u8>, delete_char: u8) {
let count = bytecount::count(input, delete_char);
if count == 0 {
output.extend_from_slice(input);
} else if count < input.len() {
output.extend(input.iter().filter(|&&b| b != delete_char).copied());
}
// If count == input.len(), all deleted, output nothing
let count = bytecount::count(input, delete_char);
if count == 0 {
output.extend_from_slice(input);
} else if count < input.len() {
output.extend(input.iter().filter(|&&b| b != delete_char).copied());
}
// If count == input.len(), all deleted, output nothing
}
/// Unified I/O processing for all operations
pub fn process_input<R, W, P>(input: &mut R, output: &mut W, processor: &P) -> UResult<()>
where
R: BufRead,
W: Write,
P: ChunkProcessor + ?Sized,
R: BufRead,
W: Write,
P: ChunkProcessor + ?Sized,
{
const BUFFER_SIZE: usize = 32768;
let mut buf = [0; BUFFER_SIZE];
let mut output_buf = Vec::with_capacity(BUFFER_SIZE);
const BUFFER_SIZE: usize = 32768;
let mut buf = [0; BUFFER_SIZE];
let mut output_buf = Vec::with_capacity(BUFFER_SIZE);
loop {
let length = match input.read(&mut buf[..]) {
Ok(0) => break,
Ok(len) => len,
Err(e) if e.kind() == std::io::ErrorKind::Interrupted => continue,
Err(e) => return Err(e.map_err_context(|| "read error".to_string())),
};
loop {
let length = match input.read(&mut buf[..]) {
Ok(0) => break,
Ok(len) => len,
Err(e) if e.kind() == std::io::ErrorKind::Interrupted => continue,
Err(e) => return Err(e.map_err_context(|| "read error".to_string())),
};
output_buf.clear();
processor.process_chunk(&buf[..length], &mut output_buf);
output_buf.clear();
processor.process_chunk(&buf[..length], &mut output_buf);
if !output_buf.is_empty() {
write_output(output, &output_buf)?;
}
}
if !output_buf.is_empty() {
write_output(output, &output_buf)?;
}
}
Ok(())
Ok(())
}
/// Helper function to handle platform-specific write operations
#[inline]
pub fn write_output<W: Write>(output: &mut W, buf: &[u8]) -> UResult<()> {
output.write_all(buf).map_err_context(|| "write error".to_string())
output
.write_all(buf)
.map_err_context(|| "write error".to_string())
}
+183 -178
View File
@@ -8,211 +8,216 @@ mod simd;
mod unicode_table;
use std::{
ffi::OsString,
io::{BufReader, Write},
ffi::OsString,
io::{BufReader, Write},
};
use clap::{Arg, ArgAction, Command, value_parser};
use operation::{
DeleteOperation, Sequence, SqueezeOperation, SymbolTranslator, TranslateOperation,
flush_output, translate_input,
DeleteOperation, Sequence, SqueezeOperation, SymbolTranslator, TranslateOperation, flush_output,
translate_input,
};
use pi_uutils_ctx::format_usage;
use simd::process_input;
use uucore::display::Quotable;
use uucore::error::{UResult, UUsageError};
use uucore::os_str_as_bytes;
use uucore::{
display::Quotable,
error::{UResult, UUsageError},
os_str_as_bytes,
};
mod options {
pub const COMPLEMENT: &str = "complement";
pub const DELETE: &str = "delete";
pub const SQUEEZE: &str = "squeeze-repeats";
pub const TRUNCATE_SET1: &str = "truncate-set1";
pub const SETS: &str = "sets";
pub const COMPLEMENT: &str = "complement";
pub const DELETE: &str = "delete";
pub const SQUEEZE: &str = "squeeze-repeats";
pub const TRUNCATE_SET1: &str = "truncate-set1";
pub const SETS: &str = "sets";
}
/// pi-uutils: context-safe in-process entry point. `argv` includes the command name;
/// clap output and all utility diagnostics are written only to scoped streams.
/// pi-uutils: context-safe in-process entry point. `argv` includes the command
/// name; clap output and all utility diagnostics are written only to scoped
/// streams.
pub fn run(argv: Vec<OsString>) -> i32 {
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
}
};
let matches = match uu_app().try_get_matches_from(argv) {
Ok(matches) => matches,
Err(err) => {
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
return 1;
}
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
return 0;
},
};
match tr_main(&matches) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
let _ = writeln!(pi_uutils_ctx::stderr(), "tr: {err}");
if code == 0 { 1 } else { code }
}
}
match tr_main(&matches) {
Ok(()) => pi_uutils_ctx::exit_code(),
Err(err) => {
let code = err.code();
let _ = writeln!(pi_uutils_ctx::stderr(), "tr: {err}");
if code == 0 { 1 } else { code }
},
}
}
fn tr_main(matches: &clap::ArgMatches) -> UResult<()> {
let delete_flag = matches.get_flag(options::DELETE);
let complement_flag = matches.get_flag(options::COMPLEMENT);
let squeeze_flag = matches.get_flag(options::SQUEEZE);
let truncate_set1_flag = matches.get_flag(options::TRUNCATE_SET1);
let delete_flag = matches.get_flag(options::DELETE);
let complement_flag = matches.get_flag(options::COMPLEMENT);
let squeeze_flag = matches.get_flag(options::SQUEEZE);
let truncate_set1_flag = matches.get_flag(options::TRUNCATE_SET1);
let sets: Vec<_> = matches
.get_many::<OsString>(options::SETS)
.into_iter()
.flatten()
.map(ToOwned::to_owned)
.collect();
let sets: Vec<_> = matches
.get_many::<OsString>(options::SETS)
.into_iter()
.flatten()
.map(ToOwned::to_owned)
.collect();
if sets.is_empty() {
return Err(UUsageError::new(1, "missing operand"));
}
if sets.is_empty() {
return Err(UUsageError::new(1, "missing operand"));
}
let sets_len = sets.len();
if !(delete_flag || squeeze_flag) && sets_len == 1 {
return Err(UUsageError::new(
1,
format!(
"missing operand after {}\nTwo strings must be given when translating.",
sets[0].quote()
),
));
}
let sets_len = sets.len();
if !(delete_flag || squeeze_flag) && sets_len == 1 {
return Err(UUsageError::new(
1,
format!(
"missing operand after {}\nTwo strings must be given when translating.",
sets[0].quote()
),
));
}
if delete_flag && squeeze_flag && sets_len == 1 {
return Err(UUsageError::new(
1,
format!(
"missing operand after {}\nTwo strings must be given when deleting and squeezing.",
sets[0].quote()
),
));
}
if delete_flag && squeeze_flag && sets_len == 1 {
return Err(UUsageError::new(
1,
format!(
"missing operand after {}\nTwo strings must be given when deleting and squeezing.",
sets[0].quote()
),
));
}
if sets_len > 1 {
if delete_flag && !squeeze_flag {
let operand = sets[1].quote();
let message = if sets_len == 2 {
format!(
"extra operand {operand}\nOnly one string may be given when deleting without squeezing repeats."
)
} else {
format!("extra operand {operand}")
};
return Err(UUsageError::new(1, message));
}
if sets_len > 2 {
return Err(UUsageError::new(
1,
format!("extra operand {}", sets[2].quote()),
));
}
}
if sets_len > 1 {
if delete_flag && !squeeze_flag {
let operand = sets[1].quote();
let message = if sets_len == 2 {
format!(
"extra operand {operand}\nOnly one string may be given when deleting without \
squeezing repeats."
)
} else {
format!("extra operand {operand}")
};
return Err(UUsageError::new(1, message));
}
if sets_len > 2 {
return Err(UUsageError::new(1, format!("extra operand {}", sets[2].quote())));
}
}
if let Some(first) = sets.first() {
let bytes = os_str_as_bytes(first)?;
let trailing_backslashes = bytes.iter().rev().take_while(|&&byte| byte == b'\\').count();
if trailing_backslashes % 2 == 1 {
let _ = writeln!(
pi_uutils_ctx::stderr(),
"tr: warning: an unescaped backslash at end of string is not portable"
);
}
}
if let Some(first) = sets.first() {
let bytes = os_str_as_bytes(first)?;
let trailing_backslashes = bytes
.iter()
.rev()
.take_while(|&&byte| byte == b'\\')
.count();
if trailing_backslashes % 2 == 1 {
let _ = writeln!(
pi_uutils_ctx::stderr(),
"tr: warning: an unescaped backslash at end of string is not portable"
);
}
}
let translating = !delete_flag && sets.len() > 1;
let mut sets_iter = sets.iter().map(OsString::as_os_str);
let (set1, set2) = Sequence::solve_set_characters(
os_str_as_bytes(sets_iter.next().unwrap_or_default())?,
os_str_as_bytes(sets_iter.next().unwrap_or_default())?,
complement_flag,
truncate_set1_flag && translating,
translating,
)?;
let translating = !delete_flag && sets.len() > 1;
let mut sets_iter = sets.iter().map(OsString::as_os_str);
let (set1, set2) = Sequence::solve_set_characters(
os_str_as_bytes(sets_iter.next().unwrap_or_default())?,
os_str_as_bytes(sets_iter.next().unwrap_or_default())?,
complement_flag,
truncate_set1_flag && translating,
translating,
)?;
// pi-uutils: replace process-global stdin/stdout with the invocation context.
let mut input = BufReader::new(pi_uutils_ctx::stdin());
let mut output = pi_uutils_ctx::stdout();
// pi-uutils: replace process-global stdin/stdout with the invocation context.
let mut input = BufReader::new(pi_uutils_ctx::stdin());
let mut output = pi_uutils_ctx::stdout();
if delete_flag {
if squeeze_flag {
let operation = DeleteOperation::new(set1).chain(SqueezeOperation::new(set2));
translate_input(&mut input, &mut output, operation)?;
} else {
process_input(&mut input, &mut output, &DeleteOperation::new(set1))?;
}
} else if squeeze_flag {
if sets_len == 1 {
translate_input(&mut input, &mut output, SqueezeOperation::new(set1))?;
} else {
let operation = TranslateOperation::new(set1, set2.clone())?
.chain(SqueezeOperation::new(set2));
translate_input(&mut input, &mut output, operation)?;
}
} else {
process_input(
&mut input,
&mut output,
&TranslateOperation::new(set1, set2)?,
)?;
}
if delete_flag {
if squeeze_flag {
let operation = DeleteOperation::new(set1).chain(SqueezeOperation::new(set2));
translate_input(&mut input, &mut output, operation)?;
} else {
process_input(&mut input, &mut output, &DeleteOperation::new(set1))?;
}
} else if squeeze_flag {
if sets_len == 1 {
translate_input(&mut input, &mut output, SqueezeOperation::new(set1))?;
} else {
let operation =
TranslateOperation::new(set1, set2.clone())?.chain(SqueezeOperation::new(set2));
translate_input(&mut input, &mut output, operation)?;
}
} else {
process_input(&mut input, &mut output, &TranslateOperation::new(set1, set2)?)?;
}
flush_output(&mut output)?;
Ok(())
flush_output(&mut output)?;
Ok(())
}
pub fn uu_app() -> Command {
Command::new("tr")
.version(env!("CARGO_PKG_VERSION"))
.about("Translate or delete characters")
.override_usage(format_usage("tr [OPTION]... SET1 [SET2]"))
.after_help(
"Translate, squeeze, and/or delete characters from standard input, writing to standard output.",
)
.infer_long_args(true)
.trailing_var_arg(true)
.arg(
Arg::new(options::COMPLEMENT)
.visible_short_alias('C')
.short('c')
.long(options::COMPLEMENT)
.help("use the complement of SET1")
.action(ArgAction::SetTrue)
.overrides_with(options::COMPLEMENT),
)
.arg(
Arg::new(options::DELETE)
.short('d')
.long(options::DELETE)
.help("delete characters in SET1, do not translate")
.action(ArgAction::SetTrue)
.overrides_with(options::DELETE),
)
.arg(
Arg::new(options::SQUEEZE)
.long(options::SQUEEZE)
.short('s')
.help("replace each sequence of a repeated character listed in the last specified SET with a single occurrence")
.action(ArgAction::SetTrue)
.overrides_with(options::SQUEEZE),
)
.arg(
Arg::new(options::TRUNCATE_SET1)
.long(options::TRUNCATE_SET1)
.short('t')
.help("first truncate SET1 to length of SET2")
.action(ArgAction::SetTrue)
.overrides_with(options::TRUNCATE_SET1),
)
.arg(
Arg::new(options::SETS)
.num_args(1..)
.value_parser(value_parser!(OsString)),
)
Command::new("tr")
.version(env!("CARGO_PKG_VERSION"))
.about("Translate or delete characters")
.override_usage(format_usage("tr [OPTION]... SET1 [SET2]"))
.after_help(
"Translate, squeeze, and/or delete characters from standard input, writing to standard \
output.",
)
.infer_long_args(true)
.trailing_var_arg(true)
.arg(
Arg::new(options::COMPLEMENT)
.visible_short_alias('C')
.short('c')
.long(options::COMPLEMENT)
.help("use the complement of SET1")
.action(ArgAction::SetTrue)
.overrides_with(options::COMPLEMENT),
)
.arg(
Arg::new(options::DELETE)
.short('d')
.long(options::DELETE)
.help("delete characters in SET1, do not translate")
.action(ArgAction::SetTrue)
.overrides_with(options::DELETE),
)
.arg(
Arg::new(options::SQUEEZE)
.long(options::SQUEEZE)
.short('s')
.help(
"replace each sequence of a repeated character listed in the last specified SET \
with a single occurrence",
)
.action(ArgAction::SetTrue)
.overrides_with(options::SQUEEZE),
)
.arg(
Arg::new(options::TRUNCATE_SET1)
.long(options::TRUNCATE_SET1)
.short('t')
.help("first truncate SET1 to length of SET2")
.action(ArgAction::SetTrue)
.overrides_with(options::TRUNCATE_SET1),
)
.arg(
Arg::new(options::SETS)
.num_args(1..)
.value_parser(value_parser!(OsString)),
)
}
+4 -4
View File
@@ -6,10 +6,10 @@
pub static BEL: u8 = 0x7;
pub static BS: u8 = 0x8;
pub static HT: u8 = 0x9;
pub static LF: u8 = 0xA;
pub static VT: u8 = 0xB;
pub static FF: u8 = 0xC;
pub static CR: u8 = 0xD;
pub static LF: u8 = 0xa;
pub static VT: u8 = 0xb;
pub static FF: u8 = 0xc;
pub static CR: u8 = 0xd;
pub static SPACE: u8 = 0x20;
pub static SPACES: &[u8] = &[HT, LF, VT, FF, CR, SPACE];
pub static BLANK: &[u8] = &[HT, SPACE];
+17
View File
@@ -0,0 +1,17 @@
[package]
name = "uu_xargs"
version = "0.8.0"
edition = "2024"
license = "MIT"
[lib]
path = "src/lib.rs"
[dependencies]
clap = { version = "4.5", features = ["cargo"] }
libc = "0.2"
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
[dev-dependencies]
parking_lot = "0.12"
tempfile = "3"
+18
View File
@@ -0,0 +1,18 @@
Copyright (c) Google Inc.
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal in
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
the Software, and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+43
View File
@@ -0,0 +1,43 @@
// Copyright 2021 Collabora, Ltd.
//
// Use of this source code is governed by a MIT-style
// license that can be found in the LICENSE file or at
// https://opensource.org/licenses/MIT.
//! Vendored, patched `xargs` from uutils/findutils, wired to run in-process as
//! a brush shell builtin via [`pi_uutils_ctx`].
//!
//! Upstream: <https://github.com/uutils/findutils>, tag `0.8.0`,
//! commit `b94b5f0122b918e33de59776f264761fec5fa94a`.
pub mod xargs;
/// In-process builtin entry point. The host installs a [`pi_uutils_ctx`] scope
/// (stdio + working directory + environment) on a dedicated blocking thread,
/// then calls this.
///
/// Unlike findutils' real `main` (which `std::process::exit`s on the result of
/// `xargs_main`), this returns the exit code so it is safe to run inside the
/// long-lived host shell process. Items are read from the context stdin,
/// output is routed through the context streams, `-a` operands resolve
/// against the shell working dir, and child processes run in the shell
/// working dir with the shell's exported environment and captured stdio.
pub fn run(argv: Vec<std::ffi::OsString>) -> i32 {
// findutils' `xargs_main` is fundamentally `&[&str]`-based — upstream's
// real `main` builds it straight from `std::env::args()`, so lossy UTF-8
// conversion matches the existing upstream behavior for arguments.
let args: Vec<String> = argv
.iter()
.map(|a| a.to_string_lossy().into_owned())
.collect();
let mut strs: Vec<&str> = args.iter().map(String::as_str).collect();
// `xargs_main` treats argv[0] as the program name and skips it. The host
// always supplies it; guard against an empty argv to avoid an index panic.
if strs.is_empty() {
strs.push("xargs");
}
xargs::xargs_main(&strs)
}
#[cfg(test)]
mod tests;
+192
View File
@@ -0,0 +1,192 @@
//! Behavioral contract tests driving [`crate::run`] under a
//! [`pi_uutils_ctx::scope`], the way the shell host does.
use std::{
collections::HashMap,
ffi::OsString,
io::{self, Write},
path::Path,
sync::{Arc, atomic::AtomicBool},
};
use parking_lot::Mutex;
/// `Send` writer that appends every write to a shared buffer so the test can
/// inspect what the utility wrote to the scope's stdout/stderr.
#[derive(Clone, Default)]
struct Sink(Arc<Mutex<Vec<u8>>>);
impl Sink {
fn contents(&self) -> Vec<u8> {
self.0.lock().clone()
}
}
impl Write for Sink {
fn write(&mut self, buf: &[u8]) -> io::Result<usize> {
self.0.lock().extend_from_slice(buf);
Ok(buf.len())
}
fn flush(&mut self) -> io::Result<()> {
Ok(())
}
}
/// Runs `xargs` with `argv` (sans the leading command name), feeding `stdin`
/// bytes, in `cwd`, with `env` as the scope's exported environment. Returns
/// `(exit code, stdout, stderr)`.
fn run_xargs(
argv: &[&str],
stdin: &[u8],
cwd: &Path,
env: &[(&str, &str)],
) -> (i32, String, String) {
let out = Sink::default();
let err = Sink::default();
let mut full_argv = vec![OsString::from("xargs")];
full_argv.extend(argv.iter().map(OsString::from));
let env: HashMap<String, String> = env
.iter()
.map(|(k, v)| ((*k).to_owned(), (*v).to_owned()))
.collect();
let code = pi_uutils_ctx::scope(
pi_uutils_ctx::ScopeIo {
stdin: Box::new(io::Cursor::new(stdin.to_vec())),
stdin_fd: None,
stdin_is_search_input: false,
stdout: Box::new(out.clone()),
stderr: Box::new(err.clone()),
cwd: cwd.to_path_buf(),
env,
cancel: Arc::new(AtomicBool::new(false)),
},
|| crate::run(full_argv),
);
(
code,
String::from_utf8(out.contents()).expect("utf8 stdout"),
String::from_utf8(err.contents()).expect("utf8 stderr"),
)
}
/// Same, with an empty environment and `.` as the working directory.
fn run_simple(argv: &[&str], stdin: &[u8]) -> (i32, String, String) {
run_xargs(argv, stdin, Path::new("."), &[])
}
#[test]
fn child_stdout_is_captured_through_ctx() {
let (code, out, err) = run_simple(&["echo"], b"a b c\n");
assert_eq!(code, 0);
assert_eq!(out, "a b c\n", "child echo output flows through ctx stdout");
assert_eq!(err, "", "clean run leaves stderr empty");
}
#[test]
fn max_args_batches_into_two_invocations() {
let (code, out, _) = run_simple(&["-n", "2", "echo"], b"a b c\n");
assert_eq!(code, 0);
assert_eq!(out, "a b\nc\n", "-n 2 splits three items into two runs");
}
#[test]
fn default_mode_honors_quotes() {
// "a b" c → exactly two arguments for the child.
let (code, out, _) = run_simple(&["sh", "-c", "echo $#", "_"], b"\"a b\" c\n");
assert_eq!(code, 0);
assert_eq!(out, "2\n", "quoted item stays a single argument");
}
#[test]
fn null_mode_preserves_spaces_and_newlines() {
let (code, out, _) = run_simple(&["-0", "echo"], b"a b\0c\nd\0");
assert_eq!(code, 0);
assert_eq!(out, "a b c\nd\n", "NUL-split items keep spaces and newlines");
}
#[test]
fn replace_places_item_mid_command() {
let (code, out, _) = run_simple(&["-I", "{}", "echo", "hello", "{}", "!"], b"world\n");
assert_eq!(code, 0);
assert_eq!(out, "hello world !\n", "-I substitutes mid-command");
}
#[test]
fn failing_child_yields_123() {
let (code, out, _) = run_simple(&["false"], b"x\n");
assert_eq!(code, 123, "any failed invocation maps to 123");
assert_eq!(out, "");
}
#[test]
fn missing_command_yields_127() {
let (code, _, err) = run_simple(&["definitely-not-a-real-command-xyz"], b"x\n");
assert_eq!(code, 127, "command not found maps to 127");
assert!(err.contains("Command not found"), "diagnostic lands on ctx stderr, got: {err:?}");
}
#[test]
fn exit_255_child_yields_124() {
let (code, _, err) = run_simple(&["sh", "-c", "exit 255", "_"], b"x\n");
assert_eq!(code, 124, "a 255 exit aborts with 124");
assert!(err.contains("255"), "diagnostic mentions the urgent exit, got: {err:?}");
}
#[test]
fn no_run_if_empty_skips_command() {
let (code, out, err) = run_simple(&["-r", "echo"], b"");
assert_eq!(code, 0);
assert_eq!(out, "", "-r with no input runs nothing");
assert_eq!(err, "");
}
#[test]
fn empty_input_without_r_runs_default_echo_once() {
// Upstream findutils 0.8.0 (like GNU) still runs the built-in echo once
// on empty input, producing a single empty line.
let (code, out, _) = run_simple(&[], b"");
assert_eq!(code, 0);
assert_eq!(out, "\n", "default echo prints one empty line");
}
#[test]
fn verbose_echoes_command_line_to_stderr() {
let (code, out, err) = run_simple(&["-t", "echo", "a"], b"b\n");
assert_eq!(code, 0);
assert_eq!(out, "a b\n");
assert_eq!(err, "echo a b\n", "-t prints the command line on stderr");
}
#[test]
fn children_run_in_scope_cwd() {
let dir = tempfile::TempDir::new().expect("tempdir");
let (code, _, err) =
run_xargs(&["sh", "-c", "touch \"$1\"", "_"], b"made.txt\n", dir.path(), &[]);
assert_eq!(code, 0, "stderr: {err:?}");
assert!(
dir.path().join("made.txt").exists(),
"relative paths in the child resolve against the scope cwd"
);
}
#[test]
fn children_see_scope_environment() {
let (code, out, _) =
run_simple_env(&["sh", "-c", "echo \"$XVAR\"", "_"], b"x\n", &[("XVAR", "hello")]);
assert_eq!(code, 0);
assert_eq!(out, "hello\n", "scope env reaches the child via env_snapshot");
}
fn run_simple_env(argv: &[&str], stdin: &[u8], env: &[(&str, &str)]) -> (i32, String, String) {
run_xargs(argv, stdin, Path::new("."), env)
}
#[test]
fn arg_file_resolves_against_scope_cwd() {
let dir = tempfile::TempDir::new().expect("tempdir");
std::fs::write(dir.path().join("items.txt"), "a b\n").expect("write items");
let (code, out, _) = run_xargs(&["-a", "items.txt", "echo"], b"", dir.path(), &[]);
assert_eq!(code, 0);
assert_eq!(out, "a b\n", "-a file opens relative to the scope cwd");
}
File diff suppressed because it is too large Load Diff
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Added
- Added context-safe in-process shell builtins for `base64`; the `md5sum`, `sha1sum`, `sha224sum`, `sha256sum`, `sha384sum`, `sha512sum`, and `b2sum` checksum family; and the common path/text utilities `basename`, `dirname`, `cut`, `tee`, `tr`, `paste`, and `comm`. The vendored uutils implementations support pipelines, shell-relative file operands, redirected output, checksum verification, and abort/timeout cancellation without spawning external binaries.
## [16.4.4] - 2026-07-11
### Fixed
-20
View File
@@ -172,15 +172,6 @@ export declare function __ompInstallTokioRuntime(): void
*/
export declare function __piNativesV16_4_4(): void
/**
* Apply conservative pre-execution rewrites to a bash command.
*
* Strips trailing `| head|tail [safe-args]` and redundant trailing `2>&1`
* from each top-level pipeline. The full rules and bail conditions live in
* `pi_shell::fixup`. Synchronous and cheap (one parse pass over the input).
*/
export declare function applyBashFixups(command: string): BashFixupResult
/**
* Apply ast-grep rewrite rules to matching files; honors `dryRun` and returns
* a promise.
@@ -417,17 +408,6 @@ export interface AstReplaceResult {
parseErrors?: Array<string>
}
/**
* Result of [`apply_bash_fixups`]: a possibly-rewritten command plus the
* substrings that were removed (in source order).
*/
export interface BashFixupResult {
/** Possibly-rewritten command. Equal to the input when no fixup fired. */
command: string
/** Substrings removed, in source order — suitable for a user-facing notice. */
stripped: Array<string>
}
export interface BlockRange {
/** 1-indexed inclusive first line of the resolved block. */
startLine: number
-1
View File
@@ -25,7 +25,6 @@ export const Shell = nativeBindings.Shell;
// functions
export const __ompInstallTokioRuntime = nativeBindings.__ompInstallTokioRuntime;
export const __piNativesV16_4_4 = nativeBindings.__piNativesV16_4_4;
export const applyBashFixups = nativeBindings.applyBashFixups;
export const astEdit = nativeBindings.astEdit;
export const astGrep = nativeBindings.astGrep;
export const astMatch = nativeBindings.astMatch;