From 5a73d65b7a70c7ff223db2f8db8929a9049fd69a Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 18:36:56 +0200 Subject: [PATCH] feat(shell): integrated additional coreutils into shell builtins - Added base64, checksum utilities (md5, sha1, sha224, sha256, sha384, sha512, b2sum), path utilities (basename, dirname), and text processing tools (cut, tee, tr, paste, comm) to the shell. - Registered new utility builtins in crates/pi-shell/src/coreutils.rs and crates/pi-shell/src/shell.rs. - Updated Cargo.toml to include the necessary dependencies for the new core utilities. --- Cargo.lock | 636 +++- crates/pi-natives/src/clipboard.rs | 2 +- crates/pi-natives/src/shell.rs | 5 +- crates/pi-shell/Cargo.toml | 18 + crates/pi-shell/src/coreutils.rs | 18 + crates/pi-shell/src/minimizer/engine.rs | 2 +- crates/pi-shell/src/shell.rs | 18 + crates/pi-uutils-ctx/src/lib.rs | 17 + crates/vendor/jaq/Cargo.toml | 26 + crates/vendor/jaq/LICENSE | 23 + crates/vendor/jaq/src/cli.rs | 236 ++ crates/vendor/jaq/src/filter.rs | 358 ++ crates/vendor/jaq/src/help.txt | 40 + crates/vendor/jaq/src/lib.rs | 333 ++ crates/vendor/jaq/src/read.rs | 89 + crates/vendor/jaq/src/tests.rs | 299 ++ crates/vendor/jaq/src/write.rs | 127 + crates/vendor/uu-b2sum/src/b2sum.rs | 35 +- crates/vendor/uu-base32/src/base32.rs | 67 +- crates/vendor/uu-base32/src/base_common.rs | 1578 +++++---- crates/vendor/uu-base64/src/base64.rs | 67 +- crates/vendor/uu-basename/src/basename.rs | 225 +- crates/vendor/uu-checksum-common/src/cli.rs | 346 +- .../vendor/uu-checksum-common/src/compute.rs | 460 ++- crates/vendor/uu-checksum-common/src/lib.rs | 328 +- .../vendor/uu-checksum-common/src/validate.rs | 1537 ++++---- crates/vendor/uu-comm/src/comm.rs | 311 +- crates/vendor/uu-cut/src/cut.rs | 1284 +++---- crates/vendor/uu-cut/src/matcher.rs | 157 +- crates/vendor/uu-cut/src/searcher.rs | 260 +- crates/vendor/uu-dirname/src/dirname.rs | 433 +-- crates/vendor/uu-paste/src/paste.rs | 181 +- crates/vendor/uu-sed/Cargo.toml | 26 + crates/vendor/uu-sed/LICENSE | 21 + crates/vendor/uu-sed/src/lib.rs | 246 ++ crates/vendor/uu-sed/src/sed/command.rs | 550 +++ crates/vendor/uu-sed/src/sed/compiler.rs | 3083 +++++++++++++++++ .../vendor/uu-sed/src/sed/delimited_parser.rs | 1315 +++++++ .../vendor/uu-sed/src/sed/error_handling.rs | 113 + crates/vendor/uu-sed/src/sed/fast_io.rs | 1436 ++++++++ crates/vendor/uu-sed/src/sed/fast_regex.rs | 703 ++++ crates/vendor/uu-sed/src/sed/in_place.rs | 320 ++ crates/vendor/uu-sed/src/sed/mod.rs | 431 +++ crates/vendor/uu-sed/src/sed/named_writer.rs | 95 + crates/vendor/uu-sed/src/sed/processor.rs | 818 +++++ .../uu-sed/src/sed/script_char_provider.rs | 139 + .../uu-sed/src/sed/script_line_provider.rs | 284 ++ crates/vendor/uu-sha1sum/src/sha1sum.rs | 5 +- crates/vendor/uu-sha224sum/src/sha224sum.rs | 5 +- crates/vendor/uu-sha256sum/src/sha256sum.rs | 5 +- crates/vendor/uu-sha384sum/src/sha384sum.rs | 5 +- crates/vendor/uu-sha512sum/src/sha512sum.rs | 5 +- crates/vendor/uu-tee/src/cli.rs | 51 +- crates/vendor/uu-tee/src/tee.rs | 44 +- crates/vendor/uu-tr/src/operation.rs | 1175 +++---- crates/vendor/uu-tr/src/simd.rs | 112 +- crates/vendor/uu-tr/src/tr.rs | 361 +- crates/vendor/uu-tr/src/unicode_table.rs | 8 +- crates/vendor/uu-xargs/Cargo.toml | 17 + crates/vendor/uu-xargs/LICENSE | 18 + crates/vendor/uu-xargs/src/lib.rs | 43 + crates/vendor/uu-xargs/src/tests.rs | 192 + crates/vendor/uu-xargs/src/xargs/mod.rs | 1369 ++++++++ packages/natives/CHANGELOG.md | 4 + packages/natives/native/index.d.ts | 20 - packages/natives/native/index.js | 1 - 66 files changed, 18167 insertions(+), 4369 deletions(-) create mode 100644 crates/vendor/jaq/Cargo.toml create mode 100644 crates/vendor/jaq/LICENSE create mode 100644 crates/vendor/jaq/src/cli.rs create mode 100644 crates/vendor/jaq/src/filter.rs create mode 100644 crates/vendor/jaq/src/help.txt create mode 100644 crates/vendor/jaq/src/lib.rs create mode 100644 crates/vendor/jaq/src/read.rs create mode 100644 crates/vendor/jaq/src/tests.rs create mode 100644 crates/vendor/jaq/src/write.rs create mode 100644 crates/vendor/uu-sed/Cargo.toml create mode 100644 crates/vendor/uu-sed/LICENSE create mode 100644 crates/vendor/uu-sed/src/lib.rs create mode 100644 crates/vendor/uu-sed/src/sed/command.rs create mode 100644 crates/vendor/uu-sed/src/sed/compiler.rs create mode 100644 crates/vendor/uu-sed/src/sed/delimited_parser.rs create mode 100644 crates/vendor/uu-sed/src/sed/error_handling.rs create mode 100644 crates/vendor/uu-sed/src/sed/fast_io.rs create mode 100644 crates/vendor/uu-sed/src/sed/fast_regex.rs create mode 100644 crates/vendor/uu-sed/src/sed/in_place.rs create mode 100644 crates/vendor/uu-sed/src/sed/mod.rs create mode 100644 crates/vendor/uu-sed/src/sed/named_writer.rs create mode 100644 crates/vendor/uu-sed/src/sed/processor.rs create mode 100644 crates/vendor/uu-sed/src/sed/script_char_provider.rs create mode 100644 crates/vendor/uu-sed/src/sed/script_line_provider.rs create mode 100644 crates/vendor/uu-xargs/Cargo.toml create mode 100644 crates/vendor/uu-xargs/LICENSE create mode 100644 crates/vendor/uu-xargs/src/lib.rs create mode 100644 crates/vendor/uu-xargs/src/tests.rs create mode 100644 crates/vendor/uu-xargs/src/xargs/mod.rs diff --git a/Cargo.lock b/Cargo.lock index bb6b3b451..a58d61314 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -152,6 +152,12 @@ dependencies = [ "nix 0.24.3", ] +[[package]] +name = "arrayref" +version = "0.3.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "76a2e8124351fda1ef8aaaa3bbd7ebbcb486bbcd4225aca0aa0d84bb2db8fecb" + [[package]] name = "arrayvec" version = "0.7.8" @@ -213,6 +219,16 @@ version = "0.22.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" +[[package]] +name = "base64-simd" +version = "0.8.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "339abbe78e73178762e23bea9dfd08e697eb3f3301cd4be981c0f78ba5859195" +dependencies = [ + "outref", + "vsimd", +] + [[package]] name = "bigdecimal" version = "0.4.10" @@ -283,6 +299,49 @@ dependencies = [ "wyz", ] +[[package]] +name = "blake2b_simd" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b79834656f71332577234b50bfc009996f7449e0c056884e6a02492ded0ca2f3" +dependencies = [ + "arrayref", + "arrayvec", + "constant_time_eq", +] + +[[package]] +name = "blake3" +version = "1.8.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0aa83c34e62843d924f905e0f5c866eb1dd6545fc4d719e803d9ba6030371fce" +dependencies = [ + "arrayref", + "arrayvec", + "cc", + "cfg-if", + "constant_time_eq", + "cpufeatures 0.3.0", +] + +[[package]] +name = "block-buffer" +version = "0.10.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71" +dependencies = [ + "generic-array", +] + +[[package]] +name = "block-buffer" +version = "0.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d2f6c7dbe95a6ed67ad9f18e57daf93a2f034c524b99fd2b76d18fdfeb6660aa" +dependencies = [ + "hybrid-array", +] + [[package]] name = "bon" version = "3.9.3" @@ -572,7 +631,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d524456ba66e72eb8b115ff89e01e497f8e6d11d78b70b1aa13c0fbd97540a81" dependencies = [ "cfg-if", - "cpufeatures", + "cpufeatures 0.3.0", "rand_core 0.10.1", ] @@ -658,6 +717,12 @@ dependencies = [ "error-code", ] +[[package]] +name = "codesnake" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2205f7f6d3de68ecf4c291c789b3edf07b6569268abd0188819086f71ae42225" + [[package]] name = "color-print" version = "0.3.7" @@ -719,6 +784,12 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "const-oid" +version = "0.10.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a6ef517f0926dd24a1582492c791b6a4818a4d94e789a334894aa15b0d12f55c" + [[package]] name = "const-random" version = "0.1.18" @@ -739,6 +810,12 @@ dependencies = [ "tiny-keccak", ] +[[package]] +name = "constant_time_eq" +version = "0.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3d52eff69cd5e647efe296129160853a42795992097e8af39800e1060caeea9b" + [[package]] name = "convert_case" version = "0.11.0" @@ -763,6 +840,15 @@ dependencies = [ "libm", ] +[[package]] +name = "cpufeatures" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280" +dependencies = [ + "libc", +] + [[package]] name = "cpufeatures" version = "0.3.0" @@ -772,6 +858,16 @@ dependencies = [ "libc", ] +[[package]] +name = "crc-fast" +version = "1.10.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e75b2483e97a5a7da73ac68a05b629f9c53cff58d8ed1c77866079e18b00dba5" +dependencies = [ + "digest 0.10.7", + "spin 0.10.0", +] + [[package]] name = "crc32fast" version = "1.5.0" @@ -812,6 +908,25 @@ version = "0.2.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5" +[[package]] +name = "crypto-common" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" +dependencies = [ + "generic-array", + "typenum", +] + +[[package]] +name = "crypto-common" +version = "0.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ce6e4c961d6cd6c9a86db418387425e8bdeaf05b3c8bc1411e6dca4c252f1453" +dependencies = [ + "hybrid-array", +] + [[package]] name = "ctor" version = "1.0.8" @@ -901,6 +1016,32 @@ dependencies = [ "parking_lot_core", ] +[[package]] +name = "data-encoding" +version = "2.11.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a4ae5f15dda3c708c0ade84bfee31ccab44a3da4f88015ed22f63732abe300c8" + +[[package]] +name = "data-encoding-macro" +version = "0.1.20" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3259c913752a86488b501ed8680446a5ed2d5aeac6e596cb23ba3800768ea32c" +dependencies = [ + "data-encoding", + "data-encoding-macro-internal", +] + +[[package]] +name = "data-encoding-macro-internal" +version = "0.1.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ccc2776f0c61eca1ca32528f85548abd1a4be8fb53d1b21c013e4f18da1e7090" +dependencies = [ + "data-encoding", + "syn", +] + [[package]] name = "defmt" version = "1.1.1" @@ -932,6 +1073,27 @@ dependencies = [ "thiserror 2.0.18", ] +[[package]] +name = "digest" +version = "0.10.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" +dependencies = [ + "block-buffer 0.10.4", + "crypto-common 0.1.7", +] + +[[package]] +name = "digest" +version = "0.11.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f1dd6dbb5841937940781866fa1281a1ff7bd3bf827091440879f9994983d5c2" +dependencies = [ + "block-buffer 0.12.1", + "const-oid", + "crypto-common 0.2.2", +] + [[package]] name = "dispatch2" version = "0.3.1" @@ -965,6 +1127,12 @@ version = "1.0.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "92773504d58c093f6de2459af4af33faa518c13451eb8f2b5698ed3d36e7c813" +[[package]] +name = "dyn-clone" +version = "1.0.20" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555" + [[package]] name = "either" version = "1.16.0" @@ -1050,6 +1218,17 @@ dependencies = [ "regex-syntax", ] +[[package]] +name = "fancy-regex" +version = "0.18.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e1e1dacd0d2082dfcf1351c4bdd566bbe89a2b263235a2b50058f1e130a47277" +dependencies = [ + "bit-set", + "regex-automata", + "regex-syntax", +] + [[package]] name = "fast-srgb8" version = "1.0.0" @@ -1185,7 +1364,7 @@ dependencies = [ "futures-core", "futures-sink", "nanorand", - "spin", + "spin 0.9.8", ] [[package]] @@ -1330,6 +1509,16 @@ dependencies = [ "slab", ] +[[package]] +name = "generic-array" +version = "0.14.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" +dependencies = [ + "typenum", + "version_check", +] + [[package]] name = "gethostname" version = "1.1.0" @@ -1542,6 +1731,12 @@ version = "0.4.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7f24254aa9a54b5c858eaee2f5bccdb46aaf0e486a595ed5fd8f86ba55232a70" +[[package]] +name = "hifijson" +version = "0.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0a7763b98ba8a24f59e698bf9ab197e7676c640d6455d1580b4ce7dc560f0f0d" + [[package]] name = "hostname" version = "0.4.2" @@ -1586,6 +1781,15 @@ dependencies = [ "markup5ever", ] +[[package]] +name = "hybrid-array" +version = "0.4.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "818356c5132c1fede50f837ca96afbe78ff42413047f4abb886217845e1b6c8c" +dependencies = [ + "typenum", +] + [[package]] name = "iana-time-zone" version = "0.1.65" @@ -2066,6 +2270,63 @@ version = "1.0.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" +[[package]] +name = "jaq" +version = "2.3.0" +dependencies = [ + "codesnake", + "hifijson", + "jaq-core", + "jaq-json", + "jaq-std", + "memmap2", + "parking_lot", + "pi-uutils-ctx", + "tempfile", + "unicode-width 0.1.14", + "yansi", +] + +[[package]] +name = "jaq-core" +version = "2.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "77526a72eb79412c29fd141767a6549bbfcb1cb40e00556fe16532d5e878e098" +dependencies = [ + "dyn-clone", + "once_cell", + "typed-arena", +] + +[[package]] +name = "jaq-json" +version = "1.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "01dbdbd07b076e8403abac68ce7744d93e2ecd953bbc44bf77bf00e1e81172bc" +dependencies = [ + "foldhash 0.1.5", + "hifijson", + "indexmap", + "jaq-core", + "jaq-std", +] + +[[package]] +name = "jaq-std" +version = "2.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2c264fe397c981705976c71f1bfe020382b9eda52ae950e57fe885e147bdd67d" +dependencies = [ + "aho-corasick", + "base64", + "chrono", + "jaq-core", + "libm", + "log", + "regex-lite", + "urlencoding", +] + [[package]] name = "jiff" version = "0.2.32" @@ -2140,6 +2401,15 @@ dependencies = [ "wasm-bindgen", ] +[[package]] +name = "keccak" +version = "0.1.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cb26cec98cce3a3d96cbb7bced3c4b16e3d13f27ec56dbd62cbc8f39cfb9d653" +dependencies = [ + "cpufeatures 0.2.17", +] + [[package]] name = "kqueue" version = "1.2.0" @@ -2260,6 +2530,16 @@ dependencies = [ "web_atoms", ] +[[package]] +name = "md-5" +version = "0.10.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d89e7ee0cfbedfc4da3340218492196241d89eefb6dab27de5df917a6d2e78cf" +dependencies = [ + "cfg-if", + "digest 0.10.7", +] + [[package]] name = "memchr" version = "2.8.3" @@ -2689,6 +2969,12 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "outref" +version = "0.5.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1a80800c0488c3a21695ea981a54918fbb37abf04f4d0720c453632255e2ff0e" + [[package]] name = "palette" version = "0.7.6" @@ -3101,6 +3387,7 @@ dependencies = [ "flume", "globset", "ignore", + "jaq", "libc", "os_pipe", "parking_lot", @@ -3114,17 +3401,34 @@ dependencies = [ "tokio", "tokio-util", "toml", + "uu_b2sum", + "uu_base64", + "uu_basename", "uu_cat", + "uu_comm", + "uu_cut", + "uu_dirname", "uu_find", "uu_head", "uu_ls", + "uu_md5sum", "uu_mkdir", "uu_mv", + "uu_paste", "uu_rm", + "uu_sed", + "uu_sha1sum", + "uu_sha224sum", + "uu_sha256sum", + "uu_sha384sum", + "uu_sha512sum", "uu_sort", "uu_tail", + "uu_tee", + "uu_tr", "uu_uniq", "uu_wc", + "uu_xargs", "windows-sys 0.61.2", "winreg 0.56.0", "xxhash-rust", @@ -3504,6 +3808,12 @@ dependencies = [ "regex-syntax", ] +[[package]] +name = "regex-lite" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cab834c73d247e67f4fae452806d17d3c7501756d98c8808d7c9c7aa7d18f973" + [[package]] name = "regex-syntax" version = "0.8.11" @@ -3713,6 +4023,38 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "sha1" +version = "0.10.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a978451301f4db1d02937a4ab3ccce137717b81826e79b7d49ffe3244a13c3b8" +dependencies = [ + "cfg-if", + "cpufeatures 0.2.17", + "digest 0.10.7", +] + +[[package]] +name = "sha2" +version = "0.10.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283" +dependencies = [ + "cfg-if", + "cpufeatures 0.2.17", + "digest 0.10.7", +] + +[[package]] +name = "sha3" +version = "0.10.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "77fd7028345d415a4034cf8777cd4f8ab1851274233b45f84e3d955502d93874" +dependencies = [ + "digest 0.10.7", + "keccak", +] + [[package]] name = "shared_library" version = "0.1.9" @@ -3778,6 +4120,15 @@ version = "0.4.12" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" +[[package]] +name = "sm3" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "da6a89ba31723d185fd7413b98c576a575f356d9b84729d8ecb6ead60000a5b6" +dependencies = [ + "digest 0.11.3", +] + [[package]] name = "smallvec" version = "1.15.2" @@ -3806,6 +4157,12 @@ dependencies = [ "lock_api", ] +[[package]] +name = "spin" +version = "0.10.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d5fe4ccb98d9c292d56fec89a5e07da7fc4cf0dc11e156b41793132775d3e591" + [[package]] name = "stable_deref_trait" version = "1.2.1" @@ -4802,6 +5159,18 @@ dependencies = [ "rustc-hash 2.1.3", ] +[[package]] +name = "typed-arena" +version = "2.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6af6ae20167a9ece4bcb41af5b80f8a1f1df981f6391189ce00fd257af04126a" + +[[package]] +name = "typenum" +version = "1.20.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" + [[package]] name = "ucd-trie" version = "0.1.7" @@ -4856,6 +5225,12 @@ version = "0.5.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "81e544489bf3d8ef66c953931f56617f423cd4b5494be343d9b9d3dda037b9a3" +[[package]] +name = "urlencoding" +version = "2.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "daf8dba3b7eb870caf1ddeed7bc9d2a049f3cfdfae7cb521b087cc33ae4c49da" + [[package]] name = "utf16_iter" version = "1.0.5" @@ -4883,6 +5258,44 @@ version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" +[[package]] +name = "uu_b2sum" +version = "0.8.0" +dependencies = [ + "clap", + "pi-uutils-ctx", + "uu_checksum_common", + "uucore 0.8.0", +] + +[[package]] +name = "uu_base32" +version = "0.8.0" +dependencies = [ + "clap", + "pi-uutils-ctx", + "uucore 0.8.0", +] + +[[package]] +name = "uu_base64" +version = "0.8.0" +dependencies = [ + "clap", + "pi-uutils-ctx", + "uu_base32", + "uucore 0.8.0", +] + +[[package]] +name = "uu_basename" +version = "0.8.0" +dependencies = [ + "clap", + "pi-uutils-ctx", + "uucore 0.8.0", +] + [[package]] name = "uu_cat" version = "0.8.0" @@ -4894,6 +5307,48 @@ dependencies = [ "uucore 0.8.0", ] +[[package]] +name = "uu_checksum_common" +version = "0.8.0" +dependencies = [ + "base64-simd", + "clap", + "hex", + "os_display", + "pi-uutils-ctx", + "uucore 0.8.0", +] + +[[package]] +name = "uu_comm" +version = "0.8.0" +dependencies = [ + "clap", + "pi-uutils-ctx", + "uucore 0.8.0", +] + +[[package]] +name = "uu_cut" +version = "0.8.0" +dependencies = [ + "bstr", + "clap", + "memchr", + "pi-uutils-ctx", + "uucore 0.8.0", +] + +[[package]] +name = "uu_dirname" +version = "0.8.0" +dependencies = [ + "clap", + "parking_lot", + "pi-uutils-ctx", + "uucore 0.8.0", +] + [[package]] name = "uu_find" version = "0.8.0" @@ -4939,6 +5394,15 @@ dependencies = [ "uutils_term_grid", ] +[[package]] +name = "uu_md5sum" +version = "0.8.0" +dependencies = [ + "clap", + "uu_checksum_common", + "uucore 0.8.0", +] + [[package]] name = "uu_mkdir" version = "0.8.0" @@ -4964,6 +5428,15 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "uu_paste" +version = "0.8.0" +dependencies = [ + "clap", + "pi-uutils-ctx", + "uucore 0.8.0", +] + [[package]] name = "uu_rm" version = "0.8.0" @@ -4977,6 +5450,71 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "uu_sed" +version = "0.1.1" +dependencies = [ + "clap", + "fancy-regex 0.18.0", + "memchr", + "memmap2", + "parking_lot", + "pi-uutils-ctx", + "regex", + "tempfile", + "uucore 0.9.0", +] + +[[package]] +name = "uu_sha1sum" +version = "0.8.0" +dependencies = [ + "clap", + "pi-uutils-ctx", + "uu_checksum_common", + "uucore 0.8.0", +] + +[[package]] +name = "uu_sha224sum" +version = "0.8.0" +dependencies = [ + "clap", + "pi-uutils-ctx", + "uu_checksum_common", + "uucore 0.8.0", +] + +[[package]] +name = "uu_sha256sum" +version = "0.8.0" +dependencies = [ + "clap", + "pi-uutils-ctx", + "uu_checksum_common", + "uucore 0.8.0", +] + +[[package]] +name = "uu_sha384sum" +version = "0.8.0" +dependencies = [ + "clap", + "pi-uutils-ctx", + "uu_checksum_common", + "uucore 0.8.0", +] + +[[package]] +name = "uu_sha512sum" +version = "0.8.0" +dependencies = [ + "clap", + "pi-uutils-ctx", + "uu_checksum_common", + "uucore 0.8.0", +] + [[package]] name = "uu_sort" version = "0.8.0" @@ -5014,6 +5552,26 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "uu_tee" +version = "0.8.0" +dependencies = [ + "clap", + "pi-uutils-ctx", + "uucore 0.8.0", +] + +[[package]] +name = "uu_tr" +version = "0.8.0" +dependencies = [ + "bytecount", + "clap", + "nom 8.0.0", + "pi-uutils-ctx", + "uucore 0.8.0", +] + [[package]] name = "uu_uniq" version = "0.8.0" @@ -5037,6 +5595,17 @@ dependencies = [ "uucore 0.8.0", ] +[[package]] +name = "uu_xargs" +version = "0.8.0" +dependencies = [ + "clap", + "libc", + "parking_lot", + "pi-uutils-ctx", + "tempfile", +] + [[package]] name = "uucore" version = "0.0.30" @@ -5065,14 +5634,22 @@ version = "0.8.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "07d779636d827cde4100f0e65ff3fd23b0b1f1195055475c6e6813d425f30c8e" dependencies = [ + "base64-simd", "bigdecimal", + "blake2b_simd", + "blake3", "bstr", "clap", + "crc-fast", + "data-encoding", + "data-encoding-macro", + "digest 0.10.7", "dunce", "fluent", "fluent-bundle", "fluent-syntax", "glob", + "hex", "icu_calendar", "icu_collator", "icu_datetime", @@ -5083,12 +5660,18 @@ dependencies = [ "jiff", "jiff-icu", "libc", + "md-5", + "memchr", "nix 0.31.3", "num-traits", "os_display", "procfs", "rustc-hash 2.1.3", "rustix", + "sha1", + "sha2", + "sha3", + "sm3", "thiserror 2.0.18", "unic-langid", "unit-prefix", @@ -5097,6 +5680,27 @@ dependencies = [ "winapi-util", "windows-sys 0.61.2", "xattr", + "z85", +] + +[[package]] +name = "uucore" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "069b34217c27f611e1589f540f58118dbf226a9e407d38ab472052ff075a1dc2" +dependencies = [ + "clap", + "fluent", + "fluent-syntax", + "libc", + "nix 0.31.3", + "os_display", + "rustc-hash 2.1.3", + "rustix", + "thiserror 2.0.18", + "unic-langid", + "uucore_procs 0.9.0", + "wild", ] [[package]] @@ -5120,6 +5724,16 @@ dependencies = [ "quote", ] +[[package]] +name = "uucore_procs" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "34337da211e7abfff7189b794afb3b5018fe356fb36474b73645458fc1201350" +dependencies = [ + "proc-macro2", + "quote", +] + [[package]] name = "uuhelp_parser" version = "0.0.30" @@ -5161,6 +5775,12 @@ version = "0.9.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" +[[package]] +name = "vsimd" +version = "0.8.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5c3082ca00d5a5ef149bb8b555a72ae84c9c59f7250f013ac822ac2e49b19c64" + [[package]] name = "walkdir" version = "2.5.0" @@ -5829,6 +6449,12 @@ dependencies = [ "linked-hash-map", ] +[[package]] +name = "yansi" +version = "1.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cfe53a6657fd280eaa890a3bc59152892ffa3e30101319d168b781ed6529b049" + [[package]] name = "yoke" version = "0.8.3" @@ -5852,6 +6478,12 @@ dependencies = [ "synstructure", ] +[[package]] +name = "z85" +version = "3.0.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c6e61e59a957b7ccee15d2049f86e8bfd6f66968fcd88f018950662d9b86e675" + [[package]] name = "zerocopy" version = "0.8.54" diff --git a/crates/pi-natives/src/clipboard.rs b/crates/pi-natives/src/clipboard.rs index c7b2251b2..fdf862c1e 100644 --- a/crates/pi-natives/src/clipboard.rs +++ b/crates/pi-natives/src/clipboard.rs @@ -273,7 +273,7 @@ mod tests { d } - /// `CF_DIBV5` as PixPin (Qt) places it, after arboard's + /// `CF_DIBV5` as `PixPin` (Qt) places it, after arboard's /// `maybe_tweak_header` rewrite: a 124-byte `BITMAPV5HEADER` carrying /// `BI_BITFIELDS` compression with the BGRA masks embedded in the header /// and pixels immediately after it. This is the exact buffer shape that diff --git a/crates/pi-natives/src/shell.rs b/crates/pi-natives/src/shell.rs index 74b58c42d..ccc0e07dd 100644 --- a/crates/pi-natives/src/shell.rs +++ b/crates/pi-natives/src/shell.rs @@ -12,8 +12,7 @@ use pi_shell::{ MinimizerResult as CoreMinimizerResult, Shell as CoreShell, ShellExecuteOptions as CoreShellExecuteOptions, ShellOptions as CoreShellOptions, ShellRunOptions as CoreShellRunOptions, ShellRunResult as CoreShellRunResult, - execute_shell as core_execute_shell, - minimizer, + execute_shell as core_execute_shell, minimizer, }; use crate::task; @@ -372,7 +371,7 @@ mod tests { /// the pre-fix bridge (`flume::unbounded` + fire-and-forget /// `ThreadsafeFunctionCallMode::NonBlocking`) the same harness accumulates /// the producer's entire surplus in the queue (measured: a 32 MiB stream - /// queued all 33_554_432 bytes while the consumer stalled). + /// queued all `33_554_432` bytes while the consumer stalled). #[tokio::test(flavor = "multi_thread")] async fn bridge_pump_bounds_queue_and_delivers_all_bytes() { const CHUNKS: usize = 512; diff --git a/crates/pi-shell/Cargo.toml b/crates/pi-shell/Cargo.toml index 9ed381ca7..adad30dd9 100644 --- a/crates/pi-shell/Cargo.toml +++ b/crates/pi-shell/Cargo.toml @@ -44,6 +44,24 @@ uu_find = { path = "../vendor/uu-find" } pi_uu_grep = { path = "../pi-uu-grep" } uu_cat = { path = "../vendor/uu-cat" } uu_uniq = { path = "../vendor/uu-uniq" } +uu_base64 = { path = "../vendor/uu-base64" } +uu_md5sum = { path = "../vendor/uu-md5sum" } +uu_sha1sum = { path = "../vendor/uu-sha1sum" } +uu_sha224sum = { path = "../vendor/uu-sha224sum" } +uu_sha256sum = { path = "../vendor/uu-sha256sum" } +uu_sha384sum = { path = "../vendor/uu-sha384sum" } +uu_sha512sum = { path = "../vendor/uu-sha512sum" } +uu_b2sum = { path = "../vendor/uu-b2sum" } +uu_basename = { path = "../vendor/uu-basename" } +uu_dirname = { path = "../vendor/uu-dirname" } +uu_cut = { path = "../vendor/uu-cut" } +uu_tee = { path = "../vendor/uu-tee" } +uu_tr = { path = "../vendor/uu-tr" } +uu_paste = { path = "../vendor/uu-paste" } +uu_comm = { path = "../vendor/uu-comm" } +uu_sed = { path = "../vendor/uu-sed" } +uu_xargs = { path = "../vendor/uu-xargs" } +jaq = { path = "../vendor/jaq" } [target.'cfg(unix)'.dependencies] libc.workspace = true diff --git a/crates/pi-shell/src/coreutils.rs b/crates/pi-shell/src/coreutils.rs index 10f1c56a2..a1372f404 100644 --- a/crates/pi-shell/src/coreutils.rs +++ b/crates/pi-shell/src/coreutils.rs @@ -209,6 +209,24 @@ uutil_builtin!(pub fn rm_builtin => uu_rm::run); uutil_builtin!(pub fn mv_builtin => uu_mv::run); uutil_builtin!(pub fn cat_builtin => uu_cat::run); uutil_builtin!(pub fn uniq_builtin => uu_uniq::run); +uutil_builtin!(pub fn base64_builtin => uu_base64::run); +uutil_builtin!(pub fn md5sum_builtin => uu_md5sum::run); +uutil_builtin!(pub fn sha1sum_builtin => uu_sha1sum::run); +uutil_builtin!(pub fn sha224sum_builtin => uu_sha224sum::run); +uutil_builtin!(pub fn sha256sum_builtin => uu_sha256sum::run); +uutil_builtin!(pub fn sha384sum_builtin => uu_sha384sum::run); +uutil_builtin!(pub fn sha512sum_builtin => uu_sha512sum::run); +uutil_builtin!(pub fn b2sum_builtin => uu_b2sum::run); +uutil_builtin!(pub fn basename_builtin => uu_basename::run); +uutil_builtin!(pub fn dirname_builtin => uu_dirname::run); +uutil_builtin!(pub fn cut_builtin => uu_cut::run); +uutil_builtin!(pub fn tee_builtin => uu_tee::run); +uutil_builtin!(pub fn tr_builtin => uu_tr::run); +uutil_builtin!(pub fn paste_builtin => uu_paste::run); +uutil_builtin!(pub fn comm_builtin => uu_comm::run); +uutil_builtin!(pub fn sed_builtin => uu_sed::run); +uutil_builtin!(pub fn xargs_builtin => uu_xargs::run); +uutil_builtin!(pub fn jq_builtin => jaq::run); #[cfg(test)] mod tests { diff --git a/crates/pi-shell/src/minimizer/engine.rs b/crates/pi-shell/src/minimizer/engine.rs index 4a3ab39c5..46b9b30ba 100644 --- a/crates/pi-shell/src/minimizer/engine.rs +++ b/crates/pi-shell/src/minimizer/engine.rs @@ -523,7 +523,7 @@ mod tests { }; static CONFIG_COUNTER: AtomicUsize = AtomicUsize::new(0); - pub(crate) static TEST_LOCK: parking_lot::Mutex<()> = parking_lot::Mutex::new(()); + pub static TEST_LOCK: parking_lot::Mutex<()> = parking_lot::Mutex::new(()); use super::*; use crate::minimizer::MinimizerOptions; diff --git a/crates/pi-shell/src/shell.rs b/crates/pi-shell/src/shell.rs index b3ebe92b8..289c72fea 100644 --- a/crates/pi-shell/src/shell.rs +++ b/crates/pi-shell/src/shell.rs @@ -622,6 +622,24 @@ async fn create_session_for_run( shell.register_builtin("fd", crate::fd::fd_builtin()); shell.register_builtin("cat", crate::coreutils::cat_builtin()); shell.register_builtin("uniq", crate::coreutils::uniq_builtin()); + shell.register_builtin("base64", crate::coreutils::base64_builtin()); + shell.register_builtin("md5sum", crate::coreutils::md5sum_builtin()); + shell.register_builtin("sha1sum", crate::coreutils::sha1sum_builtin()); + shell.register_builtin("sha224sum", crate::coreutils::sha224sum_builtin()); + shell.register_builtin("sha256sum", crate::coreutils::sha256sum_builtin()); + shell.register_builtin("sha384sum", crate::coreutils::sha384sum_builtin()); + shell.register_builtin("sha512sum", crate::coreutils::sha512sum_builtin()); + shell.register_builtin("b2sum", crate::coreutils::b2sum_builtin()); + shell.register_builtin("basename", crate::coreutils::basename_builtin()); + shell.register_builtin("dirname", crate::coreutils::dirname_builtin()); + shell.register_builtin("cut", crate::coreutils::cut_builtin()); + shell.register_builtin("tee", crate::coreutils::tee_builtin()); + shell.register_builtin("tr", crate::coreutils::tr_builtin()); + shell.register_builtin("paste", crate::coreutils::paste_builtin()); + shell.register_builtin("comm", crate::coreutils::comm_builtin()); + shell.register_builtin("sed", crate::coreutils::sed_builtin()); + shell.register_builtin("xargs", crate::coreutils::xargs_builtin()); + shell.register_builtin("jq", crate::coreutils::jq_builtin()); if !uutils_env_disabled(config, "PI_DISABLE_UUTILS_DESTRUCTIVE") { if !uutils_env_disabled(config, "PI_DISABLE_RM_BUILTIN") { shell.register_builtin("rm", crate::coreutils::rm_builtin()); diff --git a/crates/pi-uutils-ctx/src/lib.rs b/crates/pi-uutils-ctx/src/lib.rs index 05174f86e..09d258e64 100644 --- a/crates/pi-uutils-ctx/src/lib.rs +++ b/crates/pi-uutils-ctx/src/lib.rs @@ -188,6 +188,23 @@ pub fn var(key: &str) -> Option { .and_then(|ctx| ctx.env.get(key).cloned()) }) } + +/// Returns a snapshot of the scope's entire environment map (the shell's +/// exported variables), or an empty vector when no scope is installed. +/// Utilities that spawn child processes use this to build the child +/// environment (`env_clear().envs(..)`), because the shell's exported +/// variables are not present in the host process environment. +#[must_use] +pub fn env_snapshot() -> Vec<(String, String)> { + CTX.with(|c| { + c.borrow().as_ref().map_or_else(Vec::new, |ctx| { + ctx.env + .iter() + .map(|(k, v)| (k.clone(), v.clone())) + .collect() + }) + }) +} /// Returns true when scoped stdin is a shell pipe or custom stream that should /// be treated as `rg PATTERN`'s implicit input instead of searching `.`. #[must_use] diff --git a/crates/vendor/jaq/Cargo.toml b/crates/vendor/jaq/Cargo.toml new file mode 100644 index 000000000..4e214bf3b --- /dev/null +++ b/crates/vendor/jaq/Cargo.toml @@ -0,0 +1,26 @@ +[package] +name = "jaq" +version = "2.3.0" +edition = "2024" +license = "MIT" + +[lib] +path = "src/lib.rs" + +[dependencies] +# Interpreter libraries from crates.io, pinned as upstream jaq v2.3.0 pins them. +jaq-core = "2.1.1" +jaq-std = "2.1.0" +jaq-json = "1.1.1" + +codesnake = "0.2" +hifijson = "0.2.0" +memmap2 = "0.9" +tempfile = "3.3.0" +unicode-width = "0.1.13" +yansi = "1.0.1" +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[dev-dependencies] +tempfile = "3" +parking_lot = "0.12" diff --git a/crates/vendor/jaq/LICENSE b/crates/vendor/jaq/LICENSE new file mode 100644 index 000000000..31aa79387 --- /dev/null +++ b/crates/vendor/jaq/LICENSE @@ -0,0 +1,23 @@ +Permission is hereby granted, free of charge, to any +person obtaining a copy of this software and associated +documentation files (the "Software"), to deal in the +Software without restriction, including without +limitation the rights to use, copy, modify, merge, +publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software +is furnished to do so, subject to the following +conditions: + +The above copyright notice and this permission notice +shall be included in all copies or substantial portions +of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF +ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED +TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A +PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT +SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY +CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION +OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR +IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER +DEALINGS IN THE SOFTWARE. diff --git a/crates/vendor/jaq/src/cli.rs b/crates/vendor/jaq/src/cli.rs new file mode 100644 index 000000000..dc5d70f67 --- /dev/null +++ b/crates/vendor/jaq/src/cli.rs @@ -0,0 +1,236 @@ +//! Command-line argument parsing +use core::fmt; +use std::{ffi::OsString, path::PathBuf}; + +/// Remaining arguments; upstream used `std::env::ArgsOs`, but as an in-process +/// builtin the argv comes from the host, not the process. +type Args = std::vec::IntoIter; + +#[derive(Debug, Default)] +pub struct Cli { + // Input options + pub null_input: bool, + /// When the option `--slurp` is used additionally, + /// then the whole input is read into a single string. + pub raw_input: bool, + /// When input is read from files, + /// jaq yields an array for each file, whereas + /// jq produces only a single array. + pub slurp: bool, + + // Output options + pub compact_output: bool, + pub raw_output: bool, + /// This flag enables `--raw-output`. + pub join_output: bool, + pub in_place: bool, + pub sort_keys: bool, + pub color_output: bool, + pub monochrome_output: bool, + pub tab: bool, + pub indent: usize, + + // Compilation options + pub from_file: bool, + /// If this option is given multiple times, all given directories are + /// searched. + pub library_path: Vec, + + // Key-value options + pub arg: Vec<(String, String)>, + pub argjson: Vec<(String, String)>, + pub slurpfile: Vec<(String, OsString)>, + pub rawfile: Vec<(String, OsString)>, + + // Positional arguments + /// If this argument is not given, it is assumed to be `.`, the identity + /// filter. + pub filter: Option, + pub files: Vec, + pub args: Vec, + //pub jsonargs: Vec, + pub run_tests: Option>, + /// If there is some last output value `v`, + /// then the exit status code is + /// 1 if `v < true` (that is, if `v` is `false` or `null`) and + /// 0 otherwise. + /// If there is no output value, then the exit status code is 4. + /// + /// If any error occurs, then this option has no effect. + pub exit_status: bool, + pub version: bool, + pub help: bool, +} + +#[derive(Debug)] +pub enum Filter { + Inline(String), + FromFile(PathBuf), +} + +impl Cli { + fn positional(&mut self, mode: &Mode, arg: OsString) -> Result<(), Error> { + if self.filter.is_none() { + self.filter = Some(if self.from_file { + Filter::FromFile(arg.into()) + } else { + Filter::Inline(arg.into_string()?) + }) + } else { + match mode { + Mode::Files => self.files.push(arg.into()), + Mode::Args => self.args.push(arg.into_string()?), + //Mode::JsonArgs => self.jsonargs.push(arg.into_string()?), + } + } + Ok(()) + } + + fn long(&mut self, mode: &mut Mode, arg: &str, args: &mut Args) -> Result<(), Error> { + let int = |s: OsString| s.into_string().ok()?.parse().ok(); + match arg { + // handle all arguments after "--" + "" => args.try_for_each(|arg| self.positional(mode, arg))?, + + "null-input" => self.short('n', args)?, + "raw-input" => self.short('R', args)?, + "slurp" => self.short('s', args)?, + + "compact-output" => self.short('c', args)?, + "raw-output" => self.short('r', args)?, + "join-output" => self.short('j', args)?, + "in-place" => self.short('i', args)?, + "sort-keys" => self.short('S', args)?, + "color-output" => self.short('C', args)?, + "monochrome-output" => self.short('M', args)?, + "tab" => self.tab = true, + "indent" => self.indent = args.next().and_then(int).ok_or(Error::Int("--indent"))?, + "from-file" => self.short('f', args)?, + "library-path" => self.short('L', args)?, + "arg" => { + let (name, value) = parse_key_val("--arg", args)?; + self.arg.push((name, value.into_string()?)); + }, + "argjson" => { + let (name, value) = parse_key_val("--argjson", args)?; + self.argjson.push((name, value.into_string()?)); + }, + "slurpfile" => self.slurpfile.push(parse_key_val("--slurpfile", args)?), + "rawfile" => self.rawfile.push(parse_key_val("--rawfile", args)?), + + "args" => *mode = Mode::Args, + //"jsonargs" => *mode = Mode::JsonArgs, + "run-tests" => self.run_tests = Some(args.map(PathBuf::from).collect()), + "exit-status" => self.short('e', args)?, + "version" => self.short('V', args)?, + "help" => self.short('h', args)?, + + arg => Err(Error::Flag(format!("--{arg}")))?, + } + Ok(()) + } + + fn short(&mut self, arg: char, args: &mut Args) -> Result<(), Error> { + match arg { + 'n' => self.null_input = true, + 'R' => self.raw_input = true, + 's' => self.slurp = true, + + 'c' => self.compact_output = true, + 'r' => self.raw_output = true, + 'j' => self.join_output = true, + 'i' => self.in_place = true, + 'S' => self.sort_keys = true, + 'C' => self.color_output = true, + 'M' => self.monochrome_output = true, + + 'f' => self.from_file = true, + // resolve -L directories against the shell's cwd here; module + // loading happens inside unpatched jaq-core + 'L' => self + .library_path + .push(pi_uutils_ctx::resolve(args.next().ok_or(Error::Path("-L"))?)), + 'e' => self.exit_status = true, + 'V' => self.version = true, + 'h' => self.help = true, + arg => Err(Error::Flag(format!("-{arg}")))?, + } + Ok(()) + } + + pub fn parse(argv: Vec) -> Result { + let mut cli = Self { indent: 2, ..Self::default() }; + let mut mode = Mode::Files; + let mut args = argv.into_iter(); + args.next(); // skip the command name (argv[0]) + while let Some(arg) = args.next() { + match arg.to_str() { + // we've got a valid UTF-8 argument here + Some(s) => match s.strip_prefix("--") { + Some(rest) => cli.long(&mut mode, rest, &mut args)?, + None => match s.strip_prefix("-") { + Some(rest) => rest.chars().try_for_each(|c| cli.short(c, &mut args))?, + None => cli.positional(&mode, arg)?, + }, + }, + // we've got invalid UTF-8, so it is no valid flag + // note that we do not check here whether arg starts with `-`, + // because this seems to be quite difficult to do in a portable way + None => cli.positional(&mode, arg)?, + } + } + Ok(cli) + } + + pub fn color_if(&self, f: impl Fn() -> bool) -> bool { + if self.monochrome_output { + false + } else if self.color_output { + true + } else { + f() + } + } +} + +#[derive(Debug)] +pub enum Error { + Flag(String), + Utf8(OsString), + KeyValue(&'static str), + Int(&'static str), + Path(&'static str), +} + +impl fmt::Display for Error { + fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { + match self { + Self::Flag(s) => write!(f, "unknown flag: {s}"), + Self::Utf8(s) => write!(f, "invalid UTF-8: {s:?}"), + Self::KeyValue(o) => write!(f, "{o} expects a key and a value"), + Self::Int(o) => write!(f, "{o} expects an integer"), + Self::Path(o) => write!(f, "{o} expects a path"), + } + } +} + +/// Conversion of errors from [`OsString::into_string`]. +impl From for Error { + fn from(e: OsString) -> Self { + Self::Utf8(e) + } +} + +fn parse_key_val(arg: &'static str, args: &mut Args) -> Result<(String, OsString), Error> { + let err = || Error::KeyValue(arg); + let key = args.next().ok_or_else(err)?.into_string()?; + let val = args.next().ok_or_else(err)?; + Ok((key, val)) +} + +/// Interpretation of positional arguments. +enum Mode { + Args, + //JsonArgs, + Files, +} diff --git a/crates/vendor/jaq/src/filter.rs b/crates/vendor/jaq/src/filter.rs new file mode 100644 index 000000000..e9b4c2889 --- /dev/null +++ b/crates/vendor/jaq/src/filter.rs @@ -0,0 +1,358 @@ +//! Filter parsing, compilation, and execution. +use core::{ + cell::Cell, + fmt::{self, Display, Formatter}, +}; +use std::{ + io::{self, Write}, + path::PathBuf, +}; + +use jaq_core::{ + Ctx, Error as CoreError, Exn, Native, RcIter, RunPtr, UpdatePtr, ValT, compile, load, +}; + +use crate::{Cli, Error, Val, read}; + +pub type Filter = jaq_core::Filter>; + +thread_local! { + /// Exit code requested by `halt`/`halt_error` in the current invocation. + /// The overridden natives set this instead of `std::process::exit` and + /// abort the run with a sentinel error; the entry point checks it first. + static HALT: Cell> = const { Cell::new(None) }; +} + +/// Takes (and clears) the exit code requested by `halt`/`halt_error`. +pub fn take_halt() -> Option { + HALT.with(Cell::take) +} + +/// Replacements for jaq-std natives that are unsound inside a long-lived host +/// process. Prepended before `jaq_std::funs()`: the compiler resolves native +/// calls by first match, so these shadow the crates.io implementations. +/// +/// - `env`: reads the shell's exported environment, not the host process's. +/// - `halt`/`halt_error`: record the exit code and abort the run with a +/// sentinel error instead of `std::process::exit`, which would kill the +/// shell. +/// - `debug`/`stderr`: write to the ctx stderr stream directly instead of going +/// through the process-global `log` facade (whose single global logger may +/// belong to the host). +fn overrides() -> impl Iterator>> { + use jaq_core::box_iter::box_once; + use jaq_std::ValT as _; + + fn halt_with<'a>(code: i32, sentinel: &'static str) -> jaq_core::ValXs<'a, Val> { + HALT.with(|h| h.set(Some(code))); + box_once(Err(Exn::from(CoreError::str(sentinel)))) + } + + fn debug_msg(v: &Val) { + // upstream format: env_logger renders `["DEBUG:", ]\n` + let _ = writeln!(pi_uutils_ctx::stderr(), "[\"DEBUG:\", {v}]"); + } + + fn stderr_msg(v: &Val) { + // like jq, print strings raw and everything else as JSON, no newline + if let Some(s) = v.as_str() { + let _ = write!(pi_uutils_ctx::stderr(), "{s}"); + } else { + let _ = write!(pi_uutils_ctx::stderr(), "{v}"); + } + } + + let run_funs: [jaq_std::Filter>; 3] = [ + ("env", jaq_std::v(0), |_, _| { + let env = pi_uutils_ctx::env_snapshot() + .into_iter() + .map(|(k, v)| (k.into(), Val::from(v))); + box_once(Ok(Val::obj(env.collect()))) + }), + ("halt", jaq_std::v(0), |_, _| halt_with(0, "halt")), + ("halt_error", jaq_std::v(1), |_, mut cv| { + match cv.0.pop_var().as_isize() { + Some(code) => { + // upstream prints the input to stdout: raw for strings + // (no trailing newline), JSON + newline otherwise + if let Some(s) = cv.1.as_str() { + let _ = write!(pi_uutils_ctx::stdout(), "{s}"); + } else { + let _ = writeln!(pi_uutils_ctx::stdout(), "{}", cv.1); + } + halt_with(code as i32, "halt_error") + }, + None => box_once(Err(Exn::from(CoreError::typ(cv.1, "integer")))), + } + }), + ]; + + // `debug` and `stderr` are identity filters with an output effect; they + // need an update pointer so `debug |= f` keeps working. + let upd_funs: [jaq_std::Filter<(RunPtr, UpdatePtr)>; 2] = [ + ( + "debug", + jaq_std::v(0), + ( + |_, cv| { + debug_msg(&cv.1); + box_once(Ok(cv.1)) + }, + |_, cv, f| { + debug_msg(&cv.1); + f(cv.1) + }, + ), + ), + ( + "stderr", + jaq_std::v(0), + ( + |_, cv| { + stderr_msg(&cv.1); + box_once(Ok(cv.1)) + }, + |_, cv, f| { + stderr_msg(&cv.1); + f(cv.1) + }, + ), + ), + ]; + + let upd = |(name, arity, (run, update)): jaq_std::Filter<(RunPtr, UpdatePtr)>| { + (name, arity, Native::new(run).with_update(update)) + }; + let run_funs = run_funs.into_iter().map(jaq_std::run); + run_funs.chain(upd_funs.into_iter().map(upd)) +} + +pub fn parse_compile( + path: &PathBuf, + code: &str, + vars: &[String], + paths: &[PathBuf], +) -> Result<(Vec, Filter), Vec> { + use compile::Compiler; + use load::{Arena, File, Loader, import}; + + let default = ["~/.jq", "$ORIGIN/../lib/jq", "$ORIGIN/../lib"].map(|x| x.into()); + let paths = if paths.is_empty() { &default } else { paths }; + + let vars: Vec<_> = vars.iter().map(|v| format!("${v}")).collect(); + let arena = Arena::default(); + let defs = jaq_std::defs().chain(jaq_json::defs()); + let loader = Loader::new(defs).with_std_read(paths); + let path = path.into(); + let modules = loader + .load(&arena, File { path, code }) + .map_err(load_errors)?; + + let mut vals = Vec::new(); + import(&modules, |p| { + let path = p.find(paths, "json")?; + vals.push(read::json_array(path).map_err(|e| e.to_string())?); + Ok(()) + }) + .map_err(load_errors)?; + + // overrides first: native lookup is first-match-wins + let funs = overrides().chain(jaq_std::funs()).chain(jaq_json::funs()); + let compiler = Compiler::default() + .with_funs(funs) + .with_global_vars(vars.iter().map(|v| &**v)); + let filter = compiler.compile(modules).map_err(compile_errors)?; + Ok((vals, filter)) +} + +/// Run a filter with given input values and run `f` for every value output. +/// +/// This function cannot return an `Iterator` because it creates an `RcIter`. +/// This is most unfortunate. We should think about how to simplify this ... +pub(crate) fn run( + cli: &Cli, + filter: &Filter, + vars: Vec, + iter: impl Iterator>, + mut f: impl FnMut(Val) -> io::Result<()>, +) -> Result, Error> { + let mut last = None; + let iter = iter.map(|r| r.map_err(|e| e.to_string())); + + let iter = Box::new(iter) as Box>; + let null = Box::new(core::iter::once(Ok(Val::Null))) as Box>; + + let iter = RcIter::new(iter); + let null = RcIter::new(null); + + let ctx = Ctx::new(vars, &iter); + + for item in if cli.null_input { &null } else { &iter } { + // host abort/timeout: stdin reads observe the cancel flag themselves, + // but file/slurped inputs and long-running filters do not + if pi_uutils_ctx::is_cancelled() { + break; + } + let input = item.map_err(Error::Parse)?; + for output in filter.run((ctx.clone(), input)) { + if pi_uutils_ctx::is_cancelled() { + return Ok(last); + } + let output = output.map_err(Error::Jaq)?; + last = Some(output.as_bool()); + f(output)?; + } + } + Ok(last) +} + +#[derive(Debug)] +pub struct FileReports(load::File, Vec); + +impl Display for FileReports { + fn fmt(&self, f: &mut Formatter) -> fmt::Result { + let Self(file, reports) = self; + let idx = codesnake::LineIndex::new(&file.code); + reports.iter().try_for_each(|e| { + writeln!(f, "Error: {}", e.message)?; + let block = e.to_block(&idx); + writeln!(f, "{}[{}]", block.prologue(), file.path.display())?; + writeln!(f, "{}{}", block, block.epilogue()) + }) + } +} + +fn load_errors(errs: load::Errors<&str, PathBuf>) -> Vec { + use load::Error; + + let errs = errs.into_iter().map(|(file, err)| { + let code = file.code; + let err = match err { + Error::Io(errs) => errs.into_iter().map(|e| report_io(code, e)).collect(), + Error::Lex(errs) => errs.into_iter().map(|e| report_lex(code, e)).collect(), + Error::Parse(errs) => errs.into_iter().map(|e| report_parse(code, e)).collect(), + }; + FileReports(file.map_code(|s| s.into()), err) + }); + errs.collect() +} + +fn compile_errors(errs: compile::Errors<&str, PathBuf>) -> Vec { + let errs = errs.into_iter().map(|(file, errs)| { + let code = file.code; + let errs = errs.into_iter().map(|e| report_compile(code, e)).collect(); + FileReports(file.map_code(|s| s.into()), errs) + }); + errs.collect() +} + +type StringColors = Vec<(String, Option)>; + +#[derive(Debug)] +struct Report { + message: String, + labels: Vec<(core::ops::Range, StringColors, Color)>, +} + +#[derive(Clone, Debug)] +enum Color { + Yellow, + Red, +} + +impl Color { + fn apply(&self, d: impl Display) -> String { + use yansi::{Color, Paint}; + let color = match self { + Self::Yellow => Color::Yellow, + Self::Red => Color::Red, + }; + d.fg(color).to_string() + } +} + +fn report_io(code: &str, (path, error): (&str, String)) -> Report { + let path_range = load::span(code, path); + Report { + message: format!("could not load file {}: {}", path, error), + labels: [(path_range, [(error, None)].into(), Color::Red)].into(), + } +} + +fn report_lex(code: &str, (expected, found): load::lex::Error<&str>) -> Report { + // truncate found string to its first character + let found = &found[..found.char_indices().nth(1).map_or(found.len(), |(i, _)| i)]; + + let found_range = load::span(code, found); + let found = match found { + "" => [("unexpected end of input".to_string(), None)].into(), + c => [("unexpected character ", None), (c, Some(Color::Red))] + .map(|(s, c)| (s.into(), c)) + .into(), + }; + let label = (found_range, found, Color::Red); + + let labels = match expected { + load::lex::Expect::Delim(open) => { + let text = [("unclosed delimiter ", None), (open, Some(Color::Yellow))] + .map(|(s, c)| (s.into(), c)); + Vec::from([(load::span(code, open), text.into(), Color::Yellow), label]) + }, + _ => Vec::from([label]), + }; + + Report { message: format!("expected {}", expected.as_str()), labels } +} + +fn report_parse(code: &str, (expected, found): load::parse::Error<&str>) -> Report { + let found_range = load::span(code, found); + + let found = if found.is_empty() { + "unexpected end of input" + } else { + "unexpected token" + }; + let found = [(found.to_string(), None)].into(); + + Report { + message: format!("expected {}", expected.as_str()), + labels: Vec::from([(found_range, found, Color::Red)]), + } +} + +fn report_compile(code: &str, (found, undefined): compile::Error<&str>) -> Report { + use compile::Undefined::Filter; + let found_range = load::span(code, found); + let wnoa = |exp, got| format!("wrong number of arguments (expected {exp}, found {got})"); + let message = match (found, undefined) { + ("reduce", Filter(arity)) => wnoa("2", arity), + ("foreach", Filter(arity)) => wnoa("2 or 3", arity), + (_, undefined) => format!("undefined {}", undefined.as_str()), + }; + let found = [(message.clone(), None)].into(); + + Report { message, labels: Vec::from([(found_range, found, Color::Red)]) } +} + +type CodeBlock = codesnake::Block, String>; + +impl Report { + fn to_block(&self, idx: &codesnake::LineIndex) -> CodeBlock { + use codesnake::{Block, CodeWidth, Label}; + let color_maybe = |(text, color): (_, Option)| match color { + None => text, + Some(color) => color.apply(text).to_string(), + }; + let labels = self.labels.iter().cloned().map(|(range, text, color)| { + let text = text.into_iter().map(color_maybe).collect::>(); + Label::new(range) + .with_text(text.join("")) + .with_style(move |s| color.apply(s).to_string()) + }); + Block::new(idx, labels).unwrap().map_code(|c| { + let c = c.replace('\t', " "); + let w = unicode_width::UnicodeWidthStr::width(&*c); + CodeWidth::new(c, core::cmp::max(w, 1)) + }) + } +} diff --git a/crates/vendor/jaq/src/help.txt b/crates/vendor/jaq/src/help.txt new file mode 100644 index 000000000..e75cd7e86 --- /dev/null +++ b/crates/vendor/jaq/src/help.txt @@ -0,0 +1,40 @@ +Just Another Query Tool + +Usage: jaq [OPTION]... [FILTER] [ARG]... + +Arguments: + [FILTER] Filter to execute + [ARG]... Positional arguments, by default used as input files + +Input options: + -n, --null-input Use null as single input value + -R, --raw-input Read lines of the input as sequence of strings + -s, --slurp Read (slurp) all input values into one array + +Output options: + -c, --compact-output Print JSON compactly, omitting whitespace + -r, --raw-output Write strings without escaping them with quotes + -j, --join-output Do not print a newline after each value + -i, --in-place Overwrite input file with its output + -S, --sort-keys Print objects sorted by their keys + -C, --color-output Always color output + -M, --monochrome-output Do not color output + --tab Use tabs for indentation rather than spaces + --indent Use N spaces for indentation [default: 2] + +Compilation options: + -f, --from-file Read filter from a file given by filter argument + -L, --library-path Search for modules and data in given directory + +Variable options: + --arg Set variable `$A` to string `V` + --argjson Set variable `$A` to JSON value `V` + --slurpfile Set variable `$A` to array containing the JSON values in file `F` + --rawfile Set variable `$A` to string containing the contents of file `F` + --args Collect remaining positional arguments into `$ARGS.positional` + +Remaining options: + --run-tests Run tests from a file + -e, --exit-status Use the last output value as exit status code + -V, --version Print version + -h, --help Print help diff --git a/crates/vendor/jaq/src/lib.rs b/crates/vendor/jaq/src/lib.rs new file mode 100644 index 000000000..761ab4771 --- /dev/null +++ b/crates/vendor/jaq/src/lib.rs @@ -0,0 +1,333 @@ +//! Vendored, patched `jaq` CLI (jq-compatible JSON processor), wired to run +//! in-process as a shell builtin via [`pi_uutils_ctx`]. +//! +//! Upstream: , tag `v2.3.0`, +//! commit `0ce6e86a5e038a623dc894ad5cc70aaa9142daf2` (MIT). +//! +//! Only the CLI crate (`jaq/`) is vendored; the interpreter libraries +//! (`jaq-core`, `jaq-std`, `jaq-json`) come from crates.io. Patches vs +//! upstream: +//! - `main()` is restructured as [`run`], returning the exit code instead of +//! `ExitCode`/`Termination`; no `std::process::exit` anywhere. +//! - stdio goes through the [`pi_uutils_ctx`] streams; every file path operand +//! resolves through `pi_uutils_ctx::resolve` against the shell's cwd. +//! - The ctx streams are never a tty, so `--color` auto mode always resolves to +//! plain output; `-C/--color-output` still forces ANSI. Color state is +//! thread-local (see `color` in this module) instead of yansi's global +//! enable/disable, so concurrent invocations don't race. +//! - The mimalloc global allocator, env_logger, and the rustyline `repl` filter +//! are stripped (binary-only / interactive-only). +//! - jaq-std's `env`, `halt`, `halt_error`, `debug`, and `stderr` natives are +//! shadowed (first-match-wins in the compiler's native table) because the +//! crates.io implementations call `std::process::exit`, read the host process +//! environment, or log through the global `log` facade. See +//! `filter::overrides`. + +mod cli; +mod filter; +mod read; +mod write; + +use core::fmt::{self, Display, Formatter}; +use std::{ + io::{self, BufRead, Write}, + path::PathBuf, +}; + +use cli::Cli; +use filter::{FileReports, Filter}; +use jaq_core::{Ctx, RcIter, load}; +use jaq_json::Val; +use write::{print, with_stdout}; + +/// In-process builtin entry point. The host installs a [`pi_uutils_ctx`] scope +/// (stdio + working directory + environment) on a dedicated blocking thread, +/// then calls this with `argv[0]` = command name (`jq`). +pub fn run(argv: Vec) -> i32 { + color::init(); + color::set(false); + filter::take_halt(); // clear leftover state from a prior scope on this thread + + let cli = match Cli::parse(argv) { + Ok(cli) => cli, + Err(e) => { + let _ = writeln!(pi_uutils_ctx::stderr(), "Error: {e}"); + return 2; + }, + }; + + if cli.version { + let _ = writeln!( + pi_uutils_ctx::stdout(), + "{} {}", + env!("CARGO_PKG_NAME"), + env!("CARGO_PKG_VERSION") + ); + return 0; + } else if cli.help { + let _ = writeln!(pi_uutils_ctx::stdout(), "{}", include_str!("help.txt")); + return 0; + } + + // Upstream enables color when stdout is a terminal and NO_COLOR is unset. + // The ctx streams are never a terminal, so auto mode is always plain; + // only -C/--color-output (minus -M) forces ANSI. + color::set(!cli.in_place && cli.color_if(|| false)); + + let res = real_main(&cli); + // `halt`/`halt_error` abort the filter run with a sentinel error; the + // requested exit code wins over the error path below. + if let Some(code) = filter::take_halt() { + return code; + } + match res { + Ok(exit) => exit, + Err(e) => { + color::set(cli.color_if(|| false)); + let _ = write!(pi_uutils_ctx::stderr(), "{e}"); + e.report() + }, + } +} + +/// Thread-local color toggle backing yansi's process-global condition. +/// +/// `yansi::enable`/`disable` flip process-global state, which races when +/// several shell pipeline stages run jaq concurrently on different threads. +/// Instead, a process-global yansi condition (installed once) reads this +/// thread-local flag, giving each invocation its own color state. +mod color { + use std::{cell::Cell, sync::Once}; + + thread_local! { + static COLOR: Cell = const { Cell::new(false) }; + } + + pub fn init() { + static ONCE: Once = Once::new(); + ONCE.call_once(|| yansi::whenever(yansi::Condition(|| COLOR.with(Cell::get)))); + } + + pub fn set(on: bool) { + COLOR.with(|c| c.set(on)); + } +} + +fn real_main(cli: &Cli) -> Result { + if let Some(test_files) = &cli.run_tests { + return Ok(match test_files.last() { + Some(file) => { + run_tests(io::BufReader::new(std::fs::File::open(pi_uutils_ctx::resolve(file))?)) + }, + None => run_tests(io::BufReader::new(pi_uutils_ctx::stdin())), + }); + } + + let (vars, mut ctx): (Vec, Vec) = binds(cli)?.into_iter().unzip(); + + let (vals, filter) = match &cli.filter { + None => (Vec::new(), Filter::default()), + Some(filter) => { + let (path, code) = match filter { + cli::Filter::FromFile(path) => { + (path.into(), std::fs::read_to_string(pi_uutils_ctx::resolve(path))?) + }, + cli::Filter::Inline(filter) => ("".into(), filter.clone()), + }; + filter::parse_compile(&path, &code, &vars, &cli.library_path).map_err(Error::Report)? + }, + }; + ctx.extend(vals); + + let last = if cli.files.is_empty() { + let inputs = read::buffered(cli, io::BufReader::new(pi_uutils_ctx::stdin())); + with_stdout(|out| filter::run(cli, &filter, ctx, inputs, |v| print(out, cli, &v)))? + } else { + let mut last = None; + for file in &cli.files { + // Resolve the operand against the shell's cwd; all later path + // operations (open, metadata, in-place temp+rename) use the + // resolved path so nothing touches the host process cwd. + let resolved = pi_uutils_ctx::resolve(file); + let path = resolved.as_path(); + let file = + read::load_file(path).map_err(|e| Error::Io(Some(path.display().to_string()), e))?; + let inputs = read::slice(cli, &file); + if cli.in_place { + // create a temporary file where output is written to, + // in the resolved target's directory so the final rename + // stays on the same filesystem + let location = path.parent().unwrap(); + let mut tmp = tempfile::Builder::new() + .prefix("jaq") + .tempfile_in(location)?; + + last = filter::run(cli, &filter, ctx.clone(), inputs, |output| { + print(tmp.as_file_mut(), cli, &output) + })?; + + // replace the input file with the temporary file + std::mem::drop(file); + let perms = std::fs::metadata(path)?.permissions(); + tmp.persist(path).map_err(Error::Persist)?; + std::fs::set_permissions(path, perms)?; + } else { + last = with_stdout(|out| { + filter::run(cli, &filter, ctx.clone(), inputs, |v| print(out, cli, &v)) + })?; + } + } + last + }; + + if cli.exit_status { + last.map_or_else(|| Err(Error::NoOutput), |b| if b { Ok(0) } else { Err(Error::FalseOrNull) }) + } else { + Ok(0) + } +} + +fn binds(cli: &Cli) -> Result, Error> { + let arg = cli.arg.iter().map(|(k, s)| { + let s = s.to_owned(); + Ok((k.to_owned(), Val::Str(s.into()))) + }); + let argjson = cli.argjson.iter().map(|(k, s)| { + use hifijson::token::Lex; + let mut lexer = hifijson::SliceLexer::new(s.as_bytes()); + let err = |e| Error::Parse(format!("{e} (for value passed to `--argjson {k}`)")); + Ok((k.to_owned(), lexer.exactly_one(Val::parse).map_err(err)?)) + }); + let rawfile = cli.rawfile.iter().map(|(k, path)| { + let s = std::fs::read_to_string(pi_uutils_ctx::resolve(path)) + .map_err(|e| Error::Io(Some(format!("{path:?}")), e)); + Ok((k.to_owned(), Val::Str(s?.into()))) + }); + let slurpfile = cli.slurpfile.iter().map(|(k, path)| { + let a = read::json_array(path).map_err(|e| Error::Io(Some(format!("{path:?}")), e)); + Ok((k.to_owned(), a?)) + }); + + let positional = cli.args.iter().cloned().map(|s| Ok(Val::from(s))); + let positional = positional.collect::, Error>>()?; + + let var_val = arg.chain(rawfile).chain(slurpfile).chain(argjson); + let mut var_val = var_val.collect::, Error>>()?; + + var_val.push(("ARGS".to_string(), args(&positional, &var_val))); + // the shell's exported environment, not the host process environment + let env = pi_uutils_ctx::env_snapshot() + .into_iter() + .map(|(k, v)| (k.into(), Val::from(v))); + var_val.push(("ENV".to_string(), Val::obj(env.collect()))); + + Ok(var_val) +} + +fn args(positional: &[Val], named: &[(String, Val)]) -> Val { + let key = |k: &str| k.to_string().into(); + let positional = positional.iter().cloned(); + let named = named.iter().map(|(var, val)| (key(var), val.clone())); + let obj = [(key("positional"), positional.collect()), (key("named"), Val::obj(named.collect()))]; + Val::obj(obj.into_iter().collect()) +} + +#[derive(Debug)] +enum Error { + Io(Option, io::Error), + Report(Vec), + Parse(String), + Jaq(jaq_core::Error), + Persist(tempfile::PersistError), + FalseOrNull, + NoOutput, +} + +impl Display for Error { + fn fmt(&self, f: &mut Formatter) -> fmt::Result { + match self { + Self::FalseOrNull | Self::NoOutput => Ok(()), + Self::Io(prefix, e) => { + write!(f, "Error: ")?; + if let Some(p) = prefix { + write!(f, "{p}: ")?; + } + writeln!(f, "{e}") + }, + Self::Persist(e) => { + writeln!(f, "Error: {e}") + }, + Self::Report(reports) => reports.iter().try_for_each(|fr| write!(f, "{fr}")), + Self::Parse(e) => writeln!(f, "Error: failed to parse: {e}"), + Self::Jaq(e) => writeln!(f, "Error: {e}"), + } + } +} + +impl Error { + /// Upstream's `Termination` exit-code mapping, kept verbatim. + fn report(&self) -> i32 { + match self { + Self::FalseOrNull => 1, + Self::Io(..) | Self::Persist(_) => 2, + Self::Report(_) => 3, + Self::NoOutput => 4, + Self::Parse(_) | Self::Jaq(_) => 5, + } + } +} + +impl From for Error { + fn from(e: io::Error) -> Self { + Self::Io(None, e) + } +} + +fn run_test(test: load::test::Test) -> Result<(Val, Val), Error> { + let (ctx, filter) = + filter::parse_compile(&PathBuf::new(), &test.filter, &[], &[]).map_err(Error::Report)?; + + let inputs = RcIter::new(Box::new(core::iter::empty())); + let ctx = Ctx::new(ctx, &inputs); + + let json = |s: String| { + use hifijson::token::Lex; + hifijson::SliceLexer::new(s.as_bytes()) + .exactly_one(Val::parse) + .map_err(read::invalid_data) + }; + let input = json(test.input)?; + let expect: Result = test.output.into_iter().map(json).collect(); + let obtain: Result = filter.run((ctx, input)).collect(); + Ok((expect?, obtain.map_err(Error::Jaq)?)) +} + +fn run_tests(read: impl BufRead) -> i32 { + let lines = read.lines().map(Result::unwrap); + let tests = load::test::Parser::new(lines); + + let (mut passed, mut total) = (0, 0); + for test in tests { + if pi_uutils_ctx::is_cancelled() { + break; + } + let _ = writeln!(pi_uutils_ctx::stdout(), "Testing {}", test.filter); + match run_test(test) { + Err(e) => { + let _ = writeln!(pi_uutils_ctx::stderr(), "{e:?}"); + }, + Ok((expect, obtain)) if expect != obtain => { + let _ = writeln!(pi_uutils_ctx::stderr(), "expected {expect}, obtained {obtain}",); + }, + Ok(_) => passed += 1, + } + total += 1; + } + + let _ = writeln!(pi_uutils_ctx::stdout(), "{passed} out of {total} tests passed"); + + i32::from(total > passed) +} + +#[cfg(test)] +mod tests; diff --git a/crates/vendor/jaq/src/read.rs b/crates/vendor/jaq/src/read.rs new file mode 100644 index 000000000..1169516bc --- /dev/null +++ b/crates/vendor/jaq/src/read.rs @@ -0,0 +1,89 @@ +use std::{ + io::{self, BufRead}, + path::Path, +}; + +use crate::{Cli, Val}; + +/// Try to load file by memory mapping and fall back to regular loading if it +/// fails. +/// +/// The path is resolved against the shell's working directory: as an +/// in-process builtin, the host process cwd is unrelated to the shell's. +pub fn load_file(path: impl AsRef) -> io::Result>> { + let path = pi_uutils_ctx::resolve(path.as_ref()); + let file = std::fs::File::open(&path)?; + match unsafe { memmap2::Mmap::map(&file) } { + Ok(mmap) => Ok(Box::new(mmap)), + Err(_) => Ok(Box::new(std::fs::read(&path)?)), + } +} + +pub fn invalid_data(e: impl std::error::Error + Send + Sync + 'static) -> std::io::Error { + io::Error::new(io::ErrorKind::InvalidData, e) +} + +fn json_slice(slice: &[u8]) -> impl Iterator> + '_ { + let mut lexer = hifijson::SliceLexer::new(slice); + core::iter::from_fn(move || { + use hifijson::token::Lex; + Some(Val::parse(lexer.ws_token()?, &mut lexer).map_err(invalid_data)) + }) +} + +fn json_read<'a>(read: impl BufRead + 'a) -> impl Iterator> + 'a { + let mut lexer = hifijson::IterLexer::new(read.bytes()); + core::iter::from_fn(move || { + use hifijson::token::Lex; + let v = Val::parse(lexer.ws_token()?, &mut lexer); + Some(v.map_err(|e| core::mem::take(&mut lexer.error).unwrap_or_else(|| invalid_data(e)))) + }) +} + +pub fn json_array(path: impl AsRef) -> io::Result { + json_slice(&load_file(path.as_ref())?).collect() +} + +pub fn buffered<'a, R>(cli: &Cli, read: R) -> Box> + 'a> +where + R: BufRead + 'a, +{ + if cli.raw_input { + Box::new(raw_input(cli.slurp, read).map(|r| r.map(Val::from))) + } else { + Box::new(collect_if(cli.slurp, json_read(read))) + } +} + +pub fn slice<'a>(cli: &Cli, slice: &'a [u8]) -> Box> + 'a> { + if cli.raw_input { + let read = io::BufReader::new(slice); + Box::new(raw_input(cli.slurp, read).map(|r| r.map(Val::from))) + } else { + Box::new(collect_if(cli.slurp, json_slice(slice))) + } +} + +fn raw_input<'a, R>(slurp: bool, mut read: R) -> impl Iterator> + 'a +where + R: BufRead + 'a, +{ + if slurp { + let mut buf = String::new(); + let s = read.read_to_string(&mut buf).map(|_| buf); + Box::new(std::iter::once(s)) + } else { + Box::new(read.lines()) as Box> + } +} + +fn collect_if<'a, T: FromIterator + 'a, E: 'a>( + slurp: bool, + iter: impl Iterator> + 'a, +) -> Box> + 'a> { + if slurp { + Box::new(core::iter::once(iter.collect())) + } else { + Box::new(iter) + } +} diff --git a/crates/vendor/jaq/src/tests.rs b/crates/vendor/jaq/src/tests.rs new file mode 100644 index 000000000..a24cbfa37 --- /dev/null +++ b/crates/vendor/jaq/src/tests.rs @@ -0,0 +1,299 @@ +//! Behavioral contract tests driving [`crate::run`] under a +//! [`pi_uutils_ctx::scope`], the way the shell host invokes the builtin. + +use std::{ + collections::HashMap, + ffi::OsString, + io::{self, Write}, + path::PathBuf, + sync::{Arc, atomic::AtomicBool}, +}; + +use parking_lot::Mutex; + +/// `Send + Write` capture buffer for the scope's stdout/stderr. +#[derive(Clone, Default)] +struct Buf(Arc>>); + +impl Buf { + fn take_string(&self) -> String { + String::from_utf8(std::mem::take(&mut *self.0.lock())).expect("utf8 output") + } +} + +impl Write for Buf { + fn write(&mut self, buf: &[u8]) -> io::Result { + self.0.lock().extend_from_slice(buf); + Ok(buf.len()) + } + + fn flush(&mut self) -> io::Result<()> { + Ok(()) + } +} + +/// Runs `jq ` with `stdin` under a fresh scope; returns +/// `(exit code, stdout, stderr)`. +fn run_jq_in( + cwd: PathBuf, + env: HashMap, + args: &[&str], + stdin: &str, +) -> (i32, String, String) { + let out = Buf::default(); + let err = Buf::default(); + let io_ = pi_uutils_ctx::ScopeIo { + stdin: Box::new(io::Cursor::new(stdin.as_bytes().to_vec())), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(out.clone()), + stderr: Box::new(err.clone()), + cwd, + env, + cancel: Arc::new(AtomicBool::new(false)), + }; + let mut argv = vec![OsString::from("jq")]; + argv.extend(args.iter().map(OsString::from)); + let code = pi_uutils_ctx::scope(io_, || crate::run(argv)); + (code, out.take_string(), err.take_string()) +} + +fn run_jq(args: &[&str], stdin: &str) -> (i32, String, String) { + run_jq_in(PathBuf::from("."), HashMap::new(), args, stdin) +} + +#[test] +fn identity_pretty_prints() { + let (code, out, err) = run_jq(&["."], "{\"a\":1}"); + assert_eq!(code, 0); + assert_eq!(out, "{\n \"a\": 1\n}\n"); + assert_eq!(err, ""); +} + +#[test] +fn compact_output() { + let (code, out, _) = run_jq(&["-c", ".a"], "{\"a\":[1,2]}"); + assert_eq!(code, 0); + assert_eq!(out, "[1,2]\n"); +} + +#[test] +fn raw_output_strips_quotes() { + let (code, out, _) = run_jq(&["-r", ".s"], "{\"s\":\"x y\"}"); + assert_eq!(code, 0); + assert_eq!(out, "x y\n"); + + let (code, out, _) = run_jq(&[".s"], "{\"s\":\"x y\"}"); + assert_eq!(code, 0); + assert_eq!(out, "\"x y\"\n"); +} + +#[test] +fn null_input_evaluates_filter() { + let (code, out, _) = run_jq(&["-n", "1+2"], ""); + assert_eq!(code, 0); + assert_eq!(out, "3\n"); +} + +#[test] +fn slurp_collects_documents() { + let (code, out, _) = run_jq(&["-s", "length"], "{\"a\":1}\n{\"b\":2}\n"); + assert_eq!(code, 0); + assert_eq!(out, "2\n"); +} + +#[test] +fn named_arg_binds_variable() { + let (code, out, _) = run_jq(&["-n", "--arg", "k", "v", "$k"], ""); + assert_eq!(code, 0); + assert_eq!(out, "\"v\"\n"); +} + +#[test] +fn argjson_binds_json_value() { + let (code, out, _) = run_jq(&["-nc", "--argjson", "k", "[1,2]", "$k"], ""); + assert_eq!(code, 0); + assert_eq!(out, "[1,2]\n"); +} + +#[test] +fn exit_status_flag() { + // false -> 1 + let (code, out, _) = run_jq(&["-n", "-e", "false"], ""); + assert_eq!(code, 1); + assert_eq!(out, "false\n"); + + // null (missing key) -> 1 + let (code, out, _) = run_jq(&["-e", ".missing"], "{}"); + assert_eq!(code, 1); + assert_eq!(out, "null\n"); + + // truthy -> 0 + let (code, ..) = run_jq(&["-e", "."], "true"); + assert_eq!(code, 0); + + // no output at all -> 4 (jaq-specific; jq also uses 4 here) + let (code, ..) = run_jq(&["-n", "-e", "empty"], ""); + assert_eq!(code, 4); +} + +#[test] +fn compile_error_exits_3_with_diagnostic() { + let (code, out, err) = run_jq(&["("], "null"); + assert_eq!(code, 3); + assert_eq!(out, "", "compile error must not produce output"); + assert!(err.contains("Error:"), "diagnostic on stderr: {err:?}"); + assert!(err.contains(""), "names the filter source: {err:?}"); +} + +#[test] +fn runtime_error_exits_5_with_diagnostic() { + // indexing a number is a runtime (Jaq) error + let (code, out, err) = run_jq(&[".[0]"], "1"); + assert_eq!(code, 5); + assert_eq!(out, ""); + assert!(err.starts_with("Error:"), "diagnostic on stderr: {err:?}"); +} + +#[test] +fn usage_error_exits_2() { + let (code, _, err) = run_jq(&["--bogus", "."], ""); + assert_eq!(code, 2); + assert!(err.contains("unknown flag: --bogus"), "stderr: {err:?}"); +} + +#[test] +fn relative_file_operand_resolves_against_scope_cwd() { + let dir = tempfile::TempDir::new().expect("tempdir"); + std::fs::write(dir.path().join("in.json"), "{\"a\":[1,2]}").expect("write input"); + // relative operand: must resolve against ScopeIo.cwd, not the process cwd + let (code, out, err) = + run_jq_in(dir.path().to_path_buf(), HashMap::new(), &["-c", ".a", "in.json"], ""); + assert_eq!(code, 0, "stderr: {err:?}"); + assert_eq!(out, "[1,2]\n"); +} + +#[test] +fn missing_file_operand_exits_2() { + let dir = tempfile::TempDir::new().expect("tempdir"); + let (code, out, err) = + run_jq_in(dir.path().to_path_buf(), HashMap::new(), &[".", "nope.json"], ""); + assert_eq!(code, 2); + assert_eq!(out, ""); + assert!(err.contains("nope.json"), "stderr names the operand: {err:?}"); +} + +#[test] +fn in_place_edit_rewrites_relative_file() { + let dir = tempfile::TempDir::new().expect("tempdir"); + std::fs::write(dir.path().join("in.json"), "{\"a\":1}").expect("write input"); + let (code, _, err) = + run_jq_in(dir.path().to_path_buf(), HashMap::new(), &["-c", "-i", ".a", "in.json"], ""); + assert_eq!(code, 0, "stderr: {err:?}"); + let rewritten = std::fs::read_to_string(dir.path().join("in.json")).expect("read back"); + assert_eq!(rewritten, "1\n"); +} + +#[test] +fn invalid_trailing_json_on_stdin_fails() { + let (code, out, err) = run_jq(&["-c", "."], "{\"a\":1} xyz"); + assert_eq!(code, 5); + assert_eq!(out, "{\"a\":1}\n", "valid leading document is still emitted"); + assert!(err.contains("Error:"), "stderr diagnostic: {err:?}"); +} + +#[test] +fn env_var_and_dollar_env_read_scope_environment() { + let env = HashMap::from([("FOO".to_string(), "bar".to_string())]); + let (code, out, _) = run_jq_in(PathBuf::from("."), env, &["-n", "$ENV.FOO, env.FOO"], ""); + assert_eq!(code, 0); + assert_eq!(out, "\"bar\"\n\"bar\"\n", "$ENV and env read the shell env"); +} + +#[test] +fn halt_returns_instead_of_killing_process() { + let (code, out, err) = run_jq(&["-n", "1, halt, 2"], ""); + assert_eq!(code, 0, "halt exits 0"); + assert_eq!(out, "1\n", "outputs before halt are emitted, none after"); + assert_eq!(err, ""); +} + +#[test] +fn halt_error_prints_message_and_exit_code() { + let (code, out, _) = run_jq(&["-n", "\"bye\\n\" | halt_error(3)"], ""); + assert_eq!(code, 3); + assert_eq!(out, "bye\n", "string message printed raw"); +} + +#[test] +fn stderr_filter_writes_to_scope_stderr() { + let (code, out, err) = run_jq(&["-n", "\"msg\" | stderr | length"], ""); + assert_eq!(code, 0); + assert_eq!(out, "3\n", "stderr is an identity filter"); + assert_eq!(err, "msg", "raw string on stderr, no newline"); +} + +#[test] +fn debug_filter_writes_to_scope_stderr() { + let (code, out, err) = run_jq(&["-nc", "[1,2] | debug"], ""); + assert_eq!(code, 0); + assert_eq!(out, "[1,2]\n"); + assert_eq!(err, "[\"DEBUG:\", [1,2]]\n"); +} + +#[test] +fn rawfile_and_slurpfile_resolve_against_scope_cwd() { + let dir = tempfile::TempDir::new().expect("tempdir"); + std::fs::write(dir.path().join("raw.txt"), "hi").expect("write raw"); + std::fs::write(dir.path().join("vals.json"), "1 2").expect("write vals"); + let (code, out, err) = run_jq_in( + dir.path().to_path_buf(), + HashMap::new(), + &["-nc", "--rawfile", "r", "raw.txt", "--slurpfile", "v", "vals.json", "$r, $v"], + "", + ); + assert_eq!(code, 0, "stderr: {err:?}"); + assert_eq!(out, "\"hi\"\n[1,2]\n"); +} + +#[test] +fn version_flag_prints_and_exits_0() { + let (code, out, _) = run_jq(&["--version"], ""); + assert_eq!(code, 0); + assert_eq!(out, format!("jaq {}\n", env!("CARGO_PKG_VERSION"))); +} + +#[test] +fn tab_and_indent_control_pretty_printing() { + let (code, out, _) = run_jq(&["--tab", "."], "{\"a\":1}"); + assert_eq!(code, 0); + assert_eq!(out, "{\n\t\"a\": 1\n}\n"); + + let (code, out, _) = run_jq(&["--indent", "4", "."], "{\"a\":1}"); + assert_eq!(code, 0); + assert_eq!(out, "{\n \"a\": 1\n}\n"); +} + +#[test] +fn from_file_reads_filter_relative_to_scope_cwd() { + let dir = tempfile::TempDir::new().expect("tempdir"); + std::fs::write(dir.path().join("f.jq"), ".a + 1").expect("write filter"); + let (code, out, err) = + run_jq_in(dir.path().to_path_buf(), HashMap::new(), &["-f", "f.jq"], "{\"a\":1}"); + assert_eq!(code, 0, "stderr: {err:?}"); + assert_eq!(out, "2\n"); +} + +#[test] +fn join_output_omits_newlines() { + let (code, out, _) = run_jq(&["-j", ".[]"], "[\"a\",\"b\"]"); + assert_eq!(code, 0); + assert_eq!(out, "ab"); +} + +#[test] +fn positional_args_after_double_dash_args() { + let (code, out, _) = run_jq(&["-nc", "$ARGS.positional", "--args", "x", "y"], ""); + assert_eq!(code, 0); + assert_eq!(out, "[\"x\",\"y\"]\n"); +} diff --git a/crates/vendor/jaq/src/write.rs b/crates/vendor/jaq/src/write.rs new file mode 100644 index 000000000..f9e2ed930 --- /dev/null +++ b/crates/vendor/jaq/src/write.rs @@ -0,0 +1,127 @@ +use core::fmt::{self, Display, Formatter}; +use std::io::{self, Write}; + +use crate::{Cli, Val}; + +struct FormatterFn(F); + +impl fmt::Result> Display for FormatterFn { + fn fmt(&self, f: &mut Formatter) -> fmt::Result { + self.0(f) + } +} + +struct PpOpts { + compact: bool, + indent: String, + sort_keys: bool, +} + +impl PpOpts { + fn indent(&self, f: &mut Formatter, level: usize) -> fmt::Result { + if !self.compact { + write!(f, "{}", self.indent.repeat(level))?; + } + Ok(()) + } + + fn newline(&self, f: &mut Formatter) -> fmt::Result { + if !self.compact { + writeln!(f)?; + } + Ok(()) + } +} + +fn fmt_seq(fmt: &mut Formatter, opts: &PpOpts, level: usize, xs: I, f: F) -> fmt::Result +where + I: IntoIterator, + F: Fn(&mut Formatter, T) -> fmt::Result, +{ + opts.newline(fmt)?; + let mut iter = xs.into_iter().peekable(); + while let Some(x) = iter.next() { + opts.indent(fmt, level + 1)?; + f(fmt, x)?; + if iter.peek().is_some() { + write!(fmt, ",")?; + } + opts.newline(fmt)?; + } + opts.indent(fmt, level) +} + +fn fmt_val(f: &mut Formatter, opts: &PpOpts, level: usize, v: &Val) -> fmt::Result { + use yansi::Paint; + match v { + Val::Null | Val::Bool(_) | Val::Int(_) | Val::Float(_) | Val::Num(_) => v.fmt(f), + Val::Str(_) => write!(f, "{}", v.green()), + Val::Arr(a) => { + '['.bold().fmt(f)?; + if !a.is_empty() { + fmt_seq(f, opts, level, &**a, |f, x| fmt_val(f, opts, level + 1, x))?; + } + ']'.bold().fmt(f) + }, + Val::Obj(o) => { + '{'.bold().fmt(f)?; + let kv = |f: &mut Formatter, (k, val): (&std::rc::Rc, &Val)| { + write!(f, "{}:", Val::Str(k.clone()).bold())?; + if !opts.compact { + write!(f, " ")?; + } + fmt_val(f, opts, level + 1, val) + }; + if !o.is_empty() { + if opts.sort_keys { + let mut o: Vec<_> = o.iter().collect(); + o.sort_by_key(|(k, _v)| *k); + fmt_seq(f, opts, level, o, kv) + } else { + fmt_seq(f, opts, level, &**o, kv) + }? + } + '}'.bold().fmt(f) + }, + } +} + +pub fn print(w: &mut (impl Write + ?Sized), cli: &Cli, val: &Val) -> io::Result<()> { + let f = |f: &mut Formatter| { + let opts = PpOpts { + compact: cli.compact_output, + indent: if cli.tab { + String::from("\t") + } else { + " ".repeat(cli.indent) + }, + sort_keys: cli.sort_keys, + }; + fmt_val(f, &opts, 0, val) + }; + + match val { + Val::Str(s) if cli.raw_output || cli.join_output => write!(w, "{s}")?, + _ => write!(w, "{}", FormatterFn(f))?, + }; + + if cli.join_output { + // when running `jaq -jn '"prompt> " | (., input)'`, + // this flush is necessary to make "prompt> " appear first + w.flush() + } else { + writeln!(w) + } +} + +/// Runs `f` with a buffered writer over the ctx stdout stream. +/// +/// Upstream used an unbuffered lock when stdout was a terminal; the ctx +/// stream never is, so output is always buffered and flushed at the end +/// (flush errors are dropped, matching upstream's `BufWriter` drop). +pub fn with_stdout(f: impl FnOnce(&mut dyn Write) -> T) -> T { + let mut out = io::BufWriter::new(pi_uutils_ctx::stdout()); + let res = f(&mut out); + let _ = out.flush(); + res +} diff --git a/crates/vendor/uu-b2sum/src/b2sum.rs b/crates/vendor/uu-b2sum/src/b2sum.rs index f63e1330d..92bae9bd5 100644 --- a/crates/vendor/uu-b2sum/src/b2sum.rs +++ b/crates/vendor/uu-b2sum/src/b2sum.rs @@ -5,31 +5,32 @@ // spell-checker:ignore (ToDO) algo -// pi-uutils: Patched for in-process embedding via the shared `uu-checksum-common` crate, -// which redirects all standard stream I/O and file resolution through `pi-uutils-ctx`. +// pi-uutils: Patched for in-process embedding via the shared +// `uu-checksum-common` crate, which redirects all standard stream I/O and file +// resolution through `pi-uutils-ctx`. -use clap::Command; use std::ffi::OsString; +use clap::Command; use uucore::checksum::{AlgoKind, BlakeLength, parse_blake_length}; pub fn run(argv: Vec) -> i32 { - let calculate_blake2b_length = - |s: &str| parse_blake_length(AlgoKind::Blake2b, BlakeLength::String(s)); - uu_checksum_common::run_standalone_with_length( - "b2sum", - AlgoKind::Blake2b, - uu_app(), - argv, - calculate_blake2b_length, - ) + let calculate_blake2b_length = + |s: &str| parse_blake_length(AlgoKind::Blake2b, BlakeLength::String(s)); + uu_checksum_common::run_standalone_with_length( + "b2sum", + AlgoKind::Blake2b, + uu_app(), + argv, + calculate_blake2b_length, + ) } #[inline] pub fn uu_app() -> Command { - uu_checksum_common::standalone_checksum_app_with_length( - "Print or check BLAKE2b (512-bit) checksums.", - "b2sum [OPTION]... [FILE]...", - ) - .name("b2sum") + uu_checksum_common::standalone_checksum_app_with_length( + "Print or check BLAKE2b (512-bit) checksums.", + "b2sum [OPTION]... [FILE]...", + ) + .name("b2sum") } diff --git a/crates/vendor/uu-base32/src/base32.rs b/crates/vendor/uu-base32/src/base32.rs index 286430c23..87a266fb9 100644 --- a/crates/vendor/uu-base32/src/base32.rs +++ b/crates/vendor/uu-base32/src/base32.rs @@ -5,43 +5,48 @@ pub mod base_common; +use std::{ffi::OsString, io::Write}; + use clap::Command; -use std::ffi::OsString; -use std::io::Write; use uucore::encoding::Format; /// pi-uutils: safe in-process entry point using invocation-scoped streams. pub fn run(argv: Vec) -> i32 { - let matches = match uu_app().try_get_matches_from(argv) { - Ok(matches) => matches, - Err(err) => { - let rendered = err.to_string(); - if err.use_stderr() { - let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); - return 1; - } - let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); - return 0; - } - }; - let result = base_common::Config::from(&matches).and_then(|config| { - let mut input = base_common::get_input(&config)?; - base_common::handle_input(&mut input, Format::Base32, config) - }); - match result { - Ok(()) => pi_uutils_ctx::exit_code(), - Err(err) => { - let code = err.code(); - let _ = writeln!(pi_uutils_ctx::stderr(), "base32: {err}"); - if code == 0 { 1 } else { code } - } - } + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + let result = base_common::Config::from(&matches).and_then(|config| { + let mut input = base_common::get_input(&config)?; + base_common::handle_input(&mut input, Format::Base32, config) + }); + match result { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + let _ = writeln!(pi_uutils_ctx::stderr(), "base32: {err}"); + if code == 0 { 1 } else { code } + }, + } } pub fn uu_app() -> Command { - base_common::base_app( - "encode/decode data and print to standard output\nWith no FILE, or when FILE is -, read standard input.\n\nThe data are encoded as described for the base32 alphabet in RFC 4648.\nWhen decoding, the input may contain newlines in addition to the bytes of the formal base32 alphabet. Use --ignore-garbage to attempt to recover from any other non-alphabet bytes in the encoded stream.".into(), - "base32 [OPTION]... [FILE]".into(), - ) - .name("base32") + base_common::base_app( + "encode/decode data and print to standard output\nWith no FILE, or when FILE is -, read \ + standard input.\n\nThe data are encoded as described for the base32 alphabet in RFC \ + 4648.\nWhen decoding, the input may contain newlines in addition to the bytes of the \ + formal base32 alphabet. Use --ignore-garbage to attempt to recover from any other \ + non-alphabet bytes in the encoded stream." + .into(), + "base32 [OPTION]... [FILE]".into(), + ) + .name("base32") } diff --git a/crates/vendor/uu-base32/src/base_common.rs b/crates/vendor/uu-base32/src/base_common.rs index 58cea1c2c..827edec5c 100644 --- a/crates/vendor/uu-base32/src/base_common.rs +++ b/crates/vendor/uu-base32/src/base_common.rs @@ -5,949 +5,947 @@ // spell-checker:ignore hexupper lsbf msbf unpadded nopad aGVsbG8sIHdvcmxkIQ -use clap::{Arg, ArgAction, Command}; -use std::ffi::OsString; -use std::fs::File; -use std::io::{self, BufRead, BufReader, Write}; -use std::path::Path; -use uucore::display::Quotable; -use uucore::encoding::{ - BASE2LSBF, BASE2MSBF, Base32Wrapper, Base58Wrapper, Base64SimdWrapper, EncodingWrapper, Format, - SupportsFastDecodeAndEncode, Z85Wrapper, - for_base_common::{BASE32, BASE32HEX, BASE64URL, HEXUPPER_PERMISSIVE}, +use std::{ + ffi::OsString, + fs::File, + io::{self, BufRead, BufReader, Write}, + path::Path, +}; + +use clap::{Arg, ArgAction, Command}; +use uucore::{ + display::Quotable, + encoding::{ + BASE2LSBF, BASE2MSBF, Base32Wrapper, Base58Wrapper, Base64SimdWrapper, EncodingWrapper, + Format, SupportsFastDecodeAndEncode, Z85Wrapper, + for_base_common::{BASE32, BASE32HEX, BASE64URL, HEXUPPER_PERMISSIVE}, + }, + error::{FromIo, UResult, USimpleError, UUsageError, strip_errno}, + format_usage, }; -use uucore::error::{FromIo, UResult, USimpleError, UUsageError, strip_errno}; -use uucore::format_usage; pub const BASE_CMD_PARSE_ERROR: i32 = 1; -/// Encoded output will be formatted in lines of this length (the last line can be shorter) +/// Encoded output will be formatted in lines of this length (the last line can +/// be shorter) /// /// Other implementations default to 76 /// /// This default is only used if no "-w"/"--wrap" argument is passed pub const WRAP_DEFAULT: usize = 76; -// Fixed to 8 KiB (equivalent to `std::sys::io::DEFAULT_BUF_SIZE` on most targets) +// Fixed to 8 KiB (equivalent to `std::sys::io::DEFAULT_BUF_SIZE` on most +// targets) pub const DEFAULT_BUF_SIZE: usize = 8 * 1024; pub struct Config { - pub decode: bool, - pub ignore_garbage: bool, - pub wrap_cols: Option, - pub to_read: Option, + pub decode: bool, + pub ignore_garbage: bool, + pub wrap_cols: Option, + pub to_read: Option, } pub mod options { - pub static DECODE: &str = "decode"; - pub static WRAP: &str = "wrap"; - pub static IGNORE_GARBAGE: &str = "ignore-garbage"; - pub static FILE: &str = "file"; + pub static DECODE: &str = "decode"; + pub static WRAP: &str = "wrap"; + pub static IGNORE_GARBAGE: &str = "ignore-garbage"; + pub static FILE: &str = "file"; } impl Config { - pub fn from(options: &clap::ArgMatches) -> UResult { - let to_read = match options.get_many::(options::FILE) { - Some(mut values) => { - let name = values.next().unwrap(); + pub fn from(options: &clap::ArgMatches) -> UResult { + let to_read = match options.get_many::(options::FILE) { + Some(mut values) => { + let name = values.next().unwrap(); - if let Some(extra_op) = values.next() { - return Err(UUsageError::new( - BASE_CMD_PARSE_ERROR, - format!("extra operand {}", extra_op.quote()), - )); - } + if let Some(extra_op) = values.next() { + return Err(UUsageError::new( + BASE_CMD_PARSE_ERROR, + format!("extra operand {}", extra_op.quote()), + )); + } - if name == "-" { - None - } else { - Some(name.clone()) - } - } - None => None, - }; + if name == "-" { + None + } else { + Some(name.clone()) + } + }, + None => None, + }; - let wrap_cols = options - .get_one::(options::WRAP) - .map(|num| { - num.parse::().map_err(|_| { - USimpleError::new( - BASE_CMD_PARSE_ERROR, - format!("invalid wrap size: {}", num.quote()), - ) - }) - }) - .transpose()?; + let wrap_cols = options + .get_one::(options::WRAP) + .map(|num| { + num.parse::().map_err(|_| { + USimpleError::new( + BASE_CMD_PARSE_ERROR, + format!("invalid wrap size: {}", num.quote()), + ) + }) + }) + .transpose()?; - Ok(Self { - decode: options.get_flag(options::DECODE), - ignore_garbage: options.get_flag(options::IGNORE_GARBAGE), - wrap_cols, - to_read, - }) - } + Ok(Self { + decode: options.get_flag(options::DECODE), + ignore_garbage: options.get_flag(options::IGNORE_GARBAGE), + wrap_cols, + to_read, + }) + } } - pub fn base_app(about: String, usage: String) -> Command { - let cmd = Command::new("") - .version(uucore::crate_version!()) - .about(about) - .override_usage(format_usage(&usage)) - .infer_long_args(true); - uucore::clap_localization::configure_localized_command(cmd) - // Format arguments. - .arg( - Arg::new(options::DECODE) - .short('d') - .visible_short_alias('D') - .long(options::DECODE) - .help("decode data") - .action(ArgAction::SetTrue) - .overrides_with(options::DECODE), - ) - .arg( - Arg::new(options::IGNORE_GARBAGE) - .short('i') - .long(options::IGNORE_GARBAGE) - .help("when decoding, ignore non-alphabetic characters") - .action(ArgAction::SetTrue) - .overrides_with(options::IGNORE_GARBAGE), - ) - .arg( - Arg::new(options::WRAP) - .short('w') - .long(options::WRAP) - .value_name("COLS") - .help(format!("wrap encoded lines after COLS character (default {WRAP_DEFAULT}, 0 to disable wrapping)")) - .overrides_with(options::WRAP), - ) - // "multiple" arguments are used to check whether there is more than one - // file passed in. - .arg( - Arg::new(options::FILE) - .index(1) - .action(ArgAction::Append) - .value_parser(clap::value_parser!(OsString)) - .value_hint(clap::ValueHint::FilePath), - ) + let cmd = Command::new("") + .version(uucore::crate_version!()) + .about(about) + .override_usage(format_usage(&usage)) + .infer_long_args(true); + uucore::clap_localization::configure_localized_command(cmd) + // Format arguments. + .arg( + Arg::new(options::DECODE) + .short('d') + .visible_short_alias('D') + .long(options::DECODE) + .help("decode data") + .action(ArgAction::SetTrue) + .overrides_with(options::DECODE), + ) + .arg( + Arg::new(options::IGNORE_GARBAGE) + .short('i') + .long(options::IGNORE_GARBAGE) + .help("when decoding, ignore non-alphabetic characters") + .action(ArgAction::SetTrue) + .overrides_with(options::IGNORE_GARBAGE), + ) + .arg( + Arg::new(options::WRAP) + .short('w') + .long(options::WRAP) + .value_name("COLS") + .help(format!( + "wrap encoded lines after COLS character (default {WRAP_DEFAULT}, 0 to disable \ + wrapping)" + )) + .overrides_with(options::WRAP), + ) + // "multiple" arguments are used to check whether there is more than one + // file passed in. + .arg( + Arg::new(options::FILE) + .index(1) + .action(ArgAction::Append) + .value_parser(clap::value_parser!(OsString)) + .value_hint(clap::ValueHint::FilePath), + ) } pub fn get_input(config: &Config) -> UResult> { - match &config.to_read { - Some(name) => { - let file = File::open(pi_uutils_ctx::resolve(Path::new(name))) - .map_err_context(|| name.maybe_quote().to_string())?; - Ok(Box::new(BufReader::with_capacity(DEFAULT_BUF_SIZE, file))) - } - None => { - // pi-uutils: stdin belongs to this invocation, never the host process. - Ok(Box::new(BufReader::with_capacity( - DEFAULT_BUF_SIZE, - pi_uutils_ctx::stdin(), - ))) - } - } + match &config.to_read { + Some(name) => { + let file = File::open(pi_uutils_ctx::resolve(Path::new(name))) + .map_err_context(|| name.maybe_quote().to_string())?; + Ok(Box::new(BufReader::with_capacity(DEFAULT_BUF_SIZE, file))) + }, + None => { + // pi-uutils: stdin belongs to this invocation, never the host process. + Ok(Box::new(BufReader::with_capacity(DEFAULT_BUF_SIZE, pi_uutils_ctx::stdin()))) + }, + } } pub fn handle_input(input: &mut R, format: Format, config: Config) -> UResult<()> { - // Always allow padding for Base64 to avoid a full pre-scan of the input. - let supports_fast_decode_and_encode = - get_supports_fast_decode_and_encode(format, config.decode, true); + // Always allow padding for Base64 to avoid a full pre-scan of the input. + let supports_fast_decode_and_encode = + get_supports_fast_decode_and_encode(format, config.decode, true); - let supports_fast_decode_and_encode_ref = supports_fast_decode_and_encode.as_ref(); - // pi-uutils: all output is scoped to this invocation. - let mut stdout_lock = pi_uutils_ctx::stdout().lock(); - let result = match (format, config.decode) { - // Base58 must process the entire input as one big integer; keep the - // historical behavior of buffering everything for this format only. - (Format::Base58, _) => { - let mut buffered = Vec::new(); - input - .read_to_end(&mut buffered) - .map_err(|err| USimpleError::new(1, format_read_error(&err)))?; - if config.decode { - fast_decode::fast_decode_buffer( - buffered, - &mut stdout_lock, - supports_fast_decode_and_encode_ref, - config.ignore_garbage, - ) - } else { - fast_encode::fast_encode_buffer( - buffered, - &mut stdout_lock, - supports_fast_decode_and_encode_ref, - config.wrap_cols, - ) - } - } - // Streaming path for all other encodings keeps memory bounded. - (_, true) => fast_decode::fast_decode_stream( - input, - &mut stdout_lock, - supports_fast_decode_and_encode_ref, - config.ignore_garbage, - ), - (_, false) => fast_encode::fast_encode_stream( - input, - &mut stdout_lock, - supports_fast_decode_and_encode_ref, - config.wrap_cols, - ), - }; + let supports_fast_decode_and_encode_ref = supports_fast_decode_and_encode.as_ref(); + // pi-uutils: all output is scoped to this invocation. + let mut stdout_lock = pi_uutils_ctx::stdout().lock(); + let result = match (format, config.decode) { + // Base58 must process the entire input as one big integer; keep the + // historical behavior of buffering everything for this format only. + (Format::Base58, _) => { + let mut buffered = Vec::new(); + input + .read_to_end(&mut buffered) + .map_err(|err| USimpleError::new(1, format_read_error(&err)))?; + if config.decode { + fast_decode::fast_decode_buffer( + buffered, + &mut stdout_lock, + supports_fast_decode_and_encode_ref, + config.ignore_garbage, + ) + } else { + fast_encode::fast_encode_buffer( + buffered, + &mut stdout_lock, + supports_fast_decode_and_encode_ref, + config.wrap_cols, + ) + } + }, + // Streaming path for all other encodings keeps memory bounded. + (_, true) => fast_decode::fast_decode_stream( + input, + &mut stdout_lock, + supports_fast_decode_and_encode_ref, + config.ignore_garbage, + ), + (_, false) => fast_encode::fast_encode_stream( + input, + &mut stdout_lock, + supports_fast_decode_and_encode_ref, + config.wrap_cols, + ), + }; - // Ensure any pending stdout buffer is flushed even if decoding failed; GNU basenc - // keeps already-decoded bytes visible before reporting the error. - match (result, stdout_lock.flush()) { - (res, Ok(())) => res, - (Ok(_), Err(err)) => Err(err.into()), - (Err(original), Err(_)) => Err(original), - } + // Ensure any pending stdout buffer is flushed even if decoding failed; GNU + // basenc keeps already-decoded bytes visible before reporting the error. + match (result, stdout_lock.flush()) { + (res, Ok(())) => res, + (Ok(_), Err(err)) => Err(err.into()), + (Err(original), Err(_)) => Err(original), + } } pub fn get_supports_fast_decode_and_encode( - format: Format, - decode: bool, - has_padding: bool, + format: Format, + decode: bool, + has_padding: bool, ) -> Box { - const BASE16_VALID_DECODING_MULTIPLE: usize = 2; - const BASE2_VALID_DECODING_MULTIPLE: usize = 8; - const BASE32_VALID_DECODING_MULTIPLE: usize = 8; - const BASE64_VALID_DECODING_MULTIPLE: usize = 4; + const BASE16_VALID_DECODING_MULTIPLE: usize = 2; + const BASE2_VALID_DECODING_MULTIPLE: usize = 8; + const BASE32_VALID_DECODING_MULTIPLE: usize = 8; + const BASE64_VALID_DECODING_MULTIPLE: usize = 4; - const BASE16_UNPADDED_MULTIPLE: usize = 1; - const BASE2_UNPADDED_MULTIPLE: usize = 1; - const BASE32_UNPADDED_MULTIPLE: usize = 5; - const BASE64_UNPADDED_MULTIPLE: usize = 3; + const BASE16_UNPADDED_MULTIPLE: usize = 1; + const BASE2_UNPADDED_MULTIPLE: usize = 1; + const BASE32_UNPADDED_MULTIPLE: usize = 5; + const BASE64_UNPADDED_MULTIPLE: usize = 3; - match format { - Format::Base16 => Box::from(EncodingWrapper::new( - HEXUPPER_PERMISSIVE, - BASE16_VALID_DECODING_MULTIPLE, - BASE16_UNPADDED_MULTIPLE, - // spell-checker:disable-next-line - b"0123456789ABCDEFabcdef", - )), - Format::Base2Lsbf => Box::from(EncodingWrapper::new( - BASE2LSBF, - BASE2_VALID_DECODING_MULTIPLE, - BASE2_UNPADDED_MULTIPLE, - // spell-checker:disable-next-line - b"01", - )), - Format::Base2Msbf => Box::from(EncodingWrapper::new( - BASE2MSBF, - BASE2_VALID_DECODING_MULTIPLE, - BASE2_UNPADDED_MULTIPLE, - // spell-checker:disable-next-line - b"01", - )), - Format::Base32 => Box::from(Base32Wrapper::new( - BASE32, - BASE32_VALID_DECODING_MULTIPLE, - BASE32_UNPADDED_MULTIPLE, - // spell-checker:disable-next-line - b"ABCDEFGHIJKLMNOPQRSTUVWXYZ234567=", - )), - Format::Base32Hex => Box::from(Base32Wrapper::new( - BASE32HEX, - BASE32_VALID_DECODING_MULTIPLE, - BASE32_UNPADDED_MULTIPLE, - // spell-checker:disable-next-line - b"0123456789ABCDEFGHIJKLMNOPQRSTUV=", - )), - Format::Base64 => { - let alphabet: &[u8] = if has_padding { - &b"abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789+/="[..] - } else { - &b"abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789+/"[..] - }; - let use_padding = !decode || has_padding; - Box::from(Base64SimdWrapper::new( - use_padding, - BASE64_VALID_DECODING_MULTIPLE, - BASE64_UNPADDED_MULTIPLE, - alphabet, - )) - } - Format::Base64Url => Box::from(EncodingWrapper::new( - BASE64URL, - BASE64_VALID_DECODING_MULTIPLE, - BASE64_UNPADDED_MULTIPLE, - // spell-checker:disable-next-line - b"abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789=_-", - )), - Format::Z85 => Box::from(Z85Wrapper {}), - Format::Base58 => Box::from(Base58Wrapper {}), - } + match format { + Format::Base16 => Box::from(EncodingWrapper::new( + HEXUPPER_PERMISSIVE, + BASE16_VALID_DECODING_MULTIPLE, + BASE16_UNPADDED_MULTIPLE, + // spell-checker:disable-next-line + b"0123456789ABCDEFabcdef", + )), + Format::Base2Lsbf => Box::from(EncodingWrapper::new( + BASE2LSBF, + BASE2_VALID_DECODING_MULTIPLE, + BASE2_UNPADDED_MULTIPLE, + // spell-checker:disable-next-line + b"01", + )), + Format::Base2Msbf => Box::from(EncodingWrapper::new( + BASE2MSBF, + BASE2_VALID_DECODING_MULTIPLE, + BASE2_UNPADDED_MULTIPLE, + // spell-checker:disable-next-line + b"01", + )), + Format::Base32 => Box::from(Base32Wrapper::new( + BASE32, + BASE32_VALID_DECODING_MULTIPLE, + BASE32_UNPADDED_MULTIPLE, + // spell-checker:disable-next-line + b"ABCDEFGHIJKLMNOPQRSTUVWXYZ234567=", + )), + Format::Base32Hex => Box::from(Base32Wrapper::new( + BASE32HEX, + BASE32_VALID_DECODING_MULTIPLE, + BASE32_UNPADDED_MULTIPLE, + // spell-checker:disable-next-line + b"0123456789ABCDEFGHIJKLMNOPQRSTUV=", + )), + Format::Base64 => { + let alphabet: &[u8] = if has_padding { + &b"abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789+/="[..] + } else { + &b"abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789+/"[..] + }; + let use_padding = !decode || has_padding; + Box::from(Base64SimdWrapper::new( + use_padding, + BASE64_VALID_DECODING_MULTIPLE, + BASE64_UNPADDED_MULTIPLE, + alphabet, + )) + }, + Format::Base64Url => Box::from(EncodingWrapper::new( + BASE64URL, + BASE64_VALID_DECODING_MULTIPLE, + BASE64_UNPADDED_MULTIPLE, + // spell-checker:disable-next-line + b"abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789=_-", + )), + Format::Z85 => Box::from(Z85Wrapper {}), + Format::Base58 => Box::from(Base58Wrapper {}), + } } pub mod fast_encode { - use crate::base_common::WRAP_DEFAULT; - use std::{ - cmp::min, - collections::VecDeque, - io::{self, BufRead, Write}, - num::NonZeroUsize, - }; - use uucore::{ - encoding::SupportsFastDecodeAndEncode, - error::{UResult, USimpleError}, - }; + use std::{ + cmp::min, + collections::VecDeque, + io::{self, BufRead, Write}, + num::NonZeroUsize, + }; - struct LineWrapping { - line_length: NonZeroUsize, - print_buffer: Vec, - } + use uucore::{ + encoding::SupportsFastDecodeAndEncode, + error::{UResult, USimpleError}, + }; - // Start of helper functions - fn encode_in_chunks_to_buffer( - supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode, - read_buffer: &[u8], - encoded_buffer: &mut VecDeque, - ) -> UResult<()> { - supports_fast_decode_and_encode.encode_to_vec_deque(read_buffer, encoded_buffer)?; - Ok(()) - } + use crate::base_common::WRAP_DEFAULT; - fn write_without_line_breaks( - encoded_buffer: &mut VecDeque, - output: &mut dyn Write, - is_cleanup: bool, - empty_wrap: bool, - ) -> io::Result<()> { - // TODO - // `encoded_buffer` only has to be a VecDeque if line wrapping is enabled - // (`make_contiguous` should be a no-op here) - // Refactoring could avoid this call - output.write_all(encoded_buffer.make_contiguous())?; + struct LineWrapping { + line_length: NonZeroUsize, + print_buffer: Vec, + } - if is_cleanup { - if !empty_wrap { - output.write_all(b"\n")?; - } - } else { - encoded_buffer.clear(); - } + // Start of helper functions + fn encode_in_chunks_to_buffer( + supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode, + read_buffer: &[u8], + encoded_buffer: &mut VecDeque, + ) -> UResult<()> { + supports_fast_decode_and_encode.encode_to_vec_deque(read_buffer, encoded_buffer)?; + Ok(()) + } - Ok(()) - } + fn write_without_line_breaks( + encoded_buffer: &mut VecDeque, + output: &mut dyn Write, + is_cleanup: bool, + empty_wrap: bool, + ) -> io::Result<()> { + // TODO + // `encoded_buffer` only has to be a VecDeque if line wrapping is enabled + // (`make_contiguous` should be a no-op here) + // Refactoring could avoid this call + output.write_all(encoded_buffer.make_contiguous())?; - fn write_with_line_breaks( - &mut LineWrapping { - ref line_length, - ref mut print_buffer, - }: &mut LineWrapping, - encoded_buffer: &mut VecDeque, - output: &mut dyn Write, - is_cleanup: bool, - ) -> io::Result<()> { - let line_length = line_length.get(); + if is_cleanup { + if !empty_wrap { + output.write_all(b"\n")?; + } + } else { + encoded_buffer.clear(); + } - let make_contiguous_result = encoded_buffer.make_contiguous(); + Ok(()) + } - let chunks_exact = make_contiguous_result.chunks_exact(line_length); + fn write_with_line_breaks( + &mut LineWrapping { ref line_length, ref mut print_buffer }: &mut LineWrapping, + encoded_buffer: &mut VecDeque, + output: &mut dyn Write, + is_cleanup: bool, + ) -> io::Result<()> { + let line_length = line_length.get(); - let mut bytes_added_to_print_buffer = 0; + let make_contiguous_result = encoded_buffer.make_contiguous(); - for sl in chunks_exact { - bytes_added_to_print_buffer += sl.len(); + let chunks_exact = make_contiguous_result.chunks_exact(line_length); - print_buffer.extend_from_slice(sl); - print_buffer.push(b'\n'); - } + let mut bytes_added_to_print_buffer = 0; - output.write_all(print_buffer)?; + for sl in chunks_exact { + bytes_added_to_print_buffer += sl.len(); - // Remove the bytes that were just printed from `encoded_buffer` - drop(encoded_buffer.drain(..bytes_added_to_print_buffer)); + print_buffer.extend_from_slice(sl); + print_buffer.push(b'\n'); + } - if is_cleanup { - if encoded_buffer.is_empty() { - // Do not write a newline in this case, because two trailing newlines should never be printed - } else { - // Print the partial line, since this is cleanup and no more data is coming - output.write_all(encoded_buffer.make_contiguous())?; - output.write_all(b"\n")?; - } - } else { - print_buffer.clear(); - } + output.write_all(print_buffer)?; - Ok(()) - } + // Remove the bytes that were just printed from `encoded_buffer` + drop(encoded_buffer.drain(..bytes_added_to_print_buffer)); - fn write_to_output( - line_wrapping: &mut Option, - encoded_buffer: &mut VecDeque, - output: &mut dyn Write, - is_cleanup: bool, - empty_wrap: bool, - ) -> io::Result<()> { - // Write all data in `encoded_buffer` to `output` - if let &mut Some(ref mut li) = line_wrapping { - write_with_line_breaks(li, encoded_buffer, output, is_cleanup)?; - } else { - write_without_line_breaks(encoded_buffer, output, is_cleanup, empty_wrap)?; - } + if is_cleanup { + if encoded_buffer.is_empty() { + // Do not write a newline in this case, because two trailing + // newlines should never be printed + } else { + // Print the partial line, since this is cleanup and no more data is coming + output.write_all(encoded_buffer.make_contiguous())?; + output.write_all(b"\n")?; + } + } else { + print_buffer.clear(); + } - Ok(()) - } - // End of helper functions + Ok(()) + } - pub fn fast_encode_buffer( - input: Vec, - output: &mut dyn Write, - supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode, - wrap: Option, - ) -> UResult<()> { - // Based on performance testing + fn write_to_output( + line_wrapping: &mut Option, + encoded_buffer: &mut VecDeque, + output: &mut dyn Write, + is_cleanup: bool, + empty_wrap: bool, + ) -> io::Result<()> { + // Write all data in `encoded_buffer` to `output` + if let &mut Some(ref mut li) = line_wrapping { + write_with_line_breaks(li, encoded_buffer, output, is_cleanup)?; + } else { + write_without_line_breaks(encoded_buffer, output, is_cleanup, empty_wrap)?; + } - const ENCODE_IN_CHUNKS_OF_SIZE_MULTIPLE: usize = 1_024; + Ok(()) + } + // End of helper functions - let encode_in_chunks_of_size = - supports_fast_decode_and_encode.unpadded_multiple() * ENCODE_IN_CHUNKS_OF_SIZE_MULTIPLE; + pub fn fast_encode_buffer( + input: Vec, + output: &mut dyn Write, + supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode, + wrap: Option, + ) -> UResult<()> { + // Based on performance testing - assert!(encode_in_chunks_of_size > 0); + const ENCODE_IN_CHUNKS_OF_SIZE_MULTIPLE: usize = 1_024; - // The "data-encoding" crate supports line wrapping, but not arbitrary line wrapping, only certain widths, so - // line wrapping must be handled here. - // https://github.com/ia0/data-encoding/blob/4f42ad7ef242f6d243e4de90cd1b46a57690d00e/lib/src/lib.rs#L1710 - let mut line_wrapping = match wrap { - // Line wrapping is disabled because "-w"/"--wrap" was passed with "0" - Some(0) => None, - // A custom line wrapping value was passed - Some(an) => Some(LineWrapping { - line_length: NonZeroUsize::new(an).unwrap(), - print_buffer: Vec::::new(), - }), - // Line wrapping was not set, so the default is used - None => Some(LineWrapping { - line_length: NonZeroUsize::new(WRAP_DEFAULT).unwrap(), - print_buffer: Vec::::new(), - }), - }; + let encode_in_chunks_of_size = + supports_fast_decode_and_encode.unpadded_multiple() * ENCODE_IN_CHUNKS_OF_SIZE_MULTIPLE; - let input_size = input.len(); + assert!(encode_in_chunks_of_size > 0); - // Start of buffers - // Data that was read from `input` but has not been encoded yet - let mut leftover_buffer = VecDeque::::new(); + // The "data-encoding" crate supports line wrapping, but not arbitrary line + // wrapping, only certain widths, so line wrapping must be handled here. + // https://github.com/ia0/data-encoding/blob/4f42ad7ef242f6d243e4de90cd1b46a57690d00e/lib/src/lib.rs#L1710 + let mut line_wrapping = match wrap { + // Line wrapping is disabled because "-w"/"--wrap" was passed with "0" + Some(0) => None, + // A custom line wrapping value was passed + Some(an) => Some(LineWrapping { + line_length: NonZeroUsize::new(an).unwrap(), + print_buffer: Vec::::new(), + }), + // Line wrapping was not set, so the default is used + None => Some(LineWrapping { + line_length: NonZeroUsize::new(WRAP_DEFAULT).unwrap(), + print_buffer: Vec::::new(), + }), + }; - // Encoded data that needs to be written to `output` - let mut encoded_buffer = VecDeque::::new(); - // End of buffers + let input_size = input.len(); - input - .iter() - .enumerate() - .step_by(encode_in_chunks_of_size) - .filter_map(|(idx, _)| { - // The part of `input_buffer` that was actually filled by the call - // to `read` - let buffer = &input[idx..min(input_size, idx + encode_in_chunks_of_size)]; + // Start of buffers + // Data that was read from `input` but has not been encoded yet + let mut leftover_buffer = VecDeque::::new(); - if buffer.len() < encode_in_chunks_of_size { - leftover_buffer.extend(buffer); - assert!(leftover_buffer.len() < encode_in_chunks_of_size); - None - } else { - Some(buffer) - } - }) - .for_each(|read_buffer| { - // Encode data in chunks, then place it in `encoded_buffer` - assert_eq!(read_buffer.len(), encode_in_chunks_of_size); - encode_in_chunks_to_buffer( - supports_fast_decode_and_encode, - read_buffer, - &mut encoded_buffer, - ) - .unwrap(); - // Write all data in `encoded_buffer` to `output` - write_to_output( - &mut line_wrapping, - &mut encoded_buffer, - output, - false, - wrap == Some(0), - ) - .unwrap(); - }); + // Encoded data that needs to be written to `output` + let mut encoded_buffer = VecDeque::::new(); + // End of buffers - // Cleanup - // `input` has finished producing data, so the data remaining in the buffers needs to be encoded and printed - { - // Encode all remaining unencoded bytes, placing them in `encoded_buffer` - supports_fast_decode_and_encode - .encode_to_vec_deque(leftover_buffer.make_contiguous(), &mut encoded_buffer)?; + input + .iter() + .enumerate() + .step_by(encode_in_chunks_of_size) + .filter_map(|(idx, _)| { + // The part of `input_buffer` that was actually filled by the call + // to `read` + let buffer = &input[idx..min(input_size, idx + encode_in_chunks_of_size)]; - // Write all data in `encoded_buffer` to output - // `is_cleanup` triggers special cleanup-only logic - write_to_output( - &mut line_wrapping, - &mut encoded_buffer, - output, - true, - wrap == Some(0), - )?; - } - Ok(()) - } + if buffer.len() < encode_in_chunks_of_size { + leftover_buffer.extend(buffer); + assert!(leftover_buffer.len() < encode_in_chunks_of_size); + None + } else { + Some(buffer) + } + }) + .for_each(|read_buffer| { + // Encode data in chunks, then place it in `encoded_buffer` + assert_eq!(read_buffer.len(), encode_in_chunks_of_size); + encode_in_chunks_to_buffer( + supports_fast_decode_and_encode, + read_buffer, + &mut encoded_buffer, + ) + .unwrap(); + // Write all data in `encoded_buffer` to `output` + write_to_output( + &mut line_wrapping, + &mut encoded_buffer, + output, + false, + wrap == Some(0), + ) + .unwrap(); + }); - /// Encodes all data read from `input` into Base32 using a fast, chunked - /// implementation and writes the result to `output`. - /// - /// The `supports_fast_decode_and_encode` parameter supplies an optimized - /// encoder and determines the chunk size used for bulk processing. When - /// `wrap` is: - /// - `Some(0)`: no line wrapping is performed, - /// - `Some(n)`: lines are wrapped every `n` characters, - /// - `None`: the default wrap width is applied. - /// - /// Remaining bytes are encoded and flushed at the end. I/O or encoding - /// failures are propagated via `UResult`. - pub fn fast_encode_stream( - input: &mut dyn BufRead, - output: &mut dyn Write, - supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode, - wrap: Option, - ) -> UResult<()> { - const ENCODE_IN_CHUNKS_OF_SIZE_MULTIPLE: usize = 1_024; + // Cleanup + // `input` has finished producing data, so the data remaining in the buffers + // needs to be encoded and printed + { + // Encode all remaining unencoded bytes, placing them in `encoded_buffer` + supports_fast_decode_and_encode + .encode_to_vec_deque(leftover_buffer.make_contiguous(), &mut encoded_buffer)?; - let encode_in_chunks_of_size = - supports_fast_decode_and_encode.unpadded_multiple() * ENCODE_IN_CHUNKS_OF_SIZE_MULTIPLE; + // Write all data in `encoded_buffer` to output + // `is_cleanup` triggers special cleanup-only logic + write_to_output(&mut line_wrapping, &mut encoded_buffer, output, true, wrap == Some(0))?; + } + Ok(()) + } - assert!(encode_in_chunks_of_size > 0); + /// Encodes all data read from `input` into Base32 using a fast, chunked + /// implementation and writes the result to `output`. + /// + /// The `supports_fast_decode_and_encode` parameter supplies an optimized + /// encoder and determines the chunk size used for bulk processing. When + /// `wrap` is: + /// - `Some(0)`: no line wrapping is performed, + /// - `Some(n)`: lines are wrapped every `n` characters, + /// - `None`: the default wrap width is applied. + /// + /// Remaining bytes are encoded and flushed at the end. I/O or encoding + /// failures are propagated via `UResult`. + pub fn fast_encode_stream( + input: &mut dyn BufRead, + output: &mut dyn Write, + supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode, + wrap: Option, + ) -> UResult<()> { + const ENCODE_IN_CHUNKS_OF_SIZE_MULTIPLE: usize = 1_024; - let mut line_wrapping = match wrap { - Some(0) => None, - Some(an) => Some(LineWrapping { - line_length: NonZeroUsize::new(an).unwrap(), - print_buffer: Vec::::new(), - }), - None => Some(LineWrapping { - line_length: NonZeroUsize::new(WRAP_DEFAULT).unwrap(), - print_buffer: Vec::::new(), - }), - }; + let encode_in_chunks_of_size = + supports_fast_decode_and_encode.unpadded_multiple() * ENCODE_IN_CHUNKS_OF_SIZE_MULTIPLE; - // Buffers - let mut encoded_buffer = VecDeque::::new(); - let mut leftover_buffer = Vec::::with_capacity(encode_in_chunks_of_size); + assert!(encode_in_chunks_of_size > 0); - loop { - let read_buffer = input - .fill_buf() - .map_err(|err| USimpleError::new(1, super::format_read_error(&err)))?; - if read_buffer.is_empty() { - break; - } + let mut line_wrapping = match wrap { + Some(0) => None, + Some(an) => Some(LineWrapping { + line_length: NonZeroUsize::new(an).unwrap(), + print_buffer: Vec::::new(), + }), + None => Some(LineWrapping { + line_length: NonZeroUsize::new(WRAP_DEFAULT).unwrap(), + print_buffer: Vec::::new(), + }), + }; - let mut consumed = 0; + // Buffers + let mut encoded_buffer = VecDeque::::new(); + let mut leftover_buffer = Vec::::with_capacity(encode_in_chunks_of_size); - if !leftover_buffer.is_empty() { - let needed = encode_in_chunks_of_size - leftover_buffer.len(); - let take = needed.min(read_buffer.len()); - leftover_buffer.extend_from_slice(&read_buffer[..take]); - consumed += take; + loop { + let read_buffer = input + .fill_buf() + .map_err(|err| USimpleError::new(1, super::format_read_error(&err)))?; + if read_buffer.is_empty() { + break; + } - if leftover_buffer.len() == encode_in_chunks_of_size { - encode_in_chunks_to_buffer( - supports_fast_decode_and_encode, - leftover_buffer.as_slice(), - &mut encoded_buffer, - )?; - leftover_buffer.clear(); + let mut consumed = 0; - write_to_output( - &mut line_wrapping, - &mut encoded_buffer, - output, - false, - wrap == Some(0), - )?; - } - } + if !leftover_buffer.is_empty() { + let needed = encode_in_chunks_of_size - leftover_buffer.len(); + let take = needed.min(read_buffer.len()); + leftover_buffer.extend_from_slice(&read_buffer[..take]); + consumed += take; - let remaining = &read_buffer[consumed..]; - let full_chunk_bytes = - (remaining.len() / encode_in_chunks_of_size) * encode_in_chunks_of_size; + if leftover_buffer.len() == encode_in_chunks_of_size { + encode_in_chunks_to_buffer( + supports_fast_decode_and_encode, + leftover_buffer.as_slice(), + &mut encoded_buffer, + )?; + leftover_buffer.clear(); - if full_chunk_bytes > 0 { - for chunk in remaining[..full_chunk_bytes].chunks_exact(encode_in_chunks_of_size) { - encode_in_chunks_to_buffer( - supports_fast_decode_and_encode, - chunk, - &mut encoded_buffer, - )?; - write_to_output( - &mut line_wrapping, - &mut encoded_buffer, - output, - false, - wrap == Some(0), - )?; - } - consumed += full_chunk_bytes; - } + write_to_output( + &mut line_wrapping, + &mut encoded_buffer, + output, + false, + wrap == Some(0), + )?; + } + } - if consumed < read_buffer.len() { - leftover_buffer.extend_from_slice(&read_buffer[consumed..]); - consumed = read_buffer.len(); - } + let remaining = &read_buffer[consumed..]; + let full_chunk_bytes = + (remaining.len() / encode_in_chunks_of_size) * encode_in_chunks_of_size; - input.consume(consumed); + if full_chunk_bytes > 0 { + for chunk in remaining[..full_chunk_bytes].chunks_exact(encode_in_chunks_of_size) { + encode_in_chunks_to_buffer( + supports_fast_decode_and_encode, + chunk, + &mut encoded_buffer, + )?; + write_to_output( + &mut line_wrapping, + &mut encoded_buffer, + output, + false, + wrap == Some(0), + )?; + } + consumed += full_chunk_bytes; + } - // `leftover_buffer` should never exceed one partial chunk. - debug_assert!(leftover_buffer.len() < encode_in_chunks_of_size); - } + if consumed < read_buffer.len() { + leftover_buffer.extend_from_slice(&read_buffer[consumed..]); + consumed = read_buffer.len(); + } - // Encode any remaining bytes and flush - supports_fast_decode_and_encode - .encode_to_vec_deque(&leftover_buffer, &mut encoded_buffer)?; + input.consume(consumed); - write_to_output( - &mut line_wrapping, - &mut encoded_buffer, - output, - true, - wrap == Some(0), - )?; + // `leftover_buffer` should never exceed one partial chunk. + debug_assert!(leftover_buffer.len() < encode_in_chunks_of_size); + } - Ok(()) - } + // Encode any remaining bytes and flush + supports_fast_decode_and_encode.encode_to_vec_deque(&leftover_buffer, &mut encoded_buffer)?; + + write_to_output(&mut line_wrapping, &mut encoded_buffer, output, true, wrap == Some(0))?; + + Ok(()) + } } pub mod fast_decode { - use std::io::{self, BufRead, Write}; - use uucore::{ - encoding::SupportsFastDecodeAndEncode, - error::{UResult, USimpleError}, - }; + use std::io::{self, BufRead, Write}; - // Start of helper functions - fn alphabet_lookup(alphabet: &[u8]) -> [bool; 256] { - // Precompute O(1) membership checks so we can validate every byte before decoding. - let mut table = [false; 256]; + use uucore::{ + encoding::SupportsFastDecodeAndEncode, + error::{UResult, USimpleError}, + }; - for &byte in alphabet { - table[usize::from(byte)] = true; - } + // Start of helper functions + fn alphabet_lookup(alphabet: &[u8]) -> [bool; 256] { + // Precompute O(1) membership checks so we can validate every byte before + // decoding. + let mut table = [false; 256]; - table - } + for &byte in alphabet { + table[usize::from(byte)] = true; + } - fn decode_in_chunks_to_buffer( - supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode, - read_buffer_filtered: &[u8], - decoded_buffer: &mut Vec, - ) -> UResult<()> { - supports_fast_decode_and_encode.decode_into_vec(read_buffer_filtered, decoded_buffer)?; - Ok(()) - } + table + } - fn write_to_output(decoded_buffer: &mut Vec, output: &mut dyn Write) -> io::Result<()> { - // Write all data in `decoded_buffer` to `output` - output.write_all(decoded_buffer.as_slice())?; + fn decode_in_chunks_to_buffer( + supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode, + read_buffer_filtered: &[u8], + decoded_buffer: &mut Vec, + ) -> UResult<()> { + supports_fast_decode_and_encode.decode_into_vec(read_buffer_filtered, decoded_buffer)?; + Ok(()) + } - decoded_buffer.clear(); + fn write_to_output(decoded_buffer: &mut Vec, output: &mut dyn Write) -> io::Result<()> { + // Write all data in `decoded_buffer` to `output` + output.write_all(decoded_buffer.as_slice())?; - Ok(()) - } + decoded_buffer.clear(); - fn flush_ready_chunks( - buffer: &mut Vec, - block_limit: usize, - valid_multiple: usize, - supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode, - decoded_buffer: &mut Vec, - output: &mut dyn Write, - ) -> UResult<()> { - // While at least one full decode block is buffered, keep draining - // it and never yield more than block_limit per chunk. - while buffer.len() >= valid_multiple { - let take = buffer.len().min(block_limit); - let aligned_take = take - (take % valid_multiple); + Ok(()) + } - if aligned_take < valid_multiple { - break; - } + fn flush_ready_chunks( + buffer: &mut Vec, + block_limit: usize, + valid_multiple: usize, + supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode, + decoded_buffer: &mut Vec, + output: &mut dyn Write, + ) -> UResult<()> { + // While at least one full decode block is buffered, keep draining + // it and never yield more than block_limit per chunk. + while buffer.len() >= valid_multiple { + let take = buffer.len().min(block_limit); + let aligned_take = take - (take % valid_multiple); - decode_in_chunks_to_buffer( - supports_fast_decode_and_encode, - &buffer[..aligned_take], - decoded_buffer, - )?; + if aligned_take < valid_multiple { + break; + } - write_to_output(decoded_buffer, output)?; + decode_in_chunks_to_buffer( + supports_fast_decode_and_encode, + &buffer[..aligned_take], + decoded_buffer, + )?; - buffer.drain(..aligned_take); - } + write_to_output(decoded_buffer, output)?; - Ok(()) - } - // End of helper functions + buffer.drain(..aligned_take); + } - pub fn fast_decode_buffer( - input: Vec, - output: &mut dyn Write, - supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode, - ignore_garbage: bool, - ) -> UResult<()> { - const DECODE_IN_CHUNKS_OF_SIZE_MULTIPLE: usize = 1_024; + Ok(()) + } + // End of helper functions - let alphabet = supports_fast_decode_and_encode.alphabet(); - let alphabet_table = alphabet_lookup(alphabet); - let valid_multiple = supports_fast_decode_and_encode.valid_decoding_multiple(); - let decode_in_chunks_of_size = valid_multiple * DECODE_IN_CHUNKS_OF_SIZE_MULTIPLE; + pub fn fast_decode_buffer( + input: Vec, + output: &mut dyn Write, + supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode, + ignore_garbage: bool, + ) -> UResult<()> { + const DECODE_IN_CHUNKS_OF_SIZE_MULTIPLE: usize = 1_024; - assert!(decode_in_chunks_of_size > 0); - assert!(valid_multiple > 0); + let alphabet = supports_fast_decode_and_encode.alphabet(); + let alphabet_table = alphabet_lookup(alphabet); + let valid_multiple = supports_fast_decode_and_encode.valid_decoding_multiple(); + let decode_in_chunks_of_size = valid_multiple * DECODE_IN_CHUNKS_OF_SIZE_MULTIPLE; - // Start of buffers + assert!(decode_in_chunks_of_size > 0); + assert!(valid_multiple > 0); - // Decoded data that needs to be written to `output` - let mut decoded_buffer = Vec::::new(); + // Start of buffers - // End of buffers + // Decoded data that needs to be written to `output` + let mut decoded_buffer = Vec::::new(); - let mut buffer = Vec::with_capacity(decode_in_chunks_of_size); + // End of buffers - let supports_partial_decode = supports_fast_decode_and_encode.supports_partial_decode(); + let mut buffer = Vec::with_capacity(decode_in_chunks_of_size); - for &byte in &input { - if byte == b'\n' || byte == b'\r' { - continue; - } + let supports_partial_decode = supports_fast_decode_and_encode.supports_partial_decode(); - if alphabet_table[usize::from(byte)] { - buffer.push(byte); - } else if ignore_garbage { - continue; - } else { - return Err(USimpleError::new(1, "error: invalid input")); - } + for &byte in &input { + if byte == b'\n' || byte == b'\r' { + continue; + } - if supports_partial_decode { - flush_ready_chunks( - &mut buffer, - decode_in_chunks_of_size, - valid_multiple, - supports_fast_decode_and_encode, - &mut decoded_buffer, - output, - )?; - } else if buffer.len() == decode_in_chunks_of_size { - decode_in_chunks_to_buffer( - supports_fast_decode_and_encode, - &buffer, - &mut decoded_buffer, - )?; - write_to_output(&mut decoded_buffer, output)?; - buffer.clear(); - } - } + if alphabet_table[usize::from(byte)] { + buffer.push(byte); + } else if ignore_garbage { + continue; + } else { + return Err(USimpleError::new(1, "error: invalid input")); + } - if supports_partial_decode { - flush_ready_chunks( - &mut buffer, - decode_in_chunks_of_size, - valid_multiple, - supports_fast_decode_and_encode, - &mut decoded_buffer, - output, - )?; - } + if supports_partial_decode { + flush_ready_chunks( + &mut buffer, + decode_in_chunks_of_size, + valid_multiple, + supports_fast_decode_and_encode, + &mut decoded_buffer, + output, + )?; + } else if buffer.len() == decode_in_chunks_of_size { + decode_in_chunks_to_buffer( + supports_fast_decode_and_encode, + &buffer, + &mut decoded_buffer, + )?; + write_to_output(&mut decoded_buffer, output)?; + buffer.clear(); + } + } - if !buffer.is_empty() { - let mut owned_chunk: Option> = None; - let mut had_invalid_tail = false; + if supports_partial_decode { + flush_ready_chunks( + &mut buffer, + decode_in_chunks_of_size, + valid_multiple, + supports_fast_decode_and_encode, + &mut decoded_buffer, + output, + )?; + } - if let Some(pad_result) = supports_fast_decode_and_encode.pad_remainder(&buffer) { - had_invalid_tail = pad_result.had_invalid_tail; - owned_chunk = Some(pad_result.chunk); - } + if !buffer.is_empty() { + let mut owned_chunk: Option> = None; + let mut had_invalid_tail = false; - let final_chunk = owned_chunk.as_deref().unwrap_or(&buffer); + if let Some(pad_result) = supports_fast_decode_and_encode.pad_remainder(&buffer) { + had_invalid_tail = pad_result.had_invalid_tail; + owned_chunk = Some(pad_result.chunk); + } - supports_fast_decode_and_encode.decode_into_vec(final_chunk, &mut decoded_buffer)?; - write_to_output(&mut decoded_buffer, output)?; + let final_chunk = owned_chunk.as_deref().unwrap_or(&buffer); - if had_invalid_tail { - return Err(USimpleError::new(1, "error: invalid input")); - } - } + supports_fast_decode_and_encode.decode_into_vec(final_chunk, &mut decoded_buffer)?; + write_to_output(&mut decoded_buffer, output)?; - Ok(()) - } + if had_invalid_tail { + return Err(USimpleError::new(1, "error: invalid input")); + } + } - pub fn fast_decode_stream( - input: &mut dyn BufRead, - output: &mut dyn Write, - supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode, - ignore_garbage: bool, - ) -> UResult<()> { - const DECODE_IN_CHUNKS_OF_SIZE_MULTIPLE: usize = 1_024; + Ok(()) + } - let alphabet = supports_fast_decode_and_encode.alphabet(); - let alphabet_table = alphabet_lookup(alphabet); - let valid_multiple = supports_fast_decode_and_encode.valid_decoding_multiple(); - let decode_in_chunks_of_size = valid_multiple * DECODE_IN_CHUNKS_OF_SIZE_MULTIPLE; + pub fn fast_decode_stream( + input: &mut dyn BufRead, + output: &mut dyn Write, + supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode, + ignore_garbage: bool, + ) -> UResult<()> { + const DECODE_IN_CHUNKS_OF_SIZE_MULTIPLE: usize = 1_024; - assert!(decode_in_chunks_of_size > 0); - assert!(valid_multiple > 0); + let alphabet = supports_fast_decode_and_encode.alphabet(); + let alphabet_table = alphabet_lookup(alphabet); + let valid_multiple = supports_fast_decode_and_encode.valid_decoding_multiple(); + let decode_in_chunks_of_size = valid_multiple * DECODE_IN_CHUNKS_OF_SIZE_MULTIPLE; - let supports_partial_decode = supports_fast_decode_and_encode.supports_partial_decode(); + assert!(decode_in_chunks_of_size > 0); + assert!(valid_multiple > 0); - let mut buffer = Vec::with_capacity(decode_in_chunks_of_size); - let mut decoded_buffer = Vec::::new(); + let supports_partial_decode = supports_fast_decode_and_encode.supports_partial_decode(); - loop { - let read_buffer = input - .fill_buf() - .map_err(|err| USimpleError::new(1, super::format_read_error(&err)))?; - let read_len = read_buffer.len(); - if read_len == 0 { - break; - } + let mut buffer = Vec::with_capacity(decode_in_chunks_of_size); + let mut decoded_buffer = Vec::::new(); - for &byte in read_buffer { - if byte == b'\n' || byte == b'\r' { - continue; - } + loop { + let read_buffer = input + .fill_buf() + .map_err(|err| USimpleError::new(1, super::format_read_error(&err)))?; + let read_len = read_buffer.len(); + if read_len == 0 { + break; + } - if alphabet_table[usize::from(byte)] { - buffer.push(byte); - } else if ignore_garbage { - continue; - } else { - if supports_partial_decode { - flush_ready_chunks( - &mut buffer, - decode_in_chunks_of_size, - valid_multiple, - supports_fast_decode_and_encode, - &mut decoded_buffer, - output, - )?; - } else { - while buffer.len() >= decode_in_chunks_of_size { - decode_in_chunks_to_buffer( - supports_fast_decode_and_encode, - &buffer[..decode_in_chunks_of_size], - &mut decoded_buffer, - )?; - write_to_output(&mut decoded_buffer, output)?; - buffer.drain(..decode_in_chunks_of_size); - } - } - return Err(USimpleError::new(1, "error: invalid input")); - } + for &byte in read_buffer { + if byte == b'\n' || byte == b'\r' { + continue; + } - if supports_partial_decode { - flush_ready_chunks( - &mut buffer, - decode_in_chunks_of_size, - valid_multiple, - supports_fast_decode_and_encode, - &mut decoded_buffer, - output, - )?; - } else if buffer.len() == decode_in_chunks_of_size { - decode_in_chunks_to_buffer( - supports_fast_decode_and_encode, - &buffer, - &mut decoded_buffer, - )?; - write_to_output(&mut decoded_buffer, output)?; - buffer.clear(); - } - } + if alphabet_table[usize::from(byte)] { + buffer.push(byte); + } else if ignore_garbage { + continue; + } else { + if supports_partial_decode { + flush_ready_chunks( + &mut buffer, + decode_in_chunks_of_size, + valid_multiple, + supports_fast_decode_and_encode, + &mut decoded_buffer, + output, + )?; + } else { + while buffer.len() >= decode_in_chunks_of_size { + decode_in_chunks_to_buffer( + supports_fast_decode_and_encode, + &buffer[..decode_in_chunks_of_size], + &mut decoded_buffer, + )?; + write_to_output(&mut decoded_buffer, output)?; + buffer.drain(..decode_in_chunks_of_size); + } + } + return Err(USimpleError::new(1, "error: invalid input")); + } - input.consume(read_len); - } + if supports_partial_decode { + flush_ready_chunks( + &mut buffer, + decode_in_chunks_of_size, + valid_multiple, + supports_fast_decode_and_encode, + &mut decoded_buffer, + output, + )?; + } else if buffer.len() == decode_in_chunks_of_size { + decode_in_chunks_to_buffer( + supports_fast_decode_and_encode, + &buffer, + &mut decoded_buffer, + )?; + write_to_output(&mut decoded_buffer, output)?; + buffer.clear(); + } + } - if supports_partial_decode { - flush_ready_chunks( - &mut buffer, - decode_in_chunks_of_size, - valid_multiple, - supports_fast_decode_and_encode, - &mut decoded_buffer, - output, - )?; - } + input.consume(read_len); + } - if !buffer.is_empty() { - let mut owned_chunk: Option> = None; - let mut had_invalid_tail = false; + if supports_partial_decode { + flush_ready_chunks( + &mut buffer, + decode_in_chunks_of_size, + valid_multiple, + supports_fast_decode_and_encode, + &mut decoded_buffer, + output, + )?; + } - if let Some(pad_result) = supports_fast_decode_and_encode.pad_remainder(&buffer) { - had_invalid_tail = pad_result.had_invalid_tail; - owned_chunk = Some(pad_result.chunk); - } + if !buffer.is_empty() { + let mut owned_chunk: Option> = None; + let mut had_invalid_tail = false; - let final_chunk = owned_chunk.as_deref().unwrap_or(&buffer); + if let Some(pad_result) = supports_fast_decode_and_encode.pad_remainder(&buffer) { + had_invalid_tail = pad_result.had_invalid_tail; + owned_chunk = Some(pad_result.chunk); + } - supports_fast_decode_and_encode.decode_into_vec(final_chunk, &mut decoded_buffer)?; - write_to_output(&mut decoded_buffer, output)?; + let final_chunk = owned_chunk.as_deref().unwrap_or(&buffer); - if had_invalid_tail { - return Err(USimpleError::new(1, "error: invalid input")); - } - } + supports_fast_decode_and_encode.decode_into_vec(final_chunk, &mut decoded_buffer)?; + write_to_output(&mut decoded_buffer, output)?; - Ok(()) - } + if had_invalid_tail { + return Err(USimpleError::new(1, "error: invalid input")); + } + } + + Ok(()) + } } fn format_read_error(error: &io::Error) -> String { - format!("read error: {}", strip_errno(error)) + format!("read error: {}", strip_errno(error)) } -/// Determines if the input buffer contains any padding ('=') ignoring trailing whitespace. +/// Determines if the input buffer contains any padding ('=') ignoring trailing +/// whitespace. #[cfg(test)] fn read_and_has_padding(input: &mut R) -> UResult<(bool, Vec)> { - let mut buf = Vec::new(); - input - .read_to_end(&mut buf) - .map_err(|err| USimpleError::new(1, format_read_error(&err)))?; + let mut buf = Vec::new(); + input + .read_to_end(&mut buf) + .map_err(|err| USimpleError::new(1, format_read_error(&err)))?; - // Treat the stream as padded if any '=' exists (GNU coreutils continues decoding - // even when padding bytes are followed by more data). - let has_padding = buf.contains(&b'='); + // Treat the stream as padded if any '=' exists (GNU coreutils continues + // decoding even when padding bytes are followed by more data). + let has_padding = buf.contains(&b'='); - Ok((has_padding, buf)) + Ok((has_padding, buf)) } #[cfg(test)] mod tests { - use crate::base_common::read_and_has_padding; - use std::io::Cursor; + use std::io::Cursor; - #[test] - fn test_has_padding() { - let test_cases = vec![ - ("aGVsbG8sIHdvcmxkIQ==", true), - ("aGVsbG8sIHdvcmxkIQ== ", true), - ("aGVsbG8sIHdvcmxkIQ==\n", true), - ("aGVsbG8sIHdvcmxkIQ== \n", true), - ("aGVsbG8sIHdvcmxkIQ=", true), - ("aGVsbG8sIHdvcmxkIQ= ", true), - ("MTIzNA==MTIzNA", true), - ("MTIzNA==\nMTIzNA", true), - ("aGVsbG8sIHdvcmxkIQ \n", false), - ("aGVsbG8sIHdvcmxkIQ", false), - ]; + use crate::base_common::read_and_has_padding; - for (input, expected) in test_cases { - let mut cursor = Cursor::new(input.as_bytes()); - assert_eq!( - read_and_has_padding(&mut cursor).unwrap().0, - expected, - "Failed for input: '{input}'" - ); - } - } + #[test] + fn test_has_padding() { + let test_cases = vec![ + ("aGVsbG8sIHdvcmxkIQ==", true), + ("aGVsbG8sIHdvcmxkIQ== ", true), + ("aGVsbG8sIHdvcmxkIQ==\n", true), + ("aGVsbG8sIHdvcmxkIQ== \n", true), + ("aGVsbG8sIHdvcmxkIQ=", true), + ("aGVsbG8sIHdvcmxkIQ= ", true), + ("MTIzNA==MTIzNA", true), + ("MTIzNA==\nMTIzNA", true), + ("aGVsbG8sIHdvcmxkIQ \n", false), + ("aGVsbG8sIHdvcmxkIQ", false), + ]; + + for (input, expected) in test_cases { + let mut cursor = Cursor::new(input.as_bytes()); + assert_eq!( + read_and_has_padding(&mut cursor).unwrap().0, + expected, + "Failed for input: '{input}'" + ); + } + } } diff --git a/crates/vendor/uu-base64/src/base64.rs b/crates/vendor/uu-base64/src/base64.rs index fffced8dc..b3753e9f6 100644 --- a/crates/vendor/uu-base64/src/base64.rs +++ b/crates/vendor/uu-base64/src/base64.rs @@ -3,44 +3,49 @@ // For the full copyright and license information, please view the LICENSE // file that was distributed with this source code. +use std::{ffi::OsString, io::Write}; + use clap::Command; -use std::ffi::OsString; -use std::io::Write; use uu_base32::base_common; use uucore::encoding::Format; /// pi-uutils: safe in-process entry point using invocation-scoped streams. pub fn run(argv: Vec) -> i32 { - let matches = match uu_app().try_get_matches_from(argv) { - Ok(matches) => matches, - Err(err) => { - let rendered = err.to_string(); - if err.use_stderr() { - let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); - return 1; - } - let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); - return 0; - } - }; - let result = base_common::Config::from(&matches).and_then(|config| { - let mut input = base_common::get_input(&config)?; - base_common::handle_input(&mut input, Format::Base64, config) - }); - match result { - Ok(()) => pi_uutils_ctx::exit_code(), - Err(err) => { - let code = err.code(); - let _ = writeln!(pi_uutils_ctx::stderr(), "base64: {err}"); - if code == 0 { 1 } else { code } - } - } + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + let result = base_common::Config::from(&matches).and_then(|config| { + let mut input = base_common::get_input(&config)?; + base_common::handle_input(&mut input, Format::Base64, config) + }); + match result { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + let _ = writeln!(pi_uutils_ctx::stderr(), "base64: {err}"); + if code == 0 { 1 } else { code } + }, + } } pub fn uu_app() -> Command { - base_common::base_app( - "encode/decode data and print to standard output\nWith no FILE, or when FILE is -, read standard input.\n\nThe data are encoded as described for the base64 alphabet in RFC 3548.\nWhen decoding, the input may contain newlines in addition to the bytes of the formal base64 alphabet. Use --ignore-garbage to attempt to recover from any other non-alphabet bytes in the encoded stream.".into(), - "base64 [OPTION]... [FILE]".into(), - ) - .name("base64") + base_common::base_app( + "encode/decode data and print to standard output\nWith no FILE, or when FILE is -, read \ + standard input.\n\nThe data are encoded as described for the base64 alphabet in RFC \ + 3548.\nWhen decoding, the input may contain newlines in addition to the bytes of the \ + formal base64 alphabet. Use --ignore-garbage to attempt to recover from any other \ + non-alphabet bytes in the encoded stream." + .into(), + "base64 [OPTION]... [FILE]".into(), + ) + .name("base64") } diff --git a/crates/vendor/uu-basename/src/basename.rs b/crates/vendor/uu-basename/src/basename.rs index be4a431f3..1cd1ae3dc 100644 --- a/crates/vendor/uu-basename/src/basename.rs +++ b/crates/vendor/uu-basename/src/basename.rs @@ -6,25 +6,25 @@ // spell-checker:ignore (ToDO) fullname // pi-uutils: Patched for in-process embedding in the shell. -// All I/O is routed through thread-local stream buffers provided by `pi-uutils-ctx`. -// Command-line arguments are parsed and errors are mapped without process-global -// termination or stdout/stderr pollution. +// All I/O is routed through thread-local stream buffers provided by +// `pi-uutils-ctx`. Command-line arguments are parsed and errors are mapped +// without process-global termination or stdout/stderr pollution. -use clap::builder::ValueParser; -use clap::{Arg, ArgAction, ArgMatches, Command}; -use std::ffi::OsString; -use std::io::Write; -use std::path::PathBuf; -use uucore::display::Quotable; -use uucore::error::{UResult, UUsageError}; +use std::{ffi::OsString, io::Write, path::PathBuf}; + +use clap::{Arg, ArgAction, ArgMatches, Command, builder::ValueParser}; use pi_uutils_ctx::format_usage; -use uucore::line_ending::LineEnding; +use uucore::{ + display::Quotable, + error::{UResult, UUsageError}, + line_ending::LineEnding, +}; pub mod options { - pub static MULTIPLE: &str = "multiple"; - pub static NAME: &str = "name"; - pub static SUFFIX: &str = "suffix"; - pub static ZERO: &str = "zero"; + pub static MULTIPLE: &str = "multiple"; + pub static NAME: &str = "name"; + pub static SUFFIX: &str = "suffix"; + pub static ZERO: &str = "zero"; } /// In-process builtin entry point. Unlike upstream's `uumain`, this parses the @@ -56,117 +56,114 @@ pub fn run(argv: Vec) -> i32 { } fn basename_main(matches: &ArgMatches) -> UResult<()> { - let line_ending = LineEnding::from_zero_flag(matches.get_flag(options::ZERO)); + let line_ending = LineEnding::from_zero_flag(matches.get_flag(options::ZERO)); - let mut name_args = matches - .get_many::(options::NAME) - .unwrap_or_default() - .collect::>(); - if name_args.is_empty() { - return Err(UUsageError::new( - 1, - "missing operand".to_string(), - )); - } - let multiple_paths = matches.get_one::(options::SUFFIX).is_some() - || matches.get_flag(options::MULTIPLE); - let suffix = if multiple_paths { - matches - .get_one::(options::SUFFIX) - .cloned() - .unwrap_or_default() - } else { - // "simple format" - match name_args.len() { - 0 => panic!("already checked"), - 1 => OsString::default(), - 2 => name_args.pop().unwrap().clone(), - _ => { - return Err(UUsageError::new( - 1, - format!("extra operand {}", name_args[2].quote()), - )); - } - } - }; + let mut name_args = matches + .get_many::(options::NAME) + .unwrap_or_default() + .collect::>(); + if name_args.is_empty() { + return Err(UUsageError::new(1, "missing operand".to_string())); + } + let multiple_paths = + matches.get_one::(options::SUFFIX).is_some() || matches.get_flag(options::MULTIPLE); + let suffix = if multiple_paths { + matches + .get_one::(options::SUFFIX) + .cloned() + .unwrap_or_default() + } else { + // "simple format" + match name_args.len() { + 0 => panic!("already checked"), + 1 => OsString::default(), + 2 => name_args.pop().unwrap().clone(), + _ => { + return Err(UUsageError::new(1, format!("extra operand {}", name_args[2].quote()))); + }, + } + }; - // - // Main Program Processing - // - let mut out = pi_uutils_ctx::stdout(); - for path in name_args { - out.write_all(&basename(path, &suffix)?)?; - write!(out, "{line_ending}")?; - } + // + // Main Program Processing + // + let mut out = pi_uutils_ctx::stdout(); + for path in name_args { + out.write_all(&basename(path, &suffix)?)?; + write!(out, "{line_ending}")?; + } - Ok(()) + Ok(()) } pub fn uu_app() -> Command { - Command::new("basename") - .version(uucore::crate_version!()) - .about("Print NAME with any leading directory components removed\nIf specified, also remove a trailing SUFFIX") - .override_usage(format_usage("basename [-z] NAME [SUFFIX]\n basename OPTION... NAME...")) - .infer_long_args(true) - .arg( - Arg::new(options::MULTIPLE) - .short('a') - .long(options::MULTIPLE) - .help("support multiple arguments and treat each as a NAME") - .action(ArgAction::SetTrue) - .overrides_with(options::MULTIPLE), - ) - .arg( - Arg::new(options::NAME) - .action(ArgAction::Append) - .value_parser(ValueParser::os_string()) - .value_hint(clap::ValueHint::AnyPath) - .hide(true) - .trailing_var_arg(true), - ) - .arg( - Arg::new(options::SUFFIX) - .short('s') - .long(options::SUFFIX) - .value_name("SUFFIX") - .value_parser(ValueParser::os_string()) - .help("remove a trailing SUFFIX; implies -a") - .overrides_with(options::SUFFIX), - ) - .arg( - Arg::new(options::ZERO) - .short('z') - .long(options::ZERO) - .help("end each output line with NUL, not newline") - .action(ArgAction::SetTrue) - .overrides_with(options::ZERO), - ) + Command::new("basename") + .version(uucore::crate_version!()) + .about( + "Print NAME with any leading directory components removed\nIf specified, also remove a \ + trailing SUFFIX", + ) + .override_usage(format_usage("basename [-z] NAME [SUFFIX]\n basename OPTION... NAME...")) + .infer_long_args(true) + .arg( + Arg::new(options::MULTIPLE) + .short('a') + .long(options::MULTIPLE) + .help("support multiple arguments and treat each as a NAME") + .action(ArgAction::SetTrue) + .overrides_with(options::MULTIPLE), + ) + .arg( + Arg::new(options::NAME) + .action(ArgAction::Append) + .value_parser(ValueParser::os_string()) + .value_hint(clap::ValueHint::AnyPath) + .hide(true) + .trailing_var_arg(true), + ) + .arg( + Arg::new(options::SUFFIX) + .short('s') + .long(options::SUFFIX) + .value_name("SUFFIX") + .value_parser(ValueParser::os_string()) + .help("remove a trailing SUFFIX; implies -a") + .overrides_with(options::SUFFIX), + ) + .arg( + Arg::new(options::ZERO) + .short('z') + .long(options::ZERO) + .help("end each output line with NUL, not newline") + .action(ArgAction::SetTrue) + .overrides_with(options::ZERO), + ) } // We return a Vec. Returning a seemingly more proper `OsString` would // require back and forth conversions as we need a &[u8] for printing anyway. fn basename(fullname: &OsString, suffix: &OsString) -> UResult> { - let fullname_bytes = uucore::os_str_as_bytes(fullname)?; + let fullname_bytes = uucore::os_str_as_bytes(fullname)?; - // Handle special case where path ends with /. - if fullname_bytes.ends_with(b"/.") { - return Ok(b".".into()); - } + // Handle special case where path ends with /. + if fullname_bytes.ends_with(b"/.") { + return Ok(b".".into()); + } - // Convert to path buffer and get last path component - let pb = PathBuf::from(fullname); + // Convert to path buffer and get last path component + let pb = PathBuf::from(fullname); - pb.components().next_back().map_or(Ok([].into()), |c| { - let name = c.as_os_str(); - let name_bytes = uucore::os_str_as_bytes(name)?; - if name == suffix { - Ok(name_bytes.into()) - } else { - let suffix_bytes = uucore::os_str_as_bytes(suffix)?; - Ok(name_bytes - .strip_suffix(suffix_bytes) - .unwrap_or(name_bytes) - .into()) - } - }) + pb.components().next_back().map_or(Ok([].into()), |c| { + let name = c.as_os_str(); + let name_bytes = uucore::os_str_as_bytes(name)?; + if name == suffix { + Ok(name_bytes.into()) + } else { + let suffix_bytes = uucore::os_str_as_bytes(suffix)?; + Ok(name_bytes + .strip_suffix(suffix_bytes) + .unwrap_or(name_bytes) + .into()) + } + }) } diff --git a/crates/vendor/uu-checksum-common/src/cli.rs b/crates/vendor/uu-checksum-common/src/cli.rs index 6d99b1f9f..513dbefad 100644 --- a/crates/vendor/uu-checksum-common/src/cli.rs +++ b/crates/vendor/uu-checksum-common/src/cli.rs @@ -8,210 +8,214 @@ use uucore::checksum::SUPPORTED_ALGORITHMS; /// List of all options that can be encountered in checksum utils pub mod options { - // cksum-specific - pub const ALGORITHM: &str = "algorithm"; - pub const DEBUG: &str = "debug"; + // cksum-specific + pub const ALGORITHM: &str = "algorithm"; + pub const DEBUG: &str = "debug"; - // positional arg - pub const FILE: &str = "file"; + // positional arg + pub const FILE: &str = "file"; - pub const UNTAGGED: &str = "untagged"; - pub const TAG: &str = "tag"; - pub const LENGTH: &str = "length"; - pub const RAW: &str = "raw"; - pub const BASE64: &str = "base64"; - pub const CHECK: &str = "check"; - pub const TEXT: &str = "text"; - pub const BINARY: &str = "binary"; - pub const ZERO: &str = "zero"; + pub const UNTAGGED: &str = "untagged"; + pub const TAG: &str = "tag"; + pub const LENGTH: &str = "length"; + pub const RAW: &str = "raw"; + pub const BASE64: &str = "base64"; + pub const CHECK: &str = "check"; + pub const TEXT: &str = "text"; + pub const BINARY: &str = "binary"; + pub const ZERO: &str = "zero"; - // check-specific - pub const STRICT: &str = "strict"; - pub const STATUS: &str = "status"; - pub const WARN: &str = "warn"; - pub const IGNORE_MISSING: &str = "ignore-missing"; - pub const QUIET: &str = "quiet"; + // check-specific + pub const STRICT: &str = "strict"; + pub const STATUS: &str = "status"; + pub const WARN: &str = "warn"; + pub const IGNORE_MISSING: &str = "ignore-missing"; + pub const QUIET: &str = "quiet"; } /// `ChecksumCommand` is a convenience trait to more easily declare checksum /// CLI interfaces with pub trait ChecksumCommand { - fn with_algo(self) -> Self; + fn with_algo(self) -> Self; - fn with_length(self) -> Self; + fn with_length(self) -> Self; - fn with_check_and_opts(self) -> Self; + fn with_check_and_opts(self) -> Self; - fn with_binary(self) -> Self; + fn with_binary(self) -> Self; - fn with_text(self, is_default: bool) -> Self; + fn with_text(self, is_default: bool) -> Self; - fn with_tag(self, is_default: bool) -> Self; + fn with_tag(self, is_default: bool) -> Self; - fn with_untagged(self) -> Self; + fn with_untagged(self) -> Self; - fn with_raw(self) -> Self; + fn with_raw(self) -> Self; - fn with_base64(self) -> Self; + fn with_base64(self) -> Self; - fn with_zero(self) -> Self; + fn with_zero(self) -> Self; - fn with_debug(self) -> Self; + fn with_debug(self) -> Self; } impl ChecksumCommand for Command { - fn with_algo(self) -> Self { - self.arg( - Arg::new(options::ALGORITHM) - .long(options::ALGORITHM) - .short('a') - .help("select the digest type to use. See DIGEST below") - .value_name("ALGORITHM") - .value_parser(SUPPORTED_ALGORITHMS), - ) - } + fn with_algo(self) -> Self { + self.arg( + Arg::new(options::ALGORITHM) + .long(options::ALGORITHM) + .short('a') + .help("select the digest type to use. See DIGEST below") + .value_name("ALGORITHM") + .value_parser(SUPPORTED_ALGORITHMS), + ) + } - fn with_length(self) -> Self { - self.arg( - Arg::new(options::LENGTH) - .long(options::LENGTH) - .short('l') - .help("digest length in bits; must not exceed the maximum and must be a multiple of 8 for BLAKE2b") - .action(ArgAction::Set), - ) - } + fn with_length(self) -> Self { + self.arg( + Arg::new(options::LENGTH) + .long(options::LENGTH) + .short('l') + .help( + "digest length in bits; must not exceed the maximum and must be a multiple of 8 \ + for BLAKE2b", + ) + .action(ArgAction::Set), + ) + } - fn with_check_and_opts(self) -> Self { - self.arg( - Arg::new(options::CHECK) - .short('c') - .long(options::CHECK) - .help("read checksums from the FILEs and check them") - .action(ArgAction::SetTrue), - ) - .arg( - Arg::new(options::WARN) - .short('w') - .long("warn") - .help("warn about improperly formatted checksum lines") - .action(ArgAction::SetTrue) - .overrides_with_all([options::STATUS, options::QUIET]), - ) - .arg( - Arg::new(options::STATUS) - .long("status") - .help("don't output anything, status code shows success") - .action(ArgAction::SetTrue) - .overrides_with_all([options::WARN, options::QUIET]), - ) - .arg( - Arg::new(options::QUIET) - .long(options::QUIET) - .help("don't print OK for each successfully verified file") - .action(ArgAction::SetTrue) - .overrides_with_all([options::STATUS, options::WARN]), - ) - .arg( - Arg::new(options::IGNORE_MISSING) - .long(options::IGNORE_MISSING) - .help("don't fail or report status for missing files") - .action(ArgAction::SetTrue), - ) - .arg( - Arg::new(options::STRICT) - .long(options::STRICT) - .help("exit non-zero for improperly formatted checksum lines") - .action(ArgAction::SetTrue), - ) - } + fn with_check_and_opts(self) -> Self { + self + .arg( + Arg::new(options::CHECK) + .short('c') + .long(options::CHECK) + .help("read checksums from the FILEs and check them") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::WARN) + .short('w') + .long("warn") + .help("warn about improperly formatted checksum lines") + .action(ArgAction::SetTrue) + .overrides_with_all([options::STATUS, options::QUIET]), + ) + .arg( + Arg::new(options::STATUS) + .long("status") + .help("don't output anything, status code shows success") + .action(ArgAction::SetTrue) + .overrides_with_all([options::WARN, options::QUIET]), + ) + .arg( + Arg::new(options::QUIET) + .long(options::QUIET) + .help("don't print OK for each successfully verified file") + .action(ArgAction::SetTrue) + .overrides_with_all([options::STATUS, options::WARN]), + ) + .arg( + Arg::new(options::IGNORE_MISSING) + .long(options::IGNORE_MISSING) + .help("don't fail or report status for missing files") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::STRICT) + .long(options::STRICT) + .help("exit non-zero for improperly formatted checksum lines") + .action(ArgAction::SetTrue), + ) + } - fn with_binary(self) -> Self { - self.arg( - Arg::new(options::BINARY) - .long(options::BINARY) - .short('b') - .hide(true) - .overrides_with(options::TEXT) - .action(ArgAction::SetTrue), - ) - } + fn with_binary(self) -> Self { + self.arg( + Arg::new(options::BINARY) + .long(options::BINARY) + .short('b') + .hide(true) + .overrides_with(options::TEXT) + .action(ArgAction::SetTrue), + ) + } - fn with_text(self, is_default: bool) -> Self { - let mut arg = Arg::new(options::TEXT) - .long(options::TEXT) - .short('t') - .action(ArgAction::SetTrue); + fn with_text(self, is_default: bool) -> Self { + let mut arg = Arg::new(options::TEXT) + .long(options::TEXT) + .short('t') + .action(ArgAction::SetTrue); - arg = if is_default { - arg.help("read in text mode (default)") - } else { - arg.hide(true) - }; + arg = if is_default { + arg.help("read in text mode (default)") + } else { + arg.hide(true) + }; - self.arg(arg) - } + self.arg(arg) + } - fn with_tag(self, default: bool) -> Self { - let mut arg = Arg::new(options::TAG) - .long(options::TAG) - .action(ArgAction::SetTrue); + fn with_tag(self, default: bool) -> Self { + let mut arg = Arg::new(options::TAG) + .long(options::TAG) + .action(ArgAction::SetTrue); - arg = if default { - arg.help("create a BSD style checksum (default)") - } else { - arg.help("create a BSD style checksum") - }; + arg = if default { + arg.help("create a BSD style checksum (default)") + } else { + arg.help("create a BSD style checksum") + }; - self.arg(arg) - } + self.arg(arg) + } - fn with_untagged(self) -> Self { - self.arg( - Arg::new(options::UNTAGGED) - .long(options::UNTAGGED) - .help("create a reversed style checksum, without digest type") - .overrides_with(options::TAG) - .action(ArgAction::SetTrue), - ) - } + fn with_untagged(self) -> Self { + self.arg( + Arg::new(options::UNTAGGED) + .long(options::UNTAGGED) + .help("create a reversed style checksum, without digest type") + .overrides_with(options::TAG) + .action(ArgAction::SetTrue), + ) + } - fn with_raw(self) -> Self { - self.arg( - Arg::new(options::RAW) - .long(options::RAW) - .help("emit a raw binary digest, not hexadecimal") - .action(ArgAction::SetTrue), - ) - } + fn with_raw(self) -> Self { + self.arg( + Arg::new(options::RAW) + .long(options::RAW) + .help("emit a raw binary digest, not hexadecimal") + .action(ArgAction::SetTrue), + ) + } - fn with_base64(self) -> Self { - self.arg( - Arg::new(options::BASE64) - .long(options::BASE64) - .help("emit base64-encoded digests, not hexadecimal") - .action(ArgAction::SetTrue) - // Even though this could easily just override an earlier '--raw', - // GNU cksum does not permit these flags to be combined: - .conflicts_with(options::RAW), - ) - } + fn with_base64(self) -> Self { + self.arg( + Arg::new(options::BASE64) + .long(options::BASE64) + .help("emit base64-encoded digests, not hexadecimal") + .action(ArgAction::SetTrue) + // Even though this could easily just override an earlier '--raw', + // GNU cksum does not permit these flags to be combined: + .conflicts_with(options::RAW), + ) + } - fn with_zero(self) -> Self { - self.arg( - Arg::new(options::ZERO) - .long(options::ZERO) - .short('z') - .help("end each output line with NUL, not newline, and disable file name escaping") - .action(ArgAction::SetTrue), - ) - } + fn with_zero(self) -> Self { + self.arg( + Arg::new(options::ZERO) + .long(options::ZERO) + .short('z') + .help("end each output line with NUL, not newline, and disable file name escaping") + .action(ArgAction::SetTrue), + ) + } - fn with_debug(self) -> Self { - self.arg( - Arg::new(options::DEBUG) - .long(options::DEBUG) - .help("print CPU hardware capability detection info used by cksum") - .action(ArgAction::SetTrue), - ) - } + fn with_debug(self) -> Self { + self.arg( + Arg::new(options::DEBUG) + .long(options::DEBUG) + .help("print CPU hardware capability detection info used by cksum") + .action(ArgAction::SetTrue), + ) + } } diff --git a/crates/vendor/uu-checksum-common/src/compute.rs b/crates/vendor/uu-checksum-common/src/compute.rs index e4164bd11..18f6a0bc2 100644 --- a/crates/vendor/uu-checksum-common/src/compute.rs +++ b/crates/vendor/uu-checksum-common/src/compute.rs @@ -5,17 +5,22 @@ // spell-checker:ignore bitlen -use std::ffi::OsStr; -use std::fs::File; -use std::io::{BufReader, Read, Write}; -use std::path::Path; - -use uucore::checksum::{ - AlgoKind, ChecksumError, ReadingMode, SizedAlgoKind, digest_reader, escape_filename, +use std::{ + ffi::OsStr, + fs::File, + io::{BufReader, Read, Write}, + path::Path, }; -use uucore::error::{FromIo, UResult, USimpleError}; -use uucore::line_ending::LineEnding; -use uucore::sum::DigestOutput; + +use uucore::{ + checksum::{ + AlgoKind, ChecksumError, ReadingMode, SizedAlgoKind, digest_reader, escape_filename, + }, + error::{FromIo, UResult, USimpleError}, + line_ending::LineEnding, + sum::DigestOutput, +}; + use crate::report_error; /// Use the same buffer size as GNU when reading a file to create a checksum @@ -28,200 +33,192 @@ const READ_BUFFER_SIZE: usize = 32 * 1024; /// deprecated anyway, it was decided in #9168 to ignore the difference when /// computing the checksum. pub struct ChecksumComputeOptions { - /// Which algorithm to use to compute the digest. - pub algo_kind: SizedAlgoKind, + /// Which algorithm to use to compute the digest. + pub algo_kind: SizedAlgoKind, - /// Printing format to use for each checksum. - pub output_format: OutputFormat, + /// Printing format to use for each checksum. + pub output_format: OutputFormat, - /// Whether to finish lines with '\n' or '\0'. - pub line_ending: LineEnding, + /// Whether to finish lines with '\n' or '\0'. + pub line_ending: LineEnding, } /// Whether to write the digest as hexadecimal or encoded in base64. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum DigestFormat { - Hexadecimal, - Base64, + Hexadecimal, + Base64, } impl DigestFormat { - #[inline] - fn is_base64(self) -> bool { - self == Self::Base64 - } + #[inline] + fn is_base64(self) -> bool { + self == Self::Base64 + } } /// Holds the representation that shall be used for printing a checksum line #[derive(Debug, PartialEq, Eq)] pub enum OutputFormat { - /// Raw digest - Raw, + /// Raw digest + Raw, - /// Selected for older algorithms which had their custom formatting - /// - /// Default for crc, sysv, bsd - Legacy, + /// Selected for older algorithms which had their custom formatting + /// + /// Default for crc, sysv, bsd + Legacy, - /// `$ALGO_NAME ($FILENAME) = $DIGEST` - Tagged(DigestFormat), + /// `$ALGO_NAME ($FILENAME) = $DIGEST` + Tagged(DigestFormat), - /// '$DIGEST $FLAG$FILENAME' - /// where 'flag' depends on the reading mode - /// - /// Default for standalone checksum utilities - Untagged(DigestFormat, ReadingMode), + /// '$DIGEST $FLAG$FILENAME' + /// where 'flag' depends on the reading mode + /// + /// Default for standalone checksum utilities + Untagged(DigestFormat, ReadingMode), } impl OutputFormat { - #[inline] - fn is_raw(&self) -> bool { - *self == Self::Raw - } + #[inline] + fn is_raw(&self) -> bool { + *self == Self::Raw + } - /// Find the correct output format for cksum. - pub fn from_cksum(algo: AlgoKind, tag: bool, binary: bool, raw: bool, base64: bool) -> Self { - // Raw output format takes precedence over anything else. - if raw { - return Self::Raw; - } + /// Find the correct output format for cksum. + pub fn from_cksum(algo: AlgoKind, tag: bool, binary: bool, raw: bool, base64: bool) -> Self { + // Raw output format takes precedence over anything else. + if raw { + return Self::Raw; + } - // Then, if the algo is legacy, takes precedence over the rest - if algo.is_legacy() { - return Self::Legacy; - } + // Then, if the algo is legacy, takes precedence over the rest + if algo.is_legacy() { + return Self::Legacy; + } - let digest_format = if base64 { - DigestFormat::Base64 - } else { - DigestFormat::Hexadecimal - }; + let digest_format = if base64 { + DigestFormat::Base64 + } else { + DigestFormat::Hexadecimal + }; - // After that, decide between tagged and untagged output - if tag { - Self::Tagged(digest_format) - } else { - let reading_mode = if binary { - ReadingMode::Binary - } else { - ReadingMode::Text - }; - Self::Untagged(digest_format, reading_mode) - } - } + // After that, decide between tagged and untagged output + if tag { + Self::Tagged(digest_format) + } else { + let reading_mode = if binary { + ReadingMode::Binary + } else { + ReadingMode::Text + }; + Self::Untagged(digest_format, reading_mode) + } + } - /// Find the correct output format for a standalone checksum util (b2sum, - /// md5sum, etc) - /// - /// Since standalone utils can't use the Raw or Legacy output format, it is - /// decided only using the --tag, --binary and --text arguments. - pub fn from_standalone(text: bool, tag: bool) -> Self { - if tag { - Self::Tagged(DigestFormat::Hexadecimal) - } else { - Self::Untagged( - DigestFormat::Hexadecimal, - if text { - ReadingMode::Text - } else { - ReadingMode::Binary - }, - ) - } - } + /// Find the correct output format for a standalone checksum util (b2sum, + /// md5sum, etc) + /// + /// Since standalone utils can't use the Raw or Legacy output format, it is + /// decided only using the --tag, --binary and --text arguments. + pub fn from_standalone(text: bool, tag: bool) -> Self { + if tag { + Self::Tagged(DigestFormat::Hexadecimal) + } else { + Self::Untagged( + DigestFormat::Hexadecimal, + if text { + ReadingMode::Text + } else { + ReadingMode::Binary + }, + ) + } + } } fn print_legacy_checksum( - options: &ChecksumComputeOptions, - filename: &OsStr, - sum: &DigestOutput, - size: usize, + options: &ChecksumComputeOptions, + filename: &OsStr, + sum: &DigestOutput, + size: usize, ) { - debug_assert!(options.algo_kind.is_legacy()); - debug_assert!(matches!(sum, DigestOutput::U16(_) | DigestOutput::Crc(_))); + debug_assert!(options.algo_kind.is_legacy()); + debug_assert!(matches!(sum, DigestOutput::U16(_) | DigestOutput::Crc(_))); - let (escaped_filename, prefix) = if options.line_ending == LineEnding::Nul { - (filename.to_string_lossy().to_string(), "") - } else { - escape_filename(filename) - }; + let (escaped_filename, prefix) = if options.line_ending == LineEnding::Nul { + (filename.to_string_lossy().to_string(), "") + } else { + escape_filename(filename) + }; - // Print the sum - match (options.algo_kind, sum) { - (SizedAlgoKind::Sysv, DigestOutput::U16(sum)) => { - let _ = write!( - pi_uutils_ctx::stdout(), - "{prefix}{sum} {}", - size.div_ceil(options.algo_kind.bitlen()), - ); - } - (SizedAlgoKind::Bsd, DigestOutput::U16(sum)) => { - // The BSD checksum output is 5 digit integer - let bsd_width = 5; - let _ = write!( - pi_uutils_ctx::stdout(), - "{prefix}{sum:0bsd_width$} {:bsd_width$}", - size.div_ceil(options.algo_kind.bitlen()), - ); - } - (SizedAlgoKind::Crc | SizedAlgoKind::Crc32b, DigestOutput::Crc(sum)) => { - let _ = write!(pi_uutils_ctx::stdout(), "{prefix}{sum} {size}"); - } - (algo, output) => unreachable!("Bug: Invalid legacy checksum ({algo:?}, {output:?})"), - } + // Print the sum + match (options.algo_kind, sum) { + (SizedAlgoKind::Sysv, DigestOutput::U16(sum)) => { + let _ = write!( + pi_uutils_ctx::stdout(), + "{prefix}{sum} {}", + size.div_ceil(options.algo_kind.bitlen()), + ); + }, + (SizedAlgoKind::Bsd, DigestOutput::U16(sum)) => { + // The BSD checksum output is 5 digit integer + let bsd_width = 5; + let _ = write!( + pi_uutils_ctx::stdout(), + "{prefix}{sum:0bsd_width$} {:bsd_width$}", + size.div_ceil(options.algo_kind.bitlen()), + ); + }, + (SizedAlgoKind::Crc | SizedAlgoKind::Crc32b, DigestOutput::Crc(sum)) => { + let _ = write!(pi_uutils_ctx::stdout(), "{prefix}{sum} {size}"); + }, + (algo, output) => unreachable!("Bug: Invalid legacy checksum ({algo:?}, {output:?})"), + } - // Print the filename after a space if not stdin - if escaped_filename != "-" { - let _ = write!(pi_uutils_ctx::stdout(), " "); - let _dropped_result = pi_uutils_ctx::stdout().write_all(escaped_filename.as_bytes()); - } + // Print the filename after a space if not stdin + if escaped_filename != "-" { + let _ = write!(pi_uutils_ctx::stdout(), " "); + let _dropped_result = pi_uutils_ctx::stdout().write_all(escaped_filename.as_bytes()); + } } fn print_tagged_checksum(options: &ChecksumComputeOptions, filename: &OsStr, sum: &String) { - let (escaped_filename, prefix) = if options.line_ending == LineEnding::Nul { - (filename.to_string_lossy().to_string(), "") - } else { - escape_filename(filename) - }; + let (escaped_filename, prefix) = if options.line_ending == LineEnding::Nul { + (filename.to_string_lossy().to_string(), "") + } else { + escape_filename(filename) + }; - // Print algo name and opening parenthesis. - let _ = write!( - pi_uutils_ctx::stdout(), - "{prefix}{} (", - options.algo_kind.to_tag() - ); + // Print algo name and opening parenthesis. + let _ = write!(pi_uutils_ctx::stdout(), "{prefix}{} (", options.algo_kind.to_tag()); - // Print filename - let _dropped_result = pi_uutils_ctx::stdout().write_all(escaped_filename.as_bytes()); + // Print filename + let _dropped_result = pi_uutils_ctx::stdout().write_all(escaped_filename.as_bytes()); - // Print closing parenthesis and sum - let _ = write!(pi_uutils_ctx::stdout(), ") = {sum}"); + // Print closing parenthesis and sum + let _ = write!(pi_uutils_ctx::stdout(), ") = {sum}"); } fn print_untagged_checksum( - options: &ChecksumComputeOptions, - filename: &OsStr, - sum: &String, - reading_mode: ReadingMode, + options: &ChecksumComputeOptions, + filename: &OsStr, + sum: &String, + reading_mode: ReadingMode, ) { - let (escaped_filename, prefix) = if options.line_ending == LineEnding::Nul { - (filename.to_string_lossy().to_string(), "") - } else { - escape_filename(filename) - }; + let (escaped_filename, prefix) = if options.line_ending == LineEnding::Nul { + (filename.to_string_lossy().to_string(), "") + } else { + escape_filename(filename) + }; - // Print checksum and reading mode flag - let _ = write!( - pi_uutils_ctx::stdout(), - "{prefix}{sum} {}", - match reading_mode { - ReadingMode::Binary => '*', - ReadingMode::Text => ' ', - } - ); + // Print checksum and reading mode flag + let _ = write!(pi_uutils_ctx::stdout(), "{prefix}{sum} {}", match reading_mode { + ReadingMode::Binary => '*', + ReadingMode::Text => ' ', + }); - // Print filename - let _dropped_result = pi_uutils_ctx::stdout().write_all(escaped_filename.as_bytes()); + // Print filename + let _dropped_result = pi_uutils_ctx::stdout().write_all(escaped_filename.as_bytes()); } /// Calculate checksum @@ -229,89 +226,86 @@ fn print_untagged_checksum( /// # Arguments /// /// * `options` - CLI options for the assigning checksum algorithm -/// * `files` - A iterator of [`OsStr`] which is a bunch of files that are using for calculating checksum +/// * `files` - A iterator of [`OsStr`] which is a bunch of files that are using +/// for calculating checksum pub fn perform_checksum_computation<'a, I>(options: ChecksumComputeOptions, files: I) -> UResult<()> where - I: Iterator, + I: Iterator, { - let mut files = files.peekable(); + let mut files = files.peekable(); - while let Some(filename) = files.next() { - // Check that in raw mode, we are not provided with several files. - if options.output_format.is_raw() && files.peek().is_some() { - return Err(Box::new(ChecksumError::RawMultipleFiles)); - } + while let Some(filename) = files.next() { + // Check that in raw mode, we are not provided with several files. + if options.output_format.is_raw() && files.peek().is_some() { + return Err(Box::new(ChecksumError::RawMultipleFiles)); + } - let filepath = Path::new(filename); - let resolved_filepath = pi_uutils_ctx::resolve(filepath); - let stdin_buf; - let file_buf; - if resolved_filepath.is_dir() { - report_error(&USimpleError::new(1, format!("{}: Is a directory", filepath.display()))); - continue; - } + let filepath = Path::new(filename); + let resolved_filepath = pi_uutils_ctx::resolve(filepath); + let stdin_buf; + let file_buf; + if resolved_filepath.is_dir() { + report_error(&USimpleError::new(1, format!("{}: Is a directory", filepath.display()))); + continue; + } - // Handle the file input - let mut file = BufReader::with_capacity( - READ_BUFFER_SIZE, - if filename == "-" { - stdin_buf = pi_uutils_ctx::stdin(); - Box::new(stdin_buf) as Box - } else { - file_buf = match File::open(&resolved_filepath) { - Ok(file) => file, - Err(err) => { - report_error(&err.map_err_context(|| filepath.to_string_lossy().into())); - continue; - } - }; - Box::new(file_buf) as Box - }, - ); + // Handle the file input + let mut file = BufReader::with_capacity( + READ_BUFFER_SIZE, + if filename == "-" { + stdin_buf = pi_uutils_ctx::stdin(); + Box::new(stdin_buf) as Box + } else { + file_buf = match File::open(&resolved_filepath) { + Ok(file) => file, + Err(err) => { + report_error(&err.map_err_context(|| filepath.to_string_lossy().into())); + continue; + }, + }; + Box::new(file_buf) as Box + }, + ); - let mut digest = options.algo_kind.create_digest(); + let mut digest = options.algo_kind.create_digest(); - // Always compute the "binary" version of the digest, i.e. on Windows, - // never handle CRLFs specifically. - let (digest_output, sz) = digest_reader(&mut digest, &mut file, ReadingMode::Binary) - .map_err_context(|| "failed to read input".to_string())?; + // Always compute the "binary" version of the digest, i.e. on Windows, + // never handle CRLFs specifically. + let (digest_output, sz) = digest_reader(&mut digest, &mut file, ReadingMode::Binary) + .map_err_context(|| "failed to read input".to_string())?; - // Encodes the sum if df is Base64, leaves as-is otherwise. - let encode_sum = |sum: DigestOutput, df: DigestFormat| { - if df.is_base64() { - sum.to_base64() - } else { - sum.to_hex() - } - }; + // Encodes the sum if df is Base64, leaves as-is otherwise. + let encode_sum = |sum: DigestOutput, df: DigestFormat| { + if df.is_base64() { + sum.to_base64() + } else { + sum.to_hex() + } + }; - match options.output_format { - OutputFormat::Raw => { - // Cannot handle multiple files anyway, output immediately. - digest_output.write_raw(pi_uutils_ctx::stdout())?; - return Ok(()); - } - OutputFormat::Legacy => { - print_legacy_checksum(&options, filename, &digest_output, sz); - } - OutputFormat::Tagged(digest_format) => { - print_tagged_checksum( - &options, - filename, - &encode_sum(digest_output, digest_format)?, - ); - } - OutputFormat::Untagged(digest_format, reading_mode) => { - print_untagged_checksum( - &options, - filename, - &encode_sum(digest_output, digest_format)?, - reading_mode, - ); - } - } + match options.output_format { + OutputFormat::Raw => { + // Cannot handle multiple files anyway, output immediately. + digest_output.write_raw(pi_uutils_ctx::stdout())?; + return Ok(()); + }, + OutputFormat::Legacy => { + print_legacy_checksum(&options, filename, &digest_output, sz); + }, + OutputFormat::Tagged(digest_format) => { + print_tagged_checksum(&options, filename, &encode_sum(digest_output, digest_format)?); + }, + OutputFormat::Untagged(digest_format, reading_mode) => { + print_untagged_checksum( + &options, + filename, + &encode_sum(digest_output, digest_format)?, + reading_mode, + ); + }, + } - let _ = write!(pi_uutils_ctx::stdout(), "{}", options.line_ending); - } - Ok(()) + let _ = write!(pi_uutils_ctx::stdout(), "{}", options.line_ending); + } + Ok(()) } diff --git a/crates/vendor/uu-checksum-common/src/lib.rs b/crates/vendor/uu-checksum-common/src/lib.rs index b197977ce..6b2c0e2a7 100644 --- a/crates/vendor/uu-checksum-common/src/lib.rs +++ b/crates/vendor/uu-checksum-common/src/lib.rs @@ -6,218 +6,236 @@ // pi-uutils: vendored from uutils/coreutils 0.8.0 checksum_common and patched // to use invocation-scoped I/O and cwd resolution for in-process builtins. -use std::borrow::Borrow; -use std::cell::RefCell; -use std::ffi::OsString; -use std::io::Write; +use std::{borrow::Borrow, cell::RefCell, ffi::OsString, io::Write}; -use clap::builder::ValueParser; -use clap::{Arg, ArgAction, ArgMatches, Command, ValueHint}; -use uucore::checksum::{AlgoKind, ChecksumError, SizedAlgoKind}; -use uucore::error::{UError, UResult}; -use uucore::line_ending::LineEnding; +use clap::{Arg, ArgAction, ArgMatches, Command, ValueHint, builder::ValueParser}; +use uucore::{ + checksum::{AlgoKind, ChecksumError, SizedAlgoKind}, + error::{UError, UResult}, + line_ending::LineEnding, +}; mod cli; mod compute; mod validate; -pub use cli::{options, ChecksumCommand}; +pub use cli::{ChecksumCommand, options}; pub use compute::{ChecksumComputeOptions, DigestFormat, OutputFormat}; pub use validate::{ChecksumValidateOptions, ChecksumVerbose}; thread_local! { - static COMMAND_NAME: RefCell<&'static str> = const { RefCell::new("checksum") }; + static COMMAND_NAME: RefCell<&'static str> = const { RefCell::new("checksum") }; } pub(crate) fn command_name() -> &'static str { - COMMAND_NAME.with(|name| *name.borrow()) + COMMAND_NAME.with(|name| *name.borrow()) } pub(crate) fn report_error(error: &dyn std::fmt::Display) { - let _ = writeln!(pi_uutils_ctx::stderr(), "{}: {error}", command_name()); - pi_uutils_ctx::set_exit_code(1); + let _ = writeln!(pi_uutils_ctx::stderr(), "{}: {error}", command_name()); + pi_uutils_ctx::set_exit_code(1); } pub(crate) fn report_warning(message: &str) { - let _ = writeln!(pi_uutils_ctx::stderr(), "{}: {message}", command_name()); + let _ = writeln!(pi_uutils_ctx::stderr(), "{}: {message}", command_name()); } /// Generate a context-safe standalone checksum wrapper. #[macro_export] macro_rules! declare_standalone { - ($bin:literal, $kind:expr) => { - pub fn run(argv: Vec<::std::ffi::OsString>) -> i32 { - ::uu_checksum_common::run_standalone($bin, $kind, uu_app(), argv) - } + ($bin:literal, $kind:expr) => { + pub fn run(argv: Vec<::std::ffi::OsString>) -> i32 { + ::uu_checksum_common::run_standalone($bin, $kind, uu_app(), argv) + } - #[inline] - pub fn uu_app() -> ::clap::Command { - let (about, usage) = ::uu_checksum_common::standalone_strings($bin); - ::uu_checksum_common::standalone_checksum_app(about, usage).name($bin) - } - }; + #[inline] + pub fn uu_app() -> ::clap::Command { + let (about, usage) = ::uu_checksum_common::standalone_strings($bin); + ::uu_checksum_common::standalone_checksum_app(about, usage).name($bin) + } + }; } /// English descriptions used by standalone wrappers (localization is /// intentionally literalized because embedded commands have no global locale). pub fn standalone_strings(bin: &str) -> (&'static str, &'static str) { - match bin { - "md5sum" => ("Print or check the MD5 checksums", "md5sum [OPTIONS] [FILE]..."), - "sha1sum" => ("Print or check SHA1 (160-bit) checksums", "sha1sum [OPTION]... [FILE]..."), - "sha224sum" => ("Print or check SHA224 (224-bit) checksums", "sha224sum [OPTION]... [FILE]..."), - "sha256sum" => ("Print or check SHA256 (256-bit) checksums", "sha256sum [OPTION]... [FILE]..."), - "sha384sum" => ("Print or check SHA384 (384-bit) checksums", "sha384sum [OPTION]... [FILE]..."), - "sha512sum" => ("Print or check SHA512 (512-bit) checksums", "sha512sum [OPTION]... [FILE]..."), - "b2sum" => ("Print or check BLAKE2b (512-bit) checksums", "b2sum [OPTION]... [FILE]..."), - _ => ("Print or check checksums", "checksum [OPTION]... [FILE]..."), - } + match bin { + "md5sum" => ("Print or check the MD5 checksums", "md5sum [OPTIONS] [FILE]..."), + "sha1sum" => ("Print or check SHA1 (160-bit) checksums", "sha1sum [OPTION]... [FILE]..."), + "sha224sum" => { + ("Print or check SHA224 (224-bit) checksums", "sha224sum [OPTION]... [FILE]...") + }, + "sha256sum" => { + ("Print or check SHA256 (256-bit) checksums", "sha256sum [OPTION]... [FILE]...") + }, + "sha384sum" => { + ("Print or check SHA384 (384-bit) checksums", "sha384sum [OPTION]... [FILE]...") + }, + "sha512sum" => { + ("Print or check SHA512 (512-bit) checksums", "sha512sum [OPTION]... [FILE]...") + }, + "b2sum" => ("Print or check BLAKE2b (512-bit) checksums", "b2sum [OPTION]... [FILE]..."), + _ => ("Print or check checksums", "checksum [OPTION]... [FILE]..."), + } } -pub fn run_standalone( - bin: &'static str, - algo: AlgoKind, - cmd: Command, - argv: Vec, -) -> i32 { - run_with_optional_length(bin, algo, cmd, argv, None) +pub fn run_standalone(bin: &'static str, algo: AlgoKind, cmd: Command, argv: Vec) -> i32 { + run_with_optional_length(bin, algo, cmd, argv, None) } /// Context-safe entrypoint for b2sum and other standalone hashes supporting /// `--length`. The validator is applied only when that option is present. pub fn run_standalone_with_length( - bin: &'static str, - algo: AlgoKind, - cmd: Command, - argv: Vec, - validate_len: fn(&str) -> UResult, + bin: &'static str, + algo: AlgoKind, + cmd: Command, + argv: Vec, + validate_len: fn(&str) -> UResult, ) -> i32 { - run_with_optional_length(bin, algo, cmd, argv, Some(validate_len)) + run_with_optional_length(bin, algo, cmd, argv, Some(validate_len)) } fn run_with_optional_length( - bin: &'static str, - algo: AlgoKind, - cmd: Command, - argv: Vec, - validate_len: Option UResult>, + bin: &'static str, + algo: AlgoKind, + cmd: Command, + argv: Vec, + validate_len: Option UResult>, ) -> i32 { - COMMAND_NAME.with(|name| *name.borrow_mut() = bin); - let matches = match cmd.try_get_matches_from(argv) { - Ok(matches) => matches, - Err(err) => { - let rendered = err.to_string(); - if err.use_stderr() { - let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); - return 2; - } - let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); - return 0; - } - }; - let length = match validate_len { - Some(validate_len) => match matches - .get_one::(options::LENGTH) - .map(String::as_str) - .map(validate_len) - .transpose() - { - Ok(length) => length, - Err(err) => return finish_error(bin, err), - }, - None => None, - }; - let text = !matches.get_flag(options::BINARY); - let tag = matches.get_flag(options::TAG); - let format = OutputFormat::from_standalone(text, tag); - match checksum_main(Some(algo), length, matches, format) { - Ok(()) => pi_uutils_ctx::exit_code(), - Err(err) => finish_error(bin, err), - } + COMMAND_NAME.with(|name| *name.borrow_mut() = bin); + let matches = match cmd.try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 2; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + let length = match validate_len { + Some(validate_len) => match matches + .get_one::(options::LENGTH) + .map(String::as_str) + .map(validate_len) + .transpose() + { + Ok(length) => length, + Err(err) => return finish_error(bin, err), + }, + None => None, + }; + let text = !matches.get_flag(options::BINARY); + let tag = matches.get_flag(options::TAG); + let format = OutputFormat::from_standalone(text, tag); + match checksum_main(Some(algo), length, matches, format) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => finish_error(bin, err), + } } fn finish_error(bin: &str, err: Box) -> i32 { - let code = err.code(); - let message = err.to_string(); - if !message.is_empty() { - let _ = writeln!(pi_uutils_ctx::stderr(), "{bin}: {message}"); - } - if code == 0 { 1 } else { code } + let code = err.code(); + let message = err.to_string(); + if !message.is_empty() { + let _ = writeln!(pi_uutils_ctx::stderr(), "{bin}: {message}"); + } + if code == 0 { 1 } else { code } } pub fn default_checksum_app(about: impl Into, usage: impl Into) -> Command { - Command::new("") - .version("0.8.0") - .about(about.into()) - .override_usage(usage.into()) - .infer_long_args(true) - .args_override_self(true) - .after_help("With no FILE or when FILE is -, read standard input") - .arg( - Arg::new(options::FILE) - .hide(true) - .action(ArgAction::Append) - .value_parser(ValueParser::os_string()) - .default_value("-") - .hide_default_value(true) - .value_hint(ValueHint::FilePath), - ) + Command::new("") + .version("0.8.0") + .about(about.into()) + .override_usage(usage.into()) + .infer_long_args(true) + .args_override_self(true) + .after_help("With no FILE or when FILE is -, read standard input") + .arg( + Arg::new(options::FILE) + .hide(true) + .action(ArgAction::Append) + .value_parser(ValueParser::os_string()) + .default_value("-") + .hide_default_value(true) + .value_hint(ValueHint::FilePath), + ) } pub fn standalone_checksum_app_with_length( - about: impl Into, - usage: impl Into, + about: impl Into, + usage: impl Into, ) -> Command { - default_checksum_app(about, usage) - .with_binary().with_check_and_opts().with_length().with_tag(false).with_text(true).with_zero() + default_checksum_app(about, usage) + .with_binary() + .with_check_and_opts() + .with_length() + .with_tag(false) + .with_text(true) + .with_zero() } -pub fn standalone_checksum_app( - about: impl Into, - usage: impl Into, -) -> Command { - default_checksum_app(about, usage) - .with_binary().with_check_and_opts().with_tag(false).with_text(true).with_zero() +pub fn standalone_checksum_app(about: impl Into, usage: impl Into) -> Command { + default_checksum_app(about, usage) + .with_binary() + .with_check_and_opts() + .with_tag(false) + .with_text(true) + .with_zero() } pub fn checksum_main( - algo: Option, - length: Option, - matches: ArgMatches, - output_format: OutputFormat, + algo: Option, + length: Option, + matches: ArgMatches, + output_format: OutputFormat, ) -> UResult<()> { - let check = matches.get_flag(options::CHECK); - let check_flag = |flag| match (check, matches.get_flag(flag)) { - (_, false) => Ok(false), - (true, true) => Ok(true), - (false, true) => Err(ChecksumError::CheckOnlyFlag(flag.into())), - }; - let ignore_missing = check_flag(options::IGNORE_MISSING)?; - let warn = check_flag(options::WARN)?; - let quiet = check_flag(options::QUIET)?; - let strict = check_flag(options::STRICT)?; - let status = check_flag(options::STATUS)?; - let text_flag = matches.get_flag(options::TEXT); - let binary_flag = matches.get_flag(options::BINARY); - let tag = matches.get_flag(options::TAG); - let files = matches.get_many::(options::FILE).unwrap().map(Borrow::borrow); + let check = matches.get_flag(options::CHECK); + let check_flag = |flag| match (check, matches.get_flag(flag)) { + (_, false) => Ok(false), + (true, true) => Ok(true), + (false, true) => Err(ChecksumError::CheckOnlyFlag(flag.into())), + }; + let ignore_missing = check_flag(options::IGNORE_MISSING)?; + let warn = check_flag(options::WARN)?; + let quiet = check_flag(options::QUIET)?; + let strict = check_flag(options::STRICT)?; + let status = check_flag(options::STATUS)?; + let text_flag = matches.get_flag(options::TEXT); + let binary_flag = matches.get_flag(options::BINARY); + let tag = matches.get_flag(options::TAG); + let files = matches + .get_many::(options::FILE) + .unwrap() + .map(Borrow::borrow); - if text_flag && tag { return Err(ChecksumError::TextAfterTag.into()); } - if check { - if algo.is_some_and(AlgoKind::is_legacy) { return Err(ChecksumError::AlgorithmNotSupportedWithCheck.into()); } - if tag { return Err(ChecksumError::TagCheck.into()); } - if binary_flag || text_flag { return Err(ChecksumError::BinaryTextConflict.into()); } - let opts = ChecksumValidateOptions { - ignore_missing, - strict, - verbose: ChecksumVerbose::new(status, quiet, warn), - }; - return validate::perform_checksum_validation(files, algo, length, opts); - } + if text_flag && tag { + return Err(ChecksumError::TextAfterTag.into()); + } + if check { + if algo.is_some_and(AlgoKind::is_legacy) { + return Err(ChecksumError::AlgorithmNotSupportedWithCheck.into()); + } + if tag { + return Err(ChecksumError::TagCheck.into()); + } + if binary_flag || text_flag { + return Err(ChecksumError::BinaryTextConflict.into()); + } + let opts = ChecksumValidateOptions { + ignore_missing, + strict, + verbose: ChecksumVerbose::new(status, quiet, warn), + }; + return validate::perform_checksum_validation(files, algo, length, opts); + } - let algo = SizedAlgoKind::from_unsized(algo.unwrap_or(AlgoKind::Crc), length)?; - let opts = ChecksumComputeOptions { - algo_kind: algo, - output_format, - line_ending: LineEnding::from_zero_flag(matches.get_flag(options::ZERO)), - }; - compute::perform_checksum_computation(opts, files) + let algo = SizedAlgoKind::from_unsized(algo.unwrap_or(AlgoKind::Crc), length)?; + let opts = ChecksumComputeOptions { + algo_kind: algo, + output_format, + line_ending: LineEnding::from_zero_flag(matches.get_flag(options::ZERO)), + }; + compute::perform_checksum_computation(opts, files) } diff --git a/crates/vendor/uu-checksum-common/src/validate.rs b/crates/vendor/uu-checksum-common/src/validate.rs index 8ab1ceeb8..36281207c 100644 --- a/crates/vendor/uu-checksum-common/src/validate.rs +++ b/crates/vendor/uu-checksum-common/src/validate.rs @@ -3,993 +3,1002 @@ // For the full copyright and license information, please view the LICENSE // file that was distributed with this source code. -// spell-checker:ignore rsplit hexdigit bitlen invalidchecksum inva idchecksum xffname +// spell-checker:ignore rsplit hexdigit bitlen invalidchecksum inva idchecksum +// xffname -use std::ffi::OsStr; -use std::fmt::Display; -use std::fs::File; -use std::io::{self, BufReader, Read, Write}; +use std::{ + ffi::OsStr, + fmt::Display, + fs::File, + io::{self, BufReader, Read, Write}, +}; use os_display::Quotable; - -use uucore::checksum::{ - AlgoKind, BlakeLength, ChecksumError, ReadingMode, ShaLength, SizedAlgoKind, digest_reader, - parse_blake_length, unescape_filename, +use uucore::{ + checksum::{ + AlgoKind, BlakeLength, ChecksumError, ReadingMode, ShaLength, SizedAlgoKind, digest_reader, + parse_blake_length, unescape_filename, + }, + error::{FromIo, UError, UIoError, UResult, USimpleError}, + os_str_as_bytes, os_str_from_bytes, + quoting_style::{QuotingStyle, locale_aware_escape_name}, + read_os_string_lines, + sum::{self, Blake2b, Blake3, DigestOutput}, }; -use uucore::error::{FromIo, UError, UIoError, UResult, USimpleError}; -use uucore::quoting_style::{QuotingStyle, locale_aware_escape_name}; -use uucore::sum::{self, Blake2b, Blake3, DigestOutput}; -use uucore::{os_str_as_bytes, os_str_from_bytes, read_os_string_lines}; + use crate::{command_name, report_error, report_warning}; /// To what level should checksum validation print logging info. #[derive(Debug, PartialEq, Eq, PartialOrd, Clone, Copy, Default)] pub enum ChecksumVerbose { - Status, - Quiet, - #[default] - Normal, - Warning, + Status, + Quiet, + #[default] + Normal, + Warning, } impl ChecksumVerbose { - pub fn new(status: bool, quiet: bool, warn: bool) -> Self { - use ChecksumVerbose::*; + pub fn new(status: bool, quiet: bool, warn: bool) -> Self { + use ChecksumVerbose::*; - // Assume only one of the three booleans will be enabled at once. - // This is ensured by clap's overriding arguments. - match (status, quiet, warn) { - (true, _, _) => Status, - (_, true, _) => Quiet, - (_, _, true) => Warning, - _ => Normal, - } - } + // Assume only one of the three booleans will be enabled at once. + // This is ensured by clap's overriding arguments. + match (status, quiet, warn) { + (true, ..) => Status, + (_, true, _) => Quiet, + (_, _, true) => Warning, + _ => Normal, + } + } - #[inline] - pub fn over_status(self) -> bool { - self > Self::Status - } + #[inline] + pub fn over_status(self) -> bool { + self > Self::Status + } - #[inline] - pub fn over_quiet(self) -> bool { - self > Self::Quiet - } + #[inline] + pub fn over_quiet(self) -> bool { + self > Self::Quiet + } - #[inline] - pub fn at_least_warning(self) -> bool { - self >= Self::Warning - } + #[inline] + pub fn at_least_warning(self) -> bool { + self >= Self::Warning + } } /// This struct regroups CLI flags. #[derive(Debug, Default, Clone, Copy)] pub struct ChecksumValidateOptions { - pub ignore_missing: bool, - pub strict: bool, - pub verbose: ChecksumVerbose, + pub ignore_missing: bool, + pub strict: bool, + pub verbose: ChecksumVerbose, } /// This structure holds the count of checksum test lines' outcomes. #[derive(Default)] struct ChecksumResult { - /// Number of lines in the file where the computed checksum MATCHES - /// the expectation. - pub correct: u32, - /// Number of lines in the file where the computed checksum DIFFERS - /// from the expectation. - pub failed_cksum: u32, - pub failed_open_file: u32, - /// Number of improperly formatted lines. - pub bad_format: u32, - /// Total number of non-empty, non-comment lines. - pub total: u32, + /// Number of lines in the file where the computed checksum MATCHES + /// the expectation. + pub correct: u32, + /// Number of lines in the file where the computed checksum DIFFERS + /// from the expectation. + pub failed_cksum: u32, + pub failed_open_file: u32, + /// Number of improperly formatted lines. + pub bad_format: u32, + /// Total number of non-empty, non-comment lines. + pub total: u32, } impl ChecksumResult { - #[inline] - fn total_properly_formatted(&self) -> u32 { - self.total - self.bad_format - } + #[inline] + fn total_properly_formatted(&self) -> u32 { + self.total - self.bad_format + } } /// Represents a reason for which the processing of a checksum line /// could not proceed to digest comparison. enum LineCheckError { - /// a generic UError was encountered in sub-functions - UError(Box), - /// the computed checksum digest differs from the expected one - DigestMismatch, - /// the line is empty or is a comment - Skipped, - /// the line has a formatting error - ImproperlyFormatted, - /// file exists but is impossible to read - CantOpenFile, - /// there is nothing at the given path - FileNotFound, - /// the given path leads to a directory - FileIsDirectory, + /// a generic UError was encountered in sub-functions + UError(Box), + /// the computed checksum digest differs from the expected one + DigestMismatch, + /// the line is empty or is a comment + Skipped, + /// the line has a formatting error + ImproperlyFormatted, + /// file exists but is impossible to read + CantOpenFile, + /// there is nothing at the given path + FileNotFound, + /// the given path leads to a directory + FileIsDirectory, } impl From> for LineCheckError { - fn from(value: Box) -> Self { - Self::UError(value) - } + fn from(value: Box) -> Self { + Self::UError(value) + } } impl From for LineCheckError { - fn from(value: ChecksumError) -> Self { - Self::UError(Box::new(value)) - } + fn from(value: ChecksumError) -> Self { + Self::UError(Box::new(value)) + } } /// Represents an error that was encountered when processing a checksum file. enum FileCheckError { - /// a generic UError was encountered in sub-functions - UError(Box), - /// reading of the checksum file failed - CantOpenChecksumFile, - /// processing of the file is considered as a failure regarding the - /// provided flags. This however does not stop the processing of - /// further files. - Failed, + /// a generic UError was encountered in sub-functions + UError(Box), + /// reading of the checksum file failed + CantOpenChecksumFile, + /// processing of the file is considered as a failure regarding the + /// provided flags. This however does not stop the processing of + /// further files. + Failed, } impl From> for FileCheckError { - fn from(value: Box) -> Self { - Self::UError(value) - } + fn from(value: Box) -> Self { + Self::UError(value) + } } impl From for FileCheckError { - fn from(value: ChecksumError) -> Self { - Self::UError(Box::new(value)) - } + fn from(value: ChecksumError) -> Self { + Self::UError(Box::new(value)) + } } fn print_cksum_report(res: &ChecksumResult) { - if res.bad_format > 0 { - report_warning(&format!("WARNING: {} line(s) are improperly formatted", res.bad_format)); - } + if res.bad_format > 0 { + report_warning(&format!("WARNING: {} line(s) are improperly formatted", res.bad_format)); + } - if res.failed_cksum > 0 { - report_warning(&format!("WARNING: {} computed checksum(s) did NOT match", res.failed_cksum)); - } + if res.failed_cksum > 0 { + report_warning(&format!("WARNING: {} computed checksum(s) did NOT match", res.failed_cksum)); + } - if res.failed_open_file > 0 { - report_warning(&format!("WARNING: {} listed file(s) could not be read", res.failed_open_file)); - } + if res.failed_open_file > 0 { + report_warning(&format!( + "WARNING: {} listed file(s) could not be read", + res.failed_open_file + )); + } } /// Print a "no properly formatted lines" message in stderr #[inline] fn log_no_properly_formatted(filename: impl Display) { - let _ = writeln!( - pi_uutils_ctx::stderr(), - "{}: {}", - command_name(), - format!("{}: no properly formatted checksum lines found", filename) - ); + let _ = writeln!( + pi_uutils_ctx::stderr(), + "{}: {filename}: no properly formatted checksum lines found", + command_name() + ); } /// Print a "no file was verified" message in stderr #[inline] fn log_no_file_verified(filename: impl Display) { - let _ = writeln!( - pi_uutils_ctx::stderr(), - "{}: {}", - command_name(), - format!("{}: no file was verified", filename) - ); + let _ = + writeln!(pi_uutils_ctx::stderr(), "{}: {filename}: no file was verified", command_name()); } /// Represents the different outcomes that can happen to a file /// that is being checked. #[derive(Debug, Clone, Copy)] enum FileChecksumResult { - Ok, - Failed, - CantOpen, + Ok, + Failed, + CantOpen, } impl FileChecksumResult { - /// Creates a `FileChecksumResult` from a digest comparison that - /// either succeeded or failed. - fn from_bool(checksum_correct: bool) -> Self { - if checksum_correct { - Self::Ok - } else { - Self::Failed - } - } + /// Creates a `FileChecksumResult` from a digest comparison that + /// either succeeded or failed. + fn from_bool(checksum_correct: bool) -> Self { + if checksum_correct { + Self::Ok + } else { + Self::Failed + } + } - /// The cli options might prevent to display on the outcome of the - /// comparison on STDOUT. - fn can_display(self, verbose: ChecksumVerbose) -> bool { - match self { - Self::Ok => verbose.over_quiet(), - Self::Failed => verbose.over_status(), - Self::CantOpen => true, - } - } + /// The cli options might prevent to display on the outcome of the + /// comparison on STDOUT. + fn can_display(self, verbose: ChecksumVerbose) -> bool { + match self { + Self::Ok => verbose.over_quiet(), + Self::Failed => verbose.over_status(), + Self::CantOpen => true, + } + } } impl Display for FileChecksumResult { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - Self::Ok => write!(f, "OK"), - Self::Failed => write!(f, "FAILED"), - Self::CantOpen => write!(f, "FAILED open or read"), - } - } + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Ok => write!(f, "OK"), + Self::Failed => write!(f, "FAILED"), + Self::CantOpen => write!(f, "FAILED open or read"), + } + } } /// Write to the given buffer the checksum validation status of a file which /// name might contain non-utf-8 characters. fn write_file_report( - mut w: W, - filename: &[u8], - result: FileChecksumResult, - prefix: &str, - verbose: ChecksumVerbose, + mut w: W, + filename: &[u8], + result: FileChecksumResult, + prefix: &str, + verbose: ChecksumVerbose, ) { - if result.can_display(verbose) { - let _ = write!(w, "{prefix}"); - let _ = w.write_all(filename); - let _ = writeln!(w, ": {result}"); - } + if result.can_display(verbose) { + let _ = write!(w, "{prefix}"); + let _ = w.write_all(filename); + let _ = writeln!(w, ": {result}"); + } } #[derive(Debug, PartialEq, Eq, Clone, Copy)] enum LineFormat { - AlgoBased, - SingleSpace, - Untagged, + AlgoBased, + SingleSpace, + Untagged, } impl LineFormat { - /// parse [tagged output format] - /// Normally the format is simply space separated but openssl does not - /// respect the gnu definition. - /// - /// [tagged output format]: https://www.gnu.org/software/coreutils/manual/html_node/cksum-output-modes.html#cksum-output-modes-1 - fn parse_algo_based(line: &[u8]) -> Option { - // r"\MD5 (a\\ b) = abc123", - // BLAKE2b(44)= a45a4c4883cce4b50d844fab460414cc2080ca83690e74d850a9253e757384366382625b218c8585daee80f34dc9eb2f2fde5fb959db81cd48837f9216e7b0fa - let trimmed = line.trim_ascii_start(); - let algo_start = usize::from(trimmed.starts_with(b"\\")); - let rest = &trimmed[algo_start..]; + /// parse [tagged output format] + /// Normally the format is simply space separated but openssl does not + /// respect the gnu definition. + /// + /// [tagged output format]: https://www.gnu.org/software/coreutils/manual/html_node/cksum-output-modes.html#cksum-output-modes-1 + fn parse_algo_based(line: &[u8]) -> Option { + // r"\MD5 (a\\ b) = abc123", + // BLAKE2b(44)= + // a45a4c4883cce4b50d844fab460414cc2080ca83690e74d850a9253e757384366382625b218c8585daee80f34dc9eb2f2fde5fb959db81cd48837f9216e7b0fa + let trimmed = line.trim_ascii_start(); + let algo_start = usize::from(trimmed.starts_with(b"\\")); + let rest = &trimmed[algo_start..]; - enum SubCase { - Posix, - OpenSSL, - } - // find the next parenthesis using byte search (not next whitespace) because openssl's - // tagged format does not put a space before (filename) + enum SubCase { + Posix, + OpenSSL, + } + // find the next parenthesis using byte search (not next whitespace) because + // openssl's tagged format does not put a space before (filename) - let par_idx = rest.iter().position(|&b| b == b'(')?; - let sub_case = if rest[par_idx - 1] == b' ' { - SubCase::Posix - } else { - SubCase::OpenSSL - }; + let par_idx = rest.iter().position(|&b| b == b'(')?; + let sub_case = if rest[par_idx - 1] == b' ' { + SubCase::Posix + } else { + SubCase::OpenSSL + }; - let algo_substring = match sub_case { - SubCase::Posix => &rest[..par_idx - 1], - SubCase::OpenSSL => &rest[..par_idx], - }; - let mut algo_parts = algo_substring.splitn(2, |&b| b == b'-'); - let algo = algo_parts.next()?; + let algo_substring = match sub_case { + SubCase::Posix => &rest[..par_idx - 1], + SubCase::OpenSSL => &rest[..par_idx], + }; + let mut algo_parts = algo_substring.splitn(2, |&b| b == b'-'); + let algo = algo_parts.next()?; - // Parse algo_bits if present - let algo_bits = algo_parts - .next() - .and_then(|s| std::str::from_utf8(s).ok()?.parse::().ok()); + // Parse algo_bits if present + let algo_bits = algo_parts + .next() + .and_then(|s| std::str::from_utf8(s).ok()?.parse::().ok()); - // Check algo format: uppercase ASCII or digits or "BLAKE2b" - let is_valid_algo = algo == b"BLAKE2b" - || algo - .iter() - .all(|&b| b.is_ascii_uppercase() || b.is_ascii_digit()); - if !is_valid_algo { - return None; - } - // SAFETY: we just validated the contents of algo, we can unsafely make a - // String from it - let algo_utf8 = unsafe { String::from_utf8_unchecked(algo.to_vec()) }; - // stripping '(' not ' (' since we matched on ( not whitespace because of openssl. - let after_paren = rest.get(par_idx + 1..)?; - let (filename, checksum) = match sub_case { - SubCase::Posix => ByteSliceExt::rsplit_once(after_paren, b") = ")?, - SubCase::OpenSSL => ByteSliceExt::rsplit_once(after_paren, b")= ")?, - }; + // Check algo format: uppercase ASCII or digits or "BLAKE2b" + let is_valid_algo = algo == b"BLAKE2b" + || algo + .iter() + .all(|&b| b.is_ascii_uppercase() || b.is_ascii_digit()); + if !is_valid_algo { + return None; + } + // SAFETY: we just validated the contents of algo, we can unsafely make a + // String from it + let algo_utf8 = unsafe { String::from_utf8_unchecked(algo.to_vec()) }; + // stripping '(' not ' (' since we matched on ( not whitespace because of + // openssl. + let after_paren = rest.get(par_idx + 1..)?; + let (filename, checksum) = match sub_case { + SubCase::Posix => ByteSliceExt::rsplit_once(after_paren, b") = ")?, + SubCase::OpenSSL => ByteSliceExt::rsplit_once(after_paren, b")= ")?, + }; - let checksum_utf8 = Self::validate_checksum_format(checksum)?; + let checksum_utf8 = Self::validate_checksum_format(checksum)?; - Some(LineInfo { - algo_name: Some(algo_utf8), - algo_bit_len: algo_bits, - checksum: checksum_utf8, - filename: filename.to_vec(), - format: Self::AlgoBased, - }) - } + Some(LineInfo { + algo_name: Some(algo_utf8), + algo_bit_len: algo_bits, + checksum: checksum_utf8, + filename: filename.to_vec(), + format: Self::AlgoBased, + }) + } - #[allow(rustdoc::invalid_html_tags)] - /// parse [untagged output format] - /// The format is simple, either " " or - /// " *" - /// - /// [untagged output format]: https://www.gnu.org/software/coreutils/manual/html_node/cksum-output-modes.html#cksum-output-modes-1 - fn parse_untagged(line: &[u8]) -> Option { - let space_idx = line.iter().position(|&b| b == b' ')?; - let checksum = &line[..space_idx]; + #[allow(rustdoc::invalid_html_tags)] + /// parse [untagged output format] + /// The format is simple, either " " or + /// " *" + /// + /// [untagged output format]: https://www.gnu.org/software/coreutils/manual/html_node/cksum-output-modes.html#cksum-output-modes-1 + fn parse_untagged(line: &[u8]) -> Option { + let space_idx = line.iter().position(|&b| b == b' ')?; + let checksum = &line[..space_idx]; - let checksum_utf8 = Self::validate_checksum_format(checksum)?; + let checksum_utf8 = Self::validate_checksum_format(checksum)?; - let rest = &line[space_idx..]; - let filename = rest - .strip_prefix(b" ") - .or_else(|| rest.strip_prefix(b" *"))?; + let rest = &line[space_idx..]; + let filename = rest + .strip_prefix(b" ") + .or_else(|| rest.strip_prefix(b" *"))?; - Some(LineInfo { - algo_name: None, - algo_bit_len: None, - checksum: checksum_utf8, - filename: filename.to_vec(), - format: Self::Untagged, - }) - } + Some(LineInfo { + algo_name: None, + algo_bit_len: None, + checksum: checksum_utf8, + filename: filename.to_vec(), + format: Self::Untagged, + }) + } - #[allow(rustdoc::invalid_html_tags)] - /// parse [untagged output format] - /// Normally the format is simple, either " " or - /// " *" - /// But the bsd tests expect special single space behavior where - /// checksum and filename are separated only by a space, meaning the second - /// space or asterisk is part of the file name. - /// This parser accounts for this variation - /// - /// [untagged output format]: https://www.gnu.org/software/coreutils/manual/html_node/cksum-output-modes.html#cksum-output-modes-1 - fn parse_single_space(line: &[u8]) -> Option { - // Find first space - let space_idx = line.iter().position(|&b| b == b' ')?; - let checksum = &line[..space_idx]; - if !checksum.iter().all(|&b| b.is_ascii_hexdigit()) || checksum.is_empty() { - return None; - } - // SAFETY: we just validated the contents of checksum, we can unsafely make a - // String from it - let checksum_utf8 = unsafe { String::from_utf8_unchecked(checksum.to_vec()) }; + #[allow(rustdoc::invalid_html_tags)] + /// parse [untagged output format] + /// Normally the format is simple, either " " or + /// " *" + /// But the bsd tests expect special single space behavior where + /// checksum and filename are separated only by a space, meaning the second + /// space or asterisk is part of the file name. + /// This parser accounts for this variation + /// + /// [untagged output format]: https://www.gnu.org/software/coreutils/manual/html_node/cksum-output-modes.html#cksum-output-modes-1 + fn parse_single_space(line: &[u8]) -> Option { + // Find first space + let space_idx = line.iter().position(|&b| b == b' ')?; + let checksum = &line[..space_idx]; + if !checksum.iter().all(|&b| b.is_ascii_hexdigit()) || checksum.is_empty() { + return None; + } + // SAFETY: we just validated the contents of checksum, we can unsafely make a + // String from it + let checksum_utf8 = unsafe { String::from_utf8_unchecked(checksum.to_vec()) }; - let filename = line.get(space_idx + 1..)?; // Skip single space + let filename = line.get(space_idx + 1..)?; // Skip single space - Some(LineInfo { - algo_name: None, - algo_bit_len: None, - checksum: checksum_utf8, - filename: filename.to_vec(), - format: Self::SingleSpace, - }) - } + Some(LineInfo { + algo_name: None, + algo_bit_len: None, + checksum: checksum_utf8, + filename: filename.to_vec(), + format: Self::SingleSpace, + }) + } - /// Ensure that the given checksum is syntactically valid (that it is either - /// hexadecimal or base64 encoded). - fn validate_checksum_format(checksum: &[u8]) -> Option { - if checksum.is_empty() { - return None; - } + /// Ensure that the given checksum is syntactically valid (that it is either + /// hexadecimal or base64 encoded). + fn validate_checksum_format(checksum: &[u8]) -> Option { + if checksum.is_empty() { + return None; + } - let mut is_base64 = false; + let mut is_base64 = false; - for index in 0..checksum.len() { - match checksum[index..] { - // ASCII alphanumeric - [b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9', ..] => (), - // Base64 special character - [b'+' | b'/', ..] => is_base64 = true, - // Base64 end of string padding - [b'='] | [b'=', b'='] | [b'=', b'=', b'='] => { - is_base64 = true; - break; - } - // Any other character means the checksum is wrong - _ => return None, - } - } + for index in 0..checksum.len() { + match checksum[index..] { + // ASCII alphanumeric + [b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9', ..] => (), + // Base64 special character + [b'+' | b'/', ..] => is_base64 = true, + // Base64 end of string padding + [b'='] | [b'=', b'='] | [b'=', b'=', b'='] => { + is_base64 = true; + break; + }, + // Any other character means the checksum is wrong + _ => return None, + } + } - // If base64 characters were encountered, make sure the checksum has a - // length multiple of 4. - // - // This check is not enough because it may allow base64-encoded - // checksums that are fully alphanumeric. Another check happens later - // when we are provided with a length hint to detect ambiguous - // base64-encoded checksums. - if is_base64 && !checksum.len().is_multiple_of(4) { - return None; - } + // If base64 characters were encountered, make sure the checksum has a + // length multiple of 4. + // + // This check is not enough because it may allow base64-encoded + // checksums that are fully alphanumeric. Another check happens later + // when we are provided with a length hint to detect ambiguous + // base64-encoded checksums. + if is_base64 && !checksum.len().is_multiple_of(4) { + return None; + } - // SAFETY: we just validated the contents of checksum, we can unsafely make a - // String from it - Some(unsafe { String::from_utf8_unchecked(checksum.to_vec()) }) - } + // SAFETY: we just validated the contents of checksum, we can unsafely make a + // String from it + Some(unsafe { String::from_utf8_unchecked(checksum.to_vec()) }) + } } // Helper trait for byte slice operations trait ByteSliceExt { - /// Look for a pattern from right to left, return surrounding parts if found. - fn rsplit_once(&self, pattern: &[u8]) -> Option<(&Self, &Self)>; + /// Look for a pattern from right to left, return surrounding parts if found. + fn rsplit_once(&self, pattern: &[u8]) -> Option<(&Self, &Self)>; } impl ByteSliceExt for [u8] { - fn rsplit_once(&self, pattern: &[u8]) -> Option<(&Self, &Self)> { - let pos = self - .windows(pattern.len()) - .rev() - .position(|w| w == pattern)?; - Some(( - &self[..self.len() - pattern.len() - pos], - &self[self.len() - pos..], - )) - } + fn rsplit_once(&self, pattern: &[u8]) -> Option<(&Self, &Self)> { + let pos = self + .windows(pattern.len()) + .rev() + .position(|w| w == pattern)?; + Some((&self[..self.len() - pattern.len() - pos], &self[self.len() - pos..])) + } } /// Hold the data extracted from a checksum line. struct LineInfo { - algo_name: Option, - algo_bit_len: Option, - checksum: String, - filename: Vec, - format: LineFormat, + algo_name: Option, + algo_bit_len: Option, + checksum: String, + filename: Vec, + format: LineFormat, } impl LineInfo { - /// Returns a `LineInfo` parsed from a checksum line. - /// The function will run 3 parsers against the line and select the first one that matches - /// to populate the fields of the struct. - /// However, there is a catch to handle regarding the handling of `cached_line_format`. - /// In case of non-algo-based format, if `cached_line_format` is Some, it must take the priority - /// over the detected format. Otherwise, we must set it the the detected format. - /// This specific behavior is emphasized by the test - /// `test_md5sum::test_check_md5sum_only_one_space`. - fn parse(s: impl AsRef, cached_line_format: &mut Option) -> Option { - let line_bytes = os_str_as_bytes(s.as_ref()).ok()?; + /// Returns a `LineInfo` parsed from a checksum line. + /// The function will run 3 parsers against the line and select the first one + /// that matches to populate the fields of the struct. + /// However, there is a catch to handle regarding the handling of + /// `cached_line_format`. In case of non-algo-based format, if + /// `cached_line_format` is Some, it must take the priority + /// over the detected format. Otherwise, we must set it the the detected + /// format. This specific behavior is emphasized by the test + /// `test_md5sum::test_check_md5sum_only_one_space`. + fn parse(s: impl AsRef, cached_line_format: &mut Option) -> Option { + let line_bytes = os_str_as_bytes(s.as_ref()).ok()?; - if let Some(info) = LineFormat::parse_algo_based(line_bytes) { - return Some(info); - } - if let Some(cached_format) = cached_line_format { - match cached_format { - LineFormat::Untagged => LineFormat::parse_untagged(line_bytes), - LineFormat::SingleSpace => LineFormat::parse_single_space(line_bytes), - LineFormat::AlgoBased => unreachable!("we never catch the algo based format"), - } - } else if let Some(info) = LineFormat::parse_untagged(line_bytes) { - *cached_line_format = Some(LineFormat::Untagged); - Some(info) - } else if let Some(info) = LineFormat::parse_single_space(line_bytes) { - *cached_line_format = Some(LineFormat::SingleSpace); - Some(info) - } else { - None - } - } + if let Some(info) = LineFormat::parse_algo_based(line_bytes) { + return Some(info); + } + if let Some(cached_format) = cached_line_format { + match cached_format { + LineFormat::Untagged => LineFormat::parse_untagged(line_bytes), + LineFormat::SingleSpace => LineFormat::parse_single_space(line_bytes), + LineFormat::AlgoBased => unreachable!("we never catch the algo based format"), + } + } else if let Some(info) = LineFormat::parse_untagged(line_bytes) { + *cached_line_format = Some(LineFormat::Untagged); + Some(info) + } else if let Some(info) = LineFormat::parse_single_space(line_bytes) { + *cached_line_format = Some(LineFormat::SingleSpace); + Some(info) + } else { + None + } + } } /// Extract the expected digest from the checksum string and decode it fn get_raw_expected_digest(checksum: &str, bit_len_hint: Option) -> Option> { - // If the length of the digest is not a multiple of 2, then it must be - // improperly formatted (1 byte is 2 hex digits, and base64 strings should - // always be a multiple of 4). - if !checksum.len().is_multiple_of(2) { - return None; - } + // If the length of the digest is not a multiple of 2, then it must be + // improperly formatted (1 byte is 2 hex digits, and base64 strings should + // always be a multiple of 4). + if !checksum.len().is_multiple_of(2) { + return None; + } - let byte_len_hint = bit_len_hint.map(|n| n.div_ceil(8)); + let byte_len_hint = bit_len_hint.map(|n| n.div_ceil(8)); - let checks_hint = |len| byte_len_hint.is_none_or(|hint| hint == len); + let checks_hint = |len| byte_len_hint.is_none_or(|hint| hint == len); - // If the length of the string matches the one to be expected (in case it's - // given) AND the digest can be decoded as hexadecimal, just go with it. - if checks_hint(checksum.len() / 2) { - if let Ok(raw_ck) = hex::decode(checksum) { - return Some(raw_ck); - } - } + // If the length of the string matches the one to be expected (in case it's + // given) AND the digest can be decoded as hexadecimal, just go with it. + if checks_hint(checksum.len() / 2) + && let Ok(raw_ck) = hex::decode(checksum) + { + return Some(raw_ck); + } - // If the checksum cannot be decoded as hexadecimal, interpret it as Base64 - // instead. + // If the checksum cannot be decoded as hexadecimal, interpret it as Base64 + // instead. - // But first, verify the encoded checksum length, which should be a - // multiple of 4. - // - // It is important to check it before trying to decode, because the - // forgiving mode of decoding will ignore if padding characters '=' are - // MISSING, but to match GNU's behavior, we must reject it. - if !checksum.len().is_multiple_of(4) { - return None; - } + // But first, verify the encoded checksum length, which should be a + // multiple of 4. + // + // It is important to check it before trying to decode, because the + // forgiving mode of decoding will ignore if padding characters '=' are + // MISSING, but to match GNU's behavior, we must reject it. + if !checksum.len().is_multiple_of(4) { + return None; + } - // Perform the decoding and be FORGIVING about it, to allow for checksums - // with INVALID padding to still be decoded. This is enforced by - // `test_untagged_base64_matching_tag` in `test_cksum.rs` + // Perform the decoding and be FORGIVING about it, to allow for checksums + // with INVALID padding to still be decoded. This is enforced by + // `test_untagged_base64_matching_tag` in `test_cksum.rs` - base64_simd::forgiving_decode_to_vec(checksum.as_bytes()) - .ok() - .filter(|raw| checks_hint(raw.len())) + base64_simd::forgiving_decode_to_vec(checksum.as_bytes()) + .ok() + .filter(|raw| checks_hint(raw.len())) } -/// Returns a reader that reads from the specified file, or from stdin if `filename_to_check` is "-". +/// Returns a reader that reads from the specified file, or from stdin if +/// `filename_to_check` is "-". fn get_file_to_check( - filename: &OsStr, - opts: ChecksumValidateOptions, + filename: &OsStr, + opts: ChecksumValidateOptions, ) -> Result, LineCheckError> { - let filename_bytes = os_str_as_bytes(filename).map_err(|e| LineCheckError::UError(e.into()))?; + let filename_bytes = os_str_as_bytes(filename).map_err(|e| LineCheckError::UError(e.into()))?; - if filename == "-" { - Ok(Box::new(pi_uutils_ctx::stdin())) // Use stdin if "-" is specified in the checksum file - } else { - let failed_open = || { - write_file_report( - pi_uutils_ctx::stdout(), - filename_bytes, - FileChecksumResult::CantOpen, - "", - opts.verbose, - ); - }; - let print_error = |err: io::Error| { - report_error(&err.map_err_context(|| locale_aware_escape_name(filename, QuotingStyle::SHELL_ESCAPE).to_string_lossy().to_string())); - }; - match File::open(pi_uutils_ctx::resolve(filename)) { - Ok(f) => { - if f.metadata() - .map_err(|_| LineCheckError::CantOpenFile)? - .is_dir() - { - print_error(io::Error::new( - io::ErrorKind::IsADirectory, - "Is a directory", - )); - // also regarded as a failed open - failed_open(); - Err(LineCheckError::FileIsDirectory) - } else { - Ok(Box::new(f)) - } - } - Err(err) => { - if !opts.ignore_missing { - // yes, we have both stderr and stdout here - print_error(err); - failed_open(); - } - // we could not open the file but we want to continue - Err(LineCheckError::FileNotFound) - } - } - } + if filename == "-" { + Ok(Box::new(pi_uutils_ctx::stdin())) // Use stdin if "-" is specified in the checksum file + } else { + let failed_open = || { + write_file_report( + pi_uutils_ctx::stdout(), + filename_bytes, + FileChecksumResult::CantOpen, + "", + opts.verbose, + ); + }; + let print_error = |err: io::Error| { + report_error(&err.map_err_context(|| { + locale_aware_escape_name(filename, QuotingStyle::SHELL_ESCAPE) + .to_string_lossy() + .to_string() + })); + }; + match File::open(pi_uutils_ctx::resolve(filename)) { + Ok(f) => { + if f + .metadata() + .map_err(|_| LineCheckError::CantOpenFile)? + .is_dir() + { + print_error(io::Error::new(io::ErrorKind::IsADirectory, "Is a directory")); + // also regarded as a failed open + failed_open(); + Err(LineCheckError::FileIsDirectory) + } else { + Ok(Box::new(f)) + } + }, + Err(err) => { + if !opts.ignore_missing { + // yes, we have both stderr and stdout here + print_error(err); + failed_open(); + } + // we could not open the file but we want to continue + Err(LineCheckError::FileNotFound) + }, + } + } } /// Returns a reader to the list of checksums fn get_input_file(filename: &OsStr) -> UResult> { - match File::open(pi_uutils_ctx::resolve(filename)) { - Ok(f) => { - if f.metadata()?.is_dir() { - Err(io::Error::other( - format!("{}: Is a directory", filename.maybe_quote()), - ) - .into()) - } else { - Ok(Box::new(f)) - } - } - Err(_) => Err(io::Error::other(format!( - "{}: {}", - filename.maybe_quote(), - "No such file or directory" - )) - .into()), - } + match File::open(pi_uutils_ctx::resolve(filename)) { + Ok(f) => { + if f.metadata()?.is_dir() { + Err(io::Error::other(format!("{}: Is a directory", filename.maybe_quote())).into()) + } else { + Ok(Box::new(f)) + } + }, + Err(_) => Err( + io::Error::other(format!("{}: {}", filename.maybe_quote(), "No such file or directory")) + .into(), + ), + } } -/// Gets the algorithm name and length from the `LineInfo` if the algo-based format is matched. +/// Gets the algorithm name and length from the `LineInfo` if the algo-based +/// format is matched. fn identify_algo_name_and_length( - line_info: &LineInfo, - algo_name_input: Option, - last_algo: &mut Option, + line_info: &LineInfo, + algo_name_input: Option, + last_algo: &mut Option, ) -> Result<(AlgoKind, Option), LineCheckError> { - use AlgoKind as ak; - let algo_from_line = line_info.algo_name.clone().unwrap_or_default(); - let Ok(line_algo) = AlgoKind::from_cksum(algo_from_line.to_lowercase()) else { - // Unknown algorithm - return Err(LineCheckError::ImproperlyFormatted); - }; - *last_algo = Some(algo_from_line); + use AlgoKind as ak; + let algo_from_line = line_info.algo_name.clone().unwrap_or_default(); + let Ok(line_algo) = AlgoKind::from_cksum(algo_from_line.to_lowercase()) else { + // Unknown algorithm + return Err(LineCheckError::ImproperlyFormatted); + }; + *last_algo = Some(algo_from_line); - // check if we are called with XXXsum (example: md5sum) but we detected a - // different algo parsing the file (for example SHA1 (f) = d...) - // - // Also handle the case cksum -s sm3 but the file contains other formats - if let Some(algo_name_input) = algo_name_input { - match (algo_name_input, line_algo) { - (l, r) if l == r => (), - // Edge case for SHA2, which matches SHA(224|256|384|512) - (ak::Sha2, ak::Sha224 | ak::Sha256 | ak::Sha384 | ak::Sha512) => (), - _ => return Err(LineCheckError::ImproperlyFormatted), - } - } + // check if we are called with XXXsum (example: md5sum) but we detected a + // different algo parsing the file (for example SHA1 (f) = d...) + // + // Also handle the case cksum -s sm3 but the file contains other formats + if let Some(algo_name_input) = algo_name_input { + match (algo_name_input, line_algo) { + (l, r) if l == r => (), + // Edge case for SHA2, which matches SHA(224|256|384|512) + (ak::Sha2, ak::Sha224 | ak::Sha256 | ak::Sha384 | ak::Sha512) => (), + _ => return Err(LineCheckError::ImproperlyFormatted), + } + } - let bytes = if let Some(bitlen) = line_info.algo_bit_len { - match line_algo { - algo @ (ak::Blake2b | ak::Blake3) => { - match parse_blake_length(algo, BlakeLength::Int(bitlen)) { - Ok(len) => Some(len), - Err(_) => return Err(LineCheckError::ImproperlyFormatted), - } - } - ak::Sha2 | ak::Sha3 if [224, 256, 384, 512].contains(&bitlen) => Some(bitlen), - ak::Shake128 | ak::Shake256 => Some(bitlen), - // Either - // the algo based line is provided with a bit length with an - // algorithm that does not support it (only Blake2b, Blake3, sha2, - // and sha3 do). - // - // eg: MD5-128 (foo.txt) = fffffffff - // ^ This is illegal - // OR - // the given length is wrong because it's not a multiple of 8. - _ => return Err(LineCheckError::ImproperlyFormatted), - } - } else if line_algo == ak::Blake2b { - // Default length with BLAKE2b, - Some(Blake2b::DEFAULT_BYTE_SIZE) - } else if line_algo == ak::Blake3 { - // Default length with BLAKE3, - Some(Blake3::DEFAULT_BYTE_SIZE) - } else { - None - }; + let bytes = if let Some(bitlen) = line_info.algo_bit_len { + match line_algo { + algo @ (ak::Blake2b | ak::Blake3) => { + match parse_blake_length(algo, BlakeLength::Int(bitlen)) { + Ok(len) => Some(len), + Err(_) => return Err(LineCheckError::ImproperlyFormatted), + } + }, + ak::Sha2 | ak::Sha3 if [224, 256, 384, 512].contains(&bitlen) => Some(bitlen), + ak::Shake128 | ak::Shake256 => Some(bitlen), + // Either + // the algo based line is provided with a bit length with an + // algorithm that does not support it (only Blake2b, Blake3, sha2, + // and sha3 do). + // + // eg: MD5-128 (foo.txt) = fffffffff + // ^ This is illegal + // OR + // the given length is wrong because it's not a multiple of 8. + _ => return Err(LineCheckError::ImproperlyFormatted), + } + } else if line_algo == ak::Blake2b { + // Default length with BLAKE2b, + Some(Blake2b::DEFAULT_BYTE_SIZE) + } else if line_algo == ak::Blake3 { + // Default length with BLAKE3, + Some(Blake3::DEFAULT_BYTE_SIZE) + } else { + None + }; - Ok((line_algo, bytes)) + Ok((line_algo, bytes)) } /// Given a filename and an algorithm, compute the digest and compare it with /// the expected one. fn compute_and_check_digest_from_file( - filename: &[u8], - expected_checksum: &[u8], - algo: SizedAlgoKind, - opts: ChecksumValidateOptions, + filename: &[u8], + expected_checksum: &[u8], + algo: SizedAlgoKind, + opts: ChecksumValidateOptions, ) -> Result<(), LineCheckError> { - let (filename_to_check_unescaped, prefix) = unescape_filename(filename); - let real_filename_to_check = os_str_from_bytes(&filename_to_check_unescaped)?; + let (filename_to_check_unescaped, prefix) = unescape_filename(filename); + let real_filename_to_check = os_str_from_bytes(&filename_to_check_unescaped)?; - // Open the input file - let file_to_check = get_file_to_check(&real_filename_to_check, opts)?; - let mut file_reader = BufReader::new(file_to_check); + // Open the input file + let file_to_check = get_file_to_check(&real_filename_to_check, opts)?; + let mut file_reader = BufReader::new(file_to_check); - // Read the file and calculate the checksum - let mut digest = algo.create_digest(); + // Read the file and calculate the checksum + let mut digest = algo.create_digest(); - // Set binary to false because --binary is not supported with --check + // Set binary to false because --binary is not supported with --check - let (calculated_checksum, _) = - match digest_reader(&mut digest, &mut file_reader, ReadingMode::Text) { - Ok(result) => result, - Err(err) => { - report_error(&err.map_err_context(|| locale_aware_escape_name(&real_filename_to_check, QuotingStyle::SHELL_ESCAPE).to_string_lossy().to_string())); + let (calculated_checksum, _) = + match digest_reader(&mut digest, &mut file_reader, ReadingMode::Text) { + Ok(result) => result, + Err(err) => { + report_error(&err.map_err_context(|| { + locale_aware_escape_name(&real_filename_to_check, QuotingStyle::SHELL_ESCAPE) + .to_string_lossy() + .to_string() + })); - write_file_report( - pi_uutils_ctx::stdout(), - filename, - FileChecksumResult::CantOpen, - prefix, - opts.verbose, - ); - return Err(LineCheckError::CantOpenFile); - } - }; + write_file_report( + pi_uutils_ctx::stdout(), + filename, + FileChecksumResult::CantOpen, + prefix, + opts.verbose, + ); + return Err(LineCheckError::CantOpenFile); + }, + }; - // Do the checksum validation - let checksum_correct = match calculated_checksum { - DigestOutput::Vec(data) => data == expected_checksum, - DigestOutput::Crc(n) => n.to_be_bytes() == expected_checksum, - DigestOutput::U16(n) => n.to_be_bytes() == expected_checksum, - }; - write_file_report( - pi_uutils_ctx::stdout(), - filename, - FileChecksumResult::from_bool(checksum_correct), - prefix, - opts.verbose, - ); + // Do the checksum validation + let checksum_correct = match calculated_checksum { + DigestOutput::Vec(data) => data == expected_checksum, + DigestOutput::Crc(n) => n.to_be_bytes() == expected_checksum, + DigestOutput::U16(n) => n.to_be_bytes() == expected_checksum, + }; + write_file_report( + pi_uutils_ctx::stdout(), + filename, + FileChecksumResult::from_bool(checksum_correct), + prefix, + opts.verbose, + ); - if checksum_correct { - Ok(()) - } else { - Err(LineCheckError::DigestMismatch) - } + if checksum_correct { + Ok(()) + } else { + Err(LineCheckError::DigestMismatch) + } } /// Check a digest checksum with non-algo based pre-treatment. fn process_algo_based_line( - line_info: &LineInfo, - cli_algo_kind: Option, - opts: ChecksumValidateOptions, - last_algo: &mut Option, + line_info: &LineInfo, + cli_algo_kind: Option, + opts: ChecksumValidateOptions, + last_algo: &mut Option, ) -> Result<(), LineCheckError> { - let filename_to_check = line_info.filename.as_slice(); + let filename_to_check = line_info.filename.as_slice(); - let (algo_kind, algo_len) = identify_algo_name_and_length(line_info, cli_algo_kind, last_algo)?; + let (algo_kind, algo_len) = identify_algo_name_and_length(line_info, cli_algo_kind, last_algo)?; - // If the digest bitlen is known, we can check the format of the expected - // checksum with it. - let digest_bit_length_hint = match (algo_kind, algo_len) { - (AlgoKind::Blake2b | AlgoKind::Blake3, Some(byte_len)) => Some(byte_len * 8), - (AlgoKind::Shake128 | AlgoKind::Shake256, Some(bit_len)) => Some(bit_len), - (AlgoKind::Shake128, None) => Some(sum::Shake128::DEFAULT_BIT_SIZE), - (AlgoKind::Shake256, None) => Some(sum::Shake256::DEFAULT_BIT_SIZE), - _ => None, - }; + // If the digest bitlen is known, we can check the format of the expected + // checksum with it. + let digest_bit_length_hint = match (algo_kind, algo_len) { + (AlgoKind::Blake2b | AlgoKind::Blake3, Some(byte_len)) => Some(byte_len * 8), + (AlgoKind::Shake128 | AlgoKind::Shake256, Some(bit_len)) => Some(bit_len), + (AlgoKind::Shake128, None) => Some(sum::Shake128::DEFAULT_BIT_SIZE), + (AlgoKind::Shake256, None) => Some(sum::Shake256::DEFAULT_BIT_SIZE), + _ => None, + }; - let expected_checksum = get_raw_expected_digest(&line_info.checksum, digest_bit_length_hint) - .ok_or(LineCheckError::ImproperlyFormatted)?; + let expected_checksum = get_raw_expected_digest(&line_info.checksum, digest_bit_length_hint) + .ok_or(LineCheckError::ImproperlyFormatted)?; - let algo = SizedAlgoKind::from_unsized(algo_kind, algo_len) - .map_err(|_| LineCheckError::ImproperlyFormatted)?; + let algo = SizedAlgoKind::from_unsized(algo_kind, algo_len) + .map_err(|_| LineCheckError::ImproperlyFormatted)?; - compute_and_check_digest_from_file(filename_to_check, &expected_checksum, algo, opts) + compute_and_check_digest_from_file(filename_to_check, &expected_checksum, algo, opts) } /// Check a digest checksum with non-algo based pre-treatment. fn process_non_algo_based_line( - line_number: usize, - line_info: &LineInfo, - cli_algo_kind: AlgoKind, - cli_algo_length: Option, - opts: ChecksumValidateOptions, + line_number: usize, + line_info: &LineInfo, + cli_algo_kind: AlgoKind, + cli_algo_length: Option, + opts: ChecksumValidateOptions, ) -> Result<(), LineCheckError> { - use AlgoKind as ak; - let mut filename_to_check = line_info.filename.as_slice(); - if filename_to_check.starts_with(b"*") - && line_number == 0 - && line_info.format == LineFormat::SingleSpace - { - // Remove the leading asterisk if present - only for the first line - filename_to_check = &filename_to_check[1..]; - } + use AlgoKind as ak; + let mut filename_to_check = line_info.filename.as_slice(); + if filename_to_check.starts_with(b"*") + && line_number == 0 + && line_info.format == LineFormat::SingleSpace + { + // Remove the leading asterisk if present - only for the first line + filename_to_check = &filename_to_check[1..]; + } - let expected_digest_sum = cli_algo_kind.expected_digest_bit_len(); - let expected_checksum = get_raw_expected_digest(&line_info.checksum, expected_digest_sum) - .ok_or(LineCheckError::ImproperlyFormatted)?; + let expected_digest_sum = cli_algo_kind.expected_digest_bit_len(); + let expected_checksum = get_raw_expected_digest(&line_info.checksum, expected_digest_sum) + .ok_or(LineCheckError::ImproperlyFormatted)?; - // When a specific algorithm name is input, use it and use the provided - // bits except when dealing with blake2b, sha2 and sha3, where we will - // detect the length. - let algo_byte_len = match cli_algo_kind { - ak::Blake2b | ak::Blake3 => Some(expected_checksum.len()), - ak::Sha2 | ak::Sha3 => { - // multiplication by 8 to get the number of bits - Some( - ShaLength::try_from(expected_checksum.len() * 8) - .map_err(|_| LineCheckError::ImproperlyFormatted)? - .as_usize(), - ) - } - _ => cli_algo_length, - }; + // When a specific algorithm name is input, use it and use the provided + // bits except when dealing with blake2b, sha2 and sha3, where we will + // detect the length. + let algo_byte_len = match cli_algo_kind { + ak::Blake2b | ak::Blake3 => Some(expected_checksum.len()), + ak::Sha2 | ak::Sha3 => { + // multiplication by 8 to get the number of bits + Some( + ShaLength::try_from(expected_checksum.len() * 8) + .map_err(|_| LineCheckError::ImproperlyFormatted)? + .as_usize(), + ) + }, + _ => cli_algo_length, + }; - let algo = SizedAlgoKind::from_unsized(cli_algo_kind, algo_byte_len)?; + let algo = SizedAlgoKind::from_unsized(cli_algo_kind, algo_byte_len)?; - compute_and_check_digest_from_file(filename_to_check, &expected_checksum, algo, opts) + compute_and_check_digest_from_file(filename_to_check, &expected_checksum, algo, opts) } -/// Parses a checksum line, detect the algorithm to use, read the file and produce -/// its digest, and compare it to the expected value. +/// Parses a checksum line, detect the algorithm to use, read the file and +/// produce its digest, and compare it to the expected value. /// /// Returns `Ok(bool)` if the comparison happened, bool indicates if the digest /// matched the expected. /// If the comparison didn't happen, return a `LineChecksumError`. fn process_checksum_line( - line: &OsStr, - i: usize, - cli_algo_name: Option, - cli_algo_length: Option, - opts: ChecksumValidateOptions, - cached_line_format: &mut Option, - last_algo: &mut Option, + line: &OsStr, + i: usize, + cli_algo_name: Option, + cli_algo_length: Option, + opts: ChecksumValidateOptions, + cached_line_format: &mut Option, + last_algo: &mut Option, ) -> Result<(), LineCheckError> { - let line_bytes = os_str_as_bytes(line).map_err(|e| LineCheckError::UError(Box::new(e)))?; + let line_bytes = os_str_as_bytes(line).map_err(|e| LineCheckError::UError(Box::new(e)))?; - // Early return on empty or commented lines. - if line.is_empty() || line_bytes.starts_with(b"#") { - return Err(LineCheckError::Skipped); - } + // Early return on empty or commented lines. + if line.is_empty() || line_bytes.starts_with(b"#") { + return Err(LineCheckError::Skipped); + } - // Use `LineInfo` to extract the data of a line. - // Then, depending on its format, apply a different pre-treatment. - let Some(line_info) = LineInfo::parse(line, cached_line_format) else { - return Err(LineCheckError::ImproperlyFormatted); - }; + // Use `LineInfo` to extract the data of a line. + // Then, depending on its format, apply a different pre-treatment. + let Some(line_info) = LineInfo::parse(line, cached_line_format) else { + return Err(LineCheckError::ImproperlyFormatted); + }; - if line_info.format == LineFormat::AlgoBased { - process_algo_based_line(&line_info, cli_algo_name, opts, last_algo) - } else if let Some(cli_algo) = cli_algo_name { - // If we match a non-algo based parser, we expect a cli argument - // to give us the algorithm to use - process_non_algo_based_line(i, &line_info, cli_algo, cli_algo_length, opts) - } else { - // We have no clue of what algorithm to use - Err(LineCheckError::ImproperlyFormatted) - } + if line_info.format == LineFormat::AlgoBased { + process_algo_based_line(&line_info, cli_algo_name, opts, last_algo) + } else if let Some(cli_algo) = cli_algo_name { + // If we match a non-algo based parser, we expect a cli argument + // to give us the algorithm to use + process_non_algo_based_line(i, &line_info, cli_algo, cli_algo_length, opts) + } else { + // We have no clue of what algorithm to use + Err(LineCheckError::ImproperlyFormatted) + } } fn process_checksum_file( - filename_input: &OsStr, - cli_algo_kind: Option, - cli_algo_length: Option, - opts: ChecksumValidateOptions, + filename_input: &OsStr, + cli_algo_kind: Option, + cli_algo_length: Option, + opts: ChecksumValidateOptions, ) -> Result<(), FileCheckError> { - let mut res = ChecksumResult::default(); + let mut res = ChecksumResult::default(); - let input_is_stdin = filename_input == OsStr::new("-"); + let input_is_stdin = filename_input == OsStr::new("-"); - let file: Box = if input_is_stdin { - // Use stdin if "-" is specified - Box::new(pi_uutils_ctx::stdin()) - } else { - match get_input_file(filename_input) { - Ok(f) => f, - Err(e) => { - // Could not read the file, show the error and continue to the next file - let _ = writeln!(pi_uutils_ctx::stderr(), "{}: {e}", command_name()); - return Err(FileCheckError::CantOpenChecksumFile); - } - } - }; + let file: Box = if input_is_stdin { + // Use stdin if "-" is specified + Box::new(pi_uutils_ctx::stdin()) + } else { + match get_input_file(filename_input) { + Ok(f) => f, + Err(e) => { + // Could not read the file, show the error and continue to the next file + let _ = writeln!(pi_uutils_ctx::stderr(), "{}: {e}", command_name()); + return Err(FileCheckError::CantOpenChecksumFile); + }, + } + }; - let reader = BufReader::new(file); + let reader = BufReader::new(file); - // cached_line_format is used to ensure that several non algo-based checksum line - // will use the same parser. - let mut cached_line_format = None; - // last_algo caches the algorithm used in the last line to print a warning - // message for the current line if improperly formatted. - // Behavior tested in gnu_cksum_c::test_warn - let mut last_algo = None; + // cached_line_format is used to ensure that several non algo-based checksum + // line will use the same parser. + let mut cached_line_format = None; + // last_algo caches the algorithm used in the last line to print a warning + // message for the current line if improperly formatted. + // Behavior tested in gnu_cksum_c::test_warn + let mut last_algo = None; - for (i, line_res) in read_os_string_lines(reader).enumerate() { - let line = line_res.map_err(|e| { - USimpleError::new( - UIoError::from(e).code(), - format!("{}: read error", filename_input.maybe_quote()), - ) - })?; + for (i, line_res) in read_os_string_lines(reader).enumerate() { + let line = line_res.map_err(|e| { + USimpleError::new( + UIoError::from(e).code(), + format!("{}: read error", filename_input.maybe_quote()), + ) + })?; - let line_result = process_checksum_line( - &line, - i, - cli_algo_kind, - cli_algo_length, - opts, - &mut cached_line_format, - &mut last_algo, - ); + let line_result = process_checksum_line( + &line, + i, + cli_algo_kind, + cli_algo_length, + opts, + &mut cached_line_format, + &mut last_algo, + ); - // Match a first time to elude critical UErrors, and increment the total - // in all cases except on skipped. - use LineCheckError::*; - match line_result { - Err(UError(e)) => return Err(e.into()), - Err(Skipped) => (), - _ => res.total += 1, - } + // Match a first time to elude critical UErrors, and increment the total + // in all cases except on skipped. + use LineCheckError::*; + match line_result { + Err(UError(e)) => return Err(e.into()), + Err(Skipped) => (), + _ => res.total += 1, + } - // Match a second time to update the right field of `res`. - match line_result { - Ok(()) => res.correct += 1, - Err(DigestMismatch) => res.failed_cksum += 1, - Err(ImproperlyFormatted) => { - res.bad_format += 1; + // Match a second time to update the right field of `res`. + match line_result { + Ok(()) => res.correct += 1, + Err(DigestMismatch) => res.failed_cksum += 1, + Err(ImproperlyFormatted) => { + res.bad_format += 1; - if opts.verbose.at_least_warning() { - let algo = if let Some(algo_name_input) = cli_algo_kind { - algo_name_input.to_uppercase() - } else if let Some(algo) = &last_algo { - algo.as_str() - } else { - "Unknown algorithm" - }; - let _ = writeln!( - pi_uutils_ctx::stderr(), - "{}: {}", - command_name(), - format!("{}: line {}: improperly formatted {} checksum line", filename_input.maybe_quote(), i + 1, algo) - ); - } - } - Err(CantOpenFile | FileIsDirectory) => res.failed_open_file += 1, - Err(FileNotFound) if !opts.ignore_missing => res.failed_open_file += 1, - _ => (), - } - } + if opts.verbose.at_least_warning() { + let algo = if let Some(algo_name_input) = cli_algo_kind { + algo_name_input.to_uppercase() + } else if let Some(algo) = &last_algo { + algo.as_str() + } else { + "Unknown algorithm" + }; + let _ = writeln!( + pi_uutils_ctx::stderr(), + "{}: {}: line {}: improperly formatted {} checksum line", + command_name(), + filename_input.maybe_quote(), + i + 1, + algo + ); + } + }, + Err(CantOpenFile | FileIsDirectory) => res.failed_open_file += 1, + Err(FileNotFound) if !opts.ignore_missing => res.failed_open_file += 1, + _ => (), + } + } - let filename_display = || { - if input_is_stdin { - "standard input".maybe_quote() - } else { - filename_input.maybe_quote() - } - }; + let filename_display = || { + if input_is_stdin { + "standard input".maybe_quote() + } else { + filename_input.maybe_quote() + } + }; - // not a single line correctly formatted found - // return an error - if res.total_properly_formatted() == 0 { - if opts.verbose.over_status() { - log_no_properly_formatted(filename_display()); - } - return Err(FileCheckError::Failed); - } + // not a single line correctly formatted found + // return an error + if res.total_properly_formatted() == 0 { + if opts.verbose.over_status() { + log_no_properly_formatted(filename_display()); + } + return Err(FileCheckError::Failed); + } - // if any incorrectly formatted line, show it - if opts.verbose.over_status() { - print_cksum_report(&res); - } + // if any incorrectly formatted line, show it + if opts.verbose.over_status() { + print_cksum_report(&res); + } - if opts.ignore_missing && res.correct == 0 { - // we have only bad format - // and we had ignore-missing - if opts.verbose.over_status() { - log_no_file_verified(filename_display()); - } - return Err(FileCheckError::Failed); - } + if opts.ignore_missing && res.correct == 0 { + // we have only bad format + // and we had ignore-missing + if opts.verbose.over_status() { + log_no_file_verified(filename_display()); + } + return Err(FileCheckError::Failed); + } - // strict means that we should have an exit code. - if opts.strict && res.bad_format > 0 { - return Err(FileCheckError::Failed); - } + // strict means that we should have an exit code. + if opts.strict && res.bad_format > 0 { + return Err(FileCheckError::Failed); + } - // If a file was missing, return an error unless we explicitly ignore it. - if res.failed_open_file > 0 && !opts.ignore_missing { - return Err(FileCheckError::Failed); - } + // If a file was missing, return an error unless we explicitly ignore it. + if res.failed_open_file > 0 && !opts.ignore_missing { + return Err(FileCheckError::Failed); + } - // Obviously, if a checksum failed at some point, report the error. - if res.failed_cksum > 0 { - return Err(FileCheckError::Failed); - } + // Obviously, if a checksum failed at some point, report the error. + if res.failed_cksum > 0 { + return Err(FileCheckError::Failed); + } - Ok(()) + Ok(()) } /// Do the checksum validation (can be strict or not) pub fn perform_checksum_validation<'a, I>( - files: I, - algo_kind: Option, - length_input: Option, - opts: ChecksumValidateOptions, + files: I, + algo_kind: Option, + length_input: Option, + opts: ChecksumValidateOptions, ) -> UResult<()> where - I: Iterator, + I: Iterator, { - let mut failed = false; + let mut failed = false; - // if cksum has several input files, it will print the result for each file - for filename_input in files { - use FileCheckError::*; - match process_checksum_file(filename_input, algo_kind, length_input, opts) { - Err(UError(e)) => return Err(e), - Err(Failed | CantOpenChecksumFile) => failed = true, - Ok(_) => (), - } - } + // if cksum has several input files, it will print the result for each file + for filename_input in files { + use FileCheckError::*; + match process_checksum_file(filename_input, algo_kind, length_input, opts) { + Err(UError(e)) => return Err(e), + Err(Failed | CantOpenChecksumFile) => failed = true, + Ok(_) => (), + } + } - if failed { - Err(USimpleError::new(1, "")) - } else { - Ok(()) - } + if failed { + Err(USimpleError::new(1, "")) + } else { + Ok(()) + } } diff --git a/crates/vendor/uu-comm/src/comm.rs b/crates/vendor/uu-comm/src/comm.rs index 8e39022f5..38806720f 100644 --- a/crates/vendor/uu-comm/src/comm.rs +++ b/crates/vendor/uu-comm/src/comm.rs @@ -4,17 +4,21 @@ // file that was distributed with this source code. // Vendored from uutils/coreutils 0.8.0 and patched for pi-uutils context I/O. -use std::cmp::Ordering; -use std::ffi::{OsStr, OsString}; -use std::fs::{self, File}; -use std::io::{self, BufRead, BufReader, BufWriter, Read, Write}; -use std::path::Path; +use std::{ + cmp::Ordering, + ffi::{OsStr, OsString}, + fs::{self, File}, + io::{self, BufRead, BufReader, BufWriter, Read, Write}, + path::Path, +}; use clap::{Arg, ArgAction, ArgMatches, Command}; use pi_uutils_ctx::format_usage; -use uucore::display::Quotable; -use uucore::error::{FromIo, UResult, USimpleError}; -use uucore::line_ending::LineEnding; +use uucore::{ + display::Quotable, + error::{FromIo, UResult, USimpleError}, + line_ending::LineEnding, +}; mod options { pub const COLUMN_1: &str = "1"; @@ -30,21 +34,30 @@ mod options { } #[derive(Clone, Copy)] -enum FileNumber { One, Two } +enum FileNumber { + One, + Two, +} impl FileNumber { - fn as_str(self) -> &'static str { match self { Self::One => "1", Self::Two => "2" } } + fn as_str(self) -> &'static str { + match self { + Self::One => "1", + Self::Two => "2", + } + } } struct OrderChecker { - last_line: Vec, - file_num: FileNumber, + last_line: Vec, + file_num: FileNumber, check_order: bool, - has_error: bool, + has_error: bool, } impl OrderChecker { fn new(file_num: FileNumber, check_order: bool) -> Self { Self { last_line: Vec::new(), file_num, check_order, has_error: false } } + fn verify_order(&mut self, line: &[u8]) -> bool { if self.last_line.is_empty() { self.last_line = line.to_vec(); @@ -52,7 +65,11 @@ impl OrderChecker { } let ordered = line >= self.last_line.as_slice(); if !ordered && !self.has_error { - let _ = writeln!(pi_uutils_ctx::stderr(), "comm: file {} is not in sorted order", self.file_num.as_str()); + let _ = writeln!( + pi_uutils_ctx::stderr(), + "comm: file {} is not in sorted order", + self.file_num.as_str() + ); self.has_error = true; } self.last_line.clear(); @@ -63,15 +80,18 @@ impl OrderChecker { struct LineReader { line_ending: u8, - input: Box, + input: Box, } impl LineReader { fn new(input: Box, line_ending: LineEnding) -> Self { Self { input, line_ending: line_ending.into() } } + fn read_line(&mut self, buf: &mut Vec) -> io::Result { let result = self.input.read_until(self.line_ending, buf)?; - if result != 0 && !buf.ends_with(&[self.line_ending]) { buf.push(self.line_ending); } + if result != 0 && !buf.ends_with(&[self.line_ending]) { + buf.push(self.line_ending); + } Ok(result) } } @@ -79,91 +99,184 @@ impl LineReader { fn files_identical(path1: &Path, path2: &Path) -> io::Result { let m1 = fs::metadata(path1)?; let m2 = fs::metadata(path2)?; - if !m1.is_file() || !m2.is_file() || m1.len() != m2.len() { return Ok(false); } + if !m1.is_file() || !m2.is_file() || m1.len() != m2.len() { + return Ok(false); + } let mut a = BufReader::new(File::open(path1)?); let mut b = BufReader::new(File::open(path2)?); let mut ba = [0; 8192]; let mut bb = [0; 8192]; loop { - let na = loop { match a.read(&mut ba) { Err(e) if e.kind() == io::ErrorKind::Interrupted => {}, r => break r? } }; - let nb = loop { match b.read(&mut bb) { Err(e) if e.kind() == io::ErrorKind::Interrupted => {}, r => break r? } }; - if na != nb || ba[..na] != bb[..nb] { return Ok(false); } - if na == 0 { return Ok(true); } + let na = loop { + match a.read(&mut ba) { + Err(e) if e.kind() == io::ErrorKind::Interrupted => {}, + r => break r?, + } + }; + let nb = loop { + match b.read(&mut bb) { + Err(e) if e.kind() == io::ErrorKind::Interrupted => {}, + r => break r?, + } + }; + if na != nb || ba[..na] != bb[..nb] { + return Ok(false); + } + if na == 0 { + return Ok(true); + } } } fn write_delimited(writer: &mut impl Write, delim: &[u8], line: &[u8]) -> UResult<()> { - writer.write_all(delim).map_err_context(|| "write error".to_string())?; - writer.write_all(line).map_err_context(|| "write error".to_string()) + writer + .write_all(delim) + .map_err_context(|| "write error".to_string())?; + writer + .write_all(line) + .map_err_context(|| "write error".to_string()) } -fn compare(a: &mut LineReader, b: &mut LineReader, name1: &OsStr, name2: &OsStr, delim: &str, opts: &ArgMatches, identical: bool) -> UResult { +fn compare( + a: &mut LineReader, + b: &mut LineReader, + name1: &OsStr, + name2: &OsStr, + delim: &str, + opts: &ArgMatches, + identical: bool, +) -> UResult { let col2 = delim.repeat(usize::from(!opts.get_flag(options::COLUMN_1))); - let col3 = delim.repeat(usize::from(!opts.get_flag(options::COLUMN_1)) + usize::from(!opts.get_flag(options::COLUMN_2))); + let col3 = delim.repeat( + usize::from(!opts.get_flag(options::COLUMN_1)) + + usize::from(!opts.get_flag(options::COLUMN_2)), + ); let mut writer = BufWriter::new(pi_uutils_ctx::stdout()); let (mut ra, mut rb) = (Vec::new(), Vec::new()); - let mut na = a.read_line(&mut ra).map_err_context(|| name1.maybe_quote().to_string())?; - let mut nb = b.read_line(&mut rb).map_err_context(|| name2.maybe_quote().to_string())?; + let mut na = a + .read_line(&mut ra) + .map_err_context(|| name1.maybe_quote().to_string())?; + let mut nb = b + .read_line(&mut rb) + .map_err_context(|| name2.maybe_quote().to_string())?; let (mut n1, mut n2, mut n3) = (0usize, 0usize, 0usize); let explicit = opts.get_flag(options::CHECK_ORDER); let should_check = !opts.get_flag(options::NO_CHECK_ORDER) && (explicit || !identical); - let (mut c1, mut c2) = (OrderChecker::new(FileNumber::One, explicit), OrderChecker::new(FileNumber::Two, explicit)); + let (mut c1, mut c2) = + (OrderChecker::new(FileNumber::One, explicit), OrderChecker::new(FileNumber::Two, explicit)); let mut delayed_error = false; while na != 0 || nb != 0 { - let ord = match (na, nb) { (0, _) => Ordering::Greater, (_, 0) => Ordering::Less, _ => ra.cmp(&rb) }; + let ord = match (na, nb) { + (0, _) => Ordering::Greater, + (_, 0) => Ordering::Less, + _ => ra.cmp(&rb), + }; match ord { Ordering::Less => { - if should_check && !c1.verify_order(&ra) { break; } - if !opts.get_flag(options::COLUMN_1) { writer.write_all(&ra).map_err_context(|| "write error".to_string())?; } - ra.clear(); na = a.read_line(&mut ra).map_err_context(|| name1.maybe_quote().to_string())?; n1 += 1; + if should_check && !c1.verify_order(&ra) { + break; + } + if !opts.get_flag(options::COLUMN_1) { + writer + .write_all(&ra) + .map_err_context(|| "write error".to_string())?; + } + ra.clear(); + na = a + .read_line(&mut ra) + .map_err_context(|| name1.maybe_quote().to_string())?; + n1 += 1; }, Ordering::Greater => { - if should_check && !c2.verify_order(&rb) { break; } - if !opts.get_flag(options::COLUMN_2) { write_delimited(&mut writer, col2.as_bytes(), &rb)?; } - rb.clear(); nb = b.read_line(&mut rb).map_err_context(|| name2.maybe_quote().to_string())?; n2 += 1; + if should_check && !c2.verify_order(&rb) { + break; + } + if !opts.get_flag(options::COLUMN_2) { + write_delimited(&mut writer, col2.as_bytes(), &rb)?; + } + rb.clear(); + nb = b + .read_line(&mut rb) + .map_err_context(|| name2.maybe_quote().to_string())?; + n2 += 1; }, Ordering::Equal => { - if should_check && (!c1.verify_order(&ra) || !c2.verify_order(&rb)) { break; } - if !opts.get_flag(options::COLUMN_3) { write_delimited(&mut writer, col3.as_bytes(), &ra)?; } - ra.clear(); rb.clear(); - na = a.read_line(&mut ra).map_err_context(|| name1.maybe_quote().to_string())?; - nb = b.read_line(&mut rb).map_err_context(|| name2.maybe_quote().to_string())?; n3 += 1; + if should_check && (!c1.verify_order(&ra) || !c2.verify_order(&rb)) { + break; + } + if !opts.get_flag(options::COLUMN_3) { + write_delimited(&mut writer, col3.as_bytes(), &ra)?; + } + ra.clear(); + rb.clear(); + na = a + .read_line(&mut ra) + .map_err_context(|| name1.maybe_quote().to_string())?; + nb = b + .read_line(&mut rb) + .map_err_context(|| name2.maybe_quote().to_string())?; + n3 += 1; }, } - if (c1.has_error || c2.has_error) && !explicit { delayed_error = true; } + if (c1.has_error || c2.has_error) && !explicit { + delayed_error = true; + } } if opts.get_flag(options::TOTAL) { let ending = LineEnding::from_zero_flag(opts.get_flag(options::ZERO_TERMINATED)); - write!(writer, "{n1}{delim}{n2}{delim}{n3}{delim}total{ending}").map_err_context(|| "write error".to_string())?; + write!(writer, "{n1}{delim}{n2}{delim}{n3}{delim}total{ending}") + .map_err_context(|| "write error".to_string())?; } - writer.flush().map_err_context(|| "write error".to_string())?; + writer + .flush() + .map_err_context(|| "write error".to_string())?; if should_check && (c1.has_error || c2.has_error) { - if delayed_error { let _ = writeln!(pi_uutils_ctx::stderr(), "comm: input is not in sorted order"); } + if delayed_error { + let _ = writeln!(pi_uutils_ctx::stderr(), "comm: input is not in sorted order"); + } Ok(false) - } else { Ok(true) } + } else { + Ok(true) + } } fn open_file(name: &OsStr, ending: LineEnding) -> io::Result { - if name == "-" { return Ok(LineReader::new(Box::new(BufReader::new(pi_uutils_ctx::stdin())), ending)); } + if name == "-" { + return Ok(LineReader::new(Box::new(BufReader::new(pi_uutils_ctx::stdin())), ending)); + } let resolved = pi_uutils_ctx::resolve(name); - if fs::metadata(&resolved)?.is_dir() { return Err(io::Error::other("is a directory")); } + if fs::metadata(&resolved)?.is_dir() { + return Err(io::Error::other("is a directory")); + } Ok(LineReader::new(Box::new(BufReader::new(File::open(resolved)?)), ending)) } fn comm_main(matches: &ArgMatches) -> UResult { let name1 = matches.get_one::(options::FILE_1).unwrap(); let name2 = matches.get_one::(options::FILE_2).unwrap(); - if name1 == "-" && name2 == "-" { return Err(USimpleError::new(1, "standard input is specified twice")); } + if name1 == "-" && name2 == "-" { + return Err(USimpleError::new(1, "standard input is specified twice")); + } let ending = LineEnding::from_zero_flag(matches.get_flag(options::ZERO_TERMINATED)); let mut f1 = open_file(name1, ending).map_err_context(|| name1.maybe_quote().to_string())?; let mut f2 = open_file(name2, ending).map_err_context(|| name2.maybe_quote().to_string())?; - let delimiters: Vec<_> = matches.get_many::(options::DELIMITER).unwrap().collect(); + let delimiters: Vec<_> = matches + .get_many::(options::DELIMITER) + .unwrap() + .collect(); if delimiters[1..].iter().any(|d| *d != delimiters[0]) { return Err(USimpleError::new(1, "multiple conflicting output delimiters specified")); } - let delim = if delimiters[0].is_empty() { "\0" } else { delimiters[0] }; - let identical = if name1 == "-" || name2 == "-" { false } else { - files_identical(&pi_uutils_ctx::resolve(name1), &pi_uutils_ctx::resolve(name2)).unwrap_or(false) + let delim = if delimiters[0].is_empty() { + "\0" + } else { + delimiters[0] + }; + let identical = if name1 == "-" || name2 == "-" { + false + } else { + files_identical(&pi_uutils_ctx::resolve(name1), &pi_uutils_ctx::resolve(name2)) + .unwrap_or(false) }; compare(&mut f1, &mut f2, name1, name2, delim, matches, identical) } @@ -174,14 +287,22 @@ pub fn run(argv: Vec) -> i32 { Ok(m) => m, Err(e) => { let rendered = e.to_string(); - if e.use_stderr() { let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); return 1; } - let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); return 0; + if e.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; }, }; match comm_main(&matches) { Ok(true) => pi_uutils_ctx::exit_code(), Ok(false) => 1, - Err(e) => { let code = e.code(); let _ = writeln!(pi_uutils_ctx::stderr(), "comm: {e}"); if code == 0 { 1 } else { code } }, + Err(e) => { + let code = e.code(); + let _ = writeln!(pi_uutils_ctx::stderr(), "comm: {e}"); + if code == 0 { 1 } else { code } + }, } } @@ -190,15 +311,73 @@ pub fn uu_app() -> Command { .version(uucore::crate_version!()) .about("Compare sorted files FILE1 and FILE2 line by line.") .override_usage(format_usage("comm [OPTION]... FILE1 FILE2")) - .infer_long_args(true).args_override_self(true) - .arg(Arg::new(options::COLUMN_1).short('1').help("suppress column 1 (lines unique to FILE1)").action(ArgAction::SetTrue)) - .arg(Arg::new(options::COLUMN_2).short('2').help("suppress column 2 (lines unique to FILE2)").action(ArgAction::SetTrue)) - .arg(Arg::new(options::COLUMN_3).short('3').help("suppress column 3 (lines that appear in both files)").action(ArgAction::SetTrue)) - .arg(Arg::new(options::DELIMITER).long(options::DELIMITER).help("separate columns with STR").value_name("STR").default_value("\t").allow_hyphen_values(true).action(ArgAction::Append).hide_default_value(true)) - .arg(Arg::new(options::ZERO_TERMINATED).long(options::ZERO_TERMINATED).short('z').overrides_with(options::ZERO_TERMINATED).help("line delimiter is NUL, not newline").action(ArgAction::SetTrue)) - .arg(Arg::new(options::FILE_1).required(true).value_hint(clap::ValueHint::FilePath).value_parser(clap::value_parser!(OsString))) - .arg(Arg::new(options::FILE_2).required(true).value_hint(clap::ValueHint::FilePath).value_parser(clap::value_parser!(OsString))) - .arg(Arg::new(options::TOTAL).long(options::TOTAL).help("output a summary").action(ArgAction::SetTrue)) - .arg(Arg::new(options::CHECK_ORDER).long(options::CHECK_ORDER).help("check that input is correctly sorted, even if all input lines are pairable").action(ArgAction::SetTrue)) - .arg(Arg::new(options::NO_CHECK_ORDER).long(options::NO_CHECK_ORDER).help("do not check that input is correctly sorted").action(ArgAction::SetTrue).conflicts_with(options::CHECK_ORDER)) + .infer_long_args(true) + .args_override_self(true) + .arg( + Arg::new(options::COLUMN_1) + .short('1') + .help("suppress column 1 (lines unique to FILE1)") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::COLUMN_2) + .short('2') + .help("suppress column 2 (lines unique to FILE2)") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::COLUMN_3) + .short('3') + .help("suppress column 3 (lines that appear in both files)") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::DELIMITER) + .long(options::DELIMITER) + .help("separate columns with STR") + .value_name("STR") + .default_value("\t") + .allow_hyphen_values(true) + .action(ArgAction::Append) + .hide_default_value(true), + ) + .arg( + Arg::new(options::ZERO_TERMINATED) + .long(options::ZERO_TERMINATED) + .short('z') + .overrides_with(options::ZERO_TERMINATED) + .help("line delimiter is NUL, not newline") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::FILE_1) + .required(true) + .value_hint(clap::ValueHint::FilePath) + .value_parser(clap::value_parser!(OsString)), + ) + .arg( + Arg::new(options::FILE_2) + .required(true) + .value_hint(clap::ValueHint::FilePath) + .value_parser(clap::value_parser!(OsString)), + ) + .arg( + Arg::new(options::TOTAL) + .long(options::TOTAL) + .help("output a summary") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::CHECK_ORDER) + .long(options::CHECK_ORDER) + .help("check that input is correctly sorted, even if all input lines are pairable") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::NO_CHECK_ORDER) + .long(options::NO_CHECK_ORDER) + .help("do not check that input is correctly sorted") + .action(ArgAction::SetTrue) + .conflicts_with(options::CHECK_ORDER), + ) } diff --git a/crates/vendor/uu-cut/src/cut.rs b/crates/vendor/uu-cut/src/cut.rs index f916134a6..85b090b5a 100644 --- a/crates/vendor/uu-cut/src/cut.rs +++ b/crates/vendor/uu-cut/src/cut.rs @@ -5,438 +5,444 @@ // spell-checker:ignore (ToDO) delim sourcefiles undelimited +use std::{ + ffi::OsString, + fs::File, + io::{BufRead, BufReader, BufWriter, Read, Write}, + path::Path, +}; + use bstr::io::BufReadExt; use clap::{Arg, ArgAction, ArgMatches, Command, builder::ValueParser}; -use std::ffi::OsString; -use std::fs::File; -use std::io::{BufRead, BufReader, BufWriter, Read, Write}; -use std::path::Path; -use uucore::display::Quotable; -use uucore::error::{FromIo, UResult, USimpleError}; -use uucore::line_ending::LineEnding; -use uucore::os_str_as_bytes; +use matcher::{ExactMatcher, Matcher, WhitespaceMatcher}; +use pi_uutils_ctx::format_usage; +use uucore::{ + display::Quotable, + error::{FromIo, UResult, USimpleError}, + line_ending::LineEnding, + os_str_as_bytes, + ranges::Range, +}; use self::searcher::Searcher; -use matcher::{ExactMatcher, Matcher, WhitespaceMatcher}; -use uucore::ranges::Range; -use pi_uutils_ctx::format_usage; mod matcher; mod searcher; struct Options<'a> { - out_delimiter: Option<&'a [u8]>, - line_ending: LineEnding, - field_opts: Option>, + out_delimiter: Option<&'a [u8]>, + line_ending: LineEnding, + field_opts: Option>, } enum Delimiter<'a> { - Whitespace, - Slice(&'a [u8]), + Whitespace, + Slice(&'a [u8]), } struct FieldOptions<'a> { - delimiter: Delimiter<'a>, - only_delimited: bool, + delimiter: Delimiter<'a>, + only_delimited: bool, } enum Mode<'a> { - Bytes(Vec, Options<'a>), - Characters(Vec, Options<'a>), - Fields(Vec, Options<'a>), + Bytes(Vec, Options<'a>), + Characters(Vec, Options<'a>), + Fields(Vec, Options<'a>), } impl Default for Delimiter<'_> { - fn default() -> Self { - Self::Slice(b"\t") - } + fn default() -> Self { + Self::Slice(b"\t") + } } impl<'a> From<&'a OsString> for Delimiter<'a> { - fn from(s: &'a OsString) -> Self { - Self::Slice(os_str_as_bytes(s).unwrap()) - } + fn from(s: &'a OsString) -> Self { + Self::Slice(os_str_as_bytes(s).unwrap()) + } } fn list_to_ranges(list: &str, complement: bool) -> Result, String> { - if complement { - Range::from_list(list).map(|r| uucore::ranges::complement(&r)) - } else { - Range::from_list(list) - } + if complement { + Range::from_list(list).map(|r| uucore::ranges::complement(&r)) + } else { + Range::from_list(list) + } } fn cut_bytes( - reader: R, - out: &mut W, - ranges: &[Range], - opts: &Options, + reader: R, + out: &mut W, + ranges: &[Range], + opts: &Options, ) -> UResult<()> { - let newline_char = opts.line_ending.into(); - let mut buf_in = BufReader::new(reader); - let out_delim = opts.out_delimiter.unwrap_or(b"\t"); + let newline_char = opts.line_ending.into(); + let mut buf_in = BufReader::new(reader); + let out_delim = opts.out_delimiter.unwrap_or(b"\t"); - let result = buf_in.for_byte_record(newline_char, |line| { - let mut print_delim = false; - for &Range { low, high } in ranges { - if low > line.len() { - break; - } - if print_delim { - out.write_all(out_delim)?; - } else if opts.out_delimiter.is_some() { - print_delim = true; - } - // change `low` from 1-indexed value to 0-index value - let low = low - 1; - let high = high.min(line.len()); - out.write_all(&line[low..high])?; - } - out.write_all(&[newline_char])?; - Ok(true) - }); + let result = buf_in.for_byte_record(newline_char, |line| { + let mut print_delim = false; + for &Range { low, high } in ranges { + if low > line.len() { + break; + } + if print_delim { + out.write_all(out_delim)?; + } else if opts.out_delimiter.is_some() { + print_delim = true; + } + // change `low` from 1-indexed value to 0-index value + let low = low - 1; + let high = high.min(line.len()); + out.write_all(&line[low..high])?; + } + out.write_all(&[newline_char])?; + Ok(true) + }); - if let Err(e) = result { - return Err(USimpleError::new(1, e.to_string())); - } + if let Err(e) = result { + return Err(USimpleError::new(1, e.to_string())); + } - Ok(()) + Ok(()) } /// Output delimiter is explicitly specified fn cut_fields_explicit_out_delim( - reader: R, - out: &mut W, - matcher: &M, - ranges: &[Range], - only_delimited: bool, - newline_char: u8, - out_delim: &[u8], + reader: R, + out: &mut W, + matcher: &M, + ranges: &[Range], + only_delimited: bool, + newline_char: u8, + out_delim: &[u8], ) -> UResult<()> { - let mut buf_in = BufReader::new(reader); + let mut buf_in = BufReader::new(reader); - let result = buf_in.for_byte_record_with_terminator(newline_char, |line| { - let mut fields_pos = 1; - let mut low_idx = 0; - let mut delim_search = Searcher::new(matcher, line).peekable(); - let mut print_delim = false; + let result = buf_in.for_byte_record_with_terminator(newline_char, |line| { + let mut fields_pos = 1; + let mut low_idx = 0; + let mut delim_search = Searcher::new(matcher, line).peekable(); + let mut print_delim = false; - if delim_search.peek().is_none() { - if !only_delimited { - // Always write the entire line, even if it doesn't end with `newline_char` - out.write_all(line)?; - if line.is_empty() || line[line.len() - 1] != newline_char { - out.write_all(&[newline_char])?; - } - } + if delim_search.peek().is_none() { + if !only_delimited { + // Always write the entire line, even if it doesn't end with `newline_char` + out.write_all(line)?; + if line.is_empty() || line[line.len() - 1] != newline_char { + out.write_all(&[newline_char])?; + } + } - return Ok(true); - } + return Ok(true); + } - for &Range { low, high } in ranges { - if low - fields_pos > 0 { - // current field is not in the range, so jump to the field corresponding to the - // beginning of the range if any - low_idx = match delim_search.nth(low - fields_pos - 1) { - Some((_, last)) => last, - None => break, - }; - } + for &Range { low, high } in ranges { + if low - fields_pos > 0 { + // current field is not in the range, so jump to the field corresponding to the + // beginning of the range if any + low_idx = match delim_search.nth(low - fields_pos - 1) { + Some((_, last)) => last, + None => break, + }; + } - // at this point, current field is the first in the range - for _ in 0..=high - low { - // skip printing delimiter if this is the first matching field for this line - if print_delim { - out.write_all(out_delim)?; - } else { - print_delim = true; - } + // at this point, current field is the first in the range + for _ in 0..=high - low { + // skip printing delimiter if this is the first matching field for this line + if print_delim { + out.write_all(out_delim)?; + } else { + print_delim = true; + } - if let Some((first, last)) = delim_search.next() { - // print the current field up to the next field delim - let segment = &line[low_idx..first]; + if let Some((first, last)) = delim_search.next() { + // print the current field up to the next field delim + let segment = &line[low_idx..first]; - out.write_all(segment)?; + out.write_all(segment)?; - low_idx = last; - fields_pos = high + 1; - } else { - // this is the last field in the line, so print the rest - let segment = &line[low_idx..]; + low_idx = last; + fields_pos = high + 1; + } else { + // this is the last field in the line, so print the rest + let segment = &line[low_idx..]; - out.write_all(segment)?; + out.write_all(segment)?; - if line[line.len() - 1] == newline_char { - return Ok(true); - } - break; - } - } - } + if line[line.len() - 1] == newline_char { + return Ok(true); + } + break; + } + } + } - out.write_all(&[newline_char])?; - Ok(true) - }); + out.write_all(&[newline_char])?; + Ok(true) + }); - if let Err(e) = result { - return Err(USimpleError::new(1, e.to_string())); - } + if let Err(e) = result { + return Err(USimpleError::new(1, e.to_string())); + } - Ok(()) + Ok(()) } /// Output delimiter is the same as input delimiter fn cut_fields_implicit_out_delim( - reader: R, - out: &mut W, - matcher: &M, - ranges: &[Range], - only_delimited: bool, - newline_char: u8, + reader: R, + out: &mut W, + matcher: &M, + ranges: &[Range], + only_delimited: bool, + newline_char: u8, ) -> UResult<()> { - let mut buf_in = BufReader::new(reader); + let mut buf_in = BufReader::new(reader); - let result = buf_in.for_byte_record_with_terminator(newline_char, |line| { - let mut fields_pos = 1; - let mut low_idx = 0; - let mut delim_search = Searcher::new(matcher, line).peekable(); - let mut print_delim = false; + let result = buf_in.for_byte_record_with_terminator(newline_char, |line| { + let mut fields_pos = 1; + let mut low_idx = 0; + let mut delim_search = Searcher::new(matcher, line).peekable(); + let mut print_delim = false; - if delim_search.peek().is_none() { - if !only_delimited { - // Always write the entire line, even if it doesn't end with `newline_char` - out.write_all(line)?; - if line.is_empty() || line[line.len() - 1] != newline_char { - out.write_all(&[newline_char])?; - } - } + if delim_search.peek().is_none() { + if !only_delimited { + // Always write the entire line, even if it doesn't end with `newline_char` + out.write_all(line)?; + if line.is_empty() || line[line.len() - 1] != newline_char { + out.write_all(&[newline_char])?; + } + } - return Ok(true); - } + return Ok(true); + } - for &Range { low, high } in ranges { - if low - fields_pos > 0 { - if let Some((first, last)) = delim_search.nth(low - fields_pos - 1) { - low_idx = if print_delim { first } else { last } - } else { - break; - } - } + for &Range { low, high } in ranges { + if low - fields_pos > 0 { + if let Some((first, last)) = delim_search.nth(low - fields_pos - 1) { + low_idx = if print_delim { first } else { last } + } else { + break; + } + } - if let Some((first, _)) = delim_search.nth(high - low) { - let segment = &line[low_idx..first]; + if let Some((first, _)) = delim_search.nth(high - low) { + let segment = &line[low_idx..first]; - out.write_all(segment)?; + out.write_all(segment)?; - print_delim = true; - low_idx = first; - fields_pos = high + 1; - } else { - let segment = &line[low_idx..line.len()]; + print_delim = true; + low_idx = first; + fields_pos = high + 1; + } else { + let segment = &line[low_idx..line.len()]; - out.write_all(segment)?; + out.write_all(segment)?; - if line[line.len() - 1] == newline_char { - return Ok(true); - } - break; - } - } - out.write_all(&[newline_char])?; - Ok(true) - }); + if line[line.len() - 1] == newline_char { + return Ok(true); + } + break; + } + } + out.write_all(&[newline_char])?; + Ok(true) + }); - if let Err(e) = result { - return Err(USimpleError::new(1, e.to_string())); - } + if let Err(e) = result { + return Err(USimpleError::new(1, e.to_string())); + } - Ok(()) + Ok(()) } /// Streams and filters fields where the record terminator and /// field delimiter are the same character (specified by `newline_char`) fn cut_fields_newline_char_delim( - reader: R, - out: &mut W, - ranges: &[Range], - newline_char: u8, - out_delim: &[u8], - only_delimited: bool, + reader: R, + out: &mut W, + ranges: &[Range], + newline_char: u8, + out_delim: &[u8], + only_delimited: bool, ) -> UResult<()> { - let mut reader = BufReader::new(reader); - let mut line = Vec::new(); + let mut reader = BufReader::new(reader); + let mut line = Vec::new(); - // We start at 1 because 'cut' field indexing is 1-based - let mut current_field_idx = 1; - let mut first_field_printed = false; - let mut has_data = false; - let mut suppressed = false; + // We start at 1 because 'cut' field indexing is 1-based + let mut current_field_idx = 1; + let mut first_field_printed = false; + let mut has_data = false; + let mut suppressed = false; - let mut range_idx = 0; + let mut range_idx = 0; - loop { - line.clear(); + loop { + line.clear(); - let is_selected = range_idx < ranges.len() && current_field_idx >= ranges[range_idx].low; - let needs_data = is_selected || current_field_idx == 1; + let is_selected = range_idx < ranges.len() && current_field_idx >= ranges[range_idx].low; + let needs_data = is_selected || current_field_idx == 1; - let mut has_processed_data = false; + let mut has_processed_data = false; - if needs_data { - // Standard read: copies bytes into `line` - loop { - let buf = reader.fill_buf()?; - if buf.is_empty() { - break; - } + if needs_data { + // Standard read: copies bytes into `line` + loop { + let buf = reader.fill_buf()?; + if buf.is_empty() { + break; + } - has_processed_data = true; + has_processed_data = true; - if let Some(pos) = memchr::memchr(newline_char, buf) { - let amt = pos + 1; - line.extend_from_slice(&buf[..amt]); - reader.consume(amt); + if let Some(pos) = memchr::memchr(newline_char, buf) { + let amt = pos + 1; + line.extend_from_slice(&buf[..amt]); + reader.consume(amt); - break; - } - let len = buf.len(); - line.extend_from_slice(buf); - reader.consume(len); - } - } else { - // Zero-allocation skip: scans the buffer and advances the cursor without copying - loop { - let buf = reader.fill_buf()?; - if buf.is_empty() { - break; // EOF - } + break; + } + let len = buf.len(); + line.extend_from_slice(buf); + reader.consume(len); + } + } else { + // Zero-allocation skip: scans the buffer and advances the cursor without + // copying + loop { + let buf = reader.fill_buf()?; + if buf.is_empty() { + break; // EOF + } - has_processed_data = true; + has_processed_data = true; - if let Some(pos) = memchr::memchr(newline_char, buf) { - let bytes_to_consume = pos + 1; - reader.consume(bytes_to_consume); - break; - } + if let Some(pos) = memchr::memchr(newline_char, buf) { + let bytes_to_consume = pos + 1; + reader.consume(bytes_to_consume); + break; + } - let len = buf.len(); - reader.consume(len); - } - } + let len = buf.len(); + reader.consume(len); + } + } - if !has_processed_data { - break; - } - has_data = true; + if !has_processed_data { + break; + } + has_data = true; - // To comply with -s when the stream consists of only a single field. - if current_field_idx == 1 { - let is_eof_next = reader.fill_buf()?.is_empty(); + // To comply with -s when the stream consists of only a single field. + if current_field_idx == 1 { + let is_eof_next = reader.fill_buf()?.is_empty(); - if is_eof_next && line.last() != Some(&newline_char) { - if only_delimited { - suppressed = true; - } else { - // GNU cut prints the whole line if no delimiter is found. - out.write_all(&line)?; - } - break; - } - } + if is_eof_next && line.last() != Some(&newline_char) { + if only_delimited { + suppressed = true; + } else { + // GNU cut prints the whole line if no delimiter is found. + out.write_all(&line)?; + } + break; + } + } - if range_idx < ranges.len() && current_field_idx > ranges[range_idx].high { - range_idx += 1; + if range_idx < ranges.len() && current_field_idx > ranges[range_idx].high { + range_idx += 1; - // EARLY EXIT: If we've exhausted all ranges, stop reading the stream entirely. - if range_idx == ranges.len() { - break; - } - } + // EARLY EXIT: If we've exhausted all ranges, stop reading the stream entirely. + if range_idx == ranges.len() { + break; + } + } - // Check if the current field falls inside the current active range - let is_selected = range_idx < ranges.len() && current_field_idx >= ranges[range_idx].low; + // Check if the current field falls inside the current active range + let is_selected = range_idx < ranges.len() && current_field_idx >= ranges[range_idx].low; - if is_selected { - if first_field_printed { - out.write_all(out_delim)?; - } + if is_selected { + if first_field_printed { + out.write_all(out_delim)?; + } - let has_newline = line.last() == Some(&newline_char); - let content = if has_newline { - &line[..line.len() - 1] - } else { - &line[..] - }; + let has_newline = line.last() == Some(&newline_char); + let content = if has_newline { + &line[..line.len() - 1] + } else { + &line[..] + }; - out.write_all(content)?; - first_field_printed = true; - } + out.write_all(content)?; + first_field_printed = true; + } - current_field_idx += 1; - } + current_field_idx += 1; + } - if has_data && !suppressed { - out.write_all(&[newline_char])?; - } + if has_data && !suppressed { + out.write_all(&[newline_char])?; + } - Ok(()) + Ok(()) } fn cut_fields( - reader: R, - out: &mut W, - ranges: &[Range], - opts: &Options, + reader: R, + out: &mut W, + ranges: &[Range], + opts: &Options, ) -> UResult<()> { - let newline_char = opts.line_ending.into(); - let field_opts = opts.field_opts.as_ref().unwrap(); // it is safe to unwrap() here - field_opts will always be Some() for cut_fields() call - match field_opts.delimiter { - Delimiter::Slice(delim) if delim == [newline_char] => { - let out_delim = opts.out_delimiter.unwrap_or(delim); - cut_fields_newline_char_delim( - reader, - out, - ranges, - newline_char, - out_delim, - field_opts.only_delimited, - ) - } - Delimiter::Slice(delim) => { - let matcher = ExactMatcher::new(delim); - match opts.out_delimiter { - Some(out_delim) => cut_fields_explicit_out_delim( - reader, - out, - &matcher, - ranges, - field_opts.only_delimited, - newline_char, - out_delim, - ), - None => cut_fields_implicit_out_delim( - reader, - out, - &matcher, - ranges, - field_opts.only_delimited, - newline_char, - ), - } - } - Delimiter::Whitespace => { - let matcher = WhitespaceMatcher {}; - cut_fields_explicit_out_delim( - reader, - out, - &matcher, - ranges, - field_opts.only_delimited, - newline_char, - opts.out_delimiter.unwrap_or(b"\t"), - ) - } - } + let newline_char = opts.line_ending.into(); + let field_opts = opts.field_opts.as_ref().unwrap(); // it is safe to unwrap() here - field_opts will always be Some() for cut_fields() call + match field_opts.delimiter { + Delimiter::Slice(delim) if delim == [newline_char] => { + let out_delim = opts.out_delimiter.unwrap_or(delim); + cut_fields_newline_char_delim( + reader, + out, + ranges, + newline_char, + out_delim, + field_opts.only_delimited, + ) + }, + Delimiter::Slice(delim) => { + let matcher = ExactMatcher::new(delim); + match opts.out_delimiter { + Some(out_delim) => cut_fields_explicit_out_delim( + reader, + out, + &matcher, + ranges, + field_opts.only_delimited, + newline_char, + out_delim, + ), + None => cut_fields_implicit_out_delim( + reader, + out, + &matcher, + ranges, + field_opts.only_delimited, + newline_char, + ), + } + }, + Delimiter::Whitespace => { + let matcher = WhitespaceMatcher {}; + cut_fields_explicit_out_delim( + reader, + out, + &matcher, + ranges, + field_opts.only_delimited, + newline_char, + opts.out_delimiter.unwrap_or(b"\t"), + ) + }, + } } // pi-uutils: route standard streams through the invocation context, and @@ -444,111 +450,113 @@ fn cut_fields( // original operand for diagnostics. fn cut_files<'a, I>(filenames: I, mode: &Mode) where - I: IntoIterator, + I: IntoIterator, { - let mut stdin_read = false; - let mut out = BufWriter::new(pi_uutils_ctx::stdout()); + let mut stdin_read = false; + let mut out = BufWriter::new(pi_uutils_ctx::stdout()); - for filename in filenames { - if filename == "-" { - if stdin_read { - continue; - } - let result = match mode { - Mode::Bytes(ranges, opts) | Mode::Characters(ranges, opts) => - cut_bytes(pi_uutils_ctx::stdin(), &mut out, ranges, opts), - Mode::Fields(ranges, opts) => - cut_fields(pi_uutils_ctx::stdin(), &mut out, ranges, opts), - }; - if let Err(err) = result { - let _ = writeln!(pi_uutils_ctx::stderr(), "cut: {err}"); - pi_uutils_ctx::set_exit_code(1); - } - stdin_read = true; - } else { - let result = File::open(pi_uutils_ctx::resolve(Path::new(filename))) - .map_err_context(|| filename.maybe_quote().to_string()) - .and_then(|file| match mode { - Mode::Bytes(ranges, opts) | Mode::Characters(ranges, opts) => - cut_bytes(file, &mut out, ranges, opts), - Mode::Fields(ranges, opts) => cut_fields(file, &mut out, ranges, opts), - }); - if let Err(err) = result { - let _ = writeln!(pi_uutils_ctx::stderr(), "cut: {err}"); - pi_uutils_ctx::set_exit_code(1); - } - } - } + for filename in filenames { + if filename == "-" { + if stdin_read { + continue; + } + let result = match mode { + Mode::Bytes(ranges, opts) | Mode::Characters(ranges, opts) => { + cut_bytes(pi_uutils_ctx::stdin(), &mut out, ranges, opts) + }, + Mode::Fields(ranges, opts) => { + cut_fields(pi_uutils_ctx::stdin(), &mut out, ranges, opts) + }, + }; + if let Err(err) = result { + let _ = writeln!(pi_uutils_ctx::stderr(), "cut: {err}"); + pi_uutils_ctx::set_exit_code(1); + } + stdin_read = true; + } else { + let result = File::open(pi_uutils_ctx::resolve(Path::new(filename))) + .map_err_context(|| filename.maybe_quote().to_string()) + .and_then(|file| match mode { + Mode::Bytes(ranges, opts) | Mode::Characters(ranges, opts) => { + cut_bytes(file, &mut out, ranges, opts) + }, + Mode::Fields(ranges, opts) => cut_fields(file, &mut out, ranges, opts), + }); + if let Err(err) = result { + let _ = writeln!(pi_uutils_ctx::stderr(), "cut: {err}"); + pi_uutils_ctx::set_exit_code(1); + } + } + } - if let Err(err) = out.flush().map_err_context(|| "write error".to_string()) { - let _ = writeln!(pi_uutils_ctx::stderr(), "cut: {err}"); - pi_uutils_ctx::set_exit_code(1); - } + if let Err(err) = out.flush().map_err_context(|| "write error".to_string()) { + let _ = writeln!(pi_uutils_ctx::stderr(), "cut: {err}"); + pi_uutils_ctx::set_exit_code(1); + } } -/// Get delimiter and output delimiter from `-d`/`--delimiter` and `--output-delimiter` options respectively -/// Allow either delimiter to have a value that is neither UTF-8 nor ASCII to align with GNU behavior +/// Get delimiter and output delimiter from `-d`/`--delimiter` and +/// `--output-delimiter` options respectively Allow either delimiter to have a +/// value that is neither UTF-8 nor ASCII to align with GNU behavior fn get_delimiters(matches: &ArgMatches) -> UResult<(Delimiter<'_>, Option<&[u8]>)> { - let whitespace_delimited = matches.get_flag(options::WHITESPACE_DELIMITED); - let delim_opt = matches.get_one::(options::DELIMITER); - let delim = match delim_opt { - Some(_) if whitespace_delimited => { - return Err(USimpleError::new( - 1, - "invalid input: Only one of --delimiter (-d) or -w option can be specified", - )); - } - Some(os_string) => { - if os_string.is_empty() { - Delimiter::Slice(b"\0") - } else { - // For delimiter `-d` option value - allow both UTF-8 (possibly multi-byte) characters - // and Non UTF-8 (and not ASCII) single byte "characters", like `b"\xAD"` to align with GNU behavior - let bytes = os_str_as_bytes(os_string)?; - if os_string.to_str().is_some_and(|s| s.chars().count() > 1) - || os_string.to_str().is_none() && bytes.len() > 1 - { - return Err(USimpleError::new( - 1, - "the delimiter must be a single character", - )); - } - Delimiter::from(os_string) - } - } - None => { - if whitespace_delimited { - Delimiter::Whitespace - } else { - Delimiter::default() - } - } - }; - let out_delim = matches - .get_one::(options::OUTPUT_DELIMITER) - .map(|os_string| { - if os_string.is_empty() { - b"\0" - } else { - os_str_as_bytes(os_string).unwrap() - } - }); - Ok((delim, out_delim)) + let whitespace_delimited = matches.get_flag(options::WHITESPACE_DELIMITED); + let delim_opt = matches.get_one::(options::DELIMITER); + let delim = match delim_opt { + Some(_) if whitespace_delimited => { + return Err(USimpleError::new( + 1, + "invalid input: Only one of --delimiter (-d) or -w option can be specified", + )); + }, + Some(os_string) => { + if os_string.is_empty() { + Delimiter::Slice(b"\0") + } else { + // For delimiter `-d` option value - allow both UTF-8 (possibly multi-byte) + // characters and Non UTF-8 (and not ASCII) single byte "characters", like + // `b"\xAD"` to align with GNU behavior + let bytes = os_str_as_bytes(os_string)?; + if os_string.to_str().is_some_and(|s| s.chars().count() > 1) + || os_string.to_str().is_none() && bytes.len() > 1 + { + return Err(USimpleError::new(1, "the delimiter must be a single character")); + } + Delimiter::from(os_string) + } + }, + None => { + if whitespace_delimited { + Delimiter::Whitespace + } else { + Delimiter::default() + } + }, + }; + let out_delim = matches + .get_one::(options::OUTPUT_DELIMITER) + .map(|os_string| { + if os_string.is_empty() { + b"\0" + } else { + os_str_as_bytes(os_string).unwrap() + } + }); + Ok((delim, out_delim)) } mod options { - pub const BYTES: &str = "bytes"; - pub const CHARACTERS: &str = "characters"; - pub const DELIMITER: &str = "delimiter"; - pub const FIELDS: &str = "fields"; - pub const ZERO_TERMINATED: &str = "zero-terminated"; - pub const ONLY_DELIMITED: &str = "only-delimited"; - pub const OUTPUT_DELIMITER: &str = "output-delimiter"; - pub const WHITESPACE_DELIMITED: &str = "whitespace-delimited"; - pub const COMPLEMENT: &str = "complement"; - pub const FILE: &str = "file"; - // ignored option - pub const NOTHING: &str = "nothing"; + pub const BYTES: &str = "bytes"; + pub const CHARACTERS: &str = "characters"; + pub const DELIMITER: &str = "delimiter"; + pub const FIELDS: &str = "fields"; + pub const ZERO_TERMINATED: &str = "zero-terminated"; + pub const ONLY_DELIMITED: &str = "only-delimited"; + pub const OUTPUT_DELIMITER: &str = "output-delimiter"; + pub const WHITESPACE_DELIMITED: &str = "whitespace-delimited"; + pub const COMPLEMENT: &str = "complement"; + pub const FILE: &str = "file"; + // ignored option + pub const NOTHING: &str = "nothing"; } // pi-uutils: replace the terminating uucore entry macro and localization-aware @@ -556,233 +564,233 @@ mod options { /// Run `cut` against the streams and working directory installed by /// `pi-uutils-ctx`. pub fn run(argv: Vec) -> i32 { - // GNU cut accepts `-d=` as a delimiter spelling. Clap otherwise parses it - // as an empty value assigned to `-d`. - let argv = argv - .into_iter() - .map(|arg| if arg == "-d=" { OsString::from("--delimiter==") } else { arg }) - .collect::>(); - let matches = match uu_app().try_get_matches_from(argv) { - Ok(matches) => matches, - Err(err) => { - let rendered = err.to_string(); - if err.use_stderr() { - let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); - return 1; - } - let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); - return 0; - } - }; + // GNU cut accepts `-d=` as a delimiter spelling. Clap otherwise parses it + // as an empty value assigned to `-d`. + let argv = argv + .into_iter() + .map(|arg| { + if arg == "-d=" { + OsString::from("--delimiter==") + } else { + arg + } + }) + .collect::>(); + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; - match cut_main(&matches) { - Ok(()) => pi_uutils_ctx::exit_code(), - Err(err) => { - let code = err.code(); - let _ = writeln!(pi_uutils_ctx::stderr(), "cut: {err}"); - if code == 0 { 1 } else { code } - } - } + match cut_main(&matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + let _ = writeln!(pi_uutils_ctx::stderr(), "cut: {err}"); + if code == 0 { 1 } else { code } + }, + } } fn cut_main(matches: &ArgMatches) -> UResult<()> { - let complement = matches.get_flag(options::COMPLEMENT); - let only_delimited = matches.get_flag(options::ONLY_DELIMITED); + let complement = matches.get_flag(options::COMPLEMENT); + let only_delimited = matches.get_flag(options::ONLY_DELIMITED); - let (delimiter, out_delimiter) = get_delimiters(&matches)?; - let line_ending = LineEnding::from_zero_flag(matches.get_flag(options::ZERO_TERMINATED)); + let (delimiter, out_delimiter) = get_delimiters(matches)?; + let line_ending = LineEnding::from_zero_flag(matches.get_flag(options::ZERO_TERMINATED)); - // Only one, and only one of cutting mode arguments, i.e. `-b`, `-c`, `-f`, - // is expected. The number of those arguments is used for parsing a cutting - // mode and handling the error cases. - let mode_args_count = [ - matches.indices_of(options::BYTES), - matches.indices_of(options::CHARACTERS), - matches.indices_of(options::FIELDS), - ] - .into_iter() - .map(|indices| indices.unwrap_or_default().count()) - .sum(); + // Only one, and only one of cutting mode arguments, i.e. `-b`, `-c`, `-f`, + // is expected. The number of those arguments is used for parsing a cutting + // mode and handling the error cases. + let mode_args_count = [ + matches.indices_of(options::BYTES), + matches.indices_of(options::CHARACTERS), + matches.indices_of(options::FIELDS), + ] + .into_iter() + .map(|indices| indices.unwrap_or_default().count()) + .sum(); - let mode_parse = match ( - mode_args_count, - matches.get_one::(options::BYTES), - matches.get_one::(options::CHARACTERS), - matches.get_one::(options::FIELDS), - ) { - (1, Some(byte_ranges), None, None) => { - list_to_ranges(byte_ranges, complement).map(|ranges| { - Mode::Bytes( - ranges, - Options { - out_delimiter, - line_ending, - field_opts: None, - }, - ) - }) - } + let mode_parse = match ( + mode_args_count, + matches.get_one::(options::BYTES), + matches.get_one::(options::CHARACTERS), + matches.get_one::(options::FIELDS), + ) { + (1, Some(byte_ranges), None, None) => list_to_ranges(byte_ranges, complement).map(|ranges| { + Mode::Bytes(ranges, Options { out_delimiter, line_ending, field_opts: None }) + }), - (1, None, Some(char_ranges), None) => { - list_to_ranges(char_ranges, complement).map(|ranges| { - Mode::Characters( - ranges, - Options { - out_delimiter, - line_ending, - field_opts: None, - }, - ) - }) - } + (1, None, Some(char_ranges), None) => list_to_ranges(char_ranges, complement).map(|ranges| { + Mode::Characters(ranges, Options { out_delimiter, line_ending, field_opts: None }) + }), - (1, None, None, Some(field_ranges)) => { - list_to_ranges(field_ranges, complement).map(|ranges| { - Mode::Fields( - ranges, - Options { - out_delimiter, - line_ending, - field_opts: Some(FieldOptions { - delimiter, - only_delimited, - }), - }, - ) - }) - } + (1, None, None, Some(field_ranges)) => { + list_to_ranges(field_ranges, complement).map(|ranges| { + Mode::Fields(ranges, Options { + out_delimiter, + line_ending, + field_opts: Some(FieldOptions { delimiter, only_delimited }), + }) + }) + }, - (2.., _, _, _) => Err("invalid usage: expects no more than one of --fields (-f), --chars (-c) or --bytes (-b)".to_owned()), - _ => Err("invalid usage: expects one of --fields (-f), --chars (-c) or --bytes (-b)".to_owned()), - }; + (2.., ..) => Err( + "invalid usage: expects no more than one of --fields (-f), --chars (-c) or --bytes (-b)" + .to_owned(), + ), + _ => { + Err("invalid usage: expects one of --fields (-f), --chars (-c) or --bytes (-b)".to_owned()) + }, + }; - let mode_parse = match mode_parse { - Err(_) => mode_parse, - Ok(mode) => match mode { - Mode::Bytes(_, _) | Mode::Characters(_, _) - if matches.contains_id(options::DELIMITER) => - { - Err("invalid input: The '--delimiter' ('-d') option can only be used when printing a sequence of fields".to_owned()) - } - Mode::Bytes(_, _) | Mode::Characters(_, _) - if matches.get_flag(options::WHITESPACE_DELIMITED) => - { - Err("invalid input: The '-w' option can only be used when printing a sequence of fields".to_owned()) - } - Mode::Bytes(_, _) | Mode::Characters(_, _) - if matches.get_flag(options::ONLY_DELIMITED) => - { - Err("invalid input: The '--only-delimited' ('-s') option can only be used when printing a sequence of fields".to_owned()) - } - _ => Ok(mode), - }, - }; + let mode_parse = match mode_parse { + Err(_) => mode_parse, + Ok(mode) => match mode { + Mode::Bytes(..) | Mode::Characters(..) if matches.contains_id(options::DELIMITER) => Err( + "invalid input: The '--delimiter' ('-d') option can only be used when printing a \ + sequence of fields" + .to_owned(), + ), + Mode::Bytes(..) | Mode::Characters(..) + if matches.get_flag(options::WHITESPACE_DELIMITED) => + { + Err( + "invalid input: The '-w' option can only be used when printing a sequence of fields" + .to_owned(), + ) + }, + Mode::Bytes(..) | Mode::Characters(..) if matches.get_flag(options::ONLY_DELIMITED) => { + Err( + "invalid input: The '--only-delimited' ('-s') option can only be used when \ + printing a sequence of fields" + .to_owned(), + ) + }, + _ => Ok(mode), + }, + }; - let mode = mode_parse.map_err(|e| USimpleError::new(1, e))?; - #[allow(clippy::unwrap_used, reason = "clap provides '-' by default")] - let files = matches.get_many::(options::FILE).unwrap(); + let mode = mode_parse.map_err(|e| USimpleError::new(1, e))?; + #[allow(clippy::unwrap_used, reason = "clap provides '-' by default")] + let files = matches.get_many::(options::FILE).unwrap(); - cut_files(files, &mode); + cut_files(files, &mode); - Ok(()) + Ok(()) } pub fn uu_app() -> Command { - Command::new("cut") - .version(env!("CARGO_PKG_VERSION")) - .override_usage(format_usage("cut OPTION... [FILE]...")) - .about("Print specified byte or field columns from each line of stdin or input files") - .after_help("Each invocation must specify exactly one of --bytes, --characters, or --fields. Use - as a file operand to read standard input.") - .infer_long_args(true) - // While `args_override_self(true)` for some arguments, such as `-d` - // and `--output-delimiter`, is consistent to the behavior of GNU cut, - // arguments related to cutting mode, i.e. `-b`, `-c`, `-f`, should - // cause an error when there is more than one of them, as described in - // the manual of GNU cut: "Use one, and only one of -b, -c or -f". - // `ArgAction::Append` is used on `-b`, `-c`, `-f` arguments, so that - // the occurrences of those could be counted and be handled accordingly. - .args_override_self(true) - .arg( - Arg::new(options::BYTES) - .short('b') - .long(options::BYTES) - .help("filter byte columns from the input source") - .allow_hyphen_values(true) - .value_name("LIST") - .action(ArgAction::Append), - ) - .arg( - Arg::new(options::CHARACTERS) - .short('c') - .long(options::CHARACTERS) - .help("alias for character mode") - .allow_hyphen_values(true) - .value_name("LIST") - .action(ArgAction::Append), - ) - .arg( - Arg::new(options::DELIMITER) - .short('d') - .long(options::DELIMITER) - .value_parser(ValueParser::os_string()) - .help("specify the delimiter character that separates fields in the input source (default: Tab)") - .value_name("DELIM"), - ) - .arg( - Arg::new(options::WHITESPACE_DELIMITED) - .short('w') - .help("use any amount of whitespace (Space, Tab) to separate fields (FreeBSD extension)") - .value_name("WHITESPACE") - .action(ArgAction::SetTrue), - ) - .arg( - Arg::new(options::FIELDS) - .short('f') - .long(options::FIELDS) - .help("filter field columns from the input source") - .allow_hyphen_values(true) - .value_name("LIST") - .action(ArgAction::Append), - ) - .arg( - Arg::new(options::COMPLEMENT) - .long(options::COMPLEMENT) - .help("invert the filter, displaying all but the selected columns") - .action(ArgAction::SetTrue), - ) - .arg( - Arg::new(options::ONLY_DELIMITED) - .short('s') - .long(options::ONLY_DELIMITED) - .help("in field mode, only print lines which contain the delimiter") - .action(ArgAction::SetTrue), - ) - .arg( - Arg::new(options::ZERO_TERMINATED) - .short('z') - .long(options::ZERO_TERMINATED) - .help("filter records separated by NUL instead of newline") - .action(ArgAction::SetTrue), - ) - .arg( - Arg::new(options::OUTPUT_DELIMITER) - .long(options::OUTPUT_DELIMITER) - .value_parser(ValueParser::os_string()) - .help("in field mode, replace the delimiter in output lines with this argument") - .value_name("NEW_DELIM"), - ) - .arg( - Arg::new(options::FILE) - .hide(true) - .action(ArgAction::Append) - .value_hint(clap::ValueHint::FilePath) - .default_value("-") - .value_parser(clap::value_parser!(OsString)), - ) - .arg( - Arg::new(options::NOTHING) - .short('n') - .help("(ignored)") - .action(ArgAction::SetTrue), - ) + Command::new("cut") + .version(env!("CARGO_PKG_VERSION")) + .override_usage(format_usage("cut OPTION... [FILE]...")) + .about("Print specified byte or field columns from each line of stdin or input files") + .after_help( + "Each invocation must specify exactly one of --bytes, --characters, or --fields. Use - \ + as a file operand to read standard input.", + ) + .infer_long_args(true) + // While `args_override_self(true)` for some arguments, such as `-d` + // and `--output-delimiter`, is consistent to the behavior of GNU cut, + // arguments related to cutting mode, i.e. `-b`, `-c`, `-f`, should + // cause an error when there is more than one of them, as described in + // the manual of GNU cut: "Use one, and only one of -b, -c or -f". + // `ArgAction::Append` is used on `-b`, `-c`, `-f` arguments, so that + // the occurrences of those could be counted and be handled accordingly. + .args_override_self(true) + .arg( + Arg::new(options::BYTES) + .short('b') + .long(options::BYTES) + .help("filter byte columns from the input source") + .allow_hyphen_values(true) + .value_name("LIST") + .action(ArgAction::Append), + ) + .arg( + Arg::new(options::CHARACTERS) + .short('c') + .long(options::CHARACTERS) + .help("alias for character mode") + .allow_hyphen_values(true) + .value_name("LIST") + .action(ArgAction::Append), + ) + .arg( + Arg::new(options::DELIMITER) + .short('d') + .long(options::DELIMITER) + .value_parser(ValueParser::os_string()) + .help( + "specify the delimiter character that separates fields in the input source \ + (default: Tab)", + ) + .value_name("DELIM"), + ) + .arg( + Arg::new(options::WHITESPACE_DELIMITED) + .short('w') + .help( + "use any amount of whitespace (Space, Tab) to separate fields (FreeBSD extension)", + ) + .value_name("WHITESPACE") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::FIELDS) + .short('f') + .long(options::FIELDS) + .help("filter field columns from the input source") + .allow_hyphen_values(true) + .value_name("LIST") + .action(ArgAction::Append), + ) + .arg( + Arg::new(options::COMPLEMENT) + .long(options::COMPLEMENT) + .help("invert the filter, displaying all but the selected columns") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::ONLY_DELIMITED) + .short('s') + .long(options::ONLY_DELIMITED) + .help("in field mode, only print lines which contain the delimiter") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::ZERO_TERMINATED) + .short('z') + .long(options::ZERO_TERMINATED) + .help("filter records separated by NUL instead of newline") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::OUTPUT_DELIMITER) + .long(options::OUTPUT_DELIMITER) + .value_parser(ValueParser::os_string()) + .help("in field mode, replace the delimiter in output lines with this argument") + .value_name("NEW_DELIM"), + ) + .arg( + Arg::new(options::FILE) + .hide(true) + .action(ArgAction::Append) + .value_hint(clap::ValueHint::FilePath) + .default_value("-") + .value_parser(clap::value_parser!(OsString)), + ) + .arg( + Arg::new(options::NOTHING) + .short('n') + .help("(ignored)") + .action(ArgAction::SetTrue), + ) } diff --git a/crates/vendor/uu-cut/src/matcher.rs b/crates/vendor/uu-cut/src/matcher.rs index c1be9fb5e..72973edf0 100644 --- a/crates/vendor/uu-cut/src/matcher.rs +++ b/crates/vendor/uu-cut/src/matcher.rs @@ -6,112 +6,113 @@ use memchr::{memchr, memchr2}; // Find the next matching byte sequence positions -// Return (first, last) where haystack[first..last] corresponds to the matched pattern +// Return (first, last) where haystack[first..last] corresponds to the matched +// pattern pub trait Matcher { - fn next_match(&self, haystack: &[u8]) -> Option<(usize, usize)>; + fn next_match(&self, haystack: &[u8]) -> Option<(usize, usize)>; } // Matches for the exact byte sequence pattern pub struct ExactMatcher<'a> { - needle: &'a [u8], + needle: &'a [u8], } impl<'a> ExactMatcher<'a> { - pub fn new(needle: &'a [u8]) -> Self { - assert!(!needle.is_empty()); - Self { needle } - } + pub fn new(needle: &'a [u8]) -> Self { + assert!(!needle.is_empty()); + Self { needle } + } } impl Matcher for ExactMatcher<'_> { - fn next_match(&self, haystack: &[u8]) -> Option<(usize, usize)> { - let mut pos = 0usize; - loop { - let match_idx = memchr(self.needle[0], &haystack[pos..])?; - let match_idx = match_idx + pos; // account for starting from pos + fn next_match(&self, haystack: &[u8]) -> Option<(usize, usize)> { + let mut pos = 0usize; + loop { + let match_idx = memchr(self.needle[0], &haystack[pos..])?; + let match_idx = match_idx + pos; // account for starting from pos - if self.needle.len() == 1 || haystack[match_idx + 1..].starts_with(&self.needle[1..]) { - return Some((match_idx, match_idx + self.needle.len())); - } + if self.needle.len() == 1 || haystack[match_idx + 1..].starts_with(&self.needle[1..]) { + return Some((match_idx, match_idx + self.needle.len())); + } - pos = match_idx + 1; - } - } + pos = match_idx + 1; + } + } } // Matches for any number of SPACE or TAB pub struct WhitespaceMatcher {} impl Matcher for WhitespaceMatcher { - fn next_match(&self, haystack: &[u8]) -> Option<(usize, usize)> { - let match_idx = memchr2(b' ', b'\t', haystack)?; - let mut skip = match_idx + 1; + fn next_match(&self, haystack: &[u8]) -> Option<(usize, usize)> { + let match_idx = memchr2(b' ', b'\t', haystack)?; + let mut skip = match_idx + 1; - while skip < haystack.len() { - match haystack[skip] { - b' ' | b'\t' => skip += 1, - _ => break, - } - } + while skip < haystack.len() { + match haystack[skip] { + b' ' | b'\t' => skip += 1, + _ => break, + } + } - Some((match_idx, skip)) - } + Some((match_idx, skip)) + } } #[cfg(test)] mod matcher_tests { - use super::*; + use super::*; - #[test] - fn test_exact_matcher_single_byte() { - let matcher = ExactMatcher::new(":".as_bytes()); - // spell-checker:disable - assert_eq!(matcher.next_match("".as_bytes()), None); - assert_eq!(matcher.next_match(":".as_bytes()), Some((0, 1))); - assert_eq!(matcher.next_match(":abcxyz".as_bytes()), Some((0, 1))); - assert_eq!(matcher.next_match("abc:xyz".as_bytes()), Some((3, 4))); - assert_eq!(matcher.next_match("abcxyz:".as_bytes()), Some((6, 7))); - assert_eq!(matcher.next_match("abcxyz".as_bytes()), None); - // spell-checker:enable - } + #[test] + fn test_exact_matcher_single_byte() { + let matcher = ExactMatcher::new(":".as_bytes()); + // spell-checker:disable + assert_eq!(matcher.next_match("".as_bytes()), None); + assert_eq!(matcher.next_match(":".as_bytes()), Some((0, 1))); + assert_eq!(matcher.next_match(":abcxyz".as_bytes()), Some((0, 1))); + assert_eq!(matcher.next_match("abc:xyz".as_bytes()), Some((3, 4))); + assert_eq!(matcher.next_match("abcxyz:".as_bytes()), Some((6, 7))); + assert_eq!(matcher.next_match("abcxyz".as_bytes()), None); + // spell-checker:enable + } - #[test] - fn test_exact_matcher_multi_bytes() { - let matcher = ExactMatcher::new("<>".as_bytes()); - // spell-checker:disable - assert_eq!(matcher.next_match("".as_bytes()), None); - assert_eq!(matcher.next_match("<>".as_bytes()), Some((0, 2))); - assert_eq!(matcher.next_match("<>abcxyz".as_bytes()), Some((0, 2))); - assert_eq!(matcher.next_match("abc<>xyz".as_bytes()), Some((3, 5))); - assert_eq!(matcher.next_match("abcxyz<>".as_bytes()), Some((6, 8))); - assert_eq!(matcher.next_match("abcxyz".as_bytes()), None); - // spell-checker:enable - } + #[test] + fn test_exact_matcher_multi_bytes() { + let matcher = ExactMatcher::new("<>".as_bytes()); + // spell-checker:disable + assert_eq!(matcher.next_match("".as_bytes()), None); + assert_eq!(matcher.next_match("<>".as_bytes()), Some((0, 2))); + assert_eq!(matcher.next_match("<>abcxyz".as_bytes()), Some((0, 2))); + assert_eq!(matcher.next_match("abc<>xyz".as_bytes()), Some((3, 5))); + assert_eq!(matcher.next_match("abcxyz<>".as_bytes()), Some((6, 8))); + assert_eq!(matcher.next_match("abcxyz".as_bytes()), None); + // spell-checker:enable + } - #[test] - fn test_whitespace_matcher_single_space() { - let matcher = WhitespaceMatcher {}; - // spell-checker:disable - assert_eq!(matcher.next_match("".as_bytes()), None); - assert_eq!(matcher.next_match(" ".as_bytes()), Some((0, 1))); - assert_eq!(matcher.next_match("\tabcxyz".as_bytes()), Some((0, 1))); - assert_eq!(matcher.next_match("abc\txyz".as_bytes()), Some((3, 4))); - assert_eq!(matcher.next_match("abcxyz ".as_bytes()), Some((6, 7))); - assert_eq!(matcher.next_match("abcxyz".as_bytes()), None); - // spell-checker:enable - } + #[test] + fn test_whitespace_matcher_single_space() { + let matcher = WhitespaceMatcher {}; + // spell-checker:disable + assert_eq!(matcher.next_match("".as_bytes()), None); + assert_eq!(matcher.next_match(" ".as_bytes()), Some((0, 1))); + assert_eq!(matcher.next_match("\tabcxyz".as_bytes()), Some((0, 1))); + assert_eq!(matcher.next_match("abc\txyz".as_bytes()), Some((3, 4))); + assert_eq!(matcher.next_match("abcxyz ".as_bytes()), Some((6, 7))); + assert_eq!(matcher.next_match("abcxyz".as_bytes()), None); + // spell-checker:enable + } - #[test] - fn test_whitespace_matcher_multi_spaces() { - let matcher = WhitespaceMatcher {}; - // spell-checker:disable - assert_eq!(matcher.next_match("".as_bytes()), None); - assert_eq!(matcher.next_match(" \t ".as_bytes()), Some((0, 3))); - assert_eq!(matcher.next_match("\t\tabcxyz".as_bytes()), Some((0, 2))); - assert_eq!(matcher.next_match("abc \txyz".as_bytes()), Some((3, 5))); - assert_eq!(matcher.next_match("abcxyz ".as_bytes()), Some((6, 8))); - assert_eq!(matcher.next_match("abcxyz".as_bytes()), None); - // spell-checker:enable - } + #[test] + fn test_whitespace_matcher_multi_spaces() { + let matcher = WhitespaceMatcher {}; + // spell-checker:disable + assert_eq!(matcher.next_match("".as_bytes()), None); + assert_eq!(matcher.next_match(" \t ".as_bytes()), Some((0, 3))); + assert_eq!(matcher.next_match("\t\tabcxyz".as_bytes()), Some((0, 2))); + assert_eq!(matcher.next_match("abc \txyz".as_bytes()), Some((3, 5))); + assert_eq!(matcher.next_match("abcxyz ".as_bytes()), Some((6, 8))); + assert_eq!(matcher.next_match("abcxyz".as_bytes()), None); + // spell-checker:enable + } } diff --git a/crates/vendor/uu-cut/src/searcher.rs b/crates/vendor/uu-cut/src/searcher.rs index a25fa7909..58af473f2 100644 --- a/crates/vendor/uu-cut/src/searcher.rs +++ b/crates/vendor/uu-cut/src/searcher.rs @@ -9,172 +9,168 @@ use super::matcher::Matcher; // Generic searcher that relies on a specific matcher pub struct Searcher<'a, 'b, M: Matcher> { - matcher: &'a M, - haystack: &'b [u8], - position: usize, + matcher: &'a M, + haystack: &'b [u8], + position: usize, } impl<'a, 'b, M: Matcher> Searcher<'a, 'b, M> { - pub fn new(matcher: &'a M, haystack: &'b [u8]) -> Self { - Self { - matcher, - haystack, - position: 0, - } - } + pub fn new(matcher: &'a M, haystack: &'b [u8]) -> Self { + Self { matcher, haystack, position: 0 } + } } // Iterate over field delimiters -// Returns (first, last) positions of each sequence, where `haystack[first..last]` -// corresponds to the delimiter. +// Returns (first, last) positions of each sequence, where +// `haystack[first..last]` corresponds to the delimiter. impl Iterator for Searcher<'_, '_, M> { - type Item = (usize, usize); + type Item = (usize, usize); - fn next(&mut self) -> Option { - let (first, last) = self.matcher.next_match(&self.haystack[self.position..])?; - let result = (first + self.position, last + self.position); - self.position += last; + fn next(&mut self) -> Option { + let (first, last) = self.matcher.next_match(&self.haystack[self.position..])?; + let result = (first + self.position, last + self.position); + self.position += last; - Some(result) - } + Some(result) + } } #[cfg(test)] mod exact_searcher_tests { - use super::super::matcher::ExactMatcher; - use super::*; + use super::{super::matcher::ExactMatcher, *}; - #[test] - fn test_normal() { - let matcher = ExactMatcher::new("a".as_bytes()); - let iter = Searcher::new(&matcher, "a.a.a".as_bytes()); - let items: Vec<(usize, usize)> = iter.collect(); - assert_eq!(vec![(0, 1), (2, 3), (4, 5)], items); - } + #[test] + fn test_normal() { + let matcher = ExactMatcher::new("a".as_bytes()); + let iter = Searcher::new(&matcher, "a.a.a".as_bytes()); + let items: Vec<(usize, usize)> = iter.collect(); + assert_eq!(vec![(0, 1), (2, 3), (4, 5)], items); + } - #[test] - fn test_empty() { - let matcher = ExactMatcher::new("a".as_bytes()); - let iter = Searcher::new(&matcher, "".as_bytes()); - let items: Vec<(usize, usize)> = iter.collect(); - assert!(items.is_empty()); - } + #[test] + fn test_empty() { + let matcher = ExactMatcher::new("a".as_bytes()); + let iter = Searcher::new(&matcher, "".as_bytes()); + let items: Vec<(usize, usize)> = iter.collect(); + assert!(items.is_empty()); + } - fn test_multibyte(line: &[u8], expected: &[(usize, usize)]) { - let matcher = ExactMatcher::new("ab".as_bytes()); - let iter = Searcher::new(&matcher, line); - let items: Vec<(usize, usize)> = iter.collect(); - assert_eq!(expected, items); - } + fn test_multibyte(line: &[u8], expected: &[(usize, usize)]) { + let matcher = ExactMatcher::new("ab".as_bytes()); + let iter = Searcher::new(&matcher, line); + let items: Vec<(usize, usize)> = iter.collect(); + assert_eq!(expected, items); + } - #[test] - fn test_multibyte_normal() { - test_multibyte("...ab...ab...".as_bytes(), &[(3, 5), (8, 10)]); - } + #[test] + fn test_multibyte_normal() { + test_multibyte("...ab...ab...".as_bytes(), &[(3, 5), (8, 10)]); + } - #[test] - fn test_multibyte_needle_head_at_end() { - test_multibyte("a".as_bytes(), &[]); - } + #[test] + fn test_multibyte_needle_head_at_end() { + test_multibyte("a".as_bytes(), &[]); + } - #[test] - fn test_multibyte_starting_needle() { - test_multibyte("ab...ab...".as_bytes(), &[(0, 2), (5, 7)]); - } + #[test] + fn test_multibyte_starting_needle() { + test_multibyte("ab...ab...".as_bytes(), &[(0, 2), (5, 7)]); + } - #[test] - fn test_multibyte_trailing_needle() { - test_multibyte("...ab...ab".as_bytes(), &[(3, 5), (8, 10)]); - } + #[test] + fn test_multibyte_trailing_needle() { + test_multibyte("...ab...ab".as_bytes(), &[(3, 5), (8, 10)]); + } - #[test] - fn test_multibyte_first_byte_false_match() { - test_multibyte("aA..aCaC..ab..aD".as_bytes(), &[(10, 12)]); - } + #[test] + fn test_multibyte_first_byte_false_match() { + test_multibyte("aA..aCaC..ab..aD".as_bytes(), &[(10, 12)]); + } - #[test] - fn test_searcher_with_exact_matcher() { - let matcher = ExactMatcher::new("<>".as_bytes()); - let haystack = "<><>a<>b<><>cd<><>".as_bytes(); - let mut searcher = Searcher::new(&matcher, haystack); - assert_eq!(searcher.next(), Some((0, 2))); - assert_eq!(searcher.next(), Some((2, 4))); - assert_eq!(searcher.next(), Some((5, 7))); - assert_eq!(searcher.next(), Some((8, 10))); - assert_eq!(searcher.next(), Some((10, 12))); - assert_eq!(searcher.next(), Some((14, 16))); - assert_eq!(searcher.next(), Some((16, 18))); - assert_eq!(searcher.next(), None); - assert_eq!(searcher.next(), None); - } + #[test] + fn test_searcher_with_exact_matcher() { + let matcher = ExactMatcher::new("<>".as_bytes()); + let haystack = "<><>a<>b<><>cd<><>".as_bytes(); + let mut searcher = Searcher::new(&matcher, haystack); + assert_eq!(searcher.next(), Some((0, 2))); + assert_eq!(searcher.next(), Some((2, 4))); + assert_eq!(searcher.next(), Some((5, 7))); + assert_eq!(searcher.next(), Some((8, 10))); + assert_eq!(searcher.next(), Some((10, 12))); + assert_eq!(searcher.next(), Some((14, 16))); + assert_eq!(searcher.next(), Some((16, 18))); + assert_eq!(searcher.next(), None); + assert_eq!(searcher.next(), None); + } } #[cfg(test)] mod whitespace_searcher_tests { - use super::super::matcher::WhitespaceMatcher; - use super::*; + use super::{super::matcher::WhitespaceMatcher, *}; - #[test] - fn test_space() { - let matcher = WhitespaceMatcher {}; - let iter = Searcher::new(&matcher, " . . ".as_bytes()); - let items: Vec<(usize, usize)> = iter.collect(); - assert_eq!(vec![(0, 1), (2, 3), (4, 5)], items); - } + #[test] + fn test_space() { + let matcher = WhitespaceMatcher {}; + let iter = Searcher::new(&matcher, " . . ".as_bytes()); + let items: Vec<(usize, usize)> = iter.collect(); + assert_eq!(vec![(0, 1), (2, 3), (4, 5)], items); + } - #[test] - fn test_tab() { - let matcher = WhitespaceMatcher {}; - let iter = Searcher::new(&matcher, "\t.\t.\t".as_bytes()); - let items: Vec<(usize, usize)> = iter.collect(); - assert_eq!(vec![(0, 1), (2, 3), (4, 5)], items); - } + #[test] + fn test_tab() { + let matcher = WhitespaceMatcher {}; + let iter = Searcher::new(&matcher, "\t.\t.\t".as_bytes()); + let items: Vec<(usize, usize)> = iter.collect(); + assert_eq!(vec![(0, 1), (2, 3), (4, 5)], items); + } - #[test] - fn test_empty() { - let matcher = WhitespaceMatcher {}; - let iter = Searcher::new(&matcher, "".as_bytes()); - let items: Vec<(usize, usize)> = iter.collect(); - assert!(items.is_empty()); - } + #[test] + fn test_empty() { + let matcher = WhitespaceMatcher {}; + let iter = Searcher::new(&matcher, "".as_bytes()); + let items: Vec<(usize, usize)> = iter.collect(); + assert!(items.is_empty()); + } - fn test_multispace(line: &[u8], expected: &[(usize, usize)]) { - let matcher = WhitespaceMatcher {}; - let iter = Searcher::new(&matcher, line); - let items: Vec<(usize, usize)> = iter.collect(); - assert_eq!(expected, items); - } + fn test_multispace(line: &[u8], expected: &[(usize, usize)]) { + let matcher = WhitespaceMatcher {}; + let iter = Searcher::new(&matcher, line); + let items: Vec<(usize, usize)> = iter.collect(); + assert_eq!(expected, items); + } - #[test] - fn test_multispace_normal() { - test_multispace( - "... ... \t...\t ... \t ...".as_bytes(), - &[(3, 5), (8, 10), (13, 15), (18, 21)], - ); - } + #[test] + fn test_multispace_normal() { + test_multispace("... ... \t...\t ... \t ...".as_bytes(), &[ + (3, 5), + (8, 10), + (13, 15), + (18, 21), + ]); + } - #[test] - fn test_multispace_begin() { - test_multispace(" \t\t...".as_bytes(), &[(0, 3)]); - } + #[test] + fn test_multispace_begin() { + test_multispace(" \t\t...".as_bytes(), &[(0, 3)]); + } - #[test] - fn test_multispace_end() { - test_multispace("...\t ".as_bytes(), &[(3, 6)]); - } + #[test] + fn test_multispace_end() { + test_multispace("...\t ".as_bytes(), &[(3, 6)]); + } - #[test] - fn test_searcher_with_whitespace_matcher() { - let matcher = WhitespaceMatcher {}; - let haystack = "\t a b \t cd\t\t".as_bytes(); - let mut searcher = Searcher::new(&matcher, haystack); - assert_eq!(searcher.next(), Some((0, 2))); - assert_eq!(searcher.next(), Some((3, 4))); - assert_eq!(searcher.next(), Some((5, 8))); - assert_eq!(searcher.next(), Some((10, 12))); - assert_eq!(searcher.next(), None); - assert_eq!(searcher.next(), None); - } + #[test] + fn test_searcher_with_whitespace_matcher() { + let matcher = WhitespaceMatcher {}; + let haystack = "\t a b \t cd\t\t".as_bytes(); + let mut searcher = Searcher::new(&matcher, haystack); + assert_eq!(searcher.next(), Some((0, 2))); + assert_eq!(searcher.next(), Some((3, 4))); + assert_eq!(searcher.next(), Some((5, 8))); + assert_eq!(searcher.next(), Some((10, 12))); + assert_eq!(searcher.next(), None); + assert_eq!(searcher.next(), None); + } } diff --git a/crates/vendor/uu-dirname/src/dirname.rs b/crates/vendor/uu-dirname/src/dirname.rs index 603f827a9..c2364ee0f 100644 --- a/crates/vendor/uu-dirname/src/dirname.rs +++ b/crates/vendor/uu-dirname/src/dirname.rs @@ -5,16 +5,15 @@ // pi-uutils: modified for in-process embedding using pi-uutils-ctx streams. -use clap::{Arg, ArgAction, Command, ArgMatches}; -use std::borrow::Cow; -use std::ffi::OsString; -use std::io::Write; -use uucore::error::{UResult, UUsageError}; +use std::{borrow::Cow, ffi::OsString, io::Write}; + +use clap::{Arg, ArgAction, ArgMatches, Command}; use pi_uutils_ctx::format_usage; +use uucore::error::{UResult, UUsageError}; mod options { - pub const ZERO: &str = "zero"; - pub const DIR: &str = "dir"; + pub const ZERO: &str = "zero"; + pub const DIR: &str = "dir"; } /// Perform dirname as pure string manipulation per POSIX/GNU behavior. @@ -38,63 +37,63 @@ mod options { /// /// See issue #8910 and similar fix in basename (#8373, commit c5268a897). fn dirname_string_manipulation(path_bytes: &[u8]) -> Cow<'_, [u8]> { - if path_bytes.is_empty() { - return Cow::Borrowed(b"."); - } + if path_bytes.is_empty() { + return Cow::Borrowed(b"."); + } - let mut bytes = path_bytes; + let mut bytes = path_bytes; - // Step 1: Strip trailing slashes (but not if the entire path is slashes) - let all_slashes = bytes.iter().all(|&b| b == b'/'); - if all_slashes { - return Cow::Borrowed(b"/"); - } + // Step 1: Strip trailing slashes (but not if the entire path is slashes) + let all_slashes = bytes.iter().all(|&b| b == b'/'); + if all_slashes { + return Cow::Borrowed(b"/"); + } - while bytes.len() > 1 && bytes.ends_with(b"/") { - bytes = &bytes[..bytes.len() - 1]; - } + while bytes.len() > 1 && bytes.ends_with(b"/") { + bytes = &bytes[..bytes.len() - 1]; + } - // Step 2: Check if it ends with `/.` and strip the `/+.` pattern - if bytes.ends_with(b".") && bytes.len() >= 2 { - let dot_pos = bytes.len() - 1; - if bytes[dot_pos - 1] == b'/' { - // Find where the slashes before the dot start - let mut slash_start = dot_pos - 1; - while slash_start > 0 && bytes[slash_start - 1] == b'/' { - slash_start -= 1; - } - // Return the stripped result - if slash_start == 0 { - // Result would be empty - return if path_bytes.starts_with(b"/") { - Cow::Borrowed(b"/") - } else { - Cow::Borrowed(b".") - }; - } - return Cow::Borrowed(&bytes[..slash_start]); - } - } + // Step 2: Check if it ends with `/.` and strip the `/+.` pattern + if bytes.ends_with(b".") && bytes.len() >= 2 { + let dot_pos = bytes.len() - 1; + if bytes[dot_pos - 1] == b'/' { + // Find where the slashes before the dot start + let mut slash_start = dot_pos - 1; + while slash_start > 0 && bytes[slash_start - 1] == b'/' { + slash_start -= 1; + } + // Return the stripped result + if slash_start == 0 { + // Result would be empty + return if path_bytes.starts_with(b"/") { + Cow::Borrowed(b"/") + } else { + Cow::Borrowed(b".") + }; + } + return Cow::Borrowed(&bytes[..slash_start]); + } + } - // Step 3: Normal dirname - find last / and remove everything after it - if let Some(last_slash_pos) = bytes.iter().rposition(|&b| b == b'/') { - // Found a slash, remove everything after it - let mut result = &bytes[..last_slash_pos]; + // Step 3: Normal dirname - find last / and remove everything after it + if let Some(last_slash_pos) = bytes.iter().rposition(|&b| b == b'/') { + // Found a slash, remove everything after it + let mut result = &bytes[..last_slash_pos]; - // Strip trailing slashes from result (but keep at least one if at the start) - while result.len() > 1 && result.ends_with(b"/") { - result = &result[..result.len() - 1]; - } + // Strip trailing slashes from result (but keep at least one if at the start) + while result.len() > 1 && result.ends_with(b"/") { + result = &result[..result.len() - 1]; + } - if result.is_empty() { - return Cow::Borrowed(b"/"); - } + if result.is_empty() { + return Cow::Borrowed(b"/"); + } - return Cow::Borrowed(result); - } + return Cow::Borrowed(result); + } - // No slash found, return "." - Cow::Borrowed(b".") + // No slash found, return "." + Cow::Borrowed(b".") } /// In-process builtin entry point. Unlike upstream's `uumain`, this parses the @@ -102,196 +101,198 @@ fn dirname_string_manipulation(path_bytes: &[u8]) -> Cow<'_, [u8]> { /// streams, and maps the `UResult` to an exit code, so it is safe to run inside /// the host shell process. pub fn run(argv: Vec) -> i32 { - let matches = match uu_app().try_get_matches_from(argv) { - Ok(matches) => matches, - Err(err) => { - let rendered = err.to_string(); - if err.use_stderr() { - let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); - return 1; - } - let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); - return 0; - } - }; - match dirname_main(&matches) { - Ok(()) => pi_uutils_ctx::exit_code(), - Err(err) => { - let code = err.code(); - let _ = writeln!(pi_uutils_ctx::stderr(), "dirname: {err}"); - if code == 0 { 1 } else { code } - } - } + let matches = match uu_app().try_get_matches_from(argv) { + Ok(matches) => matches, + Err(err) => { + let rendered = err.to_string(); + if err.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + match dirname_main(&matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(err) => { + let code = err.code(); + let _ = writeln!(pi_uutils_ctx::stderr(), "dirname: {err}"); + if code == 0 { 1 } else { code } + }, + } } fn dirname_main(matches: &ArgMatches) -> UResult<()> { - let dirnames: Vec = matches - .get_many::(options::DIR) - .unwrap_or_default() - .cloned() - .collect(); + let dirnames: Vec = matches + .get_many::(options::DIR) + .unwrap_or_default() + .cloned() + .collect(); - if dirnames.is_empty() { - return Err(UUsageError::new(1, "missing operand".to_string())); - } + if dirnames.is_empty() { + return Err(UUsageError::new(1, "missing operand".to_string())); + } - let line_ending = if matches.get_flag(options::ZERO) { - b"\0" as &[u8] - } else { - b"\n" as &[u8] - }; + let line_ending = if matches.get_flag(options::ZERO) { + b"\0" as &[u8] + } else { + b"\n" as &[u8] + }; - let mut stdout = pi_uutils_ctx::stdout(); + let mut stdout = pi_uutils_ctx::stdout(); - for path in &dirnames { - let path_bytes = uucore::os_str_as_bytes(path.as_os_str())?; - let result = dirname_string_manipulation(path_bytes); + for path in &dirnames { + let path_bytes = uucore::os_str_as_bytes(path.as_os_str())?; + let result = dirname_string_manipulation(path_bytes); - stdout.write_all(&result)?; - stdout.write_all(line_ending)?; - } + stdout.write_all(&result)?; + stdout.write_all(line_ending)?; + } - Ok(()) + Ok(()) } pub fn uu_app() -> Command { - Command::new("dirname") - .about("Strip last component from file name") - .version(uucore::crate_version!()) - .override_usage(format_usage("dirname [OPTION] NAME...")) - .args_override_self(true) - .infer_long_args(true) - .after_help("Output each NAME with its last non-slash component and trailing slashes\n removed; if NAME contains no /'s, output '.' (meaning the current directory).") - .arg( - Arg::new(options::ZERO) - .long(options::ZERO) - .short('z') - .help("separate output with NUL rather than newline") - .action(ArgAction::SetTrue), - ) - .arg( - Arg::new(options::DIR) - .hide(true) - .action(ArgAction::Append) - .value_hint(clap::ValueHint::AnyPath) - .value_parser(clap::value_parser!(OsString)), - ) + Command::new("dirname") + .about("Strip last component from file name") + .version(uucore::crate_version!()) + .override_usage(format_usage("dirname [OPTION] NAME...")) + .args_override_self(true) + .infer_long_args(true) + .after_help( + "Output each NAME with its last non-slash component and trailing slashes\n removed; if \ + NAME contains no /'s, output '.' (meaning the current directory).", + ) + .arg( + Arg::new(options::ZERO) + .long(options::ZERO) + .short('z') + .help("separate output with NUL rather than newline") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::DIR) + .hide(true) + .action(ArgAction::Append) + .value_hint(clap::ValueHint::AnyPath) + .value_parser(clap::value_parser!(OsString)), + ) } #[cfg(test)] mod tests { - use super::*; - use std::sync::Arc; - use std::collections::HashMap; - use std::path::PathBuf; - use pi_uutils_ctx::ScopeIo; - use parking_lot::Mutex; + use std::{collections::HashMap, path::PathBuf, sync::Arc}; - fn run_test(args: Vec<&str>) -> (i32, String, String) { - let stdout_buf = Arc::new(Mutex::new(Vec::new())); - let stderr_buf = Arc::new(Mutex::new(Vec::new())); + use parking_lot::Mutex; + use pi_uutils_ctx::ScopeIo; - #[derive(Clone)] - struct SharedWriter { - buf: Arc>>, - } - impl Write for SharedWriter { - fn write(&mut self, buf: &[u8]) -> std::io::Result { - self.buf.lock().write(buf) - } - fn flush(&mut self) -> std::io::Result<()> { - self.buf.lock().flush() - } - } + use super::*; - let io = ScopeIo { - stdin: Box::new(std::io::empty()), - stdin_fd: None, - stdin_is_search_input: false, - stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), - stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), - cwd: PathBuf::from("."), - env: HashMap::new(), - cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), - }; + fn run_test(args: Vec<&str>) -> (i32, String, String) { + let stdout_buf = Arc::new(Mutex::new(Vec::new())); + let stderr_buf = Arc::new(Mutex::new(Vec::new())); - let argv: Vec = std::iter::once("dirname") - .chain(args) - .map(OsString::from) - .collect(); + #[derive(Clone)] + struct SharedWriter { + buf: Arc>>, + } + impl Write for SharedWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.buf.lock().write(buf) + } - let code = pi_uutils_ctx::scope(io, || { - run(argv) - }); + fn flush(&mut self) -> std::io::Result<()> { + self.buf.lock().flush() + } + } - let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); - let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); + let io = ScopeIo { + stdin: Box::new(std::io::empty()), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }), + stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }), + cwd: PathBuf::from("."), + env: HashMap::new(), + cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)), + }; - (code, out_str, err_str) - } + let argv: Vec = std::iter::once("dirname") + .chain(args) + .map(OsString::from) + .collect(); - #[test] - fn test_normal() { - let (code, stdout, stderr) = run_test(vec!["foo/bar"]); - assert_eq!(code, 0); - assert_eq!(stdout, "foo\n"); - assert_eq!(stderr, ""); - } + let code = pi_uutils_ctx::scope(io, || run(argv)); - #[test] - fn test_trailing_slash() { - let (code, stdout, stderr) = run_test(vec!["foo/bar/"]); - assert_eq!(code, 0); - assert_eq!(stdout, "foo\n"); - assert_eq!(stderr, ""); - } + let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap(); + let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap(); - #[test] - fn test_root() { - let (code, stdout, stderr) = run_test(vec!["/"]); - assert_eq!(code, 0); - assert_eq!(stdout, "/\n"); - assert_eq!(stderr, ""); - } + (code, out_str, err_str) + } - #[test] - fn test_multiple() { - let (code, stdout, stderr) = run_test(vec!["a/b", "c/d/e"]); - assert_eq!(code, 0); - assert_eq!(stdout, "a\nc/d\n"); - assert_eq!(stderr, ""); - } + #[test] + fn test_normal() { + let (code, stdout, stderr) = run_test(vec!["foo/bar"]); + assert_eq!(code, 0); + assert_eq!(stdout, "foo\n"); + assert_eq!(stderr, ""); + } - #[test] - fn test_zero_delimited() { - let (code, stdout, stderr) = run_test(vec!["-z", "a/b", "c/d/e"]); - assert_eq!(code, 0); - assert_eq!(stdout, "a\0c/d\0"); - assert_eq!(stderr, ""); - } + #[test] + fn test_trailing_slash() { + let (code, stdout, stderr) = run_test(vec!["foo/bar/"]); + assert_eq!(code, 0); + assert_eq!(stdout, "foo\n"); + assert_eq!(stderr, ""); + } - #[test] - fn test_help() { - let (code, stdout, stderr) = run_test(vec!["--help"]); - assert_eq!(code, 0); - assert!(stdout.contains("Usage:")); - assert!(stdout.contains("Strip last component")); - assert_eq!(stderr, ""); - } + #[test] + fn test_root() { + let (code, stdout, stderr) = run_test(vec!["/"]); + assert_eq!(code, 0); + assert_eq!(stdout, "/\n"); + assert_eq!(stderr, ""); + } - #[test] - fn test_invalid_arg() { - let (code, stdout, stderr) = run_test(vec!["--invalid-flag"]); - assert_eq!(code, 1); - assert_eq!(stdout, ""); - assert!(stderr.contains("unexpected argument")); - } + #[test] + fn test_multiple() { + let (code, stdout, stderr) = run_test(vec!["a/b", "c/d/e"]); + assert_eq!(code, 0); + assert_eq!(stdout, "a\nc/d\n"); + assert_eq!(stderr, ""); + } - #[test] - fn test_missing_operand() { - let (code, stdout, stderr) = run_test(vec![]); - assert_eq!(code, 1); - assert_eq!(stdout, ""); - assert!(stderr.contains("missing operand")); - } + #[test] + fn test_zero_delimited() { + let (code, stdout, stderr) = run_test(vec!["-z", "a/b", "c/d/e"]); + assert_eq!(code, 0); + assert_eq!(stdout, "a\0c/d\0"); + assert_eq!(stderr, ""); + } + + #[test] + fn test_help() { + let (code, stdout, stderr) = run_test(vec!["--help"]); + assert_eq!(code, 0); + assert!(stdout.contains("Usage:")); + assert!(stdout.contains("Strip last component")); + assert_eq!(stderr, ""); + } + + #[test] + fn test_invalid_arg() { + let (code, stdout, stderr) = run_test(vec!["--invalid-flag"]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert!(stderr.contains("unexpected argument")); + } + + #[test] + fn test_missing_operand() { + let (code, stdout, stderr) = run_test(vec![]); + assert_eq!(code, 1); + assert_eq!(stdout, ""); + assert!(stderr.contains("missing operand")); + } } diff --git a/crates/vendor/uu-paste/src/paste.rs b/crates/vendor/uu-paste/src/paste.rs index cd40ad8bd..40fad949e 100644 --- a/crates/vendor/uu-paste/src/paste.rs +++ b/crates/vendor/uu-paste/src/paste.rs @@ -3,16 +3,21 @@ // For the full copyright and license information, please view the LICENSE // file that was distributed with this source code. +use std::{ + cell::RefCell, + ffi::OsString, + fs::File, + io::{BufRead, BufReader, Read, Write}, + iter::Cycle, + rc::Rc, + slice::Iter, +}; + use clap::{Arg, ArgAction, Command}; -use std::cell::RefCell; -use std::ffi::OsString; -use std::fs::File; -use std::io::{BufRead, BufReader, Read, Write}; -use std::iter::Cycle; -use std::rc::Rc; -use std::slice::Iter; -use uucore::error::{UResult, USimpleError, strip_errno}; -use uucore::i18n::charmap::mb_char_len; +use uucore::{ + error::{UResult, USimpleError, strip_errno}, + i18n::charmap::mb_char_len, +}; mod options { pub const DELIMITER: &str = "delimiters"; @@ -39,8 +44,16 @@ pub fn run(argv: Vec) -> i32 { let serial = matches.get_flag(options::SERIAL); let delimiters = matches.get_one::(options::DELIMITER).unwrap(); - let files = matches.get_many::(options::FILE).unwrap().cloned().collect(); - let line_ending = if matches.get_flag(options::ZERO_TERMINATED) { b'\0' } else { b'\n' }; + let files = matches + .get_many::(options::FILE) + .unwrap() + .cloned() + .collect(); + let line_ending = if matches.get_flag(options::ZERO_TERMINATED) { + b'\0' + } else { + b'\n' + }; match paste(files, serial, delimiters, line_ending) { Ok(()) => pi_uutils_ctx::exit_code(), @@ -58,13 +71,46 @@ pub fn uu_app() -> Command { .about("Merge lines of files") .override_usage(pi_uutils_ctx::format_usage("paste [OPTION]... [FILE]...")) .infer_long_args(true) - .arg(Arg::new(options::SERIAL).long(options::SERIAL).short('s').help("paste one file at a time instead of in parallel").action(ArgAction::SetTrue)) - .arg(Arg::new(options::DELIMITER).long(options::DELIMITER).short('d').help("reuse characters from LIST instead of TABs").value_name("LIST").default_value("\t").hide_default_value(true).value_parser(clap::value_parser!(OsString))) - .arg(Arg::new(options::FILE).value_name("FILE").action(ArgAction::Append).default_value("-").value_hint(clap::ValueHint::FilePath).value_parser(clap::value_parser!(OsString))) - .arg(Arg::new(options::ZERO_TERMINATED).long(options::ZERO_TERMINATED).short('z').help("line delimiter is NUL, not newline").action(ArgAction::SetTrue)) + .arg( + Arg::new(options::SERIAL) + .long(options::SERIAL) + .short('s') + .help("paste one file at a time instead of in parallel") + .action(ArgAction::SetTrue), + ) + .arg( + Arg::new(options::DELIMITER) + .long(options::DELIMITER) + .short('d') + .help("reuse characters from LIST instead of TABs") + .value_name("LIST") + .default_value("\t") + .hide_default_value(true) + .value_parser(clap::value_parser!(OsString)), + ) + .arg( + Arg::new(options::FILE) + .value_name("FILE") + .action(ArgAction::Append) + .default_value("-") + .value_hint(clap::ValueHint::FilePath) + .value_parser(clap::value_parser!(OsString)), + ) + .arg( + Arg::new(options::ZERO_TERMINATED) + .long(options::ZERO_TERMINATED) + .short('z') + .help("line delimiter is NUL, not newline") + .action(ArgAction::SetTrue), + ) } -fn paste(filenames: Vec, serial: bool, delimiters: &OsString, line_ending: u8) -> UResult<()> { +fn paste( + filenames: Vec, + serial: bool, + delimiters: &OsString, + line_ending: u8, +) -> UResult<()> { let delimiters = parse_delimiters(delimiters)?; // pi-uutils: all `-` operands share the scoped stdin and consume it in order. let stdin = Rc::new(RefCell::new(BufReader::new(pi_uutils_ctx::stdin()))); @@ -94,7 +140,9 @@ fn paste(filenames: Vec, serial: bool, delimiters: &OsString, line_end for source in &mut sources { output.clear(); loop { - if source.read_until(line_ending, &mut output)? == 0 { break; } + if source.read_until(line_ending, &mut output)? == 0 { + break; + } remove_trailing_line_ending(line_ending, &mut output); delimiter_state.write_delimiter(&mut output); } @@ -118,7 +166,9 @@ fn paste(filenames: Vec, serial: bool, delimiters: &OsString, line_end } delimiter_state.write_delimiter(&mut output); } - if eof_count == source_count { break; } + if eof_count == source_count { + break; + } delimiter_state.remove_trailing_delimiter(&mut output); stdout.write_all(&output)?; stdout.write_all(&[line_ending])?; @@ -128,18 +178,26 @@ fn paste(filenames: Vec, serial: bool, delimiters: &OsString, line_end Ok(()) } -fn write_single_input_source(writer: &mut impl Write, mut source: InputSource, line_ending: u8) -> UResult<()> { +fn write_single_input_source( + writer: &mut impl Write, + mut source: InputSource, + line_ending: u8, +) -> UResult<()> { let mut buffer = [0_u8; 8192]; let mut has_data = false; let mut last_byte = line_ending; loop { let count = source.read(&mut buffer)?; - if count == 0 { break; } + if count == 0 { + break; + } has_data = true; last_byte = buffer[count - 1]; writer.write_all(&buffer[..count])?; } - if has_data && last_byte != line_ending { writer.write_all(&[line_ending])?; } + if has_data && last_byte != line_ending { + writer.write_all(&[line_ending])?; + } Ok(()) } @@ -151,14 +209,29 @@ fn parse_delimiters(delimiters: &OsString) -> UResult]>> { if bytes[i] == b'\\' { i += 1; if i >= bytes.len() { - return Err(USimpleError::new(1, format!("delimiter list ends with an unescaped backslash: {}", delimiters.to_string_lossy()))); + return Err(USimpleError::new( + 1, + format!( + "delimiter list ends with an unescaped backslash: {}", + delimiters.to_string_lossy() + ), + )); } match bytes[i] { - b'0' => result.push(Box::new([])), b'\\' => result.push(Box::new([b'\\'])), - b'n' => result.push(Box::new([b'\n'])), b't' => result.push(Box::new([b'\t'])), - b'b' => result.push(Box::new([b'\x08'])), b'f' => result.push(Box::new([b'\x0c'])), - b'r' => result.push(Box::new([b'\r'])), b'v' => result.push(Box::new([b'\x0b'])), - _ => { let len = mb_char_len(&bytes[i..]).min(bytes.len() - i); result.push(Box::from(&bytes[i..i + len])); i += len; continue; } + b'0' => result.push(Box::new([])), + b'\\' => result.push(Box::new(*b"\\")), + b'n' => result.push(Box::new(*b"\n")), + b't' => result.push(Box::new(*b"\t")), + b'b' => result.push(Box::new(*b"\x08")), + b'f' => result.push(Box::new(*b"\x0c")), + b'r' => result.push(Box::new(*b"\r")), + b'v' => result.push(Box::new(*b"\x0b")), + _ => { + let len = mb_char_len(&bytes[i..]).min(bytes.len() - i); + result.push(Box::from(&bytes[i..i + len])); + i += len; + continue; + }, } i += 1; } else { @@ -171,13 +244,19 @@ fn parse_delimiters(delimiters: &OsString) -> UResult]>> { } fn remove_trailing_line_ending(line_ending: u8, output: &mut Vec) { - if output.last() == Some(&line_ending) { output.pop(); } + if output.last() == Some(&line_ending) { + output.pop(); + } } enum DelimiterState<'a> { NoDelimiters, OneDelimiter(&'a [u8]), - MultipleDelimiters { current: &'a [u8], delimiters: &'a [Box<[u8]>], iterator: Cycle>> }, + MultipleDelimiters { + current: &'a [u8], + delimiters: &'a [Box<[u8]>], + iterator: Cycle>>, + }, } impl<'a> DelimiterState<'a> { @@ -186,21 +265,40 @@ impl<'a> DelimiterState<'a> { [] => Self::NoDelimiters, [only] if only.is_empty() => Self::NoDelimiters, [only] => Self::OneDelimiter(only), - [first, ..] => Self::MultipleDelimiters { current: first, delimiters, iterator: delimiters.iter().cycle() }, + [first, ..] => Self::MultipleDelimiters { + current: first, + delimiters, + iterator: delimiters.iter().cycle(), + }, } } + fn reset_to_first_delimiter(&mut self) { - if let Self::MultipleDelimiters { delimiters, iterator, .. } = self { *iterator = delimiters.iter().cycle(); } + if let Self::MultipleDelimiters { delimiters, iterator, .. } = self { + *iterator = delimiters.iter().cycle(); + } } + fn remove_trailing_delimiter(&self, output: &mut Vec) { - let len = match self { Self::NoDelimiters => return, Self::OneDelimiter(d) => d.len(), Self::MultipleDelimiters { current, .. } => current.len() }; - if len > 0 { output.truncate(output.len().saturating_sub(len)); } + let len = match self { + Self::NoDelimiters => return, + Self::OneDelimiter(d) => d.len(), + Self::MultipleDelimiters { current, .. } => current.len(), + }; + if len > 0 { + output.truncate(output.len().saturating_sub(len)); + } } + fn write_delimiter(&mut self, output: &mut Vec) { match self { Self::NoDelimiters => {}, Self::OneDelimiter(d) => output.extend_from_slice(d), - Self::MultipleDelimiters { current, iterator, .. } => { let d = iterator.next().unwrap(); output.extend_from_slice(d); *current = d; }, + Self::MultipleDelimiters { current, iterator, .. } => { + let d = iterator.next().unwrap(); + output.extend_from_slice(d); + *current = d; + }, } } } @@ -214,13 +312,24 @@ impl InputSource { fn read(&mut self, buf: &mut [u8]) -> UResult { Ok(match self { Self::File(reader) => reader.read(buf)?, - Self::StandardInput(stdin) => stdin.try_borrow_mut().map_err(|err| USimpleError::new(1, format!("standard input is already borrowed: {err}")))?.read(buf)?, + Self::StandardInput(stdin) => stdin + .try_borrow_mut() + .map_err(|err| { + USimpleError::new(1, format!("standard input is already borrowed: {err}")) + })? + .read(buf)?, }) } + fn read_until(&mut self, byte: u8, buf: &mut Vec) -> UResult { Ok(match self { Self::File(reader) => reader.read_until(byte, buf)?, - Self::StandardInput(stdin) => stdin.try_borrow_mut().map_err(|err| USimpleError::new(1, format!("standard input is already borrowed: {err}")))?.read_until(byte, buf)?, + Self::StandardInput(stdin) => stdin + .try_borrow_mut() + .map_err(|err| { + USimpleError::new(1, format!("standard input is already borrowed: {err}")) + })? + .read_until(byte, buf)?, }) } } diff --git a/crates/vendor/uu-sed/Cargo.toml b/crates/vendor/uu-sed/Cargo.toml new file mode 100644 index 000000000..c12810c2f --- /dev/null +++ b/crates/vendor/uu-sed/Cargo.toml @@ -0,0 +1,26 @@ +# Vendored from uutils/sed commit b37e23fa987888572e02e4e9b6906b3ede749bc6, +# patched for in-process embedding through pi-uutils-ctx. +[package] +name = "uu_sed" +version = "0.1.1" +edition = "2024" +license = "MIT" +description = "sed ~ (uutils) stream editor for filtering and transforming text (vendored + patched for in-process embedding)" + +[lib] +path = "src/lib.rs" + +[dependencies] +clap = { version = "4.5", features = ["wrap_help", "cargo"] } +fancy-regex = "0.18" +memchr = "2.7" +regex = "1.11" +tempfile = "3" +uucore = { version = "0.9.0", features = ["libc"] } +pi-uutils-ctx = { path = "../../pi-uutils-ctx" } + +[target.'cfg(unix)'.dependencies] +memmap2 = "0.9" + +[dev-dependencies] +parking_lot = "0.12" diff --git a/crates/vendor/uu-sed/LICENSE b/crates/vendor/uu-sed/LICENSE new file mode 100644 index 000000000..c66459ce3 --- /dev/null +++ b/crates/vendor/uu-sed/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2025 Diomidis Spinellis + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/crates/vendor/uu-sed/src/lib.rs b/crates/vendor/uu-sed/src/lib.rs new file mode 100644 index 000000000..f26a81ae9 --- /dev/null +++ b/crates/vendor/uu-sed/src/lib.rs @@ -0,0 +1,246 @@ +// This file is part of the uutils sed package. +// +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +//! Vendored, patched `sed` from uutils/sed, wired to run in-process as a +//! shell builtin via [`pi_uutils_ctx`]. +//! +//! Upstream: +//! Pinned commit: `b37e23fa987888572e02e4e9b6906b3ede749bc6` (default-branch +//! HEAD, 2026-07-10, version 0.1.1). +//! +//! Patches applied for in-process embedding: +//! - all stdio goes through the `pi_uutils_ctx` streams, +//! - every path operand resolves against the shell working directory via +//! `pi_uutils_ctx::resolve`, +//! - no `std::process::exit`: `q`/`Q` exit codes flow through +//! `pi_uutils_ctx::set_exit_code`, clap errors are rendered manually, +//! - the `s///e` shell escape spawns with the shell's cwd and piped stdio, +//! - output is never assumed to be a terminal (no `-l` width auto-detect, no +//! tty-triggered unbuffered mode). + +pub mod sed; + +use std::{ffi::OsString, io::Write}; + +/// In-process builtin entry point. The host installs a [`pi_uutils_ctx`] +/// scope (stdio + working directory + environment) on a dedicated blocking +/// thread, then calls this. +/// +/// Unlike upstream's `main` (which `std::process::exit`s on the result of +/// `uumain`), this returns the exit code so it is safe to run inside the +/// long-lived host shell process. +pub fn run(argv: Vec) -> i32 { + // A reused blocking thread may still hold `w`/`s///w` writers registered + // by a previous invocation that failed before flushing; drop them. + sed::named_writer::reset(); + + let matches = match sed::uu_app().try_get_matches_from(sed::normalize_args(argv)) { + Ok(m) => m, + Err(e) => { + let rendered = e.to_string(); + if e.use_stderr() { + let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); + return 1; + } + let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); + return 0; + }, + }; + + // Upstream prints help and exits 1 when invoked without any argument. + if !matches.args_present() { + let _ = write!(pi_uutils_ctx::stdout(), "{}", sed::uu_app().render_help()); + return 1; + } + + match sed::sed_main(&matches) { + Ok(()) => pi_uutils_ctx::exit_code(), + Err(e) => { + let code = e.code(); + let _ = writeln!(pi_uutils_ctx::stderr(), "sed: {e}"); + if code == 0 { 1 } else { code } + }, + } +} + +#[cfg(test)] +mod tests { + use std::{ + collections::HashMap, + ffi::OsString, + io::{self, Write}, + path::PathBuf, + sync::{Arc, atomic::AtomicBool}, + }; + + use parking_lot::Mutex; + + use super::run; + + /// `Send` writer capturing everything a run writes to a scope stream. + #[derive(Clone, Default)] + struct SharedBuf(Arc>>); + + impl Write for SharedBuf { + fn write(&mut self, buf: &[u8]) -> io::Result { + self.0.lock().extend_from_slice(buf); + Ok(buf.len()) + } + + fn flush(&mut self) -> io::Result<()> { + Ok(()) + } + } + + impl SharedBuf { + fn take(&self) -> String { + String::from_utf8(self.0.lock().clone()).expect("utf8 stream") + } + } + + /// Drive `run()` under a pi-uutils-ctx scope with `cwd` as the shell + /// working directory; returns (exit code, stdout, stderr). + fn run_sed_in(cwd: PathBuf, stdin: &[u8], args: &[&str]) -> (i32, String, String) { + let stdout = SharedBuf::default(); + let stderr = SharedBuf::default(); + let argv: Vec = std::iter::once("sed") + .chain(args.iter().copied()) + .map(OsString::from) + .collect(); + let code = pi_uutils_ctx::scope( + pi_uutils_ctx::ScopeIo { + stdin: Box::new(io::Cursor::new(stdin.to_vec())), + stdin_fd: None, + stdin_is_search_input: false, + stdout: Box::new(stdout.clone()), + stderr: Box::new(stderr.clone()), + cwd, + env: HashMap::new(), + cancel: Arc::new(AtomicBool::new(false)), + }, + || run(argv), + ); + (code, stdout.take(), stderr.take()) + } + + fn run_sed(stdin: &[u8], args: &[&str]) -> (i32, String, String) { + run_sed_in(PathBuf::from("."), stdin, args) + } + + #[test] + fn substitutes_basic_from_stdin() { + let (code, out, err) = run_sed(b"hello\n", &["s/hello/world/"]); + assert_eq!(code, 0); + assert_eq!(out, "world\n"); + assert!(err.is_empty(), "unexpected stderr: {err}"); + } + + #[test] + fn quiet_prints_address_range() { + let (code, out, _) = run_sed(b"a\nb\nc\nd\n", &["-n", "2,3p"]); + assert_eq!(code, 0); + assert_eq!(out, "b\nc\n"); + } + + #[test] + fn substitution_global_flag() { + let (code, out, _) = run_sed(b"aaa\n", &["s/a/b/g"]); + assert_eq!(code, 0); + assert_eq!(out, "bbb\n"); + } + + #[test] + fn substitution_numbered_occurrence() { + let (code, out, _) = run_sed(b"aaa\n", &["s/a/b/2"]); + assert_eq!(code, 0); + assert_eq!(out, "aba\n"); + } + + #[test] + fn ere_capture_groups_swap() { + let (code, out, _) = run_sed(b"john smith\n", &["-E", r"s/([a-z]+) ([a-z]+)/\2 \1/"]); + assert_eq!(code, 0); + assert_eq!(out, "smith john\n"); + } + + #[test] + fn bre_backreference_in_pattern() { + let (code, out, _) = run_sed(b"abab\nabcd\n", &["-n", r"/\(ab\)\1/p"]); + assert_eq!(code, 0); + assert_eq!(out, "abab\n"); + } + + #[test] + fn hold_space_tac() { + let (code, out, _) = run_sed(b"1\n2\n3\n", &["1!G;h;$!d"]); + assert_eq!(code, 0); + assert_eq!(out, "3\n2\n1\n"); + } + + #[test] + fn transliterates() { + let (code, out, _) = run_sed(b"abcabc\n", &["y/abc/xyz/"]); + assert_eq!(code, 0); + assert_eq!(out, "xyzxyz\n"); + } + + #[test] + fn multiple_expressions_compose_in_order() { + let (code, out, _) = run_sed(b"a\n", &["-e", "s/a/b/", "-e", "s/b/c/"]); + assert_eq!(code, 0); + assert_eq!(out, "c\n"); + } + + #[test] + fn q_with_operand_propagates_exit_code() { + let (code, out, _) = run_sed(b"one\ntwo\nthree\n", &["2q42"]); + assert_eq!(code, 42); + assert_eq!(out, "one\ntwo\n"); + } + + #[test] + fn q_stops_before_later_lines() { + let (code, out, _) = run_sed(b"one\ntwo\n", &["1q"]); + assert_eq!(code, 0); + assert_eq!(out, "one\n"); + } + + #[test] + fn in_place_edits_relative_path_against_scope_cwd() { + let dir = tempfile::tempdir().unwrap(); + std::fs::write(dir.path().join("file.txt"), "x marks\n").unwrap(); + let (code, out, err) = + run_sed_in(dir.path().to_path_buf(), b"", &["-i", "s/x/y/", "file.txt"]); + assert_eq!(code, 0, "stderr: {err}"); + assert!(out.is_empty(), "in-place edit must not print: {out}"); + assert_eq!(std::fs::read_to_string(dir.path().join("file.txt")).unwrap(), "y marks\n"); + } + + #[test] + fn in_place_backup_suffix_keeps_original() { + let dir = tempfile::tempdir().unwrap(); + std::fs::write(dir.path().join("file.txt"), "x marks\n").unwrap(); + let (code, _, err) = + run_sed_in(dir.path().to_path_buf(), b"", &["-i.bak", "s/x/y/", "file.txt"]); + assert_eq!(code, 0, "stderr: {err}"); + assert_eq!(std::fs::read_to_string(dir.path().join("file.txt")).unwrap(), "y marks\n"); + assert_eq!(std::fs::read_to_string(dir.path().join("file.txt.bak")).unwrap(), "x marks\n"); + } + + #[test] + fn null_data_mode_substitutes_per_record() { + let (code, out, _) = run_sed(b"a\0b\0", &["-z", "s/a/X/"]); + assert_eq!(code, 0); + assert_eq!(out, "X\0b\0"); + } + + #[test] + fn unknown_option_diagnoses_on_stderr() { + let (code, out, err) = run_sed(b"", &["--definitely-not-an-option", "p"]); + assert_ne!(code, 0); + assert!(out.is_empty(), "usage errors must not write stdout: {out}"); + assert!(!err.is_empty(), "expected a diagnostic on stderr"); + } +} diff --git a/crates/vendor/uu-sed/src/sed/command.rs b/crates/vendor/uu-sed/src/sed/command.rs new file mode 100644 index 000000000..59a62fcfc --- /dev/null +++ b/crates/vendor/uu-sed/src/sed/command.rs @@ -0,0 +1,550 @@ +// Definitions for the compiled code data structures +// +// SPDX-License-Identifier: MIT +// Copyright (c) 2025 Diomidis Spinellis +// +// This file is part of the uutils sed package. +// It is licensed under the MIT License. +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +use std::path::PathBuf; // For file descriptors and equivalent +use std::{cell::RefCell, collections::HashMap, rc::Rc}; + +use uucore::error::UResult; + +use crate::sed::{ + error_handling::{ScriptLocation, runtime_error}, + fast_regex::{Captures, Match, Regex}, + named_writer::NamedWriter, + script_char_provider::ScriptCharProvider, + script_line_provider::ScriptLineProvider, +}; + +#[derive(Debug, Default, Clone)] +/// Compilation and processing options provided mostly through the +/// command-line interface +pub struct ProcessingContext { + // Command-line flags with corresponding names + pub all_output_files: bool, + pub debug: bool, + pub regex_extended: bool, + pub follow_symlinks: bool, + pub in_place: bool, + pub in_place_suffix: Option, + pub length: usize, + pub quiet: bool, + pub posix: bool, + pub separate: bool, + pub sandbox: bool, + pub unbuffered: bool, + pub null_data: bool, + + // Other context + /// Currently processed input file name (not script) in quoted form + pub input_name: String, + /// Current input line number + pub line_number: usize, + /// True if this is the last address of a range + pub last_address: bool, + /// True if the line read is the last line + pub last_line: bool, + /// True if the file is the last file of the ones specified + pub last_file: bool, + /// Stop processing further input. + pub stop_processing: bool, + /// Previously compiled RE, saved for reuse when specifying an empty RE + pub saved_regex: Option, + /// Modification of input processing action + // This is required to avoid doubly borrowing the reader in the 'N' + // command. + pub input_action: Option, + /// Hold space + pub hold: StringSpace, + /// Nesting of { } at compile time + pub parsed_block_nesting: usize, + /// Command associated with each label + pub label_to_command_map: HashMap>>, + /// Commands with a (latchable and resetable) address range + pub range_commands: Vec>>, + /// True if a substitution was made as specified in the t command + pub substitution_made: bool, + /// Elements to append at the end of each command processing cycle + pub append_elements: Vec, +} + +#[derive(Clone, Debug)] +/// Elements that shall be appended at the end of each command processing cycle +pub enum AppendElement { + Text(Rc), // The specified text string + Path(PathBuf), // The contents of the specified file path +} + +#[derive(Clone, Debug, Default, PartialEq)] +/// A space mirroring IOChunk, but only with a String +pub struct StringSpace { + pub content: String, // Line content without newline + pub has_newline: bool, // True if \n-terminated +} + +#[derive(Debug)] +/// Types of address specifications that precede commands +pub enum Address { + Re(Option), // Line that matches (optional) regex + Line(usize), // Specific line + RelLine(usize), // Relative line + Last, // Last line + StepMatch(usize), // Lines matching specified step from first + StepEnd(usize), // Range ending at specified step from first +} + +#[derive(Debug)] +/// A single part of an RE replacement +pub enum ReplacementPart { + Literal(String), // Normal text + WholeMatch, // & + Group(u32), // \1 to \9 +} + +// The maximum value allowed in regex quantifier +pub const RE_DUP_MAX: usize = 32767; + +/// Regex modes (BRE or ERE) +#[derive(Copy, Clone, Debug)] +pub enum RegexMode { + Basic, + Extended, +} + +#[derive(Debug)] +/// All specified replacements for an RE +pub struct ReplacementTemplate { + pub parts: Vec, + pub max_group_number: usize, // Highest used group number (e.g. 8 for \8) +} + +impl Default for ReplacementTemplate { + /// Create an empty template. + fn default() -> Self { + ReplacementTemplate::new(Vec::new()) + } +} + +impl ReplacementTemplate { + /// Construct from the parts + pub fn new(parts: Vec) -> Self { + let max_group_number = parts + .iter() + .filter_map(|part| match part { + ReplacementPart::Group(n) => Some(*n), + _ => None, + }) + .max() + .unwrap_or(0); + + Self { parts, max_group_number: max_group_number.try_into().unwrap() } + } + + /// Apply the template to the given RE captures. + /// Example: + /// let result = regex.replace_all(input, |caps: &Captures| { + /// template.apply_captures(&command, caps) }); + /// Returns an error if a backreference in the template was not matched by + /// the RE. + pub fn apply_captures(&self, command: &Command, caps: &Captures) -> UResult { + let mut result = String::new(); + + // Invalid group numbers may end here through (unkown at compile time) + // reused REs. + if self.max_group_number > caps.len() - 1 { + return runtime_error( + &command.location, + format!("invalid reference \\{} on command's RHS", self.max_group_number), + ); + } + + for part in &self.parts { + match part { + ReplacementPart::Literal(s) => result.push_str(s), + + ReplacementPart::WholeMatch => { + result.push_str(caps.get(0)?.map(|m| m.as_str()).unwrap_or_default()); + }, + + ReplacementPart::Group(n) => { + let i: usize = (*n).try_into().unwrap(); + result.push_str(caps.get(i)?.map(|m| m.as_str()).unwrap_or_default()); + }, + } + } + + Ok(result) + } + + /// Apply the template to the given RE single match. + pub fn apply_match(&self, m: &Match) -> String { + let mut result = String::new(); + + for part in &self.parts { + match part { + ReplacementPart::Literal(s) => result.push_str(s), + + ReplacementPart::WholeMatch => result.push_str(m.as_str()), + + ReplacementPart::Group(_) => { + panic!("unexpected Regex group replacement") + }, + } + } + result + } +} + +#[derive(Debug, Default)] +/// Substitution command +pub struct Substitution { + pub regex: Option, // Regular expression + pub replacement: ReplacementTemplate, // Specified broken-down replacement + pub occurrence: usize, // Which occurrence to substitute + pub print_flag: bool, // True if 'p' flag + pub ignore_case: bool, // True if 'I' flag + pub execute: bool, // True if 'e' flag (GNU extension) + pub multiline: bool, // True if 'm' or 'M' flag (GNU extension) + pub write_file: Option>>, // Writer to file if 'w' flag is used +} + +/// The block of the first and most common Unicode characters: +/// ASCII, Latin Extended, Greek, Curillic, Coptic, Arabic, etc. +/// It comprises all UCS-2 characters. We use a fast lookup array for these. +const COMMON_UNICODE: usize = 2048; + +#[derive(Debug)] +/// Transliteration command (y) +pub struct Transliteration { + fast: [char; COMMON_UNICODE], + slow: HashMap, +} + +impl Default for Transliteration { + /// Create a new Transliteration with identity mapping for the fast-path. + fn default() -> Self { + let mut fast = ['\0'; COMMON_UNICODE]; + for (i, slot) in fast.iter_mut().enumerate() { + *slot = char::from_u32(i as u32).unwrap_or('\0'); + } + Self { fast, slow: HashMap::new() } + } +} + +impl Transliteration { + /// Create through character mappings from `source` to `target`. + pub fn from_strings(source: &str, target: &str) -> Self { + let mut result = Self::default(); + for (from, to) in source.chars().zip(target.chars()) { + result.insert(from, to); + } + result + } + + /// Set a transliteration mapping from one character to another. + fn insert(&mut self, from: char, to: char) { + let cp = from as usize; + if cp < COMMON_UNICODE { + self.fast[cp] = to; + } else { + self.slow.insert(from, to); + } + } + + /// Look up a character transliteration. + pub fn lookup(&self, ch: char) -> char { + let cp = ch as usize; + if cp < COMMON_UNICODE { + self.fast[cp] + } else { + self.slow.get(&ch).copied().unwrap_or(ch) + } + } +} + +#[derive(Debug)] +/// An internally compiled command. +pub struct Command { + pub code: char, // Command code + pub addr1: Option
, // Start address + pub addr2: Option
, // End address + pub non_select: bool, // True if '!' + pub start_line: Option, // Start line number (or None if unlatched) + pub data: CommandData, // Command-specific data + pub next: Option>>, // Pointer to next command + pub location: ScriptLocation, // Command's definition location +} + +impl Default for Command { + fn default() -> Self { + Command { + code: '_', + addr1: None, + addr2: None, + non_select: false, + start_line: None, + data: CommandData::None, + next: None, + location: ScriptLocation::default(), + } + } +} + +impl Command { + /// Construct with position information from the given providers. + pub fn at_position(lines: &ScriptLineProvider, line: &ScriptCharProvider) -> Self { + Command { location: ScriptLocation::at_position(lines, line), ..Default::default() } + } +} + +#[derive(Debug)] +/// Command-specific data +/// After parsing, t, b Label elements are converted into BranchTarget ones. +pub enum CommandData { + None, + BranchTarget(Option>>), // Commands for 'b', 't', '{' + Label(Option), // Label name for 'b', 't', ':' + Path(PathBuf), // File path for 'r' + NamedWriter(Rc>), // File output for 'w' + Number(usize), // Number for 'l', 'q', 'Q' (GNU) + Substitution(Box), // Substitute command 's' + Text(Rc), // Text for 'a', 'c', 'i' + Transliteration(Box), // Transliteration command 'y' +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +/// Flag for space modifications +pub enum SpaceFlag { + Append, // Append to contents + Replace, // Replace contents +} + +#[derive(Debug, Clone)] +/// Action to execute after reading a new input line +pub struct InputAction { + /// Next command to execute (rather than commands from start) + pub next_command: Option>>, + /// Data to prepend to the read contents + pub prepend: String, +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::sed::fast_io::IOChunk; + + // Return the captures for the RE applied to the specified string + fn caps_for<'a>(re: &str, chunk: &'a mut IOChunk) -> Captures<'a> { + Regex::new(re) + .unwrap() + .captures(chunk) + .unwrap() + .expect("captures") + } + + #[test] + // s/foo// + fn test_empty_template() { + let template = ReplacementTemplate::default(); + let input = &mut IOChunk::new_from_str("foo"); + let caps = caps_for("foo", input); + let cmd = Command::default(); + + let result = template.apply_captures(&cmd, &caps).unwrap(); + assert_eq!(result, ""); + } + + #[test] + // s/abc/hello/ + fn test_literal_only() { + let template = ReplacementTemplate::new(vec![ReplacementPart::Literal("hello".into())]); + let input = &mut IOChunk::new_from_str("abc"); + let caps = caps_for("abc", input); + let cmd = Command::default(); + + let result = template.apply_captures(&cmd, &caps).unwrap(); + assert_eq!(result, "hello"); + } + + #[test] + // s/foo\d+/got: &/ + fn test_whole_match() { + let template = ReplacementTemplate::new(vec![ + ReplacementPart::Literal("got: ".into()), + ReplacementPart::WholeMatch, + ]); + let input = &mut IOChunk::new_from_str("foo42"); + let caps = caps_for(r"foo\d+", input); + let cmd = Command::default(); + + let result = template.apply_captures(&cmd, &caps).unwrap(); + assert_eq!(result, "got: foo42"); + } + + #[test] + // s/foo(\d+)/number: \1/ + fn test_backreference() { + let template = ReplacementTemplate::new(vec![ + ReplacementPart::Literal("number: ".into()), + ReplacementPart::Group(1), + ]); + let input = &mut IOChunk::new_from_str("foo42"); + let caps = caps_for(r"foo(\d+)", input); + let cmd = Command::default(); + + let result = template.apply_captures(&cmd, &caps).unwrap(); + assert_eq!(result, "number: 42"); + } + + #[test] + // s/(\w+):(\d+)/key: \1, value: \2/ + fn test_multiple_parts() { + let template = ReplacementTemplate::new(vec![ + ReplacementPart::Literal("key: ".into()), + ReplacementPart::Group(1), + ReplacementPart::Literal(", value: ".into()), + ReplacementPart::Group(2), + ]); + let input = &mut IOChunk::new_from_str("x:123"); + let caps = caps_for(r"(\w+):(\d+)", input); + let cmd = Command::default(); + + let result = template.apply_captures(&cmd, &caps).unwrap(); + assert_eq!(result, "key: x, value: 123"); + } + + #[test] + // s/(\w+):(\d+)/key: \1, value: \3/ + fn test_invalid_group() { + let template = ReplacementTemplate::new(vec![ + ReplacementPart::Literal("key: ".into()), + ReplacementPart::Group(1), + ReplacementPart::Literal(", value: ".into()), + ReplacementPart::Group(3), + ]); + let input = &mut IOChunk::new_from_str("x:123"); + let caps = caps_for(r"(\w+):(\d+)", input); + let cmd = Command::default(); + + let result = template.apply_captures(&cmd, &caps); + assert!(result.is_err()); + + let msg = result.unwrap_err().to_string(); + assert!(msg.contains("invalid reference \\3")); + } + + // max_group_number + #[test] + fn test_max_group_number_with_groups() { + let template = ReplacementTemplate::new(vec![ + ReplacementPart::Literal("a".into()), + ReplacementPart::Group(2), + ReplacementPart::WholeMatch, + ReplacementPart::Group(5), + ReplacementPart::Literal("z".into()), + ]); + assert_eq!(template.max_group_number, 5); + } + + #[test] + fn test_max_group_number_without_groups() { + let template = ReplacementTemplate::new(vec![ + ReplacementPart::Literal("no".into()), + ReplacementPart::WholeMatch, + ReplacementPart::Literal("groups".into()), + ]); + assert_eq!(template.max_group_number, 0); + } + + // Transliteration + // Creation and internal functions + #[test] + fn test_identity_lookup_fast_path() { + let t = Transliteration::default(); + assert_eq!(t.lookup('A'), 'A'); + assert_eq!(t.lookup('z'), 'z'); + assert_eq!(t.lookup('\u{07FF}'), '\u{07FF}'); // highest 2-byte UTF-8 char + } + + #[test] + fn test_identity_lookup_slow_path() { + let t = Transliteration::default(); + assert_eq!(t.lookup('\u{0800}'), '\u{0800}'); // just outside fast path + assert_eq!(t.lookup('\u{1F600}'), '\u{1F600}'); // 😀 + } + + #[test] + fn test_insert_and_lookup_fast_path() { + let mut t = Transliteration::default(); + t.insert('a', 'α'); + t.insert('b', 'β'); + assert_eq!(t.lookup('a'), 'α'); + assert_eq!(t.lookup('b'), 'β'); + assert_eq!(t.lookup('c'), 'c'); // unchanged + } + + #[test] + fn test_insert_and_lookup_slow_path() { + let mut t = Transliteration::default(); + t.insert('🦀', 'c'); // U+1F980 Crab emoji -> 'c' + assert_eq!(t.lookup('🦀'), 'c'); + assert_eq!(t.lookup('🦁'), '🦁'); // unchanged + } + + #[test] + fn test_overwrite_mapping() { + let mut t = Transliteration::default(); + t.insert('x', '1'); + assert_eq!(t.lookup('x'), '1'); + t.insert('x', '2'); + assert_eq!(t.lookup('x'), '2'); + } + + #[test] + fn test_all_fast_path_mapped_to_space() { + let mut t = Transliteration::default(); + for cp in 0..COMMON_UNICODE { + if let Some(ch) = char::from_u32(cp as u32) { + t.insert(ch, ' '); + } + } + assert_eq!(t.lookup('A'), ' '); + assert_eq!(t.lookup('\u{07FF}'), ' '); + } + + // from_strings + #[test] + fn test_basic_transliteration() { + let t = Transliteration::from_strings("abcδ", "1234"); + + assert_eq!(t.lookup('a'), '1'); + assert_eq!(t.lookup('b'), '2'); + assert_eq!(t.lookup('c'), '3'); + assert_eq!(t.lookup('δ'), '4'); + assert_eq!(t.lookup('e'), 'e'); // not mapped, fallback + } + + #[test] + fn test_unicode_slow_path() { + let source = "é漢🦀"; + let target = "e文c"; + let t = Transliteration::from_strings(source, target); + + assert_eq!(t.lookup('é'), 'e'); + assert_eq!(t.lookup('漢'), '文'); + assert_eq!(t.lookup('🦀'), 'c'); + assert_eq!(t.lookup('x'), 'x'); // fast fallback + assert_eq!(t.lookup('文'), '文'); // slow fallback + } + + #[test] + fn test_overwrite_fast_path() { + let t = Transliteration::from_strings("aa", "12"); + assert_eq!(t.lookup('a'), '2'); // last mapping wins + } +} diff --git a/crates/vendor/uu-sed/src/sed/compiler.rs b/crates/vendor/uu-sed/src/sed/compiler.rs new file mode 100644 index 000000000..6f35f21e7 --- /dev/null +++ b/crates/vendor/uu-sed/src/sed/compiler.rs @@ -0,0 +1,3083 @@ +// Compile the scripts into the internal representation of commands +// +// SPDX-License-Identifier: MIT +// Copyright (c) 2025 Diomidis Spinellis +// +// This file is part of the uutils sed package. +// It is licensed under the MIT License. +// For the full copyright and license information, please view the LICENSE +// file that was distributed with this source code. + +use std::{cell::RefCell, mem, path::PathBuf, rc::Rc}; + +use uucore::error::{UResult, USimpleError}; + +use crate::sed::{ + command::{ + Address, Command, CommandData, ProcessingContext, RegexMode, ReplacementPart, + ReplacementTemplate, Substitution, Transliteration, + }, + delimited_parser::{parse_char_escape, parse_regex, parse_transliteration}, + error_handling::{ScriptLocation, compilation_error, semantic_error}, + fast_regex::Regex, + named_writer::NamedWriter, + script_char_provider::ScriptCharProvider, + script_line_provider::{ScriptLineProvider, ScriptValue}, +}; + +const DEFAULT_OUTPUT_WIDTH: usize = 60; + +const ERR_ADDRESS_0_USAGE: &str = + "address 0 can only be used with ~step, a second regular expression, or a read command"; +const ERR_SANDBOX: &str = "command not allowed with --sandbox"; + +const ERR_UNKNOWN_OPTION_TO_S: &str = "unknown option to 's'"; + +// Handling required after processing a command +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum CommandHandling { + GetNext, // Get next command and process that: ! + Return, // Return from the sequence parser: } + Continue, // Continue sequence parsing: all other commands +} + +/// The type of functions that compile individual commands +type CommandHandler = fn( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Command, + context: &mut ProcessingContext, +) -> UResult; + +// Command specification +#[derive(Debug, Clone, Copy)] +struct CommandSpec { + n_addr: usize, // Number of supported addresses + handler: CommandHandler, // Argument-specific command compilation handler +} + +/// Compile the scripts into an executable data structure. +pub fn compile( + scripts: Vec, + context: &mut ProcessingContext, +) -> UResult>>> { + let mut make_providers = ScriptLineProvider::new(scripts); + + let mut empty_line = ScriptCharProvider::new(""); + let result = compile_sequence(&mut make_providers, &mut empty_line, context)?; + + // Comment-out the following to show the compiled script. + #[cfg(any())] + dbg!(&result); + + // Link branch commands to the target label commands. + populate_label_map(result.clone(), context)?; + populate_range_commands(result.clone(), context); + resolve_branch_targets(result.clone(), context)?; + + // Link the ends of command blocks to their following commands. + // This converts the tree into a graph, so it must be the last + // conversion that traverses the structure as a tree. + if context.parsed_block_nesting > 0 { + return Err(USimpleError::new(1, "unmatched `{'")); + } + patch_block_endings(result.clone()); + + Ok(result) +} + +/// For every Command in the top-level `head` chain, look for +/// `CommandData::BranchTarget(Some(sub_head))` '{' commands. +/// Recursively patch the sub-chain, then splice its tail back to the +/// original “next” pointer of the *parent* (falling back to its own +/// parent_next if its own next was `None`). +fn patch_block_endings(head: Option>>) { + fn patch_block_endings_to_parent( + mut cur: Option>>, + parent_next: Option>>, + ) { + while let Some(rc_cmd) = cur { + // Borrow mutably just long enough to inspect/rewire this node + let cmd = rc_cmd.borrow_mut(); + // Save this node’s own next pointer + let own_next = cmd.next.clone(); + // Decide what “splice target” to use: + // - if this node has its own_next, use that + // - otherwise, fall back to parent_next + let splice_target = own_next.clone().or(parent_next.clone()); + + // If it has a sub-block, recurse and then patch its tail + if let CommandData::BranchTarget(Some(ref sub_head)) = cmd.data + && cmd.code == '{' + { + // 1) recurse into the sub-chain, passing splice_target + patch_block_endings_to_parent(Some(sub_head.clone()), splice_target.clone()); + + // 2) find the tail of that sub-chain + let mut tail = sub_head.clone(); + loop { + let next_in_sub = tail.borrow().next.clone(); + match next_in_sub { + Some(n) => tail = n, + None => break, + } + } + + // 3) splice the tail’s `.next` to splice_target + tail.borrow_mut().next.clone_from(&splice_target); + } + + // drop the borrow before moving on + drop(cmd); + + // advance to the next sibling in this level + cur = own_next; + } + } + + // top-level has no parent, so pass None + patch_block_endings_to_parent(head, None); +} + +/// Populate the context's label map with references to associated commands. +fn populate_label_map( + mut cur: Option>>, + context: &mut ProcessingContext, +) -> UResult<()> { + while let Some(rc_cmd) = cur.take() { + // Borrow mutably just long enough to inspect/rewire this node + let cmd = rc_cmd.borrow_mut(); + + // Extract any label to insert after borrow ends + let maybe_label = match &cmd.data { + CommandData::BranchTarget(Some(sub_head)) => { + populate_label_map(Some(sub_head.clone()), context)?; + None + }, + CommandData::Label(Some(label)) => Some(label.clone()), + _ => None, + }; + + if let Some(label) = maybe_label + && cmd.code == ':' + { + if context.label_to_command_map.contains_key(&label) { + return semantic_error(&cmd.location, format!("duplicate label `{label}'")); + } + context.label_to_command_map.insert(label, rc_cmd.clone()); + } + + cur.clone_from(&cmd.next); + } + Ok(()) +} + +/// Populate the context's address range command list with references to +/// associated commands. +fn populate_range_commands(mut cur: Option>>, context: &mut ProcessingContext) { + while let Some(rc_cmd) = cur.take() { + // Borrow mutably just long enough to inspect/rewire this node + let cmd = rc_cmd.borrow_mut(); + + // Recursively process blocks. + if let CommandData::BranchTarget(Some(sub_head)) = &cmd.data { + populate_range_commands(Some(Rc::clone(sub_head)), context); + } + + if cmd.addr2.is_some() { + // Save detected range command. + context.range_commands.push(Rc::clone(&rc_cmd)); + } + + cur.clone_from(&cmd.next); + } +} + +/// Replace branch labels with references to the corresponding commands. +/// Raise an error on undefined labels. +fn resolve_branch_targets( + mut cur: Option>>, + context: &mut ProcessingContext, +) -> UResult<()> { + while let Some(rc_cmd) = cur.take() { + // Borrow mutably just long enough to inspect/rewire this node + let mut cmd = rc_cmd.borrow_mut(); + + // Recurse into blocks + if let CommandData::BranchTarget(Some(sub_head)) = &cmd.data { + resolve_branch_targets(Some(sub_head.clone()), context)?; + } + + // Only for 't' or 'b' commands: + if matches!(cmd.code, 't' | 'b') { + // Take ownership of the current data + let old_data = mem::replace(&mut cmd.data, CommandData::None); + + // Build the replacement + let new_data = match old_data { + CommandData::Label(Some(label)) => { + let target = context + .label_to_command_map + .get(&label) + .cloned() + .ok_or_else(|| { + semantic_error::<()>(&cmd.location, format!("undefined label `{label}'")) + .unwrap_err() + })?; + CommandData::BranchTarget(Some(target)) + }, + CommandData::Label(None) => CommandData::BranchTarget(None), + other => other, // put back anything else unchanged + }; + + // Store it back + cmd.data = new_data; + } + + // Advance to the next sibling + cur.clone_from(&cmd.next); + } + Ok(()) +} + +/// Compile provided scripts into a sequence of commands. +fn compile_sequence( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, + context: &mut ProcessingContext, +) -> UResult>>> { + let mut head: Option>> = None; + let mut tail: Option>> = None; + + loop { + line.eat_spaces(); + + // According to POSIX: "If the first two characters in the script are + // "#n", the default output shall be suppressed". + if !line.eol() && line.current() == '#' && lines.get_line_number() == 1 && line.get_pos() == 0 + { + line.advance(); + if !line.eol() && line.current() == 'n' { + context.quiet = true; + } + // Ignore rest of line + while !line.eol() { + line.advance(); + } + } + + if line.eol() || line.current() == '#' { + match lines.next_line()? { + None => { + return Ok(head); + }, + Some(line_string) => { + *line = ScriptCharProvider::new(&line_string); + }, + } + continue; + } else if line.current() == ';' { + line.advance(); + continue; + } + + let mut cmd = Rc::new(RefCell::new(Command::at_position(lines, line))); + let n_addr = compile_address_range(lines, line, &mut cmd, context)?; + line.eat_spaces(); + let mut cmd_spec = get_verified_cmd_spec(lines, line, n_addr, context.posix)?; + // Compile the command according to its specification. + let mut cmd_mut = cmd.borrow_mut(); + cmd_mut.code = line.current(); + match (cmd_spec.handler)(lines, line, &mut cmd_mut, context)? { + CommandHandling::GetNext => { + cmd_spec = get_verified_cmd_spec(lines, line, n_addr, context.posix)?; + cmd_mut.code = line.current(); + (cmd_spec.handler)(lines, line, &mut cmd_mut, context)?; + }, + CommandHandling::Return => return Ok(head), + CommandHandling::Continue => (), + } + drop(cmd_mut); + + if let Some(ref t) = tail { + // there's already a tail: link it + t.borrow_mut().next = Some(cmd.clone()); + } else { + // first element: set head + head = Some(cmd.clone()); + } + tail = Some(cmd); + } +} + +/// Return true if c is a valid character for specifying a context address +fn is_address_char(c: char) -> bool { + matches!(c, '0'..='9' | '/' | '\\' | '$') +} + +/// Compile a command's optional address range into cmd. +/// Return the number of addresses encountered. +fn compile_address_range( + lines: &ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Rc>, + context: &ProcessingContext, +) -> UResult { + let mut n_addr = 0; + let mut cmd = cmd.borrow_mut(); + + let mut is_line0 = false; + + line.eat_spaces(); + if !line.eol() && is_address_char(line.current()) { + let addr1 = compile_address(lines, line, context)?; + is_line0 = matches!(addr1, Address::Line(0)); + cmd.addr1 = Some(addr1); + if is_line0 && context.posix { + // 0 starting address is a GNU extension. + return compilation_error(lines, line, "address 0 is invalid in POSIX mode"); + } + n_addr += 1; + } + + line.eat_spaces(); + if n_addr == 1 && !line.eol() && matches!(line.current(), ',' | '~') { + let is_step_match = line.current() == '~'; // E.g. 0~2: Pick even-numbered lines + line.advance(); + line.eat_spaces(); + let is_step_end = if line.current() == '~' { + // E.g. /foo/,~10: Start at foo, include all lines until multiple of 10 is + // reached. + line.advance(); + line.eat_spaces(); + true + } else { + false + }; + + if (is_step_match || is_step_end) && context.posix { + // ~ steps are a GNU extension. + return compilation_error(lines, line, "~step is invalid in POSIX mode"); + } + + // Look for second address. + if !line.eol() { + let addr2 = compile_address(lines, line, context)?; + // Set step_n to the number specified in the (required numeric) address. + let step_n = if is_step_match || is_step_end { + match addr2 { + Address::Line(n) => n, + _ => { + return compilation_error( + lines, + line, + "~step can only be specified on numeric addresses", + ); + }, + } + } else { + 0 // dummy, not used + }; + + if is_line0 && !matches!(addr2, Address::Re(_)) && !is_step_match { + return compilation_error(lines, line, ERR_ADDRESS_0_USAGE); + } + + // If needed, transform Address::Line into Address::Step*. + cmd.addr2 = if is_step_match { + Some(Address::StepMatch(step_n)) + } else if is_step_end { + Some(Address::StepEnd(step_n)) + } else { + Some(addr2) + }; + n_addr += 1; + } + } + + // Zero-address read command check + if is_line0 && n_addr == 1 { + // After retrieval of first address, subsequent spaces + // are consumed unconditionally. By now, the position + // must be in non-whitespace character or EOL. + if line.eol() || line.current() != 'r' { + return compilation_error(lines, line, ERR_ADDRESS_0_USAGE); + } + } + + Ok(n_addr) +} + +/// Read the line's remaining characters as a file path and return it. +fn read_file_path(lines: &ScriptLineProvider, line: &mut ScriptCharProvider) -> UResult { + line.advance(); // Skip the command/w character + line.eat_spaces(); // Skip any leading whitespace + + let mut path = String::new(); + while !line.eol() { + path.push(line.current()); + line.advance(); + } + + if path.is_empty() { + compilation_error(lines, line, "missing file path") + } else { + // Patched for pi-uutils-ctx embedding: resolve `r`/`w`/`s///w` file + // operands against the shell working directory. + Ok(pi_uutils_ctx::resolve(path)) + } +} + +/// Compile and return a single range address specification. +// Due to their irregular syntax ~ addresses are returned as Line() and adjusted +// in compile_address_range(). +fn compile_address( + lines: &ScriptLineProvider, + line: &mut ScriptCharProvider, + context: &ProcessingContext, +) -> UResult
{ + let mut icase = false; + + if line.eol() { + return compilation_error(lines, line, "expected context address"); + } + + match line.current() { + '\\' | '/' => { + // Regular expression + if line.current() == '\\' { + // The next character is an arbitrary delimiter + line.advance(); + } + let regex_mode = if context.regex_extended { + RegexMode::Extended + } else { + RegexMode::Basic + }; + let re = parse_regex(lines, line, regex_mode)?; + // Skip over delimiter + line.advance(); + + line.eat_spaces(); + if !line.eol() && line.current() == 'I' { + icase = true; + line.advance(); + } + + Ok(Address::Re(compile_regex(lines, line, &re, context, icase, false)?)) + }, + '$' => { + line.advance(); + Ok(Address::Last) + }, + '+' => { + line.advance(); + let number = parse_number(lines, line, true)?.unwrap(); + Ok(Address::RelLine(number)) + }, + c if c.is_ascii_digit() => { + let number = parse_number(lines, line, true)?.unwrap(); + Ok(Address::Line(number)) + }, + _ => panic!("invalid context address"), + } +} + +/// Parse and return the decimal number at the current line position. +/// Advance the line to first non-digit or EOL. +/// Issue an error if the number is required. +fn parse_number( + lines: &ScriptLineProvider, + line: &mut ScriptCharProvider, + required: bool, +) -> UResult> { + let mut num_str = String::new(); + + while !line.eol() && line.current().is_ascii_digit() { + num_str.push(line.current()); + line.advance(); + } + + if num_str.is_empty() { + if required { + return compilation_error(lines, line, "number expected"); + } + return Ok(None); + } + + num_str + .parse::() + .map_err(|_| format!("invalid number '{num_str}'")) + .map_err(|msg| compilation_error::(lines, line, msg).unwrap_err()) + .map(Some) +} + +/// Parse the end of a command, failing with an error on extra characters. +fn parse_command_ending( + lines: &ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Command, +) -> UResult<()> { + if !line.eol() && line.current() == ';' { + line.advance(); + return Ok(()); + } + + if !line.eol() { + return compilation_error( + lines, + line, + format!("extra characters at the end of the {} command", cmd.code), + ); + } + + Ok(()) +} + +/// Convert a primitive BRE pattern to a safe ERE-compatible pattern string. +/// - Replaces `\(`, `\)`, `\?`, `\+`, `\|`, `\{` and `\}` with `(`, `)`, `?`, +/// `+`, `|`, `{` and `}`. +/// - Puts single-digit back-references in non-capturing groups.. +/// - Escapes ERE-only metacharacters: `+ ? { } | ( )`. +/// - Leaves all other characters as-is. +fn bre_to_ere(pattern: &str) -> String { + let mut result = String::with_capacity(pattern.len()); + let mut chars = pattern.chars().peekable(); + + let mut at_beginning = true; + let mut previous: Option = None; + while let Some(c) = chars.next() { + if c == '\\' { + match chars.peek() { + Some('(') => { + chars.next(); + result.push('('); // Group start + }, + Some(')') => { + chars.next(); + result.push(')'); // Group end + }, + Some('?') => { + chars.next(); + result.push('?'); // Quantifier 0 or 1 + }, + Some('+') => { + chars.next(); + result.push('+'); // Quantifier 1 or more + }, + Some('|') => { + chars.next(); + result.push('|'); // Alternation operator + }, + Some('{') => { + chars.next(); + result.push('{'); // Brace quantifier start + }, + Some('}') => { + chars.next(); + result.push('}'); // Brace quantifier end + }, + Some(v) if v.is_ascii_digit() => { + // Back-reference. In sed BREs these are single-digit + // (\1-\9) whereas fancy_regex supports multi-digit + // back-references. Put them in a non-capturing group + // to avoid having the number extend beyond the single + // digit. Example: In sed \11 matches group 1 followed + // by '1', not group 11. + result.push_str(&format!(r"(?:\{v})")); + chars.next(); + }, + Some(&next) => { + // Preserve other escaped characters. + chars.next(); + result.push('\\'); + result.push(next); + }, + None => { + // Trailing backslash; keep it. + result.push('\\'); + }, + } + } else { + match c { + '+' | '?' | '{' | '}' | '|' | '(' | ')' => { + // Escape unsupported ERE metacharacters. + result.push('\\'); + result.push(c); + }, + '^' if !at_beginning && previous != Some('[') => { + // In BREs ^ has special meaning at the beginning + // and as bracket negation. This heuristic escapes + // all other uses, which per POSIX are valid in EREs. + // "the ERE "a^b" is valid, but can never match because + // the 'a' prevents the expression "^b" from matching + // starting at the first character." + // POSIX 9.4.9 ERE Expression Anchoring + result.push('\\'); + result.push(c); + }, + '$' if chars.peek().is_some() => { + // Similarly for $ appearing not at the end. + result.push('\\'); + result.push(c); + }, + _ => result.push(c), + } + } + at_beginning = false; + previous = Some(c); + } + + result +} + +/// Compile the provided regular expression string into a corresponding engine. +/// An empty pattern results in None, which means that the last RE employed +/// at runtime will be used. +fn compile_regex( + lines: &ScriptLineProvider, + line: &ScriptCharProvider, + pattern: &str, + context: &ProcessingContext, + icase: bool, + multiline: bool, +) -> UResult> { + if pattern.is_empty() { + return Ok(None); + } + + // Convert basic to extended regular expression if needed. + let pattern = if context.regex_extended { + pattern + } else { + &bre_to_ere(pattern) + }; + + let mut modifiers = String::new(); + if icase { + modifiers.push('i'); + } + if multiline { + modifiers.push('m'); + } + let pattern = if modifiers.is_empty() { + pattern.to_string() + } else { + format!("(?{modifiers}){pattern}") + }; + + // Compile into engine. + let compiled = Regex::new(&pattern).map_err(|e| { + compilation_error::(lines, line, format!("invalid regex '{pattern}': {e}")) + .unwrap_err() + })?; + + Ok(Some(compiled)) +} + +/// Compile a regular expression replacement string. +pub fn compile_replacement( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, +) -> UResult { + let mut parts = Vec::new(); + let mut literal = String::new(); + + let delimiter = line.current(); + line.advance(); + + loop { + while !line.eol() { + match line.current() { + '\\' => { + line.advance(); + + // Line input_action + if line.eol() { + if let Some(next_line_string) = lines.next_line()? { + literal.push('\n'); + *line = ScriptCharProvider::new(&next_line_string); + continue; + } + return compilation_error( + lines, + line, + "unterminated substitute replacement (unexpected EOF)", + ); + } + + match line.current() { + // \0 - \9 + c @ '0'..='9' => { + let ref_num = c.to_digit(10).unwrap(); + + if !literal.is_empty() { + parts.push(ReplacementPart::Literal(std::mem::take(&mut literal))); + } + if ref_num == 0 { + parts.push(ReplacementPart::WholeMatch); + } else { + parts.push(ReplacementPart::Group(ref_num)); + } + line.advance(); + }, + + // Literal \ and & + '\\' | '&' => { + literal.push(line.current()); + line.advance(); + }, + + // Literal delimiter + v if v == delimiter => { + literal.push(line.current()); + line.advance(); + }, + + // other escape sequences + _ => { + if let Some(decoded) = parse_char_escape(line) { + literal.push(decoded); + } else { + literal.push('\\'); + literal.push(line.current()); + line.advance(); + } + }, + } + }, + + '&' => { + if !literal.is_empty() { + parts.push(ReplacementPart::Literal(std::mem::take(&mut literal))); + } + parts.push(ReplacementPart::WholeMatch); + line.advance(); + }, + + '\n' => { + return compilation_error( + lines, + line, + "unescaped newline inside substitute replacement", + ); + }, + + c if c == delimiter => { + line.advance(); // skip closing delimiter + if !literal.is_empty() { + parts.push(ReplacementPart::Literal(literal)); + } + return Ok(ReplacementTemplate::new(parts)); + }, + + c => { + literal.push(c); + line.advance(); + }, + } + } + + // Fetch next line for continued replacement string + if let Some(next_line_string) = lines.next_line()? { + *line = ScriptCharProvider::new(&next_line_string); + } else { + return compilation_error(lines, line, "unterminated substitute replacement"); + } + } +} + +// Handles s +fn compile_subst_command( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Command, + context: &mut ProcessingContext, +) -> UResult { + line.advance(); // move past 's' + + let delimiter = line.current(); + if delimiter == '\0' || delimiter == '\\' { + return compilation_error( + lines, + line, + "substitute pattern cannot be delimited by newline or backslash", + ); + } + + let regex_mode = if context.regex_extended { + RegexMode::Extended + } else { + RegexMode::Basic + }; + let pattern = parse_regex(lines, line, regex_mode)?; + let mut subst = Box::new(Substitution::default()); + + subst.replacement = compile_replacement(lines, line)?; + compile_subst_flags(lines, line, &mut subst, context.posix, context.sandbox)?; + + if pattern.is_empty() && (subst.ignore_case || subst.multiline) { + return compilation_error( + lines, + line, + "cannot specify modifiers on an empty regular expression", + ); + } + + // Compile regex with now known modifier flags. + subst.regex = compile_regex(lines, line, &pattern, context, subst.ignore_case, subst.multiline)?; + + // Catch invalid group references at compile time, if possible. + if let Some(regex) = &subst.regex + && subst.replacement.max_group_number > regex.captures_len() - 1 + { + return compilation_error( + lines, + line, + format!("invalid reference \\{} on `s' command's RHS", subst.replacement.max_group_number), + ); + } + cmd.data = CommandData::Substitution(subst); + + parse_command_ending(lines, line, cmd)?; + Ok(CommandHandling::Continue) +} + +// Handles y +fn compile_trans_command( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Command, + _context: &mut ProcessingContext, +) -> UResult { + line.advance(); // move past 'y' + + let delimiter = line.current(); + if delimiter == '\0' || delimiter == '\\' { + return compilation_error( + lines, + line, + "transliteration string cannot be delimited by newline or backslash", + ); + } + + let source = parse_transliteration(lines, line)?; + let target = parse_transliteration(lines, line)?; + if source.chars().count() != target.chars().count() { + return compilation_error(lines, line, "transliteration strings are not the same length"); + } + + let transliteration = Box::new(Transliteration::from_strings(&source, &target)); + cmd.data = CommandData::Transliteration(transliteration); + + line.advance(); // move past last delimiter + parse_command_ending(lines, line, cmd)?; + Ok(CommandHandling::Continue) +} + +/// Parse the substitution command's optional flags +pub fn compile_subst_flags( + lines: &ScriptLineProvider, + line: &mut ScriptCharProvider, + subst: &mut Substitution, + posix: bool, + sandbox: bool, +) -> UResult<()> { + let mut seen_g_or_n = false; + + subst.occurrence = 1; // default + subst.print_flag = false; + subst.ignore_case = false; + subst.execute = false; + subst.multiline = false; + subst.write_file = None; + + loop { + line.eat_spaces(); + if line.eol() { + break; + } + + match line.current() { + 'g' => { + if seen_g_or_n { + return compilation_error( + lines, + line, + "multiple 'g' or numeric flags in substitute command", + ); + } + seen_g_or_n = true; + subst.occurrence = 0; + line.advance(); + }, + + 'p' => { + subst.print_flag = true; + line.advance(); + }, + + 'i' | 'I' => { + if posix { + return compilation_error(lines, line, ERR_UNKNOWN_OPTION_TO_S); + } + subst.ignore_case = true; + line.advance(); + }, + + 'm' | 'M' => { + if posix { + return compilation_error(lines, line, ERR_UNKNOWN_OPTION_TO_S); + } + subst.multiline = true; + line.advance(); + }, + + 'e' => { + if posix || sandbox { + return compilation_error( + lines, + line, + "the 'e' substitute flag is not allowed with --posix or --sandbox", + ); + } + subst.execute = true; + line.advance(); + }, + + _c @ '1'..='9' => { + if seen_g_or_n { + return compilation_error( + lines, + line, + "multiple 'g' or numeric flags in substitute command", + ); + } + + let mut number = 0usize; + while !line.eol() && line.current().is_ascii_digit() { + number = number + .checked_mul(10) + .and_then(|n| n.checked_add(line.current().to_digit(10).unwrap() as usize)) + .ok_or_else(|| { + compilation_error::<()>(lines, line, "overflow in numeric substitute flag") + .unwrap_err() + })?; + line.advance(); + } + + subst.occurrence = number; + seen_g_or_n = true; + }, + + 'w' => { + if sandbox { + return compilation_error(lines, line, ERR_SANDBOX); + } + let location = ScriptLocation::at_position(lines, line); + let path = read_file_path(lines, line)?; + subst.write_file = Some(NamedWriter::new(path, location)?); + return Ok(()); // 'w' is the last flag allowed + }, + + ';' | '\n' => break, + + other => { + return compilation_error(lines, line, format!("invalid substitute flag: '{other}'")); + }, + } + } + + Ok(()) +} + +// Handles } +fn compile_end_group_command( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Command, + context: &mut ProcessingContext, +) -> UResult { + if context.parsed_block_nesting == 0 { + return compilation_error(lines, line, "unexpected `}'"); + } + context.parsed_block_nesting -= 1; + line.advance(); + line.eat_spaces(); + parse_command_ending(lines, line, cmd)?; + Ok(CommandHandling::Return) +} + +// Handles ! +fn compile_negation_command( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Command, + _context: &mut ProcessingContext, +) -> UResult { + line.advance(); + line.eat_spaces(); + if cmd.non_select { + return compilation_error(lines, line, "negation already applied"); + } + cmd.non_select = true; + Ok(CommandHandling::GetNext) +} + +/// Compile a command that doesn't take any arguments +// Handles d D g G h H l n N p P q x = +fn compile_empty_command( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Command, + _context: &mut ProcessingContext, +) -> UResult { + line.advance(); // Skip the command character + line.eat_spaces(); // Skip any trailing whitespace + + parse_command_ending(lines, line, cmd)?; + Ok(CommandHandling::Continue) +} + +// Handles r +fn compile_read_file_command( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Command, + context: &mut ProcessingContext, +) -> UResult { + if context.sandbox { + return compilation_error(lines, line, ERR_SANDBOX); + } + let path = read_file_path(lines, line)?; + cmd.data = CommandData::Path(path); + Ok(CommandHandling::Continue) +} + +// Handles w +fn compile_write_file_command( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Command, + context: &mut ProcessingContext, +) -> UResult { + if context.sandbox { + return compilation_error(lines, line, ERR_SANDBOX); + } + let location = ScriptLocation::at_position(lines, line); + let path = read_file_path(lines, line)?; + cmd.data = CommandData::NamedWriter(NamedWriter::new(path, location)?); + Ok(CommandHandling::Continue) +} + +// Handles { +fn compile_block_command( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Command, + context: &mut ProcessingContext, +) -> UResult { + line.advance(); // move past '{' + context.parsed_block_nesting += 1; + let block_body = compile_sequence(lines, line, context)?; + cmd.data = CommandData::BranchTarget(block_body); + Ok(CommandHandling::Continue) +} + +// Handles b, t, : +fn compile_label_command( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Command, + _context: &mut ProcessingContext, +) -> UResult { + /// Return true if `c` is in the POSIX portable filename character set. + fn is_portable_filename_char(c: char) -> bool { + c.is_ascii_alphanumeric() // A–Z, a–z, 0–9 + || matches!(c, '.' | '_' | '-') + } + + line.advance(); // Skip the command character + line.eat_spaces(); // Skip any leading whitespace + + let mut label = String::new(); + while !line.eol() && is_portable_filename_char(line.current()) { + label.push(line.current()); + line.advance(); + } + + if label.is_empty() { + if cmd.code == ':' { + return compilation_error(lines, line, "empty label"); + } + cmd.data = CommandData::Label(None); + } else { + cmd.data = CommandData::Label(Some(label)); + } + + line.eat_spaces(); // Skip any trailing whitespace + parse_command_ending(lines, line, cmd)?; + Ok(CommandHandling::Continue) +} + +/// Return the default `l` command output width. +// Patched for pi-uutils-ctx embedding: the context streams are never a +// terminal, so upstream's terminal_size() width auto-detection is dropped. +fn output_width() -> usize { + DEFAULT_OUTPUT_WIDTH +} + +/// Compile commands that take a number as an argument. +// Handles l q Q +fn compile_number_command( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Command, + _context: &mut ProcessingContext, +) -> UResult { + line.advance(); // Skip the command character + line.eat_spaces(); // Skip any leading whitespace + + match parse_number(lines, line, false)? { + Some(n) => { + cmd.data = CommandData::Number(n); + }, + None => match cmd.code { + 'q' | 'Q' => { + cmd.data = CommandData::Number(0); + }, + 'l' => { + cmd.data = CommandData::Number(output_width()); + }, + _ => panic!("invalid number-expecting command"), + }, + } + + line.eat_spaces(); // Skip any trailing whitespace + parse_command_ending(lines, line, cmd)?; + Ok(CommandHandling::Continue) +} + +/// Compile commands that take text as an argument. +// Handles a, c, i +// According to POSIX, these commands expect \ followed by text. +// As a GNU extension the initial \ can be ommitted, and from then on +// character escapes are honored. +fn compile_text_command( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Command, + context: &mut ProcessingContext, +) -> UResult { + line.advance(); // Skip the command character. + line.eat_spaces(); // Skip any leading whitespace. + if context.posix { + compile_text_command_posix(lines, line, cmd, context) + } else { + compile_text_command_gnu(lines, line, cmd, context) + } +} + +/// Compile commands that take text as an argument (GNU syntax). +// Handles a, c, i; after the command and initial whitespace have been consumed. +// According to POSIX, these commands expect \ followed by text. +// As a GNU extension the initial \ can be ommitted, and from then on +// character escapes are honored. +fn compile_text_command_gnu( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Command, + _context: &mut ProcessingContext, +) -> UResult { + // True after a \ at the end of a line + let mut escaped_newline = false; + + if line.eol() { + return compilation_error( + lines, + line, + format!("command `{}' expects \\ followed by text", cmd.code), + ); + } + + // Skip optional \. + if !line.eol() && line.current() == '\\' { + line.advance(); + escaped_newline = line.eol(); + } + + // Gather replacement text. Stop on a non-escaped newline. + let mut text = String::new(); + 'text_content: loop { + if escaped_newline { + match lines.next_line()? { + None => { + break 'text_content; + }, + Some(line_string) => { + *line = ScriptCharProvider::new(&line_string); + }, + } + escaped_newline = false; + } + + // Non-escaped newline + if line.eol() { + text.push('\n'); + break 'text_content; + } + + if line.current() == '\\' { + line.advance(); + + if line.eol() { + escaped_newline = true; + text.push('\n'); + continue 'text_content; + } + + if let Some(decoded) = parse_char_escape(line) { + text.push(decoded); + } else { + // Invalid escapes result in the escaped character. + text.push(line.current()); + line.advance(); + } + } else { + text.push(line.current()); + line.advance(); + } + } + cmd.data = CommandData::Text(Rc::from(text)); + Ok(CommandHandling::Continue) +} + +/// Compile commands that take text as an argument (POSIX syntax). +// Handles a, c, i; after the command and initial whitespace have been consumed. +// According to POSIX, these commands expect \ followed by text. +fn compile_text_command_posix( + lines: &mut ScriptLineProvider, + line: &mut ScriptCharProvider, + cmd: &mut Command, + _context: &mut ProcessingContext, +) -> UResult { + if line.eol() || line.current() != '\\' { + return compilation_error( + lines, + line, + format!("command `{}' expects \\ followed by text", cmd.code), + ); + } + + line.advance(); // Skip \. + line.eat_spaces(); // Skip any whitespace at the end of \. + if !line.eol() { + return compilation_error( + lines, + line, + format!("extra characters after \\ at the end of `{}' command", cmd.code), + ); + } + + let mut text = String::new(); + while let Some(line) = lines.next_line()? { + if line.ends_with('\\') { + // Line ends with \ to escape \n; remove the trailing \. + text.push_str(&line[..line.len() - 1]); + text.push('\n'); + } else { + text.push_str(&line); + text.push('\n'); + break; + } + } + + if text.is_empty() { + compilation_error(lines, line, "incomplete command")?; + } + + cmd.data = CommandData::Text(Rc::from(text)); + Ok(CommandHandling::Continue) +} + +// Return the specification for the command letter at the current line position +// checking for diverse errors. +fn get_verified_cmd_spec( + lines: &ScriptLineProvider, + line: &ScriptCharProvider, + n_addr: usize, + posix: bool, +) -> UResult { + if line.eol() { + return compilation_error(lines, line, "command expected"); + } + + let ch = line.current(); + let cmd_spec = get_cmd_spec(lines, line, ch, posix)?; + + if n_addr > cmd_spec.n_addr { + return compilation_error( + lines, + line, + format!("command {} expects up to {} address(es), found {}", ch, cmd_spec.n_addr, n_addr), + ); + } + + Ok(cmd_spec) +} + +// Look up a command addresses and handler by its command code. +fn get_cmd_spec( + lines: &ScriptLineProvider, + line: &ScriptCharProvider, + cmd_code: char, + posix: bool, +) -> UResult { + match cmd_code { + '!' => Ok(CommandSpec { n_addr: 2, handler: compile_negation_command }), + '=' => Ok(CommandSpec { n_addr: if posix { 1 } else { 2 }, handler: compile_empty_command }), + ':' => Ok(CommandSpec { n_addr: 0, handler: compile_label_command }), + '{' => Ok(CommandSpec { n_addr: 2, handler: compile_block_command }), + '}' => Ok(CommandSpec { n_addr: 0, handler: compile_end_group_command }), + 'a' | 'i' => { + Ok(CommandSpec { n_addr: if posix { 1 } else { 2 }, handler: compile_text_command }) + }, + 'b' | 't' => Ok(CommandSpec { n_addr: 2, handler: compile_label_command }), + 'c' => Ok(CommandSpec { n_addr: 2, handler: compile_text_command }), + 'd' | 'D' | 'g' | 'G' | 'h' | 'H' | 'n' | 'N' | 'p' | 'P' | 'x' => { + Ok(CommandSpec { n_addr: 2, handler: compile_empty_command }) + }, + 'z' if !posix => Ok(CommandSpec { n_addr: 2, handler: compile_empty_command }), + 'l' => Ok(CommandSpec { n_addr: 2, handler: compile_number_command }), + 'q' => { + Ok(CommandSpec { n_addr: if posix { 1 } else { 2 }, handler: compile_number_command }) + }, + // Q is a GNU extension + 'Q' => Ok(CommandSpec { n_addr: 1, handler: compile_number_command }), + 'r' => { + Ok(CommandSpec { n_addr: if posix { 1 } else { 2 }, handler: compile_read_file_command }) + }, + 's' => Ok(CommandSpec { n_addr: 2, handler: compile_subst_command }), + 'w' => Ok(CommandSpec { n_addr: 2, handler: compile_write_file_command }), + 'y' => Ok(CommandSpec { n_addr: 2, handler: compile_trans_command }), + _ => compilation_error(lines, line, format!("invalid command code `{cmd_code}'")), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::sed::fast_io::IOChunk; + + // Return an empty line provider and a char provider for the specified str. + fn make_providers(input: &str) -> (ScriptLineProvider, ScriptCharProvider) { + let lines = ScriptLineProvider::new(vec![]); // Empty for tests + let line = ScriptCharProvider::new(input); + (lines, line) + } + + fn make_line_provider(lines: &[&str]) -> ScriptLineProvider { + let input = lines + .iter() + .map(|s| ScriptValue::StringVal((*s).to_string())) + .collect(); + ScriptLineProvider::new(input) + } + + fn make_char_provider(input: &str) -> ScriptCharProvider { + ScriptCharProvider::new(input) + } + + /// Return a default ProcessingContext for use in tests. + pub fn ctx() -> ProcessingContext { + ProcessingContext::default() + } + + // get_cmd_spec + #[test] + fn test_lookup_empty_command() { + let (lines, line) = make_providers("123abc"); + let cmd = get_cmd_spec(&lines, &line, 'd', false).unwrap(); + assert_eq!(cmd.n_addr, 2); + } + + #[test] + fn test_lookup_text_command() { + let (lines, line) = make_providers("123abc"); + let cmd = get_cmd_spec(&lines, &line, 'a', false).unwrap(); + assert_eq!(cmd.n_addr, 2); + } + + #[test] + fn test_lookup_nonselect_command() { + let (lines, line) = make_providers("123abc"); + let cmd = get_cmd_spec(&lines, &line, '!', false).unwrap(); + assert_eq!(cmd.n_addr, 2); + } + + #[test] + fn test_lookup_endgroup_command() { + let (lines, line) = make_providers("123abc"); + let cmd = get_cmd_spec(&lines, &line, '}', false).unwrap(); + assert_eq!(cmd.n_addr, 0); + } + + #[test] + fn test_lookup_invalid_command() { + let (lines, line) = make_providers("123abc"); + let result = get_cmd_spec(&lines, &line, 'Z', false); + assert!(result.is_err()); + } + + // Utility to create a ScriptCharProvider from a &str + fn char_provider_from(s: &str) -> ScriptCharProvider { + ScriptCharProvider::new(s) + } + + // compilation_error + #[test] + fn test_compilation_error_message_format() { + let lines = ScriptLineProvider::with_active_state("test.sed", 42); + let mut line = char_provider_from("whatever"); + line.advance(); // move to position 1 + line.advance(); // move to position 2 + line.advance(); // move to position 3 + line.advance(); // now at position 4 + + let msg = "unexpected token"; + let result: UResult<()> = compilation_error(&lines, &line, msg); + + assert!(result.is_err()); + + let err = result.unwrap_err(); + let msg = err.to_string(); + + assert!(msg.contains("test.sed:42:5: error: unexpected token")); + } + + #[test] + fn test_compilation_error_with_format_message() { + let lines = ScriptLineProvider::with_active_state("input.txt", 3); + let line = char_provider_from("x"); + // We're at position 0 + + let result: UResult<()> = + compilation_error(&lines, &line, format!("invalid command '{}'", 'x')); + + assert!(result.is_err()); + + let err = result.unwrap_err(); + let msg = err.to_string(); + + assert_eq!(msg, "input.txt:3:1: error: invalid command 'x'"); + } + + // get_verified_cmd_spec + #[test] + fn test_missing_command_character() { + let lines = ScriptLineProvider::with_active_state("test.sed", 1); + let line = char_provider_from(""); + let result = get_verified_cmd_spec(&lines, &line, 0, ctx().posix); + + assert!(result.is_err()); + let msg = result.unwrap_err().to_string(); + assert!(msg.contains("test.sed:1:1: error: command expected")); + } + + #[test] + fn test_invalid_command_character() { + let lines = ScriptLineProvider::with_active_state("script.sed", 2); + let line = char_provider_from("@"); + let result = get_verified_cmd_spec(&lines, &line, 0, ctx().posix); + + assert!(result.is_err()); + let msg = result.unwrap_err().to_string(); + assert!(msg.contains("script.sed:2:1: error: invalid command code `@'")); + } + + #[test] + fn test_too_many_addresses() { + let lines = ScriptLineProvider::with_active_state("input.sed", 3); + let line = char_provider_from("q"); // q takes one address + let result = get_verified_cmd_spec(&lines, &line, 2, true); + + assert!(result.is_err()); + let msg = result.unwrap_err().to_string(); + assert!(msg.contains("input.sed:3:1: error: command q expects up to 1 address(es), found 2")); + } + + #[test] + fn test_valid_command_spec() { + let lines = ScriptLineProvider::with_active_state("input.sed", 4); + let line = char_provider_from("a"); // valid command + let result = get_verified_cmd_spec(&lines, &line, 2, ctx().posix); + assert!(result.is_ok()); + let spec = result.unwrap(); + assert_eq!(spec.n_addr, 2); + } + + #[test] + fn test_invalid_address_range_posix() { + let lines = ScriptLineProvider::with_active_state("input.sed", 1); + let line = char_provider_from("i"); // valid command + let result = get_verified_cmd_spec(&lines, &line, 2, true); + assert!(result.is_err()); + let msg = result.unwrap_err().to_string(); + assert!(msg.contains("input.sed:1:1: error: command i expects up to 1 address(es), found 2")); + } + + // parse_number + #[test] + fn test_parse_number_basic() { + let (lines, mut chars) = make_providers("123abc"); + assert_eq!(parse_number(&lines, &mut chars, true).unwrap(), Some(123)); + assert_eq!(chars.current(), 'a'); // Should stop at first non-digit + } + + #[test] + fn test_parse_optional_number_missing() { + let (lines, mut chars) = make_providers(" ;"); + assert_eq!(parse_number(&lines, &mut chars, false).unwrap(), None); + } + + #[test] + fn test_parse_number_invalid() { + let (lines, mut chars) = make_providers("537654897563495734653453434534534534545"); + let err = parse_number(&lines, &mut chars, true).unwrap_err(); + assert!(err.to_string().contains("invalid number")); + } + + #[test] + fn test_parse_required_number_missing() { + let (lines, mut chars) = make_providers(""); + let err = parse_number(&lines, &mut chars, true).unwrap_err(); + assert!(err.to_string().contains("number expected")); + } + + // compile_re + fn dummy_providers() -> (ScriptLineProvider, ScriptCharProvider) { + make_providers("dummy input") + } + + #[test] + fn test_compile_re_basic() { + let (lines, chars) = dummy_providers(); + let regex = compile_regex(&lines, &chars, "abc", &ctx(), false, false) + .unwrap() + .expect("regex should be present"); + assert!(regex.is_match(&mut IOChunk::new_from_str("abc")).unwrap()); + assert!(!regex.is_match(&mut IOChunk::new_from_str("ABC")).unwrap()); + } + + #[test] + fn test_compile_re_extended() { + let (lines, chars) = make_providers("acaa\nbbb\nccc"); + let mut ctx = ctx(); + ctx.regex_extended = true; + let regex = compile_regex(&lines, &chars, "cc{0,}", &ctx, false, false) + .unwrap() + .expect("regex should be present"); + assert!( + regex + .is_match(&mut IOChunk::new_from_str("acaa\nccc")) + .unwrap() + ); + } + + #[test] + fn test_compile_re_case_insensitive() { + let (lines, chars) = dummy_providers(); + let regex = compile_regex(&lines, &chars, "abc", &ctx(), true, false) + .unwrap() + .expect("regex should be present"); + assert!(regex.is_match(&mut IOChunk::new_from_str("abc")).unwrap()); + assert!(regex.is_match(&mut IOChunk::new_from_str("ABC")).unwrap()); + assert!(regex.is_match(&mut IOChunk::new_from_str("AbC")).unwrap()); + } + + #[test] + fn test_compile_re_invalid() { + let (lines, chars) = dummy_providers(); + let result = compile_regex(&lines, &chars, "a[d", &ctx(), false, false); + assert!(result.is_err()); // Should fail due to open bracketed expression + } + + #[test] + fn test_compile_re_multiline_start() { + let (lines, chars) = dummy_providers(); + let regex = compile_regex(&lines, &chars, "^bar", &ctx(), false, true) + .unwrap() + .expect("regex should be present"); + assert!( + regex + .is_match(&mut IOChunk::new_from_str("foo\nbar")) + .unwrap() + ); + } + + #[test] + fn test_compile_re_multiline_end() { + let (lines, chars) = dummy_providers(); + let regex = compile_regex(&lines, &chars, "foo$", &ctx(), false, true) + .unwrap() + .expect("regex should be present"); + assert!( + regex + .is_match(&mut IOChunk::new_from_str("foo\nbar")) + .unwrap() + ); + } + + // compile_address + #[test] + fn test_compile_addr_line_number() { + let (lines, mut chars) = make_providers("42"); + let addr = compile_address(&lines, &mut chars, &ctx()).unwrap(); + assert!(matches!(addr, Address::Line(42))); + } + + #[test] + fn test_compile_addr_relative_line() { + let (lines, mut chars) = make_providers("+7"); + let addr = compile_address(&lines, &mut chars, &ctx()).unwrap(); + assert!(matches!(addr, Address::RelLine(7))); + } + + #[test] + fn test_compile_addr_last_line() { + let (lines, mut chars) = make_providers("$"); + let addr = compile_address(&lines, &mut chars, &ctx()).unwrap(); + assert!(matches!(addr, Address::Last)); + } + + #[test] + fn test_compile_addr_regex() { + let (lines, mut chars) = make_providers("/hello/"); + let addr = compile_address(&lines, &mut chars, &ctx()).unwrap(); + + let Address::Re(Some(re)) = addr else { + panic!("expected Address::Re(Some(_))"); + }; + + assert!(re.is_match(&mut IOChunk::new_from_str("hello")).unwrap()); + } + + #[test] + fn test_compile_addr_regex_backref_match() { + let (lines, mut chars) = make_providers(r"/he\(.\)\1o/"); + let addr = compile_address(&lines, &mut chars, &ctx()).unwrap(); + + match addr { + Address::Re(Some(re)) => { + assert!(re.is_match(&mut IOChunk::new_from_str("hello")).unwrap()); + }, + _ => panic!("expected Address::Re(Some(_))"), + } + } + + #[test] + fn test_compile_addr_regex_backref_no_match() { + let (lines, mut chars) = make_providers(r"/he\(.\)\1o/"); + let addr = compile_address(&lines, &mut chars, &ctx()).unwrap(); + + match addr { + Address::Re(Some(re)) => { + assert!(!re.is_match(&mut IOChunk::new_from_str("helio")).unwrap()); + }, + _ => panic!("expected Address::Re(Some(_))"), + } + } + + #[test] + fn test_compile_addr_regex_other_delimiter() { + let (lines, mut chars) = make_providers("\\#hello#"); + let addr = compile_address(&lines, &mut chars, &ctx()).unwrap(); + + match addr { + Address::Re(Some(re)) => { + assert!(re.is_match(&mut IOChunk::new_from_str("hello")).unwrap()); + }, + _ => panic!("expected Address::Re(Some(_))"), + } + } + + #[test] + fn test_compile_addr_regex_with_modifier() { + let (lines, mut chars) = make_providers("/hello/I"); + let addr = compile_address(&lines, &mut chars, &ctx()).unwrap(); + + match addr { + Address::Re(Some(re)) => { + // Case-insensitive + assert!(re.is_match(&mut IOChunk::new_from_str("HELLO")).unwrap()); + }, + _ => panic!("expected Address::Re(Some(_))"), + } + } + + // compile_address_range + #[test] + fn test_compile_single_line_address() { + let (lines, mut chars) = make_providers("42"); + let mut cmd = Rc::new(RefCell::new(Command::default())); + let n_addr = compile_address_range(&lines, &mut chars, &mut cmd, &ctx()).unwrap(); + + assert_eq!(n_addr, 1); + assert!(matches!(cmd.borrow().addr1, Some(Address::Line(42)))); + } + + #[test] + fn test_compile_relative_address_range() { + let (lines, mut chars) = make_providers("2,+3"); + let mut cmd = Rc::new(RefCell::new(Command::default())); + let n_addr = compile_address_range(&lines, &mut chars, &mut cmd, &ctx()).unwrap(); + + assert_eq!(n_addr, 2); + + assert!(matches!(cmd.borrow().addr1, Some(Address::Line(2)))); + assert!(matches!(cmd.borrow().addr2, Some(Address::RelLine(3)))); + } + + #[test] + fn test_compile_step_match_address() { + let (lines, mut chars) = make_providers("0~2"); + let mut cmd = Rc::new(RefCell::new(Command::default())); + let n_addr = compile_address_range(&lines, &mut chars, &mut cmd, &ctx()).unwrap(); + + assert_eq!(n_addr, 2); + assert!(matches!(cmd.borrow().addr1, Some(Address::Line(0)))); + assert!(matches!(cmd.borrow().addr2, Some(Address::StepMatch(2)))); + } + + #[test] + fn test_compile_step_end_address() { + let (lines, mut chars) = make_providers("1,~10"); + let mut cmd = Rc::new(RefCell::new(Command::default())); + let n_addr = compile_address_range(&lines, &mut chars, &mut cmd, &ctx()).unwrap(); + + assert_eq!(n_addr, 2); + assert!(matches!(cmd.borrow().addr1, Some(Address::Line(1)))); + assert!(matches!(cmd.borrow().addr2, Some(Address::StepEnd(10)))); + } + + #[test] + fn test_compile_last_address() { + let (lines, mut chars) = make_providers("$"); + let mut cmd = Rc::new(RefCell::new(Command::default())); + let n_addr = compile_address_range(&lines, &mut chars, &mut cmd, &ctx()).unwrap(); + + assert_eq!(n_addr, 1); + assert!(matches!(cmd.borrow().addr1, Some(Address::Last))); + } + + #[test] + fn test_compile_absolute_address_range() { + let (lines, mut chars) = make_providers("5,10"); + let mut cmd = Rc::new(RefCell::new(Command::default())); + let n_addr = compile_address_range(&lines, &mut chars, &mut cmd, &ctx()).unwrap(); + + assert_eq!(n_addr, 2); + assert!(matches!(cmd.borrow().addr1, Some(Address::Line(5)))); + assert!(matches!(cmd.borrow().addr2, Some(Address::Line(10)))); + } + + #[test] + fn test_compile_regex_address() { + let (lines, mut chars) = make_providers("/foo/"); + let mut cmd = Rc::new(RefCell::new(Command::default())); + let n_addr = compile_address_range(&lines, &mut chars, &mut cmd, &ctx()).unwrap(); + + assert_eq!(n_addr, 1); + + match cmd.borrow().addr1.as_ref().unwrap() { + Address::Re(Some(re)) => { + assert!(re.is_match(&mut IOChunk::new_from_str("foo")).unwrap()); + assert!(!re.is_match(&mut IOChunk::new_from_str("bar")).unwrap()); + }, + _ => panic!("expected regex address"), + } + } + + #[test] + fn test_compile_regex_address_range_other_delimiter() { + let (lines, mut chars) = make_providers("\\#foo# , \\|bar|"); + let mut cmd = Rc::new(RefCell::new(Command::default())); + let n_addr = compile_address_range(&lines, &mut chars, &mut cmd, &ctx()).unwrap(); + + assert_eq!(n_addr, 2); + + match cmd.borrow().addr1.as_ref().unwrap() { + Address::Re(Some(re)) => { + assert!(re.is_match(&mut IOChunk::new_from_str("foo")).unwrap()); + assert!(!re.is_match(&mut IOChunk::new_from_str("bar")).unwrap()); + }, + _ => panic!("expected regex address"), + } + + match cmd.borrow().addr2.as_ref().unwrap() { + Address::Re(Some(re)) => { + assert!(re.is_match(&mut IOChunk::new_from_str("bar")).unwrap()); + assert!(!re.is_match(&mut IOChunk::new_from_str("foo")).unwrap()); + }, + _ => panic!("expected regex address"), + } + } + + #[test] + fn test_compile_regex_with_modifier() { + let (lines, mut chars) = make_providers("/foo/I"); + let mut cmd = Rc::new(RefCell::new(Command::default())); + let n_addr = compile_address_range(&lines, &mut chars, &mut cmd, &ctx()).unwrap(); + + assert_eq!(n_addr, 1); + + match cmd.borrow().addr1.as_ref().unwrap() { + Address::Re(Some(re)) => { + assert!(re.is_match(&mut IOChunk::new_from_str("FOO")).unwrap()); + assert!(re.is_match(&mut IOChunk::new_from_str("foo")).unwrap()); + }, + _ => panic!("expected regex address"), + } + } + + #[test] + fn test_compile_address_range_error_propagation() { + let (lines, mut chars) = make_providers("1,/abc"); + let mut cmd = Rc::new(RefCell::new(Command::default())); + let result = compile_address_range(&lines, &mut chars, &mut cmd, &ctx()); + + assert!(result.is_err()); + let msg = result.unwrap_err().to_string(); + assert!(msg.contains("unterminated regular expression")); + } + + // compile_sequence + fn empty_line() -> ScriptCharProvider { + ScriptCharProvider::new("") + } + + #[test] + fn test_zero_addr_r_accepted() { + for input in ["0r", "0 r"] { + let (lines, mut chars) = make_providers(input); + let mut cmd = Rc::new(RefCell::new(Command::default())); + let n_addr = compile_address_range(&lines, &mut chars, &mut cmd, &ctx()).unwrap(); + + assert_eq!(n_addr, 1); + assert!(matches!(cmd.borrow().addr1, Some(Address::Line(0)))); + assert_eq!(chars.current(), 'r'); + } + } + + // Zero-address with no commands + #[test] + fn test_zero_addr_no_commands() { + let (lines, mut chars) = make_providers("0"); + let mut cmd = Rc::new(RefCell::new(Command::default())); + let result = compile_address_range(&lines, &mut chars, &mut cmd, &ctx()); + + assert!(result.is_err()); + assert!( + result + .unwrap_err() + .to_string() + .contains(ERR_ADDRESS_0_USAGE) + ); + } + + // Zero-address with a command other than 'r' must still be rejected. + #[test] + fn test_zero_addr_non_r_rejected() { + let (lines, mut chars) = make_providers("0p"); + let mut cmd = Rc::new(RefCell::new(Command::default())); + let result = compile_address_range(&lines, &mut chars, &mut cmd, &ctx()); + + assert!(result.is_err()); + assert!( + result + .unwrap_err() + .to_string() + .contains(ERR_ADDRESS_0_USAGE) + ); + } + + #[test] + fn test_compile_sequence_empty_input() { + let mut provider = make_line_provider(&[]); + let mut opts = ctx(); + + let result = compile_sequence(&mut provider, &mut empty_line(), &mut opts).unwrap(); + assert!(result.is_none()); + } + + #[test] + fn test_compile_sequence_comment_only() { + let mut provider = make_line_provider(&["# comment", " ", ";;"]); + let mut opts = ctx(); + + let result = compile_sequence(&mut provider, &mut empty_line(), &mut opts).unwrap(); + assert!(result.is_none()); + } + + #[test] + fn test_compile_sequence_single_command() { + let mut provider = make_line_provider(&["42q"]); + let mut opts = ctx(); + + let result = compile_sequence(&mut provider, &mut empty_line(), &mut opts).unwrap(); + let binding = result.unwrap(); + let cmd = binding.borrow(); + + assert_eq!(cmd.code, 'q'); + assert!(!cmd.non_select); + + assert!(matches!(cmd.addr1, Some(Address::Line(42)))); + assert!(cmd.next.is_none()); + } + + #[test] + fn test_compile_sequence_non_selected_single_command() { + let mut provider = make_line_provider(&["42!p"]); + let mut opts = ctx(); + + let result = compile_sequence(&mut provider, &mut empty_line(), &mut opts).unwrap(); + let binding = result.unwrap(); + let cmd = binding.borrow(); + + assert_eq!(cmd.code, 'p'); + assert!(cmd.non_select); + + assert!(matches!(cmd.addr1, Some(Address::Line(42)))); + assert!(cmd.next.is_none()); + } + + #[test] + fn test_compile_sequence_multiple_lines() { + let mut provider = make_line_provider(&["1q", "2d"]); + let mut opts = ctx(); + + let result = compile_sequence(&mut provider, &mut empty_line(), &mut opts).unwrap(); + let binding = result.unwrap(); + let first = binding.borrow(); + + assert_eq!(first.code, 'q'); + let binding = first.next.clone().unwrap(); + let second = binding.borrow(); + assert_eq!(second.code, 'd'); + assert!(second.next.is_none()); + } + + #[test] + fn test_compile_sequence_single_line_multiple_commands() { + let mut provider = make_line_provider(&["1q;2d"]); + let mut opts = ctx(); + + let result = compile_sequence(&mut provider, &mut empty_line(), &mut opts).unwrap(); + let binding = result.unwrap(); + let first = binding.borrow(); + + assert_eq!(first.code, 'q'); + let binding = first.next.clone().unwrap(); + let second = binding.borrow(); + assert_eq!(second.code, 'd'); + assert!(second.next.is_none()); + } + + // compile + #[test] + fn test_compile_single_command() { + let scripts = vec![ScriptValue::StringVal("1q".to_string())]; + let mut opts = ProcessingContext::default(); + + let result = compile(scripts, &mut opts).unwrap(); + let binding = result.unwrap(); + let cmd = binding.borrow(); + + assert_eq!(cmd.code, 'q'); + + assert!(matches!(cmd.addr1, Some(Address::Line(1)))); + + assert_eq!(cmd.location.line_number, 1); + assert_eq!(cmd.location.column_number, 1); + assert_eq!(cmd.location.input_name.as_ref(), "