feat(builtins): sweep of GNU/BSD compat fixes + str performance opts

Addresses a broad audit of built-in shell utilities against their real counterparts: timeout gains signal delivery, -s/-k/--preserve-status/--foreground/-v, and GNU exit codes; diff defaults to normal format and gains -w/-b/-B/-i/-c/-x/-L/-s/--strip-trailing-cr and proper -r gating; find fixes -newerXY timestamp comparison direction, anchors -regex to whole paths, and gains BSD -perm +mode, -type lists, -size T/P suffixes, -E/-x/-s flags; date gains BSD -r epoch, -v adjustments, -j -f strptime, and non-greedy -I; tail/head accept obsolete -N/+N at any position with any file count; rg resolves case flags by last occurrence and gains --path-separator and clean -0 output; stat prints integer epochs for %X/%Y/%Z and gains BSD -s/-x/-t; cksum is registered as a builtin; truncate implements -o/--io-blocks and b/= size suffixes; sleep/timeout accept infinity; yes/errno/kill accept hyphen-prefixed operands; nohup -- cmd no longer runs --; which gains BSD -s.
This commit is contained in:
can1357
2026-08-20 01:45:28 +02:00
parent 219aae1303
commit edc0caeb0f
51 changed files with 5909 additions and 980 deletions
Generated
+23 -14
View File
@@ -2661,6 +2661,15 @@ dependencies = [
"zerocopy",
]
[[package]]
name = "hash32"
version = "0.3.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "47d60b12902ba28e2730cd37e95b8c9223af2808df9e902d4df49588d1470606"
dependencies = [
"byteorder",
]
[[package]]
name = "hashbrown"
version = "0.14.5"
@@ -2695,6 +2704,16 @@ version = "0.17.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a"
[[package]]
name = "heapless"
version = "0.9.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "25ba4bd83f9415b58b4ed8dc5714c76e626a105be4646c02630ad730ad3b5aa4"
dependencies = [
"hash32",
"stable_deref_trait",
]
[[package]]
name = "heck"
version = "0.5.0"
@@ -5026,10 +5045,10 @@ dependencies = [
"tokio",
"tokio-util",
"tracing",
"unicode-width 0.2.2",
"uucore",
"uutils_term_grid",
"windows-sys 0.61.2",
"xutf",
"yansi",
]
@@ -5069,6 +5088,7 @@ dependencies = [
"grep-pcre2",
"grep-regex",
"grep-searcher",
"heapless",
"html-to-markdown-rs",
"icy_sixel",
"ignore",
@@ -5107,11 +5127,6 @@ dependencies = [
"tokio-util",
"toml",
"uiautomation",
"unicode-normalization",
"unicode-properties",
"unicode-script",
"unicode-segmentation",
"unicode-width 0.2.2",
"windows-sys 0.61.2",
"winreg 0.56.0",
"x11rb",
@@ -7527,12 +7542,6 @@ version = "0.1.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e70f2a8b45122e719eb623c01822704c4e0907e7e426a05927e1a1cfff5b75d0"
[[package]]
name = "unicode-script"
version = "0.5.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9fb421b350c9aff471779e262955939f565ec18b86c15364e6bdf0d662ca7c1f"
[[package]]
name = "unicode-segmentation"
version = "1.13.3"
@@ -8843,9 +8852,9 @@ checksum = "e450f9b2ed1dff33c94c12589a87338689467b9c4f5d8a5710bd09a847d2c8a7"
[[package]]
name = "xutf"
version = "1.2.0"
version = "1.4.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "fa2ab198275c47f70ceb92678c000b1bb9468a52297c5ce20c5ab5e99fa2c33d"
checksum = "a14f3ca7038796715a5f5fd1c22aac21423b4f86fd9ef5585ef1744a5cfc573e"
[[package]]
name = "xxhash-rust"
+1 -5
View File
@@ -25,7 +25,6 @@ repository = "https://github.com/can1357/oh-my-pi"
[patch.crates-io]
brush-core = { path = "crates/vendor/brush-core" }
[profile.release]
opt-level = 3
lto = "fat"
@@ -232,16 +231,13 @@ clap = { version = "4", features = ["derive"] }
pdf-inspector = "1"
regex = "1"
similar = "3.1.0"
unicode-segmentation = "1.13"
unicode-normalization = "0.1"
unicode-properties = "=0.1.3" # Unicode 16 - must match fancy-regex/HF/CPython oracles (utok)
unicode-width = "0.2"
fontdue = { version = "0.9", default-features = false }
# ──────────────────────────────────────────────────────────────────────────────
# Data Structures - Collections
# ──────────────────────────────────────────────────────────────────────────────
phf = { version = "0.13", features = ["macros"] }
heapless = "0.9"
smallvec = { version = "1.15.1", features = [
"serde",
"write",
+40 -12
View File
File diff suppressed because one or more lines are too long
+28 -9
View File
@@ -207,6 +207,13 @@
"devDependencies": {
"@napi-rs/cli": "catalog:",
"@types/bun": "catalog:",
"cli-truncate": "6.1.1",
"diff": "9.0.0",
"gpt-tokenizer": "4.0.0",
"mitata": "1.0.34",
"slice-ansi": "9.0.0",
"string-width": "8.2.2",
"wrap-ansi": "10.0.0",
},
},
"packages/omptype": {
@@ -967,7 +974,7 @@
"ansi-regex": ["ansi-regex@6.3.0", "", {}, "sha512-WpDfL7NO6j7tH88IDBNVdUJxDh9nmCteAVW9dsep846XdwF4naCBK+/tGLX3KJgcpgMRXCFlTM2hKGoK9FsdrQ=="],
"ansi-styles": ["ansi-styles@4.3.0", "", { "dependencies": { "color-convert": "^2.0.1" } }, "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg=="],
"ansi-styles": ["ansi-styles@6.2.3", "", {}, "sha512-4Dj6M28JB+oAH8kFkTLUo+a2jwOFkuqb3yucU0CANcRRUbxS0cP0nZYCGjcc3BNXwRIsUVmDGgzawme7zvJHvg=="],
"argparse": ["argparse@2.0.1", "", {}, "sha512-8+9WqebbFzpX9OR+Wa6O29asIogeRMzcGtAINdpMHHyAg10f05aSFVBbcEqGf/PXw1EjAZ+q2/bEBg3DvurK3Q=="],
@@ -1007,6 +1014,8 @@
"cli-progress": ["cli-progress@3.12.0", "", { "dependencies": { "string-width": "^4.2.3" } }, "sha512-tRkV3HJ1ASwm19THiiLIXLO7Im7wlTuKnvkYaTkyoAPefqjNg7W7DHKUlGRxy9vxDvbyCYQkQozvptuMkGCg8A=="],
"cli-truncate": ["cli-truncate@6.1.1", "", { "dependencies": { "slice-ansi": "^9.0.0", "string-width": "^8.2.0" } }, "sha512-06p9vyLahLa4zkGcgsGxU6iEkSOiuI4fhCH6Emhe2lPAcoUv73n72DnODsnHA+5wwXGnV0n9M9/qOQJSjYhFhw=="],
"cli-width": ["cli-width@4.1.0", "", {}, "sha512-ouuZd4/dm2Sw5Gmqy6bGyNNNe1qt9RpmxveLSO7KcgsTnU7RXfsw+/bukWGo1abgBiMAic068rclZsO4IWmmxQ=="],
"clipanion": ["clipanion@4.0.0-rc.4", "", { "dependencies": { "typanion": "^3.8.0" } }, "sha512-CXkMQxU6s9GklO/1f714dkKBMu1lopS1WFF0B8o4AxPykR1hpozxSiUZ5ZUeBjfPgCWqbcNOtZVFhB8Lkfp1+Q=="],
@@ -1119,6 +1128,8 @@
"gopd": ["gopd@1.2.0", "", {}, "sha512-ZUKRh6/kUFoAiTAtTYPZJ3hw9wNxx+BIBOijnlG9PnrJsCcSjs1wyyD6vJpaYtgnzDrKYRSqf3OO6Rfa93xsRg=="],
"gpt-tokenizer": ["gpt-tokenizer@4.0.0", "", {}, "sha512-YAWIyzvuVUHEfW7tFfFAxH8qQb+Q3RU9nYOTy7skMNX5qzU6Q8jxTHZLyO56ug1vYvCR7wndzpd3jwD86/mhjQ=="],
"graceful-fs": ["graceful-fs@4.2.11", "", {}, "sha512-RbJ5/jmFcNNCcDV5o9eTnBLJ/HszWV0P73bc+Ff4nS/rJj+YaS6IGyiOL0VoBYX+l1Wrl3k63h/KrH+nhJ0XvQ=="],
"guid-typescript": ["guid-typescript@1.0.9", "", {}, "sha512-Y8T4vYhEfwJOTbouREvG+3XDsjr8E3kIr7uf+JZ0BYloFsttiHU0WfvANVsR7TxNUJa/WpCnw/Ino/p+DeBhBQ=="],
@@ -1135,7 +1146,7 @@
"internmap": ["internmap@2.0.3", "", {}, "sha512-5Hh7Y1wQbvY5ooGgPbDaL5iYLAPzMTUrjMulskHLH6wnv/A+1q5rgEaiuqEjB+oxGXIVZs1FF+R/KPN3ZSQYYg=="],
"is-fullwidth-code-point": ["is-fullwidth-code-point@3.0.0", "", {}, "sha512-zymm5+u+sCsSWyD9qNaejV3DFvhCKclKdizYaJUuHA83RLjb7nSuGnddCHGv0hk+KY7BMAlsWeK4Ueg6EV6XQg=="],
"is-fullwidth-code-point": ["is-fullwidth-code-point@5.1.0", "", { "dependencies": { "get-east-asian-width": "^1.3.1" } }, "sha512-5XHYaSyiqADb4RnZ1Bdad6cPp8Toise4TzEjcOYDHZkTCbKgiUl7WTUCpNWHuxmDt91wnsZBc9xinNzopv3JMQ=="],
"is-what": ["is-what@4.1.16", "", {}, "sha512-ZhMwEosbFJkA0YhFnNDgTM4ZxDRsS6HqTo7qsZM08fehyRYIYa0yHu5R6mgo1n/8MgaPBXiPimPD77baVFYg+A=="],
@@ -1203,6 +1214,8 @@
"minizlib": ["minizlib@2.1.2", "", { "dependencies": { "minipass": "^3.0.0", "yallist": "^4.0.0" } }, "sha512-bAxsR8BVfj60DWXHE3u30oHzfl4G7khkSuPW+qvpd7jFRHm7dLxOjUk1EHACJ/hxLY8phGJ0YhYHZo7jil7Qdg=="],
"mitata": ["mitata@1.0.34", "", {}, "sha512-Mc3zrtNBKIMeHSCQ0XqRLo1vbdIx1wvFV9c8NJAiyho6AjNfMY8bVhbS12bwciUdd1t4rj8099CH3N3NFahaUA=="],
"mitt": ["mitt@3.0.1", "", {}, "sha512-vKivATfr97l2/QBCYAkXYDbrIWPM2IIKEl7YPhjCvKlG3kE2gm+uBo6nEXK3M5/Ffh/FLpKExzOQ3JJoJGFKBw=="],
"mkdirp": ["mkdirp@1.0.4", "", { "bin": { "mkdirp": "bin/cmd.js" } }, "sha512-vVqVZQyf3WLx2Shd0qJ9xuvqgAyKPLAiqITEtqW0oIUjzo3PePDd6fW9iFz30ef7Ysp/oiWqbhszeGWW2T6Gzw=="],
@@ -1311,6 +1324,8 @@
"signal-exit": ["signal-exit@4.1.0", "", {}, "sha512-bzyZ1e88w9O1iNJbKnOlvYTrWPDl46O1bG0D3XInv+9tkPrxrN8jUUTiFlDkkmKWgn1M6CfIA13SuGqOa9Korw=="],
"slice-ansi": ["slice-ansi@9.0.0", "", { "dependencies": { "ansi-styles": "^6.2.3", "is-fullwidth-code-point": "^5.1.0" } }, "sha512-SO/3iYL5S3W57LLEniscOGPZgOqZUPCx6d3dB+52B80yJ0XstzsC/eV8gnA4tM3MHDrKz+OCFSLNjswdSC+/bA=="],
"solid-js": ["solid-js@1.9.14", "", { "dependencies": { "csstype": "^3.1.0", "seroval": "~1.5.4", "seroval-plugins": "~1.5.4" } }, "sha512-sAEXC0Kk0S1EDg+8ysEWJDbYhA3RRoEjwuySUGlKIemeo0I5YZfOyumNjNs9Sv3y2nmhD+0rW66ag2HsMuQiGQ=="],
"solid-refresh": ["solid-refresh@0.6.3", "", { "dependencies": { "@babel/generator": "^7.23.6", "@babel/helper-module-imports": "^7.22.15", "@babel/types": "^7.23.6" }, "peerDependencies": { "solid-js": "^1.3" } }, "sha512-F3aPsX6hVw9ttm5LYlth8Q15x6MlI/J3Dn+o3EQyRTtTxidepSTwAYdozt01/YA+7ObcciagGEyXIopGZzQtbA=="],
@@ -1375,7 +1390,7 @@
"webdriver-bidi-protocol": ["webdriver-bidi-protocol@0.4.2", "", {}, "sha512-VSV+fzfChirL3e7jay2yUC7B4HQCGtEWEg/MSSQbK+qWbqeGlRLlXTzPpYr3XGUvbpDHumWZBJxgesg4N7dbtA=="],
"wrap-ansi": ["wrap-ansi@9.0.2", "", { "dependencies": { "ansi-styles": "^6.2.1", "string-width": "^7.0.0", "strip-ansi": "^7.1.0" } }, "sha512-42AtmgqjV+X1VpdOfyTGOYRi0/zsoLqtXQckTmqTeybT+BDIbM/Guxo7x3pE2vtpr1ok6xRqM9OpBe+Jyoqyww=="],
"wrap-ansi": ["wrap-ansi@10.0.0", "", { "dependencies": { "ansi-styles": "^6.2.3", "string-width": "^8.2.0", "strip-ansi": "^7.1.2" } }, "sha512-SGcvg80f0wUy2/fXES19feHMz8E0JoXv2uNgHOu4Dgi2OrCy1lqwFYEJz1BLbDI0exjPMe/ZdzZ/YpGECBG/aQ=="],
"ws": ["ws@8.21.3", "", { "peerDependencies": { "bufferutil": "^4.0.1", "utf-8-validate": ">=5.0.2" }, "optionalPeers": ["bufferutil", "utf-8-validate"] }, "sha512-201TZ/kPWxoPr/OKWjquZR1SWKXcvxdH+e1xrx89b3YbmzLMFCLfnaG1HFIgWzJOEWZ7MvpK++odZufgYR50Rw=="],
@@ -1469,10 +1484,14 @@
"babel-plugin-jsx-dom-expressions/@babel/helper-module-imports": ["@babel/helper-module-imports@7.18.6", "", { "dependencies": { "@babel/types": "^7.18.6" } }, "sha512-0NFvs3VkuSYbFi1x2Vd6tKrywq+z/cLeYC/RJNFrIX/30Bf5aiGYbtvGXolEktzJH8o5E5KJ3tT+nkxuuZFVlA=="],
"chalk/ansi-styles": ["ansi-styles@4.3.0", "", { "dependencies": { "color-convert": "^2.0.1" } }, "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg=="],
"cli-progress/string-width": ["string-width@4.2.3", "", { "dependencies": { "emoji-regex": "^8.0.0", "is-fullwidth-code-point": "^3.0.0", "strip-ansi": "^6.0.1" } }, "sha512-wKyQRQpjJ0sIp62ErSZdGsjMJWsap5oRNihHhu6G7JVO/9jIB6UyevL+tXuOqrng8j/cxKTWyWUwvSTriiZz/g=="],
"cliui/string-width": ["string-width@7.2.0", "", { "dependencies": { "emoji-regex": "^10.3.0", "get-east-asian-width": "^1.0.0", "strip-ansi": "^7.1.0" } }, "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ=="],
"cliui/wrap-ansi": ["wrap-ansi@9.0.2", "", { "dependencies": { "ansi-styles": "^6.2.1", "string-width": "^7.0.0", "strip-ansi": "^7.1.0" } }, "sha512-42AtmgqjV+X1VpdOfyTGOYRi0/zsoLqtXQckTmqTeybT+BDIbM/Guxo7x3pE2vtpr1ok6xRqM9OpBe+Jyoqyww=="],
"fastembed/onnxruntime-node": ["onnxruntime-node@1.21.0", "", { "dependencies": { "global-agent": "^3.0.0", "onnxruntime-common": "1.21.0", "tar": "^7.0.1" }, "os": [ "linux", "win32", "darwin", ] }, "sha512-NeaCX6WW2L8cRCSqy3bInlo5ojjQqu2fD3D+9W5qb5irwxhEyWKXeH2vZ8W9r6VxaMPUan+4/7NDwZMtouZxEw=="],
"fs-minipass/minipass": ["minipass@3.3.6", "", { "dependencies": { "yallist": "^4.0.0" } }, "sha512-DxiNidxSEK+tHG6zOIklvNOwm3hvCrbUrdtzY74U6HKTJxvIDfOUL5W5P2Ghd3DTkhhKPYGqeNUIh5qcM4YBfw=="],
@@ -1485,10 +1504,6 @@
"vite/lightningcss": ["lightningcss@1.33.0", "", { "dependencies": { "detect-libc": "^2.0.3" }, "optionalDependencies": { "lightningcss-android-arm64": "1.33.0", "lightningcss-darwin-arm64": "1.33.0", "lightningcss-darwin-x64": "1.33.0", "lightningcss-freebsd-x64": "1.33.0", "lightningcss-linux-arm-gnueabihf": "1.33.0", "lightningcss-linux-arm64-gnu": "1.33.0", "lightningcss-linux-arm64-musl": "1.33.0", "lightningcss-linux-x64-gnu": "1.33.0", "lightningcss-linux-x64-musl": "1.33.0", "lightningcss-win32-arm64-msvc": "1.33.0", "lightningcss-win32-x64-msvc": "1.33.0" } }, "sha512-WkUDrojuJs0xkgGf2udWxa3yGBRxPtxUkB79i6aCZLRgc7PM8fZe9TosfPDcvEpQZbuFASnHYmRLBLUbmLOIIA=="],
"wrap-ansi/ansi-styles": ["ansi-styles@6.2.3", "", {}, "sha512-4Dj6M28JB+oAH8kFkTLUo+a2jwOFkuqb3yucU0CANcRRUbxS0cP0nZYCGjcc3BNXwRIsUVmDGgzawme7zvJHvg=="],
"wrap-ansi/string-width": ["string-width@7.2.0", "", { "dependencies": { "emoji-regex": "^10.3.0", "get-east-asian-width": "^1.0.0", "strip-ansi": "^7.1.0" } }, "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ=="],
"@huggingface/transformers/onnxruntime-node/global-agent": ["global-agent@3.0.0", "", { "dependencies": { "boolean": "^3.0.1", "es6-error": "^4.1.1", "matcher": "^3.0.0", "roarr": "^2.15.3", "semver": "^7.3.2", "serialize-error": "^7.0.1" } }, "sha512-PT6XReJ+D07JvGoxQMkT6qji/jVNfX/h364XHZOWeRzy64sSFr+xJ5OX7LI3b4MPQzdL4H8Y8M0xzPpsVMwA8Q=="],
"@huggingface/transformers/onnxruntime-node/onnxruntime-common": ["onnxruntime-common@1.24.3", "", {}, "sha512-GeuPZO6U/LBJXvwdaqHbuUmoXiEdeCjWi/EG7Y1HNnDwJYuk6WUbNXpF6luSUY8yASul3cmUlLGrCCL1ZgVXqA=="],
@@ -1507,6 +1522,8 @@
"@typescript/analyze-trace/yargs/yargs-parser": ["yargs-parser@20.2.9", "", {}, "sha512-y11nGElTIV+CT3Zv9t7VKl+Q3hTQoT9a1Qzezhhl6Rp21gJ/IVTW7Z3y9EWXhuUBC2Shnf+DX0antecpAwSP8w=="],
"cli-progress/string-width/is-fullwidth-code-point": ["is-fullwidth-code-point@3.0.0", "", {}, "sha512-zymm5+u+sCsSWyD9qNaejV3DFvhCKclKdizYaJUuHA83RLjb7nSuGnddCHGv0hk+KY7BMAlsWeK4Ueg6EV6XQg=="],
"cli-progress/string-width/strip-ansi": ["strip-ansi@6.0.1", "", { "dependencies": { "ansi-regex": "^5.0.1" } }, "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A=="],
"cliui/string-width/emoji-regex": ["emoji-regex@10.6.0", "", {}, "sha512-toUI84YS5YmxW219erniWD0CIVOo46xGKColeNQRgOzDorgBi1v4D71/OFzgD9GO2UGKIv1C3Sp8DAn0+j5w7A=="],
@@ -1539,8 +1556,6 @@
"vite/lightningcss/lightningcss-win32-x64-msvc": ["lightningcss-win32-x64-msvc@1.33.0", "", { "os": "win32", "cpu": "x64" }, "sha512-OlEICDx/Xl0FqSp4bry8zFnCvGpig3Gl4gCquvYwHuqJKEC1+n9NgDniFvqHGmMv1ZkqDJrDqKKSykTDX+ehuA=="],
"wrap-ansi/string-width/emoji-regex": ["emoji-regex@10.6.0", "", {}, "sha512-toUI84YS5YmxW219erniWD0CIVOo46xGKColeNQRgOzDorgBi1v4D71/OFzgD9GO2UGKIv1C3Sp8DAn0+j5w7A=="],
"@huggingface/transformers/onnxruntime-node/global-agent/matcher": ["matcher@3.0.0", "", { "dependencies": { "escape-string-regexp": "^4.0.0" } }, "sha512-OkeDaAZ/bQCxeFAozM55PKcKU0yJMPGifLwV4Qgjitu+5MoAfSQN4lsLJeXZ1b8w0x+/Emda6MZgXS1jvsapng=="],
"@huggingface/transformers/onnxruntime-node/global-agent/serialize-error": ["serialize-error@7.0.1", "", { "dependencies": { "type-fest": "^0.13.1" } }, "sha512-8I8TjW5KMOKsZQTvoxjuSIa7foAwPWGOts+6o7sgjz41/qMD9VQHEDxi6PBvK2l0MXUmqZyNpUK+T2tQaaElvw=="],
@@ -1549,6 +1564,8 @@
"@typescript/analyze-trace/yargs/cliui/wrap-ansi": ["wrap-ansi@7.0.0", "", { "dependencies": { "ansi-styles": "^4.0.0", "string-width": "^4.1.0", "strip-ansi": "^6.0.0" } }, "sha512-YVGIj2kamLSTxw6NsZjoBxfSwsn0ycdesmc4p+Q21c5zPuZ1pl+NfxVdxPtdHvmNVOQ6XSYG4AUtyt/Fi7D16Q=="],
"@typescript/analyze-trace/yargs/string-width/is-fullwidth-code-point": ["is-fullwidth-code-point@3.0.0", "", {}, "sha512-zymm5+u+sCsSWyD9qNaejV3DFvhCKclKdizYaJUuHA83RLjb7nSuGnddCHGv0hk+KY7BMAlsWeK4Ueg6EV6XQg=="],
"@typescript/analyze-trace/yargs/string-width/strip-ansi": ["strip-ansi@6.0.1", "", { "dependencies": { "ansi-regex": "^5.0.1" } }, "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A=="],
"cli-progress/string-width/strip-ansi/ansi-regex": ["ansi-regex@5.0.1", "", {}, "sha512-quJQXlTSUGL2LH9SUXo8VwsY4soanhgo6LNSm84E1LBcE8s3O0wpdiRzyR9z/ZZJMlMWv37qOOb9pdJlMUEKFQ=="],
@@ -1569,6 +1586,8 @@
"@typescript/analyze-trace/yargs/cliui/strip-ansi/ansi-regex": ["ansi-regex@5.0.1", "", {}, "sha512-quJQXlTSUGL2LH9SUXo8VwsY4soanhgo6LNSm84E1LBcE8s3O0wpdiRzyR9z/ZZJMlMWv37qOOb9pdJlMUEKFQ=="],
"@typescript/analyze-trace/yargs/cliui/wrap-ansi/ansi-styles": ["ansi-styles@4.3.0", "", { "dependencies": { "color-convert": "^2.0.1" } }, "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg=="],
"@typescript/analyze-trace/yargs/string-width/strip-ansi/ansi-regex": ["ansi-regex@5.0.1", "", {}, "sha512-quJQXlTSUGL2LH9SUXo8VwsY4soanhgo6LNSm84E1LBcE8s3O0wpdiRzyR9z/ZZJMlMWv37qOOb9pdJlMUEKFQ=="],
"fastembed/onnxruntime-node/global-agent/serialize-error/type-fest": ["type-fest@0.13.1", "", {}, "sha512-34R7HTnG0XIJcBSn5XhDd7nNFPRcXYRZrBB2O2jdKqYODldSzBAqzsWoZYYvduky73toYS/ESqxPvkDf/F0XMg=="],
+1 -1
View File
@@ -97,7 +97,7 @@ smallvec.workspace = true
similar = "3.1.0"
tempfile = "3.15.0"
tokio-util = "0.7"
unicode-width = "0.2.0"
xutf.workspace = true
libc = "0.2.172"
memmap2 = "0.9"
uutils_term_grid = "0.8"
+314 -1
View File
@@ -11,6 +11,7 @@ use std::{
io::{self, BufReader, Read, Write},
};
use brush_core::{ShellExtensions, builtins::Registration};
use clap::{Arg, ArgAction, ArgMatches, Command, ValueHint, builder::ValueParser};
use os_display::Quotable;
use uucore::{
@@ -18,6 +19,7 @@ use uucore::{
AlgoKind, BlakeLength, ChecksumError, ReadingMode, ShaLength, SizedAlgoKind,
digest_reader, escape_filename, parse_blake_length, unescape_filename, SUPPORTED_ALGORITHMS,
},
hardware::{HasHardwareFeatures as _, SimdPolicy},
line_ending::LineEnding,
os_str_from_bytes,
quoting_style::{QuotingStyle, locale_aware_escape_name},
@@ -25,7 +27,7 @@ use uucore::{
sum::{self, Blake2b, Blake3, DigestOutput},
};
use crate::host::{Host, os_bytes};
use crate::host::{Host, Utility, matches_parser, os_bytes, util};
#[derive(Debug, Clone)]
struct Failure(String);
@@ -415,6 +417,164 @@ pub(crate) fn run(
}
}
/// Parsed `cksum` invocation.
pub(crate) struct Cksum {
matches: ArgMatches,
}
matches_parser!(Cksum, cksum_app);
impl Utility for Cksum {
const NAME: &'static str = "cksum";
fn run(self, host: &mut Host) -> i32 {
match run_cksum(host, self.matches) {
Ok(()) => host.exit_code(),
Err(error) => {
if !error.0.is_empty() {
host.error(error, 1);
}
1
},
}
}
}
/// Builds the clap command for the GNU `cksum` multi-algorithm front-end.
fn cksum_app() -> Command {
default_checksum_app(
"Print or verify checksums; without --algorithm, prints the POSIX CRC and byte count",
"cksum [OPTION]... [FILE]...",
)
.name("cksum")
.with_algo()
.with_untagged()
.with_tag(true)
.with_length()
.with_raw()
.with_check_and_opts()
.with_base64()
.with_text(false)
.with_binary()
.with_zero()
.with_debug()
}
/// Creates the `cksum` builtin registration.
pub(crate) fn cksum_builtin<SE: ShellExtensions>() -> Registration<SE> {
util::<Cksum, SE>()
}
/// Sanitizes `--length` against `--algorithm`, mirroring GNU `cksum`.
fn sanitize_cksum_length(
host: &mut Host,
algo: Option<AlgoKind>,
input_length: Option<&str>,
) -> ExecResult<Option<usize>> {
match (algo, input_length) {
// No provided length is not a problem so far.
(_, None) => Ok(None),
// For SHA2 and SHA3, if a length is provided, ensure it is correct.
(Some(algo @ (AlgoKind::Sha2 | AlgoKind::Sha3)), Some(len)) => {
// Positive overflow while parsing counts as an invalid number,
// but a number still; it gets the extra reminder of the accepted
// inputs, unlike a plain parse failure.
let parsed = match len.parse::<usize>() {
Ok(parsed) => Some(parsed),
Err(error) if *error.kind() == std::num::IntErrorKind::PosOverflow => None,
Err(_) => return Err(failure(ChecksumError::InvalidLength(len.into()))),
};
match parsed {
Some(parsed @ (224 | 256 | 384 | 512)) => Ok(Some(parsed)),
_ => {
host.error(ChecksumError::InvalidLength(len.into()), 1);
Err(failure(ChecksumError::InvalidLengthForSha(algo.to_uppercase().into())))
},
}
},
// SHAKE128 and SHAKE256 algorithms optionally take a bit length. No
// validation is performed on this length, any value is valid.
(Some(AlgoKind::Shake128 | AlgoKind::Shake256), Some(len)) => match len.parse::<usize>() {
Ok(0) => Ok(None),
Ok(parsed) => Ok(Some(parsed)),
Err(_) => Err(failure(ChecksumError::InvalidLength(len.into()))),
},
// For BLAKE, if a length is provided, validate it.
(Some(algo @ (AlgoKind::Blake2b | AlgoKind::Blake3)), Some(len)) => {
parse_blake_length(algo, BlakeLength::String(len)).map(Some).map_err(failure)
},
// For any other provided algorithm, check if length is 0.
// Otherwise, this is an error.
(_, Some(len)) if len.parse::<u32>() == Ok(0) => Ok(None),
(_, Some(_)) => Err(failure(ChecksumError::LengthOnlyForBlake2bSha2Sha3)),
}
}
/// Prints CPU hardware capability detection info, matching GNU `cksum
/// --debug`.
fn print_cpu_debug_info(host: &mut Host) {
let features = SimdPolicy::detect();
let mut print_feature = |name: &str, available: bool| {
if available {
let _ = writeln!(host.stderr, "using {name} hardware support");
} else {
let _ = writeln!(host.stderr, "{name} support not detected");
}
};
// x86/x86_64
print_feature("avx512", features.has_avx512());
print_feature("avx2", features.has_avx2());
print_feature("pclmul", features.has_pclmul());
// ARM aarch64
if cfg!(target_arch = "aarch64") {
print_feature("vmull", features.has_vmull());
}
}
/// Runs one parsed `cksum` invocation. Unlike the standalone utilities, the
/// algorithm comes from `--algorithm` (default: legacy POSIX CRC), output
/// defaults to tagged, and `--raw`/`--base64` are accepted.
fn run_cksum(host: &mut Host, matches: ArgMatches) -> ExecResult<()> {
let algo = matches
.get_one::<String>(options::ALGORITHM)
.map(AlgoKind::from_cksum)
.transpose()
.map_err(failure)?;
let input_length = matches.get_one::<String>(options::LENGTH).map(String::as_str);
let length = sanitize_cksum_length(host, algo, input_length)?;
let tag = !matches.get_flag(options::UNTAGGED);
let binary = matches.get_flag(options::BINARY);
let text = matches.get_flag(options::TEXT);
// Specifying --text without ever mentioning --untagged fails.
if text && tag {
return Err(failure(ChecksumError::TextWithoutUntagged));
}
let output_format = OutputFormat::from_cksum(
algo.unwrap_or(AlgoKind::Crc),
tag,
binary,
matches.get_flag(options::RAW),
matches.get_flag(options::BASE64),
);
if matches.get_flag(options::DEBUG) {
print_cpu_debug_info(host);
}
checksum_main(host, algo, length, matches, output_format)
}
/// Use the same buffer size as GNU when reading a file to create a checksum
/// from it: 32 KiB.
const READ_BUFFER_SIZE: usize = 32 * 1024;
@@ -1953,5 +2113,158 @@ mod tests {
assert_eq!(&buffer, expected);
}
}
mod cksum_front_end {
//! `cksum` is the GNU multi-algorithm front-end; without these the
//! builtin would shadow the system binary while rejecting or
//! misprinting invocations the real `cksum` accepts.
use std::fs;
use super::super::Cksum;
use crate::host::run_util;
/// Failure mode: default invocation must keep the POSIX CRC format
/// (`<crc> <size>`), not a hex digest.
#[test]
fn default_is_posix_crc_output() {
let (code, capture) = run_util::<Cksum>(&[], "hi", "/");
assert_eq!(code, 0);
assert_eq!(capture.out(), "2352138605 2\n");
}
/// Failure mode: `cksum somefile` printing no filename or the wrong
/// CRC would silently diverge from `/usr/bin/cksum`.
#[test]
fn file_operand_appends_the_filename() {
let dir = tempfile::tempdir().unwrap();
fs::write(dir.path().join("input"), b"hi").unwrap();
let (code, capture) = run_util::<Cksum>(&["input"], "", dir.path());
assert_eq!(code, 0);
assert_eq!(capture.out(), "2352138605 2 input\n");
}
/// Failure mode: `-a sha256` is the flagship GNU extension; it must
/// parse and produce BSD-tagged output by default.
#[test]
fn algorithm_selects_tagged_sha256() {
let (code, capture) = run_util::<Cksum>(&["-a", "sha256"], "hi", "/");
assert_eq!(code, 0);
assert_eq!(
capture.out(),
"SHA256 (-) = 8f434346648f6b96df89dda901c5176b10a6d83961dd3c1ac88b59b2dc327aa4\n"
);
}
/// Failure mode: `--untagged` must switch to the two-space coreutils
/// format so output can be fed back to `sha256sum -c`.
#[test]
fn untagged_prints_coreutils_format() {
let (code, capture) = run_util::<Cksum>(&["-a", "sha256", "--untagged"], "hi", "/");
assert_eq!(code, 0);
assert_eq!(
capture.out(),
"8f434346648f6b96df89dda901c5176b10a6d83961dd3c1ac88b59b2dc327aa4 -\n"
);
}
/// Failure mode: `--base64` must encode the digest, not error or
/// print hex.
#[test]
fn base64_encodes_the_digest() {
let (code, capture) = run_util::<Cksum>(&["-a", "sha256", "--base64"], "hi", "/");
assert_eq!(code, 0);
assert_eq!(capture.out(), "SHA256 (-) = j0NDRmSPa5bfid2pAcUXaxCm2Dlh3TwayItZstwyeqQ=\n");
}
/// Failure mode: `-a blake2b -l N` (the one length-taking algorithm
/// agents use) must honor the bit length in the tag.
#[test]
fn blake2b_length_is_honored() {
let (code, capture) = run_util::<Cksum>(&["-a", "blake2b", "-l", "8"], "abc", "/");
assert_eq!(code, 0);
assert_eq!(capture.out(), "BLAKE2b-8 (-) = 6b\n");
}
/// Failure mode: GNU rejects `--length` for non-length algorithms;
/// silently ignoring it would hide user error.
#[test]
fn length_requires_a_length_algorithm() {
let (code, capture) = run_util::<Cksum>(&["-l", "16"], "", "/");
assert_eq!(code, 1);
assert!(
capture.err().contains("--length is only supported with"),
"{}",
capture.err()
);
}
/// Failure mode: `--text` without `--untagged` is a GNU usage error
/// (`--text mode is only supported with --untagged`), even though the
/// standalone `*sum` utilities accept `-t` freely.
#[test]
fn text_without_untagged_is_rejected() {
let (code, capture) = run_util::<Cksum>(&["-a", "sha256", "-t"], "", "/");
assert_eq!(code, 1);
assert!(
capture.err().contains("--text mode is only supported with --untagged"),
"{}",
capture.err()
);
}
/// Failure mode: verification must accept tagged lines produced by
/// the compute side and exit 0.
#[test]
fn check_verifies_tagged_lines() {
let dir = tempfile::tempdir().unwrap();
fs::write(dir.path().join("data"), b"hi").unwrap();
fs::write(
dir.path().join("list"),
b"SHA256 (data) = 8f434346648f6b96df89dda901c5176b10a6d83961dd3c1ac88b59b2dc327aa4\n",
)
.unwrap();
let (code, capture) = run_util::<Cksum>(&["-c", "list"], "", dir.path());
assert_eq!(code, 0, "stderr: {}", capture.err());
assert_eq!(capture.out(), "data: OK\n");
}
/// Failure mode: real GNU 9.x treats `-c -b` as a fatal usage error
/// ("meaningless when verifying checksums", exit 1, nothing
/// verified); the builtin must not silently verify anyway.
#[test]
fn check_with_binary_stays_fatal_like_gnu() {
let dir = tempfile::tempdir().unwrap();
fs::write(dir.path().join("data"), b"hi").unwrap();
fs::write(
dir.path().join("list"),
b"SHA256 (data) = 8f434346648f6b96df89dda901c5176b10a6d83961dd3c1ac88b59b2dc327aa4\n",
)
.unwrap();
let (code, capture) = run_util::<Cksum>(&["-c", "-b", "list"], "", dir.path());
assert_eq!(code, 1);
assert!(
capture
.err()
.contains("the --binary and --text options are meaningless when verifying checksums"),
"{}",
capture.err()
);
assert!(!capture.out().contains("OK"), "must not verify: {}", capture.out());
}
/// Failure mode: legacy algorithms cannot be verified; GNU errors out
/// rather than parsing the list.
#[test]
fn check_rejects_legacy_algorithms() {
let (code, capture) = run_util::<Cksum>(&["-a", "crc", "-c"], "", "/");
assert_eq!(code, 1);
assert!(
capture.err().contains("--check is not supported with --algorithm"),
"{}",
capture.err()
);
}
}
}
+546 -16
View File
@@ -764,7 +764,7 @@ use std::{
ffi::OsString,
fs::File,
io::{BufRead, BufReader, BufWriter, Read, Write},
path::PathBuf,
path::{Path, PathBuf},
sync::LazyLock,
};
#[cfg(any(
@@ -780,7 +780,7 @@ use std::ffi::{CStr, CString};
use clap::{Arg, ArgAction, ArgMatches, Command};
use jiff::{
Timestamp, Zoned,
Span, Timestamp, Zoned,
fmt::strtime::{self, BrokenDownTime, Config, PosixCustom},
tz::{Offset, TimeZone, TimeZoneDatabase},
};
@@ -807,6 +807,9 @@ const OPT_SET: &str = "set";
const OPT_REFERENCE: &str = "reference";
const OPT_UNIVERSAL: &str = "universal";
const OPT_UNIVERSAL_2: &str = "utc";
// BSD compatibility options (no GNU equivalents).
const OPT_BSD_ADJUST: &str = "bsd-adjust";
const OPT_BSD_PARSE_ONLY: &str = "bsd-parse-only";
/// Settings for this program, parsed from the command line
struct Settings {
@@ -849,6 +852,8 @@ enum DateSource {
FileMtime(PathBuf),
Stdin,
Human(String),
/// BSD `date -j -f FMT VALUE`: VALUE parsed with the strptime format FMT.
Strptime { format: String, value: String },
Resolution,
}
@@ -1035,6 +1040,247 @@ fn parse_military_timezone_with_offset(s: &str) -> Option<(i32, DayDelta)> {
Some((hours_from_midnight, day_delta))
}
/// Rewrite `-I`/`--iso-8601` so the optional ISO precision only binds when it
/// is attached (`-Ihours`, `--iso-8601=hours`), matching GNU getopt. Without
/// this, clap's optional-value handling greedily consumes a following
/// `+FORMAT` operand (`date -I +%s`).
///
/// `argv[0]` is the command name. Scanning stops at `--`, and a token that is
/// the value of a preceding option is never rewritten.
fn rewrite_date_argv(argv: Vec<OsString>) -> Vec<OsString> {
/// Long options that consume a separate value token.
const VALUE_LONGS: &[&str] = &["date", "file", "reference", "set", "rfc-3339"];
/// Long flags that take no value (`iso-8601` is handled separately).
const FLAG_LONGS: &[&str] =
&["debug", "resolution", "rfc-email", "rfc-2822", "rfc-822", "universal", "utc", "uct"];
/// Short options that consume a value, attached or separate.
const VALUE_SHORTS: &[char] = &['d', 'f', 'r', 's', 'v'];
let mut out = Vec::with_capacity(argv.len());
let mut argv = argv.into_iter();
if let Some(name) = argv.next() {
out.push(name);
}
let mut skip_value = false;
let mut opts_ended = false;
for arg in argv {
if skip_value || opts_ended {
skip_value = false;
out.push(arg);
continue;
}
let Some(token) = arg.to_str() else {
out.push(arg);
continue;
};
if token == "--" {
opts_ended = true;
out.push(arg);
} else if let Some(long) = token.strip_prefix("--") {
// `infer_long_args` is enabled, so any unambiguous prefix names
// the option.
let (name, value) = match long.split_once('=') {
Some((name, value)) => (name, Some(value)),
None => (long, None),
};
let ambiguous = VALUE_LONGS
.iter()
.chain(FLAG_LONGS)
.any(|other| other.starts_with(name));
if !name.is_empty() && "iso-8601".starts_with(name) && !ambiguous {
out.push(format!("--iso-8601={}", value.unwrap_or(DATE)).into());
} else {
if value.is_none()
&& VALUE_LONGS.iter().filter(|l| l.starts_with(name)).count() == 1
&& !FLAG_LONGS.iter().any(|l| l.starts_with(name))
&& !"iso-8601".starts_with(name)
{
skip_value = true;
}
out.push(arg);
}
} else if let Some(cluster) = token.strip_prefix('-').filter(|rest| !rest.is_empty()) {
// Walk a short-option cluster (`-uI`, `-ud @0`, `-Ihours`).
let mut rewrote = false;
for (index, ch) in cluster.char_indices() {
if ch == 'I' {
let flags = &cluster[..index];
let rest = &cluster[index + ch.len_utf8()..];
if !flags.is_empty() {
out.push(format!("-{flags}").into());
}
let spec = if rest.is_empty() { DATE } else { rest };
out.push(format!("--iso-8601={spec}").into());
rewrote = true;
break;
}
if VALUE_SHORTS.contains(&ch) {
// The remainder of the token (or the next token when the
// remainder is empty) is this option's value.
if cluster[index + ch.len_utf8()..].is_empty() {
skip_value = true;
}
break;
}
}
if !rewrote {
out.push(arg);
}
} else {
out.push(arg);
}
}
out
}
/// Unit letter of a BSD `-v` adjustment.
#[derive(Clone, Copy)]
enum BsdAdjustUnit {
Year,
Month,
Week,
Day,
Hour,
Minute,
Second,
}
/// One BSD `-v` adjustment: a relative offset (`+1d`, `-2m`) or an absolute
/// field set (`1d` sets the day of the month).
#[derive(Clone, Copy)]
enum BsdAdjustment {
Offset(i64, BsdAdjustUnit),
Set(i64, BsdAdjustUnit),
}
/// Parse one BSD `-v` argument of the form `[+|-]VAL[ymwdHMS]`.
///
/// BSD is case-sensitive only where it is ambiguous (`m` month vs `M`
/// minute); the unambiguous letters are accepted in either case. Weekday and
/// month names (`-vsun`, `-vjan`) are not supported.
fn parse_bsd_adjustment(spec: &str) -> Option<BsdAdjustment> {
let (sign, rest) = match *spec.as_bytes().first()? {
b'+' => (Some(1), &spec[1..]),
b'-' => (Some(-1), &spec[1..]),
_ => (None, spec),
};
let mut chars = rest.chars();
let unit = match chars.next_back()? {
'y' | 'Y' => BsdAdjustUnit::Year,
'm' => BsdAdjustUnit::Month,
'w' | 'W' => BsdAdjustUnit::Week,
'd' | 'D' => BsdAdjustUnit::Day,
'H' | 'h' => BsdAdjustUnit::Hour,
'M' => BsdAdjustUnit::Minute,
'S' | 's' => BsdAdjustUnit::Second,
_ => return None,
};
let digits = chars.as_str();
if digits.is_empty() || !digits.bytes().all(|b| b.is_ascii_digit()) {
return None;
}
let value: i64 = digits.parse().ok()?;
Some(match sign {
Some(sign) => BsdAdjustment::Offset(sign * value, unit),
None => BsdAdjustment::Set(value, unit),
})
}
/// Apply BSD `-v` adjustments to `date` in command-line order.
fn apply_bsd_adjustments(mut date: Zoned, adjustments: &[BsdAdjustment]) -> Result<Zoned, String> {
for adjustment in adjustments {
date = match *adjustment {
BsdAdjustment::Offset(value, unit) => {
let span = Span::new();
let span = match unit {
BsdAdjustUnit::Year => span.try_years(value),
BsdAdjustUnit::Month => span.try_months(value),
BsdAdjustUnit::Week => span.try_weeks(value),
BsdAdjustUnit::Day => span.try_days(value),
BsdAdjustUnit::Hour => span.try_hours(value),
BsdAdjustUnit::Minute => span.try_minutes(value),
BsdAdjustUnit::Second => span.try_seconds(value),
}
.map_err(|error| format!("invalid adjustment ({error})"))?;
date
.checked_add(span)
.map_err(|error| format!("cannot adjust date ({error})"))?
},
BsdAdjustment::Set(value, unit) => {
let narrow = |unit: char| {
i8::try_from(value).map_err(|_| format!("invalid adjustment: '{value}{unit}'"))
};
let with = date.with();
let with = match unit {
BsdAdjustUnit::Year => {
// BSD windows two-digit years: 69-99 => 19xx, 0-68 => 20xx.
let year = match value {
0..=68 => value + 2000,
69..=99 => value + 1900,
_ => value,
};
let year = i16::try_from(year)
.map_err(|_| format!("invalid adjustment: '{value}y'"))?;
with.year(year)
},
BsdAdjustUnit::Month => with.month(narrow('m')?),
BsdAdjustUnit::Week => {
return Err(format!(
"unsupported adjustment: '{value}w' (setting the week is not \
implemented; use an offset like '+{value}w')"
));
},
BsdAdjustUnit::Day => with.day(narrow('d')?),
BsdAdjustUnit::Hour => with.hour(narrow('H')?),
BsdAdjustUnit::Minute => with.minute(narrow('M')?),
BsdAdjustUnit::Second => with.second(narrow('S')?),
};
with
.build()
.map_err(|error| format!("cannot adjust date ({error})"))?
},
};
}
Ok(date)
}
/// BSD `date -j -f FMT VALUE`: parse VALUE with the strptime format FMT.
///
/// Fields the format does not mention keep the current date/time's values,
/// matching BSD `date`, which seeds the broken-down time from
/// `localtime(now)` before calling strptime(3).
fn parse_bsd_strptime(format: &str, value: &str, now: &Zoned) -> Result<Zoned, String> {
let convert_error =
|error: jiff::Error| format!("failed conversion of '{value}' using format '{format}' ({error})");
let broken = BrokenDownTime::parse(format, value).map_err(convert_error)?;
// `%s` (or a complete civil datetime plus an offset) pins an instant.
if let Ok(timestamp) = broken.to_timestamp() {
return Ok(timestamp.to_zoned(now.time_zone().clone()));
}
let base = now.datetime();
let date = broken
.to_date()
.or_else(|_| {
jiff::civil::Date::new(
broken.year().unwrap_or(base.year()),
broken.month().unwrap_or(base.month()),
broken.day().unwrap_or(base.day()),
)
})
.map_err(convert_error)?;
let time = jiff::civil::Time::new(
broken.hour().unwrap_or(base.hour()),
broken.minute().unwrap_or(base.minute()),
broken.second().unwrap_or(base.second()),
broken.subsec_nanosecond().unwrap_or(base.subsec_nanosecond()),
)
.map_err(convert_error)?;
date
.to_datetime(time)
.to_zoned(now.time_zone().clone())
.map_err(convert_error)
}
/// Parsed `date` invocation.
pub(crate) struct Date {
@@ -1064,6 +1310,10 @@ impl From<std::io::Error> for DateError {
impl Utility for Date {
const NAME: &'static str = "date";
fn rewrite_argv(argv: Vec<OsString>) -> Result<Vec<OsString>, String> {
Ok(rewrite_date_argv(argv))
}
fn run(self, host: &mut Host) -> i32 {
match date_main(host, &self.matches) {
Ok(()) => host.exit_code(),
@@ -1152,7 +1402,44 @@ fn locale_default_format(_locale: &str) -> Option<String> {
#[allow(clippy::cognitive_complexity)]
fn date_main(host: &mut Host, matches: &ArgMatches) -> Result<(), DateError> {
let date_source = if let Some(date_os) = matches.get_one::<OsString>(OPT_DATE) {
let bsd_parse_only = matches.get_flag(OPT_BSD_PARSE_ONLY);
let adjustments: Vec<BsdAdjustment> = matches
.get_many::<String>(OPT_BSD_ADJUST)
.into_iter()
.flatten()
.map(|spec| {
parse_bsd_adjustment(spec)
.ok_or_else(|| DateError::new(1, format!("invalid adjustment: '{spec}'")))
})
.collect::<Result<_, _>>()?;
// Positional operands: at most one `+FORMAT`, plus (in the BSD `-j -f`
// form) the date value to parse.
let mut operands: Vec<&String> = matches
.get_many::<String>(OPT_FORMAT)
.map(Iterator::collect)
.unwrap_or_default();
// BSD: with `-j`, `-f` is the strptime(3) input format for the date
// operand rather than GNU's `--file=DATEFILE`.
let strptime_format = if bsd_parse_only {
matches.get_one::<String>(OPT_FILE)
} else {
None
};
let strptime_value = if strptime_format.is_some() {
if operands.first().is_some_and(|operand| !operand.starts_with('+')) {
Some(operands.remove(0))
} else {
return Err(DateError::new(1, "'-j -f FORMAT' requires a date operand to parse"));
}
} else {
None
};
let date_source = if let (Some(format), Some(value)) = (strptime_format, strptime_value) {
DateSource::Strptime { format: format.clone(), value: value.clone() }
} else if let Some(date_os) = matches.get_one::<OsString>(OPT_DATE) {
// Convert OsString to String, handling invalid UTF-8 with GNU-compatible error
let date = date_os.to_str().ok_or_else(|| {
let bytes = date_os.as_encoded_bytes();
@@ -1165,8 +1452,18 @@ fn date_main(host: &mut Host, matches: &ArgMatches) -> Result<(), DateError> {
"-" => DateSource::Stdin,
_ => DateSource::File(file.into()),
}
} else if let Some(file) = matches.get_one::<String>(OPT_REFERENCE) {
DateSource::FileMtime(file.into())
} else if let Some(reference) = matches.get_one::<String>(OPT_REFERENCE) {
// `-r` doubles as GNU `--reference=FILE` and BSD `-r SECONDS`.
// Precedence: an existing file always wins (GNU semantics are
// primary); a purely numeric operand naming no existing file is
// seconds since the epoch (BSD), i.e. GNU `-d @SECONDS`.
let digits = reference.strip_prefix('-').unwrap_or(reference);
let numeric = !digits.is_empty() && digits.bytes().all(|b| b.is_ascii_digit());
if numeric && !host.resolve(Path::new(reference)).exists() {
DateSource::Human(format!("@{reference}"))
} else {
DateSource::FileMtime(reference.into())
}
} else if matches.get_flag(OPT_RESOLUTION) {
DateSource::Resolution
} else {
@@ -1174,18 +1471,15 @@ fn date_main(host: &mut Host, matches: &ArgMatches) -> Result<(), DateError> {
};
// Check for extra operands (multiple positional arguments)
if let Some(formats) = matches.get_many::<String>(OPT_FORMAT) {
let format_args: Vec<&String> = formats.collect();
if format_args.len() > 1 {
return Err(DateError::new(1, format!("extra operand '{}'", format_args[1])));
}
if operands.len() > 1 {
return Err(DateError::new(1, format!("extra operand '{}'", operands[1])));
}
let format = if let Some(form) = matches.get_one::<String>(OPT_FORMAT) {
let format = if let Some(form) = operands.first() {
if !form.starts_with('+') {
// if an optional Format String was found but the user has not provided an input
// date GNU prints an invalid date Error
if !matches!(date_source, DateSource::Human(_)) {
if !matches!(date_source, DateSource::Human(_) | DateSource::Strptime { .. }) {
return Err(DateError::new(1, format!("invalid date '{form}'")));
}
// If the user did provide an input date with the --date flag and the Format
@@ -1198,8 +1492,7 @@ fn date_main(host: &mut Host, matches: &ArgMatches) -> Result<(), DateError> {
),
));
}
let form = form[1..].to_string();
Format::Custom(form)
Format::Custom(form[1..].to_string())
} else if let Some(fmt) = matches
.get_many::<String>(OPT_ISO_8601)
.map(|mut iter| iter.next().unwrap_or(&DATE.to_string()).as_str().into())
@@ -1427,6 +1720,11 @@ fn date_main(host: &mut Host, matches: &ArgMatches) -> Result<(), DateError> {
let iter = std::iter::once(Ok(date));
Box::new(iter)
},
DateSource::Strptime { format, value } => {
let date = parse_bsd_strptime(format, value, &now)
.map_err(|message| DateError::new(1, message))?;
Box::new(std::iter::once(Ok(date)))
},
DateSource::Now => {
let iter = std::iter::once(Ok(now.clone()));
Box::new(iter)
@@ -1445,6 +1743,18 @@ fn date_main(host: &mut Host, matches: &ArgMatches) -> Result<(), DateError> {
}
match date {
Ok(date) => {
// BSD `-v` adjustments apply to the base date in argv order.
let date = if adjustments.is_empty() {
date
} else {
match apply_bsd_adjustments(date, &adjustments) {
Ok(date) => date,
Err(message) => {
let _ = stdout.flush();
return Err(DateError::new(1, message));
},
}
};
let date = if settings.utc {
date.with_time_zone(TimeZone::UTC)
} else {
@@ -1509,7 +1819,10 @@ fn uu_app() -> Command {
.value_name("DATEFILE")
.value_hint(clap::ValueHint::FilePath)
.conflicts_with(OPT_DATE)
.help("like --date; once for each line of DATEFILE"),
.help(
"like --date; once for each line of DATEFILE\n(BSD: with -j, the strptime(3) \
input format for the date operand)",
),
)
.arg(
Arg::new(OPT_ISO_8601)
@@ -1517,7 +1830,12 @@ fn uu_app() -> Command {
.long(OPT_ISO_8601)
.value_name("FMT")
.value_parser(ShortcutValueParser::new([DATE, HOURS, MINUTES, SECONDS, NS]))
// The optional precision binds only when attached (`-Ihours`,
// `--iso-8601=hours`): `rewrite_date_argv` normalizes every
// spelling to the `=` form, so a following `+FORMAT` operand
// is never consumed as the value (GNU getopt behavior).
.num_args(0..=1)
.require_equals(true)
.default_missing_value(OPT_DATE)
.help(
"output date/time in ISO 8601 format.\nFMT='date' for date only (the \
@@ -1567,8 +1885,12 @@ fn uu_app() -> Command {
.long(OPT_REFERENCE)
.value_name("FILE")
.value_hint(clap::ValueHint::AnyPath)
.allow_hyphen_values(true)
.conflicts_with_all([OPT_DATE, OPT_FILE, OPT_RESOLUTION])
.help("display the last modification time of FILE"),
.help(
"display the last modification time of FILE\n(BSD: when FILE is numeric and \
no such file exists,\ndisplay the date at that many seconds since the epoch)",
),
)
.arg(
Arg::new(OPT_SET)
@@ -1588,6 +1910,26 @@ fn uu_app() -> Command {
.help("print or set Coordinated Universal Time (UTC)")
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_BSD_PARSE_ONLY)
.short('j')
.help(
"BSD compatibility: do not try to set the system clock;\nwith -f, parse the \
date operand using the strptime(3)\nformat given to -f",
)
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(OPT_BSD_ADJUST)
.short('v')
.value_name("[+|-]VAL[ymwdHMS]")
.allow_hyphen_values(true)
.action(ArgAction::Append)
.help(
"BSD compatibility: adjust ('+'/'-') or set (no sign) the\ndisplayed date; \
may be given multiple times, applied in order",
),
)
.arg(Arg::new(OPT_FORMAT).num_args(0..))
}
@@ -1982,6 +2324,194 @@ mod tests {
assert_eq!(strip_parenthesized_comments("a(b(c)d)e"), "ae");
assert_eq!(strip_parenthesized_comments("a(b)c(d"), "ac");
}
/// Defends: BSD `-r <epoch>` must print that instant, not fail trying to
/// open `./<epoch>` as a reference file.
#[test]
fn bsd_reference_epoch_when_no_such_file() {
let (code, capture) =
run_util::<Date>(&["-u", "-r", "1700000000", "+%Y-%m-%dT%H:%M:%S"], "", "/");
assert_eq!(code, 0);
assert_eq!(capture.out(), "2023-11-14T22:13:20\n");
assert_eq!(capture.err(), "");
}
/// Defends: an existing file always wins over the BSD numeric-epoch
/// reading of `-r` (GNU `--reference` semantics are primary).
#[test]
fn reference_prefers_existing_file_over_epoch() {
let dir = std::env::temp_dir().join(format!("pi-date-r-{}", std::process::id()));
std::fs::create_dir_all(&dir).unwrap();
let file = dir.join("1700000000");
std::fs::write(&file, b"x").unwrap();
let (code, capture) = run_util::<Date>(
&["-u", "-r", "1700000000", "+%s"],
"",
dir.to_str().unwrap(),
);
let mtime = std::fs::metadata(&file).unwrap().modified().unwrap();
let expected = Timestamp::try_from(mtime).unwrap().as_second();
std::fs::remove_dir_all(&dir).unwrap();
assert_eq!(code, 0);
assert_eq!(capture.out(), format!("{expected}\n"));
}
/// Defends: BSD `-v` offsets (`+1d`, `-2m`) must parse and apply in argv
/// order instead of being rejected as unknown options.
#[test]
fn bsd_adjustments_apply_in_order() {
let (code, capture) = run_util::<Date>(
&["-u", "-r", "1700000000", "-v+1d", "-v-2m", "+%F %T"],
"",
"/",
);
assert_eq!(code, 0);
assert_eq!(capture.out(), "2023-09-15 22:13:20\n");
assert_eq!(capture.err(), "");
}
/// Defends: the unsigned `-v` form sets a field absolutely (`-v1d` = first
/// of the month) rather than offsetting.
#[test]
fn bsd_adjustment_sets_fields_absolutely() {
let (code, capture) = run_util::<Date>(
&["-u", "-r", "1700000000", "-v1d", "-v5H", "+%F %T"],
"",
"/",
);
assert_eq!(code, 0);
assert_eq!(capture.out(), "2023-11-01 05:13:20\n");
assert_eq!(capture.err(), "");
}
/// Defends: a malformed `-v` argument is a clean diagnostic, not a panic
/// or a silently ignored adjustment.
#[test]
fn bsd_adjustment_rejects_unknown_unit() {
let (code, capture) = run_util::<Date>(&["-v+1x", "+%F"], "", "/");
assert_eq!(code, 1);
assert!(capture.err().contains("invalid adjustment"), "stderr: {}", capture.err());
}
/// Defends: with `-j`, `-f` is the BSD strptime input format for the date
/// operand, not GNU `--file=DATEFILE`.
#[test]
fn bsd_j_f_parses_with_strptime_format() {
let (code, capture) = run_util::<Date>(
&["-u", "-j", "-f", "%Y-%m-%d %H:%M:%S", "2026-01-01 00:00:00", "+%s"],
"",
"/",
);
assert_eq!(code, 0);
assert_eq!(capture.out(), "1767225600\n");
assert_eq!(capture.err(), "");
}
/// Defends: fields missing from the `-j -f` format are seeded from "now"
/// (BSD strptime semantics), so a date-only format keeps the given date.
#[test]
fn bsd_j_f_fills_missing_fields_from_now() {
let (code, capture) =
run_util::<Date>(&["-u", "-j", "-f", "%Y-%m-%d", "2026-01-01", "+%F"], "", "/");
assert_eq!(code, 0);
assert_eq!(capture.out(), "2026-01-01\n");
assert_eq!(capture.err(), "");
}
/// Defends: `-j -f` without a date operand is a diagnostic, not a silent
/// fallback to GNU `--file` behavior.
#[test]
fn bsd_j_f_requires_date_operand() {
let (code, capture) = run_util::<Date>(&["-j", "-f", "%Y", "+%F"], "", "/");
assert_eq!(code, 1);
assert!(
capture.err().contains("requires a date operand"),
"stderr: {}",
capture.err()
);
}
/// Defends: bare `-j` parses as a no-op (never sets the clock) instead of
/// being rejected, and `-f` keeps GNU file semantics without `-j`.
#[test]
fn bsd_j_alone_is_a_no_op() {
let (code, capture) = run_util::<Date>(&["-u", "-j", "-r", "0", "+%F"], "", "/");
assert_eq!(code, 0);
assert_eq!(capture.out(), "1970-01-01\n");
assert_eq!(capture.err(), "");
}
/// Defends: `date -I +%s` must treat `+%s` as the output format instead of
/// greedily consuming it as the ISO precision value.
#[test]
fn iso_flag_does_not_consume_format_operand() {
let (code, capture) = run_util::<Date>(&["-u", "-d", "@0", "-I", "+%s"], "", "/");
assert_eq!(code, 0);
assert_eq!(capture.out(), "0\n");
assert_eq!(capture.err(), "");
}
/// Defends: bare `-I` still defaults to date precision after the
/// non-greedy rewrite.
#[test]
fn iso_flag_defaults_to_date_precision() {
let (code, capture) = run_util::<Date>(&["-u", "-d", "@0", "-I"], "", "/");
assert_eq!(code, 0);
assert_eq!(capture.out(), "1970-01-01\n");
assert_eq!(capture.err(), "");
}
/// Defends: the attached form `-Ihours` keeps binding the precision.
#[test]
fn iso_flag_accepts_attached_precision() {
let (code, capture) = run_util::<Date>(&["-u", "-d", "@0", "-Ihours"], "", "/");
assert_eq!(code, 0);
assert_eq!(capture.out(), "1970-01-01T00+00:00\n");
assert_eq!(capture.err(), "");
}
/// Defends: the argv rewrite only touches genuine `-I`/`--iso-8601`
/// options — never option values, post-`--` operands, or other flags.
#[test]
fn rewrites_iso_argv_forms_conservatively() {
let argv = |args: &[&str]| -> Vec<OsString> { args.iter().map(OsString::from).collect() };
assert_eq!(
rewrite_date_argv(argv(&["date", "-I", "+%s"])),
argv(&["date", "--iso-8601=date", "+%s"])
);
assert_eq!(rewrite_date_argv(argv(&["date", "-Ihours"])), argv(&["date", "--iso-8601=hours"]));
assert_eq!(
rewrite_date_argv(argv(&["date", "--iso-8601", "+%s"])),
argv(&["date", "--iso-8601=date", "+%s"])
);
assert_eq!(
rewrite_date_argv(argv(&["date", "--iso", "-u"])),
argv(&["date", "--iso-8601=date", "-u"])
);
// `-I` as the value of another option is untouched.
assert_eq!(rewrite_date_argv(argv(&["date", "-d", "-I"])), argv(&["date", "-d", "-I"]));
// Everything after `--` is an operand.
assert_eq!(rewrite_date_argv(argv(&["date", "--", "-I"])), argv(&["date", "--", "-I"]));
// Clustered flags before `-I` are preserved.
assert_eq!(
rewrite_date_argv(argv(&["date", "-uI"])),
argv(&["date", "-u", "--iso-8601=date"])
);
}
}
#[cfg(not(any(unix, windows)))]
File diff suppressed because it is too large Load Diff
+47 -5
View File
@@ -168,6 +168,7 @@ fn command() -> Command {
Arg::new(ARG_QUERY)
.value_name("NAME-OR-NUMBER")
.num_args(0..)
.allow_hyphen_values(true)
.action(ArgAction::Append),
)
}
@@ -189,13 +190,21 @@ fn print_entry(host: &mut Host, name: &str, number: i32) {
/// Looks up one name or number argument; returns false on failure.
fn lookup(host: &mut Host, arg: &str) -> bool {
if let Ok(number) = arg.parse::<i32>() {
// Reverse lookup: first-listed name for the number is canonical.
match ERRNOS.iter().find(|(_, value)| *value == number) {
// Kernel convention returns errors as negative errno values; resolve
// `-2` the same as `2`. Reverse lookup: first-listed name for the
// number is canonical.
match number
.checked_abs()
.and_then(|number| ERRNOS.iter().find(|(_, value)| *value == number))
{
Some((name, value)) => {
print_entry(host, name, *value);
true
},
None => false,
None => {
let _ = writeln!(host.stderr, "errno: unknown errno {arg}");
false
},
}
} else if let Some((name, value)) = ERRNOS
.iter()
@@ -277,11 +286,44 @@ mod tests {
}
#[test]
fn unknown_number_fails_silently() {
fn unknown_number_fails_with_stderr() {
// Failure mode: unknown numeric lookups exiting 1 with no diagnostic.
let (code, stdout, stderr) = run_errno(&["99999"]);
assert_eq!(code, 1);
assert!(stdout.is_empty());
assert!(stderr.is_empty());
assert!(stderr.contains("unknown errno 99999"), "stderr: {stderr:?}");
}
#[test]
fn negative_number_resolves_by_absolute_value() {
// Failure mode: clap rejecting `errno -2` as an unknown option instead
// of resolving the kernel-style negative errno.
let number = format!("-{}", libc::ENOENT);
let (code, stdout, stderr) = run_errno(&[&number]);
assert_eq!(code, 0);
assert!(stdout.starts_with(&format!("ENOENT {} ", libc::ENOENT)), "stdout: {stdout:?}");
assert!(stderr.is_empty(), "stderr: {stderr:?}");
}
#[test]
fn unknown_negative_number_fails_with_stderr() {
let (code, stdout, stderr) = run_errno(&["-99999"]);
assert_eq!(code, 1);
assert!(stdout.is_empty());
assert!(stderr.contains("unknown errno -99999"), "stderr: {stderr:?}");
}
#[test]
fn short_flags_still_win_over_hyphen_operands() {
// Failure mode: allow_hyphen_values swallowing `-l` as a query.
let (code, stdout, _) = run_errno(&["-l"]);
assert_eq!(code, 0);
assert!(
stdout
.lines()
.any(|line| line.starts_with(&format!("ENOENT {} ", libc::ENOENT))),
"-l no longer lists: {stdout:?}"
);
}
#[test]
+2
View File
@@ -204,6 +204,8 @@ pub fn utility_builtins<SE: brush_core::ShellExtensions>()
m.push(("basename", basename::basename_builtin::<SE>()));
#[cfg(feature = "util.cat")]
m.push(("cat", cat::cat_builtin::<SE>()));
#[cfg(feature = "util.cksum")]
m.push(("cksum", cksum::cksum_builtin::<SE>()));
#[cfg(feature = "util.cmp")]
m.push(("cmp", cmp::cmp_builtin::<SE>()));
#[cfg(feature = "util.comm")]
+340 -45
View File
@@ -1847,7 +1847,9 @@ pub mod matchers {
match chars.next() {
Some('-') => (ComparisonType::AtLeast, chars.as_str()),
Some('/') => (ComparisonType::AnyOf, chars.as_str()),
// GNU spells "any of these bits" as /mode; BSD find spells
// it +mode. Accept both.
Some('/') | Some('+') => (ComparisonType::AnyOf, chars.as_str()),
_ => (ComparisonType::Exact, pattern),
}
}
@@ -1882,7 +1884,7 @@ pub mod matchers {
pub fn new(pattern: &str) -> Result<Self, Box<dyn Error>> {
let (comparison_type, pattern) = parsing::split_comparison_type(pattern);
let file_pattern = parsing::parse_mode(pattern, false)?;
let dir_pattern = parsing::parse_mode(pattern, false)?;
let dir_pattern = parsing::parse_mode(pattern, true)?;
Ok(Self { comparison_type, file_pattern, dir_pattern })
}
@@ -2686,7 +2688,7 @@ pub mod matchers {
use std::{error::Error, fmt, str::FromStr};
use onig::{Regex, RegexOptions, Syntax};
use onig::{Regex, RegexOptions, SearchOptions, Syntax};
use super::{Matcher, MatcherIO, WalkEntry};
@@ -2783,9 +2785,15 @@ pub mod matchers {
impl Matcher for RegexMatcher {
fn matches(&self, file_info: &WalkEntry, _: &mut MatcherIO) -> bool {
let path = file_info.display_path().to_string_lossy();
// `-regex` must match the WHOLE path (POSIX/GNU/BSD), not a
// substring: anchor the match at the start of the path and
// require it to end at the end of the path (backtracking
// retries alternatives that stop short).
self
.regex
.is_match(file_info.display_path().to_string_lossy().as_ref())
.match_with_options(path.as_ref(), 0, SearchOptions::SEARCH_OPTION_WHOLE_STRING, None)
.is_some()
}
}
}
@@ -2859,6 +2867,8 @@ pub mod matchers {
KibiByte,
MebiByte,
GibiByte,
TebiByte,
PebiByte,
}
impl FromStr for Unit {
@@ -2872,10 +2882,12 @@ pub mod matchers {
"k" => Self::KibiByte,
"M" => Self::MebiByte,
"G" => Self::GibiByte,
"T" => Self::TebiByte,
"P" => Self::PebiByte,
_ => {
return Err(From::from(format!(
"Invalid suffix {s} for -size. Only allowed values are <nothing>, b, c, w, k, M or \
G"
"Invalid suffix {s} for -size. Only allowed values are <nothing>, b, c, w, k, M, G, \
T or P"
)));
},
})
@@ -2894,6 +2906,8 @@ pub mod matchers {
Unit::KibiByte => 10,
Unit::MebiByte => 20,
Unit::GibiByte => 30,
Unit::TebiByte => 40,
Unit::PebiByte => 50,
};
// Skip pointless arithmetic.
if bits_to_shift == 0 {
@@ -3111,13 +3125,12 @@ pub mod matchers {
}
}
/// This matcher checks whether the file is newer than the file time of any
/// combination of two comparison types from the target file's
/// `NewerOptionType`.
/// This matcher checks whether the X timestamp of the file being
/// considered is newer than the Y timestamp of the reference file,
/// captured once when the matcher is built (`-newerXY reference`).
pub struct NewerOptionMatcher {
x_option: NewerOptionType,
y_option: NewerOptionType,
given_modification_time: SystemTime,
reference_time: SystemTime,
}
impl NewerOptionMatcher {
@@ -3125,20 +3138,18 @@ pub mod matchers {
let metadata = fs::metadata(host.resolve(path_to_file))?;
let x_option = NewerOptionType::from_str(x_option);
let y_option = NewerOptionType::from_str(y_option);
Ok(Self { x_option, y_option, given_modification_time: metadata.modified()? })
let reference_time = y_option.get_file_time(&metadata)?;
Ok(Self { x_option, reference_time })
}
fn matches_impl(&self, file_info: &WalkEntry) -> Result<bool, Box<dyn Error>> {
let x_option_time = self.x_option.get_file_time(file_info.metadata()?)?;
let y_option_time = self.y_option.get_file_time(file_info.metadata()?)?;
// duration_since returns Err when x_option_time is strictly
// newer than the reference time.
Ok(self
.given_modification_time
.reference_time
.duration_since(x_option_time)
.is_err()
&& self
.given_modification_time
.duration_since(y_option_time)
.is_err())
}
}
@@ -3149,9 +3160,8 @@ pub mod matchers {
Err(e) => {
writeln!(
&mut matcher_io.host().stderr,
"Error getting {:?} and {:?} time for {}: {}",
"Error getting {:?} time for {}: {}",
self.x_option,
self.y_option,
file_info.path().to_string_lossy(),
e
)
@@ -3390,12 +3400,16 @@ pub mod matchers {
use super::{FileType, Follow, Matcher, MatcherIO, WalkEntry};
/// This matcher checks the type of the file.
/// This matcher checks the type of the file against a list of accepted
/// types (GNU findutils 4.9+ accepts comma-separated lists, e.g. `f,d`).
pub struct TypeMatcher {
file_type: FileType,
file_types: Vec<FileType>,
}
fn parse(type_string: &str) -> Result<FileType, Box<dyn Error>> {
/// Parses one type letter. `Ok(None)` means the letter is accepted but
/// can never match here (BSD `w` — whiteouts don't exist on this
/// platform's walk results).
fn parse_one(type_string: &str) -> Result<Option<FileType>, Box<dyn Error>> {
let file_type = match type_string {
"f" => FileType::Regular,
"d" => FileType::Directory,
@@ -3404,35 +3418,50 @@ pub mod matchers {
"c" => FileType::CharDevice,
"p" => FileType::Fifo, // named pipe (FIFO)
"s" => FileType::Socket,
// w: whiteout (BSD); accepted but never produced by the walker
"w" => return Ok(None),
// D: door (Solaris)
"D" => return Err(From::from(format!("Type argument {type_string} not supported yet"))),
_ => return Err(From::from(format!("Unrecognised type argument {type_string}"))),
};
Ok(file_type)
Ok(Some(file_type))
}
fn parse(type_string: &str) -> Result<Vec<FileType>, Box<dyn Error>> {
let mut file_types = Vec::new();
for part in type_string.split(',') {
if part.is_empty() {
return Err(From::from(format!("Unrecognised type argument {type_string}")));
}
if let Some(file_type) = parse_one(part)? {
file_types.push(file_type);
}
}
Ok(file_types)
}
impl TypeMatcher {
pub fn new(type_string: &str) -> Result<Self, Box<dyn Error>> {
let file_type = parse(type_string)?;
Ok(Self { file_type })
let file_types = parse(type_string)?;
Ok(Self { file_types })
}
}
impl Matcher for TypeMatcher {
fn matches(&self, file_info: &WalkEntry, _: &mut MatcherIO) -> bool {
file_info.file_type() == self.file_type
self.file_types.contains(&file_info.file_type())
}
}
/// Like [TypeMatcher], but toggles whether symlinks are followed.
pub struct XtypeMatcher {
file_type: FileType,
file_types: Vec<FileType>,
}
impl XtypeMatcher {
pub fn new(type_string: &str) -> Result<Self, Box<dyn Error>> {
let file_type = parse(type_string)?;
Ok(Self { file_type })
let file_types = parse(type_string)?;
Ok(Self { file_types })
}
}
@@ -3450,9 +3479,9 @@ pub mod matchers {
.map(FileType::from);
match file_type {
Ok(file_type) if file_type == self.file_type => true,
Ok(file_type) if self.file_types.contains(&file_type) => true,
// Since GNU find 4.10, ELOOP will match -xtype l
Err(e) if self.file_type.is_symlink() && e.is_loop() => true,
Err(e) if self.file_types.iter().any(|t| t.is_symlink()) && e.is_loop() => true,
_ => false,
}
}
@@ -3558,7 +3587,7 @@ pub mod matchers {
};
use ::regex::Regex;
use chrono::{DateTime, Datelike, NaiveDateTime, Utc};
use chrono::{DateTime, Datelike, Local, NaiveDate, NaiveDateTime, TimeZone, Utc};
pub use entry::{FileType, WalkEntry, WalkError};
use fs::FileSystemMatcher;
use ls::Ls;
@@ -3872,13 +3901,40 @@ pub mod matchers {
)))
}
/// This is a function that converts a specific string format into a timestamp.
/// It allows converting a time string of
/// "(week abbreviation) (date), (year) (time)" to a Unix timestamp.
/// such as: "jan 01, 2025 00:00:01" -> 1735689601000
/// When (time) is not provided, it will be automatically filled in as 00:00:00
/// such as: "jan 01, 2025" = "jan 01, 2025 00:00:00" -> 1735689600000
/// Converts a `-newerXt`-style reference time string into a Unix timestamp
/// (milliseconds).
///
/// Accepts, in order:
/// - `@N[.N]` seconds since the epoch (GNU extension)
/// - RFC 3339 datetimes with an explicit offset, e.g.
/// "2026-01-01T00:00:00Z"
/// - ISO-style naive dates/datetimes ("2026-01-01",
/// "2026-01-01 12:30[:45]", with ` ` or `T` separators), interpreted in
/// local time like GNU find
/// - "(month abbreviation) (date), (year) (time)" strings, e.g.
/// "jan 01, 2025 00:00:01" (time defaults to 00:00:00)
fn parse_date_str_to_timestamps(date_str: &str) -> Option<i64> {
if let Some(epoch) = date_str.strip_prefix('@')
&& let Ok(seconds) = epoch.parse::<f64>()
{
return Some((seconds * 1000.0) as i64);
}
if let Ok(datetime) = DateTime::parse_from_rfc3339(date_str) {
return Some(datetime.timestamp_millis());
}
let naive = NaiveDate::parse_from_str(date_str, "%Y-%m-%d")
.ok()
.and_then(|date| date.and_hms_opt(0, 0, 0))
.or_else(|| NaiveDateTime::parse_from_str(date_str, "%Y-%m-%d %H:%M:%S").ok())
.or_else(|| NaiveDateTime::parse_from_str(date_str, "%Y-%m-%dT%H:%M:%S").ok())
.or_else(|| NaiveDateTime::parse_from_str(date_str, "%Y-%m-%d %H:%M").ok())
.or_else(|| NaiveDateTime::parse_from_str(date_str, "%Y-%m-%dT%H:%M").ok());
if let Some(naive) = naive {
return Some(Local.from_local_datetime(&naive).earliest()?.timestamp_millis());
}
let regex_pattern =
r"^(?P<month_day>\w{3} \d{2})?(?:, (?P<year>\d{4}))?(?: (?P<time>\d{2}:\d{2}:\d{2}))?$";
let re = Regex::new(regex_pattern);
@@ -4602,6 +4658,7 @@ fn parse_args(args: &[&str], host: &mut Host) -> Result<ParsedInfo, Box<dyn Erro
let mut paths = vec![];
let mut i = 0;
let mut config = Config::default();
let mut extended_regex = false;
while i < args.len() {
match args[i] {
@@ -4611,6 +4668,12 @@ fn parse_args(args: &[&str], host: &mut Host) -> Result<ParsedInfo, Box<dyn Erro
"-H" => config.follow = Follow::Roots,
"-L" => config.follow = Follow::Always,
"-P" => config.follow = Follow::Never,
// BSD find leading flags (macOS muscle memory).
"-E" => extended_regex = true,
// -x is the BSD spelling of -xdev.
"-x" => config.same_file_system = true,
// -s sorts output lexicographically.
"-s" => config.sorted_output = true,
"--" => {
// End of flags
i += 1;
@@ -4634,7 +4697,16 @@ fn parse_args(args: &[&str], host: &mut Host) -> Result<ParsedInfo, Box<dyn Erro
if i == paths_start {
paths.push(".".to_string());
}
let matcher = matchers::build_top_level_matcher(&args[i..], &mut config, host)?;
let matcher = if extended_regex {
// BSD -E selects POSIX extended regular expressions; GNU spells that
// -regextype posix-extended, which must precede any -regex/-iregex.
let mut expression = Vec::with_capacity(args.len() - i + 2);
expression.extend(["-regextype", "posix-extended"]);
expression.extend_from_slice(&args[i..]);
matchers::build_top_level_matcher(&expression, &mut config, host)?
} else {
matchers::build_top_level_matcher(&args[i..], &mut config, host)?
};
if let Some(new_paths) = &config.new_paths {
if paths.len() == 1 && paths[0] == "." {
paths = new_paths.to_vec();
@@ -4878,9 +4950,9 @@ Early alpha implementation. Currently the only expressions supported are
-files0-from
-regex pattern
-iregex pattern
-type type_char
currently type_char can only be f (for file) or d (for directory)
-size [+-]N[bcwkMG]
-type type_char[,type_char...]
type_char is one of f d l b c p s w
-size [+-]N[bcwkMGTP]
-delete
-prune
-not
@@ -4897,7 +4969,7 @@ Early alpha implementation. Currently the only expressions supported are
-ctime [+-]N
-atime [+-]N
-mtime [+-]N
-perm [-/]{{octal|u=rwx,go=w}}
-perm [-/+]{{octal|u=rwx,go=w}}
-newer path_to_file
-exec[dir] executable [args] [{{}}] [more args] ;
-sorted
@@ -4940,7 +5012,7 @@ fn rewrite_bsd_invocation(args: &[&str], host: &mut Host) -> Option<Vec<String>>
let mut i = 0;
while i < rewritten.len() {
match rewritten[i].as_str() {
"-O0" | "-O1" | "-O2" | "-O3" | "-H" | "-L" | "-P" => i += 1,
"-O0" | "-O1" | "-O2" | "-O3" | "-H" | "-L" | "-P" | "-x" | "-s" => i += 1,
"--" => {
i += 1;
break;
@@ -5118,4 +5190,227 @@ mod tests {
assert_eq!(capture.err(), "");
assert_eq!(capture.out(), "sub\n");
}
/// Failure mode: `-newerXY ref` compared both X and Y timestamps of the
/// CANDIDATE against the reference's mtime, instead of comparing the
/// candidate's X timestamp against the reference's Y timestamp.
#[cfg(unix)]
#[test]
fn newer_xy_compares_candidate_x_against_reference_y() {
use std::{
fs::FileTimes,
time::{Duration, SystemTime},
};
let dir = tempfile::tempdir().unwrap();
let root = fs::canonicalize(dir.path()).unwrap();
let now = SystemTime::now();
let old = now - Duration::from_secs(2000);
let mid = now - Duration::from_secs(1000);
let write_with_times = |name: &str, accessed: SystemTime, modified: SystemTime| {
let path = root.join(name);
fs::write(&path, b"x").unwrap();
let file = fs::File::options().write(true).open(&path).unwrap();
file
.set_times(FileTimes::new().set_accessed(accessed).set_modified(modified))
.unwrap();
};
write_with_times("ref", mid, mid);
// atime newer than ref's mtime, but mtime older: -neweram must match.
// (The old code also demanded a newer mtime and rejected this file.)
write_with_times("hit", now, old);
// atime older than ref's mtime: -neweram must not match.
write_with_times("miss", old, now);
let (code, capture) = run(
&root,
&[
root.display().to_string(),
"-type".into(),
"f".into(),
"-neweram".into(),
"ref".into(),
],
);
assert_eq!(code, 0, "stderr: {}", capture.err());
assert_eq!(capture.err(), "");
assert_eq!(capture.out(), format!("{}\n", root.join("hit").display()));
}
/// Failure mode: `-newermt` rejected ISO dates like `2026-01-01` with
/// "cannot figure out how to interpret ... as a date or time".
#[test]
fn newermt_accepts_iso_dates() {
let (_dir, root) = fixture();
let (code, capture) = run(
&root,
&[
root.display().to_string(),
"-name".into(),
"a.txt".into(),
"-newermt".into(),
"2000-01-01".into(),
],
);
assert_eq!(code, 0, "stderr: {}", capture.err());
assert_eq!(capture.err(), "");
assert_eq!(capture.out(), format!("{}\n", root.join("a.txt").display()));
let (code, capture) = run(
&root,
&[
root.display().to_string(),
"-newermt".into(),
"3000-01-01".into(),
],
);
assert_eq!(code, 0, "stderr: {}", capture.err());
assert_eq!(capture.err(), "");
assert_eq!(capture.out(), "");
}
/// Failure mode: BSD `-perm +mode` (any of the bits set) was parsed as an
/// exact-mode pattern and failed with a parse error.
#[cfg(unix)]
#[test]
fn perm_plus_mode_matches_any_set_bits() {
use std::os::unix::fs::PermissionsExt;
let (_dir, root) = fixture();
fs::set_permissions(root.join("a.txt"), fs::Permissions::from_mode(0o755)).unwrap();
fs::set_permissions(root.join("b.md"), fs::Permissions::from_mode(0o644)).unwrap();
fs::set_permissions(root.join("c.rs"), fs::Permissions::from_mode(0o600)).unwrap();
let (code, capture) = run(
&root,
&[
root.display().to_string(),
"-type".into(),
"f".into(),
"-perm".into(),
"+111".into(),
],
);
assert_eq!(code, 0, "stderr: {}", capture.err());
assert_eq!(capture.err(), "");
assert_eq!(capture.out(), format!("{}\n", root.join("a.txt").display()));
}
/// Failure mode: GNU `-type f,d` lists were rejected with "Unrecognised
/// type argument f,d".
#[test]
fn type_accepts_comma_separated_list() {
let (_dir, root) = fixture();
fs::create_dir(root.join("sub")).unwrap();
let (code, capture) = run(&root, &[root.display().to_string(), "-type".into(), "f,d".into()]);
assert_eq!(code, 0, "stderr: {}", capture.err());
assert_eq!(capture.err(), "");
let mut matches: Vec<PathBuf> = capture.out().lines().map(PathBuf::from).collect();
matches.sort();
assert_eq!(matches, vec![
root.clone(),
root.join("a.txt"),
root.join("b.md"),
root.join("c.rs"),
root.join("sub"),
]);
}
/// Failure mode: BSD `-type w` (whiteout) errored instead of parsing and
/// matching nothing.
#[test]
fn type_w_parses_and_matches_nothing() {
let (_dir, root) = fixture();
let (code, capture) = run(&root, &[root.display().to_string(), "-type".into(), "w".into()]);
assert_eq!(code, 0, "stderr: {}", capture.err());
assert_eq!(capture.err(), "");
assert_eq!(capture.out(), "");
}
/// Failure mode: `-regex` matched substrings of the path instead of
/// requiring the pattern to span the whole path.
#[test]
fn regex_matches_whole_path_only() {
let (_dir, root) = fixture();
let (code, capture) =
run(&root, &[root.display().to_string(), "-regex".into(), r".*\.rs".into()]);
assert_eq!(code, 0, "stderr: {}", capture.err());
assert_eq!(capture.out(), format!("{}\n", root.join("c.rs").display()));
let (code, capture) =
run(&root, &[root.display().to_string(), "-regex".into(), r"c\.rs".into()]);
assert_eq!(code, 0, "stderr: {}", capture.err());
assert_eq!(capture.out(), "");
}
/// Failure mode: an alternation whose shorter branch matches a path prefix
/// would win and the full-path match was missed (no backtracking retry).
#[test]
fn regex_full_match_prefers_longest_alternative() {
let (_dir, root) = fixture();
let (code, capture) = run(
&root,
&[
root.display().to_string(),
"-regextype".into(),
"posix-extended".into(),
"-regex".into(),
r".*/c|.*/c\.rs".into(),
],
);
assert_eq!(code, 0, "stderr: {}", capture.err());
assert_eq!(capture.err(), "");
assert_eq!(capture.out(), format!("{}\n", root.join("c.rs").display()));
}
/// Failure mode: `-size` rejected the `T` and `P` suffixes accepted by
/// modern GNU and BSD find.
#[test]
fn size_accepts_t_and_p_suffixes() {
let (_dir, root) = fixture();
let (code, capture) = run(
&root,
&[
root.display().to_string(),
"-type".into(),
"f".into(),
"-size".into(),
"-2T".into(),
],
);
assert_eq!(code, 0, "stderr: {}", capture.err());
assert_eq!(capture.err(), "");
let mut matches: Vec<PathBuf> = capture.out().lines().map(PathBuf::from).collect();
matches.sort();
assert_eq!(matches, vec![root.join("a.txt"), root.join("b.md"), root.join("c.rs")]);
let (code, capture) =
run(&root, &[root.display().to_string(), "-size".into(), "+1P".into()]);
assert_eq!(code, 0, "stderr: {}", capture.err());
assert_eq!(capture.err(), "");
assert_eq!(capture.out(), "");
}
/// Failure mode: BSD leading flags `-x` and `-s` were treated as unknown
/// predicates and the invocation failed to parse.
#[test]
fn bsd_leading_flags_x_and_s_parse() {
let (_dir, root) = fixture();
let (code, capture) = run(
&root,
&[
"-s".into(),
"-x".into(),
root.display().to_string(),
"-type".into(),
"f".into(),
],
);
assert_eq!(code, 0, "stderr: {}", capture.err());
assert_eq!(capture.err(), "");
// -s guarantees lexicographically sorted output.
let matches: Vec<PathBuf> = capture.out().lines().map(PathBuf::from).collect();
assert_eq!(matches, vec![root.join("a.txt"), root.join("b.md"), root.join("c.rs")]);
}
}
+187 -40
View File
@@ -175,10 +175,13 @@ fn process_num_block(
let mut quiet = false;
let mut verbose = false;
let mut zero_terminated = false;
// Lowercase suffixes are byte multipliers (obsolete BSD `-Nc`/`-Nb`/`-Nk`/`-Nm`);
// uppercase suffixes mirror the modern `-n NUM<suffix>` form and scale the
// line count (`head -10K` == `head -n 10240`).
let mut multiplier = None;
let mut line_multiplier: usize = 1;
let mut c = last_char;
loop {
// note that here, we only match lower case 'k', 'c', and 'm'
match c {
// we want to preserve order
// this also saves us 1 heap allocation
@@ -195,6 +198,18 @@ fn process_num_block(
'b' => multiplier = Some(512),
'k' => multiplier = Some(1024),
'm' => multiplier = Some(1024 * 1024),
'K' => {
line_multiplier = 1024;
multiplier = None;
},
'M' => {
line_multiplier = 1024 * 1024;
multiplier = None;
},
'G' => {
line_multiplier = 1024 * 1024 * 1024;
multiplier = None;
},
'\0' => {},
_ => return Err(ParseError),
}
@@ -220,6 +235,7 @@ fn process_num_block(
options.push(OsString::from(format!("{num}")));
} else {
options.push(OsString::from("-n"));
let num = num.saturating_mul(line_multiplier);
options.push(OsString::from(format!("{num}")));
}
Ok(options)
@@ -266,6 +282,8 @@ mod tests {
assert_eq!(obsolete("-1k"), obsolete_result(&["-c", "1024"]));
assert_eq!(obsolete("-2b"), obsolete_result(&["-c", "1024"]));
assert_eq!(obsolete("-1mmk"), obsolete_result(&["-c", "1024"]));
assert_eq!(obsolete("-10K"), obsolete_result(&["-n", "10240"]));
assert_eq!(obsolete("-1M"), obsolete_result(&["-n", "1048576"]));
assert_eq!(obsolete("-1vz"), obsolete_result(&["-v", "-z", "-n", "1"]));
assert_eq!(
obsolete("-1vzqvq"),
@@ -1020,33 +1038,81 @@ impl Mode {
}
}
fn arg_iterate<'a>(
mut args: impl Iterator<Item = OsString> + 'a,
) -> HeadResult<Box<dyn Iterator<Item = OsString> + 'a>> {
/// True when `token` is an option that takes its value from the *next* argv
/// token, so that value must never be mistaken for an obsolete `-NUM` form
/// (e.g. the `-5` in `head -n -5 file`).
fn consumes_separate_value(token: &str) -> bool {
if let Some(long) = token.strip_prefix("--") {
if long.is_empty() || long.contains('=') {
return false;
}
// clap infers unambiguous long-option prefixes.
return ["lines", "bytes"].iter().any(|name| name.starts_with(long));
}
let Some(cluster) = token.strip_prefix('-') else {
return false;
};
let mut chars = cluster.chars();
while let Some(c) = chars.next() {
match c {
// Value-taking shorts: a trailing `-n`/`-c` consumes the next
// token; anything after them in the cluster is an attached value.
'n' | 'c' => return chars.next().is_none(),
'q' | 'v' | 'z' => {},
_ => return false,
}
}
false
}
/// Rewrites every obsolete `-NUM[suffix]` token (before `--`) into modern
/// options, wherever it appears among flags and operands: GNU/BSD accept
/// `head -q -5 file`, `head file -5`, and `head -5 -20 file`.
fn arg_iterate(argv: Vec<OsString>) -> HeadResult<Vec<OsString>> {
let mut rewritten = Vec::with_capacity(argv.len() + 1);
let mut iter = argv.into_iter();
// argv[0] is always present
let first = args.next().unwrap();
if let Some(second) = args.next() {
if let Some(s) = second.to_str() {
if let Some(v) = parse::parse_obsolete(s) {
match v {
Ok(iter) => Ok(Box::new(vec![first].into_iter().chain(iter).chain(args))),
Err(parse::ParseError) => {
Err(HeadError::ParseError(format!("bad argument format: {}", s.quote())))
rewritten.extend(iter.next());
let mut skip_value = false;
let mut seen_ddash = false;
for arg in iter {
if skip_value || seen_ddash {
skip_value = false;
rewritten.push(arg);
continue;
}
let Some(token) = arg.to_str() else {
// Non-UTF-8 can't be an obsolete option like "-5"; treat it as a
// regular file argument.
rewritten.push(arg);
continue;
};
if token == "--" {
seen_ddash = true;
rewritten.push(arg);
continue;
}
if matches!(token.as_bytes(), [b'-', b'0'..=b'9', ..]) {
match parse::parse_obsolete(token) {
Some(Ok(options)) => {
rewritten.extend(options);
continue;
},
Some(Err(parse::ParseError)) => {
return Err(HeadError::ParseError(format!(
"bad argument format: {}",
token.quote()
)));
},
None => {},
}
} else {
// The second argument contains non-UTF-8 sequences, so it can't be an obsolete
// option like "-5". Treat it as a regular file argument.
Ok(Box::new(vec![first, second].into_iter().chain(args)))
}
} else {
// The second argument contains non-UTF-8 sequences, so it can't be an obsolete
// option like "-5". Treat it as a regular file argument.
Ok(Box::new(vec![first, second].into_iter().chain(args)))
if token.len() > 1 && token.starts_with('-') {
skip_value = consumes_separate_value(token);
}
} else {
Ok(Box::new(vec![first].into_iter()))
rewritten.push(arg);
}
Ok(rewritten)
}
#[derive(Debug, PartialEq, Default)]
@@ -1301,9 +1367,7 @@ impl Utility for Head {
// Normalize GNU's obsolete `-NUM` syntax before clap sees argv; clap
// otherwise treats it as an unknown short-option cluster.
fn rewrite_argv(argv: Vec<OsString>) -> Result<Vec<OsString>, String> {
arg_iterate(argv.into_iter())
.map(Iterator::collect)
.map_err(|err| err.to_string())
arg_iterate(argv).map_err(|err| err.to_string())
}
@@ -1316,14 +1380,24 @@ impl Utility for Head {
},
};
let print_headers = (options.files.len() > 1 && !options.quiet) || options.verbose;
// GNU head only emits the blank separator line before a header when a
// previous file actually produced output; open failures print nothing
// and must not flip `first`.
let mut first = true;
fn print_header(out: &mut impl Write, name: &[u8], first: &mut bool) {
if !*first {
let _ = writeln!(out);
}
let _ = out.write_all(b"==> ");
let _ = out.write_all(name);
let _ = out.write_all(b" <==\n");
*first = false;
}
for file in &options.files {
let result = if file == "-" {
if (options.files.len() > 1 && !options.quiet) || options.verbose {
if !first {
let _ = writeln!(host.stdout);
}
let _ = writeln!(host.stdout, "==> standard input <==");
if print_headers {
print_header(&mut host.stdout, b"standard input", &mut first);
}
let mut input = io::BufReader::with_capacity(BUF_SIZE, &mut host.stdin);
match options.mode {
@@ -1344,25 +1418,23 @@ impl Utility for Head {
} else {
let resolved = host.resolve(file);
if resolved.is_dir() {
// GNU prints the header before reporting the read error,
// and that header counts as produced output.
if print_headers {
print_header(&mut host.stdout, file.as_encoded_bytes(), &mut first);
}
host.error(format!("error reading {}: Is a directory", file.quote()), 1);
first = false;
continue;
}
let mut input = match File::open(&resolved) {
Ok(input) => input,
Err(err) => {
host.error(format!("cannot open {} for reading: {err}", file.quote()), 1);
first = false;
continue;
},
};
if (options.files.len() > 1 && !options.quiet) || options.verbose {
if !first {
let _ = writeln!(host.stdout);
}
let _ = write!(host.stdout, "==> ");
let _ = host.stdout.write_all(file.as_encoded_bytes());
let _ = writeln!(host.stdout, " <==");
if print_headers {
print_header(&mut host.stdout, file.as_encoded_bytes(), &mut first);
}
head_file(&mut input, &mut host.stdout, &options)
};
@@ -1372,10 +1444,14 @@ impl Utility for Head {
} else {
PathBuf::from(file)
};
// A dead pipe ends the whole invocation; any other I/O error
// only fails this operand, and GNU keeps going.
let broken_pipe = err.kind() == io::ErrorKind::BrokenPipe;
host.error(HeadError::Io { name, err }, 1);
if broken_pipe {
return 1;
}
first = false;
}
}
host.exit_code()
}
@@ -1434,6 +1510,77 @@ mod tests {
assert_eq!(capture.err(), "head: bad argument format: '-123FooBar'\n");
}
fn rewritten(argv: &[&str]) -> Vec<String> {
Head::rewrite_argv(argv.iter().map(OsString::from).collect())
.unwrap()
.into_iter()
.map(|arg| arg.to_str().unwrap().to_owned())
.collect()
}
// Failure mode: obsolete `-NUM` was only recognized as argv[1], so
// `head -q -5 file`, `head file -5`, and repeated counts were parse errors.
#[test]
fn obsolete_num_is_rewritten_at_any_position() {
assert_eq!(rewritten(&["head", "-q", "-5", "f"]), ["head", "-q", "-n", "5", "f"]);
assert_eq!(rewritten(&["head", "-v", "-20", "f"]), ["head", "-v", "-n", "20", "f"]);
assert_eq!(rewritten(&["head", "f", "-5"]), ["head", "f", "-n", "5"]);
assert_eq!(
rewritten(&["head", "-5", "-20", "f"]),
["head", "-n", "5", "-n", "20", "f"]
);
assert_eq!(rewritten(&["head", "-5qz", "f"]), ["head", "-q", "-z", "-n", "5", "f"]);
}
// Failure mode: `-5` following a value-taking option is that option's
// value, and rewriting it would corrupt the invocation.
#[test]
fn option_values_and_post_ddash_operands_are_not_rewritten() {
assert_eq!(rewritten(&["head", "-n", "-5", "f"]), ["head", "-n", "-5", "f"]);
assert_eq!(rewritten(&["head", "-c", "-5", "f"]), ["head", "-c", "-5", "f"]);
assert_eq!(rewritten(&["head", "--lines", "-5", "f"]), ["head", "--lines", "-5", "f"]);
assert_eq!(rewritten(&["head", "--", "-5"]), ["head", "--", "-5"]);
assert_eq!(rewritten(&["head", "-n5", "-", "f"]), ["head", "-n5", "-", "f"]);
}
// Failure mode: uppercase suffixes in the obsolete form were rejected
// even though `head -n 10K` accepts them.
#[test]
fn obsolete_uppercase_suffixes_scale_lines() {
assert_eq!(options("-10K").unwrap().mode, Mode::FirstLines(10 * 1024));
assert_eq!(options("-1M").unwrap().mode, Mode::FirstLines(1024 * 1024));
assert_eq!(options("-1G").unwrap().mode, Mode::FirstLines(1024 * 1024 * 1024));
// Lowercase suffixes keep their historical byte meaning.
assert_eq!(options("-1k").unwrap().mode, Mode::FirstBytes(1024));
}
// Failure mode: an unreadable operand flipped the separator state and the
// next header gained a spurious leading blank line; a mid-list open error
// must also not abort the remaining operands.
#[test]
fn open_error_produces_no_separator_and_processing_continues() {
let dir = tempdir().unwrap();
std::fs::write(dir.path().join("f1"), b"a\n").unwrap();
std::fs::write(dir.path().join("f2"), b"b\n").unwrap();
let (code, capture) = run_util::<Head>(&["-n", "1", "missing", "f1", "f2"], "", dir.path());
assert_eq!(code, 1);
assert_eq!(capture.out(), "==> f1 <==\na\n\n==> f2 <==\nb\n");
assert!(capture.err().contains("cannot open 'missing' for reading"));
}
// Failure mode: the `==> dir <==` header was suppressed before the
// Is-a-directory diagnostic; GNU prints it and counts it as output.
#[test]
fn directory_operand_prints_header_before_error() {
let dir = tempdir().unwrap();
std::fs::create_dir(dir.path().join("d")).unwrap();
std::fs::write(dir.path().join("f"), b"a\n").unwrap();
let (code, capture) = run_util::<Head>(&["-n", "1", "d", "f"], "", dir.path());
assert_eq!(code, 1);
assert_eq!(capture.out(), "==> d <==\n\n==> f <==\na\n");
assert_eq!(capture.err(), "head: error reading 'd': Is a directory\n");
}
#[test]
fn defaults_to_ten_lines_from_stdin() {
let input = (1..=12).map(|n| format!("{n}\n")).collect::<String>();
+14 -5
View File
@@ -443,11 +443,20 @@ pub(crate) fn os_bytes_lossy(value: &std::ffi::OsStr) -> std::borrow::Cow<'_, [u
/// Parses a GNU-style duration: a decimal number with an optional `s`/`m`/`h`/`d`
/// suffix, as accepted by `sleep` and `timeout`.
///
/// GNU also accepts `inf`/`infinity` (optionally signed `+`, any case);
/// infinite and overflowing values saturate to [`Duration::MAX`]. Callers
/// treat such durations as "sleep until cancelled". Sub-millisecond precision
/// is preserved: GNU `sleep 0.0001` really sleeps 100 microseconds.
pub(crate) fn parse_duration(input: &str) -> Option<Duration> {
let trimmed = input.trim();
if trimmed.is_empty() {
return None;
}
let unsigned = trimmed.strip_prefix('+').unwrap_or(trimmed);
if unsigned.eq_ignore_ascii_case("inf") || unsigned.eq_ignore_ascii_case("infinity") {
return Some(Duration::MAX);
}
let (number, multiplier) = match trimmed.chars().last()? {
's' => (&trimmed[..trimmed.len() - 1], 1.0),
'm' => (&trimmed[..trimmed.len() - 1], 60.0),
@@ -457,14 +466,14 @@ pub(crate) fn parse_duration(input: &str) -> Option<Duration> {
_ => (trimmed, 1.0),
};
let value = number.parse::<f64>().ok()?;
if value.is_sign_negative() {
if value.is_nan() || value.is_sign_negative() {
return None;
}
let millis = value * multiplier * 1000.0;
if !millis.is_finite() || millis < 0.0 {
return None;
if value.is_infinite() {
return Some(Duration::MAX);
}
Some(Duration::from_millis(millis.round() as u64))
// Only overflow remains once NaN and negatives are excluded; saturate.
Duration::try_from_secs_f64(value * multiplier).map_or(Some(Duration::MAX), Some)
}
+1 -1
View File
@@ -624,7 +624,7 @@ mod filter {
});
Block::new(idx, labels).unwrap().map_code(|c| {
let c = c.replace('\t', " ");
let w = unicode_width::UnicodeWidthStr::width(&*c);
let w = xutf::width_str(&c);
CodeWidth::new(c, core::cmp::max(w, 1))
})
}
+172 -20
View File
@@ -37,6 +37,13 @@ pub(crate) struct KillCommand {
impl builtins::Command for KillCommand {
type Error = brush_core::Error;
fn new<I>(args: I) -> std::result::Result<Self, clap::Error>
where
I: IntoIterator<Item = String>,
{
Self::try_parse_from(rewrite_attached_short_options(args))
}
#[allow(unknown_lints, reason = "unused_async_trait_impl is unknown to the pinned CI nightly")]
#[allow(
clippy::unused_async_trait_impl,
@@ -342,6 +349,63 @@ impl builtins::Command for KillCommand {
}
}
/// Splits attached short-option values before clap sees the argv: `-sKILL`
/// and `-s9` become `-s <spec>`, `-n9` becomes `-n 9`, and `-l9`/`-L137`
/// become `-l <spec>` (bash splits `-s<name>` the same way; the digit forms
/// are the /bin/kill spellings). A token whose whole body already names a
/// signal (`-sigkill`, `-SIGKILL`, `-9`) is left intact, matching BSD kill
/// and the manual sigspec pre-parse in `execute`. Rewriting stops at `--` or
/// the first operand, so negative-PID operands survive untouched.
fn rewrite_attached_short_options(args: impl IntoIterator<Item = String>) -> Vec<String> {
let mut out: Vec<String> = Vec::new();
let mut args = args.into_iter();
// The first element is the command name itself.
out.extend(args.next());
let mut skip_value = false;
for arg in &mut args {
if skip_value {
skip_value = false;
out.push(arg);
continue;
}
if arg == "--" {
out.push(arg);
break;
}
if arg == "-s" || arg == "-n" {
skip_value = true;
out.push(arg);
continue;
}
if arg == "-l" || arg == "-L" {
out.push(arg);
continue;
}
if let Some((option, value)) = split_attached(&arg) {
out.push(option);
out.push(value);
continue;
}
out.push(arg);
break;
}
out.extend(args);
out
}
/// Splits one attached-value option token, or `None` for anything that must
/// pass through untouched (whole sigspecs, operands, malformed tokens).
fn split_attached(arg: &str) -> Option<(String, String)> {
let rest = arg.get(2..).filter(|rest| !rest.is_empty())?;
let split = match arg.get(..2)? {
"-l" | "-L" => true,
"-s" => KillSignal::parse(&arg[1..]).is_err(),
"-n" => rest.bytes().all(|byte| byte.is_ascii_digit()),
_ => false,
};
split.then(|| (arg[..2].to_string(), rest.to_string()))
}
/// Whether signalling `target` would reach the shell or one of its ancestors.
///
/// `target` follows `kill(2)`: a positive value is a pid, `0` is the caller's own
@@ -375,26 +439,7 @@ fn print_kill_signals<'a>(
.map(|()| ExecutionResult::success());
}
for value in signals {
enum PrintedSignal {
Name(&'static str),
Number(i32),
}
let signal = if let Ok(number) = value.parse::<i32>() {
TrapSignal::try_from(number).map(|signal| {
PrintedSignal::Name(
signal
.as_str()
.strip_prefix("SIG")
.unwrap_or(signal.as_str()),
)
})
} else {
TrapSignal::try_from(value.as_str()).map(|signal| {
i32::try_from(signal)
.map_or(PrintedSignal::Name(signal.as_str()), PrintedSignal::Number)
})
};
match signal {
match printed_signal(value) {
Ok(PrintedSignal::Name(name)) => writeln!(context.stdout(), "{name}")?,
Ok(PrintedSignal::Number(number)) => writeln!(context.stdout(), "{number}")?,
Err(err) => {
@@ -406,6 +451,34 @@ fn print_kill_signals<'a>(
Ok(result)
}
/// How `kill -l <operand>` renders one operand: numbers become names and
/// names become numbers.
enum PrintedSignal {
Name(&'static str),
Number(i32),
}
fn printed_signal(value: &str) -> std::result::Result<PrintedSignal, brush_core::Error> {
if let Ok(number) = value.parse::<i32>() {
// bash also maps the exit status of a signal-killed process back to
// its signal: `kill -l 137` prints `KILL` (137 = 128 + 9), while an
// unmappable value like 128 or 265 keeps its own diagnostic.
let signal = TrapSignal::try_from(number).or_else(|err| {
if number > 128 {
TrapSignal::try_from(number - 128).map_err(|_| err)
} else {
Err(err)
}
})?;
Ok(PrintedSignal::Name(
signal.as_str().strip_prefix("SIG").unwrap_or(signal.as_str()),
))
} else {
let signal = TrapSignal::try_from(value)?;
Ok(i32::try_from(signal).map_or(PrintedSignal::Name(signal.as_str()), PrintedSignal::Number))
}
}
#[cfg(test)]
impl KillCommand {
fn listed_signals(&self) -> impl Iterator<Item = &String> {
@@ -447,6 +520,85 @@ mod tests {
fn lists_pre_marker_operands_without_marker() {
assert_eq!(listed(&["kill", "-l", "TERM", "HUP"]), ["TERM", "HUP"]);
}
fn parsed(args: &[&str]) -> KillCommand {
use brush_core::builtins::Command as _;
KillCommand::new(args.iter().map(ToString::to_string)).unwrap()
}
/// `kill -s9`/`-sKILL` used to land in the positional args and die with
/// "invalid signal name"; the attached value must reach `-s`.
#[test]
fn attached_signal_name_values_split() {
let cmd = parsed(&["kill", "-s9", "123"]);
assert_eq!(cmd.signal_name.as_deref(), Some("9"));
assert_eq!(cmd.args, ["123"]);
let cmd = parsed(&["kill", "-sKILL", "123"]);
assert_eq!(cmd.signal_name.as_deref(), Some("KILL"));
assert_eq!(cmd.args, ["123"]);
}
/// A whole-token signal spec such as `-sigkill` (BSD kill accepts it)
/// must not be misread as `-s igkill`.
#[test]
fn sig_prefixed_spec_stays_whole() {
let cmd = parsed(&["kill", "-sigkill", "123"]);
assert_eq!(cmd.signal_name, None);
assert_eq!(cmd.args, ["-sigkill", "123"]);
}
/// `kill -n9` used to fail to parse; the digits must reach `-n`.
#[test]
fn attached_signal_number_splits() {
let cmd = parsed(&["kill", "-n9", "123"]);
assert_eq!(cmd.signal_number, Some(9));
assert_eq!(cmd.args, ["123"]);
}
/// `kill -l9` and `kill -L137` used to be clap parse errors; they must
/// behave as `-l` with the value as its listing operand.
#[test]
fn attached_list_operand_splits() {
let cmd = parsed(&["kill", "-l9"]);
assert!(cmd.list_signals);
assert_eq!(cmd.listed_signals().cloned().collect::<Vec<_>>(), ["9"]);
let cmd = parsed(&["kill", "-L137"]);
assert!(cmd.list_signals);
assert_eq!(cmd.listed_signals().cloned().collect::<Vec<_>>(), ["137"]);
}
/// Rewriting must stop at `--` and at the first operand so option-like
/// operands (and negative PIDs) are never split.
#[test]
fn rewrite_leaves_operand_region_alone() {
let rewritten = rewrite_attached_short_options(
["kill", "--", "-s9"].map(String::from),
);
assert_eq!(rewritten, ["kill", "--", "-s9"]);
let rewritten = rewrite_attached_short_options(
["kill", "-9", "-s9"].map(String::from),
);
assert_eq!(rewritten, ["kill", "-9", "-s9"]);
let rewritten = rewrite_attached_short_options(
["kill", "-s", "KILL", "-123"].map(String::from),
);
assert_eq!(rewritten, ["kill", "-s", "KILL", "-123"]);
}
/// bash maps exit statuses above 128 back to the terminating signal:
/// `kill -l 137` prints `KILL`, while 128 and 265 stay invalid.
#[test]
fn list_maps_exit_statuses_above_128() {
assert!(matches!(printed_signal("137"), Ok(PrintedSignal::Name("KILL"))));
assert!(matches!(printed_signal("9"), Ok(PrintedSignal::Name("KILL"))));
assert!(matches!(printed_signal("129"), Ok(PrintedSignal::Name("HUP"))));
assert!(printed_signal("128").is_err());
assert!(printed_signal("265").is_err());
}
}
/// A `kill` signal argument: a real signal, or the "does this process
+2 -1
View File
@@ -124,7 +124,8 @@ mod base64;
mod basename;
#[cfg(feature = "util.cat")]
mod cat;
/// Shared checksum machinery behind `md5sum`, `sha*sum`, and `b2sum`.
/// The `cksum` builtin plus the shared checksum machinery behind `md5sum`,
/// `sha*sum`, and `b2sum`.
#[cfg(feature = "util.cksum")]
mod cksum;
#[cfg(feature = "util.md5sum")]
+112 -1
View File
@@ -19,22 +19,74 @@ use crate::host::quote_arg;
#[derive(Parser)]
#[command(disable_help_flag = true)]
pub(crate) struct NohupCommand {
/// `--help` was the first argument; set by [`NohupCommand::from_argv`].
#[clap(skip)]
help: bool,
/// `--version` was the first argument; set by [`NohupCommand::from_argv`].
#[clap(skip)]
version: bool,
#[arg(num_args = 0.., trailing_var_arg = true, allow_hyphen_values = true)]
command: Vec<String>,
}
impl NohupCommand {
/// Parses `argv` (without the command name) the way GNU nohup does:
/// `--help`/`--version` are recognized only as the first argument, and a
/// single leading `--` ends option processing, so `nohup -- --help` runs
/// a command named `--help` and `nohup -- --` runs one named `--`.
fn from_argv(mut argv: Vec<String>) -> Self {
match argv.first().map(String::as_str) {
Some("--help") => {
return Self { help: true, version: false, command: Vec::new() };
},
Some("--version") => {
return Self { help: false, version: true, command: Vec::new() };
},
Some("--") => {
argv.remove(0);
},
_ => {},
}
Self { help: false, version: false, command: argv }
}
}
impl builtins::Command for NohupCommand {
type Error = brush_core::Error;
/// Bypasses clap: clap silently eats the first `--` even inside a
/// `trailing_var_arg` capture, which loses the distinction between
/// `nohup --help` (help) and `nohup -- --help` (run `--help`).
fn new<I>(args: I) -> std::result::Result<Self, clap::Error>
where
I: IntoIterator<Item = String>,
{
// The first element is the command name itself.
Ok(Self::from_argv(args.into_iter().skip(1).collect()))
}
fn execute<SE: brush_core::ShellExtensions>(
&self,
context: ExecutionContext<'_, SE>,
) -> impl Future<Output = std::result::Result<ExecutionResult, brush_core::Error>> + Send {
let command = self.command.clone();
let (help, version) = (self.help, self.version);
async move {
if context.is_cancelled() {
return Ok(ExecutionExitCode::Interrupted.into());
}
if help {
let _ = write!(context.stdout(), "{NOHUP_HELP}");
return Ok(ExecutionResult::success());
}
if version {
let _ = writeln!(
context.stdout(),
"nohup (pi-builtins) {}",
env!("CARGO_PKG_VERSION")
);
return Ok(ExecutionResult::success());
}
// coreutils `nohup` with no operand fails with exit code 125.
if command.is_empty() {
return Ok(report_missing_operand(context.stderr()));
@@ -60,6 +112,15 @@ impl builtins::Command for NohupCommand {
}
}
const NOHUP_HELP: &str = "\
Usage: nohup COMMAND [ARG]...
or: nohup OPTION
Run COMMAND immune to the shell's teardown, in a new process group.
--help display this help and exit
--version output version information and exit
";
fn report_missing_operand(mut stderr: impl Write) -> ExecutionResult {
let _ = writeln!(stderr, "nohup: missing operand");
ExecutionResult::new(125)
@@ -78,7 +139,57 @@ fn rebuild_command_line(command: &[String]) -> String {
#[cfg(test)]
mod tests {
use super::{rebuild_command_line, report_missing_operand};
use super::{NohupCommand, rebuild_command_line, report_missing_operand};
fn parsed(argv: &[&str]) -> NohupCommand {
NohupCommand::from_argv(argv.iter().map(ToString::to_string).collect())
}
/// `nohup -- cmd args` must run `cmd`; the leading `--` is an option
/// terminator, not part of the operand vector.
#[test]
fn leading_dashdash_ends_options() {
let cmd = parsed(&["--", "sleep", "1"]);
assert!(!cmd.help && !cmd.version);
assert_eq!(cmd.command, ["sleep", "1"]);
}
/// Only the first `--` terminates options: `nohup -- -- x` runs a command
/// literally named `--`, and `nohup -- --help` runs one named `--help`.
#[test]
fn dashdash_protects_operands_including_help() {
assert_eq!(parsed(&["--", "--", "x"]).command, ["--", "x"]);
let cmd = parsed(&["--", "--help"]);
assert!(!cmd.help);
assert_eq!(cmd.command, ["--help"]);
}
/// A mid-command `--` belongs to the operand, never to nohup itself.
#[test]
fn mid_command_dashdash_is_preserved() {
assert_eq!(parsed(&["echo", "a", "--", "b"]).command, ["echo", "a", "--", "b"]);
}
/// `--help`/`--version` used to be executed as commands (exit 127);
/// GNU nohup prints to stdout and exits 0.
#[test]
fn leading_help_and_version_are_options() {
let cmd = parsed(&["--help"]);
assert!(cmd.help && !cmd.version && cmd.command.is_empty());
let cmd = parsed(&["--version"]);
assert!(cmd.version && !cmd.help && cmd.command.is_empty());
}
/// The builtin entry point receives argv including the command name and
/// must skip it before option handling.
#[test]
fn new_skips_command_name() {
use brush_core::builtins::Command as _;
let cmd = NohupCommand::new(["nohup", "--", "sleep", "1"].map(String::from))
.expect("nohup argv parsing is infallible");
assert_eq!(cmd.command, ["sleep", "1"]);
}
#[test]
fn missing_operand_reports_diagnostic_and_exit_code() {
+182 -16
View File
@@ -113,15 +113,15 @@ pub(crate) struct Rg {
no_fixed_strings: bool,
/// Search case-insensitively.
#[arg(short = 'i', long = "ignore-case")]
#[arg(short = 'i', long = "ignore-case", overrides_with_all = ["case_sensitive", "smart_case"])]
ignore_case: bool,
/// Search case-sensitively.
#[arg(short = 's', long = "case-sensitive")]
#[arg(short = 's', long = "case-sensitive", overrides_with_all = ["ignore_case", "smart_case"])]
case_sensitive: bool,
/// Search case-insensitively when the pattern is all lowercase.
#[arg(short = 'S', long = "smart-case")]
#[arg(short = 'S', long = "smart-case", overrides_with_all = ["ignore_case", "case_sensitive"])]
smart_case: bool,
/// Invert matching.
@@ -297,17 +297,21 @@ pub(crate) struct Rg {
context: Option<usize>,
/// Show line numbers.
#[arg(short = 'n', long = "line-number")]
#[arg(short = 'n', long = "line-number", overrides_with = "no_line_number")]
line_number: bool,
/// Suppress line numbers.
#[arg(short = 'N', long = "no-line-number")]
#[arg(short = 'N', long = "no-line-number", overrides_with = "line_number")]
no_line_number: bool,
/// Show column numbers.
#[arg(long = "column")]
#[arg(long = "column", overrides_with = "no_column")]
column: bool,
/// Do not show column numbers.
#[arg(long = "no-column", overrides_with = "column")]
no_column: bool,
/// Show the zero-based byte offset for each result.
#[arg(short = 'b', long = "byte-offset", overrides_with = "no_byte_offset")]
byte_offset: bool,
@@ -464,6 +468,19 @@ pub(crate) struct Rg {
#[arg(long = "no-stats")]
_no_stats: bool,
/// Never read configuration files (accepted; this builtin never reads any).
#[arg(long = "no-config")]
_no_config: bool,
/// Number of search threads (accepted; this builtin searches in-process,
/// serially).
#[arg(short = 'j', long = "threads", value_name = "NUM")]
_threads: Option<usize>,
/// Print SEPARATOR instead of '/' in printed file paths.
#[arg(long = "path-separator", value_name = "SEPARATOR")]
path_separator: Option<String>,
/// Arguments: PATTERN followed by PATHs unless -e/-f/--files is used.
#[arg(value_name = "ARGS")]
args: Vec<OsString>,
@@ -520,6 +537,7 @@ struct SearchOptions {
max_columns: Option<usize>,
max_columns_preview: bool,
null_paths: bool,
path_separator: Option<u8>,
no_messages: bool,
replacement: Option<Vec<u8>>,
json: bool,
@@ -545,7 +563,7 @@ struct RgSink<'a, M: Matcher, W: Write> {
impl<M: Matcher, W: Write> RgSink<'_, M, W> {
fn write_path(&mut self) -> io::Result<()> {
if let Some(name) = self.display {
self.out.write_all(name)?;
write_display_bytes(&mut *self.out, name, self.opts.path_separator)?;
if self.opts.null_paths {
self.out.write_all(b"\0")?;
}
@@ -562,8 +580,10 @@ impl<M: Matcher, W: Write> RgSink<'_, M, W> {
) -> io::Result<()> {
if self.display.is_some() {
self.write_path()?;
if !self.opts.null_paths {
self.out.write_all(&[separator])?;
}
}
if self.opts.line_number
&& let Some(number) = line_number
{
@@ -780,18 +800,24 @@ impl<M: Matcher, W: Write> Sink for RgSink<'_, M, W> {
if self.opts.files_with_matches {
if self.any_match {
self.write_path()?;
if !self.opts.null_paths {
self.out.write_all(b"\n")?;
}
}
} else if self.opts.files_without_match {
if !self.any_match {
self.write_path()?;
if !self.opts.null_paths {
self.out.write_all(b"\n")?;
}
}
} else if self.opts.count || self.opts.count_matches {
if self.display.is_some() {
self.write_path()?;
if !self.opts.null_paths {
self.out.write_all(b":")?;
}
}
let count = if self.opts.count_matches {
self.match_count
} else {
@@ -811,6 +837,42 @@ fn trim_ascii_start(bytes: &[u8]) -> &[u8] {
&bytes[start..]
}
/// Writes a display path, substituting `separator` for `/` when requested via
/// `--path-separator`.
fn write_display_bytes<W: Write>(out: &mut W, bytes: &[u8], separator: Option<u8>) -> io::Result<()> {
let Some(separator) = separator else {
return out.write_all(bytes);
};
let mut rest = bytes;
while let Some(pos) = rest.iter().position(|&byte| byte == b'/') {
out.write_all(&rest[..pos])?;
out.write_all(&[separator])?;
rest = &rest[pos + 1..];
}
out.write_all(rest)
}
fn parse_path_separator(spec: Option<&str>) -> Result<Option<u8>, String> {
match spec {
None | Some("") => Ok(None),
Some(separator) if separator.len() == 1 => Ok(Some(separator.as_bytes()[0])),
Some(separator) => Err(format!(
"error parsing flag --path-separator: a path separator must be exactly one byte, but the given separator is {} bytes",
separator.len()
)),
}
}
/// Line numbers follow real ripgrep's piped behavior: off unless requested
/// (`-n`), or implied by `--column`/`--vimgrep`. Real rg enables them by
/// default only on a tty; the builtin's output is always consumed piped.
fn effective_line_number(cli: &Rg) -> bool {
if cli.no_line_number {
return false;
}
cli.line_number || cli.column || cli.vimgrep
}
fn first_column<M: Matcher>(matcher: &M, line: &[u8]) -> io::Result<Option<usize>> {
Ok(matcher
.find(line)
@@ -843,8 +905,8 @@ fn build_rust_matcher(patterns: &[String], cli: &Rg) -> Result<RegexMatcher, gre
let crlf = cli.crlf && !cli.no_crlf && !cli.null_data;
let mut builder = RegexMatcherBuilder::new();
builder
.case_insensitive(cli.ignore_case && !cli.case_sensitive)
.case_smart(cli.smart_case && !cli.ignore_case && !cli.case_sensitive)
.case_insensitive(cli.ignore_case)
.case_smart(cli.smart_case)
.word(cli.word_regexp && !cli.line_regexp)
.whole_line(cli.line_regexp)
.fixed_strings(cli.fixed_strings && !cli.no_fixed_strings)
@@ -864,8 +926,8 @@ fn build_pcre_matcher(host: &Host, patterns: &[String], cli: &Rg) -> Result<Pcre
let unicode = !cli.no_unicode;
let mut builder = PcreMatcherBuilder::new();
builder
.caseless(cli.ignore_case && !cli.case_sensitive)
.case_smart(cli.smart_case && !cli.ignore_case && !cli.case_sensitive)
.caseless(cli.ignore_case)
.case_smart(cli.smart_case)
.word(cli.word_regexp && !cli.line_regexp)
.whole_line(cli.line_regexp)
.fixed_strings(cli.fixed_strings && !cli.no_fixed_strings)
@@ -1000,6 +1062,9 @@ fn search_options(cli: &Rg) -> SearchOptions {
max_columns: cli.max_columns,
max_columns_preview: cli.max_columns_preview && !cli.no_max_columns_preview,
null_paths: cli.null,
// Validated --path-separator and the line-number default are applied in
// run() once paths are known.
path_separator: None,
no_messages: cli.no_messages && !cli.messages,
replacement: cli
.replacement
@@ -1504,7 +1569,13 @@ fn collect_filtered_files(host: &mut Host, cli: &Rg, root: &Path) -> Result<Vec<
Ok(files)
}
fn list_files<W: Write>(host: &mut Host, cli: &Rg, paths: &[OsString], out: &mut W) -> SearchOutcome {
fn list_files<W: Write>(
host: &mut Host,
cli: &Rg,
paths: &[OsString],
path_separator: Option<u8>,
out: &mut W,
) -> SearchOutcome {
let mut any = false;
let mut had_error = false;
let mut processed_operand = false;
@@ -1534,13 +1605,14 @@ fn list_files<W: Write>(host: &mut Host, cli: &Rg, paths: &[OsString], out: &mut
}
for path in files {
let display = display_path(operand.as_os_str(), &resolved, &path);
let _ = out.write_all(display.as_os_str().as_encoded_bytes());
let _ =
write_display_bytes(out, display.as_os_str().as_encoded_bytes(), path_separator);
let _ = out.write_all(if cli.null { b"\0" } else { b"\n" });
any = true;
}
},
Ok(meta) if meta.is_file() => {
let _ = out.write_all(operand.as_encoded_bytes());
let _ = write_display_bytes(out, operand.as_encoded_bytes(), path_separator);
let _ = out.write_all(if cli.null { b"\0" } else { b"\n" });
any = true;
},
@@ -1745,7 +1817,14 @@ impl Utility for Rg {
fn run(self, host: &mut Host) -> i32 {
let cli = self;
let opts = search_options(&cli);
let mut opts = search_options(&cli);
match parse_path_separator(cli.path_separator.as_deref()) {
Ok(separator) => opts.path_separator = separator,
Err(error) => {
let _ = writeln!(host.stderr, "rg: {error}");
return 2;
},
}
if opts.json
&& (cli.files
|| cli.type_list
@@ -1786,8 +1865,9 @@ impl Utility for Rg {
&mut paths,
!cli.files && !pattern_stdin_consumed && host.stdin_is_search_input(),
);
opts.line_number = effective_line_number(&cli);
if cli.files {
let outcome = list_files(host, &cli, &paths, &mut out);
let outcome = list_files(host, &cli, &paths, opts.path_separator, &mut out);
let _ = out.flush();
return if outcome.had_error {
2
@@ -1930,6 +2010,92 @@ mod tests {
assert_eq!(code, 0, "{}", capture.err());
assert_eq!(capture.out(), "hit\n");
}
#[test]
fn line_numbers_stay_off_when_piped_unless_requested() {
// Defends: the builtin's output is always consumed piped, where real
// rg omits line numbers; `1:` prefixes sprouting by default break
// text consumers. `-n` opts in, `-N` beats `-n`.
let tree = tempfile::tempdir().unwrap();
std::fs::write(tree.path().join("a.txt"), "miss\nhit\n").unwrap();
let (code, capture) = run_util::<Rg>(&["hit", "."], "", tree.path());
assert_eq!(code, 0, "{}", capture.err());
assert_eq!(capture.out(), "a.txt:hit\n");
let (code, capture) = run_util::<Rg>(&["-n", "hit", "."], "", tree.path());
assert_eq!(code, 0, "{}", capture.err());
assert_eq!(capture.out(), "a.txt:2:hit\n");
let (code, capture) = run_util::<Rg>(&["-n", "-N", "hit", "."], "", tree.path());
assert_eq!(code, 0, "{}", capture.err());
assert_eq!(capture.out(), "a.txt:hit\n");
}
#[test]
fn null_terminates_paths_without_trailing_newline() {
// Defends: `rg -l0 | xargs -0` must see `path\0`, not `path\0\n`.
let tree = tempfile::tempdir().unwrap();
std::fs::write(tree.path().join("a.txt"), "hit\n").unwrap();
let (code, capture) = run_util::<Rg>(&["-l0", "hit", "."], "", tree.path());
assert_eq!(code, 0, "{}", capture.err());
assert_eq!(capture.out(), "a.txt\0");
// Match lines: `path\0text` with no `:` after the NUL; with -n the
// line number follows the NUL (`path\0N:text`).
let (code, capture) = run_util::<Rg>(&["-0", "hit", "."], "", tree.path());
assert_eq!(code, 0, "{}", capture.err());
assert_eq!(capture.out(), "a.txt\0hit\n");
let (code, capture) = run_util::<Rg>(&["-n0", "hit", "."], "", tree.path());
assert_eq!(code, 0, "{}", capture.err());
assert_eq!(capture.out(), "a.txt\01:hit\n");
// --files-without-match also emits bare `path\0`. (Exit code for this
// mode is a pre-existing divergence outside this test's contract.)
let (_code, capture) =
run_util::<Rg>(&["--files-without-match", "-0", "nope", "."], "", tree.path());
assert_eq!(capture.out(), "a.txt\0");
}
#[test]
fn last_case_flag_wins() {
// Defends: real rg resolves -s/-i/-S by last occurrence, not by a
// fixed precedence.
let (code, out, err) = run(&["-s", "-i", "HIT", "-"], "hit\n");
assert_eq!(code, 0, "{err}");
assert_eq!(out, "hit\n");
let (code, out, _) = run(&["-i", "-s", "HIT", "-"], "hit\n");
assert_eq!(code, 1);
assert_eq!(out, "");
// -i then -S: smart case wins, and an uppercase pattern stays sensitive.
let (code, _, _) = run(&["-i", "-S", "HIT", "-"], "hit\n");
assert_eq!(code, 1);
}
#[test]
fn compat_flags_are_accepted() {
// Defends: `--no-config`/`-j`/`--threads`/`--no-column` must not be
// clap parse errors.
let (code, out, err) = run(&["--no-config", "-j2", "-S", "hit", "-"], "hit\n");
assert_eq!(code, 0, "{err}");
assert_eq!(out, "hit\n");
let (code, out, err) =
run(&["--threads", "4", "-n", "--column", "--no-column", "hit", "-"], "hit\n");
assert_eq!(code, 0, "{err}");
assert_eq!(out, "1:hit\n");
}
#[test]
fn path_separator_replaces_slash() {
// Defends: --path-separator must rewrite `/` in printed paths and
// reject multi-byte separators like real rg.
let tree = tempfile::tempdir().unwrap();
std::fs::create_dir(tree.path().join("sub")).unwrap();
std::fs::write(tree.path().join("sub/a.txt"), "hit\n").unwrap();
let (code, capture) =
run_util::<Rg>(&["--path-separator", "|", "hit", "."], "", tree.path());
assert_eq!(code, 0, "{}", capture.err());
assert_eq!(capture.out(), "sub|a.txt:hit\n");
let (code, capture) =
run_util::<Rg>(&["--path-separator", "::", "hit", "."], "", tree.path());
assert_eq!(code, 2);
assert!(capture.err().contains("exactly one byte"));
}
}
use brush_core::{ShellExtensions, builtins::Registration};
+72 -5
View File
@@ -12,7 +12,9 @@ use crate::host::parse_duration;
#[derive(Parser)]
#[command(disable_help_flag = true)]
pub(crate) struct SleepCommand {
#[arg(required = true)]
// GNU reports `sleep -1` as an invalid time interval (exit 1), not as an
// unknown option; let hyphenated operands through to `parse_duration`.
#[arg(required = true, allow_hyphen_values = true)]
durations: Vec<String>,
}
@@ -28,13 +30,14 @@ impl builtins::Command for SleepCommand {
if context.is_cancelled() {
return Ok(ExecutionExitCode::Interrupted.into());
}
let mut total = Duration::from_millis(0);
let mut total = Duration::ZERO;
for duration in &durations {
let Some(parsed) = parse_duration(duration) else {
let _ = writeln!(context.stderr(), "sleep: invalid time interval '{duration}'");
return Ok(ExecutionResult::new(1));
};
total += parsed;
// `infinity` parses as `Duration::MAX`; keep the sum saturating.
total = total.saturating_add(parsed);
}
let sleep = time::sleep(total);
tokio::pin!(sleep);
@@ -69,8 +72,10 @@ mod tests {
assert_eq!(parse_duration("0.001"), Some(Duration::from_millis(1)));
assert_eq!(parse_duration("0.001s"), Some(Duration::from_millis(1)));
assert_eq!(parse_duration("0.001m"), Some(Duration::from_millis(60)));
assert_eq!(parse_duration("0.000001h"), Some(Duration::from_millis(4)));
assert_eq!(parse_duration("0.00000001d"), Some(Duration::from_millis(1)));
// Sub-millisecond precision must survive: GNU sleep honors 100µs.
assert_eq!(parse_duration("0.0001"), Some(Duration::from_micros(100)));
assert_eq!(parse_duration("0.000001h"), Some(Duration::from_micros(3600)));
assert_eq!(parse_duration("0.00000001d"), Some(Duration::from_micros(864)));
let mut shell = Shell::builder().build().await.expect("test shell should build");
let command = SleepCommand { durations: vec!["0.001".into(), "0.001s".into()] };
@@ -87,6 +92,68 @@ mod tests {
assert!(result.is_success());
}
#[tokio::test]
async fn infinity_operand_parses_and_sleep_is_cancellable() {
// GNU accepts `inf`/`infinity`, any case, with an optional `+` sign.
for spec in ["infinity", "inf", "INFINITY", "Inf", "+infinity", "+inf"] {
assert_eq!(parse_duration(spec), Some(Duration::MAX), "spec {spec:?}");
}
assert_eq!(parse_duration("nan"), None);
assert_eq!(parse_duration("-inf"), None);
// `sleep infinity` must block until cancelled rather than erroring out.
let token = CancellationToken::new();
let mut params = ExecutionParameters::default();
params.set_cancel_token(token.clone());
let mut shell = Shell::builder().build().await.expect("test shell should build");
let command = SleepCommand { durations: vec!["infinity".into()] };
let context = ExecutionContext {
shell: &mut shell,
command_name: "sleep".into(),
params,
};
let execution = async {
let (result, ()) = tokio::join!(command.execute(context), async {
tokio::task::yield_now().await;
token.cancel();
});
result
};
let result = time::timeout(Duration::from_millis(100), execution)
.await
.expect("cancelled infinite sleep should return promptly")
.expect("sleep execution should succeed");
assert_eq!(
u8::from(result.exit_code),
u8::from(ExecutionExitCode::Interrupted)
);
}
#[tokio::test]
async fn hyphenated_operand_is_an_invalid_interval_not_an_unknown_flag() {
// `sleep -1` must not die in clap with an unknown-option error; GNU
// reports an invalid time interval and exits 1.
let command = <SleepCommand as Command>::new(["sleep".into(), "-1".into()])
.expect("hyphenated operand should reach the builtin");
let (mut stderr_reader, stderr_writer) = std::io::pipe().expect("stderr pipe should open");
let mut params = ExecutionParameters::default();
params.set_fd(OpenFiles::STDERR_FD, OpenFile::from(stderr_writer));
let mut shell = Shell::builder().build().await.expect("test shell should build");
let context = ExecutionContext {
shell: &mut shell,
command_name: "sleep".into(),
params,
};
let result = command.execute(context).await.expect("sleep execution should succeed");
let mut stderr = String::new();
stderr_reader.read_to_string(&mut stderr).expect("stderr should be readable");
assert_eq!(u8::from(result.exit_code), 1);
assert_eq!(stderr, "sleep: invalid time interval '-1'\n");
}
#[tokio::test]
async fn invalid_duration_reports_original_diagnostic_and_exit_code() {
let (mut stderr_reader, stderr_writer) = std::io::pipe().expect("stderr pipe should open");
+439 -114
View File
@@ -157,7 +157,8 @@ for details about the options it supports.";
pub const FORMAT: &str = "format";
pub const PRINTF: &str = "printf";
pub const TERSE: &str = "terse";
pub const BSD_TIME_WARNING: &str = "bsd-time-warning";
pub const BSD_SHELL: &str = "bsd-shell";
pub const BSD_TIMEFMT: &str = "bsd-timefmt";
pub const FILES: &str = "files";
}
@@ -267,7 +268,7 @@ for details about the options it supports.";
Unsigned(u64),
UnsignedHex(u64),
UnsignedOct(u32),
Float(f64),
Timestamp { sec: i64, nsec: u32 },
Unknown,
}
@@ -400,6 +401,7 @@ for details about the options it supports.";
show_fs: bool,
from_user: bool,
files: Vec<OsString>,
time_format: Option<String>,
#[cfg_attr(not(unix), allow(dead_code))]
mount_list: OnceCell<Option<Vec<OsString>>>,
#[cfg_attr(not(unix), allow(dead_code))]
@@ -479,8 +481,8 @@ for details about the options it supports.";
OutputType::UnsignedHex(num) => {
print_unsigned_hex(out, *num, flags, width, precision, padding_char);
},
OutputType::Float(num) => {
print_float(out, *num, flags, width, precision, padding_char);
OutputType::Timestamp { sec, nsec } => {
print_timestamp(out, *sec, *nsec, flags, width, precision, padding_char);
},
OutputType::Unknown => {
let _ = write!(out, "?");
@@ -698,48 +700,26 @@ for details about the options it supports.";
pad_and_print(out, &extended, flags.left, width, padding_char);
}
/// Truncate a float to the given number of digits after the decimal point.
fn precision_trunc(num: f64, precision: Precision) -> String {
// GNU `stat` doesn't round, it just seems to truncate to the
// given precision:
//
// $ stat -c "%.5Y" /dev/pts/ptmx
// 1736344012.76399
// $ stat -c "%.4Y" /dev/pts/ptmx
// 1736344012.7639
// $ stat -c "%.3Y" /dev/pts/ptmx
// 1736344012.763
//
// Contrast this with `printf`, which seems to round the
// numbers:
//
// $ printf "%.5f\n" 1736344012.76399
// 1736344012.76399
// $ printf "%.4f\n" 1736344012.76399
// 1736344012.7640
// $ printf "%.3f\n" 1736344012.76399
// 1736344012.764
//
let num_str = num.to_string();
let n = num_str.len();
match (num_str.find('.'), precision) {
(None, Precision::NotSpecified) => num_str,
(None, Precision::NoNumber) => num_str,
(None, Precision::Number(0)) => num_str,
(None, Precision::Number(p)) => format!("{num_str}.{zeros}", zeros = "0".repeat(p)),
(Some(i), Precision::NotSpecified) => num_str[..i].to_string(),
(Some(_), Precision::NoNumber) => num_str,
(Some(i), Precision::Number(0)) => num_str[..i].to_string(),
(Some(i), Precision::Number(p)) if p < n - i => num_str[..i + 1 + p].to_string(),
(Some(i), Precision::Number(p)) => {
format!("{num_str}{zeros}", zeros = "0".repeat(p - (n - i - 1)))
/// Formats an epoch timestamp with GNU `stat`'s truncation rules: no
/// precision prints whole seconds (so `stat -c %Y` survives shell
/// arithmetic), a bare `.` prints all nine fractional digits, and an
/// explicit precision truncates or zero-pads the fraction.
fn timestamp_string(sec: i64, nsec: u32, precision: Precision) -> String {
match precision {
Precision::NotSpecified | Precision::Number(0) => sec.to_string(),
Precision::NoNumber => format!("{sec}.{nsec:09}"),
Precision::Number(p) if p <= 9 => {
let frac = format!("{nsec:09}");
format!("{sec}.{}", &frac[..p])
},
Precision::Number(p) => format!("{sec}.{nsec:09}{:0<pad$}", "", pad = p - 9),
}
}
fn print_float(
fn print_timestamp(
out: &mut dyn Write,
num: f64,
sec: i64,
nsec: u32,
flags: Flags,
width: usize,
precision: Precision,
@@ -752,8 +732,7 @@ for details about the options it supports.";
} else {
""
};
let num_str = precision_trunc(num, precision);
let extended = format!("{prefix}{num_str}");
let extended = format!("{prefix}{}", timestamp_string(sec, nsec, precision));
pad_and_print(out, &extended, flags.left, width, padding_char);
}
@@ -1128,6 +1107,7 @@ for details about the options it supports.";
show_fs,
from_user: !format_str.is_empty(),
files,
time_format: matches.get_one::<String>(options::BSD_TIMEFMT).cloned(),
mount_list: OnceCell::new(),
mount_list_needed,
default_tokens,
@@ -1135,6 +1115,12 @@ for details about the options it supports.";
})
}
/// The `strftime` format for human-readable time directives; BSD `-t`
/// overrides the GNU default.
fn time_fmt(&self) -> &str {
self.time_format.as_deref().unwrap_or(PRETTY_DATETIME_FORMAT)
}
#[cfg(unix)]
fn find_mount_point<P: AsRef<Path>>(
&self,
@@ -1273,37 +1259,36 @@ for details about the options it supports.";
},
// time of file birth, human-readable; - if unknown
'w' => OutputType::Str(pretty_time(meta, MetadataTimeField::Birth)),
'w' => OutputType::Str(pretty_time(meta, MetadataTimeField::Birth, self.time_fmt())),
// time of file birth, seconds since Epoch; 0 if unknown
'W' => OutputType::Integer(
metadata_get_time(meta, MetadataTimeField::Birth)
.map_or(0, |x| system_time_to_sec(x).0),
),
'W' => {
let (sec, nsec) = metadata_get_time(meta, MetadataTimeField::Birth)
.map_or((0, 0), system_time_to_sec);
OutputType::Timestamp { sec, nsec }
},
// time of last access, human-readable
'x' => OutputType::Str(pretty_time(meta, MetadataTimeField::Access)),
'x' => OutputType::Str(pretty_time(meta, MetadataTimeField::Access, self.time_fmt())),
// time of last access, seconds since Epoch
'X' => {
let (sec, nsec) = metadata_get_time(meta, MetadataTimeField::Access)
.map_or((0, 0), system_time_to_sec);
OutputType::Float(sec as f64 + nsec as f64 / 1_000_000_000.0)
OutputType::Timestamp { sec, nsec }
},
// time of last data modification, human-readable
'y' => OutputType::Str(pretty_time(meta, MetadataTimeField::Modification)),
'y' => OutputType::Str(pretty_time(meta, MetadataTimeField::Modification, self.time_fmt())),
// time of last data modification, seconds since Epoch
'Y' => {
let (sec, nsec) = metadata_get_time(meta, MetadataTimeField::Modification)
.map_or((0, 0), system_time_to_sec);
OutputType::Float(sec as f64 + nsec as f64 / 1_000_000_000.0)
OutputType::Timestamp { sec, nsec }
},
// time of last status change, human-readable
'z' => OutputType::Str(pretty_time(meta, MetadataTimeField::Change)),
'z' => OutputType::Str(pretty_time(meta, MetadataTimeField::Change, self.time_fmt())),
// time of last status change, seconds since Epoch
'Z' => {
let (sec, nsec) = metadata_get_time(meta, MetadataTimeField::Change)
.map_or((0, 0), system_time_to_sec);
OutputType::Float(sec as f64 + nsec as f64 / 1_000_000_000.0)
OutputType::Timestamp { sec, nsec }
},
'R' => OutputType::UnsignedHex(meta.rdev()),
'r' if flag.major => OutputType::Unsigned(major(meta.rdev() as _) as u64),
@@ -1442,11 +1427,13 @@ for details about the options it supports.";
/// GNU's `-f` is `--file-system`; parsed as GNU, a BSD invocation prints
/// filesystem info for each real operand and errors on the format operand.
/// An invocation is treated as BSD when a `-f` cluster (optionally with the
/// BSD boolean flags `L`/`n`/`q`/`F`) carries a format value containing
/// `%` — GNU filesystem mode would have to target a file literally named
/// like a format string, which never happens in practice. Detected
/// BSD boolean flags `L`/`n`/`q`/`F`/`s`/`x`) carries a format value
/// containing `%` — GNU filesystem mode would have to target a file
/// literally named like a format string, which never happens in practice —
/// or when a cluster of BSD boolean flags contains the BSD-only output
/// styles `-s` (shell assignments) or `-x` (Linux-like verbose). Detected
/// invocations are rewritten to the GNU equivalent (`-c`/`--printf` plus a
/// translated format) before clap parsing.
/// translated format, or hidden style/timefmt options) before clap parsing.
///
/// Returns `None` when the invocation is not BSD-shaped, `Some(Err(_))`
/// when it is BSD-shaped but uses an option or directive with no GNU
@@ -1464,12 +1451,22 @@ for details about the options it supports.";
if cluster.is_empty() || cluster.starts_with('-') {
continue;
}
// `-s` / `-x` are BSD-only output styles: a cluster of BSD boolean
// flags containing one marks the invocation (GNU stat has neither).
if cluster
.chars()
.all(|c| matches!(c, 'L' | 'n' | 'q' | 'F' | 's' | 'x'))
&& cluster.chars().any(|c| matches!(c, 's' | 'x'))
{
detected = true;
break;
}
let Some(fpos) = cluster.find('f') else {
continue;
};
if !cluster[..fpos]
.chars()
.all(|c| matches!(c, 'L' | 'n' | 'q' | 'F'))
.all(|c| matches!(c, 'L' | 'n' | 'q' | 'F' | 's' | 'x'))
{
continue;
}
@@ -1490,12 +1487,22 @@ for details about the options it supports.";
Some(bsd_to_gnu_argv(argv, &toks))
}
/// Output style selected by a BSD invocation.
enum BsdStyle {
/// `-f <fmt>`: caller-supplied BSD format string.
Custom(String),
/// `-s`: eval-able `st_dev=… st_ino=…` shell assignments.
Shell,
/// `-x`: Linux-like verbose block.
Verbose,
}
/// Parses a detected BSD invocation and produces the equivalent GNU argv.
fn bsd_to_gnu_argv(argv: &[OsString], toks: &[Cow<'_, str>]) -> Result<Vec<OsString>, String> {
let mut follow = false;
let mut no_newline = false;
let mut format = None;
let mut timefmt_ignored = false;
let mut style: Option<BsdStyle> = None;
let mut timefmt: Option<String> = None;
let mut files: Vec<OsString> = Vec::new();
let mut i = 1;
@@ -1522,6 +1529,9 @@ for details about the options it supports.";
// `-q` (suppress error messages) and `-F` (ls -F type
// decorations) have no GNU counterpart worth emulating.
'q' | 'F' => {},
// Output styles; like BSD, the last one seen wins.
's' => style = Some(BsdStyle::Shell),
'x' => style = Some(BsdStyle::Verbose),
c @ ('f' | 't') => {
// The rest of the cluster is the attached value,
// otherwise the next token is.
@@ -1535,9 +1545,9 @@ for details about the options it supports.";
}
};
if c == 'f' {
format = Some(value);
style = Some(BsdStyle::Custom(value));
} else {
timefmt_ignored = true;
timefmt = Some(value);
}
break;
},
@@ -1552,27 +1562,37 @@ for details about the options it supports.";
i += 1 + usize::from(consumed_next);
}
let Some(format) = format else {
return Err("BSD-style '-f' expects a format string".to_string());
};
let translated = translate_bsd_format(&format, no_newline)?;
let mut out: Vec<OsString> = Vec::with_capacity(files.len() + 5);
let mut out: Vec<OsString> = Vec::with_capacity(files.len() + 7);
out.push(argv[0].clone());
if follow {
out.push("-L".into());
}
if timefmt_ignored {
out.push("--bsd-time-warning".into());
match style {
// `-s` renders directly from the metadata (the full octal
// `st_mode` and `st_flags` have no GNU format directive); its
// timestamps are epoch integers regardless of `-t`, as on BSD.
Some(BsdStyle::Shell) => out.push("--bsd-shell".into()),
Some(BsdStyle::Verbose) => {
out.push("--bsd-timefmt".into());
out.push(timefmt.unwrap_or_else(|| BSD_VERBOSE_TIMEFMT.into()).into());
out.push(if no_newline { "--printf".into() } else { "-c".into() });
out.push(BSD_VERBOSE_FORMAT.into());
},
Some(BsdStyle::Custom(format)) => {
let translated = translate_bsd_format(&format, no_newline)?;
if let Some(timefmt) = timefmt {
out.push("--bsd-timefmt".into());
out.push(timefmt.into());
}
// `--printf` suppresses the mandatory trailing newline (BSD `-n`); the
// translator escapes literal backslashes so text survives printf mode.
out.push(if no_newline {
"--printf".into()
} else {
"-c".into()
});
// `--printf` suppresses the mandatory trailing newline (BSD
// `-n`); the translator escapes literal backslashes so text
// survives printf mode.
out.push(if no_newline { "--printf".into() } else { "-c".into() });
out.push(translated.into());
},
None => return Err("BSD-style '-f' expects a format string".to_string()),
}
out.push("--".into());
out.extend(files);
Ok(out)
}
@@ -1739,6 +1759,154 @@ for details about the options it supports.";
format!("unsupported BSD format directive '{directive}'")
}
/// GNU-language rendering of BSD `stat -x` ("Linux-like" verbose output).
const BSD_VERBOSE_FORMAT: &str = concat!(
" File: \"%n\"\n",
" Size: %-11s FileType: %F\n",
" Mode: (%04a/%.10A) Uid: (%5u/%8U) Gid: (%5g/%8G)\n",
"Device: %Hd,%Ld Inode: %i Links: %h\n",
"Access: %x\n",
"Modify: %y\n",
"Change: %z\n",
" Birth: %w",
);
/// BSD `stat -x` renders timestamps `ctime(3)`-style.
const BSD_VERBOSE_TIMEFMT: &str = "%a %b %e %H:%M:%S %Y";
/// BSD `stat -s`: one eval-able line of `st_*=value` shell assignments per
/// file, rendered directly from the metadata.
fn bsd_shell_exec(matches: &ArgMatches, host: &mut Host) -> i32 {
let files: Vec<OsString> = matches
.get_many::<OsString>(options::FILES)
.map(|v| v.cloned().collect())
.unwrap_or_default();
if files.is_empty() {
host.error(StatError::MissingOperand, 1);
return 1;
}
let follow = matches.get_flag(options::DEREFERENCE);
let mut ret = 0;
for file in &files {
let display_name = file.to_string_lossy();
let resolved = host.resolve(file);
let result = if follow {
fs::metadata(&resolved)
} else {
fs::symlink_metadata(&resolved)
};
match result {
Ok(meta) => {
let _ = writeln!(host.stdout, "{}", bsd_shell_line(&meta, &resolved));
},
Err(e) => {
let _ = writeln!(&mut host.stderr, "stat: {}", StatError::CannotStat {
file: display_name.quote().to_string(),
error: e.to_string(),
});
ret = 1;
},
}
}
ret
}
#[cfg(unix)]
fn bsd_shell_line(meta: &Metadata, _resolved: &Path) -> String {
#[cfg(target_os = "macos")]
let flags = std::os::macos::fs::MetadataExt::st_flags(meta);
#[cfg(not(target_os = "macos"))]
let flags = 0u32;
let birth = metadata_get_time(meta, MetadataTimeField::Birth)
.map_or(0, |t| system_time_to_sec(t).0);
format!(
"st_dev={} st_ino={} st_mode=0{:o} st_nlink={} st_uid={} st_gid={} st_rdev={} \
st_size={} st_atime={} st_mtime={} st_ctime={} st_birthtime={birth} st_blksize={} \
st_blocks={} st_flags={flags}",
meta.dev(),
meta.ino(),
meta.mode(),
meta.nlink(),
meta.uid(),
meta.gid(),
meta.rdev(),
meta.len(),
meta.atime(),
meta.mtime(),
meta.ctime(),
meta.blksize(),
meta.blocks(),
)
}
#[cfg(windows)]
fn bsd_shell_line(meta: &Metadata, resolved: &Path) -> String {
let ids = win::handle_info(resolved, !meta.file_type().is_symlink());
let (dev, ino, nlink) = ids.map_or((0, 0, 1), |i| (i.volume_serial, i.file_index, i.links));
let sec = |field| win::md_time(meta, field).map_or(0, |t| system_time_to_sec(t).0);
format!(
"st_dev={dev} st_ino={ino} st_mode=0{:o} st_nlink={nlink} st_uid=0 st_gid=0 st_rdev=0 \
st_size={} st_atime={} st_mtime={} st_ctime={} st_birthtime={} st_blksize=4096 \
st_blocks={} st_flags=0",
win::synth_mode(meta),
meta.len(),
sec(win::TimeField::Access),
sec(win::TimeField::Modification),
sec(win::TimeField::Change),
sec(win::TimeField::Birth),
win::allocated_size(resolved, meta.len()).div_ceil(512),
)
}
/// GNU `-f`/`--file-system` whose first operand names no file but looks
/// like a BSD format string (contains `%` or whitespace): rather than
/// failing on a nonexistent operand, re-interpret the invocation as BSD
/// `stat -f <fmt> <file>...`. Existing-path operands always keep GNU
/// filesystem mode.
fn bsd_filesystem_fallback(matches: &ArgMatches, host: &Host) -> Option<Vec<OsString>> {
if !matches.get_flag(options::FILE_SYSTEM)
|| matches.contains_id(options::FORMAT)
|| matches.contains_id(options::PRINTF)
{
return None;
}
let files: Vec<&OsString> = matches.get_many::<OsString>(options::FILES)?.collect();
// A format plus at least one operand; a lone missing path stays a GNU
// error.
if files.len() < 2 {
return None;
}
let fmt = files[0].to_string_lossy();
if !(fmt.contains('%') || fmt.chars().any(char::is_whitespace)) {
return None;
}
if host.resolve(files[0]).symlink_metadata().is_ok() {
return None;
}
let translated = translate_bsd_format(&fmt, false).ok()?;
let mut argv: Vec<OsString> = Vec::with_capacity(files.len() + 4);
argv.push("stat".into());
if matches.get_flag(options::DEREFERENCE) {
argv.push("-L".into());
}
argv.push("-c".into());
argv.push(translated.into());
argv.push("--".into());
argv.extend(files[1..].iter().map(|f| (*f).clone()));
Some(argv)
}
/// Builds a [`Stater`] from parsed matches and runs it.
fn run_stater(matches: &ArgMatches, host: &mut Host) -> i32 {
match Stater::new(matches, host) {
Ok(stater) => stater.exec(host),
Err(error) => {
host.error(error, 1);
1
},
}
}
/// Parsed `stat` invocation.
pub(crate) struct Stat {
@@ -1758,20 +1926,21 @@ for details about the options it supports.";
}
fn run(self, host: &mut Host) -> i32 {
if self.matches.get_flag(options::BSD_TIME_WARNING) {
let _ = writeln!(
host.stderr,
"stat: warning: BSD '-t' time format is ignored; human-readable times use the GNU \
default format"
);
if self.matches.get_flag(options::BSD_SHELL) {
return bsd_shell_exec(&self.matches, host);
}
match Stater::new(&self.matches, host) {
Ok(stater) => stater.exec(host),
Err(error) => {
host.error(error, 1);
if let Some(argv) = bsd_filesystem_fallback(&self.matches, host) {
return match app().try_get_matches_from(argv) {
Ok(matches) => run_stater(&matches, host),
// The rebuilt argv is a plain `-c FORMAT -- FILE...`; a
// parse failure here is unreachable in practice.
Err(err) => {
let _ = write!(host.stderr, "{err}");
1
},
};
}
run_stater(&self.matches, host)
}
}
@@ -1804,11 +1973,17 @@ for details about the options it supports.";
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::BSD_TIME_WARNING)
.long(options::BSD_TIME_WARNING)
Arg::new(options::BSD_SHELL)
.long(options::BSD_SHELL)
.hide(true)
.action(ArgAction::SetTrue),
)
.arg(
Arg::new(options::BSD_TIMEFMT)
.long(options::BSD_TIMEFMT)
.value_name("TIMEFMT")
.hide(true),
)
.arg(
Arg::new(options::FORMAT)
.short('c')
@@ -1839,13 +2014,13 @@ for details about the options it supports.";
const PRETTY_DATETIME_FORMAT: &str = "%Y-%m-%d %H:%M:%S.%N %z";
#[cfg(unix)]
fn pretty_time(meta: &Metadata, md_time_field: MetadataTimeField) -> String {
fn pretty_time(meta: &Metadata, md_time_field: MetadataTimeField, fmt: &str) -> String {
if let Some(time) = metadata_get_time(meta, md_time_field) {
let mut tmp = Vec::new();
if format_system_time(
&mut tmp,
time,
PRETTY_DATETIME_FORMAT,
fmt,
FormatSystemTimeFallback::Float,
)
.is_ok()
@@ -1860,7 +2035,7 @@ for details about the options it supports.";
/// most intricate part of the utility and the print paths were repatched.
#[cfg(test)]
mod unit_tests {
use super::{Flags, Precision, ScanUtil, Stater, Token, group_num, precision_trunc};
use super::{Flags, Precision, ScanUtil, Stater, Token, group_num, timestamp_string};
#[test]
fn test_scanners() {
@@ -1940,12 +2115,22 @@ for details about the options it supports.";
}
#[test]
fn test_precision_trunc() {
assert_eq!(precision_trunc(123.456, Precision::NotSpecified), "123");
assert_eq!(precision_trunc(123.456, Precision::NoNumber), "123.456");
assert_eq!(precision_trunc(123.456, Precision::Number(0)), "123");
assert_eq!(precision_trunc(123.456, Precision::Number(1)), "123.4");
assert_eq!(precision_trunc(123.456, Precision::Number(5)), "123.45600");
fn test_timestamp_string() {
// `stat -c %Y` must yield integers so shell arithmetic works.
assert_eq!(timestamp_string(1712345678, 999_999_999, Precision::NotSpecified), "1712345678");
assert_eq!(timestamp_string(1712345678, 123_456_789, Precision::Number(0)), "1712345678");
// `%.Y` prints all nine fractional digits; explicit precision
// truncates (GNU semantics) or zero-pads past nine.
assert_eq!(
timestamp_string(1712345678, 123_456_789, Precision::NoNumber),
"1712345678.123456789"
);
assert_eq!(timestamp_string(1712345678, 123_456_789, Precision::Number(3)), "1712345678.123");
assert_eq!(timestamp_string(1712345678, 5, Precision::Number(3)), "1712345678.000");
assert_eq!(
timestamp_string(1712345678, 123_456_789, Precision::Number(11)),
"1712345678.12345678900"
);
}
}
/// file-status path is Unix-only (`std::os::unix`); this reimplements the
@@ -2205,13 +2390,13 @@ for details about the options it supports.";
/// `std::fs::Metadata` timestamp through the shared datetime format.
#[cfg(windows)]
fn pretty_time(meta: &Metadata, field: win::TimeField) -> String {
fn pretty_time(meta: &Metadata, field: win::TimeField, fmt: &str) -> String {
if let Some(time) = win::md_time(meta, field) {
let mut tmp = Vec::new();
if format_system_time(
&mut tmp,
time,
PRETTY_DATETIME_FORMAT,
fmt,
FormatSystemTimeFallback::Float,
)
.is_ok()
@@ -2352,35 +2537,36 @@ for details about the options it supports.";
// user name of owner
'U' => OutputType::Str("UNKNOWN".to_string()),
// time of file birth, human-readable; - if unknown
'w' => OutputType::Str(pretty_time(meta, win::TimeField::Birth)),
'w' => OutputType::Str(pretty_time(meta, win::TimeField::Birth, self.time_fmt())),
// time of file birth, seconds since Epoch; 0 if unknown
'W' => OutputType::Integer(
win::md_time(meta, win::TimeField::Birth)
.map_or(0, |x| system_time_to_sec(x).0),
),
'W' => {
let (sec, nsec) = win::md_time(meta, win::TimeField::Birth)
.map_or((0, 0), system_time_to_sec);
OutputType::Timestamp { sec, nsec }
},
// time of last access, human-readable
'x' => OutputType::Str(pretty_time(meta, win::TimeField::Access)),
'x' => OutputType::Str(pretty_time(meta, win::TimeField::Access, self.time_fmt())),
// time of last access, seconds since Epoch
'X' => {
let (sec, nsec) = win::md_time(meta, win::TimeField::Access)
.map_or((0, 0), system_time_to_sec);
OutputType::Float(sec as f64 + nsec as f64 / 1_000_000_000.0)
OutputType::Timestamp { sec, nsec }
},
// time of last data modification, human-readable
'y' => OutputType::Str(pretty_time(meta, win::TimeField::Modification)),
'y' => OutputType::Str(pretty_time(meta, win::TimeField::Modification, self.time_fmt())),
// time of last data modification, seconds since Epoch
'Y' => {
let (sec, nsec) = win::md_time(meta, win::TimeField::Modification)
.map_or((0, 0), system_time_to_sec);
OutputType::Float(sec as f64 + nsec as f64 / 1_000_000_000.0)
OutputType::Timestamp { sec, nsec }
},
// time of last status change, human-readable (write time)
'z' => OutputType::Str(pretty_time(meta, win::TimeField::Change)),
'z' => OutputType::Str(pretty_time(meta, win::TimeField::Change, self.time_fmt())),
// time of last status change, seconds since Epoch
'Z' => {
let (sec, nsec) = win::md_time(meta, win::TimeField::Change)
.map_or((0, 0), system_time_to_sec);
OutputType::Float(sec as f64 + nsec as f64 / 1_000_000_000.0)
OutputType::Timestamp { sec, nsec }
},
// rdev (no device special files on Windows)
'R' => OutputType::UnsignedHex(0),
@@ -2656,6 +2842,145 @@ mod tests {
"unexpected stderr: {stderr:?}"
);
}
#[test]
fn epoch_time_specifiers_print_integers() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("data.bin"), b"x").unwrap();
// Regression: `%X`/`%Y`/`%Z` printed floats, which broke shell
// arithmetic like `$(($(stat -c %Y a) - $(stat -c %Y b)))`.
let (code, stdout, stderr) = run_in(root, vec!["-c", "%X %Y %Z %W", "data.bin"]);
assert_eq!(code, 0);
assert_eq!(stderr, "");
let fields: Vec<&str> = stdout.split_whitespace().collect();
assert_eq!(fields.len(), 4, "unexpected stdout: {stdout:?}");
for field in fields {
assert!(field.parse::<i64>().is_ok(), "epoch fields must be integers: {stdout:?}");
}
}
#[test]
fn epoch_time_precision_prints_fraction() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("data.bin"), b"x").unwrap();
// `%.3Y` keeps three fractional digits; bare `%.Y` prints all nine.
let (code, stdout, _) = run_in(root.clone(), vec!["-c", "%.3Y", "data.bin"]);
assert_eq!(code, 0);
let (sec, frac) = stdout.trim_end().split_once('.').expect("fraction expected");
assert!(sec.parse::<i64>().is_ok(), "unexpected stdout: {stdout:?}");
assert_eq!(frac.len(), 3, "unexpected stdout: {stdout:?}");
let (_, stdout, _) = run_in(root, vec!["-c", "%.Y", "data.bin"]);
let (_, frac) = stdout.trim_end().split_once('.').expect("fraction expected");
assert_eq!(frac.len(), 9, "unexpected stdout: {stdout:?}");
}
#[test]
fn bsd_shell_format_prints_evalable_assignments() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("data.bin"), b"hello world!").unwrap();
// BSD `stat -s`: one line of `st_*=value` pairs, eval-able in sh.
let (code, stdout, stderr) = run_in(root, vec!["-s", "data.bin"]);
assert_eq!(code, 0);
assert_eq!(stderr, "");
assert_eq!(stdout.lines().count(), 1, "one line per file: {stdout:?}");
let keys: Vec<&str> = stdout
.split_whitespace()
.map(|pair| pair.split_once('=').expect("key=value pair").0)
.collect();
assert_eq!(keys, [
"st_dev",
"st_ino",
"st_mode",
"st_nlink",
"st_uid",
"st_gid",
"st_rdev",
"st_size",
"st_atime",
"st_mtime",
"st_ctime",
"st_birthtime",
"st_blksize",
"st_blocks",
"st_flags",
]);
assert!(stdout.contains(" st_size=12 "), "unexpected stdout: {stdout:?}");
let mode = stdout
.split_whitespace()
.find_map(|pair| pair.strip_prefix("st_mode="))
.unwrap();
assert!(mode.starts_with('0'), "octal mode with leading zero: {stdout:?}");
assert!(u32::from_str_radix(mode, 8).is_ok(), "octal mode: {stdout:?}");
}
#[test]
fn bsd_verbose_format_prints_linux_like_block() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("data.bin"), b"hello world!").unwrap();
let (code, stdout, stderr) = run_in(root, vec!["-x", "data.bin"]);
assert_eq!(code, 0);
assert_eq!(stderr, "");
assert!(stdout.starts_with(" File: \"data.bin\"\n"), "unexpected stdout: {stdout:?}");
assert!(stdout.contains("FileType:"), "unexpected stdout: {stdout:?}");
assert!(stdout.contains(" Mode: (0"), "unexpected stdout: {stdout:?}");
// ctime(3)-style timestamps: "Access: Wed Aug 20 10:11:12 2026".
let access = stdout.lines().find(|l| l.starts_with("Access: ")).unwrap();
let year = access.rsplit(' ').next().unwrap();
assert_eq!(year.len(), 4, "ctime-style year expected: {access:?}");
assert!(year.parse::<u32>().is_ok(), "ctime-style year expected: {access:?}");
}
#[test]
fn bsd_dash_f_size_format_prints_size() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("data.bin"), b"hello world!").unwrap();
// Acceptance: BSD `stat -f '%z bytes' file`.
let (code, stdout, stderr) = run_in(root, vec!["-f", "%z bytes", "data.bin"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "12 bytes\n", ""));
}
#[test]
fn bsd_dash_t_timefmt_formats_times() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("data.bin"), b"x").unwrap();
// BSD `-t` supplies the strftime format for `%Sm`-style directives;
// this used to be ignored with a warning.
let (code, stdout, stderr) = run_in(root, vec!["-f", "%Sm", "-t", "%Y", "data.bin"]);
assert_eq!(code, 0);
assert_eq!(stderr, "");
let year: u32 = stdout.trim_end().parse().expect("year only");
assert!((1970..=9999).contains(&year), "unexpected stdout: {stdout:?}");
}
#[test]
fn gnu_filesystem_mode_keeps_existing_path_operands() {
let (_dir, root) = canonical_tempdir();
// `stat -f <existing path>` stays GNU `--file-system` mode.
let (code, stdout, stderr) = run_in(root, vec!["-f", "."]);
assert_eq!(code, 0);
assert_eq!(stderr, "");
assert!(stdout.contains("Namelen:"), "filesystem block expected: {stdout:?}");
}
#[test]
fn bsd_dash_f_fallback_on_nonexistent_format_like_operand() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("data.bin"), b"x").unwrap();
// No `%` directive, but the operand names no file and looks like a
// format string: BSD semantics print it literally instead of failing
// with a filesystem error on a nonexistent operand.
let (code, stdout, stderr) = run_in(root, vec!["-f", "no percent here", "data.bin"]);
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "no percent here\n", ""));
}
}
#[cfg(all(test, windows))]
+333 -98
View File
@@ -2875,101 +2875,109 @@ pub(crate) fn map_output_error(error: io::Error) -> TailError {
error.into()
}
fn rewrite_tail_argv(mut argv: Vec<OsString>) -> Result<Vec<OsString>, String> {
let mut has_reverse = false;
let mut incompatible = false;
let mut unsupported = None;
for arg in argv.iter().skip(1) {
let token = arg.to_string_lossy();
if token == "--" {
break;
/// True when `token` is an option that takes its value from the *next* argv
/// token, so that value must never be mistaken for an obsolete `-N`/`+N` form
/// (e.g. the `+5` in `tail -n +5 file`).
fn consumes_separate_value(token: &str) -> bool {
if let Some(long) = token.strip_prefix("--") {
if long.is_empty() || long.contains('=') {
return false;
}
// clap infers unambiguous long-option prefixes; `--follow` requires
// `=` for its value and never consumes the next token.
return ["lines", "bytes", "pid", "sleep-interval", "max-unchanged-stats"]
.iter()
.any(|name| name.starts_with(long));
}
let Some(cluster) = token.strip_prefix('-') else {
continue;
return false;
};
if cluster.is_empty() {
continue;
}
if cluster.starts_with('-') {
if token == "--reverse" {
has_reverse = true;
} else {
unsupported = Some(token.into_owned());
}
continue;
}
for flag in cluster.chars() {
match flag {
'r' => has_reverse = true,
'n' | 'c' | 'b' | 'f' => incompatible = true,
_ => unsupported = Some(format!("-{flag}")),
let mut chars = cluster.chars();
while let Some(c) = chars.next() {
match c {
// Value-taking shorts: a trailing `-n`/`-c`/`-s` consumes the next
// token; anything after them in the cluster is an attached value.
'n' | 'c' | 's' => return chars.next().is_none(),
'q' | 'v' | 'z' | 'f' | 'F' | 'r' | 'b' => {},
_ => return false,
}
}
false
}
if has_reverse && incompatible {
return Err(
"-r with -n, -c, -b, or -f is not supported by this builtin; pipe through tac"
.to_owned(),
);
/// Rewrites every obsolete `-N[bcl][f]` / `+N[bcl][f]` token (before `--`)
/// into modern options, wherever it appears among flags and operands: GNU/BSD
/// accept `tail -20 f1 f2`, `tail -f -5 file`, and `tail -5 -q file`.
fn rewrite_tail_argv(argv: Vec<OsString>) -> Result<Vec<OsString>, String> {
let mut rewritten = Vec::with_capacity(argv.len() + 2);
let mut iter = argv.into_iter();
rewritten.extend(iter.next());
let mut follow = false;
let mut has_operand = false;
let mut skip_value = false;
let mut seen_ddash = false;
for arg in iter {
if skip_value {
skip_value = false;
rewritten.push(arg);
continue;
}
if has_reverse {
if let Some(option) = unsupported {
return Err(format!(
"-r with {option} is not supported by this builtin; pipe through tac"
));
if seen_ddash {
has_operand = true;
rewritten.push(arg);
continue;
}
return Ok(argv);
let token = arg.to_string_lossy();
if token == "--" {
seen_ddash = true;
rewritten.push(arg);
continue;
}
if argv.len() != 2 && argv.len() != 3 {
return Ok(argv);
}
let clap_ok = args::uu_app()
.try_get_matches_from(argv.clone())
.is_ok_and(|matches| Settings::from(&matches).is_ok());
let obsolete_token = argv[1].clone();
let force_obsolete_blocks = obsolete_token
.to_string_lossy()
.strip_prefix('-')
.is_some_and(|cluster| cluster.contains('b'));
if clap_ok
&& !force_obsolete_blocks
&& !obsolete_token.to_string_lossy().starts_with('+')
{
return Ok(argv);
}
match parse::parse_obsolete(&obsolete_token) {
let bytes = token.as_bytes();
// `+…` is always a candidate (`+10`, `+f`); `-…` only with a leading
// digit (`-5`, `-20f`) so options like `-n` stay untouched.
let candidate =
bytes.first() == Some(&b'+') || matches!(bytes, [b'-', b'0'..=b'9', ..]);
if candidate {
match parse::parse_obsolete(&arg) {
Some(Ok(obsolete)) => {
let mut rewritten = vec![argv.remove(0)];
if obsolete.follow {
rewritten.push(OsString::from(if argv.len() > 1 {
"--follow=name"
} else {
"--follow=descriptor"
}));
}
follow |= obsolete.follow;
rewritten.push(OsString::from(if obsolete.lines { "-n" } else { "-c" }));
rewritten.push(OsString::from(format!(
"{}{}",
if obsolete.plus { "+" } else { "" },
obsolete.num
)));
if argv.len() > 1 {
rewritten.push(argv.remove(1));
continue;
},
Some(Err(parse::ParseError::Context)) => {
return Err(format!(
"option used in invalid context -- {}",
token.chars().nth(1).unwrap_or_default()
));
},
Some(Err(parse::ParseError::InvalidEncoding)) => {
return Err(format!("bad argument encoding: {}", arg.quote()));
},
None => {},
}
}
if bytes.len() > 1 && bytes[0] == b'-' {
skip_value = consumes_separate_value(&token);
} else {
has_operand = true;
}
rewritten.push(arg);
}
if follow {
// Obsolete `f` follows by name when a file operand is present,
// matching GNU; insert up front so an explicit later -f/-F wins.
rewritten.insert(
1,
OsString::from(if has_operand { "--follow=name" } else { "--follow=descriptor" }),
);
}
Ok(rewritten)
},
Some(Err(parse::ParseError::Context)) => Err(format!(
"option used in invalid context -- {}",
obsolete_token.to_string_lossy().chars().nth(1).unwrap_or_default()
)),
Some(Err(parse::ParseError::InvalidEncoding)) => {
Err(format!("bad argument encoding: {}", obsolete_token.quote()))
},
None => Ok(argv),
}
}
/// Parsed `tail` invocation.
pub(crate) struct Tail {
@@ -2986,24 +2994,7 @@ impl Utility for Tail {
fn run(self, host: &mut Host) -> i32 {
if self.matches.get_flag(args::options::REVERSE) {
if self.matches.contains_id(args::options::LINES)
|| self.matches.contains_id(args::options::BYTES)
|| self.matches.get_flag(args::options::BLOCKS)
|| self.matches.contains_id(args::options::FOLLOW)
|| self.matches.get_flag(args::options::FOLLOW_RETRY)
{
let _ = writeln!(
host.stderr,
"tail: -r with -n, -c, -b, or -f is not supported by this builtin; pipe through tac"
);
return 1;
}
let mut argv = vec![OsString::from("tac"), OsString::from("--")];
if let Some(files) = self.matches.get_many::<OsString>(args::options::ARG_FILES) {
argv.extend(files.cloned());
}
return crate::tac::run_argv(argv, host);
return run_reverse(&self.matches, host);
}
let mut settings = match Settings::from(&self.matches) {
Ok(settings) => settings,
@@ -3032,6 +3023,154 @@ pub(crate) fn tail_builtin<SE: ShellExtensions>() -> Registration<SE> {
util::<Tail, SE>()
}
/// BSD `tail -r`: print lines in reverse order. With `-n N` the count selects
/// how many lines to show (last N, or from line N for `+N`) before reversing.
fn run_reverse(matches: &ArgMatches, host: &mut Host) -> i32 {
if matches.contains_id(args::options::BYTES)
|| matches.get_flag(args::options::BLOCKS)
|| matches.contains_id(args::options::FOLLOW)
|| matches.get_flag(args::options::FOLLOW_RETRY)
{
let _ = writeln!(
host.stderr,
"tail: -r with -c, -b, or -f is not supported by this builtin; pipe through tac"
);
return 1;
}
let mut settings = match Settings::from(matches) {
Ok(settings) => settings,
Err(error) => {
let _ = writeln!(host.stderr, "tail: {error}");
return error.code();
},
};
// Without `-n`, `-r` reverses whole inputs; when no headers are wanted
// that is exactly `tac`, so keep delegating.
let all_lines = !matches.contains_id(args::options::LINES);
if all_lines && !settings.verbose {
let mut argv = vec![OsString::from("tac"), OsString::from("--")];
if let Some(files) = matches.get_many::<OsString>(args::options::ARG_FILES) {
argv.extend(files.cloned());
}
return crate::tac::run_argv(argv, host);
}
settings.resolve_paths(host);
match reverse_main(&settings, all_lines, host) {
Ok(()) => host.exit_code(),
Err(error) => {
let code = error.code();
if code != SIGPIPE_EXIT_CODE {
let _ = writeln!(host.stderr, "tail: {error}");
}
code
},
}
}
fn reverse_main(settings: &Settings, all_lines: bool, host: &mut Host) -> TailResult<()> {
let FilterMode::Lines(signum, sep) = &settings.mode else {
unreachable!("-r with -c is rejected before dispatch");
};
let (signum, sep) = (*signum, *sep);
let mut printer = HeaderPrinter::new(settings.verbose, true);
for input in &settings.inputs {
let path = match input.kind() {
InputKind::File(path) if !(cfg!(unix) && path == &PathBuf::from(text::DEV_STDIN)) => {
Some(path)
},
InputKind::File(_) | InputKind::Stdin => None,
};
let mut data = Vec::new();
if let Some(path) = path {
if path.is_dir() {
host.fail(1);
printer.print_input(input, &mut host.stdout);
let _ = writeln!(
host.stderr,
"tail: error reading '{}': Is a directory",
input.display_name
);
continue;
}
match File::open(path) {
Ok(mut file) => {
printer.print_input(input, &mut host.stdout);
file.read_to_end(&mut data)?;
},
Err(error) if error.kind() == ErrorKind::NotFound => {
host.fail(1);
let _ = writeln!(
host.stderr,
"tail: cannot open '{}' for reading: No such file or directory",
input.display_name
);
continue;
},
Err(error) => {
host.fail(1);
let _ = writeln!(
host.stderr,
"tail: cannot open '{}' for reading: {error}",
input.display_name
);
continue;
},
}
} else {
printer.print_input(input, &mut host.stdout);
host.stdin.read_to_end(&mut data)?;
}
write_reversed_lines(&data, signum, sep, all_lines, &mut host.stdout)?;
}
Ok(())
}
/// Writes the selected lines of `data` in reverse order, BSD `tail -r` style:
/// each line keeps its trailing delimiter, so an unterminated final line leads
/// the output without one (matching `tac`).
fn write_reversed_lines(
data: &[u8],
signum: Signum,
sep: u8,
all_lines: bool,
writer: &mut impl Write,
) -> io::Result<()> {
let mut segments: Vec<&[u8]> = Vec::new();
let mut start = 0;
for end in memchr_iter(sep, data) {
segments.push(&data[start..=end]);
start = end + 1;
}
if start < data.len() {
segments.push(&data[start..]);
}
let keep: &[&[u8]] = if all_lines {
&segments[..]
} else {
match signum {
Signum::Negative(count) => {
let count = usize::try_from(count).unwrap_or(usize::MAX);
&segments[segments.len().saturating_sub(count)..]
},
Signum::MinusZero => &[],
Signum::PlusZero => &segments[..],
Signum::Positive(count) => {
// GNU-style 1-based origin: `+1` (like `+0`) selects everything.
let skip = usize::try_from(count.saturating_sub(1)).unwrap_or(usize::MAX);
&segments[skip.min(segments.len())..]
},
}
};
let mut writer = BufWriter::new(writer);
for segment in keep.iter().rev() {
writer.write_all(segment)?;
}
writer.flush()
}
fn tail_main(settings: &Settings, host: &mut Host) -> TailResult<()> {
settings.check_warnings(&mut host.stderr);
@@ -3525,13 +3664,21 @@ where
#[cfg(test)]
mod tests {
use std::{fs, io::Cursor};
use std::{ffi::OsString, fs, io::Cursor};
use clap::Parser;
use super::{Tail, Utility, forwards_thru_file};
use crate::host::{Host, run_util};
fn rewritten(argv: &[&str]) -> Vec<String> {
Tail::rewrite_argv(argv.iter().map(OsString::from).collect())
.unwrap()
.into_iter()
.map(|arg| arg.to_str().unwrap().to_owned())
.collect()
}
#[test]
fn prints_last_line_from_stdin() {
let (code, capture) = run_util::<Tail>(&["-n", "1"], "first\nlast\n", "/");
@@ -3579,14 +3726,102 @@ mod tests {
assert_eq!(parsed.run(&mut host), 0);
}
// Failure mode: obsolete `-N`/`+N` was only rewritten for `argv.len()`
// of 2 or 3 with the token at argv[1], so multi-file and flag-interleaved
// invocations were clap parse errors.
#[test]
fn reverse_with_line_count_keeps_bsd_error() {
let (code, capture) = run_util::<Tail>(&["-r", "-n", "2"], "", "/");
fn obsolete_count_rewritten_at_any_position() {
assert_eq!(rewritten(&["tail", "-20", "f1", "f2"]), ["tail", "-n", "20", "f1", "f2"]);
assert_eq!(rewritten(&["tail", "-f", "-5", "f"]), ["tail", "-f", "-n", "5", "f"]);
assert_eq!(rewritten(&["tail", "-5", "-q", "f"]), ["tail", "-n", "5", "-q", "f"]);
assert_eq!(rewritten(&["tail", "+10", "f"]), ["tail", "-n", "+10", "f"]);
assert_eq!(rewritten(&["tail", "-5c", "f"]), ["tail", "-c", "5", "f"]);
// Obsolete `f` still maps to --follow=name with a file operand.
assert_eq!(
rewritten(&["tail", "-20f", "f"]),
["tail", "--follow=name", "-n", "20", "f"]
);
assert_eq!(rewritten(&["tail", "-20f"]), ["tail", "--follow=descriptor", "-n", "20"]);
}
// Failure mode: a `-N`/`+N` token that is really an option value or a
// post-`--` operand must never be rewritten.
#[test]
fn option_values_and_post_ddash_operands_are_not_rewritten() {
assert_eq!(rewritten(&["tail", "-n", "+5", "f"]), ["tail", "-n", "+5", "f"]);
assert_eq!(rewritten(&["tail", "-c", "-5", "f"]), ["tail", "-c", "-5", "f"]);
assert_eq!(rewritten(&["tail", "--lines", "-5", "f"]), ["tail", "--lines", "-5", "f"]);
assert_eq!(rewritten(&["tail", "--", "-5"]), ["tail", "--", "-5"]);
}
// Failure mode: `tail -20 f1 f2` was rejected outright; it must print the
// last lines of every operand with GNU headers.
#[test]
fn obsolete_count_with_multiple_files_prints_headers() {
let dir = tempfile::tempdir().unwrap();
fs::write(dir.path().join("f1"), "a\nb\n").unwrap();
fs::write(dir.path().join("f2"), "c\nd\n").unwrap();
let (code, capture) = run_util::<Tail>(&["-1", "f1", "f2"], "", dir.path());
assert_eq!(code, 0);
assert_eq!(capture.out(), "==> f1 <==\nb\n\n==> f2 <==\nd\n");
assert_eq!(capture.err(), "");
}
// Failure mode: `-r` with `-n N` was rejected; BSD tail shows the last N
// lines in reverse order.
#[test]
fn reverse_with_line_count_takes_last_lines_reversed() {
let (code, capture) = run_util::<Tail>(&["-r", "-n", "2"], "a\nb\nc\n", "/");
assert_eq!(code, 0);
assert_eq!(capture.out(), "c\nb\n");
assert_eq!(capture.err(), "");
}
// Failure mode: `-rq` was rejected; with `-q` headers stay suppressed
// while each file's selection is reversed independently.
#[test]
fn reverse_quiet_suppresses_headers_across_files() {
let dir = tempfile::tempdir().unwrap();
fs::write(dir.path().join("f1"), "a\nb\n").unwrap();
fs::write(dir.path().join("f2"), "c\nd\n").unwrap();
let (code, capture) = run_util::<Tail>(&["-rq", "-n", "2", "f1", "f2"], "", dir.path());
assert_eq!(code, 0);
assert_eq!(capture.out(), "b\na\nd\nc\n");
assert_eq!(capture.err(), "");
}
// Multi-file reverse keeps GNU-style headers.
#[test]
fn reverse_with_multiple_files_prints_headers() {
let dir = tempfile::tempdir().unwrap();
fs::write(dir.path().join("f1"), "a\nb\n").unwrap();
fs::write(dir.path().join("f2"), "c\nd\n").unwrap();
let (code, capture) = run_util::<Tail>(&["-r", "-n", "1", "f1", "f2"], "", dir.path());
assert_eq!(code, 0);
assert_eq!(capture.out(), "==> f1 <==\nb\n\n==> f2 <==\nd\n");
assert_eq!(capture.err(), "");
}
// Failure mode: obsolete `-N` combined with `-r` (`tail -r -5`) must feed
// the rewritten count into the reverse path.
#[test]
fn reverse_with_obsolete_count() {
let (code, capture) = run_util::<Tail>(&["-r", "-2"], "a\nb\nc\n", "/");
assert_eq!(code, 0);
assert_eq!(capture.out(), "c\nb\n");
assert_eq!(capture.err(), "");
}
// `-r` with byte/block counts stays an explicit error rather than
// silently diverging from BSD semantics.
#[test]
fn reverse_with_byte_count_keeps_clear_error() {
let (code, capture) = run_util::<Tail>(&["-r", "-c", "5"], "", "/");
assert_eq!(code, 1);
assert_eq!(capture.out(), "");
assert_eq!(
capture.err(),
"tail: -r with -n, -c, -b, or -f is not supported by this builtin; pipe through tac\n"
"tail: -r with -c, -b, or -f is not supported by this builtin; pipe through tac\n"
);
}
+498 -39
View File
@@ -1,9 +1,14 @@
//! `timeout` builtin, moved from `pi-shell`.
use std::{future::Future, io::Write, time::Duration};
use std::{
io::Write,
sync::{Arc, Mutex},
time::Duration,
};
use brush_core::{
ExecutionContext, ExecutionExitCode, ExecutionResult, ProcessGroupPolicy, SourceInfo, builtins,
ExecutionContext, ExecutionExitCode, ExecutionResult, ProcessGroupPolicy, SourceInfo,
SpawnObserver, builtins, sys, traps::TrapSignal,
};
use clap::Parser;
use tokio::time;
@@ -11,82 +16,338 @@ use tokio_util::sync::CancellationToken;
use crate::host::{parse_duration, quote_arg};
/// GNU timeout's exit status for its own usage/internal errors.
const EXIT_TIMEOUT_FAILURE: u8 = 125;
/// GNU timeout's exit status when the time limit expired.
const EXIT_TIMED_OUT: u8 = 124;
/// 128 + SIGKILL(9): reported when the command died from SIGKILL.
const EXIT_KILLED: u8 = 137;
/// Run a command with a time limit.
#[derive(Parser)]
#[command(disable_help_flag = true)]
pub(crate) struct TimeoutCommand {
#[arg(required = true)]
struct TimeoutArgs {
/// Signal to send on expiry: a name with or without the `SIG` prefix, or
/// a number. Defaults to TERM.
#[arg(short = 's', long = "signal", value_name = "SIGNAL")]
signal: Option<String>,
/// Also send SIGKILL if the command is still running this long after the
/// initial signal.
#[arg(short = 'k', long = "kill-after", value_name = "DURATION")]
kill_after: Option<String>,
/// Exit with the command's own status even when the time limit expired.
#[arg(long)]
preserve_status: bool,
/// GNU compatibility: don't put the command in a separate process group,
/// and signal only the direct children rather than a whole group.
#[arg(long)]
foreground: bool,
/// Diagnose each signal sent to the command on stderr.
#[arg(short = 'v', long)]
verbose: bool,
// Hyphenated operands must reach `parse_duration` so `timeout -1 cmd`
// reports an invalid time interval (exit 125) like GNU, instead of a
// clap unknown-option error.
#[arg(required = true, allow_hyphen_values = true)]
duration: String,
#[arg(required = true, num_args = 1.., trailing_var_arg = true)]
// The command's own options belong to the command: `timeout 5 grep -v x`.
#[arg(required = true, num_args = 1.., trailing_var_arg = true, allow_hyphen_values = true)]
command: Vec<String>,
}
/// Holds the raw argument vector so parse failures surface as GNU timeout's
/// exit status 125, not brush's generic usage-error status 2 (which the
/// default `builtins::Command::new` path would produce).
pub(crate) struct TimeoutCommand {
argv: Vec<String>,
}
impl clap::FromArgMatches for TimeoutCommand {
fn from_arg_matches(_matches: &clap::ArgMatches) -> Result<Self, clap::Error> {
Ok(Self { argv: Vec::new() })
}
fn update_from_arg_matches(&mut self, _matches: &clap::ArgMatches) -> Result<(), clap::Error> {
Ok(())
}
}
impl clap::CommandFactory for TimeoutCommand {
fn command() -> clap::Command {
<TimeoutArgs as clap::CommandFactory>::command()
}
fn command_for_update() -> clap::Command {
<TimeoutArgs as clap::CommandFactory>::command_for_update()
}
}
impl clap::Parser for TimeoutCommand {}
/// Records the external children spawned while running the timed command.
///
/// brush's cancellation token can only SIGKILL a child (see
/// `brush_core::processes::Process::wait`), so delivering the *configured*
/// signal requires knowing the child's pid/pgid; the shell reports those
/// through its [`SpawnObserver`] hook.
#[derive(Default)]
struct SpawnRecorder(Mutex<Vec<(i32, Option<i32>)>>);
impl SpawnObserver for SpawnRecorder {
fn on_spawn(&self, pid: i32, pgid: Option<i32>) {
if let Ok(mut spawns) = self.0.lock() {
spawns.push((pid, pgid));
}
}
}
impl SpawnRecorder {
/// Sends `signal` to every recorded child — its whole process group when
/// `group` is set — and reports whether any delivery succeeded.
fn signal(&self, signal: TrapSignal, group: bool) -> bool {
let spawns = match self.0.lock() {
Ok(spawns) => spawns.clone(),
Err(_) => return false,
};
let mut sent = false;
for (pid, pgid) in spawns {
let target = match pgid {
Some(pgid) if group => -pgid,
_ => pid,
};
if sys::signal::kill_process(target, signal).is_ok() {
sent = true;
}
}
sent
}
}
/// Parses a `-s` operand: a signal name (with or without the `SIG` prefix,
/// any case) or a signal number.
fn parse_signal(spec: &str) -> Option<TrapSignal> {
let parsed = if let Ok(number) = spec.trim().parse::<i32>() {
TrapSignal::try_from(number).ok()?
} else {
TrapSignal::try_from(spec).ok()?
};
// Only real signals can be delivered; EXIT/DEBUG/ERR are shell traps.
matches!(parsed, TrapSignal::Signal(_)).then_some(parsed)
}
/// Renders a signal the way GNU timeout's diagnostics do: `TERM`, not `SIGTERM`.
fn signal_display(signal: TrapSignal) -> &'static str {
let name = signal.as_str();
name.strip_prefix("SIG").unwrap_or(name)
}
impl builtins::Command for TimeoutCommand {
type Error = brush_core::Error;
fn execute<SE: brush_core::ShellExtensions>(
fn new<I>(args: I) -> Result<Self, clap::Error>
where
I: IntoIterator<Item = String>,
{
Ok(Self { argv: args.into_iter().collect() })
}
async fn execute<SE: brush_core::ShellExtensions>(
&self,
context: ExecutionContext<'_, SE>,
) -> impl Future<Output = std::result::Result<ExecutionResult, brush_core::Error>> + Send {
let duration = self.duration.clone();
let command = self.command.clone();
async move {
) -> std::result::Result<ExecutionResult, brush_core::Error> {
if context.is_cancelled() {
return Ok(ExecutionExitCode::Interrupted.into());
}
let Some(timeout) = parse_duration(&duration) else {
let _ = writeln!(context.stderr(), "timeout: invalid time interval '{duration}'");
return Ok(ExecutionResult::new(125));
};
if command.is_empty() {
let _ = writeln!(context.stderr(), "timeout: missing command");
return Ok(ExecutionResult::new(125));
let args = match TimeoutArgs::try_parse_from(&self.argv) {
Ok(args) => args,
Err(err) => {
// clap reports `--help` as an error; that belongs on stdout
// with a success status, real usage errors exit 125.
let rendered = err.to_string();
if err.use_stderr() {
let _ = write!(context.stderr(), "{rendered}");
return Ok(ExecutionResult::new(EXIT_TIMEOUT_FAILURE));
}
let _ = write!(context.stdout(), "{rendered}");
return Ok(ExecutionResult::success());
},
};
let Some(limit) = parse_duration(&args.duration) else {
let _ = writeln!(
context.stderr(),
"timeout: invalid time interval '{}'",
args.duration
);
return Ok(ExecutionResult::new(EXIT_TIMEOUT_FAILURE));
};
let kill_after = match &args.kill_after {
Some(spec) => match parse_duration(spec) {
Some(duration) => Some(duration),
None => {
let _ =
writeln!(context.stderr(), "timeout: invalid time interval '{spec}'");
return Ok(ExecutionResult::new(EXIT_TIMEOUT_FAILURE));
},
},
None => None,
};
let signal = match &args.signal {
Some(spec) => match parse_signal(spec) {
Some(signal) => signal,
None => {
let _ = writeln!(context.stderr(), "timeout: '{spec}': invalid signal");
return Ok(ExecutionResult::new(EXIT_TIMEOUT_FAILURE));
},
},
None => TrapSignal::try_from("TERM").expect("SIGTERM must be a known signal"),
};
let child_cancel = CancellationToken::new();
let spawns = Arc::new(SpawnRecorder::default());
let mut params = context.params.clone();
params.process_group_policy = ProcessGroupPolicy::NewProcessGroup;
// GNU runs the command in its own process group and signals the whole
// group; `--foreground` keeps it in the invoking group and signals
// only the direct children.
params.process_group_policy = if args.foreground {
ProcessGroupPolicy::SameProcessGroup
} else {
ProcessGroupPolicy::NewProcessGroup
};
params.set_cancel_token(child_cancel.clone());
params.set_spawn_observer(Arc::clone(&spawns) as Arc<dyn SpawnObserver>);
let mut command_line = String::new();
for (idx, arg) in command.iter().enumerate() {
for (idx, arg) in args.command.iter().enumerate() {
if idx > 0 {
command_line.push(' ');
}
command_line.push_str(&quote_arg(arg));
}
let cancel_token = context.cancel_token();
// Grab an owned stderr handle up front: `run_future` below holds the
// shell mutably, so `context.stderr()` is unavailable once it exists.
let mut stderr = context.stderr();
let outer_cancel = context.cancel_token();
let source_info = SourceInfo::from("pi-natives:timeout");
let run_future = context.shell.run_string(command_line, &source_info, &params);
tokio::pin!(run_future);
if let Some(cancel_token) = cancel_token {
tokio::select! {
result = &mut run_future => result,
() = time::sleep(timeout) => {
child_cancel.cancel();
// Wait briefly for the child to exit after cancellation.
let _ = time::timeout(Duration::from_secs(2), &mut run_future).await;
Ok(ExecutionResult::new(124))
},
() = cancel_token.cancelled() => {
child_cancel.cancel();
Ok(ExecutionExitCode::Interrupted.into())
},
let outer_cancelled = async {
match &outer_cancel {
Some(token) => token.cancelled().await,
None => std::future::pending().await,
}
};
tokio::pin!(outer_cancelled);
// GNU: a duration of zero disables the timeout entirely.
let deadline = async {
if limit.is_zero() {
std::future::pending::<()>().await;
} else {
time::sleep(limit).await;
}
};
tokio::pin!(deadline);
tokio::select! {
result = &mut run_future => result,
() = time::sleep(timeout) => {
result = &mut run_future => return result,
() = &mut outer_cancelled => {
child_cancel.cancel();
// Wait briefly for the child to exit after cancellation.
let _ = time::timeout(Duration::from_secs(2), &mut run_future).await;
Ok(ExecutionResult::new(124))
return Ok(ExecutionExitCode::Interrupted.into());
},
() = &mut deadline => {},
}
// The limit expired: deliver the configured signal like GNU timeout.
if args.verbose {
let _ = writeln!(
stderr,
"timeout: sending signal {} to command '{}'",
signal_display(signal),
args.command[0]
);
}
let signalled = spawns.signal(signal, !args.foreground);
if !signalled {
// The operand ran in-process (a builtin, say) or the child is
// already gone; cancellation is the only remaining lever. For
// external children it degrades to SIGKILL — see `Process::wait`.
child_cancel.cancel();
}
let mut killed = signal.as_str() == "SIGKILL";
// Wait for the command to finish, escalating to SIGKILL after
// `--kill-after`. Without `-k`, GNU waits indefinitely — a command
// that catches the signal keeps running (the caller can still cancel).
// After a cancel-fallback (in-process operand), the inner shell may
// surface its own cancellation as an Interrupted error instead of the
// operand's result; that is expected retirement, not a fault.
let reap = |result: Result<ExecutionResult, brush_core::Error>| match result {
Ok(result) => Ok(Some(result)),
Err(err) if !signalled && matches!(err.kind(), brush_core::ErrorKind::Interrupted) => {
Ok(None)
},
Err(err) => Err(err),
};
let kill_deadline = async {
match kill_after {
Some(duration) => time::sleep(duration).await,
None => std::future::pending().await,
}
};
tokio::pin!(kill_deadline);
let child_result = tokio::select! {
result = &mut run_future => reap(result)?,
() = &mut outer_cancelled => {
child_cancel.cancel();
return Ok(ExecutionExitCode::Interrupted.into());
},
() = &mut kill_deadline => {
if args.verbose {
let _ = writeln!(
stderr,
"timeout: sending signal KILL to command '{}'",
args.command[0]
);
}
killed = true;
let kill = TrapSignal::try_from("KILL").expect("SIGKILL must be a known signal");
spawns.signal(kill, !args.foreground);
child_cancel.cancel();
// SIGKILL can't be resisted; bound the reaping wait anyway so
// a wedged in-process operand can't hang the builtin forever.
match time::timeout(Duration::from_secs(2), &mut run_future).await {
Ok(result) => reap(result)?,
Err(_) => None,
}
},
};
// Exit status per GNU: the command's own status under
// `--preserve-status`; 137 when it died from SIGKILL; else 124.
if args.preserve_status {
if signalled {
return Ok(child_result.unwrap_or_else(|| ExecutionResult::new(EXIT_KILLED)));
}
// Cancel-fallback path (in-process operand): the inner shell's own
// cancellation check races the operand's result, so its status is
// unreliable. Report death by the delivered signal (128+N, or 137
// after escalation) deterministically, matching GNU for a command
// taken down by the timeout signal.
let number = i32::try_from(signal).unwrap_or(15);
let code = if killed {
EXIT_KILLED
} else {
128_u8.wrapping_add(number as u8)
};
return Ok(ExecutionResult::new(code));
}
if killed {
return Ok(ExecutionResult::new(EXIT_KILLED));
}
Ok(ExecutionResult::new(EXIT_TIMED_OUT))
}
}
@@ -104,7 +365,7 @@ mod tests {
};
use clap::Parser;
use super::TimeoutCommand;
use super::{TimeoutArgs, TimeoutCommand, parse_signal, signal_display};
#[derive(Parser)]
struct StatusCommand;
@@ -142,6 +403,23 @@ mod tests {
}
}
/// An operand that ignores cancellation, so only SIGKILL escalation (the
/// cancel-token fallback plus the bounded reap) can retire it early.
#[derive(Parser)]
struct StubbornCommand;
impl builtins::Command for StubbornCommand {
type Error = brush_core::Error;
async fn execute<SE: brush_core::ShellExtensions>(
&self,
_context: ExecutionContext<'_, SE>,
) -> Result<ExecutionResult, Self::Error> {
tokio::time::sleep(Duration::from_millis(300)).await;
Ok(ExecutionResult::new(99))
}
}
async fn test_shell() -> Shell<DefaultShellExtensions> {
Shell::builder()
.builtin(
@@ -220,4 +498,185 @@ mod tests {
assert_eq!(u8::from(result.exit_code), 125);
assert_eq!(diagnostic, "timeout: invalid time interval 'invalid'\n");
}
#[tokio::test]
async fn gnu_flag_spellings_parse() {
// Failure mode: a real-world GNU invocation dying in clap.
let args = TimeoutArgs::try_parse_from([
"timeout", "-s", "INT", "-k", "2s", "10s", "cmd", "arg",
])
.expect("-s/-k spellings must parse");
assert_eq!(args.signal.as_deref(), Some("INT"));
assert_eq!(args.kill_after.as_deref(), Some("2s"));
assert_eq!(args.duration, "10s");
assert_eq!(args.command, ["cmd", "arg"]);
let args = TimeoutArgs::try_parse_from([
"timeout",
"--signal=KILL",
"--kill-after=1",
"--preserve-status",
"--foreground",
"-v",
"5",
"cmd",
])
.expect("long spellings must parse");
assert!(args.preserve_status && args.foreground && args.verbose);
// `--` before the duration ends option parsing, GNU-style.
let args = TimeoutArgs::try_parse_from(["timeout", "--", "5", "cmd"])
.expect("-- before the duration must parse");
assert_eq!(args.duration, "5");
// The command's own options must pass through untouched.
let args = TimeoutArgs::try_parse_from(["timeout", "5", "grep", "-v", "-e", "x"])
.expect("command options must not be parsed by timeout");
assert_eq!(args.command, ["grep", "-v", "-e", "x"]);
// A hyphenated duration reaches parse_duration (exit 125 later),
// instead of failing as an unknown clap option.
let args = TimeoutArgs::try_parse_from(["timeout", "-1", "cmd"])
.expect("hyphenated duration must reach the interval check");
assert_eq!(args.duration, "-1");
}
#[test]
fn signal_spellings_parse_and_display_without_prefix() {
// Failure mode: rejecting a signal spelling GNU accepts.
for spec in ["TERM", "term", "SIGTERM", "sigterm", "15", "KILL", "9", "INT", "2"] {
assert!(parse_signal(spec).is_some(), "spec {spec:?} must parse");
}
// Shell-trap pseudo-signals and unknown names are invalid for kill(2).
for spec in ["NOSUCH", "EXIT", "DEBUG", "ERR", "64", "-5"] {
assert!(parse_signal(spec).is_none(), "spec {spec:?} must be rejected");
}
let term = parse_signal("SIGTERM").expect("SIGTERM parses");
assert_eq!(signal_display(term), "TERM");
}
#[tokio::test]
async fn zero_duration_disables_the_timeout() {
// Failure mode: `timeout 0 cmd` firing instantly instead of never.
let result = run_with_deadline("timeout 0 slow-test").await;
assert_eq!(u8::from(result.exit_code), 99);
let result = run_with_deadline("timeout 0s status-test").await;
assert_eq!(u8::from(result.exit_code), 7);
}
#[tokio::test]
async fn preserve_status_reports_death_by_the_timeout_signal() {
// In-process operands retire via the cancel fallback, where the inner
// shell's result is racy; --preserve-status must deterministically
// report death by the configured signal (TERM -> 143), like GNU does
// for a command killed by the timeout signal.
let result = run_with_deadline("timeout --preserve-status 0.010 slow-test").await;
assert_eq!(u8::from(result.exit_code), 143);
}
#[tokio::test]
async fn kill_signal_reports_137() {
// GNU exits 128+9 when the command is taken down with SIGKILL.
let result = run_with_deadline("timeout -s KILL 0.010 slow-test").await;
assert_eq!(u8::from(result.exit_code), 137);
}
#[tokio::test]
async fn kill_after_escalates_and_reports_137() {
// stubborn-test ignores the initial (cancellation-based) signal; the
// -k deadline must escalate and report the SIGKILL status.
let mut shell = test_shell().await;
shell
.register_builtin(
"stubborn-test",
builtins::builtin::<StubbornCommand, DefaultShellExtensions>(),
);
let mut params = shell.default_exec_params();
// Keep the inner shell's interrupted notice off the test runner's
// terminal, like run_with_deadline does.
for fd in [OpenFiles::STDIN_FD, OpenFiles::STDOUT_FD, OpenFiles::STDERR_FD] {
params.set_fd(fd, brush_core::openfiles::null().expect("null device"));
}
let result = tokio::time::timeout(
Duration::from_secs(1),
shell.run_string(
"timeout -k 0.075 0.010 stubborn-test",
&SourceInfo::default(),
&params,
),
)
.await
.expect("escalation test exceeded its safety deadline")
.expect("execute escalation command");
assert_eq!(u8::from(result.exit_code), 137);
}
#[tokio::test]
async fn clap_parse_failure_exits_125_not_2() {
// GNU usage errors exit 125; brush's generic clap path would exit 2.
let result = run_with_deadline("timeout").await;
assert_eq!(u8::from(result.exit_code), 125);
}
#[tokio::test]
async fn invalid_kill_after_interval_exits_125() {
let result = run_with_deadline("timeout -k bogus 1 status-test").await;
assert_eq!(u8::from(result.exit_code), 125);
}
#[tokio::test]
async fn invalid_signal_exits_125() {
let result = run_with_deadline("timeout -s NOSUCH 1 status-test").await;
assert_eq!(u8::from(result.exit_code), 125);
}
#[tokio::test]
async fn verbose_reports_the_signal_sent() {
let mut shell = test_shell().await;
let mut stderr = tempfile::tempfile().expect("create stderr capture");
let mut params = shell.default_exec_params();
params.set_fd(
OpenFiles::STDERR_FD,
stderr.try_clone().expect("clone stderr capture").into(),
);
let result = tokio::time::timeout(
Duration::from_secs(1),
shell.run_string("timeout -v 0.010 slow-test", &SourceInfo::default(), &params),
)
.await
.expect("verbose test exceeded its safety deadline")
.expect("execute verbose command");
stderr.seek(SeekFrom::Start(0)).expect("rewind stderr capture");
let mut diagnostic = String::new();
stderr.read_to_string(&mut diagnostic).expect("read stderr capture");
assert_eq!(u8::from(result.exit_code), 124);
// The cancel-fallback may race the inner shell's own cancellation
// check, which can append its interrupted notice after our line.
assert!(
diagnostic.starts_with("timeout: sending signal TERM to command 'slow-test'\n"),
"diagnostic must lead with the GNU-style signal line: {diagnostic:?}"
);
}
#[cfg(unix)]
#[tokio::test]
async fn external_child_receives_the_configured_signal() {
// Failure mode: the timed-out external child only ever seeing SIGKILL
// (the cancel-token path) instead of the configured signal.
let result = run_with_deadline("timeout -s TERM 0.050 /bin/sleep 5").await;
assert_eq!(u8::from(result.exit_code), 124);
let result =
run_with_deadline("timeout --preserve-status 0.050 /bin/sleep 5").await;
assert_eq!(u8::from(result.exit_code), 143, "SIGTERM death is 128+15");
}
}
+223 -44
View File
@@ -26,6 +26,80 @@ enum TopSortKey {
Time,
}
/// Column keys accepted by macOS-style `-stats` (comma-separated).
#[derive(Clone, Copy, Debug, PartialEq, Eq, clap::ValueEnum)]
enum TopStat {
Pid,
#[value(alias = "uid")]
User,
#[value(name = "pstate", alias = "state")]
State,
#[value(name = "nice", alias = "ni")]
Nice,
#[value(name = "th", alias = "threads")]
Threads,
#[value(name = "vsize", alias = "virt")]
Virt,
#[value(name = "mem", alias = "rsize", alias = "res")]
Res,
#[value(alias = "time+")]
Time,
#[value(alias = "%cpu")]
Cpu,
#[value(name = "%mem", alias = "pmem")]
PctMem,
#[value(alias = "comm")]
Command,
}
/// Column order used when `-stats` is not given.
const DEFAULT_TOP_STATS: &[TopStat] = &[
TopStat::Pid,
TopStat::User,
TopStat::State,
TopStat::Nice,
TopStat::Threads,
TopStat::Virt,
TopStat::Res,
TopStat::Time,
TopStat::Cpu,
TopStat::PctMem,
TopStat::Command,
];
impl TopStat {
fn header(self) -> &'static str {
match self {
Self::Pid => "PID",
Self::User => "USER",
Self::State => "S",
Self::Nice => "NI",
Self::Threads => "TH",
Self::Virt => "VIRT",
Self::Res => "RES",
Self::Time => "TIME+",
Self::Cpu => "%CPU",
Self::PctMem => "%MEM",
Self::Command => "COMMAND",
}
}
/// Right-align width; 0 renders as-is (used for the free-form command).
fn width(self) -> usize {
match self {
Self::Pid => 7,
Self::User => 8,
Self::State => 2,
Self::Nice => 3,
Self::Threads => 4,
Self::Virt | Self::Res => 9,
Self::Time => 10,
Self::Cpu | Self::PctMem => 4,
Self::Command => 0,
}
}
}
/// Display processes.
#[derive(Parser)]
#[command(name = "top", version, about = "Display processes", disable_help_flag = false)]
@@ -80,6 +154,10 @@ pub(crate) struct TopCommand {
/// Show the complete command line instead of the executable name.
#[arg(short = 'c', long = "full-command")]
full_command: bool,
/// Columns to display, in order (comma-separated, macOS `-stats` style).
#[arg(long = "stats", value_enum, value_delimiter = ',', ignore_case = true)]
stats: Vec<TopStat>,
}
#[derive(Clone)]
@@ -96,9 +174,46 @@ struct TopProcessRow {
command: String,
}
/// Long option names `top` accepts with a macOS-style single dash.
const TOP_LONG_OPTIONS: &[&str] = &[
"batch",
"samples",
"iterations",
"delay",
"rows",
"pid",
"user",
"sort",
"full-command",
"stats",
"help",
"version",
];
/// Rewrites macOS-style single-dash long options (`-pid`, `-stats pid,cpu`)
/// into clap-style `--` options; everything else passes through untouched.
fn normalize_top_flag(arg: String) -> String {
if let Some(rest) = arg.strip_prefix('-')
&& !rest.starts_with('-')
{
let name = rest.split('=').next().unwrap_or(rest);
if name.len() > 1 && TOP_LONG_OPTIONS.contains(&name) {
return format!("-{arg}");
}
}
arg
}
impl builtins::Command for TopCommand {
type Error = brush_core::Error;
fn new<I>(args: I) -> Result<Self, clap::Error>
where
I: IntoIterator<Item = String>,
{
Self::try_parse_from(args.into_iter().map(normalize_top_flag))
}
fn execute<SE: brush_core::ShellExtensions>(
&self,
context: ExecutionContext<'_, SE>,
@@ -111,6 +226,11 @@ impl builtins::Command for TopCommand {
let sort = self.sort;
let full_command = self.full_command;
let _ = self.batch;
let stats = if self.stats.is_empty() {
DEFAULT_TOP_STATS.to_vec()
} else {
self.stats.clone()
};
async move {
if !delay.is_finite() || delay < 0.0 || delay > Duration::MAX.as_secs_f64() {
writeln!(context.stderr(), "top: invalid delay '{delay}'")?;
@@ -200,7 +320,7 @@ impl builtins::Command for TopCommand {
}
sort_top_rows(&mut rows, sort);
let output = render_top_snapshot(&rows, row_limit, sample + 1);
let output = render_top_snapshot(&rows, row_limit, sample + 1, &stats);
if let Err(err) = write!(context.stdout(), "{output}") {
if err.kind() == io::ErrorKind::BrokenPipe {
return Ok(ExecutionResult::success());
@@ -245,7 +365,46 @@ fn sort_top_rows(rows: &mut [TopProcessRow], key: TopSortKey) {
});
}
fn render_top_snapshot(rows: &[TopProcessRow], row_limit: Option<usize>, sample: u64) -> String {
fn top_cell(row: &TopProcessRow, stat: TopStat) -> String {
match stat {
TopStat::Pid => row.pid.to_string(),
TopStat::User => row
.user
.map_or_else(|| "?".to_string(), |value| value.to_string()),
TopStat::State => row.state.to_string(),
TopStat::Nice => row
.nice
.map_or_else(|| "?".to_string(), |value| value.to_string()),
TopStat::Threads => row
.threads
.map_or_else(|| "?".to_string(), |value| value.to_string()),
TopStat::Virt => row
.virtual_size
.map_or_else(|| "?".to_string(), format_top_bytes),
TopStat::Res => row
.resident_size
.map_or_else(|| "?".to_string(), format_top_bytes),
TopStat::Time => row
.cpu_time
.map_or_else(|| "?".to_string(), format_top_time),
TopStat::Cpu => format!("{:.1}", row.cpu_percent),
TopStat::PctMem => "?".to_string(),
TopStat::Command => {
if row.command.is_empty() {
"?".to_string()
} else {
row.command.clone()
}
},
}
}
fn render_top_snapshot(
rows: &[TopProcessRow],
row_limit: Option<usize>,
sample: u64,
stats: &[TopStat],
) -> String {
let mut running = 0_usize;
let mut sleeping = 0_usize;
let mut stopped = 0_usize;
@@ -295,50 +454,24 @@ fn render_top_snapshot(rows: &[TopProcessRow], row_limit: Option<usize>, sample:
format_top_bytes(resident),
format_top_bytes(virtual_size)
);
let _ = writeln!(
output,
"{:>7} {:>8} {:>2} {:>3} {:>4} {:>9} {:>9} {:>10} {:>4} {:>4} COMMAND",
"PID", "USER", "S", "NI", "TH", "VIRT", "RES", "TIME+", "%CPU", "%MEM"
);
let mut line = String::new();
for (index, stat) in stats.iter().enumerate() {
if index > 0 {
line.push(' ');
}
let _ = write!(line, "{:>width$}", stat.header(), width = stat.width());
}
let _ = writeln!(output, "{line}");
for row in rows.iter().take(row_limit.unwrap_or(usize::MAX)) {
let user = row
.user
.map_or_else(|| "?".to_string(), |value| value.to_string());
let nice = row
.nice
.map_or_else(|| "?".to_string(), |value| value.to_string());
let threads = row
.threads
.map_or_else(|| "?".to_string(), |value| value.to_string());
let virtual_size = row
.virtual_size
.map_or_else(|| "?".to_string(), format_top_bytes);
let resident_size = row
.resident_size
.map_or_else(|| "?".to_string(), format_top_bytes);
let cpu_time = row
.cpu_time
.map_or_else(|| "?".to_string(), format_top_time);
let _ = writeln!(
output,
"{:>7} {:>8} {:>2} {:>3} {:>4} {:>9} {:>9} {:>10} {:>4.1} {:>4} {}",
row.pid,
user,
row.state,
nice,
threads,
virtual_size,
resident_size,
cpu_time,
row.cpu_percent,
"?",
if row.command.is_empty() {
"?"
} else {
&row.command
line.clear();
for (index, stat) in stats.iter().enumerate() {
if index > 0 {
line.push(' ');
}
);
let _ = write!(line, "{:>width$}", top_cell(row, *stat), width = stat.width());
}
let _ = writeln!(output, "{line}");
}
output.push('\n');
output
@@ -433,10 +566,56 @@ mod tests {
row(2, "visible-two", 0.0, 0, 0),
row(1, "hidden-one", 0.0, 0, 0),
];
let output = render_top_snapshot(&rows, Some(2), 7);
let output = render_top_snapshot(&rows, Some(2), 7, DEFAULT_TOP_STATS);
assert!(output.contains("top - snapshot 7"));
assert!(output.contains("visible-three"));
assert!(output.contains("visible-two"));
assert!(!output.contains("hidden-one"));
}
#[test]
fn parses_macos_single_dash_long_options() {
use brush_core::builtins::Command as _;
let cmd = TopCommand::new(
["top", "-pid", "56943,101", "-stats", "pid,cpu,th,mem,pstate"]
.into_iter()
.map(String::from),
)
.expect("macOS-style flags must parse");
assert_eq!(cmd.pids, vec![56943, 101]);
assert_eq!(cmd.stats, vec![
TopStat::Pid,
TopStat::Cpu,
TopStat::Threads,
TopStat::Res,
TopStat::State
]);
}
#[test]
fn snapshot_renders_selected_stats_in_order() {
let rows = vec![row(42, "worker", 12.3, 4096, 61)];
let output = render_top_snapshot(&rows, None, 1, &[
TopStat::Pid,
TopStat::Cpu,
TopStat::Threads,
TopStat::Res,
TopStat::State,
]);
let header = output
.lines()
.find(|line| line.contains("PID"))
.expect("header line");
assert_eq!(header.split_whitespace().collect::<Vec<_>>(), vec![
"PID", "%CPU", "TH", "RES", "S"
]);
let row_line = output
.lines()
.find(|line| line.contains("42"))
.expect("process row");
assert_eq!(row_line.split_whitespace().collect::<Vec<_>>(), vec![
"42", "12.3", "2", "4.0k", "S"
]);
assert!(!output.contains("worker"));
}
}
+181 -41
View File
@@ -3,10 +3,10 @@
//! Ported from uutils coreutils 0.8.0.
#[cfg(unix)]
use std::os::unix::fs::FileTypeExt;
use std::os::unix::fs::{FileTypeExt, MetadataExt};
use std::{
ffi::OsString,
fs::{OpenOptions, metadata},
fs::{Metadata, OpenOptions, metadata},
io::ErrorKind,
};
@@ -19,7 +19,7 @@ use uucore::{
use crate::host::{Host, Utility, format_usage, matches_parser, util};
#[derive(Debug, Eq, PartialEq)]
#[derive(Debug, Clone, Copy, Eq, PartialEq)]
enum TruncateMode {
Absolute(u64),
Extend(u64),
@@ -51,6 +51,35 @@ impl TruncateMode {
}
}
/// The numeric value carried by this mode.
fn value(&self) -> u64 {
match self {
Self::Absolute(n)
| Self::Extend(n)
| Self::Reduce(n)
| Self::AtMost(n)
| Self::AtLeast(n)
| Self::RoundDown(n)
| Self::RoundUp(n) => *n,
}
}
/// Multiply this mode's value by `factor` (for `--io-blocks` scaling).
///
/// Returns `None` on overflow.
fn scale(&self, factor: u64) -> Option<Self> {
let value = self.value().checked_mul(factor)?;
Some(match self {
Self::Absolute(_) => Self::Absolute(value),
Self::Extend(_) => Self::Extend(value),
Self::Reduce(_) => Self::Reduce(value),
Self::AtMost(_) => Self::AtMost(value),
Self::AtLeast(_) => Self::AtLeast(value),
Self::RoundDown(_) => Self::RoundDown(value),
Self::RoundUp(_) => Self::RoundUp(value),
})
}
/// Determine whether this mode specifies an absolute size.
fn is_absolute(&self) -> bool {
matches!(self, Self::Absolute(_))
@@ -123,10 +152,7 @@ fn app() -> Command {
Arg::new(options::IO_BLOCKS)
.short('o')
.long(options::IO_BLOCKS)
.help(
"treat SIZE as the number of I/O blocks of the file rather than bytes (NOT \
IMPLEMENTED)",
)
.help("treat SIZE as the number of I/O blocks of the file rather than bytes")
.action(ArgAction::SetTrue),
)
.arg(
@@ -167,70 +193,103 @@ fn app() -> Command {
)
}
/// Truncate the named file to the specified size.
/// The I/O block size of a file, falling back to 512 when the filesystem
/// reports 0 (mirrors GNU's `ST_BLKSIZE`).
#[cfg(unix)]
fn io_blocksize(file_metadata: &Metadata) -> u64 {
match file_metadata.blksize() {
0 => 512,
blksize => blksize,
}
}
#[cfg(not(unix))]
fn io_blocksize(_file_metadata: &Metadata) -> u64 {
512
}
/// Truncate one file according to `mode`.
///
/// If `create` is true, the file is created if it does not already exist. If
/// `size` is larger than the file, it is padded with zeros; if smaller, bytes
/// beyond `size` are discarded.
fn do_file_truncate(
host: &Host,
filename: &OsString,
create: bool,
size: u64,
) -> Result<(), String> {
let resolved = host.resolve(filename);
match OpenOptions::new().write(true).create(create).open(&resolved) {
Ok(file) => file.set_len(size),
Err(error) if error.kind() == ErrorKind::NotFound && !create => Ok(()),
Err(error) => Err(error),
}
.map_err(|error| format!("cannot open {} for writing: {error}", filename.quote()))
}
/// Unless `no_create` is set, the file is created if it does not already
/// exist. If the target size is larger than the file, it is padded with
/// zeros; if smaller, bytes beyond it are discarded. When `io_blocks` is
/// set, the size is scaled by the file's I/O block size, matching GNU
/// (which scales by the block size observed after opening the file).
fn file_truncate(
host: &Host,
no_create: bool,
io_blocks: bool,
reference_size: Option<u64>,
mode: &TruncateMode,
filename: &OsString,
) -> Result<(), String> {
let resolved = host.resolve(filename);
// Get the length of the file.
let file_size = match metadata(&resolved) {
Ok(metadata) => {
// A pipe has no length. Do this here to avoid a duplicate `stat()` syscall.
// A pipe has no length, and opening it for writing would block waiting
// for a reader; refuse it before the open.
#[cfg(unix)]
if metadata.file_type().is_fifo() {
if let Ok(pre_metadata) = metadata(&resolved) {
if pre_metadata.file_type().is_fifo() {
return Err(format!(
"cannot open {} for writing: No such device or address",
filename.to_string_lossy().quote()
));
}
metadata.len()
}
let create = !no_create;
let file = match OpenOptions::new().write(true).create(create).open(&resolved) {
Ok(file) => file,
Err(error) if error.kind() == ErrorKind::NotFound && !create => return Ok(()),
Err(error) => {
return Err(format!("cannot open {} for writing: {error}", filename.quote()));
},
Err(_) => 0,
};
let file_metadata = file
.metadata()
.map_err(|error| format!("cannot fstat {}: {error}", filename.quote()))?;
let mode = if io_blocks {
let blksize = io_blocksize(&file_metadata);
mode.scale(blksize).ok_or_else(|| {
format!(
"overflow in {} * {blksize} byte blocks for file {}",
mode.value(),
filename.quote()
)
})?
} else {
*mode
};
// The reference size is either the given reference file's size, or the size
// of the file to be truncated when no reference was provided.
let actual_reference_size = reference_size.unwrap_or(file_size);
let actual_reference_size = reference_size.unwrap_or_else(|| file_metadata.len());
let Some(truncate_size) = mode.to_size(actual_reference_size) else {
return Err("division by zero".to_string());
};
do_file_truncate(host, filename, !no_create, truncate_size)
file.set_len(truncate_size).map_err(|error| {
format!(
"failed to truncate {} at {truncate_size} bytes: {error}",
filename.quote()
)
})
}
fn truncate(
host: &mut Host,
no_create: bool,
_: bool,
io_blocks: bool,
reference: Option<String>,
size: Option<String>,
filenames: &[OsString],
) -> Result<(), String> {
if io_blocks && size.is_none() {
return Err("--io-blocks was specified but --size was not".to_string());
}
let reference_size = match reference {
Some(reference_path) => {
let reference_metadata = metadata(host.resolve(&reference_path)).map_err(|error| {
@@ -254,13 +313,21 @@ fn truncate(
None => TruncateMode::Extend(0),
};
// GNU rejects rounding to a multiple of zero up front, before touching
// any file.
if matches!(mode, TruncateMode::RoundDown(0) | TruncateMode::RoundUp(0)) {
return Err("division by zero".to_string());
}
// If a reference file has been given, the truncate mode cannot be absolute.
if reference_size.is_some() && mode.is_absolute() {
return Err("you must specify a relative '--size' with '--reference'".to_string());
}
for filename in filenames {
if let Err(error) = file_truncate(host, no_create, reference_size, &mode, filename) {
if let Err(error) =
file_truncate(host, no_create, io_blocks, reference_size, &mode, filename)
{
host.error(error, 1);
}
}
@@ -269,8 +336,10 @@ fn truncate(
}
/// Decide whether a character is one of the size modifiers, like `+` or `<`.
///
/// `=` is the BSD spelling of an absolute size.
fn is_modifier(c: char) -> bool {
c == '+' || c == '-' || c == '<' || c == '>' || c == '/' || c == '%'
c == '+' || c == '-' || c == '<' || c == '>' || c == '/' || c == '%' || c == '='
}
/// Parse a size string with an optional modifier symbol as its first character.
@@ -281,7 +350,10 @@ fn parse_mode_and_size(size_string: &str) -> Result<TruncateMode, ParseSizeError
if is_modifier(c) {
size_string = &size_string[1..];
}
let allow_list = allow_list_with_all_suffixes("EgGkKmMPQRtTYZ");
let mut allow_list = allow_list_with_all_suffixes("EgGkKmMPQRtTYZ");
// `b` counts 512-byte blocks (dd-style); accepted here for agent
// convenience even though GNU truncate omits it.
allow_list.push("b".to_string());
let allow_list_ref = allow_list.iter().map(AsRef::as_ref).collect::<Vec<&str>>();
Parser::default()
.with_allow_list(&allow_list_ref)
@@ -431,8 +503,76 @@ mod tests {
assert_eq!(parse_mode_and_size(">10"), Ok(TruncateMode::AtLeast(10)));
assert_eq!(parse_mode_and_size("/10"), Ok(TruncateMode::RoundDown(10)));
assert_eq!(parse_mode_and_size("%10"), Ok(TruncateMode::RoundUp(10)));
assert_eq!(parse_mode_and_size("=10"), Ok(TruncateMode::Absolute(10)));
assert_eq!(parse_mode_and_size("1kB"), Ok(TruncateMode::Absolute(1000)));
assert!(parse_mode_and_size("1b").is_err());
// `b` counts 512-byte blocks; rejecting it broke `truncate -s 2b f`.
assert_eq!(parse_mode_and_size("2b"), Ok(TruncateMode::Absolute(1024)));
}
#[test]
fn b_suffix_counts_512_byte_blocks() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("f"), b"x").unwrap();
let (code, _, stderr) = run_in(root.clone(), &["-s", "2b", "f"]);
assert_eq!((code, stderr.as_str()), (0, ""));
assert_eq!(len(&root.join("f")), 1024);
}
#[test]
fn bsd_equals_prefix_sets_absolute_size() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("f"), b"x").unwrap();
let (code, _, stderr) = run_in(root.clone(), &["-s", "=100", "f"]);
assert_eq!((code, stderr.as_str()), (0, ""));
assert_eq!(len(&root.join("f")), 100);
}
/// Regression: `-o` used to be parsed and then silently ignored, truncating
/// to the raw byte count (real data loss for `truncate -o -s 1 f`).
#[cfg(unix)]
#[test]
fn io_blocks_scales_size_by_file_blocksize() {
use std::os::unix::fs::MetadataExt;
let (_dir, root) = canonical_tempdir();
let path = root.join("f");
fs::write(&path, vec![0u8; 4096]).unwrap();
let blksize = fs::metadata(&path).unwrap().blksize();
assert!(blksize > 1, "test needs a real filesystem block size");
let (code, _, stderr) = run_in(root.clone(), &["-o", "-s", "1", "f"]);
assert_eq!((code, stderr.as_str()), (0, ""));
assert_eq!(len(&path), blksize, "-o must scale SIZE by st_blksize, not truncate to 1 byte");
}
#[test]
fn io_blocks_without_size_is_rejected() {
let (_dir, root) = canonical_tempdir();
fs::write(root.join("ref"), b"123").unwrap();
fs::write(root.join("f"), b"x").unwrap();
let (code, _, stderr) = run_in(root.clone(), &["-o", "-r", "ref", "f"]);
assert_eq!(code, 1);
assert!(stderr.contains("--io-blocks was specified but --size was not"));
}
/// Regression: `%0`/`/0` must fail up front with GNU's error, not create
/// or modify any operand.
#[test]
fn round_to_zero_is_division_by_zero_up_front() {
let (_dir, root) = canonical_tempdir();
for size in ["%0", "/0"] {
let (code, _, stderr) = run_in(root.clone(), &["-s", size, "missing"]);
assert_eq!(code, 1, "size {size} must fail");
assert!(stderr.contains("division by zero"), "size {size}: {stderr}");
assert!(
!root.join("missing").exists(),
"size {size} must not create the operand"
);
}
}
#[test]
+1 -2
View File
@@ -534,7 +534,6 @@ use std::{
use brush_core::{ShellExtensions, builtins::Registration};
use clap::{Arg, ArgAction, ArgMatches, Command, builder::ValueParser};
use thiserror::Error;
use unicode_width::UnicodeWidthChar;
use utf8::{BufReadDecoder, BufReadDecoderError};
use uucore::{
display::Quotable,
@@ -1107,7 +1106,7 @@ fn process_chunk<
*current_len += 8;
},
_ => {
*current_len += ch.width().unwrap_or(0);
*current_len += xutf::width_char(ch);
},
}
}
+43
View File
@@ -22,6 +22,10 @@ pub(crate) struct WhichCli {
#[arg(short = 'a', long = "all")]
all: bool,
/// Silent (BSD): print nothing, report matches via the exit status only.
#[arg(short = 's')]
silent: bool,
/// Command names to locate.
#[arg(value_name = "name")]
names: Vec<String>,
@@ -32,6 +36,11 @@ impl Utility for WhichCli {
const USAGE_ERROR: u8 = 2;
fn run(self, host: &mut Host) -> i32 {
// BSD and GNU which both treat a bare `which` as a usage error (exit 1).
if self.names.is_empty() {
let _ = writeln!(host.stderr, "usage: which [-as] program ...");
return 1;
}
let path_var = host.var("PATH").unwrap_or_default().to_owned();
let mut all_found = true;
@@ -57,10 +66,12 @@ impl Utility for WhichCli {
// which(1) reports missing names via the exit status only.
all_found = false;
}
if !self.silent {
for path in matches {
let _ = writeln!(host.stdout, "{}", path.display());
}
}
}
i32::from(!all_found)
}
@@ -101,6 +112,38 @@ mod tests {
(code, capture.out())
}
/// Bare `which` used to silently exit 0; BSD/GNU which report a usage
/// error on stderr and exit 1.
#[test]
fn no_operands_is_a_usage_error() {
let temp = tempfile::tempdir().expect("temp directory should be created");
let cwd = fs::canonicalize(temp.path()).expect("temp directory should canonicalize");
let cli = WhichCli::try_parse_from(["which"]).expect("test arguments should parse");
let (mut host, capture) = Host::for_test("which", Vec::new(), &cwd);
let code = cli.run(&mut host);
assert_eq!(code, 1);
assert_eq!(capture.out(), "");
assert_eq!(capture.err(), "usage: which [-as] program ...\n");
}
/// BSD `which -s` used to be rejected by clap with exit 2; it must print
/// nothing and report found/missing purely via the exit status, including
/// in the clustered `-as` spelling.
#[test]
fn silent_flag_suppresses_output_and_keeps_exit_status() {
let temp = tempfile::tempdir().expect("temp directory should be created");
let dir = fs::canonicalize(temp.path()).expect("temp directory should canonicalize");
place_file(&dir, "tool", true);
let path_var = dir.to_string_lossy();
assert_eq!(run_which(&["-s", "tool"], &path_var, &dir), (0, String::new()));
assert_eq!(run_which(&["-s", "missing"], &path_var, &dir), (1, String::new()));
assert_eq!(run_which(&["-as", "tool"], &path_var, &dir), (0, String::new()));
assert_eq!(run_which(&["-sa", "tool"], &path_var, &dir), (0, String::new()));
assert_eq!(run_which(&["-s", "tool", "missing"], &path_var, &dir), (1, String::new()));
}
#[test]
fn finds_only_executable_files() {
let temp = tempfile::tempdir().expect("temp directory should be created");
+73 -3
View File
@@ -26,6 +26,24 @@ matches_parser!(Yes, app);
impl Utility for Yes {
const NAME: &'static str = "yes";
fn rewrite_argv(mut argv: Vec<OsString>) -> Result<Vec<OsString>, String> {
// GNU yes (gnulib `parse_gnu_standard_options_only`) recognizes
// `--help`/`--version` only as the sole argument; everything else —
// `yes -n`, `yes --no`, even `yes --help me` — is echoed verbatim.
// Insert `--` so clap treats every remaining argument as an operand.
if argv.is_empty()
|| (argv.len() == 2 && matches!(argv[1].to_str(), Some("--help" | "--version")))
{
return Ok(argv);
}
// GNU consumes one leading `--` as the operand separator; ours replaces it.
if argv.get(1).is_some_and(|arg| arg.to_str() == Some("--")) {
argv.remove(1);
}
argv.insert(1, OsString::from("--"));
Ok(argv)
}
fn run(self, host: &mut Host) -> i32 {
let mut buffer = Vec::with_capacity(BUF_SIZE);
let Some(strings) = self.matches.get_many::<OsString>("STRING") else {
@@ -59,7 +77,9 @@ fn app() -> Command {
Arg::new("STRING")
.default_value("y")
.value_parser(ValueParser::os_string())
.action(ArgAction::Append),
.action(ArgAction::Append)
.allow_hyphen_values(true)
.trailing_var_arg(true),
)
.infer_long_args(true)
}
@@ -214,8 +234,12 @@ mod tests {
budget: usize,
fail_kind: io::ErrorKind,
) -> (i32, String, String) {
let parsed = Yes::try_parse_from(std::iter::once("yes").chain(arguments.iter().copied()))
.expect("test arguments should parse");
let argv: Vec<OsString> = std::iter::once("yes")
.chain(arguments.iter().copied())
.map(OsString::from)
.collect();
let argv = Yes::rewrite_argv(argv).expect("yes rewrite is infallible");
let parsed = Yes::try_parse_from(argv).expect("test arguments should parse");
let (mut host, capture) = Host::for_test("yes", Vec::new(), Path::new("/"));
let state = Arc::new(Mutex::new(WriterState { bytes: Vec::new(), remaining: budget }));
host.stdout = OpenFile::Stream(Box::new(FailingWriter {
@@ -228,6 +252,52 @@ mod tests {
(code, stdout, capture.err())
}
#[test]
fn hyphen_operands_are_echoed_not_parsed() {
// Failure mode: clap rejecting `yes -n` / `yes --no` / `yes -1` as
// unknown options where GNU yes echoes them.
let (code, stdout, stderr) = run_with(&["-n"], 6, io::ErrorKind::BrokenPipe);
assert_eq!(code, 0);
assert_eq!(stdout, "-n\n-n\n");
assert_eq!(stderr, "");
let (code, stdout, _) = run_with(&["--no", "-1"], 16, io::ErrorKind::BrokenPipe);
assert_eq!(code, 0);
assert_eq!(stdout, "--no -1\n--no -1\n");
}
#[test]
fn help_is_special_only_as_sole_argument() {
// Failure mode: `yes --help me` rendering help; GNU echoes "--help me".
let (code, stdout, _) = run_with(&["--help", "me"], 20, io::ErrorKind::BrokenPipe);
assert_eq!(code, 0);
assert_eq!(stdout, "--help me\n--help me\n");
}
#[test]
fn version_is_special_only_as_sole_argument() {
let (code, capture) = run_util::<Yes>(&["--version"], "", "/");
assert_eq!(code, 0);
assert!(capture.out().contains("0.8.0"), "stdout: {:?}", capture.out());
// Failure mode: `yes --version x` printing the version banner.
let (code, stdout, _) = run_with(&["--version", "x"], 24, io::ErrorKind::BrokenPipe);
assert_eq!(code, 0);
assert_eq!(stdout, "--version x\n--version x\n");
}
#[test]
fn leading_double_dash_is_operand_separator() {
// Failure mode: the rewrite doubling `--` so `yes --` echoes "--".
let (code, stdout, _) = run_with(&["--"], 4, io::ErrorKind::BrokenPipe);
assert_eq!(code, 0);
assert_eq!(stdout, "y\ny\n");
let (code, stdout, _) = run_with(&["--", "--help"], 14, io::ErrorKind::BrokenPipe);
assert_eq!(code, 0);
assert_eq!(stdout, "--help\n--help\n");
}
#[test]
fn broken_pipe_is_clean_exit() {
let (code, stdout, stderr) = run_with(&[], 100, io::ErrorKind::BrokenPipe);
+1 -7
View File
@@ -28,6 +28,7 @@ ast-grep-core.workspace = true
base64.workspace = true
clap.workspace = true
globset.workspace = true
heapless.workspace = true
fontdue.workspace = true
grep-matcher.workspace = true
futures.workspace = true
@@ -63,13 +64,8 @@ syntect.workspace = true
tokio.workspace = true
tokio-util.workspace = true
toml.workspace = true
unicode-segmentation.workspace = true
unicode-normalization.workspace = true
unicode-properties.workspace = true
unicode-script.workspace = true
xutf.workspace = true
zstd.workspace = true
unicode-width.workspace = true
xxhash-rust.workspace = true
[target.'cfg(target_os = "linux")'.dependencies]
@@ -78,8 +74,6 @@ atspi = { version = "=0.30.0", features = ["tokio", "zbus"] }
pipewire = { version = "=0.9.2", optional = true }
reis = { version = "=0.5.0", features = ["tokio"] }
x11rb = { version = "=0.13.2", features = ["randr", "xinput", "xtest"] }
fancy-regex.workspace = true # utok scanner differential oracle
tiktoken-rs.workspace = true # utok openai differential oracle
xkeysym = "=0.2.1"
[target.'cfg(any(target_os = "macos", target_os = "windows"))'.dependencies]
+6 -6
View File
@@ -8,10 +8,10 @@ use std::io::Cursor;
use arboard::{Clipboard, Error as ClipboardError, ImageData};
use image::{DynamicImage, ImageFormat, RgbaImage};
use napi::bindgen_prelude::*;
use napi::{JsString, bindgen_prelude::*};
use napi_derive::napi;
use crate::task;
use crate::{js, task};
/// Clipboard image payload encoded as PNG bytes.
#[napi(object)]
@@ -135,8 +135,8 @@ fn read_raw_cf_dib() -> Option<Vec<u8>> {
/// # Errors
/// Returns an error if clipboard access fails.
#[napi]
pub fn copy_to_clipboard(text: String) -> Result<()> {
set_clipboard_text(text)
pub fn copy_to_clipboard(text: JsString) -> Result<()> {
set_clipboard_text(&js::utf8(text)?)
}
/// Linux: keep a single `arboard::Clipboard` alive for the whole process.
@@ -153,7 +153,7 @@ pub fn copy_to_clipboard(text: String) -> Result<()> {
/// (`wl-clipboard-rs` forks its own serving process) but sharing the instance
/// is harmless there.
#[cfg(target_os = "linux")]
fn set_clipboard_text(text: String) -> Result<()> {
fn set_clipboard_text(text: &str) -> Result<()> {
use std::sync::OnceLock;
use parking_lot::Mutex;
@@ -180,7 +180,7 @@ fn set_clipboard_text(text: String) -> Result<()> {
/// calling thread also avoids worker-thread `AppKit` pasteboard warnings on
/// macOS.
#[cfg(not(target_os = "linux"))]
fn set_clipboard_text(text: String) -> Result<()> {
fn set_clipboard_text(text: &str) -> Result<()> {
let mut clipboard = Clipboard::new()
.map_err(|err| Error::from_reason(format!("Failed to access clipboard: {err}")))?;
clipboard
+21 -11
View File
@@ -24,9 +24,11 @@
use std::{collections::HashMap, rc::Rc};
use napi::bindgen_prelude::*;
use napi::{JsString, bindgen_prelude::*};
use napi_derive::napi;
use crate::js;
/// UTF-16 code unit for `\n`.
const LF: u16 = 0x000a;
@@ -333,8 +335,10 @@ fn concat_tokens(tokens: &[&[u16]]) -> Vec<u16> {
/// options). Change values keep line terminators, and common runs are joined
/// from the new text.
#[napi]
pub fn diff_lines(old_text: Utf16String, new_text: Utf16String) -> Vec<DiffChange> {
diff_lines_impl(&old_text, &new_text)
pub fn diff_lines(old_text: JsString, new_text: JsString) -> Result<Vec<DiffChange>> {
let old_text = js::utf16(old_text)?;
let new_text = js::utf16(new_text)?;
Ok(diff_lines_impl(&old_text, &new_text))
}
fn diff_lines_impl(old_text: &[u16], new_text: &[u16]) -> Vec<DiffChange> {
@@ -351,8 +355,10 @@ fn diff_lines_impl(old_text: &[u16], new_text: &[u16]) -> Vec<DiffChange> {
/// Callers that map line numbers — like hashline recovery — need the counts,
/// not another copy of the text.
#[napi]
pub fn diff_line_runs(old_text: Utf16String, new_text: Utf16String) -> Vec<DiffRun> {
diff_line_runs_impl(&old_text, &new_text)
pub fn diff_line_runs(old_text: JsString, new_text: JsString) -> Result<Vec<DiffRun>> {
let old_text = js::utf16(old_text)?;
let new_text = js::utf16(new_text)?;
Ok(diff_line_runs_impl(&old_text, &new_text))
}
fn diff_line_runs_impl(old_text: &[u16], new_text: &[u16]) -> Vec<DiffRun> {
@@ -387,11 +393,13 @@ fn no_newline_marker() -> Vec<u16> {
/// semantics. `context` defaults to 4 like jsdiff.
#[napi]
pub fn structured_patch_hunks(
old_text: Utf16String,
new_text: Utf16String,
old_text: JsString,
new_text: JsString,
context: Option<u32>,
) -> Vec<PatchHunk> {
structured_patch_hunks_impl(&old_text, &new_text, context)
) -> Result<Vec<PatchHunk>> {
let old_text = js::utf16(old_text)?;
let new_text = js::utf16(new_text)?;
Ok(structured_patch_hunks_impl(&old_text, &new_text, context))
}
fn structured_patch_hunks_impl(
@@ -894,8 +902,10 @@ fn word_post_process(changes: &mut [DiffChange]) {
/// Tokens carry surrounding whitespace, equality ignores it, and the
/// post-pass dedupes whitespace across change boundaries.
#[napi]
pub fn diff_words(old_text: Utf16String, new_text: Utf16String) -> Vec<DiffChange> {
diff_words_impl(&old_text, &new_text)
pub fn diff_words(old_text: JsString, new_text: JsString) -> Result<Vec<DiffChange>> {
let old_text = js::utf16(old_text)?;
let new_text = js::utf16(new_text)?;
Ok(diff_words_impl(&old_text, &new_text))
}
fn diff_words_impl(old_text: &[u16], new_text: &[u16]) -> Vec<DiffChange> {
+9 -2
View File
@@ -5,8 +5,11 @@
//! on a persistent sidecar because they lack a process-owned in-memory name
//! registry with automatic crash recovery.
use napi::JsString;
use napi_derive::napi;
use crate::js;
#[cfg(target_os = "linux")]
mod linux;
#[cfg(all(unix, not(target_os = "linux")))]
@@ -48,9 +51,13 @@ pub struct FileLock {
impl FileLock {
/// Try to acquire `path` without blocking.
#[napi(factory)]
pub fn try_acquire(path: String) -> napi::Result<Self> {
pub fn try_acquire(path: JsString) -> napi::Result<Self> {
let path = js::utf8(path)?;
let inner = platform::try_acquire(&path).map_err(|error| {
napi::Error::from_reason(format!("Failed to acquire native file lock for {path}: {error}"))
napi::Error::from_reason(format!(
"Failed to acquire native file lock for {}: {error}",
&*path
))
})?;
Ok(Self { inner })
}
+82 -48
View File
@@ -7,11 +7,22 @@
use std::{cell::RefCell, collections::HashMap, sync::OnceLock};
use napi::{JsString, Result};
use napi_derive::napi;
use syntect::parsing::{
ParseState, Scope, ScopeStack, ScopeStackOp, SyntaxDefinition, SyntaxReference, SyntaxSet,
};
use crate::js::{self, InlineStr};
/// One theme colour: an ANSI escape sequence such as `\x1b[38;2;255;0;0m`.
///
/// Decoded inline, so a whole palette crosses the boundary without touching
/// the heap. The longest sequence a theme can produce sets attributes plus
/// truecolor foreground and background — `\x1b[1;3;4;38;2;255;255;255;48;2;
/// 255;255;255m`, 42 bytes — which the 47 usable bytes cover.
pub type Color = InlineStr<48>;
static SYNTAX_SET: OnceLock<SyntaxSet> = OnceLock::new();
static SCOPE_MATCHERS: OnceLock<ScopeMatchers> = OnceLock::new();
@@ -150,27 +161,38 @@ fn get_scope_matchers() -> &'static ScopeMatchers {
#[napi(object)]
pub struct HighlightColors {
/// ANSI color for comments.
pub comment: String,
#[napi(ts_type = "string")]
pub comment: Color,
/// ANSI color for keywords.
pub keyword: String,
#[napi(ts_type = "string")]
pub keyword: Color,
/// ANSI color for function names.
pub function: String,
#[napi(ts_type = "string")]
pub function: Color,
/// ANSI color for variables and identifiers.
pub variable: String,
#[napi(ts_type = "string")]
pub variable: Color,
/// ANSI color for string literals.
pub string: String,
#[napi(ts_type = "string")]
pub string: Color,
/// ANSI color for numeric literals.
pub number: String,
#[napi(ts_type = "string")]
pub number: Color,
/// ANSI color for type identifiers.
pub r#type: String,
#[napi(ts_type = "string")]
pub r#type: Color,
/// ANSI color for operators.
pub operator: String,
#[napi(ts_type = "string")]
pub operator: Color,
/// ANSI color for punctuation tokens.
pub punctuation: String,
#[napi(ts_type = "string")]
pub punctuation: Color,
/// ANSI color for diff inserted lines.
pub inserted: Option<String>,
#[napi(ts_type = "string")]
pub inserted: Option<Color>,
/// ANSI color for diff deleted lines.
pub deleted: Option<String>,
#[napi(ts_type = "string")]
pub deleted: Option<Color>,
}
/// Language alias mappings: (aliases, target syntax name).
@@ -383,21 +405,31 @@ fn find_syntax<'a>(ss: &'a SyntaxSet, lang: &str) -> Option<&'a SyntaxReference>
/// Highlighted code with ANSI color codes, or the original code if highlighting
/// fails.
#[napi]
pub fn highlight_code(code: String, lang: Option<String>, colors: HighlightColors) -> String {
pub fn highlight_code(
code: JsString,
lang: Option<JsString>,
colors: HighlightColors,
) -> Result<String> {
let code = js::utf8(code)?;
let lang = lang.map(js::utf8).transpose()?;
Ok(highlight_code_impl(&code, lang.as_deref(), &colors))
}
fn highlight_code_impl(code: &str, lang: Option<&str>, colors: &HighlightColors) -> String {
let inserted = colors.inserted.as_deref().unwrap_or("");
let deleted = colors.deleted.as_deref().unwrap_or("");
// Color palette as array for quick indexing
let palette = [
colors.comment.as_str(), // 0
colors.keyword.as_str(), // 1
colors.function.as_str(), // 2
colors.variable.as_str(), // 3
colors.string.as_str(), // 4
colors.number.as_str(), // 5
colors.r#type.as_str(), // 6
colors.operator.as_str(), // 7
colors.punctuation.as_str(), // 8
&*colors.comment, // 0
&*colors.keyword, // 1
&*colors.function, // 2
&*colors.variable, // 3
&*colors.string, // 4
&*colors.number, // 5
&*colors.r#type, // 6
&*colors.operator, // 7
&*colors.punctuation, // 8
inserted, // 9
deleted, // 10
];
@@ -405,7 +437,7 @@ pub fn highlight_code(code: String, lang: Option<String>, colors: HighlightColor
let ss = get_syntax_set();
// Find syntax for the language
let syntax = match &lang {
let syntax = match lang {
Some(l) => find_syntax(ss, l),
None => None,
}
@@ -415,7 +447,7 @@ pub fn highlight_code(code: String, lang: Option<String>, colors: HighlightColor
let mut scope_stack = ScopeStack::new();
let mut result = String::with_capacity(code.len() * 2);
for line in syntect::util::LinesWithEndings::from(code.as_str()) {
for line in syntect::util::LinesWithEndings::from(code) {
let Ok(ops) = parse_state.parse_line(line, ss) else {
// Parse error - append unhighlighted line and continue
result.push_str(line);
@@ -477,16 +509,19 @@ pub fn highlight_code(code: String, lang: Option<String>, colors: HighlightColor
/// Returns true if the language has either direct support or a fallback
/// mapping.
#[napi]
pub fn supports_language(lang: String) -> bool {
if is_known_alias(&lang) {
pub fn supports_language(lang: JsString) -> Result<bool> {
Ok(supports_language_impl(&js::utf8(lang)?))
}
fn supports_language_impl(lang: &str) -> bool {
if is_known_alias(lang) {
return true;
}
// Fall back to direct syntax lookup
let ss = get_syntax_set();
find_syntax(ss, &lang).is_some()
find_syntax(ss, lang).is_some()
}
/// Get list of supported languages.
#[napi]
pub fn get_supported_languages() -> Vec<String> {
@@ -500,15 +535,15 @@ mod tests {
fn test_colors() -> HighlightColors {
HighlightColors {
comment: "<c>".to_string(),
keyword: "<k>".to_string(),
function: "<f>".to_string(),
variable: "<v>".to_string(),
string: "<s>".to_string(),
number: "<n>".to_string(),
r#type: "<t>".to_string(),
operator: "<o>".to_string(),
punctuation: "<p>".to_string(),
comment: Color::new("<c>").unwrap(),
keyword: Color::new("<k>").unwrap(),
function: Color::new("<f>").unwrap(),
variable: Color::new("<v>").unwrap(),
string: Color::new("<s>").unwrap(),
number: Color::new("<n>").unwrap(),
r#type: Color::new("<t>").unwrap(),
operator: Color::new("<o>").unwrap(),
punctuation: Color::new("<p>").unwrap(),
inserted: None,
deleted: None,
}
@@ -517,14 +552,13 @@ mod tests {
#[test]
fn highlights_nix_vendored_syntax() {
assert!(get_supported_languages().contains(&"Nix".to_string()));
assert!(supports_language("nix".to_string()));
assert!(supports_language_impl("nix"));
let out = highlight_code(
let out = highlight_code_impl(
"{ pkgs ? import <nixpkgs> {} }:\nlet message = \"hello\"; in pkgs.writeText \"msg\" \
message # greeting\n"
.to_string(),
Some("nix".to_string()),
test_colors(),
message # greeting\n",
Some("nix"),
&test_colors(),
);
assert!(out.contains("<k>let"));
assert!(out.contains("<s>hello"));
@@ -534,13 +568,13 @@ mod tests {
#[test]
fn highlights_mermaid_vendored_syntax() {
assert!(get_supported_languages().contains(&"Mermaid".to_string()));
assert!(supports_language("mermaid".to_string()));
assert!(supports_language("mmd".to_string()));
assert!(supports_language_impl("mermaid"));
assert!(supports_language_impl("mmd"));
let out = highlight_code(
"graph TD\n A[\"Start\"] --> B\n %% note\n".to_string(),
Some("mermaid".to_string()),
test_colors(),
let out = highlight_code_impl(
"graph TD\n A[\"Start\"] --> B\n %% note\n",
Some("mermaid"),
&test_colors(),
);
assert!(out.contains("<k>graph"));
assert!(out.contains("<s>Start"));
+7 -6
View File
@@ -3,10 +3,10 @@
use html_to_markdown_rs::{
ConversionOptions, PreprocessingOptions, PreprocessingPreset, WarningKind, convert,
};
use napi::bindgen_prelude::*;
use napi::{JsString, bindgen_prelude::*};
use napi_derive::napi;
use crate::task;
use crate::{js::into_string, task};
/// Options for HTML to Markdown conversion.
#[napi(object)]
@@ -24,14 +24,15 @@ pub struct HtmlToMarkdownOptions {
/// Returns an error if the conversion fails or the worker task aborts.
#[napi]
pub fn html_to_markdown(
html: String,
html: JsString,
options: Option<HtmlToMarkdownOptions>,
) -> task::Promise<String> {
) -> Result<task::Promise<String>> {
let html = into_string(html)?;
let options = options.unwrap_or_default();
let clean_content = options.clean_content.unwrap_or(false);
let skip_images = options.skip_images.unwrap_or(false);
task::blocking("html_to_markdown", (), move |_| {
Ok(task::blocking("html_to_markdown", (), move |_| {
let conversion_opts = ConversionOptions {
skip_images,
preprocessing: PreprocessingOptions {
@@ -54,5 +55,5 @@ pub fn html_to_markdown(
return Err(Error::from_reason(format!("Conversion error: {}", warning.message)));
}
Ok(result.content.unwrap_or_default())
})
}))
}
+6 -3
View File
@@ -4,9 +4,11 @@
//! JavaScript-facing shapes plus conversions between walker entries and N-API
//! payloads.
use napi::bindgen_prelude::*;
use napi::{JsString, bindgen_prelude::*};
use napi_derive::napi;
use crate::js;
/// Resolved filesystem entry kind for glob filters and match metadata.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
#[napi]
@@ -75,9 +77,10 @@ pub(crate) fn map_walker_error<E: std::fmt::Display>(err: pi_walker::WalkError<E
/// Intended to be called after agent file mutations: write, edit, rename, or
/// delete.
#[napi]
pub fn invalidate_fs_scan_cache(path: Option<String>) {
pub fn invalidate_fs_scan_cache(path: Option<JsString>) -> Result<()> {
match path {
Some(path) => pi_walker::invalidate_path_string(&path),
Some(path) => pi_walker::invalidate_path_string(&js::utf8(path)?),
None => pi_walker::invalidate_all(),
}
Ok(())
}
+7 -3
View File
@@ -22,7 +22,10 @@ use napi::bindgen_prelude::*;
use napi_derive::napi;
use pi_iso::{BackendKind, ChangeKind, Diff, FileChange, IsoError, IsolationBackend};
use crate::js;
const ISO_UNAVAILABLE_PREFIX: &str = "ISO_UNAVAILABLE:";
const ISO_UNAVAILABLE_WITH_LEADING_SPACE: &str = " ISO_UNAVAILABLE:";
/// Isolation backend identifier. Numeric so the JS side can `switch` on
/// the enum without string comparisons.
@@ -174,9 +177,10 @@ pub async fn iso_diff(lower: String, merged: String) -> Result<IsoDiff> {
/// Use this to distinguish "this backend isn't installed" from a hard
/// failure when handling caught errors on the JS side.
#[napi]
pub fn iso_is_unavailable_error(message: String) -> bool {
message.starts_with(ISO_UNAVAILABLE_PREFIX)
|| message.contains(&format!(" {ISO_UNAVAILABLE_PREFIX}"))
pub fn iso_is_unavailable_error(message: napi::JsString) -> Result<bool> {
let message = js::utf8(message)?;
Ok(message.starts_with(ISO_UNAVAILABLE_PREFIX)
|| message.contains(ISO_UNAVAILABLE_WITH_LEADING_SPACE))
}
const fn to_napi_kind(kind: BackendKind) -> IsoBackendKind {
+475
View File
@@ -0,0 +1,475 @@
//! Borrowing JavaScript strings at the N-API boundary.
//!
//! Node-API never hands out a pointer into a JS string's backing store: every
//! accessor writes code units into a caller-owned buffer. The copy is
//! unavoidable, the *allocation* is not — [`utf16`] and [`utf8`] read into a
//! fixed per-thread scratch arena and hand back a guard that derefs to the
//! borrowed text, releasing its range on drop. Text that fits the arena costs
//! zero allocations; longer text spills to one owned `Vec`. Several guards can
//! be live at once (diff pairs, colour palettes) — each owns a disjoint range.
//!
//! [`utf16`] is the default: it is the JS string's own encoding, so it is the
//! only accessor that never transcodes. [`utf8`] exists for algorithms that are
//! byte- or `str`-shaped (terminal escape parsing, syntect, paths). Reach for
//! [`into_string`] only when the text must outlive the N-API callback — a value
//! moved into a worker task, a channel, or an async body.
//! Short, bounded strings — ANSI colours, font names, language ids — skip the
//! arena entirely: [`InlineStr`] decodes them into a fixed-size array that
//! lives in the struct itself, so a whole options object crosses with no
//! allocation.
use std::{
cell::{Cell, UnsafeCell},
fmt,
ops::{Deref, Range},
ptr::{self, NonNull},
slice, str,
};
use napi::{
Error, JsString, JsValue, Result, Status,
bindgen_prelude::{FromNapiValue, ToNapiValue, TypeName, ValidateNapiValue},
sys,
};
/// Scratch bytes per thread. Only the JS thread reaches this module (N-API
/// handles are not `Send`), so the footprint is effectively process-global.
const SCRATCH_LEN: usize = 64 * 1024;
/// Fixed-size bump arena backing [`Utf16`]/[`Utf8`] guards.
///
/// The base address is stable for the thread's lifetime (the array never
/// grows), so guards may hold raw pointers into it. Soundness rests on range
/// discipline, not a borrow flag: every committed range is disjoint, new reads
/// only touch bytes past `offset`, and no reference to the whole array is ever
/// formed — all access goes through raw pointers into a caller's own range.
struct Arena {
/// Stored as `u16` units purely for the 2-alignment UTF-16 fills need;
/// UTF-8 fills reinterpret the same bytes at alignment 1.
buf: UnsafeCell<[u16; SCRATCH_LEN / 2]>,
/// Bytes handed out. Fills bump it; drops roll it back (see [`Self::release`]).
offset: Cell<usize>,
/// Live scratch-backed guards. Hitting zero resets `offset`, so a non-LIFO
/// drop order leaks at most until the last guard goes away.
live: Cell<usize>,
}
thread_local! {
static ARENA: Arena = const {
Arena {
buf: UnsafeCell::new([0; SCRATCH_LEN / 2]),
offset: Cell::new(0),
live: Cell::new(0),
}
};
}
impl Arena {
const fn base(&self) -> *mut u8 {
self.buf.get().cast()
}
/// Free tail aligned to `align` (a power of two): byte offset and length.
const fn tail(&self, align: usize) -> (usize, usize) {
let start = (self.offset.get() + align - 1) & !(align - 1);
(start, SCRATCH_LEN.saturating_sub(start))
}
/// Record `start..start + len` as owned by a new guard.
fn commit(&self, start: usize, len: usize) {
self.offset.set(start + len);
self.live.set(self.live.get() + 1);
}
/// Return `start..end`. The topmost range rolls the bump pointer back
/// (LIFO drops recycle immediately); otherwise the bytes are stranded
/// until `live` reaches zero and the whole arena resets.
fn release(&self, start: usize, end: usize) {
let live = self.live.get() - 1;
self.live.set(live);
if live == 0 {
self.offset.set(0);
} else if self.offset.get() == end {
self.offset.set(start);
}
}
}
/// Text read from a JS string: borrowed from the thread's scratch arena when
/// it fits, spilled to one owned `Vec` when it does not.
enum TextRepr<T> {
/// Range inside [`ARENA`]; `Drop` releases it. `NonNull` keeps this
/// variant `!Send`, so the pointer can never outlive its thread's TLS.
Scratch { ptr: NonNull<T>, len: usize },
/// Heap spill for text longer than the arena's free tail.
Owned(Vec<T>),
}
impl<T> TextRepr<T> {
#[inline]
fn as_slice(&self) -> &[T] {
match self {
// SAFETY: the constructor committed `ptr..ptr + len` to this guard;
// the arena never moves and no other guard overlaps the range.
Self::Scratch { ptr, len } => unsafe { slice::from_raw_parts(ptr.as_ptr(), *len) },
Self::Owned(vec) => vec,
}
}
}
impl<T> Drop for TextRepr<T> {
fn drop(&mut self) {
if let Self::Scratch { ptr, len } = *self {
ARENA.with(|arena| {
let start = ptr.as_ptr().addr() - arena.base().addr();
arena.release(start, start + len * size_of::<T>());
});
}
}
}
/// Borrowed UTF-16 code units of a JS string, backed by the scratch arena.
///
/// Unlike `JsString::into_utf16`, the view excludes the NUL terminator
/// Node-API appends, so `&*guard` is exactly the string's code units.
pub struct Utf16(TextRepr<u16>);
impl Deref for Utf16 {
type Target = [u16];
#[inline]
fn deref(&self) -> &[u16] {
self.0.as_slice()
}
}
/// Borrowed UTF-8 bytes of a JS string, backed by the scratch arena.
pub struct Utf8(TextRepr<u8>);
impl Deref for Utf8 {
type Target = str;
#[inline]
fn deref(&self) -> &str {
// SAFETY: utf8 validates the bytes before constructing Utf8.
unsafe { str::from_utf8_unchecked(self.0.as_slice()) }
}
}
/// Borrow `value` as UTF-16 code units using the thread's scratch arena.
///
/// The happy path is a single N-API call into the arena's free tail; text
/// that does not fit is measured and read into an owned spill buffer.
#[inline]
pub fn utf16(value: JsString<'_>) -> Result<Utf16> {
let raw = value.value();
ARENA.with(|arena| {
let (start, avail_bytes) = arena.tail(2);
let avail = avail_bytes / 2;
if avail >= 2 {
// SAFETY: `start..start + avail_bytes` is past every committed range,
// and the base is 2-aligned with `start` aligned up.
let ptr = unsafe { arena.base().add(start) }.cast::<u16>();
let mut written = 0;
// SAFETY: `raw` is a JS string owned by the live callback; Node-API
// writes at most `avail - 1` units plus a NUL into the free tail.
let status = unsafe {
sys::napi_get_value_string_utf16(raw.env, raw.value, ptr, avail, &mut written)
};
napi::check_status!(status, "Failed to read JavaScript string")?;
if written < avail - 1 {
arena.commit(start, written * 2);
return Ok(Utf16(TextRepr::Scratch {
ptr: NonNull::new(ptr).unwrap(),
len: written,
}));
}
}
let mut len = 0;
// SAFETY: a null buffer asks Node-API for the code-unit length only.
let status = unsafe {
sys::napi_get_value_string_utf16(raw.env, raw.value, ptr::null_mut(), 0, &mut len)
};
napi::check_status!(status, "Failed to measure JavaScript string")?;
let mut buf: Vec<u16> = Vec::with_capacity(len + 1);
let mut written = 0;
// SAFETY: `buf` holds the measured length plus the NUL slot.
let status = unsafe {
sys::napi_get_value_string_utf16(
raw.env,
raw.value,
buf.as_mut_ptr(),
len + 1,
&mut written,
)
};
napi::check_status!(status, "Failed to read JavaScript string")?;
// SAFETY: Node-API initialised `written` units.
unsafe { buf.set_len(written) };
Ok(Utf16(TextRepr::Owned(buf)))
})
}
/// Borrow `value` as UTF-8 using the thread's scratch arena.
///
/// Same shape as [`utf16`], plus UTF-8 validation before the guard exists.
#[inline]
pub fn utf8(value: JsString<'_>) -> Result<Utf8> {
let raw = value.value();
ARENA.with(|arena| {
let (start, avail) = arena.tail(1);
if avail >= 2 {
// SAFETY: `start..start + avail` is past every committed range.
let ptr = unsafe { arena.base().add(start) };
let mut written = 0;
// SAFETY: `raw` is a JS string owned by the live callback; Node-API
// writes at most `avail - 1` bytes plus a NUL into the free tail.
let status = unsafe {
sys::napi_get_value_string_utf8(raw.env, raw.value, ptr.cast(), avail, &mut written)
};
napi::check_status!(status, "Failed to read JavaScript string")?;
if written < avail - 1 {
// SAFETY: Node-API initialised `written` bytes at `ptr`.
let bytes = unsafe { slice::from_raw_parts(ptr, written) };
if let Err(error) = str::from_utf8(bytes) {
return Err(Error::new(Status::InvalidArg, error.to_string()));
}
arena.commit(start, written);
return Ok(Utf8(TextRepr::Scratch {
ptr: NonNull::new(ptr).unwrap(),
len: written,
}));
}
}
let mut len = 0;
// SAFETY: a null buffer asks Node-API for the byte length only.
let status = unsafe {
sys::napi_get_value_string_utf8(raw.env, raw.value, ptr::null_mut(), 0, &mut len)
};
napi::check_status!(status, "Failed to measure JavaScript string")?;
let mut buf: Vec<u8> = Vec::with_capacity(len + 1);
let mut written = 0;
// SAFETY: `buf` holds the measured length plus the NUL slot.
let status = unsafe {
sys::napi_get_value_string_utf8(
raw.env,
raw.value,
buf.as_mut_ptr().cast(),
len + 1,
&mut written,
)
};
napi::check_status!(status, "Failed to read JavaScript string")?;
// SAFETY: Node-API initialised `written` bytes.
unsafe { buf.set_len(written) };
if let Err(error) = str::from_utf8(&buf) {
return Err(Error::new(Status::InvalidArg, error.to_string()));
}
Ok(Utf8(TextRepr::Owned(buf)))
})
}
/// Append `value`'s UTF-16 code units to `out` and return their span.
///
/// For batches: one growing buffer holds every element, so an array costs a
/// single allocation instead of one per string, and the spans can then be
/// counted in parallel while the reads themselves stay on the JS thread —
/// Node-API handles are not `Send`.
pub fn utf16_append(value: JsString<'_>, out: &mut Vec<u16>) -> Result<Range<usize>> {
let raw = value.value();
let start = out.len();
let mut len = 0;
// SAFETY: `raw` is a JS string owned by the live callback; a null buffer asks
// Node-API for the code-unit length only.
let status =
unsafe { sys::napi_get_value_string_utf16(raw.env, raw.value, ptr::null_mut(), 0, &mut len) };
napi::check_status!(status, "Failed to measure JavaScript string")?;
out.resize(start + len + 1, 0);
let mut written = 0;
// SAFETY: same string, and the tail from `start` holds the measured length
// plus the NUL slot Node-API writes.
let status = unsafe {
sys::napi_get_value_string_utf16(
raw.env,
raw.value,
out[start..].as_mut_ptr(),
len + 1,
&mut written,
)
};
napi::check_status!(status, "Failed to read JavaScript string")?;
out.truncate(start + written);
Ok(start..out.len())
}
/// Copy `value` into an owned `String`.
///
/// Only for text that outlives the N-API callback: a worker task, a channel
/// message, or an async body. Synchronous consumers must borrow instead.
pub fn into_string(value: JsString<'_>) -> Result<String> {
let raw = value.value();
// SAFETY: `raw` is a validated JS string from the current callback.
unsafe { String::from_napi_value(raw.env, raw.value) }
}
/// A JS string decoded into a fixed-capacity inline buffer, no heap involved.
///
/// Holds `N` bytes with a `u8` length, so the whole value is `N + 1` bytes and
/// an options struct full of them costs nothing to build. Node-API needs one
/// byte for its NUL terminator, leaving [`Self::CAPACITY`] usable; longer input
/// is a caller error rather than a silent truncation, so an escape sequence can
/// never arrive half-copied.
///
/// Storage is UTF-8: every consumer of these values wants `&str`, and the
/// inputs are ASCII, so this is the encoding that avoids a transcode at the
/// point of use. Use [`utf16`] for text whose consumer works in code units.
#[derive(Clone)]
pub struct InlineStr<const N: usize>(heapless::Vec<u8, N, u8>);
impl<const N: usize> InlineStr<N> {
/// Usable bytes, excluding the NUL slot Node-API requires.
pub const CAPACITY: usize = N - 1;
/// Build from Rust text, for tests and native-side defaults.
pub fn new(text: &str) -> Result<Self> {
if text.len() > Self::CAPACITY {
return Err(too_long(text.len(), Self::CAPACITY));
}
heapless::Vec::from_slice(text.as_bytes())
.map(Self)
.map_err(|_| too_long(text.len(), Self::CAPACITY))
}
}
fn too_long(len: usize, capacity: usize) -> Error {
Error::new(Status::InvalidArg, format!("string is {len} bytes, expected at most {capacity}"))
}
impl<const N: usize> Deref for InlineStr<N> {
type Target = str;
fn deref(&self) -> &str {
// SAFETY: both constructors validate the bytes as UTF-8 before storing
// them, and the buffer is immutable afterwards.
unsafe { str::from_utf8_unchecked(&self.0) }
}
}
impl<const N: usize> fmt::Debug for InlineStr<N> {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
fmt::Debug::fmt(&**self, f)
}
}
impl<const N: usize> TypeName for InlineStr<N> {
fn type_name() -> &'static str {
"String"
}
fn value_type() -> napi::ValueType {
napi::ValueType::String
}
}
impl<const N: usize> ValidateNapiValue for InlineStr<N> {}
impl<const N: usize> FromNapiValue for InlineStr<N> {
unsafe fn from_napi_value(env: sys::napi_env, napi_val: sys::napi_value) -> Result<Self> {
let mut len = 0;
// SAFETY: `napi_val` is a JS string owned by the live callback; a null
// buffer asks Node-API for the byte length only.
let status =
unsafe { sys::napi_get_value_string_utf8(env, napi_val, ptr::null_mut(), 0, &mut len) };
napi::check_status!(status, "Failed to measure JavaScript string")?;
if len > Self::CAPACITY {
return Err(too_long(len, Self::CAPACITY));
}
let mut buf: heapless::Vec<u8, N, u8> = heapless::Vec::new();
buf.resize_default(N)
.map_err(|_| too_long(len, Self::CAPACITY))?;
let mut written = 0;
// SAFETY: same string, and `buf` is filled to `N`, which holds the measured
// length plus the NUL terminator Node-API writes.
let status = unsafe {
sys::napi_get_value_string_utf8(env, napi_val, buf.as_mut_ptr().cast(), N, &mut written)
};
napi::check_status!(status, "Failed to read JavaScript string")?;
buf.truncate(written);
if let Err(error) = str::from_utf8(&buf) {
return Err(Error::new(Status::InvalidArg, error.to_string()));
}
Ok(Self(buf))
}
}
impl<const N: usize> ToNapiValue for InlineStr<N> {
unsafe fn to_napi_value(env: sys::napi_env, val: Self) -> Result<sys::napi_value> {
// SAFETY: `env` is the live callback environment.
unsafe { ToNapiValue::to_napi_value(env, &*val) }
}
}
#[cfg(test)]
mod tests {
use super::*;
fn arena() -> Arena {
Arena {
buf: UnsafeCell::new([0; SCRATCH_LEN / 2]),
offset: Cell::new(0),
live: Cell::new(0),
}
}
/// Live guards own disjoint ranges; overlap would alias the derefs (UB).
#[test]
fn commits_never_overlap_live_ranges() {
let a = arena();
let (s1, _) = a.tail(1);
a.commit(s1, 100);
let (s2, _) = a.tail(2);
assert!(s2 >= s1 + 100);
a.commit(s2, 50);
let (s3, _) = a.tail(1);
assert!(s3 >= s2 + 50);
}
/// LIFO drops recycle immediately; the next fill reuses the range.
#[test]
fn lifo_release_rolls_back() {
let a = arena();
a.commit(0, 100);
a.commit(100, 50);
a.release(100, 150);
assert_eq!(a.tail(1).0, 100);
a.release(0, 100);
assert_eq!(a.tail(1).0, 0);
}
/// Non-LIFO drops strand bytes only until the last guard goes away.
#[test]
fn arena_resets_when_last_guard_drops() {
let a = arena();
a.commit(0, 100);
a.commit(100, 50);
a.release(0, 100);
assert_eq!(a.tail(1).0, 150, "inner range stays stranded while a guard is live");
a.release(100, 150);
assert_eq!(a.tail(1).0, 0);
}
/// A utf16 fill after an odd utf8 commit must get a 2-aligned range.
#[test]
fn utf16_tail_is_aligned() {
let a = arena();
a.commit(0, 7);
let (start, len) = a.tail(2);
assert_eq!(start, 8);
assert_eq!(len, SCRATCH_LEN - 8);
}
}
+37 -27
View File
@@ -12,9 +12,12 @@
use std::borrow::Cow;
use napi::{JsString, Result};
use napi_derive::napi;
use phf::phf_map;
use crate::js;
const LOCK_MASK: u32 = 64 + 128;
// Internal sentinel codes for CSI 1;mod <letter> forms:
@@ -298,11 +301,20 @@ static LETTERS: [&str; 26] = [
/// base layout key) and modifier bits.
#[napi]
pub fn matches_kitty_sequence(
data: String,
data: JsString,
expected_codepoint: i32,
expected_modifier: u32,
) -> Result<bool> {
let data = js::utf8(data)?;
Ok(matches_kitty_sequence_inner(data.as_bytes(), expected_codepoint, expected_modifier))
}
fn matches_kitty_sequence_inner(
data: &[u8],
expected_codepoint: i32,
expected_modifier: u32,
) -> bool {
let Some(parsed) = parse_kitty_sequence_bytes(data.as_bytes()) else {
let Some(parsed) = parse_kitty_sequence_bytes(data) else {
return false;
};
@@ -378,40 +390,46 @@ const fn is_symbol_key(cp: i32) -> bool {
///
/// Returns a key id like "escape" or "ctrl+c", or None if unrecognized.
#[napi]
pub fn parse_key(data: String, kitty_protocol_active: bool) -> Option<String> {
parse_key_inner(data.as_bytes(), kitty_protocol_active).map(|s| s.into_owned())
pub fn parse_key(data: JsString, kitty_protocol_active: bool) -> Result<Option<String>> {
let data = js::utf8(data)?;
Ok(parse_key_inner(data.as_bytes(), kitty_protocol_active).map(|key| key.into_owned()))
}
/// Check if input matches a legacy escape sequence for the given key name.
///
/// Returns true only when the byte sequence maps to the exact key identifier.
#[napi]
pub fn matches_legacy_sequence(data: String, key_name: String) -> bool {
LEGACY_SEQUENCES
pub fn matches_legacy_sequence(data: JsString, key_name: JsString) -> Result<bool> {
let data = js::utf8(data)?;
let key_name = js::utf8(key_name)?;
Ok(LEGACY_SEQUENCES
.get(data.as_bytes())
.is_some_and(|&id| id == key_name)
.is_some_and(|&id| id == &*key_name))
}
/// Match input data against a key identifier string.
///
/// Returns true when the bytes represent the specified key with modifiers.
#[napi]
pub fn matches_key(data: String, key_id: String, kitty_protocol_active: bool) -> bool {
matches_key_inner(data.as_bytes(), &key_id, kitty_protocol_active)
pub fn matches_key(data: JsString, key_id: JsString, kitty_protocol_active: bool) -> Result<bool> {
let data = js::utf8(data)?;
let key_id = js::utf8(key_id)?;
Ok(matches_key_inner(data.as_bytes(), &key_id, kitty_protocol_active))
}
/// Parse a Kitty keyboard protocol sequence.
///
/// Returns a structured parse result when the input is a valid Kitty sequence.
#[napi]
pub fn parse_kitty_sequence(data: String) -> Option<ParsedKittyResult> {
parse_kitty_sequence_bytes(data.as_bytes()).map(|p| ParsedKittyResult {
codepoint: p.codepoint,
shifted_key: p.shifted_key,
base_layout_key: p.base_layout_key,
modifier: p.modifier,
event_type: optional_kitty_event_type(p.event_type),
})
pub fn parse_kitty_sequence(data: JsString) -> Result<Option<ParsedKittyResult>> {
let data = js::utf8(data)?;
Ok(parse_kitty_sequence_bytes(data.as_bytes()).map(|parsed| ParsedKittyResult {
codepoint: parsed.codepoint,
shifted_key: parsed.shifted_key,
base_layout_key: parsed.base_layout_key,
modifier: parsed.modifier,
event_type: optional_kitty_event_type(parsed.event_type),
}))
}
// =============================================================================
@@ -1590,20 +1608,12 @@ mod tests {
let plain_cyrillic_c = b"\x1b[1089::99u";
assert!(!matches_key_inner(plain_cyrillic_c, "c", true));
assert_eq!(parse_key_inner(plain_cyrillic_c, true).as_deref(), None);
assert!(!matches_kitty_sequence(
String::from_utf8_lossy(plain_cyrillic_c).into_owned(),
i32::from(b'c'),
0,
));
assert!(!matches_kitty_sequence_inner(plain_cyrillic_c, i32::from(b'c'), 0));
let ctrl_cyrillic_c = b"\x1b[1089::99;5u";
assert!(matches_key_inner(ctrl_cyrillic_c, "ctrl+c", true));
assert_eq!(parse_key_inner(ctrl_cyrillic_c, true).as_deref(), Some("ctrl+c"));
assert!(matches_kitty_sequence(
String::from_utf8_lossy(ctrl_cyrillic_c).into_owned(),
i32::from(b'c'),
MOD_CTRL,
));
assert!(matches_kitty_sequence_inner(ctrl_cyrillic_c, i32::from(b'c'), MOD_CTRL));
}
#[test]
+1
View File
@@ -39,6 +39,7 @@ pub mod grep;
pub mod highlight;
pub mod html;
pub mod iofs;
pub mod js;
pub mod keys;
pub mod live;
/// PDF inspection and Markdown conversion.
+5 -5
View File
@@ -8,14 +8,14 @@
use std::time::Duration;
use napi::{
Env, Result,
Env, JsString, Result,
bindgen_prelude::{PromiseRaw, Unknown},
};
use napi_derive::napi;
use pi_shell::process::{self as core_process, ProcessStatus as CoreProcessStatus};
pub use pi_shell::process::{KILL_SIGNAL, TERM_SIGNAL, TerminationTargets, kill_process_group};
use crate::task;
use crate::{js::into_string, task};
#[derive(Default)]
#[napi(object)]
@@ -81,11 +81,11 @@ impl Process {
/// Open stable process references whose executable path matches exactly.
#[napi]
pub fn from_path(path: String) -> Vec<Process> {
core_process::Process::from_path(path)
pub fn from_path(path: JsString) -> Result<Vec<Process>> {
Ok(core_process::Process::from_path(into_string(path)?)
.into_iter()
.map(Self::from_inner)
.collect()
.collect())
}
/// Operating-system process identifier for this process reference.
+4 -3
View File
@@ -13,6 +13,7 @@ use std::{
};
use napi::{
JsString,
bindgen_prelude::*,
threadsafe_function::{ThreadsafeFunction, ThreadsafeFunctionCallMode},
};
@@ -20,7 +21,7 @@ use napi_derive::napi;
use parking_lot::Mutex;
use portable_pty::{Child, CommandBuilder, PtySize, native_pty_system};
use crate::{ps, task};
use crate::{js::into_string, ps, task};
/// Options for running a command in a PTY session.
#[napi(object)]
@@ -178,8 +179,8 @@ impl PtySession {
/// Write raw input bytes to PTY stdin.
#[napi]
pub fn write(&self, data: String) -> Result<()> {
self.send_control(ControlMessage::Input(data))
pub fn write(&self, data: JsString) -> Result<()> {
self.send_control(ControlMessage::Input(into_string(data)?))
}
/// Resize the active PTY.
+10 -7
View File
@@ -48,10 +48,10 @@ use std::{
use base64::{Engine as _, engine::general_purpose::STANDARD};
use fontdue::{Font as TtfFace, FontSettings, Metrics};
use napi::bindgen_prelude::*;
use napi::{JsString, bindgen_prelude::*};
use napi_derive::napi;
use crate::task;
use crate::{js, task};
/// Upper bound on the frame edge: a hard stop against absurd allocations
/// (`size * size` pixel buffer), far above the 2576px production frame.
@@ -1162,14 +1162,17 @@ pub struct SnapcompactRenderOptions {
/// the selected native font has a glyph for it; renderer control codes are
/// considered renderable because they are interpreted outside font lookup.
#[napi]
pub fn snapcompact_supported_chars(font: String, chars: String) -> Result<String> {
let font = resolve_font(&font).ok_or_else(|| {
pub fn snapcompact_supported_chars(font: JsString, chars: JsString) -> Result<String> {
let font_name = js::utf8(font)?;
let font = resolve_font(&font_name).ok_or_else(|| {
Error::from_reason(format!(
"Unknown snapcompact font {font:?}: expected \"5x8\", \"8x8\", \"6x12\", \"8x13\", or \
\"silver\""
"Unknown snapcompact font {:?}: expected \"5x8\", \"8x8\", \"6x12\", \"8x13\", or \
\"silver\"",
&*font_name
))
})?;
let mut supported = String::new();
let chars = js::utf8(chars)?;
let mut supported = String::with_capacity(chars.len());
for ch in chars.chars() {
if matches!(ch as u32, DIM_ON | DIM_OFF | FULL_BLOCK | 0x0a) || font.supports(ch as u32) {
supported.push(ch);
+45 -43
View File
@@ -16,8 +16,8 @@ use std::{
use napi::{JsString, bindgen_prelude::*};
use napi_derive::napi;
use smallvec::{SmallVec, smallvec};
use unicode_segmentation::UnicodeSegmentation;
use unicode_width::{UnicodeWidthChar, UnicodeWidthStr};
use crate::js;
const MIN_TAB_WIDTH: u32 = 1;
const MAX_TAB_WIDTH: u32 = 16;
@@ -45,8 +45,7 @@ fn build_utf16_string(mut data: Vec<u16>) -> Utf16String {
while data.last() == Some(&0) {
data.pop();
}
// SAFETY: we know Utf16String == struct(Vec<u16>)
unsafe { std::mem::transmute(data) }
Utf16String::from(data)
}
// ============================================================================
@@ -642,7 +641,7 @@ fn apply_hangul_compat_jamo_delta(width: usize, c: char) -> usize {
let Some(target) = hangul_compat_jamo_target_width() else {
return width;
};
let unicode_width = UnicodeWidthChar::width(c).unwrap_or(0);
let unicode_width = xutf::width_char(c);
// The zero-width filler (U+3164 HANGUL FILLER) is an invisible placeholder.
// The target is set for *visible* jamo, so only the narrow correction
// (target 1) applies to the filler; a wide terminal renders it at its
@@ -659,7 +658,7 @@ fn apply_hangul_compat_jamo_delta(width: usize, c: char) -> usize {
}
#[inline]
fn char_width_corrected(c: char) -> Option<usize> {
fn char_width_corrected(c: char) -> usize {
// Hangul Compatibility Jamo U+3131..=U+318E render as 1 cell on some
// terminals (Terminal.app, iTerm2) but follow UAX#11 at 2 cells on others
// (Ghostty, most Linux terminals). The width is resolved at runtime from the
@@ -671,13 +670,13 @@ fn char_width_corrected(c: char) -> Option<usize> {
// Zero-width filler (U+3164): only the narrow correction applies — a
// wide terminal renders it at its Unicode width (0), not the effective
// wide target set for visible jamo. See apply_hangul_compat_jamo_delta.
let unicode_width = UnicodeWidthChar::width(c).unwrap_or(0);
let unicode_width = xutf::width_char(c);
if unicode_width == 0 && target > 1 {
return Some(unicode_width);
return unicode_width;
}
return Some(target);
return target;
}
UnicodeWidthChar::width(c)
xutf::width_char(c)
}
#[inline]
@@ -690,14 +689,14 @@ fn grapheme_width_str(g: &str, tab_width: usize) -> usize {
return 0;
};
if it.next().is_none() {
return char_width_corrected(c0).unwrap_or(0);
return char_width_corrected(c0);
}
// Multi-char grapheme: keep UnicodeWidthStr as the source of truth for
// sequence-level width rules (VS16 emoji presentation, keycaps, ZWJ emoji,
// CRLF, script ligatures). A per-char sum is not equivalent. Apply only the
// same local Compatibility Jamo delta that char_width_corrected applies to
// standalone code points; the delta is a no-op when no correction is active.
let mut width = UnicodeWidthStr::width(g);
let mut width = xutf::width_str(g);
for c in g.chars() {
width = apply_hangul_compat_jamo_delta(width, c);
}
@@ -729,7 +728,7 @@ where
}
let mut utf16_pos = 0usize;
for g in scratch.graphemes(true) {
for g in xutf::graphemes_str(scratch) {
let w = grapheme_width_str(g, tab_width);
let g_u16_len: usize = g.chars().map(|c| c.len_utf16()).sum();
@@ -1255,10 +1254,12 @@ fn wrap_text_with_ansi_impl(
/// Returns UTF-16 lines with active SGR codes carried across line boundaries.
#[napi]
pub fn wrap_text_with_ansi(text: JsString, width: u32, tab_width: u32) -> Result<Vec<Utf16String>> {
let text_u16 = text.into_utf16()?;
let text = js::utf16(text)?;
let tab_width = clamp_tab_width_for_ops(tab_width);
let lines = wrap_text_with_ansi_impl(text_u16.as_slice(), width as usize, tab_width);
Ok(lines.into_iter().map(build_utf16_string).collect())
Ok(wrap_text_with_ansi_impl(&text, width as usize, tab_width)
.into_iter()
.map(build_utf16_string)
.collect())
}
// ============================================================================
@@ -1280,30 +1281,36 @@ pub fn truncate_to_width(
let ellipsis_kind = ellipsis_kind.unwrap_or(Ellipsis::Unicode);
let pad = pad.unwrap_or(false);
let tab_width = clamp_tab_width_for_ops(tab_width);
// Keep original handle so we can return it without allocating.
let original = text;
let text = js::utf16(text)?;
Ok(truncate_to_width_impl(original, &text, max_width, ellipsis_kind, pad, tab_width))
}
let text_u16 = text.into_utf16()?;
let text = text_u16.as_slice();
fn truncate_to_width_impl<'env>(
original: JsString<'env>,
text: &[u16],
max_width: usize,
ellipsis_kind: Ellipsis,
pad: bool,
tab_width: usize,
) -> Either<JsString<'env>, Utf16String> {
// Fast path: early-exit width check
let (text_w, exceeded) = visible_width_u16_up_to(text, max_width, tab_width);
if !exceeded {
if !pad {
// Return original JsString handle: zero output allocation.
return Ok(Either::A(original));
return Either::A(original);
}
if text_w < max_width {
let mut out = Vec::with_capacity(text.len() + (max_width - text_w));
out.extend_from_slice(text);
out.resize(out.len() + (max_width - text_w), b' ' as u16);
return Ok(Either::B(build_utf16_string(out)));
return Either::B(build_utf16_string(out));
}
// Exactly fits and padding requested: return original is still fine.
return Ok(Either::A(original));
return Either::A(original);
}
// Map ellipsis kind to UTF-16 data and width
@@ -1335,7 +1342,7 @@ pub fn truncate_to_width(
if pad && w < max_width {
out.resize(out.len() + (max_width - w), b' ' as u16);
}
return Ok(Either::B(build_utf16_string(out)));
return Either::B(build_utf16_string(out));
}
// Main truncation
@@ -1439,7 +1446,7 @@ pub fn truncate_to_width(
}
}
Ok(Either::B(build_utf16_string(out)))
Either::B(build_utf16_string(out))
}
// ============================================================================
@@ -1590,19 +1597,16 @@ pub fn slice_with_width(
strict: Option<bool>,
tab_width: u32,
) -> Result<SliceResult> {
let line_u16 = line.into_utf16()?;
let line = line_u16.as_slice();
let strict = strict.unwrap_or(false);
if length == 0 {
return Ok(SliceResult { text: build_utf16_string(vec![]), width: 0 });
}
let line = js::utf16(line)?;
let strict = strict.unwrap_or(false);
let tab_width = clamp_tab_width_for_ops(tab_width);
let (out, w) =
slice_with_width_impl(line, start_col as usize, length as usize, strict, tab_width);
Ok(SliceResult { text: build_utf16_string(out), width: crate::utils::clamp_u32(w as u64) })
let (out, width) =
slice_with_width_impl(&line, start_col as usize, length as usize, strict, tab_width);
Ok(SliceResult { text: build_utf16_string(out), width: crate::utils::clamp_u32(width as u64) })
}
// ============================================================================
@@ -1812,12 +1816,10 @@ pub fn extract_segments(
strict_after: bool,
tab_width: u32,
) -> Result<ExtractSegmentsResult> {
let line_u16 = line.into_utf16()?;
let line = line_u16.as_slice();
let line = js::utf16(line)?;
let tab_width = clamp_tab_width_for_ops(tab_width);
let (before, bw, after, aw) = extract_segments_impl(
line,
let (before, before_width, after, after_width) = extract_segments_impl(
&line,
before_end as usize,
after_start as usize,
after_len as usize,
@@ -1827,9 +1829,9 @@ pub fn extract_segments(
Ok(ExtractSegmentsResult {
before: build_utf16_string(before),
before_width: crate::utils::clamp_u32(bw as u64),
before_width: crate::utils::clamp_u32(before_width as u64),
after: build_utf16_string(after),
after_width: crate::utils::clamp_u32(aw as u64),
after_width: crate::utils::clamp_u32(after_width as u64),
})
}
@@ -1842,9 +1844,9 @@ pub fn extract_segments(
/// Tabs count as a fixed-width cell.
#[napi]
pub fn visible_width(text: JsString, tab_width: u32) -> Result<u32> {
let text_u16 = text.into_utf16()?;
let text = js::utf16(text)?;
let tab_width = clamp_tab_width_for_ops(tab_width);
Ok(crate::utils::clamp_u32(visible_width_u16(text_u16.as_slice(), tab_width) as u64))
Ok(crate::utils::clamp_u32(visible_width_u16(&text, tab_width) as u64))
}
#[cfg(test)]
+29 -15
View File
@@ -22,8 +22,10 @@ use napi::{
bindgen_prelude::{Array, Either},
};
use napi_derive::napi;
use pi_shell::rayon_global_pool_available;
use rayon::prelude::*;
use crate::utok;
use crate::{js, utok};
/// Tokenizer encoding to use.
#[napi(string_enum)]
@@ -70,9 +72,9 @@ impl Encoding {
/// Count tokens in `input`.
///
/// `input` may be a single string or an array of strings; an array returns
/// the sum across all elements. Always returns a single token total — use
/// this for any aggregate budget question without paying a per-element napi
/// crossing.
/// the sum across all elements (counted in parallel when the global rayon pool
/// is available). Always returns a single token total — use this for any
/// aggregate budget question without paying a per-element napi crossing.
///
/// Measures user/model content, not wire-protocol tokens: BPE encodings
/// use ordinary encoding (no special-token handling) and the Claude
@@ -86,22 +88,34 @@ pub fn count_tokens(
) -> napi::Result<u32> {
let enc = Encoding::utok(encoding);
match input {
Either::A(js_str) => {
let text = js_str.into_utf16()?;
let (_, units) = text.as_slice().split_last().expect("napi UTF-16 buffer has a terminator");
Ok(enc.count(units))
},
Either::A(text) => Ok(enc.count(&*js::utf16(text)?)),
Either::B(array) => {
let mut total = 0u32;
// Node-API handles are thread-affine, so every element is read here on
// the JS thread — into one buffer, so the batch costs one allocation
// rather than one per string. Only the counting fans out.
let mut units = Vec::new();
let mut spans = Vec::with_capacity(array.len() as usize);
for index in 0..array.len() {
let text = array
.get::<JsString>(index)?
.ok_or_else(|| napi::Error::from_reason("array changed during token counting"))?
.into_utf16()?;
let (_, units) = text.as_slice().split_last().expect("napi UTF-16 buffer has a terminator");
total += enc.count(units);
.ok_or_else(|| napi::Error::from_reason("array changed during token counting"))?;
spans.push(js::utf16_append(text, &mut units)?);
}
Ok(total)
// Scheduling a Rayon job costs more than tokenizing a small prompt
// batch. Keep those batches on the N-API thread; large batches still
// amortize the pool handoff across enough independent strings.
const PARALLEL_BATCH_MIN: usize = 16;
Ok(if spans.len() >= PARALLEL_BATCH_MIN && rayon_global_pool_available() {
spans
.par_iter()
.map(|span| enc.count(&units[span.clone()]))
.sum()
} else {
spans
.iter()
.map(|span| enc.count(&units[span.clone()]))
.sum()
})
},
}
}
+14 -6
View File
@@ -8,11 +8,13 @@
//! bit-identical to the TS versions and integer results are exactly equal.
use napi::{
Error, Result, Status,
bindgen_prelude::{Float32Array, Float64Array, Uint32Array},
Error, JsString, Result, Status,
bindgen_prelude::{Array, Float32Array, Float64Array, Uint32Array},
};
use napi_derive::napi;
use crate::js;
fn invalid<T>(message: &str) -> Result<T> {
Err(Error::new(Status::InvalidArg, message))
}
@@ -261,20 +263,26 @@ fn jaccard_sorted(a: &[Box<str>], b: &[Box<str>]) -> f64 {
reason = "mul_add rounds differently; bit-exact with the TS loops is the contract"
)]
pub fn mmr_rerank_indices(
contents: Vec<String>,
#[napi(ts_arg_type = "Array<string>")] contents: Array,
scores: Float64Array,
lambda_param: f64,
top_k: u32,
) -> Result<Uint32Array> {
if scores.len() != contents.len() {
if scores.len() != contents.len() as usize {
return invalid("scores length must equal contents length");
}
let limit = top_k as usize;
let count = contents.len();
let count = contents.len() as usize;
if limit == 0 || count == 0 {
return Ok(Uint32Array::new(Vec::new()));
}
let sets: Vec<Vec<Box<str>>> = contents.iter().map(|text| word_set(text)).collect();
let mut sets = Vec::with_capacity(count);
for index in 0..contents.len() {
let content = contents
.get::<JsString>(index)?
.ok_or_else(|| Error::new(Status::InvalidArg, "contents changed during reranking"))?;
sets.push(word_set(&js::utf8(content)?));
}
let mut selected: Vec<u32> = Vec::with_capacity(limit.min(count));
selected.push(0);
let mut remaining: Vec<u32> = (1..count as u32).collect();
+2 -1
View File
@@ -3268,6 +3268,7 @@ mod tests {
"base64",
"basename",
"cat",
"cksum",
"cmp",
"combine",
"comm",
@@ -4513,7 +4514,7 @@ mod tests {
.expect("process substitution should not hang");
assert_eq!(result.exit_code, Some(1));
assert!(output.contains("-a\n+b\n"), "diff output missing changed lines: {output:?}");
assert!(output.contains("< a\n---\n> b\n"), "diff output missing changed lines: {output:?}");
}
#[cfg(unix)]
+12 -1
View File
@@ -512,10 +512,21 @@ fn unwrap_transparent_background_wrapper(pipeline: &ast::Pipeline) -> Option<ast
};
let mut unwrapped = simple_cmd.clone();
let suffix = unwrapped.suffix.as_mut()?;
let operand_index = suffix
let mut operand_index = suffix
.0
.iter()
.position(|item| matches!(item, CommandPrefixOrSuffixItem::Word(_)))?;
// A leading `--` only terminates the wrapper's own options
// (`nohup -- cmd &`): drop it and take the next word as the operand.
if let CommandPrefixOrSuffixItem::Word(word) = &suffix.0[operand_index]
&& word.value == "--"
{
suffix.0.remove(operand_index);
operand_index = suffix
.0
.iter()
.position(|item| matches!(item, CommandPrefixOrSuffixItem::Word(_)))?;
}
let CommandPrefixOrSuffixItem::Word(operand_word) = suffix.0.remove(operand_index) else {
return None;
};
+2
View File
@@ -17,6 +17,8 @@
### Fixed
- Fixed a prompt cancelled during turn setup (Esc while the pre-stream spinner is up, after dispatch had started) vanishing entirely: it was never persisted to the session — so the `/tree` and `/branch` selectors had nothing to rewind to — and was not returned to the editor either, while its optimistic transcript row kept lingering. A prompt dropped before reaching the agent (abort or usage-preflight denial racing setup) is now handed back: the stale transcript row is removed and the typed text and image attachments are restored to the editor for editing.
- Fixed macOS `top`-style single-dash long options in the `top` shell builtin: `top -l 2 -pid 56943 -stats pid,cpu,th,mem,pstate` previously failed with `invalid value 'id' for '--pid <PIDS>'` because clap read `-pid` as `-p id`. Single-dash long spellings now parse, and `-stats` selects and orders output columns using macOS stat keys (`pid`, `cpu`, `th`, `mem`, `pstate`, ...).
- Fixed a sweep of GNU/BSD compatibility gaps in the built-in shell utilities, found by auditing every builtin against its real counterpart: `timeout` gained `-s`/`-k`/`--preserve-status`/`--foreground`/`-v`, GNU exit codes (124/125/137), signal delivery to the child process group with `-k` SIGKILL escalation, and `timeout 0` disabling the limit; `diff` gained normal-format default output, `-w`/`-b`/`-B`/`-i`/`-x`/`-L`/`-s`/`--strip-trailing-cr`, context format (`-c`/`-C`), bundled flags (`-ru`, `-urN`), timestamped unified headers, and no longer recurses directories without `-r`; `find` fixed inverted `-newerXY` timestamp comparisons, anchored `-regex` to whole paths, and gained BSD `-perm +mode`, `-type f,d` lists, `-size` `T`/`P` suffixes, ISO dates in `-newermt`, and BSD leading flags `-E`/`-x`/`-s`; `date` gained BSD `-r <epoch>`, `-v` adjustments, and `-j -f` strptime parsing, and `-I` no longer swallows a following `+FORMAT`; `tail`/`head` accept obsolete `-N`/`+N` counts at any argv position with any file count, `tail -r -n N` works, and `head` continues past per-file I/O errors with GNU header/separator placement; `rg` resolves `-s`/`-i`/`-S` by last occurrence, accepts `--no-config`/`-j`/`--threads`/`--no-column`, implements `--path-separator`, and emits clean NUL-delimited output under `-0`/`-l0`; `stat` prints integer epochs for `%X`/`%Y`/`%Z` (bash arithmetic on `stat -c %Y` works) and gained BSD `-s`/`-x` output modes plus `-t` time formatting; `cksum` is now registered (multi-algorithm `cksum -a sha256`); `truncate` implements `-o`/`--io-blocks` (previously silently truncated to the raw byte count), accepts `b` (512-byte) suffix and BSD `=` prefix; `sleep`/`timeout` accept `infinity` and keep sub-millisecond precision; `yes` and `errno` accept hyphen-prefixed operands (`yes -n`, `errno -2`); `nohup -- cmd` no longer tries to run `--` (including backgrounded via the brush wrapper); `which` gained BSD `-s` and errors on zero operands; `kill` accepts attached values (`-s9`, `-sKILL`, `-l9`) and maps exit statuses above 128 (`kill -l 137` → `KILL`).
## [17.3.8] - 2026-08-19
+266
View File
@@ -0,0 +1,266 @@
/**
* Micro-benchmarks for native text primitives vs standard JS/Bun equivalents
* using the mitata benchmarking framework.
*
* Run with: `bun packages/natives/bench/text.ts`
*
* Every bench body pipes its result through `do_not_optimize`. Without it JSC
* dead-code-eliminates pure calls with discarded results after warmup, which
* reports sub-nanosecond phantoms (e.g. string-width at ~180 ps/iter).
*/
import cliTruncate from "cli-truncate";
import * as diff from "diff";
import { countTokens as gptCountTokens } from "gpt-tokenizer/model/gpt-4o";
import { bench, do_not_optimize, run, summary } from "mitata";
import sliceAnsi from "slice-ansi";
import stringWidth from "string-width";
import wrapAnsi from "wrap-ansi";
// The TS-side width measurer every TUI render path actually calls. Backed by
// Bun.stringWidth with a printable-ASCII fast path; the N-API `visibleWidth`
// below is only the raw binding (its ~150 ns floor is per-call FFI overhead:
// UTF-16 -> UTF-8 marshal + result box, not the width algorithm).
import { visibleWidth as tuiVisibleWidth } from "../../tui/src/utils";
import {
countTokens,
diffLines,
extractSegments,
highlightCode,
sliceWithWidth,
truncateToWidth,
visibleWidth,
wrapTextWithAnsi,
} from "../native/index.js";
const testCases = {
shortAscii: "const x = 42; // standard code snippet",
longAscii:
"This is a much longer line of text designed to test how measurement scales when lines are wider in terminal buffers. ".repeat(
4,
),
ansiStyled:
"\x1b[1m\x1b[38;2;100;200;255mfunction\x1b[0m \x1b[38;2;255;215;0mrenderTerminal\x1b[0m(\x1b[38;2;150;150;150mprops\x1b[0m: \x1b[38;2;80;250;123mTerminalProps\x1b[0m) {\x1b[38;2;98;114;164m // styled output\x1b[0m",
emojiCjk: "⚡ Status: 🚀 Deploying to 東京 (Tokyo) cluster 🎯 [5/10 completed] 🌸",
multilineAnsi: (
"\x1b[32m✔ Loaded config successfully\x1b[0m\n" +
"\x1b[34mℹ Connecting to server at 127.0.0.1:8080...\x1b[0m\n" +
"\x1b[33m⚠ Warning: high memory usage detected in worker pool\x1b[0m\n" +
"\x1b[31m✖ Error: failed to establish connection to database replica\x1b[0m\n" +
"Stack trace: at ConnectionPool.acquire (/app/src/db.ts:142:18)\n"
).repeat(3),
diffOld:
"import { a, b, c } from 'pkg';\n\nfunction main() {\n console.log('hello');\n const x = 1;\n return x + 2;\n}\n",
diffNew:
"import { a, b, c, d } from 'pkg';\n\nfunction main() {\n console.log('hello world');\n const x = 2;\n const y = 3;\n return x + y;\n}\n",
tokenArray: [
"You are a helpful assistant with access to tools.",
"User prompt: please inspect the code in src/index.ts and summarize findings.",
"System message: running tool call 'read_file' with arguments {'path': 'src/index.ts'}.",
"File content: export function run() { console.log('active'); }".repeat(5),
],
colors: {
comment: "\x1b[38;2;98;114;164m",
keyword: "\x1b[38;2;255;121;198m",
function: "\x1b[38;2;80;250;123m",
variable: "\x1b[38;2;248;248;242m",
string: "\x1b[38;2;241;250;140m",
number: "\x1b[38;2;189;147;249m",
type: "\x1b[38;2;139;233;253m",
operator: "\x1b[38;2;255;121;198m",
punctuation: "\x1b[38;2;248;248;242m",
},
};
// Each width bench cycles a pool of 64 distinct strings. This defeats
// constant-argument hoisting in pure comparators (`do_not_optimize` only
// protects the result) and mirrors a real redraw workload: a frame re-measures
// the same visible lines every paint, so pi-tui's bounded width memo hits —
// but the pool is far larger than any cache that merely fits the bench.
const WIDTH_VARIANT_COUNT = 64;
function makeWidthVariants(base: string): string[] {
const variants: string[] = new Array(WIDTH_VARIANT_COUNT);
for (let i = 0; i < WIDTH_VARIANT_COUNT; i++) {
variants[i] = `${base} ${String(i).padStart(2, "0")}`;
}
return variants;
}
const widthInputVariants = {
shortAscii: makeWidthVariants(testCases.shortAscii),
longAscii: makeWidthVariants(testCases.longAscii),
ansiStyled: makeWidthVariants(testCases.ansiStyled),
emojiCjk: makeWidthVariants(testCases.emojiCjk),
} as const;
let widthInputVariantIndex = 0;
function nextWidthInput(kind: keyof typeof widthInputVariants): string {
return widthInputVariants[kind][widthInputVariantIndex++ & (WIDTH_VARIANT_COUNT - 1)];
}
// ============================================================================
// 1. visibleWidth: pi-tui hot path vs raw N-API binding vs Bun.stringWidth vs
// string-width npm package
// ============================================================================
summary(() => {
bench("visibleWidth: short ascii (pi-tui)", () => do_not_optimize(tuiVisibleWidth(nextWidthInput("shortAscii"))));
bench("visibleWidth: short ascii (native N-API)", () =>
do_not_optimize(visibleWidth(nextWidthInput("shortAscii"), 3)),
);
bench("visibleWidth: short ascii (Bun.stringWidth)", () =>
do_not_optimize(Bun.stringWidth(nextWidthInput("shortAscii"))),
);
bench("visibleWidth: short ascii (string-width npm)", () =>
do_not_optimize(stringWidth(nextWidthInput("shortAscii"))),
);
});
summary(() => {
bench("visibleWidth: long ascii (pi-tui)", () => do_not_optimize(tuiVisibleWidth(nextWidthInput("longAscii"))));
bench("visibleWidth: long ascii (native N-API)", () =>
do_not_optimize(visibleWidth(nextWidthInput("longAscii"), 3)),
);
bench("visibleWidth: long ascii (Bun.stringWidth)", () =>
do_not_optimize(Bun.stringWidth(nextWidthInput("longAscii"))),
);
bench("visibleWidth: long ascii (string-width npm)", () =>
do_not_optimize(stringWidth(nextWidthInput("longAscii"))),
);
});
summary(() => {
bench("visibleWidth: ansi styled (pi-tui)", () => do_not_optimize(tuiVisibleWidth(nextWidthInput("ansiStyled"))));
bench("visibleWidth: ansi styled (native N-API)", () =>
do_not_optimize(visibleWidth(nextWidthInput("ansiStyled"), 3)),
);
bench("visibleWidth: ansi styled (Bun.stringWidth)", () =>
do_not_optimize(Bun.stringWidth(nextWidthInput("ansiStyled"))),
);
bench("visibleWidth: ansi styled (string-width npm)", () =>
do_not_optimize(stringWidth(nextWidthInput("ansiStyled"))),
);
});
summary(() => {
bench("visibleWidth: emoji / CJK (pi-tui)", () => do_not_optimize(tuiVisibleWidth(nextWidthInput("emojiCjk"))));
bench("visibleWidth: emoji / CJK (native N-API)", () =>
do_not_optimize(visibleWidth(nextWidthInput("emojiCjk"), 3)),
);
bench("visibleWidth: emoji / CJK (Bun.stringWidth)", () =>
do_not_optimize(Bun.stringWidth(nextWidthInput("emojiCjk"))),
);
bench("visibleWidth: emoji / CJK (string-width npm)", () =>
do_not_optimize(stringWidth(nextWidthInput("emojiCjk"))),
);
});
// ============================================================================
// 2. truncateToWidth: Native vs cli-truncate
// ============================================================================
summary(() => {
bench("truncateToWidth: long ascii (native)", () =>
do_not_optimize(truncateToWidth(testCases.longAscii, 40, 0, false, 3)),
);
bench("truncateToWidth: long ascii (cli-truncate)", () => do_not_optimize(cliTruncate(testCases.longAscii, 40)));
});
summary(() => {
bench("truncateToWidth: ansi styled (native)", () =>
do_not_optimize(truncateToWidth(testCases.ansiStyled, 40, 0, false, 3)),
);
bench("truncateToWidth: ansi styled (cli-truncate)", () => do_not_optimize(cliTruncate(testCases.ansiStyled, 40)));
});
bench("truncateToWidth: fits no-alloc (native)", () =>
do_not_optimize(truncateToWidth(testCases.shortAscii, 100, 0, false, 3)),
);
bench("truncateToWidth: pads with spaces (native)", () =>
do_not_optimize(truncateToWidth(testCases.shortAscii, 60, 0, true, 3)),
);
// ============================================================================
// 3. sliceWithWidth: Native vs slice-ansi
// ============================================================================
summary(() => {
bench("sliceWithWidth: ascii slice (native)", () =>
do_not_optimize(sliceWithWidth(testCases.shortAscii, 10, 20, false, 3)),
);
bench("sliceWithWidth: ascii slice (slice-ansi)", () => do_not_optimize(sliceAnsi(testCases.shortAscii, 10, 30)));
});
summary(() => {
bench("sliceWithWidth: ansi styled slice (native)", () =>
do_not_optimize(sliceWithWidth(testCases.ansiStyled, 15, 30, false, 3)),
);
bench("sliceWithWidth: ansi styled slice (slice-ansi)", () =>
do_not_optimize(sliceAnsi(testCases.ansiStyled, 15, 45)),
);
});
// ============================================================================
// 4. wrapTextWithAnsi: Native vs wrap-ansi
// ============================================================================
summary(() => {
bench("wrapTextWithAnsi: single line (native)", () =>
do_not_optimize(wrapTextWithAnsi(testCases.ansiStyled, 30, 3)),
);
bench("wrapTextWithAnsi: single line (wrap-ansi)", () =>
do_not_optimize(wrapAnsi(testCases.ansiStyled, 30, { hard: true })),
);
});
summary(() => {
bench("wrapTextWithAnsi: multiline logs (native)", () =>
do_not_optimize(wrapTextWithAnsi(testCases.multilineAnsi, 60, 3)),
);
bench("wrapTextWithAnsi: multiline logs (wrap-ansi)", () =>
do_not_optimize(wrapAnsi(testCases.multilineAnsi, 60, { hard: true })),
);
});
// ============================================================================
// 5. diffLines: Native vs jsdiff (diff npm package)
// ============================================================================
summary(() => {
bench("diffLines: source files (native)", () => do_not_optimize(diffLines(testCases.diffOld, testCases.diffNew)));
bench("diffLines: source files (diff npm)", () =>
do_not_optimize(diff.diffLines(testCases.diffOld, testCases.diffNew)),
);
});
// ============================================================================
// 6. countTokens: Native (o200k_base) vs gpt-tokenizer (pure JS)
// ============================================================================
summary(() => {
bench("countTokens: single string (native)", () => do_not_optimize(countTokens(testCases.longAscii)));
bench("countTokens: single string (gpt-tokenizer)", () => do_not_optimize(gptCountTokens(testCases.longAscii)));
});
summary(() => {
bench("countTokens: array of strings (native)", () => do_not_optimize(countTokens(testCases.tokenArray)));
bench("countTokens: array of strings (gpt-tokenizer)", () => {
let total = 0;
for (const s of testCases.tokenArray) total += gptCountTokens(s);
do_not_optimize(total);
});
});
// ============================================================================
// 7. Specialized native primitives (standalone)
// ============================================================================
bench("extractSegments: ansi overlay (native)", () =>
do_not_optimize(extractSegments(testCases.ansiStyled, 15, 25, 20, false, 3)),
);
bench("highlightCode: rust snippet (native)", () =>
do_not_optimize(highlightCode('fn main() { println!("hello"); }', "rust", testCases.colors)),
);
await run();
+10 -2
View File
@@ -40,11 +40,19 @@
"gen:native": "bun scripts/embed-native.ts",
"gen:native:reset": "bun scripts/embed-native.ts --reset",
"gen:npm": "bun scripts/gen-npm-packages.ts",
"bench": "bun bench/grep.ts"
"bench": "bun bench/grep.ts",
"bench:text": "bun bench/text.ts"
},
"devDependencies": {
"@napi-rs/cli": "catalog:",
"@types/bun": "catalog:"
"@types/bun": "catalog:",
"cli-truncate": "6.1.1",
"diff": "9.0.0",
"gpt-tokenizer": "4.0.0",
"mitata": "1.0.34",
"slice-ansi": "9.0.0",
"string-width": "8.2.2",
"wrap-ansi": "10.0.0"
},
"engines": {
"bun": ">=1.3.14"
+2 -2
View File
@@ -7,16 +7,16 @@ import {
astEdit,
astMatch,
blockRangeAt,
countTokens,
Encoding,
executeShell,
FileType,
countTokens,
fuzzyFind,
type GlobMatch,
GrepOutputMode,
getSupportedLanguages,
glob,
grep,
Encoding,
highlightCode,
htmlToMarkdown,
invalidateFsScanCache,
+53 -58
View File
@@ -225,9 +225,7 @@ export function getSegmenter(): Intl.Segmenter {
// added back so width matches the native truncate/slice/wrap helpers.
const OSC66_SPAN_REGEX = /\x1b\]66;([^;]*);([\s\S]*?)(?:\x07|\x1b\\)/g;
const OSC66_PREFIX = "\x1b]66;";
const ESC = "\x1b";
const TAB = "\t";
const LONG_WIDTH_FAST_PATH_MIN = 128;
const PRINTABLE_ASCII_REGEX = /^[\u0020-\u007e]*$/;
// Pin Bun.stringWidth semantics to the native width engine and guard against Bun
// default drift: strip ANSI/OSC (don't count escape bytes) and treat
@@ -243,8 +241,6 @@ const STRING_WIDTH_OPTS = { countAnsiEscapeCodes: false, ambiguousIsNarrow: true
// `setHangulCompatibilityJamoWidth`; mirror the same correction here so the TS
// width stays in parity with the native truncate/slice/wrap model — and so the
// hardware cursor column lands on the actual glyph during Korean IME input.
const HANGUL_COMPAT_JAMO_REGEX = /[\u3131-\u318e]/;
const HANGUL_COMPAT_JAMO_GLOBAL_REGEX = /[\u3131-\u318e]/g;
const HANGUL_FILLER_CODE_POINT = 0x3164;
// `Bun.stringWidth` counts every code point in the Compatibility Jamo block as
// 2 cells (even the U+3164 filler that `unicode-width` treats as zero-width).
@@ -274,19 +270,28 @@ function hangulCompatibilityJamoTargetWidth(): 1 | 2 | null {
// crates/pi-natives/src/text.rs, including the rule that the zero-width filler
// (U+3164) is never widened past the narrow correction (a wide terminal still
// renders it at its Unicode width of 0).
function correctHangulCompatibilityJamoWidth(width: number, str: string): number {
if (!HANGUL_COMPAT_JAMO_REGEX.test(str)) return width;
function correctHangulCompatibilityJamoWidth(
width: number,
compatibilityJamoCount: number,
fillerCount: number,
): number {
if (compatibilityJamoCount === 0) return width;
const target = hangulCompatibilityJamoTargetWidth();
let corrected = width;
HANGUL_COMPAT_JAMO_GLOBAL_REGEX.lastIndex = 0;
for (let m = HANGUL_COMPAT_JAMO_GLOBAL_REGEX.exec(str); m !== null; m = HANGUL_COMPAT_JAMO_GLOBAL_REGEX.exec(str)) {
const unicodeWidth = m[0].codePointAt(0) === HANGUL_FILLER_CODE_POINT ? 0 : 2;
const finalWidth = target === null || (unicodeWidth === 0 && target > 1) ? unicodeWidth : target;
corrected += finalWidth - HANGUL_COMPAT_JAMO_BUN_WIDTH;
}
return corrected;
return target === 1 ? width - compatibilityJamoCount : width - fillerCount * HANGUL_COMPAT_JAMO_BUN_WIDTH;
}
// Terminal redraws re-measure the same visible lines every frame, usually as
// the same string objects (JSC caches their hashes, so repeat lookups are
// O(1) — cheaper than even the ASCII fast scan). Strings longer than the
// length gate skip the cache entirely: hashing them costs as much as measuring
// them, and retaining them would pin large render buffers. Worst-case
// retention is MAX * MAX_LEN UTF-16 units (~2 MiB); cleared when the width
// configuration epoch changes.
const VISIBLE_WIDTH_CACHE_MAX = 2048;
const VISIBLE_WIDTH_CACHE_MAX_LEN = 512;
const visibleWidthCache = new Map<string, number>();
let visibleWidthCacheEpoch = widthConfigEpoch;
/**
* Visible width of a string in terminal columns, excluding ANSI/OSC escapes.
*
@@ -296,65 +301,51 @@ function correctHangulCompatibilityJamoWidth(width: number, str: string): number
*/
export function visibleWidth(str: string): number {
if (!str) return 0;
// Long non-escape text is faster through Bun's native scanner than through
// a JS printable-ASCII prepass. Escape-bearing strings stay on the scanner
// below so CSI/OSC-heavy render output can still bail out at the first ESC.
if (str.length >= LONG_WIDTH_FAST_PATH_MIN && !str.includes(ESC)) {
let width = Bun.stringWidth(str, STRING_WIDTH_OPTS);
let tabCount = 0;
for (let tabIndex = str.indexOf(TAB); tabIndex !== -1; tabIndex = str.indexOf(TAB, tabIndex + 1)) {
tabCount++;
const cacheable = str.length <= VISIBLE_WIDTH_CACHE_MAX_LEN;
if (cacheable) {
if (visibleWidthCacheEpoch !== widthConfigEpoch) {
visibleWidthCache.clear();
visibleWidthCacheEpoch = widthConfigEpoch;
}
if (tabCount > 0) width += tabCount * DEFAULT_TAB_WIDTH;
return correctHangulCompatibilityJamoWidth(width, str);
const cached = visibleWidthCache.get(str);
if (cached !== undefined) return cached;
}
// This regex compiles to a native ASCII scan, cheaper than Bun's width
// scanner for the overwhelmingly common source-code path.
if (PRINTABLE_ASCII_REGEX.test(str)) {
if (cacheable) {
if (visibleWidthCache.size >= VISIBLE_WIDTH_CACHE_MAX) visibleWidthCache.clear();
visibleWidthCache.set(str, str.length);
}
return str.length;
}
let tabCount = 0;
let i = 0;
for (; i < str.length; i++) {
let compatibilityJamoCount = 0;
let fillerCount = 0;
let hasEsc = false;
for (let i = 0; i < str.length; i++) {
const code = str.charCodeAt(i);
if (code < 0x20 || code > 0x7e) {
if (code === 0x09) {
tabCount++;
continue;
}
break;
}
}
if (i === str.length) {
return tabCount === 0 ? str.length : str.length + tabCount * (DEFAULT_TAB_WIDTH - 1);
}
if (tabCount === 0) {
let tabIndex = str.indexOf(TAB, i + 1);
if (tabIndex !== -1) {
tabCount = 1;
for (tabIndex = str.indexOf(TAB, tabIndex + 1); tabIndex !== -1; tabIndex = str.indexOf(TAB, tabIndex + 1)) {
tabCount++;
}
}
} else {
for (let tabIndex = str.indexOf(TAB, i + 1); tabIndex !== -1; tabIndex = str.indexOf(TAB, tabIndex + 1)) {
tabCount++;
} else if (code === 0x1b) {
hasEsc = true;
} else if (code >= 0x3131 && code <= 0x318e) {
compatibilityJamoCount++;
if (code === HANGUL_FILLER_CODE_POINT) fillerCount++;
}
}
// `Bun.stringWidth` is a JSC builtin (no per-call N-API number box, unlike
// the native scanner that traps under Bun 1.3.x GC/N-API load). It strips
// CSI/OSC to zero cells and shares the native engine's UAX#11 width tables.
let width = Bun.stringWidth(str, STRING_WIDTH_OPTS);
if (tabCount > 0) width += tabCount * DEFAULT_TAB_WIDTH;
// OSC 66: add back each stripped span as `scale * (explicit w ?? payload
// width)`. Matched rather than replaced to avoid reallocating the string.
if (str.includes(OSC66_PREFIX, i)) {
if (hasEsc && str.includes(OSC66_PREFIX)) {
OSC66_SPAN_REGEX.lastIndex = 0;
for (let m = OSC66_SPAN_REGEX.exec(str); m !== null; m = OSC66_SPAN_REGEX.exec(str)) {
let scale = 1;
let explicit: number | undefined;
for (const part of m[1].split(":")) {
// metadata keys are single chars, e.g. `s=2`, `w=5`
if (part.indexOf("=") !== 1) continue;
const value = Number.parseInt(part.slice(2), 10);
if (!Number.isFinite(value)) continue;
@@ -368,11 +359,15 @@ export function visibleWidth(str: string): number {
}
}
return correctHangulCompatibilityJamoWidth(width, str);
width = correctHangulCompatibilityJamoWidth(width, compatibilityJamoCount, fillerCount);
if (cacheable) {
if (visibleWidthCache.size >= VISIBLE_WIDTH_CACHE_MAX) visibleWidthCache.clear();
visibleWidthCache.set(str, width);
}
return width;
}
/**
* True when a row carries a Kitty OSC 66 text-sizing span (`\x1b]66;…`).
* Scaled spans must bypass wrapping/padding and, when scaled up, reserve the
* terminal rows their multicell glyphs flow into.
*/