feat(builtins): sweep of GNU/BSD compat fixes + str performance opts
Addresses a broad audit of built-in shell utilities against their real counterparts: timeout gains signal delivery, -s/-k/--preserve-status/--foreground/-v, and GNU exit codes; diff defaults to normal format and gains -w/-b/-B/-i/-c/-x/-L/-s/--strip-trailing-cr and proper -r gating; find fixes -newerXY timestamp comparison direction, anchors -regex to whole paths, and gains BSD -perm +mode, -type lists, -size T/P suffixes, -E/-x/-s flags; date gains BSD -r epoch, -v adjustments, -j -f strptime, and non-greedy -I; tail/head accept obsolete -N/+N at any position with any file count; rg resolves case flags by last occurrence and gains --path-separator and clean -0 output; stat prints integer epochs for %X/%Y/%Z and gains BSD -s/-x/-t; cksum is registered as a builtin; truncate implements -o/--io-blocks and b/= size suffixes; sleep/timeout accept infinity; yes/errno/kill accept hyphen-prefixed operands; nohup -- cmd no longer runs --; which gains BSD -s.
This commit is contained in:
Generated
+23
-14
@@ -2661,6 +2661,15 @@ dependencies = [
|
||||
"zerocopy",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "hash32"
|
||||
version = "0.3.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "47d60b12902ba28e2730cd37e95b8c9223af2808df9e902d4df49588d1470606"
|
||||
dependencies = [
|
||||
"byteorder",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "hashbrown"
|
||||
version = "0.14.5"
|
||||
@@ -2695,6 +2704,16 @@ version = "0.17.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a"
|
||||
|
||||
[[package]]
|
||||
name = "heapless"
|
||||
version = "0.9.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "25ba4bd83f9415b58b4ed8dc5714c76e626a105be4646c02630ad730ad3b5aa4"
|
||||
dependencies = [
|
||||
"hash32",
|
||||
"stable_deref_trait",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "heck"
|
||||
version = "0.5.0"
|
||||
@@ -5026,10 +5045,10 @@ dependencies = [
|
||||
"tokio",
|
||||
"tokio-util",
|
||||
"tracing",
|
||||
"unicode-width 0.2.2",
|
||||
"uucore",
|
||||
"uutils_term_grid",
|
||||
"windows-sys 0.61.2",
|
||||
"xutf",
|
||||
"yansi",
|
||||
]
|
||||
|
||||
@@ -5069,6 +5088,7 @@ dependencies = [
|
||||
"grep-pcre2",
|
||||
"grep-regex",
|
||||
"grep-searcher",
|
||||
"heapless",
|
||||
"html-to-markdown-rs",
|
||||
"icy_sixel",
|
||||
"ignore",
|
||||
@@ -5107,11 +5127,6 @@ dependencies = [
|
||||
"tokio-util",
|
||||
"toml",
|
||||
"uiautomation",
|
||||
"unicode-normalization",
|
||||
"unicode-properties",
|
||||
"unicode-script",
|
||||
"unicode-segmentation",
|
||||
"unicode-width 0.2.2",
|
||||
"windows-sys 0.61.2",
|
||||
"winreg 0.56.0",
|
||||
"x11rb",
|
||||
@@ -7527,12 +7542,6 @@ version = "0.1.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e70f2a8b45122e719eb623c01822704c4e0907e7e426a05927e1a1cfff5b75d0"
|
||||
|
||||
[[package]]
|
||||
name = "unicode-script"
|
||||
version = "0.5.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9fb421b350c9aff471779e262955939f565ec18b86c15364e6bdf0d662ca7c1f"
|
||||
|
||||
[[package]]
|
||||
name = "unicode-segmentation"
|
||||
version = "1.13.3"
|
||||
@@ -8843,9 +8852,9 @@ checksum = "e450f9b2ed1dff33c94c12589a87338689467b9c4f5d8a5710bd09a847d2c8a7"
|
||||
|
||||
[[package]]
|
||||
name = "xutf"
|
||||
version = "1.2.0"
|
||||
version = "1.4.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fa2ab198275c47f70ceb92678c000b1bb9468a52297c5ce20c5ab5e99fa2c33d"
|
||||
checksum = "a14f3ca7038796715a5f5fd1c22aac21423b4f86fd9ef5585ef1744a5cfc573e"
|
||||
|
||||
[[package]]
|
||||
name = "xxhash-rust"
|
||||
|
||||
+1
-5
@@ -25,7 +25,6 @@ repository = "https://github.com/can1357/oh-my-pi"
|
||||
|
||||
[patch.crates-io]
|
||||
brush-core = { path = "crates/vendor/brush-core" }
|
||||
|
||||
[profile.release]
|
||||
opt-level = 3
|
||||
lto = "fat"
|
||||
@@ -232,16 +231,13 @@ clap = { version = "4", features = ["derive"] }
|
||||
pdf-inspector = "1"
|
||||
regex = "1"
|
||||
similar = "3.1.0"
|
||||
unicode-segmentation = "1.13"
|
||||
unicode-normalization = "0.1"
|
||||
unicode-properties = "=0.1.3" # Unicode 16 - must match fancy-regex/HF/CPython oracles (utok)
|
||||
unicode-width = "0.2"
|
||||
fontdue = { version = "0.9", default-features = false }
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────────────
|
||||
# Data Structures - Collections
|
||||
# ──────────────────────────────────────────────────────────────────────────────
|
||||
phf = { version = "0.13", features = ["macros"] }
|
||||
heapless = "0.9"
|
||||
smallvec = { version = "1.15.1", features = [
|
||||
"serde",
|
||||
"write",
|
||||
|
||||
Generated
+40
-12
File diff suppressed because one or more lines are too long
@@ -207,6 +207,13 @@
|
||||
"devDependencies": {
|
||||
"@napi-rs/cli": "catalog:",
|
||||
"@types/bun": "catalog:",
|
||||
"cli-truncate": "6.1.1",
|
||||
"diff": "9.0.0",
|
||||
"gpt-tokenizer": "4.0.0",
|
||||
"mitata": "1.0.34",
|
||||
"slice-ansi": "9.0.0",
|
||||
"string-width": "8.2.2",
|
||||
"wrap-ansi": "10.0.0",
|
||||
},
|
||||
},
|
||||
"packages/omptype": {
|
||||
@@ -967,7 +974,7 @@
|
||||
|
||||
"ansi-regex": ["ansi-regex@6.3.0", "", {}, "sha512-WpDfL7NO6j7tH88IDBNVdUJxDh9nmCteAVW9dsep846XdwF4naCBK+/tGLX3KJgcpgMRXCFlTM2hKGoK9FsdrQ=="],
|
||||
|
||||
"ansi-styles": ["ansi-styles@4.3.0", "", { "dependencies": { "color-convert": "^2.0.1" } }, "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg=="],
|
||||
"ansi-styles": ["ansi-styles@6.2.3", "", {}, "sha512-4Dj6M28JB+oAH8kFkTLUo+a2jwOFkuqb3yucU0CANcRRUbxS0cP0nZYCGjcc3BNXwRIsUVmDGgzawme7zvJHvg=="],
|
||||
|
||||
"argparse": ["argparse@2.0.1", "", {}, "sha512-8+9WqebbFzpX9OR+Wa6O29asIogeRMzcGtAINdpMHHyAg10f05aSFVBbcEqGf/PXw1EjAZ+q2/bEBg3DvurK3Q=="],
|
||||
|
||||
@@ -1007,6 +1014,8 @@
|
||||
|
||||
"cli-progress": ["cli-progress@3.12.0", "", { "dependencies": { "string-width": "^4.2.3" } }, "sha512-tRkV3HJ1ASwm19THiiLIXLO7Im7wlTuKnvkYaTkyoAPefqjNg7W7DHKUlGRxy9vxDvbyCYQkQozvptuMkGCg8A=="],
|
||||
|
||||
"cli-truncate": ["cli-truncate@6.1.1", "", { "dependencies": { "slice-ansi": "^9.0.0", "string-width": "^8.2.0" } }, "sha512-06p9vyLahLa4zkGcgsGxU6iEkSOiuI4fhCH6Emhe2lPAcoUv73n72DnODsnHA+5wwXGnV0n9M9/qOQJSjYhFhw=="],
|
||||
|
||||
"cli-width": ["cli-width@4.1.0", "", {}, "sha512-ouuZd4/dm2Sw5Gmqy6bGyNNNe1qt9RpmxveLSO7KcgsTnU7RXfsw+/bukWGo1abgBiMAic068rclZsO4IWmmxQ=="],
|
||||
|
||||
"clipanion": ["clipanion@4.0.0-rc.4", "", { "dependencies": { "typanion": "^3.8.0" } }, "sha512-CXkMQxU6s9GklO/1f714dkKBMu1lopS1WFF0B8o4AxPykR1hpozxSiUZ5ZUeBjfPgCWqbcNOtZVFhB8Lkfp1+Q=="],
|
||||
@@ -1119,6 +1128,8 @@
|
||||
|
||||
"gopd": ["gopd@1.2.0", "", {}, "sha512-ZUKRh6/kUFoAiTAtTYPZJ3hw9wNxx+BIBOijnlG9PnrJsCcSjs1wyyD6vJpaYtgnzDrKYRSqf3OO6Rfa93xsRg=="],
|
||||
|
||||
"gpt-tokenizer": ["gpt-tokenizer@4.0.0", "", {}, "sha512-YAWIyzvuVUHEfW7tFfFAxH8qQb+Q3RU9nYOTy7skMNX5qzU6Q8jxTHZLyO56ug1vYvCR7wndzpd3jwD86/mhjQ=="],
|
||||
|
||||
"graceful-fs": ["graceful-fs@4.2.11", "", {}, "sha512-RbJ5/jmFcNNCcDV5o9eTnBLJ/HszWV0P73bc+Ff4nS/rJj+YaS6IGyiOL0VoBYX+l1Wrl3k63h/KrH+nhJ0XvQ=="],
|
||||
|
||||
"guid-typescript": ["guid-typescript@1.0.9", "", {}, "sha512-Y8T4vYhEfwJOTbouREvG+3XDsjr8E3kIr7uf+JZ0BYloFsttiHU0WfvANVsR7TxNUJa/WpCnw/Ino/p+DeBhBQ=="],
|
||||
@@ -1135,7 +1146,7 @@
|
||||
|
||||
"internmap": ["internmap@2.0.3", "", {}, "sha512-5Hh7Y1wQbvY5ooGgPbDaL5iYLAPzMTUrjMulskHLH6wnv/A+1q5rgEaiuqEjB+oxGXIVZs1FF+R/KPN3ZSQYYg=="],
|
||||
|
||||
"is-fullwidth-code-point": ["is-fullwidth-code-point@3.0.0", "", {}, "sha512-zymm5+u+sCsSWyD9qNaejV3DFvhCKclKdizYaJUuHA83RLjb7nSuGnddCHGv0hk+KY7BMAlsWeK4Ueg6EV6XQg=="],
|
||||
"is-fullwidth-code-point": ["is-fullwidth-code-point@5.1.0", "", { "dependencies": { "get-east-asian-width": "^1.3.1" } }, "sha512-5XHYaSyiqADb4RnZ1Bdad6cPp8Toise4TzEjcOYDHZkTCbKgiUl7WTUCpNWHuxmDt91wnsZBc9xinNzopv3JMQ=="],
|
||||
|
||||
"is-what": ["is-what@4.1.16", "", {}, "sha512-ZhMwEosbFJkA0YhFnNDgTM4ZxDRsS6HqTo7qsZM08fehyRYIYa0yHu5R6mgo1n/8MgaPBXiPimPD77baVFYg+A=="],
|
||||
|
||||
@@ -1203,6 +1214,8 @@
|
||||
|
||||
"minizlib": ["minizlib@2.1.2", "", { "dependencies": { "minipass": "^3.0.0", "yallist": "^4.0.0" } }, "sha512-bAxsR8BVfj60DWXHE3u30oHzfl4G7khkSuPW+qvpd7jFRHm7dLxOjUk1EHACJ/hxLY8phGJ0YhYHZo7jil7Qdg=="],
|
||||
|
||||
"mitata": ["mitata@1.0.34", "", {}, "sha512-Mc3zrtNBKIMeHSCQ0XqRLo1vbdIx1wvFV9c8NJAiyho6AjNfMY8bVhbS12bwciUdd1t4rj8099CH3N3NFahaUA=="],
|
||||
|
||||
"mitt": ["mitt@3.0.1", "", {}, "sha512-vKivATfr97l2/QBCYAkXYDbrIWPM2IIKEl7YPhjCvKlG3kE2gm+uBo6nEXK3M5/Ffh/FLpKExzOQ3JJoJGFKBw=="],
|
||||
|
||||
"mkdirp": ["mkdirp@1.0.4", "", { "bin": { "mkdirp": "bin/cmd.js" } }, "sha512-vVqVZQyf3WLx2Shd0qJ9xuvqgAyKPLAiqITEtqW0oIUjzo3PePDd6fW9iFz30ef7Ysp/oiWqbhszeGWW2T6Gzw=="],
|
||||
@@ -1311,6 +1324,8 @@
|
||||
|
||||
"signal-exit": ["signal-exit@4.1.0", "", {}, "sha512-bzyZ1e88w9O1iNJbKnOlvYTrWPDl46O1bG0D3XInv+9tkPrxrN8jUUTiFlDkkmKWgn1M6CfIA13SuGqOa9Korw=="],
|
||||
|
||||
"slice-ansi": ["slice-ansi@9.0.0", "", { "dependencies": { "ansi-styles": "^6.2.3", "is-fullwidth-code-point": "^5.1.0" } }, "sha512-SO/3iYL5S3W57LLEniscOGPZgOqZUPCx6d3dB+52B80yJ0XstzsC/eV8gnA4tM3MHDrKz+OCFSLNjswdSC+/bA=="],
|
||||
|
||||
"solid-js": ["solid-js@1.9.14", "", { "dependencies": { "csstype": "^3.1.0", "seroval": "~1.5.4", "seroval-plugins": "~1.5.4" } }, "sha512-sAEXC0Kk0S1EDg+8ysEWJDbYhA3RRoEjwuySUGlKIemeo0I5YZfOyumNjNs9Sv3y2nmhD+0rW66ag2HsMuQiGQ=="],
|
||||
|
||||
"solid-refresh": ["solid-refresh@0.6.3", "", { "dependencies": { "@babel/generator": "^7.23.6", "@babel/helper-module-imports": "^7.22.15", "@babel/types": "^7.23.6" }, "peerDependencies": { "solid-js": "^1.3" } }, "sha512-F3aPsX6hVw9ttm5LYlth8Q15x6MlI/J3Dn+o3EQyRTtTxidepSTwAYdozt01/YA+7ObcciagGEyXIopGZzQtbA=="],
|
||||
@@ -1375,7 +1390,7 @@
|
||||
|
||||
"webdriver-bidi-protocol": ["webdriver-bidi-protocol@0.4.2", "", {}, "sha512-VSV+fzfChirL3e7jay2yUC7B4HQCGtEWEg/MSSQbK+qWbqeGlRLlXTzPpYr3XGUvbpDHumWZBJxgesg4N7dbtA=="],
|
||||
|
||||
"wrap-ansi": ["wrap-ansi@9.0.2", "", { "dependencies": { "ansi-styles": "^6.2.1", "string-width": "^7.0.0", "strip-ansi": "^7.1.0" } }, "sha512-42AtmgqjV+X1VpdOfyTGOYRi0/zsoLqtXQckTmqTeybT+BDIbM/Guxo7x3pE2vtpr1ok6xRqM9OpBe+Jyoqyww=="],
|
||||
"wrap-ansi": ["wrap-ansi@10.0.0", "", { "dependencies": { "ansi-styles": "^6.2.3", "string-width": "^8.2.0", "strip-ansi": "^7.1.2" } }, "sha512-SGcvg80f0wUy2/fXES19feHMz8E0JoXv2uNgHOu4Dgi2OrCy1lqwFYEJz1BLbDI0exjPMe/ZdzZ/YpGECBG/aQ=="],
|
||||
|
||||
"ws": ["ws@8.21.3", "", { "peerDependencies": { "bufferutil": "^4.0.1", "utf-8-validate": ">=5.0.2" }, "optionalPeers": ["bufferutil", "utf-8-validate"] }, "sha512-201TZ/kPWxoPr/OKWjquZR1SWKXcvxdH+e1xrx89b3YbmzLMFCLfnaG1HFIgWzJOEWZ7MvpK++odZufgYR50Rw=="],
|
||||
|
||||
@@ -1469,10 +1484,14 @@
|
||||
|
||||
"babel-plugin-jsx-dom-expressions/@babel/helper-module-imports": ["@babel/helper-module-imports@7.18.6", "", { "dependencies": { "@babel/types": "^7.18.6" } }, "sha512-0NFvs3VkuSYbFi1x2Vd6tKrywq+z/cLeYC/RJNFrIX/30Bf5aiGYbtvGXolEktzJH8o5E5KJ3tT+nkxuuZFVlA=="],
|
||||
|
||||
"chalk/ansi-styles": ["ansi-styles@4.3.0", "", { "dependencies": { "color-convert": "^2.0.1" } }, "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg=="],
|
||||
|
||||
"cli-progress/string-width": ["string-width@4.2.3", "", { "dependencies": { "emoji-regex": "^8.0.0", "is-fullwidth-code-point": "^3.0.0", "strip-ansi": "^6.0.1" } }, "sha512-wKyQRQpjJ0sIp62ErSZdGsjMJWsap5oRNihHhu6G7JVO/9jIB6UyevL+tXuOqrng8j/cxKTWyWUwvSTriiZz/g=="],
|
||||
|
||||
"cliui/string-width": ["string-width@7.2.0", "", { "dependencies": { "emoji-regex": "^10.3.0", "get-east-asian-width": "^1.0.0", "strip-ansi": "^7.1.0" } }, "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ=="],
|
||||
|
||||
"cliui/wrap-ansi": ["wrap-ansi@9.0.2", "", { "dependencies": { "ansi-styles": "^6.2.1", "string-width": "^7.0.0", "strip-ansi": "^7.1.0" } }, "sha512-42AtmgqjV+X1VpdOfyTGOYRi0/zsoLqtXQckTmqTeybT+BDIbM/Guxo7x3pE2vtpr1ok6xRqM9OpBe+Jyoqyww=="],
|
||||
|
||||
"fastembed/onnxruntime-node": ["onnxruntime-node@1.21.0", "", { "dependencies": { "global-agent": "^3.0.0", "onnxruntime-common": "1.21.0", "tar": "^7.0.1" }, "os": [ "linux", "win32", "darwin", ] }, "sha512-NeaCX6WW2L8cRCSqy3bInlo5ojjQqu2fD3D+9W5qb5irwxhEyWKXeH2vZ8W9r6VxaMPUan+4/7NDwZMtouZxEw=="],
|
||||
|
||||
"fs-minipass/minipass": ["minipass@3.3.6", "", { "dependencies": { "yallist": "^4.0.0" } }, "sha512-DxiNidxSEK+tHG6zOIklvNOwm3hvCrbUrdtzY74U6HKTJxvIDfOUL5W5P2Ghd3DTkhhKPYGqeNUIh5qcM4YBfw=="],
|
||||
@@ -1485,10 +1504,6 @@
|
||||
|
||||
"vite/lightningcss": ["lightningcss@1.33.0", "", { "dependencies": { "detect-libc": "^2.0.3" }, "optionalDependencies": { "lightningcss-android-arm64": "1.33.0", "lightningcss-darwin-arm64": "1.33.0", "lightningcss-darwin-x64": "1.33.0", "lightningcss-freebsd-x64": "1.33.0", "lightningcss-linux-arm-gnueabihf": "1.33.0", "lightningcss-linux-arm64-gnu": "1.33.0", "lightningcss-linux-arm64-musl": "1.33.0", "lightningcss-linux-x64-gnu": "1.33.0", "lightningcss-linux-x64-musl": "1.33.0", "lightningcss-win32-arm64-msvc": "1.33.0", "lightningcss-win32-x64-msvc": "1.33.0" } }, "sha512-WkUDrojuJs0xkgGf2udWxa3yGBRxPtxUkB79i6aCZLRgc7PM8fZe9TosfPDcvEpQZbuFASnHYmRLBLUbmLOIIA=="],
|
||||
|
||||
"wrap-ansi/ansi-styles": ["ansi-styles@6.2.3", "", {}, "sha512-4Dj6M28JB+oAH8kFkTLUo+a2jwOFkuqb3yucU0CANcRRUbxS0cP0nZYCGjcc3BNXwRIsUVmDGgzawme7zvJHvg=="],
|
||||
|
||||
"wrap-ansi/string-width": ["string-width@7.2.0", "", { "dependencies": { "emoji-regex": "^10.3.0", "get-east-asian-width": "^1.0.0", "strip-ansi": "^7.1.0" } }, "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ=="],
|
||||
|
||||
"@huggingface/transformers/onnxruntime-node/global-agent": ["global-agent@3.0.0", "", { "dependencies": { "boolean": "^3.0.1", "es6-error": "^4.1.1", "matcher": "^3.0.0", "roarr": "^2.15.3", "semver": "^7.3.2", "serialize-error": "^7.0.1" } }, "sha512-PT6XReJ+D07JvGoxQMkT6qji/jVNfX/h364XHZOWeRzy64sSFr+xJ5OX7LI3b4MPQzdL4H8Y8M0xzPpsVMwA8Q=="],
|
||||
|
||||
"@huggingface/transformers/onnxruntime-node/onnxruntime-common": ["onnxruntime-common@1.24.3", "", {}, "sha512-GeuPZO6U/LBJXvwdaqHbuUmoXiEdeCjWi/EG7Y1HNnDwJYuk6WUbNXpF6luSUY8yASul3cmUlLGrCCL1ZgVXqA=="],
|
||||
@@ -1507,6 +1522,8 @@
|
||||
|
||||
"@typescript/analyze-trace/yargs/yargs-parser": ["yargs-parser@20.2.9", "", {}, "sha512-y11nGElTIV+CT3Zv9t7VKl+Q3hTQoT9a1Qzezhhl6Rp21gJ/IVTW7Z3y9EWXhuUBC2Shnf+DX0antecpAwSP8w=="],
|
||||
|
||||
"cli-progress/string-width/is-fullwidth-code-point": ["is-fullwidth-code-point@3.0.0", "", {}, "sha512-zymm5+u+sCsSWyD9qNaejV3DFvhCKclKdizYaJUuHA83RLjb7nSuGnddCHGv0hk+KY7BMAlsWeK4Ueg6EV6XQg=="],
|
||||
|
||||
"cli-progress/string-width/strip-ansi": ["strip-ansi@6.0.1", "", { "dependencies": { "ansi-regex": "^5.0.1" } }, "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A=="],
|
||||
|
||||
"cliui/string-width/emoji-regex": ["emoji-regex@10.6.0", "", {}, "sha512-toUI84YS5YmxW219erniWD0CIVOo46xGKColeNQRgOzDorgBi1v4D71/OFzgD9GO2UGKIv1C3Sp8DAn0+j5w7A=="],
|
||||
@@ -1539,8 +1556,6 @@
|
||||
|
||||
"vite/lightningcss/lightningcss-win32-x64-msvc": ["lightningcss-win32-x64-msvc@1.33.0", "", { "os": "win32", "cpu": "x64" }, "sha512-OlEICDx/Xl0FqSp4bry8zFnCvGpig3Gl4gCquvYwHuqJKEC1+n9NgDniFvqHGmMv1ZkqDJrDqKKSykTDX+ehuA=="],
|
||||
|
||||
"wrap-ansi/string-width/emoji-regex": ["emoji-regex@10.6.0", "", {}, "sha512-toUI84YS5YmxW219erniWD0CIVOo46xGKColeNQRgOzDorgBi1v4D71/OFzgD9GO2UGKIv1C3Sp8DAn0+j5w7A=="],
|
||||
|
||||
"@huggingface/transformers/onnxruntime-node/global-agent/matcher": ["matcher@3.0.0", "", { "dependencies": { "escape-string-regexp": "^4.0.0" } }, "sha512-OkeDaAZ/bQCxeFAozM55PKcKU0yJMPGifLwV4Qgjitu+5MoAfSQN4lsLJeXZ1b8w0x+/Emda6MZgXS1jvsapng=="],
|
||||
|
||||
"@huggingface/transformers/onnxruntime-node/global-agent/serialize-error": ["serialize-error@7.0.1", "", { "dependencies": { "type-fest": "^0.13.1" } }, "sha512-8I8TjW5KMOKsZQTvoxjuSIa7foAwPWGOts+6o7sgjz41/qMD9VQHEDxi6PBvK2l0MXUmqZyNpUK+T2tQaaElvw=="],
|
||||
@@ -1549,6 +1564,8 @@
|
||||
|
||||
"@typescript/analyze-trace/yargs/cliui/wrap-ansi": ["wrap-ansi@7.0.0", "", { "dependencies": { "ansi-styles": "^4.0.0", "string-width": "^4.1.0", "strip-ansi": "^6.0.0" } }, "sha512-YVGIj2kamLSTxw6NsZjoBxfSwsn0ycdesmc4p+Q21c5zPuZ1pl+NfxVdxPtdHvmNVOQ6XSYG4AUtyt/Fi7D16Q=="],
|
||||
|
||||
"@typescript/analyze-trace/yargs/string-width/is-fullwidth-code-point": ["is-fullwidth-code-point@3.0.0", "", {}, "sha512-zymm5+u+sCsSWyD9qNaejV3DFvhCKclKdizYaJUuHA83RLjb7nSuGnddCHGv0hk+KY7BMAlsWeK4Ueg6EV6XQg=="],
|
||||
|
||||
"@typescript/analyze-trace/yargs/string-width/strip-ansi": ["strip-ansi@6.0.1", "", { "dependencies": { "ansi-regex": "^5.0.1" } }, "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A=="],
|
||||
|
||||
"cli-progress/string-width/strip-ansi/ansi-regex": ["ansi-regex@5.0.1", "", {}, "sha512-quJQXlTSUGL2LH9SUXo8VwsY4soanhgo6LNSm84E1LBcE8s3O0wpdiRzyR9z/ZZJMlMWv37qOOb9pdJlMUEKFQ=="],
|
||||
@@ -1569,6 +1586,8 @@
|
||||
|
||||
"@typescript/analyze-trace/yargs/cliui/strip-ansi/ansi-regex": ["ansi-regex@5.0.1", "", {}, "sha512-quJQXlTSUGL2LH9SUXo8VwsY4soanhgo6LNSm84E1LBcE8s3O0wpdiRzyR9z/ZZJMlMWv37qOOb9pdJlMUEKFQ=="],
|
||||
|
||||
"@typescript/analyze-trace/yargs/cliui/wrap-ansi/ansi-styles": ["ansi-styles@4.3.0", "", { "dependencies": { "color-convert": "^2.0.1" } }, "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg=="],
|
||||
|
||||
"@typescript/analyze-trace/yargs/string-width/strip-ansi/ansi-regex": ["ansi-regex@5.0.1", "", {}, "sha512-quJQXlTSUGL2LH9SUXo8VwsY4soanhgo6LNSm84E1LBcE8s3O0wpdiRzyR9z/ZZJMlMWv37qOOb9pdJlMUEKFQ=="],
|
||||
|
||||
"fastembed/onnxruntime-node/global-agent/serialize-error/type-fest": ["type-fest@0.13.1", "", {}, "sha512-34R7HTnG0XIJcBSn5XhDd7nNFPRcXYRZrBB2O2jdKqYODldSzBAqzsWoZYYvduky73toYS/ESqxPvkDf/F0XMg=="],
|
||||
|
||||
@@ -97,7 +97,7 @@ smallvec.workspace = true
|
||||
similar = "3.1.0"
|
||||
tempfile = "3.15.0"
|
||||
tokio-util = "0.7"
|
||||
unicode-width = "0.2.0"
|
||||
xutf.workspace = true
|
||||
libc = "0.2.172"
|
||||
memmap2 = "0.9"
|
||||
uutils_term_grid = "0.8"
|
||||
|
||||
@@ -11,6 +11,7 @@ use std::{
|
||||
io::{self, BufReader, Read, Write},
|
||||
};
|
||||
|
||||
use brush_core::{ShellExtensions, builtins::Registration};
|
||||
use clap::{Arg, ArgAction, ArgMatches, Command, ValueHint, builder::ValueParser};
|
||||
use os_display::Quotable;
|
||||
use uucore::{
|
||||
@@ -18,6 +19,7 @@ use uucore::{
|
||||
AlgoKind, BlakeLength, ChecksumError, ReadingMode, ShaLength, SizedAlgoKind,
|
||||
digest_reader, escape_filename, parse_blake_length, unescape_filename, SUPPORTED_ALGORITHMS,
|
||||
},
|
||||
hardware::{HasHardwareFeatures as _, SimdPolicy},
|
||||
line_ending::LineEnding,
|
||||
os_str_from_bytes,
|
||||
quoting_style::{QuotingStyle, locale_aware_escape_name},
|
||||
@@ -25,7 +27,7 @@ use uucore::{
|
||||
sum::{self, Blake2b, Blake3, DigestOutput},
|
||||
};
|
||||
|
||||
use crate::host::{Host, os_bytes};
|
||||
use crate::host::{Host, Utility, matches_parser, os_bytes, util};
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
struct Failure(String);
|
||||
@@ -415,6 +417,164 @@ pub(crate) fn run(
|
||||
}
|
||||
}
|
||||
|
||||
/// Parsed `cksum` invocation.
|
||||
pub(crate) struct Cksum {
|
||||
matches: ArgMatches,
|
||||
}
|
||||
|
||||
matches_parser!(Cksum, cksum_app);
|
||||
|
||||
impl Utility for Cksum {
|
||||
const NAME: &'static str = "cksum";
|
||||
|
||||
fn run(self, host: &mut Host) -> i32 {
|
||||
match run_cksum(host, self.matches) {
|
||||
Ok(()) => host.exit_code(),
|
||||
Err(error) => {
|
||||
if !error.0.is_empty() {
|
||||
host.error(error, 1);
|
||||
}
|
||||
1
|
||||
},
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Builds the clap command for the GNU `cksum` multi-algorithm front-end.
|
||||
fn cksum_app() -> Command {
|
||||
default_checksum_app(
|
||||
"Print or verify checksums; without --algorithm, prints the POSIX CRC and byte count",
|
||||
"cksum [OPTION]... [FILE]...",
|
||||
)
|
||||
.name("cksum")
|
||||
.with_algo()
|
||||
.with_untagged()
|
||||
.with_tag(true)
|
||||
.with_length()
|
||||
.with_raw()
|
||||
.with_check_and_opts()
|
||||
.with_base64()
|
||||
.with_text(false)
|
||||
.with_binary()
|
||||
.with_zero()
|
||||
.with_debug()
|
||||
}
|
||||
|
||||
/// Creates the `cksum` builtin registration.
|
||||
pub(crate) fn cksum_builtin<SE: ShellExtensions>() -> Registration<SE> {
|
||||
util::<Cksum, SE>()
|
||||
}
|
||||
|
||||
/// Sanitizes `--length` against `--algorithm`, mirroring GNU `cksum`.
|
||||
fn sanitize_cksum_length(
|
||||
host: &mut Host,
|
||||
algo: Option<AlgoKind>,
|
||||
input_length: Option<&str>,
|
||||
) -> ExecResult<Option<usize>> {
|
||||
match (algo, input_length) {
|
||||
// No provided length is not a problem so far.
|
||||
(_, None) => Ok(None),
|
||||
|
||||
// For SHA2 and SHA3, if a length is provided, ensure it is correct.
|
||||
(Some(algo @ (AlgoKind::Sha2 | AlgoKind::Sha3)), Some(len)) => {
|
||||
// Positive overflow while parsing counts as an invalid number,
|
||||
// but a number still; it gets the extra reminder of the accepted
|
||||
// inputs, unlike a plain parse failure.
|
||||
let parsed = match len.parse::<usize>() {
|
||||
Ok(parsed) => Some(parsed),
|
||||
Err(error) if *error.kind() == std::num::IntErrorKind::PosOverflow => None,
|
||||
Err(_) => return Err(failure(ChecksumError::InvalidLength(len.into()))),
|
||||
};
|
||||
match parsed {
|
||||
Some(parsed @ (224 | 256 | 384 | 512)) => Ok(Some(parsed)),
|
||||
_ => {
|
||||
host.error(ChecksumError::InvalidLength(len.into()), 1);
|
||||
Err(failure(ChecksumError::InvalidLengthForSha(algo.to_uppercase().into())))
|
||||
},
|
||||
}
|
||||
},
|
||||
|
||||
// SHAKE128 and SHAKE256 algorithms optionally take a bit length. No
|
||||
// validation is performed on this length, any value is valid.
|
||||
(Some(AlgoKind::Shake128 | AlgoKind::Shake256), Some(len)) => match len.parse::<usize>() {
|
||||
Ok(0) => Ok(None),
|
||||
Ok(parsed) => Ok(Some(parsed)),
|
||||
Err(_) => Err(failure(ChecksumError::InvalidLength(len.into()))),
|
||||
},
|
||||
|
||||
// For BLAKE, if a length is provided, validate it.
|
||||
(Some(algo @ (AlgoKind::Blake2b | AlgoKind::Blake3)), Some(len)) => {
|
||||
parse_blake_length(algo, BlakeLength::String(len)).map(Some).map_err(failure)
|
||||
},
|
||||
|
||||
// For any other provided algorithm, check if length is 0.
|
||||
// Otherwise, this is an error.
|
||||
(_, Some(len)) if len.parse::<u32>() == Ok(0) => Ok(None),
|
||||
(_, Some(_)) => Err(failure(ChecksumError::LengthOnlyForBlake2bSha2Sha3)),
|
||||
}
|
||||
}
|
||||
|
||||
/// Prints CPU hardware capability detection info, matching GNU `cksum
|
||||
/// --debug`.
|
||||
fn print_cpu_debug_info(host: &mut Host) {
|
||||
let features = SimdPolicy::detect();
|
||||
|
||||
let mut print_feature = |name: &str, available: bool| {
|
||||
if available {
|
||||
let _ = writeln!(host.stderr, "using {name} hardware support");
|
||||
} else {
|
||||
let _ = writeln!(host.stderr, "{name} support not detected");
|
||||
}
|
||||
};
|
||||
|
||||
// x86/x86_64
|
||||
print_feature("avx512", features.has_avx512());
|
||||
print_feature("avx2", features.has_avx2());
|
||||
print_feature("pclmul", features.has_pclmul());
|
||||
|
||||
// ARM aarch64
|
||||
if cfg!(target_arch = "aarch64") {
|
||||
print_feature("vmull", features.has_vmull());
|
||||
}
|
||||
}
|
||||
|
||||
/// Runs one parsed `cksum` invocation. Unlike the standalone utilities, the
|
||||
/// algorithm comes from `--algorithm` (default: legacy POSIX CRC), output
|
||||
/// defaults to tagged, and `--raw`/`--base64` are accepted.
|
||||
fn run_cksum(host: &mut Host, matches: ArgMatches) -> ExecResult<()> {
|
||||
let algo = matches
|
||||
.get_one::<String>(options::ALGORITHM)
|
||||
.map(AlgoKind::from_cksum)
|
||||
.transpose()
|
||||
.map_err(failure)?;
|
||||
|
||||
let input_length = matches.get_one::<String>(options::LENGTH).map(String::as_str);
|
||||
let length = sanitize_cksum_length(host, algo, input_length)?;
|
||||
|
||||
let tag = !matches.get_flag(options::UNTAGGED);
|
||||
let binary = matches.get_flag(options::BINARY);
|
||||
let text = matches.get_flag(options::TEXT);
|
||||
|
||||
// Specifying --text without ever mentioning --untagged fails.
|
||||
if text && tag {
|
||||
return Err(failure(ChecksumError::TextWithoutUntagged));
|
||||
}
|
||||
|
||||
let output_format = OutputFormat::from_cksum(
|
||||
algo.unwrap_or(AlgoKind::Crc),
|
||||
tag,
|
||||
binary,
|
||||
matches.get_flag(options::RAW),
|
||||
matches.get_flag(options::BASE64),
|
||||
);
|
||||
|
||||
if matches.get_flag(options::DEBUG) {
|
||||
print_cpu_debug_info(host);
|
||||
}
|
||||
|
||||
checksum_main(host, algo, length, matches, output_format)
|
||||
}
|
||||
|
||||
/// Use the same buffer size as GNU when reading a file to create a checksum
|
||||
/// from it: 32 KiB.
|
||||
const READ_BUFFER_SIZE: usize = 32 * 1024;
|
||||
@@ -1953,5 +2113,158 @@ mod tests {
|
||||
assert_eq!(&buffer, expected);
|
||||
}
|
||||
}
|
||||
|
||||
mod cksum_front_end {
|
||||
//! `cksum` is the GNU multi-algorithm front-end; without these the
|
||||
//! builtin would shadow the system binary while rejecting or
|
||||
//! misprinting invocations the real `cksum` accepts.
|
||||
|
||||
use std::fs;
|
||||
|
||||
use super::super::Cksum;
|
||||
use crate::host::run_util;
|
||||
|
||||
/// Failure mode: default invocation must keep the POSIX CRC format
|
||||
/// (`<crc> <size>`), not a hex digest.
|
||||
#[test]
|
||||
fn default_is_posix_crc_output() {
|
||||
let (code, capture) = run_util::<Cksum>(&[], "hi", "/");
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "2352138605 2\n");
|
||||
}
|
||||
|
||||
/// Failure mode: `cksum somefile` printing no filename or the wrong
|
||||
/// CRC would silently diverge from `/usr/bin/cksum`.
|
||||
#[test]
|
||||
fn file_operand_appends_the_filename() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
fs::write(dir.path().join("input"), b"hi").unwrap();
|
||||
let (code, capture) = run_util::<Cksum>(&["input"], "", dir.path());
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "2352138605 2 input\n");
|
||||
}
|
||||
|
||||
/// Failure mode: `-a sha256` is the flagship GNU extension; it must
|
||||
/// parse and produce BSD-tagged output by default.
|
||||
#[test]
|
||||
fn algorithm_selects_tagged_sha256() {
|
||||
let (code, capture) = run_util::<Cksum>(&["-a", "sha256"], "hi", "/");
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(
|
||||
capture.out(),
|
||||
"SHA256 (-) = 8f434346648f6b96df89dda901c5176b10a6d83961dd3c1ac88b59b2dc327aa4\n"
|
||||
);
|
||||
}
|
||||
|
||||
/// Failure mode: `--untagged` must switch to the two-space coreutils
|
||||
/// format so output can be fed back to `sha256sum -c`.
|
||||
#[test]
|
||||
fn untagged_prints_coreutils_format() {
|
||||
let (code, capture) = run_util::<Cksum>(&["-a", "sha256", "--untagged"], "hi", "/");
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(
|
||||
capture.out(),
|
||||
"8f434346648f6b96df89dda901c5176b10a6d83961dd3c1ac88b59b2dc327aa4 -\n"
|
||||
);
|
||||
}
|
||||
|
||||
/// Failure mode: `--base64` must encode the digest, not error or
|
||||
/// print hex.
|
||||
#[test]
|
||||
fn base64_encodes_the_digest() {
|
||||
let (code, capture) = run_util::<Cksum>(&["-a", "sha256", "--base64"], "hi", "/");
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "SHA256 (-) = j0NDRmSPa5bfid2pAcUXaxCm2Dlh3TwayItZstwyeqQ=\n");
|
||||
}
|
||||
|
||||
/// Failure mode: `-a blake2b -l N` (the one length-taking algorithm
|
||||
/// agents use) must honor the bit length in the tag.
|
||||
#[test]
|
||||
fn blake2b_length_is_honored() {
|
||||
let (code, capture) = run_util::<Cksum>(&["-a", "blake2b", "-l", "8"], "abc", "/");
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "BLAKE2b-8 (-) = 6b\n");
|
||||
}
|
||||
|
||||
/// Failure mode: GNU rejects `--length` for non-length algorithms;
|
||||
/// silently ignoring it would hide user error.
|
||||
#[test]
|
||||
fn length_requires_a_length_algorithm() {
|
||||
let (code, capture) = run_util::<Cksum>(&["-l", "16"], "", "/");
|
||||
assert_eq!(code, 1);
|
||||
assert!(
|
||||
capture.err().contains("--length is only supported with"),
|
||||
"{}",
|
||||
capture.err()
|
||||
);
|
||||
}
|
||||
|
||||
/// Failure mode: `--text` without `--untagged` is a GNU usage error
|
||||
/// (`--text mode is only supported with --untagged`), even though the
|
||||
/// standalone `*sum` utilities accept `-t` freely.
|
||||
#[test]
|
||||
fn text_without_untagged_is_rejected() {
|
||||
let (code, capture) = run_util::<Cksum>(&["-a", "sha256", "-t"], "", "/");
|
||||
assert_eq!(code, 1);
|
||||
assert!(
|
||||
capture.err().contains("--text mode is only supported with --untagged"),
|
||||
"{}",
|
||||
capture.err()
|
||||
);
|
||||
}
|
||||
|
||||
/// Failure mode: verification must accept tagged lines produced by
|
||||
/// the compute side and exit 0.
|
||||
#[test]
|
||||
fn check_verifies_tagged_lines() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
fs::write(dir.path().join("data"), b"hi").unwrap();
|
||||
fs::write(
|
||||
dir.path().join("list"),
|
||||
b"SHA256 (data) = 8f434346648f6b96df89dda901c5176b10a6d83961dd3c1ac88b59b2dc327aa4\n",
|
||||
)
|
||||
.unwrap();
|
||||
let (code, capture) = run_util::<Cksum>(&["-c", "list"], "", dir.path());
|
||||
assert_eq!(code, 0, "stderr: {}", capture.err());
|
||||
assert_eq!(capture.out(), "data: OK\n");
|
||||
}
|
||||
|
||||
/// Failure mode: real GNU 9.x treats `-c -b` as a fatal usage error
|
||||
/// ("meaningless when verifying checksums", exit 1, nothing
|
||||
/// verified); the builtin must not silently verify anyway.
|
||||
#[test]
|
||||
fn check_with_binary_stays_fatal_like_gnu() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
fs::write(dir.path().join("data"), b"hi").unwrap();
|
||||
fs::write(
|
||||
dir.path().join("list"),
|
||||
b"SHA256 (data) = 8f434346648f6b96df89dda901c5176b10a6d83961dd3c1ac88b59b2dc327aa4\n",
|
||||
)
|
||||
.unwrap();
|
||||
let (code, capture) = run_util::<Cksum>(&["-c", "-b", "list"], "", dir.path());
|
||||
assert_eq!(code, 1);
|
||||
assert!(
|
||||
capture
|
||||
.err()
|
||||
.contains("the --binary and --text options are meaningless when verifying checksums"),
|
||||
"{}",
|
||||
capture.err()
|
||||
);
|
||||
assert!(!capture.out().contains("OK"), "must not verify: {}", capture.out());
|
||||
}
|
||||
|
||||
/// Failure mode: legacy algorithms cannot be verified; GNU errors out
|
||||
/// rather than parsing the list.
|
||||
#[test]
|
||||
fn check_rejects_legacy_algorithms() {
|
||||
let (code, capture) = run_util::<Cksum>(&["-a", "crc", "-c"], "", "/");
|
||||
assert_eq!(code, 1);
|
||||
assert!(
|
||||
capture.err().contains("--check is not supported with --algorithm"),
|
||||
"{}",
|
||||
capture.err()
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+546
-16
@@ -764,7 +764,7 @@ use std::{
|
||||
ffi::OsString,
|
||||
fs::File,
|
||||
io::{BufRead, BufReader, BufWriter, Read, Write},
|
||||
path::PathBuf,
|
||||
path::{Path, PathBuf},
|
||||
sync::LazyLock,
|
||||
};
|
||||
#[cfg(any(
|
||||
@@ -780,7 +780,7 @@ use std::ffi::{CStr, CString};
|
||||
|
||||
use clap::{Arg, ArgAction, ArgMatches, Command};
|
||||
use jiff::{
|
||||
Timestamp, Zoned,
|
||||
Span, Timestamp, Zoned,
|
||||
fmt::strtime::{self, BrokenDownTime, Config, PosixCustom},
|
||||
tz::{Offset, TimeZone, TimeZoneDatabase},
|
||||
};
|
||||
@@ -807,6 +807,9 @@ const OPT_SET: &str = "set";
|
||||
const OPT_REFERENCE: &str = "reference";
|
||||
const OPT_UNIVERSAL: &str = "universal";
|
||||
const OPT_UNIVERSAL_2: &str = "utc";
|
||||
// BSD compatibility options (no GNU equivalents).
|
||||
const OPT_BSD_ADJUST: &str = "bsd-adjust";
|
||||
const OPT_BSD_PARSE_ONLY: &str = "bsd-parse-only";
|
||||
|
||||
/// Settings for this program, parsed from the command line
|
||||
struct Settings {
|
||||
@@ -849,6 +852,8 @@ enum DateSource {
|
||||
FileMtime(PathBuf),
|
||||
Stdin,
|
||||
Human(String),
|
||||
/// BSD `date -j -f FMT VALUE`: VALUE parsed with the strptime format FMT.
|
||||
Strptime { format: String, value: String },
|
||||
Resolution,
|
||||
}
|
||||
|
||||
@@ -1035,6 +1040,247 @@ fn parse_military_timezone_with_offset(s: &str) -> Option<(i32, DayDelta)> {
|
||||
|
||||
Some((hours_from_midnight, day_delta))
|
||||
}
|
||||
/// Rewrite `-I`/`--iso-8601` so the optional ISO precision only binds when it
|
||||
/// is attached (`-Ihours`, `--iso-8601=hours`), matching GNU getopt. Without
|
||||
/// this, clap's optional-value handling greedily consumes a following
|
||||
/// `+FORMAT` operand (`date -I +%s`).
|
||||
///
|
||||
/// `argv[0]` is the command name. Scanning stops at `--`, and a token that is
|
||||
/// the value of a preceding option is never rewritten.
|
||||
fn rewrite_date_argv(argv: Vec<OsString>) -> Vec<OsString> {
|
||||
/// Long options that consume a separate value token.
|
||||
const VALUE_LONGS: &[&str] = &["date", "file", "reference", "set", "rfc-3339"];
|
||||
/// Long flags that take no value (`iso-8601` is handled separately).
|
||||
const FLAG_LONGS: &[&str] =
|
||||
&["debug", "resolution", "rfc-email", "rfc-2822", "rfc-822", "universal", "utc", "uct"];
|
||||
/// Short options that consume a value, attached or separate.
|
||||
const VALUE_SHORTS: &[char] = &['d', 'f', 'r', 's', 'v'];
|
||||
|
||||
let mut out = Vec::with_capacity(argv.len());
|
||||
let mut argv = argv.into_iter();
|
||||
if let Some(name) = argv.next() {
|
||||
out.push(name);
|
||||
}
|
||||
let mut skip_value = false;
|
||||
let mut opts_ended = false;
|
||||
for arg in argv {
|
||||
if skip_value || opts_ended {
|
||||
skip_value = false;
|
||||
out.push(arg);
|
||||
continue;
|
||||
}
|
||||
let Some(token) = arg.to_str() else {
|
||||
out.push(arg);
|
||||
continue;
|
||||
};
|
||||
if token == "--" {
|
||||
opts_ended = true;
|
||||
out.push(arg);
|
||||
} else if let Some(long) = token.strip_prefix("--") {
|
||||
// `infer_long_args` is enabled, so any unambiguous prefix names
|
||||
// the option.
|
||||
let (name, value) = match long.split_once('=') {
|
||||
Some((name, value)) => (name, Some(value)),
|
||||
None => (long, None),
|
||||
};
|
||||
let ambiguous = VALUE_LONGS
|
||||
.iter()
|
||||
.chain(FLAG_LONGS)
|
||||
.any(|other| other.starts_with(name));
|
||||
if !name.is_empty() && "iso-8601".starts_with(name) && !ambiguous {
|
||||
out.push(format!("--iso-8601={}", value.unwrap_or(DATE)).into());
|
||||
} else {
|
||||
if value.is_none()
|
||||
&& VALUE_LONGS.iter().filter(|l| l.starts_with(name)).count() == 1
|
||||
&& !FLAG_LONGS.iter().any(|l| l.starts_with(name))
|
||||
&& !"iso-8601".starts_with(name)
|
||||
{
|
||||
skip_value = true;
|
||||
}
|
||||
out.push(arg);
|
||||
}
|
||||
} else if let Some(cluster) = token.strip_prefix('-').filter(|rest| !rest.is_empty()) {
|
||||
// Walk a short-option cluster (`-uI`, `-ud @0`, `-Ihours`).
|
||||
let mut rewrote = false;
|
||||
for (index, ch) in cluster.char_indices() {
|
||||
if ch == 'I' {
|
||||
let flags = &cluster[..index];
|
||||
let rest = &cluster[index + ch.len_utf8()..];
|
||||
if !flags.is_empty() {
|
||||
out.push(format!("-{flags}").into());
|
||||
}
|
||||
let spec = if rest.is_empty() { DATE } else { rest };
|
||||
out.push(format!("--iso-8601={spec}").into());
|
||||
rewrote = true;
|
||||
break;
|
||||
}
|
||||
if VALUE_SHORTS.contains(&ch) {
|
||||
// The remainder of the token (or the next token when the
|
||||
// remainder is empty) is this option's value.
|
||||
if cluster[index + ch.len_utf8()..].is_empty() {
|
||||
skip_value = true;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
if !rewrote {
|
||||
out.push(arg);
|
||||
}
|
||||
} else {
|
||||
out.push(arg);
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// Unit letter of a BSD `-v` adjustment.
|
||||
#[derive(Clone, Copy)]
|
||||
enum BsdAdjustUnit {
|
||||
Year,
|
||||
Month,
|
||||
Week,
|
||||
Day,
|
||||
Hour,
|
||||
Minute,
|
||||
Second,
|
||||
}
|
||||
|
||||
/// One BSD `-v` adjustment: a relative offset (`+1d`, `-2m`) or an absolute
|
||||
/// field set (`1d` sets the day of the month).
|
||||
#[derive(Clone, Copy)]
|
||||
enum BsdAdjustment {
|
||||
Offset(i64, BsdAdjustUnit),
|
||||
Set(i64, BsdAdjustUnit),
|
||||
}
|
||||
|
||||
/// Parse one BSD `-v` argument of the form `[+|-]VAL[ymwdHMS]`.
|
||||
///
|
||||
/// BSD is case-sensitive only where it is ambiguous (`m` month vs `M`
|
||||
/// minute); the unambiguous letters are accepted in either case. Weekday and
|
||||
/// month names (`-vsun`, `-vjan`) are not supported.
|
||||
fn parse_bsd_adjustment(spec: &str) -> Option<BsdAdjustment> {
|
||||
let (sign, rest) = match *spec.as_bytes().first()? {
|
||||
b'+' => (Some(1), &spec[1..]),
|
||||
b'-' => (Some(-1), &spec[1..]),
|
||||
_ => (None, spec),
|
||||
};
|
||||
let mut chars = rest.chars();
|
||||
let unit = match chars.next_back()? {
|
||||
'y' | 'Y' => BsdAdjustUnit::Year,
|
||||
'm' => BsdAdjustUnit::Month,
|
||||
'w' | 'W' => BsdAdjustUnit::Week,
|
||||
'd' | 'D' => BsdAdjustUnit::Day,
|
||||
'H' | 'h' => BsdAdjustUnit::Hour,
|
||||
'M' => BsdAdjustUnit::Minute,
|
||||
'S' | 's' => BsdAdjustUnit::Second,
|
||||
_ => return None,
|
||||
};
|
||||
let digits = chars.as_str();
|
||||
if digits.is_empty() || !digits.bytes().all(|b| b.is_ascii_digit()) {
|
||||
return None;
|
||||
}
|
||||
let value: i64 = digits.parse().ok()?;
|
||||
Some(match sign {
|
||||
Some(sign) => BsdAdjustment::Offset(sign * value, unit),
|
||||
None => BsdAdjustment::Set(value, unit),
|
||||
})
|
||||
}
|
||||
|
||||
/// Apply BSD `-v` adjustments to `date` in command-line order.
|
||||
fn apply_bsd_adjustments(mut date: Zoned, adjustments: &[BsdAdjustment]) -> Result<Zoned, String> {
|
||||
for adjustment in adjustments {
|
||||
date = match *adjustment {
|
||||
BsdAdjustment::Offset(value, unit) => {
|
||||
let span = Span::new();
|
||||
let span = match unit {
|
||||
BsdAdjustUnit::Year => span.try_years(value),
|
||||
BsdAdjustUnit::Month => span.try_months(value),
|
||||
BsdAdjustUnit::Week => span.try_weeks(value),
|
||||
BsdAdjustUnit::Day => span.try_days(value),
|
||||
BsdAdjustUnit::Hour => span.try_hours(value),
|
||||
BsdAdjustUnit::Minute => span.try_minutes(value),
|
||||
BsdAdjustUnit::Second => span.try_seconds(value),
|
||||
}
|
||||
.map_err(|error| format!("invalid adjustment ({error})"))?;
|
||||
date
|
||||
.checked_add(span)
|
||||
.map_err(|error| format!("cannot adjust date ({error})"))?
|
||||
},
|
||||
BsdAdjustment::Set(value, unit) => {
|
||||
let narrow = |unit: char| {
|
||||
i8::try_from(value).map_err(|_| format!("invalid adjustment: '{value}{unit}'"))
|
||||
};
|
||||
let with = date.with();
|
||||
let with = match unit {
|
||||
BsdAdjustUnit::Year => {
|
||||
// BSD windows two-digit years: 69-99 => 19xx, 0-68 => 20xx.
|
||||
let year = match value {
|
||||
0..=68 => value + 2000,
|
||||
69..=99 => value + 1900,
|
||||
_ => value,
|
||||
};
|
||||
let year = i16::try_from(year)
|
||||
.map_err(|_| format!("invalid adjustment: '{value}y'"))?;
|
||||
with.year(year)
|
||||
},
|
||||
BsdAdjustUnit::Month => with.month(narrow('m')?),
|
||||
BsdAdjustUnit::Week => {
|
||||
return Err(format!(
|
||||
"unsupported adjustment: '{value}w' (setting the week is not \
|
||||
implemented; use an offset like '+{value}w')"
|
||||
));
|
||||
},
|
||||
BsdAdjustUnit::Day => with.day(narrow('d')?),
|
||||
BsdAdjustUnit::Hour => with.hour(narrow('H')?),
|
||||
BsdAdjustUnit::Minute => with.minute(narrow('M')?),
|
||||
BsdAdjustUnit::Second => with.second(narrow('S')?),
|
||||
};
|
||||
with
|
||||
.build()
|
||||
.map_err(|error| format!("cannot adjust date ({error})"))?
|
||||
},
|
||||
};
|
||||
}
|
||||
Ok(date)
|
||||
}
|
||||
|
||||
/// BSD `date -j -f FMT VALUE`: parse VALUE with the strptime format FMT.
|
||||
///
|
||||
/// Fields the format does not mention keep the current date/time's values,
|
||||
/// matching BSD `date`, which seeds the broken-down time from
|
||||
/// `localtime(now)` before calling strptime(3).
|
||||
fn parse_bsd_strptime(format: &str, value: &str, now: &Zoned) -> Result<Zoned, String> {
|
||||
let convert_error =
|
||||
|error: jiff::Error| format!("failed conversion of '{value}' using format '{format}' ({error})");
|
||||
let broken = BrokenDownTime::parse(format, value).map_err(convert_error)?;
|
||||
// `%s` (or a complete civil datetime plus an offset) pins an instant.
|
||||
if let Ok(timestamp) = broken.to_timestamp() {
|
||||
return Ok(timestamp.to_zoned(now.time_zone().clone()));
|
||||
}
|
||||
let base = now.datetime();
|
||||
let date = broken
|
||||
.to_date()
|
||||
.or_else(|_| {
|
||||
jiff::civil::Date::new(
|
||||
broken.year().unwrap_or(base.year()),
|
||||
broken.month().unwrap_or(base.month()),
|
||||
broken.day().unwrap_or(base.day()),
|
||||
)
|
||||
})
|
||||
.map_err(convert_error)?;
|
||||
let time = jiff::civil::Time::new(
|
||||
broken.hour().unwrap_or(base.hour()),
|
||||
broken.minute().unwrap_or(base.minute()),
|
||||
broken.second().unwrap_or(base.second()),
|
||||
broken.subsec_nanosecond().unwrap_or(base.subsec_nanosecond()),
|
||||
)
|
||||
.map_err(convert_error)?;
|
||||
date
|
||||
.to_datetime(time)
|
||||
.to_zoned(now.time_zone().clone())
|
||||
.map_err(convert_error)
|
||||
}
|
||||
|
||||
|
||||
/// Parsed `date` invocation.
|
||||
pub(crate) struct Date {
|
||||
@@ -1064,6 +1310,10 @@ impl From<std::io::Error> for DateError {
|
||||
impl Utility for Date {
|
||||
const NAME: &'static str = "date";
|
||||
|
||||
fn rewrite_argv(argv: Vec<OsString>) -> Result<Vec<OsString>, String> {
|
||||
Ok(rewrite_date_argv(argv))
|
||||
}
|
||||
|
||||
fn run(self, host: &mut Host) -> i32 {
|
||||
match date_main(host, &self.matches) {
|
||||
Ok(()) => host.exit_code(),
|
||||
@@ -1152,7 +1402,44 @@ fn locale_default_format(_locale: &str) -> Option<String> {
|
||||
|
||||
#[allow(clippy::cognitive_complexity)]
|
||||
fn date_main(host: &mut Host, matches: &ArgMatches) -> Result<(), DateError> {
|
||||
let date_source = if let Some(date_os) = matches.get_one::<OsString>(OPT_DATE) {
|
||||
let bsd_parse_only = matches.get_flag(OPT_BSD_PARSE_ONLY);
|
||||
let adjustments: Vec<BsdAdjustment> = matches
|
||||
.get_many::<String>(OPT_BSD_ADJUST)
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.map(|spec| {
|
||||
parse_bsd_adjustment(spec)
|
||||
.ok_or_else(|| DateError::new(1, format!("invalid adjustment: '{spec}'")))
|
||||
})
|
||||
.collect::<Result<_, _>>()?;
|
||||
|
||||
// Positional operands: at most one `+FORMAT`, plus (in the BSD `-j -f`
|
||||
// form) the date value to parse.
|
||||
let mut operands: Vec<&String> = matches
|
||||
.get_many::<String>(OPT_FORMAT)
|
||||
.map(Iterator::collect)
|
||||
.unwrap_or_default();
|
||||
|
||||
// BSD: with `-j`, `-f` is the strptime(3) input format for the date
|
||||
// operand rather than GNU's `--file=DATEFILE`.
|
||||
let strptime_format = if bsd_parse_only {
|
||||
matches.get_one::<String>(OPT_FILE)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
let strptime_value = if strptime_format.is_some() {
|
||||
if operands.first().is_some_and(|operand| !operand.starts_with('+')) {
|
||||
Some(operands.remove(0))
|
||||
} else {
|
||||
return Err(DateError::new(1, "'-j -f FORMAT' requires a date operand to parse"));
|
||||
}
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
let date_source = if let (Some(format), Some(value)) = (strptime_format, strptime_value) {
|
||||
DateSource::Strptime { format: format.clone(), value: value.clone() }
|
||||
} else if let Some(date_os) = matches.get_one::<OsString>(OPT_DATE) {
|
||||
// Convert OsString to String, handling invalid UTF-8 with GNU-compatible error
|
||||
let date = date_os.to_str().ok_or_else(|| {
|
||||
let bytes = date_os.as_encoded_bytes();
|
||||
@@ -1165,8 +1452,18 @@ fn date_main(host: &mut Host, matches: &ArgMatches) -> Result<(), DateError> {
|
||||
"-" => DateSource::Stdin,
|
||||
_ => DateSource::File(file.into()),
|
||||
}
|
||||
} else if let Some(file) = matches.get_one::<String>(OPT_REFERENCE) {
|
||||
DateSource::FileMtime(file.into())
|
||||
} else if let Some(reference) = matches.get_one::<String>(OPT_REFERENCE) {
|
||||
// `-r` doubles as GNU `--reference=FILE` and BSD `-r SECONDS`.
|
||||
// Precedence: an existing file always wins (GNU semantics are
|
||||
// primary); a purely numeric operand naming no existing file is
|
||||
// seconds since the epoch (BSD), i.e. GNU `-d @SECONDS`.
|
||||
let digits = reference.strip_prefix('-').unwrap_or(reference);
|
||||
let numeric = !digits.is_empty() && digits.bytes().all(|b| b.is_ascii_digit());
|
||||
if numeric && !host.resolve(Path::new(reference)).exists() {
|
||||
DateSource::Human(format!("@{reference}"))
|
||||
} else {
|
||||
DateSource::FileMtime(reference.into())
|
||||
}
|
||||
} else if matches.get_flag(OPT_RESOLUTION) {
|
||||
DateSource::Resolution
|
||||
} else {
|
||||
@@ -1174,18 +1471,15 @@ fn date_main(host: &mut Host, matches: &ArgMatches) -> Result<(), DateError> {
|
||||
};
|
||||
|
||||
// Check for extra operands (multiple positional arguments)
|
||||
if let Some(formats) = matches.get_many::<String>(OPT_FORMAT) {
|
||||
let format_args: Vec<&String> = formats.collect();
|
||||
if format_args.len() > 1 {
|
||||
return Err(DateError::new(1, format!("extra operand '{}'", format_args[1])));
|
||||
}
|
||||
if operands.len() > 1 {
|
||||
return Err(DateError::new(1, format!("extra operand '{}'", operands[1])));
|
||||
}
|
||||
|
||||
let format = if let Some(form) = matches.get_one::<String>(OPT_FORMAT) {
|
||||
let format = if let Some(form) = operands.first() {
|
||||
if !form.starts_with('+') {
|
||||
// if an optional Format String was found but the user has not provided an input
|
||||
// date GNU prints an invalid date Error
|
||||
if !matches!(date_source, DateSource::Human(_)) {
|
||||
if !matches!(date_source, DateSource::Human(_) | DateSource::Strptime { .. }) {
|
||||
return Err(DateError::new(1, format!("invalid date '{form}'")));
|
||||
}
|
||||
// If the user did provide an input date with the --date flag and the Format
|
||||
@@ -1198,8 +1492,7 @@ fn date_main(host: &mut Host, matches: &ArgMatches) -> Result<(), DateError> {
|
||||
),
|
||||
));
|
||||
}
|
||||
let form = form[1..].to_string();
|
||||
Format::Custom(form)
|
||||
Format::Custom(form[1..].to_string())
|
||||
} else if let Some(fmt) = matches
|
||||
.get_many::<String>(OPT_ISO_8601)
|
||||
.map(|mut iter| iter.next().unwrap_or(&DATE.to_string()).as_str().into())
|
||||
@@ -1427,6 +1720,11 @@ fn date_main(host: &mut Host, matches: &ArgMatches) -> Result<(), DateError> {
|
||||
let iter = std::iter::once(Ok(date));
|
||||
Box::new(iter)
|
||||
},
|
||||
DateSource::Strptime { format, value } => {
|
||||
let date = parse_bsd_strptime(format, value, &now)
|
||||
.map_err(|message| DateError::new(1, message))?;
|
||||
Box::new(std::iter::once(Ok(date)))
|
||||
},
|
||||
DateSource::Now => {
|
||||
let iter = std::iter::once(Ok(now.clone()));
|
||||
Box::new(iter)
|
||||
@@ -1445,6 +1743,18 @@ fn date_main(host: &mut Host, matches: &ArgMatches) -> Result<(), DateError> {
|
||||
}
|
||||
match date {
|
||||
Ok(date) => {
|
||||
// BSD `-v` adjustments apply to the base date in argv order.
|
||||
let date = if adjustments.is_empty() {
|
||||
date
|
||||
} else {
|
||||
match apply_bsd_adjustments(date, &adjustments) {
|
||||
Ok(date) => date,
|
||||
Err(message) => {
|
||||
let _ = stdout.flush();
|
||||
return Err(DateError::new(1, message));
|
||||
},
|
||||
}
|
||||
};
|
||||
let date = if settings.utc {
|
||||
date.with_time_zone(TimeZone::UTC)
|
||||
} else {
|
||||
@@ -1509,7 +1819,10 @@ fn uu_app() -> Command {
|
||||
.value_name("DATEFILE")
|
||||
.value_hint(clap::ValueHint::FilePath)
|
||||
.conflicts_with(OPT_DATE)
|
||||
.help("like --date; once for each line of DATEFILE"),
|
||||
.help(
|
||||
"like --date; once for each line of DATEFILE\n(BSD: with -j, the strptime(3) \
|
||||
input format for the date operand)",
|
||||
),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(OPT_ISO_8601)
|
||||
@@ -1517,7 +1830,12 @@ fn uu_app() -> Command {
|
||||
.long(OPT_ISO_8601)
|
||||
.value_name("FMT")
|
||||
.value_parser(ShortcutValueParser::new([DATE, HOURS, MINUTES, SECONDS, NS]))
|
||||
// The optional precision binds only when attached (`-Ihours`,
|
||||
// `--iso-8601=hours`): `rewrite_date_argv` normalizes every
|
||||
// spelling to the `=` form, so a following `+FORMAT` operand
|
||||
// is never consumed as the value (GNU getopt behavior).
|
||||
.num_args(0..=1)
|
||||
.require_equals(true)
|
||||
.default_missing_value(OPT_DATE)
|
||||
.help(
|
||||
"output date/time in ISO 8601 format.\nFMT='date' for date only (the \
|
||||
@@ -1567,8 +1885,12 @@ fn uu_app() -> Command {
|
||||
.long(OPT_REFERENCE)
|
||||
.value_name("FILE")
|
||||
.value_hint(clap::ValueHint::AnyPath)
|
||||
.allow_hyphen_values(true)
|
||||
.conflicts_with_all([OPT_DATE, OPT_FILE, OPT_RESOLUTION])
|
||||
.help("display the last modification time of FILE"),
|
||||
.help(
|
||||
"display the last modification time of FILE\n(BSD: when FILE is numeric and \
|
||||
no such file exists,\ndisplay the date at that many seconds since the epoch)",
|
||||
),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(OPT_SET)
|
||||
@@ -1588,6 +1910,26 @@ fn uu_app() -> Command {
|
||||
.help("print or set Coordinated Universal Time (UTC)")
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(OPT_BSD_PARSE_ONLY)
|
||||
.short('j')
|
||||
.help(
|
||||
"BSD compatibility: do not try to set the system clock;\nwith -f, parse the \
|
||||
date operand using the strptime(3)\nformat given to -f",
|
||||
)
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(OPT_BSD_ADJUST)
|
||||
.short('v')
|
||||
.value_name("[+|-]VAL[ymwdHMS]")
|
||||
.allow_hyphen_values(true)
|
||||
.action(ArgAction::Append)
|
||||
.help(
|
||||
"BSD compatibility: adjust ('+'/'-') or set (no sign) the\ndisplayed date; \
|
||||
may be given multiple times, applied in order",
|
||||
),
|
||||
)
|
||||
.arg(Arg::new(OPT_FORMAT).num_args(0..))
|
||||
}
|
||||
|
||||
@@ -1982,6 +2324,194 @@ mod tests {
|
||||
assert_eq!(strip_parenthesized_comments("a(b(c)d)e"), "ae");
|
||||
assert_eq!(strip_parenthesized_comments("a(b)c(d"), "ac");
|
||||
}
|
||||
|
||||
/// Defends: BSD `-r <epoch>` must print that instant, not fail trying to
|
||||
/// open `./<epoch>` as a reference file.
|
||||
#[test]
|
||||
fn bsd_reference_epoch_when_no_such_file() {
|
||||
let (code, capture) =
|
||||
run_util::<Date>(&["-u", "-r", "1700000000", "+%Y-%m-%dT%H:%M:%S"], "", "/");
|
||||
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "2023-11-14T22:13:20\n");
|
||||
assert_eq!(capture.err(), "");
|
||||
}
|
||||
|
||||
/// Defends: an existing file always wins over the BSD numeric-epoch
|
||||
/// reading of `-r` (GNU `--reference` semantics are primary).
|
||||
#[test]
|
||||
fn reference_prefers_existing_file_over_epoch() {
|
||||
let dir = std::env::temp_dir().join(format!("pi-date-r-{}", std::process::id()));
|
||||
std::fs::create_dir_all(&dir).unwrap();
|
||||
let file = dir.join("1700000000");
|
||||
std::fs::write(&file, b"x").unwrap();
|
||||
|
||||
let (code, capture) = run_util::<Date>(
|
||||
&["-u", "-r", "1700000000", "+%s"],
|
||||
"",
|
||||
dir.to_str().unwrap(),
|
||||
);
|
||||
let mtime = std::fs::metadata(&file).unwrap().modified().unwrap();
|
||||
let expected = Timestamp::try_from(mtime).unwrap().as_second();
|
||||
std::fs::remove_dir_all(&dir).unwrap();
|
||||
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), format!("{expected}\n"));
|
||||
}
|
||||
|
||||
/// Defends: BSD `-v` offsets (`+1d`, `-2m`) must parse and apply in argv
|
||||
/// order instead of being rejected as unknown options.
|
||||
#[test]
|
||||
fn bsd_adjustments_apply_in_order() {
|
||||
let (code, capture) = run_util::<Date>(
|
||||
&["-u", "-r", "1700000000", "-v+1d", "-v-2m", "+%F %T"],
|
||||
"",
|
||||
"/",
|
||||
);
|
||||
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "2023-09-15 22:13:20\n");
|
||||
assert_eq!(capture.err(), "");
|
||||
}
|
||||
|
||||
/// Defends: the unsigned `-v` form sets a field absolutely (`-v1d` = first
|
||||
/// of the month) rather than offsetting.
|
||||
#[test]
|
||||
fn bsd_adjustment_sets_fields_absolutely() {
|
||||
let (code, capture) = run_util::<Date>(
|
||||
&["-u", "-r", "1700000000", "-v1d", "-v5H", "+%F %T"],
|
||||
"",
|
||||
"/",
|
||||
);
|
||||
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "2023-11-01 05:13:20\n");
|
||||
assert_eq!(capture.err(), "");
|
||||
}
|
||||
|
||||
/// Defends: a malformed `-v` argument is a clean diagnostic, not a panic
|
||||
/// or a silently ignored adjustment.
|
||||
#[test]
|
||||
fn bsd_adjustment_rejects_unknown_unit() {
|
||||
let (code, capture) = run_util::<Date>(&["-v+1x", "+%F"], "", "/");
|
||||
|
||||
assert_eq!(code, 1);
|
||||
assert!(capture.err().contains("invalid adjustment"), "stderr: {}", capture.err());
|
||||
}
|
||||
|
||||
/// Defends: with `-j`, `-f` is the BSD strptime input format for the date
|
||||
/// operand, not GNU `--file=DATEFILE`.
|
||||
#[test]
|
||||
fn bsd_j_f_parses_with_strptime_format() {
|
||||
let (code, capture) = run_util::<Date>(
|
||||
&["-u", "-j", "-f", "%Y-%m-%d %H:%M:%S", "2026-01-01 00:00:00", "+%s"],
|
||||
"",
|
||||
"/",
|
||||
);
|
||||
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "1767225600\n");
|
||||
assert_eq!(capture.err(), "");
|
||||
}
|
||||
|
||||
/// Defends: fields missing from the `-j -f` format are seeded from "now"
|
||||
/// (BSD strptime semantics), so a date-only format keeps the given date.
|
||||
#[test]
|
||||
fn bsd_j_f_fills_missing_fields_from_now() {
|
||||
let (code, capture) =
|
||||
run_util::<Date>(&["-u", "-j", "-f", "%Y-%m-%d", "2026-01-01", "+%F"], "", "/");
|
||||
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "2026-01-01\n");
|
||||
assert_eq!(capture.err(), "");
|
||||
}
|
||||
|
||||
/// Defends: `-j -f` without a date operand is a diagnostic, not a silent
|
||||
/// fallback to GNU `--file` behavior.
|
||||
#[test]
|
||||
fn bsd_j_f_requires_date_operand() {
|
||||
let (code, capture) = run_util::<Date>(&["-j", "-f", "%Y", "+%F"], "", "/");
|
||||
|
||||
assert_eq!(code, 1);
|
||||
assert!(
|
||||
capture.err().contains("requires a date operand"),
|
||||
"stderr: {}",
|
||||
capture.err()
|
||||
);
|
||||
}
|
||||
|
||||
/// Defends: bare `-j` parses as a no-op (never sets the clock) instead of
|
||||
/// being rejected, and `-f` keeps GNU file semantics without `-j`.
|
||||
#[test]
|
||||
fn bsd_j_alone_is_a_no_op() {
|
||||
let (code, capture) = run_util::<Date>(&["-u", "-j", "-r", "0", "+%F"], "", "/");
|
||||
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "1970-01-01\n");
|
||||
assert_eq!(capture.err(), "");
|
||||
}
|
||||
|
||||
/// Defends: `date -I +%s` must treat `+%s` as the output format instead of
|
||||
/// greedily consuming it as the ISO precision value.
|
||||
#[test]
|
||||
fn iso_flag_does_not_consume_format_operand() {
|
||||
let (code, capture) = run_util::<Date>(&["-u", "-d", "@0", "-I", "+%s"], "", "/");
|
||||
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "0\n");
|
||||
assert_eq!(capture.err(), "");
|
||||
}
|
||||
|
||||
/// Defends: bare `-I` still defaults to date precision after the
|
||||
/// non-greedy rewrite.
|
||||
#[test]
|
||||
fn iso_flag_defaults_to_date_precision() {
|
||||
let (code, capture) = run_util::<Date>(&["-u", "-d", "@0", "-I"], "", "/");
|
||||
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "1970-01-01\n");
|
||||
assert_eq!(capture.err(), "");
|
||||
}
|
||||
|
||||
/// Defends: the attached form `-Ihours` keeps binding the precision.
|
||||
#[test]
|
||||
fn iso_flag_accepts_attached_precision() {
|
||||
let (code, capture) = run_util::<Date>(&["-u", "-d", "@0", "-Ihours"], "", "/");
|
||||
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "1970-01-01T00+00:00\n");
|
||||
assert_eq!(capture.err(), "");
|
||||
}
|
||||
|
||||
/// Defends: the argv rewrite only touches genuine `-I`/`--iso-8601`
|
||||
/// options — never option values, post-`--` operands, or other flags.
|
||||
#[test]
|
||||
fn rewrites_iso_argv_forms_conservatively() {
|
||||
let argv = |args: &[&str]| -> Vec<OsString> { args.iter().map(OsString::from).collect() };
|
||||
|
||||
assert_eq!(
|
||||
rewrite_date_argv(argv(&["date", "-I", "+%s"])),
|
||||
argv(&["date", "--iso-8601=date", "+%s"])
|
||||
);
|
||||
assert_eq!(rewrite_date_argv(argv(&["date", "-Ihours"])), argv(&["date", "--iso-8601=hours"]));
|
||||
assert_eq!(
|
||||
rewrite_date_argv(argv(&["date", "--iso-8601", "+%s"])),
|
||||
argv(&["date", "--iso-8601=date", "+%s"])
|
||||
);
|
||||
assert_eq!(
|
||||
rewrite_date_argv(argv(&["date", "--iso", "-u"])),
|
||||
argv(&["date", "--iso-8601=date", "-u"])
|
||||
);
|
||||
// `-I` as the value of another option is untouched.
|
||||
assert_eq!(rewrite_date_argv(argv(&["date", "-d", "-I"])), argv(&["date", "-d", "-I"]));
|
||||
// Everything after `--` is an operand.
|
||||
assert_eq!(rewrite_date_argv(argv(&["date", "--", "-I"])), argv(&["date", "--", "-I"]));
|
||||
// Clustered flags before `-I` are preserved.
|
||||
assert_eq!(
|
||||
rewrite_date_argv(argv(&["date", "-uI"])),
|
||||
argv(&["date", "-u", "--iso-8601=date"])
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(not(any(unix, windows)))]
|
||||
|
||||
+869
-127
File diff suppressed because it is too large
Load Diff
@@ -168,6 +168,7 @@ fn command() -> Command {
|
||||
Arg::new(ARG_QUERY)
|
||||
.value_name("NAME-OR-NUMBER")
|
||||
.num_args(0..)
|
||||
.allow_hyphen_values(true)
|
||||
.action(ArgAction::Append),
|
||||
)
|
||||
}
|
||||
@@ -189,13 +190,21 @@ fn print_entry(host: &mut Host, name: &str, number: i32) {
|
||||
/// Looks up one name or number argument; returns false on failure.
|
||||
fn lookup(host: &mut Host, arg: &str) -> bool {
|
||||
if let Ok(number) = arg.parse::<i32>() {
|
||||
// Reverse lookup: first-listed name for the number is canonical.
|
||||
match ERRNOS.iter().find(|(_, value)| *value == number) {
|
||||
// Kernel convention returns errors as negative errno values; resolve
|
||||
// `-2` the same as `2`. Reverse lookup: first-listed name for the
|
||||
// number is canonical.
|
||||
match number
|
||||
.checked_abs()
|
||||
.and_then(|number| ERRNOS.iter().find(|(_, value)| *value == number))
|
||||
{
|
||||
Some((name, value)) => {
|
||||
print_entry(host, name, *value);
|
||||
true
|
||||
},
|
||||
None => false,
|
||||
None => {
|
||||
let _ = writeln!(host.stderr, "errno: unknown errno {arg}");
|
||||
false
|
||||
},
|
||||
}
|
||||
} else if let Some((name, value)) = ERRNOS
|
||||
.iter()
|
||||
@@ -277,11 +286,44 @@ mod tests {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unknown_number_fails_silently() {
|
||||
fn unknown_number_fails_with_stderr() {
|
||||
// Failure mode: unknown numeric lookups exiting 1 with no diagnostic.
|
||||
let (code, stdout, stderr) = run_errno(&["99999"]);
|
||||
assert_eq!(code, 1);
|
||||
assert!(stdout.is_empty());
|
||||
assert!(stderr.is_empty());
|
||||
assert!(stderr.contains("unknown errno 99999"), "stderr: {stderr:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn negative_number_resolves_by_absolute_value() {
|
||||
// Failure mode: clap rejecting `errno -2` as an unknown option instead
|
||||
// of resolving the kernel-style negative errno.
|
||||
let number = format!("-{}", libc::ENOENT);
|
||||
let (code, stdout, stderr) = run_errno(&[&number]);
|
||||
assert_eq!(code, 0);
|
||||
assert!(stdout.starts_with(&format!("ENOENT {} ", libc::ENOENT)), "stdout: {stdout:?}");
|
||||
assert!(stderr.is_empty(), "stderr: {stderr:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unknown_negative_number_fails_with_stderr() {
|
||||
let (code, stdout, stderr) = run_errno(&["-99999"]);
|
||||
assert_eq!(code, 1);
|
||||
assert!(stdout.is_empty());
|
||||
assert!(stderr.contains("unknown errno -99999"), "stderr: {stderr:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn short_flags_still_win_over_hyphen_operands() {
|
||||
// Failure mode: allow_hyphen_values swallowing `-l` as a query.
|
||||
let (code, stdout, _) = run_errno(&["-l"]);
|
||||
assert_eq!(code, 0);
|
||||
assert!(
|
||||
stdout
|
||||
.lines()
|
||||
.any(|line| line.starts_with(&format!("ENOENT {} ", libc::ENOENT))),
|
||||
"-l no longer lists: {stdout:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -204,6 +204,8 @@ pub fn utility_builtins<SE: brush_core::ShellExtensions>()
|
||||
m.push(("basename", basename::basename_builtin::<SE>()));
|
||||
#[cfg(feature = "util.cat")]
|
||||
m.push(("cat", cat::cat_builtin::<SE>()));
|
||||
#[cfg(feature = "util.cksum")]
|
||||
m.push(("cksum", cksum::cksum_builtin::<SE>()));
|
||||
#[cfg(feature = "util.cmp")]
|
||||
m.push(("cmp", cmp::cmp_builtin::<SE>()));
|
||||
#[cfg(feature = "util.comm")]
|
||||
|
||||
+342
-47
@@ -1847,7 +1847,9 @@ pub mod matchers {
|
||||
|
||||
match chars.next() {
|
||||
Some('-') => (ComparisonType::AtLeast, chars.as_str()),
|
||||
Some('/') => (ComparisonType::AnyOf, chars.as_str()),
|
||||
// GNU spells "any of these bits" as /mode; BSD find spells
|
||||
// it +mode. Accept both.
|
||||
Some('/') | Some('+') => (ComparisonType::AnyOf, chars.as_str()),
|
||||
_ => (ComparisonType::Exact, pattern),
|
||||
}
|
||||
}
|
||||
@@ -1882,7 +1884,7 @@ pub mod matchers {
|
||||
pub fn new(pattern: &str) -> Result<Self, Box<dyn Error>> {
|
||||
let (comparison_type, pattern) = parsing::split_comparison_type(pattern);
|
||||
let file_pattern = parsing::parse_mode(pattern, false)?;
|
||||
let dir_pattern = parsing::parse_mode(pattern, false)?;
|
||||
let dir_pattern = parsing::parse_mode(pattern, true)?;
|
||||
Ok(Self { comparison_type, file_pattern, dir_pattern })
|
||||
}
|
||||
|
||||
@@ -2686,7 +2688,7 @@ pub mod matchers {
|
||||
|
||||
use std::{error::Error, fmt, str::FromStr};
|
||||
|
||||
use onig::{Regex, RegexOptions, Syntax};
|
||||
use onig::{Regex, RegexOptions, SearchOptions, Syntax};
|
||||
|
||||
use super::{Matcher, MatcherIO, WalkEntry};
|
||||
|
||||
@@ -2783,9 +2785,15 @@ pub mod matchers {
|
||||
|
||||
impl Matcher for RegexMatcher {
|
||||
fn matches(&self, file_info: &WalkEntry, _: &mut MatcherIO) -> bool {
|
||||
let path = file_info.display_path().to_string_lossy();
|
||||
// `-regex` must match the WHOLE path (POSIX/GNU/BSD), not a
|
||||
// substring: anchor the match at the start of the path and
|
||||
// require it to end at the end of the path (backtracking
|
||||
// retries alternatives that stop short).
|
||||
self
|
||||
.regex
|
||||
.is_match(file_info.display_path().to_string_lossy().as_ref())
|
||||
.match_with_options(path.as_ref(), 0, SearchOptions::SEARCH_OPTION_WHOLE_STRING, None)
|
||||
.is_some()
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -2859,6 +2867,8 @@ pub mod matchers {
|
||||
KibiByte,
|
||||
MebiByte,
|
||||
GibiByte,
|
||||
TebiByte,
|
||||
PebiByte,
|
||||
}
|
||||
|
||||
impl FromStr for Unit {
|
||||
@@ -2872,10 +2882,12 @@ pub mod matchers {
|
||||
"k" => Self::KibiByte,
|
||||
"M" => Self::MebiByte,
|
||||
"G" => Self::GibiByte,
|
||||
"T" => Self::TebiByte,
|
||||
"P" => Self::PebiByte,
|
||||
_ => {
|
||||
return Err(From::from(format!(
|
||||
"Invalid suffix {s} for -size. Only allowed values are <nothing>, b, c, w, k, M or \
|
||||
G"
|
||||
"Invalid suffix {s} for -size. Only allowed values are <nothing>, b, c, w, k, M, G, \
|
||||
T or P"
|
||||
)));
|
||||
},
|
||||
})
|
||||
@@ -2894,6 +2906,8 @@ pub mod matchers {
|
||||
Unit::KibiByte => 10,
|
||||
Unit::MebiByte => 20,
|
||||
Unit::GibiByte => 30,
|
||||
Unit::TebiByte => 40,
|
||||
Unit::PebiByte => 50,
|
||||
};
|
||||
// Skip pointless arithmetic.
|
||||
if bits_to_shift == 0 {
|
||||
@@ -3111,13 +3125,12 @@ pub mod matchers {
|
||||
}
|
||||
}
|
||||
|
||||
/// This matcher checks whether the file is newer than the file time of any
|
||||
/// combination of two comparison types from the target file's
|
||||
/// `NewerOptionType`.
|
||||
/// This matcher checks whether the X timestamp of the file being
|
||||
/// considered is newer than the Y timestamp of the reference file,
|
||||
/// captured once when the matcher is built (`-newerXY reference`).
|
||||
pub struct NewerOptionMatcher {
|
||||
x_option: NewerOptionType,
|
||||
y_option: NewerOptionType,
|
||||
given_modification_time: SystemTime,
|
||||
x_option: NewerOptionType,
|
||||
reference_time: SystemTime,
|
||||
}
|
||||
|
||||
impl NewerOptionMatcher {
|
||||
@@ -3125,21 +3138,19 @@ pub mod matchers {
|
||||
let metadata = fs::metadata(host.resolve(path_to_file))?;
|
||||
let x_option = NewerOptionType::from_str(x_option);
|
||||
let y_option = NewerOptionType::from_str(y_option);
|
||||
Ok(Self { x_option, y_option, given_modification_time: metadata.modified()? })
|
||||
let reference_time = y_option.get_file_time(&metadata)?;
|
||||
Ok(Self { x_option, reference_time })
|
||||
}
|
||||
|
||||
fn matches_impl(&self, file_info: &WalkEntry) -> Result<bool, Box<dyn Error>> {
|
||||
let x_option_time = self.x_option.get_file_time(file_info.metadata()?)?;
|
||||
let y_option_time = self.y_option.get_file_time(file_info.metadata()?)?;
|
||||
|
||||
// duration_since returns Err when x_option_time is strictly
|
||||
// newer than the reference time.
|
||||
Ok(self
|
||||
.given_modification_time
|
||||
.reference_time
|
||||
.duration_since(x_option_time)
|
||||
.is_err()
|
||||
&& self
|
||||
.given_modification_time
|
||||
.duration_since(y_option_time)
|
||||
.is_err())
|
||||
.is_err())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3149,9 +3160,8 @@ pub mod matchers {
|
||||
Err(e) => {
|
||||
writeln!(
|
||||
&mut matcher_io.host().stderr,
|
||||
"Error getting {:?} and {:?} time for {}: {}",
|
||||
"Error getting {:?} time for {}: {}",
|
||||
self.x_option,
|
||||
self.y_option,
|
||||
file_info.path().to_string_lossy(),
|
||||
e
|
||||
)
|
||||
@@ -3390,12 +3400,16 @@ pub mod matchers {
|
||||
|
||||
use super::{FileType, Follow, Matcher, MatcherIO, WalkEntry};
|
||||
|
||||
/// This matcher checks the type of the file.
|
||||
/// This matcher checks the type of the file against a list of accepted
|
||||
/// types (GNU findutils 4.9+ accepts comma-separated lists, e.g. `f,d`).
|
||||
pub struct TypeMatcher {
|
||||
file_type: FileType,
|
||||
file_types: Vec<FileType>,
|
||||
}
|
||||
|
||||
fn parse(type_string: &str) -> Result<FileType, Box<dyn Error>> {
|
||||
/// Parses one type letter. `Ok(None)` means the letter is accepted but
|
||||
/// can never match here (BSD `w` — whiteouts don't exist on this
|
||||
/// platform's walk results).
|
||||
fn parse_one(type_string: &str) -> Result<Option<FileType>, Box<dyn Error>> {
|
||||
let file_type = match type_string {
|
||||
"f" => FileType::Regular,
|
||||
"d" => FileType::Directory,
|
||||
@@ -3404,35 +3418,50 @@ pub mod matchers {
|
||||
"c" => FileType::CharDevice,
|
||||
"p" => FileType::Fifo, // named pipe (FIFO)
|
||||
"s" => FileType::Socket,
|
||||
// w: whiteout (BSD); accepted but never produced by the walker
|
||||
"w" => return Ok(None),
|
||||
// D: door (Solaris)
|
||||
"D" => return Err(From::from(format!("Type argument {type_string} not supported yet"))),
|
||||
_ => return Err(From::from(format!("Unrecognised type argument {type_string}"))),
|
||||
};
|
||||
Ok(file_type)
|
||||
Ok(Some(file_type))
|
||||
}
|
||||
|
||||
fn parse(type_string: &str) -> Result<Vec<FileType>, Box<dyn Error>> {
|
||||
let mut file_types = Vec::new();
|
||||
for part in type_string.split(',') {
|
||||
if part.is_empty() {
|
||||
return Err(From::from(format!("Unrecognised type argument {type_string}")));
|
||||
}
|
||||
if let Some(file_type) = parse_one(part)? {
|
||||
file_types.push(file_type);
|
||||
}
|
||||
}
|
||||
Ok(file_types)
|
||||
}
|
||||
|
||||
impl TypeMatcher {
|
||||
pub fn new(type_string: &str) -> Result<Self, Box<dyn Error>> {
|
||||
let file_type = parse(type_string)?;
|
||||
Ok(Self { file_type })
|
||||
let file_types = parse(type_string)?;
|
||||
Ok(Self { file_types })
|
||||
}
|
||||
}
|
||||
|
||||
impl Matcher for TypeMatcher {
|
||||
fn matches(&self, file_info: &WalkEntry, _: &mut MatcherIO) -> bool {
|
||||
file_info.file_type() == self.file_type
|
||||
self.file_types.contains(&file_info.file_type())
|
||||
}
|
||||
}
|
||||
|
||||
/// Like [TypeMatcher], but toggles whether symlinks are followed.
|
||||
pub struct XtypeMatcher {
|
||||
file_type: FileType,
|
||||
file_types: Vec<FileType>,
|
||||
}
|
||||
|
||||
impl XtypeMatcher {
|
||||
pub fn new(type_string: &str) -> Result<Self, Box<dyn Error>> {
|
||||
let file_type = parse(type_string)?;
|
||||
Ok(Self { file_type })
|
||||
let file_types = parse(type_string)?;
|
||||
Ok(Self { file_types })
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3450,9 +3479,9 @@ pub mod matchers {
|
||||
.map(FileType::from);
|
||||
|
||||
match file_type {
|
||||
Ok(file_type) if file_type == self.file_type => true,
|
||||
Ok(file_type) if self.file_types.contains(&file_type) => true,
|
||||
// Since GNU find 4.10, ELOOP will match -xtype l
|
||||
Err(e) if self.file_type.is_symlink() && e.is_loop() => true,
|
||||
Err(e) if self.file_types.iter().any(|t| t.is_symlink()) && e.is_loop() => true,
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
@@ -3558,7 +3587,7 @@ pub mod matchers {
|
||||
};
|
||||
|
||||
use ::regex::Regex;
|
||||
use chrono::{DateTime, Datelike, NaiveDateTime, Utc};
|
||||
use chrono::{DateTime, Datelike, Local, NaiveDate, NaiveDateTime, TimeZone, Utc};
|
||||
pub use entry::{FileType, WalkEntry, WalkError};
|
||||
use fs::FileSystemMatcher;
|
||||
use ls::Ls;
|
||||
@@ -3872,13 +3901,40 @@ pub mod matchers {
|
||||
)))
|
||||
}
|
||||
|
||||
/// This is a function that converts a specific string format into a timestamp.
|
||||
/// It allows converting a time string of
|
||||
/// "(week abbreviation) (date), (year) (time)" to a Unix timestamp.
|
||||
/// such as: "jan 01, 2025 00:00:01" -> 1735689601000
|
||||
/// When (time) is not provided, it will be automatically filled in as 00:00:00
|
||||
/// such as: "jan 01, 2025" = "jan 01, 2025 00:00:00" -> 1735689600000
|
||||
/// Converts a `-newerXt`-style reference time string into a Unix timestamp
|
||||
/// (milliseconds).
|
||||
///
|
||||
/// Accepts, in order:
|
||||
/// - `@N[.N]` seconds since the epoch (GNU extension)
|
||||
/// - RFC 3339 datetimes with an explicit offset, e.g.
|
||||
/// "2026-01-01T00:00:00Z"
|
||||
/// - ISO-style naive dates/datetimes ("2026-01-01",
|
||||
/// "2026-01-01 12:30[:45]", with ` ` or `T` separators), interpreted in
|
||||
/// local time like GNU find
|
||||
/// - "(month abbreviation) (date), (year) (time)" strings, e.g.
|
||||
/// "jan 01, 2025 00:00:01" (time defaults to 00:00:00)
|
||||
fn parse_date_str_to_timestamps(date_str: &str) -> Option<i64> {
|
||||
if let Some(epoch) = date_str.strip_prefix('@')
|
||||
&& let Ok(seconds) = epoch.parse::<f64>()
|
||||
{
|
||||
return Some((seconds * 1000.0) as i64);
|
||||
}
|
||||
|
||||
if let Ok(datetime) = DateTime::parse_from_rfc3339(date_str) {
|
||||
return Some(datetime.timestamp_millis());
|
||||
}
|
||||
|
||||
let naive = NaiveDate::parse_from_str(date_str, "%Y-%m-%d")
|
||||
.ok()
|
||||
.and_then(|date| date.and_hms_opt(0, 0, 0))
|
||||
.or_else(|| NaiveDateTime::parse_from_str(date_str, "%Y-%m-%d %H:%M:%S").ok())
|
||||
.or_else(|| NaiveDateTime::parse_from_str(date_str, "%Y-%m-%dT%H:%M:%S").ok())
|
||||
.or_else(|| NaiveDateTime::parse_from_str(date_str, "%Y-%m-%d %H:%M").ok())
|
||||
.or_else(|| NaiveDateTime::parse_from_str(date_str, "%Y-%m-%dT%H:%M").ok());
|
||||
if let Some(naive) = naive {
|
||||
return Some(Local.from_local_datetime(&naive).earliest()?.timestamp_millis());
|
||||
}
|
||||
|
||||
let regex_pattern =
|
||||
r"^(?P<month_day>\w{3} \d{2})?(?:, (?P<year>\d{4}))?(?: (?P<time>\d{2}:\d{2}:\d{2}))?$";
|
||||
let re = Regex::new(regex_pattern);
|
||||
@@ -4602,6 +4658,7 @@ fn parse_args(args: &[&str], host: &mut Host) -> Result<ParsedInfo, Box<dyn Erro
|
||||
let mut paths = vec![];
|
||||
let mut i = 0;
|
||||
let mut config = Config::default();
|
||||
let mut extended_regex = false;
|
||||
|
||||
while i < args.len() {
|
||||
match args[i] {
|
||||
@@ -4611,6 +4668,12 @@ fn parse_args(args: &[&str], host: &mut Host) -> Result<ParsedInfo, Box<dyn Erro
|
||||
"-H" => config.follow = Follow::Roots,
|
||||
"-L" => config.follow = Follow::Always,
|
||||
"-P" => config.follow = Follow::Never,
|
||||
// BSD find leading flags (macOS muscle memory).
|
||||
"-E" => extended_regex = true,
|
||||
// -x is the BSD spelling of -xdev.
|
||||
"-x" => config.same_file_system = true,
|
||||
// -s sorts output lexicographically.
|
||||
"-s" => config.sorted_output = true,
|
||||
"--" => {
|
||||
// End of flags
|
||||
i += 1;
|
||||
@@ -4634,7 +4697,16 @@ fn parse_args(args: &[&str], host: &mut Host) -> Result<ParsedInfo, Box<dyn Erro
|
||||
if i == paths_start {
|
||||
paths.push(".".to_string());
|
||||
}
|
||||
let matcher = matchers::build_top_level_matcher(&args[i..], &mut config, host)?;
|
||||
let matcher = if extended_regex {
|
||||
// BSD -E selects POSIX extended regular expressions; GNU spells that
|
||||
// -regextype posix-extended, which must precede any -regex/-iregex.
|
||||
let mut expression = Vec::with_capacity(args.len() - i + 2);
|
||||
expression.extend(["-regextype", "posix-extended"]);
|
||||
expression.extend_from_slice(&args[i..]);
|
||||
matchers::build_top_level_matcher(&expression, &mut config, host)?
|
||||
} else {
|
||||
matchers::build_top_level_matcher(&args[i..], &mut config, host)?
|
||||
};
|
||||
if let Some(new_paths) = &config.new_paths {
|
||||
if paths.len() == 1 && paths[0] == "." {
|
||||
paths = new_paths.to_vec();
|
||||
@@ -4878,9 +4950,9 @@ Early alpha implementation. Currently the only expressions supported are
|
||||
-files0-from
|
||||
-regex pattern
|
||||
-iregex pattern
|
||||
-type type_char
|
||||
currently type_char can only be f (for file) or d (for directory)
|
||||
-size [+-]N[bcwkMG]
|
||||
-type type_char[,type_char...]
|
||||
type_char is one of f d l b c p s w
|
||||
-size [+-]N[bcwkMGTP]
|
||||
-delete
|
||||
-prune
|
||||
-not
|
||||
@@ -4897,7 +4969,7 @@ Early alpha implementation. Currently the only expressions supported are
|
||||
-ctime [+-]N
|
||||
-atime [+-]N
|
||||
-mtime [+-]N
|
||||
-perm [-/]{{octal|u=rwx,go=w}}
|
||||
-perm [-/+]{{octal|u=rwx,go=w}}
|
||||
-newer path_to_file
|
||||
-exec[dir] executable [args] [{{}}] [more args] ;
|
||||
-sorted
|
||||
@@ -4940,7 +5012,7 @@ fn rewrite_bsd_invocation(args: &[&str], host: &mut Host) -> Option<Vec<String>>
|
||||
let mut i = 0;
|
||||
while i < rewritten.len() {
|
||||
match rewritten[i].as_str() {
|
||||
"-O0" | "-O1" | "-O2" | "-O3" | "-H" | "-L" | "-P" => i += 1,
|
||||
"-O0" | "-O1" | "-O2" | "-O3" | "-H" | "-L" | "-P" | "-x" | "-s" => i += 1,
|
||||
"--" => {
|
||||
i += 1;
|
||||
break;
|
||||
@@ -5118,4 +5190,227 @@ mod tests {
|
||||
assert_eq!(capture.err(), "");
|
||||
assert_eq!(capture.out(), "sub\n");
|
||||
}
|
||||
|
||||
/// Failure mode: `-newerXY ref` compared both X and Y timestamps of the
|
||||
/// CANDIDATE against the reference's mtime, instead of comparing the
|
||||
/// candidate's X timestamp against the reference's Y timestamp.
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn newer_xy_compares_candidate_x_against_reference_y() {
|
||||
use std::{
|
||||
fs::FileTimes,
|
||||
time::{Duration, SystemTime},
|
||||
};
|
||||
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let root = fs::canonicalize(dir.path()).unwrap();
|
||||
let now = SystemTime::now();
|
||||
let old = now - Duration::from_secs(2000);
|
||||
let mid = now - Duration::from_secs(1000);
|
||||
|
||||
let write_with_times = |name: &str, accessed: SystemTime, modified: SystemTime| {
|
||||
let path = root.join(name);
|
||||
fs::write(&path, b"x").unwrap();
|
||||
let file = fs::File::options().write(true).open(&path).unwrap();
|
||||
file
|
||||
.set_times(FileTimes::new().set_accessed(accessed).set_modified(modified))
|
||||
.unwrap();
|
||||
};
|
||||
|
||||
write_with_times("ref", mid, mid);
|
||||
// atime newer than ref's mtime, but mtime older: -neweram must match.
|
||||
// (The old code also demanded a newer mtime and rejected this file.)
|
||||
write_with_times("hit", now, old);
|
||||
// atime older than ref's mtime: -neweram must not match.
|
||||
write_with_times("miss", old, now);
|
||||
|
||||
let (code, capture) = run(
|
||||
&root,
|
||||
&[
|
||||
root.display().to_string(),
|
||||
"-type".into(),
|
||||
"f".into(),
|
||||
"-neweram".into(),
|
||||
"ref".into(),
|
||||
],
|
||||
);
|
||||
assert_eq!(code, 0, "stderr: {}", capture.err());
|
||||
assert_eq!(capture.err(), "");
|
||||
assert_eq!(capture.out(), format!("{}\n", root.join("hit").display()));
|
||||
}
|
||||
|
||||
/// Failure mode: `-newermt` rejected ISO dates like `2026-01-01` with
|
||||
/// "cannot figure out how to interpret ... as a date or time".
|
||||
#[test]
|
||||
fn newermt_accepts_iso_dates() {
|
||||
let (_dir, root) = fixture();
|
||||
let (code, capture) = run(
|
||||
&root,
|
||||
&[
|
||||
root.display().to_string(),
|
||||
"-name".into(),
|
||||
"a.txt".into(),
|
||||
"-newermt".into(),
|
||||
"2000-01-01".into(),
|
||||
],
|
||||
);
|
||||
assert_eq!(code, 0, "stderr: {}", capture.err());
|
||||
assert_eq!(capture.err(), "");
|
||||
assert_eq!(capture.out(), format!("{}\n", root.join("a.txt").display()));
|
||||
|
||||
let (code, capture) = run(
|
||||
&root,
|
||||
&[
|
||||
root.display().to_string(),
|
||||
"-newermt".into(),
|
||||
"3000-01-01".into(),
|
||||
],
|
||||
);
|
||||
assert_eq!(code, 0, "stderr: {}", capture.err());
|
||||
assert_eq!(capture.err(), "");
|
||||
assert_eq!(capture.out(), "");
|
||||
}
|
||||
|
||||
/// Failure mode: BSD `-perm +mode` (any of the bits set) was parsed as an
|
||||
/// exact-mode pattern and failed with a parse error.
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn perm_plus_mode_matches_any_set_bits() {
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
|
||||
let (_dir, root) = fixture();
|
||||
fs::set_permissions(root.join("a.txt"), fs::Permissions::from_mode(0o755)).unwrap();
|
||||
fs::set_permissions(root.join("b.md"), fs::Permissions::from_mode(0o644)).unwrap();
|
||||
fs::set_permissions(root.join("c.rs"), fs::Permissions::from_mode(0o600)).unwrap();
|
||||
let (code, capture) = run(
|
||||
&root,
|
||||
&[
|
||||
root.display().to_string(),
|
||||
"-type".into(),
|
||||
"f".into(),
|
||||
"-perm".into(),
|
||||
"+111".into(),
|
||||
],
|
||||
);
|
||||
assert_eq!(code, 0, "stderr: {}", capture.err());
|
||||
assert_eq!(capture.err(), "");
|
||||
assert_eq!(capture.out(), format!("{}\n", root.join("a.txt").display()));
|
||||
}
|
||||
|
||||
/// Failure mode: GNU `-type f,d` lists were rejected with "Unrecognised
|
||||
/// type argument f,d".
|
||||
#[test]
|
||||
fn type_accepts_comma_separated_list() {
|
||||
let (_dir, root) = fixture();
|
||||
fs::create_dir(root.join("sub")).unwrap();
|
||||
let (code, capture) = run(&root, &[root.display().to_string(), "-type".into(), "f,d".into()]);
|
||||
assert_eq!(code, 0, "stderr: {}", capture.err());
|
||||
assert_eq!(capture.err(), "");
|
||||
let mut matches: Vec<PathBuf> = capture.out().lines().map(PathBuf::from).collect();
|
||||
matches.sort();
|
||||
assert_eq!(matches, vec![
|
||||
root.clone(),
|
||||
root.join("a.txt"),
|
||||
root.join("b.md"),
|
||||
root.join("c.rs"),
|
||||
root.join("sub"),
|
||||
]);
|
||||
}
|
||||
|
||||
/// Failure mode: BSD `-type w` (whiteout) errored instead of parsing and
|
||||
/// matching nothing.
|
||||
#[test]
|
||||
fn type_w_parses_and_matches_nothing() {
|
||||
let (_dir, root) = fixture();
|
||||
let (code, capture) = run(&root, &[root.display().to_string(), "-type".into(), "w".into()]);
|
||||
assert_eq!(code, 0, "stderr: {}", capture.err());
|
||||
assert_eq!(capture.err(), "");
|
||||
assert_eq!(capture.out(), "");
|
||||
}
|
||||
|
||||
/// Failure mode: `-regex` matched substrings of the path instead of
|
||||
/// requiring the pattern to span the whole path.
|
||||
#[test]
|
||||
fn regex_matches_whole_path_only() {
|
||||
let (_dir, root) = fixture();
|
||||
let (code, capture) =
|
||||
run(&root, &[root.display().to_string(), "-regex".into(), r".*\.rs".into()]);
|
||||
assert_eq!(code, 0, "stderr: {}", capture.err());
|
||||
assert_eq!(capture.out(), format!("{}\n", root.join("c.rs").display()));
|
||||
|
||||
let (code, capture) =
|
||||
run(&root, &[root.display().to_string(), "-regex".into(), r"c\.rs".into()]);
|
||||
assert_eq!(code, 0, "stderr: {}", capture.err());
|
||||
assert_eq!(capture.out(), "");
|
||||
}
|
||||
|
||||
/// Failure mode: an alternation whose shorter branch matches a path prefix
|
||||
/// would win and the full-path match was missed (no backtracking retry).
|
||||
#[test]
|
||||
fn regex_full_match_prefers_longest_alternative() {
|
||||
let (_dir, root) = fixture();
|
||||
let (code, capture) = run(
|
||||
&root,
|
||||
&[
|
||||
root.display().to_string(),
|
||||
"-regextype".into(),
|
||||
"posix-extended".into(),
|
||||
"-regex".into(),
|
||||
r".*/c|.*/c\.rs".into(),
|
||||
],
|
||||
);
|
||||
assert_eq!(code, 0, "stderr: {}", capture.err());
|
||||
assert_eq!(capture.err(), "");
|
||||
assert_eq!(capture.out(), format!("{}\n", root.join("c.rs").display()));
|
||||
}
|
||||
|
||||
/// Failure mode: `-size` rejected the `T` and `P` suffixes accepted by
|
||||
/// modern GNU and BSD find.
|
||||
#[test]
|
||||
fn size_accepts_t_and_p_suffixes() {
|
||||
let (_dir, root) = fixture();
|
||||
let (code, capture) = run(
|
||||
&root,
|
||||
&[
|
||||
root.display().to_string(),
|
||||
"-type".into(),
|
||||
"f".into(),
|
||||
"-size".into(),
|
||||
"-2T".into(),
|
||||
],
|
||||
);
|
||||
assert_eq!(code, 0, "stderr: {}", capture.err());
|
||||
assert_eq!(capture.err(), "");
|
||||
let mut matches: Vec<PathBuf> = capture.out().lines().map(PathBuf::from).collect();
|
||||
matches.sort();
|
||||
assert_eq!(matches, vec![root.join("a.txt"), root.join("b.md"), root.join("c.rs")]);
|
||||
|
||||
let (code, capture) =
|
||||
run(&root, &[root.display().to_string(), "-size".into(), "+1P".into()]);
|
||||
assert_eq!(code, 0, "stderr: {}", capture.err());
|
||||
assert_eq!(capture.err(), "");
|
||||
assert_eq!(capture.out(), "");
|
||||
}
|
||||
|
||||
/// Failure mode: BSD leading flags `-x` and `-s` were treated as unknown
|
||||
/// predicates and the invocation failed to parse.
|
||||
#[test]
|
||||
fn bsd_leading_flags_x_and_s_parse() {
|
||||
let (_dir, root) = fixture();
|
||||
let (code, capture) = run(
|
||||
&root,
|
||||
&[
|
||||
"-s".into(),
|
||||
"-x".into(),
|
||||
root.display().to_string(),
|
||||
"-type".into(),
|
||||
"f".into(),
|
||||
],
|
||||
);
|
||||
assert_eq!(code, 0, "stderr: {}", capture.err());
|
||||
assert_eq!(capture.err(), "");
|
||||
// -s guarantees lexicographically sorted output.
|
||||
let matches: Vec<PathBuf> = capture.out().lines().map(PathBuf::from).collect();
|
||||
assert_eq!(matches, vec![root.join("a.txt"), root.join("b.md"), root.join("c.rs")]);
|
||||
}
|
||||
}
|
||||
|
||||
+192
-45
@@ -175,10 +175,13 @@ fn process_num_block(
|
||||
let mut quiet = false;
|
||||
let mut verbose = false;
|
||||
let mut zero_terminated = false;
|
||||
// Lowercase suffixes are byte multipliers (obsolete BSD `-Nc`/`-Nb`/`-Nk`/`-Nm`);
|
||||
// uppercase suffixes mirror the modern `-n NUM<suffix>` form and scale the
|
||||
// line count (`head -10K` == `head -n 10240`).
|
||||
let mut multiplier = None;
|
||||
let mut line_multiplier: usize = 1;
|
||||
let mut c = last_char;
|
||||
loop {
|
||||
// note that here, we only match lower case 'k', 'c', and 'm'
|
||||
match c {
|
||||
// we want to preserve order
|
||||
// this also saves us 1 heap allocation
|
||||
@@ -195,6 +198,18 @@ fn process_num_block(
|
||||
'b' => multiplier = Some(512),
|
||||
'k' => multiplier = Some(1024),
|
||||
'm' => multiplier = Some(1024 * 1024),
|
||||
'K' => {
|
||||
line_multiplier = 1024;
|
||||
multiplier = None;
|
||||
},
|
||||
'M' => {
|
||||
line_multiplier = 1024 * 1024;
|
||||
multiplier = None;
|
||||
},
|
||||
'G' => {
|
||||
line_multiplier = 1024 * 1024 * 1024;
|
||||
multiplier = None;
|
||||
},
|
||||
'\0' => {},
|
||||
_ => return Err(ParseError),
|
||||
}
|
||||
@@ -220,6 +235,7 @@ fn process_num_block(
|
||||
options.push(OsString::from(format!("{num}")));
|
||||
} else {
|
||||
options.push(OsString::from("-n"));
|
||||
let num = num.saturating_mul(line_multiplier);
|
||||
options.push(OsString::from(format!("{num}")));
|
||||
}
|
||||
Ok(options)
|
||||
@@ -266,6 +282,8 @@ mod tests {
|
||||
assert_eq!(obsolete("-1k"), obsolete_result(&["-c", "1024"]));
|
||||
assert_eq!(obsolete("-2b"), obsolete_result(&["-c", "1024"]));
|
||||
assert_eq!(obsolete("-1mmk"), obsolete_result(&["-c", "1024"]));
|
||||
assert_eq!(obsolete("-10K"), obsolete_result(&["-n", "10240"]));
|
||||
assert_eq!(obsolete("-1M"), obsolete_result(&["-n", "1048576"]));
|
||||
assert_eq!(obsolete("-1vz"), obsolete_result(&["-v", "-z", "-n", "1"]));
|
||||
assert_eq!(
|
||||
obsolete("-1vzqvq"),
|
||||
@@ -1020,33 +1038,81 @@ impl Mode {
|
||||
}
|
||||
}
|
||||
|
||||
fn arg_iterate<'a>(
|
||||
mut args: impl Iterator<Item = OsString> + 'a,
|
||||
) -> HeadResult<Box<dyn Iterator<Item = OsString> + 'a>> {
|
||||
// argv[0] is always present
|
||||
let first = args.next().unwrap();
|
||||
if let Some(second) = args.next() {
|
||||
if let Some(s) = second.to_str() {
|
||||
if let Some(v) = parse::parse_obsolete(s) {
|
||||
match v {
|
||||
Ok(iter) => Ok(Box::new(vec![first].into_iter().chain(iter).chain(args))),
|
||||
Err(parse::ParseError) => {
|
||||
Err(HeadError::ParseError(format!("bad argument format: {}", s.quote())))
|
||||
},
|
||||
}
|
||||
} else {
|
||||
// The second argument contains non-UTF-8 sequences, so it can't be an obsolete
|
||||
// option like "-5". Treat it as a regular file argument.
|
||||
Ok(Box::new(vec![first, second].into_iter().chain(args)))
|
||||
}
|
||||
} else {
|
||||
// The second argument contains non-UTF-8 sequences, so it can't be an obsolete
|
||||
// option like "-5". Treat it as a regular file argument.
|
||||
Ok(Box::new(vec![first, second].into_iter().chain(args)))
|
||||
/// True when `token` is an option that takes its value from the *next* argv
|
||||
/// token, so that value must never be mistaken for an obsolete `-NUM` form
|
||||
/// (e.g. the `-5` in `head -n -5 file`).
|
||||
fn consumes_separate_value(token: &str) -> bool {
|
||||
if let Some(long) = token.strip_prefix("--") {
|
||||
if long.is_empty() || long.contains('=') {
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
Ok(Box::new(vec![first].into_iter()))
|
||||
// clap infers unambiguous long-option prefixes.
|
||||
return ["lines", "bytes"].iter().any(|name| name.starts_with(long));
|
||||
}
|
||||
let Some(cluster) = token.strip_prefix('-') else {
|
||||
return false;
|
||||
};
|
||||
let mut chars = cluster.chars();
|
||||
while let Some(c) = chars.next() {
|
||||
match c {
|
||||
// Value-taking shorts: a trailing `-n`/`-c` consumes the next
|
||||
// token; anything after them in the cluster is an attached value.
|
||||
'n' | 'c' => return chars.next().is_none(),
|
||||
'q' | 'v' | 'z' => {},
|
||||
_ => return false,
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Rewrites every obsolete `-NUM[suffix]` token (before `--`) into modern
|
||||
/// options, wherever it appears among flags and operands: GNU/BSD accept
|
||||
/// `head -q -5 file`, `head file -5`, and `head -5 -20 file`.
|
||||
fn arg_iterate(argv: Vec<OsString>) -> HeadResult<Vec<OsString>> {
|
||||
let mut rewritten = Vec::with_capacity(argv.len() + 1);
|
||||
let mut iter = argv.into_iter();
|
||||
// argv[0] is always present
|
||||
rewritten.extend(iter.next());
|
||||
let mut skip_value = false;
|
||||
let mut seen_ddash = false;
|
||||
for arg in iter {
|
||||
if skip_value || seen_ddash {
|
||||
skip_value = false;
|
||||
rewritten.push(arg);
|
||||
continue;
|
||||
}
|
||||
let Some(token) = arg.to_str() else {
|
||||
// Non-UTF-8 can't be an obsolete option like "-5"; treat it as a
|
||||
// regular file argument.
|
||||
rewritten.push(arg);
|
||||
continue;
|
||||
};
|
||||
if token == "--" {
|
||||
seen_ddash = true;
|
||||
rewritten.push(arg);
|
||||
continue;
|
||||
}
|
||||
if matches!(token.as_bytes(), [b'-', b'0'..=b'9', ..]) {
|
||||
match parse::parse_obsolete(token) {
|
||||
Some(Ok(options)) => {
|
||||
rewritten.extend(options);
|
||||
continue;
|
||||
},
|
||||
Some(Err(parse::ParseError)) => {
|
||||
return Err(HeadError::ParseError(format!(
|
||||
"bad argument format: {}",
|
||||
token.quote()
|
||||
)));
|
||||
},
|
||||
None => {},
|
||||
}
|
||||
}
|
||||
if token.len() > 1 && token.starts_with('-') {
|
||||
skip_value = consumes_separate_value(token);
|
||||
}
|
||||
rewritten.push(arg);
|
||||
}
|
||||
Ok(rewritten)
|
||||
}
|
||||
|
||||
#[derive(Debug, PartialEq, Default)]
|
||||
@@ -1301,9 +1367,7 @@ impl Utility for Head {
|
||||
// Normalize GNU's obsolete `-NUM` syntax before clap sees argv; clap
|
||||
// otherwise treats it as an unknown short-option cluster.
|
||||
fn rewrite_argv(argv: Vec<OsString>) -> Result<Vec<OsString>, String> {
|
||||
arg_iterate(argv.into_iter())
|
||||
.map(Iterator::collect)
|
||||
.map_err(|err| err.to_string())
|
||||
arg_iterate(argv).map_err(|err| err.to_string())
|
||||
}
|
||||
|
||||
|
||||
@@ -1316,14 +1380,24 @@ impl Utility for Head {
|
||||
},
|
||||
};
|
||||
|
||||
let print_headers = (options.files.len() > 1 && !options.quiet) || options.verbose;
|
||||
// GNU head only emits the blank separator line before a header when a
|
||||
// previous file actually produced output; open failures print nothing
|
||||
// and must not flip `first`.
|
||||
let mut first = true;
|
||||
fn print_header(out: &mut impl Write, name: &[u8], first: &mut bool) {
|
||||
if !*first {
|
||||
let _ = writeln!(out);
|
||||
}
|
||||
let _ = out.write_all(b"==> ");
|
||||
let _ = out.write_all(name);
|
||||
let _ = out.write_all(b" <==\n");
|
||||
*first = false;
|
||||
}
|
||||
for file in &options.files {
|
||||
let result = if file == "-" {
|
||||
if (options.files.len() > 1 && !options.quiet) || options.verbose {
|
||||
if !first {
|
||||
let _ = writeln!(host.stdout);
|
||||
}
|
||||
let _ = writeln!(host.stdout, "==> standard input <==");
|
||||
if print_headers {
|
||||
print_header(&mut host.stdout, b"standard input", &mut first);
|
||||
}
|
||||
let mut input = io::BufReader::with_capacity(BUF_SIZE, &mut host.stdin);
|
||||
match options.mode {
|
||||
@@ -1344,25 +1418,23 @@ impl Utility for Head {
|
||||
} else {
|
||||
let resolved = host.resolve(file);
|
||||
if resolved.is_dir() {
|
||||
// GNU prints the header before reporting the read error,
|
||||
// and that header counts as produced output.
|
||||
if print_headers {
|
||||
print_header(&mut host.stdout, file.as_encoded_bytes(), &mut first);
|
||||
}
|
||||
host.error(format!("error reading {}: Is a directory", file.quote()), 1);
|
||||
first = false;
|
||||
continue;
|
||||
}
|
||||
let mut input = match File::open(&resolved) {
|
||||
Ok(input) => input,
|
||||
Err(err) => {
|
||||
host.error(format!("cannot open {} for reading: {err}", file.quote()), 1);
|
||||
first = false;
|
||||
continue;
|
||||
},
|
||||
};
|
||||
if (options.files.len() > 1 && !options.quiet) || options.verbose {
|
||||
if !first {
|
||||
let _ = writeln!(host.stdout);
|
||||
}
|
||||
let _ = write!(host.stdout, "==> ");
|
||||
let _ = host.stdout.write_all(file.as_encoded_bytes());
|
||||
let _ = writeln!(host.stdout, " <==");
|
||||
if print_headers {
|
||||
print_header(&mut host.stdout, file.as_encoded_bytes(), &mut first);
|
||||
}
|
||||
head_file(&mut input, &mut host.stdout, &options)
|
||||
};
|
||||
@@ -1372,10 +1444,14 @@ impl Utility for Head {
|
||||
} else {
|
||||
PathBuf::from(file)
|
||||
};
|
||||
// A dead pipe ends the whole invocation; any other I/O error
|
||||
// only fails this operand, and GNU keeps going.
|
||||
let broken_pipe = err.kind() == io::ErrorKind::BrokenPipe;
|
||||
host.error(HeadError::Io { name, err }, 1);
|
||||
return 1;
|
||||
if broken_pipe {
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
first = false;
|
||||
}
|
||||
host.exit_code()
|
||||
}
|
||||
@@ -1434,6 +1510,77 @@ mod tests {
|
||||
assert_eq!(capture.err(), "head: bad argument format: '-123FooBar'\n");
|
||||
}
|
||||
|
||||
fn rewritten(argv: &[&str]) -> Vec<String> {
|
||||
Head::rewrite_argv(argv.iter().map(OsString::from).collect())
|
||||
.unwrap()
|
||||
.into_iter()
|
||||
.map(|arg| arg.to_str().unwrap().to_owned())
|
||||
.collect()
|
||||
}
|
||||
|
||||
// Failure mode: obsolete `-NUM` was only recognized as argv[1], so
|
||||
// `head -q -5 file`, `head file -5`, and repeated counts were parse errors.
|
||||
#[test]
|
||||
fn obsolete_num_is_rewritten_at_any_position() {
|
||||
assert_eq!(rewritten(&["head", "-q", "-5", "f"]), ["head", "-q", "-n", "5", "f"]);
|
||||
assert_eq!(rewritten(&["head", "-v", "-20", "f"]), ["head", "-v", "-n", "20", "f"]);
|
||||
assert_eq!(rewritten(&["head", "f", "-5"]), ["head", "f", "-n", "5"]);
|
||||
assert_eq!(
|
||||
rewritten(&["head", "-5", "-20", "f"]),
|
||||
["head", "-n", "5", "-n", "20", "f"]
|
||||
);
|
||||
assert_eq!(rewritten(&["head", "-5qz", "f"]), ["head", "-q", "-z", "-n", "5", "f"]);
|
||||
}
|
||||
|
||||
// Failure mode: `-5` following a value-taking option is that option's
|
||||
// value, and rewriting it would corrupt the invocation.
|
||||
#[test]
|
||||
fn option_values_and_post_ddash_operands_are_not_rewritten() {
|
||||
assert_eq!(rewritten(&["head", "-n", "-5", "f"]), ["head", "-n", "-5", "f"]);
|
||||
assert_eq!(rewritten(&["head", "-c", "-5", "f"]), ["head", "-c", "-5", "f"]);
|
||||
assert_eq!(rewritten(&["head", "--lines", "-5", "f"]), ["head", "--lines", "-5", "f"]);
|
||||
assert_eq!(rewritten(&["head", "--", "-5"]), ["head", "--", "-5"]);
|
||||
assert_eq!(rewritten(&["head", "-n5", "-", "f"]), ["head", "-n5", "-", "f"]);
|
||||
}
|
||||
|
||||
// Failure mode: uppercase suffixes in the obsolete form were rejected
|
||||
// even though `head -n 10K` accepts them.
|
||||
#[test]
|
||||
fn obsolete_uppercase_suffixes_scale_lines() {
|
||||
assert_eq!(options("-10K").unwrap().mode, Mode::FirstLines(10 * 1024));
|
||||
assert_eq!(options("-1M").unwrap().mode, Mode::FirstLines(1024 * 1024));
|
||||
assert_eq!(options("-1G").unwrap().mode, Mode::FirstLines(1024 * 1024 * 1024));
|
||||
// Lowercase suffixes keep their historical byte meaning.
|
||||
assert_eq!(options("-1k").unwrap().mode, Mode::FirstBytes(1024));
|
||||
}
|
||||
|
||||
// Failure mode: an unreadable operand flipped the separator state and the
|
||||
// next header gained a spurious leading blank line; a mid-list open error
|
||||
// must also not abort the remaining operands.
|
||||
#[test]
|
||||
fn open_error_produces_no_separator_and_processing_continues() {
|
||||
let dir = tempdir().unwrap();
|
||||
std::fs::write(dir.path().join("f1"), b"a\n").unwrap();
|
||||
std::fs::write(dir.path().join("f2"), b"b\n").unwrap();
|
||||
let (code, capture) = run_util::<Head>(&["-n", "1", "missing", "f1", "f2"], "", dir.path());
|
||||
assert_eq!(code, 1);
|
||||
assert_eq!(capture.out(), "==> f1 <==\na\n\n==> f2 <==\nb\n");
|
||||
assert!(capture.err().contains("cannot open 'missing' for reading"));
|
||||
}
|
||||
|
||||
// Failure mode: the `==> dir <==` header was suppressed before the
|
||||
// Is-a-directory diagnostic; GNU prints it and counts it as output.
|
||||
#[test]
|
||||
fn directory_operand_prints_header_before_error() {
|
||||
let dir = tempdir().unwrap();
|
||||
std::fs::create_dir(dir.path().join("d")).unwrap();
|
||||
std::fs::write(dir.path().join("f"), b"a\n").unwrap();
|
||||
let (code, capture) = run_util::<Head>(&["-n", "1", "d", "f"], "", dir.path());
|
||||
assert_eq!(code, 1);
|
||||
assert_eq!(capture.out(), "==> d <==\n\n==> f <==\na\n");
|
||||
assert_eq!(capture.err(), "head: error reading 'd': Is a directory\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn defaults_to_ten_lines_from_stdin() {
|
||||
let input = (1..=12).map(|n| format!("{n}\n")).collect::<String>();
|
||||
|
||||
@@ -443,11 +443,20 @@ pub(crate) fn os_bytes_lossy(value: &std::ffi::OsStr) -> std::borrow::Cow<'_, [u
|
||||
|
||||
/// Parses a GNU-style duration: a decimal number with an optional `s`/`m`/`h`/`d`
|
||||
/// suffix, as accepted by `sleep` and `timeout`.
|
||||
///
|
||||
/// GNU also accepts `inf`/`infinity` (optionally signed `+`, any case);
|
||||
/// infinite and overflowing values saturate to [`Duration::MAX`]. Callers
|
||||
/// treat such durations as "sleep until cancelled". Sub-millisecond precision
|
||||
/// is preserved: GNU `sleep 0.0001` really sleeps 100 microseconds.
|
||||
pub(crate) fn parse_duration(input: &str) -> Option<Duration> {
|
||||
let trimmed = input.trim();
|
||||
if trimmed.is_empty() {
|
||||
return None;
|
||||
}
|
||||
let unsigned = trimmed.strip_prefix('+').unwrap_or(trimmed);
|
||||
if unsigned.eq_ignore_ascii_case("inf") || unsigned.eq_ignore_ascii_case("infinity") {
|
||||
return Some(Duration::MAX);
|
||||
}
|
||||
let (number, multiplier) = match trimmed.chars().last()? {
|
||||
's' => (&trimmed[..trimmed.len() - 1], 1.0),
|
||||
'm' => (&trimmed[..trimmed.len() - 1], 60.0),
|
||||
@@ -457,14 +466,14 @@ pub(crate) fn parse_duration(input: &str) -> Option<Duration> {
|
||||
_ => (trimmed, 1.0),
|
||||
};
|
||||
let value = number.parse::<f64>().ok()?;
|
||||
if value.is_sign_negative() {
|
||||
if value.is_nan() || value.is_sign_negative() {
|
||||
return None;
|
||||
}
|
||||
let millis = value * multiplier * 1000.0;
|
||||
if !millis.is_finite() || millis < 0.0 {
|
||||
return None;
|
||||
if value.is_infinite() {
|
||||
return Some(Duration::MAX);
|
||||
}
|
||||
Some(Duration::from_millis(millis.round() as u64))
|
||||
// Only overflow remains once NaN and negatives are excluded; saturate.
|
||||
Duration::try_from_secs_f64(value * multiplier).map_or(Some(Duration::MAX), Some)
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -624,7 +624,7 @@ mod filter {
|
||||
});
|
||||
Block::new(idx, labels).unwrap().map_code(|c| {
|
||||
let c = c.replace('\t', " ");
|
||||
let w = unicode_width::UnicodeWidthStr::width(&*c);
|
||||
let w = xutf::width_str(&c);
|
||||
CodeWidth::new(c, core::cmp::max(w, 1))
|
||||
})
|
||||
}
|
||||
|
||||
+172
-20
@@ -37,6 +37,13 @@ pub(crate) struct KillCommand {
|
||||
impl builtins::Command for KillCommand {
|
||||
type Error = brush_core::Error;
|
||||
|
||||
fn new<I>(args: I) -> std::result::Result<Self, clap::Error>
|
||||
where
|
||||
I: IntoIterator<Item = String>,
|
||||
{
|
||||
Self::try_parse_from(rewrite_attached_short_options(args))
|
||||
}
|
||||
|
||||
#[allow(unknown_lints, reason = "unused_async_trait_impl is unknown to the pinned CI nightly")]
|
||||
#[allow(
|
||||
clippy::unused_async_trait_impl,
|
||||
@@ -342,6 +349,63 @@ impl builtins::Command for KillCommand {
|
||||
}
|
||||
}
|
||||
|
||||
/// Splits attached short-option values before clap sees the argv: `-sKILL`
|
||||
/// and `-s9` become `-s <spec>`, `-n9` becomes `-n 9`, and `-l9`/`-L137`
|
||||
/// become `-l <spec>` (bash splits `-s<name>` the same way; the digit forms
|
||||
/// are the /bin/kill spellings). A token whose whole body already names a
|
||||
/// signal (`-sigkill`, `-SIGKILL`, `-9`) is left intact, matching BSD kill
|
||||
/// and the manual sigspec pre-parse in `execute`. Rewriting stops at `--` or
|
||||
/// the first operand, so negative-PID operands survive untouched.
|
||||
fn rewrite_attached_short_options(args: impl IntoIterator<Item = String>) -> Vec<String> {
|
||||
let mut out: Vec<String> = Vec::new();
|
||||
let mut args = args.into_iter();
|
||||
// The first element is the command name itself.
|
||||
out.extend(args.next());
|
||||
let mut skip_value = false;
|
||||
for arg in &mut args {
|
||||
if skip_value {
|
||||
skip_value = false;
|
||||
out.push(arg);
|
||||
continue;
|
||||
}
|
||||
if arg == "--" {
|
||||
out.push(arg);
|
||||
break;
|
||||
}
|
||||
if arg == "-s" || arg == "-n" {
|
||||
skip_value = true;
|
||||
out.push(arg);
|
||||
continue;
|
||||
}
|
||||
if arg == "-l" || arg == "-L" {
|
||||
out.push(arg);
|
||||
continue;
|
||||
}
|
||||
if let Some((option, value)) = split_attached(&arg) {
|
||||
out.push(option);
|
||||
out.push(value);
|
||||
continue;
|
||||
}
|
||||
out.push(arg);
|
||||
break;
|
||||
}
|
||||
out.extend(args);
|
||||
out
|
||||
}
|
||||
|
||||
/// Splits one attached-value option token, or `None` for anything that must
|
||||
/// pass through untouched (whole sigspecs, operands, malformed tokens).
|
||||
fn split_attached(arg: &str) -> Option<(String, String)> {
|
||||
let rest = arg.get(2..).filter(|rest| !rest.is_empty())?;
|
||||
let split = match arg.get(..2)? {
|
||||
"-l" | "-L" => true,
|
||||
"-s" => KillSignal::parse(&arg[1..]).is_err(),
|
||||
"-n" => rest.bytes().all(|byte| byte.is_ascii_digit()),
|
||||
_ => false,
|
||||
};
|
||||
split.then(|| (arg[..2].to_string(), rest.to_string()))
|
||||
}
|
||||
|
||||
/// Whether signalling `target` would reach the shell or one of its ancestors.
|
||||
///
|
||||
/// `target` follows `kill(2)`: a positive value is a pid, `0` is the caller's own
|
||||
@@ -375,26 +439,7 @@ fn print_kill_signals<'a>(
|
||||
.map(|()| ExecutionResult::success());
|
||||
}
|
||||
for value in signals {
|
||||
enum PrintedSignal {
|
||||
Name(&'static str),
|
||||
Number(i32),
|
||||
}
|
||||
let signal = if let Ok(number) = value.parse::<i32>() {
|
||||
TrapSignal::try_from(number).map(|signal| {
|
||||
PrintedSignal::Name(
|
||||
signal
|
||||
.as_str()
|
||||
.strip_prefix("SIG")
|
||||
.unwrap_or(signal.as_str()),
|
||||
)
|
||||
})
|
||||
} else {
|
||||
TrapSignal::try_from(value.as_str()).map(|signal| {
|
||||
i32::try_from(signal)
|
||||
.map_or(PrintedSignal::Name(signal.as_str()), PrintedSignal::Number)
|
||||
})
|
||||
};
|
||||
match signal {
|
||||
match printed_signal(value) {
|
||||
Ok(PrintedSignal::Name(name)) => writeln!(context.stdout(), "{name}")?,
|
||||
Ok(PrintedSignal::Number(number)) => writeln!(context.stdout(), "{number}")?,
|
||||
Err(err) => {
|
||||
@@ -406,6 +451,34 @@ fn print_kill_signals<'a>(
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
/// How `kill -l <operand>` renders one operand: numbers become names and
|
||||
/// names become numbers.
|
||||
enum PrintedSignal {
|
||||
Name(&'static str),
|
||||
Number(i32),
|
||||
}
|
||||
|
||||
fn printed_signal(value: &str) -> std::result::Result<PrintedSignal, brush_core::Error> {
|
||||
if let Ok(number) = value.parse::<i32>() {
|
||||
// bash also maps the exit status of a signal-killed process back to
|
||||
// its signal: `kill -l 137` prints `KILL` (137 = 128 + 9), while an
|
||||
// unmappable value like 128 or 265 keeps its own diagnostic.
|
||||
let signal = TrapSignal::try_from(number).or_else(|err| {
|
||||
if number > 128 {
|
||||
TrapSignal::try_from(number - 128).map_err(|_| err)
|
||||
} else {
|
||||
Err(err)
|
||||
}
|
||||
})?;
|
||||
Ok(PrintedSignal::Name(
|
||||
signal.as_str().strip_prefix("SIG").unwrap_or(signal.as_str()),
|
||||
))
|
||||
} else {
|
||||
let signal = TrapSignal::try_from(value)?;
|
||||
Ok(i32::try_from(signal).map_or(PrintedSignal::Name(signal.as_str()), PrintedSignal::Number))
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
impl KillCommand {
|
||||
fn listed_signals(&self) -> impl Iterator<Item = &String> {
|
||||
@@ -447,6 +520,85 @@ mod tests {
|
||||
fn lists_pre_marker_operands_without_marker() {
|
||||
assert_eq!(listed(&["kill", "-l", "TERM", "HUP"]), ["TERM", "HUP"]);
|
||||
}
|
||||
|
||||
fn parsed(args: &[&str]) -> KillCommand {
|
||||
use brush_core::builtins::Command as _;
|
||||
KillCommand::new(args.iter().map(ToString::to_string)).unwrap()
|
||||
}
|
||||
|
||||
/// `kill -s9`/`-sKILL` used to land in the positional args and die with
|
||||
/// "invalid signal name"; the attached value must reach `-s`.
|
||||
#[test]
|
||||
fn attached_signal_name_values_split() {
|
||||
let cmd = parsed(&["kill", "-s9", "123"]);
|
||||
assert_eq!(cmd.signal_name.as_deref(), Some("9"));
|
||||
assert_eq!(cmd.args, ["123"]);
|
||||
|
||||
let cmd = parsed(&["kill", "-sKILL", "123"]);
|
||||
assert_eq!(cmd.signal_name.as_deref(), Some("KILL"));
|
||||
assert_eq!(cmd.args, ["123"]);
|
||||
}
|
||||
|
||||
/// A whole-token signal spec such as `-sigkill` (BSD kill accepts it)
|
||||
/// must not be misread as `-s igkill`.
|
||||
#[test]
|
||||
fn sig_prefixed_spec_stays_whole() {
|
||||
let cmd = parsed(&["kill", "-sigkill", "123"]);
|
||||
assert_eq!(cmd.signal_name, None);
|
||||
assert_eq!(cmd.args, ["-sigkill", "123"]);
|
||||
}
|
||||
|
||||
/// `kill -n9` used to fail to parse; the digits must reach `-n`.
|
||||
#[test]
|
||||
fn attached_signal_number_splits() {
|
||||
let cmd = parsed(&["kill", "-n9", "123"]);
|
||||
assert_eq!(cmd.signal_number, Some(9));
|
||||
assert_eq!(cmd.args, ["123"]);
|
||||
}
|
||||
|
||||
/// `kill -l9` and `kill -L137` used to be clap parse errors; they must
|
||||
/// behave as `-l` with the value as its listing operand.
|
||||
#[test]
|
||||
fn attached_list_operand_splits() {
|
||||
let cmd = parsed(&["kill", "-l9"]);
|
||||
assert!(cmd.list_signals);
|
||||
assert_eq!(cmd.listed_signals().cloned().collect::<Vec<_>>(), ["9"]);
|
||||
|
||||
let cmd = parsed(&["kill", "-L137"]);
|
||||
assert!(cmd.list_signals);
|
||||
assert_eq!(cmd.listed_signals().cloned().collect::<Vec<_>>(), ["137"]);
|
||||
}
|
||||
|
||||
/// Rewriting must stop at `--` and at the first operand so option-like
|
||||
/// operands (and negative PIDs) are never split.
|
||||
#[test]
|
||||
fn rewrite_leaves_operand_region_alone() {
|
||||
let rewritten = rewrite_attached_short_options(
|
||||
["kill", "--", "-s9"].map(String::from),
|
||||
);
|
||||
assert_eq!(rewritten, ["kill", "--", "-s9"]);
|
||||
|
||||
let rewritten = rewrite_attached_short_options(
|
||||
["kill", "-9", "-s9"].map(String::from),
|
||||
);
|
||||
assert_eq!(rewritten, ["kill", "-9", "-s9"]);
|
||||
|
||||
let rewritten = rewrite_attached_short_options(
|
||||
["kill", "-s", "KILL", "-123"].map(String::from),
|
||||
);
|
||||
assert_eq!(rewritten, ["kill", "-s", "KILL", "-123"]);
|
||||
}
|
||||
|
||||
/// bash maps exit statuses above 128 back to the terminating signal:
|
||||
/// `kill -l 137` prints `KILL`, while 128 and 265 stay invalid.
|
||||
#[test]
|
||||
fn list_maps_exit_statuses_above_128() {
|
||||
assert!(matches!(printed_signal("137"), Ok(PrintedSignal::Name("KILL"))));
|
||||
assert!(matches!(printed_signal("9"), Ok(PrintedSignal::Name("KILL"))));
|
||||
assert!(matches!(printed_signal("129"), Ok(PrintedSignal::Name("HUP"))));
|
||||
assert!(printed_signal("128").is_err());
|
||||
assert!(printed_signal("265").is_err());
|
||||
}
|
||||
}
|
||||
|
||||
/// A `kill` signal argument: a real signal, or the "does this process
|
||||
|
||||
@@ -124,7 +124,8 @@ mod base64;
|
||||
mod basename;
|
||||
#[cfg(feature = "util.cat")]
|
||||
mod cat;
|
||||
/// Shared checksum machinery behind `md5sum`, `sha*sum`, and `b2sum`.
|
||||
/// The `cksum` builtin plus the shared checksum machinery behind `md5sum`,
|
||||
/// `sha*sum`, and `b2sum`.
|
||||
#[cfg(feature = "util.cksum")]
|
||||
mod cksum;
|
||||
#[cfg(feature = "util.md5sum")]
|
||||
|
||||
@@ -19,22 +19,74 @@ use crate::host::quote_arg;
|
||||
#[derive(Parser)]
|
||||
#[command(disable_help_flag = true)]
|
||||
pub(crate) struct NohupCommand {
|
||||
/// `--help` was the first argument; set by [`NohupCommand::from_argv`].
|
||||
#[clap(skip)]
|
||||
help: bool,
|
||||
/// `--version` was the first argument; set by [`NohupCommand::from_argv`].
|
||||
#[clap(skip)]
|
||||
version: bool,
|
||||
#[arg(num_args = 0.., trailing_var_arg = true, allow_hyphen_values = true)]
|
||||
command: Vec<String>,
|
||||
}
|
||||
|
||||
impl NohupCommand {
|
||||
/// Parses `argv` (without the command name) the way GNU nohup does:
|
||||
/// `--help`/`--version` are recognized only as the first argument, and a
|
||||
/// single leading `--` ends option processing, so `nohup -- --help` runs
|
||||
/// a command named `--help` and `nohup -- --` runs one named `--`.
|
||||
fn from_argv(mut argv: Vec<String>) -> Self {
|
||||
match argv.first().map(String::as_str) {
|
||||
Some("--help") => {
|
||||
return Self { help: true, version: false, command: Vec::new() };
|
||||
},
|
||||
Some("--version") => {
|
||||
return Self { help: false, version: true, command: Vec::new() };
|
||||
},
|
||||
Some("--") => {
|
||||
argv.remove(0);
|
||||
},
|
||||
_ => {},
|
||||
}
|
||||
Self { help: false, version: false, command: argv }
|
||||
}
|
||||
}
|
||||
|
||||
impl builtins::Command for NohupCommand {
|
||||
type Error = brush_core::Error;
|
||||
|
||||
/// Bypasses clap: clap silently eats the first `--` even inside a
|
||||
/// `trailing_var_arg` capture, which loses the distinction between
|
||||
/// `nohup --help` (help) and `nohup -- --help` (run `--help`).
|
||||
fn new<I>(args: I) -> std::result::Result<Self, clap::Error>
|
||||
where
|
||||
I: IntoIterator<Item = String>,
|
||||
{
|
||||
// The first element is the command name itself.
|
||||
Ok(Self::from_argv(args.into_iter().skip(1).collect()))
|
||||
}
|
||||
|
||||
fn execute<SE: brush_core::ShellExtensions>(
|
||||
&self,
|
||||
context: ExecutionContext<'_, SE>,
|
||||
) -> impl Future<Output = std::result::Result<ExecutionResult, brush_core::Error>> + Send {
|
||||
let command = self.command.clone();
|
||||
let (help, version) = (self.help, self.version);
|
||||
async move {
|
||||
if context.is_cancelled() {
|
||||
return Ok(ExecutionExitCode::Interrupted.into());
|
||||
}
|
||||
if help {
|
||||
let _ = write!(context.stdout(), "{NOHUP_HELP}");
|
||||
return Ok(ExecutionResult::success());
|
||||
}
|
||||
if version {
|
||||
let _ = writeln!(
|
||||
context.stdout(),
|
||||
"nohup (pi-builtins) {}",
|
||||
env!("CARGO_PKG_VERSION")
|
||||
);
|
||||
return Ok(ExecutionResult::success());
|
||||
}
|
||||
// coreutils `nohup` with no operand fails with exit code 125.
|
||||
if command.is_empty() {
|
||||
return Ok(report_missing_operand(context.stderr()));
|
||||
@@ -60,6 +112,15 @@ impl builtins::Command for NohupCommand {
|
||||
}
|
||||
}
|
||||
|
||||
const NOHUP_HELP: &str = "\
|
||||
Usage: nohup COMMAND [ARG]...
|
||||
or: nohup OPTION
|
||||
Run COMMAND immune to the shell's teardown, in a new process group.
|
||||
|
||||
--help display this help and exit
|
||||
--version output version information and exit
|
||||
";
|
||||
|
||||
fn report_missing_operand(mut stderr: impl Write) -> ExecutionResult {
|
||||
let _ = writeln!(stderr, "nohup: missing operand");
|
||||
ExecutionResult::new(125)
|
||||
@@ -78,7 +139,57 @@ fn rebuild_command_line(command: &[String]) -> String {
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{rebuild_command_line, report_missing_operand};
|
||||
use super::{NohupCommand, rebuild_command_line, report_missing_operand};
|
||||
|
||||
fn parsed(argv: &[&str]) -> NohupCommand {
|
||||
NohupCommand::from_argv(argv.iter().map(ToString::to_string).collect())
|
||||
}
|
||||
|
||||
/// `nohup -- cmd args` must run `cmd`; the leading `--` is an option
|
||||
/// terminator, not part of the operand vector.
|
||||
#[test]
|
||||
fn leading_dashdash_ends_options() {
|
||||
let cmd = parsed(&["--", "sleep", "1"]);
|
||||
assert!(!cmd.help && !cmd.version);
|
||||
assert_eq!(cmd.command, ["sleep", "1"]);
|
||||
}
|
||||
|
||||
/// Only the first `--` terminates options: `nohup -- -- x` runs a command
|
||||
/// literally named `--`, and `nohup -- --help` runs one named `--help`.
|
||||
#[test]
|
||||
fn dashdash_protects_operands_including_help() {
|
||||
assert_eq!(parsed(&["--", "--", "x"]).command, ["--", "x"]);
|
||||
let cmd = parsed(&["--", "--help"]);
|
||||
assert!(!cmd.help);
|
||||
assert_eq!(cmd.command, ["--help"]);
|
||||
}
|
||||
|
||||
/// A mid-command `--` belongs to the operand, never to nohup itself.
|
||||
#[test]
|
||||
fn mid_command_dashdash_is_preserved() {
|
||||
assert_eq!(parsed(&["echo", "a", "--", "b"]).command, ["echo", "a", "--", "b"]);
|
||||
}
|
||||
|
||||
/// `--help`/`--version` used to be executed as commands (exit 127);
|
||||
/// GNU nohup prints to stdout and exits 0.
|
||||
#[test]
|
||||
fn leading_help_and_version_are_options() {
|
||||
let cmd = parsed(&["--help"]);
|
||||
assert!(cmd.help && !cmd.version && cmd.command.is_empty());
|
||||
let cmd = parsed(&["--version"]);
|
||||
assert!(cmd.version && !cmd.help && cmd.command.is_empty());
|
||||
}
|
||||
|
||||
/// The builtin entry point receives argv including the command name and
|
||||
/// must skip it before option handling.
|
||||
#[test]
|
||||
fn new_skips_command_name() {
|
||||
use brush_core::builtins::Command as _;
|
||||
|
||||
let cmd = NohupCommand::new(["nohup", "--", "sleep", "1"].map(String::from))
|
||||
.expect("nohup argv parsing is infallible");
|
||||
assert_eq!(cmd.command, ["sleep", "1"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn missing_operand_reports_diagnostic_and_exit_code() {
|
||||
|
||||
+186
-20
@@ -113,15 +113,15 @@ pub(crate) struct Rg {
|
||||
no_fixed_strings: bool,
|
||||
|
||||
/// Search case-insensitively.
|
||||
#[arg(short = 'i', long = "ignore-case")]
|
||||
#[arg(short = 'i', long = "ignore-case", overrides_with_all = ["case_sensitive", "smart_case"])]
|
||||
ignore_case: bool,
|
||||
|
||||
/// Search case-sensitively.
|
||||
#[arg(short = 's', long = "case-sensitive")]
|
||||
#[arg(short = 's', long = "case-sensitive", overrides_with_all = ["ignore_case", "smart_case"])]
|
||||
case_sensitive: bool,
|
||||
|
||||
/// Search case-insensitively when the pattern is all lowercase.
|
||||
#[arg(short = 'S', long = "smart-case")]
|
||||
#[arg(short = 'S', long = "smart-case", overrides_with_all = ["ignore_case", "case_sensitive"])]
|
||||
smart_case: bool,
|
||||
|
||||
/// Invert matching.
|
||||
@@ -297,17 +297,21 @@ pub(crate) struct Rg {
|
||||
context: Option<usize>,
|
||||
|
||||
/// Show line numbers.
|
||||
#[arg(short = 'n', long = "line-number")]
|
||||
#[arg(short = 'n', long = "line-number", overrides_with = "no_line_number")]
|
||||
line_number: bool,
|
||||
|
||||
/// Suppress line numbers.
|
||||
#[arg(short = 'N', long = "no-line-number")]
|
||||
#[arg(short = 'N', long = "no-line-number", overrides_with = "line_number")]
|
||||
no_line_number: bool,
|
||||
|
||||
/// Show column numbers.
|
||||
#[arg(long = "column")]
|
||||
#[arg(long = "column", overrides_with = "no_column")]
|
||||
column: bool,
|
||||
|
||||
/// Do not show column numbers.
|
||||
#[arg(long = "no-column", overrides_with = "column")]
|
||||
no_column: bool,
|
||||
|
||||
/// Show the zero-based byte offset for each result.
|
||||
#[arg(short = 'b', long = "byte-offset", overrides_with = "no_byte_offset")]
|
||||
byte_offset: bool,
|
||||
@@ -464,6 +468,19 @@ pub(crate) struct Rg {
|
||||
#[arg(long = "no-stats")]
|
||||
_no_stats: bool,
|
||||
|
||||
/// Never read configuration files (accepted; this builtin never reads any).
|
||||
#[arg(long = "no-config")]
|
||||
_no_config: bool,
|
||||
|
||||
/// Number of search threads (accepted; this builtin searches in-process,
|
||||
/// serially).
|
||||
#[arg(short = 'j', long = "threads", value_name = "NUM")]
|
||||
_threads: Option<usize>,
|
||||
|
||||
/// Print SEPARATOR instead of '/' in printed file paths.
|
||||
#[arg(long = "path-separator", value_name = "SEPARATOR")]
|
||||
path_separator: Option<String>,
|
||||
|
||||
/// Arguments: PATTERN followed by PATHs unless -e/-f/--files is used.
|
||||
#[arg(value_name = "ARGS")]
|
||||
args: Vec<OsString>,
|
||||
@@ -520,6 +537,7 @@ struct SearchOptions {
|
||||
max_columns: Option<usize>,
|
||||
max_columns_preview: bool,
|
||||
null_paths: bool,
|
||||
path_separator: Option<u8>,
|
||||
no_messages: bool,
|
||||
replacement: Option<Vec<u8>>,
|
||||
json: bool,
|
||||
@@ -545,7 +563,7 @@ struct RgSink<'a, M: Matcher, W: Write> {
|
||||
impl<M: Matcher, W: Write> RgSink<'_, M, W> {
|
||||
fn write_path(&mut self) -> io::Result<()> {
|
||||
if let Some(name) = self.display {
|
||||
self.out.write_all(name)?;
|
||||
write_display_bytes(&mut *self.out, name, self.opts.path_separator)?;
|
||||
if self.opts.null_paths {
|
||||
self.out.write_all(b"\0")?;
|
||||
}
|
||||
@@ -562,7 +580,9 @@ impl<M: Matcher, W: Write> RgSink<'_, M, W> {
|
||||
) -> io::Result<()> {
|
||||
if self.display.is_some() {
|
||||
self.write_path()?;
|
||||
self.out.write_all(&[separator])?;
|
||||
if !self.opts.null_paths {
|
||||
self.out.write_all(&[separator])?;
|
||||
}
|
||||
}
|
||||
if self.opts.line_number
|
||||
&& let Some(number) = line_number
|
||||
@@ -780,17 +800,23 @@ impl<M: Matcher, W: Write> Sink for RgSink<'_, M, W> {
|
||||
if self.opts.files_with_matches {
|
||||
if self.any_match {
|
||||
self.write_path()?;
|
||||
self.out.write_all(b"\n")?;
|
||||
if !self.opts.null_paths {
|
||||
self.out.write_all(b"\n")?;
|
||||
}
|
||||
}
|
||||
} else if self.opts.files_without_match {
|
||||
if !self.any_match {
|
||||
self.write_path()?;
|
||||
self.out.write_all(b"\n")?;
|
||||
if !self.opts.null_paths {
|
||||
self.out.write_all(b"\n")?;
|
||||
}
|
||||
}
|
||||
} else if self.opts.count || self.opts.count_matches {
|
||||
if self.display.is_some() {
|
||||
self.write_path()?;
|
||||
self.out.write_all(b":")?;
|
||||
if !self.opts.null_paths {
|
||||
self.out.write_all(b":")?;
|
||||
}
|
||||
}
|
||||
let count = if self.opts.count_matches {
|
||||
self.match_count
|
||||
@@ -811,6 +837,42 @@ fn trim_ascii_start(bytes: &[u8]) -> &[u8] {
|
||||
&bytes[start..]
|
||||
}
|
||||
|
||||
/// Writes a display path, substituting `separator` for `/` when requested via
|
||||
/// `--path-separator`.
|
||||
fn write_display_bytes<W: Write>(out: &mut W, bytes: &[u8], separator: Option<u8>) -> io::Result<()> {
|
||||
let Some(separator) = separator else {
|
||||
return out.write_all(bytes);
|
||||
};
|
||||
let mut rest = bytes;
|
||||
while let Some(pos) = rest.iter().position(|&byte| byte == b'/') {
|
||||
out.write_all(&rest[..pos])?;
|
||||
out.write_all(&[separator])?;
|
||||
rest = &rest[pos + 1..];
|
||||
}
|
||||
out.write_all(rest)
|
||||
}
|
||||
|
||||
fn parse_path_separator(spec: Option<&str>) -> Result<Option<u8>, String> {
|
||||
match spec {
|
||||
None | Some("") => Ok(None),
|
||||
Some(separator) if separator.len() == 1 => Ok(Some(separator.as_bytes()[0])),
|
||||
Some(separator) => Err(format!(
|
||||
"error parsing flag --path-separator: a path separator must be exactly one byte, but the given separator is {} bytes",
|
||||
separator.len()
|
||||
)),
|
||||
}
|
||||
}
|
||||
|
||||
/// Line numbers follow real ripgrep's piped behavior: off unless requested
|
||||
/// (`-n`), or implied by `--column`/`--vimgrep`. Real rg enables them by
|
||||
/// default only on a tty; the builtin's output is always consumed piped.
|
||||
fn effective_line_number(cli: &Rg) -> bool {
|
||||
if cli.no_line_number {
|
||||
return false;
|
||||
}
|
||||
cli.line_number || cli.column || cli.vimgrep
|
||||
}
|
||||
|
||||
fn first_column<M: Matcher>(matcher: &M, line: &[u8]) -> io::Result<Option<usize>> {
|
||||
Ok(matcher
|
||||
.find(line)
|
||||
@@ -843,8 +905,8 @@ fn build_rust_matcher(patterns: &[String], cli: &Rg) -> Result<RegexMatcher, gre
|
||||
let crlf = cli.crlf && !cli.no_crlf && !cli.null_data;
|
||||
let mut builder = RegexMatcherBuilder::new();
|
||||
builder
|
||||
.case_insensitive(cli.ignore_case && !cli.case_sensitive)
|
||||
.case_smart(cli.smart_case && !cli.ignore_case && !cli.case_sensitive)
|
||||
.case_insensitive(cli.ignore_case)
|
||||
.case_smart(cli.smart_case)
|
||||
.word(cli.word_regexp && !cli.line_regexp)
|
||||
.whole_line(cli.line_regexp)
|
||||
.fixed_strings(cli.fixed_strings && !cli.no_fixed_strings)
|
||||
@@ -864,8 +926,8 @@ fn build_pcre_matcher(host: &Host, patterns: &[String], cli: &Rg) -> Result<Pcre
|
||||
let unicode = !cli.no_unicode;
|
||||
let mut builder = PcreMatcherBuilder::new();
|
||||
builder
|
||||
.caseless(cli.ignore_case && !cli.case_sensitive)
|
||||
.case_smart(cli.smart_case && !cli.ignore_case && !cli.case_sensitive)
|
||||
.caseless(cli.ignore_case)
|
||||
.case_smart(cli.smart_case)
|
||||
.word(cli.word_regexp && !cli.line_regexp)
|
||||
.whole_line(cli.line_regexp)
|
||||
.fixed_strings(cli.fixed_strings && !cli.no_fixed_strings)
|
||||
@@ -1000,6 +1062,9 @@ fn search_options(cli: &Rg) -> SearchOptions {
|
||||
max_columns: cli.max_columns,
|
||||
max_columns_preview: cli.max_columns_preview && !cli.no_max_columns_preview,
|
||||
null_paths: cli.null,
|
||||
// Validated --path-separator and the line-number default are applied in
|
||||
// run() once paths are known.
|
||||
path_separator: None,
|
||||
no_messages: cli.no_messages && !cli.messages,
|
||||
replacement: cli
|
||||
.replacement
|
||||
@@ -1504,7 +1569,13 @@ fn collect_filtered_files(host: &mut Host, cli: &Rg, root: &Path) -> Result<Vec<
|
||||
Ok(files)
|
||||
}
|
||||
|
||||
fn list_files<W: Write>(host: &mut Host, cli: &Rg, paths: &[OsString], out: &mut W) -> SearchOutcome {
|
||||
fn list_files<W: Write>(
|
||||
host: &mut Host,
|
||||
cli: &Rg,
|
||||
paths: &[OsString],
|
||||
path_separator: Option<u8>,
|
||||
out: &mut W,
|
||||
) -> SearchOutcome {
|
||||
let mut any = false;
|
||||
let mut had_error = false;
|
||||
let mut processed_operand = false;
|
||||
@@ -1534,13 +1605,14 @@ fn list_files<W: Write>(host: &mut Host, cli: &Rg, paths: &[OsString], out: &mut
|
||||
}
|
||||
for path in files {
|
||||
let display = display_path(operand.as_os_str(), &resolved, &path);
|
||||
let _ = out.write_all(display.as_os_str().as_encoded_bytes());
|
||||
let _ =
|
||||
write_display_bytes(out, display.as_os_str().as_encoded_bytes(), path_separator);
|
||||
let _ = out.write_all(if cli.null { b"\0" } else { b"\n" });
|
||||
any = true;
|
||||
}
|
||||
},
|
||||
Ok(meta) if meta.is_file() => {
|
||||
let _ = out.write_all(operand.as_encoded_bytes());
|
||||
let _ = write_display_bytes(out, operand.as_encoded_bytes(), path_separator);
|
||||
let _ = out.write_all(if cli.null { b"\0" } else { b"\n" });
|
||||
any = true;
|
||||
},
|
||||
@@ -1745,7 +1817,14 @@ impl Utility for Rg {
|
||||
|
||||
fn run(self, host: &mut Host) -> i32 {
|
||||
let cli = self;
|
||||
let opts = search_options(&cli);
|
||||
let mut opts = search_options(&cli);
|
||||
match parse_path_separator(cli.path_separator.as_deref()) {
|
||||
Ok(separator) => opts.path_separator = separator,
|
||||
Err(error) => {
|
||||
let _ = writeln!(host.stderr, "rg: {error}");
|
||||
return 2;
|
||||
},
|
||||
}
|
||||
if opts.json
|
||||
&& (cli.files
|
||||
|| cli.type_list
|
||||
@@ -1786,8 +1865,9 @@ impl Utility for Rg {
|
||||
&mut paths,
|
||||
!cli.files && !pattern_stdin_consumed && host.stdin_is_search_input(),
|
||||
);
|
||||
opts.line_number = effective_line_number(&cli);
|
||||
if cli.files {
|
||||
let outcome = list_files(host, &cli, &paths, &mut out);
|
||||
let outcome = list_files(host, &cli, &paths, opts.path_separator, &mut out);
|
||||
let _ = out.flush();
|
||||
return if outcome.had_error {
|
||||
2
|
||||
@@ -1930,6 +2010,92 @@ mod tests {
|
||||
assert_eq!(code, 0, "{}", capture.err());
|
||||
assert_eq!(capture.out(), "hit\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn line_numbers_stay_off_when_piped_unless_requested() {
|
||||
// Defends: the builtin's output is always consumed piped, where real
|
||||
// rg omits line numbers; `1:` prefixes sprouting by default break
|
||||
// text consumers. `-n` opts in, `-N` beats `-n`.
|
||||
let tree = tempfile::tempdir().unwrap();
|
||||
std::fs::write(tree.path().join("a.txt"), "miss\nhit\n").unwrap();
|
||||
let (code, capture) = run_util::<Rg>(&["hit", "."], "", tree.path());
|
||||
assert_eq!(code, 0, "{}", capture.err());
|
||||
assert_eq!(capture.out(), "a.txt:hit\n");
|
||||
let (code, capture) = run_util::<Rg>(&["-n", "hit", "."], "", tree.path());
|
||||
assert_eq!(code, 0, "{}", capture.err());
|
||||
assert_eq!(capture.out(), "a.txt:2:hit\n");
|
||||
let (code, capture) = run_util::<Rg>(&["-n", "-N", "hit", "."], "", tree.path());
|
||||
assert_eq!(code, 0, "{}", capture.err());
|
||||
assert_eq!(capture.out(), "a.txt:hit\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn null_terminates_paths_without_trailing_newline() {
|
||||
// Defends: `rg -l0 | xargs -0` must see `path\0`, not `path\0\n`.
|
||||
let tree = tempfile::tempdir().unwrap();
|
||||
std::fs::write(tree.path().join("a.txt"), "hit\n").unwrap();
|
||||
let (code, capture) = run_util::<Rg>(&["-l0", "hit", "."], "", tree.path());
|
||||
assert_eq!(code, 0, "{}", capture.err());
|
||||
assert_eq!(capture.out(), "a.txt\0");
|
||||
// Match lines: `path\0text` with no `:` after the NUL; with -n the
|
||||
// line number follows the NUL (`path\0N:text`).
|
||||
let (code, capture) = run_util::<Rg>(&["-0", "hit", "."], "", tree.path());
|
||||
assert_eq!(code, 0, "{}", capture.err());
|
||||
assert_eq!(capture.out(), "a.txt\0hit\n");
|
||||
let (code, capture) = run_util::<Rg>(&["-n0", "hit", "."], "", tree.path());
|
||||
assert_eq!(code, 0, "{}", capture.err());
|
||||
assert_eq!(capture.out(), "a.txt\01:hit\n");
|
||||
// --files-without-match also emits bare `path\0`. (Exit code for this
|
||||
// mode is a pre-existing divergence outside this test's contract.)
|
||||
let (_code, capture) =
|
||||
run_util::<Rg>(&["--files-without-match", "-0", "nope", "."], "", tree.path());
|
||||
assert_eq!(capture.out(), "a.txt\0");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn last_case_flag_wins() {
|
||||
// Defends: real rg resolves -s/-i/-S by last occurrence, not by a
|
||||
// fixed precedence.
|
||||
let (code, out, err) = run(&["-s", "-i", "HIT", "-"], "hit\n");
|
||||
assert_eq!(code, 0, "{err}");
|
||||
assert_eq!(out, "hit\n");
|
||||
let (code, out, _) = run(&["-i", "-s", "HIT", "-"], "hit\n");
|
||||
assert_eq!(code, 1);
|
||||
assert_eq!(out, "");
|
||||
// -i then -S: smart case wins, and an uppercase pattern stays sensitive.
|
||||
let (code, _, _) = run(&["-i", "-S", "HIT", "-"], "hit\n");
|
||||
assert_eq!(code, 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn compat_flags_are_accepted() {
|
||||
// Defends: `--no-config`/`-j`/`--threads`/`--no-column` must not be
|
||||
// clap parse errors.
|
||||
let (code, out, err) = run(&["--no-config", "-j2", "-S", "hit", "-"], "hit\n");
|
||||
assert_eq!(code, 0, "{err}");
|
||||
assert_eq!(out, "hit\n");
|
||||
let (code, out, err) =
|
||||
run(&["--threads", "4", "-n", "--column", "--no-column", "hit", "-"], "hit\n");
|
||||
assert_eq!(code, 0, "{err}");
|
||||
assert_eq!(out, "1:hit\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn path_separator_replaces_slash() {
|
||||
// Defends: --path-separator must rewrite `/` in printed paths and
|
||||
// reject multi-byte separators like real rg.
|
||||
let tree = tempfile::tempdir().unwrap();
|
||||
std::fs::create_dir(tree.path().join("sub")).unwrap();
|
||||
std::fs::write(tree.path().join("sub/a.txt"), "hit\n").unwrap();
|
||||
let (code, capture) =
|
||||
run_util::<Rg>(&["--path-separator", "|", "hit", "."], "", tree.path());
|
||||
assert_eq!(code, 0, "{}", capture.err());
|
||||
assert_eq!(capture.out(), "sub|a.txt:hit\n");
|
||||
let (code, capture) =
|
||||
run_util::<Rg>(&["--path-separator", "::", "hit", "."], "", tree.path());
|
||||
assert_eq!(code, 2);
|
||||
assert!(capture.err().contains("exactly one byte"));
|
||||
}
|
||||
}
|
||||
|
||||
use brush_core::{ShellExtensions, builtins::Registration};
|
||||
|
||||
@@ -12,7 +12,9 @@ use crate::host::parse_duration;
|
||||
#[derive(Parser)]
|
||||
#[command(disable_help_flag = true)]
|
||||
pub(crate) struct SleepCommand {
|
||||
#[arg(required = true)]
|
||||
// GNU reports `sleep -1` as an invalid time interval (exit 1), not as an
|
||||
// unknown option; let hyphenated operands through to `parse_duration`.
|
||||
#[arg(required = true, allow_hyphen_values = true)]
|
||||
durations: Vec<String>,
|
||||
}
|
||||
|
||||
@@ -28,13 +30,14 @@ impl builtins::Command for SleepCommand {
|
||||
if context.is_cancelled() {
|
||||
return Ok(ExecutionExitCode::Interrupted.into());
|
||||
}
|
||||
let mut total = Duration::from_millis(0);
|
||||
let mut total = Duration::ZERO;
|
||||
for duration in &durations {
|
||||
let Some(parsed) = parse_duration(duration) else {
|
||||
let _ = writeln!(context.stderr(), "sleep: invalid time interval '{duration}'");
|
||||
return Ok(ExecutionResult::new(1));
|
||||
};
|
||||
total += parsed;
|
||||
// `infinity` parses as `Duration::MAX`; keep the sum saturating.
|
||||
total = total.saturating_add(parsed);
|
||||
}
|
||||
let sleep = time::sleep(total);
|
||||
tokio::pin!(sleep);
|
||||
@@ -69,8 +72,10 @@ mod tests {
|
||||
assert_eq!(parse_duration("0.001"), Some(Duration::from_millis(1)));
|
||||
assert_eq!(parse_duration("0.001s"), Some(Duration::from_millis(1)));
|
||||
assert_eq!(parse_duration("0.001m"), Some(Duration::from_millis(60)));
|
||||
assert_eq!(parse_duration("0.000001h"), Some(Duration::from_millis(4)));
|
||||
assert_eq!(parse_duration("0.00000001d"), Some(Duration::from_millis(1)));
|
||||
// Sub-millisecond precision must survive: GNU sleep honors 100µs.
|
||||
assert_eq!(parse_duration("0.0001"), Some(Duration::from_micros(100)));
|
||||
assert_eq!(parse_duration("0.000001h"), Some(Duration::from_micros(3600)));
|
||||
assert_eq!(parse_duration("0.00000001d"), Some(Duration::from_micros(864)));
|
||||
|
||||
let mut shell = Shell::builder().build().await.expect("test shell should build");
|
||||
let command = SleepCommand { durations: vec!["0.001".into(), "0.001s".into()] };
|
||||
@@ -87,6 +92,68 @@ mod tests {
|
||||
assert!(result.is_success());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn infinity_operand_parses_and_sleep_is_cancellable() {
|
||||
// GNU accepts `inf`/`infinity`, any case, with an optional `+` sign.
|
||||
for spec in ["infinity", "inf", "INFINITY", "Inf", "+infinity", "+inf"] {
|
||||
assert_eq!(parse_duration(spec), Some(Duration::MAX), "spec {spec:?}");
|
||||
}
|
||||
assert_eq!(parse_duration("nan"), None);
|
||||
assert_eq!(parse_duration("-inf"), None);
|
||||
|
||||
// `sleep infinity` must block until cancelled rather than erroring out.
|
||||
let token = CancellationToken::new();
|
||||
let mut params = ExecutionParameters::default();
|
||||
params.set_cancel_token(token.clone());
|
||||
let mut shell = Shell::builder().build().await.expect("test shell should build");
|
||||
let command = SleepCommand { durations: vec!["infinity".into()] };
|
||||
let context = ExecutionContext {
|
||||
shell: &mut shell,
|
||||
command_name: "sleep".into(),
|
||||
params,
|
||||
};
|
||||
let execution = async {
|
||||
let (result, ()) = tokio::join!(command.execute(context), async {
|
||||
tokio::task::yield_now().await;
|
||||
token.cancel();
|
||||
});
|
||||
result
|
||||
};
|
||||
let result = time::timeout(Duration::from_millis(100), execution)
|
||||
.await
|
||||
.expect("cancelled infinite sleep should return promptly")
|
||||
.expect("sleep execution should succeed");
|
||||
|
||||
assert_eq!(
|
||||
u8::from(result.exit_code),
|
||||
u8::from(ExecutionExitCode::Interrupted)
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn hyphenated_operand_is_an_invalid_interval_not_an_unknown_flag() {
|
||||
// `sleep -1` must not die in clap with an unknown-option error; GNU
|
||||
// reports an invalid time interval and exits 1.
|
||||
let command = <SleepCommand as Command>::new(["sleep".into(), "-1".into()])
|
||||
.expect("hyphenated operand should reach the builtin");
|
||||
|
||||
let (mut stderr_reader, stderr_writer) = std::io::pipe().expect("stderr pipe should open");
|
||||
let mut params = ExecutionParameters::default();
|
||||
params.set_fd(OpenFiles::STDERR_FD, OpenFile::from(stderr_writer));
|
||||
let mut shell = Shell::builder().build().await.expect("test shell should build");
|
||||
let context = ExecutionContext {
|
||||
shell: &mut shell,
|
||||
command_name: "sleep".into(),
|
||||
params,
|
||||
};
|
||||
let result = command.execute(context).await.expect("sleep execution should succeed");
|
||||
let mut stderr = String::new();
|
||||
stderr_reader.read_to_string(&mut stderr).expect("stderr should be readable");
|
||||
|
||||
assert_eq!(u8::from(result.exit_code), 1);
|
||||
assert_eq!(stderr, "sleep: invalid time interval '-1'\n");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn invalid_duration_reports_original_diagnostic_and_exit_code() {
|
||||
let (mut stderr_reader, stderr_writer) = std::io::pipe().expect("stderr pipe should open");
|
||||
|
||||
+442
-117
@@ -157,7 +157,8 @@ for details about the options it supports.";
|
||||
pub const FORMAT: &str = "format";
|
||||
pub const PRINTF: &str = "printf";
|
||||
pub const TERSE: &str = "terse";
|
||||
pub const BSD_TIME_WARNING: &str = "bsd-time-warning";
|
||||
pub const BSD_SHELL: &str = "bsd-shell";
|
||||
pub const BSD_TIMEFMT: &str = "bsd-timefmt";
|
||||
pub const FILES: &str = "files";
|
||||
}
|
||||
|
||||
@@ -267,7 +268,7 @@ for details about the options it supports.";
|
||||
Unsigned(u64),
|
||||
UnsignedHex(u64),
|
||||
UnsignedOct(u32),
|
||||
Float(f64),
|
||||
Timestamp { sec: i64, nsec: u32 },
|
||||
Unknown,
|
||||
}
|
||||
|
||||
@@ -400,6 +401,7 @@ for details about the options it supports.";
|
||||
show_fs: bool,
|
||||
from_user: bool,
|
||||
files: Vec<OsString>,
|
||||
time_format: Option<String>,
|
||||
#[cfg_attr(not(unix), allow(dead_code))]
|
||||
mount_list: OnceCell<Option<Vec<OsString>>>,
|
||||
#[cfg_attr(not(unix), allow(dead_code))]
|
||||
@@ -479,8 +481,8 @@ for details about the options it supports.";
|
||||
OutputType::UnsignedHex(num) => {
|
||||
print_unsigned_hex(out, *num, flags, width, precision, padding_char);
|
||||
},
|
||||
OutputType::Float(num) => {
|
||||
print_float(out, *num, flags, width, precision, padding_char);
|
||||
OutputType::Timestamp { sec, nsec } => {
|
||||
print_timestamp(out, *sec, *nsec, flags, width, precision, padding_char);
|
||||
},
|
||||
OutputType::Unknown => {
|
||||
let _ = write!(out, "?");
|
||||
@@ -698,48 +700,26 @@ for details about the options it supports.";
|
||||
pad_and_print(out, &extended, flags.left, width, padding_char);
|
||||
}
|
||||
|
||||
/// Truncate a float to the given number of digits after the decimal point.
|
||||
fn precision_trunc(num: f64, precision: Precision) -> String {
|
||||
// GNU `stat` doesn't round, it just seems to truncate to the
|
||||
// given precision:
|
||||
//
|
||||
// $ stat -c "%.5Y" /dev/pts/ptmx
|
||||
// 1736344012.76399
|
||||
// $ stat -c "%.4Y" /dev/pts/ptmx
|
||||
// 1736344012.7639
|
||||
// $ stat -c "%.3Y" /dev/pts/ptmx
|
||||
// 1736344012.763
|
||||
//
|
||||
// Contrast this with `printf`, which seems to round the
|
||||
// numbers:
|
||||
//
|
||||
// $ printf "%.5f\n" 1736344012.76399
|
||||
// 1736344012.76399
|
||||
// $ printf "%.4f\n" 1736344012.76399
|
||||
// 1736344012.7640
|
||||
// $ printf "%.3f\n" 1736344012.76399
|
||||
// 1736344012.764
|
||||
//
|
||||
let num_str = num.to_string();
|
||||
let n = num_str.len();
|
||||
match (num_str.find('.'), precision) {
|
||||
(None, Precision::NotSpecified) => num_str,
|
||||
(None, Precision::NoNumber) => num_str,
|
||||
(None, Precision::Number(0)) => num_str,
|
||||
(None, Precision::Number(p)) => format!("{num_str}.{zeros}", zeros = "0".repeat(p)),
|
||||
(Some(i), Precision::NotSpecified) => num_str[..i].to_string(),
|
||||
(Some(_), Precision::NoNumber) => num_str,
|
||||
(Some(i), Precision::Number(0)) => num_str[..i].to_string(),
|
||||
(Some(i), Precision::Number(p)) if p < n - i => num_str[..i + 1 + p].to_string(),
|
||||
(Some(i), Precision::Number(p)) => {
|
||||
format!("{num_str}{zeros}", zeros = "0".repeat(p - (n - i - 1)))
|
||||
/// Formats an epoch timestamp with GNU `stat`'s truncation rules: no
|
||||
/// precision prints whole seconds (so `stat -c %Y` survives shell
|
||||
/// arithmetic), a bare `.` prints all nine fractional digits, and an
|
||||
/// explicit precision truncates or zero-pads the fraction.
|
||||
fn timestamp_string(sec: i64, nsec: u32, precision: Precision) -> String {
|
||||
match precision {
|
||||
Precision::NotSpecified | Precision::Number(0) => sec.to_string(),
|
||||
Precision::NoNumber => format!("{sec}.{nsec:09}"),
|
||||
Precision::Number(p) if p <= 9 => {
|
||||
let frac = format!("{nsec:09}");
|
||||
format!("{sec}.{}", &frac[..p])
|
||||
},
|
||||
Precision::Number(p) => format!("{sec}.{nsec:09}{:0<pad$}", "", pad = p - 9),
|
||||
}
|
||||
}
|
||||
|
||||
fn print_float(
|
||||
fn print_timestamp(
|
||||
out: &mut dyn Write,
|
||||
num: f64,
|
||||
sec: i64,
|
||||
nsec: u32,
|
||||
flags: Flags,
|
||||
width: usize,
|
||||
precision: Precision,
|
||||
@@ -752,8 +732,7 @@ for details about the options it supports.";
|
||||
} else {
|
||||
""
|
||||
};
|
||||
let num_str = precision_trunc(num, precision);
|
||||
let extended = format!("{prefix}{num_str}");
|
||||
let extended = format!("{prefix}{}", timestamp_string(sec, nsec, precision));
|
||||
pad_and_print(out, &extended, flags.left, width, padding_char);
|
||||
}
|
||||
|
||||
@@ -1128,6 +1107,7 @@ for details about the options it supports.";
|
||||
show_fs,
|
||||
from_user: !format_str.is_empty(),
|
||||
files,
|
||||
time_format: matches.get_one::<String>(options::BSD_TIMEFMT).cloned(),
|
||||
mount_list: OnceCell::new(),
|
||||
mount_list_needed,
|
||||
default_tokens,
|
||||
@@ -1135,6 +1115,12 @@ for details about the options it supports.";
|
||||
})
|
||||
}
|
||||
|
||||
/// The `strftime` format for human-readable time directives; BSD `-t`
|
||||
/// overrides the GNU default.
|
||||
fn time_fmt(&self) -> &str {
|
||||
self.time_format.as_deref().unwrap_or(PRETTY_DATETIME_FORMAT)
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
fn find_mount_point<P: AsRef<Path>>(
|
||||
&self,
|
||||
@@ -1273,37 +1259,36 @@ for details about the options it supports.";
|
||||
},
|
||||
|
||||
// time of file birth, human-readable; - if unknown
|
||||
'w' => OutputType::Str(pretty_time(meta, MetadataTimeField::Birth)),
|
||||
|
||||
'w' => OutputType::Str(pretty_time(meta, MetadataTimeField::Birth, self.time_fmt())),
|
||||
// time of file birth, seconds since Epoch; 0 if unknown
|
||||
'W' => OutputType::Integer(
|
||||
metadata_get_time(meta, MetadataTimeField::Birth)
|
||||
.map_or(0, |x| system_time_to_sec(x).0),
|
||||
),
|
||||
|
||||
'W' => {
|
||||
let (sec, nsec) = metadata_get_time(meta, MetadataTimeField::Birth)
|
||||
.map_or((0, 0), system_time_to_sec);
|
||||
OutputType::Timestamp { sec, nsec }
|
||||
},
|
||||
// time of last access, human-readable
|
||||
'x' => OutputType::Str(pretty_time(meta, MetadataTimeField::Access)),
|
||||
'x' => OutputType::Str(pretty_time(meta, MetadataTimeField::Access, self.time_fmt())),
|
||||
// time of last access, seconds since Epoch
|
||||
'X' => {
|
||||
let (sec, nsec) = metadata_get_time(meta, MetadataTimeField::Access)
|
||||
.map_or((0, 0), system_time_to_sec);
|
||||
OutputType::Float(sec as f64 + nsec as f64 / 1_000_000_000.0)
|
||||
OutputType::Timestamp { sec, nsec }
|
||||
},
|
||||
// time of last data modification, human-readable
|
||||
'y' => OutputType::Str(pretty_time(meta, MetadataTimeField::Modification)),
|
||||
'y' => OutputType::Str(pretty_time(meta, MetadataTimeField::Modification, self.time_fmt())),
|
||||
// time of last data modification, seconds since Epoch
|
||||
'Y' => {
|
||||
let (sec, nsec) = metadata_get_time(meta, MetadataTimeField::Modification)
|
||||
.map_or((0, 0), system_time_to_sec);
|
||||
OutputType::Float(sec as f64 + nsec as f64 / 1_000_000_000.0)
|
||||
OutputType::Timestamp { sec, nsec }
|
||||
},
|
||||
// time of last status change, human-readable
|
||||
'z' => OutputType::Str(pretty_time(meta, MetadataTimeField::Change)),
|
||||
'z' => OutputType::Str(pretty_time(meta, MetadataTimeField::Change, self.time_fmt())),
|
||||
// time of last status change, seconds since Epoch
|
||||
'Z' => {
|
||||
let (sec, nsec) = metadata_get_time(meta, MetadataTimeField::Change)
|
||||
.map_or((0, 0), system_time_to_sec);
|
||||
OutputType::Float(sec as f64 + nsec as f64 / 1_000_000_000.0)
|
||||
OutputType::Timestamp { sec, nsec }
|
||||
},
|
||||
'R' => OutputType::UnsignedHex(meta.rdev()),
|
||||
'r' if flag.major => OutputType::Unsigned(major(meta.rdev() as _) as u64),
|
||||
@@ -1442,11 +1427,13 @@ for details about the options it supports.";
|
||||
/// GNU's `-f` is `--file-system`; parsed as GNU, a BSD invocation prints
|
||||
/// filesystem info for each real operand and errors on the format operand.
|
||||
/// An invocation is treated as BSD when a `-f` cluster (optionally with the
|
||||
/// BSD boolean flags `L`/`n`/`q`/`F`) carries a format value containing
|
||||
/// `%` — GNU filesystem mode would have to target a file literally named
|
||||
/// like a format string, which never happens in practice. Detected
|
||||
/// BSD boolean flags `L`/`n`/`q`/`F`/`s`/`x`) carries a format value
|
||||
/// containing `%` — GNU filesystem mode would have to target a file
|
||||
/// literally named like a format string, which never happens in practice —
|
||||
/// or when a cluster of BSD boolean flags contains the BSD-only output
|
||||
/// styles `-s` (shell assignments) or `-x` (Linux-like verbose). Detected
|
||||
/// invocations are rewritten to the GNU equivalent (`-c`/`--printf` plus a
|
||||
/// translated format) before clap parsing.
|
||||
/// translated format, or hidden style/timefmt options) before clap parsing.
|
||||
///
|
||||
/// Returns `None` when the invocation is not BSD-shaped, `Some(Err(_))`
|
||||
/// when it is BSD-shaped but uses an option or directive with no GNU
|
||||
@@ -1464,12 +1451,22 @@ for details about the options it supports.";
|
||||
if cluster.is_empty() || cluster.starts_with('-') {
|
||||
continue;
|
||||
}
|
||||
// `-s` / `-x` are BSD-only output styles: a cluster of BSD boolean
|
||||
// flags containing one marks the invocation (GNU stat has neither).
|
||||
if cluster
|
||||
.chars()
|
||||
.all(|c| matches!(c, 'L' | 'n' | 'q' | 'F' | 's' | 'x'))
|
||||
&& cluster.chars().any(|c| matches!(c, 's' | 'x'))
|
||||
{
|
||||
detected = true;
|
||||
break;
|
||||
}
|
||||
let Some(fpos) = cluster.find('f') else {
|
||||
continue;
|
||||
};
|
||||
if !cluster[..fpos]
|
||||
.chars()
|
||||
.all(|c| matches!(c, 'L' | 'n' | 'q' | 'F'))
|
||||
.all(|c| matches!(c, 'L' | 'n' | 'q' | 'F' | 's' | 'x'))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
@@ -1490,12 +1487,22 @@ for details about the options it supports.";
|
||||
Some(bsd_to_gnu_argv(argv, &toks))
|
||||
}
|
||||
|
||||
/// Output style selected by a BSD invocation.
|
||||
enum BsdStyle {
|
||||
/// `-f <fmt>`: caller-supplied BSD format string.
|
||||
Custom(String),
|
||||
/// `-s`: eval-able `st_dev=… st_ino=…` shell assignments.
|
||||
Shell,
|
||||
/// `-x`: Linux-like verbose block.
|
||||
Verbose,
|
||||
}
|
||||
|
||||
/// Parses a detected BSD invocation and produces the equivalent GNU argv.
|
||||
fn bsd_to_gnu_argv(argv: &[OsString], toks: &[Cow<'_, str>]) -> Result<Vec<OsString>, String> {
|
||||
let mut follow = false;
|
||||
let mut no_newline = false;
|
||||
let mut format = None;
|
||||
let mut timefmt_ignored = false;
|
||||
let mut style: Option<BsdStyle> = None;
|
||||
let mut timefmt: Option<String> = None;
|
||||
let mut files: Vec<OsString> = Vec::new();
|
||||
|
||||
let mut i = 1;
|
||||
@@ -1522,6 +1529,9 @@ for details about the options it supports.";
|
||||
// `-q` (suppress error messages) and `-F` (ls -F type
|
||||
// decorations) have no GNU counterpart worth emulating.
|
||||
'q' | 'F' => {},
|
||||
// Output styles; like BSD, the last one seen wins.
|
||||
's' => style = Some(BsdStyle::Shell),
|
||||
'x' => style = Some(BsdStyle::Verbose),
|
||||
c @ ('f' | 't') => {
|
||||
// The rest of the cluster is the attached value,
|
||||
// otherwise the next token is.
|
||||
@@ -1535,9 +1545,9 @@ for details about the options it supports.";
|
||||
}
|
||||
};
|
||||
if c == 'f' {
|
||||
format = Some(value);
|
||||
style = Some(BsdStyle::Custom(value));
|
||||
} else {
|
||||
timefmt_ignored = true;
|
||||
timefmt = Some(value);
|
||||
}
|
||||
break;
|
||||
},
|
||||
@@ -1552,27 +1562,37 @@ for details about the options it supports.";
|
||||
i += 1 + usize::from(consumed_next);
|
||||
}
|
||||
|
||||
let Some(format) = format else {
|
||||
return Err("BSD-style '-f' expects a format string".to_string());
|
||||
};
|
||||
let translated = translate_bsd_format(&format, no_newline)?;
|
||||
let mut out: Vec<OsString> = Vec::with_capacity(files.len() + 5);
|
||||
|
||||
let mut out: Vec<OsString> = Vec::with_capacity(files.len() + 7);
|
||||
out.push(argv[0].clone());
|
||||
if follow {
|
||||
out.push("-L".into());
|
||||
}
|
||||
if timefmt_ignored {
|
||||
out.push("--bsd-time-warning".into());
|
||||
match style {
|
||||
// `-s` renders directly from the metadata (the full octal
|
||||
// `st_mode` and `st_flags` have no GNU format directive); its
|
||||
// timestamps are epoch integers regardless of `-t`, as on BSD.
|
||||
Some(BsdStyle::Shell) => out.push("--bsd-shell".into()),
|
||||
Some(BsdStyle::Verbose) => {
|
||||
out.push("--bsd-timefmt".into());
|
||||
out.push(timefmt.unwrap_or_else(|| BSD_VERBOSE_TIMEFMT.into()).into());
|
||||
out.push(if no_newline { "--printf".into() } else { "-c".into() });
|
||||
out.push(BSD_VERBOSE_FORMAT.into());
|
||||
},
|
||||
Some(BsdStyle::Custom(format)) => {
|
||||
let translated = translate_bsd_format(&format, no_newline)?;
|
||||
if let Some(timefmt) = timefmt {
|
||||
out.push("--bsd-timefmt".into());
|
||||
out.push(timefmt.into());
|
||||
}
|
||||
// `--printf` suppresses the mandatory trailing newline (BSD
|
||||
// `-n`); the translator escapes literal backslashes so text
|
||||
// survives printf mode.
|
||||
out.push(if no_newline { "--printf".into() } else { "-c".into() });
|
||||
out.push(translated.into());
|
||||
},
|
||||
None => return Err("BSD-style '-f' expects a format string".to_string()),
|
||||
}
|
||||
// `--printf` suppresses the mandatory trailing newline (BSD `-n`); the
|
||||
// translator escapes literal backslashes so text survives printf mode.
|
||||
out.push(if no_newline {
|
||||
"--printf".into()
|
||||
} else {
|
||||
"-c".into()
|
||||
});
|
||||
out.push(translated.into());
|
||||
out.push("--".into());
|
||||
out.extend(files);
|
||||
Ok(out)
|
||||
}
|
||||
@@ -1739,6 +1759,154 @@ for details about the options it supports.";
|
||||
format!("unsupported BSD format directive '{directive}'")
|
||||
}
|
||||
|
||||
/// GNU-language rendering of BSD `stat -x` ("Linux-like" verbose output).
|
||||
const BSD_VERBOSE_FORMAT: &str = concat!(
|
||||
" File: \"%n\"\n",
|
||||
" Size: %-11s FileType: %F\n",
|
||||
" Mode: (%04a/%.10A) Uid: (%5u/%8U) Gid: (%5g/%8G)\n",
|
||||
"Device: %Hd,%Ld Inode: %i Links: %h\n",
|
||||
"Access: %x\n",
|
||||
"Modify: %y\n",
|
||||
"Change: %z\n",
|
||||
" Birth: %w",
|
||||
);
|
||||
|
||||
/// BSD `stat -x` renders timestamps `ctime(3)`-style.
|
||||
const BSD_VERBOSE_TIMEFMT: &str = "%a %b %e %H:%M:%S %Y";
|
||||
|
||||
/// BSD `stat -s`: one eval-able line of `st_*=value` shell assignments per
|
||||
/// file, rendered directly from the metadata.
|
||||
fn bsd_shell_exec(matches: &ArgMatches, host: &mut Host) -> i32 {
|
||||
let files: Vec<OsString> = matches
|
||||
.get_many::<OsString>(options::FILES)
|
||||
.map(|v| v.cloned().collect())
|
||||
.unwrap_or_default();
|
||||
if files.is_empty() {
|
||||
host.error(StatError::MissingOperand, 1);
|
||||
return 1;
|
||||
}
|
||||
let follow = matches.get_flag(options::DEREFERENCE);
|
||||
let mut ret = 0;
|
||||
for file in &files {
|
||||
let display_name = file.to_string_lossy();
|
||||
let resolved = host.resolve(file);
|
||||
let result = if follow {
|
||||
fs::metadata(&resolved)
|
||||
} else {
|
||||
fs::symlink_metadata(&resolved)
|
||||
};
|
||||
match result {
|
||||
Ok(meta) => {
|
||||
let _ = writeln!(host.stdout, "{}", bsd_shell_line(&meta, &resolved));
|
||||
},
|
||||
Err(e) => {
|
||||
let _ = writeln!(&mut host.stderr, "stat: {}", StatError::CannotStat {
|
||||
file: display_name.quote().to_string(),
|
||||
error: e.to_string(),
|
||||
});
|
||||
ret = 1;
|
||||
},
|
||||
}
|
||||
}
|
||||
ret
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
fn bsd_shell_line(meta: &Metadata, _resolved: &Path) -> String {
|
||||
#[cfg(target_os = "macos")]
|
||||
let flags = std::os::macos::fs::MetadataExt::st_flags(meta);
|
||||
#[cfg(not(target_os = "macos"))]
|
||||
let flags = 0u32;
|
||||
let birth = metadata_get_time(meta, MetadataTimeField::Birth)
|
||||
.map_or(0, |t| system_time_to_sec(t).0);
|
||||
format!(
|
||||
"st_dev={} st_ino={} st_mode=0{:o} st_nlink={} st_uid={} st_gid={} st_rdev={} \
|
||||
st_size={} st_atime={} st_mtime={} st_ctime={} st_birthtime={birth} st_blksize={} \
|
||||
st_blocks={} st_flags={flags}",
|
||||
meta.dev(),
|
||||
meta.ino(),
|
||||
meta.mode(),
|
||||
meta.nlink(),
|
||||
meta.uid(),
|
||||
meta.gid(),
|
||||
meta.rdev(),
|
||||
meta.len(),
|
||||
meta.atime(),
|
||||
meta.mtime(),
|
||||
meta.ctime(),
|
||||
meta.blksize(),
|
||||
meta.blocks(),
|
||||
)
|
||||
}
|
||||
|
||||
#[cfg(windows)]
|
||||
fn bsd_shell_line(meta: &Metadata, resolved: &Path) -> String {
|
||||
let ids = win::handle_info(resolved, !meta.file_type().is_symlink());
|
||||
let (dev, ino, nlink) = ids.map_or((0, 0, 1), |i| (i.volume_serial, i.file_index, i.links));
|
||||
let sec = |field| win::md_time(meta, field).map_or(0, |t| system_time_to_sec(t).0);
|
||||
format!(
|
||||
"st_dev={dev} st_ino={ino} st_mode=0{:o} st_nlink={nlink} st_uid=0 st_gid=0 st_rdev=0 \
|
||||
st_size={} st_atime={} st_mtime={} st_ctime={} st_birthtime={} st_blksize=4096 \
|
||||
st_blocks={} st_flags=0",
|
||||
win::synth_mode(meta),
|
||||
meta.len(),
|
||||
sec(win::TimeField::Access),
|
||||
sec(win::TimeField::Modification),
|
||||
sec(win::TimeField::Change),
|
||||
sec(win::TimeField::Birth),
|
||||
win::allocated_size(resolved, meta.len()).div_ceil(512),
|
||||
)
|
||||
}
|
||||
|
||||
/// GNU `-f`/`--file-system` whose first operand names no file but looks
|
||||
/// like a BSD format string (contains `%` or whitespace): rather than
|
||||
/// failing on a nonexistent operand, re-interpret the invocation as BSD
|
||||
/// `stat -f <fmt> <file>...`. Existing-path operands always keep GNU
|
||||
/// filesystem mode.
|
||||
fn bsd_filesystem_fallback(matches: &ArgMatches, host: &Host) -> Option<Vec<OsString>> {
|
||||
if !matches.get_flag(options::FILE_SYSTEM)
|
||||
|| matches.contains_id(options::FORMAT)
|
||||
|| matches.contains_id(options::PRINTF)
|
||||
{
|
||||
return None;
|
||||
}
|
||||
let files: Vec<&OsString> = matches.get_many::<OsString>(options::FILES)?.collect();
|
||||
// A format plus at least one operand; a lone missing path stays a GNU
|
||||
// error.
|
||||
if files.len() < 2 {
|
||||
return None;
|
||||
}
|
||||
let fmt = files[0].to_string_lossy();
|
||||
if !(fmt.contains('%') || fmt.chars().any(char::is_whitespace)) {
|
||||
return None;
|
||||
}
|
||||
if host.resolve(files[0]).symlink_metadata().is_ok() {
|
||||
return None;
|
||||
}
|
||||
let translated = translate_bsd_format(&fmt, false).ok()?;
|
||||
let mut argv: Vec<OsString> = Vec::with_capacity(files.len() + 4);
|
||||
argv.push("stat".into());
|
||||
if matches.get_flag(options::DEREFERENCE) {
|
||||
argv.push("-L".into());
|
||||
}
|
||||
argv.push("-c".into());
|
||||
argv.push(translated.into());
|
||||
argv.push("--".into());
|
||||
argv.extend(files[1..].iter().map(|f| (*f).clone()));
|
||||
Some(argv)
|
||||
}
|
||||
|
||||
/// Builds a [`Stater`] from parsed matches and runs it.
|
||||
fn run_stater(matches: &ArgMatches, host: &mut Host) -> i32 {
|
||||
match Stater::new(matches, host) {
|
||||
Ok(stater) => stater.exec(host),
|
||||
Err(error) => {
|
||||
host.error(error, 1);
|
||||
1
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/// Parsed `stat` invocation.
|
||||
pub(crate) struct Stat {
|
||||
@@ -1758,20 +1926,21 @@ for details about the options it supports.";
|
||||
}
|
||||
|
||||
fn run(self, host: &mut Host) -> i32 {
|
||||
if self.matches.get_flag(options::BSD_TIME_WARNING) {
|
||||
let _ = writeln!(
|
||||
host.stderr,
|
||||
"stat: warning: BSD '-t' time format is ignored; human-readable times use the GNU \
|
||||
default format"
|
||||
);
|
||||
if self.matches.get_flag(options::BSD_SHELL) {
|
||||
return bsd_shell_exec(&self.matches, host);
|
||||
}
|
||||
match Stater::new(&self.matches, host) {
|
||||
Ok(stater) => stater.exec(host),
|
||||
Err(error) => {
|
||||
host.error(error, 1);
|
||||
1
|
||||
},
|
||||
if let Some(argv) = bsd_filesystem_fallback(&self.matches, host) {
|
||||
return match app().try_get_matches_from(argv) {
|
||||
Ok(matches) => run_stater(&matches, host),
|
||||
// The rebuilt argv is a plain `-c FORMAT -- FILE...`; a
|
||||
// parse failure here is unreachable in practice.
|
||||
Err(err) => {
|
||||
let _ = write!(host.stderr, "{err}");
|
||||
1
|
||||
},
|
||||
};
|
||||
}
|
||||
run_stater(&self.matches, host)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1804,11 +1973,17 @@ for details about the options it supports.";
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::BSD_TIME_WARNING)
|
||||
.long(options::BSD_TIME_WARNING)
|
||||
Arg::new(options::BSD_SHELL)
|
||||
.long(options::BSD_SHELL)
|
||||
.hide(true)
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::BSD_TIMEFMT)
|
||||
.long(options::BSD_TIMEFMT)
|
||||
.value_name("TIMEFMT")
|
||||
.hide(true),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::FORMAT)
|
||||
.short('c')
|
||||
@@ -1839,13 +2014,13 @@ for details about the options it supports.";
|
||||
const PRETTY_DATETIME_FORMAT: &str = "%Y-%m-%d %H:%M:%S.%N %z";
|
||||
|
||||
#[cfg(unix)]
|
||||
fn pretty_time(meta: &Metadata, md_time_field: MetadataTimeField) -> String {
|
||||
fn pretty_time(meta: &Metadata, md_time_field: MetadataTimeField, fmt: &str) -> String {
|
||||
if let Some(time) = metadata_get_time(meta, md_time_field) {
|
||||
let mut tmp = Vec::new();
|
||||
if format_system_time(
|
||||
&mut tmp,
|
||||
time,
|
||||
PRETTY_DATETIME_FORMAT,
|
||||
fmt,
|
||||
FormatSystemTimeFallback::Float,
|
||||
)
|
||||
.is_ok()
|
||||
@@ -1860,7 +2035,7 @@ for details about the options it supports.";
|
||||
/// most intricate part of the utility and the print paths were repatched.
|
||||
#[cfg(test)]
|
||||
mod unit_tests {
|
||||
use super::{Flags, Precision, ScanUtil, Stater, Token, group_num, precision_trunc};
|
||||
use super::{Flags, Precision, ScanUtil, Stater, Token, group_num, timestamp_string};
|
||||
|
||||
#[test]
|
||||
fn test_scanners() {
|
||||
@@ -1940,12 +2115,22 @@ for details about the options it supports.";
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_precision_trunc() {
|
||||
assert_eq!(precision_trunc(123.456, Precision::NotSpecified), "123");
|
||||
assert_eq!(precision_trunc(123.456, Precision::NoNumber), "123.456");
|
||||
assert_eq!(precision_trunc(123.456, Precision::Number(0)), "123");
|
||||
assert_eq!(precision_trunc(123.456, Precision::Number(1)), "123.4");
|
||||
assert_eq!(precision_trunc(123.456, Precision::Number(5)), "123.45600");
|
||||
fn test_timestamp_string() {
|
||||
// `stat -c %Y` must yield integers so shell arithmetic works.
|
||||
assert_eq!(timestamp_string(1712345678, 999_999_999, Precision::NotSpecified), "1712345678");
|
||||
assert_eq!(timestamp_string(1712345678, 123_456_789, Precision::Number(0)), "1712345678");
|
||||
// `%.Y` prints all nine fractional digits; explicit precision
|
||||
// truncates (GNU semantics) or zero-pads past nine.
|
||||
assert_eq!(
|
||||
timestamp_string(1712345678, 123_456_789, Precision::NoNumber),
|
||||
"1712345678.123456789"
|
||||
);
|
||||
assert_eq!(timestamp_string(1712345678, 123_456_789, Precision::Number(3)), "1712345678.123");
|
||||
assert_eq!(timestamp_string(1712345678, 5, Precision::Number(3)), "1712345678.000");
|
||||
assert_eq!(
|
||||
timestamp_string(1712345678, 123_456_789, Precision::Number(11)),
|
||||
"1712345678.12345678900"
|
||||
);
|
||||
}
|
||||
}
|
||||
/// file-status path is Unix-only (`std::os::unix`); this reimplements the
|
||||
@@ -2205,13 +2390,13 @@ for details about the options it supports.";
|
||||
|
||||
/// `std::fs::Metadata` timestamp through the shared datetime format.
|
||||
#[cfg(windows)]
|
||||
fn pretty_time(meta: &Metadata, field: win::TimeField) -> String {
|
||||
fn pretty_time(meta: &Metadata, field: win::TimeField, fmt: &str) -> String {
|
||||
if let Some(time) = win::md_time(meta, field) {
|
||||
let mut tmp = Vec::new();
|
||||
if format_system_time(
|
||||
&mut tmp,
|
||||
time,
|
||||
PRETTY_DATETIME_FORMAT,
|
||||
fmt,
|
||||
FormatSystemTimeFallback::Float,
|
||||
)
|
||||
.is_ok()
|
||||
@@ -2352,35 +2537,36 @@ for details about the options it supports.";
|
||||
// user name of owner
|
||||
'U' => OutputType::Str("UNKNOWN".to_string()),
|
||||
// time of file birth, human-readable; - if unknown
|
||||
'w' => OutputType::Str(pretty_time(meta, win::TimeField::Birth)),
|
||||
'w' => OutputType::Str(pretty_time(meta, win::TimeField::Birth, self.time_fmt())),
|
||||
// time of file birth, seconds since Epoch; 0 if unknown
|
||||
'W' => OutputType::Integer(
|
||||
win::md_time(meta, win::TimeField::Birth)
|
||||
.map_or(0, |x| system_time_to_sec(x).0),
|
||||
),
|
||||
'W' => {
|
||||
let (sec, nsec) = win::md_time(meta, win::TimeField::Birth)
|
||||
.map_or((0, 0), system_time_to_sec);
|
||||
OutputType::Timestamp { sec, nsec }
|
||||
},
|
||||
// time of last access, human-readable
|
||||
'x' => OutputType::Str(pretty_time(meta, win::TimeField::Access)),
|
||||
'x' => OutputType::Str(pretty_time(meta, win::TimeField::Access, self.time_fmt())),
|
||||
// time of last access, seconds since Epoch
|
||||
'X' => {
|
||||
let (sec, nsec) = win::md_time(meta, win::TimeField::Access)
|
||||
.map_or((0, 0), system_time_to_sec);
|
||||
OutputType::Float(sec as f64 + nsec as f64 / 1_000_000_000.0)
|
||||
OutputType::Timestamp { sec, nsec }
|
||||
},
|
||||
// time of last data modification, human-readable
|
||||
'y' => OutputType::Str(pretty_time(meta, win::TimeField::Modification)),
|
||||
'y' => OutputType::Str(pretty_time(meta, win::TimeField::Modification, self.time_fmt())),
|
||||
// time of last data modification, seconds since Epoch
|
||||
'Y' => {
|
||||
let (sec, nsec) = win::md_time(meta, win::TimeField::Modification)
|
||||
.map_or((0, 0), system_time_to_sec);
|
||||
OutputType::Float(sec as f64 + nsec as f64 / 1_000_000_000.0)
|
||||
OutputType::Timestamp { sec, nsec }
|
||||
},
|
||||
// time of last status change, human-readable (write time)
|
||||
'z' => OutputType::Str(pretty_time(meta, win::TimeField::Change)),
|
||||
'z' => OutputType::Str(pretty_time(meta, win::TimeField::Change, self.time_fmt())),
|
||||
// time of last status change, seconds since Epoch
|
||||
'Z' => {
|
||||
let (sec, nsec) = win::md_time(meta, win::TimeField::Change)
|
||||
.map_or((0, 0), system_time_to_sec);
|
||||
OutputType::Float(sec as f64 + nsec as f64 / 1_000_000_000.0)
|
||||
OutputType::Timestamp { sec, nsec }
|
||||
},
|
||||
// rdev (no device special files on Windows)
|
||||
'R' => OutputType::UnsignedHex(0),
|
||||
@@ -2656,6 +2842,145 @@ mod tests {
|
||||
"unexpected stderr: {stderr:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn epoch_time_specifiers_print_integers() {
|
||||
let (_dir, root) = canonical_tempdir();
|
||||
fs::write(root.join("data.bin"), b"x").unwrap();
|
||||
|
||||
// Regression: `%X`/`%Y`/`%Z` printed floats, which broke shell
|
||||
// arithmetic like `$(($(stat -c %Y a) - $(stat -c %Y b)))`.
|
||||
let (code, stdout, stderr) = run_in(root, vec!["-c", "%X %Y %Z %W", "data.bin"]);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stderr, "");
|
||||
let fields: Vec<&str> = stdout.split_whitespace().collect();
|
||||
assert_eq!(fields.len(), 4, "unexpected stdout: {stdout:?}");
|
||||
for field in fields {
|
||||
assert!(field.parse::<i64>().is_ok(), "epoch fields must be integers: {stdout:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn epoch_time_precision_prints_fraction() {
|
||||
let (_dir, root) = canonical_tempdir();
|
||||
fs::write(root.join("data.bin"), b"x").unwrap();
|
||||
|
||||
// `%.3Y` keeps three fractional digits; bare `%.Y` prints all nine.
|
||||
let (code, stdout, _) = run_in(root.clone(), vec!["-c", "%.3Y", "data.bin"]);
|
||||
assert_eq!(code, 0);
|
||||
let (sec, frac) = stdout.trim_end().split_once('.').expect("fraction expected");
|
||||
assert!(sec.parse::<i64>().is_ok(), "unexpected stdout: {stdout:?}");
|
||||
assert_eq!(frac.len(), 3, "unexpected stdout: {stdout:?}");
|
||||
|
||||
let (_, stdout, _) = run_in(root, vec!["-c", "%.Y", "data.bin"]);
|
||||
let (_, frac) = stdout.trim_end().split_once('.').expect("fraction expected");
|
||||
assert_eq!(frac.len(), 9, "unexpected stdout: {stdout:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bsd_shell_format_prints_evalable_assignments() {
|
||||
let (_dir, root) = canonical_tempdir();
|
||||
fs::write(root.join("data.bin"), b"hello world!").unwrap();
|
||||
|
||||
// BSD `stat -s`: one line of `st_*=value` pairs, eval-able in sh.
|
||||
let (code, stdout, stderr) = run_in(root, vec!["-s", "data.bin"]);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stderr, "");
|
||||
assert_eq!(stdout.lines().count(), 1, "one line per file: {stdout:?}");
|
||||
let keys: Vec<&str> = stdout
|
||||
.split_whitespace()
|
||||
.map(|pair| pair.split_once('=').expect("key=value pair").0)
|
||||
.collect();
|
||||
assert_eq!(keys, [
|
||||
"st_dev",
|
||||
"st_ino",
|
||||
"st_mode",
|
||||
"st_nlink",
|
||||
"st_uid",
|
||||
"st_gid",
|
||||
"st_rdev",
|
||||
"st_size",
|
||||
"st_atime",
|
||||
"st_mtime",
|
||||
"st_ctime",
|
||||
"st_birthtime",
|
||||
"st_blksize",
|
||||
"st_blocks",
|
||||
"st_flags",
|
||||
]);
|
||||
assert!(stdout.contains(" st_size=12 "), "unexpected stdout: {stdout:?}");
|
||||
let mode = stdout
|
||||
.split_whitespace()
|
||||
.find_map(|pair| pair.strip_prefix("st_mode="))
|
||||
.unwrap();
|
||||
assert!(mode.starts_with('0'), "octal mode with leading zero: {stdout:?}");
|
||||
assert!(u32::from_str_radix(mode, 8).is_ok(), "octal mode: {stdout:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bsd_verbose_format_prints_linux_like_block() {
|
||||
let (_dir, root) = canonical_tempdir();
|
||||
fs::write(root.join("data.bin"), b"hello world!").unwrap();
|
||||
|
||||
let (code, stdout, stderr) = run_in(root, vec!["-x", "data.bin"]);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stderr, "");
|
||||
assert!(stdout.starts_with(" File: \"data.bin\"\n"), "unexpected stdout: {stdout:?}");
|
||||
assert!(stdout.contains("FileType:"), "unexpected stdout: {stdout:?}");
|
||||
assert!(stdout.contains(" Mode: (0"), "unexpected stdout: {stdout:?}");
|
||||
// ctime(3)-style timestamps: "Access: Wed Aug 20 10:11:12 2026".
|
||||
let access = stdout.lines().find(|l| l.starts_with("Access: ")).unwrap();
|
||||
let year = access.rsplit(' ').next().unwrap();
|
||||
assert_eq!(year.len(), 4, "ctime-style year expected: {access:?}");
|
||||
assert!(year.parse::<u32>().is_ok(), "ctime-style year expected: {access:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bsd_dash_f_size_format_prints_size() {
|
||||
let (_dir, root) = canonical_tempdir();
|
||||
fs::write(root.join("data.bin"), b"hello world!").unwrap();
|
||||
|
||||
// Acceptance: BSD `stat -f '%z bytes' file`.
|
||||
let (code, stdout, stderr) = run_in(root, vec!["-f", "%z bytes", "data.bin"]);
|
||||
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "12 bytes\n", ""));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bsd_dash_t_timefmt_formats_times() {
|
||||
let (_dir, root) = canonical_tempdir();
|
||||
fs::write(root.join("data.bin"), b"x").unwrap();
|
||||
|
||||
// BSD `-t` supplies the strftime format for `%Sm`-style directives;
|
||||
// this used to be ignored with a warning.
|
||||
let (code, stdout, stderr) = run_in(root, vec!["-f", "%Sm", "-t", "%Y", "data.bin"]);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stderr, "");
|
||||
let year: u32 = stdout.trim_end().parse().expect("year only");
|
||||
assert!((1970..=9999).contains(&year), "unexpected stdout: {stdout:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gnu_filesystem_mode_keeps_existing_path_operands() {
|
||||
let (_dir, root) = canonical_tempdir();
|
||||
|
||||
// `stat -f <existing path>` stays GNU `--file-system` mode.
|
||||
let (code, stdout, stderr) = run_in(root, vec!["-f", "."]);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stderr, "");
|
||||
assert!(stdout.contains("Namelen:"), "filesystem block expected: {stdout:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bsd_dash_f_fallback_on_nonexistent_format_like_operand() {
|
||||
let (_dir, root) = canonical_tempdir();
|
||||
fs::write(root.join("data.bin"), b"x").unwrap();
|
||||
|
||||
// No `%` directive, but the operand names no file and looks like a
|
||||
// format string: BSD semantics print it literally instead of failing
|
||||
// with a filesystem error on a nonexistent operand.
|
||||
let (code, stdout, stderr) = run_in(root, vec!["-f", "no percent here", "data.bin"]);
|
||||
assert_eq!((code, stdout.as_str(), stderr.as_str()), (0, "no percent here\n", ""));
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(all(test, windows))]
|
||||
|
||||
+342
-107
@@ -2875,101 +2875,109 @@ pub(crate) fn map_output_error(error: io::Error) -> TailError {
|
||||
error.into()
|
||||
}
|
||||
|
||||
fn rewrite_tail_argv(mut argv: Vec<OsString>) -> Result<Vec<OsString>, String> {
|
||||
let mut has_reverse = false;
|
||||
let mut incompatible = false;
|
||||
let mut unsupported = None;
|
||||
/// True when `token` is an option that takes its value from the *next* argv
|
||||
/// token, so that value must never be mistaken for an obsolete `-N`/`+N` form
|
||||
/// (e.g. the `+5` in `tail -n +5 file`).
|
||||
fn consumes_separate_value(token: &str) -> bool {
|
||||
if let Some(long) = token.strip_prefix("--") {
|
||||
if long.is_empty() || long.contains('=') {
|
||||
return false;
|
||||
}
|
||||
// clap infers unambiguous long-option prefixes; `--follow` requires
|
||||
// `=` for its value and never consumes the next token.
|
||||
return ["lines", "bytes", "pid", "sleep-interval", "max-unchanged-stats"]
|
||||
.iter()
|
||||
.any(|name| name.starts_with(long));
|
||||
}
|
||||
let Some(cluster) = token.strip_prefix('-') else {
|
||||
return false;
|
||||
};
|
||||
let mut chars = cluster.chars();
|
||||
while let Some(c) = chars.next() {
|
||||
match c {
|
||||
// Value-taking shorts: a trailing `-n`/`-c`/`-s` consumes the next
|
||||
// token; anything after them in the cluster is an attached value.
|
||||
'n' | 'c' | 's' => return chars.next().is_none(),
|
||||
'q' | 'v' | 'z' | 'f' | 'F' | 'r' | 'b' => {},
|
||||
_ => return false,
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
for arg in argv.iter().skip(1) {
|
||||
/// Rewrites every obsolete `-N[bcl][f]` / `+N[bcl][f]` token (before `--`)
|
||||
/// into modern options, wherever it appears among flags and operands: GNU/BSD
|
||||
/// accept `tail -20 f1 f2`, `tail -f -5 file`, and `tail -5 -q file`.
|
||||
fn rewrite_tail_argv(argv: Vec<OsString>) -> Result<Vec<OsString>, String> {
|
||||
let mut rewritten = Vec::with_capacity(argv.len() + 2);
|
||||
let mut iter = argv.into_iter();
|
||||
rewritten.extend(iter.next());
|
||||
let mut follow = false;
|
||||
let mut has_operand = false;
|
||||
let mut skip_value = false;
|
||||
let mut seen_ddash = false;
|
||||
for arg in iter {
|
||||
if skip_value {
|
||||
skip_value = false;
|
||||
rewritten.push(arg);
|
||||
continue;
|
||||
}
|
||||
if seen_ddash {
|
||||
has_operand = true;
|
||||
rewritten.push(arg);
|
||||
continue;
|
||||
}
|
||||
let token = arg.to_string_lossy();
|
||||
if token == "--" {
|
||||
break;
|
||||
}
|
||||
let Some(cluster) = token.strip_prefix('-') else {
|
||||
continue;
|
||||
};
|
||||
if cluster.is_empty() {
|
||||
seen_ddash = true;
|
||||
rewritten.push(arg);
|
||||
continue;
|
||||
}
|
||||
if cluster.starts_with('-') {
|
||||
if token == "--reverse" {
|
||||
has_reverse = true;
|
||||
} else {
|
||||
unsupported = Some(token.into_owned());
|
||||
}
|
||||
continue;
|
||||
}
|
||||
for flag in cluster.chars() {
|
||||
match flag {
|
||||
'r' => has_reverse = true,
|
||||
'n' | 'c' | 'b' | 'f' => incompatible = true,
|
||||
_ => unsupported = Some(format!("-{flag}")),
|
||||
let bytes = token.as_bytes();
|
||||
// `+…` is always a candidate (`+10`, `+f`); `-…` only with a leading
|
||||
// digit (`-5`, `-20f`) so options like `-n` stay untouched.
|
||||
let candidate =
|
||||
bytes.first() == Some(&b'+') || matches!(bytes, [b'-', b'0'..=b'9', ..]);
|
||||
if candidate {
|
||||
match parse::parse_obsolete(&arg) {
|
||||
Some(Ok(obsolete)) => {
|
||||
follow |= obsolete.follow;
|
||||
rewritten.push(OsString::from(if obsolete.lines { "-n" } else { "-c" }));
|
||||
rewritten.push(OsString::from(format!(
|
||||
"{}{}",
|
||||
if obsolete.plus { "+" } else { "" },
|
||||
obsolete.num
|
||||
)));
|
||||
continue;
|
||||
},
|
||||
Some(Err(parse::ParseError::Context)) => {
|
||||
return Err(format!(
|
||||
"option used in invalid context -- {}",
|
||||
token.chars().nth(1).unwrap_or_default()
|
||||
));
|
||||
},
|
||||
Some(Err(parse::ParseError::InvalidEncoding)) => {
|
||||
return Err(format!("bad argument encoding: {}", arg.quote()));
|
||||
},
|
||||
None => {},
|
||||
}
|
||||
}
|
||||
if bytes.len() > 1 && bytes[0] == b'-' {
|
||||
skip_value = consumes_separate_value(&token);
|
||||
} else {
|
||||
has_operand = true;
|
||||
}
|
||||
rewritten.push(arg);
|
||||
}
|
||||
|
||||
if has_reverse && incompatible {
|
||||
return Err(
|
||||
"-r with -n, -c, -b, or -f is not supported by this builtin; pipe through tac"
|
||||
.to_owned(),
|
||||
if follow {
|
||||
// Obsolete `f` follows by name when a file operand is present,
|
||||
// matching GNU; insert up front so an explicit later -f/-F wins.
|
||||
rewritten.insert(
|
||||
1,
|
||||
OsString::from(if has_operand { "--follow=name" } else { "--follow=descriptor" }),
|
||||
);
|
||||
}
|
||||
if has_reverse {
|
||||
if let Some(option) = unsupported {
|
||||
return Err(format!(
|
||||
"-r with {option} is not supported by this builtin; pipe through tac"
|
||||
));
|
||||
}
|
||||
return Ok(argv);
|
||||
}
|
||||
|
||||
if argv.len() != 2 && argv.len() != 3 {
|
||||
return Ok(argv);
|
||||
}
|
||||
let clap_ok = args::uu_app()
|
||||
.try_get_matches_from(argv.clone())
|
||||
.is_ok_and(|matches| Settings::from(&matches).is_ok());
|
||||
let obsolete_token = argv[1].clone();
|
||||
let force_obsolete_blocks = obsolete_token
|
||||
.to_string_lossy()
|
||||
.strip_prefix('-')
|
||||
.is_some_and(|cluster| cluster.contains('b'));
|
||||
if clap_ok
|
||||
&& !force_obsolete_blocks
|
||||
&& !obsolete_token.to_string_lossy().starts_with('+')
|
||||
{
|
||||
return Ok(argv);
|
||||
}
|
||||
match parse::parse_obsolete(&obsolete_token) {
|
||||
Some(Ok(obsolete)) => {
|
||||
let mut rewritten = vec![argv.remove(0)];
|
||||
if obsolete.follow {
|
||||
rewritten.push(OsString::from(if argv.len() > 1 {
|
||||
"--follow=name"
|
||||
} else {
|
||||
"--follow=descriptor"
|
||||
}));
|
||||
}
|
||||
rewritten.push(OsString::from(if obsolete.lines { "-n" } else { "-c" }));
|
||||
rewritten.push(OsString::from(format!(
|
||||
"{}{}",
|
||||
if obsolete.plus { "+" } else { "" },
|
||||
obsolete.num
|
||||
)));
|
||||
if argv.len() > 1 {
|
||||
rewritten.push(argv.remove(1));
|
||||
}
|
||||
Ok(rewritten)
|
||||
},
|
||||
Some(Err(parse::ParseError::Context)) => Err(format!(
|
||||
"option used in invalid context -- {}",
|
||||
obsolete_token.to_string_lossy().chars().nth(1).unwrap_or_default()
|
||||
)),
|
||||
Some(Err(parse::ParseError::InvalidEncoding)) => {
|
||||
Err(format!("bad argument encoding: {}", obsolete_token.quote()))
|
||||
},
|
||||
None => Ok(argv),
|
||||
}
|
||||
Ok(rewritten)
|
||||
}
|
||||
/// Parsed `tail` invocation.
|
||||
pub(crate) struct Tail {
|
||||
@@ -2986,24 +2994,7 @@ impl Utility for Tail {
|
||||
|
||||
fn run(self, host: &mut Host) -> i32 {
|
||||
if self.matches.get_flag(args::options::REVERSE) {
|
||||
if self.matches.contains_id(args::options::LINES)
|
||||
|| self.matches.contains_id(args::options::BYTES)
|
||||
|| self.matches.get_flag(args::options::BLOCKS)
|
||||
|| self.matches.contains_id(args::options::FOLLOW)
|
||||
|| self.matches.get_flag(args::options::FOLLOW_RETRY)
|
||||
{
|
||||
let _ = writeln!(
|
||||
host.stderr,
|
||||
"tail: -r with -n, -c, -b, or -f is not supported by this builtin; pipe through tac"
|
||||
);
|
||||
return 1;
|
||||
}
|
||||
|
||||
let mut argv = vec![OsString::from("tac"), OsString::from("--")];
|
||||
if let Some(files) = self.matches.get_many::<OsString>(args::options::ARG_FILES) {
|
||||
argv.extend(files.cloned());
|
||||
}
|
||||
return crate::tac::run_argv(argv, host);
|
||||
return run_reverse(&self.matches, host);
|
||||
}
|
||||
let mut settings = match Settings::from(&self.matches) {
|
||||
Ok(settings) => settings,
|
||||
@@ -3032,6 +3023,154 @@ pub(crate) fn tail_builtin<SE: ShellExtensions>() -> Registration<SE> {
|
||||
util::<Tail, SE>()
|
||||
}
|
||||
|
||||
/// BSD `tail -r`: print lines in reverse order. With `-n N` the count selects
|
||||
/// how many lines to show (last N, or from line N for `+N`) before reversing.
|
||||
fn run_reverse(matches: &ArgMatches, host: &mut Host) -> i32 {
|
||||
if matches.contains_id(args::options::BYTES)
|
||||
|| matches.get_flag(args::options::BLOCKS)
|
||||
|| matches.contains_id(args::options::FOLLOW)
|
||||
|| matches.get_flag(args::options::FOLLOW_RETRY)
|
||||
{
|
||||
let _ = writeln!(
|
||||
host.stderr,
|
||||
"tail: -r with -c, -b, or -f is not supported by this builtin; pipe through tac"
|
||||
);
|
||||
return 1;
|
||||
}
|
||||
|
||||
let mut settings = match Settings::from(matches) {
|
||||
Ok(settings) => settings,
|
||||
Err(error) => {
|
||||
let _ = writeln!(host.stderr, "tail: {error}");
|
||||
return error.code();
|
||||
},
|
||||
};
|
||||
|
||||
// Without `-n`, `-r` reverses whole inputs; when no headers are wanted
|
||||
// that is exactly `tac`, so keep delegating.
|
||||
let all_lines = !matches.contains_id(args::options::LINES);
|
||||
if all_lines && !settings.verbose {
|
||||
let mut argv = vec![OsString::from("tac"), OsString::from("--")];
|
||||
if let Some(files) = matches.get_many::<OsString>(args::options::ARG_FILES) {
|
||||
argv.extend(files.cloned());
|
||||
}
|
||||
return crate::tac::run_argv(argv, host);
|
||||
}
|
||||
settings.resolve_paths(host);
|
||||
|
||||
match reverse_main(&settings, all_lines, host) {
|
||||
Ok(()) => host.exit_code(),
|
||||
Err(error) => {
|
||||
let code = error.code();
|
||||
if code != SIGPIPE_EXIT_CODE {
|
||||
let _ = writeln!(host.stderr, "tail: {error}");
|
||||
}
|
||||
code
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn reverse_main(settings: &Settings, all_lines: bool, host: &mut Host) -> TailResult<()> {
|
||||
let FilterMode::Lines(signum, sep) = &settings.mode else {
|
||||
unreachable!("-r with -c is rejected before dispatch");
|
||||
};
|
||||
let (signum, sep) = (*signum, *sep);
|
||||
let mut printer = HeaderPrinter::new(settings.verbose, true);
|
||||
for input in &settings.inputs {
|
||||
let path = match input.kind() {
|
||||
InputKind::File(path) if !(cfg!(unix) && path == &PathBuf::from(text::DEV_STDIN)) => {
|
||||
Some(path)
|
||||
},
|
||||
InputKind::File(_) | InputKind::Stdin => None,
|
||||
};
|
||||
let mut data = Vec::new();
|
||||
if let Some(path) = path {
|
||||
if path.is_dir() {
|
||||
host.fail(1);
|
||||
printer.print_input(input, &mut host.stdout);
|
||||
let _ = writeln!(
|
||||
host.stderr,
|
||||
"tail: error reading '{}': Is a directory",
|
||||
input.display_name
|
||||
);
|
||||
continue;
|
||||
}
|
||||
match File::open(path) {
|
||||
Ok(mut file) => {
|
||||
printer.print_input(input, &mut host.stdout);
|
||||
file.read_to_end(&mut data)?;
|
||||
},
|
||||
Err(error) if error.kind() == ErrorKind::NotFound => {
|
||||
host.fail(1);
|
||||
let _ = writeln!(
|
||||
host.stderr,
|
||||
"tail: cannot open '{}' for reading: No such file or directory",
|
||||
input.display_name
|
||||
);
|
||||
continue;
|
||||
},
|
||||
Err(error) => {
|
||||
host.fail(1);
|
||||
let _ = writeln!(
|
||||
host.stderr,
|
||||
"tail: cannot open '{}' for reading: {error}",
|
||||
input.display_name
|
||||
);
|
||||
continue;
|
||||
},
|
||||
}
|
||||
} else {
|
||||
printer.print_input(input, &mut host.stdout);
|
||||
host.stdin.read_to_end(&mut data)?;
|
||||
}
|
||||
write_reversed_lines(&data, signum, sep, all_lines, &mut host.stdout)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Writes the selected lines of `data` in reverse order, BSD `tail -r` style:
|
||||
/// each line keeps its trailing delimiter, so an unterminated final line leads
|
||||
/// the output without one (matching `tac`).
|
||||
fn write_reversed_lines(
|
||||
data: &[u8],
|
||||
signum: Signum,
|
||||
sep: u8,
|
||||
all_lines: bool,
|
||||
writer: &mut impl Write,
|
||||
) -> io::Result<()> {
|
||||
let mut segments: Vec<&[u8]> = Vec::new();
|
||||
let mut start = 0;
|
||||
for end in memchr_iter(sep, data) {
|
||||
segments.push(&data[start..=end]);
|
||||
start = end + 1;
|
||||
}
|
||||
if start < data.len() {
|
||||
segments.push(&data[start..]);
|
||||
}
|
||||
let keep: &[&[u8]] = if all_lines {
|
||||
&segments[..]
|
||||
} else {
|
||||
match signum {
|
||||
Signum::Negative(count) => {
|
||||
let count = usize::try_from(count).unwrap_or(usize::MAX);
|
||||
&segments[segments.len().saturating_sub(count)..]
|
||||
},
|
||||
Signum::MinusZero => &[],
|
||||
Signum::PlusZero => &segments[..],
|
||||
Signum::Positive(count) => {
|
||||
// GNU-style 1-based origin: `+1` (like `+0`) selects everything.
|
||||
let skip = usize::try_from(count.saturating_sub(1)).unwrap_or(usize::MAX);
|
||||
&segments[skip.min(segments.len())..]
|
||||
},
|
||||
}
|
||||
};
|
||||
let mut writer = BufWriter::new(writer);
|
||||
for segment in keep.iter().rev() {
|
||||
writer.write_all(segment)?;
|
||||
}
|
||||
writer.flush()
|
||||
}
|
||||
|
||||
fn tail_main(settings: &Settings, host: &mut Host) -> TailResult<()> {
|
||||
settings.check_warnings(&mut host.stderr);
|
||||
|
||||
@@ -3525,13 +3664,21 @@ where
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use std::{fs, io::Cursor};
|
||||
use std::{ffi::OsString, fs, io::Cursor};
|
||||
|
||||
use clap::Parser;
|
||||
|
||||
use super::{Tail, Utility, forwards_thru_file};
|
||||
use crate::host::{Host, run_util};
|
||||
|
||||
fn rewritten(argv: &[&str]) -> Vec<String> {
|
||||
Tail::rewrite_argv(argv.iter().map(OsString::from).collect())
|
||||
.unwrap()
|
||||
.into_iter()
|
||||
.map(|arg| arg.to_str().unwrap().to_owned())
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prints_last_line_from_stdin() {
|
||||
let (code, capture) = run_util::<Tail>(&["-n", "1"], "first\nlast\n", "/");
|
||||
@@ -3579,14 +3726,102 @@ mod tests {
|
||||
assert_eq!(parsed.run(&mut host), 0);
|
||||
}
|
||||
|
||||
// Failure mode: obsolete `-N`/`+N` was only rewritten for `argv.len()`
|
||||
// of 2 or 3 with the token at argv[1], so multi-file and flag-interleaved
|
||||
// invocations were clap parse errors.
|
||||
#[test]
|
||||
fn reverse_with_line_count_keeps_bsd_error() {
|
||||
let (code, capture) = run_util::<Tail>(&["-r", "-n", "2"], "", "/");
|
||||
fn obsolete_count_rewritten_at_any_position() {
|
||||
assert_eq!(rewritten(&["tail", "-20", "f1", "f2"]), ["tail", "-n", "20", "f1", "f2"]);
|
||||
assert_eq!(rewritten(&["tail", "-f", "-5", "f"]), ["tail", "-f", "-n", "5", "f"]);
|
||||
assert_eq!(rewritten(&["tail", "-5", "-q", "f"]), ["tail", "-n", "5", "-q", "f"]);
|
||||
assert_eq!(rewritten(&["tail", "+10", "f"]), ["tail", "-n", "+10", "f"]);
|
||||
assert_eq!(rewritten(&["tail", "-5c", "f"]), ["tail", "-c", "5", "f"]);
|
||||
// Obsolete `f` still maps to --follow=name with a file operand.
|
||||
assert_eq!(
|
||||
rewritten(&["tail", "-20f", "f"]),
|
||||
["tail", "--follow=name", "-n", "20", "f"]
|
||||
);
|
||||
assert_eq!(rewritten(&["tail", "-20f"]), ["tail", "--follow=descriptor", "-n", "20"]);
|
||||
}
|
||||
|
||||
// Failure mode: a `-N`/`+N` token that is really an option value or a
|
||||
// post-`--` operand must never be rewritten.
|
||||
#[test]
|
||||
fn option_values_and_post_ddash_operands_are_not_rewritten() {
|
||||
assert_eq!(rewritten(&["tail", "-n", "+5", "f"]), ["tail", "-n", "+5", "f"]);
|
||||
assert_eq!(rewritten(&["tail", "-c", "-5", "f"]), ["tail", "-c", "-5", "f"]);
|
||||
assert_eq!(rewritten(&["tail", "--lines", "-5", "f"]), ["tail", "--lines", "-5", "f"]);
|
||||
assert_eq!(rewritten(&["tail", "--", "-5"]), ["tail", "--", "-5"]);
|
||||
}
|
||||
|
||||
// Failure mode: `tail -20 f1 f2` was rejected outright; it must print the
|
||||
// last lines of every operand with GNU headers.
|
||||
#[test]
|
||||
fn obsolete_count_with_multiple_files_prints_headers() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
fs::write(dir.path().join("f1"), "a\nb\n").unwrap();
|
||||
fs::write(dir.path().join("f2"), "c\nd\n").unwrap();
|
||||
let (code, capture) = run_util::<Tail>(&["-1", "f1", "f2"], "", dir.path());
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "==> f1 <==\nb\n\n==> f2 <==\nd\n");
|
||||
assert_eq!(capture.err(), "");
|
||||
}
|
||||
|
||||
// Failure mode: `-r` with `-n N` was rejected; BSD tail shows the last N
|
||||
// lines in reverse order.
|
||||
#[test]
|
||||
fn reverse_with_line_count_takes_last_lines_reversed() {
|
||||
let (code, capture) = run_util::<Tail>(&["-r", "-n", "2"], "a\nb\nc\n", "/");
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "c\nb\n");
|
||||
assert_eq!(capture.err(), "");
|
||||
}
|
||||
|
||||
// Failure mode: `-rq` was rejected; with `-q` headers stay suppressed
|
||||
// while each file's selection is reversed independently.
|
||||
#[test]
|
||||
fn reverse_quiet_suppresses_headers_across_files() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
fs::write(dir.path().join("f1"), "a\nb\n").unwrap();
|
||||
fs::write(dir.path().join("f2"), "c\nd\n").unwrap();
|
||||
let (code, capture) = run_util::<Tail>(&["-rq", "-n", "2", "f1", "f2"], "", dir.path());
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "b\na\nd\nc\n");
|
||||
assert_eq!(capture.err(), "");
|
||||
}
|
||||
|
||||
// Multi-file reverse keeps GNU-style headers.
|
||||
#[test]
|
||||
fn reverse_with_multiple_files_prints_headers() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
fs::write(dir.path().join("f1"), "a\nb\n").unwrap();
|
||||
fs::write(dir.path().join("f2"), "c\nd\n").unwrap();
|
||||
let (code, capture) = run_util::<Tail>(&["-r", "-n", "1", "f1", "f2"], "", dir.path());
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "==> f1 <==\nb\n\n==> f2 <==\nd\n");
|
||||
assert_eq!(capture.err(), "");
|
||||
}
|
||||
|
||||
// Failure mode: obsolete `-N` combined with `-r` (`tail -r -5`) must feed
|
||||
// the rewritten count into the reverse path.
|
||||
#[test]
|
||||
fn reverse_with_obsolete_count() {
|
||||
let (code, capture) = run_util::<Tail>(&["-r", "-2"], "a\nb\nc\n", "/");
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(capture.out(), "c\nb\n");
|
||||
assert_eq!(capture.err(), "");
|
||||
}
|
||||
|
||||
// `-r` with byte/block counts stays an explicit error rather than
|
||||
// silently diverging from BSD semantics.
|
||||
#[test]
|
||||
fn reverse_with_byte_count_keeps_clear_error() {
|
||||
let (code, capture) = run_util::<Tail>(&["-r", "-c", "5"], "", "/");
|
||||
assert_eq!(code, 1);
|
||||
assert_eq!(capture.out(), "");
|
||||
assert_eq!(
|
||||
capture.err(),
|
||||
"tail: -r with -n, -c, -b, or -f is not supported by this builtin; pipe through tac\n"
|
||||
"tail: -r with -c, -b, or -f is not supported by this builtin; pipe through tac\n"
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
@@ -1,9 +1,14 @@
|
||||
//! `timeout` builtin, moved from `pi-shell`.
|
||||
|
||||
use std::{future::Future, io::Write, time::Duration};
|
||||
use std::{
|
||||
io::Write,
|
||||
sync::{Arc, Mutex},
|
||||
time::Duration,
|
||||
};
|
||||
|
||||
use brush_core::{
|
||||
ExecutionContext, ExecutionExitCode, ExecutionResult, ProcessGroupPolicy, SourceInfo, builtins,
|
||||
ExecutionContext, ExecutionExitCode, ExecutionResult, ProcessGroupPolicy, SourceInfo,
|
||||
SpawnObserver, builtins, sys, traps::TrapSignal,
|
||||
};
|
||||
use clap::Parser;
|
||||
use tokio::time;
|
||||
@@ -11,82 +16,338 @@ use tokio_util::sync::CancellationToken;
|
||||
|
||||
use crate::host::{parse_duration, quote_arg};
|
||||
|
||||
/// GNU timeout's exit status for its own usage/internal errors.
|
||||
const EXIT_TIMEOUT_FAILURE: u8 = 125;
|
||||
/// GNU timeout's exit status when the time limit expired.
|
||||
const EXIT_TIMED_OUT: u8 = 124;
|
||||
/// 128 + SIGKILL(9): reported when the command died from SIGKILL.
|
||||
const EXIT_KILLED: u8 = 137;
|
||||
|
||||
/// Run a command with a time limit.
|
||||
#[derive(Parser)]
|
||||
#[command(disable_help_flag = true)]
|
||||
pub(crate) struct TimeoutCommand {
|
||||
#[arg(required = true)]
|
||||
struct TimeoutArgs {
|
||||
/// Signal to send on expiry: a name with or without the `SIG` prefix, or
|
||||
/// a number. Defaults to TERM.
|
||||
#[arg(short = 's', long = "signal", value_name = "SIGNAL")]
|
||||
signal: Option<String>,
|
||||
/// Also send SIGKILL if the command is still running this long after the
|
||||
/// initial signal.
|
||||
#[arg(short = 'k', long = "kill-after", value_name = "DURATION")]
|
||||
kill_after: Option<String>,
|
||||
/// Exit with the command's own status even when the time limit expired.
|
||||
#[arg(long)]
|
||||
preserve_status: bool,
|
||||
/// GNU compatibility: don't put the command in a separate process group,
|
||||
/// and signal only the direct children rather than a whole group.
|
||||
#[arg(long)]
|
||||
foreground: bool,
|
||||
/// Diagnose each signal sent to the command on stderr.
|
||||
#[arg(short = 'v', long)]
|
||||
verbose: bool,
|
||||
// Hyphenated operands must reach `parse_duration` so `timeout -1 cmd`
|
||||
// reports an invalid time interval (exit 125) like GNU, instead of a
|
||||
// clap unknown-option error.
|
||||
#[arg(required = true, allow_hyphen_values = true)]
|
||||
duration: String,
|
||||
#[arg(required = true, num_args = 1.., trailing_var_arg = true)]
|
||||
// The command's own options belong to the command: `timeout 5 grep -v x`.
|
||||
#[arg(required = true, num_args = 1.., trailing_var_arg = true, allow_hyphen_values = true)]
|
||||
command: Vec<String>,
|
||||
}
|
||||
|
||||
/// Holds the raw argument vector so parse failures surface as GNU timeout's
|
||||
/// exit status 125, not brush's generic usage-error status 2 (which the
|
||||
/// default `builtins::Command::new` path would produce).
|
||||
pub(crate) struct TimeoutCommand {
|
||||
argv: Vec<String>,
|
||||
}
|
||||
|
||||
impl clap::FromArgMatches for TimeoutCommand {
|
||||
fn from_arg_matches(_matches: &clap::ArgMatches) -> Result<Self, clap::Error> {
|
||||
Ok(Self { argv: Vec::new() })
|
||||
}
|
||||
|
||||
fn update_from_arg_matches(&mut self, _matches: &clap::ArgMatches) -> Result<(), clap::Error> {
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
impl clap::CommandFactory for TimeoutCommand {
|
||||
fn command() -> clap::Command {
|
||||
<TimeoutArgs as clap::CommandFactory>::command()
|
||||
}
|
||||
|
||||
fn command_for_update() -> clap::Command {
|
||||
<TimeoutArgs as clap::CommandFactory>::command_for_update()
|
||||
}
|
||||
}
|
||||
|
||||
impl clap::Parser for TimeoutCommand {}
|
||||
|
||||
/// Records the external children spawned while running the timed command.
|
||||
///
|
||||
/// brush's cancellation token can only SIGKILL a child (see
|
||||
/// `brush_core::processes::Process::wait`), so delivering the *configured*
|
||||
/// signal requires knowing the child's pid/pgid; the shell reports those
|
||||
/// through its [`SpawnObserver`] hook.
|
||||
#[derive(Default)]
|
||||
struct SpawnRecorder(Mutex<Vec<(i32, Option<i32>)>>);
|
||||
|
||||
impl SpawnObserver for SpawnRecorder {
|
||||
fn on_spawn(&self, pid: i32, pgid: Option<i32>) {
|
||||
if let Ok(mut spawns) = self.0.lock() {
|
||||
spawns.push((pid, pgid));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl SpawnRecorder {
|
||||
/// Sends `signal` to every recorded child — its whole process group when
|
||||
/// `group` is set — and reports whether any delivery succeeded.
|
||||
fn signal(&self, signal: TrapSignal, group: bool) -> bool {
|
||||
let spawns = match self.0.lock() {
|
||||
Ok(spawns) => spawns.clone(),
|
||||
Err(_) => return false,
|
||||
};
|
||||
let mut sent = false;
|
||||
for (pid, pgid) in spawns {
|
||||
let target = match pgid {
|
||||
Some(pgid) if group => -pgid,
|
||||
_ => pid,
|
||||
};
|
||||
if sys::signal::kill_process(target, signal).is_ok() {
|
||||
sent = true;
|
||||
}
|
||||
}
|
||||
sent
|
||||
}
|
||||
}
|
||||
|
||||
/// Parses a `-s` operand: a signal name (with or without the `SIG` prefix,
|
||||
/// any case) or a signal number.
|
||||
fn parse_signal(spec: &str) -> Option<TrapSignal> {
|
||||
let parsed = if let Ok(number) = spec.trim().parse::<i32>() {
|
||||
TrapSignal::try_from(number).ok()?
|
||||
} else {
|
||||
TrapSignal::try_from(spec).ok()?
|
||||
};
|
||||
// Only real signals can be delivered; EXIT/DEBUG/ERR are shell traps.
|
||||
matches!(parsed, TrapSignal::Signal(_)).then_some(parsed)
|
||||
}
|
||||
|
||||
/// Renders a signal the way GNU timeout's diagnostics do: `TERM`, not `SIGTERM`.
|
||||
fn signal_display(signal: TrapSignal) -> &'static str {
|
||||
let name = signal.as_str();
|
||||
name.strip_prefix("SIG").unwrap_or(name)
|
||||
}
|
||||
|
||||
impl builtins::Command for TimeoutCommand {
|
||||
type Error = brush_core::Error;
|
||||
|
||||
fn execute<SE: brush_core::ShellExtensions>(
|
||||
fn new<I>(args: I) -> Result<Self, clap::Error>
|
||||
where
|
||||
I: IntoIterator<Item = String>,
|
||||
{
|
||||
Ok(Self { argv: args.into_iter().collect() })
|
||||
}
|
||||
|
||||
async fn execute<SE: brush_core::ShellExtensions>(
|
||||
&self,
|
||||
context: ExecutionContext<'_, SE>,
|
||||
) -> impl Future<Output = std::result::Result<ExecutionResult, brush_core::Error>> + Send {
|
||||
let duration = self.duration.clone();
|
||||
let command = self.command.clone();
|
||||
async move {
|
||||
if context.is_cancelled() {
|
||||
return Ok(ExecutionExitCode::Interrupted.into());
|
||||
}
|
||||
let Some(timeout) = parse_duration(&duration) else {
|
||||
let _ = writeln!(context.stderr(), "timeout: invalid time interval '{duration}'");
|
||||
return Ok(ExecutionResult::new(125));
|
||||
};
|
||||
if command.is_empty() {
|
||||
let _ = writeln!(context.stderr(), "timeout: missing command");
|
||||
return Ok(ExecutionResult::new(125));
|
||||
}
|
||||
|
||||
let child_cancel = CancellationToken::new();
|
||||
let mut params = context.params.clone();
|
||||
params.process_group_policy = ProcessGroupPolicy::NewProcessGroup;
|
||||
params.set_cancel_token(child_cancel.clone());
|
||||
|
||||
let mut command_line = String::new();
|
||||
for (idx, arg) in command.iter().enumerate() {
|
||||
if idx > 0 {
|
||||
command_line.push(' ');
|
||||
}
|
||||
command_line.push_str("e_arg(arg));
|
||||
}
|
||||
|
||||
let cancel_token = context.cancel_token();
|
||||
let source_info = SourceInfo::from("pi-natives:timeout");
|
||||
let run_future = context.shell.run_string(command_line, &source_info, ¶ms);
|
||||
tokio::pin!(run_future);
|
||||
|
||||
if let Some(cancel_token) = cancel_token {
|
||||
tokio::select! {
|
||||
result = &mut run_future => result,
|
||||
() = time::sleep(timeout) => {
|
||||
child_cancel.cancel();
|
||||
// Wait briefly for the child to exit after cancellation.
|
||||
let _ = time::timeout(Duration::from_secs(2), &mut run_future).await;
|
||||
Ok(ExecutionResult::new(124))
|
||||
},
|
||||
() = cancel_token.cancelled() => {
|
||||
child_cancel.cancel();
|
||||
Ok(ExecutionExitCode::Interrupted.into())
|
||||
},
|
||||
}
|
||||
} else {
|
||||
tokio::select! {
|
||||
result = &mut run_future => result,
|
||||
() = time::sleep(timeout) => {
|
||||
child_cancel.cancel();
|
||||
// Wait briefly for the child to exit after cancellation.
|
||||
let _ = time::timeout(Duration::from_secs(2), &mut run_future).await;
|
||||
Ok(ExecutionResult::new(124))
|
||||
},
|
||||
}
|
||||
}
|
||||
) -> std::result::Result<ExecutionResult, brush_core::Error> {
|
||||
if context.is_cancelled() {
|
||||
return Ok(ExecutionExitCode::Interrupted.into());
|
||||
}
|
||||
let args = match TimeoutArgs::try_parse_from(&self.argv) {
|
||||
Ok(args) => args,
|
||||
Err(err) => {
|
||||
// clap reports `--help` as an error; that belongs on stdout
|
||||
// with a success status, real usage errors exit 125.
|
||||
let rendered = err.to_string();
|
||||
if err.use_stderr() {
|
||||
let _ = write!(context.stderr(), "{rendered}");
|
||||
return Ok(ExecutionResult::new(EXIT_TIMEOUT_FAILURE));
|
||||
}
|
||||
let _ = write!(context.stdout(), "{rendered}");
|
||||
return Ok(ExecutionResult::success());
|
||||
},
|
||||
};
|
||||
let Some(limit) = parse_duration(&args.duration) else {
|
||||
let _ = writeln!(
|
||||
context.stderr(),
|
||||
"timeout: invalid time interval '{}'",
|
||||
args.duration
|
||||
);
|
||||
return Ok(ExecutionResult::new(EXIT_TIMEOUT_FAILURE));
|
||||
};
|
||||
let kill_after = match &args.kill_after {
|
||||
Some(spec) => match parse_duration(spec) {
|
||||
Some(duration) => Some(duration),
|
||||
None => {
|
||||
let _ =
|
||||
writeln!(context.stderr(), "timeout: invalid time interval '{spec}'");
|
||||
return Ok(ExecutionResult::new(EXIT_TIMEOUT_FAILURE));
|
||||
},
|
||||
},
|
||||
None => None,
|
||||
};
|
||||
let signal = match &args.signal {
|
||||
Some(spec) => match parse_signal(spec) {
|
||||
Some(signal) => signal,
|
||||
None => {
|
||||
let _ = writeln!(context.stderr(), "timeout: '{spec}': invalid signal");
|
||||
return Ok(ExecutionResult::new(EXIT_TIMEOUT_FAILURE));
|
||||
},
|
||||
},
|
||||
None => TrapSignal::try_from("TERM").expect("SIGTERM must be a known signal"),
|
||||
};
|
||||
|
||||
let child_cancel = CancellationToken::new();
|
||||
let spawns = Arc::new(SpawnRecorder::default());
|
||||
let mut params = context.params.clone();
|
||||
// GNU runs the command in its own process group and signals the whole
|
||||
// group; `--foreground` keeps it in the invoking group and signals
|
||||
// only the direct children.
|
||||
params.process_group_policy = if args.foreground {
|
||||
ProcessGroupPolicy::SameProcessGroup
|
||||
} else {
|
||||
ProcessGroupPolicy::NewProcessGroup
|
||||
};
|
||||
params.set_cancel_token(child_cancel.clone());
|
||||
params.set_spawn_observer(Arc::clone(&spawns) as Arc<dyn SpawnObserver>);
|
||||
|
||||
let mut command_line = String::new();
|
||||
for (idx, arg) in args.command.iter().enumerate() {
|
||||
if idx > 0 {
|
||||
command_line.push(' ');
|
||||
}
|
||||
command_line.push_str("e_arg(arg));
|
||||
}
|
||||
|
||||
// Grab an owned stderr handle up front: `run_future` below holds the
|
||||
// shell mutably, so `context.stderr()` is unavailable once it exists.
|
||||
let mut stderr = context.stderr();
|
||||
let outer_cancel = context.cancel_token();
|
||||
let source_info = SourceInfo::from("pi-natives:timeout");
|
||||
let run_future = context.shell.run_string(command_line, &source_info, ¶ms);
|
||||
tokio::pin!(run_future);
|
||||
|
||||
let outer_cancelled = async {
|
||||
match &outer_cancel {
|
||||
Some(token) => token.cancelled().await,
|
||||
None => std::future::pending().await,
|
||||
}
|
||||
};
|
||||
tokio::pin!(outer_cancelled);
|
||||
|
||||
// GNU: a duration of zero disables the timeout entirely.
|
||||
let deadline = async {
|
||||
if limit.is_zero() {
|
||||
std::future::pending::<()>().await;
|
||||
} else {
|
||||
time::sleep(limit).await;
|
||||
}
|
||||
};
|
||||
tokio::pin!(deadline);
|
||||
|
||||
tokio::select! {
|
||||
result = &mut run_future => return result,
|
||||
() = &mut outer_cancelled => {
|
||||
child_cancel.cancel();
|
||||
return Ok(ExecutionExitCode::Interrupted.into());
|
||||
},
|
||||
() = &mut deadline => {},
|
||||
}
|
||||
|
||||
// The limit expired: deliver the configured signal like GNU timeout.
|
||||
if args.verbose {
|
||||
let _ = writeln!(
|
||||
stderr,
|
||||
"timeout: sending signal {} to command '{}'",
|
||||
signal_display(signal),
|
||||
args.command[0]
|
||||
);
|
||||
}
|
||||
let signalled = spawns.signal(signal, !args.foreground);
|
||||
if !signalled {
|
||||
// The operand ran in-process (a builtin, say) or the child is
|
||||
// already gone; cancellation is the only remaining lever. For
|
||||
// external children it degrades to SIGKILL — see `Process::wait`.
|
||||
child_cancel.cancel();
|
||||
}
|
||||
let mut killed = signal.as_str() == "SIGKILL";
|
||||
|
||||
// Wait for the command to finish, escalating to SIGKILL after
|
||||
// `--kill-after`. Without `-k`, GNU waits indefinitely — a command
|
||||
// that catches the signal keeps running (the caller can still cancel).
|
||||
// After a cancel-fallback (in-process operand), the inner shell may
|
||||
// surface its own cancellation as an Interrupted error instead of the
|
||||
// operand's result; that is expected retirement, not a fault.
|
||||
let reap = |result: Result<ExecutionResult, brush_core::Error>| match result {
|
||||
Ok(result) => Ok(Some(result)),
|
||||
Err(err) if !signalled && matches!(err.kind(), brush_core::ErrorKind::Interrupted) => {
|
||||
Ok(None)
|
||||
},
|
||||
Err(err) => Err(err),
|
||||
};
|
||||
let kill_deadline = async {
|
||||
match kill_after {
|
||||
Some(duration) => time::sleep(duration).await,
|
||||
None => std::future::pending().await,
|
||||
}
|
||||
};
|
||||
tokio::pin!(kill_deadline);
|
||||
|
||||
let child_result = tokio::select! {
|
||||
result = &mut run_future => reap(result)?,
|
||||
() = &mut outer_cancelled => {
|
||||
child_cancel.cancel();
|
||||
return Ok(ExecutionExitCode::Interrupted.into());
|
||||
},
|
||||
() = &mut kill_deadline => {
|
||||
if args.verbose {
|
||||
let _ = writeln!(
|
||||
stderr,
|
||||
"timeout: sending signal KILL to command '{}'",
|
||||
args.command[0]
|
||||
);
|
||||
}
|
||||
killed = true;
|
||||
let kill = TrapSignal::try_from("KILL").expect("SIGKILL must be a known signal");
|
||||
spawns.signal(kill, !args.foreground);
|
||||
child_cancel.cancel();
|
||||
// SIGKILL can't be resisted; bound the reaping wait anyway so
|
||||
// a wedged in-process operand can't hang the builtin forever.
|
||||
match time::timeout(Duration::from_secs(2), &mut run_future).await {
|
||||
Ok(result) => reap(result)?,
|
||||
Err(_) => None,
|
||||
}
|
||||
},
|
||||
};
|
||||
|
||||
// Exit status per GNU: the command's own status under
|
||||
// `--preserve-status`; 137 when it died from SIGKILL; else 124.
|
||||
if args.preserve_status {
|
||||
if signalled {
|
||||
return Ok(child_result.unwrap_or_else(|| ExecutionResult::new(EXIT_KILLED)));
|
||||
}
|
||||
// Cancel-fallback path (in-process operand): the inner shell's own
|
||||
// cancellation check races the operand's result, so its status is
|
||||
// unreliable. Report death by the delivered signal (128+N, or 137
|
||||
// after escalation) deterministically, matching GNU for a command
|
||||
// taken down by the timeout signal.
|
||||
let number = i32::try_from(signal).unwrap_or(15);
|
||||
let code = if killed {
|
||||
EXIT_KILLED
|
||||
} else {
|
||||
128_u8.wrapping_add(number as u8)
|
||||
};
|
||||
return Ok(ExecutionResult::new(code));
|
||||
}
|
||||
if killed {
|
||||
return Ok(ExecutionResult::new(EXIT_KILLED));
|
||||
}
|
||||
Ok(ExecutionResult::new(EXIT_TIMED_OUT))
|
||||
}
|
||||
}
|
||||
|
||||
@@ -104,7 +365,7 @@ mod tests {
|
||||
};
|
||||
use clap::Parser;
|
||||
|
||||
use super::TimeoutCommand;
|
||||
use super::{TimeoutArgs, TimeoutCommand, parse_signal, signal_display};
|
||||
|
||||
#[derive(Parser)]
|
||||
struct StatusCommand;
|
||||
@@ -142,6 +403,23 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
/// An operand that ignores cancellation, so only SIGKILL escalation (the
|
||||
/// cancel-token fallback plus the bounded reap) can retire it early.
|
||||
#[derive(Parser)]
|
||||
struct StubbornCommand;
|
||||
|
||||
impl builtins::Command for StubbornCommand {
|
||||
type Error = brush_core::Error;
|
||||
|
||||
async fn execute<SE: brush_core::ShellExtensions>(
|
||||
&self,
|
||||
_context: ExecutionContext<'_, SE>,
|
||||
) -> Result<ExecutionResult, Self::Error> {
|
||||
tokio::time::sleep(Duration::from_millis(300)).await;
|
||||
Ok(ExecutionResult::new(99))
|
||||
}
|
||||
}
|
||||
|
||||
async fn test_shell() -> Shell<DefaultShellExtensions> {
|
||||
Shell::builder()
|
||||
.builtin(
|
||||
@@ -220,4 +498,185 @@ mod tests {
|
||||
assert_eq!(u8::from(result.exit_code), 125);
|
||||
assert_eq!(diagnostic, "timeout: invalid time interval 'invalid'\n");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn gnu_flag_spellings_parse() {
|
||||
// Failure mode: a real-world GNU invocation dying in clap.
|
||||
let args = TimeoutArgs::try_parse_from([
|
||||
"timeout", "-s", "INT", "-k", "2s", "10s", "cmd", "arg",
|
||||
])
|
||||
.expect("-s/-k spellings must parse");
|
||||
assert_eq!(args.signal.as_deref(), Some("INT"));
|
||||
assert_eq!(args.kill_after.as_deref(), Some("2s"));
|
||||
assert_eq!(args.duration, "10s");
|
||||
assert_eq!(args.command, ["cmd", "arg"]);
|
||||
|
||||
let args = TimeoutArgs::try_parse_from([
|
||||
"timeout",
|
||||
"--signal=KILL",
|
||||
"--kill-after=1",
|
||||
"--preserve-status",
|
||||
"--foreground",
|
||||
"-v",
|
||||
"5",
|
||||
"cmd",
|
||||
])
|
||||
.expect("long spellings must parse");
|
||||
assert!(args.preserve_status && args.foreground && args.verbose);
|
||||
|
||||
// `--` before the duration ends option parsing, GNU-style.
|
||||
let args = TimeoutArgs::try_parse_from(["timeout", "--", "5", "cmd"])
|
||||
.expect("-- before the duration must parse");
|
||||
assert_eq!(args.duration, "5");
|
||||
|
||||
// The command's own options must pass through untouched.
|
||||
let args = TimeoutArgs::try_parse_from(["timeout", "5", "grep", "-v", "-e", "x"])
|
||||
.expect("command options must not be parsed by timeout");
|
||||
assert_eq!(args.command, ["grep", "-v", "-e", "x"]);
|
||||
|
||||
// A hyphenated duration reaches parse_duration (exit 125 later),
|
||||
// instead of failing as an unknown clap option.
|
||||
let args = TimeoutArgs::try_parse_from(["timeout", "-1", "cmd"])
|
||||
.expect("hyphenated duration must reach the interval check");
|
||||
assert_eq!(args.duration, "-1");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn signal_spellings_parse_and_display_without_prefix() {
|
||||
// Failure mode: rejecting a signal spelling GNU accepts.
|
||||
for spec in ["TERM", "term", "SIGTERM", "sigterm", "15", "KILL", "9", "INT", "2"] {
|
||||
assert!(parse_signal(spec).is_some(), "spec {spec:?} must parse");
|
||||
}
|
||||
// Shell-trap pseudo-signals and unknown names are invalid for kill(2).
|
||||
for spec in ["NOSUCH", "EXIT", "DEBUG", "ERR", "64", "-5"] {
|
||||
assert!(parse_signal(spec).is_none(), "spec {spec:?} must be rejected");
|
||||
}
|
||||
let term = parse_signal("SIGTERM").expect("SIGTERM parses");
|
||||
assert_eq!(signal_display(term), "TERM");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn zero_duration_disables_the_timeout() {
|
||||
// Failure mode: `timeout 0 cmd` firing instantly instead of never.
|
||||
let result = run_with_deadline("timeout 0 slow-test").await;
|
||||
assert_eq!(u8::from(result.exit_code), 99);
|
||||
|
||||
let result = run_with_deadline("timeout 0s status-test").await;
|
||||
assert_eq!(u8::from(result.exit_code), 7);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn preserve_status_reports_death_by_the_timeout_signal() {
|
||||
// In-process operands retire via the cancel fallback, where the inner
|
||||
// shell's result is racy; --preserve-status must deterministically
|
||||
// report death by the configured signal (TERM -> 143), like GNU does
|
||||
// for a command killed by the timeout signal.
|
||||
let result = run_with_deadline("timeout --preserve-status 0.010 slow-test").await;
|
||||
|
||||
assert_eq!(u8::from(result.exit_code), 143);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn kill_signal_reports_137() {
|
||||
// GNU exits 128+9 when the command is taken down with SIGKILL.
|
||||
let result = run_with_deadline("timeout -s KILL 0.010 slow-test").await;
|
||||
|
||||
assert_eq!(u8::from(result.exit_code), 137);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn kill_after_escalates_and_reports_137() {
|
||||
// stubborn-test ignores the initial (cancellation-based) signal; the
|
||||
// -k deadline must escalate and report the SIGKILL status.
|
||||
let mut shell = test_shell().await;
|
||||
shell
|
||||
.register_builtin(
|
||||
"stubborn-test",
|
||||
builtins::builtin::<StubbornCommand, DefaultShellExtensions>(),
|
||||
);
|
||||
let mut params = shell.default_exec_params();
|
||||
// Keep the inner shell's interrupted notice off the test runner's
|
||||
// terminal, like run_with_deadline does.
|
||||
for fd in [OpenFiles::STDIN_FD, OpenFiles::STDOUT_FD, OpenFiles::STDERR_FD] {
|
||||
params.set_fd(fd, brush_core::openfiles::null().expect("null device"));
|
||||
}
|
||||
let result = tokio::time::timeout(
|
||||
Duration::from_secs(1),
|
||||
shell.run_string(
|
||||
"timeout -k 0.075 0.010 stubborn-test",
|
||||
&SourceInfo::default(),
|
||||
¶ms,
|
||||
),
|
||||
)
|
||||
.await
|
||||
.expect("escalation test exceeded its safety deadline")
|
||||
.expect("execute escalation command");
|
||||
|
||||
assert_eq!(u8::from(result.exit_code), 137);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn clap_parse_failure_exits_125_not_2() {
|
||||
// GNU usage errors exit 125; brush's generic clap path would exit 2.
|
||||
let result = run_with_deadline("timeout").await;
|
||||
|
||||
assert_eq!(u8::from(result.exit_code), 125);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn invalid_kill_after_interval_exits_125() {
|
||||
let result = run_with_deadline("timeout -k bogus 1 status-test").await;
|
||||
|
||||
assert_eq!(u8::from(result.exit_code), 125);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn invalid_signal_exits_125() {
|
||||
let result = run_with_deadline("timeout -s NOSUCH 1 status-test").await;
|
||||
|
||||
assert_eq!(u8::from(result.exit_code), 125);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn verbose_reports_the_signal_sent() {
|
||||
let mut shell = test_shell().await;
|
||||
let mut stderr = tempfile::tempfile().expect("create stderr capture");
|
||||
let mut params = shell.default_exec_params();
|
||||
params.set_fd(
|
||||
OpenFiles::STDERR_FD,
|
||||
stderr.try_clone().expect("clone stderr capture").into(),
|
||||
);
|
||||
|
||||
let result = tokio::time::timeout(
|
||||
Duration::from_secs(1),
|
||||
shell.run_string("timeout -v 0.010 slow-test", &SourceInfo::default(), ¶ms),
|
||||
)
|
||||
.await
|
||||
.expect("verbose test exceeded its safety deadline")
|
||||
.expect("execute verbose command");
|
||||
stderr.seek(SeekFrom::Start(0)).expect("rewind stderr capture");
|
||||
let mut diagnostic = String::new();
|
||||
stderr.read_to_string(&mut diagnostic).expect("read stderr capture");
|
||||
|
||||
assert_eq!(u8::from(result.exit_code), 124);
|
||||
// The cancel-fallback may race the inner shell's own cancellation
|
||||
// check, which can append its interrupted notice after our line.
|
||||
assert!(
|
||||
diagnostic.starts_with("timeout: sending signal TERM to command 'slow-test'\n"),
|
||||
"diagnostic must lead with the GNU-style signal line: {diagnostic:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn external_child_receives_the_configured_signal() {
|
||||
// Failure mode: the timed-out external child only ever seeing SIGKILL
|
||||
// (the cancel-token path) instead of the configured signal.
|
||||
let result = run_with_deadline("timeout -s TERM 0.050 /bin/sleep 5").await;
|
||||
assert_eq!(u8::from(result.exit_code), 124);
|
||||
|
||||
let result =
|
||||
run_with_deadline("timeout --preserve-status 0.050 /bin/sleep 5").await;
|
||||
assert_eq!(u8::from(result.exit_code), 143, "SIGTERM death is 128+15");
|
||||
}
|
||||
}
|
||||
|
||||
+223
-44
@@ -26,6 +26,80 @@ enum TopSortKey {
|
||||
Time,
|
||||
}
|
||||
|
||||
/// Column keys accepted by macOS-style `-stats` (comma-separated).
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq, clap::ValueEnum)]
|
||||
enum TopStat {
|
||||
Pid,
|
||||
#[value(alias = "uid")]
|
||||
User,
|
||||
#[value(name = "pstate", alias = "state")]
|
||||
State,
|
||||
#[value(name = "nice", alias = "ni")]
|
||||
Nice,
|
||||
#[value(name = "th", alias = "threads")]
|
||||
Threads,
|
||||
#[value(name = "vsize", alias = "virt")]
|
||||
Virt,
|
||||
#[value(name = "mem", alias = "rsize", alias = "res")]
|
||||
Res,
|
||||
#[value(alias = "time+")]
|
||||
Time,
|
||||
#[value(alias = "%cpu")]
|
||||
Cpu,
|
||||
#[value(name = "%mem", alias = "pmem")]
|
||||
PctMem,
|
||||
#[value(alias = "comm")]
|
||||
Command,
|
||||
}
|
||||
|
||||
/// Column order used when `-stats` is not given.
|
||||
const DEFAULT_TOP_STATS: &[TopStat] = &[
|
||||
TopStat::Pid,
|
||||
TopStat::User,
|
||||
TopStat::State,
|
||||
TopStat::Nice,
|
||||
TopStat::Threads,
|
||||
TopStat::Virt,
|
||||
TopStat::Res,
|
||||
TopStat::Time,
|
||||
TopStat::Cpu,
|
||||
TopStat::PctMem,
|
||||
TopStat::Command,
|
||||
];
|
||||
|
||||
impl TopStat {
|
||||
fn header(self) -> &'static str {
|
||||
match self {
|
||||
Self::Pid => "PID",
|
||||
Self::User => "USER",
|
||||
Self::State => "S",
|
||||
Self::Nice => "NI",
|
||||
Self::Threads => "TH",
|
||||
Self::Virt => "VIRT",
|
||||
Self::Res => "RES",
|
||||
Self::Time => "TIME+",
|
||||
Self::Cpu => "%CPU",
|
||||
Self::PctMem => "%MEM",
|
||||
Self::Command => "COMMAND",
|
||||
}
|
||||
}
|
||||
|
||||
/// Right-align width; 0 renders as-is (used for the free-form command).
|
||||
fn width(self) -> usize {
|
||||
match self {
|
||||
Self::Pid => 7,
|
||||
Self::User => 8,
|
||||
Self::State => 2,
|
||||
Self::Nice => 3,
|
||||
Self::Threads => 4,
|
||||
Self::Virt | Self::Res => 9,
|
||||
Self::Time => 10,
|
||||
Self::Cpu | Self::PctMem => 4,
|
||||
Self::Command => 0,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Display processes.
|
||||
#[derive(Parser)]
|
||||
#[command(name = "top", version, about = "Display processes", disable_help_flag = false)]
|
||||
@@ -80,6 +154,10 @@ pub(crate) struct TopCommand {
|
||||
/// Show the complete command line instead of the executable name.
|
||||
#[arg(short = 'c', long = "full-command")]
|
||||
full_command: bool,
|
||||
|
||||
/// Columns to display, in order (comma-separated, macOS `-stats` style).
|
||||
#[arg(long = "stats", value_enum, value_delimiter = ',', ignore_case = true)]
|
||||
stats: Vec<TopStat>,
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
@@ -96,9 +174,46 @@ struct TopProcessRow {
|
||||
command: String,
|
||||
}
|
||||
|
||||
/// Long option names `top` accepts with a macOS-style single dash.
|
||||
const TOP_LONG_OPTIONS: &[&str] = &[
|
||||
"batch",
|
||||
"samples",
|
||||
"iterations",
|
||||
"delay",
|
||||
"rows",
|
||||
"pid",
|
||||
"user",
|
||||
"sort",
|
||||
"full-command",
|
||||
"stats",
|
||||
"help",
|
||||
"version",
|
||||
];
|
||||
|
||||
/// Rewrites macOS-style single-dash long options (`-pid`, `-stats pid,cpu`)
|
||||
/// into clap-style `--` options; everything else passes through untouched.
|
||||
fn normalize_top_flag(arg: String) -> String {
|
||||
if let Some(rest) = arg.strip_prefix('-')
|
||||
&& !rest.starts_with('-')
|
||||
{
|
||||
let name = rest.split('=').next().unwrap_or(rest);
|
||||
if name.len() > 1 && TOP_LONG_OPTIONS.contains(&name) {
|
||||
return format!("-{arg}");
|
||||
}
|
||||
}
|
||||
arg
|
||||
}
|
||||
|
||||
impl builtins::Command for TopCommand {
|
||||
type Error = brush_core::Error;
|
||||
|
||||
fn new<I>(args: I) -> Result<Self, clap::Error>
|
||||
where
|
||||
I: IntoIterator<Item = String>,
|
||||
{
|
||||
Self::try_parse_from(args.into_iter().map(normalize_top_flag))
|
||||
}
|
||||
|
||||
fn execute<SE: brush_core::ShellExtensions>(
|
||||
&self,
|
||||
context: ExecutionContext<'_, SE>,
|
||||
@@ -111,6 +226,11 @@ impl builtins::Command for TopCommand {
|
||||
let sort = self.sort;
|
||||
let full_command = self.full_command;
|
||||
let _ = self.batch;
|
||||
let stats = if self.stats.is_empty() {
|
||||
DEFAULT_TOP_STATS.to_vec()
|
||||
} else {
|
||||
self.stats.clone()
|
||||
};
|
||||
async move {
|
||||
if !delay.is_finite() || delay < 0.0 || delay > Duration::MAX.as_secs_f64() {
|
||||
writeln!(context.stderr(), "top: invalid delay '{delay}'")?;
|
||||
@@ -200,7 +320,7 @@ impl builtins::Command for TopCommand {
|
||||
}
|
||||
|
||||
sort_top_rows(&mut rows, sort);
|
||||
let output = render_top_snapshot(&rows, row_limit, sample + 1);
|
||||
let output = render_top_snapshot(&rows, row_limit, sample + 1, &stats);
|
||||
if let Err(err) = write!(context.stdout(), "{output}") {
|
||||
if err.kind() == io::ErrorKind::BrokenPipe {
|
||||
return Ok(ExecutionResult::success());
|
||||
@@ -245,7 +365,46 @@ fn sort_top_rows(rows: &mut [TopProcessRow], key: TopSortKey) {
|
||||
});
|
||||
}
|
||||
|
||||
fn render_top_snapshot(rows: &[TopProcessRow], row_limit: Option<usize>, sample: u64) -> String {
|
||||
fn top_cell(row: &TopProcessRow, stat: TopStat) -> String {
|
||||
match stat {
|
||||
TopStat::Pid => row.pid.to_string(),
|
||||
TopStat::User => row
|
||||
.user
|
||||
.map_or_else(|| "?".to_string(), |value| value.to_string()),
|
||||
TopStat::State => row.state.to_string(),
|
||||
TopStat::Nice => row
|
||||
.nice
|
||||
.map_or_else(|| "?".to_string(), |value| value.to_string()),
|
||||
TopStat::Threads => row
|
||||
.threads
|
||||
.map_or_else(|| "?".to_string(), |value| value.to_string()),
|
||||
TopStat::Virt => row
|
||||
.virtual_size
|
||||
.map_or_else(|| "?".to_string(), format_top_bytes),
|
||||
TopStat::Res => row
|
||||
.resident_size
|
||||
.map_or_else(|| "?".to_string(), format_top_bytes),
|
||||
TopStat::Time => row
|
||||
.cpu_time
|
||||
.map_or_else(|| "?".to_string(), format_top_time),
|
||||
TopStat::Cpu => format!("{:.1}", row.cpu_percent),
|
||||
TopStat::PctMem => "?".to_string(),
|
||||
TopStat::Command => {
|
||||
if row.command.is_empty() {
|
||||
"?".to_string()
|
||||
} else {
|
||||
row.command.clone()
|
||||
}
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn render_top_snapshot(
|
||||
rows: &[TopProcessRow],
|
||||
row_limit: Option<usize>,
|
||||
sample: u64,
|
||||
stats: &[TopStat],
|
||||
) -> String {
|
||||
let mut running = 0_usize;
|
||||
let mut sleeping = 0_usize;
|
||||
let mut stopped = 0_usize;
|
||||
@@ -295,50 +454,24 @@ fn render_top_snapshot(rows: &[TopProcessRow], row_limit: Option<usize>, sample:
|
||||
format_top_bytes(resident),
|
||||
format_top_bytes(virtual_size)
|
||||
);
|
||||
let _ = writeln!(
|
||||
output,
|
||||
"{:>7} {:>8} {:>2} {:>3} {:>4} {:>9} {:>9} {:>10} {:>4} {:>4} COMMAND",
|
||||
"PID", "USER", "S", "NI", "TH", "VIRT", "RES", "TIME+", "%CPU", "%MEM"
|
||||
);
|
||||
let mut line = String::new();
|
||||
for (index, stat) in stats.iter().enumerate() {
|
||||
if index > 0 {
|
||||
line.push(' ');
|
||||
}
|
||||
let _ = write!(line, "{:>width$}", stat.header(), width = stat.width());
|
||||
}
|
||||
let _ = writeln!(output, "{line}");
|
||||
|
||||
for row in rows.iter().take(row_limit.unwrap_or(usize::MAX)) {
|
||||
let user = row
|
||||
.user
|
||||
.map_or_else(|| "?".to_string(), |value| value.to_string());
|
||||
let nice = row
|
||||
.nice
|
||||
.map_or_else(|| "?".to_string(), |value| value.to_string());
|
||||
let threads = row
|
||||
.threads
|
||||
.map_or_else(|| "?".to_string(), |value| value.to_string());
|
||||
let virtual_size = row
|
||||
.virtual_size
|
||||
.map_or_else(|| "?".to_string(), format_top_bytes);
|
||||
let resident_size = row
|
||||
.resident_size
|
||||
.map_or_else(|| "?".to_string(), format_top_bytes);
|
||||
let cpu_time = row
|
||||
.cpu_time
|
||||
.map_or_else(|| "?".to_string(), format_top_time);
|
||||
let _ = writeln!(
|
||||
output,
|
||||
"{:>7} {:>8} {:>2} {:>3} {:>4} {:>9} {:>9} {:>10} {:>4.1} {:>4} {}",
|
||||
row.pid,
|
||||
user,
|
||||
row.state,
|
||||
nice,
|
||||
threads,
|
||||
virtual_size,
|
||||
resident_size,
|
||||
cpu_time,
|
||||
row.cpu_percent,
|
||||
"?",
|
||||
if row.command.is_empty() {
|
||||
"?"
|
||||
} else {
|
||||
&row.command
|
||||
line.clear();
|
||||
for (index, stat) in stats.iter().enumerate() {
|
||||
if index > 0 {
|
||||
line.push(' ');
|
||||
}
|
||||
);
|
||||
let _ = write!(line, "{:>width$}", top_cell(row, *stat), width = stat.width());
|
||||
}
|
||||
let _ = writeln!(output, "{line}");
|
||||
}
|
||||
output.push('\n');
|
||||
output
|
||||
@@ -433,10 +566,56 @@ mod tests {
|
||||
row(2, "visible-two", 0.0, 0, 0),
|
||||
row(1, "hidden-one", 0.0, 0, 0),
|
||||
];
|
||||
let output = render_top_snapshot(&rows, Some(2), 7);
|
||||
let output = render_top_snapshot(&rows, Some(2), 7, DEFAULT_TOP_STATS);
|
||||
assert!(output.contains("top - snapshot 7"));
|
||||
assert!(output.contains("visible-three"));
|
||||
assert!(output.contains("visible-two"));
|
||||
assert!(!output.contains("hidden-one"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_macos_single_dash_long_options() {
|
||||
use brush_core::builtins::Command as _;
|
||||
let cmd = TopCommand::new(
|
||||
["top", "-pid", "56943,101", "-stats", "pid,cpu,th,mem,pstate"]
|
||||
.into_iter()
|
||||
.map(String::from),
|
||||
)
|
||||
.expect("macOS-style flags must parse");
|
||||
assert_eq!(cmd.pids, vec![56943, 101]);
|
||||
assert_eq!(cmd.stats, vec![
|
||||
TopStat::Pid,
|
||||
TopStat::Cpu,
|
||||
TopStat::Threads,
|
||||
TopStat::Res,
|
||||
TopStat::State
|
||||
]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snapshot_renders_selected_stats_in_order() {
|
||||
let rows = vec![row(42, "worker", 12.3, 4096, 61)];
|
||||
let output = render_top_snapshot(&rows, None, 1, &[
|
||||
TopStat::Pid,
|
||||
TopStat::Cpu,
|
||||
TopStat::Threads,
|
||||
TopStat::Res,
|
||||
TopStat::State,
|
||||
]);
|
||||
let header = output
|
||||
.lines()
|
||||
.find(|line| line.contains("PID"))
|
||||
.expect("header line");
|
||||
assert_eq!(header.split_whitespace().collect::<Vec<_>>(), vec![
|
||||
"PID", "%CPU", "TH", "RES", "S"
|
||||
]);
|
||||
let row_line = output
|
||||
.lines()
|
||||
.find(|line| line.contains("42"))
|
||||
.expect("process row");
|
||||
assert_eq!(row_line.split_whitespace().collect::<Vec<_>>(), vec![
|
||||
"42", "12.3", "2", "4.0k", "S"
|
||||
]);
|
||||
assert!(!output.contains("worker"));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3,10 +3,10 @@
|
||||
//! Ported from uutils coreutils 0.8.0.
|
||||
|
||||
#[cfg(unix)]
|
||||
use std::os::unix::fs::FileTypeExt;
|
||||
use std::os::unix::fs::{FileTypeExt, MetadataExt};
|
||||
use std::{
|
||||
ffi::OsString,
|
||||
fs::{OpenOptions, metadata},
|
||||
fs::{Metadata, OpenOptions, metadata},
|
||||
io::ErrorKind,
|
||||
};
|
||||
|
||||
@@ -19,7 +19,7 @@ use uucore::{
|
||||
|
||||
use crate::host::{Host, Utility, format_usage, matches_parser, util};
|
||||
|
||||
#[derive(Debug, Eq, PartialEq)]
|
||||
#[derive(Debug, Clone, Copy, Eq, PartialEq)]
|
||||
enum TruncateMode {
|
||||
Absolute(u64),
|
||||
Extend(u64),
|
||||
@@ -51,6 +51,35 @@ impl TruncateMode {
|
||||
}
|
||||
}
|
||||
|
||||
/// The numeric value carried by this mode.
|
||||
fn value(&self) -> u64 {
|
||||
match self {
|
||||
Self::Absolute(n)
|
||||
| Self::Extend(n)
|
||||
| Self::Reduce(n)
|
||||
| Self::AtMost(n)
|
||||
| Self::AtLeast(n)
|
||||
| Self::RoundDown(n)
|
||||
| Self::RoundUp(n) => *n,
|
||||
}
|
||||
}
|
||||
|
||||
/// Multiply this mode's value by `factor` (for `--io-blocks` scaling).
|
||||
///
|
||||
/// Returns `None` on overflow.
|
||||
fn scale(&self, factor: u64) -> Option<Self> {
|
||||
let value = self.value().checked_mul(factor)?;
|
||||
Some(match self {
|
||||
Self::Absolute(_) => Self::Absolute(value),
|
||||
Self::Extend(_) => Self::Extend(value),
|
||||
Self::Reduce(_) => Self::Reduce(value),
|
||||
Self::AtMost(_) => Self::AtMost(value),
|
||||
Self::AtLeast(_) => Self::AtLeast(value),
|
||||
Self::RoundDown(_) => Self::RoundDown(value),
|
||||
Self::RoundUp(_) => Self::RoundUp(value),
|
||||
})
|
||||
}
|
||||
|
||||
/// Determine whether this mode specifies an absolute size.
|
||||
fn is_absolute(&self) -> bool {
|
||||
matches!(self, Self::Absolute(_))
|
||||
@@ -123,10 +152,7 @@ fn app() -> Command {
|
||||
Arg::new(options::IO_BLOCKS)
|
||||
.short('o')
|
||||
.long(options::IO_BLOCKS)
|
||||
.help(
|
||||
"treat SIZE as the number of I/O blocks of the file rather than bytes (NOT \
|
||||
IMPLEMENTED)",
|
||||
)
|
||||
.help("treat SIZE as the number of I/O blocks of the file rather than bytes")
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
.arg(
|
||||
@@ -167,70 +193,103 @@ fn app() -> Command {
|
||||
)
|
||||
}
|
||||
|
||||
/// Truncate the named file to the specified size.
|
||||
///
|
||||
/// If `create` is true, the file is created if it does not already exist. If
|
||||
/// `size` is larger than the file, it is padded with zeros; if smaller, bytes
|
||||
/// beyond `size` are discarded.
|
||||
fn do_file_truncate(
|
||||
host: &Host,
|
||||
filename: &OsString,
|
||||
create: bool,
|
||||
size: u64,
|
||||
) -> Result<(), String> {
|
||||
let resolved = host.resolve(filename);
|
||||
|
||||
match OpenOptions::new().write(true).create(create).open(&resolved) {
|
||||
Ok(file) => file.set_len(size),
|
||||
Err(error) if error.kind() == ErrorKind::NotFound && !create => Ok(()),
|
||||
Err(error) => Err(error),
|
||||
/// The I/O block size of a file, falling back to 512 when the filesystem
|
||||
/// reports 0 (mirrors GNU's `ST_BLKSIZE`).
|
||||
#[cfg(unix)]
|
||||
fn io_blocksize(file_metadata: &Metadata) -> u64 {
|
||||
match file_metadata.blksize() {
|
||||
0 => 512,
|
||||
blksize => blksize,
|
||||
}
|
||||
.map_err(|error| format!("cannot open {} for writing: {error}", filename.quote()))
|
||||
}
|
||||
|
||||
#[cfg(not(unix))]
|
||||
fn io_blocksize(_file_metadata: &Metadata) -> u64 {
|
||||
512
|
||||
}
|
||||
|
||||
/// Truncate one file according to `mode`.
|
||||
///
|
||||
/// Unless `no_create` is set, the file is created if it does not already
|
||||
/// exist. If the target size is larger than the file, it is padded with
|
||||
/// zeros; if smaller, bytes beyond it are discarded. When `io_blocks` is
|
||||
/// set, the size is scaled by the file's I/O block size, matching GNU
|
||||
/// (which scales by the block size observed after opening the file).
|
||||
fn file_truncate(
|
||||
host: &Host,
|
||||
no_create: bool,
|
||||
io_blocks: bool,
|
||||
reference_size: Option<u64>,
|
||||
mode: &TruncateMode,
|
||||
filename: &OsString,
|
||||
) -> Result<(), String> {
|
||||
let resolved = host.resolve(filename);
|
||||
|
||||
// Get the length of the file.
|
||||
let file_size = match metadata(&resolved) {
|
||||
Ok(metadata) => {
|
||||
// A pipe has no length. Do this here to avoid a duplicate `stat()` syscall.
|
||||
#[cfg(unix)]
|
||||
if metadata.file_type().is_fifo() {
|
||||
return Err(format!(
|
||||
"cannot open {} for writing: No such device or address",
|
||||
filename.to_string_lossy().quote()
|
||||
));
|
||||
}
|
||||
metadata.len()
|
||||
// A pipe has no length, and opening it for writing would block waiting
|
||||
// for a reader; refuse it before the open.
|
||||
#[cfg(unix)]
|
||||
if let Ok(pre_metadata) = metadata(&resolved) {
|
||||
if pre_metadata.file_type().is_fifo() {
|
||||
return Err(format!(
|
||||
"cannot open {} for writing: No such device or address",
|
||||
filename.to_string_lossy().quote()
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
let create = !no_create;
|
||||
let file = match OpenOptions::new().write(true).create(create).open(&resolved) {
|
||||
Ok(file) => file,
|
||||
Err(error) if error.kind() == ErrorKind::NotFound && !create => return Ok(()),
|
||||
Err(error) => {
|
||||
return Err(format!("cannot open {} for writing: {error}", filename.quote()));
|
||||
},
|
||||
Err(_) => 0,
|
||||
};
|
||||
|
||||
let file_metadata = file
|
||||
.metadata()
|
||||
.map_err(|error| format!("cannot fstat {}: {error}", filename.quote()))?;
|
||||
|
||||
let mode = if io_blocks {
|
||||
let blksize = io_blocksize(&file_metadata);
|
||||
mode.scale(blksize).ok_or_else(|| {
|
||||
format!(
|
||||
"overflow in {} * {blksize} byte blocks for file {}",
|
||||
mode.value(),
|
||||
filename.quote()
|
||||
)
|
||||
})?
|
||||
} else {
|
||||
*mode
|
||||
};
|
||||
|
||||
// The reference size is either the given reference file's size, or the size
|
||||
// of the file to be truncated when no reference was provided.
|
||||
let actual_reference_size = reference_size.unwrap_or(file_size);
|
||||
let actual_reference_size = reference_size.unwrap_or_else(|| file_metadata.len());
|
||||
let Some(truncate_size) = mode.to_size(actual_reference_size) else {
|
||||
return Err("division by zero".to_string());
|
||||
};
|
||||
|
||||
do_file_truncate(host, filename, !no_create, truncate_size)
|
||||
file.set_len(truncate_size).map_err(|error| {
|
||||
format!(
|
||||
"failed to truncate {} at {truncate_size} bytes: {error}",
|
||||
filename.quote()
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
fn truncate(
|
||||
host: &mut Host,
|
||||
no_create: bool,
|
||||
_: bool,
|
||||
io_blocks: bool,
|
||||
reference: Option<String>,
|
||||
size: Option<String>,
|
||||
filenames: &[OsString],
|
||||
) -> Result<(), String> {
|
||||
if io_blocks && size.is_none() {
|
||||
return Err("--io-blocks was specified but --size was not".to_string());
|
||||
}
|
||||
|
||||
let reference_size = match reference {
|
||||
Some(reference_path) => {
|
||||
let reference_metadata = metadata(host.resolve(&reference_path)).map_err(|error| {
|
||||
@@ -254,13 +313,21 @@ fn truncate(
|
||||
None => TruncateMode::Extend(0),
|
||||
};
|
||||
|
||||
// GNU rejects rounding to a multiple of zero up front, before touching
|
||||
// any file.
|
||||
if matches!(mode, TruncateMode::RoundDown(0) | TruncateMode::RoundUp(0)) {
|
||||
return Err("division by zero".to_string());
|
||||
}
|
||||
|
||||
// If a reference file has been given, the truncate mode cannot be absolute.
|
||||
if reference_size.is_some() && mode.is_absolute() {
|
||||
return Err("you must specify a relative '--size' with '--reference'".to_string());
|
||||
}
|
||||
|
||||
for filename in filenames {
|
||||
if let Err(error) = file_truncate(host, no_create, reference_size, &mode, filename) {
|
||||
if let Err(error) =
|
||||
file_truncate(host, no_create, io_blocks, reference_size, &mode, filename)
|
||||
{
|
||||
host.error(error, 1);
|
||||
}
|
||||
}
|
||||
@@ -269,8 +336,10 @@ fn truncate(
|
||||
}
|
||||
|
||||
/// Decide whether a character is one of the size modifiers, like `+` or `<`.
|
||||
///
|
||||
/// `=` is the BSD spelling of an absolute size.
|
||||
fn is_modifier(c: char) -> bool {
|
||||
c == '+' || c == '-' || c == '<' || c == '>' || c == '/' || c == '%'
|
||||
c == '+' || c == '-' || c == '<' || c == '>' || c == '/' || c == '%' || c == '='
|
||||
}
|
||||
|
||||
/// Parse a size string with an optional modifier symbol as its first character.
|
||||
@@ -281,7 +350,10 @@ fn parse_mode_and_size(size_string: &str) -> Result<TruncateMode, ParseSizeError
|
||||
if is_modifier(c) {
|
||||
size_string = &size_string[1..];
|
||||
}
|
||||
let allow_list = allow_list_with_all_suffixes("EgGkKmMPQRtTYZ");
|
||||
let mut allow_list = allow_list_with_all_suffixes("EgGkKmMPQRtTYZ");
|
||||
// `b` counts 512-byte blocks (dd-style); accepted here for agent
|
||||
// convenience even though GNU truncate omits it.
|
||||
allow_list.push("b".to_string());
|
||||
let allow_list_ref = allow_list.iter().map(AsRef::as_ref).collect::<Vec<&str>>();
|
||||
Parser::default()
|
||||
.with_allow_list(&allow_list_ref)
|
||||
@@ -431,8 +503,76 @@ mod tests {
|
||||
assert_eq!(parse_mode_and_size(">10"), Ok(TruncateMode::AtLeast(10)));
|
||||
assert_eq!(parse_mode_and_size("/10"), Ok(TruncateMode::RoundDown(10)));
|
||||
assert_eq!(parse_mode_and_size("%10"), Ok(TruncateMode::RoundUp(10)));
|
||||
assert_eq!(parse_mode_and_size("=10"), Ok(TruncateMode::Absolute(10)));
|
||||
assert_eq!(parse_mode_and_size("1kB"), Ok(TruncateMode::Absolute(1000)));
|
||||
assert!(parse_mode_and_size("1b").is_err());
|
||||
// `b` counts 512-byte blocks; rejecting it broke `truncate -s 2b f`.
|
||||
assert_eq!(parse_mode_and_size("2b"), Ok(TruncateMode::Absolute(1024)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn b_suffix_counts_512_byte_blocks() {
|
||||
let (_dir, root) = canonical_tempdir();
|
||||
fs::write(root.join("f"), b"x").unwrap();
|
||||
|
||||
let (code, _, stderr) = run_in(root.clone(), &["-s", "2b", "f"]);
|
||||
assert_eq!((code, stderr.as_str()), (0, ""));
|
||||
assert_eq!(len(&root.join("f")), 1024);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bsd_equals_prefix_sets_absolute_size() {
|
||||
let (_dir, root) = canonical_tempdir();
|
||||
fs::write(root.join("f"), b"x").unwrap();
|
||||
|
||||
let (code, _, stderr) = run_in(root.clone(), &["-s", "=100", "f"]);
|
||||
assert_eq!((code, stderr.as_str()), (0, ""));
|
||||
assert_eq!(len(&root.join("f")), 100);
|
||||
}
|
||||
|
||||
/// Regression: `-o` used to be parsed and then silently ignored, truncating
|
||||
/// to the raw byte count (real data loss for `truncate -o -s 1 f`).
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn io_blocks_scales_size_by_file_blocksize() {
|
||||
use std::os::unix::fs::MetadataExt;
|
||||
|
||||
let (_dir, root) = canonical_tempdir();
|
||||
let path = root.join("f");
|
||||
fs::write(&path, vec![0u8; 4096]).unwrap();
|
||||
let blksize = fs::metadata(&path).unwrap().blksize();
|
||||
assert!(blksize > 1, "test needs a real filesystem block size");
|
||||
|
||||
let (code, _, stderr) = run_in(root.clone(), &["-o", "-s", "1", "f"]);
|
||||
assert_eq!((code, stderr.as_str()), (0, ""));
|
||||
assert_eq!(len(&path), blksize, "-o must scale SIZE by st_blksize, not truncate to 1 byte");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn io_blocks_without_size_is_rejected() {
|
||||
let (_dir, root) = canonical_tempdir();
|
||||
fs::write(root.join("ref"), b"123").unwrap();
|
||||
fs::write(root.join("f"), b"x").unwrap();
|
||||
|
||||
let (code, _, stderr) = run_in(root.clone(), &["-o", "-r", "ref", "f"]);
|
||||
assert_eq!(code, 1);
|
||||
assert!(stderr.contains("--io-blocks was specified but --size was not"));
|
||||
}
|
||||
|
||||
/// Regression: `%0`/`/0` must fail up front with GNU's error, not create
|
||||
/// or modify any operand.
|
||||
#[test]
|
||||
fn round_to_zero_is_division_by_zero_up_front() {
|
||||
let (_dir, root) = canonical_tempdir();
|
||||
|
||||
for size in ["%0", "/0"] {
|
||||
let (code, _, stderr) = run_in(root.clone(), &["-s", size, "missing"]);
|
||||
assert_eq!(code, 1, "size {size} must fail");
|
||||
assert!(stderr.contains("division by zero"), "size {size}: {stderr}");
|
||||
assert!(
|
||||
!root.join("missing").exists(),
|
||||
"size {size} must not create the operand"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -534,7 +534,6 @@ use std::{
|
||||
use brush_core::{ShellExtensions, builtins::Registration};
|
||||
use clap::{Arg, ArgAction, ArgMatches, Command, builder::ValueParser};
|
||||
use thiserror::Error;
|
||||
use unicode_width::UnicodeWidthChar;
|
||||
use utf8::{BufReadDecoder, BufReadDecoderError};
|
||||
use uucore::{
|
||||
display::Quotable,
|
||||
@@ -1107,7 +1106,7 @@ fn process_chunk<
|
||||
*current_len += 8;
|
||||
},
|
||||
_ => {
|
||||
*current_len += ch.width().unwrap_or(0);
|
||||
*current_len += xutf::width_char(ch);
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
@@ -22,6 +22,10 @@ pub(crate) struct WhichCli {
|
||||
#[arg(short = 'a', long = "all")]
|
||||
all: bool,
|
||||
|
||||
/// Silent (BSD): print nothing, report matches via the exit status only.
|
||||
#[arg(short = 's')]
|
||||
silent: bool,
|
||||
|
||||
/// Command names to locate.
|
||||
#[arg(value_name = "name")]
|
||||
names: Vec<String>,
|
||||
@@ -32,6 +36,11 @@ impl Utility for WhichCli {
|
||||
const USAGE_ERROR: u8 = 2;
|
||||
|
||||
fn run(self, host: &mut Host) -> i32 {
|
||||
// BSD and GNU which both treat a bare `which` as a usage error (exit 1).
|
||||
if self.names.is_empty() {
|
||||
let _ = writeln!(host.stderr, "usage: which [-as] program ...");
|
||||
return 1;
|
||||
}
|
||||
let path_var = host.var("PATH").unwrap_or_default().to_owned();
|
||||
let mut all_found = true;
|
||||
|
||||
@@ -57,8 +66,10 @@ impl Utility for WhichCli {
|
||||
// which(1) reports missing names via the exit status only.
|
||||
all_found = false;
|
||||
}
|
||||
for path in matches {
|
||||
let _ = writeln!(host.stdout, "{}", path.display());
|
||||
if !self.silent {
|
||||
for path in matches {
|
||||
let _ = writeln!(host.stdout, "{}", path.display());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -101,6 +112,38 @@ mod tests {
|
||||
(code, capture.out())
|
||||
}
|
||||
|
||||
/// Bare `which` used to silently exit 0; BSD/GNU which report a usage
|
||||
/// error on stderr and exit 1.
|
||||
#[test]
|
||||
fn no_operands_is_a_usage_error() {
|
||||
let temp = tempfile::tempdir().expect("temp directory should be created");
|
||||
let cwd = fs::canonicalize(temp.path()).expect("temp directory should canonicalize");
|
||||
let cli = WhichCli::try_parse_from(["which"]).expect("test arguments should parse");
|
||||
let (mut host, capture) = Host::for_test("which", Vec::new(), &cwd);
|
||||
let code = cli.run(&mut host);
|
||||
|
||||
assert_eq!(code, 1);
|
||||
assert_eq!(capture.out(), "");
|
||||
assert_eq!(capture.err(), "usage: which [-as] program ...\n");
|
||||
}
|
||||
|
||||
/// BSD `which -s` used to be rejected by clap with exit 2; it must print
|
||||
/// nothing and report found/missing purely via the exit status, including
|
||||
/// in the clustered `-as` spelling.
|
||||
#[test]
|
||||
fn silent_flag_suppresses_output_and_keeps_exit_status() {
|
||||
let temp = tempfile::tempdir().expect("temp directory should be created");
|
||||
let dir = fs::canonicalize(temp.path()).expect("temp directory should canonicalize");
|
||||
place_file(&dir, "tool", true);
|
||||
let path_var = dir.to_string_lossy();
|
||||
|
||||
assert_eq!(run_which(&["-s", "tool"], &path_var, &dir), (0, String::new()));
|
||||
assert_eq!(run_which(&["-s", "missing"], &path_var, &dir), (1, String::new()));
|
||||
assert_eq!(run_which(&["-as", "tool"], &path_var, &dir), (0, String::new()));
|
||||
assert_eq!(run_which(&["-sa", "tool"], &path_var, &dir), (0, String::new()));
|
||||
assert_eq!(run_which(&["-s", "tool", "missing"], &path_var, &dir), (1, String::new()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn finds_only_executable_files() {
|
||||
let temp = tempfile::tempdir().expect("temp directory should be created");
|
||||
|
||||
@@ -26,6 +26,24 @@ matches_parser!(Yes, app);
|
||||
impl Utility for Yes {
|
||||
const NAME: &'static str = "yes";
|
||||
|
||||
fn rewrite_argv(mut argv: Vec<OsString>) -> Result<Vec<OsString>, String> {
|
||||
// GNU yes (gnulib `parse_gnu_standard_options_only`) recognizes
|
||||
// `--help`/`--version` only as the sole argument; everything else —
|
||||
// `yes -n`, `yes --no`, even `yes --help me` — is echoed verbatim.
|
||||
// Insert `--` so clap treats every remaining argument as an operand.
|
||||
if argv.is_empty()
|
||||
|| (argv.len() == 2 && matches!(argv[1].to_str(), Some("--help" | "--version")))
|
||||
{
|
||||
return Ok(argv);
|
||||
}
|
||||
// GNU consumes one leading `--` as the operand separator; ours replaces it.
|
||||
if argv.get(1).is_some_and(|arg| arg.to_str() == Some("--")) {
|
||||
argv.remove(1);
|
||||
}
|
||||
argv.insert(1, OsString::from("--"));
|
||||
Ok(argv)
|
||||
}
|
||||
|
||||
fn run(self, host: &mut Host) -> i32 {
|
||||
let mut buffer = Vec::with_capacity(BUF_SIZE);
|
||||
let Some(strings) = self.matches.get_many::<OsString>("STRING") else {
|
||||
@@ -59,7 +77,9 @@ fn app() -> Command {
|
||||
Arg::new("STRING")
|
||||
.default_value("y")
|
||||
.value_parser(ValueParser::os_string())
|
||||
.action(ArgAction::Append),
|
||||
.action(ArgAction::Append)
|
||||
.allow_hyphen_values(true)
|
||||
.trailing_var_arg(true),
|
||||
)
|
||||
.infer_long_args(true)
|
||||
}
|
||||
@@ -214,8 +234,12 @@ mod tests {
|
||||
budget: usize,
|
||||
fail_kind: io::ErrorKind,
|
||||
) -> (i32, String, String) {
|
||||
let parsed = Yes::try_parse_from(std::iter::once("yes").chain(arguments.iter().copied()))
|
||||
.expect("test arguments should parse");
|
||||
let argv: Vec<OsString> = std::iter::once("yes")
|
||||
.chain(arguments.iter().copied())
|
||||
.map(OsString::from)
|
||||
.collect();
|
||||
let argv = Yes::rewrite_argv(argv).expect("yes rewrite is infallible");
|
||||
let parsed = Yes::try_parse_from(argv).expect("test arguments should parse");
|
||||
let (mut host, capture) = Host::for_test("yes", Vec::new(), Path::new("/"));
|
||||
let state = Arc::new(Mutex::new(WriterState { bytes: Vec::new(), remaining: budget }));
|
||||
host.stdout = OpenFile::Stream(Box::new(FailingWriter {
|
||||
@@ -228,6 +252,52 @@ mod tests {
|
||||
(code, stdout, capture.err())
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hyphen_operands_are_echoed_not_parsed() {
|
||||
// Failure mode: clap rejecting `yes -n` / `yes --no` / `yes -1` as
|
||||
// unknown options where GNU yes echoes them.
|
||||
let (code, stdout, stderr) = run_with(&["-n"], 6, io::ErrorKind::BrokenPipe);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stdout, "-n\n-n\n");
|
||||
assert_eq!(stderr, "");
|
||||
|
||||
let (code, stdout, _) = run_with(&["--no", "-1"], 16, io::ErrorKind::BrokenPipe);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stdout, "--no -1\n--no -1\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn help_is_special_only_as_sole_argument() {
|
||||
// Failure mode: `yes --help me` rendering help; GNU echoes "--help me".
|
||||
let (code, stdout, _) = run_with(&["--help", "me"], 20, io::ErrorKind::BrokenPipe);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stdout, "--help me\n--help me\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn version_is_special_only_as_sole_argument() {
|
||||
let (code, capture) = run_util::<Yes>(&["--version"], "", "/");
|
||||
assert_eq!(code, 0);
|
||||
assert!(capture.out().contains("0.8.0"), "stdout: {:?}", capture.out());
|
||||
|
||||
// Failure mode: `yes --version x` printing the version banner.
|
||||
let (code, stdout, _) = run_with(&["--version", "x"], 24, io::ErrorKind::BrokenPipe);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stdout, "--version x\n--version x\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn leading_double_dash_is_operand_separator() {
|
||||
// Failure mode: the rewrite doubling `--` so `yes --` echoes "--".
|
||||
let (code, stdout, _) = run_with(&["--"], 4, io::ErrorKind::BrokenPipe);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stdout, "y\ny\n");
|
||||
|
||||
let (code, stdout, _) = run_with(&["--", "--help"], 14, io::ErrorKind::BrokenPipe);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stdout, "--help\n--help\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn broken_pipe_is_clean_exit() {
|
||||
let (code, stdout, stderr) = run_with(&[], 100, io::ErrorKind::BrokenPipe);
|
||||
|
||||
@@ -28,6 +28,7 @@ ast-grep-core.workspace = true
|
||||
base64.workspace = true
|
||||
clap.workspace = true
|
||||
globset.workspace = true
|
||||
heapless.workspace = true
|
||||
fontdue.workspace = true
|
||||
grep-matcher.workspace = true
|
||||
futures.workspace = true
|
||||
@@ -63,13 +64,8 @@ syntect.workspace = true
|
||||
tokio.workspace = true
|
||||
tokio-util.workspace = true
|
||||
toml.workspace = true
|
||||
unicode-segmentation.workspace = true
|
||||
unicode-normalization.workspace = true
|
||||
unicode-properties.workspace = true
|
||||
unicode-script.workspace = true
|
||||
xutf.workspace = true
|
||||
zstd.workspace = true
|
||||
unicode-width.workspace = true
|
||||
xxhash-rust.workspace = true
|
||||
|
||||
[target.'cfg(target_os = "linux")'.dependencies]
|
||||
@@ -78,8 +74,6 @@ atspi = { version = "=0.30.0", features = ["tokio", "zbus"] }
|
||||
pipewire = { version = "=0.9.2", optional = true }
|
||||
reis = { version = "=0.5.0", features = ["tokio"] }
|
||||
x11rb = { version = "=0.13.2", features = ["randr", "xinput", "xtest"] }
|
||||
fancy-regex.workspace = true # utok scanner differential oracle
|
||||
tiktoken-rs.workspace = true # utok openai differential oracle
|
||||
xkeysym = "=0.2.1"
|
||||
|
||||
[target.'cfg(any(target_os = "macos", target_os = "windows"))'.dependencies]
|
||||
|
||||
@@ -8,10 +8,10 @@ use std::io::Cursor;
|
||||
|
||||
use arboard::{Clipboard, Error as ClipboardError, ImageData};
|
||||
use image::{DynamicImage, ImageFormat, RgbaImage};
|
||||
use napi::bindgen_prelude::*;
|
||||
use napi::{JsString, bindgen_prelude::*};
|
||||
use napi_derive::napi;
|
||||
|
||||
use crate::task;
|
||||
use crate::{js, task};
|
||||
|
||||
/// Clipboard image payload encoded as PNG bytes.
|
||||
#[napi(object)]
|
||||
@@ -135,8 +135,8 @@ fn read_raw_cf_dib() -> Option<Vec<u8>> {
|
||||
/// # Errors
|
||||
/// Returns an error if clipboard access fails.
|
||||
#[napi]
|
||||
pub fn copy_to_clipboard(text: String) -> Result<()> {
|
||||
set_clipboard_text(text)
|
||||
pub fn copy_to_clipboard(text: JsString) -> Result<()> {
|
||||
set_clipboard_text(&js::utf8(text)?)
|
||||
}
|
||||
|
||||
/// Linux: keep a single `arboard::Clipboard` alive for the whole process.
|
||||
@@ -153,7 +153,7 @@ pub fn copy_to_clipboard(text: String) -> Result<()> {
|
||||
/// (`wl-clipboard-rs` forks its own serving process) but sharing the instance
|
||||
/// is harmless there.
|
||||
#[cfg(target_os = "linux")]
|
||||
fn set_clipboard_text(text: String) -> Result<()> {
|
||||
fn set_clipboard_text(text: &str) -> Result<()> {
|
||||
use std::sync::OnceLock;
|
||||
|
||||
use parking_lot::Mutex;
|
||||
@@ -180,7 +180,7 @@ fn set_clipboard_text(text: String) -> Result<()> {
|
||||
/// calling thread also avoids worker-thread `AppKit` pasteboard warnings on
|
||||
/// macOS.
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
fn set_clipboard_text(text: String) -> Result<()> {
|
||||
fn set_clipboard_text(text: &str) -> Result<()> {
|
||||
let mut clipboard = Clipboard::new()
|
||||
.map_err(|err| Error::from_reason(format!("Failed to access clipboard: {err}")))?;
|
||||
clipboard
|
||||
|
||||
@@ -24,9 +24,11 @@
|
||||
|
||||
use std::{collections::HashMap, rc::Rc};
|
||||
|
||||
use napi::bindgen_prelude::*;
|
||||
use napi::{JsString, bindgen_prelude::*};
|
||||
use napi_derive::napi;
|
||||
|
||||
use crate::js;
|
||||
|
||||
/// UTF-16 code unit for `\n`.
|
||||
const LF: u16 = 0x000a;
|
||||
|
||||
@@ -333,8 +335,10 @@ fn concat_tokens(tokens: &[&[u16]]) -> Vec<u16> {
|
||||
/// options). Change values keep line terminators, and common runs are joined
|
||||
/// from the new text.
|
||||
#[napi]
|
||||
pub fn diff_lines(old_text: Utf16String, new_text: Utf16String) -> Vec<DiffChange> {
|
||||
diff_lines_impl(&old_text, &new_text)
|
||||
pub fn diff_lines(old_text: JsString, new_text: JsString) -> Result<Vec<DiffChange>> {
|
||||
let old_text = js::utf16(old_text)?;
|
||||
let new_text = js::utf16(new_text)?;
|
||||
Ok(diff_lines_impl(&old_text, &new_text))
|
||||
}
|
||||
|
||||
fn diff_lines_impl(old_text: &[u16], new_text: &[u16]) -> Vec<DiffChange> {
|
||||
@@ -351,8 +355,10 @@ fn diff_lines_impl(old_text: &[u16], new_text: &[u16]) -> Vec<DiffChange> {
|
||||
/// Callers that map line numbers — like hashline recovery — need the counts,
|
||||
/// not another copy of the text.
|
||||
#[napi]
|
||||
pub fn diff_line_runs(old_text: Utf16String, new_text: Utf16String) -> Vec<DiffRun> {
|
||||
diff_line_runs_impl(&old_text, &new_text)
|
||||
pub fn diff_line_runs(old_text: JsString, new_text: JsString) -> Result<Vec<DiffRun>> {
|
||||
let old_text = js::utf16(old_text)?;
|
||||
let new_text = js::utf16(new_text)?;
|
||||
Ok(diff_line_runs_impl(&old_text, &new_text))
|
||||
}
|
||||
|
||||
fn diff_line_runs_impl(old_text: &[u16], new_text: &[u16]) -> Vec<DiffRun> {
|
||||
@@ -387,11 +393,13 @@ fn no_newline_marker() -> Vec<u16> {
|
||||
/// semantics. `context` defaults to 4 like jsdiff.
|
||||
#[napi]
|
||||
pub fn structured_patch_hunks(
|
||||
old_text: Utf16String,
|
||||
new_text: Utf16String,
|
||||
old_text: JsString,
|
||||
new_text: JsString,
|
||||
context: Option<u32>,
|
||||
) -> Vec<PatchHunk> {
|
||||
structured_patch_hunks_impl(&old_text, &new_text, context)
|
||||
) -> Result<Vec<PatchHunk>> {
|
||||
let old_text = js::utf16(old_text)?;
|
||||
let new_text = js::utf16(new_text)?;
|
||||
Ok(structured_patch_hunks_impl(&old_text, &new_text, context))
|
||||
}
|
||||
|
||||
fn structured_patch_hunks_impl(
|
||||
@@ -894,8 +902,10 @@ fn word_post_process(changes: &mut [DiffChange]) {
|
||||
/// Tokens carry surrounding whitespace, equality ignores it, and the
|
||||
/// post-pass dedupes whitespace across change boundaries.
|
||||
#[napi]
|
||||
pub fn diff_words(old_text: Utf16String, new_text: Utf16String) -> Vec<DiffChange> {
|
||||
diff_words_impl(&old_text, &new_text)
|
||||
pub fn diff_words(old_text: JsString, new_text: JsString) -> Result<Vec<DiffChange>> {
|
||||
let old_text = js::utf16(old_text)?;
|
||||
let new_text = js::utf16(new_text)?;
|
||||
Ok(diff_words_impl(&old_text, &new_text))
|
||||
}
|
||||
|
||||
fn diff_words_impl(old_text: &[u16], new_text: &[u16]) -> Vec<DiffChange> {
|
||||
|
||||
@@ -5,8 +5,11 @@
|
||||
//! on a persistent sidecar because they lack a process-owned in-memory name
|
||||
//! registry with automatic crash recovery.
|
||||
|
||||
use napi::JsString;
|
||||
use napi_derive::napi;
|
||||
|
||||
use crate::js;
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
mod linux;
|
||||
#[cfg(all(unix, not(target_os = "linux")))]
|
||||
@@ -48,9 +51,13 @@ pub struct FileLock {
|
||||
impl FileLock {
|
||||
/// Try to acquire `path` without blocking.
|
||||
#[napi(factory)]
|
||||
pub fn try_acquire(path: String) -> napi::Result<Self> {
|
||||
pub fn try_acquire(path: JsString) -> napi::Result<Self> {
|
||||
let path = js::utf8(path)?;
|
||||
let inner = platform::try_acquire(&path).map_err(|error| {
|
||||
napi::Error::from_reason(format!("Failed to acquire native file lock for {path}: {error}"))
|
||||
napi::Error::from_reason(format!(
|
||||
"Failed to acquire native file lock for {}: {error}",
|
||||
&*path
|
||||
))
|
||||
})?;
|
||||
Ok(Self { inner })
|
||||
}
|
||||
|
||||
@@ -7,11 +7,22 @@
|
||||
|
||||
use std::{cell::RefCell, collections::HashMap, sync::OnceLock};
|
||||
|
||||
use napi::{JsString, Result};
|
||||
use napi_derive::napi;
|
||||
use syntect::parsing::{
|
||||
ParseState, Scope, ScopeStack, ScopeStackOp, SyntaxDefinition, SyntaxReference, SyntaxSet,
|
||||
};
|
||||
|
||||
use crate::js::{self, InlineStr};
|
||||
|
||||
/// One theme colour: an ANSI escape sequence such as `\x1b[38;2;255;0;0m`.
|
||||
///
|
||||
/// Decoded inline, so a whole palette crosses the boundary without touching
|
||||
/// the heap. The longest sequence a theme can produce sets attributes plus
|
||||
/// truecolor foreground and background — `\x1b[1;3;4;38;2;255;255;255;48;2;
|
||||
/// 255;255;255m`, 42 bytes — which the 47 usable bytes cover.
|
||||
pub type Color = InlineStr<48>;
|
||||
|
||||
static SYNTAX_SET: OnceLock<SyntaxSet> = OnceLock::new();
|
||||
static SCOPE_MATCHERS: OnceLock<ScopeMatchers> = OnceLock::new();
|
||||
|
||||
@@ -150,27 +161,38 @@ fn get_scope_matchers() -> &'static ScopeMatchers {
|
||||
#[napi(object)]
|
||||
pub struct HighlightColors {
|
||||
/// ANSI color for comments.
|
||||
pub comment: String,
|
||||
#[napi(ts_type = "string")]
|
||||
pub comment: Color,
|
||||
/// ANSI color for keywords.
|
||||
pub keyword: String,
|
||||
#[napi(ts_type = "string")]
|
||||
pub keyword: Color,
|
||||
/// ANSI color for function names.
|
||||
pub function: String,
|
||||
#[napi(ts_type = "string")]
|
||||
pub function: Color,
|
||||
/// ANSI color for variables and identifiers.
|
||||
pub variable: String,
|
||||
#[napi(ts_type = "string")]
|
||||
pub variable: Color,
|
||||
/// ANSI color for string literals.
|
||||
pub string: String,
|
||||
#[napi(ts_type = "string")]
|
||||
pub string: Color,
|
||||
/// ANSI color for numeric literals.
|
||||
pub number: String,
|
||||
#[napi(ts_type = "string")]
|
||||
pub number: Color,
|
||||
/// ANSI color for type identifiers.
|
||||
pub r#type: String,
|
||||
#[napi(ts_type = "string")]
|
||||
pub r#type: Color,
|
||||
/// ANSI color for operators.
|
||||
pub operator: String,
|
||||
#[napi(ts_type = "string")]
|
||||
pub operator: Color,
|
||||
/// ANSI color for punctuation tokens.
|
||||
pub punctuation: String,
|
||||
#[napi(ts_type = "string")]
|
||||
pub punctuation: Color,
|
||||
/// ANSI color for diff inserted lines.
|
||||
pub inserted: Option<String>,
|
||||
#[napi(ts_type = "string")]
|
||||
pub inserted: Option<Color>,
|
||||
/// ANSI color for diff deleted lines.
|
||||
pub deleted: Option<String>,
|
||||
#[napi(ts_type = "string")]
|
||||
pub deleted: Option<Color>,
|
||||
}
|
||||
|
||||
/// Language alias mappings: (aliases, target syntax name).
|
||||
@@ -383,29 +405,39 @@ fn find_syntax<'a>(ss: &'a SyntaxSet, lang: &str) -> Option<&'a SyntaxReference>
|
||||
/// Highlighted code with ANSI color codes, or the original code if highlighting
|
||||
/// fails.
|
||||
#[napi]
|
||||
pub fn highlight_code(code: String, lang: Option<String>, colors: HighlightColors) -> String {
|
||||
pub fn highlight_code(
|
||||
code: JsString,
|
||||
lang: Option<JsString>,
|
||||
colors: HighlightColors,
|
||||
) -> Result<String> {
|
||||
let code = js::utf8(code)?;
|
||||
let lang = lang.map(js::utf8).transpose()?;
|
||||
Ok(highlight_code_impl(&code, lang.as_deref(), &colors))
|
||||
}
|
||||
|
||||
fn highlight_code_impl(code: &str, lang: Option<&str>, colors: &HighlightColors) -> String {
|
||||
let inserted = colors.inserted.as_deref().unwrap_or("");
|
||||
let deleted = colors.deleted.as_deref().unwrap_or("");
|
||||
|
||||
// Color palette as array for quick indexing
|
||||
let palette = [
|
||||
colors.comment.as_str(), // 0
|
||||
colors.keyword.as_str(), // 1
|
||||
colors.function.as_str(), // 2
|
||||
colors.variable.as_str(), // 3
|
||||
colors.string.as_str(), // 4
|
||||
colors.number.as_str(), // 5
|
||||
colors.r#type.as_str(), // 6
|
||||
colors.operator.as_str(), // 7
|
||||
colors.punctuation.as_str(), // 8
|
||||
inserted, // 9
|
||||
deleted, // 10
|
||||
&*colors.comment, // 0
|
||||
&*colors.keyword, // 1
|
||||
&*colors.function, // 2
|
||||
&*colors.variable, // 3
|
||||
&*colors.string, // 4
|
||||
&*colors.number, // 5
|
||||
&*colors.r#type, // 6
|
||||
&*colors.operator, // 7
|
||||
&*colors.punctuation, // 8
|
||||
inserted, // 9
|
||||
deleted, // 10
|
||||
];
|
||||
|
||||
let ss = get_syntax_set();
|
||||
|
||||
// Find syntax for the language
|
||||
let syntax = match &lang {
|
||||
let syntax = match lang {
|
||||
Some(l) => find_syntax(ss, l),
|
||||
None => None,
|
||||
}
|
||||
@@ -415,7 +447,7 @@ pub fn highlight_code(code: String, lang: Option<String>, colors: HighlightColor
|
||||
let mut scope_stack = ScopeStack::new();
|
||||
let mut result = String::with_capacity(code.len() * 2);
|
||||
|
||||
for line in syntect::util::LinesWithEndings::from(code.as_str()) {
|
||||
for line in syntect::util::LinesWithEndings::from(code) {
|
||||
let Ok(ops) = parse_state.parse_line(line, ss) else {
|
||||
// Parse error - append unhighlighted line and continue
|
||||
result.push_str(line);
|
||||
@@ -477,16 +509,19 @@ pub fn highlight_code(code: String, lang: Option<String>, colors: HighlightColor
|
||||
/// Returns true if the language has either direct support or a fallback
|
||||
/// mapping.
|
||||
#[napi]
|
||||
pub fn supports_language(lang: String) -> bool {
|
||||
if is_known_alias(&lang) {
|
||||
pub fn supports_language(lang: JsString) -> Result<bool> {
|
||||
Ok(supports_language_impl(&js::utf8(lang)?))
|
||||
}
|
||||
|
||||
fn supports_language_impl(lang: &str) -> bool {
|
||||
if is_known_alias(lang) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Fall back to direct syntax lookup
|
||||
let ss = get_syntax_set();
|
||||
find_syntax(ss, &lang).is_some()
|
||||
find_syntax(ss, lang).is_some()
|
||||
}
|
||||
|
||||
/// Get list of supported languages.
|
||||
#[napi]
|
||||
pub fn get_supported_languages() -> Vec<String> {
|
||||
@@ -500,15 +535,15 @@ mod tests {
|
||||
|
||||
fn test_colors() -> HighlightColors {
|
||||
HighlightColors {
|
||||
comment: "<c>".to_string(),
|
||||
keyword: "<k>".to_string(),
|
||||
function: "<f>".to_string(),
|
||||
variable: "<v>".to_string(),
|
||||
string: "<s>".to_string(),
|
||||
number: "<n>".to_string(),
|
||||
r#type: "<t>".to_string(),
|
||||
operator: "<o>".to_string(),
|
||||
punctuation: "<p>".to_string(),
|
||||
comment: Color::new("<c>").unwrap(),
|
||||
keyword: Color::new("<k>").unwrap(),
|
||||
function: Color::new("<f>").unwrap(),
|
||||
variable: Color::new("<v>").unwrap(),
|
||||
string: Color::new("<s>").unwrap(),
|
||||
number: Color::new("<n>").unwrap(),
|
||||
r#type: Color::new("<t>").unwrap(),
|
||||
operator: Color::new("<o>").unwrap(),
|
||||
punctuation: Color::new("<p>").unwrap(),
|
||||
inserted: None,
|
||||
deleted: None,
|
||||
}
|
||||
@@ -517,14 +552,13 @@ mod tests {
|
||||
#[test]
|
||||
fn highlights_nix_vendored_syntax() {
|
||||
assert!(get_supported_languages().contains(&"Nix".to_string()));
|
||||
assert!(supports_language("nix".to_string()));
|
||||
assert!(supports_language_impl("nix"));
|
||||
|
||||
let out = highlight_code(
|
||||
let out = highlight_code_impl(
|
||||
"{ pkgs ? import <nixpkgs> {} }:\nlet message = \"hello\"; in pkgs.writeText \"msg\" \
|
||||
message # greeting\n"
|
||||
.to_string(),
|
||||
Some("nix".to_string()),
|
||||
test_colors(),
|
||||
message # greeting\n",
|
||||
Some("nix"),
|
||||
&test_colors(),
|
||||
);
|
||||
assert!(out.contains("<k>let"));
|
||||
assert!(out.contains("<s>hello"));
|
||||
@@ -534,13 +568,13 @@ mod tests {
|
||||
#[test]
|
||||
fn highlights_mermaid_vendored_syntax() {
|
||||
assert!(get_supported_languages().contains(&"Mermaid".to_string()));
|
||||
assert!(supports_language("mermaid".to_string()));
|
||||
assert!(supports_language("mmd".to_string()));
|
||||
assert!(supports_language_impl("mermaid"));
|
||||
assert!(supports_language_impl("mmd"));
|
||||
|
||||
let out = highlight_code(
|
||||
"graph TD\n A[\"Start\"] --> B\n %% note\n".to_string(),
|
||||
Some("mermaid".to_string()),
|
||||
test_colors(),
|
||||
let out = highlight_code_impl(
|
||||
"graph TD\n A[\"Start\"] --> B\n %% note\n",
|
||||
Some("mermaid"),
|
||||
&test_colors(),
|
||||
);
|
||||
assert!(out.contains("<k>graph"));
|
||||
assert!(out.contains("<s>Start"));
|
||||
|
||||
@@ -3,10 +3,10 @@
|
||||
use html_to_markdown_rs::{
|
||||
ConversionOptions, PreprocessingOptions, PreprocessingPreset, WarningKind, convert,
|
||||
};
|
||||
use napi::bindgen_prelude::*;
|
||||
use napi::{JsString, bindgen_prelude::*};
|
||||
use napi_derive::napi;
|
||||
|
||||
use crate::task;
|
||||
use crate::{js::into_string, task};
|
||||
|
||||
/// Options for HTML to Markdown conversion.
|
||||
#[napi(object)]
|
||||
@@ -24,14 +24,15 @@ pub struct HtmlToMarkdownOptions {
|
||||
/// Returns an error if the conversion fails or the worker task aborts.
|
||||
#[napi]
|
||||
pub fn html_to_markdown(
|
||||
html: String,
|
||||
html: JsString,
|
||||
options: Option<HtmlToMarkdownOptions>,
|
||||
) -> task::Promise<String> {
|
||||
) -> Result<task::Promise<String>> {
|
||||
let html = into_string(html)?;
|
||||
let options = options.unwrap_or_default();
|
||||
let clean_content = options.clean_content.unwrap_or(false);
|
||||
let skip_images = options.skip_images.unwrap_or(false);
|
||||
|
||||
task::blocking("html_to_markdown", (), move |_| {
|
||||
Ok(task::blocking("html_to_markdown", (), move |_| {
|
||||
let conversion_opts = ConversionOptions {
|
||||
skip_images,
|
||||
preprocessing: PreprocessingOptions {
|
||||
@@ -54,5 +55,5 @@ pub fn html_to_markdown(
|
||||
return Err(Error::from_reason(format!("Conversion error: {}", warning.message)));
|
||||
}
|
||||
Ok(result.content.unwrap_or_default())
|
||||
})
|
||||
}))
|
||||
}
|
||||
|
||||
@@ -4,9 +4,11 @@
|
||||
//! JavaScript-facing shapes plus conversions between walker entries and N-API
|
||||
//! payloads.
|
||||
|
||||
use napi::bindgen_prelude::*;
|
||||
use napi::{JsString, bindgen_prelude::*};
|
||||
use napi_derive::napi;
|
||||
|
||||
use crate::js;
|
||||
|
||||
/// Resolved filesystem entry kind for glob filters and match metadata.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
#[napi]
|
||||
@@ -75,9 +77,10 @@ pub(crate) fn map_walker_error<E: std::fmt::Display>(err: pi_walker::WalkError<E
|
||||
/// Intended to be called after agent file mutations: write, edit, rename, or
|
||||
/// delete.
|
||||
#[napi]
|
||||
pub fn invalidate_fs_scan_cache(path: Option<String>) {
|
||||
pub fn invalidate_fs_scan_cache(path: Option<JsString>) -> Result<()> {
|
||||
match path {
|
||||
Some(path) => pi_walker::invalidate_path_string(&path),
|
||||
Some(path) => pi_walker::invalidate_path_string(&js::utf8(path)?),
|
||||
None => pi_walker::invalidate_all(),
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -22,7 +22,10 @@ use napi::bindgen_prelude::*;
|
||||
use napi_derive::napi;
|
||||
use pi_iso::{BackendKind, ChangeKind, Diff, FileChange, IsoError, IsolationBackend};
|
||||
|
||||
use crate::js;
|
||||
|
||||
const ISO_UNAVAILABLE_PREFIX: &str = "ISO_UNAVAILABLE:";
|
||||
const ISO_UNAVAILABLE_WITH_LEADING_SPACE: &str = " ISO_UNAVAILABLE:";
|
||||
|
||||
/// Isolation backend identifier. Numeric so the JS side can `switch` on
|
||||
/// the enum without string comparisons.
|
||||
@@ -174,9 +177,10 @@ pub async fn iso_diff(lower: String, merged: String) -> Result<IsoDiff> {
|
||||
/// Use this to distinguish "this backend isn't installed" from a hard
|
||||
/// failure when handling caught errors on the JS side.
|
||||
#[napi]
|
||||
pub fn iso_is_unavailable_error(message: String) -> bool {
|
||||
message.starts_with(ISO_UNAVAILABLE_PREFIX)
|
||||
|| message.contains(&format!(" {ISO_UNAVAILABLE_PREFIX}"))
|
||||
pub fn iso_is_unavailable_error(message: napi::JsString) -> Result<bool> {
|
||||
let message = js::utf8(message)?;
|
||||
Ok(message.starts_with(ISO_UNAVAILABLE_PREFIX)
|
||||
|| message.contains(ISO_UNAVAILABLE_WITH_LEADING_SPACE))
|
||||
}
|
||||
|
||||
const fn to_napi_kind(kind: BackendKind) -> IsoBackendKind {
|
||||
|
||||
@@ -0,0 +1,475 @@
|
||||
//! Borrowing JavaScript strings at the N-API boundary.
|
||||
//!
|
||||
//! Node-API never hands out a pointer into a JS string's backing store: every
|
||||
//! accessor writes code units into a caller-owned buffer. The copy is
|
||||
//! unavoidable, the *allocation* is not — [`utf16`] and [`utf8`] read into a
|
||||
//! fixed per-thread scratch arena and hand back a guard that derefs to the
|
||||
//! borrowed text, releasing its range on drop. Text that fits the arena costs
|
||||
//! zero allocations; longer text spills to one owned `Vec`. Several guards can
|
||||
//! be live at once (diff pairs, colour palettes) — each owns a disjoint range.
|
||||
//!
|
||||
//! [`utf16`] is the default: it is the JS string's own encoding, so it is the
|
||||
//! only accessor that never transcodes. [`utf8`] exists for algorithms that are
|
||||
//! byte- or `str`-shaped (terminal escape parsing, syntect, paths). Reach for
|
||||
//! [`into_string`] only when the text must outlive the N-API callback — a value
|
||||
//! moved into a worker task, a channel, or an async body.
|
||||
//! Short, bounded strings — ANSI colours, font names, language ids — skip the
|
||||
//! arena entirely: [`InlineStr`] decodes them into a fixed-size array that
|
||||
//! lives in the struct itself, so a whole options object crosses with no
|
||||
//! allocation.
|
||||
|
||||
use std::{
|
||||
cell::{Cell, UnsafeCell},
|
||||
fmt,
|
||||
ops::{Deref, Range},
|
||||
ptr::{self, NonNull},
|
||||
slice, str,
|
||||
};
|
||||
|
||||
use napi::{
|
||||
Error, JsString, JsValue, Result, Status,
|
||||
bindgen_prelude::{FromNapiValue, ToNapiValue, TypeName, ValidateNapiValue},
|
||||
sys,
|
||||
};
|
||||
|
||||
/// Scratch bytes per thread. Only the JS thread reaches this module (N-API
|
||||
/// handles are not `Send`), so the footprint is effectively process-global.
|
||||
const SCRATCH_LEN: usize = 64 * 1024;
|
||||
|
||||
/// Fixed-size bump arena backing [`Utf16`]/[`Utf8`] guards.
|
||||
///
|
||||
/// The base address is stable for the thread's lifetime (the array never
|
||||
/// grows), so guards may hold raw pointers into it. Soundness rests on range
|
||||
/// discipline, not a borrow flag: every committed range is disjoint, new reads
|
||||
/// only touch bytes past `offset`, and no reference to the whole array is ever
|
||||
/// formed — all access goes through raw pointers into a caller's own range.
|
||||
struct Arena {
|
||||
/// Stored as `u16` units purely for the 2-alignment UTF-16 fills need;
|
||||
/// UTF-8 fills reinterpret the same bytes at alignment 1.
|
||||
buf: UnsafeCell<[u16; SCRATCH_LEN / 2]>,
|
||||
/// Bytes handed out. Fills bump it; drops roll it back (see [`Self::release`]).
|
||||
offset: Cell<usize>,
|
||||
/// Live scratch-backed guards. Hitting zero resets `offset`, so a non-LIFO
|
||||
/// drop order leaks at most until the last guard goes away.
|
||||
live: Cell<usize>,
|
||||
}
|
||||
|
||||
thread_local! {
|
||||
static ARENA: Arena = const {
|
||||
Arena {
|
||||
buf: UnsafeCell::new([0; SCRATCH_LEN / 2]),
|
||||
offset: Cell::new(0),
|
||||
live: Cell::new(0),
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
impl Arena {
|
||||
const fn base(&self) -> *mut u8 {
|
||||
self.buf.get().cast()
|
||||
}
|
||||
|
||||
/// Free tail aligned to `align` (a power of two): byte offset and length.
|
||||
const fn tail(&self, align: usize) -> (usize, usize) {
|
||||
let start = (self.offset.get() + align - 1) & !(align - 1);
|
||||
(start, SCRATCH_LEN.saturating_sub(start))
|
||||
}
|
||||
|
||||
/// Record `start..start + len` as owned by a new guard.
|
||||
fn commit(&self, start: usize, len: usize) {
|
||||
self.offset.set(start + len);
|
||||
self.live.set(self.live.get() + 1);
|
||||
}
|
||||
|
||||
/// Return `start..end`. The topmost range rolls the bump pointer back
|
||||
/// (LIFO drops recycle immediately); otherwise the bytes are stranded
|
||||
/// until `live` reaches zero and the whole arena resets.
|
||||
fn release(&self, start: usize, end: usize) {
|
||||
let live = self.live.get() - 1;
|
||||
self.live.set(live);
|
||||
if live == 0 {
|
||||
self.offset.set(0);
|
||||
} else if self.offset.get() == end {
|
||||
self.offset.set(start);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Text read from a JS string: borrowed from the thread's scratch arena when
|
||||
/// it fits, spilled to one owned `Vec` when it does not.
|
||||
enum TextRepr<T> {
|
||||
/// Range inside [`ARENA`]; `Drop` releases it. `NonNull` keeps this
|
||||
/// variant `!Send`, so the pointer can never outlive its thread's TLS.
|
||||
Scratch { ptr: NonNull<T>, len: usize },
|
||||
/// Heap spill for text longer than the arena's free tail.
|
||||
Owned(Vec<T>),
|
||||
}
|
||||
|
||||
impl<T> TextRepr<T> {
|
||||
#[inline]
|
||||
fn as_slice(&self) -> &[T] {
|
||||
match self {
|
||||
// SAFETY: the constructor committed `ptr..ptr + len` to this guard;
|
||||
// the arena never moves and no other guard overlaps the range.
|
||||
Self::Scratch { ptr, len } => unsafe { slice::from_raw_parts(ptr.as_ptr(), *len) },
|
||||
Self::Owned(vec) => vec,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl<T> Drop for TextRepr<T> {
|
||||
fn drop(&mut self) {
|
||||
if let Self::Scratch { ptr, len } = *self {
|
||||
ARENA.with(|arena| {
|
||||
let start = ptr.as_ptr().addr() - arena.base().addr();
|
||||
arena.release(start, start + len * size_of::<T>());
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Borrowed UTF-16 code units of a JS string, backed by the scratch arena.
|
||||
///
|
||||
/// Unlike `JsString::into_utf16`, the view excludes the NUL terminator
|
||||
/// Node-API appends, so `&*guard` is exactly the string's code units.
|
||||
pub struct Utf16(TextRepr<u16>);
|
||||
|
||||
impl Deref for Utf16 {
|
||||
type Target = [u16];
|
||||
|
||||
#[inline]
|
||||
fn deref(&self) -> &[u16] {
|
||||
self.0.as_slice()
|
||||
}
|
||||
}
|
||||
|
||||
/// Borrowed UTF-8 bytes of a JS string, backed by the scratch arena.
|
||||
pub struct Utf8(TextRepr<u8>);
|
||||
|
||||
impl Deref for Utf8 {
|
||||
type Target = str;
|
||||
|
||||
#[inline]
|
||||
fn deref(&self) -> &str {
|
||||
// SAFETY: utf8 validates the bytes before constructing Utf8.
|
||||
unsafe { str::from_utf8_unchecked(self.0.as_slice()) }
|
||||
}
|
||||
}
|
||||
|
||||
/// Borrow `value` as UTF-16 code units using the thread's scratch arena.
|
||||
///
|
||||
/// The happy path is a single N-API call into the arena's free tail; text
|
||||
/// that does not fit is measured and read into an owned spill buffer.
|
||||
#[inline]
|
||||
pub fn utf16(value: JsString<'_>) -> Result<Utf16> {
|
||||
let raw = value.value();
|
||||
ARENA.with(|arena| {
|
||||
let (start, avail_bytes) = arena.tail(2);
|
||||
let avail = avail_bytes / 2;
|
||||
if avail >= 2 {
|
||||
// SAFETY: `start..start + avail_bytes` is past every committed range,
|
||||
// and the base is 2-aligned with `start` aligned up.
|
||||
let ptr = unsafe { arena.base().add(start) }.cast::<u16>();
|
||||
let mut written = 0;
|
||||
// SAFETY: `raw` is a JS string owned by the live callback; Node-API
|
||||
// writes at most `avail - 1` units plus a NUL into the free tail.
|
||||
let status = unsafe {
|
||||
sys::napi_get_value_string_utf16(raw.env, raw.value, ptr, avail, &mut written)
|
||||
};
|
||||
napi::check_status!(status, "Failed to read JavaScript string")?;
|
||||
if written < avail - 1 {
|
||||
arena.commit(start, written * 2);
|
||||
return Ok(Utf16(TextRepr::Scratch {
|
||||
ptr: NonNull::new(ptr).unwrap(),
|
||||
len: written,
|
||||
}));
|
||||
}
|
||||
}
|
||||
|
||||
let mut len = 0;
|
||||
// SAFETY: a null buffer asks Node-API for the code-unit length only.
|
||||
let status = unsafe {
|
||||
sys::napi_get_value_string_utf16(raw.env, raw.value, ptr::null_mut(), 0, &mut len)
|
||||
};
|
||||
napi::check_status!(status, "Failed to measure JavaScript string")?;
|
||||
let mut buf: Vec<u16> = Vec::with_capacity(len + 1);
|
||||
let mut written = 0;
|
||||
// SAFETY: `buf` holds the measured length plus the NUL slot.
|
||||
let status = unsafe {
|
||||
sys::napi_get_value_string_utf16(
|
||||
raw.env,
|
||||
raw.value,
|
||||
buf.as_mut_ptr(),
|
||||
len + 1,
|
||||
&mut written,
|
||||
)
|
||||
};
|
||||
napi::check_status!(status, "Failed to read JavaScript string")?;
|
||||
// SAFETY: Node-API initialised `written` units.
|
||||
unsafe { buf.set_len(written) };
|
||||
Ok(Utf16(TextRepr::Owned(buf)))
|
||||
})
|
||||
}
|
||||
|
||||
/// Borrow `value` as UTF-8 using the thread's scratch arena.
|
||||
///
|
||||
/// Same shape as [`utf16`], plus UTF-8 validation before the guard exists.
|
||||
#[inline]
|
||||
pub fn utf8(value: JsString<'_>) -> Result<Utf8> {
|
||||
let raw = value.value();
|
||||
ARENA.with(|arena| {
|
||||
let (start, avail) = arena.tail(1);
|
||||
if avail >= 2 {
|
||||
// SAFETY: `start..start + avail` is past every committed range.
|
||||
let ptr = unsafe { arena.base().add(start) };
|
||||
let mut written = 0;
|
||||
// SAFETY: `raw` is a JS string owned by the live callback; Node-API
|
||||
// writes at most `avail - 1` bytes plus a NUL into the free tail.
|
||||
let status = unsafe {
|
||||
sys::napi_get_value_string_utf8(raw.env, raw.value, ptr.cast(), avail, &mut written)
|
||||
};
|
||||
napi::check_status!(status, "Failed to read JavaScript string")?;
|
||||
if written < avail - 1 {
|
||||
// SAFETY: Node-API initialised `written` bytes at `ptr`.
|
||||
let bytes = unsafe { slice::from_raw_parts(ptr, written) };
|
||||
if let Err(error) = str::from_utf8(bytes) {
|
||||
return Err(Error::new(Status::InvalidArg, error.to_string()));
|
||||
}
|
||||
arena.commit(start, written);
|
||||
return Ok(Utf8(TextRepr::Scratch {
|
||||
ptr: NonNull::new(ptr).unwrap(),
|
||||
len: written,
|
||||
}));
|
||||
}
|
||||
}
|
||||
|
||||
let mut len = 0;
|
||||
// SAFETY: a null buffer asks Node-API for the byte length only.
|
||||
let status = unsafe {
|
||||
sys::napi_get_value_string_utf8(raw.env, raw.value, ptr::null_mut(), 0, &mut len)
|
||||
};
|
||||
napi::check_status!(status, "Failed to measure JavaScript string")?;
|
||||
let mut buf: Vec<u8> = Vec::with_capacity(len + 1);
|
||||
let mut written = 0;
|
||||
// SAFETY: `buf` holds the measured length plus the NUL slot.
|
||||
let status = unsafe {
|
||||
sys::napi_get_value_string_utf8(
|
||||
raw.env,
|
||||
raw.value,
|
||||
buf.as_mut_ptr().cast(),
|
||||
len + 1,
|
||||
&mut written,
|
||||
)
|
||||
};
|
||||
napi::check_status!(status, "Failed to read JavaScript string")?;
|
||||
// SAFETY: Node-API initialised `written` bytes.
|
||||
unsafe { buf.set_len(written) };
|
||||
if let Err(error) = str::from_utf8(&buf) {
|
||||
return Err(Error::new(Status::InvalidArg, error.to_string()));
|
||||
}
|
||||
Ok(Utf8(TextRepr::Owned(buf)))
|
||||
})
|
||||
}
|
||||
|
||||
/// Append `value`'s UTF-16 code units to `out` and return their span.
|
||||
///
|
||||
/// For batches: one growing buffer holds every element, so an array costs a
|
||||
/// single allocation instead of one per string, and the spans can then be
|
||||
/// counted in parallel while the reads themselves stay on the JS thread —
|
||||
/// Node-API handles are not `Send`.
|
||||
pub fn utf16_append(value: JsString<'_>, out: &mut Vec<u16>) -> Result<Range<usize>> {
|
||||
let raw = value.value();
|
||||
let start = out.len();
|
||||
|
||||
let mut len = 0;
|
||||
// SAFETY: `raw` is a JS string owned by the live callback; a null buffer asks
|
||||
// Node-API for the code-unit length only.
|
||||
let status =
|
||||
unsafe { sys::napi_get_value_string_utf16(raw.env, raw.value, ptr::null_mut(), 0, &mut len) };
|
||||
napi::check_status!(status, "Failed to measure JavaScript string")?;
|
||||
|
||||
out.resize(start + len + 1, 0);
|
||||
let mut written = 0;
|
||||
// SAFETY: same string, and the tail from `start` holds the measured length
|
||||
// plus the NUL slot Node-API writes.
|
||||
let status = unsafe {
|
||||
sys::napi_get_value_string_utf16(
|
||||
raw.env,
|
||||
raw.value,
|
||||
out[start..].as_mut_ptr(),
|
||||
len + 1,
|
||||
&mut written,
|
||||
)
|
||||
};
|
||||
napi::check_status!(status, "Failed to read JavaScript string")?;
|
||||
out.truncate(start + written);
|
||||
Ok(start..out.len())
|
||||
}
|
||||
|
||||
/// Copy `value` into an owned `String`.
|
||||
///
|
||||
/// Only for text that outlives the N-API callback: a worker task, a channel
|
||||
/// message, or an async body. Synchronous consumers must borrow instead.
|
||||
pub fn into_string(value: JsString<'_>) -> Result<String> {
|
||||
let raw = value.value();
|
||||
// SAFETY: `raw` is a validated JS string from the current callback.
|
||||
unsafe { String::from_napi_value(raw.env, raw.value) }
|
||||
}
|
||||
|
||||
/// A JS string decoded into a fixed-capacity inline buffer, no heap involved.
|
||||
///
|
||||
/// Holds `N` bytes with a `u8` length, so the whole value is `N + 1` bytes and
|
||||
/// an options struct full of them costs nothing to build. Node-API needs one
|
||||
/// byte for its NUL terminator, leaving [`Self::CAPACITY`] usable; longer input
|
||||
/// is a caller error rather than a silent truncation, so an escape sequence can
|
||||
/// never arrive half-copied.
|
||||
///
|
||||
/// Storage is UTF-8: every consumer of these values wants `&str`, and the
|
||||
/// inputs are ASCII, so this is the encoding that avoids a transcode at the
|
||||
/// point of use. Use [`utf16`] for text whose consumer works in code units.
|
||||
#[derive(Clone)]
|
||||
pub struct InlineStr<const N: usize>(heapless::Vec<u8, N, u8>);
|
||||
|
||||
impl<const N: usize> InlineStr<N> {
|
||||
/// Usable bytes, excluding the NUL slot Node-API requires.
|
||||
pub const CAPACITY: usize = N - 1;
|
||||
|
||||
/// Build from Rust text, for tests and native-side defaults.
|
||||
pub fn new(text: &str) -> Result<Self> {
|
||||
if text.len() > Self::CAPACITY {
|
||||
return Err(too_long(text.len(), Self::CAPACITY));
|
||||
}
|
||||
heapless::Vec::from_slice(text.as_bytes())
|
||||
.map(Self)
|
||||
.map_err(|_| too_long(text.len(), Self::CAPACITY))
|
||||
}
|
||||
}
|
||||
|
||||
fn too_long(len: usize, capacity: usize) -> Error {
|
||||
Error::new(Status::InvalidArg, format!("string is {len} bytes, expected at most {capacity}"))
|
||||
}
|
||||
|
||||
impl<const N: usize> Deref for InlineStr<N> {
|
||||
type Target = str;
|
||||
|
||||
fn deref(&self) -> &str {
|
||||
// SAFETY: both constructors validate the bytes as UTF-8 before storing
|
||||
// them, and the buffer is immutable afterwards.
|
||||
unsafe { str::from_utf8_unchecked(&self.0) }
|
||||
}
|
||||
}
|
||||
|
||||
impl<const N: usize> fmt::Debug for InlineStr<N> {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
fmt::Debug::fmt(&**self, f)
|
||||
}
|
||||
}
|
||||
|
||||
impl<const N: usize> TypeName for InlineStr<N> {
|
||||
fn type_name() -> &'static str {
|
||||
"String"
|
||||
}
|
||||
|
||||
fn value_type() -> napi::ValueType {
|
||||
napi::ValueType::String
|
||||
}
|
||||
}
|
||||
|
||||
impl<const N: usize> ValidateNapiValue for InlineStr<N> {}
|
||||
|
||||
impl<const N: usize> FromNapiValue for InlineStr<N> {
|
||||
unsafe fn from_napi_value(env: sys::napi_env, napi_val: sys::napi_value) -> Result<Self> {
|
||||
let mut len = 0;
|
||||
// SAFETY: `napi_val` is a JS string owned by the live callback; a null
|
||||
// buffer asks Node-API for the byte length only.
|
||||
let status =
|
||||
unsafe { sys::napi_get_value_string_utf8(env, napi_val, ptr::null_mut(), 0, &mut len) };
|
||||
napi::check_status!(status, "Failed to measure JavaScript string")?;
|
||||
if len > Self::CAPACITY {
|
||||
return Err(too_long(len, Self::CAPACITY));
|
||||
}
|
||||
|
||||
let mut buf: heapless::Vec<u8, N, u8> = heapless::Vec::new();
|
||||
buf.resize_default(N)
|
||||
.map_err(|_| too_long(len, Self::CAPACITY))?;
|
||||
let mut written = 0;
|
||||
// SAFETY: same string, and `buf` is filled to `N`, which holds the measured
|
||||
// length plus the NUL terminator Node-API writes.
|
||||
let status = unsafe {
|
||||
sys::napi_get_value_string_utf8(env, napi_val, buf.as_mut_ptr().cast(), N, &mut written)
|
||||
};
|
||||
napi::check_status!(status, "Failed to read JavaScript string")?;
|
||||
buf.truncate(written);
|
||||
if let Err(error) = str::from_utf8(&buf) {
|
||||
return Err(Error::new(Status::InvalidArg, error.to_string()));
|
||||
}
|
||||
Ok(Self(buf))
|
||||
}
|
||||
}
|
||||
|
||||
impl<const N: usize> ToNapiValue for InlineStr<N> {
|
||||
unsafe fn to_napi_value(env: sys::napi_env, val: Self) -> Result<sys::napi_value> {
|
||||
// SAFETY: `env` is the live callback environment.
|
||||
unsafe { ToNapiValue::to_napi_value(env, &*val) }
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn arena() -> Arena {
|
||||
Arena {
|
||||
buf: UnsafeCell::new([0; SCRATCH_LEN / 2]),
|
||||
offset: Cell::new(0),
|
||||
live: Cell::new(0),
|
||||
}
|
||||
}
|
||||
|
||||
/// Live guards own disjoint ranges; overlap would alias the derefs (UB).
|
||||
#[test]
|
||||
fn commits_never_overlap_live_ranges() {
|
||||
let a = arena();
|
||||
let (s1, _) = a.tail(1);
|
||||
a.commit(s1, 100);
|
||||
let (s2, _) = a.tail(2);
|
||||
assert!(s2 >= s1 + 100);
|
||||
a.commit(s2, 50);
|
||||
let (s3, _) = a.tail(1);
|
||||
assert!(s3 >= s2 + 50);
|
||||
}
|
||||
|
||||
/// LIFO drops recycle immediately; the next fill reuses the range.
|
||||
#[test]
|
||||
fn lifo_release_rolls_back() {
|
||||
let a = arena();
|
||||
a.commit(0, 100);
|
||||
a.commit(100, 50);
|
||||
a.release(100, 150);
|
||||
assert_eq!(a.tail(1).0, 100);
|
||||
a.release(0, 100);
|
||||
assert_eq!(a.tail(1).0, 0);
|
||||
}
|
||||
|
||||
/// Non-LIFO drops strand bytes only until the last guard goes away.
|
||||
#[test]
|
||||
fn arena_resets_when_last_guard_drops() {
|
||||
let a = arena();
|
||||
a.commit(0, 100);
|
||||
a.commit(100, 50);
|
||||
a.release(0, 100);
|
||||
assert_eq!(a.tail(1).0, 150, "inner range stays stranded while a guard is live");
|
||||
a.release(100, 150);
|
||||
assert_eq!(a.tail(1).0, 0);
|
||||
}
|
||||
|
||||
/// A utf16 fill after an odd utf8 commit must get a 2-aligned range.
|
||||
#[test]
|
||||
fn utf16_tail_is_aligned() {
|
||||
let a = arena();
|
||||
a.commit(0, 7);
|
||||
let (start, len) = a.tail(2);
|
||||
assert_eq!(start, 8);
|
||||
assert_eq!(len, SCRATCH_LEN - 8);
|
||||
}
|
||||
}
|
||||
@@ -12,9 +12,12 @@
|
||||
|
||||
use std::borrow::Cow;
|
||||
|
||||
use napi::{JsString, Result};
|
||||
use napi_derive::napi;
|
||||
use phf::phf_map;
|
||||
|
||||
use crate::js;
|
||||
|
||||
const LOCK_MASK: u32 = 64 + 128;
|
||||
|
||||
// Internal sentinel codes for CSI 1;mod <letter> forms:
|
||||
@@ -298,11 +301,20 @@ static LETTERS: [&str; 26] = [
|
||||
/// base layout key) and modifier bits.
|
||||
#[napi]
|
||||
pub fn matches_kitty_sequence(
|
||||
data: String,
|
||||
data: JsString,
|
||||
expected_codepoint: i32,
|
||||
expected_modifier: u32,
|
||||
) -> Result<bool> {
|
||||
let data = js::utf8(data)?;
|
||||
Ok(matches_kitty_sequence_inner(data.as_bytes(), expected_codepoint, expected_modifier))
|
||||
}
|
||||
|
||||
fn matches_kitty_sequence_inner(
|
||||
data: &[u8],
|
||||
expected_codepoint: i32,
|
||||
expected_modifier: u32,
|
||||
) -> bool {
|
||||
let Some(parsed) = parse_kitty_sequence_bytes(data.as_bytes()) else {
|
||||
let Some(parsed) = parse_kitty_sequence_bytes(data) else {
|
||||
return false;
|
||||
};
|
||||
|
||||
@@ -378,40 +390,46 @@ const fn is_symbol_key(cp: i32) -> bool {
|
||||
///
|
||||
/// Returns a key id like "escape" or "ctrl+c", or None if unrecognized.
|
||||
#[napi]
|
||||
pub fn parse_key(data: String, kitty_protocol_active: bool) -> Option<String> {
|
||||
parse_key_inner(data.as_bytes(), kitty_protocol_active).map(|s| s.into_owned())
|
||||
pub fn parse_key(data: JsString, kitty_protocol_active: bool) -> Result<Option<String>> {
|
||||
let data = js::utf8(data)?;
|
||||
Ok(parse_key_inner(data.as_bytes(), kitty_protocol_active).map(|key| key.into_owned()))
|
||||
}
|
||||
|
||||
/// Check if input matches a legacy escape sequence for the given key name.
|
||||
///
|
||||
/// Returns true only when the byte sequence maps to the exact key identifier.
|
||||
#[napi]
|
||||
pub fn matches_legacy_sequence(data: String, key_name: String) -> bool {
|
||||
LEGACY_SEQUENCES
|
||||
pub fn matches_legacy_sequence(data: JsString, key_name: JsString) -> Result<bool> {
|
||||
let data = js::utf8(data)?;
|
||||
let key_name = js::utf8(key_name)?;
|
||||
Ok(LEGACY_SEQUENCES
|
||||
.get(data.as_bytes())
|
||||
.is_some_and(|&id| id == key_name)
|
||||
.is_some_and(|&id| id == &*key_name))
|
||||
}
|
||||
|
||||
/// Match input data against a key identifier string.
|
||||
///
|
||||
/// Returns true when the bytes represent the specified key with modifiers.
|
||||
#[napi]
|
||||
pub fn matches_key(data: String, key_id: String, kitty_protocol_active: bool) -> bool {
|
||||
matches_key_inner(data.as_bytes(), &key_id, kitty_protocol_active)
|
||||
pub fn matches_key(data: JsString, key_id: JsString, kitty_protocol_active: bool) -> Result<bool> {
|
||||
let data = js::utf8(data)?;
|
||||
let key_id = js::utf8(key_id)?;
|
||||
Ok(matches_key_inner(data.as_bytes(), &key_id, kitty_protocol_active))
|
||||
}
|
||||
|
||||
/// Parse a Kitty keyboard protocol sequence.
|
||||
///
|
||||
/// Returns a structured parse result when the input is a valid Kitty sequence.
|
||||
#[napi]
|
||||
pub fn parse_kitty_sequence(data: String) -> Option<ParsedKittyResult> {
|
||||
parse_kitty_sequence_bytes(data.as_bytes()).map(|p| ParsedKittyResult {
|
||||
codepoint: p.codepoint,
|
||||
shifted_key: p.shifted_key,
|
||||
base_layout_key: p.base_layout_key,
|
||||
modifier: p.modifier,
|
||||
event_type: optional_kitty_event_type(p.event_type),
|
||||
})
|
||||
pub fn parse_kitty_sequence(data: JsString) -> Result<Option<ParsedKittyResult>> {
|
||||
let data = js::utf8(data)?;
|
||||
Ok(parse_kitty_sequence_bytes(data.as_bytes()).map(|parsed| ParsedKittyResult {
|
||||
codepoint: parsed.codepoint,
|
||||
shifted_key: parsed.shifted_key,
|
||||
base_layout_key: parsed.base_layout_key,
|
||||
modifier: parsed.modifier,
|
||||
event_type: optional_kitty_event_type(parsed.event_type),
|
||||
}))
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
@@ -1590,20 +1608,12 @@ mod tests {
|
||||
let plain_cyrillic_c = b"\x1b[1089::99u";
|
||||
assert!(!matches_key_inner(plain_cyrillic_c, "c", true));
|
||||
assert_eq!(parse_key_inner(plain_cyrillic_c, true).as_deref(), None);
|
||||
assert!(!matches_kitty_sequence(
|
||||
String::from_utf8_lossy(plain_cyrillic_c).into_owned(),
|
||||
i32::from(b'c'),
|
||||
0,
|
||||
));
|
||||
assert!(!matches_kitty_sequence_inner(plain_cyrillic_c, i32::from(b'c'), 0));
|
||||
|
||||
let ctrl_cyrillic_c = b"\x1b[1089::99;5u";
|
||||
assert!(matches_key_inner(ctrl_cyrillic_c, "ctrl+c", true));
|
||||
assert_eq!(parse_key_inner(ctrl_cyrillic_c, true).as_deref(), Some("ctrl+c"));
|
||||
assert!(matches_kitty_sequence(
|
||||
String::from_utf8_lossy(ctrl_cyrillic_c).into_owned(),
|
||||
i32::from(b'c'),
|
||||
MOD_CTRL,
|
||||
));
|
||||
assert!(matches_kitty_sequence_inner(ctrl_cyrillic_c, i32::from(b'c'), MOD_CTRL));
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -39,6 +39,7 @@ pub mod grep;
|
||||
pub mod highlight;
|
||||
pub mod html;
|
||||
pub mod iofs;
|
||||
pub mod js;
|
||||
pub mod keys;
|
||||
pub mod live;
|
||||
/// PDF inspection and Markdown conversion.
|
||||
|
||||
@@ -8,14 +8,14 @@
|
||||
use std::time::Duration;
|
||||
|
||||
use napi::{
|
||||
Env, Result,
|
||||
Env, JsString, Result,
|
||||
bindgen_prelude::{PromiseRaw, Unknown},
|
||||
};
|
||||
use napi_derive::napi;
|
||||
use pi_shell::process::{self as core_process, ProcessStatus as CoreProcessStatus};
|
||||
pub use pi_shell::process::{KILL_SIGNAL, TERM_SIGNAL, TerminationTargets, kill_process_group};
|
||||
|
||||
use crate::task;
|
||||
use crate::{js::into_string, task};
|
||||
|
||||
#[derive(Default)]
|
||||
#[napi(object)]
|
||||
@@ -81,11 +81,11 @@ impl Process {
|
||||
|
||||
/// Open stable process references whose executable path matches exactly.
|
||||
#[napi]
|
||||
pub fn from_path(path: String) -> Vec<Process> {
|
||||
core_process::Process::from_path(path)
|
||||
pub fn from_path(path: JsString) -> Result<Vec<Process>> {
|
||||
Ok(core_process::Process::from_path(into_string(path)?)
|
||||
.into_iter()
|
||||
.map(Self::from_inner)
|
||||
.collect()
|
||||
.collect())
|
||||
}
|
||||
|
||||
/// Operating-system process identifier for this process reference.
|
||||
|
||||
@@ -13,6 +13,7 @@ use std::{
|
||||
};
|
||||
|
||||
use napi::{
|
||||
JsString,
|
||||
bindgen_prelude::*,
|
||||
threadsafe_function::{ThreadsafeFunction, ThreadsafeFunctionCallMode},
|
||||
};
|
||||
@@ -20,7 +21,7 @@ use napi_derive::napi;
|
||||
use parking_lot::Mutex;
|
||||
use portable_pty::{Child, CommandBuilder, PtySize, native_pty_system};
|
||||
|
||||
use crate::{ps, task};
|
||||
use crate::{js::into_string, ps, task};
|
||||
|
||||
/// Options for running a command in a PTY session.
|
||||
#[napi(object)]
|
||||
@@ -178,8 +179,8 @@ impl PtySession {
|
||||
|
||||
/// Write raw input bytes to PTY stdin.
|
||||
#[napi]
|
||||
pub fn write(&self, data: String) -> Result<()> {
|
||||
self.send_control(ControlMessage::Input(data))
|
||||
pub fn write(&self, data: JsString) -> Result<()> {
|
||||
self.send_control(ControlMessage::Input(into_string(data)?))
|
||||
}
|
||||
|
||||
/// Resize the active PTY.
|
||||
|
||||
@@ -48,10 +48,10 @@ use std::{
|
||||
|
||||
use base64::{Engine as _, engine::general_purpose::STANDARD};
|
||||
use fontdue::{Font as TtfFace, FontSettings, Metrics};
|
||||
use napi::bindgen_prelude::*;
|
||||
use napi::{JsString, bindgen_prelude::*};
|
||||
use napi_derive::napi;
|
||||
|
||||
use crate::task;
|
||||
use crate::{js, task};
|
||||
|
||||
/// Upper bound on the frame edge: a hard stop against absurd allocations
|
||||
/// (`size * size` pixel buffer), far above the 2576px production frame.
|
||||
@@ -1162,14 +1162,17 @@ pub struct SnapcompactRenderOptions {
|
||||
/// the selected native font has a glyph for it; renderer control codes are
|
||||
/// considered renderable because they are interpreted outside font lookup.
|
||||
#[napi]
|
||||
pub fn snapcompact_supported_chars(font: String, chars: String) -> Result<String> {
|
||||
let font = resolve_font(&font).ok_or_else(|| {
|
||||
pub fn snapcompact_supported_chars(font: JsString, chars: JsString) -> Result<String> {
|
||||
let font_name = js::utf8(font)?;
|
||||
let font = resolve_font(&font_name).ok_or_else(|| {
|
||||
Error::from_reason(format!(
|
||||
"Unknown snapcompact font {font:?}: expected \"5x8\", \"8x8\", \"6x12\", \"8x13\", or \
|
||||
\"silver\""
|
||||
"Unknown snapcompact font {:?}: expected \"5x8\", \"8x8\", \"6x12\", \"8x13\", or \
|
||||
\"silver\"",
|
||||
&*font_name
|
||||
))
|
||||
})?;
|
||||
let mut supported = String::new();
|
||||
let chars = js::utf8(chars)?;
|
||||
let mut supported = String::with_capacity(chars.len());
|
||||
for ch in chars.chars() {
|
||||
if matches!(ch as u32, DIM_ON | DIM_OFF | FULL_BLOCK | 0x0a) || font.supports(ch as u32) {
|
||||
supported.push(ch);
|
||||
|
||||
@@ -16,8 +16,8 @@ use std::{
|
||||
use napi::{JsString, bindgen_prelude::*};
|
||||
use napi_derive::napi;
|
||||
use smallvec::{SmallVec, smallvec};
|
||||
use unicode_segmentation::UnicodeSegmentation;
|
||||
use unicode_width::{UnicodeWidthChar, UnicodeWidthStr};
|
||||
|
||||
use crate::js;
|
||||
|
||||
const MIN_TAB_WIDTH: u32 = 1;
|
||||
const MAX_TAB_WIDTH: u32 = 16;
|
||||
@@ -45,8 +45,7 @@ fn build_utf16_string(mut data: Vec<u16>) -> Utf16String {
|
||||
while data.last() == Some(&0) {
|
||||
data.pop();
|
||||
}
|
||||
// SAFETY: we know Utf16String == struct(Vec<u16>)
|
||||
unsafe { std::mem::transmute(data) }
|
||||
Utf16String::from(data)
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
@@ -642,7 +641,7 @@ fn apply_hangul_compat_jamo_delta(width: usize, c: char) -> usize {
|
||||
let Some(target) = hangul_compat_jamo_target_width() else {
|
||||
return width;
|
||||
};
|
||||
let unicode_width = UnicodeWidthChar::width(c).unwrap_or(0);
|
||||
let unicode_width = xutf::width_char(c);
|
||||
// The zero-width filler (U+3164 HANGUL FILLER) is an invisible placeholder.
|
||||
// The target is set for *visible* jamo, so only the narrow correction
|
||||
// (target 1) applies to the filler; a wide terminal renders it at its
|
||||
@@ -659,7 +658,7 @@ fn apply_hangul_compat_jamo_delta(width: usize, c: char) -> usize {
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn char_width_corrected(c: char) -> Option<usize> {
|
||||
fn char_width_corrected(c: char) -> usize {
|
||||
// Hangul Compatibility Jamo U+3131..=U+318E render as 1 cell on some
|
||||
// terminals (Terminal.app, iTerm2) but follow UAX#11 at 2 cells on others
|
||||
// (Ghostty, most Linux terminals). The width is resolved at runtime from the
|
||||
@@ -671,13 +670,13 @@ fn char_width_corrected(c: char) -> Option<usize> {
|
||||
// Zero-width filler (U+3164): only the narrow correction applies — a
|
||||
// wide terminal renders it at its Unicode width (0), not the effective
|
||||
// wide target set for visible jamo. See apply_hangul_compat_jamo_delta.
|
||||
let unicode_width = UnicodeWidthChar::width(c).unwrap_or(0);
|
||||
let unicode_width = xutf::width_char(c);
|
||||
if unicode_width == 0 && target > 1 {
|
||||
return Some(unicode_width);
|
||||
return unicode_width;
|
||||
}
|
||||
return Some(target);
|
||||
return target;
|
||||
}
|
||||
UnicodeWidthChar::width(c)
|
||||
xutf::width_char(c)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
@@ -690,14 +689,14 @@ fn grapheme_width_str(g: &str, tab_width: usize) -> usize {
|
||||
return 0;
|
||||
};
|
||||
if it.next().is_none() {
|
||||
return char_width_corrected(c0).unwrap_or(0);
|
||||
return char_width_corrected(c0);
|
||||
}
|
||||
// Multi-char grapheme: keep UnicodeWidthStr as the source of truth for
|
||||
// sequence-level width rules (VS16 emoji presentation, keycaps, ZWJ emoji,
|
||||
// CRLF, script ligatures). A per-char sum is not equivalent. Apply only the
|
||||
// same local Compatibility Jamo delta that char_width_corrected applies to
|
||||
// standalone code points; the delta is a no-op when no correction is active.
|
||||
let mut width = UnicodeWidthStr::width(g);
|
||||
let mut width = xutf::width_str(g);
|
||||
for c in g.chars() {
|
||||
width = apply_hangul_compat_jamo_delta(width, c);
|
||||
}
|
||||
@@ -729,7 +728,7 @@ where
|
||||
}
|
||||
|
||||
let mut utf16_pos = 0usize;
|
||||
for g in scratch.graphemes(true) {
|
||||
for g in xutf::graphemes_str(scratch) {
|
||||
let w = grapheme_width_str(g, tab_width);
|
||||
|
||||
let g_u16_len: usize = g.chars().map(|c| c.len_utf16()).sum();
|
||||
@@ -1255,10 +1254,12 @@ fn wrap_text_with_ansi_impl(
|
||||
/// Returns UTF-16 lines with active SGR codes carried across line boundaries.
|
||||
#[napi]
|
||||
pub fn wrap_text_with_ansi(text: JsString, width: u32, tab_width: u32) -> Result<Vec<Utf16String>> {
|
||||
let text_u16 = text.into_utf16()?;
|
||||
let text = js::utf16(text)?;
|
||||
let tab_width = clamp_tab_width_for_ops(tab_width);
|
||||
let lines = wrap_text_with_ansi_impl(text_u16.as_slice(), width as usize, tab_width);
|
||||
Ok(lines.into_iter().map(build_utf16_string).collect())
|
||||
Ok(wrap_text_with_ansi_impl(&text, width as usize, tab_width)
|
||||
.into_iter()
|
||||
.map(build_utf16_string)
|
||||
.collect())
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
@@ -1280,30 +1281,36 @@ pub fn truncate_to_width(
|
||||
let ellipsis_kind = ellipsis_kind.unwrap_or(Ellipsis::Unicode);
|
||||
let pad = pad.unwrap_or(false);
|
||||
let tab_width = clamp_tab_width_for_ops(tab_width);
|
||||
|
||||
// Keep original handle so we can return it without allocating.
|
||||
let original = text;
|
||||
let text = js::utf16(text)?;
|
||||
Ok(truncate_to_width_impl(original, &text, max_width, ellipsis_kind, pad, tab_width))
|
||||
}
|
||||
|
||||
let text_u16 = text.into_utf16()?;
|
||||
let text = text_u16.as_slice();
|
||||
|
||||
fn truncate_to_width_impl<'env>(
|
||||
original: JsString<'env>,
|
||||
text: &[u16],
|
||||
max_width: usize,
|
||||
ellipsis_kind: Ellipsis,
|
||||
pad: bool,
|
||||
tab_width: usize,
|
||||
) -> Either<JsString<'env>, Utf16String> {
|
||||
// Fast path: early-exit width check
|
||||
let (text_w, exceeded) = visible_width_u16_up_to(text, max_width, tab_width);
|
||||
if !exceeded {
|
||||
if !pad {
|
||||
// Return original JsString handle: zero output allocation.
|
||||
return Ok(Either::A(original));
|
||||
return Either::A(original);
|
||||
}
|
||||
|
||||
if text_w < max_width {
|
||||
let mut out = Vec::with_capacity(text.len() + (max_width - text_w));
|
||||
out.extend_from_slice(text);
|
||||
out.resize(out.len() + (max_width - text_w), b' ' as u16);
|
||||
return Ok(Either::B(build_utf16_string(out)));
|
||||
return Either::B(build_utf16_string(out));
|
||||
}
|
||||
|
||||
// Exactly fits and padding requested: return original is still fine.
|
||||
return Ok(Either::A(original));
|
||||
return Either::A(original);
|
||||
}
|
||||
|
||||
// Map ellipsis kind to UTF-16 data and width
|
||||
@@ -1335,7 +1342,7 @@ pub fn truncate_to_width(
|
||||
if pad && w < max_width {
|
||||
out.resize(out.len() + (max_width - w), b' ' as u16);
|
||||
}
|
||||
return Ok(Either::B(build_utf16_string(out)));
|
||||
return Either::B(build_utf16_string(out));
|
||||
}
|
||||
|
||||
// Main truncation
|
||||
@@ -1439,7 +1446,7 @@ pub fn truncate_to_width(
|
||||
}
|
||||
}
|
||||
|
||||
Ok(Either::B(build_utf16_string(out)))
|
||||
Either::B(build_utf16_string(out))
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
@@ -1590,19 +1597,16 @@ pub fn slice_with_width(
|
||||
strict: Option<bool>,
|
||||
tab_width: u32,
|
||||
) -> Result<SliceResult> {
|
||||
let line_u16 = line.into_utf16()?;
|
||||
let line = line_u16.as_slice();
|
||||
let strict = strict.unwrap_or(false);
|
||||
|
||||
if length == 0 {
|
||||
return Ok(SliceResult { text: build_utf16_string(vec![]), width: 0 });
|
||||
}
|
||||
|
||||
let line = js::utf16(line)?;
|
||||
let strict = strict.unwrap_or(false);
|
||||
let tab_width = clamp_tab_width_for_ops(tab_width);
|
||||
let (out, w) =
|
||||
slice_with_width_impl(line, start_col as usize, length as usize, strict, tab_width);
|
||||
|
||||
Ok(SliceResult { text: build_utf16_string(out), width: crate::utils::clamp_u32(w as u64) })
|
||||
let (out, width) =
|
||||
slice_with_width_impl(&line, start_col as usize, length as usize, strict, tab_width);
|
||||
Ok(SliceResult { text: build_utf16_string(out), width: crate::utils::clamp_u32(width as u64) })
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
@@ -1812,12 +1816,10 @@ pub fn extract_segments(
|
||||
strict_after: bool,
|
||||
tab_width: u32,
|
||||
) -> Result<ExtractSegmentsResult> {
|
||||
let line_u16 = line.into_utf16()?;
|
||||
let line = line_u16.as_slice();
|
||||
|
||||
let line = js::utf16(line)?;
|
||||
let tab_width = clamp_tab_width_for_ops(tab_width);
|
||||
let (before, bw, after, aw) = extract_segments_impl(
|
||||
line,
|
||||
let (before, before_width, after, after_width) = extract_segments_impl(
|
||||
&line,
|
||||
before_end as usize,
|
||||
after_start as usize,
|
||||
after_len as usize,
|
||||
@@ -1827,9 +1829,9 @@ pub fn extract_segments(
|
||||
|
||||
Ok(ExtractSegmentsResult {
|
||||
before: build_utf16_string(before),
|
||||
before_width: crate::utils::clamp_u32(bw as u64),
|
||||
before_width: crate::utils::clamp_u32(before_width as u64),
|
||||
after: build_utf16_string(after),
|
||||
after_width: crate::utils::clamp_u32(aw as u64),
|
||||
after_width: crate::utils::clamp_u32(after_width as u64),
|
||||
})
|
||||
}
|
||||
|
||||
@@ -1842,9 +1844,9 @@ pub fn extract_segments(
|
||||
/// Tabs count as a fixed-width cell.
|
||||
#[napi]
|
||||
pub fn visible_width(text: JsString, tab_width: u32) -> Result<u32> {
|
||||
let text_u16 = text.into_utf16()?;
|
||||
let text = js::utf16(text)?;
|
||||
let tab_width = clamp_tab_width_for_ops(tab_width);
|
||||
Ok(crate::utils::clamp_u32(visible_width_u16(text_u16.as_slice(), tab_width) as u64))
|
||||
Ok(crate::utils::clamp_u32(visible_width_u16(&text, tab_width) as u64))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
|
||||
@@ -22,8 +22,10 @@ use napi::{
|
||||
bindgen_prelude::{Array, Either},
|
||||
};
|
||||
use napi_derive::napi;
|
||||
use pi_shell::rayon_global_pool_available;
|
||||
use rayon::prelude::*;
|
||||
|
||||
use crate::utok;
|
||||
use crate::{js, utok};
|
||||
|
||||
/// Tokenizer encoding to use.
|
||||
#[napi(string_enum)]
|
||||
@@ -70,9 +72,9 @@ impl Encoding {
|
||||
/// Count tokens in `input`.
|
||||
///
|
||||
/// `input` may be a single string or an array of strings; an array returns
|
||||
/// the sum across all elements. Always returns a single token total — use
|
||||
/// this for any aggregate budget question without paying a per-element napi
|
||||
/// crossing.
|
||||
/// the sum across all elements (counted in parallel when the global rayon pool
|
||||
/// is available). Always returns a single token total — use this for any
|
||||
/// aggregate budget question without paying a per-element napi crossing.
|
||||
///
|
||||
/// Measures user/model content, not wire-protocol tokens: BPE encodings
|
||||
/// use ordinary encoding (no special-token handling) and the Claude
|
||||
@@ -86,22 +88,34 @@ pub fn count_tokens(
|
||||
) -> napi::Result<u32> {
|
||||
let enc = Encoding::utok(encoding);
|
||||
match input {
|
||||
Either::A(js_str) => {
|
||||
let text = js_str.into_utf16()?;
|
||||
let (_, units) = text.as_slice().split_last().expect("napi UTF-16 buffer has a terminator");
|
||||
Ok(enc.count(units))
|
||||
},
|
||||
Either::A(text) => Ok(enc.count(&*js::utf16(text)?)),
|
||||
Either::B(array) => {
|
||||
let mut total = 0u32;
|
||||
// Node-API handles are thread-affine, so every element is read here on
|
||||
// the JS thread — into one buffer, so the batch costs one allocation
|
||||
// rather than one per string. Only the counting fans out.
|
||||
let mut units = Vec::new();
|
||||
let mut spans = Vec::with_capacity(array.len() as usize);
|
||||
for index in 0..array.len() {
|
||||
let text = array
|
||||
.get::<JsString>(index)?
|
||||
.ok_or_else(|| napi::Error::from_reason("array changed during token counting"))?
|
||||
.into_utf16()?;
|
||||
let (_, units) = text.as_slice().split_last().expect("napi UTF-16 buffer has a terminator");
|
||||
total += enc.count(units);
|
||||
.ok_or_else(|| napi::Error::from_reason("array changed during token counting"))?;
|
||||
spans.push(js::utf16_append(text, &mut units)?);
|
||||
}
|
||||
Ok(total)
|
||||
// Scheduling a Rayon job costs more than tokenizing a small prompt
|
||||
// batch. Keep those batches on the N-API thread; large batches still
|
||||
// amortize the pool handoff across enough independent strings.
|
||||
const PARALLEL_BATCH_MIN: usize = 16;
|
||||
Ok(if spans.len() >= PARALLEL_BATCH_MIN && rayon_global_pool_available() {
|
||||
spans
|
||||
.par_iter()
|
||||
.map(|span| enc.count(&units[span.clone()]))
|
||||
.sum()
|
||||
} else {
|
||||
spans
|
||||
.iter()
|
||||
.map(|span| enc.count(&units[span.clone()]))
|
||||
.sum()
|
||||
})
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
@@ -8,11 +8,13 @@
|
||||
//! bit-identical to the TS versions and integer results are exactly equal.
|
||||
|
||||
use napi::{
|
||||
Error, Result, Status,
|
||||
bindgen_prelude::{Float32Array, Float64Array, Uint32Array},
|
||||
Error, JsString, Result, Status,
|
||||
bindgen_prelude::{Array, Float32Array, Float64Array, Uint32Array},
|
||||
};
|
||||
use napi_derive::napi;
|
||||
|
||||
use crate::js;
|
||||
|
||||
fn invalid<T>(message: &str) -> Result<T> {
|
||||
Err(Error::new(Status::InvalidArg, message))
|
||||
}
|
||||
@@ -261,20 +263,26 @@ fn jaccard_sorted(a: &[Box<str>], b: &[Box<str>]) -> f64 {
|
||||
reason = "mul_add rounds differently; bit-exact with the TS loops is the contract"
|
||||
)]
|
||||
pub fn mmr_rerank_indices(
|
||||
contents: Vec<String>,
|
||||
#[napi(ts_arg_type = "Array<string>")] contents: Array,
|
||||
scores: Float64Array,
|
||||
lambda_param: f64,
|
||||
top_k: u32,
|
||||
) -> Result<Uint32Array> {
|
||||
if scores.len() != contents.len() {
|
||||
if scores.len() != contents.len() as usize {
|
||||
return invalid("scores length must equal contents length");
|
||||
}
|
||||
let limit = top_k as usize;
|
||||
let count = contents.len();
|
||||
let count = contents.len() as usize;
|
||||
if limit == 0 || count == 0 {
|
||||
return Ok(Uint32Array::new(Vec::new()));
|
||||
}
|
||||
let sets: Vec<Vec<Box<str>>> = contents.iter().map(|text| word_set(text)).collect();
|
||||
let mut sets = Vec::with_capacity(count);
|
||||
for index in 0..contents.len() {
|
||||
let content = contents
|
||||
.get::<JsString>(index)?
|
||||
.ok_or_else(|| Error::new(Status::InvalidArg, "contents changed during reranking"))?;
|
||||
sets.push(word_set(&js::utf8(content)?));
|
||||
}
|
||||
let mut selected: Vec<u32> = Vec::with_capacity(limit.min(count));
|
||||
selected.push(0);
|
||||
let mut remaining: Vec<u32> = (1..count as u32).collect();
|
||||
|
||||
@@ -3268,6 +3268,7 @@ mod tests {
|
||||
"base64",
|
||||
"basename",
|
||||
"cat",
|
||||
"cksum",
|
||||
"cmp",
|
||||
"combine",
|
||||
"comm",
|
||||
@@ -4513,7 +4514,7 @@ mod tests {
|
||||
.expect("process substitution should not hang");
|
||||
|
||||
assert_eq!(result.exit_code, Some(1));
|
||||
assert!(output.contains("-a\n+b\n"), "diff output missing changed lines: {output:?}");
|
||||
assert!(output.contains("< a\n---\n> b\n"), "diff output missing changed lines: {output:?}");
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
|
||||
+12
-1
@@ -512,10 +512,21 @@ fn unwrap_transparent_background_wrapper(pipeline: &ast::Pipeline) -> Option<ast
|
||||
};
|
||||
let mut unwrapped = simple_cmd.clone();
|
||||
let suffix = unwrapped.suffix.as_mut()?;
|
||||
let operand_index = suffix
|
||||
let mut operand_index = suffix
|
||||
.0
|
||||
.iter()
|
||||
.position(|item| matches!(item, CommandPrefixOrSuffixItem::Word(_)))?;
|
||||
// A leading `--` only terminates the wrapper's own options
|
||||
// (`nohup -- cmd &`): drop it and take the next word as the operand.
|
||||
if let CommandPrefixOrSuffixItem::Word(word) = &suffix.0[operand_index]
|
||||
&& word.value == "--"
|
||||
{
|
||||
suffix.0.remove(operand_index);
|
||||
operand_index = suffix
|
||||
.0
|
||||
.iter()
|
||||
.position(|item| matches!(item, CommandPrefixOrSuffixItem::Word(_)))?;
|
||||
}
|
||||
let CommandPrefixOrSuffixItem::Word(operand_word) = suffix.0.remove(operand_index) else {
|
||||
return None;
|
||||
};
|
||||
|
||||
@@ -17,6 +17,8 @@
|
||||
### Fixed
|
||||
|
||||
- Fixed a prompt cancelled during turn setup (Esc while the pre-stream spinner is up, after dispatch had started) vanishing entirely: it was never persisted to the session — so the `/tree` and `/branch` selectors had nothing to rewind to — and was not returned to the editor either, while its optimistic transcript row kept lingering. A prompt dropped before reaching the agent (abort or usage-preflight denial racing setup) is now handed back: the stale transcript row is removed and the typed text and image attachments are restored to the editor for editing.
|
||||
- Fixed macOS `top`-style single-dash long options in the `top` shell builtin: `top -l 2 -pid 56943 -stats pid,cpu,th,mem,pstate` previously failed with `invalid value 'id' for '--pid <PIDS>'` because clap read `-pid` as `-p id`. Single-dash long spellings now parse, and `-stats` selects and orders output columns using macOS stat keys (`pid`, `cpu`, `th`, `mem`, `pstate`, ...).
|
||||
- Fixed a sweep of GNU/BSD compatibility gaps in the built-in shell utilities, found by auditing every builtin against its real counterpart: `timeout` gained `-s`/`-k`/`--preserve-status`/`--foreground`/`-v`, GNU exit codes (124/125/137), signal delivery to the child process group with `-k` SIGKILL escalation, and `timeout 0` disabling the limit; `diff` gained normal-format default output, `-w`/`-b`/`-B`/`-i`/`-x`/`-L`/`-s`/`--strip-trailing-cr`, context format (`-c`/`-C`), bundled flags (`-ru`, `-urN`), timestamped unified headers, and no longer recurses directories without `-r`; `find` fixed inverted `-newerXY` timestamp comparisons, anchored `-regex` to whole paths, and gained BSD `-perm +mode`, `-type f,d` lists, `-size` `T`/`P` suffixes, ISO dates in `-newermt`, and BSD leading flags `-E`/`-x`/`-s`; `date` gained BSD `-r <epoch>`, `-v` adjustments, and `-j -f` strptime parsing, and `-I` no longer swallows a following `+FORMAT`; `tail`/`head` accept obsolete `-N`/`+N` counts at any argv position with any file count, `tail -r -n N` works, and `head` continues past per-file I/O errors with GNU header/separator placement; `rg` resolves `-s`/`-i`/`-S` by last occurrence, accepts `--no-config`/`-j`/`--threads`/`--no-column`, implements `--path-separator`, and emits clean NUL-delimited output under `-0`/`-l0`; `stat` prints integer epochs for `%X`/`%Y`/`%Z` (bash arithmetic on `stat -c %Y` works) and gained BSD `-s`/`-x` output modes plus `-t` time formatting; `cksum` is now registered (multi-algorithm `cksum -a sha256`); `truncate` implements `-o`/`--io-blocks` (previously silently truncated to the raw byte count), accepts `b` (512-byte) suffix and BSD `=` prefix; `sleep`/`timeout` accept `infinity` and keep sub-millisecond precision; `yes` and `errno` accept hyphen-prefixed operands (`yes -n`, `errno -2`); `nohup -- cmd` no longer tries to run `--` (including backgrounded via the brush wrapper); `which` gained BSD `-s` and errors on zero operands; `kill` accepts attached values (`-s9`, `-sKILL`, `-l9`) and maps exit statuses above 128 (`kill -l 137` → `KILL`).
|
||||
|
||||
## [17.3.8] - 2026-08-19
|
||||
|
||||
|
||||
@@ -0,0 +1,266 @@
|
||||
/**
|
||||
* Micro-benchmarks for native text primitives vs standard JS/Bun equivalents
|
||||
* using the mitata benchmarking framework.
|
||||
*
|
||||
* Run with: `bun packages/natives/bench/text.ts`
|
||||
*
|
||||
* Every bench body pipes its result through `do_not_optimize`. Without it JSC
|
||||
* dead-code-eliminates pure calls with discarded results after warmup, which
|
||||
* reports sub-nanosecond phantoms (e.g. string-width at ~180 ps/iter).
|
||||
*/
|
||||
|
||||
import cliTruncate from "cli-truncate";
|
||||
import * as diff from "diff";
|
||||
import { countTokens as gptCountTokens } from "gpt-tokenizer/model/gpt-4o";
|
||||
import { bench, do_not_optimize, run, summary } from "mitata";
|
||||
import sliceAnsi from "slice-ansi";
|
||||
import stringWidth from "string-width";
|
||||
import wrapAnsi from "wrap-ansi";
|
||||
// The TS-side width measurer every TUI render path actually calls. Backed by
|
||||
// Bun.stringWidth with a printable-ASCII fast path; the N-API `visibleWidth`
|
||||
// below is only the raw binding (its ~150 ns floor is per-call FFI overhead:
|
||||
// UTF-16 -> UTF-8 marshal + result box, not the width algorithm).
|
||||
import { visibleWidth as tuiVisibleWidth } from "../../tui/src/utils";
|
||||
import {
|
||||
countTokens,
|
||||
diffLines,
|
||||
extractSegments,
|
||||
highlightCode,
|
||||
sliceWithWidth,
|
||||
truncateToWidth,
|
||||
visibleWidth,
|
||||
wrapTextWithAnsi,
|
||||
} from "../native/index.js";
|
||||
|
||||
const testCases = {
|
||||
shortAscii: "const x = 42; // standard code snippet",
|
||||
longAscii:
|
||||
"This is a much longer line of text designed to test how measurement scales when lines are wider in terminal buffers. ".repeat(
|
||||
4,
|
||||
),
|
||||
ansiStyled:
|
||||
"\x1b[1m\x1b[38;2;100;200;255mfunction\x1b[0m \x1b[38;2;255;215;0mrenderTerminal\x1b[0m(\x1b[38;2;150;150;150mprops\x1b[0m: \x1b[38;2;80;250;123mTerminalProps\x1b[0m) {\x1b[38;2;98;114;164m // styled output\x1b[0m",
|
||||
emojiCjk: "⚡ Status: 🚀 Deploying to 東京 (Tokyo) cluster 🎯 [5/10 completed] 🌸",
|
||||
multilineAnsi: (
|
||||
"\x1b[32m✔ Loaded config successfully\x1b[0m\n" +
|
||||
"\x1b[34mℹ Connecting to server at 127.0.0.1:8080...\x1b[0m\n" +
|
||||
"\x1b[33m⚠ Warning: high memory usage detected in worker pool\x1b[0m\n" +
|
||||
"\x1b[31m✖ Error: failed to establish connection to database replica\x1b[0m\n" +
|
||||
"Stack trace: at ConnectionPool.acquire (/app/src/db.ts:142:18)\n"
|
||||
).repeat(3),
|
||||
diffOld:
|
||||
"import { a, b, c } from 'pkg';\n\nfunction main() {\n console.log('hello');\n const x = 1;\n return x + 2;\n}\n",
|
||||
diffNew:
|
||||
"import { a, b, c, d } from 'pkg';\n\nfunction main() {\n console.log('hello world');\n const x = 2;\n const y = 3;\n return x + y;\n}\n",
|
||||
tokenArray: [
|
||||
"You are a helpful assistant with access to tools.",
|
||||
"User prompt: please inspect the code in src/index.ts and summarize findings.",
|
||||
"System message: running tool call 'read_file' with arguments {'path': 'src/index.ts'}.",
|
||||
"File content: export function run() { console.log('active'); }".repeat(5),
|
||||
],
|
||||
colors: {
|
||||
comment: "\x1b[38;2;98;114;164m",
|
||||
keyword: "\x1b[38;2;255;121;198m",
|
||||
function: "\x1b[38;2;80;250;123m",
|
||||
variable: "\x1b[38;2;248;248;242m",
|
||||
string: "\x1b[38;2;241;250;140m",
|
||||
number: "\x1b[38;2;189;147;249m",
|
||||
type: "\x1b[38;2;139;233;253m",
|
||||
operator: "\x1b[38;2;255;121;198m",
|
||||
punctuation: "\x1b[38;2;248;248;242m",
|
||||
},
|
||||
};
|
||||
|
||||
// Each width bench cycles a pool of 64 distinct strings. This defeats
|
||||
// constant-argument hoisting in pure comparators (`do_not_optimize` only
|
||||
// protects the result) and mirrors a real redraw workload: a frame re-measures
|
||||
// the same visible lines every paint, so pi-tui's bounded width memo hits —
|
||||
// but the pool is far larger than any cache that merely fits the bench.
|
||||
const WIDTH_VARIANT_COUNT = 64;
|
||||
|
||||
function makeWidthVariants(base: string): string[] {
|
||||
const variants: string[] = new Array(WIDTH_VARIANT_COUNT);
|
||||
for (let i = 0; i < WIDTH_VARIANT_COUNT; i++) {
|
||||
variants[i] = `${base} ${String(i).padStart(2, "0")}`;
|
||||
}
|
||||
return variants;
|
||||
}
|
||||
|
||||
const widthInputVariants = {
|
||||
shortAscii: makeWidthVariants(testCases.shortAscii),
|
||||
longAscii: makeWidthVariants(testCases.longAscii),
|
||||
ansiStyled: makeWidthVariants(testCases.ansiStyled),
|
||||
emojiCjk: makeWidthVariants(testCases.emojiCjk),
|
||||
} as const;
|
||||
let widthInputVariantIndex = 0;
|
||||
|
||||
function nextWidthInput(kind: keyof typeof widthInputVariants): string {
|
||||
return widthInputVariants[kind][widthInputVariantIndex++ & (WIDTH_VARIANT_COUNT - 1)];
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// 1. visibleWidth: pi-tui hot path vs raw N-API binding vs Bun.stringWidth vs
|
||||
// string-width npm package
|
||||
// ============================================================================
|
||||
|
||||
summary(() => {
|
||||
bench("visibleWidth: short ascii (pi-tui)", () => do_not_optimize(tuiVisibleWidth(nextWidthInput("shortAscii"))));
|
||||
bench("visibleWidth: short ascii (native N-API)", () =>
|
||||
do_not_optimize(visibleWidth(nextWidthInput("shortAscii"), 3)),
|
||||
);
|
||||
bench("visibleWidth: short ascii (Bun.stringWidth)", () =>
|
||||
do_not_optimize(Bun.stringWidth(nextWidthInput("shortAscii"))),
|
||||
);
|
||||
bench("visibleWidth: short ascii (string-width npm)", () =>
|
||||
do_not_optimize(stringWidth(nextWidthInput("shortAscii"))),
|
||||
);
|
||||
});
|
||||
|
||||
summary(() => {
|
||||
bench("visibleWidth: long ascii (pi-tui)", () => do_not_optimize(tuiVisibleWidth(nextWidthInput("longAscii"))));
|
||||
bench("visibleWidth: long ascii (native N-API)", () =>
|
||||
do_not_optimize(visibleWidth(nextWidthInput("longAscii"), 3)),
|
||||
);
|
||||
bench("visibleWidth: long ascii (Bun.stringWidth)", () =>
|
||||
do_not_optimize(Bun.stringWidth(nextWidthInput("longAscii"))),
|
||||
);
|
||||
bench("visibleWidth: long ascii (string-width npm)", () =>
|
||||
do_not_optimize(stringWidth(nextWidthInput("longAscii"))),
|
||||
);
|
||||
});
|
||||
|
||||
summary(() => {
|
||||
bench("visibleWidth: ansi styled (pi-tui)", () => do_not_optimize(tuiVisibleWidth(nextWidthInput("ansiStyled"))));
|
||||
bench("visibleWidth: ansi styled (native N-API)", () =>
|
||||
do_not_optimize(visibleWidth(nextWidthInput("ansiStyled"), 3)),
|
||||
);
|
||||
bench("visibleWidth: ansi styled (Bun.stringWidth)", () =>
|
||||
do_not_optimize(Bun.stringWidth(nextWidthInput("ansiStyled"))),
|
||||
);
|
||||
bench("visibleWidth: ansi styled (string-width npm)", () =>
|
||||
do_not_optimize(stringWidth(nextWidthInput("ansiStyled"))),
|
||||
);
|
||||
});
|
||||
|
||||
summary(() => {
|
||||
bench("visibleWidth: emoji / CJK (pi-tui)", () => do_not_optimize(tuiVisibleWidth(nextWidthInput("emojiCjk"))));
|
||||
bench("visibleWidth: emoji / CJK (native N-API)", () =>
|
||||
do_not_optimize(visibleWidth(nextWidthInput("emojiCjk"), 3)),
|
||||
);
|
||||
bench("visibleWidth: emoji / CJK (Bun.stringWidth)", () =>
|
||||
do_not_optimize(Bun.stringWidth(nextWidthInput("emojiCjk"))),
|
||||
);
|
||||
bench("visibleWidth: emoji / CJK (string-width npm)", () =>
|
||||
do_not_optimize(stringWidth(nextWidthInput("emojiCjk"))),
|
||||
);
|
||||
});
|
||||
|
||||
// ============================================================================
|
||||
// 2. truncateToWidth: Native vs cli-truncate
|
||||
// ============================================================================
|
||||
|
||||
summary(() => {
|
||||
bench("truncateToWidth: long ascii (native)", () =>
|
||||
do_not_optimize(truncateToWidth(testCases.longAscii, 40, 0, false, 3)),
|
||||
);
|
||||
bench("truncateToWidth: long ascii (cli-truncate)", () => do_not_optimize(cliTruncate(testCases.longAscii, 40)));
|
||||
});
|
||||
|
||||
summary(() => {
|
||||
bench("truncateToWidth: ansi styled (native)", () =>
|
||||
do_not_optimize(truncateToWidth(testCases.ansiStyled, 40, 0, false, 3)),
|
||||
);
|
||||
bench("truncateToWidth: ansi styled (cli-truncate)", () => do_not_optimize(cliTruncate(testCases.ansiStyled, 40)));
|
||||
});
|
||||
|
||||
bench("truncateToWidth: fits no-alloc (native)", () =>
|
||||
do_not_optimize(truncateToWidth(testCases.shortAscii, 100, 0, false, 3)),
|
||||
);
|
||||
bench("truncateToWidth: pads with spaces (native)", () =>
|
||||
do_not_optimize(truncateToWidth(testCases.shortAscii, 60, 0, true, 3)),
|
||||
);
|
||||
|
||||
// ============================================================================
|
||||
// 3. sliceWithWidth: Native vs slice-ansi
|
||||
// ============================================================================
|
||||
|
||||
summary(() => {
|
||||
bench("sliceWithWidth: ascii slice (native)", () =>
|
||||
do_not_optimize(sliceWithWidth(testCases.shortAscii, 10, 20, false, 3)),
|
||||
);
|
||||
bench("sliceWithWidth: ascii slice (slice-ansi)", () => do_not_optimize(sliceAnsi(testCases.shortAscii, 10, 30)));
|
||||
});
|
||||
|
||||
summary(() => {
|
||||
bench("sliceWithWidth: ansi styled slice (native)", () =>
|
||||
do_not_optimize(sliceWithWidth(testCases.ansiStyled, 15, 30, false, 3)),
|
||||
);
|
||||
bench("sliceWithWidth: ansi styled slice (slice-ansi)", () =>
|
||||
do_not_optimize(sliceAnsi(testCases.ansiStyled, 15, 45)),
|
||||
);
|
||||
});
|
||||
|
||||
// ============================================================================
|
||||
// 4. wrapTextWithAnsi: Native vs wrap-ansi
|
||||
// ============================================================================
|
||||
|
||||
summary(() => {
|
||||
bench("wrapTextWithAnsi: single line (native)", () =>
|
||||
do_not_optimize(wrapTextWithAnsi(testCases.ansiStyled, 30, 3)),
|
||||
);
|
||||
bench("wrapTextWithAnsi: single line (wrap-ansi)", () =>
|
||||
do_not_optimize(wrapAnsi(testCases.ansiStyled, 30, { hard: true })),
|
||||
);
|
||||
});
|
||||
|
||||
summary(() => {
|
||||
bench("wrapTextWithAnsi: multiline logs (native)", () =>
|
||||
do_not_optimize(wrapTextWithAnsi(testCases.multilineAnsi, 60, 3)),
|
||||
);
|
||||
bench("wrapTextWithAnsi: multiline logs (wrap-ansi)", () =>
|
||||
do_not_optimize(wrapAnsi(testCases.multilineAnsi, 60, { hard: true })),
|
||||
);
|
||||
});
|
||||
|
||||
// ============================================================================
|
||||
// 5. diffLines: Native vs jsdiff (diff npm package)
|
||||
// ============================================================================
|
||||
|
||||
summary(() => {
|
||||
bench("diffLines: source files (native)", () => do_not_optimize(diffLines(testCases.diffOld, testCases.diffNew)));
|
||||
bench("diffLines: source files (diff npm)", () =>
|
||||
do_not_optimize(diff.diffLines(testCases.diffOld, testCases.diffNew)),
|
||||
);
|
||||
});
|
||||
|
||||
// ============================================================================
|
||||
// 6. countTokens: Native (o200k_base) vs gpt-tokenizer (pure JS)
|
||||
// ============================================================================
|
||||
|
||||
summary(() => {
|
||||
bench("countTokens: single string (native)", () => do_not_optimize(countTokens(testCases.longAscii)));
|
||||
bench("countTokens: single string (gpt-tokenizer)", () => do_not_optimize(gptCountTokens(testCases.longAscii)));
|
||||
});
|
||||
|
||||
summary(() => {
|
||||
bench("countTokens: array of strings (native)", () => do_not_optimize(countTokens(testCases.tokenArray)));
|
||||
bench("countTokens: array of strings (gpt-tokenizer)", () => {
|
||||
let total = 0;
|
||||
for (const s of testCases.tokenArray) total += gptCountTokens(s);
|
||||
do_not_optimize(total);
|
||||
});
|
||||
});
|
||||
|
||||
// ============================================================================
|
||||
// 7. Specialized native primitives (standalone)
|
||||
// ============================================================================
|
||||
|
||||
bench("extractSegments: ansi overlay (native)", () =>
|
||||
do_not_optimize(extractSegments(testCases.ansiStyled, 15, 25, 20, false, 3)),
|
||||
);
|
||||
|
||||
bench("highlightCode: rust snippet (native)", () =>
|
||||
do_not_optimize(highlightCode('fn main() { println!("hello"); }', "rust", testCases.colors)),
|
||||
);
|
||||
|
||||
await run();
|
||||
@@ -40,11 +40,19 @@
|
||||
"gen:native": "bun scripts/embed-native.ts",
|
||||
"gen:native:reset": "bun scripts/embed-native.ts --reset",
|
||||
"gen:npm": "bun scripts/gen-npm-packages.ts",
|
||||
"bench": "bun bench/grep.ts"
|
||||
"bench": "bun bench/grep.ts",
|
||||
"bench:text": "bun bench/text.ts"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@napi-rs/cli": "catalog:",
|
||||
"@types/bun": "catalog:"
|
||||
"@types/bun": "catalog:",
|
||||
"cli-truncate": "6.1.1",
|
||||
"diff": "9.0.0",
|
||||
"gpt-tokenizer": "4.0.0",
|
||||
"mitata": "1.0.34",
|
||||
"slice-ansi": "9.0.0",
|
||||
"string-width": "8.2.2",
|
||||
"wrap-ansi": "10.0.0"
|
||||
},
|
||||
"engines": {
|
||||
"bun": ">=1.3.14"
|
||||
|
||||
@@ -7,16 +7,16 @@ import {
|
||||
astEdit,
|
||||
astMatch,
|
||||
blockRangeAt,
|
||||
countTokens,
|
||||
Encoding,
|
||||
executeShell,
|
||||
FileType,
|
||||
countTokens,
|
||||
fuzzyFind,
|
||||
type GlobMatch,
|
||||
GrepOutputMode,
|
||||
getSupportedLanguages,
|
||||
glob,
|
||||
grep,
|
||||
Encoding,
|
||||
highlightCode,
|
||||
htmlToMarkdown,
|
||||
invalidateFsScanCache,
|
||||
|
||||
+54
-59
@@ -225,9 +225,7 @@ export function getSegmenter(): Intl.Segmenter {
|
||||
// added back so width matches the native truncate/slice/wrap helpers.
|
||||
const OSC66_SPAN_REGEX = /\x1b\]66;([^;]*);([\s\S]*?)(?:\x07|\x1b\\)/g;
|
||||
const OSC66_PREFIX = "\x1b]66;";
|
||||
const ESC = "\x1b";
|
||||
const TAB = "\t";
|
||||
const LONG_WIDTH_FAST_PATH_MIN = 128;
|
||||
const PRINTABLE_ASCII_REGEX = /^[\u0020-\u007e]*$/;
|
||||
|
||||
// Pin Bun.stringWidth semantics to the native width engine and guard against Bun
|
||||
// default drift: strip ANSI/OSC (don't count escape bytes) and treat
|
||||
@@ -243,8 +241,6 @@ const STRING_WIDTH_OPTS = { countAnsiEscapeCodes: false, ambiguousIsNarrow: true
|
||||
// `setHangulCompatibilityJamoWidth`; mirror the same correction here so the TS
|
||||
// width stays in parity with the native truncate/slice/wrap model — and so the
|
||||
// hardware cursor column lands on the actual glyph during Korean IME input.
|
||||
const HANGUL_COMPAT_JAMO_REGEX = /[\u3131-\u318e]/;
|
||||
const HANGUL_COMPAT_JAMO_GLOBAL_REGEX = /[\u3131-\u318e]/g;
|
||||
const HANGUL_FILLER_CODE_POINT = 0x3164;
|
||||
// `Bun.stringWidth` counts every code point in the Compatibility Jamo block as
|
||||
// 2 cells (even the U+3164 filler that `unicode-width` treats as zero-width).
|
||||
@@ -274,19 +270,28 @@ function hangulCompatibilityJamoTargetWidth(): 1 | 2 | null {
|
||||
// crates/pi-natives/src/text.rs, including the rule that the zero-width filler
|
||||
// (U+3164) is never widened past the narrow correction (a wide terminal still
|
||||
// renders it at its Unicode width of 0).
|
||||
function correctHangulCompatibilityJamoWidth(width: number, str: string): number {
|
||||
if (!HANGUL_COMPAT_JAMO_REGEX.test(str)) return width;
|
||||
function correctHangulCompatibilityJamoWidth(
|
||||
width: number,
|
||||
compatibilityJamoCount: number,
|
||||
fillerCount: number,
|
||||
): number {
|
||||
if (compatibilityJamoCount === 0) return width;
|
||||
const target = hangulCompatibilityJamoTargetWidth();
|
||||
let corrected = width;
|
||||
HANGUL_COMPAT_JAMO_GLOBAL_REGEX.lastIndex = 0;
|
||||
for (let m = HANGUL_COMPAT_JAMO_GLOBAL_REGEX.exec(str); m !== null; m = HANGUL_COMPAT_JAMO_GLOBAL_REGEX.exec(str)) {
|
||||
const unicodeWidth = m[0].codePointAt(0) === HANGUL_FILLER_CODE_POINT ? 0 : 2;
|
||||
const finalWidth = target === null || (unicodeWidth === 0 && target > 1) ? unicodeWidth : target;
|
||||
corrected += finalWidth - HANGUL_COMPAT_JAMO_BUN_WIDTH;
|
||||
}
|
||||
return corrected;
|
||||
return target === 1 ? width - compatibilityJamoCount : width - fillerCount * HANGUL_COMPAT_JAMO_BUN_WIDTH;
|
||||
}
|
||||
|
||||
// Terminal redraws re-measure the same visible lines every frame, usually as
|
||||
// the same string objects (JSC caches their hashes, so repeat lookups are
|
||||
// O(1) — cheaper than even the ASCII fast scan). Strings longer than the
|
||||
// length gate skip the cache entirely: hashing them costs as much as measuring
|
||||
// them, and retaining them would pin large render buffers. Worst-case
|
||||
// retention is MAX * MAX_LEN UTF-16 units (~2 MiB); cleared when the width
|
||||
// configuration epoch changes.
|
||||
const VISIBLE_WIDTH_CACHE_MAX = 2048;
|
||||
const VISIBLE_WIDTH_CACHE_MAX_LEN = 512;
|
||||
const visibleWidthCache = new Map<string, number>();
|
||||
let visibleWidthCacheEpoch = widthConfigEpoch;
|
||||
|
||||
/**
|
||||
* Visible width of a string in terminal columns, excluding ANSI/OSC escapes.
|
||||
*
|
||||
@@ -296,65 +301,51 @@ function correctHangulCompatibilityJamoWidth(width: number, str: string): number
|
||||
*/
|
||||
export function visibleWidth(str: string): number {
|
||||
if (!str) return 0;
|
||||
|
||||
// Long non-escape text is faster through Bun's native scanner than through
|
||||
// a JS printable-ASCII prepass. Escape-bearing strings stay on the scanner
|
||||
// below so CSI/OSC-heavy render output can still bail out at the first ESC.
|
||||
if (str.length >= LONG_WIDTH_FAST_PATH_MIN && !str.includes(ESC)) {
|
||||
let width = Bun.stringWidth(str, STRING_WIDTH_OPTS);
|
||||
let tabCount = 0;
|
||||
for (let tabIndex = str.indexOf(TAB); tabIndex !== -1; tabIndex = str.indexOf(TAB, tabIndex + 1)) {
|
||||
tabCount++;
|
||||
const cacheable = str.length <= VISIBLE_WIDTH_CACHE_MAX_LEN;
|
||||
if (cacheable) {
|
||||
if (visibleWidthCacheEpoch !== widthConfigEpoch) {
|
||||
visibleWidthCache.clear();
|
||||
visibleWidthCacheEpoch = widthConfigEpoch;
|
||||
}
|
||||
if (tabCount > 0) width += tabCount * DEFAULT_TAB_WIDTH;
|
||||
return correctHangulCompatibilityJamoWidth(width, str);
|
||||
const cached = visibleWidthCache.get(str);
|
||||
if (cached !== undefined) return cached;
|
||||
}
|
||||
|
||||
// This regex compiles to a native ASCII scan, cheaper than Bun's width
|
||||
// scanner for the overwhelmingly common source-code path.
|
||||
if (PRINTABLE_ASCII_REGEX.test(str)) {
|
||||
if (cacheable) {
|
||||
if (visibleWidthCache.size >= VISIBLE_WIDTH_CACHE_MAX) visibleWidthCache.clear();
|
||||
visibleWidthCache.set(str, str.length);
|
||||
}
|
||||
return str.length;
|
||||
}
|
||||
|
||||
let tabCount = 0;
|
||||
let i = 0;
|
||||
for (; i < str.length; i++) {
|
||||
let compatibilityJamoCount = 0;
|
||||
let fillerCount = 0;
|
||||
let hasEsc = false;
|
||||
for (let i = 0; i < str.length; i++) {
|
||||
const code = str.charCodeAt(i);
|
||||
if (code < 0x20 || code > 0x7e) {
|
||||
if (code === 0x09) {
|
||||
tabCount++;
|
||||
continue;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (i === str.length) {
|
||||
return tabCount === 0 ? str.length : str.length + tabCount * (DEFAULT_TAB_WIDTH - 1);
|
||||
}
|
||||
|
||||
if (tabCount === 0) {
|
||||
let tabIndex = str.indexOf(TAB, i + 1);
|
||||
if (tabIndex !== -1) {
|
||||
tabCount = 1;
|
||||
for (tabIndex = str.indexOf(TAB, tabIndex + 1); tabIndex !== -1; tabIndex = str.indexOf(TAB, tabIndex + 1)) {
|
||||
tabCount++;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for (let tabIndex = str.indexOf(TAB, i + 1); tabIndex !== -1; tabIndex = str.indexOf(TAB, tabIndex + 1)) {
|
||||
if (code === 0x09) {
|
||||
tabCount++;
|
||||
} else if (code === 0x1b) {
|
||||
hasEsc = true;
|
||||
} else if (code >= 0x3131 && code <= 0x318e) {
|
||||
compatibilityJamoCount++;
|
||||
if (code === HANGUL_FILLER_CODE_POINT) fillerCount++;
|
||||
}
|
||||
}
|
||||
|
||||
// `Bun.stringWidth` is a JSC builtin (no per-call N-API number box, unlike
|
||||
// the native scanner that traps under Bun 1.3.x GC/N-API load). It strips
|
||||
// CSI/OSC to zero cells and shares the native engine's UAX#11 width tables.
|
||||
let width = Bun.stringWidth(str, STRING_WIDTH_OPTS);
|
||||
if (tabCount > 0) width += tabCount * DEFAULT_TAB_WIDTH;
|
||||
|
||||
// OSC 66: add back each stripped span as `scale * (explicit w ?? payload
|
||||
// width)`. Matched rather than replaced to avoid reallocating the string.
|
||||
if (str.includes(OSC66_PREFIX, i)) {
|
||||
if (hasEsc && str.includes(OSC66_PREFIX)) {
|
||||
OSC66_SPAN_REGEX.lastIndex = 0;
|
||||
for (let m = OSC66_SPAN_REGEX.exec(str); m !== null; m = OSC66_SPAN_REGEX.exec(str)) {
|
||||
let scale = 1;
|
||||
let explicit: number | undefined;
|
||||
for (const part of m[1].split(":")) {
|
||||
// metadata keys are single chars, e.g. `s=2`, `w=5`
|
||||
if (part.indexOf("=") !== 1) continue;
|
||||
const value = Number.parseInt(part.slice(2), 10);
|
||||
if (!Number.isFinite(value)) continue;
|
||||
@@ -368,11 +359,15 @@ export function visibleWidth(str: string): number {
|
||||
}
|
||||
}
|
||||
|
||||
return correctHangulCompatibilityJamoWidth(width, str);
|
||||
width = correctHangulCompatibilityJamoWidth(width, compatibilityJamoCount, fillerCount);
|
||||
if (cacheable) {
|
||||
if (visibleWidthCache.size >= VISIBLE_WIDTH_CACHE_MAX) visibleWidthCache.clear();
|
||||
visibleWidthCache.set(str, width);
|
||||
}
|
||||
return width;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when a row carries a Kitty OSC 66 text-sizing span (`\x1b]66;…`).
|
||||
* Scaled spans must bypass wrapping/padding and, when scaled up, reserve the
|
||||
* terminal rows their multicell glyphs flow into.
|
||||
*/
|
||||
|
||||
Reference in New Issue
Block a user