Merge upstream/main into fix/advisor-refusal-fallback

This commit is contained in:
Mantas Vidutis
2026-08-05 11:43:16 -07:00
287 changed files with 13802 additions and 8090 deletions
+6 -1
View File
@@ -307,7 +307,12 @@ jobs:
- name: Test workspace packages and repo scripts (TS)
env:
OMP_TEST_CONCURRENCY: "4"
run: bun run ci:test:ts:workspace
run: |
bun run ci:test:ts:workspace
# Not `test:scripts`: scripts/musl-release.test.ts fails on main
# (its install.sh smoke-check executes a fake binary), so running
# the whole group here would red this job on an unrelated break.
bun test scripts/release.test.ts
test_coding_agent_singleton:
name: Test coding-agent singleton/global-state (TS)
Generated
+25 -25
View File
@@ -514,9 +514,9 @@ checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6"
[[package]]
name = "base64"
version = "0.23.0"
version = "0.23.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b25655df2c3cdd83c5e5b293b88acd880332b2ddadd7c30ac43144fdc0033da9"
checksum = "ac07cdecf99051d9a5238b80f35af32cdeba5b336e55d957b318b50137e18da5"
[[package]]
name = "base64-simd"
@@ -1963,9 +1963,9 @@ dependencies = [
[[package]]
name = "encoding_rs_io"
version = "0.1.7"
version = "0.1.8"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1cc3c5651fb62ab8aa3103998dade57efdd028544bd300516baa31840c252a83"
checksum = "fba3fe847045ecff794b9c138293a80db914678c453ad63fbf0c6a9eb6e00b22"
dependencies = [
"encoding_rs",
]
@@ -2573,9 +2573,9 @@ checksum = "e4eba85ea1d0a966a983acd07deee566e67395d2d96b6fb39e62b5a833f1eb0b"
[[package]]
name = "globset"
version = "0.4.19"
version = "0.4.20"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e47d37d2ae4464254884b60ab7071be2b876a9c35b696bd018ddcc76847309cd"
checksum = "07c34a9410465b45bd9787443bc7370f37735bad04b0f0cd57ff1a3186c98988"
dependencies = [
"aho-corasick",
"bstr",
@@ -2783,13 +2783,13 @@ checksum = "c9356095b4b41197bba32173600e1582792cda618f65d12f68e2e77d273413c5"
[[package]]
name = "html-to-markdown-rs"
version = "3.10.2"
version = "3.10.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bf23a50d1e4f5ca342b75308a309466882f672a324654a363e26ff723a40e791"
checksum = "1b0fdf3ba00130a03686f79af007ba066f13e2b0ea3d736ede049cfd8b55015a"
dependencies = [
"ahash",
"astral-tl",
"base64 0.23.0",
"base64 0.23.1",
"bitflags 2.13.1",
"html-escape",
"html5ever",
@@ -3148,9 +3148,9 @@ dependencies = [
[[package]]
name = "ignore"
version = "0.4.32"
version = "0.4.33"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0b17771570a2b94107741a7b033f19132c2eee21d59d21b24d2ced26500bd66e"
checksum = "00b69833ed729dc5aa7d19541d96d6cf8e9137194207a04916d658e43168402f"
dependencies = [
"crossbeam-deque",
"globset",
@@ -3537,9 +3537,9 @@ checksum = "e2db585e1d738fc771bf08a151420d3ed193d9d895a36df7f6f8a9456b911ddc"
[[package]]
name = "kqueue"
version = "1.2.0"
version = "1.2.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "273c0752728918e0ac4976f2b275b6fefb9ecd400585dec929419f3844cd87b5"
checksum = "8d763e5b24120b4ddf50de6c92308156765aabfbbccebf401da7cff2d70a41ea"
dependencies = [
"kqueue-sys",
"libc",
@@ -4899,7 +4899,7 @@ dependencies = [
[[package]]
name = "pi-ast"
version = "17.2.8"
version = "17.2.9"
dependencies = [
"anyhow",
"ast-grep-core",
@@ -4968,19 +4968,19 @@ dependencies = [
[[package]]
name = "pi-iso"
version = "17.2.8"
version = "17.2.9"
dependencies = [
"async-trait",
"libc",
"parking_lot",
"similar 3.1.1",
"similar 3.1.2",
"tokio",
"windows-sys 0.61.2",
]
[[package]]
name = "pi-natives"
version = "17.2.8"
version = "17.2.9"
dependencies = [
"anyhow",
"arboard",
@@ -5050,7 +5050,7 @@ dependencies = [
[[package]]
name = "pi-shell"
version = "17.2.8"
version = "17.2.9"
dependencies = [
"anyhow",
"brush-builtins",
@@ -5137,7 +5137,7 @@ dependencies = [
[[package]]
name = "pi-voice"
version = "17.2.8"
version = "17.2.9"
dependencies = [
"audiopus_sys",
"bytes",
@@ -5151,7 +5151,7 @@ dependencies = [
[[package]]
name = "pi-walker"
version = "17.2.8"
version = "17.2.9"
dependencies = [
"dashmap",
"globset",
@@ -5169,7 +5169,7 @@ dependencies = [
"clap",
"parking_lot",
"pi-uutils-ctx",
"similar 3.1.1",
"similar 3.1.2",
"tempfile",
]
@@ -5714,9 +5714,9 @@ dependencies = [
[[package]]
name = "regex-automata"
version = "0.4.16"
version = "0.4.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8fcfdb36bda0c880c5931cdc7a2bcdc8ba4556847b9d912bca70bc94708711ad"
checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2"
dependencies = [
"aho-corasick",
"memchr",
@@ -6238,9 +6238,9 @@ checksum = "bbbb5d9659141646ae647b42fe094daf6c6192d1620870b449d9557f748b2daa"
[[package]]
name = "similar"
version = "3.1.1"
version = "3.1.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e6505efef05804732ed8a3f2d4f279429eb485bd69d5b0cc6b19cc02005cda16"
checksum = "85ee016af5d736b69fc89e19254540fa4b5f5492853fb5503920f084011c78b6"
dependencies = [
"bstr",
]
+2 -1
View File
@@ -3,7 +3,7 @@ members = ["crates/pi-*", "crates/vendor/*"]
resolver = "3"
[workspace.package]
version = "17.2.8"
version = "17.2.9"
edition = "2024"
license = "MIT"
authors = ["Can Boluk"]
@@ -138,6 +138,7 @@ option_if_let_else = "allow" # match/if-let-else often clearer
enum_glob_use = "allow"
items_after_statements = "allow" # Sometimes more readable
wildcard_imports = "allow" # Cleaner for preludes and test modules
redundant_pub_crate = "allow"
# ──────────────────────────────────────────────────────────────────────────────
# Variables & Type Inference
-1
View File
@@ -614,7 +614,6 @@ For architecture and contribution guidelines, see [packages/coding-agent/DEVELOP
| **[@oh-my-pi/hashline](packages/hashline)** | Line-anchored patch language and applier behind the `edit` tool |
| **[@oh-my-pi/pi-mnemopi](packages/mnemopi)** | Local SQLite memory engine for Oh My Pi agents |
| **[@oh-my-pi/snapcompact](packages/snapcompact)** | Bitmap-frame context compression package and SQuAD eval suite |
| **[@oh-my-pi/swarm-extension](packages/swarm-extension)** | Swarm orchestration extension package |
| **[@oh-my-pi/browser-relay](packages/browser-relay)** | Chrome extension that lets the browser tool drive your existing tabs |
| **[@oh-my-pi/pi-metaharness](packages/metaharness)** | Unified benchmark runners, Harbor run storage, REST/SSE API, live dashboard |
| **[@oh-my-pi/typescript-edit-benchmark](packages/typescript-edit-benchmark)** | Edit benchmark suite built on TypeScript source mutations |
+46 -52
View File
@@ -21,7 +21,7 @@
},
"packages/agent": {
"name": "@oh-my-pi/pi-agent-core",
"version": "17.2.8",
"version": "17.2.9",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
@@ -40,7 +40,7 @@
},
"packages/ai": {
"name": "@oh-my-pi/pi-ai",
"version": "17.2.8",
"version": "17.2.9",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/omptype": "catalog:",
@@ -64,7 +64,7 @@
},
"packages/catalog": {
"name": "@oh-my-pi/pi-catalog",
"version": "17.2.8",
"version": "17.2.9",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/omptype": "catalog:",
@@ -78,7 +78,7 @@
},
"packages/coding-agent": {
"name": "@oh-my-pi/pi-coding-agent",
"version": "17.2.8",
"version": "17.2.9",
"bin": {
"omp": "src/cli.ts",
},
@@ -152,7 +152,7 @@
},
"packages/hashline": {
"name": "@oh-my-pi/hashline",
"version": "17.2.8",
"version": "17.2.9",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"lru-cache": "catalog:",
@@ -196,7 +196,7 @@
},
"packages/mnemopi": {
"name": "@oh-my-pi/pi-mnemopi",
"version": "17.2.8",
"version": "17.2.9",
"bin": {
"mnemopi": "src/cli.ts",
},
@@ -223,7 +223,7 @@
},
"packages/natives": {
"name": "@oh-my-pi/pi-natives",
"version": "17.2.8",
"version": "17.2.9",
"devDependencies": {
"@napi-rs/cli": "catalog:",
"@types/bun": "catalog:",
@@ -231,7 +231,7 @@
},
"packages/omptype": {
"name": "@oh-my-pi/omptype",
"version": "17.2.8",
"version": "17.2.9",
"devDependencies": {
"@ark/attest": "0.56.3",
"@ark/schema": "0.56.2",
@@ -245,7 +245,7 @@
},
"packages/snapcompact": {
"name": "@oh-my-pi/snapcompact",
"version": "17.2.8",
"version": "17.2.9",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-natives": "catalog:",
@@ -258,7 +258,7 @@
},
"packages/stats": {
"name": "@oh-my-pi/omp-stats",
"version": "17.2.8",
"version": "17.2.9",
"bin": {
"omp-stats": "./src/index.ts",
},
@@ -283,25 +283,9 @@
"postcss": "catalog:",
},
},
"packages/swarm-extension": {
"name": "@oh-my-pi/swarm-extension",
"version": "17.2.8",
"bin": {
"omp-swarm": "src/cli.ts",
},
"dependencies": {
"@oh-my-pi/pi-utils": "workspace:*",
},
"devDependencies": {
"@types/bun": "^1.3.14",
},
"peerDependencies": {
"@oh-my-pi/pi-coding-agent": "^16",
},
},
"packages/tui": {
"name": "@oh-my-pi/pi-tui",
"version": "17.2.8",
"version": "17.2.9",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
@@ -340,7 +324,7 @@
},
"packages/utils": {
"name": "@oh-my-pi/pi-utils",
"version": "17.2.8",
"version": "17.2.9",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"handlebars": "catalog:",
@@ -353,7 +337,7 @@
},
"packages/wire": {
"name": "@oh-my-pi/pi-wire",
"version": "17.2.8",
"version": "17.2.9",
"devDependencies": {
"@types/bun": "catalog:",
},
@@ -394,19 +378,19 @@
"@huggingface/transformers": "^4.2.0",
"@mozilla/readability": "^0.6.0",
"@napi-rs/cli": "3.7.2",
"@oh-my-pi/hashline": "17.2.8",
"@oh-my-pi/omp-stats": "17.2.8",
"@oh-my-pi/omptype": "17.2.8",
"@oh-my-pi/pi-agent-core": "17.2.8",
"@oh-my-pi/pi-ai": "17.2.8",
"@oh-my-pi/pi-catalog": "17.2.8",
"@oh-my-pi/pi-coding-agent": "17.2.8",
"@oh-my-pi/pi-mnemopi": "17.2.8",
"@oh-my-pi/pi-natives": "17.2.8",
"@oh-my-pi/pi-tui": "17.2.8",
"@oh-my-pi/pi-utils": "17.2.8",
"@oh-my-pi/pi-wire": "17.2.8",
"@oh-my-pi/snapcompact": "17.2.8",
"@oh-my-pi/hashline": "17.2.9",
"@oh-my-pi/omp-stats": "17.2.9",
"@oh-my-pi/omptype": "17.2.9",
"@oh-my-pi/pi-agent-core": "17.2.9",
"@oh-my-pi/pi-ai": "17.2.9",
"@oh-my-pi/pi-catalog": "17.2.9",
"@oh-my-pi/pi-coding-agent": "17.2.9",
"@oh-my-pi/pi-mnemopi": "17.2.9",
"@oh-my-pi/pi-natives": "17.2.9",
"@oh-my-pi/pi-tui": "17.2.9",
"@oh-my-pi/pi-utils": "17.2.9",
"@oh-my-pi/pi-wire": "17.2.9",
"@oh-my-pi/snapcompact": "17.2.9",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/api-logs": "^0.220.0",
"@opentelemetry/context-async-hooks": "^2.9.0",
@@ -777,13 +761,13 @@
"@octokit/auth-token": ["@octokit/auth-token@6.0.0", "", {}, "sha512-P4YJBPdPSpWTQ1NU4XYdvHvXJJDxM6YwpS0FZHRgP7YFkdVxsWcpWGy/NVqlAA7PcPCnMacXlRm1y2PFZRWL/w=="],
"@octokit/core": ["@octokit/core@7.0.6", "", { "dependencies": { "@octokit/auth-token": "^6.0.0", "@octokit/graphql": "^9.0.3", "@octokit/request": "^10.0.6", "@octokit/request-error": "^7.0.2", "@octokit/types": "^16.0.0", "before-after-hook": "^4.0.0", "universal-user-agent": "^7.0.0" } }, "sha512-DhGl4xMVFGVIyMwswXeyzdL4uXD5OGILGX5N8Y+f6W7LhC1Ze2poSNrkF/fedpVDHEEZ+PHFW0vL14I+mm8K3Q=="],
"@octokit/core": ["@octokit/core@7.0.7", "", { "dependencies": { "@octokit/auth-token": "^6.0.0", "@octokit/graphql": "^9.0.4", "@octokit/request": "^10.0.13", "@octokit/request-error": "^7.1.1", "@octokit/types": "^17.0.0", "before-after-hook": "^4.0.0", "universal-user-agent": "^7.0.0" } }, "sha512-DcB0M3KFgr9ECI328lhBMVsyFT2DnmNucSBTqEN3exyNKUzkkpUSCHmTRcunF41Eou2TIQKW4seewri8ON9bSA=="],
"@octokit/endpoint": ["@octokit/endpoint@11.0.3", "", { "dependencies": { "@octokit/types": "^16.0.0", "universal-user-agent": "^7.0.2" } }, "sha512-FWFlNxghg4HrXkD3ifYbS/IdL/mDHjh9QcsNyhQjN8dplUoZbejsdpmuqdA76nxj2xoWPs7p8uX2SNr9rYu0Ag=="],
"@octokit/graphql": ["@octokit/graphql@9.0.3", "", { "dependencies": { "@octokit/request": "^10.0.6", "@octokit/types": "^16.0.0", "universal-user-agent": "^7.0.0" } }, "sha512-grAEuupr/C1rALFnXTv6ZQhFuL1D8G5y8CN04RgrO4FIPMrtm+mcZzFG7dcBm+nq+1ppNixu+Jd78aeJOYxlGA=="],
"@octokit/graphql": ["@octokit/graphql@9.0.4", "", { "dependencies": { "@octokit/request": "^10.0.13", "@octokit/types": "^17.0.0", "universal-user-agent": "^7.0.0" } }, "sha512-5s15CCiY8XXQ+FG+b1YQcl6Z2FA++nwAz/tg2VUrTmnMncP+2nnGUEYANImdnxsA2Fnq+Mbl7hDjUTw7cFAwcg=="],
"@octokit/openapi-types": ["@octokit/openapi-types@27.0.0", "", {}, "sha512-whrdktVs1h6gtR+09+QsNk2+FO+49j6ga1c55YZudfEG+oKJVvJLQi3zkOm5JjiUXAagWK2tI2kTGKJ2Ys7MGA=="],
"@octokit/openapi-types": ["@octokit/openapi-types@28.0.0", "", {}, "sha512-0rFyLuyHvIj6uuZWuDslxkowFYdPXoNIkeAv4b27dzm2Tf4vGWXnPsMcxs7d65kLdMERgP3wc1AEPlqMz8e1cQ=="],
"@octokit/plugin-paginate-rest": ["@octokit/plugin-paginate-rest@14.0.0", "", { "dependencies": { "@octokit/types": "^16.0.0" }, "peerDependencies": { "@octokit/core": ">=6" } }, "sha512-fNVRE7ufJiAA3XUrha2omTA39M6IXIc6GIZLvlbsm8QOQCYvpq/LkMNGyFlB1d8hTDzsAXa3OKtybdMAYsV/fw=="],
@@ -791,13 +775,13 @@
"@octokit/plugin-rest-endpoint-methods": ["@octokit/plugin-rest-endpoint-methods@17.0.0", "", { "dependencies": { "@octokit/types": "^16.0.0" }, "peerDependencies": { "@octokit/core": ">=6" } }, "sha512-B5yCyIlOJFPqUUeiD0cnBJwWJO8lkJs5d8+ze9QDP6SvfiXSz1BF+91+0MeI1d2yxgOhU/O+CvtiZ9jSkHhFAw=="],
"@octokit/request": ["@octokit/request@10.0.11", "", { "dependencies": { "@octokit/endpoint": "^11.0.3", "@octokit/request-error": "^7.0.2", "@octokit/types": "^16.0.0", "content-type": "^2.0.0", "json-with-bigint": "^3.5.3", "universal-user-agent": "^7.0.2" } }, "sha512-+s7HUxjfFqOMS9VlIwDffq0MikjSAK0gSpG73W+meAvVAvX4MBrHYTK5Bj3Uot55qFT4gzUtfzE4mGWY4Br8/Q=="],
"@octokit/request": ["@octokit/request@10.0.13", "", { "dependencies": { "@octokit/endpoint": "^11.0.3", "@octokit/request-error": "^7.1.1", "@octokit/types": "^17.0.0", "content-type": "^2.0.0", "json-with-bigint": "^3.5.3", "universal-user-agent": "^7.0.2" } }, "sha512-v2269YxL9Yf+x3d+gRI63FP0vFQEiWgLyBzxe/Y+0yFDg2B/Tzf5dhh9VNfccVAQnfcfwQWyk/y6Bn7rUXXs7A=="],
"@octokit/request-error": ["@octokit/request-error@7.1.0", "", { "dependencies": { "@octokit/types": "^16.0.0" } }, "sha512-KMQIfq5sOPpkQYajXHwnhjCC0slzCNScLHs9JafXc4RAJI+9f+jNDlBNaIMTvazOPLgb4BnlhGJOTbnN0wIjPw=="],
"@octokit/request-error": ["@octokit/request-error@7.1.1", "", { "dependencies": { "@octokit/types": "^17.0.0" } }, "sha512-+eaY7G2VVpSf2pc5Gn1+mph837V/d/TYTJAgWL9Tb0ogGYcpN3IlAVFgjL+Vv93F/sevrxkvsYCedtpLdcFLzA=="],
"@octokit/rest": ["@octokit/rest@22.0.1", "", { "dependencies": { "@octokit/core": "^7.0.6", "@octokit/plugin-paginate-rest": "^14.0.0", "@octokit/plugin-request-log": "^6.0.0", "@octokit/plugin-rest-endpoint-methods": "^17.0.0" } }, "sha512-Jzbhzl3CEexhnivb1iQ0KJ7s5vvjMWcmRtq5aUsKmKDrRW6z3r84ngmiFKFvpZjpiU/9/S6ITPFRpn5s/3uQJw=="],
"@octokit/types": ["@octokit/types@16.0.0", "", { "dependencies": { "@octokit/openapi-types": "^27.0.0" } }, "sha512-sKq+9r1Mm4efXW1FCk7hFSeJo4QKreL/tTbR0rz/qx/r1Oa2VV83LTA/H/MuCOX7uCIJmQVRKBcbmWoySjAnSg=="],
"@octokit/types": ["@octokit/types@17.0.0", "", { "dependencies": { "@octokit/openapi-types": "^28.0.0" } }, "sha512-ByP1v7YL5SMveFPP7+sj0/ZuWCOOg/Chs4NafOMpq6WNIM/hdGY0S7C0TCGDBWu1aGmOxmUIhMx3cO+IdwYZ1Q=="],
"@oh-my-pi/browser-relay": ["@oh-my-pi/browser-relay@workspace:packages/browser-relay"],
@@ -831,8 +815,6 @@
"@oh-my-pi/snapcompact": ["@oh-my-pi/snapcompact@workspace:packages/snapcompact"],
"@oh-my-pi/swarm-extension": ["@oh-my-pi/swarm-extension@workspace:packages/swarm-extension"],
"@oh-my-pi/typescript-edit-benchmark": ["@oh-my-pi/typescript-edit-benchmark@workspace:packages/typescript-edit-benchmark"],
"@opentelemetry/api": ["@opentelemetry/api@1.9.1", "", {}, "sha512-gLyJlPHPZYdAk1JENA9LeHejZe1Ti77/pTeFm/nMXmQH/HFZlcS/O2XJB+L8fkbrNSqhdtlvjBVjxwUYanNH5Q=="],
@@ -1081,7 +1063,7 @@
"base64-js": ["base64-js@1.5.1", "", {}, "sha512-AKpaYlHn8t4SVbOHCy+b5+KKgvR4vrsD8vbvrbiQJps7fKDTkjkDry6ji0rUJjC0kzbNePLwzxq8iypo41qeWA=="],
"baseline-browser-mapping": ["baseline-browser-mapping@2.11.9", "", { "bin": { "baseline-browser-mapping": "dist/cli.cjs" } }, "sha512-cp447VUsGS07+n1Dqf7YSQ8maeJrjEhaDxTm1ZefbqDtypHBC5GzGMQbklR6IPR13Y8OAJRHZWEMtZipJLCttg=="],
"baseline-browser-mapping": ["baseline-browser-mapping@2.11.10", "", { "bin": { "baseline-browser-mapping": "dist/cli.cjs" } }, "sha512-35JEvJ5/KKlbCHjMCsONI2w6HE88STjVdHk+C7d8LtcFxUjZR1KeLP9izofn2qs0KUxX5r4z73bwH/rd+JHacw=="],
"before-after-hook": ["before-after-hook@4.0.0", "", {}, "sha512-q6tR3RPqIB1pMiTRMFcZwuG5T8vwp+vUvEG0vuI6B+Rikh5BfPp2fQ82c925FOs+b0lcFQ8CFrL+KbilfZFhOQ=="],
@@ -1563,7 +1545,7 @@
"through2": ["through2@4.0.2", "", { "dependencies": { "readable-stream": "3" } }, "sha512-iOqSav00cVxEEICeD7TjLB1sueEL+81Wpzp2bY17uZjZN0pWZPuo4suZ/61VujxmqSGFfgOcNuTZ85QJwNZQpw=="],
"tinyexec": ["tinyexec@1.2.4", "", {}, "sha512-SHf/r48b7vOrjve9PxJo3MN5v5yuyjHvdUcrQffT3WXMUfnGmHDVbC4k3sHJaJTgZCwpUplIaAo5ANtMyp3YHg=="],
"tinyexec": ["tinyexec@1.3.0", "", {}, "sha512-QKAl9m8gWWGHV8jZcPeym6j+XULi6tOf1mT83WYJ4Lk2ytW/uwAWkrP0uFsdoYMdueVJ0qs26wZ+23xeB4ibNQ=="],
"tinyglobby": ["tinyglobby@0.2.17", "", { "dependencies": { "fdir": "^6.5.0", "picomatch": "^4.0.4" } }, "sha512-wXR/dYpcqKmfWpEdZjiKJOwCNFndD0DMnrW/cYjVGttEkBfVgcLFHoNrlj47mjOVic9yyNu65alsgF4NQyTa2g=="],
@@ -1667,6 +1649,12 @@
"@napi-rs/wasm-tools-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.9.2", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-3U4+MIWHImeyu1wnmVygh5WlgfYDtyf0k8AbLhMFxOipihf6nrWC4syIm/SwEeec0mNSafiiNnMJwbza/Is6Lw=="],
"@octokit/endpoint/@octokit/types": ["@octokit/types@16.0.0", "", { "dependencies": { "@octokit/openapi-types": "^27.0.0" } }, "sha512-sKq+9r1Mm4efXW1FCk7hFSeJo4QKreL/tTbR0rz/qx/r1Oa2VV83LTA/H/MuCOX7uCIJmQVRKBcbmWoySjAnSg=="],
"@octokit/plugin-paginate-rest/@octokit/types": ["@octokit/types@16.0.0", "", { "dependencies": { "@octokit/openapi-types": "^27.0.0" } }, "sha512-sKq+9r1Mm4efXW1FCk7hFSeJo4QKreL/tTbR0rz/qx/r1Oa2VV83LTA/H/MuCOX7uCIJmQVRKBcbmWoySjAnSg=="],
"@octokit/plugin-rest-endpoint-methods/@octokit/types": ["@octokit/types@16.0.0", "", { "dependencies": { "@octokit/openapi-types": "^27.0.0" } }, "sha512-sKq+9r1Mm4efXW1FCk7hFSeJo4QKreL/tTbR0rz/qx/r1Oa2VV83LTA/H/MuCOX7uCIJmQVRKBcbmWoySjAnSg=="],
"@opentelemetry/exporter-metrics-otlp-http/@opentelemetry/core": ["@opentelemetry/core@2.9.0", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-m2nckMT80NnmjTYSPjJQObBJ+8dgkoajEOUbznL8AHZ3T3yHRk2P7gI1PhEBc1+lOnrYE9UWrWHqJDsmqjmNbw=="],
"@opentelemetry/exporter-metrics-otlp-http/@opentelemetry/resources": ["@opentelemetry/resources@2.9.0", "", { "dependencies": { "@opentelemetry/core": "2.9.0", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-jyA5MBLQ+Dkl3+JsZkUoUvL7yHvU64kLsvpXKarWm6347Sl1t1bXFTFykUePNpT5WH5pm9a2Qtt03iIYQhZ1Fg=="],
@@ -1779,6 +1767,12 @@
"@napi-rs/tar-wasm32-wasi/@emnapi/core/@emnapi/wasi-threads": ["@emnapi/wasi-threads@1.2.2", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-c95qOXkHdydNKhscBTebqEC1CVAZpyqOfVfBzQ1qgzyl3gfeldUjIggDbIZgDKsHLgnsM+igH7TJ/eAasaVuMA=="],
"@octokit/endpoint/@octokit/types/@octokit/openapi-types": ["@octokit/openapi-types@27.0.0", "", {}, "sha512-whrdktVs1h6gtR+09+QsNk2+FO+49j6ga1c55YZudfEG+oKJVvJLQi3zkOm5JjiUXAagWK2tI2kTGKJ2Ys7MGA=="],
"@octokit/plugin-paginate-rest/@octokit/types/@octokit/openapi-types": ["@octokit/openapi-types@27.0.0", "", {}, "sha512-whrdktVs1h6gtR+09+QsNk2+FO+49j6ga1c55YZudfEG+oKJVvJLQi3zkOm5JjiUXAagWK2tI2kTGKJ2Ys7MGA=="],
"@octokit/plugin-rest-endpoint-methods/@octokit/types/@octokit/openapi-types": ["@octokit/openapi-types@27.0.0", "", {}, "sha512-whrdktVs1h6gtR+09+QsNk2+FO+49j6ga1c55YZudfEG+oKJVvJLQi3zkOm5JjiUXAagWK2tI2kTGKJ2Ys7MGA=="],
"@rolldown/binding-wasm32-wasi/@emnapi/core/@emnapi/wasi-threads": ["@emnapi/wasi-threads@2.0.1", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-9DsSk+o5NBX0CCJT8s0EROGSGxjR/tKu6aBTaVyq+SjAEQH4XcdcRxPBRzsBLizTTJ49MJjF+jgu3qnO9GLQcQ=="],
"@typescript/analyze-trace/yargs/cliui": ["cliui@7.0.4", "", { "dependencies": { "string-width": "^4.2.0", "strip-ansi": "^6.0.0", "wrap-ansi": "^7.0.0" } }, "sha512-OcRE68cOsVMXp1Yvonl/fzkQOyjLSu/8bhPDfQt0e0/Eb283TKP20Fs2MqoPsr9SwA595rRCA+QMzYc9nBP+JQ=="],
+1 -1
View File
@@ -467,7 +467,7 @@ pub fn normalize_role_macos(native: &str) -> String {
.to_ascii_lowercase()
}
#[cfg(any(target_os = "windows", test))]
pub(crate) fn normalize_role_uia(native: &str) -> String {
pub fn normalize_role_uia(native: &str) -> String {
match native {
"Edit" => "textfield",
"Document" => "textarea",
+6 -6
View File
@@ -237,14 +237,14 @@ pub fn parse_modifiers(mods: &[String]) -> CoreResult<Modifiers> {
#[cfg(test)]
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(crate) enum KeyDirection {
pub enum KeyDirection {
Press,
Release,
Click,
}
#[cfg(test)]
pub(crate) fn execute_chord_with<E>(
pub fn execute_chord_with<E>(
keys: &[KeyName],
mut emit: impl FnMut(KeyName, KeyDirection) -> Result<(), E>,
) -> Result<(), E> {
@@ -263,10 +263,10 @@ pub(crate) fn execute_chord_with<E>(
}
let mut first_error = None;
for &key in pressed.iter().rev() {
if let Err(error) = emit(key, KeyDirection::Release) {
if first_error.is_none() {
first_error = Some(error);
}
if let Err(error) = emit(key, KeyDirection::Release)
&& first_error.is_none()
{
first_error = Some(error);
}
}
first_error.map_or(Ok(()), Err)
@@ -4,7 +4,7 @@
//! exercised by the host test suite on every platform.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub(crate) enum EventKind {
pub enum EventKind {
MouseClick,
MouseMove,
MouseScroll,
@@ -26,34 +26,34 @@ impl EventKind {
}
}
pub(crate) fn is_chromium_class(class: &str) -> bool {
pub fn is_chromium_class(class: &str) -> bool {
class
.strip_prefix("Chrome_WidgetWin_")
.is_some_and(|suffix| !suffix.is_empty())
}
pub(crate) fn is_winui3_class(class: &str) -> bool {
pub fn is_winui3_class(class: &str) -> bool {
class == "WinUIDesktopWin32WindowClass"
}
pub(crate) fn is_wpf_class(class: &str) -> bool {
pub fn is_wpf_class(class: &str) -> bool {
class
.strip_prefix("HwndWrapper[")
.is_some_and(|body| !body.is_empty() && body.ends_with(']'))
}
pub(crate) fn is_tk_class(class: &str) -> bool {
pub fn is_tk_class(class: &str) -> bool {
class == "TkTopLevel"
|| class
.strip_prefix("TkTopLevel.")
.is_some_and(|suffix| !suffix.is_empty())
}
pub(crate) fn is_gtk_class(class: &str) -> bool {
pub fn is_gtk_class(class: &str) -> bool {
class == "gdkWindowToplevel" || class == "gdkSurfaceToplevel"
}
pub(crate) fn is_vcl_class(class: &str) -> bool {
pub fn is_vcl_class(class: &str) -> bool {
class
.strip_prefix("SAL")
.is_some_and(|suffix| !suffix.is_empty())
@@ -61,7 +61,7 @@ pub(crate) fn is_vcl_class(class: &str) -> bool {
/// Returns the empirical reason that a posted event would be accepted by
/// Win32 but silently ignored by the target toolkit.
pub(crate) fn would_be_silently_dropped(class: &str, kind: EventKind) -> Option<&'static str> {
pub fn would_be_silently_dropped(class: &str, kind: EventKind) -> Option<&'static str> {
use EventKind::{KeyCombo, Keystroke, MouseClick, MouseMove, MouseScroll, TextInput};
if is_chromium_class(class) {
+1 -1
View File
@@ -2,7 +2,7 @@
mod ax;
#[cfg(target_os = "windows")]
mod capture;
pub(crate) mod delivery;
pub mod delivery;
#[cfg(target_os = "windows")]
mod input;
+259 -20
View File
@@ -3,7 +3,7 @@
//! Searches for files and directories whose paths match a query string via
//! subsequence scoring. Uses `pi-walker` for directory traversal and caching.
use std::path::Path;
use std::{cmp::Ordering, collections::BinaryHeap, path::Path};
use napi::bindgen_prelude::*;
use napi_derive::napi;
@@ -159,6 +159,97 @@ fn path_depth(path: &str) -> usize {
path.trim_end_matches('/').matches('/').count()
}
/// A scored match carrying its precomputed depth, ordered worst-first.
///
/// The ordering is the exact inverse of the final result comparator (score
/// descending, then `path_depth` ascending, then `path` ascending), so the
/// greatest element of a `BinaryHeap<RankedMatch>` is the candidate that must
/// be evicted first, and `into_sorted_vec` yields the final best-first order.
struct RankedMatch {
depth: usize,
entry: FuzzyFindMatch,
}
impl RankedMatch {
fn new(entry: FuzzyFindMatch) -> Self {
let depth = path_depth(&entry.path);
Self { depth, entry }
}
}
impl Ord for RankedMatch {
fn cmp(&self, other: &Self) -> Ordering {
other
.entry
.score
.cmp(&self.entry.score)
.then_with(|| self.depth.cmp(&other.depth))
.then_with(|| self.entry.path.cmp(&other.entry.path))
}
}
impl PartialOrd for RankedMatch {
fn partial_cmp(&self, other: &Self) -> Option<Ordering> {
Some(self.cmp(other))
}
}
impl PartialEq for RankedMatch {
fn eq(&self, other: &Self) -> bool {
self.cmp(other) == Ordering::Equal
}
}
impl Eq for RankedMatch {}
/// Bounded collector retaining at most `capacity` best matches while counting
/// every hit, so `totalMatches` stays exact even when it exceeds `maxResults`.
struct TopMatches {
capacity: usize,
total: u64,
heap: BinaryHeap<RankedMatch>,
}
impl TopMatches {
fn new(capacity: usize) -> Self {
Self { capacity, total: 0, heap: BinaryHeap::with_capacity(capacity.min(256)) }
}
fn push(&mut self, entry: FuzzyFindMatch) {
self.total = self.total.saturating_add(1);
if self.capacity == 0 {
return;
}
let candidate = RankedMatch::new(entry);
if self.heap.len() < self.capacity {
self.heap.push(candidate);
return;
}
// The root is the worst retained candidate; replace it only when the new
// candidate outranks it under the final comparator.
if self.heap.peek().is_some_and(|worst| candidate < *worst) {
self.heap.pop();
self.heap.push(candidate);
}
}
/// Exact number of scoring hits, clamped to the `u32` wire type.
const fn total_matches(&self) -> u32 {
crate::utils::clamp_u32(self.total)
}
/// Retained matches ordered by score descending, then shallower paths, then
/// path ascending.
fn into_sorted_matches(self) -> Vec<FuzzyFindMatch> {
self
.heap
.into_sorted_vec()
.into_iter()
.map(|ranked| ranked.entry)
.collect()
}
}
struct FuzzyFindConfig {
query: String,
path: String,
@@ -168,14 +259,18 @@ struct FuzzyFindConfig {
cache: Option<bool>,
}
fn score_entries(
entries: &[iofs::GlobMatch],
fn score_entries<I>(
entries: I,
query_lower: &str,
normalized_query: &str,
query_chars: &[char],
max_results: usize,
ct: &task::CancelToken,
) -> Result<Vec<FuzzyFindMatch>> {
let mut scored = Vec::with_capacity(entries.len().min(256));
) -> Result<TopMatches>
where
I: IntoIterator<Item = iofs::GlobMatch>,
{
let mut scored = TopMatches::new(max_results);
for entry in entries {
ct.heartbeat()?;
if entry.file_type == iofs::FileType::Symlink {
@@ -189,7 +284,7 @@ fn score_entries(
continue;
}
let mut path = entry.path.clone();
let mut path = entry.path;
if is_directory {
path.push('/');
}
@@ -229,21 +324,17 @@ fn fuzzy_find_sync(config: FuzzyFindConfig, ct: task::CancelToken) -> Result<Fuz
.empty_recheck(pi_walker::EmptyRecheck::Configured)
.collect_with_heartbeat(|| ct.heartbeat())
.map_err(iofs::map_walker_error)?;
let entries: Vec<iofs::GlobMatch> = outcome
.entries
.into_iter()
.map(iofs::GlobMatch::from)
.collect();
let mut scored = score_entries(&entries, &query_lower, &normalized_query, &query_chars, &ct)?;
let scored = score_entries(
outcome.entries.into_iter().map(iofs::GlobMatch::from),
&query_lower,
&normalized_query,
&query_chars,
max_results,
&ct,
)?;
scored.sort_by(|a, b| {
b.score
.cmp(&a.score)
.then_with(|| path_depth(&a.path).cmp(&path_depth(&b.path)))
.then_with(|| a.path.cmp(&b.path))
});
let total_matches = crate::utils::clamp_u32(scored.len() as u64);
let matches = scored.into_iter().take(max_results).collect();
let total_matches = scored.total_matches();
let matches = scored.into_sorted_matches();
Ok(FuzzyFindResult { matches, total_matches })
}
@@ -381,4 +472,152 @@ mod tests {
"expected cwd-root scripts/ to rank first, got {paths:?}"
);
}
#[cfg(unix)]
#[test]
fn fuzzy_find_reports_exact_total_beyond_max_results() {
let root = TempDirGuard::new();
for index in 0..12 {
fs::write(root.path().join(format!("needle-{index}.txt")), "needle\n")
.expect("write fixture file");
}
let result = fuzzy_find_sync(
FuzzyFindConfig {
query: "needle".to_string(),
path: root.path().to_string_lossy().into_owned(),
hidden: Some(true),
gitignore: Some(false),
max_results: Some(3),
cache: Some(false),
},
task::CancelToken::default(),
)
.expect("fuzzy find succeeds");
assert_eq!(result.matches.len(), 3, "retained matches must honor maxResults");
assert_eq!(result.total_matches, 12, "total must count every hit, not the retained ones");
let paths: Vec<&str> = result
.matches
.iter()
.map(|entry| entry.path.as_str())
.collect();
assert_eq!(
paths,
vec!["needle-0.txt", "needle-1.txt", "needle-10.txt"],
"bounded retention must keep the same order as the full sort"
);
}
#[test]
fn bounded_retention_matches_reference_ordering_and_total() {
use super::{FuzzyFindMatch, TopMatches, path_depth};
// Score ties across depths and directories are the cases where a bounded
// heap can diverge from the full sort, so cover them explicitly.
let candidates = [
("packages/ai/scripts/", true, 130u32),
("scripts/", true, 130),
(".omp/skills/opt/scripts/", true, 130),
("src/scripts.ts", false, 120),
("src/deep/nested/scripts.ts", false, 120),
("a/scripts.ts", false, 120),
("notes/script-notes.md", false, 80),
("z.txt", false, 51),
];
let mut reference: Vec<(u32, usize, String)> = candidates
.iter()
.map(|(path, _, score)| (*score, path_depth(path), (*path).to_string()))
.collect();
reference.sort_by(|a, b| {
b.0.cmp(&a.0)
.then_with(|| a.1.cmp(&b.1))
.then_with(|| a.2.cmp(&b.2))
});
for max_results in 1..=candidates.len() + 2 {
let mut bounded = TopMatches::new(max_results);
for (path, is_directory, score) in candidates {
bounded.push(FuzzyFindMatch { path: path.to_string(), is_directory, score });
}
let total = bounded.total_matches();
let bounded_paths: Vec<String> = bounded
.into_sorted_matches()
.into_iter()
.map(|entry| entry.path)
.collect();
let expected_paths: Vec<String> = reference
.iter()
.take(max_results)
.map(|(_, _, path)| path.clone())
.collect();
assert_eq!(total, candidates.len() as u32, "total must count every pushed hit");
assert_eq!(
bounded_paths, expected_paths,
"bounded order must match the full sort for max_results={max_results}"
);
}
}
#[test]
fn bounded_retention_matches_full_sort_on_large_corpus() {
use super::{FuzzyFindMatch, TopMatches, path_depth};
const CANDIDATE_COUNT: usize = 100_000;
const MAX_RESULTS: usize = 128;
let mut reference = Vec::with_capacity(CANDIDATE_COUNT);
let mut bounded = TopMatches::new(MAX_RESULTS);
for index in 0..CANDIDATE_COUNT {
let depth = index % 7;
let path = format!("{}{index:06}-item.txt", "nested/".repeat(depth));
let score = 50 + (index % 83) as u32;
reference.push((score, path_depth(&path), path.clone()));
bounded.push(FuzzyFindMatch { path, is_directory: false, score });
assert!(
bounded.heap.len() <= MAX_RESULTS,
"retention exceeded maxResults after candidate {index}"
);
}
assert_eq!(reference.len(), CANDIDATE_COUNT);
assert_eq!(bounded.heap.len(), MAX_RESULTS);
assert_eq!(bounded.total_matches(), CANDIDATE_COUNT as u32);
reference.sort_by(|a, b| {
b.0.cmp(&a.0)
.then_with(|| a.1.cmp(&b.1))
.then_with(|| a.2.cmp(&b.2))
});
let expected: Vec<String> = reference
.into_iter()
.take(MAX_RESULTS)
.map(|(_, _, path)| path)
.collect();
let actual: Vec<String> = bounded
.into_sorted_matches()
.into_iter()
.map(|entry| entry.path)
.collect();
assert_eq!(actual, expected, "bounded top-K must match the complete baseline sort");
}
#[test]
fn bounded_retention_counts_hits_with_zero_capacity() {
use super::{FuzzyFindMatch, TopMatches};
let mut bounded = TopMatches::new(0);
for index in 0..5 {
bounded.push(FuzzyFindMatch {
path: format!("file-{index}.txt"),
is_directory: false,
score: 10,
});
}
assert_eq!(bounded.total_matches(), 5);
assert!(bounded.into_sorted_matches().is_empty());
}
}
+1 -1
View File
@@ -255,7 +255,7 @@ fn create_windows_napi_tokio_runtime() -> Option<tokio::runtime::Runtime> {
/// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in
/// `packages/natives/native/index.js` (which derives the name from
/// `package.json#version`).
#[napi(js_name = "__piNativesV17_2_8")]
#[napi(js_name = "__piNativesV17_2_9")]
pub const fn pi_natives_version_sentinel() {}
/// Native module entry point: install crash diagnostics before any tool can
+3 -3
View File
@@ -1937,7 +1937,7 @@ mod tests {
// Treat the child's pid as protected (standing in for the harness/an
// ancestor). The sweep must refuse to signal it.
let protected: HashSet<i32> = [child_pid].into_iter().collect();
let protected: HashSet<i32> = HashSet::from([child_pid]);
let signaled = root.signal_tree_excluding(KILL_SIGNAL, &protected);
assert_eq!(signaled, 0, "a protected root must never be signalled");
@@ -1965,8 +1965,8 @@ mod tests {
#[test]
fn protected_subtree_is_pruned_not_just_the_pid() {
// root(1) -> host(2, protected) -> worker(3); root(1) -> real_child(4).
let parents: HashMap<i32, i32> = [(2, 1), (3, 2), (4, 1)].into_iter().collect();
let protected: HashSet<i32> = [2].into_iter().collect();
let parents: HashMap<i32, i32> = HashMap::from([(2, 1), (3, 2), (4, 1)]);
let protected: HashSet<i32> = HashSet::from([2]);
assert!(
pid_in_protected_subtree(2, &protected, &parents),
+6 -6
View File
@@ -8583,12 +8583,12 @@ replace = [{ pattern = "^.+$", replacement = "PWD" }]
/// A segment that carries a file redirect is still segmented, and the brush
/// `Display` reconstruction the runner executes must round-trip through
/// brush's own parser **without losing the redirect**. `echo hidden
/// >/dev/null` suppresses its own stdout: if the reconstruction dropped the
/// redirect, `hidden` would leak into the captured output. Proves the
/// reconstruction path is semantically sound for the redirect-bearing
/// shapes the per-stage whitelist accepts (not just syntactically
/// parseable).
/// brush's own parser **without losing the redirect**.
/// `echo hidden >/dev/null` suppresses its own stdout: if the reconstruction
/// dropped the redirect, `hidden` would leak into the captured output.
/// Proves the reconstruction path is semantically sound for the
/// redirect-bearing shapes the per-stage whitelist accepts (not just
/// syntactically parseable).
#[cfg(unix)]
#[tokio::test(flavor = "multi_thread")]
async fn segmented_chain_with_redirect_executes_correctly() {
+5 -83
View File
@@ -242,87 +242,9 @@ features = [
version = "0.11.0"
[lints.clippy]
bool_to_int_with_if = "allow"
cognitive_complexity = "allow"
collapsible_else_if = "allow"
collapsible_if = "allow"
expect_used = "deny"
format_push_string = "deny"
if_not_else = "allow"
if_same_then_else = "allow"
match_same_arms = "allow"
missing_errors_doc = "allow"
multiple_crate_versions = "allow"
multiple_unsafe_ops_per_block = "deny"
must_use_candidate = "allow"
option_if_let_else = "allow"
panic = "deny"
panic_in_result_fn = "deny"
redundant_closure_for_method_calls = "allow"
redundant_else = "allow"
redundant_pub_crate = "allow"
result_large_err = "allow"
similar_names = "allow"
string_lit_chars_any = "deny"
string_slice = "deny"
struct_excessive_bools = "allow"
tests_outside_test_module = "deny"
todo = "deny"
undocumented_unsafe_blocks = "deny"
unwrap_in_result = "deny"
unwrap_used = "deny"
[lints.clippy.all]
level = "deny"
priority = -1
[lints.clippy.cargo]
level = "deny"
priority = -1
[lints.clippy.nursery]
level = "deny"
priority = -1
[lints.clippy.pedantic]
level = "deny"
priority = -1
[lints.clippy.perf]
level = "deny"
priority = -1
all = { level = "allow", priority = -1 }
nursery = { level = "allow", priority = -1 }
pedantic = { level = "allow", priority = -1 }
cargo = { level = "allow", priority = -1 }
[lints.rust]
unnameable_types = "deny"
unsafe_op_in_unsafe_fn = "deny"
unused_attributes = "deny"
unused_lifetimes = "deny"
unused_macro_rules = "deny"
[lints.rust.future_incompatible]
level = "deny"
priority = 0
[lints.rust.missing_docs]
level = "deny"
priority = 0
[lints.rust.nonstandard_style]
level = "deny"
priority = 0
[lints.rust.rust_2018_idioms]
level = "deny"
priority = -1
[lints.rust.unknown_lints]
level = "allow"
priority = -100
[lints.rust.warnings]
level = "deny"
priority = 0
[lints.rustdoc.all]
level = "deny"
priority = -1
unfulfilled_lint_expectations = { level = "allow", priority = -1 }
+4 -84
View File
@@ -212,87 +212,7 @@ version = "1.23.1"
features = ["js"]
[lints.clippy]
bool_to_int_with_if = "allow"
cognitive_complexity = "allow"
collapsible_else_if = "allow"
collapsible_if = "allow"
expect_used = "deny"
format_push_string = "deny"
if_not_else = "allow"
if_same_then_else = "allow"
match_same_arms = "allow"
missing_errors_doc = "allow"
multiple_crate_versions = "allow"
multiple_unsafe_ops_per_block = "deny"
must_use_candidate = "allow"
option_if_let_else = "allow"
panic = "deny"
panic_in_result_fn = "deny"
redundant_closure_for_method_calls = "allow"
redundant_else = "allow"
redundant_pub_crate = "allow"
result_large_err = "allow"
similar_names = "allow"
string_lit_chars_any = "deny"
string_slice = "deny"
struct_excessive_bools = "allow"
tests_outside_test_module = "deny"
todo = "deny"
undocumented_unsafe_blocks = "deny"
unwrap_in_result = "deny"
unwrap_used = "deny"
[lints.clippy.all]
level = "deny"
priority = -1
[lints.clippy.cargo]
level = "deny"
priority = -1
[lints.clippy.nursery]
level = "deny"
priority = -1
[lints.clippy.pedantic]
level = "deny"
priority = -1
[lints.clippy.perf]
level = "deny"
priority = -1
[lints.rust]
unnameable_types = "deny"
unsafe_op_in_unsafe_fn = "deny"
unused_attributes = "deny"
unused_lifetimes = "deny"
unused_macro_rules = "deny"
[lints.rust.future_incompatible]
level = "deny"
priority = 0
[lints.rust.missing_docs]
level = "deny"
priority = 0
[lints.rust.nonstandard_style]
level = "deny"
priority = 0
[lints.rust.rust_2018_idioms]
level = "deny"
priority = -1
[lints.rust.unknown_lints]
level = "allow"
priority = -100
[lints.rust.warnings]
level = "deny"
priority = 0
[lints.rustdoc.all]
level = "deny"
priority = -1
all = { level = "allow", priority = -1 }
nursery = { level = "allow", priority = -1 }
pedantic = { level = "allow", priority = -1 }
cargo = { level = "allow", priority = -1 }
+2 -3
View File
@@ -327,9 +327,8 @@ mod tests {
fn ext_sort_spills_to_files_and_sorts() {
let input: String = (0..200u32).rev().map(|i| format!("{i:04}\n")).collect();
let mut settings = GlobalSettings::default();
settings.buffer_size = 64;
settings.buffer_size_is_explicit = true;
let settings =
GlobalSettings { buffer_size: 64, buffer_size_is_explicit: true, ..Default::default() };
let out_dir = tempfile::tempdir().expect("temp dir");
let out_path = out_dir.path().join("sorted.txt");
+1 -1
View File
@@ -37,7 +37,7 @@ The bash tool has the `exec` approval tier. `bash.patterns` rules can explicitly
## 2) Optional interception (blocked-command path)
If `bashInterceptor.enabled` is true, `BashTool` loads rules from settings (`getBashInterceptorRules()`) and runs `checkBashInterception()` against the command — checking both the original and the cwd-normalized form (after a leading `cd … &&` is extracted) when they differ. Rule syntax is unchanged: each rule checks the complete input first, then raw flat command fragments separated by unquoted/unescaped `&&`, `||`, `;`, `|`, `&`, or newlines, then those fragments with leading `NAME=value` assignments removed.
If `bashInterceptor.enabled` is true, `BashTool` loads rules from settings (`getBashInterceptorRules()`) and runs `checkBashInterception()` against the command — checking both the original and the cwd-normalized form (after a leading `cd … &&` is extracted) when they differ. Rule syntax is unchanged: each rule checks the complete input first, then raw flat command fragments separated by unquoted/unescaped `&&`, `||`, `;`, `|`, `|&`, `&`, or newlines, then those fragments with leading `NAME=value` assignments removed. Fragments that receive piped stdin from `|` or `|&` are excluded from the fragment candidates, including across blank/comment continuation lines, because a stdin-consuming stage cannot be replaced by a path-based dedicated tool.
Interception behavior:
+1 -1
View File
@@ -12,7 +12,7 @@ This document covers the current extension runtime in:
For discovery paths and filesystem loading rules, see [`extension-loading.md`](./extension-loading.md).
For packaged user-facing extension CLIs/features such as `packages/swarm-extension`, see [`user-facing-packages.md`](./user-facing-packages.md).
For packaged user-facing extension CLIs/features, see [`user-facing-packages.md`](./user-facing-packages.md).
## What an extension is
+1 -1
View File
@@ -39,7 +39,7 @@ OMP also translates these current tool-native sources:
- VS Code: project-only `.vscode/mcp.json` using `mcp.servers`
- installed Claude marketplace plugins and OMP extension packages that declare MCP servers
For translated providers with both scopes, a same-named user entry is encountered before its project entry. OMP-native config is the exception: its project entry precedes its active-profile user entry. Cross-provider priority is listed in [Discovery and precedence](#discovery-and-precedence).
For Claude Code, Codex, Gemini CLI, Cursor, and Windsurf, the project entry is encountered before its same-named user entry — matching OMP-native config, whose project entry precedes its active-profile user entry — so a project `enabled: false` suppresses a same-named user server. OpenCode currently encounters the user entry first. Cross-provider priority is listed in [Discovery and precedence](#discovery-and-precedence).
### Profiles
+1 -1
View File
@@ -146,7 +146,7 @@ The backend settings `eval.py` / `eval.js` default to `true`; `eval.rb` / `eval.
The tool's session-scoped schema lists only enabled runtimes. If Python preflight fails while another runtime is enabled, `eval` remains available for that runtime and a `py` call reports a Python-backend availability error with enabled alternatives.
Python prelude helpers include `agent(prompt, *, agent="task", model=None, label=None, schema=None, schema_mode=None, isolated=None, apply=None, merge=None, handle=False)`. It synchronously calls the host bridge and returns final text, or parsed data when `schema` is supplied. `schema_mode` selects permissive or strict structured-output handling; the isolation/apply/merge flags control task worktree behavior. With `handle=True`, it returns a DAG node dict (`{"text", "output", "handle", "id", "agent"}`) whose handle is the recoverable `agent://<id>` URI; parsed output is also stored under `"data"` when available.
Python prelude helpers include `agent(prompt, *, agent="task", label=None, schema=None, schema_mode=None, isolated=None, apply=None, merge=None, handle=False)`. It synchronously calls the host bridge and returns final text, or parsed data when `schema` is supplied. `schema_mode` selects permissive or strict structured-output handling; the isolation/apply/merge flags control task worktree behavior. With `handle=True`, it returns a DAG node dict (`{"text", "output", "handle", "id", "agent"}`) whose handle is the recoverable `agent://<id>` URI; parsed output is also stored under `"data"` when available.
## Execution flow and cancellation/timeout
+2 -2
View File
@@ -111,7 +111,7 @@ git add file && git commit -m "message"
GIT_AUTHOR_NAME=Dev git commit -m "message"
```
An anchored rule such as `^\s*git\s+commit\b` can therefore match the `git commit` command in both examples. Quoted, escaped, and commented text is not treated as a command. Heredocs, parameter expansion, command substitution, backticks, grouping, and malformed quoting retain only the complete-command check; the interceptor deliberately does not attempt to become a full shell parser.
An anchored rule such as `^\s*git\s+commit\b` can therefore match the `git commit` command in both examples. A stage that consumes another command's stdout through an unquoted `|` or `|&` (for example `grep x` in `printf 'x\n' | grep x`) is **not** treated as an interception candidate: it reads piped stdin, which the path-based dedicated tools cannot supply, so only a standalone or first-stage command is matched. Blank and comment-only continuation lines after the pipe preserve that context. Quoted, escaped, and commented text is not treated as a command. Heredocs, parameter expansion, command substitution, backticks, grouping, and malformed quoting retain only the complete-command check; the interceptor deliberately does not attempt to become a full shell parser.
### Interaction and selection guide
@@ -127,7 +127,7 @@ Choose the setting by the desired outcome:
1. `BashTool.execute()` in `packages/coding-agent/src/tools/bash.ts` reads `command`, validates `env`, and defaults `timeout` to `300`.
2. If `cwd` is absent, it rewrites a leading `cd <path> && ...` into the structured `cwd` field and strips that prefix from `command`.
3. If `async: true` is requested while `async.enabled` is off, it throws `ToolError` before any execution.
4. If `bashInterceptor.enabled` is on, `checkBashInterception()` runs against both the original command and the `cd`-stripped command. For each form, configured regexes still check the complete input first, then each flat command separated by unquoted/unescaped `&&`, `||`, `;`, `|`, `&`, or newlines, followed by versions of those fragments without leading `NAME=value` assignments. A matching enabled rule throws before URL expansion or execution.
4. If `bashInterceptor.enabled` is on, `checkBashInterception()` runs against both the original command and the `cd`-stripped command. For each form, configured regexes still check the complete input first, then each flat command separated by unquoted/unescaped `&&`, `||`, `;`, `|`, `|&`, `&`, or newlines (excluding stages that consume piped stdin from `|` or `|&`, including across blank/comment continuations), followed by versions of those fragments without leading `NAME=value` assignments. A matching enabled rule throws before URL expansion or execution.
5. `expandInternalUrls()` rewrites supported internal URLs inside `command`, each `env` value, and protocol-looking `cwd` values. Command replacements are shell-escaped; `env` and `cwd` replacements use raw filesystem/string values because they are not interpolated into shell text.
6. `resolveToCwd()` resolves `cwd` against `session.cwd`; `fs.stat()` verifies that the target exists and is a directory.
7. `timeout: 0` disables the deadline. Otherwise `clampTimeout("bash", requestedTimeoutSec, tools.maxTimeout)` applies a positive global ceiling (when configured), then `TOOL_TIMEOUTS.bash` (`min: 1`, `max: 3600`). When clamped, `#buildCompletedResult()` / `#buildBackgroundStartResult()` append a notice line.
+2 -2
View File
@@ -149,9 +149,9 @@ A stateless, tool-free one-shot model call:
Runs one subagent through `runStructuredSubagent(...)`:
- JS supports the preferred `await agent(prompt, { agent?, model?, label?, schema?, schemaMode?, isolated?, apply?, merge?, handle? })`; legacy positional slots are still implemented.
- JS supports the preferred `await agent(prompt, { agent?, label?, schema?, schemaMode?, isolated?, apply?, merge?, handle? })`; legacy positional slots are still implemented.
- Python/Ruby/Julia use keyword arguments (`schema_mode` outside JS).
- `agent` defaults from the current spawn policy. `model` may pin a selector/fallback chain. `schema` overrides agent/session schemas; `schemaMode`/`schema_mode` chooses `permissive` or `strict`.
- `agent` defaults from the current spawn policy; the selected agent's frontmatter model and settings always apply (there is no per-call model override — `model` is not accepted). `schema` overrides agent/session schemas; `schemaMode`/`schema_mode` chooses `permissive` or `strict`.
- `isolated` requests isolation. `apply` controls whether captured changes are integrated; `merge=false` selects patch mode while the normal setting controls branch mode.
- `handle=true` returns `{ text, output, handle, id, agent }`, optional parsed `data`, and isolation metadata instead of only output/data.
- Eval subagents are one-shot (`keepAlive=false`), are unregistered/disposed after completion, and **do not share the caller's eval executor** (`shareEvalSession=false`). Their code mutations therefore do not appear in the caller's retained VM/kernel.
-12
View File
@@ -22,18 +22,6 @@ Sources: [`python/robomp/README.md`](../python/robomp/README.md), [`python/robom
- Root commands: `bun run robomp:install` installs the Python package for host development; `bun run robomp:serve` runs it on the host; `bun run robomp:build`/`bun run robomp:rebuild`, `bun run robomp:up`, `bun run robomp:down`, `bun run robomp:restart`, `bun run robomp:logs`, `bun run robomp:dev`, and `bun run robomp:reset` manage the container deployment.
- Prerequisites: Docker Compose v2, a host-reachable LiteLLM-style model proxy, container model configuration, a GitHub webhook endpoint, and a bot PAT with write access to every allowlisted repository. The default two-container deployment keeps the PAT in an HMAC-authenticated `gh-proxy` sidecar rather than the orchestrator.
### `packages/swarm-extension` — swarm orchestration
Sources: [`packages/swarm-extension/README.md`](../packages/swarm-extension/README.md), [`packages/swarm-extension/package.json`](../packages/swarm-extension/package.json), [`packages/swarm-extension/src/cli.ts`](../packages/swarm-extension/src/cli.ts), [`packages/swarm-extension/src/extension.ts`](../packages/swarm-extension/src/extension.ts).
- Package: `@oh-my-pi/swarm-extension`; bin: `omp-swarm`.
- Feature: multi-agent DAG orchestration from YAML swarms, supporting `pipeline`, `parallel`, and `sequential` modes.
- Standalone CLI: `omp-swarm path/to/swarm.yaml` runs until completion or process termination.
- TUI extension mode: add the package path to `extensions`, then use `/swarm run <file.yaml>`, `/swarm status <name>`, or `/swarm help`.
- Inputs: YAML under top-level `swarm` with `name`, `workspace`, `mode`, optional `target_count`/`model`, and `agents` with `role`, `task`, optional `model`, `waits_for`, and `reports_to`.
- Side effects/output: creates the workspace if needed and persists state/logs under `<workspace>/.swarm_<name>/`.
- Limits/errors: validates the YAML definition, dependency graph, and cycles before execution; standalone runs have no built-in timeout.
### `packages/stats` — local usage dashboard
Sources: [`packages/stats/README.md`](../packages/stats/README.md), [`packages/stats/package.json`](../packages/stats/package.json), [`packages/coding-agent/src/cli/stats-cli.ts`](../packages/coding-agent/src/cli/stats-cli.ts).
+14 -14
View File
@@ -26,19 +26,19 @@
"@huggingface/transformers": "^4.2.0",
"@mozilla/readability": "^0.6.0",
"@napi-rs/cli": "3.7.2",
"@oh-my-pi/hashline": "17.2.8",
"@oh-my-pi/omp-stats": "17.2.8",
"@oh-my-pi/omptype": "17.2.8",
"@oh-my-pi/pi-agent-core": "17.2.8",
"@oh-my-pi/pi-ai": "17.2.8",
"@oh-my-pi/pi-catalog": "17.2.8",
"@oh-my-pi/pi-coding-agent": "17.2.8",
"@oh-my-pi/pi-mnemopi": "17.2.8",
"@oh-my-pi/pi-natives": "17.2.8",
"@oh-my-pi/pi-tui": "17.2.8",
"@oh-my-pi/pi-utils": "17.2.8",
"@oh-my-pi/pi-wire": "17.2.8",
"@oh-my-pi/snapcompact": "17.2.8",
"@oh-my-pi/hashline": "17.2.9",
"@oh-my-pi/omp-stats": "17.2.9",
"@oh-my-pi/omptype": "17.2.9",
"@oh-my-pi/pi-agent-core": "17.2.9",
"@oh-my-pi/pi-ai": "17.2.9",
"@oh-my-pi/pi-catalog": "17.2.9",
"@oh-my-pi/pi-coding-agent": "17.2.9",
"@oh-my-pi/pi-mnemopi": "17.2.9",
"@oh-my-pi/pi-natives": "17.2.9",
"@oh-my-pi/pi-tui": "17.2.9",
"@oh-my-pi/pi-utils": "17.2.9",
"@oh-my-pi/pi-wire": "17.2.9",
"@oh-my-pi/snapcompact": "17.2.9",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/api-logs": "^0.220.0",
"@opentelemetry/context-async-hooks": "^2.9.0",
@@ -117,7 +117,7 @@
"build:native": "bun --cwd=packages/natives run build",
"test": "bun scripts/ci-test-ts.ts local",
"test:ts": "bun scripts/ci-test-ts.ts local-ts",
"test:scripts": "bun test scripts/ci-release-build-binaries.test.ts scripts/musl-release.test.ts scripts/ci-release-publish.test.ts",
"test:scripts": "bun test scripts/ci-release-build-binaries.test.ts scripts/musl-release.test.ts scripts/ci-release-publish.test.ts scripts/release.test.ts",
"test:rs": "bun scripts/run-rs-task.ts test:rs",
"check": "bun run --parallel check:ts check:rs",
"check:ts": "bun run check:tools && bun run --workspaces --if-present check",
+6
View File
@@ -2,6 +2,12 @@
## [Unreleased]
## [17.2.9] - 2026-08-05
### Fixed
- Preserved queued steering and follow-up messages when a continuation is cancelled before or during pre-dequeue hooks, and propagated the caller's cancellation signal through every continuation model-call loop.
## [17.2.6] - 2026-08-03
### Fixed
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-agent-core",
"version": "17.2.8",
"version": "17.2.9",
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+5 -5
View File
@@ -1016,7 +1016,7 @@ async function runLoopBody(
// Skip when the run is already externally aborted — dequeuing would strand
// the messages in a run that is about to die.
try {
pendingMessages = signal?.aborted ? [] : (await config.getSteeringMessages?.()) || [];
pendingMessages = signal?.aborted ? [] : (await config.getSteeringMessages?.(signal)) || [];
} catch (error) {
stream.push({ type: "turn_start" });
emitInputMessages(stream, messagesToEmit);
@@ -1075,7 +1075,7 @@ async function runLoopBody(
let gateResult: AgentPreModelCallResult;
try {
if (config.syncContextBeforeModelCall) {
await config.syncContextBeforeModelCall(currentContext);
await config.syncContextBeforeModelCall(currentContext, signal);
}
if (!directiveResolvedForTurn) {
@@ -1421,7 +1421,7 @@ async function runLoopBody(
// instantly aborts — message lands in history, agent never responds. The
// mid-batch interrupt poll only peeks (hasSteeringMessages), so the queue
// still owns every message until this dequeue.
const steering = signal?.aborted ? [] : (await config.getSteeringMessages?.()) || [];
const steering = signal?.aborted ? [] : (await config.getSteeringMessages?.(signal)) || [];
if (hasMoreToolCalls) {
// Mid-work: fold any non-interrupting asides into the next turn alongside steering.
const asides = signal?.aborted ? [] : resolveAsides(await config.getAsideMessages?.());
@@ -1450,9 +1450,9 @@ async function runLoopBody(
// Re-poll steering too: a steer can land between the stop-boundary dequeue
// above and this yield point (e.g. queued while onBeforeYield ran). Without
// this poll it would strand in the queue until the next manual prompt.
const lateSteering = signal?.aborted ? [] : (await config.getSteeringMessages?.()) || [];
const lateSteering = signal?.aborted ? [] : (await config.getSteeringMessages?.(signal)) || [];
const asideMessages = signal?.aborted ? [] : resolveAsides(await config.getAsideMessages?.());
const followUpMessages = signal?.aborted ? [] : (await config.getFollowUpMessages?.()) || [];
const followUpMessages = signal?.aborted ? [] : (await config.getFollowUpMessages?.(signal)) || [];
if (lateSteering.length > 0 || asideMessages.length > 0 || followUpMessages.length > 0) {
// Set as pending so the inner loop processes them before stopping.
pendingMessages = [...lateSteering, ...asideMessages, ...followUpMessages];
+150 -57
View File
@@ -426,6 +426,8 @@ export class Agent {
#asideMessageProvider?: () => AsideMessage[] | Promise<AsideMessage[]>;
#telemetry?: AgentLoopConfig["telemetry"];
#appendOnlyContext?: AppendOnlyContextManager;
#beforeQueuedMessageDequeueHooks = new Set<(signal?: AbortSignal) => Promise<void> | void>();
#beforeModelCallHooks = new Set<(signal?: AbortSignal) => Promise<void> | void>();
/** Buffered Cursor tool results with text length at time of call (for correct ordering) */
#cursorToolResultBuffer: CursorToolResultEntry[] = [];
@@ -784,6 +786,40 @@ export class Agent {
return () => this.#listeners.delete(fn);
}
/** Register an independently removable hook that runs before queued messages are consumed. */
addBeforeQueuedMessageDequeueHook(hook: (signal?: AbortSignal) => Promise<void> | void): () => void {
const registration = (signal?: AbortSignal) => hook(signal);
this.#beforeQueuedMessageDequeueHooks.add(registration);
return () => this.#beforeQueuedMessageDequeueHooks.delete(registration);
}
/** Register an independently removable hook that runs immediately before each model call. */
addBeforeModelCallHook(hook: (signal?: AbortSignal) => Promise<void> | void): () => void {
const registration = (signal?: AbortSignal) => hook(signal);
this.#beforeModelCallHooks.add(registration);
return () => this.#beforeModelCallHooks.delete(registration);
}
async #runBeforeModelCallHooks(signal?: AbortSignal): Promise<void> {
for (const hook of this.#beforeModelCallHooks) await hook(signal);
}
async #runBeforeQueuedMessageDequeueHooks(signal?: AbortSignal): Promise<void> {
for (const hook of this.#beforeQueuedMessageDequeueHooks) await hook(signal);
}
async #dequeueSteeringMessagesAfterHooks(signal?: AbortSignal): Promise<AgentMessage[]> {
if (signal?.aborted || this.#steeringQueue.length === 0) return [];
await this.#runBeforeQueuedMessageDequeueHooks(signal);
return signal?.aborted ? [] : this.#dequeueSteeringMessages();
}
async #dequeueFollowUpMessagesAfterHooks(signal?: AbortSignal): Promise<AgentMessage[]> {
if (signal?.aborted || this.#followUpQueue.length === 0) return [];
await this.#runBeforeQueuedMessageDequeueHooks(signal);
return signal?.aborted ? [] : this.#dequeueFollowUpMessages();
}
setProviderResponseInterceptor(fn: SimpleStreamOptions["onResponse"] | undefined): void {
this.#onResponse = fn;
}
@@ -1137,48 +1173,90 @@ export class Agent {
/**
* Continue from current context (used for retries and resuming queued messages).
*/
async continue() {
#continuationDequeueSignal(signal?: AbortSignal): AbortSignal | undefined {
const signals: AbortSignal[] = [];
if (this.#abortController) signals.push(this.#abortController.signal);
if (signal) signals.push(signal);
if (this.#deadline !== undefined) {
const delay = this.#deadline - Date.now();
if (delay <= 0) {
const controller = new AbortController();
controller.abort(new DOMException("Deadline exceeded", "TimeoutError"));
signals.push(controller.signal);
} else {
signals.push(AbortSignal.timeout(delay));
}
}
if (signals.length === 0) return undefined;
return signals.length === 1 ? signals[0] : AbortSignal.any(signals);
}
async continue(signal?: AbortSignal) {
if (this.#state.isStreaming) {
throw new AgentBusyError();
}
const messages = this.#state.messages;
if (messages.length === 0) {
// An empty transcript has nothing to resume, but a queued steer/follow-up
// must still be delivered as the opening turn — mirroring the assistant-tail
// branch below. Throwing here leaves the message undeliverable, and idle-drain
// callers (AgentSession#scheduleQueuedMessageDrain) re-arm continue() on every
// microtask because hasQueuedMessages() never clears, spinning an unbounded
// allocation loop until OOM (issue #6344).
const queuedSteering = this.#dequeueSteeringMessages();
if (queuedSteering.length > 0) {
await this.#runLoop(queuedSteering, { skipInitialSteeringPoll: true });
return;
const { promise, resolve } = Promise.withResolvers<void>();
this.#runningPrompt = promise;
this.#resolveRunningPrompt = resolve;
const continuationAbortController = new AbortController();
this.#abortController = continuationAbortController;
this.#state.isStreaming = true;
this.#state.streamMessage = null;
this.#state.error = undefined;
try {
const dequeueSignal = this.#continuationDequeueSignal(signal);
const messages = this.#state.messages;
if (messages.length === 0) {
// An empty transcript has nothing to resume, but a queued steer/follow-up
// must still be delivered as the opening turn — mirroring the assistant-tail
// branch below. Throwing here leaves the message undeliverable, and idle-drain
// callers (AgentSession#scheduleQueuedMessageDrain) re-arm continue() on every
// microtask because hasQueuedMessages() never clears, spinning an unbounded
// allocation loop until OOM (issue #6344).
const queuedSteering = await this.#dequeueSteeringMessagesAfterHooks(dequeueSignal);
if (queuedSteering.length > 0) {
await this.#runLoop(queuedSteering, { skipInitialSteeringPoll: true }, signal, true);
return;
}
const queuedFollowUp = await this.#dequeueFollowUpMessagesAfterHooks(dequeueSignal);
if (queuedFollowUp.length > 0) {
await this.#runLoop(queuedFollowUp, undefined, signal, true);
return;
}
throw new Error("No messages to continue from");
}
const queuedFollowUp = this.#dequeueFollowUpMessages();
if (queuedFollowUp.length > 0) {
await this.#runLoop(queuedFollowUp);
return;
if (messages[messages.length - 1].role === "assistant") {
const queuedSteering = await this.#dequeueSteeringMessagesAfterHooks(dequeueSignal);
if (queuedSteering.length > 0) {
await this.#runLoop(queuedSteering, { skipInitialSteeringPoll: true }, signal, true);
return;
}
const queuedFollowUp = await this.#dequeueFollowUpMessagesAfterHooks(dequeueSignal);
if (queuedFollowUp.length > 0) {
await this.#runLoop(queuedFollowUp, undefined, signal, true);
return;
}
throw new Error("Cannot continue from message role: assistant");
}
await this.#runLoop(undefined, undefined, signal, true);
} finally {
resolve();
if (this.#abortController === continuationAbortController) {
this.#state.isStreaming = false;
this.#state.streamMessage = null;
this.#state.pendingToolCalls.clear();
this.#abortController = undefined;
if (this.#runningPrompt === promise) {
this.#runningPrompt = undefined;
this.#resolveRunningPrompt = undefined;
}
}
throw new Error("No messages to continue from");
}
if (messages[messages.length - 1].role === "assistant") {
const queuedSteering = this.#dequeueSteeringMessages();
if (queuedSteering.length > 0) {
await this.#runLoop(queuedSteering, { skipInitialSteeringPoll: true });
return;
}
const queuedFollowUp = this.#dequeueFollowUpMessages();
if (queuedFollowUp.length > 0) {
await this.#runLoop(queuedFollowUp);
return;
}
throw new Error("Cannot continue from message role: assistant");
}
await this.#runLoop(undefined);
}
/**
@@ -1186,17 +1264,29 @@ export class Agent {
* If messages are provided, starts a new conversation turn with those messages.
* Otherwise, continues from existing context.
*/
async #runLoop(messages?: AgentMessage[], options?: AgentPromptOptions & { skipInitialSteeringPoll?: boolean }) {
async #runLoop(
messages?: AgentMessage[],
options?: AgentPromptOptions & { skipInitialSteeringPoll?: boolean },
continuationSignal?: AbortSignal,
runStateClaimed = false,
) {
const model = this.#state.model;
if (!model) throw new Error("No model configured");
let skipInitialSteeringPoll = options?.skipInitialSteeringPoll === true;
using _ = new EventLoopKeepalive();
const { promise, resolve } = Promise.withResolvers<void>();
this.#runningPrompt = promise;
this.#resolveRunningPrompt = resolve;
this.#abortController = new AbortController();
if (!runStateClaimed) {
const { promise, resolve } = Promise.withResolvers<void>();
this.#runningPrompt = promise;
this.#resolveRunningPrompt = resolve;
this.#abortController = new AbortController();
}
const resolveRun = this.#resolveRunningPrompt;
const loopAbortController = this.#abortController;
if (!loopAbortController) throw new Error("Agent run state was not initialized");
const loopSignal = continuationSignal
? AbortSignal.any([loopAbortController.signal, continuationSignal])
: loopAbortController.signal;
this.#state.isStreaming = true;
this.#state.streamMessage = null;
this.#state.error = undefined;
@@ -1315,7 +1405,8 @@ export class Agent {
onSseEvent: this.#onSseEvent,
getApiKey: this.getApiKey,
getToolContext: this.#getToolContext,
syncContextBeforeModelCall: async context => {
syncContextBeforeModelCall: async (context, signal) => {
await this.#runBeforeModelCallHooks(signal);
if (this.#listeners.size > 0) {
await Bun.sleep(0);
}
@@ -1362,12 +1453,12 @@ export class Agent {
getReasoning: () => this.#state.thinkingLevel,
getDisableReasoning: () => this.#state.disableReasoning,
getServiceTier: this.#serviceTierResolver,
getSteeringMessages: async () => {
getSteeringMessages: async signal => {
if (skipInitialSteeringPoll) {
skipInitialSteeringPoll = false;
return [];
}
return this.#dequeueSteeringMessages();
return this.#dequeueSteeringMessagesAfterHooks(signal);
},
hasSteeringMessages: () => {
if (this.#steeringQueue.length === 0) {
@@ -1392,7 +1483,7 @@ export class Agent {
},
waitForSteeringMessages: signal => this.#waitForSteeringMessages(signal),
hasIrcInterrupts: this.hasIrcInterrupts,
getFollowUpMessages: async () => this.#dequeueFollowUpMessages(),
getFollowUpMessages: signal => this.#dequeueFollowUpMessagesAfterHooks(signal),
getAsideMessages: async () => (await this.#asideMessageProvider?.()) ?? [],
onBeforeYield: () => this.#onBeforeYield?.(),
telemetry: this.#telemetry,
@@ -1404,8 +1495,8 @@ export class Agent {
try {
const stream = messages
? agentLoop(messages, context, config, this.#abortController.signal, this.streamFn)
: agentLoopContinue(context, config, this.#abortController.signal, this.streamFn);
? agentLoop(messages, context, config, loopSignal, this.streamFn)
: agentLoopContinue(context, config, loopSignal, this.streamFn);
for await (const event of stream) {
if (event.type === "turn_start") turnOpen = true;
@@ -1472,15 +1563,15 @@ export class Agent {
if (!onlyEmpty) {
this.appendMessage(partial);
} else {
if (this.#abortController?.signal.aborted) {
if (loopSignal.aborted) {
throw new Error("Request was aborted");
}
}
}
} catch (err) {
const stoppedForAbort = this.#abortController?.signal.aborted === true;
const stoppedForAbort = loopSignal.aborted;
const errorMessage = stoppedForAbort
? abortReasonText(this.#abortController?.signal)
? abortReasonText(loopSignal)
: err instanceof Error
? err.message
: String(err);
@@ -1582,13 +1673,15 @@ export class Agent {
this.#emit({ type: "agent_end", messages: [errorMsg] });
}
} finally {
this.#state.isStreaming = false;
this.#state.streamMessage = null;
this.#state.pendingToolCalls.clear();
this.#abortController = undefined;
this.#resolveRunningPrompt?.();
this.#runningPrompt = undefined;
this.#resolveRunningPrompt = undefined;
resolveRun?.();
if (this.#abortController === loopAbortController) {
this.#state.isStreaming = false;
this.#state.streamMessage = null;
this.#state.pendingToolCalls.clear();
this.#abortController = undefined;
this.#runningPrompt = undefined;
this.#resolveRunningPrompt = undefined;
}
}
}
+2 -2
View File
@@ -11,7 +11,7 @@
*/
export class CompactionCancelledError extends Error {
readonly name = "CompactionCancelledError" as const;
override readonly name = "CompactionCancelledError" as const;
constructor(message = "Compaction cancelled") {
super(message);
@@ -27,7 +27,7 @@ export class CompactionCancelledError extends Error {
* ordinary summarization errors and must not fall through to another provider.
*/
export class NativeCompactionError extends Error {
readonly name = "NativeCompactionError" as const;
override readonly name = "NativeCompactionError" as const;
constructor(cause: unknown) {
super(cause instanceof Error ? cause.message : String(cause), { cause });
+3 -3
View File
@@ -240,7 +240,7 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
* mid-batch interrupt poll uses {@link hasSteeringMessages} instead and
* never consumes the queue.
*/
getSteeringMessages?: () => Promise<AgentMessage[]>;
getSteeringMessages?: (signal?: AbortSignal) => Promise<AgentMessage[]>;
/**
* Peeks whether steering messages are queued, without consuming them.
@@ -285,7 +285,7 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
* If messages are returned, they're added to the context and the agent
* continues with another turn.
*/
getFollowUpMessages?: () => Promise<AgentMessage[]>;
getFollowUpMessages?: (signal?: AbortSignal) => Promise<AgentMessage[]>;
/**
* Returns non-interrupting "aside" messages to inject at a step boundary.
*
@@ -319,7 +319,7 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
* Mutate the agent context here; use `beforeModelCall` to inspect the
* provider-bound context.
*/
syncContextBeforeModelCall?: (context: AgentContext) => void | Promise<void>;
syncContextBeforeModelCall?: (context: AgentContext, signal?: AbortSignal) => void | Promise<void>;
/**
* Asked after the complete provider context has been built, including
+225 -2
View File
@@ -1,5 +1,5 @@
import { describe, expect, it } from "bun:test";
import { Agent, type AgentEvent, type AgentTool, ThinkingLevel } from "@oh-my-pi/pi-agent-core";
import { Agent, AgentBusyError, type AgentEvent, type AgentTool, ThinkingLevel } from "@oh-my-pi/pi-agent-core";
import { type SimpleStreamOptions, type ToolResultMessage, z } from "@oh-my-pi/pi-ai";
import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock";
import { kCursorExecResolved } from "@oh-my-pi/pi-ai/utils/block-symbols";
@@ -240,6 +240,213 @@ describe("Agent", () => {
}
});
it("removes duplicate queued-message hooks independently", async () => {
const mock = createMockModel({ responses: [{ content: ["first"] }, { content: ["second"] }] });
const agent = new Agent({ streamFn: mock.stream });
agent.replaceMessages([createAssistantMessage([{ type: "text", text: "ready" }])]);
let calls = 0;
const signals: Array<AbortSignal | undefined> = [];
const hook = (signal?: AbortSignal) => {
calls++;
signals.push(signal);
};
const removeFirst = agent.addBeforeQueuedMessageDequeueHook(hook);
const removeSecond = agent.addBeforeQueuedMessageDequeueHook(hook);
const controller = new AbortController();
removeFirst();
agent.followUp({ role: "user", content: "first turn", timestamp: Date.now() });
await agent.continue(controller.signal);
expect(calls).toBe(1);
expect(signals).toEqual([controller.signal]);
removeSecond();
agent.followUp({ role: "user", content: "second turn", timestamp: Date.now() });
await agent.continue();
expect(calls).toBe(1);
});
it("continue() leaves queued messages owned when its signal is already aborted", async () => {
const agent = new Agent();
agent.replaceMessages([createAssistantMessage([{ type: "text", text: "ready" }])]);
agent.followUp({ role: "user", content: "stay queued", timestamp: Date.now() });
const controller = new AbortController();
controller.abort();
await expect(agent.continue(controller.signal)).rejects.toThrow("Cannot continue from message role: assistant");
expect(agent.peekFollowUpQueue()).toHaveLength(1);
});
it("keeps follow-up ownership when the deadline expires during a dequeue hook", async () => {
const mock = createMockModel({ responses: [{ content: ["done"] }] });
const agent = new Agent({ streamFn: mock.stream, deadline: Date.now() + 25 });
let hookSignal: AbortSignal | undefined;
agent.addBeforeQueuedMessageDequeueHook(async signal => {
if (!signal) throw new Error("Expected the active loop signal");
hookSignal = signal;
if (signal.aborted) return;
const { promise, resolve } = Promise.withResolvers<void>();
signal.addEventListener("abort", () => resolve(), { once: true });
await promise;
});
agent.followUp({ role: "user", content: "stay queued after deadline", timestamp: Date.now() });
await agent.prompt("start");
expect(hookSignal?.aborted).toBe(true);
expect(agent.peekFollowUpQueue()).toHaveLength(1);
});
it("keeps queued work when continue() reaches its deadline inside a dequeue hook", async () => {
const agent = new Agent({ deadline: Date.now() + 25 });
agent.replaceMessages([createAssistantMessage([{ type: "text", text: "ready" }])]);
agent.addBeforeQueuedMessageDequeueHook(async signal => {
if (!signal) throw new Error("Expected the deadline-aware dequeue signal");
if (signal.aborted) return;
const { promise, resolve } = Promise.withResolvers<void>();
signal.addEventListener("abort", () => resolve(), { once: true });
await promise;
});
agent.followUp({ role: "user", content: "stay queued before run loop", timestamp: Date.now() });
await expect(agent.continue()).rejects.toThrow("Cannot continue from message role: assistant");
expect(agent.peekFollowUpQueue()).toHaveLength(1);
});
it("claims an abortable busy state while continue() awaits dequeue hooks", async () => {
const agent = new Agent();
agent.replaceMessages([createAssistantMessage([{ type: "text", text: "ready" }])]);
agent.followUp({ role: "user", content: "stay queued", timestamp: Date.now() });
const hookStarted = Promise.withResolvers<void>();
agent.addBeforeQueuedMessageDequeueHook(async signal => {
if (!signal) throw new Error("Expected continuation dequeue signal");
hookStarted.resolve();
if (signal.aborted) return;
const { promise, resolve } = Promise.withResolvers<void>();
signal.addEventListener("abort", () => resolve(), { once: true });
await promise;
});
const continuing = agent.continue();
await hookStarted.promise;
let idleResolved = false;
const idle = agent.waitForIdle().then(() => {
idleResolved = true;
});
await Promise.resolve();
expect(agent.state.isStreaming).toBe(true);
expect(idleResolved).toBe(false);
await expect(agent.prompt("must not overlap")).rejects.toBeInstanceOf(AgentBusyError);
agent.abort("cancel dequeue");
await expect(continuing).rejects.toThrow("Cannot continue from message role: assistant");
await idle;
expect(idleResolved).toBe(true);
expect(agent.state.isStreaming).toBe(false);
expect(agent.peekFollowUpQueue()).toHaveLength(1);
});
it("does not clear a successor prompt after continue() releases idle waiters", async () => {
const firstStarted = Promise.withResolvers<void>();
const releaseFirst = Promise.withResolvers<void>();
const secondStarted = Promise.withResolvers<void>();
const releaseSecond = Promise.withResolvers<void>();
const mock = createMockModel({
responses: [
async () => {
firstStarted.resolve();
await releaseFirst.promise;
return { content: ["continued"] };
},
async () => {
secondStarted.resolve();
await releaseSecond.promise;
return { content: ["successor"] };
},
],
});
const agent = new Agent({ streamFn: mock.stream });
agent.replaceMessages([createAssistantMessage([{ type: "text", text: "ready" }])]);
agent.followUp({ role: "user", content: "continue", timestamp: Date.now() });
const continuing = agent.continue();
await firstStarted.promise;
const successor = agent.waitForIdle().then(() => agent.prompt("next prompt"));
releaseFirst.resolve();
await secondStarted.promise;
await continuing;
expect(agent.state.isStreaming).toBe(true);
releaseSecond.resolve();
await successor;
expect(agent.state.isStreaming).toBe(false);
});
it("resolves a predecessor idle waiter when agent_end starts a successor", async () => {
const secondStarted = Promise.withResolvers<void>();
const releaseSecond = Promise.withResolvers<void>();
const mock = createMockModel({
responses: [
{ content: ["first"] },
async () => {
secondStarted.resolve();
await releaseSecond.promise;
return { content: ["second"] };
},
],
});
const agent = new Agent({ streamFn: mock.stream });
let successor: Promise<void> | undefined;
agent.subscribe(event => {
if (event.type === "agent_end" && !successor) {
successor = agent.prompt("successor");
}
});
const predecessor = agent.prompt("predecessor");
let predecessorIdleResolved = false;
void agent.waitForIdle().then(() => {
predecessorIdleResolved = true;
});
await secondStarted.promise;
await predecessor;
expect(agent.state.isStreaming).toBe(true);
releaseSecond.resolve();
await successor;
await Promise.resolve();
expect(predecessorIdleResolved).toBe(true);
expect(agent.state.isStreaming).toBe(false);
});
it("classifies an in-flight continuation cancellation as aborted", async () => {
const providerStarted = Promise.withResolvers<AbortSignal>();
const agent = new Agent({
streamFn: (_model, _context, options) => {
const signal = options?.signal;
if (!signal) throw new Error("Expected provider abort signal");
providerStarted.resolve(signal);
const stream = new AssistantMessageEventStream();
signal.addEventListener("abort", () => stream.fail(new Error("provider aborted")), { once: true });
return stream;
},
});
agent.replaceMessages([createAssistantMessage([{ type: "text", text: "ready" }])]);
agent.followUp({ role: "user", content: "cancel this continuation", timestamp: Date.now() });
const controller = new AbortController();
const running = agent.continue(controller.signal);
await providerStarted.promise;
controller.abort("caller cancelled");
await running;
const finalMessage = agent.state.messages.at(-1);
expect(finalMessage?.role).toBe("assistant");
if (finalMessage?.role !== "assistant") throw new Error("Expected aborted assistant message");
expect(finalMessage.stopReason).toBe("aborted");
expect(finalMessage.errorMessage).toBe("caller cancelled");
});
it("continue() should process queued follow-up messages after an assistant turn", async () => {
const mock = createMockModel({ responses: [{ content: ["Processed"] }] });
const agent = new Agent({ streamFn: mock.stream });
@@ -276,6 +483,12 @@ describe("Agent", () => {
responses: [{ content: ["Processed 1"] }, { content: ["Processed 2"] }],
});
const agent = new Agent({ streamFn: mock.stream });
let dequeueHooks = 0;
const dequeueSignals: Array<AbortSignal | undefined> = [];
agent.addBeforeQueuedMessageDequeueHook(signal => {
dequeueHooks++;
dequeueSignals.push(signal);
});
agent.replaceMessages([
{
@@ -297,11 +510,16 @@ describe("Agent", () => {
timestamp: Date.now() + 1,
});
await expect(agent.continue()).resolves.toBeUndefined();
const controller = new AbortController();
await expect(agent.continue(controller.signal)).resolves.toBeUndefined();
const recentMessages = agent.state.messages.slice(-4);
expect(recentMessages.map(m => m.role)).toEqual(["user", "assistant", "user", "assistant"]);
expect(mock.calls.length).toBe(2);
expect(dequeueHooks).toBe(2);
expect(dequeueSignals).toHaveLength(2);
controller.abort();
expect(dequeueSignals.every(signal => signal?.aborted === true)).toBe(true);
});
it("delivers a steer that lands at the yield boundary instead of stranding it", async () => {
@@ -856,6 +1074,10 @@ describe("Agent", () => {
},
streamFn: mock.stream,
});
let beforeModelCalls = 0;
agent.addBeforeModelCallHook(() => {
beforeModelCalls++;
});
const unsubscribe = agent.subscribe(event => {
if (event.type === "message_end" && event.message.role === "toolResult") {
@@ -875,6 +1097,7 @@ describe("Agent", () => {
{ systemPrompt: "prompt-one", toolNames: ["alpha"] },
{ systemPrompt: "prompt-two", toolNames: ["alpha", "beta"] },
]);
expect(beforeModelCalls).toBe(2);
});
it("prompt() drops stale forced toolChoice after same-turn tool refresh", async () => {
+11
View File
@@ -2,6 +2,17 @@
## [Unreleased]
## [17.2.9] - 2026-08-05
### Fixed
- Fixed GitHub Copilot requests failing with a raw `HTTP 400 model_not_available_for_integrator` on roughly half of all turns for recently rolled-out models. Copilot's fleet is not uniform — part of it rejects models that `/models` advertises on the same host — and the transient classifier matched only the older `model_not_supported` code at a fixed envelope depth, so these rejections surfaced as terminal errors instead of entering the existing retry path. Model-availability 400s are now recognized at any envelope depth and rerolled on a flat delay with a dedicated 8-attempt budget on the OpenAI transports; every other retryable failure keeps its previous backoff and attempt count.
- Fixed Cursor reads with inline OMP range selectors reporting the returned slice length as the source file's `totalLines`, which made sequential reads of an unchanged file appear inconsistent ([#7590](https://github.com/can1357/oh-my-pi/issues/7590)).
- Made model-scoped usage health ignore Codex accounts that cannot use the requested plan-gated model while retaining conservative unknown-state handling and independent usage-window resets.
- Fixed OpenAI Codex usage telemetry blocking explicitly allowed ChatGPT Team credentials when a weekly `used_percent` rounded to 100, which could route multi-account sessions to an actually exhausted sibling instead ([#7617](https://github.com/can1357/oh-my-pi/issues/7617)).
- Fixed OpenAI Codex GPT-5.x requests sending optional `reasoning.summary`, `reasoning.context`, and `text.verbosity` controls by default, reducing Codex `server_error` disconnects from unsupported request shapes. ([#4949](https://github.com/can1357/oh-my-pi/issues/4949))
- Classified concurrent-request caps separately from quota exhaustion so they use a short retry backoff without burning a credential, and rotate credentials for account-scoped 403 caps such as Devin's overall message limit.
## [17.2.7] - 2026-08-03
### Changed
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-ai",
"version": "17.2.8",
"version": "17.2.9",
"description": "Unified LLM API with automatic model discovery and provider configuration",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+8 -5
View File
@@ -3,7 +3,7 @@ import type { OAuthAccess } from "./auth-storage";
import * as AIError from "./error";
import { isAuthRetryableError, isInvalidatedOAuthTokenError } from "./error/auth-classify";
import { isUsageLimit } from "./error/flags";
import { isUsageLimitOutcome } from "./error/rate-limit";
import { isConcurrencyCapExclusion, isUsageLimitOutcome } from "./error/rate-limit";
/**
* Context passed to an {@link ApiKeyResolver} on each resolution attempt.
@@ -93,11 +93,14 @@ export const AUTH_RETRY_MAX_ATTEMPTS = 64;
function isDirectCredentialRotationError(error: unknown): boolean {
if (isUsageLimit(error) || isInvalidatedOAuthTokenError(error)) return true;
const status = AIError.status(error);
// 403: the token is valid but access was denied, so refreshing the same
// credential can't help — rotate straight through the sibling pool.
if (status === 403) return true;
const message = error instanceof Error ? error.message : typeof error === "string" ? error : undefined;
if (status === undefined && message !== undefined && extractHttpStatusFromError({ message }) === 403) return true;
// A 403 normally means a valid token lacks access, so rotate through
// siblings. A concurrency-cap 403 is transient instead; do not burn a
// sibling before the caller's backoff layer can retry it.
const isForbidden =
status === 403 ||
(status === undefined && message !== undefined && extractHttpStatusFromError({ message }) === 403);
if (isForbidden && !isConcurrencyCapExclusion(status, message)) return true;
return isUsageLimitOutcome(status, message);
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+5 -4
View File
@@ -1,6 +1,6 @@
import { extractHttpStatusFromError } from "@oh-my-pi/pi-utils";
import { isOAuthExpiry, isUsageLimit } from "./flags";
import { isUsageLimitOutcome } from "./rate-limit";
import { isConcurrencyCapExclusion, isUsageLimitOutcome } from "./rate-limit";
/**
* Whether an OAuth refresh failure is definitive (the credential must be
@@ -38,9 +38,10 @@ export function isAuthRetryableError(error: unknown): boolean {
if (isUsageLimit(error)) return true;
if (isInvalidatedOAuthTokenError(error)) return true;
const httpStatus = extractHttpStatusFromError(error);
if (httpStatus === 401 || httpStatus === 403) return true;
const message = error instanceof Error ? error.message : typeof error === "string" ? error : undefined;
const embeddedStatus = message ? extractHttpStatusFromError({ message }) : undefined;
if (embeddedStatus === 401 || embeddedStatus === 403) return true;
return isUsageLimitOutcome(httpStatus ?? embeddedStatus, message);
const status = httpStatus ?? embeddedStatus;
if (isConcurrencyCapExclusion(status, message)) return false;
if (status === 401 || status === 403) return true;
return isUsageLimitOutcome(status, message);
}
+74 -18
View File
@@ -7,7 +7,13 @@ import {
ProviderHttpError,
STREAM_ENVELOPE_ERROR_PREFIX,
} from "./classes";
import { isOpaqueStatusBody, isUsageLimitStatus, matchesUsageLimitText, parseRateLimitReason } from "./rate-limit";
import {
isAccountScopedCapText,
isOpaqueStatusBody,
isUsageLimitStatus,
matchesUsageLimitText,
parseRateLimitReason,
} from "./rate-limit";
export const Flag = {
Class: 0x1000,
@@ -107,10 +113,17 @@ const STALE_RESPONSE_ITEM_DETAIL_PATTERN = /not[ _]?found|invalid|expired|stale|
export const LLAMA_CPP_TOOL_CALL_PARSE_PATTERN =
/failed to parse tool call arguments as json|\[json\.exception\.parse_error\.101\]/i;
// Copilot routing flap: HTTP 400 `model_not_supported` (structural code on the
// error, also surfaced in text). Treated as transient — a retry usually lands
// on a backend that has the model.
const COPILOT_MODEL_NOT_SUPPORTED_PATTERN = /model_not_supported/i;
// Copilot fleet skew: HTTP 400 rejecting a model that `/models` advertised on
// the very same host. Two codes appear in the wild — `model_not_supported`
// (per-OAuth-client rollout gap) and `model_not_available_for_integrator`
// (replicas whose integrator allowlist predates the model). Both flap
// request-to-request, so a retry usually lands on a backend that has the model.
const COPILOT_TRANSIENT_MODEL_CODES: Record<string, true> = {
model_not_supported: true,
model_not_available_for_integrator: true,
};
const COPILOT_MODEL_UNAVAILABLE_PATTERN =
/model_not_supported|model_not_available_for_integrator|not available for integrator/i;
// Anthropic strict-tool grammar too large / schema too complex (400 invalid_request_error).
// Feature-gated deployments (Azure Foundry, Baseten, …) reject `strict: true`
// tools outright when the hosted model lacks structured outputs, e.g.
@@ -332,21 +345,40 @@ function classifyText(errorMessage: string | undefined, errorStatus: number | un
const isOpaque = isOpaqueStatusBody(cleanMessage);
const isLimitStatus = isUsageLimitStatus(statusClean);
const reason = parseRateLimitReason(cleanMessage);
// Concurrency caps (e.g. Vertex "Online prediction concurrent requests
// quota exceeded") are shed-and-backoff, not credential-rotatable —
// exclude them even when the quota-worded phrasing matches the generic
// usage-limit text matcher, whose `quota.?exceeded` arm would otherwise
// set Flag.UsageLimit and burn a healthy sibling credential. HTTP 402 is
// excluded from this gate: it is categorically an account-billing cap, so
// a 402 whose body merely mentions concurrency still classifies as a
// usage limit, mirroring isUsageLimitOutcome.
const isBillingCapStatus = statusClean === 402;
const concurrencyExcluded = reason === "CONCURRENT_LIMIT" && !isBillingCapStatus;
if (
matchesUsageLimitText(cleanMessage) ||
(isLimitStatus && (isOpaque || parseRateLimitReason(cleanMessage) === "QUOTA_EXHAUSTED"))
!concurrencyExcluded &&
(matchesUsageLimitText(cleanMessage) ||
((statusClean === 403 || statusClean === undefined) && isAccountScopedCapText(cleanMessage)) ||
(isLimitStatus &&
(isOpaque || reason === "QUOTA_EXHAUSTED" || (isBillingCapStatus && reason === "CONCURRENT_LIMIT"))))
) {
kinds |= Flag.UsageLimit;
}
if (isTimeoutText(errorMessage)) kinds |= Flag.Transient | Flag.Timeout;
else if (isTransientErrorText(errorMessage)) kinds |= Flag.Transient;
// A concurrency cap (e.g. Vertex "Online prediction concurrent requests
// quota exceeded") is transient — shed-and-backoff. The bare wording need
// not match TRANSIENT_TRANSPORT_PATTERN, so flag it explicitly to keep
// AIError.retriable from treating the temporary cap as terminal.
if (reason === "CONCURRENT_LIMIT") kinds |= Flag.Transient;
if ((api === "openai-responses" || api === "openai-codex-responses") && isStaleResponsesText(errorMessage)) {
kinds |= Flag.StaleResponsesItem;
}
// Copilot per-client routing flap is transient.
if (statusClean === 400 && COPILOT_MODEL_NOT_SUPPORTED_PATTERN.test(cleanMessage)) kinds |= Flag.Transient;
// Copilot fleet-skew model rejection is transient.
if (statusClean === 400 && COPILOT_MODEL_UNAVAILABLE_PATTERN.test(cleanMessage)) kinds |= Flag.Transient;
if (matchesStrictToolsRejection(cleanMessage, statusClean)) kinds |= Flag.Grammar;
if (matchesFastModeUnsupported(cleanMessage, statusClean)) kinds |= Flag.FastModeUnsupported;
}
@@ -398,7 +430,10 @@ export function classify(error: unknown, api?: Api): number {
if (code === "overloaded_error" || code === "rate_limit_error") {
linkKinds |= Flag.Transient;
}
if (codeStatus === 401 || codeStatus === 403) {
if (
(codeStatus === 401 || codeStatus === 403) &&
!(codeStatus === 403 && parseRateLimitReason(link.message) === "CONCURRENT_LIMIT")
) {
linkKinds |= Flag.AuthFailed;
} else if (codeStatus === 429) {
if ((linkKinds & Flag.UsageLimit) === 0) {
@@ -460,16 +495,37 @@ export function isFastModeUnsupported(error: unknown): boolean {
}
/**
* GitHub Copilot 400 `model_not_supported` routing flap — transient. Reads the
* structural `code` (and falls back to {@link Flag.Transient} text classification).
* Depth-bounded search for a provider error `code`. SDK error objects keep the
* parsed response body on `.error`, and Copilot's body is itself
* `{ error: { code } }`, so the code sits up to two envelopes below the thrown
* error depending on which SDK produced it.
*/
function providerErrorCode(error: object): string | undefined {
let node: object = error;
for (let depth = 0; depth < 3; depth++) {
if ("code" in node && typeof node.code === "string") return node.code;
if (!("error" in node)) return undefined;
const nested: unknown = node.error;
if (!nested || typeof nested !== "object") return undefined;
node = nested;
}
return undefined;
}
/**
* GitHub Copilot 400 rejecting a model its own `/models` catalog advertises —
* transient fleet skew, not a malformed request. Reads the structural `code`
* through the SDK/body envelopes, then falls back to the stringified body both
* SDK families put in `message` (shapes drift; the wire text does not).
*/
export function isCopilotTransientModelError(error: unknown): boolean {
if (status(error) === 400 && error && typeof error === "object") {
const info = error as { code?: unknown; error?: { code?: unknown } | null };
const code = typeof info.code === "string" ? info.code : info.error?.code;
if (code === "model_not_supported") return true;
}
return false;
if (!error || typeof error !== "object" || status(error) !== 400) return false;
const code = providerErrorCode(error);
// `Object.hasOwn`, not a bare index: `code` is provider-controlled, and a
// prototype key (`__proto__`, `toString`, …) would otherwise read truthy.
if (code !== undefined && Object.hasOwn(COPILOT_TRANSIENT_MODEL_CODES, code)) return true;
const message: unknown = "message" in error ? error.message : undefined;
return typeof message === "string" && COPILOT_MODEL_UNAVAILABLE_PATTERN.test(message);
}
export function classifyMessage(message: {
+1
View File
@@ -9,5 +9,6 @@ export * from "./format";
export * from "./gateway";
export * from "./oauth";
export * from "./provider";
export * from "./rate-limit";
export * from "./retryable";
export * from "./validation";
+64 -5
View File
@@ -6,12 +6,14 @@
export type RateLimitReason =
| "QUOTA_EXHAUSTED"
| "RATE_LIMIT_EXCEEDED"
| "CONCURRENT_LIMIT"
| "MODEL_CAPACITY_EXHAUSTED"
| "SERVER_ERROR"
| "UNKNOWN";
const QUOTA_EXHAUSTED_BACKOFF_MS = 30 * 60 * 1000; // 30 min
const RATE_LIMIT_EXCEEDED_BACKOFF_MS = 30 * 1000; // 30s
const CONCURRENT_LIMIT_BACKOFF_MS = 5 * 1000; // 5s
const MODEL_CAPACITY_BASE_MS = 45 * 1000; // 45s base
const MODEL_CAPACITY_JITTER_MS = 30 * 1000; // ±15s
const SERVER_ERROR_BACKOFF_MS = 20 * 1000; // 20s
@@ -26,12 +28,24 @@ const OPENROUTER_DAILY_FREE_LIMIT_PATTERN = /\bfree[-_ ]models[-_ ]per[-_ ]day\b
// before classifying explicit details; an otherwise opaque status is transient
// model capacity, while quota/rate-limit/server wording remains authoritative.
const RESOURCE_EXHAUSTED_PATTERN = /resource.?exhausted/gi;
const CONCURRENT_LIMIT_PATTERN =
// Require an actual cap signal near "concurrent". "Too many concurrent
// requests" is itself a cap signal; bare feature rejections such as
// "concurrent invocation is not supported" remain excluded.
/\btoo many\s+concurren\w*\s+(?:requests?|invocations?)\b|\bconcurren\w*\b[^\n]{0,60}\b(?:limit|quota|exceed\w*|reach\w*)\b|\b(?:limit|quota|exceed\w*|reach\w*)\b[^\n]{0,60}\bconcurren\w*\b|\bconcurren[a-z]*[-_](?:[a-z]+[_-])*(?:limit|quota|exceed\w*|reach\w*)/i;
const ACCOUNT_SCOPED_403_PATTERN =
// The bare "limit will reset" / "will reset in" phrasing also appears on
// statusless per-minute transients ("Rate limit will reset in 30 seconds"),
// so gate the reset-window alternative on account-specific wording (Devin's
// "Your limit will reset in …"); the overall/account qualifiers arm above
// already covers the rest.
/\b(?:overall|account|organization|team|workspace)\b[^\n]{0,40}\b(?:message |request )?rate.?limit\b|\byour\b[^\n]{0,30}\b(?:limit )?will reset\b/i;
/**
* Classify a rate-limit error message into a reason category.
* Priority order: explicit details in a resource-exhausted error > QUOTA
* (Antigravity "quota will reset") > MODEL_CAPACITY > QUOTA (account) >
* RATE_LIMIT > QUOTA (generic) > SERVER_ERROR > bare resource-exhausted > UNKNOWN.
* (Antigravity "quota will reset") > CONCURRENT_LIMIT > MODEL_CAPACITY >
* QUOTA (account) > RATE_LIMIT > QUOTA (generic) > SERVER_ERROR > bare resource-exhausted > UNKNOWN.
*
* Bare "resource exhausted" / "resource_exhausted" maps to MODEL_CAPACITY (transient, short wait).
* Explicit details such as "quota exceeded" retain their normal classification.
@@ -50,6 +64,10 @@ export function parseRateLimitReason(errorMessage: string): RateLimitReason {
return "QUOTA_EXHAUSTED";
}
if (CONCURRENT_LIMIT_PATTERN.test(errorMessage)) {
return "CONCURRENT_LIMIT";
}
if (lower.includes("capacity") || lower.includes("overloaded") || lower.includes("529") || lower.includes("503")) {
return "MODEL_CAPACITY_EXHAUSTED";
}
@@ -111,6 +129,8 @@ export function calculateRateLimitBackoffMs(reason: RateLimitReason): number {
return QUOTA_EXHAUSTED_BACKOFF_MS;
case "RATE_LIMIT_EXCEEDED":
return RATE_LIMIT_EXCEEDED_BACKOFF_MS;
case "CONCURRENT_LIMIT":
return CONCURRENT_LIMIT_BACKOFF_MS;
case "MODEL_CAPACITY_EXHAUSTED":
return MODEL_CAPACITY_BASE_MS + Math.random() * MODEL_CAPACITY_JITTER_MS;
case "SERVER_ERROR":
@@ -148,18 +168,37 @@ export function isUsageLimitStatus(status: number | undefined): boolean {
* 3. Body is absent or {@link isOpaqueStatusBody opaque} (just the status,
* empty JSON, HTTP framing only) → rotate conservatively: the server
* gave us nothing else to go on.
* 4. Body has content → defer to {@link parseRateLimitReason}. Only
* `QUOTA_EXHAUSTED` rotates; `RATE_LIMIT_EXCEEDED` (`Too many requests`,
* 4. Body has content → defer to {@link parseRateLimitReason}. `QUOTA_EXHAUSTED`
* rotates; for the categorical 402 billing cap a `CONCURRENT_LIMIT` body
* also rotates (the cap is concurrent-worded but the status is still an
* exhausted billing cap). `RATE_LIMIT_EXCEEDED` (`Too many requests`,
* per-minute caps), `MODEL_CAPACITY_EXHAUSTED` (`Service overloaded`),
* `SERVER_ERROR`, and `UNKNOWN` (`Please retry in 5s`) stay in the
* provider's own backoff layer so transient 429s don't burn sibling
* credentials.
*/
export function isUsageLimitOutcome(status: number | undefined, message: string | undefined): boolean {
// Concurrency caps are shed-and-backoff, not credential-rotatable — but only
// for quota-worded 429 / other statuses. HTTP 402 is categorically an
// account-billing cap, so a 402 whose body happens to mention concurrency is
// still an exhausted billing cap and must rotate; gate the exclusion on the
// status not being that categorical billing cap.
const isBillingCapStatus = status === 402;
if (isConcurrencyCapExclusion(status, message)) return false;
if (message && matchesUsageLimitText(message)) return true;
// A 403 is normally an auth failure, but several providers deliver an
// account-scoped cap with it (Devin/Codeium Connect `permission_denied`,
// GitHub Copilot). Devin's end-of-stream Connect trailer carries no HTTP
// status at all (it arrives as a `permission_denied` ValidationError), so
// accept an undefined status too — but only when the body names a cap that
// resets, never on a bare 403, which stays an auth failure.
if ((status === 403 || status === undefined) && message && isAccountScopedCapText(message)) return true;
if (!isUsageLimitStatus(status)) return false;
if (!message || isOpaqueStatusBody(message)) return true;
return parseRateLimitReason(message) === "QUOTA_EXHAUSTED";
const reason = parseRateLimitReason(message);
// For the categorical 402 billing cap a concurrency-worded body is still an
// exhausted cap (rotate); for 429 / other only QUOTA_EXHAUSTED rotates.
return reason === "QUOTA_EXHAUSTED" || (isBillingCapStatus && reason === "CONCURRENT_LIMIT");
}
/**
@@ -190,3 +229,23 @@ export function matchesUsageLimitText(errorMessage: string): boolean {
OPENROUTER_DAILY_FREE_LIMIT_PATTERN.test(errorMessage)
);
}
/**
* Account-scoped cap phrasing delivered on a 403 (or a statusless Connect
* trailer): "Reached overall message rate limit", "Your limit will reset in …".
* Kept separate from {@link matchesUsageLimitText} because the bare wording is
* ambiguous without the 403 / statusless-account context; consumed by both
* {@link isUsageLimitOutcome} (rotation decision) and `flags.ts` (Flag.UsageLimit).
*/
export function isAccountScopedCapText(message: string): boolean {
return ACCOUNT_SCOPED_403_PATTERN.test(message);
}
/**
* A concurrency cap on a non-billing status is shed-and-backoff, not
* credential-rotatable. This mirrors the exclusion in {@link isUsageLimitOutcome}
* for the 403 auth-retry entry points. A 402 remains a categorical billing cap.
*/
export function isConcurrencyCapExclusion(status: number | undefined, message: string | undefined): boolean {
return message !== undefined && parseRateLimitReason(message) === "CONCURRENT_LIMIT" && status !== 402;
}
+1 -1
View File
@@ -33,7 +33,7 @@ function isTransientTransportMessage(message: string): boolean {
export interface ProviderRetryableHooks {
/** Provider id of the failing request, used to gate provider-specific checks. */
provider?: string;
/** Provider-specific transient predicate (e.g. Copilot `model_not_supported`). */
/** Provider-specific transient predicate (e.g. Copilot model-availability 400s). */
isProviderTransient?: (error: Error) => boolean;
}
+15 -3
View File
@@ -1531,6 +1531,13 @@ async function* observeDecodedAnthropicSdkEvents(
const PROVIDER_MAX_RETRIES = 10;
/**
* Flat delay between attempts when Copilot 400s a model its own `/models`
* catalog advertises. Part of the fleet carries the model and part doesn't, so
* the retry is a reroll rather than a wait for capacity to free up.
*/
const COPILOT_MODEL_FLAP_RETRY_DELAY_MS = 400;
/**
* How long `ping` keepalives may keep extending the idle deadline without any
* semantic stream progress, as a multiple of the idle timeout. Anthropic pings
@@ -1561,8 +1568,8 @@ function shouldIgnoreAnthropicPreambleEvent(eventType: unknown): boolean {
/**
* Whether an Anthropic (or Copilot-over-Anthropic) stream error should be
* retried. The classification lives in {@link AIError.isProviderRetryableError};
* this wrapper injects the Copilot-specific `model_not_supported` transient
* check, which the error module must not import directly.
* this wrapper injects the Copilot-specific model-availability transient check,
* which the error module must not import directly.
*/
export function isProviderRetryableError(error: unknown, provider?: string): boolean {
return AIError.isProviderRetryableError(error, {
@@ -2709,7 +2716,12 @@ const streamAnthropicOnce = (
throw streamFailure;
}
providerRetryAttempt++;
const backoffDelayMs = calculateAnthropicRetryDelayMs(providerRetryAttempt - 1);
// Copilot's model-availability 400 is a per-request replica reroll, not
// upstream backpressure — the exponential curve would just add dead
// time to a coin flip that the next attempt is as likely to win.
const backoffDelayMs = AIError.isCopilotTransientModelError(streamFailure)
? COPILOT_MODEL_FLAP_RETRY_DELAY_MS
: calculateAnthropicRetryDelayMs(providerRetryAttempt - 1);
// Honor the server's retry hint (`retry-after-ms`/`retry-after`) on
// 429/529-style failures: retrying sooner than the server asked is a
// guaranteed failure that just burns the retry budget.
@@ -47,6 +47,36 @@ export function piReadPath(readPath: string, offset?: number, limit?: number): s
return count === undefined ? `${readPath}:raw:${start}-` : `${readPath}:raw:${start}+${count}`;
}
const READ_RANGE_CHUNK_RE = /^L?(\d+)(?:(\.\.|[-+])L?(\d+)?)?$/i;
function isReadRangeList(value: string): boolean {
return value.split(",").every(chunk => {
const match = READ_RANGE_CHUNK_RE.exec(chunk);
if (!match) return false;
const start = Number.parseInt(match[1]!, 10);
if (start < 1) return false;
const separator = match[2];
if (!separator) return true;
const end = match[3] ? Number.parseInt(match[3], 10) : undefined;
if (separator === "+") return end !== undefined && end >= 1;
return end === undefined || end >= start;
});
}
/**
* Whether a read path ends in an OMP line selector, including compound `raw`
* forms. Cursor uses this only to describe the operation already executed by
* the coding-agent read tool; the selector remains embedded in the path.
*/
export function piReadPathHasRange(readPath: string): boolean {
const chunks = readPath.split(":");
const last = chunks.at(-1);
if (last && isReadRangeList(last)) return true;
if (last?.toLowerCase() !== "raw") return false;
const preceding = chunks.at(-2);
return preceding !== undefined && isReadRangeList(preceding);
}
/**
* The same range as {@link piReadPath}, rendered for a transcript block rather
* than for execution.
+16 -9
View File
@@ -216,6 +216,7 @@ import {
piLimit,
piLsPath,
piReadDisplayPath,
piReadPathHasRange,
piTimeout,
} from "./cursor/exec-modern";
@@ -1322,7 +1323,7 @@ async function handleExecServerMessage(
buildReadResultFromToolResult(
args.path,
toolResult,
args.offset !== undefined || args.limit !== undefined,
args.offset !== undefined || args.limit !== undefined || piReadPathHasRange(args.path),
),
reason => buildReadRejectedResult(args.path, reason),
error => buildReadErrorResult(args.path, error),
@@ -2439,15 +2440,21 @@ function toolResultDetailBoolean(toolResult: ToolResultMessage, key: string): bo
/**
* The file's own line count, when the tool recorded one.
*
* `details.meta.truncation.totalLines` is the whole file; the flat
* `details.truncation.totalLines` counts from the window's start line and is
* deliberately not consulted here. Absent for a read that returned the file
* whole, where the payload IS the file and counting it is exact.
* Read results expose the source-wide count directly when known. Older tool
* results carry it at `details.meta.truncation.totalLines`; the flat
* `details.truncation.totalLines` counts from a window's start and is
* deliberately not consulted here.
*/
function readTotalLinesFromDetails(toolResult: ToolResultMessage): number | undefined {
if (!toolResult.details || typeof toolResult.details !== "object") return undefined;
const meta = (toolResult.details as { meta?: { truncation?: { totalLines?: unknown } } }).meta;
const totalLines = meta?.truncation?.totalLines;
const details = toolResult.details;
if (!details || typeof details !== "object") return undefined;
const direct = "totalLines" in details ? details.totalLines : undefined;
if (typeof direct === "number" && Number.isFinite(direct)) return direct;
const meta = "meta" in details ? details.meta : undefined;
if (!meta || typeof meta !== "object") return undefined;
const truncation = "truncation" in meta ? meta.truncation : undefined;
if (!truncation || typeof truncation !== "object") return undefined;
const totalLines = "totalLines" in truncation ? truncation.totalLines : undefined;
return typeof totalLines === "number" && Number.isFinite(totalLines) ? totalLines : undefined;
}
@@ -2467,7 +2474,7 @@ function buildReadResultFromToolResult(path: string, toolResult: ToolResultMessa
// whole file. Under a composed window it is the window's, and answering a
// 20-line page of a 100-line file with `total_lines: 20` tells a paginating
// server it has reached the end.
const totalLines = readTotalLinesFromDetails(toolResult) ?? (text ? text.split("\n").length : 0);
const totalLines = readTotalLinesFromDetails(toolResult) ?? (rangeApplied ? 0 : text ? text.split("\n").length : 0);
return create(ReadResultSchema, {
result: {
case: "success",
@@ -81,6 +81,7 @@ export {
piLsPath,
piReadDisplayPath,
piReadPath,
piReadPathHasRange,
piTimeout,
} from "../cursor-pi-args";
+1 -1
View File
@@ -67,7 +67,7 @@ export type MockApi = typeof MOCK_API;
export type MockContent =
| string
| { type: "text"; text: string }
| { type: "thinking"; thinking: string }
| { type: "thinking"; thinking: string; thinkingSignature?: string }
| {
type: "toolCall";
/** Optional explicit id; auto-generated when omitted. */
@@ -128,7 +128,7 @@ import { redactSensitiveInObject, transformMessages } from "./transform-messages
export interface OpenAICodexResponsesOptions extends StreamOptions {
reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
reasoningSummary?: "auto" | "concise" | "detailed" | null;
/** `reasoning.context` replay scope; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */
/** Explicit `reasoning.context` replay scope. Omitted by default so Codex applies its native request policy. */
reasoningContext?: CodexReasoningContext;
textVerbosity?: "low" | "medium" | "high";
codexMode?: boolean;
@@ -1530,7 +1530,7 @@ export async function buildTransformedCodexRequestBody(
}
const codexOptions: CodexRequestOptions = {
reasoningEffort: options?.reasoning,
reasoningSummary: options?.reasoningSummary === undefined ? "auto" : options.reasoningSummary,
reasoningSummary: options?.reasoningSummary,
reasoningContext: options?.reasoningContext,
textVerbosity: options?.textVerbosity,
include: options?.include,
@@ -33,7 +33,7 @@ export interface CodexRequestOptions {
/** User-facing effort; maps 1:1 onto the wire tier of the same name. */
reasoningEffort?: CodexCallerEffort | "none";
reasoningSummary?: ReasoningConfig["summary"] | null;
/** Explicit `reasoning.context` override; defaults to `all_turns` when unset. Gated to gpt-5.4+ Codex models (older ids reject it, so it is suppressed and `context` omitted). Note that under Responses Lite (`responsesLite`), the server strictly requires `reasoning.context` to be `all_turns`, which overrides this option and forces `all_turns`. */
/** Explicit `reasoning.context` override. Omitted by default; Responses Lite forces `all_turns` as required by that transport. */
reasoningContext?: CodexReasoningContext;
textVerbosity?: "low" | "medium" | "high";
include?: string[];
@@ -145,13 +145,12 @@ function getReasoningConfig(
const config: ReasoningConfig = {
effort: effort === "none" ? "none" : mapCodexWireEffort(model, effort),
};
// `reasoning.summary` is accepted only from gpt-5.4 onward; earlier Codex ids
// (gpt-5.1-codex, gpt-5.3-codex, gpt-5.3-codex-spark) reject it with
// "Unsupported parameter: 'reasoning.summary' is not supported with this model".
// Mirrors the all_turns gate: an explicit summary is suppressed on unsupported
// ids, letting the server skip the human-readable summary stream.
if (options.reasoningSummary !== null && supportsCodexReasoningSummary(model.id)) {
config.summary = options.reasoningSummary ?? "detailed";
if (
options.reasoningSummary !== undefined &&
options.reasoningSummary !== null &&
supportsCodexReasoningSummary(model.id)
) {
config.summary = options.reasoningSummary;
}
return config;
}
@@ -444,21 +443,14 @@ export async function transformRequestBody(
...body.reasoning,
...reasoningConfig,
};
// Default reasoning replay to `all_turns`, mirroring codex-rs; an
// explicit `reasoningContext` overrides the default. The `all_turns`
// value is only accepted from gpt-5.4 onward — earlier Codex ids
// (gpt-5.1-codex, gpt-5.3-codex, gpt-5.3-codex-spark) reject it with
// "Unsupported value: 'all_turns' is not supported with this model".
// For those, drop `context` so the server applies its `current_turn`
// default. The version gate is authoritative: even an explicit
// `all_turns` override is suppressed on unsupported models, while
// `current_turn`/`auto` (universally supported) always pass through.
// Note: Responses Lite forces `all_turns` to satisfy the transport's server invariant.
const context = responsesLite ? "all_turns" : (options.reasoningContext ?? "all_turns");
if (context === "all_turns" && !supportsAllTurnsReasoningContext(model.id)) {
delete body.reasoning.context;
} else {
body.reasoning.context = context;
// Responses Lite requires `all_turns`; the full transport leaves context to the server unless explicitly set.
const context = responsesLite ? "all_turns" : options.reasoningContext;
if (context !== undefined) {
if (context === "all_turns" && !supportsAllTurnsReasoningContext(model.id)) {
delete body.reasoning.context;
} else {
body.reasoning.context = context;
}
}
} else {
delete body.reasoning;
@@ -481,10 +473,12 @@ export async function transformRequestBody(
delete body.stream_options;
}
body.text = {
...body.text,
verbosity: options.textVerbosity || "medium",
};
if (options.textVerbosity !== undefined) {
body.text = {
...body.text,
verbosity: options.textVerbosity,
};
}
const include = Array.isArray(options.include) ? [...options.include] : [];
include.push("reasoning.encrypted_content");
@@ -11,7 +11,7 @@
*
* Activated when a {@link Model} has `transport: "pi-native"` set; the
* dispatch hook lives in `streamSimple()` (see `../stream.ts`). Used by
* containerized omp deployments (robomp slots, the swarm extension) that
* containerized omp deployments (such as robomp slots) that
* route every LLM call through a credential-holding sidecar so the slot
* itself stays credential-free.
*/
@@ -4,7 +4,7 @@
* Where the OpenAI / Anthropic / Responses route modules translate foreign
* wire shapes through pi-ai's canonical {@link Context}, this module accepts
* the canonical shape *directly* — for clients that already speak pi-ai
* (containerized omp, the swarm extension, robomp's sidecar auth-gateway).
* (containerized omp, robomp's sidecar auth-gateway).
* Skipping the wire-format → Context → wire-format round-trip cuts
* per-request CPU but, more importantly, avoids the quantization that those
* translations impose on first-class pi-ai fields (service tier, cache
+1 -1
View File
@@ -33,7 +33,7 @@ class DevinOAuthFlow extends OAuthCallbackFlow {
});
}
generateState(): string {
override generateState(): string {
return crypto.randomUUID();
}
+3 -3
View File
@@ -21,7 +21,7 @@ import { createAuthRetryKeyState, isApiKeyResolver, resolveNextAuthRetryKey } fr
import * as AIError from "./error";
import { ProviderHttpError } from "./error";
import { isInvalidatedOAuthTokenError } from "./error/auth-classify";
import { isUsageLimitOutcome } from "./error/rate-limit";
import { isConcurrencyCapExclusion, isUsageLimitOutcome } from "./error/rate-limit";
import type { BedrockOptions } from "./providers/amazon-bedrock";
import type { AnthropicOptions } from "./providers/anthropic";
import { coworkFetch } from "./providers/cowork-fetch";
@@ -995,7 +995,7 @@ function isRetryableUpstreamError(error: unknown, status: number | undefined, me
// instead of burning siblings.
if (AIError.isUsageLimit(error)) return true;
if (isInvalidatedOAuthTokenError(error)) return true;
if (status === 401 || status === 403) return true;
if (status === 401 || (status === 403 && !isConcurrencyCapExclusion(status, message))) return true;
return isUsageLimitOutcome(status, message);
}
@@ -1705,7 +1705,7 @@ function mapOptionsForApi<TApi extends Api>(
serviceTier: options?.serviceTier,
preferWebsockets: options?.preferWebsockets,
codexCompaction: options?.codexCompaction,
reasoningSummary: options?.hideThinkingSummary ? null : "detailed",
reasoningSummary: options?.hideThinkingSummary ? null : undefined,
textVerbosity: options?.textVerbosity,
});
+30 -19
View File
@@ -263,11 +263,10 @@ function buildUsageAmount(window: ParsedUsageWindow): UsageAmount {
};
}
function buildUsageStatus(usedFraction?: number, limitReached?: boolean): UsageLimit["status"] {
if (limitReached) return "exhausted";
if (usedFraction === undefined) return "unknown";
if (usedFraction >= 1) return "exhausted";
if (usedFraction >= 0.9) return "warning";
function buildUsageStatus(args: { usedFraction?: number; explicitlyAllowed: boolean }): UsageLimit["status"] {
if (args.usedFraction === undefined) return "unknown";
if (args.usedFraction >= 1) return args.explicitlyAllowed ? "warning" : "exhausted";
if (args.usedFraction >= 0.9) return "warning";
return "ok";
}
@@ -276,6 +275,8 @@ function buildUsageLimit(args: {
window: ParsedUsageWindow;
accountId?: string;
planType?: string;
allowed?: boolean;
limitReached?: boolean;
nowMs: number;
}): UsageLimit {
const usageWindow = buildUsageWindow(args.window, args.key, args.nowMs);
@@ -290,16 +291,14 @@ function buildUsageLimit(args: {
},
window: usageWindow,
amount,
// Each chat window's status reflects ONLY its own usage. The account-level
// `rate_limit.limit_reached` flag is intentionally not applied here: Codex
// returns a single shared flag for the whole account, so threading it into
// both the primary (5h) and secondary (weekly) windows marked a window with
// real headroom `exhausted` purely because a different window (or a separate
// metered feature) was at its limit, which over-blocked sibling accounts
// during credential selection. `usedFraction >= 1` already marks a window
// that is genuinely full; a real enforced limit not reflected in
// `used_percent` is caught when the live request returns usage_limit_reached.
status: buildUsageStatus(amount.usedFraction),
// The shared account-level rejection flag cannot identify which window
// is binding, but an explicit positive verdict applies to both windows.
// Preserve 100% as a warning when Codex still allows requests; live
// usage_limit_reached responses remain authoritative for blocking.
status: buildUsageStatus({
usedFraction: amount.usedFraction,
explicitlyAllowed: args.allowed === true && args.limitReached === false,
}),
};
}
function additionalLimitSlug(args: { limitName?: string; meteredFeature?: string }): string {
@@ -331,6 +330,8 @@ function buildAdditionalUsageLimit(args: {
accountId?: string;
limitName?: string;
meteredFeature?: string;
allowed?: boolean;
limitReached?: boolean;
nowMs: number;
}): UsageLimit {
const usageWindow = buildUsageWindow(args.window, args.key, args.nowMs);
@@ -348,10 +349,12 @@ function buildAdditionalUsageLimit(args: {
},
window: usageWindow,
amount,
// The additional meter exposes one account-level flag for both windows.
// Status must follow this window's own usage or a full weekly meter marks
// the shorter window exhausted and schedules a premature retry.
status: buildUsageStatus(amount.usedFraction),
// A positive meter verdict is authoritative even when the advisory
// percentage rounds to 100; negative shared verdicts remain window-local.
status: buildUsageStatus({
usedFraction: amount.usedFraction,
explicitlyAllowed: args.allowed === true && args.limitReached === false,
}),
};
}
@@ -449,6 +452,8 @@ export const openaiCodexUsageProvider: UsageProvider = {
window: parsed.primary,
accountId,
planType,
allowed: parsed.allowed,
limitReached: parsed.limitReached,
nowMs,
}),
);
@@ -460,6 +465,8 @@ export const openaiCodexUsageProvider: UsageProvider = {
window: parsed.secondary,
accountId,
planType,
allowed: parsed.allowed,
limitReached: parsed.limitReached,
nowMs,
}),
);
@@ -478,6 +485,8 @@ export const openaiCodexUsageProvider: UsageProvider = {
accountId,
limitName: extra.limitName,
meteredFeature: extra.meteredFeature,
allowed: extra.allowed,
limitReached: extra.limitReached,
nowMs,
}),
);
@@ -492,6 +501,8 @@ export const openaiCodexUsageProvider: UsageProvider = {
accountId,
limitName: extra.limitName,
meteredFeature: extra.meteredFeature,
allowed: extra.allowed,
limitReached: extra.limitReached,
nowMs,
}),
);
+6 -5
View File
@@ -95,10 +95,11 @@ export async function finalizeErrorMessage(
* Rewrite error message for GitHub Copilot request failures.
* Must run AFTER finalizeErrorMessage since it replaces the message entirely.
*
* 400 `model_not_supported` = Copilot routing rollout gap for our OAuth client.
* A preview model (gpt-5.3-codex, gpt-5.4*, ...) flaps between 200 and
* 400 because only some of Copilot's backends have the model. After the
* in-request retry exhausts, surface guidance rather than the raw error.
* 400 model-unavailable = Copilot fleet skew. A model that `/models` advertises
* (claude-sonnet-4.6, claude-opus-4.6, gpt-5.4, gpt-5.3-codex, ...)
* flaps between 200 and 400 because only part of Copilot's fleet has it
* in the integrator allowlist. After the in-request retry exhausts,
* surface guidance rather than the raw error.
* 401 = token invalid/expired → credential removal is safe, prompt re-login.
* 403 = token valid but access denied (plan, model policy, org restriction) →
* do NOT reuse the auth-failed string (which triggers credential removal).
@@ -113,7 +114,7 @@ export function rewriteCopilotError(errorMessage: string, error: unknown, provid
return `GitHub Copilot access denied (HTTP 403). Your account may not have access to this model or feature. Check your Copilot plan or model policy settings.`;
}
if (isCopilotTransientModelError(error)) {
return `GitHub Copilot rejected this model (HTTP 400 model_not_supported) after retries. This is a known intermittent rollout gap for preview models on OAuth clients other than VS Code. Try again in a few seconds, switch to a GA model (gpt-5-mini, gpt-5.2), or run this model from VS Code.`;
return `GitHub Copilot rejected this model (HTTP 400) after retries: only part of its fleet currently serves this model id, even though /models advertises it. Try again in a few seconds or switch to a model Copilot serves fleet-wide (claude-opus-4.7, claude-sonnet-4.5, gpt-4.1).`;
}
return errorMessage;
}
+24 -5
View File
@@ -7,14 +7,25 @@ import { getHeadersFromError, getRetryAfterMsFromHeaders } from "./retry-after";
// home). Re-exported here so existing `../utils/retry` importers keep working.
export { isCopilotTransientModelError };
const COPILOT_MODEL_RETRY_MAX_ATTEMPTS = 3;
// Copilot's model-availability flap is a per-request coin flip across fleet
// replicas, not backpressure. Measured per-attempt rejection rates for models
// mid-rollout reach ~70% (gpt-5.4, 2026-08-04), and a live 10-turn run needed 6
// attempts on one turn — so a small budget just pushes the failure up to the
// agent-level retry, which restarts the whole turn. Eight attempts on a flat
// delay keep the residual near 5% at p=0.7 and under 1% at p=0.5, bounded at
// ~2.8s of dead time in the pathological case.
const COPILOT_MODEL_RETRY_MAX_ATTEMPTS = 8;
// Transport blips and status-bearing failures keep the pre-flap budget: they are
// not coin flips, so a longer ramp only delays surfacing a persistent fault.
const COPILOT_GENERIC_RETRY_MAX_ATTEMPTS = 3;
const COPILOT_MODEL_RETRY_BASE_DELAY_MS = 400;
/** Longest server-requested backoff we are willing to sit out before giving up. */
const COPILOT_RETRY_AFTER_MAX_WAIT_MS = 30_000;
/**
* Wrap an initial Copilot request so transient `model_not_supported` 400s are
* retried a small number of times. No-op for non-Copilot providers.
* Wrap an initial Copilot request so transient model-availability 400s
* (`model_not_supported`, `model_not_available_for_integrator`) are retried a
* small number of times. No-op for non-Copilot providers.
*
* The callback **MUST** create a fresh in-flight request each invocation — a
* once-consumed AsyncIterable cannot be re-iterated.
@@ -38,8 +49,16 @@ export async function callWithCopilotModelRetry<T>(
if (options.signal?.aborted) throw error;
const transientModelError = isCopilotTransientModelError(error);
if (!transientModelError && !isRetryableError(error)) throw error;
if (attempt === COPILOT_MODEL_RETRY_MAX_ATTEMPTS - 1) break;
let delayMs = retryBaseDelayMs * (attempt + 1);
// Budget is per failure kind, counted over attempts already spent: the
// eight-attempt allowance only covers the cheap model-availability reroll.
const maxAttempts = transientModelError
? COPILOT_MODEL_RETRY_MAX_ATTEMPTS
: COPILOT_GENERIC_RETRY_MAX_ATTEMPTS;
if (attempt >= maxAttempts - 1) break;
// Reroll the model flap on a flat delay: a ramp only adds dead time to a
// coin flip the next attempt is equally likely to win. Generic retryable
// failures (429/5xx/transport) keep the linear backoff below.
let delayMs = transientModelError ? retryBaseDelayMs : retryBaseDelayMs * (attempt + 1);
if (!transientModelError) {
const errorStatus = status(error);
if (errorStatus !== undefined) {
+18
View File
@@ -105,4 +105,22 @@ describe("isProviderRetryableError", () => {
expect(isProviderRetryableError(err, "anthropic")).toBe(false);
expect(isProviderRetryableError(err)).toBe(false);
});
it("retries Copilot's model_not_available_for_integrator 400 from the Anthropic messages proxy", () => {
// Shape thrown by @anthropic-ai/sdk against api.githubcopilot.com/v1/messages:
// the parsed body lands on `.error` and is itself `{ error: { code } }`.
const body = {
error: {
message:
'The requested model is not available for integrator "copilot-language-server". Available models: [gpt-4.1 claude-opus-4.7]. Verify the correct Copilot-Integration-Id header is being sent.',
code: "model_not_available_for_integrator",
param: "model",
type: "invalid_request_error",
},
};
const err = new Error(`400 ${JSON.stringify(body)}`);
Object.assign(err, { status: 400, error: body });
expect(isProviderRetryableError(err, "github-copilot")).toBe(true);
expect(isProviderRetryableError(err, "anthropic")).toBe(false);
});
});
+26
View File
@@ -275,6 +275,32 @@ describe("withAuth", () => {
expect(contexts.map(ctx => ctx.lastChance)).toEqual([false, true, true, true]);
});
it("leaves a 403 concurrency cap to the transient retry layer", async () => {
const keys: string[] = [];
const contexts: ApiKeyResolveContext[] = [];
const pool = ["k0", "k1", "k2", "k3"];
let resolveIndex = 0;
const concurrencyCap = Object.assign(new Error("concurrent requests limit reached"), { status: 403 });
await expect(
withAuth(
ctx => {
contexts.push(ctx);
return ctx.error === undefined ? pool[0] : pool[++resolveIndex];
},
async key => {
keys.push(key);
throw concurrencyCap;
},
),
).rejects.toBe(concurrencyCap);
// The outer transient retry/backoff layer owns concurrency caps. The auth
// retry layer must not refresh or select a sibling credential.
expect(keys).toEqual(["k0"]);
expect(contexts.map(ctx => ctx.lastChance)).toEqual([false]);
});
it("surfaces the last 403 when every sibling is denied", async () => {
const errors = [authError(403), authError(403)];
const resolved = ["k0", "k1", "k0"];
@@ -418,6 +418,37 @@ describe("AuthStorage codex oauth ranking", () => {
expect(apiKey).toBe("api-acct-healthy");
});
test("selects an explicitly allowed 100% Team account over a rejected exhausted sibling", async () => {
if (!authStorage) throw new Error("test setup failed");
await authStorage.set("openai-codex", [
{ type: "oauth", ...createCredential("acct-exhausted", "exhausted@example.com") },
{ type: "oauth", ...createCredential("acct-team", "team@example.com") },
]);
usageByAccount.set(
"acct-exhausted",
createCodexUsageReport({
accountId: "acct-exhausted",
primary: { usedFraction: 1, resetInMs: 3 * 24 * HOUR_MS },
secondary: { usedFraction: 1, resetInMs: 3 * 24 * HOUR_MS },
metadata: { allowed: false, limitReached: true, planType: "prolite" },
}),
);
const teamReport = createCodexUsageReport({
accountId: "acct-team",
primary: { usedFraction: 0.2, resetInMs: HOUR_MS },
secondary: { usedFraction: 1, resetInMs: 6 * 24 * HOUR_MS },
metadata: { allowed: true, limitReached: false, planType: "team" },
});
const teamSecondary = teamReport.limits.find(limit => limit.id === "openai-codex:secondary");
if (!teamSecondary) throw new Error("expected Team weekly usage limit");
teamSecondary.status = "warning";
usageByAccount.set("acct-team", teamReport);
expect(await authStorage.getApiKey("openai-codex", "allowed-team-at-100-percent")).toBe("api-acct-team");
});
test("temporarily blocks only the exhausted Codex OAuth credential after a quota 429", async () => {
if (!authStorage) throw new Error("test setup failed");
@@ -2027,6 +2058,90 @@ describe("AuthStorage codex oauth ranking", () => {
expect(apiKey).toBe("api-acct-pro");
});
test("ignores plan-ineligible headroom when reporting Spark model health", async () => {
if (!authStorage) throw new Error("test setup failed");
await authStorage.set("openai-codex", [
{ type: "oauth", ...createCredential("acct-free", "free@example.com") },
{ type: "oauth", ...createCredential("acct-pro", "pro@example.com") },
]);
usageByAccount.set(
"acct-free",
addSparkUsage(
createCodexUsageReport({
accountId: "acct-free",
primary: { usedFraction: 0.05, resetInMs: 30 * 60 * 1000 },
secondary: { usedFraction: 0.05, resetInMs: 6 * 24 * 60 * 60 * 1000 },
metadata: { planType: "free", email: "free@example.com" },
}),
0.05,
0.05,
),
);
usageByAccount.set(
"acct-pro",
addSparkUsage(
createCodexUsageReport({
accountId: "acct-pro",
primary: { usedFraction: 1, resetInMs: 2 * HOUR_MS },
secondary: { usedFraction: 1, resetInMs: 6 * 24 * 60 * 60 * 1000 },
metadata: { planType: "pro", email: "pro@example.com", limitReached: true },
}),
1,
1,
),
);
const health = await authStorage.getModelUsageHealth("openai-codex", {
modelId: "gpt-5.3-codex-spark",
reserveFraction: 0.1,
});
expect(health.state).toBe("depleted");
expect(health.accounts).toHaveLength(1);
expect(health.accounts[0]?.state).toBe("depleted");
});
test("reports an all-plan-ineligible Codex pool as depleted", async () => {
if (!authStorage) throw new Error("test setup failed");
await authStorage.set("openai-codex", [
{ type: "oauth", ...createCredential("acct-free", "free@example.com") },
{ type: "oauth", ...createCredential("acct-plus", "plus@example.com") },
]);
usageByAccount.set(
"acct-free",
createCodexUsageReport({
accountId: "acct-free",
primary: { usedFraction: 0.05, resetInMs: 30 * 60 * 1000 },
secondary: { usedFraction: 0.05, resetInMs: 6 * 24 * 60 * 60 * 1000 },
metadata: { planType: "free", email: "free@example.com" },
}),
);
usageByAccount.set(
"acct-plus",
createCodexUsageReport({
accountId: "acct-plus",
primary: { usedFraction: 0.05, resetInMs: 30 * 60 * 1000 },
secondary: { usedFraction: 0.05, resetInMs: 6 * 24 * 60 * 60 * 1000 },
metadata: { planType: "plus", email: "plus@example.com" },
}),
);
const paidHealth = await authStorage.getModelUsageHealth("openai-codex", {
modelId: "gpt-5.6-sol",
reserveFraction: 0.1,
});
const proHealth = await authStorage.getModelUsageHealth("openai-codex", {
modelId: "gpt-5.3-codex-spark",
reserveFraction: 0.1,
});
expect(paidHealth.state).toBe("healthy");
expect(paidHealth.accounts).toHaveLength(1);
expect(proHealth).toEqual({ state: "depleted", accounts: [] });
});
test("routes codex spark to a single Plus account when no Pro is connected", async () => {
if (!authStorage) throw new Error("test setup failed");
+127 -11
View File
@@ -1,14 +1,29 @@
import { describe, expect, it } from "bun:test";
import { afterEach, describe, expect, it, vi } from "bun:test";
import { scheduler } from "node:timers/promises";
import { callWithCopilotModelRetry, isCopilotTransientModelError } from "@oh-my-pi/pi-ai/utils/retry";
import { isRetryableError } from "@oh-my-pi/pi-utils";
type ErrorShape = { status: number; code?: string; error?: { code?: string; message?: string }; message: string };
afterEach(() => {
vi.restoreAllMocks();
});
function copilotError({ status, code, error, message }: ErrorShape): Error {
type ErrorShape = {
status: number;
code?: string;
error?: { code?: string; message?: string } | { error: { code?: string; message?: string } };
message: string;
headers?: Record<string, string>;
};
function copilotError({ status, code, error, message, headers }: ErrorShape): Error {
const err = new Error(message);
(err as unknown as ErrorShape).status = status;
if (code !== undefined) (err as unknown as ErrorShape).code = code;
if (error !== undefined) (err as unknown as ErrorShape).error = error;
// Single sanctioned assertion point: `Error` carries no provider fields, and
// every test reads them back through the real classifier.
const shaped = err as unknown as ErrorShape;
shaped.status = status;
if (code !== undefined) shaped.code = code;
if (error !== undefined) shaped.error = error;
if (headers !== undefined) shaped.headers = headers;
return err;
}
@@ -31,6 +46,31 @@ describe("isCopilotTransientModelError", () => {
expect(isCopilotTransientModelError(err)).toBe(true);
});
it("matches 400 model_not_available_for_integrator nested two envelopes deep (Anthropic SDK shape)", () => {
// api.githubcopilot.com/v1/messages: the SDK stores the parsed body on
// `.error`, and that body is itself `{ error: { code } }`.
const err = copilotError({
status: 400,
error: {
error: {
code: "model_not_available_for_integrator",
message: 'The requested model is not available for integrator "copilot-language-server".',
},
},
message:
'400 {"error":{"message":"The requested model is not available for integrator \\"copilot-language-server\\". Available models: [gpt-4.1 claude-opus-4.7]","code":"model_not_available_for_integrator","param":"model","type":"invalid_request_error"}}',
});
expect(isCopilotTransientModelError(err)).toBe(true);
});
it("falls back to the stringified body when no envelope exposes a code", () => {
const err = copilotError({
status: 400,
message: '400 {"error":{"message":"The requested model is not available for integrator \\"x\\"."}}',
});
expect(isCopilotTransientModelError(err)).toBe(true);
});
it("does not match other 400 codes", () => {
const err = copilotError({
status: 400,
@@ -40,6 +80,13 @@ describe("isCopilotTransientModelError", () => {
expect(isCopilotTransientModelError(err)).toBe(false);
});
it("does not match 400 codes that collide with Object.prototype keys", () => {
for (const code of ["__proto__", "constructor", "toString", "hasOwnProperty"]) {
const err = copilotError({ status: 400, code, message: "bad request" });
expect(isCopilotTransientModelError(err)).toBe(false);
}
});
it("does not match 401/403/500 regardless of code", () => {
for (const status of [401, 403, 500]) {
const err = copilotError({
@@ -74,7 +121,7 @@ describe("callWithCopilotModelRetry", () => {
expect(calls).toBe(1);
});
it("retries up to 3 attempts for Copilot transient errors and eventually throws the last error", async () => {
it("retries up to 8 attempts for Copilot transient errors and eventually throws the last error", async () => {
let calls = 0;
const err = copilotError({ status: 400, code: "model_not_supported", message: "transient" });
await expect(
@@ -86,7 +133,7 @@ describe("callWithCopilotModelRetry", () => {
{ provider: "github-copilot", retryBaseDelayMs: 0 },
),
).rejects.toBe(err);
expect(calls).toBe(3);
expect(calls).toBe(8);
});
it("succeeds on the second attempt when the first is transient", async () => {
@@ -141,9 +188,7 @@ describe("callWithCopilotModelRetry", () => {
async () => {
calls += 1;
if (calls === 1) {
const err = copilotError({ status: 429, message: "rate limited" });
(err as unknown as { headers: Record<string, string> }).headers = { "retry-after": "0.01" };
throw err;
throw copilotError({ status: 429, message: "rate limited", headers: { "retry-after": "0.01" } });
}
return "ok" as const;
},
@@ -153,6 +198,21 @@ describe("callWithCopilotModelRetry", () => {
expect(calls).toBe(2);
});
it("does not stretch a persistent Retry-After 429 across the flap budget", async () => {
let calls = 0;
const err = copilotError({ status: 429, message: "rate limited", headers: { "retry-after": "0.01" } });
await expect(
callWithCopilotModelRetry(
async () => {
calls += 1;
throw err;
},
{ provider: "github-copilot", retryBaseDelayMs: 0 },
),
).rejects.toBe(err);
expect(calls).toBe(3);
});
it("still retries status-less transport blips with the linear backoff", async () => {
let calls = 0;
const result = await callWithCopilotModelRetry(
@@ -171,6 +231,62 @@ describe("callWithCopilotModelRetry", () => {
expect(calls).toBe(2);
});
it("caps persistent generic retryable failures at the pre-flap budget", async () => {
let calls = 0;
const err = new Error(
'HTTP2StreamReset fetching "https://api.example.com/x". For more information, pass `verbose: true` in the second argument to fetch()',
);
await expect(
callWithCopilotModelRetry(
async () => {
calls += 1;
throw err;
},
{ provider: "github-copilot", retryBaseDelayMs: 0 },
),
).rejects.toBe(err);
expect(calls).toBe(3);
});
it("keeps the flat delay and the larger budget scoped to model flaps", async () => {
const flatWaits: number[] = [];
const rampWaits: number[] = [];
const record = (into: number[]) => {
const spy = vi.spyOn(scheduler, "wait");
spy.mockImplementation(async (delay?: number) => {
into.push(delay ?? 0);
});
return spy;
};
record(flatWaits);
await expect(
callWithCopilotModelRetry(
async () => {
throw copilotError({ status: 400, code: "model_not_available_for_integrator", message: "flap" });
},
{ provider: "github-copilot", retryBaseDelayMs: 100 },
),
).rejects.toBeInstanceOf(Error);
vi.restoreAllMocks();
record(rampWaits);
await expect(
callWithCopilotModelRetry(
async () => {
throw new Error(
'HTTP2StreamReset fetching "https://api.example.com/x". For more information, pass `verbose: true` in the second argument to fetch()',
);
},
{ provider: "github-copilot", retryBaseDelayMs: 100 },
),
).rejects.toBeInstanceOf(Error);
vi.restoreAllMocks();
expect(flatWaits).toEqual([100, 100, 100, 100, 100, 100, 100]);
expect(rampWaits).toEqual([100, 200]);
});
it("stops retrying when the caller aborts during backoff", async () => {
const controller = new AbortController();
controller.abort();
@@ -1680,6 +1680,31 @@ describe("Cursor legacy read frame: range reporting", () => {
if (wholeAnswer.value.result.case !== "success") throw new Error(`got ${wholeAnswer.value.result.case}`);
expect(wholeAnswer.value.result.value.rangeApplied).toBe(false);
});
it("treats a path-embedded selector as ranged without reporting the slice as the file total", async () => {
const slice = Array.from({ length: 55 }, (_, index) => `line ${index + 301}`).join("\n");
const { frames } = await dispatchExec(
buildExecMessage({
case: "readArgs",
value: create(ReadArgsSchema, {
path: "/repo/plan.md:raw:301-",
toolCallId: "c-inline",
}),
}),
{
execHandlers: {
async read() {
return toolResult(slice, { details: { fileSize: 21_015 } });
},
},
},
);
const answer = soleResult(frames);
if (answer.case !== "readResult") throw new Error(`got ${answer.case}`);
if (answer.value.result.case !== "success") throw new Error(`got ${answer.value.result.case}`);
expect(answer.value.result.value.totalLines).toBe(0);
expect(answer.value.result.value.rangeApplied).toBe(true);
expect(answer.value.result.value.fileSize).toBe(21_015n);
});
it("carries the composed selector into the synthesized call", async () => {
// A bare path beside a ranged result makes the slice look like the whole
@@ -0,0 +1,125 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic";
import type { Context, Model } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { OPENCODE_HEADERS } from "@oh-my-pi/pi-catalog/wire/github-copilot";
afterEach(() => {
vi.restoreAllMocks();
});
function makeCopilotClaudeModel(): Model<"anthropic-messages"> {
return buildModel({
id: "claude-sonnet-4.6",
name: "Claude Sonnet 4.6",
api: "anthropic-messages",
provider: "github-copilot",
baseUrl: "https://api.githubcopilot.com",
headers: { ...OPENCODE_HEADERS },
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128_000,
maxTokens: 16_000,
});
}
const testContext: Context = {
messages: [{ role: "user", content: "hello", timestamp: Date.now() }],
};
/**
* Verbatim body served by `api.githubcopilot.com/v1/messages` when the request
* lands on a fleet replica whose integrator allowlist predates the model, even
* though `/models` on the same host advertises it.
*/
const FLEET_SKEW_BODY = {
error: {
message:
'The requested model is not available for integrator "copilot-language-server". Available models: [gpt-4.1 claude-opus-4.7 claude-sonnet-4.5]. Verify the correct Copilot-Integration-Id header is being sent.',
code: "model_not_available_for_integrator",
param: "model",
type: "invalid_request_error",
},
};
const SSE_EVENTS = [
{
type: "message_start",
message: {
id: "msg_fleet",
type: "message",
role: "assistant",
model: "claude-sonnet-4.6",
content: [],
stop_reason: null,
stop_sequence: null,
usage: { input_tokens: 7, output_tokens: 0 },
},
},
{ type: "content_block_start", index: 0, content_block: { type: "text", text: "" } },
{ type: "content_block_delta", index: 0, delta: { type: "text_delta", text: "second try" } },
{ type: "content_block_stop", index: 0 },
{
type: "message_delta",
delta: { stop_reason: "end_turn", stop_sequence: null },
usage: { output_tokens: 3 },
},
{ type: "message_stop" },
];
function sseResponse(): Response {
const body = `${SSE_EVENTS.map(event => `event: ${event.type}\ndata: ${JSON.stringify(event)}\n`).join("\n")}\n`;
return new Response(body, { status: 200, headers: { "Content-Type": "text/event-stream" } });
}
describe("GitHub Copilot Anthropic fleet skew", () => {
it("retries the model-availability 400 and completes on the next replica", async () => {
let attempts = 0;
const fetchMock = vi.fn(async () => {
attempts += 1;
if (attempts === 1) {
return new Response(JSON.stringify(FLEET_SKEW_BODY), {
status: 400,
headers: { "Content-Type": "application/json" },
});
}
return sseResponse();
});
const result = await streamAnthropic(makeCopilotClaudeModel(), testContext, {
apiKey: "ghu_test_copilot_token",
fetch: fetchMock as unknown as typeof fetch,
providerRetryWait: async () => {},
}).result();
expect(attempts).toBe(2);
expect(result.stopReason).toBe("stop");
expect(result.errorMessage).toBeUndefined();
expect(result.content).toMatchObject([{ type: "text", text: "second try" }]);
// Billing is per user prompt, not per wire attempt: a turn that burned an
// extra gateway-rejected attempt must still report one premium request.
expect(result.usage.premiumRequests).toBe(1);
});
it("surfaces fleet-skew guidance once every retry lands on a stale replica", async () => {
const fetchMock = vi.fn(
async () =>
new Response(JSON.stringify(FLEET_SKEW_BODY), {
status: 400,
headers: { "Content-Type": "application/json" },
}),
);
const result = await streamAnthropic(makeCopilotClaudeModel(), testContext, {
apiKey: "ghu_test_copilot_token",
fetch: fetchMock as unknown as typeof fetch,
providerRetryWait: async () => {},
}).result();
expect(result.stopReason).toBe("error");
expect(result.errorMessage).toContain("only part of its fleet");
// Every attempt is a fresh wire request, not a replayed promise.
expect(fetchMock.mock.calls.length).toBeGreaterThan(1);
});
});
+10 -8
View File
@@ -33,14 +33,16 @@ describe("rewriteCopilotError", () => {
expect(result).not.toContain("/login github-copilot");
});
it("rewrites 400 model_not_supported with rollout-gap guidance", () => {
const err = new Error("400 The requested model is not supported.");
(err as unknown as { status: number; code: string }).status = 400;
(err as unknown as { status: number; code: string }).code = "model_not_supported";
const result = rewriteCopilotError("original", err, "github-copilot");
expect(result).toContain("HTTP 400 model_not_supported");
expect(result).toContain("rollout gap");
expect(result).not.toContain("authentication failed");
it("rewrites 400 model-unavailable codes with fleet-skew guidance", () => {
for (const code of ["model_not_supported", "model_not_available_for_integrator"]) {
const err = new Error("400 The requested model is not available.");
(err as unknown as { status: number; code: string }).status = 400;
(err as unknown as { status: number; code: string }).code = code;
const result = rewriteCopilotError("original", err, "github-copilot");
expect(result).toContain("HTTP 400");
expect(result).toContain("only part of its fleet");
expect(result).not.toContain("authentication failed");
}
});
it("leaves non-copilot 400 model_not_supported untouched", () => {
@@ -128,71 +128,58 @@ function createCodexFetchMock(sse: string, onRequest: (captured: CapturedCodexRe
}) as FetchImpl;
}
describe("openai-codex reasoning.context", () => {
it("defaults to all_turns on gpt-5.4+ models and forwards explicit overrides", async () => {
const model = createCodexModel("gpt-5.4");
describe("openai-codex optional response controls", () => {
it("omits optional controls on full requests and forwards explicit controls", async () => {
const model = createCodexModel("gpt-5.5");
const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" });
expect(defaulted.reasoning?.context).toBe("all_turns");
expect(defaulted.reasoning).toEqual({ effort: "medium" });
expect("summary" in (defaulted.reasoning ?? {})).toBe(false);
expect("context" in (defaulted.reasoning ?? {})).toBe(false);
expect("text" in defaulted).toBe(false);
expect("stream_options" in defaulted).toBe(false);
const explicit = await transformRequestBody({ model: model.id }, model, {
reasoningEffort: "medium",
reasoningContext: "current_turn",
reasoningSummary: "concise",
reasoningContext: "all_turns",
textVerbosity: "low",
});
expect(explicit.reasoning?.context).toBe("current_turn");
expect(explicit.reasoning).toEqual({
effort: "medium",
summary: "concise",
context: "all_turns",
});
expect(explicit.text).toEqual({ verbosity: "low" });
expect(explicit.stream_options).toEqual({ reasoning_summary_delivery: "sequential_cutoff" });
});
it("keeps the all_turns default for the lite transport on supported models", async () => {
it("omits reasoning.summary when explicitly suppressed", async () => {
const model = createCodexModel("gpt-5.5");
const lite = await transformRequestBody({ model: model.id }, model, {
const suppressed = await transformRequestBody({ model: model.id }, model, {
reasoningEffort: "medium",
responsesLite: true,
reasoningSummary: null,
});
expect(lite.reasoning?.context).toBe("all_turns");
const overridden = await transformRequestBody({ model: model.id }, model, {
reasoningEffort: "medium",
responsesLite: true,
reasoningContext: "auto",
});
expect(overridden.reasoning?.context).toBe("all_turns");
expect(suppressed.reasoning).toEqual({ effort: "medium" });
expect("summary" in (suppressed.reasoning ?? {})).toBe(false);
expect("stream_options" in suppressed).toBe(false);
});
it("enforces reasoning.context to be all_turns for the lite transport even when effort is unset or none", async () => {
it("forces reasoning.context to all_turns for Responses Lite", async () => {
const model = createCodexModel("gpt-5.5");
// Case 1: reasoningEffort is undefined (missing effort)
const missingEffort = await transformRequestBody({ model: model.id }, model, {
responsesLite: true,
});
expect(missingEffort.reasoning?.context).toBe("all_turns");
expect(missingEffort.reasoning?.effort).toBeUndefined();
expect(missingEffort.reasoning).toEqual({ context: "all_turns" });
// Case 2: reasoningEffort is explicitly "none" (effort set to off)
const noneEffort = await transformRequestBody({ model: model.id }, model, {
reasoningEffort: "none",
responsesLite: true,
});
expect(noneEffort.reasoning?.context).toBe("all_turns");
expect(noneEffort.reasoning?.effort).toBe("none");
// Case 3: Conflicting explicit reasoningContext with missing effort under Lite
const conflictingUnsetEffort = await transformRequestBody({ model: model.id }, model, {
responsesLite: true,
reasoningContext: "current_turn",
});
expect(conflictingUnsetEffort.reasoning?.context).toBe("all_turns");
expect(noneEffort.reasoning).toEqual({ effort: "none", context: "all_turns" });
// Case 4: Conflicting explicit reasoningContext with "none" effort under Lite
const conflictingNoneEffort = await transformRequestBody({ model: model.id }, model, {
reasoningEffort: "none",
responsesLite: true,
reasoningContext: "current_turn",
});
expect(conflictingNoneEffort.reasoning?.context).toBe("all_turns");
// Case 5: responsesLite is false and reasoningEffort is undefined (regular request with no effort)
const plainRequest = await transformRequestBody({ model: model.id }, model, {
responsesLite: false,
});
@@ -202,73 +189,35 @@ describe("openai-codex reasoning.context", () => {
// gpt-5.1-codex / gpt-5.3-codex / gpt-5.3-codex-spark reject `all_turns`
// ("Unsupported value: 'all_turns' is not supported with this model").
it.each(["gpt-5.1-codex", "gpt-5.3-codex", "gpt-5.3-codex-spark"])(
"omits the all_turns default for pre-5.4 model %s",
"omits unsupported all_turns context for pre-5.4 model %s",
async modelId => {
const model = createCodexModel(modelId);
const forced = await transformRequestBody({ model: model.id }, model, {
reasoningEffort: "medium",
reasoningContext: "all_turns",
});
expect(forced.reasoning).toEqual({ effort: "medium" });
const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" });
expect(defaulted.reasoning).toBeDefined();
expect(defaulted.reasoning?.context).toBeUndefined();
expect("context" in (defaulted.reasoning ?? {})).toBe(false);
// A supported override (current_turn/auto) is still honored.
const overridden = await transformRequestBody({ model: model.id }, model, {
reasoningEffort: "medium",
reasoningContext: "current_turn",
});
expect(overridden.reasoning?.context).toBe("current_turn");
expect(overridden.reasoning).toEqual({ effort: "medium", context: "current_turn" });
},
);
it("suppresses an explicit all_turns override on a pre-5.4 model", async () => {
const model = createCodexModel("gpt-5.3-codex-spark");
const forced = await transformRequestBody({ model: model.id }, model, {
reasoningEffort: "medium",
reasoningContext: "all_turns",
});
expect(forced.reasoning).toBeDefined();
expect(forced.reasoning?.context).toBeUndefined();
});
});
describe("openai-codex reasoning.summary", () => {
it("sends summary on gpt-5.4+ models and honors explicit levels", async () => {
const model = createCodexModel("gpt-5.4");
const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" });
expect(defaulted.reasoning?.summary).toBe("detailed");
const explicit = await transformRequestBody({ model: model.id }, model, {
reasoningEffort: "medium",
reasoningSummary: "concise",
});
expect(explicit.reasoning?.summary).toBe("concise");
const suppressed = await transformRequestBody({ model: model.id }, model, {
reasoningEffort: "medium",
reasoningSummary: null,
});
expect("summary" in (suppressed.reasoning ?? {})).toBe(false);
});
// gpt-5.1-codex / gpt-5.3-codex / gpt-5.3-codex-spark reject `reasoning.summary`
// ("Unsupported parameter: 'reasoning.summary' is not supported with this model").
it.each(["gpt-5.1-codex", "gpt-5.3-codex", "gpt-5.3-codex-spark"])(
"omits reasoning.summary for pre-5.4 model %s",
async modelId => {
const model = createCodexModel(modelId);
const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" });
expect(defaulted.reasoning).toBeDefined();
expect("summary" in (defaulted.reasoning ?? {})).toBe(false);
// Even an explicit summary level is suppressed on unsupported ids.
const forced = await transformRequestBody({ model: model.id }, model, {
reasoningEffort: "medium",
reasoningSummary: "detailed",
});
expect("summary" in (forced.reasoning ?? {})).toBe(false);
expect(forced.reasoning).toEqual({ effort: "medium" });
expect("stream_options" in forced).toBe(false);
},
);
});
@@ -816,7 +765,10 @@ describe("openai-codex concurrent reasoning summaries", () => {
it("sends stream_options only when a summary is requested and supported", async () => {
const terra = createCodexModel("gpt-5.6-terra");
const withSummary = await transformRequestBody({ model: terra.id }, terra, { reasoningEffort: "medium" });
const withSummary = await transformRequestBody({ model: terra.id }, terra, {
reasoningEffort: "medium",
reasoningSummary: "detailed",
});
expect(withSummary.stream_options).toEqual({ reasoning_summary_delivery: "sequential_cutoff" });
expect(withSummary.reasoning?.summary).toBe("detailed");
@@ -830,7 +782,10 @@ describe("openai-codex concurrent reasoning summaries", () => {
expect(noReasoning.stream_options).toBeUndefined();
const legacy = createCodexModel("gpt-5.1-codex");
const unsupported = await transformRequestBody({ model: legacy.id }, legacy, { reasoningEffort: "medium" });
const unsupported = await transformRequestBody({ model: legacy.id }, legacy, {
reasoningEffort: "medium",
reasoningSummary: "detailed",
});
expect(unsupported.stream_options).toBeUndefined();
});
@@ -889,6 +844,7 @@ describe("openai-codex concurrent reasoning summaries", () => {
apiKey: createCodexTestToken(),
fetch: fetchMock,
reasoning: "medium",
reasoningSummary: "detailed",
});
const thinkingDeltas: string[] = [];
for await (const event of stream) {
@@ -1060,6 +1016,7 @@ describe("openai-codex concurrent reasoning summaries", () => {
apiKey: createCodexTestToken(),
fetch: fetchMock,
reasoning: "medium",
reasoningSummary: "detailed",
});
const thinkingDeltas: string[] = [];
for await (const event of stream) {
@@ -1215,6 +1172,7 @@ describe("openai-codex concurrent reasoning summaries", () => {
apiKey: createCodexTestToken(),
fetch: fetchMock,
reasoning: "medium",
reasoningSummary: "detailed",
});
const deltasByBlock = new Map<number, string>();
for await (const event of stream) {
+62 -35
View File
@@ -19,6 +19,7 @@ import type {
} from "@oh-my-pi/pi-ai/types";
import { __resetProxyCache } from "@oh-my-pi/pi-ai/utils/proxy";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { Effort } from "@oh-my-pi/pi-catalog/effort";
import * as piUtils from "@oh-my-pi/pi-utils";
import { withEnv } from "./helpers";
@@ -409,7 +410,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
this.emitCodexResponse({ messageId: "msg_opaque", responseId: "resp_opaque", text: "pong" });
}
}
@@ -501,6 +502,32 @@ describe("openai-codex streaming", () => {
expect(capturedText).toEqual({ verbosity: "low" });
});
it("omits optional response controls from default SimpleStreamOptions", async () => {
const tempDir = TempDir.createSync("@pi-codex-stream-");
setAgentDir(tempDir.path());
const token = createCodexTestToken();
const context = createCodexTestContext();
const model = { ...createCodexTestModel("https://chatgpt.com/backend-api"), preferWebsockets: false };
let capturedBody: Record<string, unknown> | undefined;
const fetchMock: FetchImpl = async (_input, init) => {
capturedBody = JSON.parse(decodeCodexRequestBody(init?.body)) as Record<string, unknown>;
return new Response(createCompletedCodexSse("Hello"), {
status: 200,
headers: { "content-type": "text/event-stream" },
});
};
const result = await streamSimple(model, context, {
apiKey: token,
fetch: fetchMock,
reasoning: Effort.Medium,
}).result();
expect(result.stopReason).toBe("stop");
expect(capturedBody?.reasoning).toEqual({ effort: "medium" });
expect(capturedBody?.text).toBeUndefined();
});
async function runCodexSseEvents(events: unknown[]) {
const token = createCodexTestToken();
const context = createCodexTestContext();
@@ -1316,7 +1343,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
const added = encodeWebSocketMessage({
type: "response.output_item.added",
item: { type: "message", id: "msg_ws", role: "assistant", status: "in_progress", content: [] },
@@ -1374,7 +1401,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
this.emitCodexResponse({ messageId: "msg_obs", responseId: "resp_obs", text: "Observed" });
}
}
@@ -1431,7 +1458,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
this.sendJson({
type: "response.done",
response: {
@@ -1484,7 +1511,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
this.sendJson({
type: "response.done",
response: {
@@ -1638,7 +1665,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(data: string): void {
override send(data: string): void {
sentRequests.push(JSON.parse(data) as Record<string, unknown>);
this.emitCodexResponse({ messageId: "msg_lite", responseId: "resp_lite", text: "Hi" });
}
@@ -2876,7 +2903,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(data: string): void {
override send(data: string): void {
websocketRequestCount += 1;
const body: unknown = JSON.parse(data);
if (websocketRequestCount === 1) {
@@ -3064,7 +3091,7 @@ describe("openai-codex streaming", () => {
});
}
send(_data: string): void {
override send(_data: string): void {
websocketRequestCount += 1;
this.emitCodexResponse({
messageId: `msg_pre_turn_${websocketRequestCount}`,
@@ -3184,7 +3211,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(data: string): void {
override send(data: string): void {
sentRequests.push(JSON.parse(data) as Record<string, unknown>);
this.sendJson({
type: "response.output_item.added",
@@ -3269,7 +3296,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(data: string): void {
override send(data: string): void {
sentRequests.push(JSON.parse(data) as Record<string, unknown>);
const responseIndex = sentRequests.length;
this.emitCodexResponse({
@@ -3409,7 +3436,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(data: string): void {
override send(data: string): void {
sentRequests.push(JSON.parse(data) as Record<string, unknown>);
if (sentRequests.length === 1) {
this.sendJson({
@@ -3508,7 +3535,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
this.sendJson({
type: "response.completed",
response: {
@@ -3564,7 +3591,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
this.#sendCount += 1;
if (this.#sendCount === 1) {
this.emitCodexResponse({
@@ -3653,7 +3680,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(data: string): void {
override send(data: string): void {
sentRequests.push(JSON.parse(data) as Record<string, unknown>);
const responseIndex = sentRequests.length;
this.emitCodexResponse({
@@ -3770,7 +3797,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(data: string): void {
override send(data: string): void {
const request = JSON.parse(data) as Record<string, unknown>;
sentRequests.push(request);
const requestIndex = sentRequests.length;
@@ -3878,7 +3905,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(data: string): void {
override send(data: string): void {
const request = JSON.parse(data) as Record<string, unknown>;
sentRequests.push(request);
const requestIndex = sentRequests.length;
@@ -3996,7 +4023,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
this.emitCodexResponse({ messageId: "msg_v2", responseId: "resp_v2", text: "Hello v2" });
}
}
@@ -4053,7 +4080,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
sendCount += 1;
}
}
@@ -4113,7 +4140,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
sendCount += 1;
this.sendJson({
type: "response.output_item.added",
@@ -4143,7 +4170,7 @@ describe("openai-codex streaming", () => {
}, 2);
}
close(): void {
override close(): void {
if (interval) clearInterval(interval);
super.close();
}
@@ -4190,7 +4217,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
sendCount += 1;
this.sendJson({
type: "response.output_item.added",
@@ -4213,7 +4240,7 @@ describe("openai-codex streaming", () => {
}
}
close(): void {
override close(): void {
closeCount += 1;
super.close();
}
@@ -4257,7 +4284,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
if (this.#index === 0) {
// First attempt: a function call whose arguments are only whitespace.
// A completed reasoning item lands in nativeOutputItems before the
@@ -4313,7 +4340,7 @@ describe("openai-codex streaming", () => {
});
}
close(): void {
override close(): void {
closeCount += 1;
super.close();
}
@@ -4384,7 +4411,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
sendCount += 1;
this.sendJson({
type: "response.output_item.added",
@@ -4436,7 +4463,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
// Every frame lands in the connection queue synchronously, before the
// consumer microtask drains any of them; the close event used to wipe
// the queued terminal event and turn success into a transport error.
@@ -4479,7 +4506,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
this.sendJson({
type: "response.output_item.added",
item: { type: "function_call", id: "fc_limit", call_id: "call_limit", name: "todo", arguments: "" },
@@ -4534,13 +4561,13 @@ describe("openai-codex streaming", () => {
this.emit("open", new Event("open"));
}
close(): void {
override close(): void {
const wasPending = this.readyState === MockWebSocket.CONNECTING;
super.close();
if (wasPending) this.emit("close", { code: 1000 } as unknown as Event);
}
send(): void {
override send(): void {
this.emitCodexResponse({ messageId: "msg_join", responseId: "resp_join", text: "Joined" });
}
}
@@ -4597,7 +4624,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
sendCount += 1;
this.sendJson({
type: "response.output_item.added",
@@ -4669,7 +4696,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(data: string): void {
override send(data: string): void {
const request = JSON.parse(data) as { type?: string };
const requestType = typeof request.type === "string" ? request.type : "";
sentTypesByConnection[this.#connectionIndex]?.push(requestType);
@@ -4796,7 +4823,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(): void {
override send(): void {
this.sendJson({
type: "response.output_item.added",
item: {
@@ -4886,7 +4913,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(data: string): void {
override send(data: string): void {
this.#sendCount += 1;
const request = JSON.parse(data) as { type?: string };
requestTypes.push(typeof request.type === "string" ? request.type : "");
@@ -4979,7 +5006,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(data: string): void {
override send(data: string): void {
sendCount += 1;
const request = JSON.parse(data) as Record<string, unknown>;
expect(typeof request.type).toBe("string");
@@ -5272,7 +5299,7 @@ describe("openai-codex streaming", () => {
this.scheduleOpen();
}
send(_data: string): void {
override send(_data: string): void {
sendCount += 1;
if (sendCount === 1) {
this.emitCodexResponse({
@@ -67,6 +67,24 @@ describe("openai-codex usage parser", () => {
expect(main?.[0].amount.usedFraction).toBeCloseTo(0.04, 5);
});
it("keeps an explicitly allowed Team window usable at 100% reported usage", async () => {
const payload = makePayload();
payload.plan_type = "team";
payload.rate_limit.secondary_window.used_percent = 100;
const report = await openaiCodexUsageProvider.fetchUsage(
{
provider: "openai-codex",
credential: { type: "oauth", accessToken: accessTokenFixture, accountId: "acct-1", email: "u@example.com" },
},
{ fetch: fakeFetch(payload) },
);
const secondary = report?.limits.find(limit => limit.id === "openai-codex:secondary");
expect(secondary?.amount.usedFraction).toBe(1);
expect(secondary?.status).toBe("warning");
expect(report?.metadata).toMatchObject({ planType: "team", allowed: true, limitReached: false });
});
it("surfaces additional_rate_limits as spark UsageLimit entries the widget can detect", async () => {
const report = await openaiCodexUsageProvider.fetchUsage(
{
+107 -1
View File
@@ -1,8 +1,9 @@
import { describe, expect, it } from "bun:test";
import { ProviderHttpError } from "@oh-my-pi/pi-ai/error";
import { isUsageLimit } from "@oh-my-pi/pi-ai/error/flags";
import { classify, Flag, is, isUsageLimit, retriable } from "@oh-my-pi/pi-ai/error/flags";
import {
calculateRateLimitBackoffMs,
isConcurrencyCapExclusion,
isUsageLimitOutcome,
isUsageLimitStatus,
parseRateLimitReason,
@@ -58,6 +59,30 @@ describe("parseRateLimitReason", () => {
expect(parseRateLimitReason("Requests per minute limit reached")).toBe("RATE_LIMIT_EXCEEDED");
});
it("classifies concurrent request caps separately from rate limits and quota exhaustion", () => {
expect(parseRateLimitReason("Number of concurrent requests exceeded")).toBe("CONCURRENT_LIMIT");
expect(parseRateLimitReason("Maximum concurrent invocation limit reached")).toBe("CONCURRENT_LIMIT");
expect(parseRateLimitReason("concurrent_limit_exceeded")).toBe("CONCURRENT_LIMIT");
expect(parseRateLimitReason("concurrent_requests_limit_reached")).toBe("CONCURRENT_LIMIT");
expect(parseRateLimitReason("concurrency_quota_exceeded")).toBe("CONCURRENT_LIMIT");
expect(parseRateLimitReason("Too many concurrent requests")).toBe("CONCURRENT_LIMIT");
expect(parseRateLimitReason("Too many concurrent invocations")).toBe("CONCURRENT_LIMIT");
expect(parseRateLimitReason("Rate limit reached for gpt-4o")).toBe("RATE_LIMIT_EXCEEDED");
expect(parseRateLimitReason("Your quota will reset at 07-28")).toBe("QUOTA_EXHAUSTED");
});
// Deterministic 4xx feature rejections worded with bare concurrency nouns
// ("concurrent request/invocation is not supported") must not classify as a
// concurrency cap — doing so would set Flag.Transient and retry the rejection
// instead of surfacing it. A cap needs an explicit limit/quota/exceeded/reached
// signal near "concurrent".
it("does not classify bare concurrency feature rejections as CONCURRENT_LIMIT", () => {
expect(parseRateLimitReason("Concurrent invocation is not supported")).not.toBe("CONCURRENT_LIMIT");
expect(parseRateLimitReason("Only one concurrent request is supported")).not.toBe("CONCURRENT_LIMIT");
// The deterministic rejection must surface as a hard error, not be retried.
expect(is(classify("Concurrent invocation is not supported"), Flag.Transient)).toBe(false);
});
it("classifies overloaded 529 as MODEL_CAPACITY_EXHAUSTED", () => {
expect(parseRateLimitReason("Service overloaded 529")).toBe("MODEL_CAPACITY_EXHAUSTED");
});
@@ -266,6 +291,30 @@ describe("isUsageLimitOutcome", () => {
expect(isUsageLimitOutcome(429, message)).toBe(true);
});
it("rotates only account-scoped cap 403s and statusless trailers", () => {
const devinTrailer =
"Devin stream error permission_denied: Reached overall message rate limit. Please try again later. Your limit will reset in 13 minutes.";
// HTTP 403 with the account-scoped body rotates.
expect(isUsageLimitOutcome(403, devinTrailer)).toBe(true);
// Devin's Connect trailer carries no HTTP status (a permission_denied
// ValidationError), so it must rotate on an undefined status too —
// otherwise the exhausted credential is retried as a transient failure.
expect(isUsageLimitOutcome(undefined, devinTrailer)).toBe(true);
expect(isUsageLimit(devinTrailer)).toBe(true);
expect(isUsageLimitOutcome(403, "Forbidden")).toBe(false);
});
// A statusless per-minute reset-window transient ("Rate limit will reset in
// 30 seconds") is ordinary throttling (RATE_LIMIT_EXCEEDED), not an account
// usage cap. The reset-window alternative is gated on account scope so it stays
// in the backoff lane instead of rotating the credential.
it("does not rotate on a statusless per-minute reset-window transient", () => {
const message = "Rate limit will reset in 30 seconds";
expect(parseRateLimitReason(message)).toBe("RATE_LIMIT_EXCEEDED");
expect(isUsageLimitOutcome(undefined, message)).toBe(false);
expect(isUsageLimit(message)).toBe(false);
});
it("rotates on xAI Grok Build 402 usage-balance exhaustion regardless of status", () => {
const message = "402 Grok Build usage balance exhausted";
expect(isUsageLimitOutcome(402, message)).toBe(true);
@@ -285,6 +334,59 @@ describe("isUsageLimitOutcome", () => {
expect(isUsageLimitOutcome(401, "Invalid API key")).toBe(false);
expect(isUsageLimitOutcome(400, "invalid_request_error: model unsupported")).toBe(false);
});
// Vertex returns "Online prediction concurrent requests quota exceeded" for a
// concurrent-request cap. The generic USAGE_LIMIT_PATTERN matches
// `quota.?exceeded`, but this is a concurrency cap (5s backoff, no rotation),
// not account quota exhaustion. CONCURRENT_LIMIT must take precedence so the
// credential is not burned.
it("does not rotate on Vertex quota-worded concurrency caps", () => {
const message = "Online prediction concurrent requests quota exceeded";
expect(parseRateLimitReason(message)).toBe("CONCURRENT_LIMIT");
expect(isUsageLimitOutcome(429, message)).toBe(false);
expect(isUsageLimit(message)).toBe(false);
});
it("excludes non-billing concurrency caps from credential rotation", () => {
const message = "concurrent requests limit reached";
expect(isConcurrencyCapExclusion(403, message)).toBe(true);
expect(isConcurrencyCapExclusion(undefined, message)).toBe(true);
expect(isConcurrencyCapExclusion(402, message)).toBe(false);
expect(isConcurrencyCapExclusion(403, "Forbidden")).toBe(false);
const classified = classify(new ProviderHttpError(message, 403));
expect(is(classified, Flag.AuthFailed)).toBe(false);
expect(is(classified, Flag.Transient)).toBe(true);
});
// The same bare concurrency wording can reach turn recovery without a
// preserved HTTP status (Vertex/Bedrock paths that bypass API-key
// resolution). The body misses TRANSIENT_TRANSPORT_PATTERN, so without an
// explicit Flag.Transient the temporary cap classifies as terminal and is
// never retried. It must stay shed-and-backoff (transient/retriable).
it("keeps statusless concurrency caps transient and retriable", () => {
const message = "Online prediction concurrent requests quota exceeded";
const id = classify(message);
expect(is(id, Flag.Transient)).toBe(true);
expect(retriable(id)).toBe(true);
});
// HTTP 402 is categorically an account-billing cap, so a 402 whose body is
// worded as a concurrency cap still rotates — the billing-cap status wins
// over the concurrency exclusion. The identical concurrency wording on a
// quota-worded 429 stays non-rotatable (5s backoff). This pins the
// 402-billing-cap > concurrency-exclusion precedence in both the rotation
// decision (isUsageLimitOutcome) and the Flag.UsageLimit classification
// (isUsageLimit).
it("rotates on 402 concurrency-worded billing caps but not 429 concurrency caps", () => {
const message = "concurrent requests limit reached";
expect(parseRateLimitReason(message)).toBe("CONCURRENT_LIMIT");
// 402 billing cap wins: rotate.
expect(isUsageLimitOutcome(402, message)).toBe(true);
expect(isUsageLimit(Object.assign(new Error(message), { status: 402 }))).toBe(true);
// 429 concurrency cap: shed-and-backoff, do not rotate.
expect(isUsageLimitOutcome(429, message)).toBe(false);
expect(isUsageLimit(Object.assign(new Error(message), { status: 429 }))).toBe(false);
});
});
describe("calculateRateLimitBackoffMs", () => {
@@ -295,4 +397,8 @@ describe("calculateRateLimitBackoffMs", () => {
expect(ms).toBeLessThanOrEqual(75_000);
}
});
it("returns a short backoff for CONCURRENT_LIMIT", () => {
expect(calculateRateLimitBackoffMs("CONCURRENT_LIMIT")).toBe(5_000);
});
});
@@ -120,6 +120,41 @@ describe("streamSimple resolver auth retry", () => {
expect((contexts[1]!.error as { status?: number }).status).toBe(401);
});
it("surfaces a 403 concurrency cap for transient backoff without rotating credentials", async () => {
const keys: unknown[] = [];
const contexts: ApiKeyResolveContext[] = [];
const concurrencyCap = Object.assign(new Error("concurrent requests limit reached"), { status: 403 });
registerCustomApi(
API,
(_model: Model<Api>, _context: Context, options?: SimpleStreamOptions) => {
pushKey(keys, options);
const stream = new AssistantMessageEventStream();
queueMicrotask(() => stream.fail(concurrencyCap));
return stream;
},
SOURCE_ID,
);
const stream = streamSimple(model(), context, {
apiKey: async ctx => {
contexts.push(ctx);
return ctx.error === undefined ? "old-key" : ctx.lastChance ? "sibling-key" : "refresh-key";
},
});
await expect(
(async () => {
for await (const _event of stream) {
// drain
}
})(),
).rejects.toBe(concurrencyCap);
expect(keys).toEqual(["old-key"]);
expect(contexts.map(ctx => ({ lastChance: ctx.lastChance, hasError: ctx.error !== undefined }))).toEqual([
{ lastChance: false, hasError: false },
]);
});
it("buffers the start event and retries on a 401 error event before content", async () => {
const keys: unknown[] = [];
const eventTypes: string[] = [];
+6
View File
@@ -2,6 +2,12 @@
## [Unreleased]
## [17.2.9] - 2026-08-05
### Fixed
- Fixed Amazon Bedrock catalog generation omitting AWS GovCloud `us-gov.*` Claude inference-profile IDs, so selectors like `amazon-bedrock/us-gov.anthropic.claude-sonnet-4-5-…` resolve instead of failing model lookup (or misrouting commercial `us.*` geos onto `us-east-1` with GovCloud credentials).
## [17.2.7] - 2026-08-03
### Fixed
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-catalog",
"version": "17.2.8",
"version": "17.2.9",
"description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+325 -1
View File
@@ -10488,6 +10488,330 @@
"contextWindow": 262000,
"maxTokens": 262000
},
"us-gov.anthropic.claude-fable-5": {
"id": "us-gov.anthropic.claude-fable-5",
"name": "Claude Fable 5 (GovCloud)",
"api": "bedrock-converse-stream",
"provider": "amazon-bedrock",
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 10,
"output": 50,
"cacheRead": 1,
"cacheWrite": 12.5
},
"contextWindow": 1000000,
"maxTokens": 128000,
"thinking": {
"mode": "anthropic-adaptive",
"efforts": [
"low",
"medium",
"high",
"max"
],
"supportsDisplay": true
}
},
"us-gov.anthropic.claude-haiku-4-5-20251001-v1:0": {
"id": "us-gov.anthropic.claude-haiku-4-5-20251001-v1:0",
"name": "Claude Haiku 4.5 (GovCloud)",
"api": "bedrock-converse-stream",
"provider": "amazon-bedrock",
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 1,
"output": 5,
"cacheRead": 0.1,
"cacheWrite": 1.25
},
"contextWindow": 200000,
"maxTokens": 64000,
"thinking": {
"mode": "budget",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"us-gov.anthropic.claude-opus-4-1-20250805-v1:0": {
"id": "us-gov.anthropic.claude-opus-4-1-20250805-v1:0",
"name": "Claude Opus 4.1 (GovCloud)",
"api": "bedrock-converse-stream",
"provider": "amazon-bedrock",
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 15,
"output": 75,
"cacheRead": 1.5,
"cacheWrite": 18.75
},
"contextWindow": 200000,
"maxTokens": 32000,
"thinking": {
"mode": "budget",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"us-gov.anthropic.claude-opus-4-5-20251101-v1:0": {
"id": "us-gov.anthropic.claude-opus-4-5-20251101-v1:0",
"name": "Claude Opus 4.5 (GovCloud)",
"api": "bedrock-converse-stream",
"provider": "amazon-bedrock",
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 5,
"output": 25,
"cacheRead": 0.5,
"cacheWrite": 6.25
},
"contextWindow": 200000,
"maxTokens": 64000,
"thinking": {
"mode": "anthropic-budget-effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"us-gov.anthropic.claude-opus-4-6-v1": {
"id": "us-gov.anthropic.claude-opus-4-6-v1",
"name": "Claude Opus 4.6 (GovCloud)",
"api": "bedrock-converse-stream",
"provider": "amazon-bedrock",
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 5,
"output": 25,
"cacheRead": 0.5,
"cacheWrite": 6.25
},
"contextWindow": 1000000,
"maxTokens": 128000,
"thinking": {
"mode": "anthropic-adaptive",
"efforts": [
"low",
"medium",
"high",
"max"
]
}
},
"us-gov.anthropic.claude-opus-4-7": {
"id": "us-gov.anthropic.claude-opus-4-7",
"name": "Claude Opus 4.7 (GovCloud)",
"api": "bedrock-converse-stream",
"provider": "amazon-bedrock",
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 5,
"output": 25,
"cacheRead": 0.5,
"cacheWrite": 6.25
},
"contextWindow": 1000000,
"maxTokens": 128000,
"thinking": {
"mode": "anthropic-adaptive",
"efforts": [
"low",
"medium",
"high",
"max"
],
"supportsDisplay": true
}
},
"us-gov.anthropic.claude-opus-4-8": {
"id": "us-gov.anthropic.claude-opus-4-8",
"name": "Claude Opus 4.8 (GovCloud)",
"api": "bedrock-converse-stream",
"provider": "amazon-bedrock",
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 5,
"output": 25,
"cacheRead": 0.5,
"cacheWrite": 6.25
},
"contextWindow": 1000000,
"maxTokens": 128000,
"thinking": {
"mode": "anthropic-adaptive",
"efforts": [
"low",
"medium",
"high",
"max"
],
"supportsDisplay": true
}
},
"us-gov.anthropic.claude-opus-5": {
"id": "us-gov.anthropic.claude-opus-5",
"name": "Claude Opus 5 (GovCloud)",
"api": "bedrock-converse-stream",
"provider": "amazon-bedrock",
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 5,
"output": 25,
"cacheRead": 0.5,
"cacheWrite": 6.25
},
"contextWindow": 1000000,
"maxTokens": 128000,
"thinking": {
"mode": "anthropic-adaptive",
"efforts": [
"low",
"medium",
"high",
"max"
],
"supportsDisplay": true
}
},
"us-gov.anthropic.claude-sonnet-4-5-20250929-v1:0": {
"id": "us-gov.anthropic.claude-sonnet-4-5-20250929-v1:0",
"name": "Claude Sonnet 4.5 (GovCloud)",
"api": "bedrock-converse-stream",
"provider": "amazon-bedrock",
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 3,
"output": 15,
"cacheRead": 0.3,
"cacheWrite": 3.75
},
"contextWindow": 200000,
"maxTokens": 64000,
"thinking": {
"mode": "budget",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"us-gov.anthropic.claude-sonnet-4-6": {
"id": "us-gov.anthropic.claude-sonnet-4-6",
"name": "Claude Sonnet 4.6 (GovCloud)",
"api": "bedrock-converse-stream",
"provider": "amazon-bedrock",
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 3,
"output": 15,
"cacheRead": 0.3,
"cacheWrite": 3.75
},
"contextWindow": 1000000,
"maxTokens": 64000,
"thinking": {
"mode": "budget",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"us-gov.anthropic.claude-sonnet-5": {
"id": "us-gov.anthropic.claude-sonnet-5",
"name": "Claude Sonnet 5 (GovCloud)",
"api": "bedrock-converse-stream",
"provider": "amazon-bedrock",
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 2,
"output": 10,
"cacheRead": 0.2,
"cacheWrite": 2.5
},
"contextWindow": 1000000,
"maxTokens": 128000,
"thinking": {
"mode": "anthropic-adaptive",
"efforts": [
"low",
"medium",
"high",
"max"
],
"supportsDisplay": true
}
},
"us.amazon.nova-lite-v1:0": {
"id": "us.amazon.nova-lite-v1:0",
"name": "Nova Lite",
@@ -105772,4 +106096,4 @@
}
}
}
}
}
@@ -5608,14 +5608,25 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_BEDROCK: readonly ModelsDevProviderDescrip
id: crossRegionId,
name: toModelName(m.name, crossRegionId),
};
// Also emit EU variants for Claude models
// Also emit EU and AWS GovCloud (`us-gov.`) geo inference-profile
// variants for Claude models. GovCloud accounts list system profiles
// under the `us-gov.` prefix (e.g. us-gov.anthropic.claude-sonnet-4-5-…);
// without these rows the catalog only has commercial geos (`us.`/`eu.`/…)
// and model resolution rejects the GovCloud id (or misroutes commercial
// geos onto us-east-1 with GovCloud credentials → 403).
if (modelId.startsWith("anthropic.claude-")) {
const displayName = toModelName(m.name, modelId);
return [
bedrockModel,
{
...bedrockModel,
id: `eu.${modelId}`,
name: `${toModelName(m.name, modelId)} (EU)`,
name: `${displayName} (EU)`,
},
{
...bedrockModel,
id: `us-gov.${modelId}`,
name: `${displayName} (GovCloud)`,
},
];
}
@@ -4,18 +4,23 @@ import { MODELS_DEV_PROVIDER_DESCRIPTORS, mapModelsDevToModels } from "@oh-my-pi
import type { ModelSpec } from "@oh-my-pi/pi-catalog/types";
import { dropUnsupportedBedrockGeoIds } from "../scripts/generated-policies";
// AWS's Bedrock model card for Claude Opus 5 lists exactly these Programmatic
// Access IDs — the bare model ID plus the us./eu./au. Geo and global.
// inference profiles. Japan is explicitly marked unsupported for Geo
// AWS's Bedrock model card for Claude Opus 5 lists these commercial/geo
// Programmatic Access IDs — the bare model ID plus the us./eu./au. Geo and
// global inference profiles. Japan is explicitly marked unsupported for Geo
// inference in the same card's regional-availability table, so no `jp.`
// profile exists for this model (unlike several Opus 4.x generations).
// https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-anthropic-claude-opus-5.html
//
// The catalog also synthesizes `us-gov.*` Claude geo profiles for AWS GovCloud
// (same path as the derived `eu.*` row) so GovCloud selectors resolve without
// requiring a full inference-profile ARN.
const AWS_DOCUMENTED_OPUS_5_IDS = [
"anthropic.claude-opus-5",
"us.anthropic.claude-opus-5",
"eu.anthropic.claude-opus-5",
"au.anthropic.claude-opus-5",
"global.anthropic.claude-opus-5",
"us-gov.anthropic.claude-opus-5",
];
// A representative `stencil.so` "amazon-bedrock" payload for Claude Opus 5.
@@ -78,10 +83,10 @@ describe("Amazon Bedrock Claude Opus 5", () => {
);
const opus5Ids = dropUnsupportedBedrockGeoIds(mapped).map(model => model.id);
// Set semantics: the descriptor also derives an `eu.` variant from the
// bare `anthropic.` row, so `eu.` legitimately arrives from both that
// derivation and the standalone stencil.so row (deduped downstream by
// the generator). We assert the documented ID coverage, not row count.
// Set semantics: the descriptor also derives `eu.` and `us-gov.` variants
// from the bare `anthropic.` row, so `eu.` legitimately arrives from both
// that derivation and the standalone stencil.so row (deduped downstream
// by the generator). We assert the documented ID coverage, not row count.
expect(new Set(opus5Ids)).toEqual(new Set(AWS_DOCUMENTED_OPUS_5_IDS));
// `stencil.so` lists `jp.anthropic.claude-opus-5`, but Bedrock has no such
// inference profile for this model and would reject it, so the generation
@@ -0,0 +1,62 @@
import { describe, expect, test } from "bun:test";
import { MODELS_DEV_PROVIDER_DESCRIPTORS, mapModelsDevToModels } from "@oh-my-pi/pi-catalog/provider-models";
/**
* Contract: bare Anthropic Claude foundation rows from models.dev/stencil.so
* must produce a `us-gov.<foundation-id>` Bedrock inference-profile selector.
* GovCloud accounts expose system profiles under that geo prefix; without it,
* `omp --model amazon-bedrock/us-gov.…` fails model resolution even though
* AWS CLI and ARN-based selectors work.
*/
const CLAUDE_FOUNDATION_ID = "anthropic.claude-sonnet-4-5-20250929-v1:0";
const BEDROCK_CLAUDE_FIXTURE = {
"amazon-bedrock": {
models: {
[CLAUDE_FOUNDATION_ID]: {
name: "Claude Sonnet 4.5",
tool_call: true,
reasoning: true,
limit: { context: 200_000, output: 64_000 },
cost: { input: 3, output: 15, cache_read: 0.3, cache_write: 3.75 },
modalities: { input: ["text", "image"] },
},
// Non-Claude Bedrock model must not get a us-gov sibling from the Claude transform.
"amazon.nova-pro-v1:0": {
name: "Nova Pro",
tool_call: true,
reasoning: false,
limit: { context: 300_000, output: 10_000 },
cost: { input: 0.8, output: 3.2, cache_read: 0, cache_write: 0 },
modalities: { input: ["text", "image"] },
},
},
},
};
describe("Amazon Bedrock GovCloud (us-gov) catalog mapping", () => {
test("bare Claude foundation ids emit a us-gov geo inference-profile selector", () => {
const mapped = mapModelsDevToModels(BEDROCK_CLAUDE_FIXTURE, MODELS_DEV_PROVIDER_DESCRIPTORS).filter(
model => model.provider === "amazon-bedrock",
);
const ids = mapped.map(model => model.id);
expect(ids).toContain(`us-gov.${CLAUDE_FOUNDATION_ID}`);
expect(ids).toContain(`eu.${CLAUDE_FOUNDATION_ID}`);
const gov = mapped.find(model => model.id === `us-gov.${CLAUDE_FOUNDATION_ID}`);
expect(gov).toBeDefined();
expect(gov?.api).toBe("bedrock-converse-stream");
expect(gov?.name).toContain("GovCloud");
});
test("non-Claude Bedrock models do not receive synthesized us-gov variants", () => {
const mapped = mapModelsDevToModels(BEDROCK_CLAUDE_FIXTURE, MODELS_DEV_PROVIDER_DESCRIPTORS).filter(
model => model.provider === "amazon-bedrock",
);
const ids = mapped.map(model => model.id);
expect(ids.some(id => id.startsWith("us-gov.amazon."))).toBe(false);
expect(ids).not.toContain("us-gov.amazon.nova-pro-v1:0");
});
});
+60 -1
View File
@@ -7,6 +7,57 @@
- Fixed an advisor refusal skipping the model fallback chain. `AdvisorRuntime` treated a classifier refusal as terminal once its one stripped-reasoning resend failed, returning before the `onTurnError` hook that owns fallback (`#recoverAdvisorTurn`), so a `Refusal (cyber)` on one model disabled the advisor even with a configured chain. The refusal path now takes the same fallback pass the primary turn-recovery path already allows, and only reports the advisor unavailable when the host declines to switch.
- Bounded advisor refusal recovery to one attempt per model. The cascade walks the fallback chain to exhaustion, but a model switch re-arms `#includeThinking` via `#syncModelIdentity`, so a chain whose keys point back at each other (A→B, B→A) would strip-and-resend against the same pair forever. Each cascade now visits a model at most once; a successful turn or a reset starts a fresh walk.
- Fixed `/advisor status` throwing when the roster is empty but an advisor is live. `formatAdvisorStatus` dereferenced `stats.advisors[0]` after a guard that only covered the inactive case, so a status call landing in the window where `#advisorStatuses` is cleared for a rebuild hit `undefined.contextWindow`. Reporting status now never throws.
## [17.2.9] - 2026-08-05
### Breaking Changes
- Renamed `compareVersions` to `compareChangelogEntries` in `@oh-my-pi/pi-coding-agent/utils/changelog`. The function signature and behavior are unchanged; update imports to use the new name.
### Added
- Added automatic detection of common Ungoogled Chromium Linux installations for the browser tool.
### Changed
- Reworked the Ctrl+S Agent Hub into a responsive fullscreen roster and selected-agent inspector with aggregate status/usage, per-agent task/model/activity/usage/lineage details, roster and spawn-tree views, stable ordering, bounded large-roster rendering, asynchronous persisted-session discovery, restored task/timestamp metadata for historical agents, and consistent keyboard and mouse navigation.
- Restored the legacy project-scoped session directory naming scheme and removed its automatic migration ([#7646](https://github.com/can1357/oh-my-pi/issues/7646)).
- Routed Bun install-cache pruning in `update-cli` through the shared `compareVersions` utility (`@oh-my-pi/pi-utils`), removing a duplicate local comparator that rounded large numeric version identifiers via `Number`.
### Fixed
- Retried concurrent-request caps with a short backoff without deleting valid Copilot credentials or rotating through sibling accounts.
- Fixed the default `textVerbosity` setting being forwarded to OpenAI Codex requests unless the user explicitly configures it, preserving Codex's native response-control defaults. ([#4949](https://github.com/can1357/oh-my-pi/issues/4949))
- Reduced streaming CPU usage by coalescing the cumulative `message_update` deltas of a turn at the event-controller dispatch boundary: at most one streaming-state rebuild runs per ~33ms window instead of one per token, cutting the per-token handler work that dominated the CPU profile of streaming sessions (especially at high token rates) while preserving per-delta speech output. Subscriber dispatch is serialized so a rapid stream tail (`message_update` → `message_end` → `agent_end`) cannot overtake the coalesced flush. ([#7443](https://github.com/can1357/oh-my-pi/issues/7443))
- Fixed translated MCP importers (Claude Code, Cursor, Gemini CLI, Windsurf, VS Code) silently dropping a server's `enabled: false` flag, so a server disabled at the source config stayed mounted; the flag is now propagated and honored like Codex, OpenCode, and native `mcp.json`. These importers now also load project entries before same-named user entries (matching native/Codex) so a project `enabled: false` suppresses a same-named user server ([#7652](https://github.com/can1357/oh-my-pi/issues/7652)).
- Removed the per-call `model` override from the eval `agent()` helper (all runtimes), completing the earlier task-tool removal (`9f8aa87dbf`). Subagents always use their selected agent's frontmatter model and settings; a legacy `model` argument is silently ignored, so an explicit `model: "default"` can no longer route children onto the parent session model ([#6438](https://github.com/can1357/oh-my-pi/issues/6438)).
- Fixed legacy Pi extension validation rejecting plugins such as `remote-pi` that import the package-root `convertToPng` image helper. ([#7610](https://github.com/can1357/oh-my-pi/issues/7610))
- Fixed the legacy session-directory migration silently deleting a live session's transcript when its filename collided with an existing entry in the destination: colliding entries are now preserved in place, the legacy directory is only removed when empty, and collisions/migration failures are logged ([#7593](https://github.com/can1357/oh-my-pi/issues/7593)).
- Fixed `PUPPETEER_EXECUTABLE_PATH` being ignored when a system Chrome installation was detected, preventing Windows users from selecting a compatible headless browser for the shared browser daemon ([#7601](https://github.com/can1357/oh-my-pi/issues/7601)).
- Fixed `openai-models-list` discovery ignoring server-advertised input modalities, so custom virtual tier IDs absent from the bundled catalog showed `images: no` even when the `/v1/models` response reported `input: ["text","image"]` ([#7583](https://github.com/can1357/oh-my-pi/issues/7583)).
- Exposed exact source line counts in read results when selector-based reads reach EOF, allowing protocol bridges to distinguish a returned slice from the complete file ([#7590](https://github.com/can1357/oh-my-pi/issues/7590)).
- Fixed `grep`/`glob` silently collapsing a semicolon-delimited `path` list to one literal path when the joined string was too long for the OS to name (`ENAMETOOLONG`) — a list of bare filenames past `NAME_MAX` or absolute paths past `PATH_MAX` failed with `Path not found: <whole list>` even though every entry existed. The multipath probe now treats `ENAMETOOLONG` as a definitively non-existent single path so the split proceeds, and `glob` surfaces a clean `Path not found` instead of leaking the raw errno ([#7597](https://github.com/can1357/oh-my-pi/issues/7597)).
- Fixed `--mode json` (and text) print mode truncating a large final record (e.g. a multi-MB `agent_end`) when the process exited before stdout drained, while still exiting 0. Per-event writes are now serialized on their own completion callbacks and shutdown blocks on the last one, so the terminal record is delivered in full ([#7635](https://github.com/can1357/oh-my-pi/issues/7635)).
- Fixed text print mode treating buffered partial responses as replay-unsafe, allowing transient mid-stream connection failures to retry without exposing duplicated output ([#7625](https://github.com/can1357/oh-my-pi/issues/7625)).
- Fixed Hindsight `autoRecall` intermittently not reaching the model: two recall paths shared the `hasRecalledForFirstTurn` flag, and the `agent_start` event path could consume it first and inject only via an unawaited background prompt rebuild that a fast turn outran. `beforeAgentStartPrompt` (awaited before the turn builds) is now the sole injection path ([#7568](https://github.com/can1357/oh-my-pi/issues/7568)).
- Fixed `read memory://<id>` returning a confusing "Unknown memory namespace" error under `memory.backend=hindsight` (Hindsight stores memories server-side and has no `memory://` addressing); the handler now returns a corrective pointer to `recall`/`reflect` so a stray read — steered by the shared `recall` tool description — self-corrects in one turn ([#7587](https://github.com/can1357/oh-my-pi/issues/7587)).
- Fixed extension/custom/hook tool wrappers stripping schema methods off `parameters`: `applyToolProxy` bound every callable property, and binding a schema (a plain function carrying `toJsonSchema`/`assert`) dropped those properties, breaking wire-schema detection and crashing the status-line token estimator with `JSON.stringify(schema) === undefined`. Prototype methods are still bound; own data properties and schema callables now pass through untouched.
- Fixed bug where `agent()` calls in eval cells ignored turn cancellation and continued running indefinitely
- Fixed the built-in `tail` printing `tail: Broken pipe` and failing when a downstream pipeline reader exited early (e.g. `tail -c N file.jsonl | jq …` with jq aborting on a parse error); it now exits silently with 141 (128+SIGPIPE) like a real tail, in every output path including `--follow`.
- Fixed the in-process ps shell builtin rejecting common procps/BSD format specifiers (`ps -o tpgid,...` failed with `unknown output format specifier`); added `tpgid`, `pri`, `flags`, real/effective user and group columns, `wchan`, fault counters, `sz`, and the STAT `+` foreground flag.
- Fixed Herdr rejecting the macOS development launcher because its foreground process was reported as `bun` instead of `omp`.
- Completed usage-aware model fallback across startup, queued turns, same-turn tool continuations, ACP/TUI confirmation cancellation, eligible account reselection, cooldown restoration, and isolated subagent settings so low-usage handoffs remain lossless and cannot consume cancelled queued work.
- Fixed Agent Hub opening and selection becoming O(all rows) on large rosters: row rendering is now lazy around the selected viewport, and observer lookup is O(1) by id instead of copy-sorting every session per row.
- Fixed persisted Agent Hub rows dropping an explicit caller model role when a subagent used a model override, preserving role provenance after restart.
- Fixed the bash interceptor blocking `grep`/`cat`/`find` used as a downstream pipeline stage (e.g. `printf 'x\n' | grep x`); a stage consuming piped stdin cannot be replaced by a path-based dedicated tool, so it is no longer matched, while standalone and first-stage searches stay intercepted ([#7496](https://github.com/can1357/oh-my-pi/issues/7496)).
- Fixed floating rejections from cmux browser guest JavaScript terminating the main process and every active session; attributable rejections now fail the browser run as tool errors while unrelated process rejections retain the fatal path ([#7365](https://github.com/can1357/oh-my-pi/issues/7365)).
- Fixed the Windows bash tool silently taking down the whole omp process when a command blocked until its timeout: cancelling a timed-out run walked the spawned child's descendant tree from raw `th32ParentProcessID` links, and a recycled pid matching the harness's stale recorded parent pid could enumerate omp as a false descendant and `TerminateProcess` it, killing the session with no `session_exit` record. Run-cancellation sweeps now refuse to signal the harness or any process collected beneath it, while still reaping the timed-out target when it owns a recycled ancestor pid ([#7452](https://github.com/can1357/oh-my-pi/issues/7452)).
- Fixed the unexpected-stop guard (`features.unexpectedStopDetection`) never firing for thinking-only stops: `isUnexpectedStopCandidate` only counted non-whitespace `text` blocks, so a `stopReason: "stop"` turn whose sole content was a signed `thinking` block (a trapped response or a truncated reasoning fragment from reasoning models) bypassed classification and silently ended the turn mid-task. Such stops are now candidates and are classified on their thinking text ([#7499](https://github.com/can1357/oh-my-pi/issues/7499)).
- Fixed Task cancellation hanging forever when a child ignored abort or stalled during cleanup ([#7483](https://github.com/can1357/oh-my-pi/issues/7483)).
- Fixed LSP diagnostics being dropped when servers normalize file URI percent-encoding or Windows path casing.
- Fixed WSL sessions missing Agent Skills stored in the Windows host profile's `.agents/skills` directory. ([#3779](https://github.com/can1357/oh-my-pi/issues/3779))
- Fixed `omp setup python` to validate the same configured or discovered interpreter used by the Python eval runtime.
- Fixed self-update misclassifying glibc Linux hosts with an installed musl loader as musl hosts, which could download an unusable musl binary instead of the glibc release.
- Fixed a crash where opening the Agent Hub after a resume and moving the selection triggered an unbounded `ExtensionExitError` unhandled-rejection storm and exit 129. The postmortem module bound the native hard-exit at first evaluation; when the bundler deferred that evaluation into a `withHostGuard` window it froze the guard's throwing replacement, poisoning every later signal/fatal exit. The native exit is now resolved per call, and the guard stamps its replacement with the native primitive it shadows so mid-guard signals still exit ([#7393](https://github.com/can1357/oh-my-pi/issues/7393)).
## [17.2.8] - 2026-08-04
@@ -19,6 +70,7 @@
### Changed
- Replaced arktype with @oh-my-pi/omptype for tool parameter and config schemas, significantly improving startup performance with ~100x faster schema construction. Config schema errors are now reported via OmpErrors using the same path/problem structure.
- Replaced arktype with `@oh-my-pi/omptype` across all tool parameter and config schemas: ~100x faster schema construction removes the arktype startup tax (the `scope({}, { jitless: true })` workarounds are gone). Config schema errors now report via `OmpErrors` entries with the same `path`/`problem` shape.
### Fixed
@@ -98,6 +150,13 @@
- Fixed heavily branched conversation trees shifting linear continuations into disconnected columns.
- Fixed plugin installation validation failures for legacy compatibility shims.
- Removed hard-coded references to disabled or absent agents in system and tool prompts.
### Added
- Added resumable session details to fatal crash output, including an `omp --resume <session-id>` command for every persisted live agent session.
### Fixed
- Fixed unobserved promise continuations from browser helpers such as `tab.waitForResponse()` wedging or killing the tab worker when they reject; browser facade promises now retain native promise behavior while observing every `then`, `catch`, and `finally` continuation, and late user continuation errors are logged instead of dropped after the run ends.
## [17.2.4] - 2026-08-01
@@ -5361,7 +5420,7 @@
- Fixed command-fixup notices to list all stripped segments instead of reporting only one
- Fixed summarized `read` output stalling agents on elided regions by appending an explicit footer like `[NN lines across MM elided regions; read <path>:raw or a line range like <path>:1-9999 for verbatim content]`. The footer fires whenever the structural summarizer elided at least one span, so the model gets a concrete recovery selector instead of having to guess from a bare `...` / `{ .. }` marker. Surfaces `elidedLines` on `ReadToolDetails.summary` alongside the existing `elidedSpans`. ([#1046](https://github.com/can1357/oh-my-pi/issues/1046))
- Updated the `read` tool prompt to describe the new elision footer and instruct the model to follow `:raw` (or an explicit line range) when the elided body is actually needed, rather than guessing.
- Fixed plugin extensions failing to load when their `peerDependencies` reference internal `pi-*` packages under any scope other than `@mariozechner` (e.g. `Cannot find module '@earendil-works/pi-tui'` from `@juicesharp/rpiv-ask-user-question`, or `Cannot find module '@oh-my-pi/pi-utils'` from `@oh-my-pi/swarm-extension`). The legacy-pi specifier shim now treats `@mariozechner`, `@earendil-works`, **and** the canonical `@oh-my-pi` itself as aliases for the same set of bundled in-process packages (`pi-agent-core`, `pi-ai`, `pi-coding-agent`, `pi-natives`, `pi-tui`, `pi-utils`), and additionally rewrites the upstream-only `pi-ai/oauth` subpath onto our `pi-ai/utils/oauth` layout. Restored the `Key` runtime helper export on `@oh-my-pi/pi-tui` to match upstream — plugins using `Key.enter` / `Key.ctrl("c")` (e.g. `@plannotator/pi-extension`, `@juicesharp/rpiv-ask-user-question`) no longer fail with `Export named 'Key' not found`. End-to-end verified against `@juicesharp/rpiv-ask-user-question`, `@oh-my-pi/swarm-extension`, and `@plannotator/pi-extension` — each now loads cleanly with all of its tools/commands/handlers registered. Plugins importing any of those scopes are remapped to the omp binary's own copy at load time, so peer deps are no longer dragged in from npm and there is exactly one module instance per package regardless of which scope name the plugin's manifest happened to declare.
- Fixed plugin extensions failing to load when their `peerDependencies` reference internal `pi-*` packages under any scope other than `@mariozechner` (e.g. `Cannot find module '@earendil-works/pi-tui'` from `@juicesharp/rpiv-ask-user-question`). The legacy-pi specifier shim now treats `@mariozechner`, `@earendil-works`, **and** the canonical `@oh-my-pi` itself as aliases for the same set of bundled in-process packages (`pi-agent-core`, `pi-ai`, `pi-coding-agent`, `pi-natives`, `pi-tui`, `pi-utils`), and additionally rewrites the upstream-only `pi-ai/oauth` subpath onto our `pi-ai/utils/oauth` layout. Restored the `Key` runtime helper export on `@oh-my-pi/pi-tui` to match upstream …
- Fixed `omp commit` hanging after a successful commit instead of returning to the shell. The command now mirrors the `runPrintMode` exit pattern and calls `postmortem.quit(0)` once the pipeline resolves so lingering HTTP/2 keep-alive sockets, the Settings autosave timer, and other AgentSession background handles don't keep the event loop pinned. ([#1041](https://github.com/can1357/oh-my-pi/issues/1041))
- Fixed hashline payload parsing to silently treat truly-blank lines as empty `~`-prefixed payload lines when more payload follows in the same run. The previous behavior broke at the blank ("payload line has no preceding +, <, or = operation.") even though the intent is obvious — the only ambiguity is between in-payload blanks and end-of-section blanks, and a one-line lookahead resolves it: blanks that precede a non-payload op still end the run cleanly as section separators. Recovers the common case of forgetting the leading separator on a blank inserted line without changing how trailing blanks between ops behave.
- Rewrote the hashline edit prompt examples to use an ASCII-only `TITLE = "Mr"` → `"Mrs"` / `"Dr"` motif instead of the previous `" • "` and `"·"` separators. Some agents had been copying the middle-dot literal characters into real edits as if they were format scaffolding (e.g. emitting payload lines like `~ ·`), since the demo inserts were near-twins of the existing string. The new example keeps every original op shape (single-line replace, multiline replace, insert AFTER/BEFORE, append, delete, blank, plus both anti-patterns) but uses content that is obviously domain-specific and clearly distinct from any payload separator. Pure prompt change; no parser, schema, or runtime behavior is affected.
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-coding-agent",
"version": "17.2.8",
"version": "17.2.9",
"description": "Coding agent CLI with read, bash, edit, write tools and session management",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+11 -2
View File
@@ -36,7 +36,16 @@ mkdir -p "$launch_dir"
OMP_LAUNCH_CWD=$PWD
export OMP_LAUNCH_CWD
cd "$launch_dir"
# Herdr 0.7.5 validates the foreground process name. The macOS shell can
# preserve OMP's identity while still running the development CLI through Bun.
run_bun() {
if [ "${HERDR_ENV:-}" = 1 ] && [ "$(uname -s)" = Darwin ]; then
exec -a omp bun "$@"
fi
exec bun "$@"
}
if [ -n "${PI_TIMING:-}" ]; then
exec bun --preload "$preload" --preload "$timing_preload" "$cli" "$@"
run_bun --preload "$preload" --preload "$timing_preload" "$cli" "$@"
fi
exec bun --preload "$preload" "$cli" "$@"
run_bun --preload "$preload" "$cli" "$@"
@@ -94,6 +94,12 @@ export interface AsyncJobDeliveryState {
pendingJobIds: string[];
}
export interface AsyncJobReapResult {
settled: boolean;
pendingJobIds: string[];
completion: Promise<void>;
}
export interface AsyncJobRegisterOptions {
id?: string;
/** Registry id of the agent that owns this job; used to scope cancelAll. */
@@ -490,6 +496,26 @@ export class AsyncJobManager {
}
}
/**
* Cancel every job owned by `ownerId`, then wait only until `deadlineAt`.
* The returned completion keeps waiting for actual process settlement when
* the deadline expires, so callers can move that cleanup out of the
* user-visible Task wait without losing ownership of the live work.
*/
async cancelAndReapOwnerJobs(ownerId: string, deadlineAt: number): Promise<AsyncJobReapResult> {
this.cancelAll({ ownerId });
const timeoutMs = Math.max(0, deadlineAt - Date.now());
const settled = await this.waitForOwnerJobs(ownerId, { timeoutMs });
if (settled) {
return { settled: true, pendingJobIds: [], completion: Promise.resolve() };
}
const pendingJobIds = this.getAllJobs({ ownerId })
.filter(job => job.status === "running" || job.status === "cancelled")
.map(job => job.id);
const completion = this.waitForOwnerJobs(ownerId).then(() => {});
return { settled: false, pendingJobIds, completion };
}
async #waitForAllUntil(deadline: number): Promise<boolean> {
const promises = Array.from(this.#jobs.values()).map(job => job.promise);
if (promises.length === 0) return true;
+3 -3
View File
@@ -136,11 +136,11 @@ export async function runStatsCommand(cmd: StatsCommandArgs): Promise<void> {
}
// Start the dashboard server
const { port } = await startServer(cmd.port);
console.log(chalk.green(`Dashboard available at: http://localhost:${port}`));
const { hostname, port } = await startServer(cmd.port);
const url = `http://${hostname}:${port}`;
console.log(chalk.green(`Dashboard available at: ${url}`));
// Open browser
const url = `http://localhost:${port}`;
openPath(url);
console.log("Press Ctrl+C to stop\n");
+2 -56
View File
@@ -10,7 +10,7 @@ import * as os from "node:os";
import * as path from "node:path";
import { Transform } from "node:stream";
import { pipeline } from "node:stream/promises";
import { $env, $which, APP_NAME, isEnoent, VERSION } from "@oh-my-pi/pi-utils";
import { $env, $which, APP_NAME, compareVersions, isEnoent, VERSION } from "@oh-my-pi/pi-utils";
import { $ } from "bun";
import chalk from "chalk";
import { theme } from "../modes/theme/theme";
@@ -498,24 +498,6 @@ async function getLatestRelease(): Promise<ReleaseInfo> {
};
}
/**
* Compare semver versions. Returns:
* - negative if a < b
* - 0 if a == b
* - positive if a > b
*/
function compareVersions(a: string, b: string): number {
const pa = a.split(".").map(Number);
const pb = b.split(".").map(Number);
for (let i = 0; i < Math.max(pa.length, pb.length); i++) {
const na = pa[i] || 0;
const nb = pb[i] || 0;
if (na !== nb) return na - nb;
}
return 0;
}
interface BunInstallCachePruneResult {
scannedPackages: number;
removedEntries: number;
@@ -532,42 +514,6 @@ function stripBunCacheVersionSuffix(name: string): string {
return metadataIndex === -1 ? name : name.slice(0, metadataIndex);
}
function compareSemverIdentifier(a: string, b: string): number {
const aNumber = /^\d+$/.test(a);
const bNumber = /^\d+$/.test(b);
if (aNumber && bNumber) return Number(a) - Number(b);
if (aNumber) return -1;
if (bNumber) return 1;
return a.localeCompare(b);
}
function compareSemverLikeVersions(a: string, b: string): number {
const [aCoreWithPrerelease] = a.split("+", 1);
const [bCoreWithPrerelease] = b.split("+", 1);
const [aCore, aPrerelease] = aCoreWithPrerelease.split("-", 2);
const [bCore, bPrerelease] = bCoreWithPrerelease.split("-", 2);
const aParts = aCore.split(".");
const bParts = bCore.split(".");
for (let i = 0; i < Math.max(aParts.length, bParts.length); i++) {
const diff = Number(aParts[i] ?? 0) - Number(bParts[i] ?? 0);
if (diff !== 0 && Number.isFinite(diff)) return diff;
}
if (!aPrerelease && !bPrerelease) return 0;
if (!aPrerelease) return 1;
if (!bPrerelease) return -1;
const aPrereleaseParts = aPrerelease.split(".");
const bPrereleaseParts = bPrerelease.split(".");
for (let i = 0; i < Math.max(aPrereleaseParts.length, bPrereleaseParts.length); i++) {
const aPart = aPrereleaseParts[i];
const bPart = bPrereleaseParts[i];
if (aPart === undefined) return -1;
if (bPart === undefined) return 1;
const diff = compareSemverIdentifier(aPart, bPart);
if (diff !== 0) return diff;
}
return 0;
}
async function readdirIfExists(dir: string): Promise<fs.Dirent[]> {
try {
return await fs.promises.readdir(dir, { withFileTypes: true });
@@ -689,7 +635,7 @@ export async function pruneBunInstallCache(
scannedPackages++;
let latestVersion: string | undefined;
for (const version of group.actualDirs.keys()) {
if (!latestVersion || compareSemverLikeVersions(version, latestVersion) > 0) latestVersion = version;
if (!latestVersion || compareVersions(version, latestVersion) > 0) latestVersion = version;
}
if (!latestVersion) continue;
for (const [version, paths] of group.actualDirs) {
@@ -109,11 +109,11 @@ export class ConfigError extends Error {
this.#message = message;
}
get message(): string {
override get message(): string {
return this.#message;
}
toString(): string {
override toString(): string {
return this.message;
}
}
@@ -607,12 +607,12 @@ export class KeybindingsManager extends TuiKeybindingsManager {
this.setUserBindings(mergeKeybindingsConfig(inheritedConfig, profileConfig));
}
setUserBindings(userBindings: KeybindingsConfig): void {
override setUserBindings(userBindings: KeybindingsConfig): void {
this.#userBindings = userBindings;
super.setUserBindings(userBindings);
}
getKeys(keybinding: Keybinding): KeyId[] {
override getKeys(keybinding: Keybinding): KeyId[] {
const keys = super.getKeys(keybinding);
const fallbackKey = getFallbackKey(keybinding);
if (fallbackKey === undefined || this.#userBindings[keybinding] !== undefined) return keys;
@@ -620,7 +620,7 @@ export class KeybindingsManager extends TuiKeybindingsManager {
return removeKey(keys, fallbackKey);
}
getResolvedBindings(): KeybindingsConfig {
override getResolvedBindings(): KeybindingsConfig {
const resolved = super.getResolvedBindings();
resolved[FOLLOW_UP_KEYBINDING] = keyConfigValue(this.getKeys(FOLLOW_UP_KEYBINDING));
return resolved;
@@ -724,6 +724,31 @@ export async function discoverLlamaCppModelRuntimeMetadata(
}
}
/**
* Read image-input support from an OpenAI-compatible `/v1/models` row. Handles
* direct `input` arrays, Synthetic-style top-level `input_modalities`, and
* OpenRouter-style `architecture.input_modalities`; returns undefined when none
* is present so the bundled reference (or the `["text"]` default) can take over.
*/
function extractOpenAIModelsListInputCapabilities(item: {
input?: unknown;
input_modalities?: unknown;
architecture?: unknown;
}): ("text" | "image")[] | undefined {
const modalities = new Set<string>();
const collect = (value: unknown): void => {
if (!Array.isArray(value)) return;
for (const entry of value) {
if (typeof entry === "string") modalities.add(entry.toLowerCase());
}
};
collect(item.input);
collect(item.input_modalities);
if (isRecord(item.architecture)) collect(item.architecture.input_modalities);
if (modalities.size === 0) return undefined;
return modalities.has("image") ? ["text", "image"] : ["text"];
}
export async function discoverOpenAIModelsList(
providerConfig: DiscoveryProviderConfig,
ctx: DiscoveryContext,
@@ -752,7 +777,14 @@ export async function discoverOpenAIModelsList(
}
headers = h;
return (await res.json()) as {
data?: Array<{ id?: string; max_model_len?: unknown; context_length?: unknown }>;
data?: Array<{
id?: string;
max_model_len?: unknown;
context_length?: unknown;
input?: unknown;
input_modalities?: unknown;
architecture?: unknown;
}>;
};
}),
nativeMetadataPromise,
@@ -796,7 +828,9 @@ export async function discoverOpenAIModelsList(
baseUrl,
reasoning: reference?.reasoning ?? false,
thinking: inheritReferenceThinking(undefined, reference, providerConfig.provider),
input: nativeMetadataForModel?.input ?? reference?.input ?? ["text"],
input: nativeMetadataForModel?.input ??
extractOpenAIModelsListInputCapabilities(item) ??
reference?.input ?? ["text"],
...(providerConfig.discovery.type === "lm-studio" ? { imageInputDecoder: "stb" as const } : {}),
// Proxy/gateway pricing is provider-specific and rarely matches
// upstream bundled catalogs, so keep costs local-unknown even
@@ -1715,7 +1715,10 @@ export class ModelRegistry {
return resolveOllamaModelCacheProviderId(providerConfig.provider, providerConfig.baseUrl);
}
if (providerConfig.discovery.type === "openai-models-list") {
return `${providerConfig.provider}:openai-models-list-context-v2`;
// context-v3 invalidates rows cached before server-advertised input
// modalities were parsed from `/v1/models`; warm v2 rows pinned
// vision-capable ids at `input: ["text"]` until a forced refresh.
return `${providerConfig.provider}:openai-models-list-context-v3`;
}
if (providerConfig.discovery.type === "litellm") {
// rich-v2 invalidates rows cached before reseller usage-suffix stripping
@@ -932,6 +932,28 @@ function normalizeModelPatternList(value: string | string[] | undefined): string
return patterns.map(pattern => pattern.trim()).filter(Boolean);
}
/**
* Extract the first explicit model-role alias from a raw model selection.
*
* This intentionally runs before role expansion so callers can retain the
* source identity (`@smol`, `pi/slow`, or `*`) even when it resolves to a
* concrete provider/model or inherited fallback. Bare role names and explicit
* provider/model selectors are not role aliases.
*/
export function resolveExplicitModelRole(
value: string | string[] | undefined,
settings?: ModelRoleLookup,
): string | undefined {
for (const pattern of normalizeModelPatternList(value)) {
const prefixLength = modelRoleAliasPrefixLength(pattern);
if (prefixLength === undefined) continue;
const { base } = splitThinkingSuffix(pattern, prefixLength, MAX_THINKING_SUFFIX_OPTIONS);
const role = getModelRoleAlias(base, settings);
if (role) return role;
}
return undefined;
}
function isSessionInheritedAgentPattern(value: string): boolean {
return (
value === DEFAULT_MODEL_ROLE ||
@@ -1068,6 +1090,8 @@ export function resolveConfiguredModelPatterns(
});
}
export interface AgentModelPatternResolutionOptions {
/** Highest-priority request selector, when supplied by a caller. */
requestModel?: string | string[];
settingsOverride?: string | string[];
agentModel?: string | string[];
settings?: Settings;
@@ -1075,11 +1099,25 @@ export interface AgentModelPatternResolutionOptions {
fallbackModelPattern?: string;
}
export function resolveAgentModelPatterns(options: AgentModelPatternResolutionOptions): string[] {
const { settingsOverride, agentModel, settings, activeModelPattern, fallbackModelPattern } = options;
interface EffectiveAgentModelSelection {
source?: string | string[];
patterns: string[];
}
function resolveEffectiveAgentModelSelection(
options: AgentModelPatternResolutionOptions,
): EffectiveAgentModelSelection {
const { requestModel, settingsOverride, agentModel, settings, activeModelPattern, fallbackModelPattern } = options;
const requestPatterns = resolveConfiguredModelPatterns(requestModel, settings);
if (requestPatterns.length > 0) {
return { source: requestModel, patterns: requestPatterns };
}
const overridePatterns = resolveConfiguredModelPatterns(settingsOverride, settings);
if (overridePatterns.length > 0) return overridePatterns;
if (overridePatterns.length > 0) {
return { source: settingsOverride, patterns: overridePatterns };
}
const normalizedAgentPatterns = normalizeModelPatternList(agentModel);
const configuredAgentPatterns = resolveConfiguredModelPatterns(agentModel, settings);
@@ -1090,14 +1128,23 @@ export function resolveAgentModelPatterns(options: AgentModelPatternResolutionOp
singleAgentPattern === formatModelRoleAlias("task") ||
singleAgentPattern === `${LEGACY_MODEL_ROLE_ALIAS_PREFIX}task`
) {
return configuredAgentPatterns;
return { source: agentModel, patterns: configuredAgentPatterns };
}
if (!agentInheritsSessionModel) return configuredAgentPatterns;
if (!agentInheritsSessionModel) return { source: agentModel, patterns: configuredAgentPatterns };
}
const fallback =
activeModelPattern?.trim() || fallbackModelPattern?.trim() || settings?.getModelRole("default")?.trim() || "";
return resolveConfiguredModelPatterns(fallback, settings);
return { patterns: resolveConfiguredModelPatterns(fallback, settings) };
}
/** Return the raw selector source that supplies the effective agent patterns. */
export function resolveAgentModelSource(options: AgentModelPatternResolutionOptions): string | string[] | undefined {
return resolveEffectiveAgentModelSelection(options).source;
}
export function resolveAgentModelPatterns(options: AgentModelPatternResolutionOptions): string[] {
return resolveEffectiveAgentModelSelection(options).patterns;
}
/** Default prewalk hand-off target when no explicit target is configured. */
export const DEFAULT_PREWALK_TARGET = "@smol";
+71 -3
View File
@@ -28,9 +28,77 @@ const DISPLAY_NAME = "Agent Dirs (.agent/.agents)";
const PRIORITY = 70;
const AGENT_DIR_CANDIDATES = [".agent", ".agents"] as const;
/** User-level paths: ~/.agent/<segments> and ~/.agents/<segments>. */
function getUserPathCandidates(ctx: LoadContext, ...segments: string[]): string[] {
return AGENT_DIR_CANDIDATES.map(baseDir => path.join(ctx.home, baseDir, ...segments));
interface UserPathCandidateOptions {
platform?: NodeJS.Platform;
env?: NodeJS.ProcessEnv;
windowsUserProfile?: () => string | undefined;
wslPath?: (windowsPath: string) => string | undefined;
}
const WINDOWS_DRIVE_PROFILE_PATTERN = /^([A-Za-z]):[\\/](.*)$/;
function isWsl(platform: NodeJS.Platform, env: NodeJS.ProcessEnv): boolean {
return platform === "linux" && Boolean(env.WSL_DISTRO_NAME || env.WSL_INTEROP);
}
function convertWindowsPathToDefaultWslMount(windowsPath: string): string | undefined {
const trimmed = windowsPath.trim();
if (trimmed.length === 0) return undefined;
if (path.isAbsolute(trimmed)) return path.normalize(trimmed);
const match = WINDOWS_DRIVE_PROFILE_PATTERN.exec(trimmed);
if (!match) return undefined;
const [, drive, rest] = match;
const segments = rest.replace(/\\/g, "/").split("/").filter(Boolean);
return path.join("/mnt", drive.toLowerCase(), ...segments);
}
function resolveWithWslPath(windowsPath: string): string | undefined {
try {
const result = Bun.spawnSync(["wslpath", "-u", windowsPath], { stdout: "pipe", stderr: "ignore" });
if (result.exitCode !== 0) return undefined;
const resolved = result.stdout.toString().trim();
return resolved.length > 0 ? resolved : undefined;
} catch {
return undefined;
}
}
function resolveWindowsUserProfile(): string | undefined {
try {
const result = Bun.spawnSync(["cmd.exe", "/d", "/c", "echo", "%USERPROFILE%"], {
stdout: "pipe",
stderr: "ignore",
});
if (result.exitCode !== 0) return undefined;
const resolved = result.stdout.toString().trim();
return resolved.length > 0 && resolved !== "%USERPROFILE%" ? resolved : undefined;
} catch {
return undefined;
}
}
/** Resolve the Windows host profile home exposed to WSL, if available. */
export function getWslWindowsHomeCandidate(options: UserPathCandidateOptions = {}): string | undefined {
const platform = options.platform ?? process.platform;
const env = options.env ?? process.env;
if (!isWsl(platform, env)) return undefined;
const userProfile = env.USERPROFILE ?? (options.windowsUserProfile ?? resolveWindowsUserProfile)();
if (!userProfile) return undefined;
return (options.wslPath ?? resolveWithWslPath)(userProfile) ?? convertWindowsPathToDefaultWslMount(userProfile);
}
function getUserHomeCandidates(ctx: LoadContext): string[] {
const homes = [ctx.home];
const wslHome = getWslWindowsHomeCandidate();
if (wslHome && !homes.includes(wslHome)) homes.push(wslHome);
return homes;
}
/** User-level paths: ~/.agent[s]/<segments>, plus the Windows host profile under WSL. */
export function getUserPathCandidates(ctx: LoadContext, ...segments: string[]): string[] {
return getUserHomeCandidates(ctx).flatMap(home =>
AGENT_DIR_CANDIDATES.map(baseDir => path.join(home, baseDir, ...segments)),
);
}
/**
@@ -90,6 +90,7 @@ async function loadMCPServers(ctx: LoadContext): Promise<LoadResult<MCPServer>>
const serverConfig = config as Record<string, unknown>;
return {
name,
enabled: typeof serverConfig.enabled === "boolean" ? serverConfig.enabled : undefined,
timeout: typeof serverConfig.timeout === "number" ? serverConfig.timeout : undefined,
command: serverConfig.command as string | undefined,
args: serverConfig.args as string[] | undefined,
@@ -102,17 +103,19 @@ async function loadMCPServers(ctx: LoadContext): Promise<LoadResult<MCPServer>>
});
};
for (let i = 0; i < userPaths.length; i++) {
const servers = parseMcpServers(contents[i], userPaths[i].path, userPaths[i].level);
// Load project entries before user entries so a project `enabled: false`
// claims its dedupe key before a same-named user server can survive (#7654).
const projectOffset = userPaths.length;
for (let i = 0; i < projectPaths.length; i++) {
const servers = parseMcpServers(contents[projectOffset + i], projectPaths[i].path, projectPaths[i].level);
if (servers.length > 0) {
items.push(...servers);
break;
}
}
const projectOffset = userPaths.length;
for (let i = 0; i < projectPaths.length; i++) {
const servers = parseMcpServers(contents[projectOffset + i], projectPaths[i].path, projectPaths[i].level);
for (let i = 0; i < userPaths.length; i++) {
const servers = parseMcpServers(contents[i], userPaths[i].path, userPaths[i].level);
if (servers.length > 0) {
items.push(...servers);
break;
@@ -57,6 +57,7 @@ function parseMCPServers(
const serverConfig = config as Record<string, unknown>;
items.push({
name,
enabled: typeof serverConfig.enabled === "boolean" ? serverConfig.enabled : undefined,
command: serverConfig.command as string | undefined,
args: serverConfig.args as string[] | undefined,
env: serverConfig.env as Record<string, string> | undefined,
@@ -86,15 +87,17 @@ async function loadMCPServers(ctx: LoadContext): Promise<LoadResult<MCPServer>>
const projectContentPromise = projectPath ? readFile(projectPath) : Promise.resolve(null);
if (userContent && userPath) {
const result = parseMCPServers(userContent, userPath, "user");
// Load project entries before user entries so a project `enabled: false`
// claims its dedupe key before a same-named user server can survive (#7654).
const projectContent = await projectContentPromise;
if (projectContent && projectPath) {
const result = parseMCPServers(projectContent, projectPath, "project");
items.push(...result.items);
if (result.warning) warnings.push(result.warning);
}
const projectContent = await projectContentPromise;
if (projectContent && projectPath) {
const result = parseMCPServers(projectContent, projectPath, "project");
if (userContent && userPath) {
const result = parseMCPServers(userContent, userPath, "user");
items.push(...result.items);
if (result.warning) warnings.push(result.warning);
}
+11 -8
View File
@@ -48,14 +48,8 @@ async function loadMCPServers(ctx: LoadContext): Promise<LoadResult<MCPServer>>
const items: MCPServer[] = [];
const warnings: string[] = [];
// User-level: ~/.gemini/settings.json → mcpServers
const userPath = getUserPath(ctx, "gemini", "settings.json");
if (userPath) {
const result = await loadMCPFromSettings(ctx, userPath, "user");
items.push(...result.items);
if (result.warnings) warnings.push(...result.warnings);
}
// Load project entries before user entries so a project `enabled: false`
// claims its dedupe key before a same-named user server can survive (#7654).
// Project-level: .gemini/settings.json → mcpServers
const projectPath = getProjectPath(ctx, "gemini", "settings.json");
if (projectPath) {
@@ -64,6 +58,14 @@ async function loadMCPServers(ctx: LoadContext): Promise<LoadResult<MCPServer>>
if (result.warnings) warnings.push(...result.warnings);
}
// User-level: ~/.gemini/settings.json → mcpServers
const userPath = getUserPath(ctx, "gemini", "settings.json");
if (userPath) {
const result = await loadMCPFromSettings(ctx, userPath, "user");
items.push(...result.items);
if (result.warnings) warnings.push(...result.warnings);
}
return { items, warnings };
}
@@ -102,6 +104,7 @@ async function loadMCPFromSettings(
items.push({
name,
enabled: typeof raw.enabled === "boolean" ? raw.enabled : undefined,
command: typeof raw.command === "string" ? raw.command : undefined,
args: Array.isArray(raw.args) ? (raw.args as string[]) : undefined,
env: raw.env && typeof raw.env === "object" ? (raw.env as Record<string, string>) : undefined,
@@ -83,6 +83,7 @@ async function loadMCPConfig(
const server: MCPServer = {
name,
enabled: typeof expanded.enabled === "boolean" ? expanded.enabled : undefined,
command: typeof expanded.command === "string" ? expanded.command : undefined,
args: Array.isArray(expanded.args) ? (expanded.args as string[]) : undefined,
env: expanded.env && typeof expanded.env === "object" ? (expanded.env as Record<string, string>) : undefined,
@@ -48,6 +48,7 @@ function parseServerConfig(
return {
server: {
name,
enabled: typeof server.enabled === "boolean" ? server.enabled : undefined,
command: server.command as string | undefined,
args: server.args as string[] | undefined,
env: server.env as Record<string, string> | undefined,
@@ -71,10 +72,11 @@ async function loadMCPServers(ctx: LoadContext): Promise<LoadResult<MCPServer>>
]);
const projectContent = projectPath ? await readFile(projectPath) : null;
// Load project entries before user entries so a project `enabled: false`
// claims its dedupe key before a same-named user server can survive (#7654).
const configs: Array<{ content: string | null; path: string | null; scope: "user" | "project" }> = [
{ content: userContent, path: userPath, scope: "user" },
{ content: projectContent, path: projectPath, scope: "project" },
{ content: userContent, path: userPath, scope: "user" },
];
for (const { content, path, scope } of configs) {
@@ -87,11 +87,11 @@ export class HashlineFilesystem extends Filesystem {
return resolvePlanPath(this.session, relativePath);
}
canonicalPath(relativePath: string): string {
override canonicalPath(relativePath: string): string {
return canonicalSnapshotKey(this.resolveAbsolute(relativePath));
}
allowTagPathRecovery(authoredPath: string, resolvedPath: string): boolean {
override allowTagPathRecovery(authoredPath: string, resolvedPath: string): boolean {
// Internal-URL authored targets (`local://`, `vault://`, …) are approved
// at the lower "read" privilege; never let one redirect onto a "write".
if (isInternalUrlPath(authoredPath)) return false;
@@ -125,7 +125,7 @@ export class HashlineFilesystem extends Filesystem {
return content;
}
async readBinary(relativePath: string): Promise<Uint8Array | undefined> {
override async readBinary(relativePath: string): Promise<Uint8Array | undefined> {
const absolutePath = this.resolveAbsolute(relativePath);
if (isNotebookPath(absolutePath)) return undefined;
try {
@@ -136,7 +136,7 @@ export class HashlineFilesystem extends Filesystem {
}
}
async preflightWrite(relativePath: string, options?: PreflightWriteOptions): Promise<void> {
override async preflightWrite(relativePath: string, options?: PreflightWriteOptions): Promise<void> {
const fileOp = options?.fileOp;
if (fileOp?.kind === "rem") {
enforcePlanModeWrite(this.session, relativePath, { op: "delete" });
@@ -149,7 +149,7 @@ export class HashlineFilesystem extends Filesystem {
enforcePlanModeWrite(this.session, relativePath, { op: "update" });
}
async delete(relativePath: string): Promise<void> {
override async delete(relativePath: string): Promise<void> {
enforcePlanModeWrite(this.session, relativePath, { op: "delete" });
const absolutePath = this.resolveAbsolute(relativePath);
try {
@@ -168,7 +168,7 @@ export class HashlineFilesystem extends Filesystem {
invalidateFsScanAfterWrite(absolutePath);
}
async move(fromRelative: string, toRelative: string, content?: string): Promise<void> {
override async move(fromRelative: string, toRelative: string, content?: string): Promise<void> {
enforcePlanModeWrite(this.session, fromRelative, { op: "update", move: toRelative });
const fromAbsolute = this.resolveAbsolute(fromRelative);
const toAbsolute = this.resolveAbsolute(toRelative);
@@ -240,7 +240,7 @@ export class HashlineFilesystem extends Filesystem {
return { text: content };
}
async exists(relativePath: string): Promise<boolean> {
override async exists(relativePath: string): Promise<boolean> {
const absolutePath = this.resolveAbsolute(relativePath);
return Bun.file(absolutePath).exists();
}
@@ -23,7 +23,6 @@ export const EVAL_AGENT_BRIDGE_NAME = "__agent__";
const agentArgsSchema = type({
prompt: "string>0",
"agent?": "string>0",
"model?": "string>0|string>0[]",
"label?": "string",
"schema?": "unknown",
"schemaMode?": "'permissive' | 'strict'",
@@ -31,12 +30,12 @@ const agentArgsSchema = type({
"apply?": "boolean",
"merge?": "boolean",
"handle?": "boolean",
"+": "delete",
});
interface EvalAgentArgs {
prompt: string;
agent?: string;
model?: string | string[];
label?: string;
schema?: unknown;
schemaMode?: StructuredSubagentSchemaMode;
@@ -148,7 +147,6 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption
invocationKind: "eval",
assignment: parsed.prompt,
...(parsed.agent !== undefined ? { agent: parsed.agent } : {}),
...(parsed.model !== undefined ? { model: parsed.model } : {}),
...(Object.hasOwn(parsed, "schema") ? { outputSchema: parsed.schema } : {}),
...(parsed.schemaMode !== undefined ? { schemaMode: parsed.schemaMode } : {}),
...(parsed.label !== undefined ? { identity: { label: parsed.label } } : {}),
+4 -4
View File
@@ -519,14 +519,11 @@ function completion(prompt::String; model="default", system=nothing, schema=noth
return schema === nothing ? text : Main.json_parse(string(text))
end
function agent(prompt::String; agent="task", model=nothing, label=nothing, schema=nothing, schema_mode=nothing, isolated=nothing, apply=nothing, merge=nothing, handle=false, kwargs...)
function agent(prompt::String; agent="task", label=nothing, schema=nothing, schema_mode=nothing, isolated=nothing, apply=nothing, merge=nothing, handle=false, kwargs...)
args_dict = Dict{String, Any}("prompt" => prompt)
if agent !== nothing
args_dict["agent"] = agent
end
if model !== nothing
args_dict["model"] = model
end
if label !== nothing
args_dict["label"] = label
end
@@ -545,6 +542,9 @@ function agent(prompt::String; agent="task", model=nothing, label=nothing, schem
if merge !== nothing
args_dict["merge"] = Bool(merge)
end
if haskey(kwargs, :model)
error("agent() no longer accepts a per-call model override; the selected agent's frontmatter model is used")
end
handle_result = handle
for (k, v) in kwargs
args_dict[string(k)] = v
@@ -104,8 +104,8 @@ if (!globalThis.__omp_js_prelude_loaded__) {
"agent",
opts,
rest,
["agent", "model", "label", "schema", "isolated", "apply", "merge", "schemaMode"],
"{ agent, model, label, schema, isolated, apply, merge, schemaMode, handle }",
["agent", "label", "schema", "isolated", "apply", "merge", "schemaMode"],
"{ agent, label, schema, isolated, apply, merge, schemaMode, handle }",
);
const { handle, ...callArgs } = o;
const res = await globalThis.__omp_call_tool__("__agent__", { prompt, ...callArgs, handle: Boolean(handle) });
@@ -488,7 +488,6 @@ if "__omp_prelude_loaded__" not in globals():
prompt,
*,
agent="task",
model=None,
label=None,
schema=None,
schema_mode=None,
@@ -506,8 +505,6 @@ if "__omp_prelude_loaded__" not in globals():
args = {"prompt": prompt}
if agent is not None:
args["agent"] = agent
if model is not None:
args["model"] = model
if label is not None:
args["label"] = label
if schema is not None:
+1 -2
View File
@@ -392,10 +392,9 @@ unless defined?($__omp_prelude_loaded) && $__omp_prelude_loaded
schema.nil? ? text : JSON.parse(text)
end
def agent(prompt, agent: "task", model: nil, label: nil, schema: nil, schema_mode: nil, isolated: nil, apply: nil, merge: nil, handle: false)
def agent(prompt, agent: "task", label: nil, schema: nil, schema_mode: nil, isolated: nil, apply: nil, merge: nil, handle: false)
args = { "prompt" => prompt }
args["agent"] = agent unless agent.nil?
args["model"] = model unless model.nil?
args["label"] = label unless label.nil?
args["schema"] = schema unless schema.nil?
args["schemaMode"] = schema_mode unless schema_mode.nil?
@@ -54,6 +54,7 @@ import { ReadTool } from "../tools/read";
import { formatBytes } from "../tools/render-utils";
import { WriteTool } from "../tools/write";
import { EventBus } from "../utils/event-bus";
import { convertImageToPng } from "../utils/image-loading";
import { discoverExtensionPaths, loadExtensionFromFactory, loadExtensions } from "./extensions";
import { ExtensionRuntime } from "./extensions/loader";
import type { ExtensionFactory, ToolDefinition } from "./extensions/types";
@@ -384,6 +385,28 @@ async function executeLegacyBashOperations(
}
}
/**
* Convert an image attachment to PNG using the legacy package-root contract.
*
* Invalid or unsupported image data returns `null`, matching Pi's historical
* helper instead of surfacing Bun's decoder error to extensions.
*/
export async function convertToPng(
base64Data: string,
mimeType: string,
): Promise<{ data: string; mimeType: string } | null> {
if (mimeType === "image/png") {
return { data: base64Data, mimeType };
}
try {
const converted = await convertImageToPng({ type: "image", data: base64Data, mimeType });
return { data: converted.data, mimeType: converted.mimeType };
} catch {
return null;
}
}
/** Format the active shortcut for legacy extensions that render keybinding hints. */
export function keyText(action: Keybinding): string {
return formatKeyHints(getKeybindings().getKeys(action));
+2 -22
View File
@@ -411,24 +411,6 @@ export class HindsightSessionState {
}
}
async maybeRecallOnAgentStart(): Promise<void> {
if (!this.config.autoRecall || this.hasRecalledForFirstTurn) return;
const messages = extractMessages(this.session.sessionManager);
const lastUser = messages.findLast(m => m.role === "user");
if (!lastUser) return;
const query = composeRecallQuery(lastUser.content, messages, this.config.recallContextTurns);
const truncated = truncateRecallQuery(query, lastUser.content, this.config.recallMaxQueryChars);
const { context, ok } = await this.recallForContext(truncated);
if (!ok) return;
this.hasRecalledForFirstTurn = true;
if (!context) return;
this.lastRecallSnippet = context;
await this.#refreshBaseSystemPromptAfter("recall");
}
async beforeAgentStartPrompt(promptText: string): Promise<string | undefined> {
if (this.config.mentalModelsEnabled && this.mentalModelsLoadPromise && this.mentalModelsLoadedAt === undefined) {
await Promise.race([this.mentalModelsLoadPromise, Bun.sleep(MENTAL_MODEL_FIRST_TURN_DEADLINE_MS)]);
@@ -509,9 +491,7 @@ export class HindsightSessionState {
attachSessionListeners(): void {
this.unsubscribe?.();
this.unsubscribe = this.session.subscribe(event => {
if (event.type === "agent_start") {
void this.maybeRecallOnAgentStart();
} else if (event.type === "agent_end") {
if (event.type === "agent_end") {
void this.maybeRetainOnAgentEnd();
// Drain any queued tool-initiated retain calls now that the turn
// is settled. The queue is also debounced/size-bounded, but
@@ -540,7 +520,7 @@ export class HindsightSessionState {
this.retainQueue.dispose();
}
async #refreshBaseSystemPromptAfter(reason: "recall" | "MM load" | "MM reload" | "MM TTL reload"): Promise<void> {
async #refreshBaseSystemPromptAfter(reason: "MM load" | "MM reload" | "MM TTL reload"): Promise<void> {
try {
await this.session.refreshBaseSystemPrompt();
} catch (err) {
@@ -222,6 +222,18 @@ function mnemopiSessionStatesFromRegistry(): MnemopiSessionState[] {
return states;
}
function memoryBackendFromContext(context?: ResolveContext): string | undefined {
if (!context?.settings || typeof context.settings !== "object") return undefined;
try {
const get = Reflect.get(context.settings, "get");
if (typeof get !== "function") return undefined;
const backend = Reflect.apply(get, context.settings, ["memory.backend"]);
return typeof backend === "string" ? backend : undefined;
} catch {
return undefined;
}
}
/**
* Look up a mnemopi memory row by id across every live session's scoped banks.
* First hit wins; returns `null` when the id is not stored anywhere in scope.
@@ -290,6 +302,23 @@ export class MemoryProtocolHandler implements ProtocolHandler {
// clipped recall preview before overwriting it (issue #4443).
if (namespace !== MEMORY_NAMESPACE) {
const mnemopiStates = mnemopiSessionStatesFromRegistry();
const hindsightActive =
memoryBackendFromContext(context) === "hindsight" ||
(mnemopiStates.length === 0 &&
AgentRegistry.global()
.list()
.some(ref => ref.session?.getHindsightSessionState?.()));
if (hindsightActive) {
// Hindsight keeps memories server-side and exposes no
// `memory://<id>` addressing, yet the shared `recall` tool
// description still steers a follow-up `read memory://<id>`.
// Return a corrective pointer so that stray read self-corrects in
// one turn instead of derailing on the generic namespace error
// (issue #7587).
throw new Error(
"Hindsight memories are not addressable via memory://. Recall results are final — use `recall` to search or `reflect` to synthesize. `read memory://<id>` is only available with memory.backend=mnemopi.",
);
}
if (mnemopiStates.length === 0) {
throw new Error(
`Unknown memory namespace: ${namespace}. Supported: ${MEMORY_NAMESPACE} (file-backed memory summary), or a mnemopi memory id when memory.backend=mnemopi is active.`,
+2 -2
View File
@@ -17,7 +17,7 @@ import type {
ServerConfig,
WorkspaceEdit,
} from "./types";
import { detectLanguageId, fileToUri } from "./utils";
import { detectLanguageId, EquivalentUriMap, fileToUri } from "./utils";
// =============================================================================
// Client State
@@ -787,7 +787,7 @@ export async function getOrCreateClient(
proc,
config,
requestId: 0,
diagnostics: new Map(),
diagnostics: new EquivalentUriMap(),
diagnosticsVersion: 0,
dynamicCapabilityRegistrations: new Map(),
openFiles: new Map(),

Some files were not shown because too many files have changed in this diff Show More