Merge branch 'main' into fix/xai-oauth-credits-rotation
This commit is contained in:
@@ -622,10 +622,11 @@ jobs:
|
||||
with:
|
||||
node-version: "24"
|
||||
registry-url: "https://registry.npmjs.org"
|
||||
# Keep npm aligned with trusted publishing setup (>= 11.16.0).
|
||||
# npm runs under Bun when invoked by the release script; npm 12
|
||||
# requires a newer emulated Node version than Bun 1.3 provides.
|
||||
- name: Ensure npm supports trusted publishing
|
||||
if: ${{ !inputs.skip_npm }}
|
||||
run: npm install -g npm@latest
|
||||
run: npm install -g npm@11.17.0
|
||||
- name: Cache bun dependencies
|
||||
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
@@ -775,9 +776,10 @@ jobs:
|
||||
with:
|
||||
node-version: "24"
|
||||
registry-url: "https://registry.npmjs.org"
|
||||
# Keep npm aligned with trusted publishing setup (>= 11.16.0).
|
||||
# npm runs under Bun when invoked by the release script; npm 12
|
||||
# requires a newer emulated Node version than Bun 1.3 provides.
|
||||
- name: Ensure npm supports trusted publishing
|
||||
run: npm install -g npm@latest
|
||||
run: npm install -g npm@11.17.0
|
||||
- name: Cache bun dependencies
|
||||
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
|
||||
@@ -72,7 +72,7 @@ For each candidate issue, read the title, body, and **all comments** (comments o
|
||||
| `providers` | Provider-related behavior (generic provider scope) |
|
||||
|
||||
**Provider labels** (apply only when a specific provider is explicitly involved):
|
||||
`provider:anthropic`, `provider:bedrock`, `provider:brave`, `provider:cerebras`, `provider:cloudflare`, `provider:codex`, `provider:copilot`, `provider:cursor`, `provider:exa`, `provider:gemini`, `provider:gitlab`, `provider:groq`, `provider:huggingface`, `provider:jina`, `provider:kimi`, `provider:litellm`, `provider:minimax`, `provider:mistral`, `provider:moonshot`, `provider:nanogpt`, `provider:nvidia`, `provider:openai`, `provider:opencode`, `provider:openrouter`, `provider:perplexity`, `provider:qianfan`, `provider:qwen`, `provider:synthetic`, `provider:together`, `provider:venice`, `provider:vercel`, `provider:xai`, `provider:xiaomi`, `provider:zai`
|
||||
`provider:anthropic`, `provider:bedrock`, `provider:brave`, `provider:cerebras`, `provider:cloudflare`, `provider:codex`, `provider:copilot`, `provider:cursor`, `provider:exa`, `provider:gemini`, `provider:gitlab`, `provider:groq`, `provider:huggingface`, `provider:jina`, `provider:kimi`, `provider:litellm`, `provider:minimax`, `provider:mistral`, `provider:moonshot`, `provider:nanogpt`, `provider:novita`, `provider:nvidia`, `provider:openai`, `provider:opencode`, `provider:openrouter`, `provider:perplexity`, `provider:qianfan`, `provider:qwen`, `provider:synthetic`, `provider:together`, `provider:venice`, `provider:vercel`, `provider:xai`, `provider:xiaomi`, `provider:zai`
|
||||
|
||||
**Platform labels** (apply only when platform materially affects reproduction/root cause):
|
||||
| Label | Signals |
|
||||
|
||||
Generated
+24
-23
@@ -1850,9 +1850,9 @@ checksum = "b9e0384b61958566e926dc50660321d12159025e767c18e043daf26b70104c39"
|
||||
|
||||
[[package]]
|
||||
name = "ignore"
|
||||
version = "0.4.27"
|
||||
version = "0.4.28"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fe112b004901c62c2faa11f4f75e9864e0cc5af8da71c9115d184a3aa888749f"
|
||||
checksum = "2adf14691c72bcfc1058740436a35bdd3ae9c07d1a941ef00b749e9ea16aefa7"
|
||||
dependencies = [
|
||||
"crossbeam-deque",
|
||||
"globset",
|
||||
@@ -1951,9 +1951,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "inotify"
|
||||
version = "0.11.3"
|
||||
version = "0.11.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "dd854a95a4ac672fed8c054136039fd32c22cf039ff09ead7280afe920486483"
|
||||
checksum = "153be1941a183ec9ccd095ddbe17a8b8d435ef6c76e9e02451b933c3999af2c8"
|
||||
dependencies = [
|
||||
"bitflags 2.13.0",
|
||||
"inotify-sys",
|
||||
@@ -2026,9 +2026,9 @@ checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
|
||||
|
||||
[[package]]
|
||||
name = "jiff"
|
||||
version = "0.2.31"
|
||||
version = "0.2.32"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ccfe6121cbe750cf81efa362d85c0bde7ea298ec43092d3a193baca59cdbd634"
|
||||
checksum = "961d16382652bfdd8c6f68b223b26a8c93e0d475c672f414411db31c6c5c900e"
|
||||
dependencies = [
|
||||
"defmt",
|
||||
"jiff-static",
|
||||
@@ -2053,9 +2053,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "jiff-static"
|
||||
version = "0.2.31"
|
||||
version = "0.2.32"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e165e897f662d428f3cd3828a919dbe067c2d42bb1031eede74ef9d27ecdedd2"
|
||||
checksum = "d0879bd39df99c4c5e2c6615ccc026391a423dde10532c573e6086eb94a802cc"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -2064,9 +2064,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "jiff-tzdb"
|
||||
version = "0.1.7"
|
||||
version = "0.1.8"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6142247df1a93c2b3587402a19710be3e6e942f1581a1702e76408f2c21d6590"
|
||||
checksum = "142bd39932ad231f10513df9ab62661fead8719872150b7ad02a2df79f4e141e"
|
||||
|
||||
[[package]]
|
||||
name = "jiff-tzdb-platform"
|
||||
@@ -2881,7 +2881,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pi-ast"
|
||||
version = "16.3.12"
|
||||
version = "16.3.15"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"ast-grep-core",
|
||||
@@ -2950,7 +2950,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pi-iso"
|
||||
version = "16.3.12"
|
||||
version = "16.3.15"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"libc",
|
||||
@@ -2962,13 +2962,14 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pi-natives"
|
||||
version = "16.3.12"
|
||||
version = "16.3.15"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"arboard",
|
||||
"ast-grep-core",
|
||||
"base64",
|
||||
"clap",
|
||||
"clipboard-win",
|
||||
"flume",
|
||||
"fontdue",
|
||||
"globset",
|
||||
@@ -3014,7 +3015,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pi-shell"
|
||||
version = "16.3.12"
|
||||
version = "16.3.15"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"brush-builtins",
|
||||
@@ -3063,7 +3064,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pi-walker"
|
||||
version = "16.3.12"
|
||||
version = "16.3.15"
|
||||
dependencies = [
|
||||
"dashmap",
|
||||
"globset",
|
||||
@@ -3403,9 +3404,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "regex"
|
||||
version = "1.12.4"
|
||||
version = "1.13.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f1292b7759ae1cb9ec195452d1390a074f0cd8541ab7a5a8c31cd6db45d4a6ba"
|
||||
checksum = "2a0e75113e14dc5acb068cd0786884f214f1312650a3d36d269f5c4f3cdee8a2"
|
||||
dependencies = [
|
||||
"aho-corasick",
|
||||
"memchr",
|
||||
@@ -3415,9 +3416,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "regex-automata"
|
||||
version = "0.4.14"
|
||||
version = "0.4.15"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6e1dd4122fc1595e8162618945476892eefca7b88c52820e74af6262213cae8f"
|
||||
checksum = "1f388202e4b80542a0921078cc23b6333bcf1409c1e3f86404cae4766a6131db"
|
||||
dependencies = [
|
||||
"aho-corasick",
|
||||
"memchr",
|
||||
@@ -5765,18 +5766,18 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "zerocopy"
|
||||
version = "0.8.53"
|
||||
version = "0.8.54"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "75726053136156d419e285b9b7eddaaea9e3fea6ce32eed44a89901f0bd98de1"
|
||||
checksum = "b7cbbc0a705a0fd05cc3676525980d2bf5a9bc4adac6d6475209a7887cf59d19"
|
||||
dependencies = [
|
||||
"zerocopy-derive",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zerocopy-derive"
|
||||
version = "0.8.53"
|
||||
version = "0.8.54"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4714fd92cf900833d49538023a9b3915155210801d1c1169eba513b2addefd71"
|
||||
checksum = "e2e817b7b52d0c7358d3246da9d69935ebb18116b2b102b4230dac079b4862f5"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
|
||||
+2
-1
@@ -4,7 +4,7 @@ exclude = ["crates/vendor/brush-core", "crates/vendor/brush-builtins"]
|
||||
resolver = "3"
|
||||
|
||||
[workspace.package]
|
||||
version = "16.3.12"
|
||||
version = "16.3.15"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
authors = ["Can Boluk"]
|
||||
@@ -269,6 +269,7 @@ napi-derive = "3"
|
||||
# Terminal & PTY
|
||||
# ──────────────────────────────────────────────────────────────────────────────
|
||||
arboard = { version = "3.6.1", features = ["wayland-data-control"] }
|
||||
clipboard-win = "5.4"
|
||||
icy_sixel = "0.5"
|
||||
portable-pty = "0.9"
|
||||
|
||||
|
||||
@@ -292,7 +292,7 @@ Anthropic `oauth` · OpenAI · OpenAI Codex `oauth` · Google Gemini · Google A
|
||||
|
||||
Subscription-routed. `/login` attaches the session.
|
||||
|
||||
Cursor `oauth` · GitHub Copilot `oauth` · GitLab Duo · Kimi Code `plan` · Moonshot · MiniMax Coding Plan `plan` · MiniMax Coding Plan CN `plan` · Alibaba Coding Plan `plan` · Qwen Portal · Z.AI / GLM Coding Plan `plan` · Xiaomi MiMo · Qianfan · NanoGPT · Venice · Kilo · ZenMux · OpenCode Go · OpenCode Zen
|
||||
Cursor `oauth` · GitHub Copilot `oauth` · GitLab Duo · Kimi Code `plan` · Moonshot · MiniMax Coding Plan `plan` · MiniMax Coding Plan CN `plan` · Alibaba Coding Plan `plan` · Qwen Portal · Z.AI / GLM Coding Plan `plan` · Xiaomi MiMo · Qianfan · NanoGPT · Novita · Venice · Kilo · ZenMux · OpenCode Go · OpenCode Zen
|
||||
|
||||
### Run it yourself
|
||||
|
||||
|
||||
@@ -21,7 +21,7 @@
|
||||
},
|
||||
"packages/agent": {
|
||||
"name": "@oh-my-pi/pi-agent-core",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"dependencies": {
|
||||
"@oh-my-pi/pi-ai": "catalog:",
|
||||
"@oh-my-pi/pi-catalog": "catalog:",
|
||||
@@ -39,7 +39,7 @@
|
||||
},
|
||||
"packages/ai": {
|
||||
"name": "@oh-my-pi/pi-ai",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"dependencies": {
|
||||
"@bufbuild/protobuf": "catalog:",
|
||||
"@oh-my-pi/pi-catalog": "catalog:",
|
||||
@@ -55,7 +55,7 @@
|
||||
},
|
||||
"packages/catalog": {
|
||||
"name": "@oh-my-pi/pi-catalog",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"dependencies": {
|
||||
"@bufbuild/protobuf": "catalog:",
|
||||
"@oh-my-pi/pi-utils": "catalog:",
|
||||
@@ -69,7 +69,7 @@
|
||||
},
|
||||
"packages/coding-agent": {
|
||||
"name": "@oh-my-pi/pi-coding-agent",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"bin": {
|
||||
"omp": "src/cli.ts",
|
||||
},
|
||||
@@ -137,7 +137,7 @@
|
||||
},
|
||||
"packages/hashline": {
|
||||
"name": "@oh-my-pi/hashline",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"dependencies": {
|
||||
"diff": "catalog:",
|
||||
"lru-cache": "catalog:",
|
||||
@@ -148,7 +148,7 @@
|
||||
},
|
||||
"packages/mnemopi": {
|
||||
"name": "@oh-my-pi/pi-mnemopi",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"bin": {
|
||||
"mnemopi": "src/cli.ts",
|
||||
},
|
||||
@@ -174,7 +174,7 @@
|
||||
},
|
||||
"packages/natives": {
|
||||
"name": "@oh-my-pi/pi-natives",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"devDependencies": {
|
||||
"@napi-rs/cli": "catalog:",
|
||||
"@types/bun": "catalog:",
|
||||
@@ -182,7 +182,7 @@
|
||||
},
|
||||
"packages/snapcompact": {
|
||||
"name": "@oh-my-pi/snapcompact",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"dependencies": {
|
||||
"@oh-my-pi/pi-ai": "catalog:",
|
||||
"@oh-my-pi/pi-natives": "catalog:",
|
||||
@@ -195,7 +195,7 @@
|
||||
},
|
||||
"packages/stats": {
|
||||
"name": "@oh-my-pi/omp-stats",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"bin": {
|
||||
"omp-stats": "./src/index.ts",
|
||||
},
|
||||
@@ -221,7 +221,7 @@
|
||||
},
|
||||
"packages/swarm-extension": {
|
||||
"name": "@oh-my-pi/swarm-extension",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"bin": {
|
||||
"omp-swarm": "src/cli.ts",
|
||||
},
|
||||
@@ -247,7 +247,7 @@
|
||||
},
|
||||
"packages/tui": {
|
||||
"name": "@oh-my-pi/pi-tui",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"dependencies": {
|
||||
"@oh-my-pi/pi-natives": "catalog:",
|
||||
"@oh-my-pi/pi-utils": "catalog:",
|
||||
@@ -288,7 +288,7 @@
|
||||
},
|
||||
"packages/utils": {
|
||||
"name": "@oh-my-pi/pi-utils",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"dependencies": {
|
||||
"@oh-my-pi/pi-natives": "catalog:",
|
||||
"handlebars": "catalog:",
|
||||
@@ -301,7 +301,7 @@
|
||||
},
|
||||
"packages/wire": {
|
||||
"name": "@oh-my-pi/pi-wire",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"devDependencies": {
|
||||
"@types/bun": "catalog:",
|
||||
},
|
||||
@@ -338,18 +338,18 @@
|
||||
"@huggingface/transformers": "^4.2.0",
|
||||
"@mozilla/readability": "^0.6.0",
|
||||
"@napi-rs/cli": "3.7.0",
|
||||
"@oh-my-pi/hashline": "16.3.12",
|
||||
"@oh-my-pi/omp-stats": "16.3.12",
|
||||
"@oh-my-pi/pi-agent-core": "16.3.12",
|
||||
"@oh-my-pi/pi-ai": "16.3.12",
|
||||
"@oh-my-pi/pi-catalog": "16.3.12",
|
||||
"@oh-my-pi/pi-coding-agent": "16.3.12",
|
||||
"@oh-my-pi/pi-mnemopi": "16.3.12",
|
||||
"@oh-my-pi/pi-natives": "16.3.12",
|
||||
"@oh-my-pi/pi-tui": "16.3.12",
|
||||
"@oh-my-pi/pi-utils": "16.3.12",
|
||||
"@oh-my-pi/pi-wire": "16.3.12",
|
||||
"@oh-my-pi/snapcompact": "16.3.12",
|
||||
"@oh-my-pi/hashline": "16.3.15",
|
||||
"@oh-my-pi/omp-stats": "16.3.15",
|
||||
"@oh-my-pi/pi-agent-core": "16.3.15",
|
||||
"@oh-my-pi/pi-ai": "16.3.15",
|
||||
"@oh-my-pi/pi-catalog": "16.3.15",
|
||||
"@oh-my-pi/pi-coding-agent": "16.3.15",
|
||||
"@oh-my-pi/pi-mnemopi": "16.3.15",
|
||||
"@oh-my-pi/pi-natives": "16.3.15",
|
||||
"@oh-my-pi/pi-tui": "16.3.15",
|
||||
"@oh-my-pi/pi-utils": "16.3.15",
|
||||
"@oh-my-pi/pi-wire": "16.3.15",
|
||||
"@oh-my-pi/snapcompact": "16.3.15",
|
||||
"@opentelemetry/api": "^1.9.1",
|
||||
"@opentelemetry/context-async-hooks": "^2.7.1",
|
||||
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
|
||||
@@ -791,7 +791,7 @@
|
||||
|
||||
"@opentelemetry/sdk-trace-node": ["@opentelemetry/sdk-trace-node@2.9.0", "", { "dependencies": { "@opentelemetry/context-async-hooks": "2.9.0", "@opentelemetry/core": "2.9.0", "@opentelemetry/sdk-trace-base": "2.9.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-ec9a7ps37huy5itYk0MalaZdSLlM6AXWp/FhtEjgMpp5leEGojBDvAl/UWttQnkMZOvFHKzRESn8TD3yKTF5nQ=="],
|
||||
|
||||
"@opentelemetry/semantic-conventions": ["@opentelemetry/semantic-conventions@1.41.1", "", {}, "sha512-/UhIkaZgPutTFmQ7RnIJGgDXZmtEJ7Dvi86xNTFWcnRxVRNk/aotsqDJYeEvDP+FSMB2SdW+pQzNMcWP0rwuNA=="],
|
||||
"@opentelemetry/semantic-conventions": ["@opentelemetry/semantic-conventions@1.42.0", "", {}, "sha512-icc5xCzndZfhuJMy5oqk5AvloWquR7jtae74qzpkKkhGp8BivK+oCcEXgGnjCdTfp8hA44l+w8gE8yYJbocJJw=="],
|
||||
|
||||
"@oxc-project/types": ["@oxc-project/types@0.138.0", "", {}, "sha512-1a7ZKmrRTCoN1XMZ4L0PyyqrMnrNlLyPuOkdSX2MZg7IiIGRUyurNhAm73ptDOraoBcIordsIGKNPKUzy3ZmfA=="],
|
||||
|
||||
@@ -811,7 +811,7 @@
|
||||
|
||||
"@protobufjs/pool": ["@protobufjs/pool@1.1.0", "", {}, "sha512-0kELaGSIDBKvcgS4zkjz1PeddatrjYcmMWOlAuAPwAeccUrPHdUqo/J6LiymHHEiJT5NrF1UVwxY14f+fy4WQw=="],
|
||||
|
||||
"@protobufjs/utf8": ["@protobufjs/utf8@1.1.1", "", {}, "sha512-oOAWABowe8EAbMyWKM0tYDKi8Yaox52D+HWZhAIJqQXbqe0xI/GV7FhLWqlEKreMkfDjshR5FKgi3mnle0h6Eg=="],
|
||||
"@protobufjs/utf8": ["@protobufjs/utf8@1.1.2", "", {}, "sha512-b1UQwcEZ4yCnMCD8DAL1VlbvBJE9/IX4FTIp7BG1xYpf29SLazLSrqUkj4w7Y5y7cCVP6E5tcqqcI0xemPkHug=="],
|
||||
|
||||
"@puppeteer/browsers": ["@puppeteer/browsers@3.0.6", "", { "dependencies": { "modern-tar": "^0.7.6", "yargs": "^18.0.0" }, "peerDependencies": { "proxy-agent": ">=8.0.1", "yauzl": "^2.10.0 || ^3.4.0" }, "optionalPeers": ["proxy-agent", "yauzl"], "bin": { "browsers": "lib/main-cli.js" } }, "sha512-B/gKoqlFkzhvzsI6jo9K1cZz9o5ypviVv/xu8CwA4grZzyVwN+XfkT+tu8T1zrauuEXv6VhS2oGX+6NL95WcKA=="],
|
||||
|
||||
@@ -963,7 +963,7 @@
|
||||
|
||||
"brace-expansion": ["brace-expansion@5.0.7", "", { "dependencies": { "balanced-match": "^4.0.2" } }, "sha512-7oFy703dxfY3/NLxC1fh2SUCQ0H9rmAY+5EpDVfXjUTTs+HEwR2nYaqLv+GWcTsumwxPfiz6CzCNkwXwBUwqCA=="],
|
||||
|
||||
"browserslist": ["browserslist@4.28.4", "", { "dependencies": { "baseline-browser-mapping": "^2.10.38", "caniuse-lite": "^1.0.30001799", "electron-to-chromium": "^1.5.376", "node-releases": "^2.0.48", "update-browserslist-db": "^1.2.3" }, "bin": { "browserslist": "cli.js" } }, "sha512-MTc8i/x9jBQd1iMw2CFGS+rwMa07eYjLR0CCTLDACl9xhxy+nIs3KeML/biicXtk9JrZ6dnnTatmc7ErPXIxqw=="],
|
||||
"browserslist": ["browserslist@4.28.5", "", { "dependencies": { "baseline-browser-mapping": "^2.10.42", "caniuse-lite": "^1.0.30001800", "electron-to-chromium": "^1.5.387", "node-releases": "^2.0.50", "update-browserslist-db": "^1.2.3" }, "bin": { "browserslist": "cli.js" } }, "sha512-Cu2E6QejHWzuDMTkuwgpABFgDfZrXLQq5V13YOACZx4mFAG4IwGTbTfHPMr4WtxlHoXSM8FIuRwYYCz5XiabaQ=="],
|
||||
|
||||
"bun-types": ["bun-types@1.3.14", "", { "dependencies": { "@types/node": "*" } }, "sha512-4N0ig0fEomHt5R0KCFWjovxow98rIoRwKolrYdCcknNwMekCXRnWEUvgu5soYV8QXtVsrUD8B95MBOZGPvr6KQ=="],
|
||||
|
||||
|
||||
@@ -26,7 +26,7 @@ grep-searcher.workspace = true
|
||||
html-to-markdown-rs.workspace = true
|
||||
icy_sixel.workspace = true
|
||||
ignore.workspace = true
|
||||
image.workspace = true
|
||||
image = { workspace = true, features = ["bmp"] }
|
||||
inferno.workspace = true
|
||||
memmap2.workspace = true
|
||||
napi.workspace = true
|
||||
@@ -61,6 +61,7 @@ libc.workspace = true
|
||||
|
||||
[target.'cfg(windows)'.dependencies]
|
||||
windows-sys = { workspace = true, features = ["Wdk_Storage_FileSystem", "Win32_Security"] }
|
||||
clipboard-win.workspace = true
|
||||
winreg.workspace = true
|
||||
|
||||
[build-dependencies]
|
||||
|
||||
@@ -30,7 +30,14 @@ fn encode_png(image: ImageData<'_>) -> Result<Vec<u8>> {
|
||||
let bytes = image.bytes.into_owned();
|
||||
let buffer = RgbaImage::from_raw(width, height, bytes)
|
||||
.ok_or_else(|| Error::from_reason("Clipboard image buffer size mismatch"))?;
|
||||
let capacity = width.saturating_mul(height).saturating_mul(4) as usize;
|
||||
rgba_to_png(buffer)
|
||||
}
|
||||
|
||||
fn rgba_to_png(buffer: RgbaImage) -> Result<Vec<u8>> {
|
||||
let capacity = (buffer
|
||||
.width()
|
||||
.saturating_mul(buffer.height())
|
||||
.saturating_mul(4)) as usize;
|
||||
let mut output = Vec::with_capacity(capacity);
|
||||
DynamicImage::ImageRgba8(buffer)
|
||||
.write_to(&mut Cursor::new(&mut output), ImageFormat::Png)
|
||||
@@ -38,6 +45,88 @@ fn encode_png(image: ImageData<'_>) -> Result<Vec<u8>> {
|
||||
Ok(output)
|
||||
}
|
||||
|
||||
/// Decode a packed DIB clipboard payload (`CF_DIB`: a `BITMAPINFOHEADER`-family
|
||||
/// header, optional bitfield masks and palette, then the pixel array) into PNG
|
||||
/// bytes.
|
||||
///
|
||||
/// The payload is wrapped in a synthesized `BITMAPFILEHEADER` and decoded
|
||||
/// through the BMP *file* path so the explicit `bfOffBits` pins the pixel
|
||||
/// offset. This matters: the header-less decode path arboard uses mis-places
|
||||
/// the pixel offset for V4/V5 headers with `BI_BITFIELDS` compression (it
|
||||
/// skips 12 trailing mask bytes that those headers embed instead), which is
|
||||
/// why Qt-based screenshot tools (`PixPin`, `Snipaste`, ...) fail through
|
||||
/// arboard in the first place (#3426).
|
||||
#[cfg_attr(
|
||||
not(windows),
|
||||
allow(
|
||||
dead_code,
|
||||
reason = "reached only by the Windows clipboard fallback; kept target-independent so unit \
|
||||
tests cover it on every host"
|
||||
)
|
||||
)]
|
||||
fn dib_to_png(dib: &[u8]) -> Result<Vec<u8>> {
|
||||
const FILE_HEADER_SIZE: u64 = 14;
|
||||
const INFO_HEADER_SIZE: u64 = 40;
|
||||
const BI_BITFIELDS: u32 = 3;
|
||||
|
||||
if dib.len() < INFO_HEADER_SIZE as usize {
|
||||
return Err(Error::from_reason("Clipboard DIB shorter than BITMAPINFOHEADER"));
|
||||
}
|
||||
let u32_at =
|
||||
|at: usize| u32::from_le_bytes(dib[at..at + 4].try_into().expect("bounds checked above"));
|
||||
let header_size = u64::from(u32_at(0));
|
||||
if header_size < INFO_HEADER_SIZE || header_size > dib.len() as u64 {
|
||||
return Err(Error::from_reason("Clipboard DIB header size out of range"));
|
||||
}
|
||||
let bit_count = u16::from_le_bytes([dib[14], dib[15]]);
|
||||
let compression = u32_at(16);
|
||||
let colors_used = u64::from(u32_at(32));
|
||||
|
||||
// A plain BITMAPINFOHEADER with BI_BITFIELDS is trailed by three DWORD
|
||||
// masks; larger (V2..V5) headers embed the masks in the header itself.
|
||||
let mask_bytes: u64 = if header_size == INFO_HEADER_SIZE && compression == BI_BITFIELDS {
|
||||
12
|
||||
} else {
|
||||
0
|
||||
};
|
||||
let palette_entries: u64 = if colors_used != 0 {
|
||||
colors_used
|
||||
} else if bit_count <= 8 {
|
||||
1u64 << bit_count
|
||||
} else {
|
||||
0
|
||||
};
|
||||
let pixel_offset =
|
||||
u32::try_from(FILE_HEADER_SIZE + header_size + mask_bytes + palette_entries * 4)
|
||||
.map_err(|_| Error::from_reason("Clipboard DIB layout overflow"))?;
|
||||
let file_size = u32::try_from(FILE_HEADER_SIZE + dib.len() as u64)
|
||||
.map_err(|_| Error::from_reason("Clipboard DIB too large"))?;
|
||||
|
||||
let mut bmp = Vec::with_capacity(FILE_HEADER_SIZE as usize + dib.len());
|
||||
bmp.extend_from_slice(b"BM");
|
||||
bmp.extend_from_slice(&file_size.to_le_bytes());
|
||||
bmp.extend_from_slice(&0u32.to_le_bytes());
|
||||
bmp.extend_from_slice(&pixel_offset.to_le_bytes());
|
||||
bmp.extend_from_slice(dib);
|
||||
|
||||
let decoded = image::load_from_memory_with_format(&bmp, ImageFormat::Bmp)
|
||||
.map_err(|err| Error::from_reason(format!("Failed to decode clipboard DIB: {err}")))?;
|
||||
rgba_to_png(decoded.into_rgba8())
|
||||
}
|
||||
|
||||
/// Read the raw `CF_DIB` bytes from the Windows clipboard.
|
||||
///
|
||||
/// Windows synthesizes `CF_DIB` from whatever bitmap formats are present, so
|
||||
/// it is available whenever the clipboard holds any image at all.
|
||||
#[cfg(windows)]
|
||||
fn read_raw_cf_dib() -> Option<Vec<u8>> {
|
||||
let clip = clipboard_win::Clipboard::new_attempts(10).ok()?;
|
||||
let mut dib = Vec::new();
|
||||
clipboard_win::raw::get_vec(clipboard_win::formats::CF_DIB, &mut dib).ok()?;
|
||||
drop(clip);
|
||||
(!dib.is_empty()).then_some(dib)
|
||||
}
|
||||
|
||||
/// Copy plain text to the system clipboard.
|
||||
///
|
||||
/// # Parameters
|
||||
@@ -120,7 +209,160 @@ pub fn read_image_from_clipboard() -> task::Promise<Option<ClipboardImage>> {
|
||||
}))
|
||||
},
|
||||
Err(ClipboardError::ContentNotAvailable) => Ok(None),
|
||||
Err(err) => Err(Error::from_reason(format!("Failed to read clipboard image: {err}"))),
|
||||
Err(err) => {
|
||||
// arboard rejects the CF_DIBV5 payloads Qt-based screenshot
|
||||
// tools (PixPin, Snipaste, ...) produce; decode the raw CF_DIB
|
||||
// ourselves before surfacing the error (#3426). A fallback
|
||||
// decode failure keeps the original arboard error.
|
||||
#[cfg(windows)]
|
||||
if let Some(bytes) = read_raw_cf_dib().and_then(|dib| dib_to_png(&dib).ok()) {
|
||||
return Ok(Some(ClipboardImage {
|
||||
data: Uint8Array::from(bytes),
|
||||
mime_type: "image/png".to_string(),
|
||||
}));
|
||||
}
|
||||
Err(Error::from_reason(format!("Failed to read clipboard image: {err}")))
|
||||
},
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::dib_to_png;
|
||||
|
||||
fn push32(v: u32, out: &mut Vec<u8>) {
|
||||
out.extend_from_slice(&v.to_le_bytes());
|
||||
}
|
||||
|
||||
fn push16(v: u16, out: &mut Vec<u8>) {
|
||||
out.extend_from_slice(&v.to_le_bytes());
|
||||
}
|
||||
|
||||
/// 2x2 bottom-up BGRA pixel array: memory rows are [red, green] (bottom)
|
||||
/// then [blue, white] (top), all with alpha 0xff.
|
||||
const PIXELS_2X2: [u8; 16] = [
|
||||
0x00, 0x00, 0xff, 0xff, // (0,1) red
|
||||
0x00, 0xff, 0x00, 0xff, // (1,1) green
|
||||
0xff, 0x00, 0x00, 0xff, // (0,0) blue
|
||||
0xff, 0xff, 0xff, 0xff, // (1,0) white
|
||||
];
|
||||
|
||||
/// `CF_DIB` as Qt's clipboard writer emits it for 32-bit content: a plain
|
||||
/// `BITMAPINFOHEADER` with `BI_BITFIELDS` compression and three DWORD
|
||||
/// masks between header and pixels.
|
||||
fn qt_cf_dib(width: u32, height: u32, pixels_bgra: &[u8], compression: u32) -> Vec<u8> {
|
||||
let mut d = Vec::with_capacity(52 + pixels_bgra.len());
|
||||
push32(40, &mut d); // biSize
|
||||
push32(width, &mut d);
|
||||
push32(height, &mut d); // positive: bottom-up
|
||||
push16(1, &mut d); // biPlanes
|
||||
push16(32, &mut d); // biBitCount
|
||||
push32(compression, &mut d);
|
||||
push32(pixels_bgra.len() as u32, &mut d); // biSizeImage
|
||||
push32(0, &mut d); // biXPelsPerMeter
|
||||
push32(0, &mut d); // biYPelsPerMeter
|
||||
push32(0, &mut d); // biClrUsed
|
||||
push32(0, &mut d); // biClrImportant
|
||||
if compression == 3 {
|
||||
push32(0x00ff_0000, &mut d); // red mask
|
||||
push32(0x0000_ff00, &mut d); // green mask
|
||||
push32(0x0000_00ff, &mut d); // blue mask
|
||||
}
|
||||
d.extend_from_slice(pixels_bgra);
|
||||
d
|
||||
}
|
||||
|
||||
/// `CF_DIBV5` as PixPin (Qt) places it, after arboard's
|
||||
/// `maybe_tweak_header` rewrite: a 124-byte `BITMAPV5HEADER` carrying
|
||||
/// `BI_BITFIELDS` compression with the BGRA masks embedded in the header
|
||||
/// and pixels immediately after it. This is the exact buffer shape that
|
||||
/// arboard's header-less BMP decode rejects with `ConversionFailure`
|
||||
/// (issue #3426); the file-header wrap must decode it.
|
||||
fn pixpin_dibv5_tweaked(width: u32, height: u32, pixels_bgra: &[u8]) -> Vec<u8> {
|
||||
let mut d = Vec::with_capacity(124 + pixels_bgra.len());
|
||||
push32(124, &mut d); // bV5Size
|
||||
push32(width, &mut d);
|
||||
push32(height, &mut d);
|
||||
push16(1, &mut d); // bV5Planes
|
||||
push16(32, &mut d); // bV5BitCount
|
||||
push32(3, &mut d); // bV5Compression = BI_BITFIELDS (arboard-tweaked)
|
||||
push32(0, &mut d); // bV5SizeImage
|
||||
push32(0, &mut d); // bV5XPelsPerMeter
|
||||
push32(0, &mut d); // bV5YPelsPerMeter
|
||||
push32(0, &mut d); // bV5ClrUsed
|
||||
push32(0, &mut d); // bV5ClrImportant
|
||||
push32(0x00ff_0000, &mut d); // bV5RedMask
|
||||
push32(0x0000_ff00, &mut d); // bV5GreenMask
|
||||
push32(0x0000_00ff, &mut d); // bV5BlueMask
|
||||
push32(0xff00_0000, &mut d); // bV5AlphaMask
|
||||
push32(0x7352_4742, &mut d); // bV5CSType = LCS_sRGB
|
||||
d.extend_from_slice(&[0u8; 36]); // bV5Endpoints
|
||||
push32(0, &mut d); // bV5GammaRed
|
||||
push32(0, &mut d); // bV5GammaGreen
|
||||
push32(0, &mut d); // bV5GammaBlue
|
||||
push32(4, &mut d); // bV5Intent = LCS_GM_IMAGES
|
||||
push32(0, &mut d); // bV5ProfileData
|
||||
push32(0, &mut d); // bV5ProfileSize
|
||||
push32(0, &mut d); // bV5Reserved
|
||||
assert_eq!(d.len(), 124);
|
||||
d.extend_from_slice(pixels_bgra);
|
||||
d
|
||||
}
|
||||
|
||||
fn decode_pixels(png: &[u8]) -> (u32, u32, Vec<[u8; 4]>) {
|
||||
let img = image::load_from_memory(png).expect("fallback output must be valid PNG");
|
||||
let rgba = img.into_rgba8();
|
||||
let (w, h) = rgba.dimensions();
|
||||
let px = rgba.pixels().map(|p| p.0).collect();
|
||||
(w, h, px)
|
||||
}
|
||||
|
||||
const RED: [u8; 4] = [255, 0, 0, 255];
|
||||
const GREEN: [u8; 4] = [0, 255, 0, 255];
|
||||
const BLUE: [u8; 4] = [0, 0, 255, 255];
|
||||
const WHITE: [u8; 4] = [255, 255, 255, 255];
|
||||
|
||||
#[test]
|
||||
fn decodes_qt_cf_dib_with_bitfields_masks() {
|
||||
let dib = qt_cf_dib(2, 2, &PIXELS_2X2, 3);
|
||||
let png = dib_to_png(&dib).expect("BI_BITFIELDS CF_DIB must decode");
|
||||
let (w, h, px) = decode_pixels(&png);
|
||||
assert_eq!((w, h), (2, 2));
|
||||
// Row order flipped versus the bottom-up pixel array; BGRA -> RGBA.
|
||||
assert_eq!(px, vec![BLUE, WHITE, RED, GREEN]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decodes_pixpin_dibv5_payload_that_arboard_rejects() {
|
||||
let dib = pixpin_dibv5_tweaked(2, 2, &PIXELS_2X2);
|
||||
let png = dib_to_png(&dib).expect("V5 BI_BITFIELDS DIB must decode");
|
||||
let (w, h, px) = decode_pixels(&png);
|
||||
assert_eq!((w, h), (2, 2));
|
||||
assert_eq!(px, vec![BLUE, WHITE, RED, GREEN]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decodes_plain_bi_rgb_dib() {
|
||||
// The common "copy image" payload: BI_RGB, 32-bit, no masks. The
|
||||
// fourth byte is unused per the DIB contract — zero it to prove the
|
||||
// decode still yields opaque pixels.
|
||||
let mut pixels = PIXELS_2X2;
|
||||
for alpha in pixels.iter_mut().skip(3).step_by(4) {
|
||||
*alpha = 0;
|
||||
}
|
||||
let dib = qt_cf_dib(2, 2, &pixels, 0);
|
||||
let png = dib_to_png(&dib).expect("BI_RGB CF_DIB must decode");
|
||||
let (w, h, px) = decode_pixels(&png);
|
||||
assert_eq!((w, h), (2, 2));
|
||||
assert_eq!(px, vec![BLUE, WHITE, RED, GREEN]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_malformed_dib() {
|
||||
assert!(dib_to_png(&[0u8; 12]).is_err(), "short buffer must not decode");
|
||||
let mut oversized_header = qt_cf_dib(2, 2, &PIXELS_2X2, 3);
|
||||
oversized_header[0..4].copy_from_slice(&0xffff_ffffu32.to_le_bytes());
|
||||
assert!(dib_to_png(&oversized_header).is_err(), "header size beyond buffer must not decode");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -248,7 +248,7 @@ fn create_windows_napi_tokio_runtime() -> Option<tokio::runtime::Runtime> {
|
||||
/// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in
|
||||
/// `packages/natives/native/index.js` (which derives the name from
|
||||
/// `package.json#version`).
|
||||
#[napi(js_name = "__piNativesV16_3_12")]
|
||||
#[napi(js_name = "__piNativesV16_3_15")]
|
||||
pub const fn pi_natives_version_sentinel() {}
|
||||
|
||||
/// Native module entry point: install crash diagnostics before any tool can
|
||||
|
||||
+132
-33
@@ -5,7 +5,7 @@ use std::{collections::HashMap, sync::Arc};
|
||||
use napi::{
|
||||
Env, Result,
|
||||
bindgen_prelude::*,
|
||||
threadsafe_function::{ThreadsafeFunction, ThreadsafeFunctionCallMode},
|
||||
threadsafe_function::{ThreadsafeFunction, UnknownReturnValue},
|
||||
};
|
||||
use napi_derive::napi;
|
||||
use pi_shell::{
|
||||
@@ -216,7 +216,7 @@ impl Shell {
|
||||
env: &'env Env,
|
||||
options: ShellRunOptions<'env>,
|
||||
#[napi(ts_arg_type = "((error: Error | null, chunk: string) => void) | undefined | null")]
|
||||
on_chunk: Option<ThreadsafeFunction<String>>,
|
||||
on_chunk: Option<ThreadsafeFunction<String, UnknownReturnValue>>,
|
||||
) -> Result<PromiseRaw<'env, ShellRunResult>> {
|
||||
let cancel_token = task::CancelToken::new(options.timeout_ms, options.signal);
|
||||
let inner = Arc::clone(&self.inner);
|
||||
@@ -269,7 +269,7 @@ pub fn execute_shell<'env>(
|
||||
env: &'env Env,
|
||||
options: ShellExecuteOptions<'env>,
|
||||
#[napi(ts_arg_type = "((error: Error | null, chunk: string) => void) | undefined | null")]
|
||||
on_chunk: Option<ThreadsafeFunction<String>>,
|
||||
on_chunk: Option<ThreadsafeFunction<String, UnknownReturnValue>>,
|
||||
) -> Result<PromiseRaw<'env, ShellRunResult>> {
|
||||
let cancel_token = task::CancelToken::new(options.timeout_ms, options.signal);
|
||||
let exec_options = CoreShellExecuteOptions {
|
||||
@@ -294,42 +294,66 @@ pub fn execute_shell<'env>(
|
||||
})
|
||||
}
|
||||
|
||||
/// Capacity (in chunks) of the queue between the pipe readers and the JS
|
||||
/// forwarding pump. One queued chunk is at most one pipe read (≤64 KiB), so
|
||||
/// the Rust side of the bridge holds ~4 MiB worst case before the readers'
|
||||
/// `send_async` parks — which in turn parks the child on its stdout/stderr
|
||||
/// pipe (ordinary pipe backpressure) instead of buffering the surplus in
|
||||
/// process memory (#4078).
|
||||
const BRIDGE_QUEUE_CHUNKS: usize = 64;
|
||||
|
||||
fn bridge_chunks(
|
||||
on_chunk: Option<ThreadsafeFunction<String>>,
|
||||
on_chunk: Option<ThreadsafeFunction<String, UnknownReturnValue>>,
|
||||
) -> (Option<flume::Sender<String>>, Option<napi::tokio::task::JoinHandle<()>>) {
|
||||
let Some(on_chunk) = on_chunk else {
|
||||
return (None, None);
|
||||
};
|
||||
let (tx, rx) = flume::unbounded::<String>();
|
||||
let handle = napi::tokio::spawn(async move {
|
||||
// Hard cap on one coalesced batch so the JS main thread never sees a
|
||||
// multi-MB napi callback (a giant single string would stall sanitize +
|
||||
// tail-buffer maintenance for the whole copy).
|
||||
const MAX_BATCH_BYTES: usize = 64 * 1024;
|
||||
// Initial capacity sized for typical bursty pipe output. Re-allocated
|
||||
// each batch because `String` ownership is moved into the napi call.
|
||||
const INITIAL_BATCH_CAP: usize = 8 * 1024;
|
||||
let mut batch = String::with_capacity(INITIAL_BATCH_CAP);
|
||||
while let Ok(first) = rx.recv_async().await {
|
||||
batch.push_str(&first);
|
||||
// Greedily drain everything already queued. Child processes that
|
||||
// write byte-at-a-time (printf-style progress, llama-cli token
|
||||
// streams) otherwise produce one napi callback per `write(2)`,
|
||||
// saturating the JS main thread (~200% CPU observed) and leaving
|
||||
// the queue draining long after the child exits.
|
||||
while batch.len() < MAX_BATCH_BYTES {
|
||||
match rx.try_recv() {
|
||||
Ok(more) => batch.push_str(&more),
|
||||
Err(_) => break,
|
||||
}
|
||||
}
|
||||
let payload = std::mem::replace(&mut batch, String::with_capacity(INITIAL_BATCH_CAP));
|
||||
on_chunk.call(Ok(payload), ThreadsafeFunctionCallMode::NonBlocking);
|
||||
}
|
||||
});
|
||||
let (tx, rx) = flume::bounded::<String>(BRIDGE_QUEUE_CHUNKS);
|
||||
let handle = napi::tokio::spawn(pump_chunks(rx, async move |payload: String| {
|
||||
// `call_async` resolves only after the JS callback ran, so at most
|
||||
// one batch sits in the napi queue at a time and the JS event loop's
|
||||
// actual consumption rate backpressures the whole pipeline. An error
|
||||
// means the JS side is gone (env teardown) — stop forwarding.
|
||||
on_chunk.call_async(Ok(payload)).await.is_ok()
|
||||
}));
|
||||
(Some(tx), Some(handle))
|
||||
}
|
||||
|
||||
/// Drain `rx`, greedily coalescing queued chunks into ≤64 KiB batches, and
|
||||
/// feed each batch to `forward`, awaiting its completion before pulling more.
|
||||
/// Returns when `rx` disconnects (all senders dropped) or `forward` reports
|
||||
/// the consumer is gone; dropping `rx` then disconnects the channel so
|
||||
/// parked/future senders fail fast and the pipe readers keep draining the
|
||||
/// child instead of wedging it.
|
||||
async fn pump_chunks(rx: flume::Receiver<String>, mut forward: impl AsyncFnMut(String) -> bool) {
|
||||
// Hard cap on one coalesced batch so the JS main thread never sees a
|
||||
// multi-MB napi callback (a giant single string would stall sanitize +
|
||||
// tail-buffer maintenance for the whole copy).
|
||||
const MAX_BATCH_BYTES: usize = 64 * 1024;
|
||||
// Initial capacity sized for typical bursty pipe output. Re-allocated
|
||||
// each batch because `String` ownership is moved into the napi call.
|
||||
const INITIAL_BATCH_CAP: usize = 8 * 1024;
|
||||
let mut batch = String::with_capacity(INITIAL_BATCH_CAP);
|
||||
while let Ok(first) = rx.recv_async().await {
|
||||
batch.push_str(&first);
|
||||
// Greedily drain everything already queued. Child processes that
|
||||
// write byte-at-a-time (printf-style progress, llama-cli token
|
||||
// streams) otherwise produce one napi callback per `write(2)`,
|
||||
// saturating the JS main thread (~200% CPU observed) and leaving
|
||||
// the queue draining long after the child exits.
|
||||
while batch.len() < MAX_BATCH_BYTES {
|
||||
match rx.try_recv() {
|
||||
Ok(more) => batch.push_str(&more),
|
||||
Err(_) => break,
|
||||
}
|
||||
}
|
||||
let payload = std::mem::replace(&mut batch, String::with_capacity(INITIAL_BATCH_CAP));
|
||||
if !forward(payload).await {
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Result of [`apply_bash_fixups`]: a possibly-rewritten command plus the
|
||||
/// substrings that were removed (in source order).
|
||||
#[napi(object)]
|
||||
@@ -360,7 +384,6 @@ pub fn apply_bash_fixups(command: String) -> BashFixupResult {
|
||||
mod tests {
|
||||
use std::time::Duration;
|
||||
|
||||
#[cfg(unix)]
|
||||
use flume;
|
||||
use pi_shell::{
|
||||
ShellRunOptions as CoreShellRunOptions,
|
||||
@@ -368,7 +391,83 @@ mod tests {
|
||||
};
|
||||
use tokio::time;
|
||||
|
||||
use super::CoreShell;
|
||||
use super::{BRIDGE_QUEUE_CHUNKS, CoreShell, pump_chunks};
|
||||
|
||||
/// Regression for #4078: the reader→JS bridge queue must stay bounded when
|
||||
/// the JS side (here: a deliberately slow `forward`) cannot keep up with a
|
||||
/// fast producer, and backpressure must never drop or reorder chunks. On
|
||||
/// the pre-fix bridge (`flume::unbounded` + fire-and-forget
|
||||
/// `ThreadsafeFunctionCallMode::NonBlocking`) the same harness accumulates
|
||||
/// the producer's entire surplus in the queue (measured: a 32 MiB stream
|
||||
/// queued all 33_554_432 bytes while the consumer stalled).
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
async fn bridge_pump_bounds_queue_and_delivers_all_bytes() {
|
||||
const CHUNKS: usize = 512;
|
||||
const CHUNK_BYTES: usize = 4096;
|
||||
let (tx, rx) = flume::bounded::<String>(BRIDGE_QUEUE_CHUNKS);
|
||||
let producer = tokio::spawn(async move {
|
||||
let mut expected = String::with_capacity(CHUNKS * CHUNK_BYTES);
|
||||
let mut max_queued = 0usize;
|
||||
for i in 0..CHUNKS {
|
||||
let chunk = format!("[{i:06}]{}", "x".repeat(CHUNK_BYTES - 8));
|
||||
expected.push_str(&chunk);
|
||||
tx.send_async(chunk)
|
||||
.await
|
||||
.expect("pump should outlive the producer");
|
||||
max_queued = max_queued.max(tx.len());
|
||||
}
|
||||
(expected, max_queued)
|
||||
});
|
||||
|
||||
let mut received = String::with_capacity(CHUNKS * CHUNK_BYTES);
|
||||
time::timeout(
|
||||
Duration::from_secs(30),
|
||||
pump_chunks(rx, async |payload: String| {
|
||||
received.push_str(&payload);
|
||||
// Emulate a busy JS event loop: each napi callback takes a while.
|
||||
time::sleep(Duration::from_micros(500)).await;
|
||||
true
|
||||
}),
|
||||
)
|
||||
.await
|
||||
.expect("pump should finish once the producer hangs up");
|
||||
|
||||
let (expected, max_queued) = producer.await.expect("producer task");
|
||||
assert!(
|
||||
max_queued <= BRIDGE_QUEUE_CHUNKS,
|
||||
"bridge queue grew past its bound: {max_queued} chunks",
|
||||
);
|
||||
assert_eq!(received.len(), expected.len(), "bytes were dropped or duplicated");
|
||||
assert_eq!(received, expected, "chunks must arrive losslessly and in order");
|
||||
}
|
||||
|
||||
/// When the JS side dies (`forward` fails: threadsafe function aborted on
|
||||
/// env teardown), the pump must drop its receiver so parked and future
|
||||
/// sends fail fast — the pipe readers keep draining the child instead of
|
||||
/// wedging it on a full bridge queue.
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
async fn bridge_pump_death_disconnects_channel_without_blocking_senders() {
|
||||
let (tx, rx) = flume::bounded::<String>(4);
|
||||
let pump = tokio::spawn(pump_chunks(rx, async |_payload: String| false));
|
||||
let producer = tokio::spawn(async move {
|
||||
let mut disconnected = 0usize;
|
||||
for _ in 0..64 {
|
||||
if tx.send_async("x".repeat(1024)).await.is_err() {
|
||||
disconnected += 1;
|
||||
}
|
||||
}
|
||||
disconnected
|
||||
});
|
||||
let disconnected = time::timeout(Duration::from_secs(5), producer)
|
||||
.await
|
||||
.expect("sends must not park once the consumer died")
|
||||
.expect("producer task");
|
||||
assert!(disconnected > 0, "channel should disconnect after the pump stops");
|
||||
time::timeout(Duration::from_secs(5), pump)
|
||||
.await
|
||||
.expect("pump should exit after forward fails")
|
||||
.expect("pump task");
|
||||
}
|
||||
|
||||
mod child_session_action_tests {
|
||||
use pi_shell::{ChildSessionAction, child_session_action};
|
||||
|
||||
@@ -1649,7 +1649,7 @@ async fn read_output(
|
||||
let pending = &buf[..it];
|
||||
match str::from_utf8(pending) {
|
||||
Ok(text) => {
|
||||
emit_chunk(text, on_chunk.as_ref());
|
||||
emit_chunk(text, on_chunk.as_ref()).await;
|
||||
it = 0;
|
||||
break;
|
||||
},
|
||||
@@ -1658,7 +1658,7 @@ async fn read_output(
|
||||
if p > 0 {
|
||||
// SAFETY: [..p] is guaranteed valid UTF-8 by valid_up_to().
|
||||
let text = unsafe { str::from_utf8_unchecked(&pending[..p]) };
|
||||
emit_chunk(text, on_chunk.as_ref());
|
||||
emit_chunk(text, on_chunk.as_ref()).await;
|
||||
// copy p..it to the beginning of the buffer
|
||||
buf.copy_within(p..it, 0);
|
||||
it -= p;
|
||||
@@ -1667,7 +1667,7 @@ async fn read_output(
|
||||
match err.error_len() {
|
||||
Some(p) => {
|
||||
// Invalid byte sequence: emit replacement and drop those bytes.
|
||||
emit_chunk(REPLACEMENT, on_chunk.as_ref());
|
||||
emit_chunk(REPLACEMENT, on_chunk.as_ref()).await;
|
||||
// copy p..it to the beginning of the buffer
|
||||
buf.copy_within(p..it, 0);
|
||||
it -= p;
|
||||
@@ -1688,10 +1688,10 @@ async fn read_output(
|
||||
for chunk in buf[..it].utf8_chunks() {
|
||||
let valid = chunk.valid();
|
||||
if !valid.is_empty() {
|
||||
emit_chunk(valid, on_chunk.as_ref());
|
||||
emit_chunk(valid, on_chunk.as_ref()).await;
|
||||
}
|
||||
if !chunk.invalid().is_empty() {
|
||||
emit_chunk(REPLACEMENT, on_chunk.as_ref());
|
||||
emit_chunk(REPLACEMENT, on_chunk.as_ref()).await;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1777,7 +1777,7 @@ async fn read_output_buffered(
|
||||
while !pending.is_empty() {
|
||||
match str::from_utf8(&pending) {
|
||||
Ok(text) => {
|
||||
emit_chunk(text, Some(cb));
|
||||
emit_chunk(text, Some(cb)).await;
|
||||
pending.clear();
|
||||
break;
|
||||
},
|
||||
@@ -1786,12 +1786,12 @@ async fn read_output_buffered(
|
||||
if p > 0 {
|
||||
// SAFETY: [..p] is valid UTF-8 per valid_up_to().
|
||||
let text = unsafe { str::from_utf8_unchecked(&pending[..p]) };
|
||||
emit_chunk(text, Some(cb));
|
||||
emit_chunk(text, Some(cb)).await;
|
||||
pending.drain(..p);
|
||||
}
|
||||
match err.error_len() {
|
||||
Some(skip) => {
|
||||
emit_chunk(REPLACEMENT, Some(cb));
|
||||
emit_chunk(REPLACEMENT, Some(cb)).await;
|
||||
pending.drain(..skip);
|
||||
},
|
||||
None => break,
|
||||
@@ -1807,10 +1807,10 @@ async fn read_output_buffered(
|
||||
for chunk in pending.utf8_chunks() {
|
||||
let valid = chunk.valid();
|
||||
if !valid.is_empty() {
|
||||
emit_chunk(valid, Some(cb));
|
||||
emit_chunk(valid, Some(cb)).await;
|
||||
}
|
||||
if !chunk.invalid().is_empty() {
|
||||
emit_chunk(REPLACEMENT, Some(cb));
|
||||
emit_chunk(REPLACEMENT, Some(cb)).await;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1858,9 +1858,16 @@ fn read_nonblocking<T: std::os::fd::AsRawFd>(file: &T, buf: &mut [u8]) -> io::Re
|
||||
}
|
||||
}
|
||||
|
||||
fn emit_chunk(text: &str, callback: Option<&Sender<String>>) {
|
||||
/// Forward one decoded chunk to the streaming callback, honouring channel
|
||||
/// backpressure: on a bounded channel (the pi-natives JS bridge) the send
|
||||
/// parks until the consumer frees a slot — which parks the pipe reader and,
|
||||
/// transitively, the child on its stdout/stderr pipe — so a fast producer
|
||||
/// can never buffer unbounded output in memory (#4078). A disconnected
|
||||
/// receiver (consumer gone) fails immediately, so the pipe keeps draining
|
||||
/// and the child never wedges on a full pipe.
|
||||
async fn emit_chunk(text: &str, callback: Option<&Sender<String>>) {
|
||||
if let Some(callback) = callback {
|
||||
let _ = callback.send(text.to_string());
|
||||
let _ = callback.send_async(text.to_string()).await;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4140,4 +4147,39 @@ replace = [{ pattern = "^.+$", replacement = "PWD" }]
|
||||
"builtin nohup masked SIGHUP like the external tool (output: {out:?})",
|
||||
);
|
||||
}
|
||||
|
||||
/// Regression for #4078: the JS bridge hands the pipe readers a *bounded*
|
||||
/// chunk channel. With a consumer slower than the producer the readers
|
||||
/// must park on `send_async` (backpressuring the child through its pipe)
|
||||
/// rather than buffer unboundedly — and, unlike a drop-on-full design,
|
||||
/// every produced byte must still reach the consumer.
|
||||
#[cfg(unix)]
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
async fn streaming_output_backpressures_on_bounded_channel_without_loss() {
|
||||
const TOTAL_BYTES: usize = 1_048_576;
|
||||
let (tx, rx) = flume::bounded::<String>(4);
|
||||
let options = ShellExecuteOptions {
|
||||
command: format!("yes x | head -c {TOTAL_BYTES}"),
|
||||
..Default::default()
|
||||
};
|
||||
let run = tokio::spawn(execute_shell(options, Some(tx), CancelToken::default()));
|
||||
|
||||
let mut received = 0usize;
|
||||
while let Ok(chunk) = rx.recv_async().await {
|
||||
received += chunk.len();
|
||||
// Slow consumer: forces the bounded queue to fill and the readers
|
||||
// to park between chunks.
|
||||
time::sleep(Duration::from_micros(50)).await;
|
||||
}
|
||||
|
||||
let result = time::timeout(Duration::from_secs(30), run)
|
||||
.await
|
||||
.expect("command should finish despite backpressure")
|
||||
.expect("run task should not panic")
|
||||
.expect("execute should succeed");
|
||||
assert_eq!(result.exit_code, Some(0));
|
||||
assert!(!result.cancelled);
|
||||
assert!(!result.timed_out);
|
||||
assert_eq!(received, TOTAL_BYTES, "streamed bytes were dropped under backpressure");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -79,6 +79,8 @@ A named profile (`omp --profile <name>`, the `--alias` shortcut, or `OMP_PROFILE
|
||||
|
||||
The relocation is uniform across the native provider (`builtin.ts`) and the generic `config.ts` helpers, so it covers slash commands, rules, prompts, instructions, hooks, tools, extensions, settings, skills, and MCP, plus the top-level `SYSTEM.md` / `RULES.md` / `AGENTS.md` files and runtime state (sessions, blobs, `agent.db`). A profile sees only its own OMP config, never the default profile's `~/.omp/agent`.
|
||||
|
||||
Keybindings are the one exception: a named profile merges the default profile's `~/.omp/agent/keybindings.*` under its own `~/.omp/profiles/<name>/agent/keybindings.*`, with the profile file overriding per binding ([#4867](https://github.com/can1357/oh-my-pi/issues/4867)). Keybindings describe the terminal/keyboard in front of the user, which doesn't change with the active profile, so user-level remaps keep working in every profile unless the profile explicitly overrides them. The inherited file is read-only for the profile process — legacy-format migration of the default profile's file only happens when the default profile itself runs.
|
||||
|
||||
The other source bases are not profile-scoped and load identically under every profile: the external-tool bases (`~/.claude`, `~/.codex`, `~/.gemini`) belong to those tools, and the project-level bases (`<cwd>/.omp`, `<cwd>/.claude`, ...) are keyed to the working directory. Throughout this document, read `~/.omp/agent` as shorthand for the active profile's agent directory.
|
||||
|
||||
## Important constraint
|
||||
|
||||
@@ -49,6 +49,7 @@ These are consumed via `getEnvApiKey()` (`packages/ai/src/stream.ts`) unless not
|
||||
| `SYNTHETIC_API_KEY` | Synthetic auth | Using Synthetic models | |
|
||||
| `NVIDIA_API_KEY` | NVIDIA auth | Using `nvidia` provider | |
|
||||
| `NANO_GPT_API_KEY` | NanoGPT auth | Using `nanogpt` provider | |
|
||||
| `NOVITA_API_KEY` | Novita auth | Using `novita` provider | |
|
||||
| `VENICE_API_KEY` | Venice auth | Using `venice` provider | |
|
||||
| `LITELLM_API_KEY` | LiteLLM auth | Using `litellm` provider | OpenAI-compatible LiteLLM proxy key |
|
||||
| `LM_STUDIO_API_KEY` | LM Studio auth (optional) | Using `lm-studio` provider with authenticated hosts | Local LM Studio usually runs without auth; any non-empty token works when a key is required |
|
||||
|
||||
+4
-3
@@ -142,7 +142,7 @@ Also exposed:
|
||||
- `deliverAs: "nextTurn"` — stored and injected on the next user prompt
|
||||
- `triggerTurn: true` — starts a turn when idle (also honored with `deliverAs: "nextTurn"`: idle prompts immediately; while streaming the queued message schedules an internal continuation)
|
||||
|
||||
`pi.sendUserMessage(content, { deliverAs })` always goes through prompt flow; while streaming it queues as steer/follow-up.
|
||||
`pi.sendUserMessage(content, { deliverAs })` always goes through prompt flow. Omit `deliverAs` to start a normal prompt when idle; while streaming, omitted `deliverAs` queues the message as a steer. Set `deliverAs: "followUp"` to wait until the current run finishes.
|
||||
|
||||
## 2) Handler context (`ExtensionContext`)
|
||||
|
||||
@@ -311,6 +311,7 @@ Supported:
|
||||
|
||||
- dialogs: `select`, `confirm`, `input`, `editor`
|
||||
- input editing: `setEditorText`, `getEditorText`, `pasteToEditor`, `editor`
|
||||
- autocomplete stacking: `addAutocompleteProvider(factory)` wraps the built-in editor provider (factories apply in registration order and re-apply on every slash-command refresh)
|
||||
- terminal title and working message (`setTitle`, `setWorkingMessage`)
|
||||
- notifications/status/editor text/terminal input/custom overlays
|
||||
- theme listing/loading by name (`setTheme` supports string names)
|
||||
@@ -334,7 +335,7 @@ Unsupported/no-op in RPC implementation:
|
||||
|
||||
- `onTerminalInput`
|
||||
- `custom`
|
||||
- `setFooter`, `setHeader`, `setEditorComponent`
|
||||
- `setFooter`, `setHeader`, `setEditorComponent`, `addAutocompleteProvider`
|
||||
- `setWorkingMessage`
|
||||
- theme switching/loading (`setTheme` returns failure)
|
||||
- tool expansion controls are inert
|
||||
@@ -345,7 +346,7 @@ When no UI context is supplied to runner init, `ctx.hasUI` is `false` and method
|
||||
|
||||
### ACP mode
|
||||
|
||||
ACP installs an elicitation-bridged UI context (`createAcpExtensionUiContext` in `acp-agent.ts`). `ctx.hasUI` is `true` while only `select`/`confirm`/`input` round-trip (as ACP elicitations; defaults are returned when the client lacks the `elicitation.form` capability). The non-elicitation surface (widgets, editor, theming, terminal input) is stubbed no-op.
|
||||
ACP installs an elicitation-bridged UI context (`createAcpExtensionUiContext` in `acp-agent.ts`). `ctx.hasUI` is `true` while only `select`/`confirm`/`input` round-trip (as ACP elicitations; defaults are returned when the client lacks the `elicitation.form` capability). The non-elicitation surface (widgets, editor, theming, terminal input, autocomplete stacking) is stubbed no-op.
|
||||
|
||||
## Session and state patterns
|
||||
|
||||
|
||||
@@ -307,6 +307,7 @@ Our fork has architectural decisions that differ from upstream. **Do not port th
|
||||
| `FooterDataProvider` class | `StatusLineComponent` | Simpler, integrated status line |
|
||||
| `ctx.ui.setHeader()` / `ctx.ui.setFooter()` | No-op stubs in current extension contexts | Not currently wired to replace the TUI status/header UI |
|
||||
| `ctx.ui.setEditorComponent()` | Wired in interactive mode; no-op stubs in ACP/RPC/headless contexts | Custom editor replacement works in the interactive TUI; non-TUI runtimes keep stubs |
|
||||
| `ctx.ui.addAutocompleteProvider()` | Wired in interactive mode; no-op stubs in ACP/RPC/headless contexts | Factory wrapping matches upstream; omp's editor has no custom `triggerCharacters`, so wrapped providers surface at the built-in trigger points |
|
||||
| `InteractiveModeOptions` options object | Positional constructor args (options type still exported) | Keep constructor signature; update the type when upstream adds fields |
|
||||
|
||||
### Component Naming
|
||||
|
||||
@@ -105,6 +105,7 @@ Each provider has one or more environment variables that supply a key when no st
|
||||
| `huggingface` | `HUGGINGFACE_HUB_TOKEN`, then `HF_TOKEN` |
|
||||
| `moonshot` | `MOONSHOT_API_KEY` |
|
||||
| `nanogpt` | `NANO_GPT_API_KEY` |
|
||||
| `novita` | `NOVITA_API_KEY` |
|
||||
| `venice` | `VENICE_API_KEY` |
|
||||
| `vercel-ai-gateway` | `AI_GATEWAY_API_KEY` (also `VERCEL_AI_GATEWAY_API_KEY` for catalog discovery) |
|
||||
| `cloudflare-ai-gateway` | `CLOUDFLARE_AI_GATEWAY_API_KEY` |
|
||||
|
||||
+3
-2
@@ -215,8 +215,9 @@ Behavior:
|
||||
|
||||
1. optional command/template expansion (`/` commands, custom commands, file slash commands, prompt templates)
|
||||
2. if currently streaming:
|
||||
- requires `streamingBehavior: "steer" | "followUp"`
|
||||
- queues instead of throwing work away
|
||||
- `streamingBehavior: "steer" | "followUp"` chooses how `prompt()` queues
|
||||
- extension `sendUserMessage(content)` defaults to steer when `deliverAs` is omitted
|
||||
- queued messages are preserved instead of throwing work away
|
||||
3. if idle:
|
||||
- validates model + API key
|
||||
- appends user message
|
||||
|
||||
@@ -172,7 +172,7 @@ Interactive `/fork` creates a new session from the current one and switches the
|
||||
2. Flushes pending writes.
|
||||
3. Calls `SessionManager.fork()`.
|
||||
4. Copies artifacts directory from old session namespace to new namespace (best-effort; non-ENOENT copy failures are logged, not fatal).
|
||||
5. Updates `agent.sessionId`.
|
||||
5. Updates `agent.sessionId` and inherits the previous provider prompt-cache key unless an explicit prompt-cache key is already pinned.
|
||||
6. Emits `session_switch` with `reason: "fork"`.
|
||||
|
||||
`SessionManager.fork()` behavior:
|
||||
@@ -184,6 +184,7 @@ Interactive `/fork` creates a new session from the current one and switches the
|
||||
- new timestamp
|
||||
- `cwd` unchanged
|
||||
- `parentSession` set to previous session id
|
||||
- `providerPromptCacheKey` set to the previous header's inherited key, or the previous session id when none was pinned
|
||||
- Keeps all non-header entries unchanged in the new file.
|
||||
|
||||
### Non-persistent behavior
|
||||
@@ -200,6 +201,9 @@ Startup `--fork` is resolved before normal session creation:
|
||||
2. Path-like values (`/`, `\`, or `.jsonl`) call `SessionManager.forkFrom(path, cwd, sessionDir)`.
|
||||
3. Other values resolve via `resolveResumableSession(...)`: local sessions first, then global search when `sessionDir` is not forced. Matching accepts lowercased session id prefixes, full JSONL filename prefixes, and timestamp-stripped filename id suffixes.
|
||||
4. The forked file is created in the current cwd/session-dir scope and becomes the active session manager for startup.
|
||||
5. Full-context forks automatically seed `providerPromptCacheKey` from the source header's inherited key, falling back to the source session id. Startup drops that automatic inheritance when `--model`, `--thinking`, `--system-prompt`, `--append-system-prompt`, `--tools`, or `--no-tools` changes the provider route or prompt/tool shape.
|
||||
|
||||
Use `--prompt-cache-key <key>` to pin the provider prompt-cache identity explicitly and independently from both the OMP session id and `--provider-session-id`. `--provider-session-id` continues to control provider session/routing headers and sticky credential selection; `--prompt-cache-key` controls the OpenAI Responses `prompt_cache_key` payload where supported.
|
||||
|
||||
## Resume and continue
|
||||
|
||||
|
||||
+12
-12
@@ -25,18 +25,18 @@
|
||||
"@huggingface/transformers": "^4.2.0",
|
||||
"@mozilla/readability": "^0.6.0",
|
||||
"@napi-rs/cli": "3.7.0",
|
||||
"@oh-my-pi/hashline": "16.3.12",
|
||||
"@oh-my-pi/omp-stats": "16.3.12",
|
||||
"@oh-my-pi/pi-agent-core": "16.3.12",
|
||||
"@oh-my-pi/pi-ai": "16.3.12",
|
||||
"@oh-my-pi/pi-catalog": "16.3.12",
|
||||
"@oh-my-pi/pi-coding-agent": "16.3.12",
|
||||
"@oh-my-pi/pi-mnemopi": "16.3.12",
|
||||
"@oh-my-pi/pi-natives": "16.3.12",
|
||||
"@oh-my-pi/pi-tui": "16.3.12",
|
||||
"@oh-my-pi/pi-utils": "16.3.12",
|
||||
"@oh-my-pi/pi-wire": "16.3.12",
|
||||
"@oh-my-pi/snapcompact": "16.3.12",
|
||||
"@oh-my-pi/hashline": "16.3.15",
|
||||
"@oh-my-pi/omp-stats": "16.3.15",
|
||||
"@oh-my-pi/pi-agent-core": "16.3.15",
|
||||
"@oh-my-pi/pi-ai": "16.3.15",
|
||||
"@oh-my-pi/pi-catalog": "16.3.15",
|
||||
"@oh-my-pi/pi-coding-agent": "16.3.15",
|
||||
"@oh-my-pi/pi-mnemopi": "16.3.15",
|
||||
"@oh-my-pi/pi-natives": "16.3.15",
|
||||
"@oh-my-pi/pi-tui": "16.3.15",
|
||||
"@oh-my-pi/pi-utils": "16.3.15",
|
||||
"@oh-my-pi/pi-wire": "16.3.15",
|
||||
"@oh-my-pi/snapcompact": "16.3.15",
|
||||
"@opentelemetry/api": "^1.9.1",
|
||||
"@opentelemetry/context-async-hooks": "^2.7.1",
|
||||
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed remote compaction for Codex Responses Lite models (GPT-5.6 family): both the V1 `/responses/compact` request and the V2 `compaction_trigger` stream now apply the lite rewrite (instructions as an input item, no top-level `instructions`/`tools`, `all_turns` reasoning replay on V2) and send the `x-openai-internal-codex-responses-lite` header, matching codex-rs routing compaction through `build_responses_request`.
|
||||
|
||||
## [16.3.12] - 2026-07-08
|
||||
|
||||
### Added
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"type": "module",
|
||||
"name": "@oh-my-pi/pi-agent-core",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
|
||||
"homepage": "https://omp.sh",
|
||||
"author": "Can Boluk",
|
||||
|
||||
@@ -7,10 +7,16 @@
|
||||
* compaction item as replacement history.
|
||||
*/
|
||||
|
||||
import type { Api, FetchImpl, Model } from "@oh-my-pi/pi-ai";
|
||||
import type { Api, CodexCompactionContext, FetchImpl, Model, ProviderSessionState } from "@oh-my-pi/pi-ai";
|
||||
import { isTransientStatus, ProviderHttpError } from "@oh-my-pi/pi-ai/error";
|
||||
import { applyCodexResponsesLiteShape } from "@oh-my-pi/pi-ai/providers/openai-codex/request-transformer";
|
||||
import {
|
||||
getOpenAIResponsesPromptCacheKey,
|
||||
createOpenAICodexCompactionRequestContext,
|
||||
createOpenAICodexCompatibilityMetadata,
|
||||
type OpenAICodexCompatibilityMetadata,
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-codex-responses";
|
||||
import {
|
||||
getOpenAIPromptCacheKey,
|
||||
getOpenAIResponsesRoutingSessionId,
|
||||
parseAzureDeploymentNameMap,
|
||||
resolveOpenAIRequestSetup,
|
||||
@@ -219,6 +225,8 @@ export async function requestCompactionV2Streaming(
|
||||
fetch?: FetchImpl;
|
||||
timeoutMs?: number;
|
||||
retryWait?: (delayMs: number, signal?: AbortSignal) => Promise<void>;
|
||||
providerSessionState?: Map<string, ProviderSessionState>;
|
||||
codexCompaction?: CodexCompactionContext;
|
||||
},
|
||||
): Promise<CompactionV2Response> {
|
||||
const endpoint = getCompactionV2Endpoint(model);
|
||||
@@ -228,12 +236,32 @@ export async function requestCompactionV2Streaming(
|
||||
|
||||
const fetchImpl = options?.fetch ?? globalThis.fetch;
|
||||
const retryWait = options?.retryWait ?? ((delayMs: number) => Bun.sleep(delayMs));
|
||||
const isCodexResponses = compactionV2Api(model) === "openai-codex-responses" || model.provider === "openai-codex";
|
||||
const codexMetadata = isCodexResponses
|
||||
? createOpenAICodexCompatibilityMetadata({
|
||||
sessionId: request.sessionId,
|
||||
providerSessionState: options?.providerSessionState,
|
||||
requestKind: "compaction",
|
||||
compaction: createOpenAICodexCompactionRequestContext({
|
||||
context: options?.codexCompaction,
|
||||
implementation: "responses_compaction_v2",
|
||||
}),
|
||||
})
|
||||
: undefined;
|
||||
let lastError: Error | undefined;
|
||||
|
||||
for (let attempt = 0; attempt <= V2_COMPACTION_MAX_RETRIES; attempt++) {
|
||||
const timeoutSignal = withRequestTimeout(signal, options?.timeoutMs ?? V2_COMPACTION_TIMEOUT_MS);
|
||||
try {
|
||||
return await attemptCompactionV2Streaming(endpoint, apiKey, model, request, fetchImpl, timeoutSignal);
|
||||
return await attemptCompactionV2Streaming(
|
||||
endpoint,
|
||||
apiKey,
|
||||
model,
|
||||
request,
|
||||
fetchImpl,
|
||||
timeoutSignal,
|
||||
codexMetadata,
|
||||
);
|
||||
} catch (err) {
|
||||
const error = err instanceof Error ? err : new Error(String(err));
|
||||
if (signal?.aborted) throw error;
|
||||
@@ -264,25 +292,41 @@ async function attemptCompactionV2Streaming(
|
||||
request: CompactionV2Request,
|
||||
fetchImpl: FetchImpl,
|
||||
signal?: AbortSignal,
|
||||
codexMetadata?: OpenAICodexCompatibilityMetadata,
|
||||
): Promise<CompactionV2Response> {
|
||||
// Faithful to Codex: append the compaction trigger as the final input item
|
||||
// of an otherwise-normal Responses request, then stream the result. `store`
|
||||
// stays false — compaction must never persist a server-side response object.
|
||||
const cacheOptions = { sessionId: request.sessionId, promptCacheKey: request.promptCacheKey };
|
||||
const promptCacheKey = getOpenAIResponsesPromptCacheKey(cacheOptions);
|
||||
const promptCacheKey = getOpenAIPromptCacheKey(cacheOptions);
|
||||
const body: Record<string, unknown> = {
|
||||
model: request.model,
|
||||
input: [...request.input, COMPACTION_TRIGGER_ITEM],
|
||||
instructions: request.instructions,
|
||||
stream: true,
|
||||
store: false,
|
||||
...(request.reasoning ? { reasoning: request.reasoning, include: ["reasoning.encrypted_content"] } : {}),
|
||||
...(request.reasoning
|
||||
? {
|
||||
// Lite implies gpt-5.4+, where codex-rs sends `all_turns` replay.
|
||||
reasoning: model.useResponsesLite ? { ...request.reasoning, context: "all_turns" } : request.reasoning,
|
||||
include: ["reasoning.encrypted_content"],
|
||||
}
|
||||
: {}),
|
||||
...(promptCacheKey ? { prompt_cache_key: promptCacheKey } : {}),
|
||||
...(request.tools && request.tools.length > 0 ? { tools: request.tools, tool_choice: "auto" } : {}),
|
||||
};
|
||||
if (codexMetadata) {
|
||||
body.client_metadata = codexMetadata.clientMetadata;
|
||||
}
|
||||
// Responses Lite models take the same rewrite on the compaction stream:
|
||||
// instructions/tools ride as input items (codex-rs `compact_remote_v2`
|
||||
// builds through `build_responses_request`).
|
||||
if (model.useResponsesLite) {
|
||||
applyCodexResponsesLiteShape(body);
|
||||
}
|
||||
const response = await fetchImpl(endpoint, {
|
||||
method: "POST",
|
||||
headers: buildCompactionV2Headers(model, apiKey, request),
|
||||
headers: buildCompactionV2Headers(model, apiKey, request, codexMetadata),
|
||||
body: JSON.stringify(body),
|
||||
signal,
|
||||
});
|
||||
@@ -307,11 +351,16 @@ async function attemptCompactionV2Streaming(
|
||||
return collectCompactionV2Output(response, request);
|
||||
}
|
||||
|
||||
function buildCompactionV2Headers(model: Model, apiKey: string, request: CompactionV2Request): Record<string, string> {
|
||||
function buildCompactionV2Headers(
|
||||
model: Model,
|
||||
apiKey: string,
|
||||
request: CompactionV2Request,
|
||||
codexMetadata?: OpenAICodexCompatibilityMetadata,
|
||||
): Record<string, string> {
|
||||
const api = compactionV2Api(model);
|
||||
const cacheOptions = { sessionId: request.sessionId, promptCacheKey: request.promptCacheKey };
|
||||
const routingSessionId = getOpenAIResponsesRoutingSessionId(cacheOptions);
|
||||
const promptCacheSessionId = getOpenAIResponsesPromptCacheKey(cacheOptions);
|
||||
const promptCacheSessionId = getOpenAIPromptCacheKey(cacheOptions);
|
||||
const headers: Record<string, string> =
|
||||
api === "azure-openai-responses"
|
||||
? {
|
||||
@@ -338,7 +387,11 @@ function buildCompactionV2Headers(model: Model, apiKey: string, request: Compact
|
||||
}
|
||||
headers[OPENAI_HEADERS.BETA] = OPENAI_HEADER_VALUES.BETA_RESPONSES;
|
||||
headers[OPENAI_HEADERS.ORIGINATOR] = OPENAI_HEADER_VALUES.ORIGINATOR_CODEX;
|
||||
if (model.useResponsesLite) {
|
||||
headers[OPENAI_HEADERS.RESPONSES_LITE] = "true";
|
||||
}
|
||||
}
|
||||
if (codexMetadata) Object.assign(headers, codexMetadata.headers);
|
||||
|
||||
return headers;
|
||||
}
|
||||
|
||||
@@ -9,18 +9,21 @@ import {
|
||||
type Api,
|
||||
type ApiKey,
|
||||
type AssistantMessage,
|
||||
type CodexCompactionContext,
|
||||
type Context,
|
||||
Effort,
|
||||
type FetchImpl,
|
||||
type Message,
|
||||
type MessageAttribution,
|
||||
type Model,
|
||||
type ProviderSessionState,
|
||||
type SimpleStreamOptions,
|
||||
type Tool,
|
||||
type Usage,
|
||||
withAuth,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
import { ProviderHttpError } from "@oh-my-pi/pi-ai/error";
|
||||
import { createOpenAICodexCompactionRequestContext } from "@oh-my-pi/pi-ai/providers/openai-codex-responses";
|
||||
import { convertTools } from "@oh-my-pi/pi-ai/providers/openai-responses";
|
||||
import { buildResponsesInput, resolveOpenAICompatPolicy } from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
import { preferredDialect } from "@oh-my-pi/pi-catalog/identity";
|
||||
@@ -736,6 +739,10 @@ export interface SummaryOptions {
|
||||
sessionId?: string;
|
||||
/** Prompt-cache key for remote compaction transports that support provider prefix caching. */
|
||||
promptCacheKey?: string;
|
||||
/** Mutable provider state used to keep Codex compaction on the live session identity. */
|
||||
providerSessionState?: Map<string, ProviderSessionState>;
|
||||
/** Classification shared by every provider request in this logical compaction. */
|
||||
codexCompaction?: CodexCompactionContext;
|
||||
/** Provider-visible tools for remote compaction transports that replay native tool history. */
|
||||
tools?: Tool[];
|
||||
/** Optional fetch implementation threaded into remote compaction calls. */
|
||||
@@ -755,6 +762,13 @@ export interface SummaryOptions {
|
||||
) => Promise<AssistantMessage>;
|
||||
}
|
||||
|
||||
function localCodexCompaction(options: SummaryOptions | undefined) {
|
||||
return createOpenAICodexCompactionRequestContext({
|
||||
context: options?.codexCompaction,
|
||||
implementation: "responses",
|
||||
});
|
||||
}
|
||||
|
||||
function formatPreviousSnapcompactArchive(archiveText: string): string {
|
||||
return prompt.render(snapcompactArchiveContextPrompt, { archiveText });
|
||||
}
|
||||
@@ -844,6 +858,11 @@ export async function generateSummary(
|
||||
reasoning: resolveCompactionEffort(model, options?.thinkingLevel),
|
||||
initiatorOverride: options?.initiatorOverride,
|
||||
metadata: options?.metadata,
|
||||
fetch: options?.fetch,
|
||||
sessionId: options?.sessionId,
|
||||
promptCacheKey: options?.promptCacheKey,
|
||||
providerSessionState: options?.providerSessionState,
|
||||
codexCompaction: localCodexCompaction(options),
|
||||
},
|
||||
{ telemetry: options?.telemetry, oneshotKind: "compaction_summary", completeImpl: options?.completeImpl },
|
||||
);
|
||||
@@ -1047,6 +1066,11 @@ async function generateShortSummary(
|
||||
reasoning: resolveCompactionEffort(model, options?.thinkingLevel),
|
||||
initiatorOverride: options?.initiatorOverride,
|
||||
metadata: options?.metadata,
|
||||
fetch: options?.fetch,
|
||||
sessionId: options?.sessionId,
|
||||
promptCacheKey: options?.promptCacheKey,
|
||||
providerSessionState: options?.providerSessionState,
|
||||
codexCompaction: localCodexCompaction(options),
|
||||
},
|
||||
{ telemetry: options?.telemetry, oneshotKind: "compaction_short_summary", completeImpl: options?.completeImpl },
|
||||
);
|
||||
@@ -1317,6 +1341,8 @@ export async function compact(
|
||||
thinkingLevel: options?.thinkingLevel,
|
||||
sessionId: options?.sessionId,
|
||||
promptCacheKey: options?.promptCacheKey,
|
||||
providerSessionState: options?.providerSessionState,
|
||||
codexCompaction: options?.codexCompaction,
|
||||
tools: options?.tools,
|
||||
fetch: options?.fetch,
|
||||
completeImpl: options?.completeImpl,
|
||||
@@ -1375,7 +1401,12 @@ export async function compact(
|
||||
);
|
||||
const remote = await withAuth(
|
||||
apiKey,
|
||||
key => requestCompactionV2Streaming(model, key, request, signal, { fetch: summaryOptions.fetch }),
|
||||
key =>
|
||||
requestCompactionV2Streaming(model, key, request, signal, {
|
||||
fetch: summaryOptions.fetch,
|
||||
providerSessionState: summaryOptions.providerSessionState,
|
||||
codexCompaction: summaryOptions.codexCompaction,
|
||||
}),
|
||||
{ signal },
|
||||
);
|
||||
preserveData = { ...(preserveData ?? {}), ...storeCompactionV2PreserveData(remote, model) };
|
||||
@@ -1419,7 +1450,12 @@ export async function compact(
|
||||
remoteHistory,
|
||||
summaryOptions.remoteInstructions ?? SUMMARIZATION_SYSTEM_PROMPT,
|
||||
signal,
|
||||
{ fetch: summaryOptions.fetch },
|
||||
{
|
||||
fetch: summaryOptions.fetch,
|
||||
sessionId: summaryOptions.sessionId,
|
||||
providerSessionState: summaryOptions.providerSessionState,
|
||||
codexCompaction: summaryOptions.codexCompaction,
|
||||
},
|
||||
),
|
||||
{ signal },
|
||||
);
|
||||
@@ -1495,16 +1531,9 @@ export async function compact(
|
||||
const shortSummary = usedRemoteCompaction
|
||||
? "Remote compaction"
|
||||
: await generateShortSummary(recentMessages, summary, model, reserveTokens, apiKey, signal, {
|
||||
...summaryOptions,
|
||||
extraContext: options?.extraContext,
|
||||
remoteEndpoint: summaryOptions.remoteEndpoint,
|
||||
initiatorOverride: summaryOptions.initiatorOverride,
|
||||
metadata: summaryOptions.metadata,
|
||||
telemetry: summaryOptions.telemetry,
|
||||
// Same propagation as summaryOptions above — generateShortSummary
|
||||
// resolves its own reasoning via resolveCompactionEffort.
|
||||
thinkingLevel: options?.thinkingLevel,
|
||||
fetch: summaryOptions.fetch,
|
||||
completeImpl: summaryOptions.completeImpl,
|
||||
});
|
||||
|
||||
// Compute file lists and append to summary
|
||||
@@ -1567,6 +1596,11 @@ async function generateTurnPrefixSummary(
|
||||
reasoning: resolveCompactionEffort(model, options?.thinkingLevel),
|
||||
initiatorOverride: options?.initiatorOverride,
|
||||
metadata: options?.metadata,
|
||||
fetch: options?.fetch,
|
||||
sessionId: options?.sessionId,
|
||||
promptCacheKey: options?.promptCacheKey,
|
||||
providerSessionState: options?.providerSessionState,
|
||||
codexCompaction: localCodexCompaction(options),
|
||||
},
|
||||
{ telemetry: options?.telemetry, oneshotKind: "compaction_turn_prefix", completeImpl: options?.completeImpl },
|
||||
);
|
||||
|
||||
@@ -16,9 +16,22 @@
|
||||
*/
|
||||
|
||||
import { ProviderHttpError } from "@oh-my-pi/pi-ai/error";
|
||||
import { applyCodexResponsesLiteShape } from "@oh-my-pi/pi-ai/providers/openai-codex/request-transformer";
|
||||
import {
|
||||
createOpenAICodexCompactionRequestContext,
|
||||
createOpenAICodexCompatibilityMetadata,
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-codex-responses";
|
||||
import { parseAzureDeploymentNameMap, parseTextSignature } from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
import { transformMessages } from "@oh-my-pi/pi-ai/providers/transform-messages";
|
||||
import type { Api, AssistantMessage, FetchImpl, Message, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import type {
|
||||
Api,
|
||||
AssistantMessage,
|
||||
CodexCompactionContext,
|
||||
FetchImpl,
|
||||
Message,
|
||||
Model,
|
||||
ProviderSessionState,
|
||||
} from "@oh-my-pi/pi-ai/types";
|
||||
import {
|
||||
getOpenAIResponsesHistoryItems,
|
||||
getOpenAIResponsesHistoryPayload,
|
||||
@@ -460,7 +473,13 @@ export async function requestOpenAiRemoteCompaction(
|
||||
compactInput: Array<Record<string, unknown>>,
|
||||
instructions: string,
|
||||
signal?: AbortSignal,
|
||||
opts?: { fetch?: FetchImpl; timeoutMs?: number },
|
||||
opts?: {
|
||||
fetch?: FetchImpl;
|
||||
timeoutMs?: number;
|
||||
sessionId?: string;
|
||||
providerSessionState?: Map<string, ProviderSessionState>;
|
||||
codexCompaction?: CodexCompactionContext;
|
||||
},
|
||||
): Promise<OpenAiRemoteCompactionResponse> {
|
||||
const endpoint = resolveOpenAiCompactEndpoint(model);
|
||||
const requestModel = resolveOpenAiCompactModel(model);
|
||||
@@ -473,6 +492,8 @@ export async function requestOpenAiRemoteCompaction(
|
||||
instructions,
|
||||
};
|
||||
const isAzureOpenAiResponses = (model.remoteCompaction?.api ?? model.api) === "azure-openai-responses";
|
||||
const isCodexResponses =
|
||||
model.provider === "openai-codex" || (model.remoteCompaction?.api ?? model.api) === "openai-codex-responses";
|
||||
const headers: Record<string, string> = isAzureOpenAiResponses
|
||||
? {
|
||||
"content-type": "application/json",
|
||||
@@ -486,13 +507,33 @@ export async function requestOpenAiRemoteCompaction(
|
||||
};
|
||||
|
||||
// Codex endpoints require additional auth headers
|
||||
if (model.provider === "openai-codex") {
|
||||
if (isCodexResponses) {
|
||||
const accountId = getCodexAccountId(apiKey);
|
||||
if (accountId) {
|
||||
headers[OPENAI_HEADERS.ACCOUNT_ID] = accountId;
|
||||
}
|
||||
headers[OPENAI_HEADERS.BETA] = OPENAI_HEADER_VALUES.BETA_RESPONSES;
|
||||
headers[OPENAI_HEADERS.ORIGINATOR] = OPENAI_HEADER_VALUES.ORIGINATOR_CODEX;
|
||||
Object.assign(
|
||||
headers,
|
||||
createOpenAICodexCompatibilityMetadata({
|
||||
sessionId: opts?.sessionId,
|
||||
providerSessionState: opts?.providerSessionState,
|
||||
requestKind: "compaction",
|
||||
compaction: createOpenAICodexCompactionRequestContext({
|
||||
context: opts?.codexCompaction,
|
||||
implementation: "responses_compact",
|
||||
}),
|
||||
includeInstallationHeader: true,
|
||||
}).headers,
|
||||
);
|
||||
// Responses Lite models take the same rewrite on `/responses/compact`:
|
||||
// instructions ride as an input item and the lite marker header is set
|
||||
// (codex-rs routes compaction through `build_responses_request`).
|
||||
if (model.useResponsesLite) {
|
||||
applyCodexResponsesLiteShape(request);
|
||||
headers[OPENAI_HEADERS.RESPONSES_LITE] = "true";
|
||||
}
|
||||
}
|
||||
|
||||
const response = await (opts?.fetch ?? fetch)(endpoint, {
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { afterEach, describe, expect, test, vi } from "bun:test";
|
||||
import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test";
|
||||
import {
|
||||
type CompactionPreparation,
|
||||
compact,
|
||||
@@ -18,10 +18,36 @@ import {
|
||||
shouldUseOpenAiRemoteCompaction,
|
||||
} from "@oh-my-pi/pi-agent-core/compaction/openai";
|
||||
import * as ai from "@oh-my-pi/pi-ai";
|
||||
import type { AssistantMessage, FetchImpl, Model, ToolResultMessage } from "@oh-my-pi/pi-ai/types";
|
||||
import { getOpenAICodexTransportDetails } from "@oh-my-pi/pi-ai/providers/openai-codex-responses";
|
||||
import type {
|
||||
AssistantMessage,
|
||||
CodexCompactionContext,
|
||||
FetchImpl,
|
||||
Model,
|
||||
ProviderSessionState,
|
||||
ToolResultMessage,
|
||||
} from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import type { ModelSpec } from "@oh-my-pi/pi-catalog/types";
|
||||
import { isRecord } from "@oh-my-pi/pi-utils";
|
||||
import * as piUtils from "@oh-my-pi/pi-utils";
|
||||
|
||||
const { isRecord } = piUtils;
|
||||
const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001";
|
||||
const TEST_CODEX_COMPACTION: CodexCompactionContext = {
|
||||
operationId: "compaction-operation-1",
|
||||
trigger: "auto",
|
||||
reason: "context_limit",
|
||||
phase: "pre_turn",
|
||||
strategy: "memento",
|
||||
};
|
||||
|
||||
beforeEach(() => {
|
||||
vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
function makeOpenAiModel(overrides: Partial<ModelSpec<"openai-responses">> = {}): Model<"openai-responses"> {
|
||||
return buildModel({
|
||||
@@ -393,6 +419,396 @@ describe("requestCompactionV2Streaming", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("Responses Lite remote compaction", () => {
|
||||
function makeCodexLiteModel(
|
||||
overrides: Partial<ModelSpec<"openai-codex-responses">> = {},
|
||||
): Model<"openai-codex-responses"> {
|
||||
return buildModel({
|
||||
id: "gpt-5.6-terra",
|
||||
name: "GPT-5.6 Terra",
|
||||
api: "openai-codex-responses",
|
||||
provider: "openai-codex",
|
||||
baseUrl: "https://chatgpt.example/backend-api",
|
||||
reasoning: true,
|
||||
preferWebsockets: false,
|
||||
input: ["text", "image"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 372000,
|
||||
maxTokens: 128000,
|
||||
useResponsesLite: true,
|
||||
remoteCompaction: { enabled: true, api: "openai-codex-responses", v2StreamingEnabled: true },
|
||||
...overrides,
|
||||
});
|
||||
}
|
||||
|
||||
interface CapturedLiteRequest {
|
||||
instructions?: unknown;
|
||||
tools?: unknown;
|
||||
input?: Array<Record<string, unknown>>;
|
||||
client_metadata?: unknown;
|
||||
}
|
||||
|
||||
interface CapturedLiteExchange {
|
||||
body: CapturedLiteRequest;
|
||||
headers: Headers;
|
||||
}
|
||||
|
||||
function parseCodexTurnMetadata(value: unknown): Record<string, unknown> {
|
||||
if (typeof value !== "string") throw new Error("expected x-codex-turn-metadata");
|
||||
const parsed: unknown = JSON.parse(value);
|
||||
if (!isRecord(parsed)) throw new Error("expected Codex turn metadata object");
|
||||
return parsed;
|
||||
}
|
||||
|
||||
function captureLite(init: RequestInit | undefined): CapturedLiteExchange {
|
||||
if (!init?.headers || init.headers instanceof Headers || Array.isArray(init.headers)) {
|
||||
throw new Error("Expected remote compaction to send headers as a plain object");
|
||||
}
|
||||
return {
|
||||
body: JSON.parse(String(init.body)) as CapturedLiteRequest,
|
||||
headers: new Headers(init.headers),
|
||||
};
|
||||
}
|
||||
|
||||
function captureStreamLite(init: RequestInit | undefined): CapturedLiteExchange {
|
||||
if (!init?.headers) throw new Error("Expected local compaction request headers");
|
||||
return {
|
||||
body: JSON.parse(String(init.body)) as CapturedLiteRequest,
|
||||
headers: new Headers(init.headers),
|
||||
};
|
||||
}
|
||||
|
||||
test("V1 compaction sends the lite header and input-item instructions", async () => {
|
||||
const model = makeCodexLiteModel();
|
||||
let captured: CapturedLiteExchange | undefined;
|
||||
const fetchMock: FetchImpl = async (_input, init) => {
|
||||
captured = captureLite(init);
|
||||
return Response.json({ output: [{ type: "compaction", encrypted_content: "enc" }] });
|
||||
};
|
||||
|
||||
await requestOpenAiRemoteCompaction(
|
||||
model,
|
||||
"test-key",
|
||||
[{ type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }],
|
||||
"compact instructions",
|
||||
undefined,
|
||||
{
|
||||
fetch: fetchMock,
|
||||
sessionId: "codex-compaction-session",
|
||||
providerSessionState: new Map<string, ProviderSessionState>(),
|
||||
codexCompaction: TEST_CODEX_COMPACTION,
|
||||
},
|
||||
);
|
||||
|
||||
expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true");
|
||||
expect(captured?.body.instructions).toBeUndefined();
|
||||
expect(captured?.body.tools).toBeUndefined();
|
||||
expect(captured?.body.client_metadata).toBeUndefined();
|
||||
expect(captured?.headers.get("x-codex-installation-id")).toBe(TEST_INSTALLATION_ID);
|
||||
expect(captured?.headers.get("session-id")).toBe("codex-compaction-session");
|
||||
const v1TurnMetadata = parseCodexTurnMetadata(captured?.headers.get("x-codex-turn-metadata"));
|
||||
expect(v1TurnMetadata.request_kind).toBe("compaction");
|
||||
expect(v1TurnMetadata.compaction).toEqual({
|
||||
trigger: "auto",
|
||||
reason: "context_limit",
|
||||
implementation: "responses_compact",
|
||||
phase: "pre_turn",
|
||||
strategy: "memento",
|
||||
});
|
||||
expect(captured?.body.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools: [] });
|
||||
expect(captured?.body.input?.[1]).toEqual({
|
||||
type: "message",
|
||||
role: "developer",
|
||||
content: [{ type: "input_text", text: "compact instructions" }],
|
||||
});
|
||||
});
|
||||
|
||||
test("V2 streaming compaction applies the lite rewrite and keeps the trigger last", async () => {
|
||||
const model = makeCodexLiteModel();
|
||||
const request = buildCompactionV2Request(
|
||||
model,
|
||||
[{ type: "message", role: "user", content: [{ type: "input_text", text: "real user" }] }],
|
||||
"compact instructions",
|
||||
{ sessionId: "codex-compaction-session" },
|
||||
);
|
||||
let captured: CapturedLiteExchange | undefined;
|
||||
const fetchMock: FetchImpl = async (_input, init) => {
|
||||
captured = captureLite(init);
|
||||
return sseResponse([
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: { type: "compaction", encrypted_content: "enc" },
|
||||
},
|
||||
{ type: "response.completed", response: { usage: { input_tokens: 1, output_tokens: 1, total_tokens: 2 } } },
|
||||
]);
|
||||
};
|
||||
|
||||
expect(shouldUseCompactionV2Streaming(model)).toBe(true);
|
||||
await requestCompactionV2Streaming(model, "test-key", request, undefined, {
|
||||
fetch: fetchMock,
|
||||
providerSessionState: new Map<string, ProviderSessionState>(),
|
||||
codexCompaction: TEST_CODEX_COMPACTION,
|
||||
});
|
||||
|
||||
expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true");
|
||||
expect(captured?.body.instructions).toBeUndefined();
|
||||
expect(captured?.body.tools).toBeUndefined();
|
||||
if (!isRecord(captured?.body.client_metadata)) throw new Error("expected V2 client_metadata");
|
||||
const v2ClientMetadata = captured.body.client_metadata;
|
||||
const v2TurnMetadata = parseCodexTurnMetadata(v2ClientMetadata["x-codex-turn-metadata"]);
|
||||
expect(captured.headers.get("x-codex-installation-id")).toBeNull();
|
||||
expect(v2ClientMetadata["x-codex-installation-id"]).toBe(TEST_INSTALLATION_ID);
|
||||
expect(v2ClientMetadata.session_id).toBe(captured.headers.get("session-id"));
|
||||
expect(v2ClientMetadata.thread_id).toBe(captured.headers.get("thread-id"));
|
||||
expect(v2TurnMetadata.request_kind).toBe("compaction");
|
||||
expect(v2TurnMetadata.compaction).toEqual({
|
||||
trigger: "auto",
|
||||
reason: "context_limit",
|
||||
implementation: "responses_compaction_v2",
|
||||
phase: "pre_turn",
|
||||
strategy: "memento",
|
||||
});
|
||||
expect(captured?.body.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools: [] });
|
||||
expect(captured?.body.input?.[1]).toEqual({
|
||||
type: "message",
|
||||
role: "developer",
|
||||
content: [{ type: "input_text", text: "compact instructions" }],
|
||||
});
|
||||
expect(captured?.body.input?.at(-1)).toEqual({ type: "compaction_trigger" });
|
||||
});
|
||||
|
||||
test("compact fan-out keeps local Codex summaries on one classified turn", async () => {
|
||||
const model = makeCodexLiteModel();
|
||||
const captured: CapturedLiteExchange[] = [];
|
||||
const fetchMock: FetchImpl = async (_input, init) => {
|
||||
captured.push(captureStreamLite(init));
|
||||
return sseResponse([
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
output_index: 0,
|
||||
item: { type: "message", id: "msg_summary", role: "assistant", status: "in_progress", content: [] },
|
||||
},
|
||||
{
|
||||
type: "response.content_part.added",
|
||||
output_index: 0,
|
||||
content_index: 0,
|
||||
part: { type: "output_text", text: "" },
|
||||
},
|
||||
{ type: "response.output_text.delta", output_index: 0, content_index: 0, delta: "local summary" },
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: {
|
||||
type: "message",
|
||||
id: "msg_summary",
|
||||
role: "assistant",
|
||||
status: "completed",
|
||||
content: [{ type: "output_text", text: "local summary" }],
|
||||
},
|
||||
},
|
||||
{
|
||||
type: "response.completed",
|
||||
response: {
|
||||
status: "completed",
|
||||
usage: {
|
||||
input_tokens: 8,
|
||||
output_tokens: 2,
|
||||
total_tokens: 10,
|
||||
input_tokens_details: { cached_tokens: 0 },
|
||||
},
|
||||
},
|
||||
},
|
||||
]);
|
||||
};
|
||||
const preparation: CompactionPreparation = {
|
||||
firstKeptEntryId: "kept-1",
|
||||
messagesToSummarize: [{ role: "user", content: "long history", timestamp: 1 }],
|
||||
turnPrefixMessages: [],
|
||||
recentMessages: [{ role: "user", content: "recent", timestamp: 2 }],
|
||||
isSplitTurn: false,
|
||||
tokensBefore: 100_000,
|
||||
fileOps: createFileOps(),
|
||||
settings: {
|
||||
...DEFAULT_COMPACTION_SETTINGS,
|
||||
remoteEnabled: false,
|
||||
remoteStreamingV2Enabled: false,
|
||||
},
|
||||
};
|
||||
|
||||
const result = await compact(preparation, model, "test-key", undefined, undefined, {
|
||||
fetch: fetchMock,
|
||||
sessionId: "codex-compaction-session",
|
||||
providerSessionState: new Map<string, ProviderSessionState>(),
|
||||
codexCompaction: TEST_CODEX_COMPACTION,
|
||||
});
|
||||
|
||||
expect(result.summary).toContain("local summary");
|
||||
expect(captured).toHaveLength(2);
|
||||
const turnIds: string[] = [];
|
||||
for (const exchange of captured) {
|
||||
if (!isRecord(exchange.body.client_metadata)) throw new Error("expected local client_metadata");
|
||||
const clientMetadata = exchange.body.client_metadata;
|
||||
const turnMetadata = parseCodexTurnMetadata(clientMetadata["x-codex-turn-metadata"]);
|
||||
expect(exchange.headers.get("x-codex-installation-id")).toBeNull();
|
||||
expect(clientMetadata["x-codex-installation-id"]).toBe(TEST_INSTALLATION_ID);
|
||||
expect(turnMetadata.request_kind).toBe("compaction");
|
||||
expect(turnMetadata.compaction).toEqual({
|
||||
trigger: "auto",
|
||||
reason: "context_limit",
|
||||
implementation: "responses",
|
||||
phase: "pre_turn",
|
||||
strategy: "memento",
|
||||
});
|
||||
if (typeof turnMetadata.turn_id !== "string") throw new Error("expected Codex turn id");
|
||||
turnIds.push(turnMetadata.turn_id);
|
||||
}
|
||||
expect(new Set(turnIds).size).toBe(1);
|
||||
});
|
||||
|
||||
test("local Codex compaction isolates and closes transient websocket sessions", async () => {
|
||||
const originalWebSocket = global.WebSocket;
|
||||
const sockets: AgentCompactionWebSocket[] = [];
|
||||
let responseCount = 0;
|
||||
|
||||
class AgentCompactionWebSocket {
|
||||
static readonly CONNECTING = 0;
|
||||
static readonly OPEN = 1;
|
||||
static readonly CLOSING = 2;
|
||||
static readonly CLOSED = 3;
|
||||
|
||||
readyState = AgentCompactionWebSocket.CONNECTING;
|
||||
binaryType: "blob" | "arraybuffer" | "nodebuffer" = "blob";
|
||||
onopen: ((event: Event) => void) | null = null;
|
||||
onmessage: ((event: MessageEvent) => void) | null = null;
|
||||
onerror: ((event: Event) => void) | null = null;
|
||||
onclose: ((event: Event) => void) | null = null;
|
||||
readonly handshakeHeaders = {
|
||||
"x-codex-turn-state": `agent-compaction-state-${sockets.length}`,
|
||||
};
|
||||
|
||||
constructor(
|
||||
readonly url: string,
|
||||
readonly options?: { headers?: Record<string, string> },
|
||||
) {
|
||||
sockets.push(this);
|
||||
queueMicrotask(() => {
|
||||
this.readyState = AgentCompactionWebSocket.OPEN;
|
||||
this.onopen?.(new Event("open"));
|
||||
});
|
||||
}
|
||||
|
||||
send(_data: string): void {
|
||||
responseCount += 1;
|
||||
const responseId = `response-${responseCount}`;
|
||||
const messageId = `message-${responseCount}`;
|
||||
const text = sockets[0] === this ? "main response" : "local summary";
|
||||
const events: Record<string, unknown>[] = [
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
item: { type: "message", id: messageId, role: "assistant", status: "in_progress", content: [] },
|
||||
},
|
||||
{ type: "response.content_part.added", part: { type: "output_text", text: "" } },
|
||||
{ type: "response.output_text.delta", delta: text },
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
item: {
|
||||
type: "message",
|
||||
id: messageId,
|
||||
role: "assistant",
|
||||
status: "completed",
|
||||
content: [{ type: "output_text", text }],
|
||||
},
|
||||
},
|
||||
{
|
||||
type: "response.done",
|
||||
response: {
|
||||
id: responseId,
|
||||
status: "completed",
|
||||
usage: {
|
||||
input_tokens: 8,
|
||||
output_tokens: 2,
|
||||
total_tokens: 10,
|
||||
input_tokens_details: { cached_tokens: 0 },
|
||||
},
|
||||
},
|
||||
},
|
||||
];
|
||||
for (const event of events) {
|
||||
this.onmessage?.({ data: JSON.stringify(event) } as MessageEvent);
|
||||
}
|
||||
}
|
||||
|
||||
close(): void {
|
||||
this.readyState = AgentCompactionWebSocket.CLOSED;
|
||||
}
|
||||
}
|
||||
|
||||
const providerSessionState = new Map<string, ProviderSessionState>();
|
||||
try {
|
||||
global.WebSocket = AgentCompactionWebSocket as unknown as typeof WebSocket;
|
||||
const model = makeCodexLiteModel({ preferWebsockets: true });
|
||||
const sessionId = "agent-compaction-isolation";
|
||||
const fetchMock: FetchImpl = async () => {
|
||||
throw new Error("Codex websocket compaction unexpectedly used SSE");
|
||||
};
|
||||
const main = await ai
|
||||
.streamSimple(
|
||||
model,
|
||||
{
|
||||
systemPrompt: ["You are a helpful assistant."],
|
||||
messages: [{ role: "user", content: "Start the turn", timestamp: Date.now() }],
|
||||
},
|
||||
{ apiKey: "test-key", fetch: fetchMock, sessionId, providerSessionState },
|
||||
)
|
||||
.result();
|
||||
expect(main.stopReason).toBe("stop");
|
||||
expect(sockets).toHaveLength(1);
|
||||
expect(sockets[0]?.readyState).toBe(AgentCompactionWebSocket.OPEN);
|
||||
|
||||
const preparation: CompactionPreparation = {
|
||||
firstKeptEntryId: "kept-1",
|
||||
messagesToSummarize: [{ role: "user", content: "long history", timestamp: 1 }],
|
||||
turnPrefixMessages: [],
|
||||
recentMessages: [{ role: "user", content: "recent", timestamp: 2 }],
|
||||
isSplitTurn: false,
|
||||
tokensBefore: 100_000,
|
||||
fileOps: createFileOps(),
|
||||
settings: {
|
||||
...DEFAULT_COMPACTION_SETTINGS,
|
||||
remoteEnabled: false,
|
||||
remoteStreamingV2Enabled: false,
|
||||
},
|
||||
};
|
||||
const result = await compact(preparation, model, "test-key", undefined, undefined, {
|
||||
fetch: fetchMock,
|
||||
sessionId,
|
||||
providerSessionState,
|
||||
codexCompaction: TEST_CODEX_COMPACTION,
|
||||
});
|
||||
|
||||
expect(result.summary).toContain("local summary");
|
||||
expect(sockets).toHaveLength(3);
|
||||
expect(sockets[0]?.readyState).toBe(AgentCompactionWebSocket.OPEN);
|
||||
expect(sockets[1]?.readyState).toBe(AgentCompactionWebSocket.CLOSED);
|
||||
expect(sockets[2]?.readyState).toBe(AgentCompactionWebSocket.CLOSED);
|
||||
expect(
|
||||
getOpenAICodexTransportDetails(model, {
|
||||
sessionId,
|
||||
providerSessionState,
|
||||
}),
|
||||
).toMatchObject({
|
||||
websocketConnected: true,
|
||||
hasTurnState: true,
|
||||
});
|
||||
} finally {
|
||||
for (const state of providerSessionState.values()) state.close();
|
||||
providerSessionState.clear();
|
||||
global.WebSocket = originalWebSocket;
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
test("uses configured OpenAI-compatible compaction for custom providers", async () => {
|
||||
const model = makeOpenAiModel({
|
||||
provider: "cliproxy-codex",
|
||||
|
||||
@@ -5,12 +5,74 @@
|
||||
### Fixed
|
||||
|
||||
- Fixed xAI SuperGrok multi-account rotation when an account returns HTTP 403 `run out of credits` / `personal-team-blocked:spending-limit`. That account-local cap is now classified as a usage limit so `streamSimple` auth-retry and `rotateSessionCredential` switch to a sibling `xai-oauth` credential instead of sticking to the exhausted account.
|
||||
### Added
|
||||
|
||||
- Added model-driven Codex Responses Lite: `responsesLite` now defaults to the catalog `useResponsesLite` flag (codex-rs `use_responses_lite`, set on the GPT-5.6 family), so lite requests are sent without per-call opt-in.
|
||||
- Added the full Responses Lite wire contract: lite requests move tools into a leading `{type: "additional_tools", role: "developer"}` input item and the base instructions into a developer message, omit top-level `instructions`/`tools`, and force `parallel_tool_calls: false`, mirroring codex-rs `build_responses_request`.
|
||||
- Added concurrent reasoning summaries on Codex Responses: requests with a reasoning summary send `stream_options: { reasoning_summary_delivery: "sequential_cutoff" }`, and the stream decoder consumes the matching atomic `response.reasoning_summary_text.done` events (resolved by `item_id`/`output_index`, stale dones dropped, incremental `.delta`/`.part.*` events ignored under the cutoff contract). The cutoff gate reads the post-`onPayload` wire body on both transports, and `response.reasoning_summary_text.done` now counts as websocket watchdog progress.
|
||||
- Added Novita API-key login with authenticated key validation and `NOVITA_API_KEY` discovery ([#4917](https://github.com/can1357/oh-my-pi/pull/4917) by [@jason-wu-ai](https://github.com/jason-wu-ai)).
|
||||
|
||||
### Changed
|
||||
|
||||
- Refactored Responses Lite transport to move tools and instructions into input items
|
||||
- Updated Responses Lite to force parallel tool calling off and strip image detail
|
||||
- Standardized Responses Lite activation via model-level catalog flags
|
||||
|
||||
- Recognized Pro Lite as a paid plan tier for OpenAI Codex models
|
||||
- Changed Responses Lite image handling to match current codex-rs: a lite request containing input images now stays on the lite transport with image `detail` stripped, instead of silently falling back to the full Responses shape.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed concurrent reasoning summaries to ignore legacy streaming events under cutoff contract
|
||||
- Fixed sequential-cutoff Codex reasoning summaries repeating earlier content when atomic summary snapshots are replayed or extended.
|
||||
- Fixed error classification for typed AWS credential-resolution failures (`AwsCredentialsError`) to map them to authentication failures. ([#5030](https://github.com/can1357/oh-my-pi/pull/5030) by [@usr-bin-roygbiv](https://github.com/usr-bin-roygbiv))
|
||||
|
||||
## [16.3.15] - 2026-07-09
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
- Renamed `OpenAIResponsesCacheOptions`, `normalizeOpenAIResponsesPromptCacheKey`, and `getOpenAIResponsesPromptCacheKey` to the endpoint-neutral `OpenAICacheOptions`, `normalizeOpenAIPromptCacheKey`, and `getOpenAIPromptCacheKey`.
|
||||
|
||||
### Added
|
||||
|
||||
- Added automatic prompt-cache affinity header injection for OpenAI-family chat completions
|
||||
- Added support for explicit prompt-cache affinity headers in OpenAI-family chat completions
|
||||
- Added OpenAI pro reasoning mode support: models carrying the catalog `reasoningMode: "pro"` marker (GPT-5.6 Pro aliases) send `reasoning: { mode: "pro" }` on OpenAI Responses and Codex Responses requests, alongside the configured effort. The Codex request body now honors `requestModelId` so catalog aliases request the base upstream model id.
|
||||
|
||||
### Changed
|
||||
|
||||
- Updated xAI OAuth to use a dedicated device-code flow instead of redirect/loopback server
|
||||
|
||||
### Fixed
|
||||
|
||||
- Improved account routing for GPT-5.6 models to better respect paid tier requirements
|
||||
- Refined account selection logic to correctly identify plan types from account metadata
|
||||
- Fixed OpenAI Codex multi-account routing for GPT-5.6: Sol and Luna requests now prefer Plus-or-higher accounts while Terra remains available to Free/Go accounts; local pro-mode aliases inherit their base model's Codex plan eligibility.
|
||||
- Fixed xAI Grok OAuth login to use xAI's device authorization flow: `/login` now opens the verification URL, displays the device code, and polls for approval instead of asking for a pasted redirect or linking to Hermes Agent documentation.
|
||||
|
||||
## [16.3.14] - 2026-07-09
|
||||
|
||||
### Changed
|
||||
|
||||
- Updated Codex reasoning effort mapping to support shifted wire tiers for newer models
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed the Codex Responses request transformer bypassing catalog/compat reasoning effort maps: the clamped user effort is now remapped to the provider wire tier (GPT-5.6's shifted five-tier scale sends `max` for user `xhigh` and `xhigh` for `high`), failing loudly if a map produces a value outside the Codex wire vocabulary.
|
||||
|
||||
## [16.3.13] - 2026-07-09
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed the xAI Grok OAuth (`xai-oauth`) provider to use manual code-paste login by default. `/login` now accepts a pasted authorization code or full `http://127.0.0.1:56121/callback?code=...` redirect URL without starting a local callback listener ([#3277](https://github.com/can1357/oh-my-pi/pull/3277) by [@Jaaneek](https://github.com/Jaaneek)).
|
||||
- Renamed the xAI Grok OAuth provider in login and credential prompts to "xAI Grok OAuth (SuperGrok or X Premium+)" ([#3277](https://github.com/can1357/oh-my-pi/pull/3277) by [@Jaaneek](https://github.com/Jaaneek)).
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed the generic lazy-stream idle watchdog aborting healthy `cursor-agent` streams with "Provider stream stalled while waiting for the next event" while a Cursor exec-channel local tool (shell/read/grep/write/MCP/…) legitimately ran longer than the idle budget. Provider streams now advertise consumer-side local work in flight and the watchdog slides its deadline instead of aborting; genuinely silent streams still time out. ([#4593](https://github.com/can1357/oh-my-pi/issues/4593))
|
||||
- Fixed OpenAI Codex/Responses reasoning streams so streamed thinking content is preserved when the final `output_item.done` reconstructs to an empty summary ([#4918](https://github.com/can1357/oh-my-pi/issues/4918)).
|
||||
- Fixed Anthropic streams hanging forever when generation wedges mid-stream (notably long `write` tool calls on Opus 4.8 high/xhigh) while the server keeps sending `ping` keepalives: pings now extend the idle watchdog only within a bounded window (3x the idle timeout) since the last real stream event, so a stalled tool-call stream times out and recovers instead of hanging with no retry path ([#4900](https://github.com/can1357/oh-my-pi/issues/4900)).
|
||||
|
||||
## [16.3.12] - 2026-07-08
|
||||
|
||||
### Added
|
||||
|
||||
@@ -59,6 +59,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an
|
||||
- **Qianfan** (requires `QIANFAN_API_KEY`)
|
||||
- **NVIDIA** (requires `NVIDIA_API_KEY`)
|
||||
- **NanoGPT** (requires `NANO_GPT_API_KEY`)
|
||||
- **Novita** (requires `NOVITA_API_KEY`)
|
||||
- **Hugging Face Inference**
|
||||
- **xAI**
|
||||
- **Venice** (requires `VENICE_API_KEY`)
|
||||
@@ -943,6 +944,7 @@ In Node.js environments, you can set environment variables to avoid passing API
|
||||
| Synthetic | `SYNTHETIC_API_KEY` |
|
||||
| NVIDIA | `NVIDIA_API_KEY` |
|
||||
| NanoGPT | `NANO_GPT_API_KEY` |
|
||||
| Novita | `NOVITA_API_KEY` |
|
||||
| Venice | `VENICE_API_KEY` |
|
||||
| Moonshot | `MOONSHOT_API_KEY` |
|
||||
| xAI | `XAI_API_KEY` |
|
||||
@@ -981,6 +983,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations:
|
||||
- Qianfan: `https://qianfan.baidubce.com/v2`
|
||||
- NVIDIA: `https://integrate.api.nvidia.com/v1`
|
||||
- NanoGPT: `https://nano-gpt.com/api/v1`
|
||||
- Novita: `https://api.novita.ai/openai/v1`
|
||||
- Hugging Face Inference: `https://router.huggingface.co/v1`
|
||||
- Venice: `https://api.venice.ai/api/v1`
|
||||
- Xiaomi MiMo: `https://api.xiaomimimo.com/anthropic`
|
||||
@@ -1082,7 +1085,7 @@ Credentials are saved to `agent.db` in the agent directory. `/login qianfan` ope
|
||||
|
||||
`login` supports OAuth providers (Anthropic, OpenAI Codex, GitHub Copilot, Gemini CLI, Antigravity) and API-key onboarding flows.
|
||||
|
||||
For the current API-key onboarding flows, the library covers Together, Moonshot, Qianfan, NVIDIA, NanoGPT, Hugging Face, Venice, Xiaomi, vLLM, LiteLLM, Cloudflare AI Gateway, Qwen Portal, and Ollama Cloud. Ollama remains the local runtime integration; set `OLLAMA_API_KEY` only when your local or self-hosted deployment enforces bearer auth.
|
||||
For the current API-key onboarding flows, the library covers Together, Moonshot, Qianfan, NVIDIA, NanoGPT, Novita, Hugging Face, Venice, Xiaomi, vLLM, LiteLLM, Cloudflare AI Gateway, Qwen Portal, and Ollama Cloud. Ollama remains the local runtime integration; set `OLLAMA_API_KEY` only when your local or self-hosted deployment enforces bearer auth.
|
||||
|
||||
### Programmatic OAuth
|
||||
|
||||
@@ -1114,7 +1117,7 @@ import {
|
||||
getOAuthApiKey, // (provider, credentialsMap) => { newCredentials, apiKey } | null
|
||||
|
||||
// Types
|
||||
type OAuthProvider, // includes 'anthropic', 'openai-codex', 'github-copilot', 'google-gemini-cli', 'google-antigravity', 'together', 'moonshot', 'qianfan', 'nvidia', 'nanogpt', 'huggingface', 'venice', 'xiaomi', 'vllm', 'litellm', 'cloudflare-ai-gateway', 'qwen-portal', ...
|
||||
type OAuthProvider, // includes 'anthropic', 'openai-codex', 'github-copilot', 'google-gemini-cli', 'google-antigravity', 'together', 'moonshot', 'qianfan', 'nvidia', 'nanogpt', 'novita', 'huggingface', 'venice', 'xiaomi', 'vllm', 'litellm', 'cloudflare-ai-gateway', 'qwen-portal', ...
|
||||
type OAuthCredentials,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
```
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"type": "module",
|
||||
"name": "@oh-my-pi/pi-ai",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
||||
"homepage": "https://omp.sh",
|
||||
"author": "Can Boluk",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
/**
|
||||
* Broker-aware auth-storage discovery used by both the coding-agent runtime and
|
||||
* the catalog model generator. Keeps the precedence logic (env → config.yml →
|
||||
* the catalog model generator. Keeps the precedence logic (env → config.yml/config.yaml →
|
||||
* token file → local SQLite) in one place so build-time tooling sees the same
|
||||
* credentials as the TUI.
|
||||
*/
|
||||
@@ -12,6 +12,7 @@ import {
|
||||
getConfigRootDir,
|
||||
isEnoent,
|
||||
logger,
|
||||
MAIN_CONFIG_FILENAMES,
|
||||
} from "@oh-my-pi/pi-utils";
|
||||
import { YAML } from "bun";
|
||||
import { AuthStorage } from "../auth-storage";
|
||||
@@ -72,21 +73,24 @@ interface ConfigSnapshot {
|
||||
}
|
||||
|
||||
async function readConfigYaml(agentDir: string): Promise<ConfigSnapshot> {
|
||||
const configPath = path.join(agentDir, "config.yml");
|
||||
try {
|
||||
const raw = await Bun.file(configPath).text();
|
||||
const parsed = YAML.parse(raw);
|
||||
if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return {};
|
||||
const record = parsed as Record<string, unknown>;
|
||||
const url = typeof record["auth.broker.url"] === "string" ? (record["auth.broker.url"] as string) : undefined;
|
||||
const token =
|
||||
typeof record["auth.broker.token"] === "string" ? (record["auth.broker.token"] as string) : undefined;
|
||||
return { url, token };
|
||||
} catch (err) {
|
||||
if (isEnoent(err)) return {};
|
||||
logger.warn("auth-broker config.yml unreadable", { error: String(err) });
|
||||
return {};
|
||||
for (const filename of MAIN_CONFIG_FILENAMES) {
|
||||
const configPath = path.join(agentDir, filename);
|
||||
try {
|
||||
const raw = await Bun.file(configPath).text();
|
||||
const parsed = YAML.parse(raw);
|
||||
if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return {};
|
||||
const record = parsed as Record<string, unknown>;
|
||||
const url = typeof record["auth.broker.url"] === "string" ? (record["auth.broker.url"] as string) : undefined;
|
||||
const token =
|
||||
typeof record["auth.broker.token"] === "string" ? (record["auth.broker.token"] as string) : undefined;
|
||||
return { url, token };
|
||||
} catch (err) {
|
||||
if (isEnoent(err)) continue;
|
||||
logger.warn("auth-broker config unreadable", { path: configPath, error: String(err) });
|
||||
return {};
|
||||
}
|
||||
}
|
||||
return {};
|
||||
}
|
||||
|
||||
function resolveSnapshotTtlMs(): number {
|
||||
@@ -104,7 +108,7 @@ function resolveSnapshotTtlMs(): number {
|
||||
* Resolve broker connection configuration using the same precedence as the TUI:
|
||||
*
|
||||
* 1. `OMP_AUTH_BROKER_URL` / `OMP_AUTH_BROKER_TOKEN` env vars.
|
||||
* 2. `auth.broker.url` / `auth.broker.token` in `<agentDir>/config.yml`.
|
||||
* 2. `auth.broker.url` / `auth.broker.token` in `<agentDir>/config.yml` or `<agentDir>/config.yaml`.
|
||||
* 3. `<config-root>/auth-broker.token` file (paired with a URL from env/config).
|
||||
*
|
||||
* Returns `null` when no broker URL is configured — callers should fall back to
|
||||
|
||||
@@ -112,8 +112,8 @@ function deriveSessionId(modelId: string, context: Context): string {
|
||||
parts.push(JSON.stringify({ role: first.role, content: first.content }));
|
||||
}
|
||||
const seed = parts.join("\u0000");
|
||||
// The 36-char UUID flows through unchanged: Codex's
|
||||
// `normalizeOpenAIResponsesPromptCacheKey` accepts ≤64 chars verbatim.
|
||||
// The 36-char UUID flows through unchanged:
|
||||
// `normalizeOpenAIPromptCacheKey` accepts ≤64 chars verbatim.
|
||||
return deterministicUuid(seed);
|
||||
}
|
||||
|
||||
|
||||
+106
-40
@@ -761,25 +761,84 @@ function isAbortSignalOption(
|
||||
return typeof value === "object" && value !== null && "aborted" in value && "addEventListener" in value;
|
||||
}
|
||||
|
||||
function requiresOpenAICodexProModel(provider: string, modelId: string | undefined): boolean {
|
||||
return provider === "openai-codex" && typeof modelId === "string" && modelId.includes("-spark");
|
||||
type OpenAICodexPlanRequirement = "none" | "paid" | "pro";
|
||||
type OpenAICodexPlanClass = "free" | "paid" | "pro" | "unknown";
|
||||
|
||||
const GPT_56_PAID_CODEX_MODEL_PATTERN = /^gpt-5\.6-(?:sol|luna)(?:-pro)?$/;
|
||||
const OPENAI_CODEX_PRO_PLAN_TOKENS: Record<string, true> = {
|
||||
pro: true,
|
||||
};
|
||||
const OPENAI_CODEX_PAID_PLAN_TOKENS: Record<string, true> = {
|
||||
plus: true,
|
||||
business: true,
|
||||
team: true,
|
||||
enterprise: true,
|
||||
edu: true,
|
||||
education: true,
|
||||
teacher: true,
|
||||
teachers: true,
|
||||
health: true,
|
||||
gov: true,
|
||||
government: true,
|
||||
};
|
||||
const OPENAI_CODEX_FREE_PLAN_TOKENS: Record<string, true> = {
|
||||
free: true,
|
||||
go: true,
|
||||
};
|
||||
|
||||
/**
|
||||
* Account tier needed for model-aware Codex OAuth routing.
|
||||
*
|
||||
* GPT-5.6 Terra (including its local pro-mode alias) remains available on every
|
||||
* plan. Sol and Luna pro-mode aliases inherit their base models' paid tier;
|
||||
* only Spark currently has a documented Pro-plan preference in Codex.
|
||||
*/
|
||||
function resolveOpenAICodexPlanRequirement(provider: string, modelId: string | undefined): OpenAICodexPlanRequirement {
|
||||
if (provider !== "openai-codex" || typeof modelId !== "string") return "none";
|
||||
const separator = modelId.lastIndexOf("/");
|
||||
const bareModelId = (separator === -1 ? modelId : modelId.slice(separator + 1)).toLowerCase();
|
||||
if (bareModelId.includes("-spark")) return "pro";
|
||||
if (bareModelId === "gpt-5.6" || GPT_56_PAID_CODEX_MODEL_PATTERN.test(bareModelId)) return "paid";
|
||||
return "none";
|
||||
}
|
||||
|
||||
function getUsagePlanType(report: UsageReport | null): string | undefined {
|
||||
const metadata = report?.metadata;
|
||||
if (!metadata || typeof metadata !== "object" || Array.isArray(metadata)) return undefined;
|
||||
const planType = (metadata as { planType?: unknown }).planType;
|
||||
return typeof planType === "string" ? planType.toLowerCase() : undefined;
|
||||
if (!metadata) return undefined;
|
||||
const planType = metadata.planType;
|
||||
if (typeof planType !== "string") return undefined;
|
||||
const normalized = planType
|
||||
.trim()
|
||||
.toLowerCase()
|
||||
.replace(/[\s-]+/g, "_");
|
||||
return normalized.startsWith("chatgpt_") ? normalized.slice("chatgpt_".length) : normalized;
|
||||
}
|
||||
|
||||
function getOpenAICodexPlanPriority(report: UsageReport | null): number {
|
||||
function classifyOpenAICodexPlan(report: UsageReport | null): OpenAICodexPlanClass {
|
||||
const planType = getUsagePlanType(report);
|
||||
if (!planType) return 1;
|
||||
return planType.includes("pro") ? 0 : 2;
|
||||
if (!planType) return "unknown";
|
||||
// Pro Lite is a paid Codex tier, but does not imply full Pro-only model access.
|
||||
if (planType === "prolite" || planType === "pro_lite") return "paid";
|
||||
const tokens = planType.split("_");
|
||||
if (tokens.some(token => OPENAI_CODEX_PRO_PLAN_TOKENS[token] === true)) return "pro";
|
||||
if (tokens.some(token => OPENAI_CODEX_PAID_PLAN_TOKENS[token] === true)) return "paid";
|
||||
if (tokens.some(token => OPENAI_CODEX_FREE_PLAN_TOKENS[token] === true)) return "free";
|
||||
return "unknown";
|
||||
}
|
||||
|
||||
function hasOpenAICodexProPlan(report: UsageReport | null): boolean {
|
||||
return getUsagePlanType(report)?.includes("pro") === true;
|
||||
function getOpenAICodexPlanEligibility(
|
||||
report: UsageReport | null,
|
||||
requirement: OpenAICodexPlanRequirement,
|
||||
): boolean | undefined {
|
||||
if (requirement === "none") return true;
|
||||
const planClass = classifyOpenAICodexPlan(report);
|
||||
if (planClass === "unknown") return undefined;
|
||||
return requirement === "paid" ? planClass !== "free" : planClass === "pro";
|
||||
}
|
||||
|
||||
function getOpenAICodexPlanPriority(report: UsageReport | null, requirement: OpenAICodexPlanRequirement): number {
|
||||
const eligibility = getOpenAICodexPlanEligibility(report, requirement);
|
||||
return eligibility === true ? 0 : eligibility === undefined ? 1 : 2;
|
||||
}
|
||||
|
||||
function compareUsageRankingMetric(left: number, right: number): number {
|
||||
@@ -3186,8 +3245,7 @@ export class AuthStorage {
|
||||
#compareRankedOAuthCandidatePriority(
|
||||
left: RankedOAuthCandidate,
|
||||
right: RankedOAuthCandidate,
|
||||
provider: string,
|
||||
modelId: string | undefined,
|
||||
planRequirement: OpenAICodexPlanRequirement,
|
||||
): number {
|
||||
if (left.blocked !== right.blocked) return left.blocked ? 1 : -1;
|
||||
if (left.blocked && right.blocked) {
|
||||
@@ -3196,7 +3254,7 @@ export class AuthStorage {
|
||||
if (leftBlockedUntil !== rightBlockedUntil) return leftBlockedUntil - rightBlockedUntil;
|
||||
return 0;
|
||||
}
|
||||
if (requiresOpenAICodexProModel(provider, modelId) && left.planPriority !== right.planPriority) {
|
||||
if (planRequirement !== "none" && left.planPriority !== right.planPriority) {
|
||||
return left.planPriority - right.planPriority;
|
||||
}
|
||||
if (left.hasPriorityBoost !== right.hasPriorityBoost) return left.hasPriorityBoost ? -1 : 1;
|
||||
@@ -3214,20 +3272,18 @@ export class AuthStorage {
|
||||
#compareRankedOAuthCandidates(
|
||||
left: RankedOAuthCandidate,
|
||||
right: RankedOAuthCandidate,
|
||||
provider: string,
|
||||
modelId: string | undefined,
|
||||
planRequirement: OpenAICodexPlanRequirement,
|
||||
): number {
|
||||
const priority = this.#compareRankedOAuthCandidatePriority(left, right, provider, modelId);
|
||||
const priority = this.#compareRankedOAuthCandidatePriority(left, right, planRequirement);
|
||||
return priority !== 0 ? priority : left.orderPos - right.orderPos;
|
||||
}
|
||||
|
||||
#orderRankedOAuthCandidates(
|
||||
candidates: RankedOAuthCandidate[],
|
||||
sessionId: string | undefined,
|
||||
provider: string,
|
||||
modelId: string | undefined,
|
||||
planRequirement: OpenAICodexPlanRequirement,
|
||||
): OAuthCandidate[] {
|
||||
candidates.sort((left, right) => this.#compareRankedOAuthCandidates(left, right, provider, modelId));
|
||||
candidates.sort((left, right) => this.#compareRankedOAuthCandidates(left, right, planRequirement));
|
||||
if (!sessionId) {
|
||||
return candidates.map(candidate => ({
|
||||
selection: candidate.selection,
|
||||
@@ -3252,7 +3308,7 @@ export class AuthStorage {
|
||||
for (const candidate of unblocked) {
|
||||
if (
|
||||
candidate !== previous &&
|
||||
this.#compareRankedOAuthCandidatePriority(previous, candidate, provider, modelId) !== 0
|
||||
this.#compareRankedOAuthCandidatePriority(previous, candidate, planRequirement) !== 0
|
||||
) {
|
||||
bucketIndex += 1;
|
||||
}
|
||||
@@ -3297,6 +3353,7 @@ export class AuthStorage {
|
||||
providerKey: string;
|
||||
provider: string;
|
||||
order: number[];
|
||||
planRequirement: OpenAICodexPlanRequirement;
|
||||
credentials: OAuthSelection[];
|
||||
options?: AuthApiKeyOptions;
|
||||
sessionId?: string;
|
||||
@@ -3378,7 +3435,7 @@ export class AuthStorage {
|
||||
blocked,
|
||||
blockedUntil,
|
||||
hasPriorityBoost: strategy.hasPriorityBoost?.(primary) ?? false,
|
||||
planPriority: getOpenAICodexPlanPriority(usage),
|
||||
planPriority: getOpenAICodexPlanPriority(usage, args.planRequirement),
|
||||
secondaryUsed: this.#normalizeUsageFraction(secondaryTarget),
|
||||
secondaryDrainRate: this.#computeWindowDrainRate(
|
||||
secondaryTarget,
|
||||
@@ -3390,7 +3447,7 @@ export class AuthStorage {
|
||||
orderPos,
|
||||
});
|
||||
}
|
||||
return this.#orderRankedOAuthCandidates(ranked, args.sessionId, args.provider, args.options?.modelId);
|
||||
return this.#orderRankedOAuthCandidates(ranked, args.sessionId, args.planRequirement);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -3418,8 +3475,9 @@ export class AuthStorage {
|
||||
const strategy = this.#rankingStrategyResolver?.(provider);
|
||||
const rankingContext: CredentialRankingContext = { modelId: options?.modelId };
|
||||
const blockScope = strategy?.blockScope?.(rankingContext);
|
||||
const requiresProModel = requiresOpenAICodexProModel(provider, options?.modelId);
|
||||
const checkUsage = strategy !== undefined && (credentials.length > 1 || requiresProModel);
|
||||
const planRequirement = resolveOpenAICodexPlanRequirement(provider, options?.modelId);
|
||||
const hasPlanRequirement = planRequirement !== "none";
|
||||
const checkUsage = strategy !== undefined && (credentials.length > 1 || hasPlanRequirement);
|
||||
const sessionCredential = this.#getSessionCredential(provider, sessionId);
|
||||
const sessionPreferredIndex = sessionCredential?.type === "oauth" ? sessionCredential.index : undefined;
|
||||
const sessionPreferredCredential =
|
||||
@@ -3438,12 +3496,13 @@ export class AuthStorage {
|
||||
sessionPreferredIndex !== undefined &&
|
||||
sessionPreferredCanRefreshOrUse &&
|
||||
!this.#isCredentialBlocked(provider, providerKey, sessionPreferredIndex, blockScope);
|
||||
const shouldRank = checkUsage && (!sessionPreferredIsAvailable || requiresProModel);
|
||||
const shouldRank = checkUsage && (!sessionPreferredIsAvailable || hasPlanRequirement);
|
||||
const rankingOrder = shouldRank && sessionId ? credentials.map((_credential, index) => index) : order;
|
||||
const candidates = shouldRank
|
||||
? await this.#rankOAuthSelections({
|
||||
providerKey,
|
||||
provider,
|
||||
planRequirement,
|
||||
order: rankingOrder,
|
||||
credentials,
|
||||
options,
|
||||
@@ -3457,7 +3516,7 @@ export class AuthStorage {
|
||||
.filter((selection): selection is { credential: OAuthCredential; index: number } => Boolean(selection))
|
||||
.map(selection => ({ selection, usage: null, usageChecked: false }));
|
||||
|
||||
if (sessionPreferredIndex !== undefined && !requiresProModel) {
|
||||
if (sessionPreferredIndex !== undefined && !hasPlanRequirement) {
|
||||
const sessionPreferredCandidate = candidates.findIndex(
|
||||
candidate =>
|
||||
!this.#isCredentialBlocked(provider, providerKey, candidate.selection.index, blockScope) &&
|
||||
@@ -3537,10 +3596,12 @@ export class AuthStorage {
|
||||
}),
|
||||
);
|
||||
|
||||
// Skip the Pro-plan filter when no candidate is confirmed Pro, so users with only
|
||||
// non-Pro accounts can still attempt Spark requests (e.g. trial/grandfathered access).
|
||||
const enforceProRequirement =
|
||||
requiresProModel && candidates.some(candidate => hasOpenAICodexProPlan(candidate.usage));
|
||||
// Enforce a tier only when at least one account is confirmed eligible. If
|
||||
// every report is unknown or ineligible, preserve trial/grandfathered access
|
||||
// by allowing the normal candidate fallback to attempt the request.
|
||||
const enforcePlanRequirement =
|
||||
hasPlanRequirement &&
|
||||
candidates.some(candidate => getOpenAICodexPlanEligibility(candidate.usage, planRequirement) === true);
|
||||
|
||||
const fallback = candidates[0];
|
||||
|
||||
@@ -3556,7 +3617,8 @@ export class AuthStorage {
|
||||
allowBlocked: false,
|
||||
prefetchedUsage: candidate.usage,
|
||||
usagePrechecked: candidate.usageChecked,
|
||||
enforceProRequirement,
|
||||
planRequirement,
|
||||
enforcePlanRequirement,
|
||||
strategy,
|
||||
rankingContext,
|
||||
blockScope,
|
||||
@@ -3571,7 +3633,8 @@ export class AuthStorage {
|
||||
allowBlocked: true,
|
||||
prefetchedUsage: fallback.usage,
|
||||
usagePrechecked: fallback.usageChecked,
|
||||
enforceProRequirement,
|
||||
planRequirement,
|
||||
enforcePlanRequirement,
|
||||
strategy,
|
||||
rankingContext,
|
||||
blockScope,
|
||||
@@ -3709,7 +3772,8 @@ export class AuthStorage {
|
||||
allowBlocked: boolean;
|
||||
prefetchedUsage?: UsageReport | null;
|
||||
usagePrechecked?: boolean;
|
||||
enforceProRequirement?: boolean;
|
||||
planRequirement?: OpenAICodexPlanRequirement;
|
||||
enforcePlanRequirement?: boolean;
|
||||
strategy?: CredentialRankingStrategy;
|
||||
rankingContext?: CredentialRankingContext;
|
||||
blockScope?: string;
|
||||
@@ -3722,7 +3786,8 @@ export class AuthStorage {
|
||||
allowBlocked,
|
||||
prefetchedUsage = null,
|
||||
usagePrechecked = false,
|
||||
enforceProRequirement,
|
||||
planRequirement: providedPlanRequirement,
|
||||
enforcePlanRequirement,
|
||||
strategy,
|
||||
rankingContext,
|
||||
blockScope,
|
||||
@@ -3741,12 +3806,13 @@ export class AuthStorage {
|
||||
// refresh / persist / CAS-disable addresses the row by this stable id.
|
||||
const credentialId = this.#getStoredCredentials(provider)[selection.index]?.id;
|
||||
|
||||
const requiresProModel = requiresOpenAICodexProModel(provider, options?.modelId);
|
||||
const applyProFilter = enforceProRequirement ?? requiresProModel;
|
||||
const planRequirement = providedPlanRequirement ?? resolveOpenAICodexPlanRequirement(provider, options?.modelId);
|
||||
const hasPlanRequirement = planRequirement !== "none";
|
||||
const applyPlanFilter = enforcePlanRequirement ?? hasPlanRequirement;
|
||||
let usage: UsageReport | null = null;
|
||||
let usageChecked = false;
|
||||
|
||||
if ((checkUsage && !allowBlocked) || requiresProModel) {
|
||||
if ((checkUsage && !allowBlocked) || hasPlanRequirement) {
|
||||
if (usagePrechecked) {
|
||||
usage = prefetchedUsage;
|
||||
usageChecked = true;
|
||||
@@ -3757,7 +3823,7 @@ export class AuthStorage {
|
||||
});
|
||||
usageChecked = true;
|
||||
}
|
||||
if (applyProFilter && !hasOpenAICodexProPlan(usage)) {
|
||||
if (applyPlanFilter && getOpenAICodexPlanEligibility(usage, planRequirement) !== true) {
|
||||
return undefined;
|
||||
}
|
||||
if (checkUsage && !allowBlocked && usage && strategy && rankingContext) {
|
||||
@@ -3825,7 +3891,7 @@ export class AuthStorage {
|
||||
} else {
|
||||
this.#replaceCredentialAt(provider, selection.index, updated);
|
||||
}
|
||||
if ((checkUsage && !allowBlocked) || requiresProModel) {
|
||||
if ((checkUsage && !allowBlocked) || hasPlanRequirement) {
|
||||
const sameAccount = selection.credential.accountId === updated.accountId;
|
||||
if (!usageChecked || !sameAccount) {
|
||||
usage = await this.#getUsageReport(provider, updated, {
|
||||
@@ -3834,7 +3900,7 @@ export class AuthStorage {
|
||||
});
|
||||
usageChecked = true;
|
||||
}
|
||||
if (applyProFilter && !hasOpenAICodexProPlan(usage)) {
|
||||
if (applyPlanFilter && getOpenAICodexPlanEligibility(usage, planRequirement) !== true) {
|
||||
return undefined;
|
||||
}
|
||||
if (checkUsage && !allowBlocked && usage && strategy && rankingContext) {
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { isUnexpectedSocketCloseMessage } from "@oh-my-pi/pi-utils";
|
||||
import type { Api, AssistantMessage } from "../types";
|
||||
import { AwsCredentialsError } from "./aws";
|
||||
import {
|
||||
AnthropicConnectionError,
|
||||
AnthropicConnectionTimeoutError,
|
||||
@@ -346,7 +347,9 @@ export function classify(error: unknown, api?: Api): number {
|
||||
}
|
||||
}
|
||||
|
||||
if (link instanceof AnthropicConnectionTimeoutError) {
|
||||
if (link instanceof AwsCredentialsError) {
|
||||
kinds |= Flag.AuthFailed;
|
||||
} else if (link instanceof AnthropicConnectionTimeoutError) {
|
||||
kinds |= Flag.Timeout | Flag.Transient;
|
||||
} else if (link instanceof AnthropicConnectionError) {
|
||||
kinds |= Flag.Transient;
|
||||
|
||||
@@ -1462,6 +1462,16 @@ async function* observeDecodedAnthropicSdkEvents(
|
||||
|
||||
const PROVIDER_MAX_RETRIES = 10;
|
||||
|
||||
/**
|
||||
* How long `ping` keepalives may keep extending the idle deadline without any
|
||||
* semantic stream progress, as a multiple of the idle timeout. Anthropic pings
|
||||
* across legitimate generation gaps, so pings count as liveness — but a wedged
|
||||
* upstream that pings forever while producing no events must eventually trip
|
||||
* the idle watchdog instead of hanging an active tool-call stream without a
|
||||
* recovery path (#4900).
|
||||
*/
|
||||
const PING_PROGRESS_MAX_IDLE_MULTIPLIER = 3;
|
||||
|
||||
/**
|
||||
* Log a malformed-stream-envelope anomaly without aborting the turn. The strict
|
||||
* parser would `throw new AnthropicStreamEnvelopeError(...)` here; we instead
|
||||
@@ -2007,11 +2017,20 @@ const streamAnthropicOnce = (
|
||||
}
|
||||
>();
|
||||
|
||||
// Pings keep the idle deadline alive once content is flowing, but a
|
||||
// ping before message_start must not consume the first-event watchdog:
|
||||
// it would flip the (retryable) pre-content stall classification into
|
||||
// a terminal mid-stream idle timeout.
|
||||
// Pings keep the idle deadline alive once content is flowing (Anthropic
|
||||
// bridges legitimate generation gaps with keepalives), but only within a
|
||||
// bounded window: a wedged upstream that pings forever while the model
|
||||
// produces nothing must still trip the idle watchdog, otherwise an
|
||||
// active tool-call stream hangs unrecoverably with no retry (#4900).
|
||||
// A ping before message_start must not consume the first-event watchdog
|
||||
// either: it would flip the (retryable) pre-content stall classification
|
||||
// into a terminal mid-stream idle timeout.
|
||||
let sawNonPingEvent = false;
|
||||
let lastNonPingProgressAtMs = 0;
|
||||
const pingProgressCapMs =
|
||||
idleTimeoutMs !== undefined && idleTimeoutMs > 0
|
||||
? idleTimeoutMs * PING_PROGRESS_MAX_IDLE_MULTIPLIER
|
||||
: undefined;
|
||||
const timedAnthropicStream = iterateWithIdleTimeout(anthropicStream, {
|
||||
idleTimeoutMs,
|
||||
firstItemTimeoutMs: firstEventTimeoutMs,
|
||||
@@ -2021,8 +2040,13 @@ const streamAnthropicOnce = (
|
||||
onFirstItemTimeout: () => activeAbortTracker.abortLocally(firstEventTimeoutAbortError),
|
||||
abortSignal: options?.signal,
|
||||
isProgressItem: item => {
|
||||
if ((item as AnthropicStreamEvent).type === "ping") return sawNonPingEvent;
|
||||
if ((item as AnthropicStreamEvent).type === "ping") {
|
||||
if (!sawNonPingEvent) return false;
|
||||
if (pingProgressCapMs === undefined) return true;
|
||||
return Date.now() - lastNonPingProgressAtMs < pingProgressCapMs;
|
||||
}
|
||||
sawNonPingEvent = true;
|
||||
lastNonPingProgressAtMs = Date.now();
|
||||
return true;
|
||||
},
|
||||
});
|
||||
|
||||
@@ -34,7 +34,7 @@ import {
|
||||
applyResponsesReasoningParams,
|
||||
buildResponsesInput,
|
||||
createInitialResponsesAssistantMessage,
|
||||
getOpenAIResponsesPromptCacheKey,
|
||||
getOpenAIPromptCacheKey,
|
||||
isOpenAIResponsesProgressEvent,
|
||||
parseAzureDeploymentNameMap,
|
||||
processResponsesStream,
|
||||
@@ -348,7 +348,7 @@ function buildParams(
|
||||
model: deploymentName,
|
||||
input: messages,
|
||||
stream: true,
|
||||
prompt_cache_key: getOpenAIResponsesPromptCacheKey(options),
|
||||
prompt_cache_key: getOpenAIPromptCacheKey(options),
|
||||
// Encrypted reasoning replay (applyResponsesReasoningParams) requires
|
||||
// stateless responses, matching the openai provider.
|
||||
store: false,
|
||||
|
||||
@@ -653,7 +653,8 @@ export interface UsageState {
|
||||
sawTokenDelta: boolean;
|
||||
}
|
||||
|
||||
async function handleServerMessage(
|
||||
/** Exported for tests: drives one Cursor server message through the stream (exec waits mark the stream busy). */
|
||||
export async function handleServerMessage(
|
||||
msg: AgentServerMessage,
|
||||
output: AssistantMessage,
|
||||
stream: AssistantMessageEventStream,
|
||||
@@ -675,15 +676,21 @@ async function handleServerMessage(
|
||||
} else if (msgCase === "kvServerMessage") {
|
||||
handleKvServerMessage(msg.message.value as KvServerMessage, blobStore, h2Request);
|
||||
} else if (msgCase === "execServerMessage") {
|
||||
await handleExecServerMessage(
|
||||
msg.message.value as ExecServerMessage,
|
||||
h2Request,
|
||||
execHandlers,
|
||||
onToolResult,
|
||||
requestContextTools,
|
||||
output,
|
||||
stream,
|
||||
state,
|
||||
// The server is waiting on OUR local tool result during this window — no
|
||||
// AssistantMessageEvent flows until the handler finishes. Mark the wait
|
||||
// as local work so the lazy stream idle watchdog attributes the silence
|
||||
// to the tool run instead of aborting a healthy stream (issue #4593).
|
||||
await stream.trackLocalWork(
|
||||
handleExecServerMessage(
|
||||
msg.message.value as ExecServerMessage,
|
||||
h2Request,
|
||||
execHandlers,
|
||||
onToolResult,
|
||||
requestContextTools,
|
||||
output,
|
||||
stream,
|
||||
state,
|
||||
),
|
||||
);
|
||||
} else if (msgCase === "conversationCheckpointUpdate") {
|
||||
handleConversationCheckpointUpdate(msg.message.value, output, usageState, onConversationCheckpoint);
|
||||
|
||||
@@ -12,6 +12,7 @@ import {
|
||||
$flag,
|
||||
asRecord,
|
||||
fetchWithRetry,
|
||||
getInstallId,
|
||||
logger,
|
||||
parseStreamingJson,
|
||||
readSseJson,
|
||||
@@ -24,6 +25,8 @@ import { getEnvApiKey } from "../stream";
|
||||
import type {
|
||||
Api,
|
||||
AssistantMessage,
|
||||
CodexCompactionContext,
|
||||
CodexCompactionRequestContext,
|
||||
Context,
|
||||
FetchImpl,
|
||||
Model,
|
||||
@@ -65,7 +68,7 @@ import {
|
||||
type CodexRequestOptions,
|
||||
type InputItem,
|
||||
type RequestBody,
|
||||
shouldUseCodexResponsesLite,
|
||||
resolveCodexResponsesLite,
|
||||
transformRequestBody,
|
||||
} from "./openai-codex/request-transformer";
|
||||
import { CodexApiError } from "./openai-codex/response-handler";
|
||||
@@ -88,6 +91,7 @@ import {
|
||||
appendReasoningSummaryTextDelta,
|
||||
appendResponsesToolResultMessages,
|
||||
applyOpenAIServiceTier,
|
||||
applyReasoningSummaryDone,
|
||||
buildResponsesDeltaInput,
|
||||
convertResponsesAssistantMessage,
|
||||
convertResponsesInputContent,
|
||||
@@ -95,10 +99,11 @@ import {
|
||||
encodeTextSignatureV1,
|
||||
finalizeCustomToolCallInputDone,
|
||||
finalizePendingResponsesToolCalls,
|
||||
finalizeReasoningThinking,
|
||||
finalizeToolCallArgumentsDone,
|
||||
isOpenAIResponsesProgressEvent,
|
||||
mapOpenAIResponsesStopReason,
|
||||
normalizeOpenAIResponsesPromptCacheKey,
|
||||
normalizeOpenAIPromptCacheKey,
|
||||
populateResponsesUsageFromResponse,
|
||||
promoteResponsesToolUseStopReason,
|
||||
} from "./openai-shared";
|
||||
@@ -116,17 +121,19 @@ export interface OpenAICodexResponsesOptions extends StreamOptions {
|
||||
preferWebsockets?: boolean;
|
||||
serviceTier?: ServiceTier;
|
||||
/**
|
||||
* Opt into the Responses Lite transport contract. Sends
|
||||
* Responses Lite transport override; defaults to the model's catalog
|
||||
* `useResponsesLite` flag (codex-rs `use_responses_lite`). Sends
|
||||
* `x-openai-internal-codex-responses-lite: true` on HTTP requests and on the
|
||||
* WebSocket upgrade (the marker is connection-scoped there, so lite and
|
||||
* non-lite turns never share a pooled socket), strips image detail from
|
||||
* input, and disables parallel tool calling — mirroring codex-rs.
|
||||
* non-lite turns never share a pooled socket), moves instructions/tools
|
||||
* into input items, strips image detail, and disables parallel tool
|
||||
* calling — mirroring codex-rs.
|
||||
*/
|
||||
responsesLite?: boolean;
|
||||
/**
|
||||
* Extra `client_metadata` to include in the request body on both transports.
|
||||
* The canonical Codex envelope is `client_metadata["x-codex-turn-metadata"]`
|
||||
* (JSON string of thread/turn identifiers); flat keys are also accepted.
|
||||
* Additional fields embedded in the canonical
|
||||
* `client_metadata["x-codex-turn-metadata"]` JSON blob. Reserved identity
|
||||
* keys are ignored; extras are never emitted as top-level metadata fields.
|
||||
*/
|
||||
clientMetadata?: Record<string, string>;
|
||||
/**
|
||||
@@ -138,6 +145,49 @@ export interface OpenAICodexResponsesOptions extends StreamOptions {
|
||||
onModerationMetadata?: (metadata: unknown) => void;
|
||||
}
|
||||
|
||||
/** Inputs for synthesizing Codex request identity outside the normal stream path. */
|
||||
export interface OpenAICodexCompatibilityMetadataOptions {
|
||||
sessionId?: string;
|
||||
providerSessionState?: Map<string, ProviderSessionState>;
|
||||
requestKind: OpenAICodexRequestKind;
|
||||
compaction?: CodexCompactionRequestContext;
|
||||
startNewTurn?: boolean;
|
||||
turnStartedAtUnixMs?: number;
|
||||
clientMetadata?: Readonly<Record<string, string>>;
|
||||
/** Add the direct installation header required by `/responses/compact`. */
|
||||
includeInstallationHeader?: boolean;
|
||||
}
|
||||
|
||||
/** Canonical Codex body metadata and compatibility headers for one request. */
|
||||
export interface OpenAICodexCompatibilityMetadata {
|
||||
clientMetadata: Record<string, string>;
|
||||
headers: Record<string, string>;
|
||||
}
|
||||
|
||||
/** Live Codex session state to preserve after a successful history rewrite. */
|
||||
export interface OpenAICodexCompactionResetOptions {
|
||||
providerSessionState?: Map<string, ProviderSessionState>;
|
||||
sessionId?: string;
|
||||
compaction: CodexCompactionContext;
|
||||
}
|
||||
|
||||
/** Add the selected wire implementation to one logical compaction context. */
|
||||
export function createOpenAICodexCompactionRequestContext(options: {
|
||||
context: CodexCompactionContext | undefined;
|
||||
implementation: "responses" | "responses_compaction_v2" | "responses_compact";
|
||||
}): CodexCompactionRequestContext | undefined {
|
||||
const context = options.context;
|
||||
if (!context) return undefined;
|
||||
return {
|
||||
operationId: context.operationId,
|
||||
trigger: context.trigger,
|
||||
reason: context.reason,
|
||||
implementation: options.implementation,
|
||||
phase: context.phase,
|
||||
strategy: context.strategy,
|
||||
};
|
||||
}
|
||||
|
||||
const CODEX_DEBUG = $flag("PI_CODEX_DEBUG");
|
||||
const CODEX_MAX_RETRIES = 5;
|
||||
const CODEX_RETRY_DELAY_MS = 500;
|
||||
@@ -185,7 +235,6 @@ const CODEX_RETRYABLE_EVENT_MESSAGE =
|
||||
const CODEX_PROVIDER_SESSION_STATE_KEY = "openai-codex-responses";
|
||||
const X_CODEX_TURN_STATE_HEADER = "x-codex-turn-state";
|
||||
const X_MODELS_ETAG_HEADER = "x-models-etag";
|
||||
const X_OPENAI_INTERNAL_CODEX_RESPONSES_LITE_HEADER = "x-openai-internal-codex-responses-lite";
|
||||
/** WebSocket frames cannot carry per-request HTTP headers; codex-rs mirrors the lite marker into `client_metadata` under this key. */
|
||||
const CODEX_WS_RESPONSES_LITE_CLIENT_METADATA_KEY = "ws_request_header_x_openai_internal_codex_responses_lite";
|
||||
/** `response.metadata` payload key carrying ChatGPT moderation metadata. */
|
||||
@@ -342,6 +391,237 @@ type CodexWebSocketSessionState = {
|
||||
interface CodexProviderSessionState extends ProviderSessionState {
|
||||
webSocketSessions: Map<string, CodexWebSocketSessionState>;
|
||||
webSocketPublicToPrivate: Map<string, string>;
|
||||
metadataSessions: Map<string, CodexMetadataSessionState>;
|
||||
}
|
||||
|
||||
/** Request classification encoded in Codex turn metadata. */
|
||||
export type OpenAICodexRequestKind = "turn" | "prewarm" | "compaction";
|
||||
|
||||
interface CodexMetadataSessionState {
|
||||
sessionId: string;
|
||||
threadId: string;
|
||||
windowId: string;
|
||||
turnId?: string;
|
||||
turnStartedAtUnixMs?: number;
|
||||
compactionOperationId?: string;
|
||||
reuseTurnForNextRequest?: boolean;
|
||||
}
|
||||
|
||||
interface CodexCompatibilityIdentity {
|
||||
installationId: string;
|
||||
sessionId: string;
|
||||
threadId: string;
|
||||
windowId: string;
|
||||
turnMetadataJson?: string;
|
||||
}
|
||||
|
||||
interface CodexRequestMetadata extends CodexCompatibilityIdentity {
|
||||
turnId: string;
|
||||
turnMetadataJson: string;
|
||||
clientMetadata: Record<string, string>;
|
||||
}
|
||||
|
||||
const CODEX_RESERVED_METADATA_KEYS: Record<string, true> = {
|
||||
installation_id: true,
|
||||
[OPENAI_HEADERS.INSTALLATION_ID]: true,
|
||||
session_id: true,
|
||||
thread_id: true,
|
||||
turn_id: true,
|
||||
window_id: true,
|
||||
[OPENAI_HEADERS.WINDOW_ID]: true,
|
||||
[OPENAI_HEADERS.TURN_METADATA]: true,
|
||||
[OPENAI_HEADERS.PARENT_THREAD_ID]: true,
|
||||
[OPENAI_HEADERS.SUBAGENT]: true,
|
||||
request_kind: true,
|
||||
compaction: true,
|
||||
turn_started_at_unix_ms: true,
|
||||
forked_from_thread_id: true,
|
||||
parent_thread_id: true,
|
||||
subagent_kind: true,
|
||||
thread_source: true,
|
||||
sandbox: true,
|
||||
workspaces: true,
|
||||
};
|
||||
|
||||
function createCodexMetadataSessionState(sessionId: string): CodexMetadataSessionState {
|
||||
return {
|
||||
sessionId,
|
||||
threadId: crypto.randomUUID(),
|
||||
windowId: crypto.randomUUID(),
|
||||
};
|
||||
}
|
||||
|
||||
function getOrCreateCodexMetadataSessionState(
|
||||
sessionId: string,
|
||||
providerState: CodexProviderSessionState | undefined,
|
||||
): CodexMetadataSessionState {
|
||||
if (!providerState) return createCodexMetadataSessionState(sessionId);
|
||||
const existing = providerState.metadataSessions.get(sessionId);
|
||||
if (existing) return existing;
|
||||
const created = createCodexMetadataSessionState(sessionId);
|
||||
providerState.metadataSessions.set(sessionId, created);
|
||||
return created;
|
||||
}
|
||||
|
||||
function createCodexCompatibilityIdentity(session: CodexMetadataSessionState): CodexCompatibilityIdentity {
|
||||
return {
|
||||
installationId: getInstallId(),
|
||||
sessionId: session.sessionId,
|
||||
threadId: session.threadId,
|
||||
windowId: session.windowId,
|
||||
};
|
||||
}
|
||||
|
||||
function resolveCodexStartNewTurn(
|
||||
session: CodexMetadataSessionState,
|
||||
requestKind: OpenAICodexRequestKind,
|
||||
compaction: CodexCompactionRequestContext | undefined,
|
||||
override: boolean | undefined,
|
||||
): boolean {
|
||||
if (requestKind !== "compaction") {
|
||||
if (requestKind === "turn") {
|
||||
const reuseCompactionTurn = session.reuseTurnForNextRequest === true;
|
||||
session.reuseTurnForNextRequest = false;
|
||||
session.compactionOperationId = undefined;
|
||||
if (reuseCompactionTurn) return false;
|
||||
}
|
||||
return override ?? requestKind === "turn";
|
||||
}
|
||||
if (!compaction) return override ?? false;
|
||||
const startsNewOperation = session.compactionOperationId !== compaction.operationId;
|
||||
if (startsNewOperation) session.reuseTurnForNextRequest = false;
|
||||
session.compactionOperationId = compaction.operationId;
|
||||
return override ?? (compaction.phase !== "mid_turn" && startsNewOperation);
|
||||
}
|
||||
|
||||
function toAsciiJsonString(value: Record<string, unknown>): string {
|
||||
return JSON.stringify(value).replace(
|
||||
/[\x7f-\uffff]/g,
|
||||
char => `\\u${char.charCodeAt(0).toString(16).padStart(4, "0")}`,
|
||||
);
|
||||
}
|
||||
|
||||
function createCodexRequestMetadata(
|
||||
session: CodexMetadataSessionState,
|
||||
requestKind: OpenAICodexRequestKind,
|
||||
options: {
|
||||
startNewTurn: boolean;
|
||||
turnStartedAtUnixMs?: number;
|
||||
clientMetadata?: Readonly<Record<string, string>>;
|
||||
compaction?: CodexCompactionRequestContext;
|
||||
},
|
||||
): CodexRequestMetadata {
|
||||
if (options.startNewTurn || !session.turnId) {
|
||||
session.turnId = crypto.randomUUID();
|
||||
session.turnStartedAtUnixMs = options.turnStartedAtUnixMs;
|
||||
}
|
||||
const identity = createCodexCompatibilityIdentity(session);
|
||||
const extra: Record<string, string> = {};
|
||||
const callerMetadata = options.clientMetadata;
|
||||
if (callerMetadata) {
|
||||
for (const key in callerMetadata) {
|
||||
if (!CODEX_RESERVED_METADATA_KEYS[key]) extra[key] = callerMetadata[key];
|
||||
}
|
||||
}
|
||||
const turnMetadata: Record<string, unknown> = {
|
||||
installation_id: identity.installationId,
|
||||
session_id: identity.sessionId,
|
||||
thread_id: identity.threadId,
|
||||
turn_id: session.turnId,
|
||||
window_id: identity.windowId,
|
||||
request_kind: requestKind,
|
||||
};
|
||||
if (options.compaction) {
|
||||
turnMetadata.compaction = {
|
||||
trigger: options.compaction.trigger,
|
||||
reason: options.compaction.reason,
|
||||
implementation: options.compaction.implementation,
|
||||
phase: options.compaction.phase,
|
||||
strategy: options.compaction.strategy,
|
||||
};
|
||||
}
|
||||
if (session.turnStartedAtUnixMs !== undefined) {
|
||||
turnMetadata.turn_started_at_unix_ms = session.turnStartedAtUnixMs;
|
||||
}
|
||||
for (const key in extra) turnMetadata[key] = extra[key];
|
||||
const turnMetadataJson = toAsciiJsonString(turnMetadata);
|
||||
return {
|
||||
...identity,
|
||||
turnId: session.turnId,
|
||||
turnMetadataJson,
|
||||
clientMetadata: {
|
||||
[OPENAI_HEADERS.INSTALLATION_ID]: identity.installationId,
|
||||
session_id: identity.sessionId,
|
||||
thread_id: identity.threadId,
|
||||
[OPENAI_HEADERS.WINDOW_ID]: identity.windowId,
|
||||
turn_id: session.turnId,
|
||||
[OPENAI_HEADERS.TURN_METADATA]: turnMetadataJson,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
function applyCodexCompatibilityHeaders(headers: Headers, metadata: CodexCompatibilityIdentity): void {
|
||||
headers.set(OPENAI_HEADERS.SCOPED_SESSION_ID, metadata.sessionId);
|
||||
headers.set(OPENAI_HEADERS.THREAD_ID, metadata.threadId);
|
||||
headers.set(OPENAI_HEADERS.WINDOW_ID, metadata.windowId);
|
||||
if (metadata.turnMetadataJson) {
|
||||
headers.set(OPENAI_HEADERS.TURN_METADATA, metadata.turnMetadataJson);
|
||||
} else {
|
||||
headers.delete(OPENAI_HEADERS.TURN_METADATA);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Synthesize Codex request identity for raw provider routes such as remote
|
||||
* compaction while reusing the live session's thread, window, and turn.
|
||||
*/
|
||||
export function createOpenAICodexCompatibilityMetadata(
|
||||
options: OpenAICodexCompatibilityMetadataOptions,
|
||||
): OpenAICodexCompatibilityMetadata {
|
||||
const providerState = getCodexProviderSessionState(options.providerSessionState);
|
||||
const sessionId = normalizeOpenAIPromptCacheKey(options.sessionId) ?? crypto.randomUUID();
|
||||
const session = getOrCreateCodexMetadataSessionState(sessionId, providerState);
|
||||
const startNewTurn = resolveCodexStartNewTurn(
|
||||
session,
|
||||
options.requestKind,
|
||||
options.compaction,
|
||||
options.startNewTurn,
|
||||
);
|
||||
const metadata = createCodexRequestMetadata(session, options.requestKind, {
|
||||
startNewTurn,
|
||||
turnStartedAtUnixMs: options.turnStartedAtUnixMs ?? (startNewTurn || !session.turnId ? Date.now() : undefined),
|
||||
clientMetadata: options.clientMetadata,
|
||||
compaction: options.compaction,
|
||||
});
|
||||
const headers = new Headers();
|
||||
applyCodexCompatibilityHeaders(headers, metadata);
|
||||
if (options.includeInstallationHeader) {
|
||||
headers.set(OPENAI_HEADERS.INSTALLATION_ID, metadata.installationId);
|
||||
}
|
||||
return {
|
||||
clientMetadata: { ...metadata.clientMetadata },
|
||||
headers: Object.fromEntries(headers.entries()),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Invalidate Codex history-dependent transport state after compaction while
|
||||
* retaining the session identity and live connection.
|
||||
*/
|
||||
export function resetOpenAICodexHistoryAfterCompaction(options: OpenAICodexCompactionResetOptions): void {
|
||||
const providerState = options.providerSessionState?.get(CODEX_PROVIDER_SESSION_STATE_KEY);
|
||||
if (!isCodexProviderSessionState(providerState)) return;
|
||||
for (const websocketState of providerState.webSocketSessions.values()) {
|
||||
resetCodexWebSocketAppendState(websocketState);
|
||||
if (options.compaction.phase !== "mid_turn") websocketState.turnState = undefined;
|
||||
}
|
||||
const sessionId = normalizeOpenAIPromptCacheKey(options.sessionId);
|
||||
if (!sessionId) return;
|
||||
const metadataSession = providerState.metadataSessions.get(sessionId);
|
||||
if (!metadataSession) return;
|
||||
metadataSession.windowId = crypto.randomUUID();
|
||||
metadataSession.compactionOperationId = undefined;
|
||||
metadataSession.reuseTurnForNextRequest = options.compaction.phase !== "standalone_turn";
|
||||
}
|
||||
|
||||
interface CodexRequestContext {
|
||||
@@ -352,8 +632,10 @@ interface CodexRequestContext {
|
||||
requestHeaders: Record<string, string>;
|
||||
transportSessionId?: string;
|
||||
providerSessionState?: CodexProviderSessionState;
|
||||
isolatedTransportState?: CodexProviderSessionState;
|
||||
websocketState?: CodexWebSocketSessionState;
|
||||
responsesLite: boolean;
|
||||
requestMetadata?: CodexRequestMetadata;
|
||||
transformedBody: RequestBody;
|
||||
rawRequestDump: RawHttpRequestDump;
|
||||
}
|
||||
@@ -610,23 +892,37 @@ function createCodexProviderSessionState(): CodexProviderSessionState {
|
||||
const state: CodexProviderSessionState = {
|
||||
webSocketSessions: new Map(),
|
||||
webSocketPublicToPrivate: new Map(),
|
||||
metadataSessions: new Map(),
|
||||
close: () => {
|
||||
for (const session of state.webSocketSessions.values()) {
|
||||
session.connection?.close("session_disposed");
|
||||
}
|
||||
state.webSocketSessions.clear();
|
||||
state.webSocketPublicToPrivate.clear();
|
||||
state.metadataSessions.clear();
|
||||
},
|
||||
};
|
||||
return state;
|
||||
}
|
||||
|
||||
function isCodexProviderSessionState(state: ProviderSessionState | undefined): state is CodexProviderSessionState {
|
||||
return (
|
||||
state !== undefined &&
|
||||
"webSocketSessions" in state &&
|
||||
state.webSocketSessions instanceof Map &&
|
||||
"webSocketPublicToPrivate" in state &&
|
||||
state.webSocketPublicToPrivate instanceof Map &&
|
||||
"metadataSessions" in state &&
|
||||
state.metadataSessions instanceof Map
|
||||
);
|
||||
}
|
||||
|
||||
function getCodexProviderSessionState(
|
||||
providerSessionState: Map<string, ProviderSessionState> | undefined,
|
||||
): CodexProviderSessionState | undefined {
|
||||
if (!providerSessionState) return undefined;
|
||||
const existing = providerSessionState.get(CODEX_PROVIDER_SESSION_STATE_KEY) as CodexProviderSessionState | undefined;
|
||||
if (existing) return existing;
|
||||
const existing = providerSessionState.get(CODEX_PROVIDER_SESSION_STATE_KEY);
|
||||
if (isCodexProviderSessionState(existing)) return existing;
|
||||
const created = createCodexProviderSessionState();
|
||||
providerSessionState.set(CODEX_PROVIDER_SESSION_STATE_KEY, created);
|
||||
return created;
|
||||
@@ -897,8 +1193,8 @@ async function buildCodexRequestContext(
|
||||
const accountId = getCodexAccountId(apiKey);
|
||||
const baseUrl = model.baseUrl || CODEX_BASE_URL;
|
||||
const url = resolveCodexResponsesUrl(baseUrl);
|
||||
const promptCacheKey = normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId);
|
||||
const transportSessionId = normalizeOpenAIResponsesPromptCacheKey(options?.sessionId);
|
||||
const promptCacheKey = normalizeOpenAIPromptCacheKey(options?.promptCacheKey ?? options?.sessionId);
|
||||
const transportSessionId = normalizeOpenAIPromptCacheKey(options?.sessionId);
|
||||
const transformedBody = await buildTransformedCodexRequestBody(model, context, options, promptCacheKey);
|
||||
|
||||
const requestHeaders = { ...(model.headers ?? {}), ...(options?.headers ?? {}) };
|
||||
@@ -912,19 +1208,56 @@ async function buildCodexRequestContext(
|
||||
};
|
||||
|
||||
const providerSessionState = getCodexProviderSessionState(options?.providerSessionState);
|
||||
const responsesLite = shouldUseCodexResponsesLite(transformedBody, options?.responsesLite);
|
||||
const isolatedTransportState = options?.codexCompaction ? createCodexProviderSessionState() : undefined;
|
||||
const transportProviderSessionState = isolatedTransportState ?? providerSessionState;
|
||||
const responsesLite = resolveCodexResponsesLite(model, options?.responsesLite);
|
||||
const sessionKey = getCodexWebSocketSessionKey(transportSessionId, model, accountId, apiKey, baseUrl, responsesLite);
|
||||
const publicSessionKey = transportSessionId ? `${baseUrl}:${model.id}:${transportSessionId}` : undefined;
|
||||
if (sessionKey && publicSessionKey) {
|
||||
providerSessionState?.webSocketPublicToPrivate.set(publicSessionKey, sessionKey);
|
||||
transportProviderSessionState?.webSocketPublicToPrivate.set(publicSessionKey, sessionKey);
|
||||
}
|
||||
const sharedWebsocketState =
|
||||
sessionKey && providerSessionState
|
||||
? isolatedTransportState
|
||||
? providerSessionState.webSocketSessions.get(sessionKey)
|
||||
: getCodexWebSocketSessionState(sessionKey, providerSessionState)
|
||||
: undefined;
|
||||
const websocketState =
|
||||
sessionKey && providerSessionState ? getCodexWebSocketSessionState(sessionKey, providerSessionState) : undefined;
|
||||
if (websocketState && !isCodexWithinTurnContinuation(context)) {
|
||||
// codex-rs scopes `x-codex-turn-state` to a single user turn: tool-loop
|
||||
// follow-ups echo it, a new user turn starts without it.
|
||||
sessionKey && isolatedTransportState
|
||||
? getCodexWebSocketSessionState(sessionKey, isolatedTransportState)
|
||||
: sharedWebsocketState;
|
||||
if (isolatedTransportState && websocketState && sharedWebsocketState) {
|
||||
websocketState.disableWebsocket = sharedWebsocketState.disableWebsocket;
|
||||
websocketState.turnState = sharedWebsocketState.turnState;
|
||||
websocketState.modelsEtag = sharedWebsocketState.modelsEtag;
|
||||
}
|
||||
const withinTurnContinuation = isCodexWithinTurnContinuation(context);
|
||||
const metadataSessionId = transportSessionId ?? crypto.randomUUID();
|
||||
const metadataSession = getOrCreateCodexMetadataSessionState(metadataSessionId, providerSessionState);
|
||||
const compaction = options?.codexCompaction;
|
||||
const requestKind: OpenAICodexRequestKind = compaction ? "compaction" : "turn";
|
||||
const startNewTurn = resolveCodexStartNewTurn(
|
||||
metadataSession,
|
||||
requestKind,
|
||||
compaction,
|
||||
compaction ? undefined : !withinTurnContinuation,
|
||||
);
|
||||
if (websocketState && startNewTurn) {
|
||||
// Codex scopes turn-state to one turn. Mid-turn compaction and tool-loop
|
||||
// follow-ups preserve it; new user or compaction turns start without it.
|
||||
websocketState.turnState = undefined;
|
||||
}
|
||||
const requestMetadata = createCodexRequestMetadata(metadataSession, requestKind, {
|
||||
startNewTurn,
|
||||
turnStartedAtUnixMs: compaction
|
||||
? startNewTurn || !metadataSession.turnId
|
||||
? Date.now()
|
||||
: undefined
|
||||
: getCodexTurnStartedAtUnixMs(context),
|
||||
clientMetadata: transformedBody.client_metadata,
|
||||
compaction,
|
||||
});
|
||||
transformedBody.client_metadata = requestMetadata.clientMetadata;
|
||||
return {
|
||||
apiKey,
|
||||
accountId,
|
||||
@@ -933,8 +1266,10 @@ async function buildCodexRequestContext(
|
||||
requestHeaders,
|
||||
transportSessionId,
|
||||
providerSessionState,
|
||||
isolatedTransportState,
|
||||
websocketState,
|
||||
responsesLite,
|
||||
requestMetadata,
|
||||
transformedBody,
|
||||
rawRequestDump,
|
||||
};
|
||||
@@ -945,10 +1280,10 @@ export async function buildTransformedCodexRequestBody(
|
||||
model: Model<"openai-codex-responses">,
|
||||
context: Context,
|
||||
options: OpenAICodexResponsesOptions | undefined,
|
||||
promptCacheKey = normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId),
|
||||
promptCacheKey = normalizeOpenAIPromptCacheKey(options?.promptCacheKey ?? options?.sessionId),
|
||||
): Promise<RequestBody> {
|
||||
const params: RequestBody = {
|
||||
model: model.id,
|
||||
model: model.requestModelId ?? model.id,
|
||||
input: convertMessages(model, context),
|
||||
stream: true,
|
||||
prompt_cache_key: promptCacheKey,
|
||||
@@ -1062,19 +1397,21 @@ async function openCodexWebSocketTransport(
|
||||
}> {
|
||||
const canAppendBeforeRequest = websocketState.canAppend === true;
|
||||
const chainedBody = buildCodexChainedRequestBody(requestContext.transformedBody, websocketState);
|
||||
// WebSocket frames cannot carry per-request HTTP headers, so the Responses
|
||||
// Lite marker rides in `client_metadata` on every `response.create`.
|
||||
// WebSocket frames cannot carry per-request HTTP headers. Canonical Codex
|
||||
// request identity is already in `client_metadata`; connection-scoped
|
||||
// compatibility values that can change after the upgrade ride alongside it
|
||||
// on every `response.create`.
|
||||
const websocketClientMetadata = { ...(chainedBody.client_metadata ?? {}) };
|
||||
if (requestContext.responsesLite) {
|
||||
websocketClientMetadata[CODEX_WS_RESPONSES_LITE_CLIENT_METADATA_KEY] = "true";
|
||||
}
|
||||
if (websocketState.turnState) {
|
||||
websocketClientMetadata[X_CODEX_TURN_STATE_HEADER] = websocketState.turnState;
|
||||
}
|
||||
let websocketRequest = {
|
||||
type: "response.create",
|
||||
...chainedBody,
|
||||
...(requestContext.responsesLite
|
||||
? {
|
||||
client_metadata: {
|
||||
...(chainedBody.client_metadata ?? {}),
|
||||
[CODEX_WS_RESPONSES_LITE_CLIENT_METADATA_KEY]: "true",
|
||||
},
|
||||
}
|
||||
: {}),
|
||||
client_metadata: websocketClientMetadata,
|
||||
};
|
||||
const replacementWebsocketRequest = await options?.onPayload?.(websocketRequest, model);
|
||||
if (replacementWebsocketRequest !== undefined) {
|
||||
@@ -1089,8 +1426,17 @@ async function openCodexWebSocketTransport(
|
||||
"websocket",
|
||||
websocketState,
|
||||
requestContext.responsesLite,
|
||||
requestContext.requestMetadata,
|
||||
);
|
||||
const requestBodyForState = structuredCloneJSON(requestContext.transformedBody);
|
||||
// `onPayload` may rewrite the outgoing frame (e.g. drop `stream_options`);
|
||||
// recorded state must reflect what was actually sent — the sequential-cutoff
|
||||
// summary decoder keys off it.
|
||||
if (websocketRequest.stream_options === undefined) {
|
||||
delete requestBodyForState.stream_options;
|
||||
} else {
|
||||
requestBodyForState.stream_options = websocketRequest.stream_options;
|
||||
}
|
||||
requestContext.rawRequestDump.body = websocketRequest;
|
||||
CODEX_DEBUG &&
|
||||
logger.debug("[codex] codex websocket request", {
|
||||
@@ -1126,6 +1472,16 @@ async function openCodexWebSocketTransport(
|
||||
};
|
||||
}
|
||||
|
||||
function getCodexTurnStartedAtUnixMs(context: Context): number {
|
||||
for (let i = context.messages.length - 1; i >= 0; i--) {
|
||||
const message = context.messages[i];
|
||||
if (message?.role === "user" && Number.isFinite(message.timestamp)) {
|
||||
return Math.trunc(message.timestamp);
|
||||
}
|
||||
}
|
||||
return Date.now();
|
||||
}
|
||||
|
||||
/**
|
||||
* True when the request continues the current turn (everything after the
|
||||
* last assistant message is tool results), false when a new user turn starts.
|
||||
@@ -1166,6 +1522,7 @@ async function openCodexSseTransport(
|
||||
wireBody,
|
||||
state,
|
||||
requestContext.responsesLite,
|
||||
requestContext.requestMetadata,
|
||||
requestSetup.requestSignal,
|
||||
requestSetup.firstEventTimeoutMs,
|
||||
event => options?.onSseEvent?.(event, model),
|
||||
@@ -1323,6 +1680,17 @@ class CodexStreamProcessor {
|
||||
this.startTime = init.startTime;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the request actually sent (post-`onPayload`) opted into
|
||||
* sequential-cutoff summary delivery: summaries then arrive as atomic
|
||||
* `response.reasoning_summary_text.done` events and incremental
|
||||
* `.delta`/`.part.*` events are ignored (mirrors codex-rs
|
||||
* `uses_sequential_cutoff_reasoning_summaries`).
|
||||
*/
|
||||
get #sequentialCutoffSummaries(): boolean {
|
||||
return this.runtime.requestBodyForState.stream_options?.reasoning_summary_delivery === "sequential_cutoff";
|
||||
}
|
||||
|
||||
async process(): Promise<CodexStreamCompletion> {
|
||||
const { output, stream } = this;
|
||||
stream.push({ type: "start", partial: output });
|
||||
@@ -1384,6 +1752,7 @@ class CodexStreamProcessor {
|
||||
}
|
||||
|
||||
if (eventType === "response.reasoning_summary_part.added") {
|
||||
if (this.#sequentialCutoffSummaries) return firstTokenTime;
|
||||
if (this.runtime.currentItem?.type === "reasoning") {
|
||||
appendReasoningSummaryPart(
|
||||
this.runtime.currentItem,
|
||||
@@ -1394,6 +1763,7 @@ class CodexStreamProcessor {
|
||||
}
|
||||
|
||||
if (eventType === "response.reasoning_summary_text.delta") {
|
||||
if (this.#sequentialCutoffSummaries) return firstTokenTime;
|
||||
if (this.runtime.currentItem?.type === "reasoning" && this.runtime.currentBlock?.type === "thinking") {
|
||||
appendReasoningSummaryTextDelta(
|
||||
this.runtime.currentItem,
|
||||
@@ -1407,7 +1777,46 @@ class CodexStreamProcessor {
|
||||
return firstTokenTime;
|
||||
}
|
||||
|
||||
if (eventType === "response.reasoning_summary_text.done") {
|
||||
// Outside the cutoff contract the text already streamed via `.delta`.
|
||||
if (!this.#sequentialCutoffSummaries) return firstTokenTime;
|
||||
const entry = this.runtime.openItemForEvent(rawEvent);
|
||||
if (entry?.item.type === "reasoning" && entry.block?.type === "thinking") {
|
||||
if (!firstTokenTime) firstTokenTime = performance.now();
|
||||
const summaryIndex =
|
||||
typeof rawEvent.summary_index === "number" && Number.isFinite(rawEvent.summary_index)
|
||||
? Math.trunc(rawEvent.summary_index)
|
||||
: 0;
|
||||
applyReasoningSummaryDone(
|
||||
entry.item,
|
||||
entry.block,
|
||||
typeof rawEvent.text === "string" ? rawEvent.text : "",
|
||||
summaryIndex,
|
||||
stream,
|
||||
output,
|
||||
entry.contentIndex,
|
||||
);
|
||||
}
|
||||
return firstTokenTime;
|
||||
}
|
||||
|
||||
if (eventType === "response.reasoning_text.delta") {
|
||||
const entry = this.runtime.openItemForEvent(rawEvent);
|
||||
const delta = typeof rawEvent.delta === "string" ? rawEvent.delta : "";
|
||||
if (entry?.item.type === "reasoning" && entry.block?.type === "thinking") {
|
||||
entry.block.thinking += delta;
|
||||
stream.push({
|
||||
type: "thinking_delta",
|
||||
contentIndex: entry.contentIndex,
|
||||
delta,
|
||||
partial: output,
|
||||
});
|
||||
}
|
||||
return firstTokenTime;
|
||||
}
|
||||
|
||||
if (eventType === "response.reasoning_summary_part.done") {
|
||||
if (this.#sequentialCutoffSummaries) return firstTokenTime;
|
||||
if (this.runtime.currentItem?.type === "reasoning" && this.runtime.currentBlock?.type === "thinking") {
|
||||
appendReasoningSummaryPartDone(
|
||||
this.runtime.currentItem,
|
||||
@@ -1522,13 +1931,15 @@ class CodexStreamProcessor {
|
||||
// most-recently-added block may belong to a sibling (#2619). Some Codex
|
||||
// function/custom tool items omit `id`; in that case `output_index` still
|
||||
// routes `output_item.done` to the block that received `output_item.added`.
|
||||
const itemId = typeof (item as { id?: string }).id === "string" ? (item as { id: string }).id : "";
|
||||
const itemId = "id" in item && typeof item.id === "string" ? item.id : "";
|
||||
const entry = (itemId ? runtime.openItems.get(itemId) : null) ?? runtime.openItemForEvent(rawEvent);
|
||||
const block = entry?.block ?? null;
|
||||
const contentIndex = entry?.contentIndex ?? output.content.length - 1;
|
||||
|
||||
if (item.type === "reasoning" && block?.type === "thinking") {
|
||||
block.thinking = item.summary?.map(summary => summary.text).join("\n\n") || "";
|
||||
block.thinking = finalizeReasoningThinking(item, block.thinking, {
|
||||
cumulativeSummarySnapshots: this.#sequentialCutoffSummaries,
|
||||
});
|
||||
block.thinkingSignature = JSON.stringify(item);
|
||||
stream.push({
|
||||
type: "thinking_end",
|
||||
@@ -2105,6 +2516,8 @@ export const streamOpenAICodexResponses: StreamFunction<"openai-codex-responses"
|
||||
stream.push({ type: "error", reason: "error", error: output });
|
||||
}
|
||||
stream.end();
|
||||
} finally {
|
||||
requestContext?.isolatedTransportState?.close();
|
||||
}
|
||||
})();
|
||||
|
||||
@@ -2123,10 +2536,10 @@ export async function prewarmOpenAICodexResponses(
|
||||
const accountId = getCodexAccountId(apiKey);
|
||||
const baseUrl = model.baseUrl || CODEX_BASE_URL;
|
||||
const url = resolveCodexResponsesUrl(baseUrl);
|
||||
const transportSessionId = normalizeOpenAIResponsesPromptCacheKey(options?.sessionId);
|
||||
const transportSessionId = normalizeOpenAIPromptCacheKey(options?.sessionId);
|
||||
const promptCacheKey = transportSessionId;
|
||||
const providerSessionState = getCodexProviderSessionState(options?.providerSessionState);
|
||||
const responsesLite = options?.responsesLite === true;
|
||||
const responsesLite = resolveCodexResponsesLite(model, options?.responsesLite);
|
||||
const sessionKey = getCodexWebSocketSessionKey(transportSessionId, model, accountId, apiKey, baseUrl, responsesLite);
|
||||
const publicSessionKey = transportSessionId ? `${baseUrl}:${model.id}:${transportSessionId}` : undefined;
|
||||
if (publicSessionKey && sessionKey) {
|
||||
@@ -2135,6 +2548,11 @@ export async function prewarmOpenAICodexResponses(
|
||||
if (!sessionKey || !providerSessionState) return;
|
||||
const state = getCodexWebSocketSessionState(sessionKey, providerSessionState);
|
||||
if (!shouldUseCodexWebSocket(model, state, options?.preferWebsockets)) return;
|
||||
const metadataSession = getOrCreateCodexMetadataSessionState(
|
||||
transportSessionId ?? crypto.randomUUID(),
|
||||
providerSessionState,
|
||||
);
|
||||
const requestIdentity = createCodexCompatibilityIdentity(metadataSession);
|
||||
const headers = logger.time(
|
||||
"prewarmCodex:createHeaders",
|
||||
createCodexHeaders,
|
||||
@@ -2145,6 +2563,7 @@ export async function prewarmOpenAICodexResponses(
|
||||
"websocket",
|
||||
state,
|
||||
responsesLite,
|
||||
requestIdentity,
|
||||
);
|
||||
await logger.time(
|
||||
"prewarmCodex:establishWs",
|
||||
@@ -2252,6 +2671,7 @@ export interface OpenAICodexTransportDetails {
|
||||
canAppend: boolean;
|
||||
prewarmed: boolean;
|
||||
hasSessionState: boolean;
|
||||
hasTurnState: boolean;
|
||||
lastFallbackAt?: number;
|
||||
}
|
||||
|
||||
@@ -2267,7 +2687,7 @@ function getCodexWebSocketStateForPublicSession(
|
||||
): CodexWebSocketSessionState | undefined {
|
||||
const baseUrl = options?.baseUrl || model.baseUrl || CODEX_BASE_URL;
|
||||
const providerSessionState = getCodexProviderSessionState(options?.providerSessionState);
|
||||
const normalizedSessionId = normalizeOpenAIResponsesPromptCacheKey(options?.sessionId);
|
||||
const normalizedSessionId = normalizeOpenAIPromptCacheKey(options?.sessionId);
|
||||
const publicSessionKey = normalizedSessionId ? `${baseUrl}:${model.id}:${normalizedSessionId}` : undefined;
|
||||
const privateSessionKey = publicSessionKey
|
||||
? providerSessionState?.webSocketPublicToPrivate.get(publicSessionKey)
|
||||
@@ -2314,6 +2734,7 @@ export function getOpenAICodexTransportDetails(
|
||||
canAppend: state?.canAppend ?? false,
|
||||
prewarmed: state?.prewarmed ?? false,
|
||||
hasSessionState: state !== undefined,
|
||||
hasTurnState: state?.turnState !== undefined,
|
||||
lastFallbackAt: state?.lastFallbackAt,
|
||||
};
|
||||
}
|
||||
@@ -3257,12 +3678,22 @@ async function openCodexSseEventStream(
|
||||
body: RequestBody,
|
||||
state: CodexWebSocketSessionState | undefined,
|
||||
responsesLite: boolean,
|
||||
requestMetadata: CodexRequestMetadata | undefined,
|
||||
signal: AbortSignal | undefined,
|
||||
firstEventTimeoutMs: number | undefined,
|
||||
onSseEvent?: OpenAICodexResponsesOptions["onSseEvent"],
|
||||
fetchOverride?: FetchImpl,
|
||||
): Promise<AsyncGenerator<Record<string, unknown>>> {
|
||||
const headers = createCodexHeaders(requestHeaders, accountId, apiKey, sessionId, "sse", state, responsesLite);
|
||||
const headers = createCodexHeaders(
|
||||
requestHeaders,
|
||||
accountId,
|
||||
apiKey,
|
||||
sessionId,
|
||||
"sse",
|
||||
state,
|
||||
responsesLite,
|
||||
requestMetadata,
|
||||
);
|
||||
CODEX_DEBUG &&
|
||||
logger.debug("[codex] codex request", {
|
||||
url,
|
||||
@@ -3322,6 +3753,7 @@ function createCodexHeaders(
|
||||
transport: CodexTransport = "sse",
|
||||
state?: CodexWebSocketSessionState,
|
||||
responsesLite = false,
|
||||
requestMetadata?: CodexCompatibilityIdentity,
|
||||
): Headers {
|
||||
const headers = new Headers(initHeaders ?? {});
|
||||
headers.delete("x-api-key");
|
||||
@@ -3345,6 +3777,15 @@ function createCodexHeaders(
|
||||
headers.delete(OPENAI_HEADERS.SESSION_ID);
|
||||
headers.delete("x-client-request-id");
|
||||
}
|
||||
headers.delete(OPENAI_HEADERS.INSTALLATION_ID);
|
||||
if (requestMetadata) {
|
||||
applyCodexCompatibilityHeaders(headers, requestMetadata);
|
||||
} else {
|
||||
headers.delete(OPENAI_HEADERS.SCOPED_SESSION_ID);
|
||||
headers.delete(OPENAI_HEADERS.THREAD_ID);
|
||||
headers.delete(OPENAI_HEADERS.WINDOW_ID);
|
||||
headers.delete(OPENAI_HEADERS.TURN_METADATA);
|
||||
}
|
||||
if (state?.turnState) {
|
||||
headers.set(X_CODEX_TURN_STATE_HEADER, state.turnState);
|
||||
} else {
|
||||
@@ -3356,9 +3797,9 @@ function createCodexHeaders(
|
||||
headers.delete(X_MODELS_ETAG_HEADER);
|
||||
}
|
||||
if (responsesLite) {
|
||||
headers.set(X_OPENAI_INTERNAL_CODEX_RESPONSES_LITE_HEADER, "true");
|
||||
headers.set(OPENAI_HEADERS.RESPONSES_LITE, "true");
|
||||
} else {
|
||||
headers.delete(X_OPENAI_INTERNAL_CODEX_RESPONSES_LITE_HEADER);
|
||||
headers.delete(OPENAI_HEADERS.RESPONSES_LITE);
|
||||
}
|
||||
if (transport === "sse") {
|
||||
headers.set("accept", "text/event-stream");
|
||||
@@ -3382,6 +3823,10 @@ function redactHeaders(headers: Headers): Record<string, string> {
|
||||
lower.includes("account") ||
|
||||
lower.includes("session") ||
|
||||
lower.includes("conversation") ||
|
||||
lower.includes("thread") ||
|
||||
lower.includes("window") ||
|
||||
lower.includes("installation") ||
|
||||
lower.startsWith("x-codex-turn") ||
|
||||
lower === "x-client-request-id" ||
|
||||
lower === "cookie"
|
||||
) {
|
||||
|
||||
@@ -1,25 +1,46 @@
|
||||
import type { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import { supportsAllTurnsReasoningContext, supportsCodexReasoningSummary } from "@oh-my-pi/pi-catalog/identity";
|
||||
import { requireSupportedEffort } from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import type { Api, Model } from "../../types";
|
||||
import type { Model } from "../../types";
|
||||
import { mapOpenAIReasoningEffort } from "../openai-shared";
|
||||
|
||||
/** Reasoning replay scope for the Codex Responses API (`reasoning.context`). */
|
||||
export type CodexReasoningContext = "auto" | "current_turn" | "all_turns";
|
||||
|
||||
/** User-facing effort levels accepted by Codex request options. */
|
||||
type CodexCallerEffort = "minimal" | "low" | "medium" | "high" | "xhigh";
|
||||
|
||||
/** Caller literal → catalog `Effort` bridge (the enum is nominal). */
|
||||
const EFFORT_BY_NAME: Record<CodexCallerEffort, Effort> = {
|
||||
minimal: Effort.Minimal,
|
||||
low: Effort.Low,
|
||||
medium: Effort.Medium,
|
||||
high: Effort.High,
|
||||
xhigh: Effort.XHigh,
|
||||
};
|
||||
|
||||
export interface ReasoningConfig {
|
||||
effort: "none" | "minimal" | "low" | "medium" | "high" | "xhigh";
|
||||
effort: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
|
||||
summary?: "auto" | "concise" | "detailed";
|
||||
context?: CodexReasoningContext;
|
||||
/** Pro reasoning serving mode (gpt-5.6+ catalog pro aliases). */
|
||||
mode?: "pro";
|
||||
}
|
||||
|
||||
export interface CodexRequestOptions {
|
||||
reasoningEffort?: ReasoningConfig["effort"];
|
||||
/** User-facing effort; the wire-only `max` tier is reached via the model's effort map. */
|
||||
reasoningEffort?: CodexCallerEffort | "none";
|
||||
reasoningSummary?: ReasoningConfig["summary"] | null;
|
||||
/** Explicit `reasoning.context` override; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */
|
||||
reasoningContext?: CodexReasoningContext;
|
||||
textVerbosity?: "low" | "medium" | "high";
|
||||
include?: string[];
|
||||
/** Responses Lite transport contract: strips image detail and disables parallel tool calling, mirroring codex-rs. */
|
||||
/**
|
||||
* Responses Lite transport override; defaults to the model's
|
||||
* `useResponsesLite`. Lite moves instructions/tools into input items,
|
||||
* strips image detail, and disables parallel tool calling (codex-rs
|
||||
* `use_responses_lite`).
|
||||
*/
|
||||
responsesLite?: boolean;
|
||||
}
|
||||
|
||||
@@ -32,6 +53,8 @@ export interface InputItem {
|
||||
name?: string;
|
||||
output?: unknown;
|
||||
arguments?: unknown;
|
||||
/** `additional_tools` developer item payload (Responses Lite). */
|
||||
tools?: unknown;
|
||||
}
|
||||
|
||||
export interface RequestBody {
|
||||
@@ -42,6 +65,8 @@ export interface RequestBody {
|
||||
input?: InputItem[];
|
||||
tools?: unknown;
|
||||
tool_choice?: unknown;
|
||||
/** Concurrent reasoning-summary delivery (codex-rs `StreamOptions`). */
|
||||
stream_options?: { reasoning_summary_delivery: "sequential_cutoff" };
|
||||
// Sampling controls (temperature/top_p/top_k/min_p/presence_penalty/
|
||||
// repetition_penalty/frequency_penalty/stop) are intentionally absent: the
|
||||
// Codex backend rejects every one with a 400 `Unsupported parameter`, so
|
||||
@@ -60,30 +85,52 @@ export interface RequestBody {
|
||||
[key: string]: unknown;
|
||||
}
|
||||
|
||||
function containsInputImage(value: unknown): boolean {
|
||||
if (!value || typeof value !== "object") return false;
|
||||
if ((value as { type?: unknown }).type === "input_image") return true;
|
||||
if (Array.isArray(value)) {
|
||||
for (const item of value) {
|
||||
if (containsInputImage(item)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
for (const item of Object.values(value)) {
|
||||
if (containsInputImage(item)) return true;
|
||||
}
|
||||
return false;
|
||||
/**
|
||||
* Resolve whether a Codex request uses the Responses Lite transport: an
|
||||
* explicit option wins, otherwise the model's catalog flag (codex-rs
|
||||
* `model_info.use_responses_lite`) decides.
|
||||
*/
|
||||
export function resolveCodexResponsesLite(
|
||||
model: Model<"openai-codex-responses">,
|
||||
requested: boolean | undefined,
|
||||
): boolean {
|
||||
return requested ?? model.useResponsesLite === true;
|
||||
}
|
||||
|
||||
/** Returns whether a Codex request can use the text-only Responses Lite transport. */
|
||||
export function shouldUseCodexResponsesLite(body: RequestBody, requested: boolean | undefined): boolean {
|
||||
return requested === true && !containsInputImage(body.input);
|
||||
/**
|
||||
* Clamp a user-facing effort to the model's ladder, then remap to the wire
|
||||
* tier (e.g. GPT-5.6's shifted five-tier scale sends `max` for user `xhigh`).
|
||||
* A mapped value outside the Codex wire vocabulary is a broken compat/model
|
||||
* effort map — fail loudly rather than silently sending a different tier.
|
||||
*/
|
||||
function mapCodexWireEffort(
|
||||
model: Model<"openai-codex-responses">,
|
||||
effort: CodexCallerEffort,
|
||||
): ReasoningConfig["effort"] {
|
||||
const mapped = mapOpenAIReasoningEffort(model, model.compat, requireSupportedEffort(model, EFFORT_BY_NAME[effort]));
|
||||
switch (mapped) {
|
||||
case "none":
|
||||
case "minimal":
|
||||
case "low":
|
||||
case "medium":
|
||||
case "high":
|
||||
case "xhigh":
|
||||
case "max":
|
||||
return mapped;
|
||||
default:
|
||||
throw new Error(
|
||||
`Effort map for ${model.provider}/${model.id} produced invalid Codex reasoning effort "${mapped}"`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
function getReasoningConfig(model: Model<Api>, options: CodexRequestOptions): ReasoningConfig {
|
||||
function getReasoningConfig(
|
||||
model: Model<"openai-codex-responses">,
|
||||
effort: NonNullable<CodexRequestOptions["reasoningEffort"]>,
|
||||
options: CodexRequestOptions,
|
||||
): ReasoningConfig {
|
||||
const config: ReasoningConfig = {
|
||||
effort:
|
||||
options.reasoningEffort === "none" ? "none" : requireSupportedEffort(model, options.reasoningEffort as Effort),
|
||||
effort: effort === "none" ? "none" : mapCodexWireEffort(model, effort),
|
||||
};
|
||||
// `reasoning.summary` is accepted only from gpt-5.4 onward; earlier Codex ids
|
||||
// (gpt-5.1-codex, gpt-5.3-codex, gpt-5.3-codex-spark) reject it with
|
||||
@@ -196,27 +243,65 @@ function repairToolCallPairs(input: InputItem[]): InputItem[] {
|
||||
* `detail` from every input image (message content and tool outputs) before
|
||||
* sending, letting the server choose.
|
||||
*/
|
||||
function stripImageDetails(input: InputItem[]): void {
|
||||
function stripImageDetails(input: unknown[]): void {
|
||||
for (const item of input) {
|
||||
for (const collection of [item.content, item.output]) {
|
||||
if (!item || typeof item !== "object") continue;
|
||||
const content = "content" in item ? item.content : undefined;
|
||||
const output = "output" in item ? item.output : undefined;
|
||||
for (const collection of [content, output]) {
|
||||
if (!Array.isArray(collection)) continue;
|
||||
for (const part of collection) {
|
||||
if (
|
||||
part &&
|
||||
typeof part === "object" &&
|
||||
(part as { type?: unknown }).type === "input_image" &&
|
||||
"detail" in part
|
||||
) {
|
||||
part.detail = undefined;
|
||||
}
|
||||
if (!part || typeof part !== "object") continue;
|
||||
if (!("type" in part) || part.type !== "input_image") continue;
|
||||
if ("detail" in part) part.detail = undefined;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Structural view of a Responses-style body mutated by the Lite rewrite.
|
||||
* Loose (`unknown`) property types let the turn transformer (`RequestBody`)
|
||||
* and the agent's remote-compaction payloads reuse one shaper.
|
||||
*/
|
||||
export interface CodexLiteShapedBody {
|
||||
instructions?: unknown;
|
||||
tools?: unknown;
|
||||
input?: unknown;
|
||||
parallel_tool_calls?: unknown;
|
||||
}
|
||||
|
||||
/**
|
||||
* Applies the Responses Lite body contract in place (codex-rs
|
||||
* `build_responses_request` with `use_responses_lite`): strips pinned image
|
||||
* detail, forces parallel tool calling off, moves tools into a leading
|
||||
* `additional_tools` developer item and the base instructions into a
|
||||
* developer message, then omits top-level `instructions`/`tools`. Shared by
|
||||
* normal turns and both remote-compaction paths — codex-rs routes
|
||||
* `/responses/compact` through the same builder.
|
||||
*/
|
||||
export function applyCodexResponsesLiteShape(body: CodexLiteShapedBody): void {
|
||||
const input = Array.isArray(body.input) ? body.input : [];
|
||||
stripImageDetails(input);
|
||||
body.parallel_tool_calls = false;
|
||||
const prefix: InputItem[] = [
|
||||
{ type: "additional_tools", role: "developer", tools: Array.isArray(body.tools) ? body.tools : [] },
|
||||
];
|
||||
if (typeof body.instructions === "string" && body.instructions.length > 0) {
|
||||
prefix.push({
|
||||
type: "message",
|
||||
role: "developer",
|
||||
content: [{ type: "input_text", text: body.instructions }],
|
||||
});
|
||||
}
|
||||
body.input = [...prefix, ...input];
|
||||
delete body.instructions;
|
||||
delete body.tools;
|
||||
}
|
||||
|
||||
export async function transformRequestBody(
|
||||
body: RequestBody,
|
||||
model: Model<Api>,
|
||||
model: Model<"openai-codex-responses">,
|
||||
options: CodexRequestOptions = {},
|
||||
prompt?: { developerMessages: string[] },
|
||||
): Promise<RequestBody> {
|
||||
@@ -287,20 +372,13 @@ export async function transformRequestBody(
|
||||
}
|
||||
}
|
||||
|
||||
const responsesLite = shouldUseCodexResponsesLite(body, options.responsesLite);
|
||||
const responsesLite = resolveCodexResponsesLite(model, options.responsesLite);
|
||||
if (responsesLite) {
|
||||
if (Array.isArray(body.input)) {
|
||||
stripImageDetails(body.input);
|
||||
}
|
||||
// Responses Lite does not support parallel tool calling; codex-rs forces
|
||||
// it off (`prompt.parallel_tool_calls && !use_responses_lite`).
|
||||
if (body.tools !== undefined) {
|
||||
body.parallel_tool_calls = false;
|
||||
}
|
||||
applyCodexResponsesLiteShape(body);
|
||||
}
|
||||
|
||||
if (options.reasoningEffort !== undefined) {
|
||||
const reasoningConfig = getReasoningConfig(model, options);
|
||||
const reasoningConfig = getReasoningConfig(model, options.reasoningEffort, options);
|
||||
body.reasoning = {
|
||||
...body.reasoning,
|
||||
...reasoningConfig,
|
||||
@@ -323,6 +401,23 @@ export async function transformRequestBody(
|
||||
} else {
|
||||
delete body.reasoning;
|
||||
}
|
||||
// Catalog pro aliases (`gpt-5.6-*-pro`): applied after the effort branch so
|
||||
// the mode is sent even when no effort is set (the branch above deletes
|
||||
// `body.reasoning` in that case) — mode and effort are independent fields.
|
||||
if (model.reasoningMode) {
|
||||
body.reasoning = { ...body.reasoning, mode: model.reasoningMode };
|
||||
}
|
||||
|
||||
// Concurrent reasoning summaries (codex-rs `concurrent_reasoning_summaries`
|
||||
// feature): `sequential_cutoff` lets the server stream output without
|
||||
// blocking on summary generation. Only meaningful when a summary is
|
||||
// requested; codex-rs additionally gates on its OpenAI provider check,
|
||||
// which is inherent here.
|
||||
if (body.reasoning?.summary !== undefined) {
|
||||
body.stream_options = { reasoning_summary_delivery: "sequential_cutoff" };
|
||||
} else {
|
||||
delete body.stream_options;
|
||||
}
|
||||
|
||||
body.text = {
|
||||
...body.text,
|
||||
|
||||
@@ -82,6 +82,7 @@ import {
|
||||
createInitialResponsesAssistantMessage,
|
||||
createOpenAIStrictToolsState,
|
||||
disableStrictToolsForScope,
|
||||
getOpenAIPromptCacheKey,
|
||||
getOpenAIStrictToolsScope,
|
||||
isCompiledGrammarTooLargeStrictError,
|
||||
isOpenRouterAnthropicModel,
|
||||
@@ -616,6 +617,7 @@ const streamOpenAICompletionsOnce = (
|
||||
apiKey,
|
||||
options?.headers,
|
||||
options?.initiatorOverride,
|
||||
getOpenAIPromptCacheKey(options),
|
||||
);
|
||||
const premiumRequestsTotal = copilotPremiumRequests;
|
||||
let appliedStrictTools = false;
|
||||
@@ -1359,6 +1361,7 @@ function createRequestSetup(
|
||||
apiKey?: string,
|
||||
extraHeaders?: Record<string, string>,
|
||||
initiatorOverride?: MessageAttribution,
|
||||
promptCacheSessionId?: string,
|
||||
): OpenAIRequestSetup & { baseUrl: string } {
|
||||
const apiVersion = $env.AZURE_OPENAI_API_VERSION || "2024-10-21";
|
||||
const deploymentName = parseAzureDeploymentNameMap($env.AZURE_OPENAI_DEPLOYMENT_NAME_MAP).get(model.id) ?? model.id;
|
||||
@@ -1366,6 +1369,7 @@ function createRequestSetup(
|
||||
apiKey,
|
||||
extraHeaders,
|
||||
initiatorOverride,
|
||||
promptCacheSessionId,
|
||||
messages: context.messages,
|
||||
defaultBaseUrl: "https://api.openai.com/v1",
|
||||
// Provider auth/header overlay: Kimi-code hosts require shared client
|
||||
@@ -1413,7 +1417,11 @@ function buildParams(
|
||||
context: Context,
|
||||
options: OpenAICompletionsOptions | undefined,
|
||||
toolStrictModeOverride?: ToolStrictModeOverride,
|
||||
): { params: OpenAICompletionsParams; toolStrictMode: AppliedToolStrictMode; strictToolsApplied: boolean } {
|
||||
): {
|
||||
params: OpenAICompletionsParams;
|
||||
toolStrictMode: AppliedToolStrictMode;
|
||||
strictToolsApplied: boolean;
|
||||
} {
|
||||
const initialPolicy = resolveOpenAICompatForRequest(model, options);
|
||||
const initialCompat = initialPolicy.compat as ResolvedOpenAICompat;
|
||||
|
||||
|
||||
@@ -6318,6 +6318,13 @@ export interface Reasoning {
|
||||
* - `xhigh` is supported for all models after `gpt-5.1-codex-max`.
|
||||
*/
|
||||
effort?: ReasoningEffort | null;
|
||||
/**
|
||||
* **gpt-5.6 and later models only**
|
||||
*
|
||||
* Reasoning serving mode. `pro` routes the request to the pro reasoning
|
||||
* path (more compute per response); omit for the standard path.
|
||||
*/
|
||||
mode?: "pro" | null;
|
||||
/**
|
||||
* @deprecated **Deprecated:** use `summary` instead.
|
||||
*
|
||||
|
||||
@@ -76,7 +76,7 @@ import {
|
||||
createInitialResponsesAssistantMessage,
|
||||
createOpenAIStrictToolsState,
|
||||
disableStrictToolsForScope,
|
||||
getOpenAIResponsesPromptCacheKey,
|
||||
getOpenAIPromptCacheKey,
|
||||
getOpenAIResponsesRoutingSessionId,
|
||||
getOpenAIStrictToolsScope,
|
||||
getOpenRouterResponsesSessionId,
|
||||
@@ -390,7 +390,7 @@ const streamOpenAIResponsesOnce = (
|
||||
// stable prompt-cache key independently. Side-channel calls use this to
|
||||
// avoid perturbing provider conversation state without cold-starting the cache.
|
||||
const routingSessionId = getOpenAIResponsesRoutingSessionId(options);
|
||||
const promptCacheSessionId = getOpenAIResponsesPromptCacheKey(options);
|
||||
const promptCacheSessionId = getOpenAIPromptCacheKey(options);
|
||||
const apiKey = options?.apiKey || getEnvApiKey(model.provider) || "";
|
||||
const { headers, copilotPremiumRequests, baseUrl } = resolveOpenAIRequestSetup(model, {
|
||||
apiKey,
|
||||
@@ -818,7 +818,7 @@ export function buildParams(
|
||||
}
|
||||
|
||||
const cacheRetention = resolveCacheRetention(options?.cacheRetention);
|
||||
const promptCacheKey = getOpenAIResponsesPromptCacheKey(options);
|
||||
const promptCacheKey = getOpenAIPromptCacheKey(options);
|
||||
const modelId = applyWireModelIdTransform(
|
||||
model.requestModelId ?? model.id,
|
||||
model.compat.wireModelIdMode,
|
||||
@@ -904,6 +904,13 @@ export function buildParams(
|
||||
model.thinking?.effortMap?.[effort as NonNullable<OpenAIResponsesOptions["reasoning"]>] ??
|
||||
effort,
|
||||
});
|
||||
// Catalog pro aliases (`gpt-5.6-*-pro`): merge AFTER the compat policy so the
|
||||
// mode survives every policy branch (disabled/omitted effort included) while
|
||||
// keeping whatever effort/summary the policy produced — mode and effort are
|
||||
// independent wire fields.
|
||||
if (model.reasoningMode) {
|
||||
params.reasoning = { ...params.reasoning, mode: model.reasoningMode };
|
||||
}
|
||||
|
||||
applyOpenAIGatewayRouting(params, model.compat);
|
||||
|
||||
|
||||
@@ -123,7 +123,8 @@ export interface OpenAIRequestSetupModel extends OpenAIModelIdentity {
|
||||
compat?: Pick<ResolvedOpenAISharedCompat, "promptCacheSessionHeader">;
|
||||
}
|
||||
|
||||
export interface OpenAIResponsesCacheOptions {
|
||||
/** Cache identity controls shared by OpenAI-family transports. */
|
||||
export interface OpenAICacheOptions {
|
||||
cacheRetention?: CacheRetention;
|
||||
sessionId?: string;
|
||||
promptCacheKey?: string;
|
||||
@@ -175,6 +176,14 @@ function applyCoreWeaveProjectHeader(headers: Record<string, string>): void {
|
||||
}
|
||||
}
|
||||
|
||||
function setHeaderIfAbsent(headers: Record<string, string>, name: string, value: string): void {
|
||||
const normalizedName = name.toLowerCase();
|
||||
for (const existingName in headers) {
|
||||
if (existingName.toLowerCase() === normalizedName) return;
|
||||
}
|
||||
headers[name] = value;
|
||||
}
|
||||
|
||||
export function resolveOpenAIRequestSetup(
|
||||
model: OpenAIRequestSetupModel,
|
||||
options: OpenAIRequestSetupOptions,
|
||||
@@ -257,11 +266,11 @@ export function resolveOpenAIRequestSetup(
|
||||
}
|
||||
|
||||
if (options.openAISessionId && model.provider === "openai") {
|
||||
headers.session_id ??= options.openAISessionId;
|
||||
headers["x-client-request-id"] ??= options.openAISessionId;
|
||||
setHeaderIfAbsent(headers, "session_id", options.openAISessionId);
|
||||
setHeaderIfAbsent(headers, "x-client-request-id", options.openAISessionId);
|
||||
}
|
||||
if (options.promptCacheSessionId && model.compat?.promptCacheSessionHeader) {
|
||||
headers[model.compat.promptCacheSessionHeader] ??= options.promptCacheSessionId;
|
||||
setHeaderIfAbsent(headers, model.compat.promptCacheSessionHeader, options.promptCacheSessionId);
|
||||
}
|
||||
|
||||
if (options.defaultBaseUrl !== undefined) {
|
||||
@@ -368,7 +377,8 @@ export function calculateOpenAIUsageAccounting(accounting: OpenAIUsageAccounting
|
||||
};
|
||||
}
|
||||
|
||||
export function normalizeOpenAIResponsesPromptCacheKey(sessionId: string | undefined): string | undefined {
|
||||
/** Normalize a cache identity to the wire limit accepted by OpenAI-family providers. */
|
||||
export function normalizeOpenAIPromptCacheKey(sessionId: string | undefined): string | undefined {
|
||||
return normalizeOpenAIStableId(sessionId, 64, "pc_");
|
||||
}
|
||||
|
||||
@@ -376,20 +386,21 @@ export function normalizeOpenRouterResponsesSessionId(sessionId: string | undefi
|
||||
return normalizeOpenAIStableId(sessionId, 256, "session_");
|
||||
}
|
||||
|
||||
export function getOpenAIResponsesPromptCacheKey(options: OpenAIResponsesCacheOptions | undefined): string | undefined {
|
||||
/** Resolve a prompt-cache identity, falling back to the provider session unless caching is disabled. */
|
||||
export function getOpenAIPromptCacheKey(options: OpenAICacheOptions | undefined): string | undefined {
|
||||
if (resolveCacheRetention(options?.cacheRetention) === "none") return undefined;
|
||||
return normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId);
|
||||
return normalizeOpenAIPromptCacheKey(options?.promptCacheKey ?? options?.sessionId);
|
||||
}
|
||||
|
||||
export function getOpenAIResponsesRoutingSessionId(
|
||||
options: Pick<OpenAIResponsesCacheOptions, "cacheRetention" | "sessionId"> | undefined,
|
||||
options: Pick<OpenAICacheOptions, "cacheRetention" | "sessionId"> | undefined,
|
||||
): string | undefined {
|
||||
if (resolveCacheRetention(options?.cacheRetention) === "none") return undefined;
|
||||
return normalizeOpenAIResponsesPromptCacheKey(options?.sessionId);
|
||||
return normalizeOpenAIPromptCacheKey(options?.sessionId);
|
||||
}
|
||||
|
||||
export function getOpenRouterResponsesSessionId(
|
||||
options: Pick<OpenAIResponsesCacheOptions, "cacheRetention" | "sessionId"> | undefined,
|
||||
options: Pick<OpenAICacheOptions, "cacheRetention" | "sessionId"> | undefined,
|
||||
): string | undefined {
|
||||
if (resolveCacheRetention(options?.cacheRetention) === "none") return undefined;
|
||||
return normalizeOpenRouterResponsesSessionId(options?.sessionId);
|
||||
@@ -695,13 +706,19 @@ export interface OpenAICompatPolicy {
|
||||
};
|
||||
}
|
||||
|
||||
function mapOpenAIReasoningEffort(
|
||||
/**
|
||||
* Map a user-facing effort to the provider wire value: explicit compat
|
||||
* override first, then the model's baked `thinking.effortMap`, else identity.
|
||||
* Shared by the chat-completions/Responses policy resolver and the Codex
|
||||
* request transformer.
|
||||
*/
|
||||
export function mapOpenAIReasoningEffort(
|
||||
model: Pick<Model, "thinking">,
|
||||
compat: OpenAICompatPolicyCompat,
|
||||
compat: { reasoningEffortMap?: Partial<Record<Effort, string>> } | undefined,
|
||||
effort: string,
|
||||
): string {
|
||||
const level = effort as Effort;
|
||||
return compat.reasoningEffortMap?.[level] ?? model.thinking?.effortMap?.[level] ?? effort;
|
||||
return compat?.reasoningEffortMap?.[level] ?? model.thinking?.effortMap?.[level] ?? effort;
|
||||
}
|
||||
|
||||
function isImplicitDisableWhenNotRequested(disableMode: OpenAIReasoningDisableMode): boolean {
|
||||
@@ -1059,6 +1076,7 @@ export const OPENAI_RESPONSES_PROGRESS_EVENT_TYPES: ReadonlySet<string> = new Se
|
||||
"response.output_item.added",
|
||||
"response.reasoning_summary_part.added",
|
||||
"response.reasoning_summary_text.delta",
|
||||
"response.reasoning_summary_text.done",
|
||||
"response.reasoning_summary_part.done",
|
||||
"response.reasoning_text.delta",
|
||||
"response.content_part.added",
|
||||
@@ -1684,6 +1702,41 @@ export function appendReasoningSummaryPart(
|
||||
item.summary.push(part);
|
||||
}
|
||||
|
||||
// Sequential-cutoff streams may repeat the full canonical summary as later parts.
|
||||
function foldReasoningSummary(parts: ResponseReasoningItem["summary"] | undefined): string {
|
||||
if (!parts) return "";
|
||||
let canonical = "";
|
||||
for (const part of parts) {
|
||||
const text = part.text;
|
||||
if (!text || text === canonical) continue;
|
||||
const extendsCanonical = text.startsWith(canonical) && text[canonical.length] === "\n";
|
||||
canonical = !canonical || extendsCanonical ? text : `${canonical}\n\n${text}`;
|
||||
}
|
||||
return canonical;
|
||||
}
|
||||
|
||||
/** Chooses final reasoning text without making sequential-cutoff results disagree with emitted deltas. */
|
||||
export function finalizeReasoningThinking(
|
||||
item: ResponseReasoningItem,
|
||||
streamedThinking: string,
|
||||
options: { cumulativeSummarySnapshots?: boolean } = {},
|
||||
): string {
|
||||
const summaryThinking = options.cumulativeSummarySnapshots
|
||||
? foldReasoningSummary(item.summary)
|
||||
: (item.summary?.map(part => part.text).join("\n\n") ?? "");
|
||||
if (
|
||||
options.cumulativeSummarySnapshots &&
|
||||
streamedThinking &&
|
||||
summaryThinking &&
|
||||
summaryThinking !== streamedThinking
|
||||
) {
|
||||
return streamedThinking;
|
||||
}
|
||||
if (summaryThinking) return summaryThinking;
|
||||
const contentThinking = item.content?.[0]?.type === "reasoning_text" ? (item.content[0].text ?? "") : "";
|
||||
return contentThinking || streamedThinking || "";
|
||||
}
|
||||
|
||||
export function appendReasoningSummaryTextDelta(
|
||||
item: ResponseReasoningItem,
|
||||
block: ThinkingContent,
|
||||
@@ -1715,6 +1768,36 @@ export function appendReasoningSummaryPartDone(
|
||||
stream.push({ type: "thinking_delta", contentIndex, delta: "\n\n", partial: output });
|
||||
}
|
||||
|
||||
/**
|
||||
* Applies an atomic `response.reasoning_summary_text.done` snapshot.
|
||||
*
|
||||
* Sequential-cutoff streams can replay an index or send the accumulated
|
||||
* summary as a later part. Rebuild the canonical summary and emit only its
|
||||
* append-only suffix. Divergent corrections stay buffered until finalization
|
||||
* so delta consumers never receive suffixes based on unseen replacement text.
|
||||
*/
|
||||
export function applyReasoningSummaryDone(
|
||||
item: ResponseReasoningItem,
|
||||
block: ThinkingContent,
|
||||
text: string,
|
||||
summaryIndex: number,
|
||||
stream: AssistantMessageEventStream,
|
||||
output: AssistantMessage,
|
||||
contentIndex: number,
|
||||
): void {
|
||||
item.summary = item.summary || [];
|
||||
while (item.summary.length <= summaryIndex) {
|
||||
item.summary.push({ type: "summary_text", text: "" });
|
||||
}
|
||||
item.summary[summaryIndex].text = text;
|
||||
const after = foldReasoningSummary(item.summary);
|
||||
if (!after.startsWith(block.thinking)) return;
|
||||
const delta = after.slice(block.thinking.length);
|
||||
if (!delta) return;
|
||||
block.thinking = after;
|
||||
stream.push({ type: "thinking_delta", contentIndex, delta, partial: output });
|
||||
}
|
||||
|
||||
export function appendMessageContentPart(
|
||||
item: ResponseOutputMessage,
|
||||
part: ResponseContentPartAddedEvent["part"] | undefined,
|
||||
@@ -2208,12 +2291,6 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
? lookupOpenItem({ output_index: event.output_index, item_id: item.id ?? item.call_id })
|
||||
: lookupOpenItem({ output_index: event.output_index, item_id: item.id });
|
||||
if (item.type === "reasoning") {
|
||||
const thinking =
|
||||
item.summary?.length > 0
|
||||
? item.summary.map(part => part.text).join("\n\n")
|
||||
: item.content?.[0]?.type === "reasoning_text"
|
||||
? (item.content[0].text ?? "")
|
||||
: "";
|
||||
// Prefer the routed entry; the bare itemId find misroutes when ids are
|
||||
// absent (`undefined === undefined` matches the FIRST thinking block) and
|
||||
// misses entirely when the done-event id drifts from the added-event id.
|
||||
@@ -2224,12 +2301,12 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
| ThinkingContent
|
||||
| undefined);
|
||||
if (reasoningBlock) {
|
||||
reasoningBlock.thinking = thinking;
|
||||
reasoningBlock.thinking = finalizeReasoningThinking(item, reasoningBlock.thinking);
|
||||
reasoningBlock.thinkingSignature = JSON.stringify(item);
|
||||
stream.push({
|
||||
type: "thinking_end",
|
||||
contentIndex: contentIndexOf(reasoningBlock),
|
||||
content: thinking,
|
||||
content: reasoningBlock.thinking,
|
||||
partial: output,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -157,6 +157,7 @@ let openAICompletionsProviderModulePromise: Promise<LazyProviderModule<"openai-c
|
||||
let openAIResponsesProviderModulePromise: Promise<LazyProviderModule<"openai-responses">> | undefined;
|
||||
let ollamaProviderModulePromise: Promise<LazyProviderModule<"ollama-chat">> | undefined;
|
||||
let cursorProviderModulePromise: Promise<LazyProviderModule<"cursor-agent">> | undefined;
|
||||
let cursorProviderModuleOverride: LazyProviderModule<"cursor-agent"> | undefined;
|
||||
let devinProviderModulePromise: Promise<LazyProviderModule<"devin-agent">> | undefined;
|
||||
let bedrockProviderModuleOverride: LazyProviderModule<"bedrock-converse-stream"> | undefined;
|
||||
let bedrockProviderModulePromise: Promise<LazyProviderModule<"bedrock-converse-stream">> | undefined;
|
||||
@@ -167,6 +168,12 @@ export function setBedrockProviderModule(module: BedrockProviderModule): void {
|
||||
};
|
||||
}
|
||||
|
||||
export function setCursorProviderModule(module: CursorProviderModule): void {
|
||||
cursorProviderModuleOverride = {
|
||||
stream: module.streamCursor,
|
||||
};
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Stream forwarding / error helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -245,6 +252,10 @@ function forwardStream<TApi extends Api>(
|
||||
(limits?.openAIIdleEnvFloorsFirstEvent
|
||||
? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, limits.defaultFirstEventTimeoutMs)
|
||||
: getStreamFirstEventTimeoutMs(idleTimeoutMs, limits?.defaultFirstEventTimeoutMs)));
|
||||
// Providers with a server-driven local tool bridge (e.g. the Cursor
|
||||
// exec channel) mark their stream busy while a local tool runs; the
|
||||
// watchdog must not read that silence as a provider stall (#4593).
|
||||
const localWorkSource = source instanceof EventStreamImpl ? source : undefined;
|
||||
const watchedSource = iterateWithIdleTimeout(source, {
|
||||
idleTimeoutMs,
|
||||
firstItemTimeoutMs,
|
||||
@@ -260,6 +271,7 @@ function forwardStream<TApi extends Api>(
|
||||
// `idleTimeoutMs` while we're still legitimately waiting on the model's
|
||||
// first response (slow first-token from reasoning models, cold proxies, etc.).
|
||||
isProgressItem: event => (event as AssistantMessageEvent).type !== "start",
|
||||
hasPendingLocalWork: localWorkSource ? () => localWorkSource.hasPendingLocalWork : undefined,
|
||||
});
|
||||
|
||||
for await (const event of watchedSource) {
|
||||
@@ -411,6 +423,9 @@ function loadOllamaProviderModule(): Promise<LazyProviderModule<"ollama-chat">>
|
||||
}
|
||||
|
||||
function loadCursorProviderModule(): Promise<LazyProviderModule<"cursor-agent">> {
|
||||
if (cursorProviderModuleOverride) {
|
||||
return Promise.resolve(cursorProviderModuleOverride);
|
||||
}
|
||||
cursorProviderModulePromise ||= import("./cursor").then(module => {
|
||||
const provider = module as CursorProviderModule;
|
||||
return { stream: provider.streamCursor };
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
import { createApiKeyLogin } from "./api-key-login";
|
||||
import type { ProviderDefinition } from "./types";
|
||||
|
||||
export const loginNovita = createApiKeyLogin({
|
||||
providerLabel: "Novita",
|
||||
authUrl: "https://novita.ai/settings/key-management",
|
||||
instructions: "Create or copy your API key from the Novita dashboard",
|
||||
promptMessage: "Paste your Novita API key",
|
||||
placeholder: "sk_...",
|
||||
validation: {
|
||||
kind: "models-endpoint",
|
||||
provider: "Novita",
|
||||
modelsUrl: "https://api.novita.ai/openapi/v1/billing/balance/detail",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
},
|
||||
});
|
||||
|
||||
export const novitaProvider = {
|
||||
id: "novita",
|
||||
name: "Novita",
|
||||
login: loginNovita,
|
||||
} satisfies ProviderDefinition & { readonly id: "novita" };
|
||||
@@ -1,5 +1,5 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import { isXAIAccessTokenExpiring, refreshXAIOAuthToken, validateXAIEndpoint, XAIOAuthFlow } from "../xai-oauth";
|
||||
import { isXAIAccessTokenExpiring, loginXAIOAuth, refreshXAIOAuthToken, validateXAIEndpoint } from "../xai-oauth";
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
@@ -11,6 +11,76 @@ function jwtWithExp(exp: number): string {
|
||||
return `${header}.${payload}.sig`;
|
||||
}
|
||||
|
||||
const DISCOVERY_URL = "https://auth.x.ai/.well-known/openid-configuration";
|
||||
const DEVICE_CODE_URL = "https://auth.x.ai/oauth2/device/code";
|
||||
const TOKEN_ENDPOINT = "https://auth.x.ai/oauth2/token";
|
||||
const CLIENT_ID = "b1a00492-073a-47ea-816f-4c329264a828";
|
||||
const SCOPE = "openid profile email offline_access grok-cli:access api:access";
|
||||
|
||||
const DEVICE_AUTHORIZATION = {
|
||||
device_code: "device-code-123",
|
||||
user_code: "ABCD-EFGH",
|
||||
verification_uri: "https://auth.x.ai/activate",
|
||||
verification_uri_complete: "https://auth.x.ai/activate?user_code=ABCD-EFGH",
|
||||
expires_in: 600,
|
||||
interval: 1,
|
||||
};
|
||||
|
||||
type RecordedRequest = {
|
||||
url: string;
|
||||
init: RequestInit | undefined;
|
||||
};
|
||||
|
||||
type TokenResponse = {
|
||||
body: unknown;
|
||||
status?: number;
|
||||
};
|
||||
|
||||
function jsonResponse(body: unknown, status: number = 200): Response {
|
||||
return new Response(JSON.stringify(body), {
|
||||
status,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
|
||||
function createDeviceFlowFetch(tokenResponses: readonly TokenResponse[]) {
|
||||
const requests: RecordedRequest[] = [];
|
||||
let tokenResponseIndex = 0;
|
||||
const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => {
|
||||
const url = typeof input === "string" ? input : input instanceof Request ? input.url : input.toString();
|
||||
requests.push({ url, init });
|
||||
|
||||
if (url === DISCOVERY_URL) {
|
||||
return jsonResponse({ token_endpoint: TOKEN_ENDPOINT });
|
||||
}
|
||||
if (url === DEVICE_CODE_URL) {
|
||||
return jsonResponse(DEVICE_AUTHORIZATION);
|
||||
}
|
||||
if (url === TOKEN_ENDPOINT) {
|
||||
const tokenResponse = tokenResponses[tokenResponseIndex];
|
||||
tokenResponseIndex += 1;
|
||||
if (!tokenResponse) {
|
||||
throw new Error(`Unexpected xAI token poll ${tokenResponseIndex}`);
|
||||
}
|
||||
return jsonResponse(tokenResponse.body, tokenResponse.status);
|
||||
}
|
||||
throw new Error(`Unexpected xAI OAuth request: ${url}`);
|
||||
});
|
||||
|
||||
return {
|
||||
fetchMock: fetchMock as unknown as typeof fetch,
|
||||
requests,
|
||||
};
|
||||
}
|
||||
|
||||
function requestForm(request: RecordedRequest | undefined): URLSearchParams {
|
||||
const body = request?.init?.body;
|
||||
if (!(body instanceof URLSearchParams)) {
|
||||
throw new Error("Expected an application/x-www-form-urlencoded request body");
|
||||
}
|
||||
return body;
|
||||
}
|
||||
|
||||
describe("isXAIAccessTokenExpiring", () => {
|
||||
it("returns false for an empty string", () => {
|
||||
expect(isXAIAccessTokenExpiring("")).toBe(false);
|
||||
@@ -63,97 +133,138 @@ describe("refreshXAIOAuthToken", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("XAIOAuthFlow", () => {
|
||||
it("pins the redirect URI to xAI's allowlisted loopback port", () => {
|
||||
const flow = new XAIOAuthFlow({});
|
||||
|
||||
expect(flow.redirectUri).toBe("http://127.0.0.1:56121/callback");
|
||||
});
|
||||
|
||||
it("uses pasted-code login without starting a callback server", async () => {
|
||||
const serveSpy = vi.spyOn(Bun, "serve").mockImplementation(() => {
|
||||
throw new Error("callback server should not start");
|
||||
});
|
||||
let authUrl = "";
|
||||
let tokenRequestBody = "";
|
||||
const progress: string[] = [];
|
||||
const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => {
|
||||
const url = typeof input === "string" ? input : input instanceof Request ? input.url : input.toString();
|
||||
if (url.includes("/.well-known/openid-configuration")) {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
authorization_endpoint: "https://auth.x.ai/oauth/authorize",
|
||||
token_endpoint: "https://auth.x.ai/oauth/token",
|
||||
}),
|
||||
{ status: 200, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
}
|
||||
tokenRequestBody = init?.body instanceof URLSearchParams ? init.body.toString() : String(init?.body ?? "");
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
describe("loginXAIOAuth", () => {
|
||||
it("performs the RFC 8628 device flow and returns the issued credentials", async () => {
|
||||
const now = 1_800_000_000_000;
|
||||
vi.spyOn(Date, "now").mockReturnValue(now);
|
||||
const { fetchMock, requests } = createDeviceFlowFetch([
|
||||
{
|
||||
body: {
|
||||
access_token: "access-token",
|
||||
refresh_token: "refresh-token",
|
||||
expires_in: 3600,
|
||||
}),
|
||||
{ status: 200, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
},
|
||||
},
|
||||
]);
|
||||
const authEvents: Array<{ url: string; instructions?: string }> = [];
|
||||
const progress: string[] = [];
|
||||
const onAuth = vi.fn((info: { url: string; instructions?: string }) => {
|
||||
authEvents.push(info);
|
||||
});
|
||||
const onProgress = vi.fn((message: string) => {
|
||||
progress.push(message);
|
||||
});
|
||||
const onManualCodeInput = vi.fn(async () => {
|
||||
throw new Error("device authorization must not request a pasted code");
|
||||
});
|
||||
|
||||
const flow = new XAIOAuthFlow({
|
||||
fetch: fetchMock as unknown as typeof fetch,
|
||||
onAuth: info => {
|
||||
authUrl = info.url;
|
||||
},
|
||||
onManualCodeInput: async () => {
|
||||
const parsed = new URL(authUrl);
|
||||
const redirectUri = parsed.searchParams.get("redirect_uri") ?? "";
|
||||
const state = parsed.searchParams.get("state") ?? "";
|
||||
return `${redirectUri}?code=code-xyz&state=${encodeURIComponent(state)}`;
|
||||
},
|
||||
onProgress: message => progress.push(message),
|
||||
const credentials = await loginXAIOAuth({
|
||||
fetch: fetchMock,
|
||||
onAuth,
|
||||
onProgress,
|
||||
onManualCodeInput,
|
||||
});
|
||||
|
||||
const credentials = await flow.login();
|
||||
const authorizeUrl = new URL(authUrl);
|
||||
const tokenParams = new URLSearchParams(tokenRequestBody);
|
||||
expect(requests.map(request => request.url)).toEqual([DISCOVERY_URL, DEVICE_CODE_URL, TOKEN_ENDPOINT]);
|
||||
|
||||
expect(serveSpy).not.toHaveBeenCalled();
|
||||
expect(authorizeUrl.searchParams.get("redirect_uri")).toBe("http://127.0.0.1:56121/callback");
|
||||
expect(progress).toContain("Waiting for pasted authorization code...");
|
||||
expect(tokenParams.get("code")).toBe("code-xyz");
|
||||
expect(credentials.access).toBe("access-token");
|
||||
expect(credentials.refresh).toBe("refresh-token");
|
||||
const discoveryRequest = requests[0];
|
||||
expect(discoveryRequest?.init?.method).toBe("GET");
|
||||
expect(new Headers(discoveryRequest?.init?.headers).get("Accept")).toBe("application/json");
|
||||
|
||||
const deviceRequest = requests[1];
|
||||
expect(deviceRequest?.init?.method).toBe("POST");
|
||||
const deviceHeaders = new Headers(deviceRequest?.init?.headers);
|
||||
expect(deviceHeaders.get("Content-Type")).toBe("application/x-www-form-urlencoded");
|
||||
expect(deviceHeaders.get("Accept")).toBe("application/json");
|
||||
const deviceForm = requestForm(deviceRequest);
|
||||
expect([...deviceForm.keys()].sort()).toEqual(["client_id", "scope"]);
|
||||
expect(Object.fromEntries(deviceForm)).toEqual({
|
||||
client_id: CLIENT_ID,
|
||||
scope: SCOPE,
|
||||
});
|
||||
|
||||
const tokenRequest = requests[2];
|
||||
expect(tokenRequest?.init?.method).toBe("POST");
|
||||
const tokenHeaders = new Headers(tokenRequest?.init?.headers);
|
||||
expect(tokenHeaders.get("Content-Type")).toBe("application/x-www-form-urlencoded");
|
||||
expect(tokenHeaders.get("Accept")).toBe("application/json");
|
||||
const tokenForm = requestForm(tokenRequest);
|
||||
expect([...tokenForm.keys()].sort()).toEqual(["client_id", "device_code", "grant_type"]);
|
||||
expect(Object.fromEntries(tokenForm)).toEqual({
|
||||
grant_type: "urn:ietf:params:oauth:grant-type:device_code",
|
||||
client_id: CLIENT_ID,
|
||||
device_code: DEVICE_AUTHORIZATION.device_code,
|
||||
});
|
||||
|
||||
expect(authEvents).toEqual([
|
||||
{
|
||||
url: DEVICE_AUTHORIZATION.verification_uri_complete,
|
||||
instructions: `Enter code: ${DEVICE_AUTHORIZATION.user_code}`,
|
||||
},
|
||||
]);
|
||||
expect(authEvents[0]?.instructions).not.toMatch(/hermes/i);
|
||||
expect(onManualCodeInput).not.toHaveBeenCalled();
|
||||
expect(progress).toEqual(["Waiting for xAI device authorization..."]);
|
||||
expect(credentials).toEqual({
|
||||
access: "access-token",
|
||||
refresh: "refresh-token",
|
||||
expires: now + 3_300_000,
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("XAIOAuthFlow.exchangeToken", () => {
|
||||
it("rejects when the token-exchange response is missing access_token", async () => {
|
||||
const fetchMock = vi.fn(async (input: string | URL) => {
|
||||
const url = typeof input === "string" ? input : input.toString();
|
||||
if (url.includes("/.well-known/openid-configuration")) {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
authorization_endpoint: "https://auth.x.ai/oauth/authorize",
|
||||
token_endpoint: "https://auth.x.ai/oauth/token",
|
||||
}),
|
||||
{ status: 200, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
}
|
||||
// Token-exchange response deliberately omits `access_token` to exercise
|
||||
// the missing-token rejection path. The value of `refresh_token` here is
|
||||
// a literal test marker, not a real secret — the test verifies
|
||||
// exchangeToken throws before any token would be persisted.
|
||||
return new Response(JSON.stringify({ refresh_token: "stub-refresh-token-for-test-only" }), {
|
||||
status: 200,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
});
|
||||
it("continues through authorization_pending and slow_down responses", async () => {
|
||||
const sleepSpy = vi.spyOn(Bun, "sleep").mockResolvedValue(undefined);
|
||||
const { fetchMock, requests } = createDeviceFlowFetch([
|
||||
{ status: 400, body: { error: "authorization_pending" } },
|
||||
{ status: 400, body: { error: "slow_down" } },
|
||||
{
|
||||
body: {
|
||||
access_token: "eventual-access-token",
|
||||
refresh_token: "eventual-refresh-token",
|
||||
expires_in: 3600,
|
||||
},
|
||||
},
|
||||
]);
|
||||
|
||||
const flow = new XAIOAuthFlow({ fetch: fetchMock as unknown as typeof fetch });
|
||||
await flow.generateAuthUrl("state-abc", "http://127.0.0.1:56121/callback");
|
||||
const credentials = await loginXAIOAuth({ fetch: fetchMock });
|
||||
|
||||
await expect(flow.exchangeToken("code-xyz", "state-abc", "http://127.0.0.1:56121/callback")).rejects.toThrow(
|
||||
/access_token/,
|
||||
const tokenRequests = requests.filter(request => request.url === TOKEN_ENDPOINT);
|
||||
expect(tokenRequests).toHaveLength(3);
|
||||
expect(tokenRequests.map(request => Object.fromEntries(requestForm(request)))).toEqual([
|
||||
{
|
||||
grant_type: "urn:ietf:params:oauth:grant-type:device_code",
|
||||
client_id: CLIENT_ID,
|
||||
device_code: DEVICE_AUTHORIZATION.device_code,
|
||||
},
|
||||
{
|
||||
grant_type: "urn:ietf:params:oauth:grant-type:device_code",
|
||||
client_id: CLIENT_ID,
|
||||
device_code: DEVICE_AUTHORIZATION.device_code,
|
||||
},
|
||||
{
|
||||
grant_type: "urn:ietf:params:oauth:grant-type:device_code",
|
||||
client_id: CLIENT_ID,
|
||||
device_code: DEVICE_AUTHORIZATION.device_code,
|
||||
},
|
||||
]);
|
||||
expect(sleepSpy.mock.calls).toEqual([[1000], [6000]]);
|
||||
expect(credentials.access).toBe("eventual-access-token");
|
||||
expect(credentials.refresh).toBe("eventual-refresh-token");
|
||||
});
|
||||
|
||||
it("rejects a token response that omits access_token", async () => {
|
||||
const { fetchMock, requests } = createDeviceFlowFetch([
|
||||
{
|
||||
body: {
|
||||
refresh_token: "refresh-token",
|
||||
expires_in: 3600,
|
||||
},
|
||||
},
|
||||
]);
|
||||
|
||||
await expect(loginXAIOAuth({ fetch: fetchMock })).rejects.toThrow(
|
||||
/xAI device-code token response missing access_token/,
|
||||
);
|
||||
expect(requests.filter(request => request.url === TOKEN_ENDPOINT)).toHaveLength(1);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
import * as AIError from "../../error";
|
||||
|
||||
const DEVICE_FLOW_CANCEL_MESSAGE = "Login cancelled";
|
||||
const DEVICE_FLOW_TIMEOUT_MESSAGE = "Device flow timed out";
|
||||
const DEVICE_FLOW_SLOW_DOWN_TIMEOUT_MESSAGE =
|
||||
"Device flow timed out after one or more slow_down responses. This is often caused by clock drift in WSL or VM environments. Please sync or restart the VM clock and try again.";
|
||||
const MINIMUM_DEVICE_FLOW_INTERVAL_MS = 1000;
|
||||
const DEFAULT_DEVICE_FLOW_INTERVAL_SECONDS = 5;
|
||||
const SLOW_DOWN_INTERVAL_INCREMENT_MS = 5000;
|
||||
|
||||
/** Result returned by one OAuth device-code polling attempt. */
|
||||
export type OAuthDeviceCodePollResult<T> =
|
||||
| { status: "complete"; value: T }
|
||||
| { status: "pending" }
|
||||
| { status: "slow_down" }
|
||||
| { status: "failed"; message: string };
|
||||
|
||||
/** Options for polling an RFC 8628-style OAuth device-code flow. */
|
||||
export interface OAuthDeviceCodeFlowOptions<T> {
|
||||
/** Poll the provider once and classify the response. */
|
||||
poll(): OAuthDeviceCodePollResult<T> | Promise<OAuthDeviceCodePollResult<T>>;
|
||||
/** Provider-requested polling cadence; defaults to RFC 8628's five seconds. */
|
||||
intervalSeconds?: number;
|
||||
/** Provider-issued expiry window for the device code. */
|
||||
expiresInSeconds?: number;
|
||||
/** Cancels the flow with the legacy "Login cancelled" error. */
|
||||
signal?: AbortSignal;
|
||||
}
|
||||
|
||||
async function abortableDeviceFlowSleep(ms: number, signal: AbortSignal | undefined): Promise<void> {
|
||||
if (!signal) {
|
||||
await Bun.sleep(ms);
|
||||
return;
|
||||
}
|
||||
if (signal.aborted) {
|
||||
throw new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE);
|
||||
}
|
||||
|
||||
const { promise, resolve, reject } = Promise.withResolvers<void>();
|
||||
let timer: Timer | undefined;
|
||||
const onAbort = () => {
|
||||
clearTimeout(timer);
|
||||
reject(new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE));
|
||||
};
|
||||
timer = setTimeout(() => {
|
||||
signal.removeEventListener("abort", onAbort);
|
||||
resolve();
|
||||
}, ms);
|
||||
signal.addEventListener("abort", onAbort, { once: true });
|
||||
await promise;
|
||||
}
|
||||
|
||||
/** Poll an OAuth device-code flow until completion, provider failure, timeout, or cancellation. */
|
||||
export async function pollOAuthDeviceCodeFlow<T>(options: OAuthDeviceCodeFlowOptions<T>): Promise<T> {
|
||||
const deadline =
|
||||
typeof options.expiresInSeconds === "number"
|
||||
? Date.now() + options.expiresInSeconds * 1000
|
||||
: Number.POSITIVE_INFINITY;
|
||||
let intervalMs = Math.max(
|
||||
MINIMUM_DEVICE_FLOW_INTERVAL_MS,
|
||||
Math.floor((options.intervalSeconds ?? DEFAULT_DEVICE_FLOW_INTERVAL_SECONDS) * 1000),
|
||||
);
|
||||
let slowDownResponses = 0;
|
||||
|
||||
while (Date.now() < deadline) {
|
||||
if (options.signal?.aborted) {
|
||||
throw new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE);
|
||||
}
|
||||
const result = await options.poll();
|
||||
if (result.status === "complete") {
|
||||
return result.value;
|
||||
}
|
||||
if (result.status === "failed") {
|
||||
throw new AIError.OAuthError(result.message, { kind: "polling" });
|
||||
}
|
||||
if (result.status === "slow_down") {
|
||||
slowDownResponses += 1;
|
||||
intervalMs = Math.max(MINIMUM_DEVICE_FLOW_INTERVAL_MS, intervalMs + SLOW_DOWN_INTERVAL_INCREMENT_MS);
|
||||
}
|
||||
|
||||
const remainingMs = deadline - Date.now();
|
||||
if (remainingMs <= 0) {
|
||||
break;
|
||||
}
|
||||
await abortableDeviceFlowSleep(Math.min(intervalMs, remainingMs), options.signal);
|
||||
}
|
||||
|
||||
throw new AIError.OAuthError(
|
||||
slowDownResponses > 0 ? DEVICE_FLOW_SLOW_DOWN_TIMEOUT_MESSAGE : DEVICE_FLOW_TIMEOUT_MESSAGE,
|
||||
{ kind: "timeout" },
|
||||
);
|
||||
}
|
||||
@@ -12,99 +12,9 @@ import type {
|
||||
OAuthProviderInterface,
|
||||
} from "./types";
|
||||
|
||||
export * from "./device-code";
|
||||
export type * from "./types";
|
||||
|
||||
const DEVICE_FLOW_CANCEL_MESSAGE = "Login cancelled";
|
||||
const DEVICE_FLOW_TIMEOUT_MESSAGE = "Device flow timed out";
|
||||
const DEVICE_FLOW_SLOW_DOWN_TIMEOUT_MESSAGE =
|
||||
"Device flow timed out after one or more slow_down responses. This is often caused by clock drift in WSL or VM environments. Please sync or restart the VM clock and try again.";
|
||||
const MINIMUM_DEVICE_FLOW_INTERVAL_MS = 1000;
|
||||
const DEFAULT_DEVICE_FLOW_INTERVAL_SECONDS = 5;
|
||||
const SLOW_DOWN_INTERVAL_INCREMENT_MS = 5000;
|
||||
|
||||
/** Result returned by one OAuth device-code polling attempt. */
|
||||
export type OAuthDeviceCodePollResult<T> =
|
||||
| { status: "complete"; value: T }
|
||||
| { status: "pending" }
|
||||
| { status: "slow_down" }
|
||||
| { status: "failed"; message: string };
|
||||
|
||||
/** Options for polling an RFC 8628-style OAuth device-code flow. */
|
||||
export interface OAuthDeviceCodeFlowOptions<T> {
|
||||
/** Poll the provider once and classify the response. */
|
||||
poll(): OAuthDeviceCodePollResult<T> | Promise<OAuthDeviceCodePollResult<T>>;
|
||||
/** Provider-requested polling cadence; defaults to RFC 8628's five seconds. */
|
||||
intervalSeconds?: number;
|
||||
/** Provider-issued expiry window for the device code. */
|
||||
expiresInSeconds?: number;
|
||||
/** Cancels the flow with the legacy "Login cancelled" error. */
|
||||
signal?: AbortSignal;
|
||||
}
|
||||
|
||||
async function abortableDeviceFlowSleep(ms: number, signal: AbortSignal | undefined): Promise<void> {
|
||||
if (!signal) {
|
||||
await Bun.sleep(ms);
|
||||
return;
|
||||
}
|
||||
if (signal.aborted) {
|
||||
throw new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE);
|
||||
}
|
||||
|
||||
const { promise, resolve, reject } = Promise.withResolvers<void>();
|
||||
let timer: Timer | undefined;
|
||||
const onAbort = () => {
|
||||
if (timer) clearTimeout(timer);
|
||||
reject(new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE));
|
||||
};
|
||||
timer = setTimeout(() => {
|
||||
signal.removeEventListener("abort", onAbort);
|
||||
resolve();
|
||||
}, ms);
|
||||
signal.addEventListener("abort", onAbort, { once: true });
|
||||
await promise;
|
||||
}
|
||||
|
||||
/** Poll an OAuth device-code flow until completion, provider failure, timeout, or cancellation. */
|
||||
export async function pollOAuthDeviceCodeFlow<T>(options: OAuthDeviceCodeFlowOptions<T>): Promise<T> {
|
||||
const deadline =
|
||||
typeof options.expiresInSeconds === "number"
|
||||
? Date.now() + options.expiresInSeconds * 1000
|
||||
: Number.POSITIVE_INFINITY;
|
||||
let intervalMs = Math.max(
|
||||
MINIMUM_DEVICE_FLOW_INTERVAL_MS,
|
||||
Math.floor((options.intervalSeconds ?? DEFAULT_DEVICE_FLOW_INTERVAL_SECONDS) * 1000),
|
||||
);
|
||||
let slowDownResponses = 0;
|
||||
|
||||
while (Date.now() < deadline) {
|
||||
if (options.signal?.aborted) {
|
||||
throw new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE);
|
||||
}
|
||||
const result = await options.poll();
|
||||
if (result.status === "complete") {
|
||||
return result.value;
|
||||
}
|
||||
if (result.status === "failed") {
|
||||
throw new AIError.OAuthError(result.message, { kind: "polling" });
|
||||
}
|
||||
if (result.status === "slow_down") {
|
||||
slowDownResponses += 1;
|
||||
intervalMs = Math.max(MINIMUM_DEVICE_FLOW_INTERVAL_MS, intervalMs + SLOW_DOWN_INTERVAL_INCREMENT_MS);
|
||||
}
|
||||
|
||||
const remainingMs = deadline - Date.now();
|
||||
if (remainingMs <= 0) {
|
||||
break;
|
||||
}
|
||||
await abortableDeviceFlowSleep(Math.min(intervalMs, remainingMs), options.signal);
|
||||
}
|
||||
|
||||
throw new AIError.OAuthError(
|
||||
slowDownResponses > 0 ? DEVICE_FLOW_SLOW_DOWN_TIMEOUT_MESSAGE : DEVICE_FLOW_TIMEOUT_MESSAGE,
|
||||
{ kind: "timeout" },
|
||||
);
|
||||
}
|
||||
|
||||
const builtInOAuthProviders: OAuthProviderInfo[] = PROVIDER_REGISTRY.filter(
|
||||
provider => provider.login && provider.showInLoginList !== false,
|
||||
).map(provider => ({
|
||||
|
||||
@@ -1,31 +1,22 @@
|
||||
// Ported from NousResearch/hermes-agent (MIT) — hermes_cli/auth.py xAI sections (L93-111, L2979-3160, L5286-5469).
|
||||
// Device authorization and token refresh adapted from NousResearch/hermes-agent (MIT).
|
||||
|
||||
/**
|
||||
* xAI Grok (SuperGrok or X Premium+) OAuth flow.
|
||||
* xAI Grok OAuth device authorization flow.
|
||||
*
|
||||
* Manual-code PKCE flow using `127.0.0.1:56121/callback` as the allowlisted
|
||||
* redirect URI. One token unlocks Grok-4.x
|
||||
* chat, Grok Imagine image generation, and Grok Voice TTS via subsequent
|
||||
* commits. Endpoint discovery is hardened against MITM via
|
||||
* {@link validateXAIEndpoint}: any non-HTTPS or non-`x.ai`/`*.x.ai` host is
|
||||
* rejected on every call site, not just the first.
|
||||
* Requests an RFC 8628 device code, opens xAI's verification page, and polls
|
||||
* the discovered token endpoint until the user approves the login.
|
||||
*/
|
||||
|
||||
import * as AIError from "../../error";
|
||||
import type { FetchImpl } from "../../types";
|
||||
import { OAuthCallbackFlow, type OAuthCallbackFlowOptions } from "./callback-server";
|
||||
import { generatePKCE } from "./pkce";
|
||||
import { type OAuthDeviceCodePollResult, pollOAuthDeviceCodeFlow } from "./device-code";
|
||||
import type { OAuthController, OAuthCredentials } from "./types";
|
||||
|
||||
// Hermes hermes_cli/auth.py L93-111
|
||||
const XAI_OAUTH_ISSUER = "https://auth.x.ai";
|
||||
const XAI_OAUTH_DISCOVERY_URL = `${XAI_OAUTH_ISSUER}/.well-known/openid-configuration`;
|
||||
const XAI_OAUTH_DEVICE_CODE_URL = `${XAI_OAUTH_ISSUER}/oauth2/device/code`;
|
||||
const XAI_OAUTH_CLIENT_ID = "b1a00492-073a-47ea-816f-4c329264a828";
|
||||
const XAI_OAUTH_SCOPE = "openid profile email offline_access grok-cli:access api:access";
|
||||
const XAI_OAUTH_REDIRECT_HOST = "127.0.0.1";
|
||||
const XAI_OAUTH_REDIRECT_PORT = 56121;
|
||||
const XAI_OAUTH_REDIRECT_PATH = "/callback";
|
||||
const XAI_OAUTH_DOCS_URL = "https://hermes-agent.nousresearch.com/docs/guides/xai-grok-oauth";
|
||||
|
||||
// Mirrors the 5-min skew used by anthropic.ts:160 — keeps every provider on the
|
||||
// same conservative client-side expiry window.
|
||||
@@ -35,18 +26,27 @@ const DISCOVERY_TIMEOUT_MS = 15_000;
|
||||
const TOKEN_REQUEST_TIMEOUT_MS = 20_000;
|
||||
|
||||
interface XAIOAuthDiscovery {
|
||||
authorization_endpoint: string;
|
||||
token_endpoint: string;
|
||||
}
|
||||
|
||||
interface XAIDeviceAuthorization {
|
||||
deviceCode: string;
|
||||
userCode: string;
|
||||
verificationUriComplete: string;
|
||||
expiresInSeconds: number;
|
||||
intervalSeconds: number;
|
||||
}
|
||||
|
||||
function isRecord(value: unknown): value is Record<string, unknown> {
|
||||
return typeof value === "object" && value !== null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate an xAI OIDC discovery endpoint against scheme + host.
|
||||
* Validate an xAI OIDC endpoint against its scheme and host.
|
||||
*
|
||||
* Hermes `_xai_validate_oauth_endpoint` L2997-3035. The discovery response is
|
||||
* long-lived and cached in {@link OAuthCredentials}; a single MITM during
|
||||
* initial login could substitute a malicious `token_endpoint` that would then
|
||||
* receive every future refresh_token. Rejecting non-HTTPS or non-`x.ai` /
|
||||
* `*.x.ai` hosts pins the cached endpoint to the xAI auth origin.
|
||||
* The discovery response is long-lived and its token endpoint receives every
|
||||
* future refresh token. Rejecting non-HTTPS or non-`x.ai` / `*.x.ai` hosts
|
||||
* pins that endpoint to the xAI auth origin.
|
||||
*
|
||||
* @throws Error with message `Invalid xAI <field>: <url>` when the URL fails
|
||||
* either scheme or host validation.
|
||||
@@ -68,11 +68,7 @@ export function validateXAIEndpoint(url: string, field: string): string {
|
||||
return url;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fetch xAI's OIDC discovery document and validate both endpoints.
|
||||
*
|
||||
* Hermes `_xai_oauth_discovery` L3038-3084.
|
||||
*/
|
||||
/** Fetch xAI's OIDC discovery document and validate the token endpoint. */
|
||||
async function xaiOAuthDiscovery(
|
||||
timeoutMs: number = DISCOVERY_TIMEOUT_MS,
|
||||
fetchOverride?: FetchImpl,
|
||||
@@ -111,37 +107,29 @@ async function xaiOAuthDiscovery(
|
||||
{ kind: "validation", provider: "xai", cause: error },
|
||||
);
|
||||
}
|
||||
if (!payload || typeof payload !== "object") {
|
||||
if (!isRecord(payload)) {
|
||||
throw new AIError.OAuthError("xAI OIDC discovery response was not a JSON object.", {
|
||||
kind: "validation",
|
||||
provider: "xai",
|
||||
});
|
||||
}
|
||||
const obj = payload as Record<string, unknown>;
|
||||
const authorizationEndpoint =
|
||||
typeof obj.authorization_endpoint === "string" ? obj.authorization_endpoint.trim() : "";
|
||||
const tokenEndpoint = typeof obj.token_endpoint === "string" ? obj.token_endpoint.trim() : "";
|
||||
if (!authorizationEndpoint || !tokenEndpoint) {
|
||||
throw new AIError.OAuthError("xAI OIDC discovery response was missing required endpoints.", {
|
||||
const tokenEndpoint = typeof payload.token_endpoint === "string" ? payload.token_endpoint.trim() : "";
|
||||
if (!tokenEndpoint) {
|
||||
throw new AIError.OAuthError("xAI OIDC discovery response was missing token_endpoint.", {
|
||||
kind: "validation",
|
||||
provider: "xai",
|
||||
});
|
||||
}
|
||||
validateXAIEndpoint(authorizationEndpoint, "authorization_endpoint");
|
||||
validateXAIEndpoint(tokenEndpoint, "token_endpoint");
|
||||
return {
|
||||
authorization_endpoint: authorizationEndpoint,
|
||||
token_endpoint: tokenEndpoint,
|
||||
};
|
||||
return { token_endpoint: tokenEndpoint };
|
||||
}
|
||||
|
||||
/**
|
||||
* Check whether a JWT access token is at or past its `exp` claim (with an
|
||||
* optional refresh-skew margin).
|
||||
*
|
||||
* Hermes `_xai_access_token_is_expiring` L2979-2994. Returns `false` for any
|
||||
* malformed input — this is a refresh-trigger check, not a validation, so
|
||||
* non-JWTs ("no token in cache") must NOT trigger a spurious refresh.
|
||||
* Returns `false` for malformed input because this is a refresh-trigger check,
|
||||
* not token validation.
|
||||
*/
|
||||
export function isXAIAccessTokenExpiring(jwt: string, skewSeconds: number = 0): boolean {
|
||||
try {
|
||||
@@ -151,7 +139,8 @@ export function isXAIAccessTokenExpiring(jwt: string, skewSeconds: number = 0):
|
||||
const payloadPart = parts[1];
|
||||
if (!payloadPart) return false;
|
||||
const decoded = Buffer.from(payloadPart, "base64url").toString("utf8");
|
||||
const payload = JSON.parse(decoded) as { exp?: unknown };
|
||||
const payload: unknown = JSON.parse(decoded);
|
||||
if (!isRecord(payload)) return false;
|
||||
const exp = payload.exp;
|
||||
if (typeof exp !== "number" || !Number.isFinite(exp)) return false;
|
||||
const now = Math.floor(Date.now() / 1000);
|
||||
@@ -162,161 +151,232 @@ export function isXAIAccessTokenExpiring(jwt: string, skewSeconds: number = 0):
|
||||
}
|
||||
}
|
||||
|
||||
interface BuildXAIAuthorizeUrlOptions {
|
||||
authorizationEndpoint: string;
|
||||
redirectUri: string;
|
||||
codeChallenge: string;
|
||||
state: string;
|
||||
nonce: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the xAI authorization URL.
|
||||
*
|
||||
* Hermes `_xai_oauth_build_authorize_url` L5286-5312. `plan=generic` opts the
|
||||
* consent screen into xAI's generic OAuth plan tier; without it,
|
||||
* `accounts.x.ai` rejects loopback OAuth from non-allowlisted clients.
|
||||
* `referrer=oh-my-pi` lets xAI attribute oh-my-pi-originated logins in their
|
||||
* OAuth server logs (Hermes uses `referrer=hermes-agent`; oh-my-pi mirrors the
|
||||
* pattern with its own attribution string).
|
||||
*/
|
||||
function buildXAIAuthorizeUrl(opts: BuildXAIAuthorizeUrlOptions): string {
|
||||
const params = new URLSearchParams({
|
||||
response_type: "code",
|
||||
client_id: XAI_OAUTH_CLIENT_ID,
|
||||
redirect_uri: opts.redirectUri,
|
||||
scope: XAI_OAUTH_SCOPE,
|
||||
code_challenge: opts.codeChallenge,
|
||||
code_challenge_method: "S256",
|
||||
state: opts.state,
|
||||
nonce: opts.nonce,
|
||||
plan: "generic",
|
||||
referrer: "oh-my-pi",
|
||||
});
|
||||
return `${opts.authorizationEndpoint}?${params.toString()}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* xAI Grok OAuth code flow (Hermes `_xai_oauth_loopback_login` L5315-5469).
|
||||
*/
|
||||
export class XAIOAuthFlow extends OAuthCallbackFlow {
|
||||
#verifier: string = "";
|
||||
#fetch: FetchImpl;
|
||||
|
||||
constructor(ctrl: OAuthController) {
|
||||
super(ctrl, {
|
||||
preferredPort: XAI_OAUTH_REDIRECT_PORT,
|
||||
callbackPath: XAI_OAUTH_REDIRECT_PATH,
|
||||
callbackHostname: XAI_OAUTH_REDIRECT_HOST,
|
||||
redirectUri: `http://${XAI_OAUTH_REDIRECT_HOST}:${XAI_OAUTH_REDIRECT_PORT}${XAI_OAUTH_REDIRECT_PATH}`,
|
||||
manualInputOnly: true,
|
||||
} satisfies OAuthCallbackFlowOptions);
|
||||
this.#fetch = ctrl.fetch ?? fetch;
|
||||
function parseXAIDeviceAuthorization(payload: unknown): XAIDeviceAuthorization {
|
||||
if (!isRecord(payload)) {
|
||||
throw new AIError.OAuthError("xAI device-code response was not a JSON object.", {
|
||||
kind: "validation",
|
||||
provider: "xai",
|
||||
});
|
||||
}
|
||||
|
||||
async generateAuthUrl(state: string, redirectUri: string): Promise<{ url: string; instructions?: string }> {
|
||||
const pkce = await generatePKCE();
|
||||
this.#verifier = pkce.verifier;
|
||||
const nonce = crypto.randomUUID().replace(/-/g, "");
|
||||
|
||||
const discovery = await xaiOAuthDiscovery(DISCOVERY_TIMEOUT_MS, this.#fetch);
|
||||
const url = buildXAIAuthorizeUrl({
|
||||
authorizationEndpoint: discovery.authorization_endpoint,
|
||||
redirectUri,
|
||||
codeChallenge: pkce.challenge,
|
||||
state,
|
||||
nonce,
|
||||
const deviceCode = typeof payload.device_code === "string" ? payload.device_code.trim() : "";
|
||||
const userCode = typeof payload.user_code === "string" ? payload.user_code.trim() : "";
|
||||
const verificationUri = typeof payload.verification_uri === "string" ? payload.verification_uri.trim() : "";
|
||||
const verificationUriComplete =
|
||||
typeof payload.verification_uri_complete === "string" ? payload.verification_uri_complete.trim() : "";
|
||||
const expiresInSeconds = payload.expires_in;
|
||||
const intervalSeconds = payload.interval;
|
||||
if (
|
||||
!deviceCode ||
|
||||
!userCode ||
|
||||
!verificationUri ||
|
||||
!verificationUriComplete ||
|
||||
typeof expiresInSeconds !== "number" ||
|
||||
!Number.isFinite(expiresInSeconds) ||
|
||||
expiresInSeconds <= 0 ||
|
||||
typeof intervalSeconds !== "number" ||
|
||||
!Number.isFinite(intervalSeconds) ||
|
||||
intervalSeconds <= 0
|
||||
) {
|
||||
throw new AIError.OAuthError("xAI device-code response missing or invalid required fields.", {
|
||||
kind: "validation",
|
||||
provider: "xai",
|
||||
});
|
||||
|
||||
return {
|
||||
url,
|
||||
instructions: `Complete login in your browser for xAI Grok (SuperGrok or X Premium+). Docs: ${XAI_OAUTH_DOCS_URL}`,
|
||||
};
|
||||
}
|
||||
|
||||
async exchangeToken(code: string, _state: string, redirectUri: string): Promise<OAuthCredentials> {
|
||||
const discovery = await xaiOAuthDiscovery(DISCOVERY_TIMEOUT_MS, this.#fetch);
|
||||
const tokenEndpoint = validateXAIEndpoint(discovery.token_endpoint, "token_endpoint");
|
||||
validateXAIEndpoint(verificationUri, "verification_uri");
|
||||
validateXAIEndpoint(verificationUriComplete, "verification_uri_complete");
|
||||
return {
|
||||
deviceCode,
|
||||
userCode,
|
||||
verificationUriComplete,
|
||||
expiresInSeconds,
|
||||
intervalSeconds,
|
||||
};
|
||||
}
|
||||
|
||||
const body = new URLSearchParams({
|
||||
grant_type: "authorization_code",
|
||||
client_id: XAI_OAUTH_CLIENT_ID,
|
||||
code,
|
||||
redirect_uri: redirectUri,
|
||||
code_verifier: this.#verifier,
|
||||
function parseXAITokenResponse(payload: unknown, label: string, refreshTokenFallback?: string): OAuthCredentials {
|
||||
if (!isRecord(payload)) {
|
||||
throw new AIError.OAuthError(`${label} was not a JSON object`, {
|
||||
kind: "validation",
|
||||
provider: "xai",
|
||||
});
|
||||
}
|
||||
const accessToken = typeof payload.access_token === "string" ? payload.access_token : "";
|
||||
const responseRefreshToken = typeof payload.refresh_token === "string" ? payload.refresh_token : "";
|
||||
const refreshToken = responseRefreshToken || refreshTokenFallback || "";
|
||||
const expiresInSeconds = payload.expires_in;
|
||||
if (!accessToken) {
|
||||
throw new AIError.OAuthError(`${label} missing access_token`, {
|
||||
kind: "validation",
|
||||
provider: "xai",
|
||||
});
|
||||
}
|
||||
if (!refreshToken) {
|
||||
throw new AIError.OAuthError(`${label} missing refresh_token`, {
|
||||
kind: "validation",
|
||||
provider: "xai",
|
||||
});
|
||||
}
|
||||
if (typeof expiresInSeconds !== "number" || !Number.isFinite(expiresInSeconds)) {
|
||||
throw new AIError.OAuthError(`${label} missing expires_in`, {
|
||||
kind: "validation",
|
||||
provider: "xai",
|
||||
});
|
||||
}
|
||||
return {
|
||||
access: accessToken,
|
||||
refresh: refreshToken,
|
||||
expires: Date.now() + expiresInSeconds * 1000 - ACCESS_TOKEN_CLIENT_SKEW_MS,
|
||||
};
|
||||
}
|
||||
|
||||
const response = await this.#fetch(tokenEndpoint, {
|
||||
async function requestXAIDeviceAuthorization(
|
||||
fetchImpl: FetchImpl,
|
||||
signal?: AbortSignal,
|
||||
): Promise<XAIDeviceAuthorization> {
|
||||
let response: Response;
|
||||
try {
|
||||
const timeoutSignal = AbortSignal.timeout(TOKEN_REQUEST_TIMEOUT_MS);
|
||||
response = await fetchImpl(XAI_OAUTH_DEVICE_CODE_URL, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/x-www-form-urlencoded",
|
||||
Accept: "application/json",
|
||||
},
|
||||
body,
|
||||
signal: AbortSignal.timeout(TOKEN_REQUEST_TIMEOUT_MS),
|
||||
body: new URLSearchParams({
|
||||
client_id: XAI_OAUTH_CLIENT_ID,
|
||||
scope: XAI_OAUTH_SCOPE,
|
||||
}),
|
||||
signal: signal ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal,
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
let detail = "";
|
||||
try {
|
||||
detail = (await response.text()).trim();
|
||||
} catch {
|
||||
// Ignore body-read failures; the status code is the diagnostic.
|
||||
}
|
||||
throw new AIError.OAuthError(`xAI token exchange failed: ${response.status}${detail ? ` ${detail}` : ""}`, {
|
||||
kind: "token-exchange",
|
||||
provider: "xai",
|
||||
status: response.status,
|
||||
});
|
||||
}
|
||||
|
||||
let tokenData: { access_token?: unknown; refresh_token?: unknown; expires_in?: unknown };
|
||||
try {
|
||||
tokenData = (await response.json()) as typeof tokenData;
|
||||
} catch (error) {
|
||||
throw new AIError.OAuthError(
|
||||
`xAI token exchange returned invalid JSON: ${error instanceof Error ? error.message : String(error)}`,
|
||||
{ kind: "validation", provider: "xai", cause: error },
|
||||
);
|
||||
}
|
||||
|
||||
if (typeof tokenData.access_token !== "string" || !tokenData.access_token) {
|
||||
throw new AIError.OAuthError("xAI token exchange response missing access_token", {
|
||||
kind: "validation",
|
||||
provider: "xai",
|
||||
});
|
||||
}
|
||||
if (typeof tokenData.refresh_token !== "string" || !tokenData.refresh_token) {
|
||||
throw new AIError.OAuthError("xAI token exchange response missing refresh_token", {
|
||||
kind: "validation",
|
||||
provider: "xai",
|
||||
});
|
||||
}
|
||||
if (typeof tokenData.expires_in !== "number" || !Number.isFinite(tokenData.expires_in)) {
|
||||
throw new AIError.OAuthError("xAI token exchange response missing expires_in", {
|
||||
kind: "validation",
|
||||
provider: "xai",
|
||||
});
|
||||
}
|
||||
|
||||
return {
|
||||
access: tokenData.access_token,
|
||||
refresh: tokenData.refresh_token,
|
||||
expires: Date.now() + tokenData.expires_in * 1000 - ACCESS_TOKEN_CLIENT_SKEW_MS,
|
||||
};
|
||||
} catch (error) {
|
||||
if (signal?.aborted) throw new AIError.LoginCancelledError();
|
||||
throw new AIError.OAuthError(
|
||||
`xAI device-code request failed: ${error instanceof Error ? error.message : String(error)}`,
|
||||
{ kind: "device-auth", provider: "xai", cause: error },
|
||||
);
|
||||
}
|
||||
|
||||
if (!response.ok) {
|
||||
let detail = "";
|
||||
try {
|
||||
detail = (await response.text()).trim();
|
||||
} catch {
|
||||
// Ignore body-read failures; the status code is the diagnostic.
|
||||
}
|
||||
throw new AIError.OAuthError(`xAI device-code request failed: ${response.status}${detail ? ` ${detail}` : ""}`, {
|
||||
kind: "device-auth",
|
||||
provider: "xai",
|
||||
status: response.status,
|
||||
});
|
||||
}
|
||||
|
||||
let payload: unknown;
|
||||
try {
|
||||
payload = await response.json();
|
||||
} catch (error) {
|
||||
throw new AIError.OAuthError(
|
||||
`xAI device-code response returned invalid JSON: ${error instanceof Error ? error.message : String(error)}`,
|
||||
{ kind: "validation", provider: "xai", cause: error },
|
||||
);
|
||||
}
|
||||
return parseXAIDeviceAuthorization(payload);
|
||||
}
|
||||
|
||||
async function pollXAIDeviceToken(
|
||||
tokenEndpoint: string,
|
||||
deviceCode: string,
|
||||
fetchImpl: FetchImpl,
|
||||
signal?: AbortSignal,
|
||||
): Promise<OAuthDeviceCodePollResult<OAuthCredentials>> {
|
||||
let response: Response;
|
||||
try {
|
||||
const timeoutSignal = AbortSignal.timeout(TOKEN_REQUEST_TIMEOUT_MS);
|
||||
response = await fetchImpl(tokenEndpoint, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/x-www-form-urlencoded",
|
||||
Accept: "application/json",
|
||||
},
|
||||
body: new URLSearchParams({
|
||||
grant_type: "urn:ietf:params:oauth:grant-type:device_code",
|
||||
client_id: XAI_OAUTH_CLIENT_ID,
|
||||
device_code: deviceCode,
|
||||
}),
|
||||
signal: signal ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal,
|
||||
});
|
||||
} catch (error) {
|
||||
if (signal?.aborted) throw new AIError.LoginCancelledError();
|
||||
throw new AIError.OAuthError(
|
||||
`xAI device-code token polling failed: ${error instanceof Error ? error.message : String(error)}`,
|
||||
{ kind: "polling", provider: "xai", cause: error },
|
||||
);
|
||||
}
|
||||
|
||||
let payload: unknown;
|
||||
try {
|
||||
payload = await response.json();
|
||||
} catch (error) {
|
||||
throw new AIError.OAuthError(
|
||||
`xAI device-code token polling returned invalid JSON: ${
|
||||
error instanceof Error ? error.message : String(error)
|
||||
}`,
|
||||
{ kind: "polling", provider: "xai", status: response.status, cause: error },
|
||||
);
|
||||
}
|
||||
|
||||
if (response.ok) {
|
||||
return {
|
||||
status: "complete",
|
||||
value: parseXAITokenResponse(payload, "xAI device-code token response"),
|
||||
};
|
||||
}
|
||||
if (!isRecord(payload)) {
|
||||
throw new AIError.OAuthError(`xAI device-code token polling failed: ${response.status}`, {
|
||||
kind: "polling",
|
||||
provider: "xai",
|
||||
status: response.status,
|
||||
});
|
||||
}
|
||||
|
||||
const errorCode = typeof payload.error === "string" ? payload.error : "";
|
||||
if (errorCode === "authorization_pending") return { status: "pending" };
|
||||
if (errorCode === "slow_down") return { status: "slow_down" };
|
||||
|
||||
const errorDescription = typeof payload.error_description === "string" ? payload.error_description : "";
|
||||
const detail = errorDescription || errorCode || String(response.status);
|
||||
throw new AIError.OAuthError(`xAI device-code token polling failed: ${detail}`, {
|
||||
kind: "polling",
|
||||
provider: "xai",
|
||||
status: response.status,
|
||||
});
|
||||
}
|
||||
|
||||
/** Log in to xAI Grok with the RFC 8628 device authorization grant. */
|
||||
export async function loginXAIOAuth(ctrl: OAuthController): Promise<OAuthCredentials> {
|
||||
return new XAIOAuthFlow(ctrl).login();
|
||||
const fetchImpl = ctrl.fetch ?? fetch;
|
||||
const discovery = await xaiOAuthDiscovery(DISCOVERY_TIMEOUT_MS, fetchImpl);
|
||||
const device = await requestXAIDeviceAuthorization(fetchImpl, ctrl.signal);
|
||||
ctrl.onAuth?.({
|
||||
url: device.verificationUriComplete,
|
||||
instructions: `Enter code: ${device.userCode}`,
|
||||
});
|
||||
ctrl.onProgress?.("Waiting for xAI device authorization...");
|
||||
|
||||
return pollOAuthDeviceCodeFlow({
|
||||
poll: () => pollXAIDeviceToken(discovery.token_endpoint, device.deviceCode, fetchImpl, ctrl.signal),
|
||||
intervalSeconds: device.intervalSeconds,
|
||||
expiresInSeconds: device.expiresInSeconds,
|
||||
signal: ctrl.signal,
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Refresh an xAI OAuth access token using a stored refresh_token.
|
||||
*
|
||||
* Hermes `refresh_xai_oauth_pure` L3087-3160. Re-runs OIDC discovery and
|
||||
* re-validates the cached `token_endpoint` on the refresh hot path so a
|
||||
* cached-but-poisoned endpoint cannot silently leak a refresh_token.
|
||||
* Re-runs OIDC discovery and re-validates the token endpoint before sending
|
||||
* the stored refresh token.
|
||||
*/
|
||||
export async function refreshXAIOAuthToken(refreshToken: string, fetchOverride?: FetchImpl): Promise<OAuthCredentials> {
|
||||
const fetchImpl = fetchOverride ?? fetch;
|
||||
@@ -357,34 +417,14 @@ export async function refreshXAIOAuthToken(refreshToken: string, fetchOverride?:
|
||||
});
|
||||
}
|
||||
|
||||
let data: { access_token?: unknown; refresh_token?: unknown; expires_in?: unknown };
|
||||
let payload: unknown;
|
||||
try {
|
||||
data = (await response.json()) as typeof data;
|
||||
payload = await response.json();
|
||||
} catch (error) {
|
||||
throw new AIError.OAuthError(
|
||||
`xAI token refresh returned invalid JSON: ${error instanceof Error ? error.message : String(error)}`,
|
||||
{ kind: "validation", provider: "xai", cause: error },
|
||||
);
|
||||
}
|
||||
|
||||
if (typeof data.access_token !== "string" || !data.access_token) {
|
||||
throw new AIError.OAuthError("xAI token refresh response missing access_token", {
|
||||
kind: "validation",
|
||||
provider: "xai",
|
||||
});
|
||||
}
|
||||
if (typeof data.expires_in !== "number" || !Number.isFinite(data.expires_in)) {
|
||||
throw new AIError.OAuthError("xAI token refresh response missing expires_in", {
|
||||
kind: "validation",
|
||||
provider: "xai",
|
||||
});
|
||||
}
|
||||
|
||||
const newRefresh = typeof data.refresh_token === "string" && data.refresh_token ? data.refresh_token : refreshToken;
|
||||
|
||||
return {
|
||||
access: data.access_token,
|
||||
refresh: newRefresh,
|
||||
expires: Date.now() + data.expires_in * 1000 - ACCESS_TOKEN_CLIENT_SKEW_MS,
|
||||
};
|
||||
return parseXAITokenResponse(payload, "xAI token refresh response", refreshToken);
|
||||
}
|
||||
|
||||
@@ -34,6 +34,7 @@ import { minimaxCodeCnProvider } from "./minimax-code-cn";
|
||||
import { mistralProvider } from "./mistral";
|
||||
import { moonshotProvider } from "./moonshot";
|
||||
import { nanogptProvider } from "./nanogpt";
|
||||
import { novitaProvider } from "./novita";
|
||||
import { nvidiaProvider } from "./nvidia";
|
||||
import { ollamaProvider } from "./ollama";
|
||||
import { ollamaCloudProvider } from "./ollama-cloud";
|
||||
@@ -110,6 +111,7 @@ const ALL = [
|
||||
fireworksProvider,
|
||||
togetherProvider,
|
||||
nvidiaProvider,
|
||||
novitaProvider,
|
||||
huggingfaceProvider,
|
||||
perplexityProvider,
|
||||
qianfanProvider,
|
||||
|
||||
@@ -14,5 +14,4 @@ export const xaiOauthProvider = {
|
||||
const { refreshXAIOAuthToken } = await import("./oauth/xai-oauth");
|
||||
return refreshXAIOAuthToken(credentials.refresh);
|
||||
},
|
||||
pasteCodeFlow: true,
|
||||
} as const satisfies ProviderDefinition;
|
||||
|
||||
@@ -1632,6 +1632,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
toolChoice: mapOpenAiToolChoice(options?.toolChoice),
|
||||
serviceTier: options?.serviceTier,
|
||||
preferWebsockets: options?.preferWebsockets,
|
||||
codexCompaction: options?.codexCompaction,
|
||||
reasoningSummary: options?.hideThinkingSummary ? null : "detailed",
|
||||
textVerbosity: options?.textVerbosity,
|
||||
});
|
||||
|
||||
@@ -323,6 +323,30 @@ export interface RawSseEvent {
|
||||
raw: string[];
|
||||
}
|
||||
|
||||
/** Lifecycle fields shared by every Codex compaction implementation. */
|
||||
export interface CodexCompactionContext {
|
||||
/** Stable only for one logical compaction, including parallel summary calls. */
|
||||
operationId: string;
|
||||
trigger: "manual" | "auto";
|
||||
reason: "user_requested" | "context_limit" | "model_downshift" | "comp_hash_changed";
|
||||
phase: "standalone_turn" | "pre_turn" | "mid_turn";
|
||||
strategy: "memento" | "prefix_compaction";
|
||||
}
|
||||
|
||||
/** Canonical nested metadata serialized into the Codex turn envelope. */
|
||||
export interface CodexCompactionMetadata {
|
||||
trigger: "manual" | "auto";
|
||||
reason: "user_requested" | "context_limit" | "model_downshift" | "comp_hash_changed";
|
||||
implementation: "responses" | "responses_compaction_v2" | "responses_compact";
|
||||
phase: "standalone_turn" | "pre_turn" | "mid_turn";
|
||||
strategy: "memento" | "prefix_compaction";
|
||||
}
|
||||
|
||||
/** Dispatch context combining canonical metadata with its local operation identity. */
|
||||
export interface CodexCompactionRequestContext extends CodexCompactionMetadata {
|
||||
operationId: string;
|
||||
}
|
||||
|
||||
export interface StreamOptions {
|
||||
temperature?: number;
|
||||
topP?: number;
|
||||
@@ -388,9 +412,9 @@ export interface StreamOptions {
|
||||
*/
|
||||
sessionId?: string;
|
||||
/**
|
||||
* Optional prompt-cache identity. When set, OpenAI Responses-compatible
|
||||
* providers use this for `prompt_cache_key` while keeping `sessionId` for
|
||||
* provider routing / conversation headers.
|
||||
* Optional prompt-cache identity. OpenAI-family providers use this for
|
||||
* `prompt_cache_key` payloads and cache-affinity headers such as
|
||||
* `x-grok-conv-id`; when omitted, they fall back to `sessionId`.
|
||||
*/
|
||||
promptCacheKey?: string;
|
||||
/**
|
||||
@@ -398,6 +422,8 @@ export interface StreamOptions {
|
||||
* Providers can use this to persist transport/session state between turns.
|
||||
*/
|
||||
providerSessionState?: Map<string, ProviderSessionState>;
|
||||
/** Canonical Codex compaction classification; ignored by other providers. */
|
||||
codexCompaction?: CodexCompactionRequestContext;
|
||||
/**
|
||||
* Force Gemini model-mode Interactions API transport for providers that support it.
|
||||
* When unset, those providers may still use Interactions to continue known
|
||||
|
||||
@@ -10,6 +10,14 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
|
||||
resultSettled = false;
|
||||
#failed = false;
|
||||
#error: unknown = undefined;
|
||||
/**
|
||||
* Consumer-side local operations currently in flight for this stream — a
|
||||
* provider transport waiting on a server-requested local tool bridge
|
||||
* (e.g. the Cursor exec channel) before it can send the result upstream.
|
||||
* While non-zero, event silence is attributable to our own pending work,
|
||||
* not a provider stall; idle watchdogs consult {@link hasPendingLocalWork}.
|
||||
*/
|
||||
#pendingLocalWork = 0;
|
||||
finalResultPromise: Promise<R>;
|
||||
resolveFinalResult!: (result: R) => void;
|
||||
rejectFinalResult!: (err: unknown) => void;
|
||||
@@ -116,6 +124,24 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
|
||||
result(): Promise<R> {
|
||||
return this.finalResultPromise;
|
||||
}
|
||||
|
||||
/** True while local work tracked via {@link trackLocalWork} is pending. */
|
||||
get hasPendingLocalWork(): boolean {
|
||||
return this.#pendingLocalWork > 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Track a local-work promise so idle watchdogs on this stream do not treat
|
||||
* the event silence while it is pending as a provider stall.
|
||||
*/
|
||||
async trackLocalWork<TWork>(work: Promise<TWork>): Promise<TWork> {
|
||||
this.#pendingLocalWork++;
|
||||
try {
|
||||
return await work;
|
||||
} finally {
|
||||
this.#pendingLocalWork--;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
export class AssistantMessageEventStream extends EventStream<AssistantMessageEvent, AssistantMessage> {
|
||||
|
||||
@@ -135,6 +135,16 @@ export interface IdleTimeoutIteratorOptions {
|
||||
* keepalive/no-op events from keeping a stalled tool call alive forever.
|
||||
*/
|
||||
isProgressItem?: (item: unknown) => boolean;
|
||||
/**
|
||||
* Reports consumer-side local work in flight for the stream: the provider
|
||||
* transport is waiting on a server-requested local tool bridge (e.g. the
|
||||
* Cursor exec channel) before anything can flow upstream again. While it
|
||||
* returns true, an expired idle / first-item deadline slides forward
|
||||
* instead of aborting — the silence is ours, not a provider stall. The
|
||||
* watchdog re-arms with a full budget once the local work completes, so a
|
||||
* provider that stalls afterwards is still caught.
|
||||
*/
|
||||
hasPendingLocalWork?: () => boolean;
|
||||
/**
|
||||
* Cancel iteration as soon as this signal aborts. Required for caller-driven
|
||||
* cancellation (ESC) when the underlying transport does not surface signal
|
||||
@@ -157,7 +167,7 @@ export async function* iterateWithIdleTimeout<T>(
|
||||
options: IdleTimeoutIteratorOptions,
|
||||
): AsyncGenerator<T> {
|
||||
const firstItemTimeoutMs = options.firstItemTimeoutMs ?? options.idleTimeoutMs;
|
||||
const firstItemDeadlineMs =
|
||||
let firstItemDeadlineMs =
|
||||
firstItemTimeoutMs !== undefined && firstItemTimeoutMs > 0 ? Date.now() + firstItemTimeoutMs : undefined;
|
||||
const abortSignal = options.abortSignal;
|
||||
const iterator = iterable[Symbol.asyncIterator]();
|
||||
@@ -197,6 +207,28 @@ export async function* iterateWithIdleTimeout<T>(
|
||||
};
|
||||
let lastProgressAt = Date.now();
|
||||
|
||||
const hasPendingLocalWork = (): boolean => {
|
||||
if (!options.hasPendingLocalWork) return false;
|
||||
try {
|
||||
return options.hasPendingLocalWork();
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
};
|
||||
// Local work means the current gap is attributable to the consumer side,
|
||||
// not the provider: slide the active deadline a full budget past now
|
||||
// instead of aborting. Once the work completes the watchdog resumes from
|
||||
// the last extension, so a provider that stalls afterwards is still caught.
|
||||
const extendDeadlineForLocalWork = (): void => {
|
||||
if (awaitingFirstItem) {
|
||||
if (firstItemDeadlineMs !== undefined && firstItemTimeoutMs !== undefined) {
|
||||
firstItemDeadlineMs = Date.now() + firstItemTimeoutMs;
|
||||
}
|
||||
} else {
|
||||
lastProgressAt = Date.now();
|
||||
}
|
||||
};
|
||||
|
||||
const noTimeoutEnforced =
|
||||
(firstItemTimeoutMs === undefined || firstItemTimeoutMs <= 0) &&
|
||||
(options.idleTimeoutMs === undefined || options.idleTimeoutMs <= 0);
|
||||
@@ -271,6 +303,12 @@ export async function* iterateWithIdleTimeout<T>(
|
||||
timer = setTimeout(onTimerFire, Math.max(0, deadlineMs - Date.now()));
|
||||
};
|
||||
|
||||
// The in-flight iterator.next() promise, persisted across loop iterations:
|
||||
// a deadline extension for pending local work loops without consuming it,
|
||||
// and issuing a second next() while one is outstanding would drop an item.
|
||||
let pendingNext:
|
||||
| Promise<{ kind: "next"; result: IteratorResult<T> } | { kind: "error"; error: unknown }>
|
||||
| undefined;
|
||||
try {
|
||||
let raceCount = 0;
|
||||
while (true) {
|
||||
@@ -291,21 +329,29 @@ export async function* iterateWithIdleTimeout<T>(
|
||||
if (firstItemDeadlineMs !== undefined) {
|
||||
activeTimeoutMs = firstItemDeadlineMs - Date.now();
|
||||
if (activeTimeoutMs <= 0) {
|
||||
options.onFirstItemTimeout?.();
|
||||
closeIterator();
|
||||
throw new AIError.StreamTimeoutError(options.firstItemErrorMessage ?? options.errorMessage);
|
||||
if (!hasPendingLocalWork()) {
|
||||
options.onFirstItemTimeout?.();
|
||||
closeIterator();
|
||||
throw new AIError.StreamTimeoutError(options.firstItemErrorMessage ?? options.errorMessage);
|
||||
}
|
||||
extendDeadlineForLocalWork();
|
||||
activeTimeoutMs = firstItemDeadlineMs! - Date.now();
|
||||
}
|
||||
}
|
||||
} else if (options.idleTimeoutMs !== undefined && options.idleTimeoutMs > 0) {
|
||||
activeTimeoutMs = options.idleTimeoutMs - (Date.now() - lastProgressAt);
|
||||
if (activeTimeoutMs <= 0) {
|
||||
options.onIdle?.();
|
||||
closeIterator();
|
||||
throw new AIError.StreamTimeoutError(options.errorMessage);
|
||||
if (!hasPendingLocalWork()) {
|
||||
options.onIdle?.();
|
||||
closeIterator();
|
||||
throw new AIError.StreamTimeoutError(options.errorMessage);
|
||||
}
|
||||
extendDeadlineForLocalWork();
|
||||
activeTimeoutMs = options.idleTimeoutMs;
|
||||
}
|
||||
}
|
||||
|
||||
const nextResultPromise = withRacy(iterator.next());
|
||||
pendingNext ??= withRacy(iterator.next());
|
||||
|
||||
const racers: Array<
|
||||
Promise<
|
||||
@@ -314,7 +360,7 @@ export async function* iterateWithIdleTimeout<T>(
|
||||
| { kind: "timeout" }
|
||||
| { kind: "abort" }
|
||||
>
|
||||
> = [nextResultPromise];
|
||||
> = [pendingNext];
|
||||
|
||||
const enforceTimeout = !noTimeoutEnforced && activeTimeoutMs !== undefined && activeTimeoutMs > 0;
|
||||
if (enforceTimeout) {
|
||||
@@ -333,11 +379,21 @@ export async function* iterateWithIdleTimeout<T>(
|
||||
let continuing = false;
|
||||
try {
|
||||
const outcome = await Promise.race(racers);
|
||||
if (outcome.kind === "next" || outcome.kind === "error") {
|
||||
pendingNext = undefined;
|
||||
}
|
||||
if (outcome.kind === "abort") {
|
||||
closeIterator();
|
||||
throw abortReason(abortSignal!);
|
||||
}
|
||||
if (outcome.kind === "timeout") {
|
||||
if (hasPendingLocalWork()) {
|
||||
// A local tool is still running; the provider cannot make
|
||||
// progress until we hand its result back. Keep waiting.
|
||||
extendDeadlineForLocalWork();
|
||||
continuing = true;
|
||||
continue;
|
||||
}
|
||||
if (!awaitingFirstItem) {
|
||||
options.onIdle?.();
|
||||
} else {
|
||||
|
||||
@@ -0,0 +1,233 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { streamAnthropic } from "../src/providers/anthropic";
|
||||
import type { AnthropicMessagesClientLike } from "../src/providers/anthropic-client";
|
||||
import type { Context, Model } from "../src/types";
|
||||
import { waitForDelayOrAbort } from "./helpers";
|
||||
|
||||
const model: Model<"anthropic-messages"> = buildModel({
|
||||
id: "claude-opus-4-8",
|
||||
name: "Claude Opus 4.8",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
baseUrl: "https://api.anthropic.com",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 8_192,
|
||||
});
|
||||
|
||||
const context: Context = {
|
||||
messages: [{ role: "user", content: "write a file", timestamp: Date.now() }],
|
||||
};
|
||||
|
||||
type MockAnthropicEvent = Record<string, unknown>;
|
||||
|
||||
/** `{ waitMs, event }` script step; `waitMs` elapses (fake clock) before the event is yielded. */
|
||||
type ScriptStep = { waitMs: number; event: MockAnthropicEvent | "hang-with-pings" };
|
||||
|
||||
const writeToolCallOpening: MockAnthropicEvent[] = [
|
||||
{
|
||||
type: "message_start",
|
||||
message: {
|
||||
id: "msg_ping_keepalive",
|
||||
usage: { input_tokens: 10, output_tokens: 0, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 },
|
||||
},
|
||||
},
|
||||
{
|
||||
type: "content_block_start",
|
||||
index: 0,
|
||||
content_block: { type: "tool_use", id: "toolu_ping_keepalive", name: "write", input: {} },
|
||||
},
|
||||
{
|
||||
type: "content_block_delta",
|
||||
index: 0,
|
||||
delta: { type: "input_json_delta", partial_json: '{"path":"notes.md",' },
|
||||
},
|
||||
];
|
||||
|
||||
const writeToolCallClosing: MockAnthropicEvent[] = [
|
||||
{
|
||||
type: "content_block_delta",
|
||||
index: 0,
|
||||
delta: { type: "input_json_delta", partial_json: '"content":"hello world"}' },
|
||||
},
|
||||
{ type: "content_block_stop", index: 0 },
|
||||
{
|
||||
type: "message_delta",
|
||||
delta: { stop_reason: "tool_use" },
|
||||
usage: { input_tokens: 10, output_tokens: 6, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 },
|
||||
},
|
||||
{ type: "message_stop" },
|
||||
];
|
||||
|
||||
function createScriptedClient(
|
||||
script: ScriptStep[],
|
||||
counters: { pings: number },
|
||||
onIteratorStart: () => void,
|
||||
): AnthropicMessagesClientLike {
|
||||
const create = ((_body: unknown, requestOptions?: { signal?: AbortSignal }) => {
|
||||
const signal = requestOptions?.signal;
|
||||
const response = new Response(null, { status: 200, headers: { "request-id": "req_ping_keepalive" } });
|
||||
const stream = {
|
||||
async *[Symbol.asyncIterator]() {
|
||||
onIteratorStart();
|
||||
for (const step of script) {
|
||||
if (step.event === "hang-with-pings") {
|
||||
// Wedged upstream: no semantic events ever again, but the edge
|
||||
// keeps the SSE connection alive with keepalive pings.
|
||||
while (true) {
|
||||
await waitForDelayOrAbort(step.waitMs, signal);
|
||||
counters.pings += 1;
|
||||
yield { type: "ping" };
|
||||
}
|
||||
}
|
||||
if (step.waitMs > 0) {
|
||||
await waitForDelayOrAbort(step.waitMs, signal);
|
||||
}
|
||||
if (step.event.type === "ping") counters.pings += 1;
|
||||
yield step.event;
|
||||
}
|
||||
},
|
||||
};
|
||||
return {
|
||||
async withResponse() {
|
||||
return { data: stream, response, request_id: "req_ping_keepalive" };
|
||||
},
|
||||
} as never;
|
||||
}) as unknown as AnthropicMessagesClientLike["messages"]["create"];
|
||||
return { messages: { create } } as AnthropicMessagesClientLike;
|
||||
}
|
||||
|
||||
async function drainMicrotasks(count: number): Promise<void> {
|
||||
for (let i = 0; i < count; i++) {
|
||||
await Promise.resolve();
|
||||
}
|
||||
}
|
||||
|
||||
async function drainMicrotasksUntil(predicate: () => boolean, errorMessage: string): Promise<void> {
|
||||
for (let i = 0; i < 1000; i++) {
|
||||
if (predicate()) return;
|
||||
await Promise.resolve();
|
||||
}
|
||||
throw new Error(errorMessage);
|
||||
}
|
||||
|
||||
afterEach(() => {
|
||||
vi.useRealTimers();
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
describe("anthropic ping keepalive idle cap", () => {
|
||||
it("times out a stalled tool-call stream instead of letting pings extend it forever", async () => {
|
||||
vi.useFakeTimers();
|
||||
const counters = { pings: 0 };
|
||||
let iteratorStarted = false;
|
||||
const script: ScriptStep[] = [
|
||||
...writeToolCallOpening.map(event => ({ waitMs: 0, event })),
|
||||
{ waitMs: 500, event: "hang-with-pings" as const },
|
||||
];
|
||||
const client = createScriptedClient(script, counters, () => {
|
||||
iteratorStarted = true;
|
||||
});
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
let settled = false;
|
||||
const resultPromise = streamAnthropic(model, context, {
|
||||
client,
|
||||
streamFirstEventTimeoutMs: 1_000,
|
||||
streamIdleTimeoutMs: 1_000,
|
||||
providerRetryWait,
|
||||
})
|
||||
.result()
|
||||
.then(message => {
|
||||
settled = true;
|
||||
return message;
|
||||
});
|
||||
|
||||
await drainMicrotasksUntil(() => iteratorStarted, "Anthropic mock stream never started");
|
||||
await drainMicrotasks(30);
|
||||
|
||||
// Pings arrive every 500 fake-ms while generation is wedged. Drive far
|
||||
// past the bounded keepalive window (3x idle = 3_000ms) plus one idle
|
||||
// budget; without the cap the idle deadline is reset by every ping and
|
||||
// this loop ends with the result still pending (issue #4900's hang).
|
||||
let stepsRun = 0;
|
||||
for (let step = 0; step < 40 && !settled; step++) {
|
||||
vi.advanceTimersByTime(500);
|
||||
await drainMicrotasks(30);
|
||||
stepsRun = step + 1;
|
||||
}
|
||||
|
||||
expect(settled).toBe(true);
|
||||
// Cap (3_000ms) + idle budget (1_000ms) = fires at 3_500-4_000 fake ms.
|
||||
expect(stepsRun).toBeLessThanOrEqual(9);
|
||||
// Keepalives within the window were honored before the watchdog fired.
|
||||
expect(counters.pings).toBeGreaterThanOrEqual(5);
|
||||
|
||||
const result = await resultPromise;
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorMessage).toBe("Anthropic stream stalled while waiting for the next event");
|
||||
// Mid-stream idle stalls are terminal for the provider loop (session-level
|
||||
// auto-retry owns recovery); the provider must not silently re-request.
|
||||
expect(providerRetryWait).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("keeps a slow-but-alive stream open across ping-bridged gaps within the cap", async () => {
|
||||
vi.useFakeTimers();
|
||||
const counters = { pings: 0 };
|
||||
let iteratorStarted = false;
|
||||
// Silent generation gap of 1_800ms (> 1_000ms idle budget) bridged by
|
||||
// pings at t=600 and t=1200, then semantic progress resumes and the
|
||||
// tool call completes. Pings within the cap must count as liveness.
|
||||
const script: ScriptStep[] = [
|
||||
...writeToolCallOpening.map(event => ({ waitMs: 0, event })),
|
||||
{ waitMs: 600, event: { type: "ping" } },
|
||||
{ waitMs: 600, event: { type: "ping" } },
|
||||
{ waitMs: 600, event: writeToolCallClosing[0]! },
|
||||
...writeToolCallClosing.slice(1).map(event => ({ waitMs: 0, event })),
|
||||
];
|
||||
const client = createScriptedClient(script, counters, () => {
|
||||
iteratorStarted = true;
|
||||
});
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
let settled = false;
|
||||
const resultPromise = streamAnthropic(model, context, {
|
||||
client,
|
||||
streamFirstEventTimeoutMs: 1_000,
|
||||
streamIdleTimeoutMs: 1_000,
|
||||
providerRetryWait,
|
||||
})
|
||||
.result()
|
||||
.then(message => {
|
||||
settled = true;
|
||||
return message;
|
||||
});
|
||||
|
||||
await drainMicrotasksUntil(() => iteratorStarted, "Anthropic mock stream never started");
|
||||
await drainMicrotasks(30);
|
||||
|
||||
for (let step = 0; step < 30 && !settled; step++) {
|
||||
vi.advanceTimersByTime(200);
|
||||
await drainMicrotasks(30);
|
||||
}
|
||||
|
||||
expect(settled).toBe(true);
|
||||
expect(counters.pings).toBe(2);
|
||||
|
||||
const result = await resultPromise;
|
||||
expect(result.errorMessage).toBeUndefined();
|
||||
expect(result.stopReason).toBe("toolUse");
|
||||
expect(providerRetryWait).not.toHaveBeenCalled();
|
||||
expect(JSON.parse(JSON.stringify(result.content))).toEqual([
|
||||
{
|
||||
type: "toolCall",
|
||||
id: "toolu_ping_keepalive",
|
||||
name: "write",
|
||||
arguments: { path: "notes.md", content: "hello world" },
|
||||
},
|
||||
]);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,59 @@
|
||||
import { afterEach, beforeEach, describe, expect, test } from "bun:test";
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { resolveAuthBrokerConfig } from "@oh-my-pi/pi-ai/auth-broker";
|
||||
import { removeWithRetries } from "../../utils/src/temp";
|
||||
import { withEnv } from "./helpers";
|
||||
|
||||
const SUPPRESS_AUTH_BROKER_ENV = {
|
||||
OMP_AUTH_BROKER_URL: undefined,
|
||||
OMP_AUTH_BROKER_TOKEN: undefined,
|
||||
} as const;
|
||||
|
||||
describe("resolveAuthBrokerConfig config discovery", () => {
|
||||
let agentDir = "";
|
||||
|
||||
beforeEach(async () => {
|
||||
agentDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-auth-broker-config-"));
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
if (agentDir) {
|
||||
await removeWithRetries(agentDir);
|
||||
agentDir = "";
|
||||
}
|
||||
});
|
||||
|
||||
test("resolves broker URL and token from config.yaml when config.yml is absent", async () => {
|
||||
await Bun.write(
|
||||
path.join(agentDir, "config.yaml"),
|
||||
"auth.broker.url: https://yaml-broker.example/v1\nauth.broker.token: yaml-token\n",
|
||||
);
|
||||
|
||||
await withEnv(SUPPRESS_AUTH_BROKER_ENV, async () => {
|
||||
await expect(resolveAuthBrokerConfig({ agentDir })).resolves.toEqual({
|
||||
url: "https://yaml-broker.example/v1",
|
||||
token: "yaml-token",
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
test("prefers config.yml over config.yaml when both exist", async () => {
|
||||
await Bun.write(
|
||||
path.join(agentDir, "config.yaml"),
|
||||
"auth.broker.url: https://yaml-broker.example/v1\nauth.broker.token: yaml-token\n",
|
||||
);
|
||||
await Bun.write(
|
||||
path.join(agentDir, "config.yml"),
|
||||
"auth.broker.url: https://yml-broker.example/v1\nauth.broker.token: yml-token\n",
|
||||
);
|
||||
|
||||
await withEnv(SUPPRESS_AUTH_BROKER_ENV, async () => {
|
||||
await expect(resolveAuthBrokerConfig({ agentDir })).resolves.toEqual({
|
||||
url: "https://yml-broker.example/v1",
|
||||
token: "yml-token",
|
||||
});
|
||||
});
|
||||
});
|
||||
});
|
||||
@@ -724,6 +724,157 @@ describe("AuthStorage codex oauth ranking", () => {
|
||||
expect(apiKey).toBe("api-acct-solo");
|
||||
});
|
||||
|
||||
test.each([
|
||||
["gpt-5.6-sol", "free", "plus"],
|
||||
["gpt-5.6-luna", "go", "business"],
|
||||
["gpt-5.6-sol-pro", "free", "team"],
|
||||
])("%s routes away from a less-used %s account to an eligible %s account", async (modelId, freePlan, paidPlan) => {
|
||||
if (!authStorage) throw new Error("test setup failed");
|
||||
|
||||
await authStorage.set("openai-codex", [
|
||||
{ type: "oauth", ...createCredential("acct-free", "free@example.com") },
|
||||
{ type: "oauth", ...createCredential("acct-paid", "paid@example.com") },
|
||||
]);
|
||||
|
||||
usageByAccount.set(
|
||||
"acct-free",
|
||||
createCodexUsageReport({
|
||||
accountId: "acct-free",
|
||||
primary: { usedFraction: 0.01, resetInMs: 30 * 60 * 1000 },
|
||||
secondary: { usedFraction: 0.01, resetInMs: 6 * 24 * 60 * 60 * 1000 },
|
||||
metadata: { planType: freePlan, email: "free@example.com" },
|
||||
}),
|
||||
);
|
||||
usageByAccount.set(
|
||||
"acct-paid",
|
||||
createCodexUsageReport({
|
||||
accountId: "acct-paid",
|
||||
primary: { usedFraction: 0.8, resetInMs: 30 * 60 * 1000 },
|
||||
secondary: { usedFraction: 0.8, resetInMs: 6 * 24 * 60 * 60 * 1000 },
|
||||
metadata: { planType: paidPlan, email: "paid@example.com" },
|
||||
}),
|
||||
);
|
||||
|
||||
const apiKey = await authStorage.getApiKey("openai-codex", undefined, { modelId });
|
||||
expect(apiKey).toBe("api-acct-paid");
|
||||
});
|
||||
|
||||
test.each([
|
||||
["gpt-5.6-terra", "free", "enterprise"],
|
||||
["gpt-5.6-terra-pro", "go", "pro"],
|
||||
])("%s keeps a less-used %s account in ordinary ranking ahead of %s", async (modelId, lowUsagePlan, highUsagePlan) => {
|
||||
if (!authStorage) throw new Error("test setup failed");
|
||||
|
||||
await authStorage.set("openai-codex", [
|
||||
{ type: "oauth", ...createCredential("acct-low-usage", "low-usage@example.com") },
|
||||
{ type: "oauth", ...createCredential("acct-high-usage", "high-usage@example.com") },
|
||||
]);
|
||||
|
||||
usageByAccount.set(
|
||||
"acct-low-usage",
|
||||
createCodexUsageReport({
|
||||
accountId: "acct-low-usage",
|
||||
primary: { usedFraction: 0.01, resetInMs: 30 * 60 * 1000 },
|
||||
secondary: { usedFraction: 0.01, resetInMs: 6 * 24 * 60 * 60 * 1000 },
|
||||
metadata: { planType: lowUsagePlan, email: "low-usage@example.com" },
|
||||
}),
|
||||
);
|
||||
usageByAccount.set(
|
||||
"acct-high-usage",
|
||||
createCodexUsageReport({
|
||||
accountId: "acct-high-usage",
|
||||
primary: { usedFraction: 0.8, resetInMs: 30 * 60 * 1000 },
|
||||
secondary: { usedFraction: 0.8, resetInMs: 6 * 24 * 60 * 60 * 1000 },
|
||||
metadata: { planType: highUsagePlan, email: "high-usage@example.com" },
|
||||
}),
|
||||
);
|
||||
|
||||
const apiKey = await authStorage.getApiKey("openai-codex", undefined, { modelId });
|
||||
expect(apiKey).toBe("api-acct-low-usage");
|
||||
});
|
||||
|
||||
test("reranks a Terra session on a Go account when it switches to Sol", async () => {
|
||||
if (!authStorage) throw new Error("test setup failed");
|
||||
|
||||
await authStorage.set("openai-codex", [
|
||||
{ type: "oauth", ...createCredential("acct-go", "go@example.com") },
|
||||
{ type: "oauth", ...createCredential("acct-business", "business@example.com") },
|
||||
]);
|
||||
|
||||
usageByAccount.set(
|
||||
"acct-go",
|
||||
createCodexUsageReport({
|
||||
accountId: "acct-go",
|
||||
primary: { usedFraction: 0.01, resetInMs: 30 * 60 * 1000 },
|
||||
secondary: { usedFraction: 0.01, resetInMs: 6 * 24 * 60 * 60 * 1000 },
|
||||
metadata: { planType: "go", email: "go@example.com" },
|
||||
}),
|
||||
);
|
||||
usageByAccount.set(
|
||||
"acct-business",
|
||||
createCodexUsageReport({
|
||||
accountId: "acct-business",
|
||||
primary: { usedFraction: 0.8, resetInMs: 30 * 60 * 1000 },
|
||||
secondary: { usedFraction: 0.8, resetInMs: 6 * 24 * 60 * 60 * 1000 },
|
||||
metadata: { planType: "business", email: "business@example.com" },
|
||||
}),
|
||||
);
|
||||
|
||||
let terraSession: string | undefined;
|
||||
let terraApiKey: string | undefined;
|
||||
for (let index = 0; index < 100; index += 1) {
|
||||
const sessionId = `session-terra-to-sol-${index}`;
|
||||
const apiKey = await authStorage.getApiKey("openai-codex", sessionId, {
|
||||
modelId: "gpt-5.6-terra",
|
||||
});
|
||||
if (apiKey === "api-acct-go") {
|
||||
terraSession = sessionId;
|
||||
terraApiKey = apiKey;
|
||||
break;
|
||||
}
|
||||
}
|
||||
expect(terraApiKey).toBe("api-acct-go");
|
||||
if (!terraSession) throw new Error("expected Terra to select the lower-usage Go account");
|
||||
|
||||
const solApiKey = await authStorage.getApiKey("openai-codex", terraSession, {
|
||||
modelId: "gpt-5.6-sol",
|
||||
});
|
||||
expect(solApiKey).toBe("api-acct-business");
|
||||
});
|
||||
|
||||
test("falls back by ordinary usage ranking for Sol when no account is confirmed paid", async () => {
|
||||
if (!authStorage) throw new Error("test setup failed");
|
||||
|
||||
await authStorage.set("openai-codex", [
|
||||
{ type: "oauth", ...createCredential("acct-free", "free@example.com") },
|
||||
{ type: "oauth", ...createCredential("acct-go", "go@example.com") },
|
||||
]);
|
||||
|
||||
usageByAccount.set(
|
||||
"acct-free",
|
||||
createCodexUsageReport({
|
||||
accountId: "acct-free",
|
||||
primary: { usedFraction: 0.8, resetInMs: 30 * 60 * 1000 },
|
||||
secondary: { usedFraction: 0.8, resetInMs: 6 * 24 * 60 * 60 * 1000 },
|
||||
metadata: { planType: "free", email: "free@example.com" },
|
||||
}),
|
||||
);
|
||||
usageByAccount.set(
|
||||
"acct-go",
|
||||
createCodexUsageReport({
|
||||
accountId: "acct-go",
|
||||
primary: { usedFraction: 0.01, resetInMs: 30 * 60 * 1000 },
|
||||
secondary: { usedFraction: 0.01, resetInMs: 6 * 24 * 60 * 60 * 1000 },
|
||||
metadata: { planType: "go", email: "go@example.com" },
|
||||
}),
|
||||
);
|
||||
|
||||
const apiKey = await authStorage.getApiKey("openai-codex", undefined, {
|
||||
modelId: "gpt-5.6-sol",
|
||||
});
|
||||
expect(apiKey).toBe("api-acct-go");
|
||||
});
|
||||
|
||||
test("prefers Pro accounts for codex spark models over Plus accounts", async () => {
|
||||
if (!authStorage) throw new Error("test setup failed");
|
||||
|
||||
|
||||
@@ -1,14 +1,25 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { create } from "@bufbuild/protobuf";
|
||||
import {
|
||||
type BlockState,
|
||||
buildCursorHistoryForTest,
|
||||
buildCursorSystemPromptJsons,
|
||||
emptyGrepPatternRejection,
|
||||
handleServerMessage,
|
||||
resolveExecHandler,
|
||||
streamCursor,
|
||||
type ToolCallState,
|
||||
} from "@oh-my-pi/pi-ai/providers/cursor";
|
||||
import type { Context, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { streamCursor as lazyStreamCursor, setCursorProviderModule } from "@oh-my-pi/pi-ai/providers/register-builtins";
|
||||
import type { AssistantMessage, Context, CursorExecHandlers, Model, ToolResultMessage } from "@oh-my-pi/pi-ai/types";
|
||||
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import type { AgentRunRequest } from "@oh-my-pi/pi-catalog/discovery/cursor-gen/agent_pb";
|
||||
import {
|
||||
type AgentRunRequest,
|
||||
AgentServerMessageSchema,
|
||||
ExecServerMessageSchema,
|
||||
ReadArgsSchema,
|
||||
} from "@oh-my-pi/pi-catalog/discovery/cursor-gen/agent_pb";
|
||||
|
||||
const cursorModel: Model<"cursor-agent"> = buildModel({
|
||||
id: "cursor-composer-2.5",
|
||||
@@ -361,3 +372,188 @@ describe("Cursor grepArgs empty-pattern guard (issue #4574)", () => {
|
||||
expect(emptyGrepPatternRejection("\t\n", "src/**/*.ts")).toContain('"src/**/*.ts"');
|
||||
});
|
||||
});
|
||||
|
||||
function cursorAssistantMessage(): AssistantMessage {
|
||||
return {
|
||||
role: "assistant",
|
||||
content: [],
|
||||
api: "cursor-agent",
|
||||
provider: "cursor",
|
||||
model: "cursor-composer-2.5",
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "stop",
|
||||
timestamp: 0,
|
||||
};
|
||||
}
|
||||
|
||||
function newBlockState(): BlockState {
|
||||
let textBlock: BlockState["currentTextBlock"] = null;
|
||||
let thinkingBlock: BlockState["currentThinkingBlock"] = null;
|
||||
let toolCall: ToolCallState | null = null;
|
||||
return {
|
||||
get currentTextBlock() {
|
||||
return textBlock;
|
||||
},
|
||||
get currentThinkingBlock() {
|
||||
return thinkingBlock;
|
||||
},
|
||||
get currentToolCall() {
|
||||
return toolCall;
|
||||
},
|
||||
firstTokenTime: undefined,
|
||||
setTextBlock: b => {
|
||||
textBlock = b;
|
||||
},
|
||||
setThinkingBlock: b => {
|
||||
thinkingBlock = b;
|
||||
},
|
||||
setToolCall: t => {
|
||||
toolCall = t;
|
||||
},
|
||||
setFirstTokenTime: () => {},
|
||||
};
|
||||
}
|
||||
|
||||
describe("Cursor exec local-work tracking (issue #4593)", () => {
|
||||
it("marks the stream busy for the duration of a local exec handler", async () => {
|
||||
const output = cursorAssistantMessage();
|
||||
const stream = new AssistantMessageEventStream();
|
||||
const state = newBlockState();
|
||||
const written: unknown[] = [];
|
||||
const h2Request = {
|
||||
write: (chunk: unknown) => {
|
||||
written.push(chunk);
|
||||
return true;
|
||||
},
|
||||
} as unknown as Parameters<typeof handleServerMessage>[5];
|
||||
const handlerGate = Promise.withResolvers<void>();
|
||||
const execHandlers: CursorExecHandlers = {
|
||||
async read(args) {
|
||||
await handlerGate.promise;
|
||||
return {
|
||||
role: "toolResult",
|
||||
toolCallId: args.toolCallId,
|
||||
toolName: "read",
|
||||
content: [{ type: "text", text: "file contents" }],
|
||||
isError: false,
|
||||
timestamp: 1,
|
||||
} satisfies ToolResultMessage;
|
||||
},
|
||||
};
|
||||
const serverMsg = create(AgentServerMessageSchema, {
|
||||
message: {
|
||||
case: "execServerMessage",
|
||||
value: create(ExecServerMessageSchema, {
|
||||
id: 1,
|
||||
execId: "exec-1",
|
||||
message: {
|
||||
case: "readArgs",
|
||||
value: create(ReadArgsSchema, { path: "/tmp/slow-file", toolCallId: "call-read-1" }),
|
||||
},
|
||||
}),
|
||||
},
|
||||
});
|
||||
|
||||
expect(stream.hasPendingLocalWork).toBe(false);
|
||||
const dispatch = handleServerMessage(
|
||||
serverMsg,
|
||||
output,
|
||||
stream,
|
||||
state,
|
||||
new Map(),
|
||||
h2Request,
|
||||
execHandlers,
|
||||
undefined,
|
||||
{ sawTokenDelta: false },
|
||||
[],
|
||||
);
|
||||
|
||||
// The exec round-trip is in flight: the stream must advertise local
|
||||
// work so the lazy idle watchdog defers instead of aborting.
|
||||
expect(stream.hasPendingLocalWork).toBe(true);
|
||||
|
||||
handlerGate.resolve();
|
||||
await dispatch;
|
||||
|
||||
expect(stream.hasPendingLocalWork).toBe(false);
|
||||
// The read result went back out on the exec channel.
|
||||
expect(written.length).toBe(1);
|
||||
});
|
||||
|
||||
it("survives a local exec tool outliving the lazy idle budget end to end", async () => {
|
||||
const workDone = Promise.withResolvers<void>();
|
||||
// The tracked work completes only once the lazy watchdog has consulted
|
||||
// the stream's local-work state at two expired deadlines, proving the
|
||||
// idle budget was truly exceeded while the exec tool ran.
|
||||
class ProbedStream extends AssistantMessageEventStream {
|
||||
probeCalls = 0;
|
||||
override get hasPendingLocalWork(): boolean {
|
||||
this.probeCalls++;
|
||||
if (this.probeCalls >= 2) workDone.resolve();
|
||||
return super.hasPendingLocalWork;
|
||||
}
|
||||
}
|
||||
const source = new ProbedStream();
|
||||
let providerSignal: AbortSignal | undefined;
|
||||
setCursorProviderModule({
|
||||
streamCursor: (_model, _context, options) => {
|
||||
providerSignal = options.signal;
|
||||
void (async () => {
|
||||
const partial = cursorAssistantMessage();
|
||||
source.push({ type: "start", partial });
|
||||
source.push({ type: "text_delta", contentIndex: 0, delta: "spawning local tool", partial });
|
||||
await source.trackLocalWork(workDone.promise);
|
||||
const message = cursorAssistantMessage();
|
||||
source.push({ type: "done", reason: "stop", message });
|
||||
})();
|
||||
return source;
|
||||
},
|
||||
});
|
||||
|
||||
const stream = lazyStreamCursor(cursorModel, { messages: [] }, { apiKey: "test", streamIdleTimeoutMs: 5 });
|
||||
const result = await stream.result();
|
||||
|
||||
expect(providerSignal?.aborted).toBe(false);
|
||||
expect(source.probeCalls).toBeGreaterThanOrEqual(2);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
});
|
||||
|
||||
it("still aborts a silent cursor stream with no local work in flight", async () => {
|
||||
const partial = cursorAssistantMessage();
|
||||
let providerSignal: AbortSignal | undefined;
|
||||
const source = {
|
||||
async *[Symbol.asyncIterator]() {
|
||||
yield { type: "start", partial } as const;
|
||||
yield { type: "text_delta", contentIndex: 0, delta: "hello", partial } as const;
|
||||
const stalled = Promise.withResolvers<never>();
|
||||
if (providerSignal?.aborted) {
|
||||
stalled.reject(new Error("Request was aborted"));
|
||||
}
|
||||
providerSignal?.addEventListener("abort", () => stalled.reject(new Error("Request was aborted")), {
|
||||
once: true,
|
||||
});
|
||||
await stalled.promise;
|
||||
},
|
||||
} as unknown as AssistantMessageEventStream;
|
||||
setCursorProviderModule({
|
||||
streamCursor: (_model, _context, options) => {
|
||||
providerSignal = options.signal;
|
||||
return source;
|
||||
},
|
||||
});
|
||||
|
||||
const stream = lazyStreamCursor(cursorModel, { messages: [] }, { apiKey: "test", streamIdleTimeoutMs: 10 });
|
||||
const result = await stream.result();
|
||||
|
||||
expect(providerSignal?.aborted).toBe(true);
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorMessage).toBe("Provider stream stalled while waiting for the next event");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -32,6 +32,11 @@ describe("AIError.classify — structural provider errors", () => {
|
||||
).toBe(true);
|
||||
});
|
||||
|
||||
it("classifies a typed AWS credential-resolution failure as authFailed", () => {
|
||||
const id = AIError.classify(new AIError.AwsCredentialsError("opaque provider setup failure", "resolution"));
|
||||
expect(AIError.is(id, AIError.Flag.AuthFailed)).toBe(true);
|
||||
});
|
||||
|
||||
it("maps the usage_limit_reached code to usageLimit on a 429", () => {
|
||||
const id = AIError.classify(
|
||||
new AIError.ProviderHttpError("Payment Required", 429, { code: "usage_limit_reached" }),
|
||||
|
||||
@@ -2,6 +2,7 @@ import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import type { Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import type { ModelSpec } from "@oh-my-pi/pi-catalog/types";
|
||||
import { isEnoent } from "@oh-my-pi/pi-utils";
|
||||
|
||||
export async function withEnv(
|
||||
@@ -54,7 +55,10 @@ export async function waitForDelayOrAbort(delayMs: number, signal: AbortSignal |
|
||||
}
|
||||
}
|
||||
|
||||
export function createCodexModel(id: string): Model<"openai-codex-responses"> {
|
||||
export function createCodexModel(
|
||||
id: string,
|
||||
spec?: Partial<ModelSpec<"openai-codex-responses">>,
|
||||
): Model<"openai-codex-responses"> {
|
||||
return buildModel({
|
||||
id,
|
||||
name: id,
|
||||
@@ -66,6 +70,7 @@ export function createCodexModel(id: string): Model<"openai-codex-responses"> {
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 272000,
|
||||
maxTokens: 128000,
|
||||
...spec,
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -1,12 +1,23 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
|
||||
import { streamAzureOpenAIResponses } from "@oh-my-pi/pi-ai/providers/azure-openai-responses";
|
||||
import { streamOpenAICodexResponses } from "@oh-my-pi/pi-ai/providers/openai-codex-responses";
|
||||
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
|
||||
import type { Context, Model, Tool, ToolChoice } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import * as piUtils from "@oh-my-pi/pi-utils";
|
||||
import { z } from "zod/v4";
|
||||
|
||||
const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001";
|
||||
|
||||
beforeEach(() => {
|
||||
vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
const completionsModel: Model<"openai-completions"> = buildModel({
|
||||
id: "gpt-4o-mini-test",
|
||||
name: "GPT-4o Mini Test",
|
||||
|
||||
@@ -0,0 +1,190 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { setBedrockProviderModule, streamBedrock } from "@oh-my-pi/pi-ai/providers/register-builtins";
|
||||
import type { AssistantMessage, Context, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
|
||||
import { iterateWithIdleTimeout } from "@oh-my-pi/pi-ai/utils/idle-iterator";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
|
||||
// Issue #4593: the generic lazy stream watchdog treats "no AssistantMessageEvent"
|
||||
// as "provider stalled". During a Cursor exec-channel round-trip the server is
|
||||
// waiting on OUR local tool result and legitimately sends nothing, so a local
|
||||
// tool outliving the idle budget aborted a healthy stream with "Provider stream
|
||||
// stalled while waiting for the next event". Provider streams now advertise
|
||||
// pending local work and the watchdog slides its deadline instead of aborting.
|
||||
//
|
||||
// These tests exercise the real watchdog timer against the platform clock (that
|
||||
// timer IS the unit under test), but never guess durations: the simulated local
|
||||
// work completes only once the watchdog has demonstrably reached an expired
|
||||
// deadline and consulted the local-work probe, so the tests stay causal on a
|
||||
// loaded machine. Budgets are a few milliseconds.
|
||||
|
||||
function createModel(): Model<"bedrock-converse-stream"> {
|
||||
return buildModel({
|
||||
id: "mock-bedrock",
|
||||
name: "Mock Bedrock",
|
||||
api: "bedrock-converse-stream",
|
||||
provider: "amazon-bedrock",
|
||||
baseUrl: "https://example.invalid",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 8192,
|
||||
maxTokens: 2048,
|
||||
});
|
||||
}
|
||||
|
||||
function createAssistantMessage(): AssistantMessage {
|
||||
return {
|
||||
role: "assistant",
|
||||
content: [{ type: "text", text: "ok" }],
|
||||
api: "bedrock-converse-stream",
|
||||
provider: "amazon-bedrock",
|
||||
model: "mock-bedrock",
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "stop",
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
}
|
||||
|
||||
const baseContext: Context = { messages: [] };
|
||||
|
||||
describe("idle watchdog local-work deferral (issue #4593)", () => {
|
||||
it("slides the idle deadline while consumer-side local work is pending", async () => {
|
||||
const workDone = Promise.withResolvers<void>();
|
||||
let probeCalls = 0;
|
||||
let busy = true;
|
||||
async function* source() {
|
||||
yield "first";
|
||||
// The "local tool": finishes only after the watchdog has hit an
|
||||
// expired deadline twice and deferred both times.
|
||||
await workDone.promise;
|
||||
busy = false;
|
||||
yield "second";
|
||||
}
|
||||
let idleFired = false;
|
||||
const items: string[] = [];
|
||||
for await (const item of iterateWithIdleTimeout(source(), {
|
||||
idleTimeoutMs: 5,
|
||||
errorMessage: "stalled",
|
||||
onIdle: () => {
|
||||
idleFired = true;
|
||||
},
|
||||
hasPendingLocalWork: () => {
|
||||
probeCalls++;
|
||||
if (probeCalls >= 2) workDone.resolve();
|
||||
return busy;
|
||||
},
|
||||
})) {
|
||||
items.push(item);
|
||||
}
|
||||
expect(items).toEqual(["first", "second"]);
|
||||
expect(probeCalls).toBeGreaterThanOrEqual(2);
|
||||
expect(idleFired).toBe(false);
|
||||
});
|
||||
|
||||
it("still aborts a silent stream once local work has finished", async () => {
|
||||
const workDone = Promise.withResolvers<void>();
|
||||
let busy = true;
|
||||
async function* source() {
|
||||
yield "first";
|
||||
await workDone.promise;
|
||||
busy = false;
|
||||
// The provider genuinely stalls after the local work completed.
|
||||
await new Promise<never>(() => {});
|
||||
yield "never";
|
||||
}
|
||||
const items: string[] = [];
|
||||
let error: Error | undefined;
|
||||
try {
|
||||
for await (const item of iterateWithIdleTimeout(source(), {
|
||||
idleTimeoutMs: 5,
|
||||
errorMessage: "stalled",
|
||||
hasPendingLocalWork: () => {
|
||||
workDone.resolve();
|
||||
return busy;
|
||||
},
|
||||
})) {
|
||||
items.push(item);
|
||||
}
|
||||
} catch (err) {
|
||||
error = err as Error;
|
||||
}
|
||||
expect(items).toEqual(["first"]);
|
||||
expect(error?.message).toBe("stalled");
|
||||
});
|
||||
|
||||
it("slides the first-event deadline while local work is pending", async () => {
|
||||
const workDone = Promise.withResolvers<void>();
|
||||
let probeCalls = 0;
|
||||
let busy = true;
|
||||
async function* source() {
|
||||
// Local bridge work before the model has produced any event.
|
||||
await workDone.promise;
|
||||
yield "first";
|
||||
}
|
||||
const items: string[] = [];
|
||||
for await (const item of iterateWithIdleTimeout(source(), {
|
||||
idleTimeoutMs: 5,
|
||||
firstItemTimeoutMs: 5,
|
||||
errorMessage: "stalled",
|
||||
firstItemErrorMessage: "first event timed out",
|
||||
hasPendingLocalWork: () => {
|
||||
probeCalls++;
|
||||
if (probeCalls >= 2) workDone.resolve();
|
||||
return busy;
|
||||
},
|
||||
})) {
|
||||
items.push(item);
|
||||
busy = false;
|
||||
}
|
||||
expect(items).toEqual(["first"]);
|
||||
expect(probeCalls).toBeGreaterThanOrEqual(2);
|
||||
});
|
||||
|
||||
it("does not abort a lazy provider stream while tracked local work outlives the idle budget", async () => {
|
||||
const workDone = Promise.withResolvers<void>();
|
||||
// Counts how often the lazy wrapper's watchdog consults the stream's
|
||||
// local-work state at an expired deadline; the tracked work completes
|
||||
// only after two deferrals, proving the budget was truly exceeded.
|
||||
class ProbedStream extends AssistantMessageEventStream {
|
||||
probeCalls = 0;
|
||||
override get hasPendingLocalWork(): boolean {
|
||||
this.probeCalls++;
|
||||
if (this.probeCalls >= 2) workDone.resolve();
|
||||
return super.hasPendingLocalWork;
|
||||
}
|
||||
}
|
||||
const source = new ProbedStream();
|
||||
let providerSignal: AbortSignal | undefined;
|
||||
setBedrockProviderModule({
|
||||
streamBedrock: (_model, _context, options) => {
|
||||
providerSignal = options.signal;
|
||||
void (async () => {
|
||||
const partial = createAssistantMessage();
|
||||
source.push({ type: "start", partial });
|
||||
source.push({ type: "text_delta", contentIndex: 0, delta: "running a local tool", partial });
|
||||
// Server-driven local tool run: no events flow while the
|
||||
// tracked work is pending.
|
||||
await source.trackLocalWork(workDone.promise);
|
||||
source.push({ type: "done", reason: "stop", message: createAssistantMessage() });
|
||||
})();
|
||||
return source;
|
||||
},
|
||||
});
|
||||
|
||||
const stream = streamBedrock(createModel(), baseContext, { streamIdleTimeoutMs: 5 });
|
||||
const result = await stream.result();
|
||||
|
||||
expect(providerSignal?.aborted).toBe(false);
|
||||
expect(source.probeCalls).toBeGreaterThanOrEqual(2);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.errorMessage).toBeUndefined();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,74 @@
|
||||
import { describe, expect, test, vi } from "bun:test";
|
||||
import { loginNovita } from "../src/registry/novita";
|
||||
import { getOAuthProviders } from "../src/registry/oauth";
|
||||
import type { FetchImpl } from "../src/types";
|
||||
|
||||
describe("Novita login", () => {
|
||||
test("registers Novita as an available API-key provider", () => {
|
||||
const provider = getOAuthProviders().find(item => item.id === "novita");
|
||||
expect(provider).toMatchObject({ id: "novita", name: "Novita", available: true });
|
||||
});
|
||||
|
||||
test("validates the pasted key against the authenticated balance endpoint", async () => {
|
||||
const authEvents: Array<{ url: string; instructions?: string }> = [];
|
||||
const prompts: Array<{ message: string; placeholder?: string }> = [];
|
||||
const progress: string[] = [];
|
||||
const requests: Array<{
|
||||
url: string;
|
||||
method: string | undefined;
|
||||
authorization: string | null;
|
||||
contentType: string | null;
|
||||
}> = [];
|
||||
const fetchMock: FetchImpl = vi.fn(async (input: string | URL | Request, init?: RequestInit) => {
|
||||
const headers = new Headers(init?.headers);
|
||||
requests.push({
|
||||
url: String(input),
|
||||
method: init?.method,
|
||||
authorization: headers.get("authorization"),
|
||||
contentType: headers.get("content-type"),
|
||||
});
|
||||
return Response.json({ availableBalance: "0" });
|
||||
});
|
||||
|
||||
const apiKey = await loginNovita({
|
||||
onAuth: info => authEvents.push(info),
|
||||
onPrompt: async prompt => {
|
||||
prompts.push(prompt);
|
||||
return " novita-test-key ";
|
||||
},
|
||||
onProgress: message => progress.push(message),
|
||||
fetch: fetchMock,
|
||||
});
|
||||
|
||||
expect(apiKey).toBe("novita-test-key");
|
||||
expect(authEvents).toEqual([
|
||||
{
|
||||
url: "https://novita.ai/settings/key-management",
|
||||
instructions: "Create or copy your API key from the Novita dashboard",
|
||||
},
|
||||
]);
|
||||
expect(prompts).toEqual([{ message: "Paste your Novita API key", placeholder: "sk_..." }]);
|
||||
expect(progress).toEqual(["Validating API key..."]);
|
||||
expect(requests).toEqual([
|
||||
{
|
||||
url: "https://api.novita.ai/openapi/v1/billing/balance/detail",
|
||||
method: "GET",
|
||||
authorization: "Bearer novita-test-key",
|
||||
contentType: "application/json",
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
test("rejects a key rejected by Novita", async () => {
|
||||
const fetchMock: FetchImpl = vi.fn(async () =>
|
||||
Response.json({ code: 401, reason: "UNAUTHORIZED", message: "key not found", metadata: {} }, { status: 401 }),
|
||||
);
|
||||
|
||||
await expect(
|
||||
loginNovita({
|
||||
onPrompt: async () => "invalid-novita-key",
|
||||
fetch: fetchMock,
|
||||
}),
|
||||
).rejects.toThrow("Novita API key validation failed (401)");
|
||||
});
|
||||
});
|
||||
@@ -1,4 +1,4 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
|
||||
import {
|
||||
type InputItem,
|
||||
type RequestBody,
|
||||
@@ -7,12 +7,25 @@ import {
|
||||
import {
|
||||
buildTransformedCodexRequestBody,
|
||||
convertCodexResponsesMessages,
|
||||
resetOpenAICodexHistoryAfterCompaction,
|
||||
streamOpenAICodexResponses,
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-codex-responses";
|
||||
import type { Context, FetchImpl } from "@oh-my-pi/pi-ai/types";
|
||||
import { isOpenAIResponsesProgressEvent } from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
import type { CodexCompactionRequestContext, Context, FetchImpl, ProviderSessionState } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import * as piUtils from "@oh-my-pi/pi-utils";
|
||||
import { createCodexModel } from "./helpers";
|
||||
|
||||
const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001";
|
||||
|
||||
beforeEach(() => {
|
||||
vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
function createCodexTestToken(accountId = "acc_test"): string {
|
||||
const payload = Buffer.from(
|
||||
JSON.stringify({ "https://api.openai.com/auth": { chatgpt_account_id: accountId } }),
|
||||
@@ -63,6 +76,24 @@ interface CapturedCodexRequest {
|
||||
body: Record<string, unknown>;
|
||||
}
|
||||
|
||||
function isRecord(value: unknown): value is Record<string, unknown> {
|
||||
return typeof value === "object" && value !== null && !Array.isArray(value);
|
||||
}
|
||||
|
||||
function requireRecord(value: unknown, label: string): Record<string, unknown> {
|
||||
if (!isRecord(value)) {
|
||||
throw new Error(`expected ${label} to be an object`);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
function parseTurnMetadata(clientMetadata: Record<string, unknown>): Record<string, unknown> {
|
||||
const encoded = clientMetadata["x-codex-turn-metadata"];
|
||||
if (typeof encoded !== "string") throw new Error("expected x-codex-turn-metadata");
|
||||
const decoded: unknown = JSON.parse(encoded);
|
||||
return requireRecord(decoded, "x-codex-turn-metadata");
|
||||
}
|
||||
|
||||
function createCodexFetchMock(sse: string, onRequest: (captured: CapturedCodexRequest) => void): FetchImpl {
|
||||
return (async (input: string | URL, init?: RequestInit) => {
|
||||
const url = typeof input === "string" ? input : input.toString();
|
||||
@@ -191,7 +222,7 @@ describe("openai-codex reasoning.summary", () => {
|
||||
});
|
||||
|
||||
describe("openai-codex Responses Lite input shaping", () => {
|
||||
it("keeps full Responses image details when a requested lite body contains images", async () => {
|
||||
it("strips image detail and keeps lite when the input contains images", async () => {
|
||||
const model = createCodexModel("gpt-5.1-codex");
|
||||
const makeInput = (): InputItem[] => [
|
||||
{
|
||||
@@ -211,10 +242,11 @@ describe("openai-codex Responses Lite input shaping", () => {
|
||||
];
|
||||
|
||||
const lite = await transformRequestBody({ model: model.id, input: makeInput() }, model, { responsesLite: true });
|
||||
const liteMessage = lite.input?.[0]?.content as Array<Record<string, unknown>>;
|
||||
const liteOutput = lite.input?.[2]?.output as Array<Record<string, unknown>>;
|
||||
expect(liteMessage[1]).toEqual({ type: "input_image", detail: "auto", image_url: "data:image/png;base64,AAAA" });
|
||||
expect(liteOutput[0]).toEqual({ type: "input_image", detail: "high", image_url: "data:image/png;base64,BBBB" });
|
||||
expect(lite.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools: [] });
|
||||
const liteMessage = lite.input?.[1]?.content as Array<Record<string, unknown>>;
|
||||
const liteOutput = lite.input?.[3]?.output as Array<Record<string, unknown>>;
|
||||
expect(liteMessage[1]).toEqual({ type: "input_image", image_url: "data:image/png;base64,AAAA" });
|
||||
expect(liteOutput[0]).toEqual({ type: "input_image", image_url: "data:image/png;base64,BBBB" });
|
||||
|
||||
const plain = await transformRequestBody({ model: model.id, input: makeInput() }, model, {});
|
||||
const plainMessage = plain.input?.[0]?.content as Array<Record<string, unknown>>;
|
||||
@@ -253,7 +285,7 @@ describe("openai-codex Responses Lite input shaping", () => {
|
||||
});
|
||||
});
|
||||
|
||||
it("forces parallel_tool_calls off under lite when tools are present", async () => {
|
||||
it("forces parallel_tool_calls off and moves tools into input under lite", async () => {
|
||||
const model = createCodexModel("gpt-5.1-codex");
|
||||
const tools = [{ type: "function", name: "shot", parameters: { type: "object" } }];
|
||||
|
||||
@@ -261,12 +293,57 @@ describe("openai-codex Responses Lite input shaping", () => {
|
||||
responsesLite: true,
|
||||
});
|
||||
expect(lite.parallel_tool_calls).toBe(false);
|
||||
expect(lite.tools).toBeUndefined();
|
||||
expect(lite.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools });
|
||||
|
||||
const plain = await transformRequestBody({ model: model.id, tools, parallel_tool_calls: true }, model, {});
|
||||
expect(plain.parallel_tool_calls).toBe(true);
|
||||
expect(plain.tools).toEqual(tools);
|
||||
|
||||
const noTools = await transformRequestBody({ model: model.id }, model, { responsesLite: true });
|
||||
expect(noTools.parallel_tool_calls).toBeUndefined();
|
||||
expect(noTools.parallel_tool_calls).toBe(false);
|
||||
});
|
||||
|
||||
it("moves instructions and tools into input items under lite", async () => {
|
||||
const model = createCodexModel("gpt-5.6-terra");
|
||||
const tools = [{ type: "function", name: "shot", parameters: { type: "object" } }];
|
||||
const body = await transformRequestBody(
|
||||
{
|
||||
model: model.id,
|
||||
instructions: "test instructions",
|
||||
tools,
|
||||
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "hello" }] }],
|
||||
},
|
||||
model,
|
||||
{ responsesLite: true },
|
||||
);
|
||||
|
||||
expect(body.instructions).toBeUndefined();
|
||||
expect(body.tools).toBeUndefined();
|
||||
expect(body.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools });
|
||||
expect(body.input?.[1]).toEqual({
|
||||
type: "message",
|
||||
role: "developer",
|
||||
content: [{ type: "input_text", text: "test instructions" }],
|
||||
});
|
||||
expect(body.input?.[2]).toEqual({
|
||||
type: "message",
|
||||
role: "user",
|
||||
content: [{ type: "input_text", text: "hello" }],
|
||||
});
|
||||
});
|
||||
|
||||
it("defaults lite from the model useResponsesLite flag and honors explicit opt-out", async () => {
|
||||
const model = createCodexModel("gpt-5.6-terra", { useResponsesLite: true });
|
||||
const lite = await transformRequestBody({ model: model.id, instructions: "sys" }, model, {});
|
||||
expect(lite.instructions).toBeUndefined();
|
||||
expect(lite.input?.[0]?.type).toBe("additional_tools");
|
||||
|
||||
const optOut = await transformRequestBody({ model: model.id, instructions: "sys" }, model, {
|
||||
responsesLite: false,
|
||||
});
|
||||
expect(optOut.instructions).toBe("sys");
|
||||
expect(optOut.input?.some(item => item.type === "additional_tools")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -323,15 +400,21 @@ describe("openai-codex fresh execution input shaping", () => {
|
||||
});
|
||||
|
||||
describe("openai-codex Responses Lite and client metadata wire format", () => {
|
||||
it("sends the lite header and client_metadata body field over SSE", async () => {
|
||||
it("sends canonical Codex metadata and protects reserved fields over SSE", async () => {
|
||||
const model = createCodexModel("gpt-5.1-codex");
|
||||
const clientMetadata = { "x-codex-turn-metadata": '{"thread_id":"thread_1","turn_id":"turn_1"}' };
|
||||
const context = createCodexTestContext();
|
||||
const clientMetadata = {
|
||||
workspace_kind: "repo",
|
||||
workspace_path: "東京/🚀",
|
||||
session_id: "caller-session",
|
||||
"x-codex-turn-metadata": '{"turn_id":"caller-turn"}',
|
||||
};
|
||||
let captured: CapturedCodexRequest | undefined;
|
||||
const fetchMock = createCodexFetchMock(createCodexSse(COMPLETED_CODEX_EVENTS), request => {
|
||||
captured = request;
|
||||
});
|
||||
|
||||
const result = await streamOpenAICodexResponses(model, createCodexTestContext(), {
|
||||
const result = await streamOpenAICodexResponses(model, context, {
|
||||
apiKey: createCodexTestToken(),
|
||||
fetch: fetchMock,
|
||||
responsesLite: true,
|
||||
@@ -339,10 +422,143 @@ describe("openai-codex Responses Lite and client metadata wire format", () => {
|
||||
}).result();
|
||||
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true");
|
||||
expect(captured?.body.client_metadata).toEqual(clientMetadata);
|
||||
if (!captured) throw new Error("expected a captured Codex request");
|
||||
expect(captured.headers.get("x-openai-internal-codex-responses-lite")).toBe("true");
|
||||
expect(captured.headers.get("x-codex-installation-id")).toBeNull();
|
||||
|
||||
const metadata = requireRecord(captured.body.client_metadata, "client_metadata");
|
||||
const turnMetadata = parseTurnMetadata(metadata);
|
||||
expect(metadata.workspace_kind).toBeUndefined();
|
||||
expect(metadata.workspace_path).toBeUndefined();
|
||||
expect(metadata.session_id).not.toBe("caller-session");
|
||||
expect(turnMetadata.request_kind).toBe("turn");
|
||||
expect(turnMetadata.turn_started_at_unix_ms).toBe(context.messages[0]?.timestamp);
|
||||
expect(turnMetadata.workspace_kind).toBe("repo");
|
||||
expect(turnMetadata.workspace_path).toBe("東京/🚀");
|
||||
expect(metadata["x-codex-installation-id"]).toMatch(
|
||||
/^[0-9a-f]{8}-[0-9a-f]{4}-4[0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i,
|
||||
);
|
||||
expect(metadata.session_id).toBe(turnMetadata.session_id);
|
||||
expect(metadata.thread_id).toBe(turnMetadata.thread_id);
|
||||
expect(metadata.turn_id).toBe(turnMetadata.turn_id);
|
||||
expect(metadata["x-codex-window-id"]).toBe(turnMetadata.window_id);
|
||||
expect(metadata.session_id).toBe(captured.headers.get("session-id"));
|
||||
expect(metadata.thread_id).toBe(captured.headers.get("thread-id"));
|
||||
expect(metadata["x-codex-window-id"]).toBe(captured.headers.get("x-codex-window-id"));
|
||||
expect(metadata["x-codex-turn-metadata"]).toBe(captured.headers.get("x-codex-turn-metadata"));
|
||||
const turnMetadataHeader = captured.headers.get("x-codex-turn-metadata");
|
||||
expect(turnMetadataHeader).toMatch(/^[\x20-\x7e]+$/);
|
||||
const reparsedTurnMetadata: unknown = turnMetadataHeader ? JSON.parse(turnMetadataHeader) : undefined;
|
||||
expect(requireRecord(reparsedTurnMetadata, "round-tripped turn metadata").workspace_path).toBe("東京/🚀");
|
||||
});
|
||||
it("falls back to full Responses when a lite request contains images", async () => {
|
||||
|
||||
it("keeps the installation identity stable across provider sessions", async () => {
|
||||
const model = createCodexModel("gpt-5.1-codex");
|
||||
const captured: CapturedCodexRequest[] = [];
|
||||
const fetchMock = createCodexFetchMock(createCodexSse(COMPLETED_CODEX_EVENTS), request => {
|
||||
captured.push(request);
|
||||
});
|
||||
|
||||
await streamOpenAICodexResponses(model, createCodexTestContext(), {
|
||||
apiKey: createCodexTestToken(),
|
||||
fetch: fetchMock,
|
||||
sessionId: "metadata-session-one",
|
||||
}).result();
|
||||
await streamOpenAICodexResponses(model, createCodexTestContext(), {
|
||||
apiKey: createCodexTestToken(),
|
||||
fetch: fetchMock,
|
||||
sessionId: "metadata-session-two",
|
||||
}).result();
|
||||
|
||||
const firstMetadata = requireRecord(captured[0]?.body.client_metadata, "first client_metadata");
|
||||
const secondMetadata = requireRecord(captured[1]?.body.client_metadata, "second client_metadata");
|
||||
expect(firstMetadata["x-codex-installation-id"]).toBe(secondMetadata["x-codex-installation-id"]);
|
||||
expect(firstMetadata.session_id).toBe("metadata-session-one");
|
||||
expect(secondMetadata.session_id).toBe("metadata-session-two");
|
||||
expect(firstMetadata.thread_id).not.toBe(secondMetadata.thread_id);
|
||||
});
|
||||
|
||||
it("rotates compaction turns by phase and reuses one operation across fan-out calls", async () => {
|
||||
const model = createCodexModel("gpt-5.1-codex");
|
||||
const providerSessionState = new Map<string, ProviderSessionState>();
|
||||
const captured: CapturedCodexRequest[] = [];
|
||||
const fetchMock = createCodexFetchMock(createCodexSse(COMPLETED_CODEX_EVENTS), request => {
|
||||
captured.push(request);
|
||||
});
|
||||
const send = async (codexCompaction?: CodexCompactionRequestContext): Promise<void> => {
|
||||
await streamOpenAICodexResponses(model, createCodexTestContext(), {
|
||||
apiKey: createCodexTestToken(),
|
||||
fetch: fetchMock,
|
||||
sessionId: "compaction-lifecycle-session",
|
||||
providerSessionState,
|
||||
codexCompaction,
|
||||
}).result();
|
||||
};
|
||||
const preTurn: CodexCompactionRequestContext = {
|
||||
operationId: "pre-turn-operation",
|
||||
trigger: "auto",
|
||||
reason: "context_limit",
|
||||
implementation: "responses",
|
||||
phase: "pre_turn",
|
||||
strategy: "memento",
|
||||
};
|
||||
const midTurn: CodexCompactionRequestContext = {
|
||||
...preTurn,
|
||||
operationId: "mid-turn-operation",
|
||||
phase: "mid_turn",
|
||||
};
|
||||
const standalone: CodexCompactionRequestContext = {
|
||||
...preTurn,
|
||||
operationId: "standalone-operation",
|
||||
trigger: "manual",
|
||||
reason: "user_requested",
|
||||
phase: "standalone_turn",
|
||||
};
|
||||
|
||||
await send();
|
||||
await send(preTurn);
|
||||
await send(preTurn);
|
||||
resetOpenAICodexHistoryAfterCompaction({
|
||||
providerSessionState,
|
||||
sessionId: "compaction-lifecycle-session",
|
||||
compaction: preTurn,
|
||||
});
|
||||
await send();
|
||||
await send(midTurn);
|
||||
await send(standalone);
|
||||
|
||||
const turns = captured.map((request, index) =>
|
||||
parseTurnMetadata(requireRecord(request.body.client_metadata, `client_metadata ${index}`)),
|
||||
);
|
||||
expect(turns[0]?.request_kind).toBe("turn");
|
||||
expect(turns[1]?.turn_id).not.toBe(turns[0]?.turn_id);
|
||||
expect(turns[2]?.turn_id).toBe(turns[1]?.turn_id);
|
||||
expect(turns[2]?.turn_started_at_unix_ms).toBe(turns[1]?.turn_started_at_unix_ms);
|
||||
expect(turns[3]?.request_kind).toBe("turn");
|
||||
expect(turns[3]?.turn_id).toBe(turns[1]?.turn_id);
|
||||
expect(turns[3]?.window_id).not.toBe(turns[2]?.window_id);
|
||||
expect(turns[4]?.turn_id).toBe(turns[1]?.turn_id);
|
||||
expect(turns[5]?.turn_id).not.toBe(turns[4]?.turn_id);
|
||||
expect(turns[1]?.thread_id).toBe(turns[5]?.thread_id);
|
||||
expect(turns[1]?.compaction).toEqual({
|
||||
trigger: "auto",
|
||||
reason: "context_limit",
|
||||
implementation: "responses",
|
||||
phase: "pre_turn",
|
||||
strategy: "memento",
|
||||
});
|
||||
const nestedCompaction = requireRecord(turns[1]?.compaction, "nested compaction metadata");
|
||||
expect(nestedCompaction.operationId).toBeUndefined();
|
||||
expect(nestedCompaction.operation_id).toBeUndefined();
|
||||
expect(turns[5]?.compaction).toEqual({
|
||||
trigger: "manual",
|
||||
reason: "user_requested",
|
||||
implementation: "responses",
|
||||
phase: "standalone_turn",
|
||||
strategy: "memento",
|
||||
});
|
||||
});
|
||||
it("keeps lite and strips image detail when a lite request contains images", async () => {
|
||||
const model = buildModel({
|
||||
id: "gpt-5.5",
|
||||
name: "GPT-5.5",
|
||||
@@ -382,19 +598,39 @@ describe("openai-codex Responses Lite and client metadata wire format", () => {
|
||||
).result();
|
||||
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBeNull();
|
||||
expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true");
|
||||
expect(captured?.body.input).toEqual([
|
||||
{ type: "additional_tools", role: "developer", tools: [] },
|
||||
{
|
||||
role: "user",
|
||||
content: [
|
||||
{ type: "input_text", text: "read this image" },
|
||||
{ type: "input_image", detail: "auto", image_url: "data:image/png;base64,AAAA" },
|
||||
{ type: "input_image", image_url: "data:image/png;base64,AAAA" },
|
||||
],
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
it("omits the lite header and client_metadata when not requested", async () => {
|
||||
it("sends the lite header when the model defaults to Responses Lite", async () => {
|
||||
const model = createCodexModel("gpt-5.6-terra", { useResponsesLite: true });
|
||||
let captured: CapturedCodexRequest | undefined;
|
||||
const fetchMock = createCodexFetchMock(createCodexSse(COMPLETED_CODEX_EVENTS), request => {
|
||||
captured = request;
|
||||
});
|
||||
|
||||
const result = await streamOpenAICodexResponses(model, createCodexTestContext(), {
|
||||
apiKey: createCodexTestToken(),
|
||||
fetch: fetchMock,
|
||||
}).result();
|
||||
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true");
|
||||
expect(captured?.body.instructions).toBeUndefined();
|
||||
expect(captured?.body.tools).toBeUndefined();
|
||||
expect((captured?.body.input as Array<Record<string, unknown>>)[0]?.type).toBe("additional_tools");
|
||||
});
|
||||
|
||||
it("omits the lite marker while retaining canonical client_metadata", async () => {
|
||||
const model = createCodexModel("gpt-5.1-codex");
|
||||
let captured: CapturedCodexRequest | undefined;
|
||||
const fetchMock = createCodexFetchMock(createCodexSse(COMPLETED_CODEX_EVENTS), request => {
|
||||
@@ -408,7 +644,7 @@ describe("openai-codex Responses Lite and client metadata wire format", () => {
|
||||
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBeNull();
|
||||
expect(captured?.body.client_metadata).toBeUndefined();
|
||||
expect(captured?.body.client_metadata).toBeDefined();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -467,3 +703,198 @@ describe("openai-codex websocket append with client metadata", () => {
|
||||
expect(transformed.client_metadata).toEqual({ "x-codex-turn-metadata": "{}" });
|
||||
});
|
||||
});
|
||||
|
||||
describe("openai-codex concurrent reasoning summaries", () => {
|
||||
it("counts atomic summary dones as websocket watchdog progress", () => {
|
||||
expect(isOpenAIResponsesProgressEvent({ type: "response.reasoning_summary_text.done" })).toBe(true);
|
||||
});
|
||||
|
||||
it("sends stream_options only when a summary is requested and supported", async () => {
|
||||
const terra = createCodexModel("gpt-5.6-terra");
|
||||
const withSummary = await transformRequestBody({ model: terra.id }, terra, { reasoningEffort: "medium" });
|
||||
expect(withSummary.stream_options).toEqual({ reasoning_summary_delivery: "sequential_cutoff" });
|
||||
|
||||
const suppressed = await transformRequestBody({ model: terra.id }, terra, {
|
||||
reasoningEffort: "medium",
|
||||
reasoningSummary: null,
|
||||
});
|
||||
expect(suppressed.stream_options).toBeUndefined();
|
||||
|
||||
const noReasoning = await transformRequestBody({ model: terra.id }, terra, {});
|
||||
expect(noReasoning.stream_options).toBeUndefined();
|
||||
|
||||
const legacy = createCodexModel("gpt-5.1-codex");
|
||||
const unsupported = await transformRequestBody({ model: legacy.id }, legacy, { reasoningEffort: "medium" });
|
||||
expect(unsupported.stream_options).toBeUndefined();
|
||||
});
|
||||
|
||||
it("deduplicates cumulative atomic summaries and ignores legacy deltas under sequential cutoff", async () => {
|
||||
const model = createCodexModel("gpt-5.6-terra");
|
||||
const events: Array<Record<string, unknown>> = [
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
output_index: 0,
|
||||
item: { type: "reasoning", id: "reason_1", summary: [] },
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_part.added",
|
||||
item_id: "reason_1",
|
||||
output_index: 0,
|
||||
summary_index: 0,
|
||||
part: { type: "summary_text", text: "" },
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_text.delta",
|
||||
item_id: "reason_1",
|
||||
output_index: 0,
|
||||
summary_index: 0,
|
||||
delta: "IGNORED",
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_text.done",
|
||||
item_id: "reason_1",
|
||||
output_index: 0,
|
||||
summary_index: 0,
|
||||
text: "Plan",
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_text.done",
|
||||
item_id: "reason_1",
|
||||
output_index: 0,
|
||||
summary_index: 1,
|
||||
text: "Planning details",
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_text.done",
|
||||
item_id: "reason_1",
|
||||
output_index: 0,
|
||||
summary_index: 1,
|
||||
text: "Planning details",
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_text.done",
|
||||
item_id: "reason_1",
|
||||
output_index: 0,
|
||||
summary_index: 2,
|
||||
text: "Plan\n\nPlanning details\n\nInspect",
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_text.done",
|
||||
item_id: "reason_1",
|
||||
output_index: 0,
|
||||
summary_index: 2,
|
||||
text: "Plan\n\nPlanning details\n\nInspect details",
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_text.done",
|
||||
item_id: "reason_1",
|
||||
output_index: 0,
|
||||
summary_index: 2,
|
||||
text: "Plan\n\nPlanning details\n\nInspect details",
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_text.done",
|
||||
item_id: "reason_1",
|
||||
output_index: 0,
|
||||
summary_index: 3,
|
||||
text: "Plan\n\nPlanning details\n\nInspect details",
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_text.done",
|
||||
item_id: "reason_1",
|
||||
output_index: 0,
|
||||
summary_index: 2,
|
||||
text: "Plan\n\nPlanning details\n\nReview",
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_text.done",
|
||||
item_id: "reason_1",
|
||||
output_index: 0,
|
||||
summary_index: 2,
|
||||
text: "Plan\n\nPlanning details\n\nReview output",
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_text.done",
|
||||
item_id: "reason_1",
|
||||
output_index: 0,
|
||||
summary_index: 3,
|
||||
text: "Plan\n\nPlanning details\n\nReview output",
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: {
|
||||
type: "reasoning",
|
||||
id: "reason_1",
|
||||
summary: [
|
||||
{ type: "summary_text", text: "Plan" },
|
||||
{ type: "summary_text", text: "Planning details" },
|
||||
{ type: "summary_text", text: "Plan\n\nPlanning details\n\nInspect details\n\nUnseen final" },
|
||||
{ type: "summary_text", text: "Plan\n\nPlanning details\n\nInspect details\n\nUnseen final" },
|
||||
],
|
||||
},
|
||||
},
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
output_index: 1,
|
||||
item: { type: "message", id: "msg_1", role: "assistant", status: "in_progress", content: [] },
|
||||
},
|
||||
{ type: "response.content_part.added", part: { type: "output_text", text: "" } },
|
||||
{ type: "response.output_text.delta", item_id: "msg_1", output_index: 1, delta: "Hello" },
|
||||
{
|
||||
type: "response.reasoning_summary_text.done",
|
||||
item_id: "reason_1",
|
||||
output_index: 0,
|
||||
summary_index: 4,
|
||||
text: "STALE",
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 1,
|
||||
item: {
|
||||
type: "message",
|
||||
id: "msg_1",
|
||||
role: "assistant",
|
||||
status: "completed",
|
||||
content: [{ type: "output_text", text: "Hello" }],
|
||||
},
|
||||
},
|
||||
{
|
||||
type: "response.completed",
|
||||
response: {
|
||||
status: "completed",
|
||||
usage: {
|
||||
input_tokens: 5,
|
||||
output_tokens: 3,
|
||||
total_tokens: 8,
|
||||
input_tokens_details: { cached_tokens: 0 },
|
||||
},
|
||||
},
|
||||
},
|
||||
];
|
||||
let captured: CapturedCodexRequest | undefined;
|
||||
const fetchMock = createCodexFetchMock(createCodexSse(events), request => {
|
||||
captured = request;
|
||||
});
|
||||
|
||||
const stream = streamOpenAICodexResponses(model, createCodexTestContext(), {
|
||||
apiKey: createCodexTestToken(),
|
||||
fetch: fetchMock,
|
||||
reasoning: "medium",
|
||||
});
|
||||
const thinkingDeltas: string[] = [];
|
||||
for await (const event of stream) {
|
||||
if (event.type === "thinking_delta") thinkingDeltas.push(event.delta);
|
||||
}
|
||||
const result = await stream.result();
|
||||
|
||||
expect(captured?.body.stream_options).toEqual({ reasoning_summary_delivery: "sequential_cutoff" });
|
||||
expect(thinkingDeltas).toEqual(["Plan", "\n\nPlanning details", "\n\nInspect", " details"]);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
const thinking = result.content.find(block => block.type === "thinking");
|
||||
expect(thinking?.thinking).toBe("Plan\n\nPlanning details\n\nInspect details");
|
||||
expect(thinking?.thinking).toBe(thinkingDeltas.join(""));
|
||||
const text = result.content.find(block => block.type === "text");
|
||||
expect(text?.text).toBe("Hello");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,18 +1,29 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
|
||||
import { streamSimple } from "@oh-my-pi/pi-ai";
|
||||
import {
|
||||
getOpenAICodexTransportDetails,
|
||||
getOpenAICodexWebSocketDebugStats,
|
||||
prewarmOpenAICodexResponses,
|
||||
resetOpenAICodexHistoryAfterCompaction,
|
||||
streamOpenAICodexResponses,
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-codex-responses";
|
||||
import type { Context, FetchImpl, Model, ModelSpec, ProviderSessionState } from "@oh-my-pi/pi-ai/types";
|
||||
import type {
|
||||
CodexCompactionRequestContext,
|
||||
Context,
|
||||
FetchImpl,
|
||||
Model,
|
||||
ModelSpec,
|
||||
ProviderSessionState,
|
||||
} from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { getAgentDir, setAgentDir, TempDir } from "@oh-my-pi/pi-utils";
|
||||
import * as piUtils from "@oh-my-pi/pi-utils";
|
||||
|
||||
const { getAgentDir, setAgentDir, TempDir } = piUtils;
|
||||
|
||||
const originalAgentDir = getAgentDir();
|
||||
const originalWebSocket = global.WebSocket;
|
||||
const originalCodexWebSocketV2 = Bun.env.PI_CODEX_WEBSOCKET_V2;
|
||||
const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001";
|
||||
|
||||
function restoreEnv(name: string, value: string | undefined): void {
|
||||
if (value === undefined) {
|
||||
@@ -22,6 +33,10 @@ function restoreEnv(name: string, value: string | undefined): void {
|
||||
Bun.env[name] = value;
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
global.WebSocket = originalWebSocket;
|
||||
setAgentDir(originalAgentDir);
|
||||
@@ -60,6 +75,22 @@ function createCodexTestContext(): Context {
|
||||
};
|
||||
}
|
||||
|
||||
function isRecord(value: unknown): value is Record<string, unknown> {
|
||||
return typeof value === "object" && value !== null && !Array.isArray(value);
|
||||
}
|
||||
|
||||
function requireRecord(value: unknown, label: string): Record<string, unknown> {
|
||||
if (!isRecord(value)) throw new Error(`expected ${label} to be an object`);
|
||||
return value;
|
||||
}
|
||||
|
||||
function parseTurnMetadata(clientMetadata: Record<string, unknown>): Record<string, unknown> {
|
||||
const encoded = clientMetadata["x-codex-turn-metadata"];
|
||||
if (typeof encoded !== "string") throw new Error("expected x-codex-turn-metadata");
|
||||
const decoded: unknown = JSON.parse(encoded);
|
||||
return requireRecord(decoded, "x-codex-turn-metadata");
|
||||
}
|
||||
|
||||
function createCompletedCodexSse(text: string): string {
|
||||
return `${[
|
||||
`data: ${JSON.stringify({ type: "response.content_part.added", part: { type: "output_text", text: "" } })}`,
|
||||
@@ -442,6 +473,129 @@ describe("openai-codex streaming", () => {
|
||||
expect(capturedText).toEqual({ verbosity: "low" });
|
||||
});
|
||||
|
||||
it("preserves streamed reasoning when the done item has no summary text", async () => {
|
||||
const token = createCodexTestToken();
|
||||
const model = { ...createCodexTestModel("https://chatgpt.com/backend-api"), preferWebsockets: false };
|
||||
const events = [
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
output_index: 0,
|
||||
item: { type: "reasoning", id: "rs_1", summary: [] },
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_part.added",
|
||||
output_index: 0,
|
||||
item_id: "rs_1",
|
||||
summary_index: 0,
|
||||
part: { type: "summary_text", text: "" },
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_text.delta",
|
||||
output_index: 0,
|
||||
item_id: "rs_1",
|
||||
summary_index: 0,
|
||||
delta: "streamed thinking",
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: { type: "reasoning", id: "rs_1", summary: [] },
|
||||
},
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
output_index: 1,
|
||||
item: { type: "message", id: "msg_1", role: "assistant", status: "in_progress", content: [] },
|
||||
},
|
||||
{
|
||||
type: "response.content_part.added",
|
||||
output_index: 1,
|
||||
item_id: "msg_1",
|
||||
part: { type: "output_text", text: "" },
|
||||
},
|
||||
{ type: "response.output_text.delta", output_index: 1, item_id: "msg_1", delta: "done" },
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 1,
|
||||
item: {
|
||||
type: "message",
|
||||
id: "msg_1",
|
||||
role: "assistant",
|
||||
status: "completed",
|
||||
content: [{ type: "output_text", text: "done" }],
|
||||
},
|
||||
},
|
||||
{
|
||||
type: "response.completed",
|
||||
response: {
|
||||
id: "resp_1",
|
||||
status: "completed",
|
||||
usage: {
|
||||
input_tokens: 5,
|
||||
output_tokens: 3,
|
||||
total_tokens: 8,
|
||||
input_tokens_details: { cached_tokens: 0 },
|
||||
},
|
||||
},
|
||||
},
|
||||
];
|
||||
const sse = `${events.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`;
|
||||
const fetchMock: FetchImpl = async () =>
|
||||
new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } });
|
||||
|
||||
const result = await streamOpenAICodexResponses(model, createCodexTestContext(), {
|
||||
apiKey: token,
|
||||
fetch: fetchMock,
|
||||
}).result();
|
||||
|
||||
expect(result.content.find(block => block.type === "thinking")?.thinking).toBe("streamed thinking");
|
||||
});
|
||||
|
||||
it("streams raw reasoning text deltas into the final thinking block", async () => {
|
||||
const token = createCodexTestToken();
|
||||
const model = { ...createCodexTestModel("https://chatgpt.com/backend-api"), preferWebsockets: false };
|
||||
const events = [
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
output_index: 0,
|
||||
item: { type: "reasoning", id: "rs_raw", summary: [] },
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_text.delta",
|
||||
output_index: 0,
|
||||
item_id: "rs_raw",
|
||||
delta: "raw streamed thinking",
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: { type: "reasoning", id: "rs_raw", summary: [] },
|
||||
},
|
||||
{
|
||||
type: "response.completed",
|
||||
response: {
|
||||
id: "resp_raw",
|
||||
status: "completed",
|
||||
usage: {
|
||||
input_tokens: 5,
|
||||
output_tokens: 3,
|
||||
total_tokens: 8,
|
||||
input_tokens_details: { cached_tokens: 0 },
|
||||
},
|
||||
},
|
||||
},
|
||||
];
|
||||
const sse = `${events.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`;
|
||||
const fetchMock: FetchImpl = async () =>
|
||||
new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } });
|
||||
|
||||
const result = await streamOpenAICodexResponses(model, createCodexTestContext(), {
|
||||
apiKey: token,
|
||||
fetch: fetchMock,
|
||||
}).result();
|
||||
|
||||
expect(result.content.find(block => block.type === "thinking")?.thinking).toBe("raw streamed thinking");
|
||||
});
|
||||
|
||||
it("maps end_turn=false on the terminal event to a pause_turn stop", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-codex-stream-");
|
||||
setAgentDir(tempDir.path());
|
||||
@@ -1099,7 +1253,7 @@ describe("openai-codex streaming", () => {
|
||||
sessionId: "ws-lite-session",
|
||||
providerSessionState: new Map<string, ProviderSessionState>(),
|
||||
responsesLite: true,
|
||||
clientMetadata: { "x-codex-turn-metadata": '{"thread_id":"t_1"}' },
|
||||
clientMetadata: { workspace_kind: "repo", "x-codex-turn-metadata": '{"thread_id":"caller"}' },
|
||||
},
|
||||
).result();
|
||||
|
||||
@@ -1107,10 +1261,28 @@ describe("openai-codex streaming", () => {
|
||||
expect(capturedHeaders?.["x-openai-internal-codex-responses-lite"]).toBe("true");
|
||||
expect(sentRequests).toHaveLength(1);
|
||||
expect(sentRequests[0]?.type).toBe("response.create");
|
||||
expect(sentRequests[0]?.client_metadata).toEqual({
|
||||
"x-codex-turn-metadata": '{"thread_id":"t_1"}',
|
||||
const metadata = requireRecord(sentRequests[0]?.client_metadata, "client_metadata");
|
||||
const turnMetadata = parseTurnMetadata(metadata);
|
||||
expect(metadata).toMatchObject({
|
||||
session_id: "ws-lite-session",
|
||||
ws_request_header_x_openai_internal_codex_responses_lite: "true",
|
||||
"x-codex-installation-id": TEST_INSTALLATION_ID,
|
||||
});
|
||||
expect(metadata.workspace_kind).toBeUndefined();
|
||||
expect(turnMetadata).toMatchObject({
|
||||
installation_id: TEST_INSTALLATION_ID,
|
||||
session_id: "ws-lite-session",
|
||||
thread_id: metadata.thread_id,
|
||||
turn_id: metadata.turn_id,
|
||||
window_id: metadata["x-codex-window-id"],
|
||||
request_kind: "turn",
|
||||
workspace_kind: "repo",
|
||||
});
|
||||
expect(capturedHeaders?.["x-codex-installation-id"]).toBeUndefined();
|
||||
expect(metadata.session_id).toBe(capturedHeaders?.["session-id"]);
|
||||
expect(metadata.thread_id).toBe(capturedHeaders?.["thread-id"]);
|
||||
expect(metadata["x-codex-window-id"]).toBe(capturedHeaders?.["x-codex-window-id"]);
|
||||
expect(metadata["x-codex-turn-metadata"]).toBe(capturedHeaders?.["x-codex-turn-metadata"]);
|
||||
});
|
||||
|
||||
it("streams SSE responses into AssistantMessageEventStream", async () => {
|
||||
@@ -1987,7 +2159,7 @@ describe("openai-codex streaming", () => {
|
||||
expect(fallbackDetails.fallbackCount).toBe(1);
|
||||
});
|
||||
|
||||
it("immediately falls back to SSE on fatal websocket connection errors", async () => {
|
||||
it("carries fatal websocket fallback into isolated compaction transport", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-codex-stream-");
|
||||
setAgentDir(tempDir.path());
|
||||
|
||||
@@ -2050,6 +2222,23 @@ describe("openai-codex streaming", () => {
|
||||
expect(result.role).toBe("assistant");
|
||||
expect(constructorCount).toBe(1);
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
const compacted = await streamOpenAICodexResponses(model, context, {
|
||||
fetch: fetchMock as FetchImpl,
|
||||
apiKey: token,
|
||||
sessionId: "ws-fatal-fallback-session",
|
||||
providerSessionState,
|
||||
codexCompaction: {
|
||||
operationId: "fallback-compaction",
|
||||
trigger: "auto",
|
||||
reason: "context_limit",
|
||||
implementation: "responses",
|
||||
phase: "pre_turn",
|
||||
strategy: "memento",
|
||||
},
|
||||
}).result();
|
||||
expect(compacted.stopReason).toBe("stop");
|
||||
expect(constructorCount).toBe(1);
|
||||
expect(fetchMock).toHaveBeenCalledTimes(2);
|
||||
const transportDetails = getOpenAICodexTransportDetails(model, {
|
||||
sessionId: "ws-fatal-fallback-session",
|
||||
providerSessionState,
|
||||
@@ -2059,7 +2248,7 @@ describe("openai-codex streaming", () => {
|
||||
expect(transportDetails.fallbackCount).toBe(1);
|
||||
});
|
||||
|
||||
it("captures websocket handshake metadata and replays it on later SSE requests", async () => {
|
||||
it("isolates compaction transport and preserves main mid-turn state", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-codex-stream-");
|
||||
setAgentDir(tempDir.path());
|
||||
|
||||
@@ -2076,12 +2265,21 @@ describe("openai-codex streaming", () => {
|
||||
`data: ${JSON.stringify({ type: "response.output_item.done", item: { type: "message", id: "msg_sse", role: "assistant", status: "completed", content: [{ type: "output_text", text: "Hello SSE" }] } })}`,
|
||||
`data: ${JSON.stringify({ type: "response.completed", response: { status: "completed", usage: { input_tokens: 5, output_tokens: 3, total_tokens: 8, input_tokens_details: { cached_tokens: 0 } } } })}`,
|
||||
].join("\n\n")}\n\n`;
|
||||
let firstRequest: Record<string, unknown> | undefined;
|
||||
let continuationRequest: Record<string, unknown> | undefined;
|
||||
let continuationHeaders: Headers | undefined;
|
||||
const fetchMock = vi.fn(async (_input: string | URL, init?: RequestInit) => {
|
||||
const headers = init?.headers instanceof Headers ? init.headers : new Headers(init?.headers);
|
||||
expect(headers.get("x-codex-turn-state")).toBe("ws-turn-state-1");
|
||||
expect(headers.get("x-models-etag")).toBe("models-etag-1");
|
||||
continuationHeaders = init?.headers instanceof Headers ? init.headers : new Headers(init?.headers);
|
||||
expect(continuationHeaders.get("x-codex-turn-state")).toBe("ws-turn-state-1");
|
||||
expect(continuationHeaders.get("x-models-etag")).toBe("models-etag-1");
|
||||
if (typeof init?.body !== "string") throw new Error("expected an SSE request body");
|
||||
const body: unknown = JSON.parse(init.body);
|
||||
continuationRequest = requireRecord(body, "SSE continuation request");
|
||||
return new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } });
|
||||
});
|
||||
let websocketRequestCount = 0;
|
||||
let websocketConstructorCount = 0;
|
||||
const websocketInstances: MockWebSocket[] = [];
|
||||
|
||||
class HandshakeWebSocket extends MockWebSocket {
|
||||
handshakeHeaders = {
|
||||
@@ -2092,11 +2290,29 @@ describe("openai-codex streaming", () => {
|
||||
|
||||
constructor(url: string, options?: { headers?: WsHeaders }) {
|
||||
super(url, options);
|
||||
websocketConstructorCount += 1;
|
||||
websocketInstances.push(this);
|
||||
this.scheduleOpen();
|
||||
}
|
||||
|
||||
send(): void {
|
||||
this.emitCodexResponse({ messageId: "msg_ws", responseId: "resp_ws", text: "Hello WS" });
|
||||
send(data: string): void {
|
||||
websocketRequestCount += 1;
|
||||
const body: unknown = JSON.parse(data);
|
||||
if (websocketRequestCount === 1) {
|
||||
firstRequest = requireRecord(body, "websocket request");
|
||||
}
|
||||
if (websocketRequestCount === 3) {
|
||||
this.sendJson({
|
||||
type: "response.failed",
|
||||
response: { error: { code: "invalid_request_error", message: "isolated compaction failed" } },
|
||||
});
|
||||
return;
|
||||
}
|
||||
this.emitCodexResponse({
|
||||
messageId: `msg_ws_${websocketRequestCount}`,
|
||||
responseId: `resp_ws_${websocketRequestCount}`,
|
||||
text: "Hello WS",
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2125,12 +2341,65 @@ describe("openai-codex streaming", () => {
|
||||
messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }],
|
||||
};
|
||||
const providerSessionState = new Map<string, ProviderSessionState>();
|
||||
const midTurnCompaction: CodexCompactionRequestContext = {
|
||||
operationId: "isolated-success",
|
||||
trigger: "auto",
|
||||
reason: "context_limit",
|
||||
implementation: "responses",
|
||||
phase: "mid_turn",
|
||||
strategy: "memento",
|
||||
};
|
||||
const first = await streamOpenAICodexResponses(websocketModel, context, {
|
||||
fetch: fetchMock as FetchImpl,
|
||||
apiKey: token,
|
||||
sessionId: "ws-handshake-session",
|
||||
providerSessionState,
|
||||
}).result();
|
||||
expect(websocketInstances[0]?.readyState).toBe(MockWebSocket.OPEN);
|
||||
const isolatedSuccess = await streamOpenAICodexResponses(websocketModel, createCodexTestContext(), {
|
||||
fetch: fetchMock as FetchImpl,
|
||||
apiKey: token,
|
||||
sessionId: "ws-handshake-session",
|
||||
providerSessionState,
|
||||
codexCompaction: midTurnCompaction,
|
||||
}).result();
|
||||
expect(isolatedSuccess.stopReason).toBe("stop");
|
||||
expect(websocketInstances[0]?.readyState).toBe(MockWebSocket.OPEN);
|
||||
expect(websocketInstances[1]?.readyState).toBe(MockWebSocket.CLOSED);
|
||||
expect(websocketInstances[1]?.options?.headers?.["x-codex-turn-state"]).toBe("ws-turn-state-1");
|
||||
expect(websocketInstances[1]?.options?.headers?.["x-models-etag"]).toBe("models-etag-1");
|
||||
resetOpenAICodexHistoryAfterCompaction({
|
||||
providerSessionState,
|
||||
sessionId: "ws-handshake-session",
|
||||
compaction: midTurnCompaction,
|
||||
});
|
||||
const isolatedFailure = await streamOpenAICodexResponses(websocketModel, createCodexTestContext(), {
|
||||
fetch: fetchMock as FetchImpl,
|
||||
apiKey: token,
|
||||
sessionId: "ws-handshake-session",
|
||||
providerSessionState,
|
||||
codexCompaction: {
|
||||
operationId: "isolated-failure",
|
||||
trigger: "auto",
|
||||
reason: "context_limit",
|
||||
implementation: "responses",
|
||||
phase: "mid_turn",
|
||||
strategy: "memento",
|
||||
},
|
||||
}).result();
|
||||
expect(isolatedFailure.stopReason).toBe("error");
|
||||
expect(websocketInstances[0]?.readyState).toBe(MockWebSocket.OPEN);
|
||||
expect(websocketInstances[2]?.readyState).toBe(MockWebSocket.CLOSED);
|
||||
expect(websocketConstructorCount).toBe(3);
|
||||
expect(
|
||||
getOpenAICodexTransportDetails(websocketModel, {
|
||||
sessionId: "ws-handshake-session",
|
||||
providerSessionState,
|
||||
}),
|
||||
).toMatchObject({
|
||||
websocketConnected: true,
|
||||
hasTurnState: true,
|
||||
});
|
||||
// Turn-state is scoped to the current turn, so the SSE replay must be a
|
||||
// within-turn continuation (trailing tool result) to carry the header.
|
||||
const followUp: Context = {
|
||||
@@ -2162,6 +2431,155 @@ describe("openai-codex streaming", () => {
|
||||
providerSessionState,
|
||||
}).result();
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
if (!firstRequest || !continuationRequest || !continuationHeaders) {
|
||||
throw new Error("expected both Codex transport requests");
|
||||
}
|
||||
const firstMetadata = requireRecord(firstRequest.client_metadata, "first client_metadata");
|
||||
const continuationMetadata = requireRecord(continuationRequest.client_metadata, "continuation client_metadata");
|
||||
const firstTurnMetadata = parseTurnMetadata(firstMetadata);
|
||||
const continuationTurnMetadata = parseTurnMetadata(continuationMetadata);
|
||||
expect(continuationMetadata).toMatchObject({
|
||||
"x-codex-installation-id": TEST_INSTALLATION_ID,
|
||||
session_id: firstMetadata.session_id,
|
||||
thread_id: firstMetadata.thread_id,
|
||||
turn_id: firstMetadata.turn_id,
|
||||
});
|
||||
expect(continuationTurnMetadata).toMatchObject({
|
||||
installation_id: TEST_INSTALLATION_ID,
|
||||
session_id: firstTurnMetadata.session_id,
|
||||
thread_id: firstTurnMetadata.thread_id,
|
||||
turn_id: firstTurnMetadata.turn_id,
|
||||
window_id: continuationMetadata["x-codex-window-id"],
|
||||
request_kind: "turn",
|
||||
turn_started_at_unix_ms: context.messages[0]?.timestamp,
|
||||
});
|
||||
expect(typeof continuationMetadata["x-codex-window-id"]).toBe("string");
|
||||
expect(continuationMetadata["x-codex-window-id"]).not.toBe(firstMetadata["x-codex-window-id"]);
|
||||
expect(firstMetadata.session_id).toBe(continuationHeaders.get("session-id"));
|
||||
expect(firstMetadata.thread_id).toBe(continuationHeaders.get("thread-id"));
|
||||
expect(continuationMetadata["x-codex-window-id"]).toBe(continuationHeaders.get("x-codex-window-id"));
|
||||
expect(continuationMetadata["x-codex-turn-metadata"]).toBe(continuationHeaders.get("x-codex-turn-metadata"));
|
||||
});
|
||||
|
||||
it("clears stale main turn-state after pre-turn compaction", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-codex-stream-");
|
||||
setAgentDir(tempDir.path());
|
||||
const token = createCodexTestToken();
|
||||
const websocketInstances: MockWebSocket[] = [];
|
||||
let websocketRequestCount = 0;
|
||||
|
||||
class PreTurnCompactionWebSocket extends MockWebSocket {
|
||||
handshakeHeaders = {
|
||||
"x-codex-turn-state": "stale-main-turn-state",
|
||||
"x-models-etag": "models-etag-1",
|
||||
};
|
||||
|
||||
constructor(url: string, options?: { headers?: WsHeaders }) {
|
||||
super(url, options);
|
||||
websocketInstances.push(this);
|
||||
queueMicrotask(() => {
|
||||
this.readyState = MockWebSocket.OPEN;
|
||||
this.emit("open", new Event("open"));
|
||||
});
|
||||
}
|
||||
|
||||
send(_data: string): void {
|
||||
websocketRequestCount += 1;
|
||||
this.emitCodexResponse({
|
||||
messageId: `msg_pre_turn_${websocketRequestCount}`,
|
||||
responseId: `resp_pre_turn_${websocketRequestCount}`,
|
||||
text: "Hello WS",
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
global.WebSocket = PreTurnCompactionWebSocket as unknown as typeof WebSocket;
|
||||
const websocketModel = createCodexTestModel("https://chatgpt.com/backend-api");
|
||||
const sseModel: Model<"openai-codex-responses"> = buildModel({
|
||||
id: websocketModel.id,
|
||||
name: websocketModel.name,
|
||||
api: "openai-codex-responses",
|
||||
provider: websocketModel.provider,
|
||||
baseUrl: websocketModel.baseUrl,
|
||||
reasoning: true,
|
||||
preferWebsockets: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128000,
|
||||
maxTokens: 128000,
|
||||
});
|
||||
const providerSessionState = new Map<string, ProviderSessionState>();
|
||||
const sessionId = "pre-turn-reset-session";
|
||||
let sseHeaders: Headers | undefined;
|
||||
const fetchMock = vi.fn(async (_input: string | URL, init?: RequestInit) => {
|
||||
sseHeaders = init?.headers instanceof Headers ? init.headers : new Headers(init?.headers);
|
||||
return new Response(createCompletedCodexSse("Hello SSE"), {
|
||||
headers: { "content-type": "text/event-stream" },
|
||||
});
|
||||
});
|
||||
const compaction: CodexCompactionRequestContext = {
|
||||
operationId: "pre-turn-reset-operation",
|
||||
trigger: "auto",
|
||||
reason: "context_limit",
|
||||
implementation: "responses",
|
||||
phase: "pre_turn",
|
||||
strategy: "memento",
|
||||
};
|
||||
|
||||
try {
|
||||
await streamOpenAICodexResponses(websocketModel, createCodexTestContext(), {
|
||||
apiKey: token,
|
||||
fetch: fetchMock as FetchImpl,
|
||||
sessionId,
|
||||
providerSessionState,
|
||||
}).result();
|
||||
await streamOpenAICodexResponses(websocketModel, createCodexTestContext(), {
|
||||
apiKey: token,
|
||||
fetch: fetchMock as FetchImpl,
|
||||
sessionId,
|
||||
providerSessionState,
|
||||
codexCompaction: compaction,
|
||||
}).result();
|
||||
expect(fetchMock).not.toHaveBeenCalled();
|
||||
expect(websocketInstances).toHaveLength(2);
|
||||
expect(websocketInstances[0]?.readyState).toBe(MockWebSocket.OPEN);
|
||||
expect(websocketInstances[1]?.readyState).toBe(MockWebSocket.CLOSED);
|
||||
expect(websocketInstances[1]?.options?.headers?.["x-codex-turn-state"]).toBeUndefined();
|
||||
expect(websocketInstances[1]?.options?.headers?.["x-models-etag"]).toBe("models-etag-1");
|
||||
|
||||
resetOpenAICodexHistoryAfterCompaction({
|
||||
providerSessionState,
|
||||
sessionId,
|
||||
compaction,
|
||||
});
|
||||
expect(
|
||||
getOpenAICodexTransportDetails(websocketModel, {
|
||||
sessionId,
|
||||
providerSessionState,
|
||||
}),
|
||||
).toMatchObject({
|
||||
websocketConnected: true,
|
||||
hasTurnState: false,
|
||||
});
|
||||
await streamOpenAICodexResponses(
|
||||
sseModel,
|
||||
{
|
||||
systemPrompt: ["You are a helpful assistant."],
|
||||
messages: [{ role: "user", content: "Continue after compaction", timestamp: Date.now() }],
|
||||
},
|
||||
{
|
||||
apiKey: token,
|
||||
fetch: fetchMock as FetchImpl,
|
||||
sessionId,
|
||||
providerSessionState,
|
||||
},
|
||||
).result();
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
expect(sseHeaders?.get("x-codex-turn-state")).toBeNull();
|
||||
} finally {
|
||||
for (const state of providerSessionState.values()) state.close();
|
||||
providerSessionState.clear();
|
||||
}
|
||||
});
|
||||
|
||||
it("includes service_tier in websocket payloads when requested", async () => {
|
||||
@@ -2349,6 +2767,17 @@ describe("openai-codex streaming", () => {
|
||||
expect(deltaItems[0]?.role).toBe("user");
|
||||
expect(JSON.stringify(deltaItems)).toContain("Second question");
|
||||
expect(JSON.stringify(deltaItems)).not.toContain("First answer");
|
||||
const firstMetadata = requireRecord(sentRequests[0]?.client_metadata, "first client_metadata");
|
||||
const secondMetadata = requireRecord(sentRequests[1]?.client_metadata, "second client_metadata");
|
||||
expect(secondMetadata).toMatchObject({
|
||||
"x-codex-installation-id": firstMetadata["x-codex-installation-id"],
|
||||
session_id: firstMetadata.session_id,
|
||||
thread_id: firstMetadata.thread_id,
|
||||
"x-codex-window-id": firstMetadata["x-codex-window-id"],
|
||||
});
|
||||
expect(secondMetadata.turn_id).not.toBe(firstMetadata.turn_id);
|
||||
expect(parseTurnMetadata(firstMetadata).turn_started_at_unix_ms).toBe(firstContext.messages[0]?.timestamp);
|
||||
expect(parseTurnMetadata(secondMetadata).turn_started_at_unix_ms).toBe(secondContext.messages.at(-1)?.timestamp);
|
||||
|
||||
const stats = getOpenAICodexWebSocketDebugStats(model, {
|
||||
sessionId: "ws-delta-session",
|
||||
@@ -3815,10 +4244,12 @@ describe("openai-codex streaming", () => {
|
||||
|
||||
let constructorCount = 0;
|
||||
let sendCount = 0;
|
||||
let prewarmHeaders: WsHeaders | undefined;
|
||||
class ReusableWebSocket extends MockWebSocket {
|
||||
constructor(url: string, options?: { headers?: WsHeaders }) {
|
||||
super(url, options);
|
||||
constructorCount += 1;
|
||||
prewarmHeaders = options?.headers;
|
||||
this.scheduleOpen();
|
||||
}
|
||||
|
||||
@@ -3856,6 +4287,11 @@ describe("openai-codex streaming", () => {
|
||||
sessionId: "ws-reuse-session",
|
||||
providerSessionState,
|
||||
});
|
||||
expect(prewarmHeaders?.["session-id"]).toBe("ws-reuse-session");
|
||||
expect(prewarmHeaders?.["thread-id"]).toBeDefined();
|
||||
expect(prewarmHeaders?.["x-codex-window-id"]).toBeDefined();
|
||||
expect(prewarmHeaders?.["x-codex-turn-metadata"]).toBeUndefined();
|
||||
expect(prewarmHeaders?.["x-codex-installation-id"]).toBeUndefined();
|
||||
|
||||
const firstContext: Context = {
|
||||
systemPrompt: ["You are a helpful assistant."],
|
||||
@@ -3893,6 +4329,26 @@ describe("openai-codex streaming", () => {
|
||||
expect(transportDetails.websocketConnected).toBe(true);
|
||||
expect(transportDetails.prewarmed).toBe(true);
|
||||
expect(transportDetails.canAppend).toBe(true);
|
||||
resetOpenAICodexHistoryAfterCompaction({
|
||||
providerSessionState,
|
||||
sessionId: "ws-reuse-session",
|
||||
compaction: {
|
||||
operationId: "history-rewrite",
|
||||
trigger: "auto",
|
||||
reason: "context_limit",
|
||||
phase: "pre_turn",
|
||||
strategy: "memento",
|
||||
},
|
||||
});
|
||||
expect(
|
||||
getOpenAICodexTransportDetails(model, {
|
||||
sessionId: "ws-reuse-session",
|
||||
providerSessionState,
|
||||
}),
|
||||
).toMatchObject({
|
||||
websocketConnected: true,
|
||||
canAppend: false,
|
||||
});
|
||||
});
|
||||
|
||||
it("scopes x-codex-turn-state to the current turn on SSE requests", async () => {
|
||||
|
||||
@@ -314,6 +314,36 @@ describe("openai-codex reasoning effort validation", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("openai-codex reasoning effort wire mapping", () => {
|
||||
it("shifts gpt-5.6 user efforts one wire tier up via the baked effort map", async () => {
|
||||
const model = createCodexModel("gpt-5.6-sol");
|
||||
const shifted = [
|
||||
["minimal", "low"],
|
||||
["low", "medium"],
|
||||
["medium", "high"],
|
||||
["high", "xhigh"],
|
||||
["xhigh", "max"],
|
||||
] as const;
|
||||
|
||||
for (const [requested, wire] of shifted) {
|
||||
const transformed = await transformRequestBody({ model: model.id }, model, {
|
||||
reasoningEffort: requested,
|
||||
});
|
||||
expect(transformed.reasoning?.effort).toBe(wire);
|
||||
}
|
||||
});
|
||||
|
||||
it("keeps pre-5.6 efforts unshifted and passes none through unmapped", async () => {
|
||||
const gpt55 = createCodexModel("gpt-5.5");
|
||||
const unshifted = await transformRequestBody({ model: gpt55.id }, gpt55, { reasoningEffort: "xhigh" });
|
||||
expect(unshifted.reasoning?.effort).toBe("xhigh");
|
||||
|
||||
const gpt56 = createCodexModel("gpt-5.6-sol");
|
||||
const none = await transformRequestBody({ model: gpt56.id }, gpt56, { reasoningEffort: "none" });
|
||||
expect(none.reasoning?.effort).toBe("none");
|
||||
});
|
||||
});
|
||||
|
||||
describe("openai-codex error parsing", () => {
|
||||
it("produces friendly usage-limit messages and rate limits", async () => {
|
||||
const resetAt = Math.floor(Date.now() / 1000) + 600;
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { type OpenAICompletionsOptions, streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import type { Context, FetchImpl } from "@oh-my-pi/pi-ai/types";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
|
||||
const model = getBundledModel<"openai-completions">("xai", "grok-code-fast-1");
|
||||
if (!model) throw new Error("Expected bundled xAI Grok model");
|
||||
if (model.api !== "openai-completions") throw new Error(`Expected Chat Completions model, received ${model.api}`);
|
||||
const context: Context = { messages: [{ role: "user", content: "hello", timestamp: 0 }] };
|
||||
|
||||
function chatCompletionsSse(): Response {
|
||||
const chunk = (delta: unknown, finishReason: string | null) =>
|
||||
JSON.stringify({
|
||||
id: "chatcmpl-affinity",
|
||||
object: "chat.completion.chunk",
|
||||
created: 0,
|
||||
model: model.id,
|
||||
choices: [{ index: 0, delta, finish_reason: finishReason }],
|
||||
});
|
||||
|
||||
return new Response(
|
||||
`data: ${chunk({ role: "assistant", content: "ok" }, null)}\n\ndata: ${chunk({}, "stop")}\n\ndata: [DONE]\n\n`,
|
||||
{ status: 200, headers: { "content-type": "text/event-stream" } },
|
||||
);
|
||||
}
|
||||
|
||||
async function captureRequestHeaders(options: OpenAICompletionsOptions): Promise<Headers> {
|
||||
let requestHeaders: Headers | undefined;
|
||||
const fetchMock: FetchImpl = async (input: string | URL | Request, init?: RequestInit) => {
|
||||
const request =
|
||||
input instanceof Request
|
||||
? new Request(input, init)
|
||||
: new Request(input instanceof URL ? input.href : input, init);
|
||||
requestHeaders = request.headers;
|
||||
return chatCompletionsSse();
|
||||
};
|
||||
|
||||
await streamOpenAICompletions(model, context, {
|
||||
apiKey: "test-key",
|
||||
...options,
|
||||
fetch: fetchMock,
|
||||
}).result();
|
||||
|
||||
if (!requestHeaders) throw new Error("Expected a serialized Chat Completions request");
|
||||
return requestHeaders;
|
||||
}
|
||||
|
||||
describe("openai-completions xAI cache affinity", () => {
|
||||
const cases: Array<{
|
||||
name: string;
|
||||
options: OpenAICompletionsOptions;
|
||||
expectedHeader: string | null;
|
||||
}> = [
|
||||
{
|
||||
name: "uses sessionId when no prompt cache key is provided",
|
||||
options: { sessionId: "session-fallback" },
|
||||
expectedHeader: "session-fallback",
|
||||
},
|
||||
{
|
||||
name: "keeps the prompt cache key stable across a distinct side-channel session",
|
||||
options: { promptCacheKey: "stable-cache-key", sessionId: "side-channel-session" },
|
||||
expectedHeader: "stable-cache-key",
|
||||
},
|
||||
{
|
||||
name: "omits automatic affinity when caching is disabled",
|
||||
options: {
|
||||
promptCacheKey: "disabled-cache-key",
|
||||
sessionId: "disabled-session",
|
||||
cacheRetention: "none",
|
||||
},
|
||||
expectedHeader: null,
|
||||
},
|
||||
{
|
||||
name: "preserves a caller-provided mixed-case affinity header",
|
||||
options: {
|
||||
promptCacheKey: "automatic-cache-key",
|
||||
sessionId: "automatic-session",
|
||||
headers: { "X-Grok-Conv-Id": "caller-affinity" },
|
||||
},
|
||||
expectedHeader: "caller-affinity",
|
||||
},
|
||||
];
|
||||
|
||||
for (const { name, options, expectedHeader } of cases) {
|
||||
it(name, async () => {
|
||||
const headers = await captureRequestHeaders(options);
|
||||
|
||||
expect(headers.get("x-grok-conv-id")).toBe(expectedHeader);
|
||||
});
|
||||
}
|
||||
});
|
||||
@@ -1,4 +1,4 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
|
||||
import {
|
||||
convertCodexResponsesMessages,
|
||||
streamOpenAICodexResponses,
|
||||
@@ -9,6 +9,17 @@ import type { Context, Model, ModelSpec, ProviderSessionState } from "@oh-my-pi/
|
||||
import { createOpenAIResponsesHistoryPayload, truncateResponseItemId } from "@oh-my-pi/pi-ai/utils";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import * as piUtils from "@oh-my-pi/pi-utils";
|
||||
|
||||
const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001";
|
||||
|
||||
beforeEach(() => {
|
||||
vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
function createAbortedSignal(): AbortSignal {
|
||||
const controller = new AbortController();
|
||||
|
||||
@@ -288,7 +288,13 @@ describe("processResponsesStream: lost output_item.added recovery", () => {
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: { type: "reasoning", summary: [{ type: "summary_text", text: "first" }] },
|
||||
item: {
|
||||
type: "reasoning",
|
||||
summary: [
|
||||
{ type: "summary_text", text: "Plan" },
|
||||
{ type: "summary_text", text: "Planning details" },
|
||||
],
|
||||
},
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
@@ -305,12 +311,55 @@ describe("processResponsesStream: lost output_item.added recovery", () => {
|
||||
expect(output.content).toHaveLength(2);
|
||||
const [first, second] = output.content;
|
||||
if (first?.type !== "thinking" || second?.type !== "thinking") throw new Error("expected thinking blocks");
|
||||
expect(first.thinking).toBe("first");
|
||||
expect(first.thinking).toBe("Plan\n\nPlanning details");
|
||||
expect(second.thinking).toBe("second");
|
||||
expect(first.thinkingSignature).toBeDefined();
|
||||
expect(second.thinkingSignature).toBeDefined();
|
||||
});
|
||||
|
||||
test("preserves streamed reasoning when the done item has no summary text", async () => {
|
||||
const output = makeOutput();
|
||||
const stream = { push: () => {}, end: () => {} } as never;
|
||||
|
||||
await processResponsesStream(
|
||||
makeStream([
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
output_index: 0,
|
||||
item: { type: "reasoning", id: "rs_1", summary: [] },
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_part.added",
|
||||
output_index: 0,
|
||||
item_id: "rs_1",
|
||||
summary_index: 0,
|
||||
part: { type: "summary_text", text: "" },
|
||||
},
|
||||
{
|
||||
type: "response.reasoning_summary_text.delta",
|
||||
output_index: 0,
|
||||
item_id: "rs_1",
|
||||
summary_index: 0,
|
||||
delta: "streamed thinking",
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: { type: "reasoning", id: "rs_1", summary: [] },
|
||||
},
|
||||
{ type: "response.completed", response: { id: "resp_reasoning", status: "completed" } },
|
||||
]),
|
||||
output,
|
||||
stream,
|
||||
makeModel(),
|
||||
);
|
||||
|
||||
const block = output.content[0];
|
||||
if (block?.type !== "thinking") throw new Error("expected a thinking block");
|
||||
expect(block.thinking).toBe("streamed thinking");
|
||||
expect(block.thinkingSignature).toBeDefined();
|
||||
});
|
||||
|
||||
test("treats content_filter incomplete responses as errors, not length", async () => {
|
||||
const output = makeOutput();
|
||||
const stream = { push: () => {}, end: () => {} } as never;
|
||||
|
||||
@@ -81,7 +81,6 @@ describe("provider registry auth surface", () => {
|
||||
"google-antigravity",
|
||||
"google-gemini-cli",
|
||||
"openai-codex",
|
||||
"xai-oauth",
|
||||
].sort(),
|
||||
);
|
||||
expect(PASTE_CODE_LOGIN_PROVIDERS.has("zenmux")).toBe(false);
|
||||
|
||||
@@ -2,6 +2,62 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added Grok 4.5 model family
|
||||
- Added support for Dolphin Mistral 24b Venice Edition
|
||||
- Added GLM5.2-Fast model
|
||||
- Added Zenmux variants for GPT-5.6 (Luna, Sol, and Terra)
|
||||
- Added Novita as a model provider with authoritative public catalog discovery and generated pricing, limits, modality, reasoning, and tool metadata ([#4917](https://github.com/can1357/oh-my-pi/pull/4917) by [@jason-wu-ai](https://github.com/jason-wu-ai)).
|
||||
|
||||
- Added `useResponsesLite` to `Model`/`ModelSpec` and Codex discovery parsing of the upstream `use_responses_lite` flag; regenerated `models.json` marks the GPT-5.6 family (`sol`/`terra`/`luna` and their pro aliases) for the Responses Lite transport. Added the `x-openai-internal-codex-responses-lite` marker to `OPENAI_HEADERS`.
|
||||
|
||||
### Changed
|
||||
|
||||
- Updated costs and context windows for various models in the catalog
|
||||
|
||||
## [16.3.15] - 2026-07-09
|
||||
|
||||
### Added
|
||||
|
||||
- Added support for Grok 4.5 model
|
||||
- Added `gpt-5.6` base models and `gpt-5.6-{luna,sol,terra}-pro` variants
|
||||
- Added `meta/muse-spark-1.1` model support
|
||||
- Added support for thinking modes on `poolside/laguna` models
|
||||
- Added generated GPT-5.6 Pro aliases (`gpt-5.6-{luna,sol,terra}-pro`) on the `openai` and `openai-codex` providers: each alias sends the base model id on the wire (`requestModelId`) with the new `reasoningMode: "pro"` marker, and re-derives from the current base rows on every catalog regeneration.
|
||||
|
||||
### Changed
|
||||
|
||||
- Updated cache read costs for Grok models
|
||||
- Reduced max token limit for Grok 4.3 model
|
||||
- Enabled prompt cache affinity for Grok models via the x-grok-conv-id header in OpenAI compatible endpoints
|
||||
- Enabled prompt cache affinity for Grok models via the x-grok-conv-id header
|
||||
- Marked direct xAI Grok Chat Completions models for `x-grok-conv-id` prompt-cache affinity.
|
||||
|
||||
## [16.3.14] - 2026-07-09
|
||||
|
||||
### Added
|
||||
|
||||
- Added support for GPT-5.6 (Luna, Sol, Terra) model variants
|
||||
- Enabled expanded five-tier reasoning effort scale (minimal to xhigh) for GPT-5.6 models
|
||||
- Added GPT-5.6 (Terra/Luna/Sol) support for the new `max` reasoning tier: on wire-effort APIs (OpenAI Responses, Codex, Azure, openai-compat/OpenRouter models that advertise reasoning) user efforts shift up one notch — `xhigh` sends `max`, `high` sends `xhigh` — mirroring the Claude Fable/Opus 4.7+ five-tier mapping, and the exposed ladder becomes `minimal..xhigh` with `minimal` reaching the native `low` tier. Devin's per-tier GPT-5.6 sibling rows now collapse into `gpt-5-6-{luna,sol,terra}` logical models with the same shifted routing (`xhigh` → `-max`), plus `-fast` families that keep the direct `low..xhigh` `-priority` scale since Devin serves no `-max-priority` tier.
|
||||
|
||||
## [16.3.13] - 2026-07-09
|
||||
|
||||
### Added
|
||||
|
||||
- Added support for Grok 4.5 across multiple providers
|
||||
- Added support for GPT-5.6 series models (Luna, Sol, Terra)
|
||||
- Added Aion 3.0 and 3.0 Mini models
|
||||
- Added Kuaishou KAT-Coder v2.5 models
|
||||
- Added Nex-N2-Mini and SWE-1.7 series models
|
||||
- Added Hy3 models and free variants
|
||||
|
||||
### Changed
|
||||
|
||||
- Updated cost and token configurations for various models across providers
|
||||
- Renamed several models for consistency (e.g., MiniMax M3, Gemma 4 31B, Qwen variants)
|
||||
|
||||
## [16.3.12] - 2026-07-08
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"type": "module",
|
||||
"name": "@oh-my-pi/pi-catalog",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence",
|
||||
"homepage": "https://omp.sh",
|
||||
"author": "Can Boluk",
|
||||
|
||||
@@ -38,6 +38,7 @@ import {
|
||||
isKimiK27CodeModelId,
|
||||
MODELS_DEV_PROVIDER_DESCRIPTORS,
|
||||
mapModelsDevToModels,
|
||||
projectOpenAIProReasoningAliases,
|
||||
SAKANA_FUGU_STATIC_MODELS,
|
||||
stripFireworksDeepSeekThinkingToggle,
|
||||
} from "../src/provider-models/openai-compat";
|
||||
@@ -585,6 +586,10 @@ async function generateModels() {
|
||||
const name = cleanModelName(model.name);
|
||||
return name === model.name ? model : { ...model, name };
|
||||
});
|
||||
// Re-derive the first-party gpt-5.6 pro-reasoning aliases from the current
|
||||
// base rows (stale previous-snapshot aliases are dropped inside), before the
|
||||
// policy re-bake so the aliases get the same baked thinking metadata.
|
||||
allModels = projectOpenAIProReasoningAliases(allModels);
|
||||
applyGeneratedModelPolicies(allModels);
|
||||
linkOpenAIPromotionTargets(allModels);
|
||||
// Collapse effort-tier variants AFTER the policy re-bake: live-discovery
|
||||
|
||||
@@ -541,7 +541,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
MINIMAX_PROVIDER_OR_ID_PATTERN.test(provider) || MINIMAX_PROVIDER_OR_ID_PATTERN.test(spec.id),
|
||||
emptyLengthFinishIsContextError: provider === "ollama",
|
||||
usesOpenAIToolCallIdLimit: provider === "openai",
|
||||
promptCacheSessionHeader: undefined,
|
||||
promptCacheSessionHeader: isGrok ? "x-grok-conv-id" : undefined,
|
||||
dropThinkingWhenReasoningEffort: provider === "fireworks",
|
||||
};
|
||||
|
||||
|
||||
@@ -30,6 +30,7 @@ const codexModelEntrySchema = type({
|
||||
"supported_in_api?": "unknown",
|
||||
"priority?": "unknown",
|
||||
"prefer_websockets?": "unknown",
|
||||
"use_responses_lite?": "unknown",
|
||||
});
|
||||
|
||||
const codexModelsResponseSchema = type({
|
||||
@@ -262,6 +263,7 @@ function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCo
|
||||
const reasoning = supportsReasoning(payload.default_reasoning_level, payload.supported_reasoning_levels);
|
||||
const input = normalizeInputModalities(payload.input_modalities);
|
||||
const preferWebsockets = toBoolean(payload.prefer_websockets) === true;
|
||||
const useResponsesLite = toBoolean(payload.use_responses_lite) === true;
|
||||
const priority = toFiniteNumber(payload.priority) ?? Number.MAX_SAFE_INTEGER;
|
||||
|
||||
return {
|
||||
@@ -279,6 +281,7 @@ function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCo
|
||||
contextWindow,
|
||||
maxTokens,
|
||||
...(preferWebsockets ? { preferWebsockets: true } : {}),
|
||||
...(useResponsesLite ? { useResponsesLite: true } : {}),
|
||||
...(priority !== Number.MAX_SAFE_INTEGER ? { priority } : {}),
|
||||
},
|
||||
};
|
||||
|
||||
@@ -13,6 +13,13 @@ const CURSOR_GET_USABLE_MODELS_PATH = "/agent.v1.AgentService/GetUsableModels";
|
||||
const DEFAULT_CONTEXT_WINDOW = 200_000;
|
||||
const DEFAULT_MAX_TOKENS = 64_000;
|
||||
|
||||
/**
|
||||
* Model-id families whose native catalogs (anthropic, openai/openai-codex,
|
||||
* google) are multimodal. Cursor-only or text-only families (`composer-*`,
|
||||
* `grok-code-*`) intentionally stay outside this pattern.
|
||||
*/
|
||||
const CURSOR_MULTIMODAL_ID_PATTERN = /claude|gemini|gpt-|codex/;
|
||||
|
||||
const OptionalDisplayNameSchema = type("unknown").pipe(raw => (typeof raw === "string" ? raw : undefined));
|
||||
const CursorAliasesSchema = type("unknown").pipe(raw => {
|
||||
if (Array.isArray(raw)) {
|
||||
@@ -292,7 +299,7 @@ function normalizeCursorModel(
|
||||
provider: "cursor",
|
||||
baseUrl: baseUrlOverride ?? CURSOR_DEFAULT_BASE_URL,
|
||||
reasoning,
|
||||
input: ["text"],
|
||||
input: inferInputFromCursorId(id),
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: DEFAULT_CONTEXT_WINDOW,
|
||||
maxTokens: DEFAULT_MAX_TOKENS,
|
||||
@@ -312,3 +319,18 @@ function pickModelDisplayName(model: CursorModelDetailsValue, fallbackId: string
|
||||
}
|
||||
return fallbackId;
|
||||
}
|
||||
|
||||
/**
|
||||
* Infers input modalities for Cursor models without a bundled reference.
|
||||
*
|
||||
* `GetUsableModels` carries no per-model modality metadata, so classification
|
||||
* falls back to the model family: families that are multimodal in OMP's own
|
||||
* native catalogs accept images, everything else stays text-only. Mirrors
|
||||
* `inferInputFromGeminiId` in ./gemini.ts.
|
||||
*/
|
||||
function inferInputFromCursorId(id: string): ("text" | "image")[] {
|
||||
if (CURSOR_MULTIMODAL_ID_PATTERN.test(id.toLowerCase())) {
|
||||
return ["text", "image"];
|
||||
}
|
||||
return ["text"];
|
||||
}
|
||||
|
||||
@@ -78,7 +78,7 @@ export const isMimoModelIdOrName = memo((value: string): boolean => {
|
||||
return value.toLowerCase().includes("mimo");
|
||||
});
|
||||
|
||||
const GROK_EFFORT_CAPABLE_PREFIXES = ["grok-3-mini", "grok-4.20-multi-agent", "grok-4.3"] as const;
|
||||
const GROK_EFFORT_CAPABLE_PREFIXES = ["grok-3-mini", "grok-4.20-multi-agent", "grok-4.3", "grok-4.5"] as const;
|
||||
|
||||
/**
|
||||
* Grok SKUs that expose the wire `reasoning.effort` dial. Other Grok reasoners
|
||||
|
||||
@@ -35,7 +35,7 @@ export interface ModelManagerOptions<TApi extends Api = Api, TModelsDevPayload =
|
||||
cacheDbPath?: string;
|
||||
/** Optional provider id override for cache namespacing. Defaults to providerId. */
|
||||
cacheProviderId?: string;
|
||||
/** Maximum cache age in milliseconds before considered stale. Default: 24h. */
|
||||
/** Maximum cache age in milliseconds before considered stale. Default: 2h (`DEFAULT_CACHE_TTL_MS`). */
|
||||
cacheTtlMs?: number;
|
||||
/** When true, a successful dynamic fetch is the complete provider catalog and prunes static-only models. */
|
||||
dynamicModelsAuthoritative?: boolean;
|
||||
|
||||
@@ -17,6 +17,7 @@ import {
|
||||
type ParsedModel,
|
||||
parseAnthropicModel,
|
||||
parseKnownModel,
|
||||
parseOpenAIModel,
|
||||
semverEqual,
|
||||
semverGte,
|
||||
} from "./identity/classify";
|
||||
@@ -101,12 +102,14 @@ const MIMO_REASONING_EFFORT_MAP: Readonly<EffortMap> = {
|
||||
};
|
||||
|
||||
/**
|
||||
* Effort → wire-value map for the 5-tier adaptive scale (Opus 4.7+ and
|
||||
* Fable/Mythos 5 on the Messages API). User-facing efforts shift up one notch
|
||||
* so the top tier reaches the genuine "max" and "high" lands on Anthropic's
|
||||
* recommended "xhigh" coding/agentic default.
|
||||
* Effort → wire-value map for a shifted five-tier scale (`low..max`):
|
||||
* user-facing efforts shift up one notch so the top tier reaches the genuine
|
||||
* "max" and "high" lands on the recommended "xhigh" coding/agentic default.
|
||||
* Used by Anthropic adaptive models with a real xhigh tier (Opus 4.7+ and
|
||||
* Fable/Mythos 5 on the Messages API) and by GPT-5.6+ wire-effort models,
|
||||
* which expose the same genuine `max` tier above `xhigh`.
|
||||
*/
|
||||
export const ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER: Readonly<Partial<Record<Effort, string>>> = {
|
||||
export const SHIFTED_FIVE_TIER_EFFORT_MAP: Readonly<Partial<Record<Effort, string>>> = {
|
||||
[Effort.Minimal]: "low",
|
||||
[Effort.Low]: "medium",
|
||||
[Effort.Medium]: "high",
|
||||
@@ -295,6 +298,27 @@ function isOpenAICompatReasoningApi(api: Api): boolean {
|
||||
return api === "openai-completions" || api === "openrouter";
|
||||
}
|
||||
|
||||
/**
|
||||
* GPT-5.6+ addressed through a wire `reasoning.effort`/`reasoning_effort`
|
||||
* field, where the shifted five-tier map applies. Devin (`devin-agent`)
|
||||
* selects effort by routing to per-tier sibling model ids instead and must
|
||||
* stay unmapped.
|
||||
*/
|
||||
function isGpt56PlusWireEffortModel<TApi extends Api>(spec: ModelSpec<TApi>): boolean {
|
||||
switch (spec.api) {
|
||||
case "openai-responses":
|
||||
case "openai-codex-responses":
|
||||
case "azure-openai-responses":
|
||||
case "openai-completions":
|
||||
case "openrouter":
|
||||
break;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
const parsed = parseOpenAIModel(bareModelId(spec.id));
|
||||
return parsed !== null && semverGte(parsed.version, "5.6");
|
||||
}
|
||||
|
||||
function getModelDefinedEfforts<TApi extends Api>(
|
||||
spec: ModelSpec<TApi>,
|
||||
compat: CompatOf<TApi>,
|
||||
@@ -313,6 +337,12 @@ function getModelDefinedEfforts<TApi extends Api>(
|
||||
if (isSakanaFuguReasoningModel(spec)) {
|
||||
return FUGU_REASONING_EFFORTS;
|
||||
}
|
||||
if (isGpt56PlusWireEffortModel(spec)) {
|
||||
// Normalize stale baked/discovered `low..xhigh` surfaces to the full
|
||||
// five-tier ladder so the shifted map keeps the native `low` tier
|
||||
// reachable (user `minimal`).
|
||||
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
||||
}
|
||||
return isOpenAICompatReasoningApi(spec.api) &&
|
||||
(isMinimaxM2FamilyModelId(spec.id) ||
|
||||
isOpenAIGptOssModelId(spec.id) ||
|
||||
@@ -373,7 +403,7 @@ function inferDetectedEffortMap<TApi extends Api>(
|
||||
return MINIMAX_ANTHROPIC_ADAPTIVE_EFFORT_MAP;
|
||||
}
|
||||
return anthropicModelHasRealXHighEffort(spec, parsedModel)
|
||||
? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER
|
||||
? SHIFTED_FIVE_TIER_EFFORT_MAP
|
||||
: ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER;
|
||||
}
|
||||
// GLM-5.2 coding SKUs accept `reasoning_effort`, but the effort dialect is
|
||||
@@ -397,6 +427,9 @@ function inferDetectedEffortMap<TApi extends Api>(
|
||||
if (isSakanaFuguReasoningModel(spec)) {
|
||||
return FUGU_REASONING_EFFORT_MAP;
|
||||
}
|
||||
if (isGpt56PlusWireEffortModel(spec)) {
|
||||
return SHIFTED_FIVE_TIER_EFFORT_MAP;
|
||||
}
|
||||
if (!isOpenAICompatReasoningApi(spec.api)) {
|
||||
return undefined;
|
||||
}
|
||||
@@ -446,7 +479,7 @@ function getOpenRouterAnthropicReasoningEffortMap(modelId: string): EffortMap |
|
||||
if (!isAnthropicAdaptiveGenAtLeast(parsed, "4.6")) return undefined;
|
||||
|
||||
const hasRealXHigh = isAnthropicAdaptiveGenAtLeast(parsed, "4.7");
|
||||
return hasRealXHigh ? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER;
|
||||
return hasRealXHigh ? SHIFTED_FIVE_TIER_EFFORT_MAP : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER;
|
||||
}
|
||||
|
||||
function inferSupportedEfforts<TApi extends Api>(
|
||||
@@ -474,6 +507,11 @@ function inferOpenAISupportedEfforts(model: OpenAIModel): readonly Effort[] {
|
||||
if (model.variant === "codex-mini" && semverEqual(model.version, "5.1")) {
|
||||
return GPT_5_1_CODEX_MINI_EFFORTS;
|
||||
}
|
||||
// 5.6+ exposes the full five-tier ladder: the shifted wire map spans
|
||||
// low..max, with user `minimal` reaching the native `low` tier.
|
||||
if (semverGte(model.version, "5.6")) {
|
||||
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
||||
}
|
||||
if (semverGte(model.version, "5.2")) {
|
||||
return GPT_5_2_PLUS_EFFORTS;
|
||||
}
|
||||
|
||||
+5825
-206
File diff suppressed because it is too large
Load Diff
@@ -29,6 +29,7 @@ import {
|
||||
mistralModelManagerOptions,
|
||||
moonshotModelManagerOptions,
|
||||
nanoGptModelManagerOptions,
|
||||
novitaModelManagerOptions,
|
||||
nvidiaModelManagerOptions,
|
||||
ollamaModelManagerOptions,
|
||||
openaiModelManagerOptions,
|
||||
@@ -271,6 +272,14 @@ export const CATALOG_PROVIDERS = [
|
||||
createModelManagerOptions: (config: ModelManagerConfig) => nvidiaModelManagerOptions(config),
|
||||
catalogDiscovery: { label: "NVIDIA" },
|
||||
},
|
||||
{
|
||||
id: "novita",
|
||||
defaultModel: "moonshotai/kimi-k2.7-code",
|
||||
envVars: ["NOVITA_API_KEY"],
|
||||
createModelManagerOptions: (config: ModelManagerConfig) => novitaModelManagerOptions(config),
|
||||
dynamicModelsAuthoritative: true,
|
||||
catalogDiscovery: { label: "Novita", allowUnauthenticated: true },
|
||||
},
|
||||
{
|
||||
id: "ollama",
|
||||
defaultModel: "gpt-oss:20b",
|
||||
|
||||
@@ -821,6 +821,62 @@ export function openaiModelManagerOptions(config?: OpenAIModelManagerConfig): Mo
|
||||
};
|
||||
}
|
||||
|
||||
/** First-party gpt-5.6 SKUs that accept `reasoning: { mode: "pro" }` on the Responses APIs. */
|
||||
const OPENAI_PRO_REASONING_BASE_IDS: Record<string, true> = {
|
||||
"gpt-5.6-luna": true,
|
||||
"gpt-5.6-sol": true,
|
||||
"gpt-5.6-terra": true,
|
||||
};
|
||||
const OPENAI_PRO_REASONING_PROVIDERS: Record<string, true> = { openai: true, "openai-codex": true };
|
||||
|
||||
/**
|
||||
* A row this generator pass owns: one of the derived `gpt-5.6-*-pro` alias ids
|
||||
* on `openai`/`openai-codex` that carries the generated `reasoningMode` marker.
|
||||
* A real upstream model occupying the same id has no `reasoningMode` and is
|
||||
* never touched.
|
||||
*/
|
||||
function isGeneratedOpenAIProReasoningAlias(model: ModelSpec<Api>): boolean {
|
||||
return (
|
||||
OPENAI_PRO_REASONING_PROVIDERS[model.provider] === true &&
|
||||
model.reasoningMode !== undefined &&
|
||||
model.id.endsWith("-pro") &&
|
||||
OPENAI_PRO_REASONING_BASE_IDS[model.id.slice(0, -"-pro".length)] === true
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-derive the generated pro-reasoning aliases (`gpt-5.6-*-pro`) for the
|
||||
* first-party `openai`/`openai-codex` gpt-5.6 rows. Each alias inherits the
|
||||
* base row's metadata, requests the base wire id via `requestModelId`, and
|
||||
* sets `reasoningMode: "pro"` so Responses-family request builders emit
|
||||
* `reasoning: { mode: "pro" }`. Called by the models.json generator after all
|
||||
* sources merge: stale copies of the owned aliases (previous snapshot) are
|
||||
* dropped and re-projected from the current base rows so alias metadata always
|
||||
* tracks the base, while a real upstream model that occupies an alias id wins
|
||||
* and suppresses the projection.
|
||||
*/
|
||||
export function projectOpenAIProReasoningAliases(models: readonly ModelSpec<Api>[]): ModelSpec<Api>[] {
|
||||
const kept = models.filter(model => !isGeneratedOpenAIProReasoningAlias(model));
|
||||
const ids = new Set(kept.map(model => `${model.provider}/${model.id}`));
|
||||
const out = [...kept];
|
||||
for (const model of kept) {
|
||||
if (!OPENAI_PRO_REASONING_PROVIDERS[model.provider]) continue;
|
||||
if (!OPENAI_PRO_REASONING_BASE_IDS[model.id]) continue;
|
||||
const aliasId = `${model.id}-pro`;
|
||||
const aliasKey = `${model.provider}/${aliasId}`;
|
||||
if (ids.has(aliasKey)) continue;
|
||||
ids.add(aliasKey);
|
||||
out.push({
|
||||
...model,
|
||||
id: aliasId,
|
||||
name: `${model.name} Pro`,
|
||||
requestModelId: model.id,
|
||||
reasoningMode: "pro",
|
||||
});
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 2. Groq
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -915,6 +971,102 @@ export function nvidiaModelManagerOptions(
|
||||
return createSimpleOpenAICompletionsOptions("nvidia", "https://integrate.api.nvidia.com/v1", config);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5.5 Novita
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** Novita OpenAI-compatible discovery configuration. */
|
||||
export interface NovitaModelManagerConfig {
|
||||
apiKey?: string;
|
||||
baseUrl?: string;
|
||||
fetch?: FetchImpl;
|
||||
}
|
||||
|
||||
function novitaArrayIncludes(value: unknown, expected: string): boolean {
|
||||
return Array.isArray(value) && value.some(item => item === expected);
|
||||
}
|
||||
|
||||
function isPublicNovitaModelId(id: string): boolean {
|
||||
return !id.toLowerCase().startsWith("ai_infer_test");
|
||||
}
|
||||
|
||||
// Novita reports token prices in 1/10,000 USD per million tokens.
|
||||
function toNovitaCostPerMillion(value: unknown): number {
|
||||
return toPositiveNumber(value, 0) / 10_000;
|
||||
}
|
||||
|
||||
function getNovitaCacheReadPricePerMillion(entry: OpenAICompatibleModelRecord): number {
|
||||
const pricing = entry.pricing;
|
||||
if (!isRecord(pricing)) {
|
||||
return 0;
|
||||
}
|
||||
const cacheRead = pricing.input_cache_read;
|
||||
if (!isRecord(cacheRead)) {
|
||||
return 0;
|
||||
}
|
||||
return toNovitaCostPerMillion(cacheRead.price_per_m);
|
||||
}
|
||||
|
||||
function mapNovitaModel(
|
||||
entry: OpenAICompatibleModelRecord,
|
||||
defaults: ModelSpec<"openai-completions">,
|
||||
reference: ModelSpec<"openai-completions"> | undefined,
|
||||
): ModelSpec<"openai-completions"> {
|
||||
const model = mapWithBundledReference(
|
||||
{
|
||||
...entry,
|
||||
name: entry.display_name ?? entry.title ?? entry.name,
|
||||
},
|
||||
defaults,
|
||||
reference,
|
||||
);
|
||||
return {
|
||||
...model,
|
||||
reasoning: novitaArrayIncludes(entry.features, "reasoning"),
|
||||
supportsTools: novitaArrayIncludes(entry.features, "function-calling"),
|
||||
input: toInputCapabilities(entry.input_modalities),
|
||||
cost: {
|
||||
input: toNovitaCostPerMillion(entry.input_token_price_per_m),
|
||||
output: toNovitaCostPerMillion(entry.output_token_price_per_m),
|
||||
cacheRead: getNovitaCacheReadPricePerMillion(entry),
|
||||
cacheWrite: 0,
|
||||
},
|
||||
contextWindow: toPositiveNumber(entry.context_size, model.contextWindow),
|
||||
maxTokens: toPositiveNumber(entry.max_output_tokens, model.maxTokens),
|
||||
};
|
||||
}
|
||||
|
||||
/** Builds Novita's public model-discovery manager. */
|
||||
export function novitaModelManagerOptions(
|
||||
config?: NovitaModelManagerConfig,
|
||||
): ModelManagerOptions<"openai-completions"> {
|
||||
const apiKey = config?.apiKey;
|
||||
const baseUrl = config?.baseUrl ?? "https://api.novita.ai/openai/v1";
|
||||
const references = createBundledReferenceMap<"openai-completions">("novita");
|
||||
return {
|
||||
providerId: "novita",
|
||||
dynamicModelsAuthoritative: true,
|
||||
fetchDynamicModels: async () =>
|
||||
fetchOpenAICompatibleModels({
|
||||
api: "openai-completions",
|
||||
provider: "novita",
|
||||
baseUrl,
|
||||
apiKey,
|
||||
mapModel: (entry, defaults) => mapNovitaModel(entry, defaults, references.get(defaults.id)),
|
||||
filterModel: (entry, model) => {
|
||||
const active = typeof entry.status !== "number" || entry.status === 1;
|
||||
return (
|
||||
active &&
|
||||
isPublicNovitaModelId(model.id) &&
|
||||
novitaArrayIncludes(entry.endpoints, "chat/completions") &&
|
||||
toPositiveNumber(entry.max_output_tokens, 0) > 0
|
||||
);
|
||||
},
|
||||
fetch: config?.fetch,
|
||||
}),
|
||||
};
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. xAI
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -983,6 +1135,7 @@ export const XAI_OAUTH_CURATED_MODELS: readonly XAICuratedModel[] = [
|
||||
input: ["text", "image"],
|
||||
},
|
||||
{ id: "grok-4.3", contextWindow: 1_000_000, name: "Grok 4.3", input: ["text", "image"] },
|
||||
{ id: "grok-4.5", contextWindow: 500_000, name: "Grok 4.5", input: ["text", "image"] },
|
||||
// grok-4.20-multi-agent-0309 is text-only per the bundled catalog; omit `input` for the default.
|
||||
{ id: "grok-4.20-multi-agent-0309", contextWindow: 2_000_000, name: "Grok 4.20 (Multi-Agent)" },
|
||||
{
|
||||
|
||||
@@ -691,6 +691,13 @@ export interface Model<TApi extends Api = Api> {
|
||||
* everything local (selection, caching, usage attribution) keys on `id`.
|
||||
*/
|
||||
requestModelId?: string;
|
||||
/**
|
||||
* `reasoning.mode` to send on OpenAI Responses-family requests. Set on
|
||||
* generated pro aliases (`gpt-5.6-*-pro` on `openai`/`openai-codex`) that
|
||||
* pair a base wire id (`requestModelId`) with OpenAI's pro reasoning
|
||||
* serving path. Absent everywhere else; providers omit the wire field.
|
||||
*/
|
||||
reasoningMode?: "pro";
|
||||
name: string;
|
||||
api: TApi;
|
||||
provider: Provider;
|
||||
@@ -751,6 +758,8 @@ export interface Model<TApi extends Api = Api> {
|
||||
transport?: "pi-native";
|
||||
/** Hint that websocket transport should be preferred when supported by the provider implementation. */
|
||||
preferWebsockets?: boolean;
|
||||
/** Codex Responses Lite transport: send the lite marker and carry instructions/tools as input items (mirrors codex-rs `use_responses_lite`). */
|
||||
useResponsesLite?: boolean;
|
||||
/** Preferred model to switch to when context promotion is triggered (model id or provider/id). */
|
||||
contextPromotionTarget?: string;
|
||||
/** Preferred model to use only for compaction (model id or provider/id); the active session model is unchanged. */
|
||||
|
||||
@@ -113,6 +113,14 @@ function thinkingPair(baseId: string, name: string): EffortVariantFamily {
|
||||
|
||||
type DevinTierRoutes = Partial<Record<"off" | "minimal" | "low" | "medium" | "high" | "xhigh", string>>;
|
||||
|
||||
const DEVIN_FIVE_TIER_EFFORTS: readonly Effort[] = [
|
||||
Effort.Minimal,
|
||||
Effort.Low,
|
||||
Effort.Medium,
|
||||
Effort.High,
|
||||
Effort.XHigh,
|
||||
];
|
||||
|
||||
function devinTierFamily(
|
||||
id: string,
|
||||
name: string,
|
||||
@@ -160,6 +168,44 @@ function devinTierFamily(
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* GPT-5.6 (Luna/Sol/Terra) adds a genuine `max` tier above `xhigh`, so the
|
||||
* standard family shifts every user effort up one notch (`minimal` → `-low`
|
||||
* … `xhigh` → `-max`), mirroring the Opus 4.7+ five-tier mapping. Devin
|
||||
* serves no `-max-priority` sibling, so the fast family keeps the direct
|
||||
* `low..xhigh` `-priority` scale.
|
||||
*/
|
||||
function devinGpt56Families(variant: "luna" | "sol" | "terra", name: string): readonly EffortVariantFamily[] {
|
||||
const base = `gpt-5-6-${variant}`;
|
||||
return [
|
||||
devinTierFamily(
|
||||
base,
|
||||
name,
|
||||
{
|
||||
off: `${base}-none`,
|
||||
minimal: `${base}-low`,
|
||||
low: `${base}-medium`,
|
||||
medium: `${base}-high`,
|
||||
high: `${base}-xhigh`,
|
||||
xhigh: `${base}-max`,
|
||||
},
|
||||
DEVIN_FIVE_TIER_EFFORTS,
|
||||
),
|
||||
devinTierFamily(
|
||||
`${base}-fast`,
|
||||
`${name} Fast`,
|
||||
{
|
||||
off: `${base}-none-priority`,
|
||||
low: `${base}-low-priority`,
|
||||
medium: `${base}-medium-priority`,
|
||||
high: `${base}-high-priority`,
|
||||
xhigh: `${base}-xhigh-priority`,
|
||||
},
|
||||
DEVIN_FIVE_TIER_EFFORTS,
|
||||
),
|
||||
];
|
||||
}
|
||||
|
||||
const GEMINI_3_FLASH_FAMILY_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High];
|
||||
const GEMINI_3_PRO_FAMILY_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High];
|
||||
|
||||
@@ -330,7 +376,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
|
||||
},
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
efforts: DEVIN_FIVE_TIER_EFFORTS,
|
||||
requiresEffort: true,
|
||||
},
|
||||
},
|
||||
@@ -353,7 +399,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
|
||||
},
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
efforts: DEVIN_FIVE_TIER_EFFORTS,
|
||||
requiresEffort: true,
|
||||
},
|
||||
},
|
||||
@@ -376,7 +422,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
|
||||
},
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
efforts: DEVIN_FIVE_TIER_EFFORTS,
|
||||
requiresEffort: true,
|
||||
},
|
||||
},
|
||||
@@ -399,7 +445,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
|
||||
},
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
efforts: DEVIN_FIVE_TIER_EFFORTS,
|
||||
requiresEffort: true,
|
||||
},
|
||||
},
|
||||
@@ -413,7 +459,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
|
||||
high: "MODEL_GPT_5_2_HIGH",
|
||||
xhigh: "MODEL_GPT_5_2_XHIGH",
|
||||
},
|
||||
[Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
DEVIN_FIVE_TIER_EFFORTS,
|
||||
),
|
||||
devinTierFamily(
|
||||
"gpt-5-3-codex",
|
||||
@@ -424,7 +470,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
|
||||
high: "gpt-5-3-codex-high",
|
||||
xhigh: "gpt-5-3-codex-xhigh",
|
||||
},
|
||||
[Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
DEVIN_FIVE_TIER_EFFORTS,
|
||||
),
|
||||
devinTierFamily(
|
||||
"gpt-5-3-codex-fast",
|
||||
@@ -435,7 +481,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
|
||||
high: "gpt-5-3-codex-high-priority",
|
||||
xhigh: "gpt-5-3-codex-xhigh-priority",
|
||||
},
|
||||
[Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
DEVIN_FIVE_TIER_EFFORTS,
|
||||
),
|
||||
devinTierFamily(
|
||||
"gpt-5-4",
|
||||
@@ -447,7 +493,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
|
||||
high: "gpt-5-4-high",
|
||||
xhigh: "gpt-5-4-xhigh",
|
||||
},
|
||||
[Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
DEVIN_FIVE_TIER_EFFORTS,
|
||||
),
|
||||
devinTierFamily(
|
||||
"gpt-5-4-fast",
|
||||
@@ -459,7 +505,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
|
||||
high: "gpt-5-4-high-priority",
|
||||
xhigh: "gpt-5-4-xhigh-priority",
|
||||
},
|
||||
[Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
DEVIN_FIVE_TIER_EFFORTS,
|
||||
),
|
||||
devinTierFamily(
|
||||
"gpt-5-4-mini",
|
||||
@@ -470,7 +516,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
|
||||
high: "gpt-5-4-mini-high",
|
||||
xhigh: "gpt-5-4-mini-xhigh",
|
||||
},
|
||||
[Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
DEVIN_FIVE_TIER_EFFORTS,
|
||||
),
|
||||
devinTierFamily(
|
||||
"gpt-5-5",
|
||||
@@ -482,7 +528,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
|
||||
high: "gpt-5-5-high",
|
||||
xhigh: "gpt-5-5-xhigh",
|
||||
},
|
||||
[Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
DEVIN_FIVE_TIER_EFFORTS,
|
||||
),
|
||||
devinTierFamily(
|
||||
"gpt-5-5-fast",
|
||||
@@ -494,8 +540,11 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
|
||||
high: "gpt-5-5-high-priority",
|
||||
xhigh: "gpt-5-5-xhigh-priority",
|
||||
},
|
||||
[Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
DEVIN_FIVE_TIER_EFFORTS,
|
||||
),
|
||||
...devinGpt56Families("luna", "GPT-5.6 Luna"),
|
||||
...devinGpt56Families("sol", "GPT-5.6 Sol"),
|
||||
...devinGpt56Families("terra", "GPT-5.6 Terra"),
|
||||
devinTierFamily(
|
||||
"gemini-3-1-pro",
|
||||
"Gemini 3.1 Pro",
|
||||
|
||||
@@ -10,6 +10,15 @@ export const OPENAI_HEADERS = {
|
||||
ORIGINATOR: "originator",
|
||||
SESSION_ID: "session_id",
|
||||
CONVERSATION_ID: "conversation_id",
|
||||
SCOPED_SESSION_ID: "session-id",
|
||||
THREAD_ID: "thread-id",
|
||||
INSTALLATION_ID: "x-codex-installation-id",
|
||||
WINDOW_ID: "x-codex-window-id",
|
||||
TURN_METADATA: "x-codex-turn-metadata",
|
||||
PARENT_THREAD_ID: "x-codex-parent-thread-id",
|
||||
SUBAGENT: "x-openai-subagent",
|
||||
/** Responses Lite transport marker (codex-rs `add_responses_lite_header`); value is always `"true"`. */
|
||||
RESPONSES_LITE: "x-openai-internal-codex-responses-lite",
|
||||
} as const;
|
||||
|
||||
export const OPENAI_HEADER_VALUES = {
|
||||
|
||||
@@ -52,6 +52,50 @@ describe("Codex model discovery", () => {
|
||||
});
|
||||
});
|
||||
|
||||
it("carries use_responses_lite and prefer_websockets onto the model spec", async () => {
|
||||
const fetchFn: typeof fetch = Object.assign(
|
||||
async () =>
|
||||
new Response(
|
||||
JSON.stringify({
|
||||
models: [
|
||||
{
|
||||
slug: "gpt-5.6-terra",
|
||||
display_name: "GPT-5.6-Terra",
|
||||
context_window: 372_000,
|
||||
default_reasoning_level: "medium",
|
||||
supported_reasoning_levels: ["low", "medium", "high"],
|
||||
input_modalities: ["text", "image"],
|
||||
supported_in_api: true,
|
||||
prefer_websockets: true,
|
||||
use_responses_lite: true,
|
||||
},
|
||||
{
|
||||
slug: "gpt-5.5",
|
||||
display_name: "GPT-5.5",
|
||||
context_window: 272_000,
|
||||
default_reasoning_level: "high",
|
||||
supported_reasoning_levels: ["low", "high"],
|
||||
input_modalities: ["text"],
|
||||
supported_in_api: true,
|
||||
},
|
||||
],
|
||||
}),
|
||||
),
|
||||
{ preconnect() {} },
|
||||
);
|
||||
const result = await fetchCodexModels({
|
||||
accessToken: "test-token",
|
||||
baseUrl: "https://codex.example/backend-api",
|
||||
clientVersion: "0.99.0",
|
||||
fetchFn,
|
||||
});
|
||||
|
||||
const terra = result?.models.find(model => model.id === "gpt-5.6-terra");
|
||||
expect(terra).toMatchObject({ preferWebsockets: true, useResponsesLite: true });
|
||||
const legacy = result?.models.find(model => model.id === "gpt-5.5");
|
||||
expect(legacy?.useResponsesLite).toBeUndefined();
|
||||
});
|
||||
|
||||
it("ignores pre-V2 Codex discovery cache rows", async () => {
|
||||
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-codex-v7-cache-"));
|
||||
const dbPath = path.join(tempDir, "models.db");
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
import { afterAll, beforeAll, describe, expect, it } from "bun:test";
|
||||
import * as http2 from "node:http2";
|
||||
import { create, toBinary } from "@bufbuild/protobuf";
|
||||
// Import from source, not the package specifier: the workspace `node_modules`
|
||||
// copy resolves to the primary checkout, not this worktree.
|
||||
import { fetchCursorUsableModels } from "../src/discovery/cursor";
|
||||
import { GetUsableModelsResponseSchema, ModelDetailsSchema } from "../src/discovery/cursor-gen/agent_pb";
|
||||
import type { ModelSpec } from "../src/types";
|
||||
|
||||
const FIXTURE_MODEL_IDS = [
|
||||
// Reference-less ids from families whose native catalogs are multimodal.
|
||||
"claude-opus-4-8-99999999",
|
||||
"gpt-5.5-codex-20991231",
|
||||
"gemini-4-pro-exp",
|
||||
// Reference-less ids from text-only families.
|
||||
"composer-3",
|
||||
"grok-code-fast-2",
|
||||
// Bundled-reference ids: the reference stays authoritative.
|
||||
"claude-4.5-opus-high",
|
||||
"claude-4.6-opus-high",
|
||||
"composer-1",
|
||||
];
|
||||
|
||||
let server: http2.Http2Server;
|
||||
let baseUrl: string;
|
||||
|
||||
beforeAll(async () => {
|
||||
const response = create(GetUsableModelsResponseSchema, {
|
||||
models: FIXTURE_MODEL_IDS.map(modelId => create(ModelDetailsSchema, { modelId })),
|
||||
});
|
||||
const payload = Buffer.from(toBinary(GetUsableModelsResponseSchema, response));
|
||||
|
||||
server = http2.createServer();
|
||||
server.on("stream", (stream: http2.ServerHttp2Stream, headers: http2.IncomingHttpHeaders) => {
|
||||
stream.on("data", () => {});
|
||||
stream.on("end", () => {
|
||||
if (headers[":path"] !== "/agent.v1.AgentService/GetUsableModels") {
|
||||
stream.respond({ ":status": 404 });
|
||||
stream.end();
|
||||
return;
|
||||
}
|
||||
stream.respond({ ":status": 200, "content-type": "application/proto" });
|
||||
stream.end(payload);
|
||||
});
|
||||
});
|
||||
await new Promise<void>(resolve => server.listen(0, "127.0.0.1", resolve));
|
||||
const address = server.address();
|
||||
if (!address || typeof address === "string") {
|
||||
throw new Error("expected http2 fixture server to bind a tcp port");
|
||||
}
|
||||
baseUrl = `http://127.0.0.1:${address.port}`;
|
||||
});
|
||||
|
||||
afterAll(() => {
|
||||
server?.close();
|
||||
});
|
||||
|
||||
async function discover(): Promise<Map<string, ModelSpec<"cursor-agent">>> {
|
||||
const models = await fetchCursorUsableModels({ apiKey: "test-key", baseUrl });
|
||||
expect(models).not.toBeNull();
|
||||
return new Map((models ?? []).map(model => [model.id, model]));
|
||||
}
|
||||
|
||||
describe("cursor discovery input modalities (issue #4726)", () => {
|
||||
it("classifies reference-less multimodal-family models as text+image", async () => {
|
||||
const byId = await discover();
|
||||
expect(byId.get("claude-opus-4-8-99999999")?.input).toEqual(["text", "image"]);
|
||||
expect(byId.get("gpt-5.5-codex-20991231")?.input).toEqual(["text", "image"]);
|
||||
expect(byId.get("gemini-4-pro-exp")?.input).toEqual(["text", "image"]);
|
||||
});
|
||||
|
||||
it("keeps reference-less text-only families text-only", async () => {
|
||||
const byId = await discover();
|
||||
expect(byId.get("composer-3")?.input).toEqual(["text"]);
|
||||
expect(byId.get("grok-code-fast-2")?.input).toEqual(["text"]);
|
||||
});
|
||||
|
||||
it("keeps bundled references authoritative for input modalities", async () => {
|
||||
const byId = await discover();
|
||||
// Bundled cursor references carry their own input classification; the
|
||||
// id-based inference must not override it in either direction.
|
||||
expect(byId.get("claude-4.5-opus-high")?.input).toEqual(["text", "image"]);
|
||||
expect(byId.get("claude-4.6-opus-high")?.input).toEqual(["text"]);
|
||||
expect(byId.get("composer-1")?.input).toEqual(["text"]);
|
||||
});
|
||||
|
||||
it("preserves fallback defaults for reference-less models", async () => {
|
||||
const byId = await discover();
|
||||
const spec = byId.get("claude-opus-4-8-99999999");
|
||||
expect(spec?.provider).toBe("cursor");
|
||||
expect(spec?.api).toBe("cursor-agent");
|
||||
expect(spec?.contextWindow).toBe(200_000);
|
||||
expect(spec?.maxTokens).toBe(64_000);
|
||||
expect(spec?.cost).toEqual({ input: 0, output: 0, cacheRead: 0, cacheWrite: 0 });
|
||||
});
|
||||
});
|
||||
@@ -265,6 +265,7 @@ describe("isGrokReasoningEffortCapable", () => {
|
||||
expect(isGrokReasoningEffortCapable("grok-3-mini")).toBe(true);
|
||||
expect(isGrokReasoningEffortCapable("grok-4.20-multi-agent")).toBe(true);
|
||||
expect(isGrokReasoningEffortCapable("xai-oauth/grok-4.3")).toBe(true);
|
||||
expect(isGrokReasoningEffortCapable("xai-oauth/grok-4.5")).toBe(true);
|
||||
expect(isGrokReasoningEffortCapable("openrouter/xai/grok-3-mini")).toBe(true);
|
||||
});
|
||||
|
||||
|
||||
@@ -567,6 +567,90 @@ describe("model thinking derivation", () => {
|
||||
expect(getSupportedEfforts(model)).toEqual([]);
|
||||
expect(clampThinkingLevelForModel(model, Effort.High)).toBeUndefined();
|
||||
});
|
||||
|
||||
it("bakes the GPT-5.6 shifted five-tier effort map on wire-effort APIs", () => {
|
||||
const codex = createModel({
|
||||
id: "gpt-5.6-sol",
|
||||
api: "openai-codex-responses",
|
||||
provider: "openai-codex",
|
||||
});
|
||||
|
||||
expect(codex.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
effortMap: {
|
||||
minimal: "low",
|
||||
low: "medium",
|
||||
medium: "high",
|
||||
high: "xhigh",
|
||||
xhigh: "max",
|
||||
},
|
||||
});
|
||||
|
||||
// Stale baked four-tier metadata (caches/discovery) normalizes back to
|
||||
// the five-tier ladder with the map attached — the wire-defaults
|
||||
// backfill path — and namespaced OpenRouter ids parse.
|
||||
const staleOpenRouter = createModel({
|
||||
id: "openai/gpt-5.6-terra",
|
||||
api: "openrouter",
|
||||
provider: "openrouter",
|
||||
baseUrl: "https://openrouter.ai/api/v1",
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
});
|
||||
|
||||
expect(staleOpenRouter.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
effortMap: {
|
||||
minimal: "low",
|
||||
low: "medium",
|
||||
medium: "high",
|
||||
high: "xhigh",
|
||||
xhigh: "max",
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("keeps pre-5.6 and Devin-routed GPT models off the shifted effort map", () => {
|
||||
const gpt55 = createModel({
|
||||
id: "gpt-5.5",
|
||||
api: "openai-responses",
|
||||
provider: "openai",
|
||||
});
|
||||
|
||||
expect(gpt55.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
});
|
||||
expect(gpt55.thinking?.effortMap).toBeUndefined();
|
||||
|
||||
// Devin selects effort by routing to per-tier sibling model ids, never
|
||||
// via a wire reasoning.effort field — the shifted map must not attach.
|
||||
const devin = createModel({
|
||||
id: "gpt-5-6-sol",
|
||||
api: "devin-agent",
|
||||
provider: "devin",
|
||||
baseUrl: "https://server.codeium.com",
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
effortRouting: {
|
||||
off: "gpt-5-6-sol-none",
|
||||
minimal: "gpt-5-6-sol-low",
|
||||
low: "gpt-5-6-sol-medium",
|
||||
medium: "gpt-5-6-sol-high",
|
||||
high: "gpt-5-6-sol-xhigh",
|
||||
xhigh: "gpt-5-6-sol-max",
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(devin.thinking?.effortMap).toBeUndefined();
|
||||
expect(devin.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("model thinking runtime helpers", () => {
|
||||
|
||||
@@ -0,0 +1,94 @@
|
||||
import { describe, expect, test } from "bun:test";
|
||||
import { DEFAULT_MODEL_PER_PROVIDER, PROVIDER_DESCRIPTORS } from "@oh-my-pi/pi-catalog/provider-models/descriptors";
|
||||
import { novitaModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
|
||||
|
||||
describe("Novita built-in provider", () => {
|
||||
test("registers catalog descriptor with NOVITA_API_KEY env discovery", () => {
|
||||
const descriptor = PROVIDER_DESCRIPTORS.find(item => item.providerId === "novita");
|
||||
expect(descriptor).toBeDefined();
|
||||
expect(descriptor?.defaultModel).toBe("moonshotai/kimi-k2.7-code");
|
||||
expect(descriptor?.catalogDiscovery?.envVars).toContain("NOVITA_API_KEY");
|
||||
expect(descriptor?.catalogDiscovery?.allowUnauthenticated).toBe(true);
|
||||
expect(descriptor?.dynamicModelsAuthoritative).toBe(true);
|
||||
expect(DEFAULT_MODEL_PER_PROVIDER.novita).toBe("moonshotai/kimi-k2.7-code");
|
||||
});
|
||||
|
||||
test("maps Novita model catalog metadata from the public OpenAI-compatible endpoint", async () => {
|
||||
const requests: string[] = [];
|
||||
const fetchMock = async (input: string | URL | Request): Promise<Response> => {
|
||||
requests.push(input.toString());
|
||||
return Response.json({
|
||||
data: [
|
||||
{
|
||||
id: "moonshotai/kimi-k2.7-code",
|
||||
display_name: "Kimi K2.7 Code",
|
||||
status: 1,
|
||||
context_size: 262144,
|
||||
max_output_tokens: 131072,
|
||||
input_token_price_per_m: 9500,
|
||||
output_token_price_per_m: 40000,
|
||||
pricing: {
|
||||
input_cache_read: {
|
||||
price_per_m: 1900,
|
||||
},
|
||||
},
|
||||
features: ["serverless", "function-calling", "structured-outputs", "reasoning"],
|
||||
endpoints: ["chat/completions", "anthropic"],
|
||||
input_modalities: ["text", "image", "video"],
|
||||
},
|
||||
{
|
||||
id: "qwen/qwen3-8b-fp8",
|
||||
status: 4,
|
||||
context_size: 128000,
|
||||
max_output_tokens: 20000,
|
||||
endpoints: ["chat/completions"],
|
||||
input_modalities: ["text"],
|
||||
},
|
||||
{
|
||||
id: "ai_infer_test_1",
|
||||
status: 1,
|
||||
context_size: 200000,
|
||||
max_output_tokens: 200000,
|
||||
features: ["function-calling"],
|
||||
endpoints: ["chat/completions"],
|
||||
input_modalities: ["text"],
|
||||
},
|
||||
{
|
||||
id: "minimax/m2-her",
|
||||
status: 1,
|
||||
context_size: 32000,
|
||||
features: ["serverless"],
|
||||
endpoints: ["chat/completions"],
|
||||
input_modalities: ["text"],
|
||||
},
|
||||
{
|
||||
id: "test/zero-output",
|
||||
status: 1,
|
||||
context_size: 32000,
|
||||
max_output_tokens: 0,
|
||||
features: ["serverless"],
|
||||
endpoints: ["chat/completions"],
|
||||
input_modalities: ["text"],
|
||||
},
|
||||
],
|
||||
});
|
||||
};
|
||||
|
||||
const options = novitaModelManagerOptions({ fetch: fetchMock });
|
||||
const models = await options.fetchDynamicModels?.();
|
||||
const model = models?.find(item => item.id === "moonshotai/kimi-k2.7-code");
|
||||
|
||||
expect(requests).toEqual(["https://api.novita.ai/openai/v1/models"]);
|
||||
expect(options.dynamicModelsAuthoritative).toBe(true);
|
||||
expect(models?.map(item => item.id)).toEqual(["moonshotai/kimi-k2.7-code"]);
|
||||
expect(model?.provider).toBe("novita");
|
||||
expect(model?.baseUrl).toBe("https://api.novita.ai/openai/v1");
|
||||
expect(model?.name).toBe("Kimi K2.7 Code");
|
||||
expect(model?.reasoning).toBe(true);
|
||||
expect(model?.supportsTools).toBe(true);
|
||||
expect(model?.input).toEqual(["text", "image"]);
|
||||
expect(model?.cost).toEqual({ input: 0.95, output: 4, cacheRead: 0.19, cacheWrite: 0 });
|
||||
expect(model?.contextWindow).toBe(262144);
|
||||
expect(model?.maxTokens).toBe(131072);
|
||||
});
|
||||
});
|
||||
@@ -2,6 +2,44 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed compaction aborting instead of trying an authenticated fallback model when Amazon Bedrock credential resolution fails before a request is sent. ([#5030](https://github.com/can1357/oh-my-pi/pull/5030) by [@usr-bin-roygbiv](https://github.com/usr-bin-roygbiv))
|
||||
- Fixed full-context forks cold-missing OpenAI prompt caches by persisting an inherited provider prompt-cache key separately from the new OMP session id, adding `--prompt-cache-key` for explicit cache affinity, and dropping automatic inheritance when startup changes the model, thinking level, system prompt, or tool schema. ([#5035](https://github.com/can1357/oh-my-pi/issues/5035))
|
||||
- Fixed Codex advisor requests using local `-advisor` session labels as provider session IDs; advisors now use stable UUIDv7 provider identities while keeping labeled transcript names. ([#5040](https://github.com/can1357/oh-my-pi/issues/5040))
|
||||
|
||||
## [16.3.15] - 2026-07-09
|
||||
|
||||
### Changed
|
||||
|
||||
- Integrated testing guidance directly into the main system prompt for improved workflow cohesion
|
||||
- Moved testing guidance into the main system prompt and removed the bundled Tester subagent.
|
||||
|
||||
## [16.3.14] - 2026-07-09
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed issue where unfinalized tool blocks could incorrectly pin the live-region scroll seam
|
||||
- Improved rendering of raw thinking blocks by stripping empty HTML comment noise
|
||||
- Fixed display of thinking blocks consisting entirely of hidden comment noise
|
||||
- Fixed gpt-5.6 reasoning summaries rendering literal `<!-- -->` sentinel lines in thinking blocks; empty HTML comments (and the unterminated `<!--` tail while streaming) are now dropped from the thinking display, and blocks reduced to pure comment noise are hidden entirely.
|
||||
|
||||
## [16.3.13] - 2026-07-09
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed `read` and `grep` treating empty optional `selector` fields emitted by models as invalid selectors instead of behaving like omitted selectors. ([#4879](https://github.com/can1357/oh-my-pi/issues/4879))
|
||||
- Fixed `grep` explicit line selectors on directory searches so they filter each matched file by line number instead of aborting with a single-file-only error ([#4898](https://github.com/can1357/oh-my-pi/issues/4898)).
|
||||
- Fixed Read tool previews dropping explicit `selector` arguments, so line ranges and `raw` modifiers render in terminal read call titles again ([#4899](https://github.com/can1357/oh-my-pi/issues/4899)).
|
||||
- Fixed named profiles dropping default user keybindings from `~/.omp/agent/keybindings.*`; profile keybindings now inherit those defaults and override only the keys they define ([#4867](https://github.com/can1357/oh-my-pi/issues/4867)).
|
||||
- Fixed pi extensions calling `ctx.ui.addAutocompleteProvider(...)` crashing at load with `TypeError: ... is not a function` — a failure that, for extensions guarding init in one `try/catch` (e.g. `@ff-labs/pi-fff`), aborted the extension's entire initialization. `ExtensionUIContext` now implements pi's autocomplete-provider API: interactive mode stacks each registered factory on top of the built-in editor provider (re-applied on every slash-command refresh, with throwing or malformed factories skipped), while RPC/ACP/headless contexts accept the factory as a no-op. ([#4919](https://github.com/can1357/oh-my-pi/issues/4919))
|
||||
- Fixed bundled reviewer and plan subagents to inherit their model roles' explicit thinking effort suffixes instead of pinning `high` ([#4761](https://github.com/can1357/oh-my-pi/issues/4761)).
|
||||
- Built-in provider model discovery now refreshes an expired stored OAuth credential before an online refresh needs it, instead of silently skipping the provider. The refresh is scoped to the providers actually being discovered (`refreshProvider` cannot rotate unrelated credentials), fires under `online-if-uncached` only when the model manager will actually fetch, and offline discovery stays peek-only ([#4893](https://github.com/can1357/oh-my-pi/issues/4893)).
|
||||
- Fixed Escape during an active TUI prompt requiring a second press before canceling; the first Escape now aborts the streaming turn immediately. ([#4921](https://github.com/can1357/oh-my-pi/issues/4921))
|
||||
- Fixed the streamed `write` tool's collapsed pending tail preview leaving stale rows above the first partial-result frame in the TUI; the first result now replays the viewport like the SSH placeholder seam already did ([#4477](https://github.com/can1357/oh-my-pi/issues/4477))
|
||||
- Fixed first-run setup ignoring a pre-seeded `config.yaml`: the settings loader now treats `config.yml` and `config.yaml` as equivalent existing main config files, writes back to the existing extension, and only creates canonical `config.yml` for fresh installs. ([#4914](https://github.com/can1357/oh-my-pi/issues/4914))
|
||||
- Fixed extension `sendUserMessage()` without `deliverAs` surfacing `AgentBusyError` during active streams; omitted `deliverAs` now queues a steer through the normal prompt flow, and ACP/RPC skill-command prompts queue while streaming (RPC honors the prompt command's `streamingBehavior`, defaulting to steer) ([#4923](https://github.com/can1357/oh-my-pi/issues/4923)).
|
||||
|
||||
## [16.3.12] - 2026-07-08
|
||||
|
||||
### Added
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"type": "module",
|
||||
"name": "@oh-my-pi/pi-coding-agent",
|
||||
"version": "16.3.12",
|
||||
"version": "16.3.15",
|
||||
"description": "Coding agent CLI with read, bash, edit, write tools and session management",
|
||||
"homepage": "https://omp.sh",
|
||||
"author": "Can Boluk",
|
||||
|
||||
@@ -5,6 +5,7 @@ import * as path from "node:path";
|
||||
import {
|
||||
advisorConfigFilePath,
|
||||
discoverAdvisorConfigs,
|
||||
getOrCreateAdvisorProviderSessionId,
|
||||
loadWatchdogConfigFile,
|
||||
resolveAdvisorConfigEditPath,
|
||||
saveWatchdogConfigFile,
|
||||
@@ -86,6 +87,89 @@ describe("slugifyAdvisorName", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("getOrCreateAdvisorProviderSessionId", () => {
|
||||
const primarySessionA = "018f8f5d-75b0-7cc6-8a6f-2f1c0b8e4c9d";
|
||||
const primarySessionB = "018f8f5d-75b1-7cc6-8a6f-2f1c0b8e4c9d";
|
||||
|
||||
it("returns the generated UUIDv7 instead of a local advisor label", () => {
|
||||
const generated = "0193c8f2-7b1a-7c4d-9e2f-123456789abc";
|
||||
|
||||
const providerSessionId = getOrCreateAdvisorProviderSessionId(
|
||||
new Map<string, string>(),
|
||||
primarySessionA,
|
||||
"security-advisor",
|
||||
() => generated,
|
||||
);
|
||||
|
||||
expect(providerSessionId).toBe(generated);
|
||||
expect(providerSessionId).not.toContain("-advisor");
|
||||
});
|
||||
|
||||
it("reuses the same generated UUIDv7 for repeated calls with the same primary session and slug", () => {
|
||||
const generatedIds = ["0193c8f2-7b1a-7c4d-9e2f-123456789abc", "0193c8f2-7b1b-7c4d-9e2f-123456789abc"];
|
||||
let nextGeneratedIdIndex = 0;
|
||||
const ids = new Map<string, string>();
|
||||
|
||||
const first = getOrCreateAdvisorProviderSessionId(ids, primarySessionA, "architecture", () => {
|
||||
const generated = generatedIds[nextGeneratedIdIndex];
|
||||
if (!generated) throw new Error("unexpected generator call");
|
||||
nextGeneratedIdIndex += 1;
|
||||
return generated;
|
||||
});
|
||||
const second = getOrCreateAdvisorProviderSessionId(ids, primarySessionA, "architecture", () => {
|
||||
const generated = generatedIds[nextGeneratedIdIndex];
|
||||
if (!generated) throw new Error("unexpected generator call");
|
||||
nextGeneratedIdIndex += 1;
|
||||
return generated;
|
||||
});
|
||||
|
||||
expect(first).toBe(generatedIds[0]);
|
||||
expect(second).toBe(generatedIds[0]);
|
||||
expect(nextGeneratedIdIndex).toBe(1);
|
||||
});
|
||||
|
||||
it("creates distinct UUIDv7 values for different advisor slugs or primary sessions", () => {
|
||||
const generatedIds = [
|
||||
"0193c8f2-7b1a-7c4d-9e2f-123456789abc",
|
||||
"0193c8f2-7b1b-7c4d-9e2f-123456789abc",
|
||||
"0193c8f2-7b1c-7c4d-9e2f-123456789abc",
|
||||
];
|
||||
let nextGeneratedIdIndex = 0;
|
||||
const ids = new Map<string, string>();
|
||||
const nextGeneratedId = () => {
|
||||
const generated = generatedIds[nextGeneratedIdIndex];
|
||||
if (!generated) throw new Error("unexpected generator call");
|
||||
nextGeneratedIdIndex += 1;
|
||||
return generated;
|
||||
};
|
||||
|
||||
const architecture = getOrCreateAdvisorProviderSessionId(ids, primarySessionA, "architecture", nextGeneratedId);
|
||||
const security = getOrCreateAdvisorProviderSessionId(ids, primarySessionA, "security", nextGeneratedId);
|
||||
const architectureForOtherSession = getOrCreateAdvisorProviderSessionId(
|
||||
ids,
|
||||
primarySessionB,
|
||||
"architecture",
|
||||
nextGeneratedId,
|
||||
);
|
||||
|
||||
expect(architecture).toBe(generatedIds[0]);
|
||||
expect(security).toBe(generatedIds[1]);
|
||||
expect(architectureForOtherSession).toBe(generatedIds[2]);
|
||||
expect(new Set([architecture, security, architectureForOtherSession]).size).toBe(3);
|
||||
});
|
||||
|
||||
it("rejects generated values that are not UUIDv7", () => {
|
||||
expect(() =>
|
||||
getOrCreateAdvisorProviderSessionId(
|
||||
new Map<string, string>(),
|
||||
primarySessionA,
|
||||
"architecture",
|
||||
() => "550e8400-e29b-41d4-a716-446655440000",
|
||||
),
|
||||
).toThrow("non-UUIDv7");
|
||||
});
|
||||
});
|
||||
|
||||
describe("WATCHDOG.yml file round-trip", () => {
|
||||
let tmp: string;
|
||||
beforeEach(async () => {
|
||||
|
||||
@@ -59,6 +59,34 @@ export function slugifyAdvisorName(name: string): string {
|
||||
return slug || "advisor";
|
||||
}
|
||||
|
||||
const UUID_V7_PATTERN = /^[0-9a-f]{8}-[0-9a-f]{4}-7[0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i;
|
||||
const ADVISOR_PROVIDER_SESSION_KEY_SEPARATOR = "\u0000";
|
||||
|
||||
/**
|
||||
* Returns a stable provider-facing UUIDv7 for one advisor within one primary session.
|
||||
*
|
||||
* Codex treats `session_id`/`conversation_id` as a UUID-shaped routing identity,
|
||||
* so advisor labels such as `-advisor` stay local-only.
|
||||
*/
|
||||
export function getOrCreateAdvisorProviderSessionId(
|
||||
ids: Map<string, string>,
|
||||
primarySessionId: string | undefined,
|
||||
slug: string,
|
||||
randomSessionId: () => string = () => Bun.randomUUIDv7(),
|
||||
): string | undefined {
|
||||
if (!primarySessionId) return undefined;
|
||||
const key = `${primarySessionId}${ADVISOR_PROVIDER_SESSION_KEY_SEPARATOR}${slug}`;
|
||||
const existing = ids.get(key);
|
||||
if (existing) return existing;
|
||||
|
||||
const next = randomSessionId();
|
||||
if (!UUID_V7_PATTERN.test(next)) {
|
||||
throw new Error("Advisor provider session id generator returned a non-UUIDv7 value");
|
||||
}
|
||||
ids.set(key, next);
|
||||
return next;
|
||||
}
|
||||
|
||||
/** Built tool names, for validating an advisor's `tools` list. */
|
||||
const KNOWN_TOOL_NAMES = new Set<string>(BUILTIN_TOOL_NAMES);
|
||||
|
||||
|
||||
@@ -42,6 +42,7 @@ export interface Args {
|
||||
noSession?: boolean;
|
||||
sessionDir?: string;
|
||||
providerSessionId?: string;
|
||||
providerPromptCacheKey?: string;
|
||||
fork?: string;
|
||||
/** Collab link to join at startup (set by the `join` subcommand; no CLI flag). */
|
||||
join?: string;
|
||||
|
||||
@@ -141,6 +141,9 @@ export const STRING_SETTERS: Record<string, StringSetter> = {
|
||||
"--provider-session-id": (result, value) => {
|
||||
result.providerSessionId = value;
|
||||
},
|
||||
"--prompt-cache-key": (result, value) => {
|
||||
result.providerPromptCacheKey = value;
|
||||
},
|
||||
"--session-dir": (result, value) => {
|
||||
result.sessionDir = value;
|
||||
},
|
||||
|
||||
@@ -9,7 +9,7 @@ import {
|
||||
TUI_KEYBINDINGS,
|
||||
KeybindingsManager as TuiKeybindingsManager,
|
||||
} from "@oh-my-pi/pi-tui";
|
||||
import { getAgentDir, isEnoent, logger } from "@oh-my-pi/pi-utils";
|
||||
import { getActiveProfile, getAgentDir, getProfileRootDir, isEnoent, logger } from "@oh-my-pi/pi-utils";
|
||||
import { JSONC, YAML } from "bun";
|
||||
|
||||
/**
|
||||
@@ -375,6 +375,12 @@ interface KeybindingsConfigPaths {
|
||||
writeBackPath: string;
|
||||
}
|
||||
|
||||
/** Controls inherited keybinding lookup when creating a manager for a named profile. */
|
||||
export interface KeybindingsCreateOptions {
|
||||
/** Default-profile agent directory whose keybindings are merged before profile-specific bindings. */
|
||||
inheritedAgentDir?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Load raw config from a file synchronously.
|
||||
* Returns parsed JSON/YAML or null if file doesn't exist or is invalid.
|
||||
@@ -428,6 +434,48 @@ function resolveKeybindingsConfigPaths(agentDir: string): KeybindingsConfigPaths
|
||||
return { readPath: ymlPath, writeBackPath: ymlPath };
|
||||
}
|
||||
|
||||
function mergeKeybindingsConfig(
|
||||
inheritedConfig: KeybindingsConfig,
|
||||
profileConfig: KeybindingsConfig,
|
||||
): KeybindingsConfig {
|
||||
return { ...inheritedConfig, ...profileConfig };
|
||||
}
|
||||
|
||||
function resolveInheritedAgentDir(agentDir: string, options: KeybindingsCreateOptions): string | undefined {
|
||||
const inheritedAgentDir =
|
||||
options.inheritedAgentDir ?? (getActiveProfile() ? path.join(getProfileRootDir(undefined), "agent") : undefined);
|
||||
if (!inheritedAgentDir) return undefined;
|
||||
if (path.resolve(inheritedAgentDir) === path.resolve(agentDir)) return undefined;
|
||||
return inheritedAgentDir;
|
||||
}
|
||||
|
||||
function loadMergedKeybindingsConfig(
|
||||
agentDir: string,
|
||||
options: KeybindingsCreateOptions,
|
||||
): {
|
||||
config: KeybindingsConfig;
|
||||
profilePath: string;
|
||||
inheritedPath: string | undefined;
|
||||
} {
|
||||
const profilePaths = resolveKeybindingsConfigPaths(agentDir);
|
||||
const profile = loadKeybindingsConfig(profilePaths.readPath, profilePaths.writeBackPath);
|
||||
const inheritedAgentDir = resolveInheritedAgentDir(agentDir, options);
|
||||
if (!inheritedAgentDir) {
|
||||
return { config: profile.config, profilePath: profile.persistedPath, inheritedPath: undefined };
|
||||
}
|
||||
|
||||
const inheritedPaths = resolveKeybindingsConfigPaths(inheritedAgentDir);
|
||||
// Read-only: a named-profile process must never write migration output into
|
||||
// the default profile's agent dir. Name migration still applies in-memory;
|
||||
// the on-disk migration happens when the default profile itself launches.
|
||||
const inherited = loadKeybindingsConfig(inheritedPaths.readPath, undefined);
|
||||
return {
|
||||
config: mergeKeybindingsConfig(inherited.config, profile.config),
|
||||
profilePath: profile.persistedPath,
|
||||
inheritedPath: inherited.persistedPath,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Load and migrate keybindings config.
|
||||
* Legacy JSON is read for compatibility, but successful write-back goes to YAML.
|
||||
@@ -499,22 +547,23 @@ function keyConfigValue(keys: KeyId[]): KeyId | KeyId[] {
|
||||
*/
|
||||
export class KeybindingsManager extends TuiKeybindingsManager {
|
||||
#configPath: string | undefined;
|
||||
#inheritedConfigPath: string | undefined;
|
||||
#userBindings: KeybindingsConfig;
|
||||
|
||||
constructor(userBindings: KeybindingsConfig = {}, configPath?: string) {
|
||||
constructor(userBindings: KeybindingsConfig = {}, configPath?: string, inheritedConfigPath?: string) {
|
||||
super(KEYBINDINGS, userBindings);
|
||||
this.#configPath = configPath;
|
||||
this.#inheritedConfigPath = inheritedConfigPath;
|
||||
this.#userBindings = userBindings;
|
||||
}
|
||||
|
||||
/**
|
||||
* Create from config file at agentDir/keybindings.yml.
|
||||
* Create from config files at agentDir/keybindings.yml and the default profile.
|
||||
* Legacy keybindings.json is migrated to keybindings.yml on load.
|
||||
*/
|
||||
static create(agentDir: string = getAgentDir()): KeybindingsManager {
|
||||
const { readPath, writeBackPath } = resolveKeybindingsConfigPaths(agentDir);
|
||||
const { config: userBindings, persistedPath } = KeybindingsManager.#loadFromFile(readPath, writeBackPath);
|
||||
const manager = new KeybindingsManager(userBindings, persistedPath);
|
||||
static create(agentDir: string = getAgentDir(), options: KeybindingsCreateOptions = {}): KeybindingsManager {
|
||||
const { config: userBindings, profilePath, inheritedPath } = loadMergedKeybindingsConfig(agentDir, options);
|
||||
const manager = new KeybindingsManager(userBindings, profilePath, inheritedPath);
|
||||
// Set globally so getKeybindings() returns this manager
|
||||
setKeybindings(manager);
|
||||
return manager;
|
||||
@@ -528,12 +577,15 @@ export class KeybindingsManager extends TuiKeybindingsManager {
|
||||
}
|
||||
|
||||
/**
|
||||
* Reload keybindings from the config file.
|
||||
* Reload keybindings from the config files.
|
||||
*/
|
||||
reload(): void {
|
||||
if (!this.#configPath) return;
|
||||
const { config } = KeybindingsManager.#loadFromFile(this.#configPath);
|
||||
this.setUserBindings(config);
|
||||
const { config: inheritedConfig } = this.#inheritedConfigPath
|
||||
? KeybindingsManager.#loadFromFile(this.#inheritedConfigPath)
|
||||
: { config: {} };
|
||||
const { config: profileConfig } = KeybindingsManager.#loadFromFile(this.#configPath);
|
||||
this.setUserBindings(mergeKeybindingsConfig(inheritedConfig, profileConfig));
|
||||
}
|
||||
|
||||
setUserBindings(userBindings: KeybindingsConfig): void {
|
||||
|
||||
@@ -54,6 +54,12 @@ const LOCAL_PROVIDER_PLACEHOLDERS = new Set<string>(["llama-cpp-local", "lm-stud
|
||||
* so a successful fast path does not leave an armed timeout signal for concurrent GC.
|
||||
*/
|
||||
const RUNTIME_DYNAMIC_MODEL_FETCH_TIMEOUT_MS = 15_000;
|
||||
// Built-in discovery preflight mirror of the catalog model-manager's private
|
||||
// cache timings (model-manager.ts: DEFAULT_CACHE_TTL_MS / NON_AUTHORITATIVE_RETRY_MS).
|
||||
// Built-in descriptors never override cacheTtlMs, so agreeing with these values
|
||||
// makes the OAuth-refresh preflight fire exactly when the manager will fetch.
|
||||
const BUILT_IN_DISCOVERY_CACHE_TTL_MS = 2 * 60 * 60 * 1000;
|
||||
const BUILT_IN_DISCOVERY_NON_AUTHORITATIVE_RETRY_MS = 5 * 60 * 1000;
|
||||
|
||||
import type { ApiKeyResolver, FetchImpl } from "@oh-my-pi/pi-ai";
|
||||
import { registerOAuthProvider, unregisterOAuthProviders } from "@oh-my-pi/pi-ai/oauth";
|
||||
@@ -1560,12 +1566,11 @@ export class ModelRegistry {
|
||||
): Promise<BuiltInDiscoveryResult> {
|
||||
// Skip providers already handled by configured discovery (e.g. user-configured ollama with discovery.type)
|
||||
const configuredDiscoveryProviders = new Set(this.#discoverableProviders.map(p => p.provider));
|
||||
const managerOptions = (await this.#collectBuiltInModelManagerOptions()).filter(opts => {
|
||||
if (configuredDiscoveryProviders.has(opts.providerId)) {
|
||||
return false;
|
||||
}
|
||||
return providerFilter ? providerFilter.has(opts.providerId) : true;
|
||||
});
|
||||
const managerOptions = await this.#collectBuiltInModelManagerOptions(
|
||||
strategy,
|
||||
providerFilter,
|
||||
configuredDiscoveryProviders,
|
||||
);
|
||||
if (managerOptions.length === 0) {
|
||||
return { models: [], authoritativeProviders: new Set() };
|
||||
}
|
||||
@@ -1583,7 +1588,49 @@ export class ModelRegistry {
|
||||
return { models, authoritativeProviders };
|
||||
}
|
||||
|
||||
async #collectBuiltInModelManagerOptions(): Promise<ModelManagerOptions<Api>[]> {
|
||||
async #resolveBuiltInDiscoveryApiKey(
|
||||
providerId: string,
|
||||
strategy: ModelRefreshStrategy,
|
||||
cacheProviderId: string,
|
||||
): Promise<string | undefined> {
|
||||
const peekedKey = await this.#peekApiKeyForProvider(providerId);
|
||||
if (isAuthenticated(peekedKey) || strategy === "offline") {
|
||||
return peekedKey;
|
||||
}
|
||||
const oauthCredentials = getOAuthCredentialsForProvider(this.authStorage, providerId);
|
||||
if (oauthCredentials.length === 0) {
|
||||
return peekedKey;
|
||||
}
|
||||
if (strategy === "online-if-uncached") {
|
||||
// Mirror shouldFetchRemoteSources: built-in managers use the catalog's
|
||||
// default TTL, so only refresh when the manager will actually fetch.
|
||||
const cache = readModelCache<Api>(
|
||||
cacheProviderId,
|
||||
BUILT_IN_DISCOVERY_CACHE_TTL_MS,
|
||||
Date.now,
|
||||
this.#cacheDbPath,
|
||||
);
|
||||
const cacheAgeMs = cache ? Date.now() - cache.updatedAt : Number.POSITIVE_INFINITY;
|
||||
if (cache?.fresh && (cache.authoritative || cacheAgeMs < BUILT_IN_DISCOVERY_NON_AUTHORITATIVE_RETRY_MS)) {
|
||||
return peekedKey;
|
||||
}
|
||||
}
|
||||
try {
|
||||
return await this.getApiKeyForProvider(providerId);
|
||||
} catch (error) {
|
||||
logger.debug("OAuth refresh failed during model discovery preflight", {
|
||||
provider: providerId,
|
||||
error: error instanceof Error ? error.message : String(error),
|
||||
});
|
||||
return peekedKey;
|
||||
}
|
||||
}
|
||||
|
||||
async #collectBuiltInModelManagerOptions(
|
||||
strategy: ModelRefreshStrategy,
|
||||
providerFilter: ReadonlySet<string> | undefined,
|
||||
configuredDiscoveryProviders: ReadonlySet<string>,
|
||||
): Promise<ModelManagerOptions<Api>[]> {
|
||||
const specialProviderDescriptors: Array<{
|
||||
providerId: string;
|
||||
resolveKey: (value: string | undefined) => string | undefined;
|
||||
@@ -1622,20 +1669,33 @@ export class ModelRegistry {
|
||||
},
|
||||
];
|
||||
const disabledProviders = getDisabledProviderIdsFromSettings();
|
||||
const standardProviderDescriptors = PROVIDER_DESCRIPTORS.filter(
|
||||
descriptor => !disabledProviders.has(descriptor.providerId),
|
||||
const standardProviderDescriptors = PROVIDER_DESCRIPTORS.filter(descriptor => {
|
||||
if (disabledProviders.has(descriptor.providerId)) return false;
|
||||
if (configuredDiscoveryProviders.has(descriptor.providerId)) return false;
|
||||
return providerFilter ? providerFilter.has(descriptor.providerId) : true;
|
||||
});
|
||||
const enabledSpecialProviderDescriptors = specialProviderDescriptors.filter(descriptor => {
|
||||
if (disabledProviders.has(descriptor.providerId)) return false;
|
||||
if (configuredDiscoveryProviders.has(descriptor.providerId)) return false;
|
||||
return providerFilter ? providerFilter.has(descriptor.providerId) : true;
|
||||
});
|
||||
const standardProviderKeys = await Promise.all(
|
||||
standardProviderDescriptors.map(descriptor => {
|
||||
const discoveryBaseUrl =
|
||||
this.#runtimeProviderOverrides.get(descriptor.providerId)?.baseUrl ??
|
||||
this.#providerOverrides.get(descriptor.providerId)?.baseUrl ??
|
||||
this.getProviderBaseUrl(descriptor.providerId);
|
||||
const cacheProviderId =
|
||||
descriptor.createModelManagerOptions({ baseUrl: discoveryBaseUrl, fetch: this.#fetch })
|
||||
.cacheProviderId ?? descriptor.providerId;
|
||||
return this.#resolveBuiltInDiscoveryApiKey(descriptor.providerId, strategy, cacheProviderId);
|
||||
}),
|
||||
);
|
||||
const enabledSpecialProviderDescriptors = specialProviderDescriptors.filter(
|
||||
descriptor => !disabledProviders.has(descriptor.providerId),
|
||||
const specialKeys = await Promise.all(
|
||||
enabledSpecialProviderDescriptors.map(descriptor =>
|
||||
this.#resolveBuiltInDiscoveryApiKey(descriptor.providerId, strategy, descriptor.providerId),
|
||||
),
|
||||
);
|
||||
// Use peekApiKey to avoid OAuth token refresh during discovery.
|
||||
// The token is only needed if the dynamic fetch fires (cache miss),
|
||||
// and failures there are handled gracefully.
|
||||
const peekKey = (descriptor: { providerId: string }) => this.#peekApiKeyForProvider(descriptor.providerId);
|
||||
const [standardProviderKeys, specialKeys] = await Promise.all([
|
||||
Promise.all(standardProviderDescriptors.map(peekKey)),
|
||||
Promise.all(enabledSpecialProviderDescriptors.map(peekKey)),
|
||||
]);
|
||||
const options: ModelManagerOptions<Api>[] = [];
|
||||
for (let i = 0; i < standardProviderDescriptors.length; i++) {
|
||||
const descriptor = standardProviderDescriptors[i];
|
||||
@@ -1670,7 +1730,12 @@ export class ModelRegistry {
|
||||
}
|
||||
// Append runtime model managers registered by extensions via fetchDynamicModels.
|
||||
for (const { options: managerOpts } of this.#runtimeModelManagers.values()) {
|
||||
options.push(managerOpts);
|
||||
if (
|
||||
!configuredDiscoveryProviders.has(managerOpts.providerId) &&
|
||||
(!providerFilter || providerFilter.has(managerOpts.providerId))
|
||||
) {
|
||||
options.push(managerOpts);
|
||||
}
|
||||
}
|
||||
return options;
|
||||
}
|
||||
|
||||
@@ -22,6 +22,7 @@ import {
|
||||
getProjectDir,
|
||||
isEnoent,
|
||||
logger,
|
||||
MAIN_CONFIG_FILENAMES,
|
||||
procmgr,
|
||||
setWorktreesDir,
|
||||
} from "@oh-my-pi/pi-utils";
|
||||
@@ -60,7 +61,7 @@ export interface RawSettings {
|
||||
export interface SettingsOptions {
|
||||
/** Current working directory for project settings discovery */
|
||||
cwd?: string;
|
||||
/** Agent directory for config.yml storage */
|
||||
/** Agent directory for config.yml/config.yaml storage */
|
||||
agentDir?: string;
|
||||
/** Don't persist to disk (for tests) */
|
||||
inMemory?: boolean;
|
||||
@@ -234,7 +235,7 @@ export class Settings {
|
||||
#storage: AgentStorage | null = null;
|
||||
|
||||
#configFiles: string[] = [];
|
||||
/** Global settings from config.yml */
|
||||
/** Global settings from config.yml/config.yaml */
|
||||
#global: RawSettings = {};
|
||||
/** Project settings from .claude/settings.yml etc */
|
||||
#project: RawSettings = {};
|
||||
@@ -264,7 +265,7 @@ export class Settings {
|
||||
private constructor(options: SettingsOptions = {}) {
|
||||
this.#cwd = path.normalize(options.cwd ?? getProjectDir());
|
||||
this.#agentDir = path.normalize(options.agentDir ?? getAgentDir());
|
||||
this.#configPath = options.inMemory ? null : path.join(this.#agentDir, "config.yml");
|
||||
this.#configPath = options.inMemory ? null : path.join(this.#agentDir, MAIN_CONFIG_FILENAMES[0]);
|
||||
this.#configFiles = options.configFiles?.map(file => path.resolve(this.#cwd, expandTilde(file))) ?? [];
|
||||
this.#persist = !options.inMemory && options.readOnly !== true;
|
||||
|
||||
@@ -458,6 +459,7 @@ export class Settings {
|
||||
inMemory: !this.#persist,
|
||||
});
|
||||
cloned.#storage = this.#storage;
|
||||
cloned.#configPath = this.#configPath;
|
||||
cloned.#global = structuredClone(this.#global);
|
||||
cloned.#project = this.#persist ? await cloned.#loadProjectSettings() : structuredClone(this.#project);
|
||||
cloned.#configFiles = [...this.#configFiles];
|
||||
@@ -676,16 +678,22 @@ export class Settings {
|
||||
|
||||
async #load(): Promise<Settings> {
|
||||
// Project settings load (loadCapability scans cwd) is independent of the
|
||||
// persist chain (storage open → legacy migration → global config.yml read),
|
||||
// so kick it off first and await after the persist chain completes. The
|
||||
// persist steps remain sequential: migration may write config.yml, which
|
||||
// #loadYaml then reads; migration's db fallback needs #storage opened.
|
||||
// persist chain (storage open → legacy migration → global config read), so
|
||||
// kick it off first and await after the persist chain completes. The
|
||||
// persist steps remain sequential: existing config discovery decides
|
||||
// whether migration may write config.yml before the global config is read;
|
||||
// migration's db fallback needs #storage opened.
|
||||
const projectPromise = this.#loadProjectSettings();
|
||||
|
||||
if (this.#persist) {
|
||||
this.#storage = await AgentStorage.open(getAgentDbPath(this.#agentDir));
|
||||
await this.#migrateFromLegacy();
|
||||
this.#global = await this.#loadYaml(this.#configPath!);
|
||||
const existingConfig = await this.#loadExistingMainYaml();
|
||||
if (existingConfig) {
|
||||
this.#global = existingConfig;
|
||||
} else {
|
||||
await this.#migrateFromLegacy();
|
||||
this.#global = await this.#loadYaml(this.#configPath!);
|
||||
}
|
||||
await this.#seedLastChangelogVersionMarker();
|
||||
}
|
||||
|
||||
@@ -701,8 +709,9 @@ export class Settings {
|
||||
async #loadReadOnly(): Promise<Settings> {
|
||||
const projectPromise = this.#loadProjectSettings();
|
||||
|
||||
if (this.#configPath) {
|
||||
this.#global = await this.#loadYaml(this.#configPath);
|
||||
const existingConfig = await this.#loadExistingMainYaml();
|
||||
if (existingConfig) {
|
||||
this.#global = existingConfig;
|
||||
}
|
||||
|
||||
this.#project = await projectPromise;
|
||||
@@ -712,20 +721,46 @@ export class Settings {
|
||||
}
|
||||
|
||||
async #loadYaml(filePath: string): Promise<RawSettings> {
|
||||
const loaded = await this.#loadYamlIfPresent(filePath);
|
||||
return loaded ?? {};
|
||||
}
|
||||
|
||||
async #loadYamlIfPresent(filePath: string): Promise<RawSettings | null> {
|
||||
let content: string;
|
||||
try {
|
||||
content = await Bun.file(filePath).text();
|
||||
} catch (error) {
|
||||
if (isEnoent(error)) return null;
|
||||
logger.warn("Settings: failed to load", { path: filePath, error: String(error) });
|
||||
return {};
|
||||
}
|
||||
|
||||
try {
|
||||
const content = await Bun.file(filePath).text();
|
||||
const parsed = YAML.parse(content);
|
||||
if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) {
|
||||
return {};
|
||||
}
|
||||
return this.#migrateRawSettings(parsed as RawSettings);
|
||||
} catch (error) {
|
||||
if (isEnoent(error)) return {};
|
||||
logger.warn("Settings: failed to load", { path: filePath, error: String(error) });
|
||||
return {};
|
||||
}
|
||||
}
|
||||
|
||||
async #loadExistingMainYaml(): Promise<RawSettings | null> {
|
||||
if (!this.#configPath) return null;
|
||||
for (const filename of MAIN_CONFIG_FILENAMES) {
|
||||
const configPath = path.join(this.#agentDir, filename);
|
||||
const loaded = await this.#loadYamlIfPresent(configPath);
|
||||
if (loaded) {
|
||||
this.#configPath = configPath;
|
||||
return loaded;
|
||||
}
|
||||
}
|
||||
this.#configPath = path.join(this.#agentDir, MAIN_CONFIG_FILENAMES[0]);
|
||||
return null;
|
||||
}
|
||||
|
||||
async #loadProjectSettings(): Promise<RawSettings> {
|
||||
try {
|
||||
const result = await loadCapability(settingsCapability.id, { cwd: this.#cwd });
|
||||
@@ -781,14 +816,6 @@ export class Settings {
|
||||
async #migrateFromLegacy(): Promise<void> {
|
||||
if (!this.#configPath) return;
|
||||
|
||||
// Check if config.yml already exists
|
||||
try {
|
||||
await Bun.file(this.#configPath).text();
|
||||
return; // Already exists, no migration needed
|
||||
} catch (err) {
|
||||
if (!isEnoent(err)) return;
|
||||
}
|
||||
|
||||
let settings: RawSettings = {};
|
||||
let migrated = false;
|
||||
|
||||
|
||||
@@ -207,6 +207,7 @@ const noOpUIContext: ExtensionUIContext = {
|
||||
pasteToEditor: () => {},
|
||||
getEditorText: () => "",
|
||||
editor: async () => undefined,
|
||||
addAutocompleteProvider: () => {},
|
||||
setEditorComponent: () => {},
|
||||
get theme() {
|
||||
return theme;
|
||||
|
||||
@@ -30,7 +30,7 @@ import type {
|
||||
TSchema,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
import type { OAuthCredentials, OAuthLoginCallbacks } from "@oh-my-pi/pi-ai/oauth/types";
|
||||
import type { AutocompleteItem, Component, EditorTheme, KeyId, TUI } from "@oh-my-pi/pi-tui";
|
||||
import type { AutocompleteItem, AutocompleteProvider, Component, EditorTheme, KeyId, TUI } from "@oh-my-pi/pi-tui";
|
||||
import type { logger as PiLogger } from "@oh-my-pi/pi-utils";
|
||||
import type { Type as arktype } from "arktype";
|
||||
import type * as zod from "zod/v4";
|
||||
@@ -163,6 +163,9 @@ export type ExtensionUiComponent = Component & { dispose?(): void };
|
||||
export type ExtensionUiComponentFactory = (tui: TUI, theme: Theme) => ExtensionUiComponent;
|
||||
export type ExtensionWidgetContent = string[] | ExtensionUiComponentFactory | undefined;
|
||||
|
||||
/** Wrap the current autocomplete provider with additional behavior (pi-compatible). */
|
||||
export type AutocompleteProviderFactory = (current: AutocompleteProvider) => AutocompleteProvider;
|
||||
|
||||
/**
|
||||
* UI context for extensions to request interactive UI.
|
||||
* Each mode (interactive, RPC, print) provides its own implementation.
|
||||
@@ -243,6 +246,14 @@ export interface ExtensionUIContext {
|
||||
editorOptions?: { promptStyle?: boolean },
|
||||
): Promise<string | undefined>;
|
||||
|
||||
/**
|
||||
* Stack additional autocomplete behavior on top of the built-in provider
|
||||
* (pi-compatible). Interactive mode rebuilds the editor's provider through
|
||||
* every registered factory, in registration order; headless modes (print,
|
||||
* RPC, ACP, subagents) accept and ignore the factory.
|
||||
*/
|
||||
addAutocompleteProvider(factory: AutocompleteProviderFactory): void;
|
||||
|
||||
/**
|
||||
* Set a custom editor component via factory function, or `undefined` to restore the default editor.
|
||||
*
|
||||
@@ -1107,7 +1118,7 @@ export interface ExtensionAPI {
|
||||
options?: { triggerTurn?: boolean; deliverAs?: "steer" | "followUp" | "nextTurn" },
|
||||
): void;
|
||||
|
||||
/** Send a user message to the agent, or queue it when deliverAs is set. */
|
||||
/** Send a user prompt: idle starts a turn; streaming queues as steer unless deliverAs is set. */
|
||||
sendUserMessage(
|
||||
content: string | (TextContent | ImageContent)[],
|
||||
options?: { deliverAs?: "steer" | "followUp" },
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user