diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 7dbe06010..00df0a797 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -622,10 +622,11 @@ jobs: with: node-version: "24" registry-url: "https://registry.npmjs.org" - # Keep npm aligned with trusted publishing setup (>= 11.16.0). + # npm runs under Bun when invoked by the release script; npm 12 + # requires a newer emulated Node version than Bun 1.3 provides. - name: Ensure npm supports trusted publishing if: ${{ !inputs.skip_npm }} - run: npm install -g npm@latest + run: npm install -g npm@11.17.0 - name: Cache bun dependencies uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 with: @@ -775,9 +776,10 @@ jobs: with: node-version: "24" registry-url: "https://registry.npmjs.org" - # Keep npm aligned with trusted publishing setup (>= 11.16.0). + # npm runs under Bun when invoked by the release script; npm 12 + # requires a newer emulated Node version than Bun 1.3 provides. - name: Ensure npm supports trusted publishing - run: npm install -g npm@latest + run: npm install -g npm@11.17.0 - name: Cache bun dependencies uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 with: diff --git a/.omp/commands/triage.md b/.omp/commands/triage.md index 8fa481984..83ac48821 100644 --- a/.omp/commands/triage.md +++ b/.omp/commands/triage.md @@ -72,7 +72,7 @@ For each candidate issue, read the title, body, and **all comments** (comments o | `providers` | Provider-related behavior (generic provider scope) | **Provider labels** (apply only when a specific provider is explicitly involved): -`provider:anthropic`, `provider:bedrock`, `provider:brave`, `provider:cerebras`, `provider:cloudflare`, `provider:codex`, `provider:copilot`, `provider:cursor`, `provider:exa`, `provider:gemini`, `provider:gitlab`, `provider:groq`, `provider:huggingface`, `provider:jina`, `provider:kimi`, `provider:litellm`, `provider:minimax`, `provider:mistral`, `provider:moonshot`, `provider:nanogpt`, `provider:nvidia`, `provider:openai`, `provider:opencode`, `provider:openrouter`, `provider:perplexity`, `provider:qianfan`, `provider:qwen`, `provider:synthetic`, `provider:together`, `provider:venice`, `provider:vercel`, `provider:xai`, `provider:xiaomi`, `provider:zai` +`provider:anthropic`, `provider:bedrock`, `provider:brave`, `provider:cerebras`, `provider:cloudflare`, `provider:codex`, `provider:copilot`, `provider:cursor`, `provider:exa`, `provider:gemini`, `provider:gitlab`, `provider:groq`, `provider:huggingface`, `provider:jina`, `provider:kimi`, `provider:litellm`, `provider:minimax`, `provider:mistral`, `provider:moonshot`, `provider:nanogpt`, `provider:novita`, `provider:nvidia`, `provider:openai`, `provider:opencode`, `provider:openrouter`, `provider:perplexity`, `provider:qianfan`, `provider:qwen`, `provider:synthetic`, `provider:together`, `provider:venice`, `provider:vercel`, `provider:xai`, `provider:xiaomi`, `provider:zai` **Platform labels** (apply only when platform materially affects reproduction/root cause): | Label | Signals | diff --git a/Cargo.lock b/Cargo.lock index 1b29c8dce..125f5b658 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1850,9 +1850,9 @@ checksum = "b9e0384b61958566e926dc50660321d12159025e767c18e043daf26b70104c39" [[package]] name = "ignore" -version = "0.4.27" +version = "0.4.28" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fe112b004901c62c2faa11f4f75e9864e0cc5af8da71c9115d184a3aa888749f" +checksum = "2adf14691c72bcfc1058740436a35bdd3ae9c07d1a941ef00b749e9ea16aefa7" dependencies = [ "crossbeam-deque", "globset", @@ -1951,9 +1951,9 @@ dependencies = [ [[package]] name = "inotify" -version = "0.11.3" +version = "0.11.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dd854a95a4ac672fed8c054136039fd32c22cf039ff09ead7280afe920486483" +checksum = "153be1941a183ec9ccd095ddbe17a8b8d435ef6c76e9e02451b933c3999af2c8" dependencies = [ "bitflags 2.13.0", "inotify-sys", @@ -2026,9 +2026,9 @@ checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" [[package]] name = "jiff" -version = "0.2.31" +version = "0.2.32" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ccfe6121cbe750cf81efa362d85c0bde7ea298ec43092d3a193baca59cdbd634" +checksum = "961d16382652bfdd8c6f68b223b26a8c93e0d475c672f414411db31c6c5c900e" dependencies = [ "defmt", "jiff-static", @@ -2053,9 +2053,9 @@ dependencies = [ [[package]] name = "jiff-static" -version = "0.2.31" +version = "0.2.32" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e165e897f662d428f3cd3828a919dbe067c2d42bb1031eede74ef9d27ecdedd2" +checksum = "d0879bd39df99c4c5e2c6615ccc026391a423dde10532c573e6086eb94a802cc" dependencies = [ "proc-macro2", "quote", @@ -2064,9 +2064,9 @@ dependencies = [ [[package]] name = "jiff-tzdb" -version = "0.1.7" +version = "0.1.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6142247df1a93c2b3587402a19710be3e6e942f1581a1702e76408f2c21d6590" +checksum = "142bd39932ad231f10513df9ab62661fead8719872150b7ad02a2df79f4e141e" [[package]] name = "jiff-tzdb-platform" @@ -2881,7 +2881,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "16.3.12" +version = "16.3.15" dependencies = [ "anyhow", "ast-grep-core", @@ -2950,7 +2950,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "16.3.12" +version = "16.3.15" dependencies = [ "async-trait", "libc", @@ -2962,13 +2962,14 @@ dependencies = [ [[package]] name = "pi-natives" -version = "16.3.12" +version = "16.3.15" dependencies = [ "anyhow", "arboard", "ast-grep-core", "base64", "clap", + "clipboard-win", "flume", "fontdue", "globset", @@ -3014,7 +3015,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "16.3.12" +version = "16.3.15" dependencies = [ "anyhow", "brush-builtins", @@ -3063,7 +3064,7 @@ dependencies = [ [[package]] name = "pi-walker" -version = "16.3.12" +version = "16.3.15" dependencies = [ "dashmap", "globset", @@ -3403,9 +3404,9 @@ dependencies = [ [[package]] name = "regex" -version = "1.12.4" +version = "1.13.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f1292b7759ae1cb9ec195452d1390a074f0cd8541ab7a5a8c31cd6db45d4a6ba" +checksum = "2a0e75113e14dc5acb068cd0786884f214f1312650a3d36d269f5c4f3cdee8a2" dependencies = [ "aho-corasick", "memchr", @@ -3415,9 +3416,9 @@ dependencies = [ [[package]] name = "regex-automata" -version = "0.4.14" +version = "0.4.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6e1dd4122fc1595e8162618945476892eefca7b88c52820e74af6262213cae8f" +checksum = "1f388202e4b80542a0921078cc23b6333bcf1409c1e3f86404cae4766a6131db" dependencies = [ "aho-corasick", "memchr", @@ -5765,18 +5766,18 @@ dependencies = [ [[package]] name = "zerocopy" -version = "0.8.53" +version = "0.8.54" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "75726053136156d419e285b9b7eddaaea9e3fea6ce32eed44a89901f0bd98de1" +checksum = "b7cbbc0a705a0fd05cc3676525980d2bf5a9bc4adac6d6475209a7887cf59d19" dependencies = [ "zerocopy-derive", ] [[package]] name = "zerocopy-derive" -version = "0.8.53" +version = "0.8.54" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4714fd92cf900833d49538023a9b3915155210801d1c1169eba513b2addefd71" +checksum = "e2e817b7b52d0c7358d3246da9d69935ebb18116b2b102b4230dac079b4862f5" dependencies = [ "proc-macro2", "quote", diff --git a/Cargo.toml b/Cargo.toml index 88c64391b..652fd749f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/vendor/brush-core", "crates/vendor/brush-builtins"] resolver = "3" [workspace.package] -version = "16.3.12" +version = "16.3.15" edition = "2024" license = "MIT" authors = ["Can Boluk"] @@ -269,6 +269,7 @@ napi-derive = "3" # Terminal & PTY # ────────────────────────────────────────────────────────────────────────────── arboard = { version = "3.6.1", features = ["wayland-data-control"] } +clipboard-win = "5.4" icy_sixel = "0.5" portable-pty = "0.9" diff --git a/README.md b/README.md index 15e7a878e..ec9f6074f 100644 --- a/README.md +++ b/README.md @@ -292,7 +292,7 @@ Anthropic `oauth` · OpenAI · OpenAI Codex `oauth` · Google Gemini · Google A Subscription-routed. `/login` attaches the session. -Cursor `oauth` · GitHub Copilot `oauth` · GitLab Duo · Kimi Code `plan` · Moonshot · MiniMax Coding Plan `plan` · MiniMax Coding Plan CN `plan` · Alibaba Coding Plan `plan` · Qwen Portal · Z.AI / GLM Coding Plan `plan` · Xiaomi MiMo · Qianfan · NanoGPT · Venice · Kilo · ZenMux · OpenCode Go · OpenCode Zen +Cursor `oauth` · GitHub Copilot `oauth` · GitLab Duo · Kimi Code `plan` · Moonshot · MiniMax Coding Plan `plan` · MiniMax Coding Plan CN `plan` · Alibaba Coding Plan `plan` · Qwen Portal · Z.AI / GLM Coding Plan `plan` · Xiaomi MiMo · Qianfan · NanoGPT · Novita · Venice · Kilo · ZenMux · OpenCode Go · OpenCode Zen ### Run it yourself diff --git a/bun.lock b/bun.lock index f97a68579..d179b61c4 100644 --- a/bun.lock +++ b/bun.lock @@ -21,7 +21,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "16.3.12", + "version": "16.3.15", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -39,7 +39,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "16.3.12", + "version": "16.3.15", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -55,7 +55,7 @@ }, "packages/catalog": { "name": "@oh-my-pi/pi-catalog", - "version": "16.3.12", + "version": "16.3.15", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -69,7 +69,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "16.3.12", + "version": "16.3.15", "bin": { "omp": "src/cli.ts", }, @@ -137,7 +137,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "16.3.12", + "version": "16.3.15", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -148,7 +148,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "16.3.12", + "version": "16.3.15", "bin": { "mnemopi": "src/cli.ts", }, @@ -174,7 +174,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "16.3.12", + "version": "16.3.15", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -182,7 +182,7 @@ }, "packages/snapcompact": { "name": "@oh-my-pi/snapcompact", - "version": "16.3.12", + "version": "16.3.15", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -195,7 +195,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "16.3.12", + "version": "16.3.15", "bin": { "omp-stats": "./src/index.ts", }, @@ -221,7 +221,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "16.3.12", + "version": "16.3.15", "bin": { "omp-swarm": "src/cli.ts", }, @@ -247,7 +247,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "16.3.12", + "version": "16.3.15", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -288,7 +288,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "16.3.12", + "version": "16.3.15", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "handlebars": "catalog:", @@ -301,7 +301,7 @@ }, "packages/wire": { "name": "@oh-my-pi/pi-wire", - "version": "16.3.12", + "version": "16.3.15", "devDependencies": { "@types/bun": "catalog:", }, @@ -338,18 +338,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.3.12", - "@oh-my-pi/omp-stats": "16.3.12", - "@oh-my-pi/pi-agent-core": "16.3.12", - "@oh-my-pi/pi-ai": "16.3.12", - "@oh-my-pi/pi-catalog": "16.3.12", - "@oh-my-pi/pi-coding-agent": "16.3.12", - "@oh-my-pi/pi-mnemopi": "16.3.12", - "@oh-my-pi/pi-natives": "16.3.12", - "@oh-my-pi/pi-tui": "16.3.12", - "@oh-my-pi/pi-utils": "16.3.12", - "@oh-my-pi/pi-wire": "16.3.12", - "@oh-my-pi/snapcompact": "16.3.12", + "@oh-my-pi/hashline": "16.3.15", + "@oh-my-pi/omp-stats": "16.3.15", + "@oh-my-pi/pi-agent-core": "16.3.15", + "@oh-my-pi/pi-ai": "16.3.15", + "@oh-my-pi/pi-catalog": "16.3.15", + "@oh-my-pi/pi-coding-agent": "16.3.15", + "@oh-my-pi/pi-mnemopi": "16.3.15", + "@oh-my-pi/pi-natives": "16.3.15", + "@oh-my-pi/pi-tui": "16.3.15", + "@oh-my-pi/pi-utils": "16.3.15", + "@oh-my-pi/pi-wire": "16.3.15", + "@oh-my-pi/snapcompact": "16.3.15", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -791,7 +791,7 @@ "@opentelemetry/sdk-trace-node": ["@opentelemetry/sdk-trace-node@2.9.0", "", { "dependencies": { "@opentelemetry/context-async-hooks": "2.9.0", "@opentelemetry/core": "2.9.0", "@opentelemetry/sdk-trace-base": "2.9.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-ec9a7ps37huy5itYk0MalaZdSLlM6AXWp/FhtEjgMpp5leEGojBDvAl/UWttQnkMZOvFHKzRESn8TD3yKTF5nQ=="], - "@opentelemetry/semantic-conventions": ["@opentelemetry/semantic-conventions@1.41.1", "", {}, "sha512-/UhIkaZgPutTFmQ7RnIJGgDXZmtEJ7Dvi86xNTFWcnRxVRNk/aotsqDJYeEvDP+FSMB2SdW+pQzNMcWP0rwuNA=="], + "@opentelemetry/semantic-conventions": ["@opentelemetry/semantic-conventions@1.42.0", "", {}, "sha512-icc5xCzndZfhuJMy5oqk5AvloWquR7jtae74qzpkKkhGp8BivK+oCcEXgGnjCdTfp8hA44l+w8gE8yYJbocJJw=="], "@oxc-project/types": ["@oxc-project/types@0.138.0", "", {}, "sha512-1a7ZKmrRTCoN1XMZ4L0PyyqrMnrNlLyPuOkdSX2MZg7IiIGRUyurNhAm73ptDOraoBcIordsIGKNPKUzy3ZmfA=="], @@ -811,7 +811,7 @@ "@protobufjs/pool": ["@protobufjs/pool@1.1.0", "", {}, "sha512-0kELaGSIDBKvcgS4zkjz1PeddatrjYcmMWOlAuAPwAeccUrPHdUqo/J6LiymHHEiJT5NrF1UVwxY14f+fy4WQw=="], - "@protobufjs/utf8": ["@protobufjs/utf8@1.1.1", "", {}, "sha512-oOAWABowe8EAbMyWKM0tYDKi8Yaox52D+HWZhAIJqQXbqe0xI/GV7FhLWqlEKreMkfDjshR5FKgi3mnle0h6Eg=="], + "@protobufjs/utf8": ["@protobufjs/utf8@1.1.2", "", {}, "sha512-b1UQwcEZ4yCnMCD8DAL1VlbvBJE9/IX4FTIp7BG1xYpf29SLazLSrqUkj4w7Y5y7cCVP6E5tcqqcI0xemPkHug=="], "@puppeteer/browsers": ["@puppeteer/browsers@3.0.6", "", { "dependencies": { "modern-tar": "^0.7.6", "yargs": "^18.0.0" }, "peerDependencies": { "proxy-agent": ">=8.0.1", "yauzl": "^2.10.0 || ^3.4.0" }, "optionalPeers": ["proxy-agent", "yauzl"], "bin": { "browsers": "lib/main-cli.js" } }, "sha512-B/gKoqlFkzhvzsI6jo9K1cZz9o5ypviVv/xu8CwA4grZzyVwN+XfkT+tu8T1zrauuEXv6VhS2oGX+6NL95WcKA=="], @@ -963,7 +963,7 @@ "brace-expansion": ["brace-expansion@5.0.7", "", { "dependencies": { "balanced-match": "^4.0.2" } }, "sha512-7oFy703dxfY3/NLxC1fh2SUCQ0H9rmAY+5EpDVfXjUTTs+HEwR2nYaqLv+GWcTsumwxPfiz6CzCNkwXwBUwqCA=="], - "browserslist": ["browserslist@4.28.4", "", { "dependencies": { "baseline-browser-mapping": "^2.10.38", "caniuse-lite": "^1.0.30001799", "electron-to-chromium": "^1.5.376", "node-releases": "^2.0.48", "update-browserslist-db": "^1.2.3" }, "bin": { "browserslist": "cli.js" } }, "sha512-MTc8i/x9jBQd1iMw2CFGS+rwMa07eYjLR0CCTLDACl9xhxy+nIs3KeML/biicXtk9JrZ6dnnTatmc7ErPXIxqw=="], + "browserslist": ["browserslist@4.28.5", "", { "dependencies": { "baseline-browser-mapping": "^2.10.42", "caniuse-lite": "^1.0.30001800", "electron-to-chromium": "^1.5.387", "node-releases": "^2.0.50", "update-browserslist-db": "^1.2.3" }, "bin": { "browserslist": "cli.js" } }, "sha512-Cu2E6QejHWzuDMTkuwgpABFgDfZrXLQq5V13YOACZx4mFAG4IwGTbTfHPMr4WtxlHoXSM8FIuRwYYCz5XiabaQ=="], "bun-types": ["bun-types@1.3.14", "", { "dependencies": { "@types/node": "*" } }, "sha512-4N0ig0fEomHt5R0KCFWjovxow98rIoRwKolrYdCcknNwMekCXRnWEUvgu5soYV8QXtVsrUD8B95MBOZGPvr6KQ=="], diff --git a/crates/pi-natives/Cargo.toml b/crates/pi-natives/Cargo.toml index 6495a02c8..0f230bcd6 100644 --- a/crates/pi-natives/Cargo.toml +++ b/crates/pi-natives/Cargo.toml @@ -26,7 +26,7 @@ grep-searcher.workspace = true html-to-markdown-rs.workspace = true icy_sixel.workspace = true ignore.workspace = true -image.workspace = true +image = { workspace = true, features = ["bmp"] } inferno.workspace = true memmap2.workspace = true napi.workspace = true @@ -61,6 +61,7 @@ libc.workspace = true [target.'cfg(windows)'.dependencies] windows-sys = { workspace = true, features = ["Wdk_Storage_FileSystem", "Win32_Security"] } +clipboard-win.workspace = true winreg.workspace = true [build-dependencies] diff --git a/crates/pi-natives/src/clipboard.rs b/crates/pi-natives/src/clipboard.rs index 060cae6c7..c7b2251b2 100644 --- a/crates/pi-natives/src/clipboard.rs +++ b/crates/pi-natives/src/clipboard.rs @@ -30,7 +30,14 @@ fn encode_png(image: ImageData<'_>) -> Result> { let bytes = image.bytes.into_owned(); let buffer = RgbaImage::from_raw(width, height, bytes) .ok_or_else(|| Error::from_reason("Clipboard image buffer size mismatch"))?; - let capacity = width.saturating_mul(height).saturating_mul(4) as usize; + rgba_to_png(buffer) +} + +fn rgba_to_png(buffer: RgbaImage) -> Result> { + let capacity = (buffer + .width() + .saturating_mul(buffer.height()) + .saturating_mul(4)) as usize; let mut output = Vec::with_capacity(capacity); DynamicImage::ImageRgba8(buffer) .write_to(&mut Cursor::new(&mut output), ImageFormat::Png) @@ -38,6 +45,88 @@ fn encode_png(image: ImageData<'_>) -> Result> { Ok(output) } +/// Decode a packed DIB clipboard payload (`CF_DIB`: a `BITMAPINFOHEADER`-family +/// header, optional bitfield masks and palette, then the pixel array) into PNG +/// bytes. +/// +/// The payload is wrapped in a synthesized `BITMAPFILEHEADER` and decoded +/// through the BMP *file* path so the explicit `bfOffBits` pins the pixel +/// offset. This matters: the header-less decode path arboard uses mis-places +/// the pixel offset for V4/V5 headers with `BI_BITFIELDS` compression (it +/// skips 12 trailing mask bytes that those headers embed instead), which is +/// why Qt-based screenshot tools (`PixPin`, `Snipaste`, ...) fail through +/// arboard in the first place (#3426). +#[cfg_attr( + not(windows), + allow( + dead_code, + reason = "reached only by the Windows clipboard fallback; kept target-independent so unit \ + tests cover it on every host" + ) +)] +fn dib_to_png(dib: &[u8]) -> Result> { + const FILE_HEADER_SIZE: u64 = 14; + const INFO_HEADER_SIZE: u64 = 40; + const BI_BITFIELDS: u32 = 3; + + if dib.len() < INFO_HEADER_SIZE as usize { + return Err(Error::from_reason("Clipboard DIB shorter than BITMAPINFOHEADER")); + } + let u32_at = + |at: usize| u32::from_le_bytes(dib[at..at + 4].try_into().expect("bounds checked above")); + let header_size = u64::from(u32_at(0)); + if header_size < INFO_HEADER_SIZE || header_size > dib.len() as u64 { + return Err(Error::from_reason("Clipboard DIB header size out of range")); + } + let bit_count = u16::from_le_bytes([dib[14], dib[15]]); + let compression = u32_at(16); + let colors_used = u64::from(u32_at(32)); + + // A plain BITMAPINFOHEADER with BI_BITFIELDS is trailed by three DWORD + // masks; larger (V2..V5) headers embed the masks in the header itself. + let mask_bytes: u64 = if header_size == INFO_HEADER_SIZE && compression == BI_BITFIELDS { + 12 + } else { + 0 + }; + let palette_entries: u64 = if colors_used != 0 { + colors_used + } else if bit_count <= 8 { + 1u64 << bit_count + } else { + 0 + }; + let pixel_offset = + u32::try_from(FILE_HEADER_SIZE + header_size + mask_bytes + palette_entries * 4) + .map_err(|_| Error::from_reason("Clipboard DIB layout overflow"))?; + let file_size = u32::try_from(FILE_HEADER_SIZE + dib.len() as u64) + .map_err(|_| Error::from_reason("Clipboard DIB too large"))?; + + let mut bmp = Vec::with_capacity(FILE_HEADER_SIZE as usize + dib.len()); + bmp.extend_from_slice(b"BM"); + bmp.extend_from_slice(&file_size.to_le_bytes()); + bmp.extend_from_slice(&0u32.to_le_bytes()); + bmp.extend_from_slice(&pixel_offset.to_le_bytes()); + bmp.extend_from_slice(dib); + + let decoded = image::load_from_memory_with_format(&bmp, ImageFormat::Bmp) + .map_err(|err| Error::from_reason(format!("Failed to decode clipboard DIB: {err}")))?; + rgba_to_png(decoded.into_rgba8()) +} + +/// Read the raw `CF_DIB` bytes from the Windows clipboard. +/// +/// Windows synthesizes `CF_DIB` from whatever bitmap formats are present, so +/// it is available whenever the clipboard holds any image at all. +#[cfg(windows)] +fn read_raw_cf_dib() -> Option> { + let clip = clipboard_win::Clipboard::new_attempts(10).ok()?; + let mut dib = Vec::new(); + clipboard_win::raw::get_vec(clipboard_win::formats::CF_DIB, &mut dib).ok()?; + drop(clip); + (!dib.is_empty()).then_some(dib) +} + /// Copy plain text to the system clipboard. /// /// # Parameters @@ -120,7 +209,160 @@ pub fn read_image_from_clipboard() -> task::Promise> { })) }, Err(ClipboardError::ContentNotAvailable) => Ok(None), - Err(err) => Err(Error::from_reason(format!("Failed to read clipboard image: {err}"))), + Err(err) => { + // arboard rejects the CF_DIBV5 payloads Qt-based screenshot + // tools (PixPin, Snipaste, ...) produce; decode the raw CF_DIB + // ourselves before surfacing the error (#3426). A fallback + // decode failure keeps the original arboard error. + #[cfg(windows)] + if let Some(bytes) = read_raw_cf_dib().and_then(|dib| dib_to_png(&dib).ok()) { + return Ok(Some(ClipboardImage { + data: Uint8Array::from(bytes), + mime_type: "image/png".to_string(), + })); + } + Err(Error::from_reason(format!("Failed to read clipboard image: {err}"))) + }, } }) } + +#[cfg(test)] +mod tests { + use super::dib_to_png; + + fn push32(v: u32, out: &mut Vec) { + out.extend_from_slice(&v.to_le_bytes()); + } + + fn push16(v: u16, out: &mut Vec) { + out.extend_from_slice(&v.to_le_bytes()); + } + + /// 2x2 bottom-up BGRA pixel array: memory rows are [red, green] (bottom) + /// then [blue, white] (top), all with alpha 0xff. + const PIXELS_2X2: [u8; 16] = [ + 0x00, 0x00, 0xff, 0xff, // (0,1) red + 0x00, 0xff, 0x00, 0xff, // (1,1) green + 0xff, 0x00, 0x00, 0xff, // (0,0) blue + 0xff, 0xff, 0xff, 0xff, // (1,0) white + ]; + + /// `CF_DIB` as Qt's clipboard writer emits it for 32-bit content: a plain + /// `BITMAPINFOHEADER` with `BI_BITFIELDS` compression and three DWORD + /// masks between header and pixels. + fn qt_cf_dib(width: u32, height: u32, pixels_bgra: &[u8], compression: u32) -> Vec { + let mut d = Vec::with_capacity(52 + pixels_bgra.len()); + push32(40, &mut d); // biSize + push32(width, &mut d); + push32(height, &mut d); // positive: bottom-up + push16(1, &mut d); // biPlanes + push16(32, &mut d); // biBitCount + push32(compression, &mut d); + push32(pixels_bgra.len() as u32, &mut d); // biSizeImage + push32(0, &mut d); // biXPelsPerMeter + push32(0, &mut d); // biYPelsPerMeter + push32(0, &mut d); // biClrUsed + push32(0, &mut d); // biClrImportant + if compression == 3 { + push32(0x00ff_0000, &mut d); // red mask + push32(0x0000_ff00, &mut d); // green mask + push32(0x0000_00ff, &mut d); // blue mask + } + d.extend_from_slice(pixels_bgra); + d + } + + /// `CF_DIBV5` as PixPin (Qt) places it, after arboard's + /// `maybe_tweak_header` rewrite: a 124-byte `BITMAPV5HEADER` carrying + /// `BI_BITFIELDS` compression with the BGRA masks embedded in the header + /// and pixels immediately after it. This is the exact buffer shape that + /// arboard's header-less BMP decode rejects with `ConversionFailure` + /// (issue #3426); the file-header wrap must decode it. + fn pixpin_dibv5_tweaked(width: u32, height: u32, pixels_bgra: &[u8]) -> Vec { + let mut d = Vec::with_capacity(124 + pixels_bgra.len()); + push32(124, &mut d); // bV5Size + push32(width, &mut d); + push32(height, &mut d); + push16(1, &mut d); // bV5Planes + push16(32, &mut d); // bV5BitCount + push32(3, &mut d); // bV5Compression = BI_BITFIELDS (arboard-tweaked) + push32(0, &mut d); // bV5SizeImage + push32(0, &mut d); // bV5XPelsPerMeter + push32(0, &mut d); // bV5YPelsPerMeter + push32(0, &mut d); // bV5ClrUsed + push32(0, &mut d); // bV5ClrImportant + push32(0x00ff_0000, &mut d); // bV5RedMask + push32(0x0000_ff00, &mut d); // bV5GreenMask + push32(0x0000_00ff, &mut d); // bV5BlueMask + push32(0xff00_0000, &mut d); // bV5AlphaMask + push32(0x7352_4742, &mut d); // bV5CSType = LCS_sRGB + d.extend_from_slice(&[0u8; 36]); // bV5Endpoints + push32(0, &mut d); // bV5GammaRed + push32(0, &mut d); // bV5GammaGreen + push32(0, &mut d); // bV5GammaBlue + push32(4, &mut d); // bV5Intent = LCS_GM_IMAGES + push32(0, &mut d); // bV5ProfileData + push32(0, &mut d); // bV5ProfileSize + push32(0, &mut d); // bV5Reserved + assert_eq!(d.len(), 124); + d.extend_from_slice(pixels_bgra); + d + } + + fn decode_pixels(png: &[u8]) -> (u32, u32, Vec<[u8; 4]>) { + let img = image::load_from_memory(png).expect("fallback output must be valid PNG"); + let rgba = img.into_rgba8(); + let (w, h) = rgba.dimensions(); + let px = rgba.pixels().map(|p| p.0).collect(); + (w, h, px) + } + + const RED: [u8; 4] = [255, 0, 0, 255]; + const GREEN: [u8; 4] = [0, 255, 0, 255]; + const BLUE: [u8; 4] = [0, 0, 255, 255]; + const WHITE: [u8; 4] = [255, 255, 255, 255]; + + #[test] + fn decodes_qt_cf_dib_with_bitfields_masks() { + let dib = qt_cf_dib(2, 2, &PIXELS_2X2, 3); + let png = dib_to_png(&dib).expect("BI_BITFIELDS CF_DIB must decode"); + let (w, h, px) = decode_pixels(&png); + assert_eq!((w, h), (2, 2)); + // Row order flipped versus the bottom-up pixel array; BGRA -> RGBA. + assert_eq!(px, vec![BLUE, WHITE, RED, GREEN]); + } + + #[test] + fn decodes_pixpin_dibv5_payload_that_arboard_rejects() { + let dib = pixpin_dibv5_tweaked(2, 2, &PIXELS_2X2); + let png = dib_to_png(&dib).expect("V5 BI_BITFIELDS DIB must decode"); + let (w, h, px) = decode_pixels(&png); + assert_eq!((w, h), (2, 2)); + assert_eq!(px, vec![BLUE, WHITE, RED, GREEN]); + } + + #[test] + fn decodes_plain_bi_rgb_dib() { + // The common "copy image" payload: BI_RGB, 32-bit, no masks. The + // fourth byte is unused per the DIB contract — zero it to prove the + // decode still yields opaque pixels. + let mut pixels = PIXELS_2X2; + for alpha in pixels.iter_mut().skip(3).step_by(4) { + *alpha = 0; + } + let dib = qt_cf_dib(2, 2, &pixels, 0); + let png = dib_to_png(&dib).expect("BI_RGB CF_DIB must decode"); + let (w, h, px) = decode_pixels(&png); + assert_eq!((w, h), (2, 2)); + assert_eq!(px, vec![BLUE, WHITE, RED, GREEN]); + } + + #[test] + fn rejects_malformed_dib() { + assert!(dib_to_png(&[0u8; 12]).is_err(), "short buffer must not decode"); + let mut oversized_header = qt_cf_dib(2, 2, &PIXELS_2X2, 3); + oversized_header[0..4].copy_from_slice(&0xffff_ffffu32.to_le_bytes()); + assert!(dib_to_png(&oversized_header).is_err(), "header size beyond buffer must not decode"); + } +} diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index 7699d1fb1..5b93f4516 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -248,7 +248,7 @@ fn create_windows_napi_tokio_runtime() -> Option { /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV16_3_12")] +#[napi(js_name = "__piNativesV16_3_15")] pub const fn pi_natives_version_sentinel() {} /// Native module entry point: install crash diagnostics before any tool can diff --git a/crates/pi-natives/src/shell.rs b/crates/pi-natives/src/shell.rs index 47402b5fc..e159ba6ed 100644 --- a/crates/pi-natives/src/shell.rs +++ b/crates/pi-natives/src/shell.rs @@ -5,7 +5,7 @@ use std::{collections::HashMap, sync::Arc}; use napi::{ Env, Result, bindgen_prelude::*, - threadsafe_function::{ThreadsafeFunction, ThreadsafeFunctionCallMode}, + threadsafe_function::{ThreadsafeFunction, UnknownReturnValue}, }; use napi_derive::napi; use pi_shell::{ @@ -216,7 +216,7 @@ impl Shell { env: &'env Env, options: ShellRunOptions<'env>, #[napi(ts_arg_type = "((error: Error | null, chunk: string) => void) | undefined | null")] - on_chunk: Option>, + on_chunk: Option>, ) -> Result> { let cancel_token = task::CancelToken::new(options.timeout_ms, options.signal); let inner = Arc::clone(&self.inner); @@ -269,7 +269,7 @@ pub fn execute_shell<'env>( env: &'env Env, options: ShellExecuteOptions<'env>, #[napi(ts_arg_type = "((error: Error | null, chunk: string) => void) | undefined | null")] - on_chunk: Option>, + on_chunk: Option>, ) -> Result> { let cancel_token = task::CancelToken::new(options.timeout_ms, options.signal); let exec_options = CoreShellExecuteOptions { @@ -294,42 +294,66 @@ pub fn execute_shell<'env>( }) } +/// Capacity (in chunks) of the queue between the pipe readers and the JS +/// forwarding pump. One queued chunk is at most one pipe read (≤64 KiB), so +/// the Rust side of the bridge holds ~4 MiB worst case before the readers' +/// `send_async` parks — which in turn parks the child on its stdout/stderr +/// pipe (ordinary pipe backpressure) instead of buffering the surplus in +/// process memory (#4078). +const BRIDGE_QUEUE_CHUNKS: usize = 64; + fn bridge_chunks( - on_chunk: Option>, + on_chunk: Option>, ) -> (Option>, Option>) { let Some(on_chunk) = on_chunk else { return (None, None); }; - let (tx, rx) = flume::unbounded::(); - let handle = napi::tokio::spawn(async move { - // Hard cap on one coalesced batch so the JS main thread never sees a - // multi-MB napi callback (a giant single string would stall sanitize + - // tail-buffer maintenance for the whole copy). - const MAX_BATCH_BYTES: usize = 64 * 1024; - // Initial capacity sized for typical bursty pipe output. Re-allocated - // each batch because `String` ownership is moved into the napi call. - const INITIAL_BATCH_CAP: usize = 8 * 1024; - let mut batch = String::with_capacity(INITIAL_BATCH_CAP); - while let Ok(first) = rx.recv_async().await { - batch.push_str(&first); - // Greedily drain everything already queued. Child processes that - // write byte-at-a-time (printf-style progress, llama-cli token - // streams) otherwise produce one napi callback per `write(2)`, - // saturating the JS main thread (~200% CPU observed) and leaving - // the queue draining long after the child exits. - while batch.len() < MAX_BATCH_BYTES { - match rx.try_recv() { - Ok(more) => batch.push_str(&more), - Err(_) => break, - } - } - let payload = std::mem::replace(&mut batch, String::with_capacity(INITIAL_BATCH_CAP)); - on_chunk.call(Ok(payload), ThreadsafeFunctionCallMode::NonBlocking); - } - }); + let (tx, rx) = flume::bounded::(BRIDGE_QUEUE_CHUNKS); + let handle = napi::tokio::spawn(pump_chunks(rx, async move |payload: String| { + // `call_async` resolves only after the JS callback ran, so at most + // one batch sits in the napi queue at a time and the JS event loop's + // actual consumption rate backpressures the whole pipeline. An error + // means the JS side is gone (env teardown) — stop forwarding. + on_chunk.call_async(Ok(payload)).await.is_ok() + })); (Some(tx), Some(handle)) } +/// Drain `rx`, greedily coalescing queued chunks into ≤64 KiB batches, and +/// feed each batch to `forward`, awaiting its completion before pulling more. +/// Returns when `rx` disconnects (all senders dropped) or `forward` reports +/// the consumer is gone; dropping `rx` then disconnects the channel so +/// parked/future senders fail fast and the pipe readers keep draining the +/// child instead of wedging it. +async fn pump_chunks(rx: flume::Receiver, mut forward: impl AsyncFnMut(String) -> bool) { + // Hard cap on one coalesced batch so the JS main thread never sees a + // multi-MB napi callback (a giant single string would stall sanitize + + // tail-buffer maintenance for the whole copy). + const MAX_BATCH_BYTES: usize = 64 * 1024; + // Initial capacity sized for typical bursty pipe output. Re-allocated + // each batch because `String` ownership is moved into the napi call. + const INITIAL_BATCH_CAP: usize = 8 * 1024; + let mut batch = String::with_capacity(INITIAL_BATCH_CAP); + while let Ok(first) = rx.recv_async().await { + batch.push_str(&first); + // Greedily drain everything already queued. Child processes that + // write byte-at-a-time (printf-style progress, llama-cli token + // streams) otherwise produce one napi callback per `write(2)`, + // saturating the JS main thread (~200% CPU observed) and leaving + // the queue draining long after the child exits. + while batch.len() < MAX_BATCH_BYTES { + match rx.try_recv() { + Ok(more) => batch.push_str(&more), + Err(_) => break, + } + } + let payload = std::mem::replace(&mut batch, String::with_capacity(INITIAL_BATCH_CAP)); + if !forward(payload).await { + return; + } + } +} + /// Result of [`apply_bash_fixups`]: a possibly-rewritten command plus the /// substrings that were removed (in source order). #[napi(object)] @@ -360,7 +384,6 @@ pub fn apply_bash_fixups(command: String) -> BashFixupResult { mod tests { use std::time::Duration; - #[cfg(unix)] use flume; use pi_shell::{ ShellRunOptions as CoreShellRunOptions, @@ -368,7 +391,83 @@ mod tests { }; use tokio::time; - use super::CoreShell; + use super::{BRIDGE_QUEUE_CHUNKS, CoreShell, pump_chunks}; + + /// Regression for #4078: the reader→JS bridge queue must stay bounded when + /// the JS side (here: a deliberately slow `forward`) cannot keep up with a + /// fast producer, and backpressure must never drop or reorder chunks. On + /// the pre-fix bridge (`flume::unbounded` + fire-and-forget + /// `ThreadsafeFunctionCallMode::NonBlocking`) the same harness accumulates + /// the producer's entire surplus in the queue (measured: a 32 MiB stream + /// queued all 33_554_432 bytes while the consumer stalled). + #[tokio::test(flavor = "multi_thread")] + async fn bridge_pump_bounds_queue_and_delivers_all_bytes() { + const CHUNKS: usize = 512; + const CHUNK_BYTES: usize = 4096; + let (tx, rx) = flume::bounded::(BRIDGE_QUEUE_CHUNKS); + let producer = tokio::spawn(async move { + let mut expected = String::with_capacity(CHUNKS * CHUNK_BYTES); + let mut max_queued = 0usize; + for i in 0..CHUNKS { + let chunk = format!("[{i:06}]{}", "x".repeat(CHUNK_BYTES - 8)); + expected.push_str(&chunk); + tx.send_async(chunk) + .await + .expect("pump should outlive the producer"); + max_queued = max_queued.max(tx.len()); + } + (expected, max_queued) + }); + + let mut received = String::with_capacity(CHUNKS * CHUNK_BYTES); + time::timeout( + Duration::from_secs(30), + pump_chunks(rx, async |payload: String| { + received.push_str(&payload); + // Emulate a busy JS event loop: each napi callback takes a while. + time::sleep(Duration::from_micros(500)).await; + true + }), + ) + .await + .expect("pump should finish once the producer hangs up"); + + let (expected, max_queued) = producer.await.expect("producer task"); + assert!( + max_queued <= BRIDGE_QUEUE_CHUNKS, + "bridge queue grew past its bound: {max_queued} chunks", + ); + assert_eq!(received.len(), expected.len(), "bytes were dropped or duplicated"); + assert_eq!(received, expected, "chunks must arrive losslessly and in order"); + } + + /// When the JS side dies (`forward` fails: threadsafe function aborted on + /// env teardown), the pump must drop its receiver so parked and future + /// sends fail fast — the pipe readers keep draining the child instead of + /// wedging it on a full bridge queue. + #[tokio::test(flavor = "multi_thread")] + async fn bridge_pump_death_disconnects_channel_without_blocking_senders() { + let (tx, rx) = flume::bounded::(4); + let pump = tokio::spawn(pump_chunks(rx, async |_payload: String| false)); + let producer = tokio::spawn(async move { + let mut disconnected = 0usize; + for _ in 0..64 { + if tx.send_async("x".repeat(1024)).await.is_err() { + disconnected += 1; + } + } + disconnected + }); + let disconnected = time::timeout(Duration::from_secs(5), producer) + .await + .expect("sends must not park once the consumer died") + .expect("producer task"); + assert!(disconnected > 0, "channel should disconnect after the pump stops"); + time::timeout(Duration::from_secs(5), pump) + .await + .expect("pump should exit after forward fails") + .expect("pump task"); + } mod child_session_action_tests { use pi_shell::{ChildSessionAction, child_session_action}; diff --git a/crates/pi-shell/src/shell.rs b/crates/pi-shell/src/shell.rs index 80634bf91..b3ebe92b8 100644 --- a/crates/pi-shell/src/shell.rs +++ b/crates/pi-shell/src/shell.rs @@ -1649,7 +1649,7 @@ async fn read_output( let pending = &buf[..it]; match str::from_utf8(pending) { Ok(text) => { - emit_chunk(text, on_chunk.as_ref()); + emit_chunk(text, on_chunk.as_ref()).await; it = 0; break; }, @@ -1658,7 +1658,7 @@ async fn read_output( if p > 0 { // SAFETY: [..p] is guaranteed valid UTF-8 by valid_up_to(). let text = unsafe { str::from_utf8_unchecked(&pending[..p]) }; - emit_chunk(text, on_chunk.as_ref()); + emit_chunk(text, on_chunk.as_ref()).await; // copy p..it to the beginning of the buffer buf.copy_within(p..it, 0); it -= p; @@ -1667,7 +1667,7 @@ async fn read_output( match err.error_len() { Some(p) => { // Invalid byte sequence: emit replacement and drop those bytes. - emit_chunk(REPLACEMENT, on_chunk.as_ref()); + emit_chunk(REPLACEMENT, on_chunk.as_ref()).await; // copy p..it to the beginning of the buffer buf.copy_within(p..it, 0); it -= p; @@ -1688,10 +1688,10 @@ async fn read_output( for chunk in buf[..it].utf8_chunks() { let valid = chunk.valid(); if !valid.is_empty() { - emit_chunk(valid, on_chunk.as_ref()); + emit_chunk(valid, on_chunk.as_ref()).await; } if !chunk.invalid().is_empty() { - emit_chunk(REPLACEMENT, on_chunk.as_ref()); + emit_chunk(REPLACEMENT, on_chunk.as_ref()).await; } } } @@ -1777,7 +1777,7 @@ async fn read_output_buffered( while !pending.is_empty() { match str::from_utf8(&pending) { Ok(text) => { - emit_chunk(text, Some(cb)); + emit_chunk(text, Some(cb)).await; pending.clear(); break; }, @@ -1786,12 +1786,12 @@ async fn read_output_buffered( if p > 0 { // SAFETY: [..p] is valid UTF-8 per valid_up_to(). let text = unsafe { str::from_utf8_unchecked(&pending[..p]) }; - emit_chunk(text, Some(cb)); + emit_chunk(text, Some(cb)).await; pending.drain(..p); } match err.error_len() { Some(skip) => { - emit_chunk(REPLACEMENT, Some(cb)); + emit_chunk(REPLACEMENT, Some(cb)).await; pending.drain(..skip); }, None => break, @@ -1807,10 +1807,10 @@ async fn read_output_buffered( for chunk in pending.utf8_chunks() { let valid = chunk.valid(); if !valid.is_empty() { - emit_chunk(valid, Some(cb)); + emit_chunk(valid, Some(cb)).await; } if !chunk.invalid().is_empty() { - emit_chunk(REPLACEMENT, Some(cb)); + emit_chunk(REPLACEMENT, Some(cb)).await; } } } @@ -1858,9 +1858,16 @@ fn read_nonblocking(file: &T, buf: &mut [u8]) -> io::Re } } -fn emit_chunk(text: &str, callback: Option<&Sender>) { +/// Forward one decoded chunk to the streaming callback, honouring channel +/// backpressure: on a bounded channel (the pi-natives JS bridge) the send +/// parks until the consumer frees a slot — which parks the pipe reader and, +/// transitively, the child on its stdout/stderr pipe — so a fast producer +/// can never buffer unbounded output in memory (#4078). A disconnected +/// receiver (consumer gone) fails immediately, so the pipe keeps draining +/// and the child never wedges on a full pipe. +async fn emit_chunk(text: &str, callback: Option<&Sender>) { if let Some(callback) = callback { - let _ = callback.send(text.to_string()); + let _ = callback.send_async(text.to_string()).await; } } @@ -4140,4 +4147,39 @@ replace = [{ pattern = "^.+$", replacement = "PWD" }] "builtin nohup masked SIGHUP like the external tool (output: {out:?})", ); } + + /// Regression for #4078: the JS bridge hands the pipe readers a *bounded* + /// chunk channel. With a consumer slower than the producer the readers + /// must park on `send_async` (backpressuring the child through its pipe) + /// rather than buffer unboundedly — and, unlike a drop-on-full design, + /// every produced byte must still reach the consumer. + #[cfg(unix)] + #[tokio::test(flavor = "multi_thread")] + async fn streaming_output_backpressures_on_bounded_channel_without_loss() { + const TOTAL_BYTES: usize = 1_048_576; + let (tx, rx) = flume::bounded::(4); + let options = ShellExecuteOptions { + command: format!("yes x | head -c {TOTAL_BYTES}"), + ..Default::default() + }; + let run = tokio::spawn(execute_shell(options, Some(tx), CancelToken::default())); + + let mut received = 0usize; + while let Ok(chunk) = rx.recv_async().await { + received += chunk.len(); + // Slow consumer: forces the bounded queue to fill and the readers + // to park between chunks. + time::sleep(Duration::from_micros(50)).await; + } + + let result = time::timeout(Duration::from_secs(30), run) + .await + .expect("command should finish despite backpressure") + .expect("run task should not panic") + .expect("execute should succeed"); + assert_eq!(result.exit_code, Some(0)); + assert!(!result.cancelled); + assert!(!result.timed_out); + assert_eq!(received, TOTAL_BYTES, "streamed bytes were dropped under backpressure"); + } } diff --git a/docs/config-usage.md b/docs/config-usage.md index 6606618c2..984c08ee7 100644 --- a/docs/config-usage.md +++ b/docs/config-usage.md @@ -79,6 +79,8 @@ A named profile (`omp --profile `, the `--alias` shortcut, or `OMP_PROFILE The relocation is uniform across the native provider (`builtin.ts`) and the generic `config.ts` helpers, so it covers slash commands, rules, prompts, instructions, hooks, tools, extensions, settings, skills, and MCP, plus the top-level `SYSTEM.md` / `RULES.md` / `AGENTS.md` files and runtime state (sessions, blobs, `agent.db`). A profile sees only its own OMP config, never the default profile's `~/.omp/agent`. +Keybindings are the one exception: a named profile merges the default profile's `~/.omp/agent/keybindings.*` under its own `~/.omp/profiles//agent/keybindings.*`, with the profile file overriding per binding ([#4867](https://github.com/can1357/oh-my-pi/issues/4867)). Keybindings describe the terminal/keyboard in front of the user, which doesn't change with the active profile, so user-level remaps keep working in every profile unless the profile explicitly overrides them. The inherited file is read-only for the profile process — legacy-format migration of the default profile's file only happens when the default profile itself runs. + The other source bases are not profile-scoped and load identically under every profile: the external-tool bases (`~/.claude`, `~/.codex`, `~/.gemini`) belong to those tools, and the project-level bases (`/.omp`, `/.claude`, ...) are keyed to the working directory. Throughout this document, read `~/.omp/agent` as shorthand for the active profile's agent directory. ## Important constraint diff --git a/docs/environment-variables.md b/docs/environment-variables.md index 9d323aad7..adf3d310c 100644 --- a/docs/environment-variables.md +++ b/docs/environment-variables.md @@ -49,6 +49,7 @@ These are consumed via `getEnvApiKey()` (`packages/ai/src/stream.ts`) unless not | `SYNTHETIC_API_KEY` | Synthetic auth | Using Synthetic models | | | `NVIDIA_API_KEY` | NVIDIA auth | Using `nvidia` provider | | | `NANO_GPT_API_KEY` | NanoGPT auth | Using `nanogpt` provider | | +| `NOVITA_API_KEY` | Novita auth | Using `novita` provider | | | `VENICE_API_KEY` | Venice auth | Using `venice` provider | | | `LITELLM_API_KEY` | LiteLLM auth | Using `litellm` provider | OpenAI-compatible LiteLLM proxy key | | `LM_STUDIO_API_KEY` | LM Studio auth (optional) | Using `lm-studio` provider with authenticated hosts | Local LM Studio usually runs without auth; any non-empty token works when a key is required | diff --git a/docs/extensions.md b/docs/extensions.md index 337a1d65c..7702c1767 100644 --- a/docs/extensions.md +++ b/docs/extensions.md @@ -142,7 +142,7 @@ Also exposed: - `deliverAs: "nextTurn"` — stored and injected on the next user prompt - `triggerTurn: true` — starts a turn when idle (also honored with `deliverAs: "nextTurn"`: idle prompts immediately; while streaming the queued message schedules an internal continuation) -`pi.sendUserMessage(content, { deliverAs })` always goes through prompt flow; while streaming it queues as steer/follow-up. +`pi.sendUserMessage(content, { deliverAs })` always goes through prompt flow. Omit `deliverAs` to start a normal prompt when idle; while streaming, omitted `deliverAs` queues the message as a steer. Set `deliverAs: "followUp"` to wait until the current run finishes. ## 2) Handler context (`ExtensionContext`) @@ -311,6 +311,7 @@ Supported: - dialogs: `select`, `confirm`, `input`, `editor` - input editing: `setEditorText`, `getEditorText`, `pasteToEditor`, `editor` +- autocomplete stacking: `addAutocompleteProvider(factory)` wraps the built-in editor provider (factories apply in registration order and re-apply on every slash-command refresh) - terminal title and working message (`setTitle`, `setWorkingMessage`) - notifications/status/editor text/terminal input/custom overlays - theme listing/loading by name (`setTheme` supports string names) @@ -334,7 +335,7 @@ Unsupported/no-op in RPC implementation: - `onTerminalInput` - `custom` -- `setFooter`, `setHeader`, `setEditorComponent` +- `setFooter`, `setHeader`, `setEditorComponent`, `addAutocompleteProvider` - `setWorkingMessage` - theme switching/loading (`setTheme` returns failure) - tool expansion controls are inert @@ -345,7 +346,7 @@ When no UI context is supplied to runner init, `ctx.hasUI` is `false` and method ### ACP mode -ACP installs an elicitation-bridged UI context (`createAcpExtensionUiContext` in `acp-agent.ts`). `ctx.hasUI` is `true` while only `select`/`confirm`/`input` round-trip (as ACP elicitations; defaults are returned when the client lacks the `elicitation.form` capability). The non-elicitation surface (widgets, editor, theming, terminal input) is stubbed no-op. +ACP installs an elicitation-bridged UI context (`createAcpExtensionUiContext` in `acp-agent.ts`). `ctx.hasUI` is `true` while only `select`/`confirm`/`input` round-trip (as ACP elicitations; defaults are returned when the client lacks the `elicitation.form` capability). The non-elicitation surface (widgets, editor, theming, terminal input, autocomplete stacking) is stubbed no-op. ## Session and state patterns diff --git a/docs/porting-from-pi-mono.md b/docs/porting-from-pi-mono.md index 427d47394..6234a940d 100644 --- a/docs/porting-from-pi-mono.md +++ b/docs/porting-from-pi-mono.md @@ -307,6 +307,7 @@ Our fork has architectural decisions that differ from upstream. **Do not port th | `FooterDataProvider` class | `StatusLineComponent` | Simpler, integrated status line | | `ctx.ui.setHeader()` / `ctx.ui.setFooter()` | No-op stubs in current extension contexts | Not currently wired to replace the TUI status/header UI | | `ctx.ui.setEditorComponent()` | Wired in interactive mode; no-op stubs in ACP/RPC/headless contexts | Custom editor replacement works in the interactive TUI; non-TUI runtimes keep stubs | +| `ctx.ui.addAutocompleteProvider()` | Wired in interactive mode; no-op stubs in ACP/RPC/headless contexts | Factory wrapping matches upstream; omp's editor has no custom `triggerCharacters`, so wrapped providers surface at the built-in trigger points | | `InteractiveModeOptions` options object | Positional constructor args (options type still exported) | Keep constructor signature; update the type when upstream adds fields | ### Component Naming diff --git a/docs/providers.md b/docs/providers.md index c5f65e009..cb2b4ec53 100644 --- a/docs/providers.md +++ b/docs/providers.md @@ -105,6 +105,7 @@ Each provider has one or more environment variables that supply a key when no st | `huggingface` | `HUGGINGFACE_HUB_TOKEN`, then `HF_TOKEN` | | `moonshot` | `MOONSHOT_API_KEY` | | `nanogpt` | `NANO_GPT_API_KEY` | +| `novita` | `NOVITA_API_KEY` | | `venice` | `VENICE_API_KEY` | | `vercel-ai-gateway` | `AI_GATEWAY_API_KEY` (also `VERCEL_AI_GATEWAY_API_KEY` for catalog discovery) | | `cloudflare-ai-gateway` | `CLOUDFLARE_AI_GATEWAY_API_KEY` | diff --git a/docs/sdk.md b/docs/sdk.md index a0ab6b403..c06dec3a4 100644 --- a/docs/sdk.md +++ b/docs/sdk.md @@ -215,8 +215,9 @@ Behavior: 1. optional command/template expansion (`/` commands, custom commands, file slash commands, prompt templates) 2. if currently streaming: - - requires `streamingBehavior: "steer" | "followUp"` - - queues instead of throwing work away + - `streamingBehavior: "steer" | "followUp"` chooses how `prompt()` queues + - extension `sendUserMessage(content)` defaults to steer when `deliverAs` is omitted + - queued messages are preserved instead of throwing work away 3. if idle: - validates model + API key - appends user message diff --git a/docs/session-operations-export-share-fork-resume.md b/docs/session-operations-export-share-fork-resume.md index 417827b1a..010512b69 100644 --- a/docs/session-operations-export-share-fork-resume.md +++ b/docs/session-operations-export-share-fork-resume.md @@ -172,7 +172,7 @@ Interactive `/fork` creates a new session from the current one and switches the 2. Flushes pending writes. 3. Calls `SessionManager.fork()`. 4. Copies artifacts directory from old session namespace to new namespace (best-effort; non-ENOENT copy failures are logged, not fatal). -5. Updates `agent.sessionId`. +5. Updates `agent.sessionId` and inherits the previous provider prompt-cache key unless an explicit prompt-cache key is already pinned. 6. Emits `session_switch` with `reason: "fork"`. `SessionManager.fork()` behavior: @@ -184,6 +184,7 @@ Interactive `/fork` creates a new session from the current one and switches the - new timestamp - `cwd` unchanged - `parentSession` set to previous session id + - `providerPromptCacheKey` set to the previous header's inherited key, or the previous session id when none was pinned - Keeps all non-header entries unchanged in the new file. ### Non-persistent behavior @@ -200,6 +201,9 @@ Startup `--fork` is resolved before normal session creation: 2. Path-like values (`/`, `\`, or `.jsonl`) call `SessionManager.forkFrom(path, cwd, sessionDir)`. 3. Other values resolve via `resolveResumableSession(...)`: local sessions first, then global search when `sessionDir` is not forced. Matching accepts lowercased session id prefixes, full JSONL filename prefixes, and timestamp-stripped filename id suffixes. 4. The forked file is created in the current cwd/session-dir scope and becomes the active session manager for startup. +5. Full-context forks automatically seed `providerPromptCacheKey` from the source header's inherited key, falling back to the source session id. Startup drops that automatic inheritance when `--model`, `--thinking`, `--system-prompt`, `--append-system-prompt`, `--tools`, or `--no-tools` changes the provider route or prompt/tool shape. + +Use `--prompt-cache-key ` to pin the provider prompt-cache identity explicitly and independently from both the OMP session id and `--provider-session-id`. `--provider-session-id` continues to control provider session/routing headers and sticky credential selection; `--prompt-cache-key` controls the OpenAI Responses `prompt_cache_key` payload where supported. ## Resume and continue diff --git a/package.json b/package.json index aaee74002..a63330c5c 100644 --- a/package.json +++ b/package.json @@ -25,18 +25,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.3.12", - "@oh-my-pi/omp-stats": "16.3.12", - "@oh-my-pi/pi-agent-core": "16.3.12", - "@oh-my-pi/pi-ai": "16.3.12", - "@oh-my-pi/pi-catalog": "16.3.12", - "@oh-my-pi/pi-coding-agent": "16.3.12", - "@oh-my-pi/pi-mnemopi": "16.3.12", - "@oh-my-pi/pi-natives": "16.3.12", - "@oh-my-pi/pi-tui": "16.3.12", - "@oh-my-pi/pi-utils": "16.3.12", - "@oh-my-pi/pi-wire": "16.3.12", - "@oh-my-pi/snapcompact": "16.3.12", + "@oh-my-pi/hashline": "16.3.15", + "@oh-my-pi/omp-stats": "16.3.15", + "@oh-my-pi/pi-agent-core": "16.3.15", + "@oh-my-pi/pi-ai": "16.3.15", + "@oh-my-pi/pi-catalog": "16.3.15", + "@oh-my-pi/pi-coding-agent": "16.3.15", + "@oh-my-pi/pi-mnemopi": "16.3.15", + "@oh-my-pi/pi-natives": "16.3.15", + "@oh-my-pi/pi-tui": "16.3.15", + "@oh-my-pi/pi-utils": "16.3.15", + "@oh-my-pi/pi-wire": "16.3.15", + "@oh-my-pi/snapcompact": "16.3.15", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 04b655eeb..cf30037fd 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed remote compaction for Codex Responses Lite models (GPT-5.6 family): both the V1 `/responses/compact` request and the V2 `compaction_trigger` stream now apply the lite rewrite (instructions as an input item, no top-level `instructions`/`tools`, `all_turns` reasoning replay on V2) and send the `x-openai-internal-codex-responses-lite` header, matching codex-rs routing compaction through `build_responses_request`. + ## [16.3.12] - 2026-07-08 ### Added diff --git a/packages/agent/package.json b/packages/agent/package.json index 5d945c29a..0e5b1961c 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "16.3.12", + "version": "16.3.15", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/agent/src/compaction/compaction-v2-streaming.ts b/packages/agent/src/compaction/compaction-v2-streaming.ts index 4db13bd22..14a2318f8 100644 --- a/packages/agent/src/compaction/compaction-v2-streaming.ts +++ b/packages/agent/src/compaction/compaction-v2-streaming.ts @@ -7,10 +7,16 @@ * compaction item as replacement history. */ -import type { Api, FetchImpl, Model } from "@oh-my-pi/pi-ai"; +import type { Api, CodexCompactionContext, FetchImpl, Model, ProviderSessionState } from "@oh-my-pi/pi-ai"; import { isTransientStatus, ProviderHttpError } from "@oh-my-pi/pi-ai/error"; +import { applyCodexResponsesLiteShape } from "@oh-my-pi/pi-ai/providers/openai-codex/request-transformer"; import { - getOpenAIResponsesPromptCacheKey, + createOpenAICodexCompactionRequestContext, + createOpenAICodexCompatibilityMetadata, + type OpenAICodexCompatibilityMetadata, +} from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; +import { + getOpenAIPromptCacheKey, getOpenAIResponsesRoutingSessionId, parseAzureDeploymentNameMap, resolveOpenAIRequestSetup, @@ -219,6 +225,8 @@ export async function requestCompactionV2Streaming( fetch?: FetchImpl; timeoutMs?: number; retryWait?: (delayMs: number, signal?: AbortSignal) => Promise; + providerSessionState?: Map; + codexCompaction?: CodexCompactionContext; }, ): Promise { const endpoint = getCompactionV2Endpoint(model); @@ -228,12 +236,32 @@ export async function requestCompactionV2Streaming( const fetchImpl = options?.fetch ?? globalThis.fetch; const retryWait = options?.retryWait ?? ((delayMs: number) => Bun.sleep(delayMs)); + const isCodexResponses = compactionV2Api(model) === "openai-codex-responses" || model.provider === "openai-codex"; + const codexMetadata = isCodexResponses + ? createOpenAICodexCompatibilityMetadata({ + sessionId: request.sessionId, + providerSessionState: options?.providerSessionState, + requestKind: "compaction", + compaction: createOpenAICodexCompactionRequestContext({ + context: options?.codexCompaction, + implementation: "responses_compaction_v2", + }), + }) + : undefined; let lastError: Error | undefined; for (let attempt = 0; attempt <= V2_COMPACTION_MAX_RETRIES; attempt++) { const timeoutSignal = withRequestTimeout(signal, options?.timeoutMs ?? V2_COMPACTION_TIMEOUT_MS); try { - return await attemptCompactionV2Streaming(endpoint, apiKey, model, request, fetchImpl, timeoutSignal); + return await attemptCompactionV2Streaming( + endpoint, + apiKey, + model, + request, + fetchImpl, + timeoutSignal, + codexMetadata, + ); } catch (err) { const error = err instanceof Error ? err : new Error(String(err)); if (signal?.aborted) throw error; @@ -264,25 +292,41 @@ async function attemptCompactionV2Streaming( request: CompactionV2Request, fetchImpl: FetchImpl, signal?: AbortSignal, + codexMetadata?: OpenAICodexCompatibilityMetadata, ): Promise { // Faithful to Codex: append the compaction trigger as the final input item // of an otherwise-normal Responses request, then stream the result. `store` // stays false — compaction must never persist a server-side response object. const cacheOptions = { sessionId: request.sessionId, promptCacheKey: request.promptCacheKey }; - const promptCacheKey = getOpenAIResponsesPromptCacheKey(cacheOptions); + const promptCacheKey = getOpenAIPromptCacheKey(cacheOptions); const body: Record = { model: request.model, input: [...request.input, COMPACTION_TRIGGER_ITEM], instructions: request.instructions, stream: true, store: false, - ...(request.reasoning ? { reasoning: request.reasoning, include: ["reasoning.encrypted_content"] } : {}), + ...(request.reasoning + ? { + // Lite implies gpt-5.4+, where codex-rs sends `all_turns` replay. + reasoning: model.useResponsesLite ? { ...request.reasoning, context: "all_turns" } : request.reasoning, + include: ["reasoning.encrypted_content"], + } + : {}), ...(promptCacheKey ? { prompt_cache_key: promptCacheKey } : {}), ...(request.tools && request.tools.length > 0 ? { tools: request.tools, tool_choice: "auto" } : {}), }; + if (codexMetadata) { + body.client_metadata = codexMetadata.clientMetadata; + } + // Responses Lite models take the same rewrite on the compaction stream: + // instructions/tools ride as input items (codex-rs `compact_remote_v2` + // builds through `build_responses_request`). + if (model.useResponsesLite) { + applyCodexResponsesLiteShape(body); + } const response = await fetchImpl(endpoint, { method: "POST", - headers: buildCompactionV2Headers(model, apiKey, request), + headers: buildCompactionV2Headers(model, apiKey, request, codexMetadata), body: JSON.stringify(body), signal, }); @@ -307,11 +351,16 @@ async function attemptCompactionV2Streaming( return collectCompactionV2Output(response, request); } -function buildCompactionV2Headers(model: Model, apiKey: string, request: CompactionV2Request): Record { +function buildCompactionV2Headers( + model: Model, + apiKey: string, + request: CompactionV2Request, + codexMetadata?: OpenAICodexCompatibilityMetadata, +): Record { const api = compactionV2Api(model); const cacheOptions = { sessionId: request.sessionId, promptCacheKey: request.promptCacheKey }; const routingSessionId = getOpenAIResponsesRoutingSessionId(cacheOptions); - const promptCacheSessionId = getOpenAIResponsesPromptCacheKey(cacheOptions); + const promptCacheSessionId = getOpenAIPromptCacheKey(cacheOptions); const headers: Record = api === "azure-openai-responses" ? { @@ -338,7 +387,11 @@ function buildCompactionV2Headers(model: Model, apiKey: string, request: Compact } headers[OPENAI_HEADERS.BETA] = OPENAI_HEADER_VALUES.BETA_RESPONSES; headers[OPENAI_HEADERS.ORIGINATOR] = OPENAI_HEADER_VALUES.ORIGINATOR_CODEX; + if (model.useResponsesLite) { + headers[OPENAI_HEADERS.RESPONSES_LITE] = "true"; + } } + if (codexMetadata) Object.assign(headers, codexMetadata.headers); return headers; } diff --git a/packages/agent/src/compaction/compaction.ts b/packages/agent/src/compaction/compaction.ts index 36b8feadd..772f0b232 100644 --- a/packages/agent/src/compaction/compaction.ts +++ b/packages/agent/src/compaction/compaction.ts @@ -9,18 +9,21 @@ import { type Api, type ApiKey, type AssistantMessage, + type CodexCompactionContext, type Context, Effort, type FetchImpl, type Message, type MessageAttribution, type Model, + type ProviderSessionState, type SimpleStreamOptions, type Tool, type Usage, withAuth, } from "@oh-my-pi/pi-ai"; import { ProviderHttpError } from "@oh-my-pi/pi-ai/error"; +import { createOpenAICodexCompactionRequestContext } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; import { convertTools } from "@oh-my-pi/pi-ai/providers/openai-responses"; import { buildResponsesInput, resolveOpenAICompatPolicy } from "@oh-my-pi/pi-ai/providers/openai-shared"; import { preferredDialect } from "@oh-my-pi/pi-catalog/identity"; @@ -736,6 +739,10 @@ export interface SummaryOptions { sessionId?: string; /** Prompt-cache key for remote compaction transports that support provider prefix caching. */ promptCacheKey?: string; + /** Mutable provider state used to keep Codex compaction on the live session identity. */ + providerSessionState?: Map; + /** Classification shared by every provider request in this logical compaction. */ + codexCompaction?: CodexCompactionContext; /** Provider-visible tools for remote compaction transports that replay native tool history. */ tools?: Tool[]; /** Optional fetch implementation threaded into remote compaction calls. */ @@ -755,6 +762,13 @@ export interface SummaryOptions { ) => Promise; } +function localCodexCompaction(options: SummaryOptions | undefined) { + return createOpenAICodexCompactionRequestContext({ + context: options?.codexCompaction, + implementation: "responses", + }); +} + function formatPreviousSnapcompactArchive(archiveText: string): string { return prompt.render(snapcompactArchiveContextPrompt, { archiveText }); } @@ -844,6 +858,11 @@ export async function generateSummary( reasoning: resolveCompactionEffort(model, options?.thinkingLevel), initiatorOverride: options?.initiatorOverride, metadata: options?.metadata, + fetch: options?.fetch, + sessionId: options?.sessionId, + promptCacheKey: options?.promptCacheKey, + providerSessionState: options?.providerSessionState, + codexCompaction: localCodexCompaction(options), }, { telemetry: options?.telemetry, oneshotKind: "compaction_summary", completeImpl: options?.completeImpl }, ); @@ -1047,6 +1066,11 @@ async function generateShortSummary( reasoning: resolveCompactionEffort(model, options?.thinkingLevel), initiatorOverride: options?.initiatorOverride, metadata: options?.metadata, + fetch: options?.fetch, + sessionId: options?.sessionId, + promptCacheKey: options?.promptCacheKey, + providerSessionState: options?.providerSessionState, + codexCompaction: localCodexCompaction(options), }, { telemetry: options?.telemetry, oneshotKind: "compaction_short_summary", completeImpl: options?.completeImpl }, ); @@ -1317,6 +1341,8 @@ export async function compact( thinkingLevel: options?.thinkingLevel, sessionId: options?.sessionId, promptCacheKey: options?.promptCacheKey, + providerSessionState: options?.providerSessionState, + codexCompaction: options?.codexCompaction, tools: options?.tools, fetch: options?.fetch, completeImpl: options?.completeImpl, @@ -1375,7 +1401,12 @@ export async function compact( ); const remote = await withAuth( apiKey, - key => requestCompactionV2Streaming(model, key, request, signal, { fetch: summaryOptions.fetch }), + key => + requestCompactionV2Streaming(model, key, request, signal, { + fetch: summaryOptions.fetch, + providerSessionState: summaryOptions.providerSessionState, + codexCompaction: summaryOptions.codexCompaction, + }), { signal }, ); preserveData = { ...(preserveData ?? {}), ...storeCompactionV2PreserveData(remote, model) }; @@ -1419,7 +1450,12 @@ export async function compact( remoteHistory, summaryOptions.remoteInstructions ?? SUMMARIZATION_SYSTEM_PROMPT, signal, - { fetch: summaryOptions.fetch }, + { + fetch: summaryOptions.fetch, + sessionId: summaryOptions.sessionId, + providerSessionState: summaryOptions.providerSessionState, + codexCompaction: summaryOptions.codexCompaction, + }, ), { signal }, ); @@ -1495,16 +1531,9 @@ export async function compact( const shortSummary = usedRemoteCompaction ? "Remote compaction" : await generateShortSummary(recentMessages, summary, model, reserveTokens, apiKey, signal, { + ...summaryOptions, extraContext: options?.extraContext, - remoteEndpoint: summaryOptions.remoteEndpoint, - initiatorOverride: summaryOptions.initiatorOverride, - metadata: summaryOptions.metadata, - telemetry: summaryOptions.telemetry, - // Same propagation as summaryOptions above — generateShortSummary - // resolves its own reasoning via resolveCompactionEffort. thinkingLevel: options?.thinkingLevel, - fetch: summaryOptions.fetch, - completeImpl: summaryOptions.completeImpl, }); // Compute file lists and append to summary @@ -1567,6 +1596,11 @@ async function generateTurnPrefixSummary( reasoning: resolveCompactionEffort(model, options?.thinkingLevel), initiatorOverride: options?.initiatorOverride, metadata: options?.metadata, + fetch: options?.fetch, + sessionId: options?.sessionId, + promptCacheKey: options?.promptCacheKey, + providerSessionState: options?.providerSessionState, + codexCompaction: localCodexCompaction(options), }, { telemetry: options?.telemetry, oneshotKind: "compaction_turn_prefix", completeImpl: options?.completeImpl }, ); diff --git a/packages/agent/src/compaction/openai.ts b/packages/agent/src/compaction/openai.ts index 2695e4466..fbac0a8b7 100644 --- a/packages/agent/src/compaction/openai.ts +++ b/packages/agent/src/compaction/openai.ts @@ -16,9 +16,22 @@ */ import { ProviderHttpError } from "@oh-my-pi/pi-ai/error"; +import { applyCodexResponsesLiteShape } from "@oh-my-pi/pi-ai/providers/openai-codex/request-transformer"; +import { + createOpenAICodexCompactionRequestContext, + createOpenAICodexCompatibilityMetadata, +} from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; import { parseAzureDeploymentNameMap, parseTextSignature } from "@oh-my-pi/pi-ai/providers/openai-shared"; import { transformMessages } from "@oh-my-pi/pi-ai/providers/transform-messages"; -import type { Api, AssistantMessage, FetchImpl, Message, Model } from "@oh-my-pi/pi-ai/types"; +import type { + Api, + AssistantMessage, + CodexCompactionContext, + FetchImpl, + Message, + Model, + ProviderSessionState, +} from "@oh-my-pi/pi-ai/types"; import { getOpenAIResponsesHistoryItems, getOpenAIResponsesHistoryPayload, @@ -460,7 +473,13 @@ export async function requestOpenAiRemoteCompaction( compactInput: Array>, instructions: string, signal?: AbortSignal, - opts?: { fetch?: FetchImpl; timeoutMs?: number }, + opts?: { + fetch?: FetchImpl; + timeoutMs?: number; + sessionId?: string; + providerSessionState?: Map; + codexCompaction?: CodexCompactionContext; + }, ): Promise { const endpoint = resolveOpenAiCompactEndpoint(model); const requestModel = resolveOpenAiCompactModel(model); @@ -473,6 +492,8 @@ export async function requestOpenAiRemoteCompaction( instructions, }; const isAzureOpenAiResponses = (model.remoteCompaction?.api ?? model.api) === "azure-openai-responses"; + const isCodexResponses = + model.provider === "openai-codex" || (model.remoteCompaction?.api ?? model.api) === "openai-codex-responses"; const headers: Record = isAzureOpenAiResponses ? { "content-type": "application/json", @@ -486,13 +507,33 @@ export async function requestOpenAiRemoteCompaction( }; // Codex endpoints require additional auth headers - if (model.provider === "openai-codex") { + if (isCodexResponses) { const accountId = getCodexAccountId(apiKey); if (accountId) { headers[OPENAI_HEADERS.ACCOUNT_ID] = accountId; } headers[OPENAI_HEADERS.BETA] = OPENAI_HEADER_VALUES.BETA_RESPONSES; headers[OPENAI_HEADERS.ORIGINATOR] = OPENAI_HEADER_VALUES.ORIGINATOR_CODEX; + Object.assign( + headers, + createOpenAICodexCompatibilityMetadata({ + sessionId: opts?.sessionId, + providerSessionState: opts?.providerSessionState, + requestKind: "compaction", + compaction: createOpenAICodexCompactionRequestContext({ + context: opts?.codexCompaction, + implementation: "responses_compact", + }), + includeInstallationHeader: true, + }).headers, + ); + // Responses Lite models take the same rewrite on `/responses/compact`: + // instructions ride as an input item and the lite marker header is set + // (codex-rs routes compaction through `build_responses_request`). + if (model.useResponsesLite) { + applyCodexResponsesLiteShape(request); + headers[OPENAI_HEADERS.RESPONSES_LITE] = "true"; + } } const response = await (opts?.fetch ?? fetch)(endpoint, { diff --git a/packages/agent/test/remote-compaction.test.ts b/packages/agent/test/remote-compaction.test.ts index c2cc50dd1..78d4cb97b 100644 --- a/packages/agent/test/remote-compaction.test.ts +++ b/packages/agent/test/remote-compaction.test.ts @@ -1,4 +1,4 @@ -import { afterEach, describe, expect, test, vi } from "bun:test"; +import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test"; import { type CompactionPreparation, compact, @@ -18,10 +18,36 @@ import { shouldUseOpenAiRemoteCompaction, } from "@oh-my-pi/pi-agent-core/compaction/openai"; import * as ai from "@oh-my-pi/pi-ai"; -import type { AssistantMessage, FetchImpl, Model, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; +import { getOpenAICodexTransportDetails } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; +import type { + AssistantMessage, + CodexCompactionContext, + FetchImpl, + Model, + ProviderSessionState, + ToolResultMessage, +} from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; -import { isRecord } from "@oh-my-pi/pi-utils"; +import * as piUtils from "@oh-my-pi/pi-utils"; + +const { isRecord } = piUtils; +const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001"; +const TEST_CODEX_COMPACTION: CodexCompactionContext = { + operationId: "compaction-operation-1", + trigger: "auto", + reason: "context_limit", + phase: "pre_turn", + strategy: "memento", +}; + +beforeEach(() => { + vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID); +}); + +afterEach(() => { + vi.restoreAllMocks(); +}); function makeOpenAiModel(overrides: Partial> = {}): Model<"openai-responses"> { return buildModel({ @@ -393,6 +419,396 @@ describe("requestCompactionV2Streaming", () => { }); }); +describe("Responses Lite remote compaction", () => { + function makeCodexLiteModel( + overrides: Partial> = {}, + ): Model<"openai-codex-responses"> { + return buildModel({ + id: "gpt-5.6-terra", + name: "GPT-5.6 Terra", + api: "openai-codex-responses", + provider: "openai-codex", + baseUrl: "https://chatgpt.example/backend-api", + reasoning: true, + preferWebsockets: false, + input: ["text", "image"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 372000, + maxTokens: 128000, + useResponsesLite: true, + remoteCompaction: { enabled: true, api: "openai-codex-responses", v2StreamingEnabled: true }, + ...overrides, + }); + } + + interface CapturedLiteRequest { + instructions?: unknown; + tools?: unknown; + input?: Array>; + client_metadata?: unknown; + } + + interface CapturedLiteExchange { + body: CapturedLiteRequest; + headers: Headers; + } + + function parseCodexTurnMetadata(value: unknown): Record { + if (typeof value !== "string") throw new Error("expected x-codex-turn-metadata"); + const parsed: unknown = JSON.parse(value); + if (!isRecord(parsed)) throw new Error("expected Codex turn metadata object"); + return parsed; + } + + function captureLite(init: RequestInit | undefined): CapturedLiteExchange { + if (!init?.headers || init.headers instanceof Headers || Array.isArray(init.headers)) { + throw new Error("Expected remote compaction to send headers as a plain object"); + } + return { + body: JSON.parse(String(init.body)) as CapturedLiteRequest, + headers: new Headers(init.headers), + }; + } + + function captureStreamLite(init: RequestInit | undefined): CapturedLiteExchange { + if (!init?.headers) throw new Error("Expected local compaction request headers"); + return { + body: JSON.parse(String(init.body)) as CapturedLiteRequest, + headers: new Headers(init.headers), + }; + } + + test("V1 compaction sends the lite header and input-item instructions", async () => { + const model = makeCodexLiteModel(); + let captured: CapturedLiteExchange | undefined; + const fetchMock: FetchImpl = async (_input, init) => { + captured = captureLite(init); + return Response.json({ output: [{ type: "compaction", encrypted_content: "enc" }] }); + }; + + await requestOpenAiRemoteCompaction( + model, + "test-key", + [{ type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }], + "compact instructions", + undefined, + { + fetch: fetchMock, + sessionId: "codex-compaction-session", + providerSessionState: new Map(), + codexCompaction: TEST_CODEX_COMPACTION, + }, + ); + + expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); + expect(captured?.body.instructions).toBeUndefined(); + expect(captured?.body.tools).toBeUndefined(); + expect(captured?.body.client_metadata).toBeUndefined(); + expect(captured?.headers.get("x-codex-installation-id")).toBe(TEST_INSTALLATION_ID); + expect(captured?.headers.get("session-id")).toBe("codex-compaction-session"); + const v1TurnMetadata = parseCodexTurnMetadata(captured?.headers.get("x-codex-turn-metadata")); + expect(v1TurnMetadata.request_kind).toBe("compaction"); + expect(v1TurnMetadata.compaction).toEqual({ + trigger: "auto", + reason: "context_limit", + implementation: "responses_compact", + phase: "pre_turn", + strategy: "memento", + }); + expect(captured?.body.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools: [] }); + expect(captured?.body.input?.[1]).toEqual({ + type: "message", + role: "developer", + content: [{ type: "input_text", text: "compact instructions" }], + }); + }); + + test("V2 streaming compaction applies the lite rewrite and keeps the trigger last", async () => { + const model = makeCodexLiteModel(); + const request = buildCompactionV2Request( + model, + [{ type: "message", role: "user", content: [{ type: "input_text", text: "real user" }] }], + "compact instructions", + { sessionId: "codex-compaction-session" }, + ); + let captured: CapturedLiteExchange | undefined; + const fetchMock: FetchImpl = async (_input, init) => { + captured = captureLite(init); + return sseResponse([ + { + type: "response.output_item.done", + output_index: 0, + item: { type: "compaction", encrypted_content: "enc" }, + }, + { type: "response.completed", response: { usage: { input_tokens: 1, output_tokens: 1, total_tokens: 2 } } }, + ]); + }; + + expect(shouldUseCompactionV2Streaming(model)).toBe(true); + await requestCompactionV2Streaming(model, "test-key", request, undefined, { + fetch: fetchMock, + providerSessionState: new Map(), + codexCompaction: TEST_CODEX_COMPACTION, + }); + + expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); + expect(captured?.body.instructions).toBeUndefined(); + expect(captured?.body.tools).toBeUndefined(); + if (!isRecord(captured?.body.client_metadata)) throw new Error("expected V2 client_metadata"); + const v2ClientMetadata = captured.body.client_metadata; + const v2TurnMetadata = parseCodexTurnMetadata(v2ClientMetadata["x-codex-turn-metadata"]); + expect(captured.headers.get("x-codex-installation-id")).toBeNull(); + expect(v2ClientMetadata["x-codex-installation-id"]).toBe(TEST_INSTALLATION_ID); + expect(v2ClientMetadata.session_id).toBe(captured.headers.get("session-id")); + expect(v2ClientMetadata.thread_id).toBe(captured.headers.get("thread-id")); + expect(v2TurnMetadata.request_kind).toBe("compaction"); + expect(v2TurnMetadata.compaction).toEqual({ + trigger: "auto", + reason: "context_limit", + implementation: "responses_compaction_v2", + phase: "pre_turn", + strategy: "memento", + }); + expect(captured?.body.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools: [] }); + expect(captured?.body.input?.[1]).toEqual({ + type: "message", + role: "developer", + content: [{ type: "input_text", text: "compact instructions" }], + }); + expect(captured?.body.input?.at(-1)).toEqual({ type: "compaction_trigger" }); + }); + + test("compact fan-out keeps local Codex summaries on one classified turn", async () => { + const model = makeCodexLiteModel(); + const captured: CapturedLiteExchange[] = []; + const fetchMock: FetchImpl = async (_input, init) => { + captured.push(captureStreamLite(init)); + return sseResponse([ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "message", id: "msg_summary", role: "assistant", status: "in_progress", content: [] }, + }, + { + type: "response.content_part.added", + output_index: 0, + content_index: 0, + part: { type: "output_text", text: "" }, + }, + { type: "response.output_text.delta", output_index: 0, content_index: 0, delta: "local summary" }, + { + type: "response.output_item.done", + output_index: 0, + item: { + type: "message", + id: "msg_summary", + role: "assistant", + status: "completed", + content: [{ type: "output_text", text: "local summary" }], + }, + }, + { + type: "response.completed", + response: { + status: "completed", + usage: { + input_tokens: 8, + output_tokens: 2, + total_tokens: 10, + input_tokens_details: { cached_tokens: 0 }, + }, + }, + }, + ]); + }; + const preparation: CompactionPreparation = { + firstKeptEntryId: "kept-1", + messagesToSummarize: [{ role: "user", content: "long history", timestamp: 1 }], + turnPrefixMessages: [], + recentMessages: [{ role: "user", content: "recent", timestamp: 2 }], + isSplitTurn: false, + tokensBefore: 100_000, + fileOps: createFileOps(), + settings: { + ...DEFAULT_COMPACTION_SETTINGS, + remoteEnabled: false, + remoteStreamingV2Enabled: false, + }, + }; + + const result = await compact(preparation, model, "test-key", undefined, undefined, { + fetch: fetchMock, + sessionId: "codex-compaction-session", + providerSessionState: new Map(), + codexCompaction: TEST_CODEX_COMPACTION, + }); + + expect(result.summary).toContain("local summary"); + expect(captured).toHaveLength(2); + const turnIds: string[] = []; + for (const exchange of captured) { + if (!isRecord(exchange.body.client_metadata)) throw new Error("expected local client_metadata"); + const clientMetadata = exchange.body.client_metadata; + const turnMetadata = parseCodexTurnMetadata(clientMetadata["x-codex-turn-metadata"]); + expect(exchange.headers.get("x-codex-installation-id")).toBeNull(); + expect(clientMetadata["x-codex-installation-id"]).toBe(TEST_INSTALLATION_ID); + expect(turnMetadata.request_kind).toBe("compaction"); + expect(turnMetadata.compaction).toEqual({ + trigger: "auto", + reason: "context_limit", + implementation: "responses", + phase: "pre_turn", + strategy: "memento", + }); + if (typeof turnMetadata.turn_id !== "string") throw new Error("expected Codex turn id"); + turnIds.push(turnMetadata.turn_id); + } + expect(new Set(turnIds).size).toBe(1); + }); + + test("local Codex compaction isolates and closes transient websocket sessions", async () => { + const originalWebSocket = global.WebSocket; + const sockets: AgentCompactionWebSocket[] = []; + let responseCount = 0; + + class AgentCompactionWebSocket { + static readonly CONNECTING = 0; + static readonly OPEN = 1; + static readonly CLOSING = 2; + static readonly CLOSED = 3; + + readyState = AgentCompactionWebSocket.CONNECTING; + binaryType: "blob" | "arraybuffer" | "nodebuffer" = "blob"; + onopen: ((event: Event) => void) | null = null; + onmessage: ((event: MessageEvent) => void) | null = null; + onerror: ((event: Event) => void) | null = null; + onclose: ((event: Event) => void) | null = null; + readonly handshakeHeaders = { + "x-codex-turn-state": `agent-compaction-state-${sockets.length}`, + }; + + constructor( + readonly url: string, + readonly options?: { headers?: Record }, + ) { + sockets.push(this); + queueMicrotask(() => { + this.readyState = AgentCompactionWebSocket.OPEN; + this.onopen?.(new Event("open")); + }); + } + + send(_data: string): void { + responseCount += 1; + const responseId = `response-${responseCount}`; + const messageId = `message-${responseCount}`; + const text = sockets[0] === this ? "main response" : "local summary"; + const events: Record[] = [ + { + type: "response.output_item.added", + item: { type: "message", id: messageId, role: "assistant", status: "in_progress", content: [] }, + }, + { type: "response.content_part.added", part: { type: "output_text", text: "" } }, + { type: "response.output_text.delta", delta: text }, + { + type: "response.output_item.done", + item: { + type: "message", + id: messageId, + role: "assistant", + status: "completed", + content: [{ type: "output_text", text }], + }, + }, + { + type: "response.done", + response: { + id: responseId, + status: "completed", + usage: { + input_tokens: 8, + output_tokens: 2, + total_tokens: 10, + input_tokens_details: { cached_tokens: 0 }, + }, + }, + }, + ]; + for (const event of events) { + this.onmessage?.({ data: JSON.stringify(event) } as MessageEvent); + } + } + + close(): void { + this.readyState = AgentCompactionWebSocket.CLOSED; + } + } + + const providerSessionState = new Map(); + try { + global.WebSocket = AgentCompactionWebSocket as unknown as typeof WebSocket; + const model = makeCodexLiteModel({ preferWebsockets: true }); + const sessionId = "agent-compaction-isolation"; + const fetchMock: FetchImpl = async () => { + throw new Error("Codex websocket compaction unexpectedly used SSE"); + }; + const main = await ai + .streamSimple( + model, + { + systemPrompt: ["You are a helpful assistant."], + messages: [{ role: "user", content: "Start the turn", timestamp: Date.now() }], + }, + { apiKey: "test-key", fetch: fetchMock, sessionId, providerSessionState }, + ) + .result(); + expect(main.stopReason).toBe("stop"); + expect(sockets).toHaveLength(1); + expect(sockets[0]?.readyState).toBe(AgentCompactionWebSocket.OPEN); + + const preparation: CompactionPreparation = { + firstKeptEntryId: "kept-1", + messagesToSummarize: [{ role: "user", content: "long history", timestamp: 1 }], + turnPrefixMessages: [], + recentMessages: [{ role: "user", content: "recent", timestamp: 2 }], + isSplitTurn: false, + tokensBefore: 100_000, + fileOps: createFileOps(), + settings: { + ...DEFAULT_COMPACTION_SETTINGS, + remoteEnabled: false, + remoteStreamingV2Enabled: false, + }, + }; + const result = await compact(preparation, model, "test-key", undefined, undefined, { + fetch: fetchMock, + sessionId, + providerSessionState, + codexCompaction: TEST_CODEX_COMPACTION, + }); + + expect(result.summary).toContain("local summary"); + expect(sockets).toHaveLength(3); + expect(sockets[0]?.readyState).toBe(AgentCompactionWebSocket.OPEN); + expect(sockets[1]?.readyState).toBe(AgentCompactionWebSocket.CLOSED); + expect(sockets[2]?.readyState).toBe(AgentCompactionWebSocket.CLOSED); + expect( + getOpenAICodexTransportDetails(model, { + sessionId, + providerSessionState, + }), + ).toMatchObject({ + websocketConnected: true, + hasTurnState: true, + }); + } finally { + for (const state of providerSessionState.values()) state.close(); + providerSessionState.clear(); + global.WebSocket = originalWebSocket; + } + }); +}); + test("uses configured OpenAI-compatible compaction for custom providers", async () => { const model = makeOpenAiModel({ provider: "cliproxy-codex", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index d30b103e4..e90b9e3db 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -5,12 +5,74 @@ ### Fixed - Fixed xAI SuperGrok multi-account rotation when an account returns HTTP 403 `run out of credits` / `personal-team-blocked:spending-limit`. That account-local cap is now classified as a usage limit so `streamSimple` auth-retry and `rotateSessionCredential` switch to a sibling `xai-oauth` credential instead of sticking to the exhausted account. +### Added + +- Added model-driven Codex Responses Lite: `responsesLite` now defaults to the catalog `useResponsesLite` flag (codex-rs `use_responses_lite`, set on the GPT-5.6 family), so lite requests are sent without per-call opt-in. +- Added the full Responses Lite wire contract: lite requests move tools into a leading `{type: "additional_tools", role: "developer"}` input item and the base instructions into a developer message, omit top-level `instructions`/`tools`, and force `parallel_tool_calls: false`, mirroring codex-rs `build_responses_request`. +- Added concurrent reasoning summaries on Codex Responses: requests with a reasoning summary send `stream_options: { reasoning_summary_delivery: "sequential_cutoff" }`, and the stream decoder consumes the matching atomic `response.reasoning_summary_text.done` events (resolved by `item_id`/`output_index`, stale dones dropped, incremental `.delta`/`.part.*` events ignored under the cutoff contract). The cutoff gate reads the post-`onPayload` wire body on both transports, and `response.reasoning_summary_text.done` now counts as websocket watchdog progress. +- Added Novita API-key login with authenticated key validation and `NOVITA_API_KEY` discovery ([#4917](https://github.com/can1357/oh-my-pi/pull/4917) by [@jason-wu-ai](https://github.com/jason-wu-ai)). + +### Changed + +- Refactored Responses Lite transport to move tools and instructions into input items +- Updated Responses Lite to force parallel tool calling off and strip image detail +- Standardized Responses Lite activation via model-level catalog flags + +- Recognized Pro Lite as a paid plan tier for OpenAI Codex models +- Changed Responses Lite image handling to match current codex-rs: a lite request containing input images now stays on the lite transport with image `detail` stripped, instead of silently falling back to the full Responses shape. + +### Fixed + +- Fixed concurrent reasoning summaries to ignore legacy streaming events under cutoff contract +- Fixed sequential-cutoff Codex reasoning summaries repeating earlier content when atomic summary snapshots are replayed or extended. +- Fixed error classification for typed AWS credential-resolution failures (`AwsCredentialsError`) to map them to authentication failures. ([#5030](https://github.com/can1357/oh-my-pi/pull/5030) by [@usr-bin-roygbiv](https://github.com/usr-bin-roygbiv)) + +## [16.3.15] - 2026-07-09 + +### Breaking Changes + +- Renamed `OpenAIResponsesCacheOptions`, `normalizeOpenAIResponsesPromptCacheKey`, and `getOpenAIResponsesPromptCacheKey` to the endpoint-neutral `OpenAICacheOptions`, `normalizeOpenAIPromptCacheKey`, and `getOpenAIPromptCacheKey`. + +### Added + +- Added automatic prompt-cache affinity header injection for OpenAI-family chat completions +- Added support for explicit prompt-cache affinity headers in OpenAI-family chat completions +- Added OpenAI pro reasoning mode support: models carrying the catalog `reasoningMode: "pro"` marker (GPT-5.6 Pro aliases) send `reasoning: { mode: "pro" }` on OpenAI Responses and Codex Responses requests, alongside the configured effort. The Codex request body now honors `requestModelId` so catalog aliases request the base upstream model id. + +### Changed + +- Updated xAI OAuth to use a dedicated device-code flow instead of redirect/loopback server + +### Fixed + +- Improved account routing for GPT-5.6 models to better respect paid tier requirements +- Refined account selection logic to correctly identify plan types from account metadata +- Fixed OpenAI Codex multi-account routing for GPT-5.6: Sol and Luna requests now prefer Plus-or-higher accounts while Terra remains available to Free/Go accounts; local pro-mode aliases inherit their base model's Codex plan eligibility. +- Fixed xAI Grok OAuth login to use xAI's device authorization flow: `/login` now opens the verification URL, displays the device code, and polls for approval instead of asking for a pasted redirect or linking to Hermes Agent documentation. + +## [16.3.14] - 2026-07-09 + +### Changed + +- Updated Codex reasoning effort mapping to support shifted wire tiers for newer models + +### Fixed + +- Fixed the Codex Responses request transformer bypassing catalog/compat reasoning effort maps: the clamped user effort is now remapped to the provider wire tier (GPT-5.6's shifted five-tier scale sends `max` for user `xhigh` and `xhigh` for `high`), failing loudly if a map produces a value outside the Codex wire vocabulary. + +## [16.3.13] - 2026-07-09 ### Changed - Changed the xAI Grok OAuth (`xai-oauth`) provider to use manual code-paste login by default. `/login` now accepts a pasted authorization code or full `http://127.0.0.1:56121/callback?code=...` redirect URL without starting a local callback listener ([#3277](https://github.com/can1357/oh-my-pi/pull/3277) by [@Jaaneek](https://github.com/Jaaneek)). - Renamed the xAI Grok OAuth provider in login and credential prompts to "xAI Grok OAuth (SuperGrok or X Premium+)" ([#3277](https://github.com/can1357/oh-my-pi/pull/3277) by [@Jaaneek](https://github.com/Jaaneek)). +### Fixed + +- Fixed the generic lazy-stream idle watchdog aborting healthy `cursor-agent` streams with "Provider stream stalled while waiting for the next event" while a Cursor exec-channel local tool (shell/read/grep/write/MCP/…) legitimately ran longer than the idle budget. Provider streams now advertise consumer-side local work in flight and the watchdog slides its deadline instead of aborting; genuinely silent streams still time out. ([#4593](https://github.com/can1357/oh-my-pi/issues/4593)) +- Fixed OpenAI Codex/Responses reasoning streams so streamed thinking content is preserved when the final `output_item.done` reconstructs to an empty summary ([#4918](https://github.com/can1357/oh-my-pi/issues/4918)). +- Fixed Anthropic streams hanging forever when generation wedges mid-stream (notably long `write` tool calls on Opus 4.8 high/xhigh) while the server keeps sending `ping` keepalives: pings now extend the idle watchdog only within a bounded window (3x the idle timeout) since the last real stream event, so a stalled tool-call stream times out and recovers instead of hanging with no retry path ([#4900](https://github.com/can1357/oh-my-pi/issues/4900)). + ## [16.3.12] - 2026-07-08 ### Added diff --git a/packages/ai/README.md b/packages/ai/README.md index 156baf4a4..bb47da22e 100644 --- a/packages/ai/README.md +++ b/packages/ai/README.md @@ -59,6 +59,7 @@ Unified LLM API with automatic model discovery, provider configuration, token an - **Qianfan** (requires `QIANFAN_API_KEY`) - **NVIDIA** (requires `NVIDIA_API_KEY`) - **NanoGPT** (requires `NANO_GPT_API_KEY`) +- **Novita** (requires `NOVITA_API_KEY`) - **Hugging Face Inference** - **xAI** - **Venice** (requires `VENICE_API_KEY`) @@ -943,6 +944,7 @@ In Node.js environments, you can set environment variables to avoid passing API | Synthetic | `SYNTHETIC_API_KEY` | | NVIDIA | `NVIDIA_API_KEY` | | NanoGPT | `NANO_GPT_API_KEY` | +| Novita | `NOVITA_API_KEY` | | Venice | `VENICE_API_KEY` | | Moonshot | `MOONSHOT_API_KEY` | | xAI | `XAI_API_KEY` | @@ -981,6 +983,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations: - Qianfan: `https://qianfan.baidubce.com/v2` - NVIDIA: `https://integrate.api.nvidia.com/v1` - NanoGPT: `https://nano-gpt.com/api/v1` +- Novita: `https://api.novita.ai/openai/v1` - Hugging Face Inference: `https://router.huggingface.co/v1` - Venice: `https://api.venice.ai/api/v1` - Xiaomi MiMo: `https://api.xiaomimimo.com/anthropic` @@ -1082,7 +1085,7 @@ Credentials are saved to `agent.db` in the agent directory. `/login qianfan` ope `login` supports OAuth providers (Anthropic, OpenAI Codex, GitHub Copilot, Gemini CLI, Antigravity) and API-key onboarding flows. -For the current API-key onboarding flows, the library covers Together, Moonshot, Qianfan, NVIDIA, NanoGPT, Hugging Face, Venice, Xiaomi, vLLM, LiteLLM, Cloudflare AI Gateway, Qwen Portal, and Ollama Cloud. Ollama remains the local runtime integration; set `OLLAMA_API_KEY` only when your local or self-hosted deployment enforces bearer auth. +For the current API-key onboarding flows, the library covers Together, Moonshot, Qianfan, NVIDIA, NanoGPT, Novita, Hugging Face, Venice, Xiaomi, vLLM, LiteLLM, Cloudflare AI Gateway, Qwen Portal, and Ollama Cloud. Ollama remains the local runtime integration; set `OLLAMA_API_KEY` only when your local or self-hosted deployment enforces bearer auth. ### Programmatic OAuth @@ -1114,7 +1117,7 @@ import { getOAuthApiKey, // (provider, credentialsMap) => { newCredentials, apiKey } | null // Types - type OAuthProvider, // includes 'anthropic', 'openai-codex', 'github-copilot', 'google-gemini-cli', 'google-antigravity', 'together', 'moonshot', 'qianfan', 'nvidia', 'nanogpt', 'huggingface', 'venice', 'xiaomi', 'vllm', 'litellm', 'cloudflare-ai-gateway', 'qwen-portal', ... + type OAuthProvider, // includes 'anthropic', 'openai-codex', 'github-copilot', 'google-gemini-cli', 'google-antigravity', 'together', 'moonshot', 'qianfan', 'nvidia', 'nanogpt', 'novita', 'huggingface', 'venice', 'xiaomi', 'vllm', 'litellm', 'cloudflare-ai-gateway', 'qwen-portal', ... type OAuthCredentials, } from "@oh-my-pi/pi-ai"; ``` diff --git a/packages/ai/package.json b/packages/ai/package.json index 814352b52..1124f3f13 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "16.3.12", + "version": "16.3.15", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/src/auth-broker/discover.ts b/packages/ai/src/auth-broker/discover.ts index c66ccdd20..71ef2c5fe 100644 --- a/packages/ai/src/auth-broker/discover.ts +++ b/packages/ai/src/auth-broker/discover.ts @@ -1,6 +1,6 @@ /** * Broker-aware auth-storage discovery used by both the coding-agent runtime and - * the catalog model generator. Keeps the precedence logic (env → config.yml → + * the catalog model generator. Keeps the precedence logic (env → config.yml/config.yaml → * token file → local SQLite) in one place so build-time tooling sees the same * credentials as the TUI. */ @@ -12,6 +12,7 @@ import { getConfigRootDir, isEnoent, logger, + MAIN_CONFIG_FILENAMES, } from "@oh-my-pi/pi-utils"; import { YAML } from "bun"; import { AuthStorage } from "../auth-storage"; @@ -72,21 +73,24 @@ interface ConfigSnapshot { } async function readConfigYaml(agentDir: string): Promise { - const configPath = path.join(agentDir, "config.yml"); - try { - const raw = await Bun.file(configPath).text(); - const parsed = YAML.parse(raw); - if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return {}; - const record = parsed as Record; - const url = typeof record["auth.broker.url"] === "string" ? (record["auth.broker.url"] as string) : undefined; - const token = - typeof record["auth.broker.token"] === "string" ? (record["auth.broker.token"] as string) : undefined; - return { url, token }; - } catch (err) { - if (isEnoent(err)) return {}; - logger.warn("auth-broker config.yml unreadable", { error: String(err) }); - return {}; + for (const filename of MAIN_CONFIG_FILENAMES) { + const configPath = path.join(agentDir, filename); + try { + const raw = await Bun.file(configPath).text(); + const parsed = YAML.parse(raw); + if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return {}; + const record = parsed as Record; + const url = typeof record["auth.broker.url"] === "string" ? (record["auth.broker.url"] as string) : undefined; + const token = + typeof record["auth.broker.token"] === "string" ? (record["auth.broker.token"] as string) : undefined; + return { url, token }; + } catch (err) { + if (isEnoent(err)) continue; + logger.warn("auth-broker config unreadable", { path: configPath, error: String(err) }); + return {}; + } } + return {}; } function resolveSnapshotTtlMs(): number { @@ -104,7 +108,7 @@ function resolveSnapshotTtlMs(): number { * Resolve broker connection configuration using the same precedence as the TUI: * * 1. `OMP_AUTH_BROKER_URL` / `OMP_AUTH_BROKER_TOKEN` env vars. - * 2. `auth.broker.url` / `auth.broker.token` in `/config.yml`. + * 2. `auth.broker.url` / `auth.broker.token` in `/config.yml` or `/config.yaml`. * 3. `/auth-broker.token` file (paired with a URL from env/config). * * Returns `null` when no broker URL is configured — callers should fall back to diff --git a/packages/ai/src/auth-gateway/server.ts b/packages/ai/src/auth-gateway/server.ts index c7eff02f4..ae4d47415 100644 --- a/packages/ai/src/auth-gateway/server.ts +++ b/packages/ai/src/auth-gateway/server.ts @@ -112,8 +112,8 @@ function deriveSessionId(modelId: string, context: Context): string { parts.push(JSON.stringify({ role: first.role, content: first.content })); } const seed = parts.join("\u0000"); - // The 36-char UUID flows through unchanged: Codex's - // `normalizeOpenAIResponsesPromptCacheKey` accepts ≤64 chars verbatim. + // The 36-char UUID flows through unchanged: + // `normalizeOpenAIPromptCacheKey` accepts ≤64 chars verbatim. return deterministicUuid(seed); } diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 315d43c48..fb9020f0d 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -761,25 +761,84 @@ function isAbortSignalOption( return typeof value === "object" && value !== null && "aborted" in value && "addEventListener" in value; } -function requiresOpenAICodexProModel(provider: string, modelId: string | undefined): boolean { - return provider === "openai-codex" && typeof modelId === "string" && modelId.includes("-spark"); +type OpenAICodexPlanRequirement = "none" | "paid" | "pro"; +type OpenAICodexPlanClass = "free" | "paid" | "pro" | "unknown"; + +const GPT_56_PAID_CODEX_MODEL_PATTERN = /^gpt-5\.6-(?:sol|luna)(?:-pro)?$/; +const OPENAI_CODEX_PRO_PLAN_TOKENS: Record = { + pro: true, +}; +const OPENAI_CODEX_PAID_PLAN_TOKENS: Record = { + plus: true, + business: true, + team: true, + enterprise: true, + edu: true, + education: true, + teacher: true, + teachers: true, + health: true, + gov: true, + government: true, +}; +const OPENAI_CODEX_FREE_PLAN_TOKENS: Record = { + free: true, + go: true, +}; + +/** + * Account tier needed for model-aware Codex OAuth routing. + * + * GPT-5.6 Terra (including its local pro-mode alias) remains available on every + * plan. Sol and Luna pro-mode aliases inherit their base models' paid tier; + * only Spark currently has a documented Pro-plan preference in Codex. + */ +function resolveOpenAICodexPlanRequirement(provider: string, modelId: string | undefined): OpenAICodexPlanRequirement { + if (provider !== "openai-codex" || typeof modelId !== "string") return "none"; + const separator = modelId.lastIndexOf("/"); + const bareModelId = (separator === -1 ? modelId : modelId.slice(separator + 1)).toLowerCase(); + if (bareModelId.includes("-spark")) return "pro"; + if (bareModelId === "gpt-5.6" || GPT_56_PAID_CODEX_MODEL_PATTERN.test(bareModelId)) return "paid"; + return "none"; } function getUsagePlanType(report: UsageReport | null): string | undefined { const metadata = report?.metadata; - if (!metadata || typeof metadata !== "object" || Array.isArray(metadata)) return undefined; - const planType = (metadata as { planType?: unknown }).planType; - return typeof planType === "string" ? planType.toLowerCase() : undefined; + if (!metadata) return undefined; + const planType = metadata.planType; + if (typeof planType !== "string") return undefined; + const normalized = planType + .trim() + .toLowerCase() + .replace(/[\s-]+/g, "_"); + return normalized.startsWith("chatgpt_") ? normalized.slice("chatgpt_".length) : normalized; } -function getOpenAICodexPlanPriority(report: UsageReport | null): number { +function classifyOpenAICodexPlan(report: UsageReport | null): OpenAICodexPlanClass { const planType = getUsagePlanType(report); - if (!planType) return 1; - return planType.includes("pro") ? 0 : 2; + if (!planType) return "unknown"; + // Pro Lite is a paid Codex tier, but does not imply full Pro-only model access. + if (planType === "prolite" || planType === "pro_lite") return "paid"; + const tokens = planType.split("_"); + if (tokens.some(token => OPENAI_CODEX_PRO_PLAN_TOKENS[token] === true)) return "pro"; + if (tokens.some(token => OPENAI_CODEX_PAID_PLAN_TOKENS[token] === true)) return "paid"; + if (tokens.some(token => OPENAI_CODEX_FREE_PLAN_TOKENS[token] === true)) return "free"; + return "unknown"; } -function hasOpenAICodexProPlan(report: UsageReport | null): boolean { - return getUsagePlanType(report)?.includes("pro") === true; +function getOpenAICodexPlanEligibility( + report: UsageReport | null, + requirement: OpenAICodexPlanRequirement, +): boolean | undefined { + if (requirement === "none") return true; + const planClass = classifyOpenAICodexPlan(report); + if (planClass === "unknown") return undefined; + return requirement === "paid" ? planClass !== "free" : planClass === "pro"; +} + +function getOpenAICodexPlanPriority(report: UsageReport | null, requirement: OpenAICodexPlanRequirement): number { + const eligibility = getOpenAICodexPlanEligibility(report, requirement); + return eligibility === true ? 0 : eligibility === undefined ? 1 : 2; } function compareUsageRankingMetric(left: number, right: number): number { @@ -3186,8 +3245,7 @@ export class AuthStorage { #compareRankedOAuthCandidatePriority( left: RankedOAuthCandidate, right: RankedOAuthCandidate, - provider: string, - modelId: string | undefined, + planRequirement: OpenAICodexPlanRequirement, ): number { if (left.blocked !== right.blocked) return left.blocked ? 1 : -1; if (left.blocked && right.blocked) { @@ -3196,7 +3254,7 @@ export class AuthStorage { if (leftBlockedUntil !== rightBlockedUntil) return leftBlockedUntil - rightBlockedUntil; return 0; } - if (requiresOpenAICodexProModel(provider, modelId) && left.planPriority !== right.planPriority) { + if (planRequirement !== "none" && left.planPriority !== right.planPriority) { return left.planPriority - right.planPriority; } if (left.hasPriorityBoost !== right.hasPriorityBoost) return left.hasPriorityBoost ? -1 : 1; @@ -3214,20 +3272,18 @@ export class AuthStorage { #compareRankedOAuthCandidates( left: RankedOAuthCandidate, right: RankedOAuthCandidate, - provider: string, - modelId: string | undefined, + planRequirement: OpenAICodexPlanRequirement, ): number { - const priority = this.#compareRankedOAuthCandidatePriority(left, right, provider, modelId); + const priority = this.#compareRankedOAuthCandidatePriority(left, right, planRequirement); return priority !== 0 ? priority : left.orderPos - right.orderPos; } #orderRankedOAuthCandidates( candidates: RankedOAuthCandidate[], sessionId: string | undefined, - provider: string, - modelId: string | undefined, + planRequirement: OpenAICodexPlanRequirement, ): OAuthCandidate[] { - candidates.sort((left, right) => this.#compareRankedOAuthCandidates(left, right, provider, modelId)); + candidates.sort((left, right) => this.#compareRankedOAuthCandidates(left, right, planRequirement)); if (!sessionId) { return candidates.map(candidate => ({ selection: candidate.selection, @@ -3252,7 +3308,7 @@ export class AuthStorage { for (const candidate of unblocked) { if ( candidate !== previous && - this.#compareRankedOAuthCandidatePriority(previous, candidate, provider, modelId) !== 0 + this.#compareRankedOAuthCandidatePriority(previous, candidate, planRequirement) !== 0 ) { bucketIndex += 1; } @@ -3297,6 +3353,7 @@ export class AuthStorage { providerKey: string; provider: string; order: number[]; + planRequirement: OpenAICodexPlanRequirement; credentials: OAuthSelection[]; options?: AuthApiKeyOptions; sessionId?: string; @@ -3378,7 +3435,7 @@ export class AuthStorage { blocked, blockedUntil, hasPriorityBoost: strategy.hasPriorityBoost?.(primary) ?? false, - planPriority: getOpenAICodexPlanPriority(usage), + planPriority: getOpenAICodexPlanPriority(usage, args.planRequirement), secondaryUsed: this.#normalizeUsageFraction(secondaryTarget), secondaryDrainRate: this.#computeWindowDrainRate( secondaryTarget, @@ -3390,7 +3447,7 @@ export class AuthStorage { orderPos, }); } - return this.#orderRankedOAuthCandidates(ranked, args.sessionId, args.provider, args.options?.modelId); + return this.#orderRankedOAuthCandidates(ranked, args.sessionId, args.planRequirement); } /** @@ -3418,8 +3475,9 @@ export class AuthStorage { const strategy = this.#rankingStrategyResolver?.(provider); const rankingContext: CredentialRankingContext = { modelId: options?.modelId }; const blockScope = strategy?.blockScope?.(rankingContext); - const requiresProModel = requiresOpenAICodexProModel(provider, options?.modelId); - const checkUsage = strategy !== undefined && (credentials.length > 1 || requiresProModel); + const planRequirement = resolveOpenAICodexPlanRequirement(provider, options?.modelId); + const hasPlanRequirement = planRequirement !== "none"; + const checkUsage = strategy !== undefined && (credentials.length > 1 || hasPlanRequirement); const sessionCredential = this.#getSessionCredential(provider, sessionId); const sessionPreferredIndex = sessionCredential?.type === "oauth" ? sessionCredential.index : undefined; const sessionPreferredCredential = @@ -3438,12 +3496,13 @@ export class AuthStorage { sessionPreferredIndex !== undefined && sessionPreferredCanRefreshOrUse && !this.#isCredentialBlocked(provider, providerKey, sessionPreferredIndex, blockScope); - const shouldRank = checkUsage && (!sessionPreferredIsAvailable || requiresProModel); + const shouldRank = checkUsage && (!sessionPreferredIsAvailable || hasPlanRequirement); const rankingOrder = shouldRank && sessionId ? credentials.map((_credential, index) => index) : order; const candidates = shouldRank ? await this.#rankOAuthSelections({ providerKey, provider, + planRequirement, order: rankingOrder, credentials, options, @@ -3457,7 +3516,7 @@ export class AuthStorage { .filter((selection): selection is { credential: OAuthCredential; index: number } => Boolean(selection)) .map(selection => ({ selection, usage: null, usageChecked: false })); - if (sessionPreferredIndex !== undefined && !requiresProModel) { + if (sessionPreferredIndex !== undefined && !hasPlanRequirement) { const sessionPreferredCandidate = candidates.findIndex( candidate => !this.#isCredentialBlocked(provider, providerKey, candidate.selection.index, blockScope) && @@ -3537,10 +3596,12 @@ export class AuthStorage { }), ); - // Skip the Pro-plan filter when no candidate is confirmed Pro, so users with only - // non-Pro accounts can still attempt Spark requests (e.g. trial/grandfathered access). - const enforceProRequirement = - requiresProModel && candidates.some(candidate => hasOpenAICodexProPlan(candidate.usage)); + // Enforce a tier only when at least one account is confirmed eligible. If + // every report is unknown or ineligible, preserve trial/grandfathered access + // by allowing the normal candidate fallback to attempt the request. + const enforcePlanRequirement = + hasPlanRequirement && + candidates.some(candidate => getOpenAICodexPlanEligibility(candidate.usage, planRequirement) === true); const fallback = candidates[0]; @@ -3556,7 +3617,8 @@ export class AuthStorage { allowBlocked: false, prefetchedUsage: candidate.usage, usagePrechecked: candidate.usageChecked, - enforceProRequirement, + planRequirement, + enforcePlanRequirement, strategy, rankingContext, blockScope, @@ -3571,7 +3633,8 @@ export class AuthStorage { allowBlocked: true, prefetchedUsage: fallback.usage, usagePrechecked: fallback.usageChecked, - enforceProRequirement, + planRequirement, + enforcePlanRequirement, strategy, rankingContext, blockScope, @@ -3709,7 +3772,8 @@ export class AuthStorage { allowBlocked: boolean; prefetchedUsage?: UsageReport | null; usagePrechecked?: boolean; - enforceProRequirement?: boolean; + planRequirement?: OpenAICodexPlanRequirement; + enforcePlanRequirement?: boolean; strategy?: CredentialRankingStrategy; rankingContext?: CredentialRankingContext; blockScope?: string; @@ -3722,7 +3786,8 @@ export class AuthStorage { allowBlocked, prefetchedUsage = null, usagePrechecked = false, - enforceProRequirement, + planRequirement: providedPlanRequirement, + enforcePlanRequirement, strategy, rankingContext, blockScope, @@ -3741,12 +3806,13 @@ export class AuthStorage { // refresh / persist / CAS-disable addresses the row by this stable id. const credentialId = this.#getStoredCredentials(provider)[selection.index]?.id; - const requiresProModel = requiresOpenAICodexProModel(provider, options?.modelId); - const applyProFilter = enforceProRequirement ?? requiresProModel; + const planRequirement = providedPlanRequirement ?? resolveOpenAICodexPlanRequirement(provider, options?.modelId); + const hasPlanRequirement = planRequirement !== "none"; + const applyPlanFilter = enforcePlanRequirement ?? hasPlanRequirement; let usage: UsageReport | null = null; let usageChecked = false; - if ((checkUsage && !allowBlocked) || requiresProModel) { + if ((checkUsage && !allowBlocked) || hasPlanRequirement) { if (usagePrechecked) { usage = prefetchedUsage; usageChecked = true; @@ -3757,7 +3823,7 @@ export class AuthStorage { }); usageChecked = true; } - if (applyProFilter && !hasOpenAICodexProPlan(usage)) { + if (applyPlanFilter && getOpenAICodexPlanEligibility(usage, planRequirement) !== true) { return undefined; } if (checkUsage && !allowBlocked && usage && strategy && rankingContext) { @@ -3825,7 +3891,7 @@ export class AuthStorage { } else { this.#replaceCredentialAt(provider, selection.index, updated); } - if ((checkUsage && !allowBlocked) || requiresProModel) { + if ((checkUsage && !allowBlocked) || hasPlanRequirement) { const sameAccount = selection.credential.accountId === updated.accountId; if (!usageChecked || !sameAccount) { usage = await this.#getUsageReport(provider, updated, { @@ -3834,7 +3900,7 @@ export class AuthStorage { }); usageChecked = true; } - if (applyProFilter && !hasOpenAICodexProPlan(usage)) { + if (applyPlanFilter && getOpenAICodexPlanEligibility(usage, planRequirement) !== true) { return undefined; } if (checkUsage && !allowBlocked && usage && strategy && rankingContext) { diff --git a/packages/ai/src/error/flags.ts b/packages/ai/src/error/flags.ts index a61638ea1..f6570b285 100644 --- a/packages/ai/src/error/flags.ts +++ b/packages/ai/src/error/flags.ts @@ -1,5 +1,6 @@ import { isUnexpectedSocketCloseMessage } from "@oh-my-pi/pi-utils"; import type { Api, AssistantMessage } from "../types"; +import { AwsCredentialsError } from "./aws"; import { AnthropicConnectionError, AnthropicConnectionTimeoutError, @@ -346,7 +347,9 @@ export function classify(error: unknown, api?: Api): number { } } - if (link instanceof AnthropicConnectionTimeoutError) { + if (link instanceof AwsCredentialsError) { + kinds |= Flag.AuthFailed; + } else if (link instanceof AnthropicConnectionTimeoutError) { kinds |= Flag.Timeout | Flag.Transient; } else if (link instanceof AnthropicConnectionError) { kinds |= Flag.Transient; diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 6b3570c23..a22302fd2 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -1462,6 +1462,16 @@ async function* observeDecodedAnthropicSdkEvents( const PROVIDER_MAX_RETRIES = 10; +/** + * How long `ping` keepalives may keep extending the idle deadline without any + * semantic stream progress, as a multiple of the idle timeout. Anthropic pings + * across legitimate generation gaps, so pings count as liveness — but a wedged + * upstream that pings forever while producing no events must eventually trip + * the idle watchdog instead of hanging an active tool-call stream without a + * recovery path (#4900). + */ +const PING_PROGRESS_MAX_IDLE_MULTIPLIER = 3; + /** * Log a malformed-stream-envelope anomaly without aborting the turn. The strict * parser would `throw new AnthropicStreamEnvelopeError(...)` here; we instead @@ -2007,11 +2017,20 @@ const streamAnthropicOnce = ( } >(); - // Pings keep the idle deadline alive once content is flowing, but a - // ping before message_start must not consume the first-event watchdog: - // it would flip the (retryable) pre-content stall classification into - // a terminal mid-stream idle timeout. + // Pings keep the idle deadline alive once content is flowing (Anthropic + // bridges legitimate generation gaps with keepalives), but only within a + // bounded window: a wedged upstream that pings forever while the model + // produces nothing must still trip the idle watchdog, otherwise an + // active tool-call stream hangs unrecoverably with no retry (#4900). + // A ping before message_start must not consume the first-event watchdog + // either: it would flip the (retryable) pre-content stall classification + // into a terminal mid-stream idle timeout. let sawNonPingEvent = false; + let lastNonPingProgressAtMs = 0; + const pingProgressCapMs = + idleTimeoutMs !== undefined && idleTimeoutMs > 0 + ? idleTimeoutMs * PING_PROGRESS_MAX_IDLE_MULTIPLIER + : undefined; const timedAnthropicStream = iterateWithIdleTimeout(anthropicStream, { idleTimeoutMs, firstItemTimeoutMs: firstEventTimeoutMs, @@ -2021,8 +2040,13 @@ const streamAnthropicOnce = ( onFirstItemTimeout: () => activeAbortTracker.abortLocally(firstEventTimeoutAbortError), abortSignal: options?.signal, isProgressItem: item => { - if ((item as AnthropicStreamEvent).type === "ping") return sawNonPingEvent; + if ((item as AnthropicStreamEvent).type === "ping") { + if (!sawNonPingEvent) return false; + if (pingProgressCapMs === undefined) return true; + return Date.now() - lastNonPingProgressAtMs < pingProgressCapMs; + } sawNonPingEvent = true; + lastNonPingProgressAtMs = Date.now(); return true; }, }); diff --git a/packages/ai/src/providers/azure-openai-responses.ts b/packages/ai/src/providers/azure-openai-responses.ts index 276614b2f..31090702c 100644 --- a/packages/ai/src/providers/azure-openai-responses.ts +++ b/packages/ai/src/providers/azure-openai-responses.ts @@ -34,7 +34,7 @@ import { applyResponsesReasoningParams, buildResponsesInput, createInitialResponsesAssistantMessage, - getOpenAIResponsesPromptCacheKey, + getOpenAIPromptCacheKey, isOpenAIResponsesProgressEvent, parseAzureDeploymentNameMap, processResponsesStream, @@ -348,7 +348,7 @@ function buildParams( model: deploymentName, input: messages, stream: true, - prompt_cache_key: getOpenAIResponsesPromptCacheKey(options), + prompt_cache_key: getOpenAIPromptCacheKey(options), // Encrypted reasoning replay (applyResponsesReasoningParams) requires // stateless responses, matching the openai provider. store: false, diff --git a/packages/ai/src/providers/cursor.ts b/packages/ai/src/providers/cursor.ts index b7cb3e3de..d2034de31 100644 --- a/packages/ai/src/providers/cursor.ts +++ b/packages/ai/src/providers/cursor.ts @@ -653,7 +653,8 @@ export interface UsageState { sawTokenDelta: boolean; } -async function handleServerMessage( +/** Exported for tests: drives one Cursor server message through the stream (exec waits mark the stream busy). */ +export async function handleServerMessage( msg: AgentServerMessage, output: AssistantMessage, stream: AssistantMessageEventStream, @@ -675,15 +676,21 @@ async function handleServerMessage( } else if (msgCase === "kvServerMessage") { handleKvServerMessage(msg.message.value as KvServerMessage, blobStore, h2Request); } else if (msgCase === "execServerMessage") { - await handleExecServerMessage( - msg.message.value as ExecServerMessage, - h2Request, - execHandlers, - onToolResult, - requestContextTools, - output, - stream, - state, + // The server is waiting on OUR local tool result during this window — no + // AssistantMessageEvent flows until the handler finishes. Mark the wait + // as local work so the lazy stream idle watchdog attributes the silence + // to the tool run instead of aborting a healthy stream (issue #4593). + await stream.trackLocalWork( + handleExecServerMessage( + msg.message.value as ExecServerMessage, + h2Request, + execHandlers, + onToolResult, + requestContextTools, + output, + stream, + state, + ), ); } else if (msgCase === "conversationCheckpointUpdate") { handleConversationCheckpointUpdate(msg.message.value, output, usageState, onConversationCheckpoint); diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 80baabe8d..609a570fb 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -12,6 +12,7 @@ import { $flag, asRecord, fetchWithRetry, + getInstallId, logger, parseStreamingJson, readSseJson, @@ -24,6 +25,8 @@ import { getEnvApiKey } from "../stream"; import type { Api, AssistantMessage, + CodexCompactionContext, + CodexCompactionRequestContext, Context, FetchImpl, Model, @@ -65,7 +68,7 @@ import { type CodexRequestOptions, type InputItem, type RequestBody, - shouldUseCodexResponsesLite, + resolveCodexResponsesLite, transformRequestBody, } from "./openai-codex/request-transformer"; import { CodexApiError } from "./openai-codex/response-handler"; @@ -88,6 +91,7 @@ import { appendReasoningSummaryTextDelta, appendResponsesToolResultMessages, applyOpenAIServiceTier, + applyReasoningSummaryDone, buildResponsesDeltaInput, convertResponsesAssistantMessage, convertResponsesInputContent, @@ -95,10 +99,11 @@ import { encodeTextSignatureV1, finalizeCustomToolCallInputDone, finalizePendingResponsesToolCalls, + finalizeReasoningThinking, finalizeToolCallArgumentsDone, isOpenAIResponsesProgressEvent, mapOpenAIResponsesStopReason, - normalizeOpenAIResponsesPromptCacheKey, + normalizeOpenAIPromptCacheKey, populateResponsesUsageFromResponse, promoteResponsesToolUseStopReason, } from "./openai-shared"; @@ -116,17 +121,19 @@ export interface OpenAICodexResponsesOptions extends StreamOptions { preferWebsockets?: boolean; serviceTier?: ServiceTier; /** - * Opt into the Responses Lite transport contract. Sends + * Responses Lite transport override; defaults to the model's catalog + * `useResponsesLite` flag (codex-rs `use_responses_lite`). Sends * `x-openai-internal-codex-responses-lite: true` on HTTP requests and on the * WebSocket upgrade (the marker is connection-scoped there, so lite and - * non-lite turns never share a pooled socket), strips image detail from - * input, and disables parallel tool calling — mirroring codex-rs. + * non-lite turns never share a pooled socket), moves instructions/tools + * into input items, strips image detail, and disables parallel tool + * calling — mirroring codex-rs. */ responsesLite?: boolean; /** - * Extra `client_metadata` to include in the request body on both transports. - * The canonical Codex envelope is `client_metadata["x-codex-turn-metadata"]` - * (JSON string of thread/turn identifiers); flat keys are also accepted. + * Additional fields embedded in the canonical + * `client_metadata["x-codex-turn-metadata"]` JSON blob. Reserved identity + * keys are ignored; extras are never emitted as top-level metadata fields. */ clientMetadata?: Record; /** @@ -138,6 +145,49 @@ export interface OpenAICodexResponsesOptions extends StreamOptions { onModerationMetadata?: (metadata: unknown) => void; } +/** Inputs for synthesizing Codex request identity outside the normal stream path. */ +export interface OpenAICodexCompatibilityMetadataOptions { + sessionId?: string; + providerSessionState?: Map; + requestKind: OpenAICodexRequestKind; + compaction?: CodexCompactionRequestContext; + startNewTurn?: boolean; + turnStartedAtUnixMs?: number; + clientMetadata?: Readonly>; + /** Add the direct installation header required by `/responses/compact`. */ + includeInstallationHeader?: boolean; +} + +/** Canonical Codex body metadata and compatibility headers for one request. */ +export interface OpenAICodexCompatibilityMetadata { + clientMetadata: Record; + headers: Record; +} + +/** Live Codex session state to preserve after a successful history rewrite. */ +export interface OpenAICodexCompactionResetOptions { + providerSessionState?: Map; + sessionId?: string; + compaction: CodexCompactionContext; +} + +/** Add the selected wire implementation to one logical compaction context. */ +export function createOpenAICodexCompactionRequestContext(options: { + context: CodexCompactionContext | undefined; + implementation: "responses" | "responses_compaction_v2" | "responses_compact"; +}): CodexCompactionRequestContext | undefined { + const context = options.context; + if (!context) return undefined; + return { + operationId: context.operationId, + trigger: context.trigger, + reason: context.reason, + implementation: options.implementation, + phase: context.phase, + strategy: context.strategy, + }; +} + const CODEX_DEBUG = $flag("PI_CODEX_DEBUG"); const CODEX_MAX_RETRIES = 5; const CODEX_RETRY_DELAY_MS = 500; @@ -185,7 +235,6 @@ const CODEX_RETRYABLE_EVENT_MESSAGE = const CODEX_PROVIDER_SESSION_STATE_KEY = "openai-codex-responses"; const X_CODEX_TURN_STATE_HEADER = "x-codex-turn-state"; const X_MODELS_ETAG_HEADER = "x-models-etag"; -const X_OPENAI_INTERNAL_CODEX_RESPONSES_LITE_HEADER = "x-openai-internal-codex-responses-lite"; /** WebSocket frames cannot carry per-request HTTP headers; codex-rs mirrors the lite marker into `client_metadata` under this key. */ const CODEX_WS_RESPONSES_LITE_CLIENT_METADATA_KEY = "ws_request_header_x_openai_internal_codex_responses_lite"; /** `response.metadata` payload key carrying ChatGPT moderation metadata. */ @@ -342,6 +391,237 @@ type CodexWebSocketSessionState = { interface CodexProviderSessionState extends ProviderSessionState { webSocketSessions: Map; webSocketPublicToPrivate: Map; + metadataSessions: Map; +} + +/** Request classification encoded in Codex turn metadata. */ +export type OpenAICodexRequestKind = "turn" | "prewarm" | "compaction"; + +interface CodexMetadataSessionState { + sessionId: string; + threadId: string; + windowId: string; + turnId?: string; + turnStartedAtUnixMs?: number; + compactionOperationId?: string; + reuseTurnForNextRequest?: boolean; +} + +interface CodexCompatibilityIdentity { + installationId: string; + sessionId: string; + threadId: string; + windowId: string; + turnMetadataJson?: string; +} + +interface CodexRequestMetadata extends CodexCompatibilityIdentity { + turnId: string; + turnMetadataJson: string; + clientMetadata: Record; +} + +const CODEX_RESERVED_METADATA_KEYS: Record = { + installation_id: true, + [OPENAI_HEADERS.INSTALLATION_ID]: true, + session_id: true, + thread_id: true, + turn_id: true, + window_id: true, + [OPENAI_HEADERS.WINDOW_ID]: true, + [OPENAI_HEADERS.TURN_METADATA]: true, + [OPENAI_HEADERS.PARENT_THREAD_ID]: true, + [OPENAI_HEADERS.SUBAGENT]: true, + request_kind: true, + compaction: true, + turn_started_at_unix_ms: true, + forked_from_thread_id: true, + parent_thread_id: true, + subagent_kind: true, + thread_source: true, + sandbox: true, + workspaces: true, +}; + +function createCodexMetadataSessionState(sessionId: string): CodexMetadataSessionState { + return { + sessionId, + threadId: crypto.randomUUID(), + windowId: crypto.randomUUID(), + }; +} + +function getOrCreateCodexMetadataSessionState( + sessionId: string, + providerState: CodexProviderSessionState | undefined, +): CodexMetadataSessionState { + if (!providerState) return createCodexMetadataSessionState(sessionId); + const existing = providerState.metadataSessions.get(sessionId); + if (existing) return existing; + const created = createCodexMetadataSessionState(sessionId); + providerState.metadataSessions.set(sessionId, created); + return created; +} + +function createCodexCompatibilityIdentity(session: CodexMetadataSessionState): CodexCompatibilityIdentity { + return { + installationId: getInstallId(), + sessionId: session.sessionId, + threadId: session.threadId, + windowId: session.windowId, + }; +} + +function resolveCodexStartNewTurn( + session: CodexMetadataSessionState, + requestKind: OpenAICodexRequestKind, + compaction: CodexCompactionRequestContext | undefined, + override: boolean | undefined, +): boolean { + if (requestKind !== "compaction") { + if (requestKind === "turn") { + const reuseCompactionTurn = session.reuseTurnForNextRequest === true; + session.reuseTurnForNextRequest = false; + session.compactionOperationId = undefined; + if (reuseCompactionTurn) return false; + } + return override ?? requestKind === "turn"; + } + if (!compaction) return override ?? false; + const startsNewOperation = session.compactionOperationId !== compaction.operationId; + if (startsNewOperation) session.reuseTurnForNextRequest = false; + session.compactionOperationId = compaction.operationId; + return override ?? (compaction.phase !== "mid_turn" && startsNewOperation); +} + +function toAsciiJsonString(value: Record): string { + return JSON.stringify(value).replace( + /[\x7f-\uffff]/g, + char => `\\u${char.charCodeAt(0).toString(16).padStart(4, "0")}`, + ); +} + +function createCodexRequestMetadata( + session: CodexMetadataSessionState, + requestKind: OpenAICodexRequestKind, + options: { + startNewTurn: boolean; + turnStartedAtUnixMs?: number; + clientMetadata?: Readonly>; + compaction?: CodexCompactionRequestContext; + }, +): CodexRequestMetadata { + if (options.startNewTurn || !session.turnId) { + session.turnId = crypto.randomUUID(); + session.turnStartedAtUnixMs = options.turnStartedAtUnixMs; + } + const identity = createCodexCompatibilityIdentity(session); + const extra: Record = {}; + const callerMetadata = options.clientMetadata; + if (callerMetadata) { + for (const key in callerMetadata) { + if (!CODEX_RESERVED_METADATA_KEYS[key]) extra[key] = callerMetadata[key]; + } + } + const turnMetadata: Record = { + installation_id: identity.installationId, + session_id: identity.sessionId, + thread_id: identity.threadId, + turn_id: session.turnId, + window_id: identity.windowId, + request_kind: requestKind, + }; + if (options.compaction) { + turnMetadata.compaction = { + trigger: options.compaction.trigger, + reason: options.compaction.reason, + implementation: options.compaction.implementation, + phase: options.compaction.phase, + strategy: options.compaction.strategy, + }; + } + if (session.turnStartedAtUnixMs !== undefined) { + turnMetadata.turn_started_at_unix_ms = session.turnStartedAtUnixMs; + } + for (const key in extra) turnMetadata[key] = extra[key]; + const turnMetadataJson = toAsciiJsonString(turnMetadata); + return { + ...identity, + turnId: session.turnId, + turnMetadataJson, + clientMetadata: { + [OPENAI_HEADERS.INSTALLATION_ID]: identity.installationId, + session_id: identity.sessionId, + thread_id: identity.threadId, + [OPENAI_HEADERS.WINDOW_ID]: identity.windowId, + turn_id: session.turnId, + [OPENAI_HEADERS.TURN_METADATA]: turnMetadataJson, + }, + }; +} + +function applyCodexCompatibilityHeaders(headers: Headers, metadata: CodexCompatibilityIdentity): void { + headers.set(OPENAI_HEADERS.SCOPED_SESSION_ID, metadata.sessionId); + headers.set(OPENAI_HEADERS.THREAD_ID, metadata.threadId); + headers.set(OPENAI_HEADERS.WINDOW_ID, metadata.windowId); + if (metadata.turnMetadataJson) { + headers.set(OPENAI_HEADERS.TURN_METADATA, metadata.turnMetadataJson); + } else { + headers.delete(OPENAI_HEADERS.TURN_METADATA); + } +} + +/** + * Synthesize Codex request identity for raw provider routes such as remote + * compaction while reusing the live session's thread, window, and turn. + */ +export function createOpenAICodexCompatibilityMetadata( + options: OpenAICodexCompatibilityMetadataOptions, +): OpenAICodexCompatibilityMetadata { + const providerState = getCodexProviderSessionState(options.providerSessionState); + const sessionId = normalizeOpenAIPromptCacheKey(options.sessionId) ?? crypto.randomUUID(); + const session = getOrCreateCodexMetadataSessionState(sessionId, providerState); + const startNewTurn = resolveCodexStartNewTurn( + session, + options.requestKind, + options.compaction, + options.startNewTurn, + ); + const metadata = createCodexRequestMetadata(session, options.requestKind, { + startNewTurn, + turnStartedAtUnixMs: options.turnStartedAtUnixMs ?? (startNewTurn || !session.turnId ? Date.now() : undefined), + clientMetadata: options.clientMetadata, + compaction: options.compaction, + }); + const headers = new Headers(); + applyCodexCompatibilityHeaders(headers, metadata); + if (options.includeInstallationHeader) { + headers.set(OPENAI_HEADERS.INSTALLATION_ID, metadata.installationId); + } + return { + clientMetadata: { ...metadata.clientMetadata }, + headers: Object.fromEntries(headers.entries()), + }; +} + +/** + * Invalidate Codex history-dependent transport state after compaction while + * retaining the session identity and live connection. + */ +export function resetOpenAICodexHistoryAfterCompaction(options: OpenAICodexCompactionResetOptions): void { + const providerState = options.providerSessionState?.get(CODEX_PROVIDER_SESSION_STATE_KEY); + if (!isCodexProviderSessionState(providerState)) return; + for (const websocketState of providerState.webSocketSessions.values()) { + resetCodexWebSocketAppendState(websocketState); + if (options.compaction.phase !== "mid_turn") websocketState.turnState = undefined; + } + const sessionId = normalizeOpenAIPromptCacheKey(options.sessionId); + if (!sessionId) return; + const metadataSession = providerState.metadataSessions.get(sessionId); + if (!metadataSession) return; + metadataSession.windowId = crypto.randomUUID(); + metadataSession.compactionOperationId = undefined; + metadataSession.reuseTurnForNextRequest = options.compaction.phase !== "standalone_turn"; } interface CodexRequestContext { @@ -352,8 +632,10 @@ interface CodexRequestContext { requestHeaders: Record; transportSessionId?: string; providerSessionState?: CodexProviderSessionState; + isolatedTransportState?: CodexProviderSessionState; websocketState?: CodexWebSocketSessionState; responsesLite: boolean; + requestMetadata?: CodexRequestMetadata; transformedBody: RequestBody; rawRequestDump: RawHttpRequestDump; } @@ -610,23 +892,37 @@ function createCodexProviderSessionState(): CodexProviderSessionState { const state: CodexProviderSessionState = { webSocketSessions: new Map(), webSocketPublicToPrivate: new Map(), + metadataSessions: new Map(), close: () => { for (const session of state.webSocketSessions.values()) { session.connection?.close("session_disposed"); } state.webSocketSessions.clear(); state.webSocketPublicToPrivate.clear(); + state.metadataSessions.clear(); }, }; return state; } +function isCodexProviderSessionState(state: ProviderSessionState | undefined): state is CodexProviderSessionState { + return ( + state !== undefined && + "webSocketSessions" in state && + state.webSocketSessions instanceof Map && + "webSocketPublicToPrivate" in state && + state.webSocketPublicToPrivate instanceof Map && + "metadataSessions" in state && + state.metadataSessions instanceof Map + ); +} + function getCodexProviderSessionState( providerSessionState: Map | undefined, ): CodexProviderSessionState | undefined { if (!providerSessionState) return undefined; - const existing = providerSessionState.get(CODEX_PROVIDER_SESSION_STATE_KEY) as CodexProviderSessionState | undefined; - if (existing) return existing; + const existing = providerSessionState.get(CODEX_PROVIDER_SESSION_STATE_KEY); + if (isCodexProviderSessionState(existing)) return existing; const created = createCodexProviderSessionState(); providerSessionState.set(CODEX_PROVIDER_SESSION_STATE_KEY, created); return created; @@ -897,8 +1193,8 @@ async function buildCodexRequestContext( const accountId = getCodexAccountId(apiKey); const baseUrl = model.baseUrl || CODEX_BASE_URL; const url = resolveCodexResponsesUrl(baseUrl); - const promptCacheKey = normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId); - const transportSessionId = normalizeOpenAIResponsesPromptCacheKey(options?.sessionId); + const promptCacheKey = normalizeOpenAIPromptCacheKey(options?.promptCacheKey ?? options?.sessionId); + const transportSessionId = normalizeOpenAIPromptCacheKey(options?.sessionId); const transformedBody = await buildTransformedCodexRequestBody(model, context, options, promptCacheKey); const requestHeaders = { ...(model.headers ?? {}), ...(options?.headers ?? {}) }; @@ -912,19 +1208,56 @@ async function buildCodexRequestContext( }; const providerSessionState = getCodexProviderSessionState(options?.providerSessionState); - const responsesLite = shouldUseCodexResponsesLite(transformedBody, options?.responsesLite); + const isolatedTransportState = options?.codexCompaction ? createCodexProviderSessionState() : undefined; + const transportProviderSessionState = isolatedTransportState ?? providerSessionState; + const responsesLite = resolveCodexResponsesLite(model, options?.responsesLite); const sessionKey = getCodexWebSocketSessionKey(transportSessionId, model, accountId, apiKey, baseUrl, responsesLite); const publicSessionKey = transportSessionId ? `${baseUrl}:${model.id}:${transportSessionId}` : undefined; if (sessionKey && publicSessionKey) { - providerSessionState?.webSocketPublicToPrivate.set(publicSessionKey, sessionKey); + transportProviderSessionState?.webSocketPublicToPrivate.set(publicSessionKey, sessionKey); } + const sharedWebsocketState = + sessionKey && providerSessionState + ? isolatedTransportState + ? providerSessionState.webSocketSessions.get(sessionKey) + : getCodexWebSocketSessionState(sessionKey, providerSessionState) + : undefined; const websocketState = - sessionKey && providerSessionState ? getCodexWebSocketSessionState(sessionKey, providerSessionState) : undefined; - if (websocketState && !isCodexWithinTurnContinuation(context)) { - // codex-rs scopes `x-codex-turn-state` to a single user turn: tool-loop - // follow-ups echo it, a new user turn starts without it. + sessionKey && isolatedTransportState + ? getCodexWebSocketSessionState(sessionKey, isolatedTransportState) + : sharedWebsocketState; + if (isolatedTransportState && websocketState && sharedWebsocketState) { + websocketState.disableWebsocket = sharedWebsocketState.disableWebsocket; + websocketState.turnState = sharedWebsocketState.turnState; + websocketState.modelsEtag = sharedWebsocketState.modelsEtag; + } + const withinTurnContinuation = isCodexWithinTurnContinuation(context); + const metadataSessionId = transportSessionId ?? crypto.randomUUID(); + const metadataSession = getOrCreateCodexMetadataSessionState(metadataSessionId, providerSessionState); + const compaction = options?.codexCompaction; + const requestKind: OpenAICodexRequestKind = compaction ? "compaction" : "turn"; + const startNewTurn = resolveCodexStartNewTurn( + metadataSession, + requestKind, + compaction, + compaction ? undefined : !withinTurnContinuation, + ); + if (websocketState && startNewTurn) { + // Codex scopes turn-state to one turn. Mid-turn compaction and tool-loop + // follow-ups preserve it; new user or compaction turns start without it. websocketState.turnState = undefined; } + const requestMetadata = createCodexRequestMetadata(metadataSession, requestKind, { + startNewTurn, + turnStartedAtUnixMs: compaction + ? startNewTurn || !metadataSession.turnId + ? Date.now() + : undefined + : getCodexTurnStartedAtUnixMs(context), + clientMetadata: transformedBody.client_metadata, + compaction, + }); + transformedBody.client_metadata = requestMetadata.clientMetadata; return { apiKey, accountId, @@ -933,8 +1266,10 @@ async function buildCodexRequestContext( requestHeaders, transportSessionId, providerSessionState, + isolatedTransportState, websocketState, responsesLite, + requestMetadata, transformedBody, rawRequestDump, }; @@ -945,10 +1280,10 @@ export async function buildTransformedCodexRequestBody( model: Model<"openai-codex-responses">, context: Context, options: OpenAICodexResponsesOptions | undefined, - promptCacheKey = normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId), + promptCacheKey = normalizeOpenAIPromptCacheKey(options?.promptCacheKey ?? options?.sessionId), ): Promise { const params: RequestBody = { - model: model.id, + model: model.requestModelId ?? model.id, input: convertMessages(model, context), stream: true, prompt_cache_key: promptCacheKey, @@ -1062,19 +1397,21 @@ async function openCodexWebSocketTransport( }> { const canAppendBeforeRequest = websocketState.canAppend === true; const chainedBody = buildCodexChainedRequestBody(requestContext.transformedBody, websocketState); - // WebSocket frames cannot carry per-request HTTP headers, so the Responses - // Lite marker rides in `client_metadata` on every `response.create`. + // WebSocket frames cannot carry per-request HTTP headers. Canonical Codex + // request identity is already in `client_metadata`; connection-scoped + // compatibility values that can change after the upgrade ride alongside it + // on every `response.create`. + const websocketClientMetadata = { ...(chainedBody.client_metadata ?? {}) }; + if (requestContext.responsesLite) { + websocketClientMetadata[CODEX_WS_RESPONSES_LITE_CLIENT_METADATA_KEY] = "true"; + } + if (websocketState.turnState) { + websocketClientMetadata[X_CODEX_TURN_STATE_HEADER] = websocketState.turnState; + } let websocketRequest = { type: "response.create", ...chainedBody, - ...(requestContext.responsesLite - ? { - client_metadata: { - ...(chainedBody.client_metadata ?? {}), - [CODEX_WS_RESPONSES_LITE_CLIENT_METADATA_KEY]: "true", - }, - } - : {}), + client_metadata: websocketClientMetadata, }; const replacementWebsocketRequest = await options?.onPayload?.(websocketRequest, model); if (replacementWebsocketRequest !== undefined) { @@ -1089,8 +1426,17 @@ async function openCodexWebSocketTransport( "websocket", websocketState, requestContext.responsesLite, + requestContext.requestMetadata, ); const requestBodyForState = structuredCloneJSON(requestContext.transformedBody); + // `onPayload` may rewrite the outgoing frame (e.g. drop `stream_options`); + // recorded state must reflect what was actually sent — the sequential-cutoff + // summary decoder keys off it. + if (websocketRequest.stream_options === undefined) { + delete requestBodyForState.stream_options; + } else { + requestBodyForState.stream_options = websocketRequest.stream_options; + } requestContext.rawRequestDump.body = websocketRequest; CODEX_DEBUG && logger.debug("[codex] codex websocket request", { @@ -1126,6 +1472,16 @@ async function openCodexWebSocketTransport( }; } +function getCodexTurnStartedAtUnixMs(context: Context): number { + for (let i = context.messages.length - 1; i >= 0; i--) { + const message = context.messages[i]; + if (message?.role === "user" && Number.isFinite(message.timestamp)) { + return Math.trunc(message.timestamp); + } + } + return Date.now(); +} + /** * True when the request continues the current turn (everything after the * last assistant message is tool results), false when a new user turn starts. @@ -1166,6 +1522,7 @@ async function openCodexSseTransport( wireBody, state, requestContext.responsesLite, + requestContext.requestMetadata, requestSetup.requestSignal, requestSetup.firstEventTimeoutMs, event => options?.onSseEvent?.(event, model), @@ -1323,6 +1680,17 @@ class CodexStreamProcessor { this.startTime = init.startTime; } + /** + * Whether the request actually sent (post-`onPayload`) opted into + * sequential-cutoff summary delivery: summaries then arrive as atomic + * `response.reasoning_summary_text.done` events and incremental + * `.delta`/`.part.*` events are ignored (mirrors codex-rs + * `uses_sequential_cutoff_reasoning_summaries`). + */ + get #sequentialCutoffSummaries(): boolean { + return this.runtime.requestBodyForState.stream_options?.reasoning_summary_delivery === "sequential_cutoff"; + } + async process(): Promise { const { output, stream } = this; stream.push({ type: "start", partial: output }); @@ -1384,6 +1752,7 @@ class CodexStreamProcessor { } if (eventType === "response.reasoning_summary_part.added") { + if (this.#sequentialCutoffSummaries) return firstTokenTime; if (this.runtime.currentItem?.type === "reasoning") { appendReasoningSummaryPart( this.runtime.currentItem, @@ -1394,6 +1763,7 @@ class CodexStreamProcessor { } if (eventType === "response.reasoning_summary_text.delta") { + if (this.#sequentialCutoffSummaries) return firstTokenTime; if (this.runtime.currentItem?.type === "reasoning" && this.runtime.currentBlock?.type === "thinking") { appendReasoningSummaryTextDelta( this.runtime.currentItem, @@ -1407,7 +1777,46 @@ class CodexStreamProcessor { return firstTokenTime; } + if (eventType === "response.reasoning_summary_text.done") { + // Outside the cutoff contract the text already streamed via `.delta`. + if (!this.#sequentialCutoffSummaries) return firstTokenTime; + const entry = this.runtime.openItemForEvent(rawEvent); + if (entry?.item.type === "reasoning" && entry.block?.type === "thinking") { + if (!firstTokenTime) firstTokenTime = performance.now(); + const summaryIndex = + typeof rawEvent.summary_index === "number" && Number.isFinite(rawEvent.summary_index) + ? Math.trunc(rawEvent.summary_index) + : 0; + applyReasoningSummaryDone( + entry.item, + entry.block, + typeof rawEvent.text === "string" ? rawEvent.text : "", + summaryIndex, + stream, + output, + entry.contentIndex, + ); + } + return firstTokenTime; + } + + if (eventType === "response.reasoning_text.delta") { + const entry = this.runtime.openItemForEvent(rawEvent); + const delta = typeof rawEvent.delta === "string" ? rawEvent.delta : ""; + if (entry?.item.type === "reasoning" && entry.block?.type === "thinking") { + entry.block.thinking += delta; + stream.push({ + type: "thinking_delta", + contentIndex: entry.contentIndex, + delta, + partial: output, + }); + } + return firstTokenTime; + } + if (eventType === "response.reasoning_summary_part.done") { + if (this.#sequentialCutoffSummaries) return firstTokenTime; if (this.runtime.currentItem?.type === "reasoning" && this.runtime.currentBlock?.type === "thinking") { appendReasoningSummaryPartDone( this.runtime.currentItem, @@ -1522,13 +1931,15 @@ class CodexStreamProcessor { // most-recently-added block may belong to a sibling (#2619). Some Codex // function/custom tool items omit `id`; in that case `output_index` still // routes `output_item.done` to the block that received `output_item.added`. - const itemId = typeof (item as { id?: string }).id === "string" ? (item as { id: string }).id : ""; + const itemId = "id" in item && typeof item.id === "string" ? item.id : ""; const entry = (itemId ? runtime.openItems.get(itemId) : null) ?? runtime.openItemForEvent(rawEvent); const block = entry?.block ?? null; const contentIndex = entry?.contentIndex ?? output.content.length - 1; if (item.type === "reasoning" && block?.type === "thinking") { - block.thinking = item.summary?.map(summary => summary.text).join("\n\n") || ""; + block.thinking = finalizeReasoningThinking(item, block.thinking, { + cumulativeSummarySnapshots: this.#sequentialCutoffSummaries, + }); block.thinkingSignature = JSON.stringify(item); stream.push({ type: "thinking_end", @@ -2105,6 +2516,8 @@ export const streamOpenAICodexResponses: StreamFunction<"openai-codex-responses" stream.push({ type: "error", reason: "error", error: output }); } stream.end(); + } finally { + requestContext?.isolatedTransportState?.close(); } })(); @@ -2123,10 +2536,10 @@ export async function prewarmOpenAICodexResponses( const accountId = getCodexAccountId(apiKey); const baseUrl = model.baseUrl || CODEX_BASE_URL; const url = resolveCodexResponsesUrl(baseUrl); - const transportSessionId = normalizeOpenAIResponsesPromptCacheKey(options?.sessionId); + const transportSessionId = normalizeOpenAIPromptCacheKey(options?.sessionId); const promptCacheKey = transportSessionId; const providerSessionState = getCodexProviderSessionState(options?.providerSessionState); - const responsesLite = options?.responsesLite === true; + const responsesLite = resolveCodexResponsesLite(model, options?.responsesLite); const sessionKey = getCodexWebSocketSessionKey(transportSessionId, model, accountId, apiKey, baseUrl, responsesLite); const publicSessionKey = transportSessionId ? `${baseUrl}:${model.id}:${transportSessionId}` : undefined; if (publicSessionKey && sessionKey) { @@ -2135,6 +2548,11 @@ export async function prewarmOpenAICodexResponses( if (!sessionKey || !providerSessionState) return; const state = getCodexWebSocketSessionState(sessionKey, providerSessionState); if (!shouldUseCodexWebSocket(model, state, options?.preferWebsockets)) return; + const metadataSession = getOrCreateCodexMetadataSessionState( + transportSessionId ?? crypto.randomUUID(), + providerSessionState, + ); + const requestIdentity = createCodexCompatibilityIdentity(metadataSession); const headers = logger.time( "prewarmCodex:createHeaders", createCodexHeaders, @@ -2145,6 +2563,7 @@ export async function prewarmOpenAICodexResponses( "websocket", state, responsesLite, + requestIdentity, ); await logger.time( "prewarmCodex:establishWs", @@ -2252,6 +2671,7 @@ export interface OpenAICodexTransportDetails { canAppend: boolean; prewarmed: boolean; hasSessionState: boolean; + hasTurnState: boolean; lastFallbackAt?: number; } @@ -2267,7 +2687,7 @@ function getCodexWebSocketStateForPublicSession( ): CodexWebSocketSessionState | undefined { const baseUrl = options?.baseUrl || model.baseUrl || CODEX_BASE_URL; const providerSessionState = getCodexProviderSessionState(options?.providerSessionState); - const normalizedSessionId = normalizeOpenAIResponsesPromptCacheKey(options?.sessionId); + const normalizedSessionId = normalizeOpenAIPromptCacheKey(options?.sessionId); const publicSessionKey = normalizedSessionId ? `${baseUrl}:${model.id}:${normalizedSessionId}` : undefined; const privateSessionKey = publicSessionKey ? providerSessionState?.webSocketPublicToPrivate.get(publicSessionKey) @@ -2314,6 +2734,7 @@ export function getOpenAICodexTransportDetails( canAppend: state?.canAppend ?? false, prewarmed: state?.prewarmed ?? false, hasSessionState: state !== undefined, + hasTurnState: state?.turnState !== undefined, lastFallbackAt: state?.lastFallbackAt, }; } @@ -3257,12 +3678,22 @@ async function openCodexSseEventStream( body: RequestBody, state: CodexWebSocketSessionState | undefined, responsesLite: boolean, + requestMetadata: CodexRequestMetadata | undefined, signal: AbortSignal | undefined, firstEventTimeoutMs: number | undefined, onSseEvent?: OpenAICodexResponsesOptions["onSseEvent"], fetchOverride?: FetchImpl, ): Promise>> { - const headers = createCodexHeaders(requestHeaders, accountId, apiKey, sessionId, "sse", state, responsesLite); + const headers = createCodexHeaders( + requestHeaders, + accountId, + apiKey, + sessionId, + "sse", + state, + responsesLite, + requestMetadata, + ); CODEX_DEBUG && logger.debug("[codex] codex request", { url, @@ -3322,6 +3753,7 @@ function createCodexHeaders( transport: CodexTransport = "sse", state?: CodexWebSocketSessionState, responsesLite = false, + requestMetadata?: CodexCompatibilityIdentity, ): Headers { const headers = new Headers(initHeaders ?? {}); headers.delete("x-api-key"); @@ -3345,6 +3777,15 @@ function createCodexHeaders( headers.delete(OPENAI_HEADERS.SESSION_ID); headers.delete("x-client-request-id"); } + headers.delete(OPENAI_HEADERS.INSTALLATION_ID); + if (requestMetadata) { + applyCodexCompatibilityHeaders(headers, requestMetadata); + } else { + headers.delete(OPENAI_HEADERS.SCOPED_SESSION_ID); + headers.delete(OPENAI_HEADERS.THREAD_ID); + headers.delete(OPENAI_HEADERS.WINDOW_ID); + headers.delete(OPENAI_HEADERS.TURN_METADATA); + } if (state?.turnState) { headers.set(X_CODEX_TURN_STATE_HEADER, state.turnState); } else { @@ -3356,9 +3797,9 @@ function createCodexHeaders( headers.delete(X_MODELS_ETAG_HEADER); } if (responsesLite) { - headers.set(X_OPENAI_INTERNAL_CODEX_RESPONSES_LITE_HEADER, "true"); + headers.set(OPENAI_HEADERS.RESPONSES_LITE, "true"); } else { - headers.delete(X_OPENAI_INTERNAL_CODEX_RESPONSES_LITE_HEADER); + headers.delete(OPENAI_HEADERS.RESPONSES_LITE); } if (transport === "sse") { headers.set("accept", "text/event-stream"); @@ -3382,6 +3823,10 @@ function redactHeaders(headers: Headers): Record { lower.includes("account") || lower.includes("session") || lower.includes("conversation") || + lower.includes("thread") || + lower.includes("window") || + lower.includes("installation") || + lower.startsWith("x-codex-turn") || lower === "x-client-request-id" || lower === "cookie" ) { diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index d603130c3..8c095137f 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -1,25 +1,46 @@ -import type { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { supportsAllTurnsReasoningContext, supportsCodexReasoningSummary } from "@oh-my-pi/pi-catalog/identity"; import { requireSupportedEffort } from "@oh-my-pi/pi-catalog/model-thinking"; -import type { Api, Model } from "../../types"; +import type { Model } from "../../types"; +import { mapOpenAIReasoningEffort } from "../openai-shared"; /** Reasoning replay scope for the Codex Responses API (`reasoning.context`). */ export type CodexReasoningContext = "auto" | "current_turn" | "all_turns"; +/** User-facing effort levels accepted by Codex request options. */ +type CodexCallerEffort = "minimal" | "low" | "medium" | "high" | "xhigh"; + +/** Caller literal → catalog `Effort` bridge (the enum is nominal). */ +const EFFORT_BY_NAME: Record = { + minimal: Effort.Minimal, + low: Effort.Low, + medium: Effort.Medium, + high: Effort.High, + xhigh: Effort.XHigh, +}; + export interface ReasoningConfig { - effort: "none" | "minimal" | "low" | "medium" | "high" | "xhigh"; + effort: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; summary?: "auto" | "concise" | "detailed"; context?: CodexReasoningContext; + /** Pro reasoning serving mode (gpt-5.6+ catalog pro aliases). */ + mode?: "pro"; } export interface CodexRequestOptions { - reasoningEffort?: ReasoningConfig["effort"]; + /** User-facing effort; the wire-only `max` tier is reached via the model's effort map. */ + reasoningEffort?: CodexCallerEffort | "none"; reasoningSummary?: ReasoningConfig["summary"] | null; /** Explicit `reasoning.context` override; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */ reasoningContext?: CodexReasoningContext; textVerbosity?: "low" | "medium" | "high"; include?: string[]; - /** Responses Lite transport contract: strips image detail and disables parallel tool calling, mirroring codex-rs. */ + /** + * Responses Lite transport override; defaults to the model's + * `useResponsesLite`. Lite moves instructions/tools into input items, + * strips image detail, and disables parallel tool calling (codex-rs + * `use_responses_lite`). + */ responsesLite?: boolean; } @@ -32,6 +53,8 @@ export interface InputItem { name?: string; output?: unknown; arguments?: unknown; + /** `additional_tools` developer item payload (Responses Lite). */ + tools?: unknown; } export interface RequestBody { @@ -42,6 +65,8 @@ export interface RequestBody { input?: InputItem[]; tools?: unknown; tool_choice?: unknown; + /** Concurrent reasoning-summary delivery (codex-rs `StreamOptions`). */ + stream_options?: { reasoning_summary_delivery: "sequential_cutoff" }; // Sampling controls (temperature/top_p/top_k/min_p/presence_penalty/ // repetition_penalty/frequency_penalty/stop) are intentionally absent: the // Codex backend rejects every one with a 400 `Unsupported parameter`, so @@ -60,30 +85,52 @@ export interface RequestBody { [key: string]: unknown; } -function containsInputImage(value: unknown): boolean { - if (!value || typeof value !== "object") return false; - if ((value as { type?: unknown }).type === "input_image") return true; - if (Array.isArray(value)) { - for (const item of value) { - if (containsInputImage(item)) return true; - } - return false; - } - for (const item of Object.values(value)) { - if (containsInputImage(item)) return true; - } - return false; +/** + * Resolve whether a Codex request uses the Responses Lite transport: an + * explicit option wins, otherwise the model's catalog flag (codex-rs + * `model_info.use_responses_lite`) decides. + */ +export function resolveCodexResponsesLite( + model: Model<"openai-codex-responses">, + requested: boolean | undefined, +): boolean { + return requested ?? model.useResponsesLite === true; } -/** Returns whether a Codex request can use the text-only Responses Lite transport. */ -export function shouldUseCodexResponsesLite(body: RequestBody, requested: boolean | undefined): boolean { - return requested === true && !containsInputImage(body.input); +/** + * Clamp a user-facing effort to the model's ladder, then remap to the wire + * tier (e.g. GPT-5.6's shifted five-tier scale sends `max` for user `xhigh`). + * A mapped value outside the Codex wire vocabulary is a broken compat/model + * effort map — fail loudly rather than silently sending a different tier. + */ +function mapCodexWireEffort( + model: Model<"openai-codex-responses">, + effort: CodexCallerEffort, +): ReasoningConfig["effort"] { + const mapped = mapOpenAIReasoningEffort(model, model.compat, requireSupportedEffort(model, EFFORT_BY_NAME[effort])); + switch (mapped) { + case "none": + case "minimal": + case "low": + case "medium": + case "high": + case "xhigh": + case "max": + return mapped; + default: + throw new Error( + `Effort map for ${model.provider}/${model.id} produced invalid Codex reasoning effort "${mapped}"`, + ); + } } -function getReasoningConfig(model: Model, options: CodexRequestOptions): ReasoningConfig { +function getReasoningConfig( + model: Model<"openai-codex-responses">, + effort: NonNullable, + options: CodexRequestOptions, +): ReasoningConfig { const config: ReasoningConfig = { - effort: - options.reasoningEffort === "none" ? "none" : requireSupportedEffort(model, options.reasoningEffort as Effort), + effort: effort === "none" ? "none" : mapCodexWireEffort(model, effort), }; // `reasoning.summary` is accepted only from gpt-5.4 onward; earlier Codex ids // (gpt-5.1-codex, gpt-5.3-codex, gpt-5.3-codex-spark) reject it with @@ -196,27 +243,65 @@ function repairToolCallPairs(input: InputItem[]): InputItem[] { * `detail` from every input image (message content and tool outputs) before * sending, letting the server choose. */ -function stripImageDetails(input: InputItem[]): void { +function stripImageDetails(input: unknown[]): void { for (const item of input) { - for (const collection of [item.content, item.output]) { + if (!item || typeof item !== "object") continue; + const content = "content" in item ? item.content : undefined; + const output = "output" in item ? item.output : undefined; + for (const collection of [content, output]) { if (!Array.isArray(collection)) continue; for (const part of collection) { - if ( - part && - typeof part === "object" && - (part as { type?: unknown }).type === "input_image" && - "detail" in part - ) { - part.detail = undefined; - } + if (!part || typeof part !== "object") continue; + if (!("type" in part) || part.type !== "input_image") continue; + if ("detail" in part) part.detail = undefined; } } } } +/** + * Structural view of a Responses-style body mutated by the Lite rewrite. + * Loose (`unknown`) property types let the turn transformer (`RequestBody`) + * and the agent's remote-compaction payloads reuse one shaper. + */ +export interface CodexLiteShapedBody { + instructions?: unknown; + tools?: unknown; + input?: unknown; + parallel_tool_calls?: unknown; +} + +/** + * Applies the Responses Lite body contract in place (codex-rs + * `build_responses_request` with `use_responses_lite`): strips pinned image + * detail, forces parallel tool calling off, moves tools into a leading + * `additional_tools` developer item and the base instructions into a + * developer message, then omits top-level `instructions`/`tools`. Shared by + * normal turns and both remote-compaction paths — codex-rs routes + * `/responses/compact` through the same builder. + */ +export function applyCodexResponsesLiteShape(body: CodexLiteShapedBody): void { + const input = Array.isArray(body.input) ? body.input : []; + stripImageDetails(input); + body.parallel_tool_calls = false; + const prefix: InputItem[] = [ + { type: "additional_tools", role: "developer", tools: Array.isArray(body.tools) ? body.tools : [] }, + ]; + if (typeof body.instructions === "string" && body.instructions.length > 0) { + prefix.push({ + type: "message", + role: "developer", + content: [{ type: "input_text", text: body.instructions }], + }); + } + body.input = [...prefix, ...input]; + delete body.instructions; + delete body.tools; +} + export async function transformRequestBody( body: RequestBody, - model: Model, + model: Model<"openai-codex-responses">, options: CodexRequestOptions = {}, prompt?: { developerMessages: string[] }, ): Promise { @@ -287,20 +372,13 @@ export async function transformRequestBody( } } - const responsesLite = shouldUseCodexResponsesLite(body, options.responsesLite); + const responsesLite = resolveCodexResponsesLite(model, options.responsesLite); if (responsesLite) { - if (Array.isArray(body.input)) { - stripImageDetails(body.input); - } - // Responses Lite does not support parallel tool calling; codex-rs forces - // it off (`prompt.parallel_tool_calls && !use_responses_lite`). - if (body.tools !== undefined) { - body.parallel_tool_calls = false; - } + applyCodexResponsesLiteShape(body); } if (options.reasoningEffort !== undefined) { - const reasoningConfig = getReasoningConfig(model, options); + const reasoningConfig = getReasoningConfig(model, options.reasoningEffort, options); body.reasoning = { ...body.reasoning, ...reasoningConfig, @@ -323,6 +401,23 @@ export async function transformRequestBody( } else { delete body.reasoning; } + // Catalog pro aliases (`gpt-5.6-*-pro`): applied after the effort branch so + // the mode is sent even when no effort is set (the branch above deletes + // `body.reasoning` in that case) — mode and effort are independent fields. + if (model.reasoningMode) { + body.reasoning = { ...body.reasoning, mode: model.reasoningMode }; + } + + // Concurrent reasoning summaries (codex-rs `concurrent_reasoning_summaries` + // feature): `sequential_cutoff` lets the server stream output without + // blocking on summary generation. Only meaningful when a summary is + // requested; codex-rs additionally gates on its OpenAI provider check, + // which is inherent here. + if (body.reasoning?.summary !== undefined) { + body.stream_options = { reasoning_summary_delivery: "sequential_cutoff" }; + } else { + delete body.stream_options; + } body.text = { ...body.text, diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index 8006628c2..45a56d9ce 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -82,6 +82,7 @@ import { createInitialResponsesAssistantMessage, createOpenAIStrictToolsState, disableStrictToolsForScope, + getOpenAIPromptCacheKey, getOpenAIStrictToolsScope, isCompiledGrammarTooLargeStrictError, isOpenRouterAnthropicModel, @@ -616,6 +617,7 @@ const streamOpenAICompletionsOnce = ( apiKey, options?.headers, options?.initiatorOverride, + getOpenAIPromptCacheKey(options), ); const premiumRequestsTotal = copilotPremiumRequests; let appliedStrictTools = false; @@ -1359,6 +1361,7 @@ function createRequestSetup( apiKey?: string, extraHeaders?: Record, initiatorOverride?: MessageAttribution, + promptCacheSessionId?: string, ): OpenAIRequestSetup & { baseUrl: string } { const apiVersion = $env.AZURE_OPENAI_API_VERSION || "2024-10-21"; const deploymentName = parseAzureDeploymentNameMap($env.AZURE_OPENAI_DEPLOYMENT_NAME_MAP).get(model.id) ?? model.id; @@ -1366,6 +1369,7 @@ function createRequestSetup( apiKey, extraHeaders, initiatorOverride, + promptCacheSessionId, messages: context.messages, defaultBaseUrl: "https://api.openai.com/v1", // Provider auth/header overlay: Kimi-code hosts require shared client @@ -1413,7 +1417,11 @@ function buildParams( context: Context, options: OpenAICompletionsOptions | undefined, toolStrictModeOverride?: ToolStrictModeOverride, -): { params: OpenAICompletionsParams; toolStrictMode: AppliedToolStrictMode; strictToolsApplied: boolean } { +): { + params: OpenAICompletionsParams; + toolStrictMode: AppliedToolStrictMode; + strictToolsApplied: boolean; +} { const initialPolicy = resolveOpenAICompatForRequest(model, options); const initialCompat = initialPolicy.compat as ResolvedOpenAICompat; diff --git a/packages/ai/src/providers/openai-responses-wire.ts b/packages/ai/src/providers/openai-responses-wire.ts index 5246b5eaf..7992ec319 100644 --- a/packages/ai/src/providers/openai-responses-wire.ts +++ b/packages/ai/src/providers/openai-responses-wire.ts @@ -6318,6 +6318,13 @@ export interface Reasoning { * - `xhigh` is supported for all models after `gpt-5.1-codex-max`. */ effort?: ReasoningEffort | null; + /** + * **gpt-5.6 and later models only** + * + * Reasoning serving mode. `pro` routes the request to the pro reasoning + * path (more compute per response); omit for the standard path. + */ + mode?: "pro" | null; /** * @deprecated **Deprecated:** use `summary` instead. * diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index 759fdbfdc..3aa740236 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -76,7 +76,7 @@ import { createInitialResponsesAssistantMessage, createOpenAIStrictToolsState, disableStrictToolsForScope, - getOpenAIResponsesPromptCacheKey, + getOpenAIPromptCacheKey, getOpenAIResponsesRoutingSessionId, getOpenAIStrictToolsScope, getOpenRouterResponsesSessionId, @@ -390,7 +390,7 @@ const streamOpenAIResponsesOnce = ( // stable prompt-cache key independently. Side-channel calls use this to // avoid perturbing provider conversation state without cold-starting the cache. const routingSessionId = getOpenAIResponsesRoutingSessionId(options); - const promptCacheSessionId = getOpenAIResponsesPromptCacheKey(options); + const promptCacheSessionId = getOpenAIPromptCacheKey(options); const apiKey = options?.apiKey || getEnvApiKey(model.provider) || ""; const { headers, copilotPremiumRequests, baseUrl } = resolveOpenAIRequestSetup(model, { apiKey, @@ -818,7 +818,7 @@ export function buildParams( } const cacheRetention = resolveCacheRetention(options?.cacheRetention); - const promptCacheKey = getOpenAIResponsesPromptCacheKey(options); + const promptCacheKey = getOpenAIPromptCacheKey(options); const modelId = applyWireModelIdTransform( model.requestModelId ?? model.id, model.compat.wireModelIdMode, @@ -904,6 +904,13 @@ export function buildParams( model.thinking?.effortMap?.[effort as NonNullable] ?? effort, }); + // Catalog pro aliases (`gpt-5.6-*-pro`): merge AFTER the compat policy so the + // mode survives every policy branch (disabled/omitted effort included) while + // keeping whatever effort/summary the policy produced — mode and effort are + // independent wire fields. + if (model.reasoningMode) { + params.reasoning = { ...params.reasoning, mode: model.reasoningMode }; + } applyOpenAIGatewayRouting(params, model.compat); diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index 1d40b0ac5..789f31213 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -123,7 +123,8 @@ export interface OpenAIRequestSetupModel extends OpenAIModelIdentity { compat?: Pick; } -export interface OpenAIResponsesCacheOptions { +/** Cache identity controls shared by OpenAI-family transports. */ +export interface OpenAICacheOptions { cacheRetention?: CacheRetention; sessionId?: string; promptCacheKey?: string; @@ -175,6 +176,14 @@ function applyCoreWeaveProjectHeader(headers: Record): void { } } +function setHeaderIfAbsent(headers: Record, name: string, value: string): void { + const normalizedName = name.toLowerCase(); + for (const existingName in headers) { + if (existingName.toLowerCase() === normalizedName) return; + } + headers[name] = value; +} + export function resolveOpenAIRequestSetup( model: OpenAIRequestSetupModel, options: OpenAIRequestSetupOptions, @@ -257,11 +266,11 @@ export function resolveOpenAIRequestSetup( } if (options.openAISessionId && model.provider === "openai") { - headers.session_id ??= options.openAISessionId; - headers["x-client-request-id"] ??= options.openAISessionId; + setHeaderIfAbsent(headers, "session_id", options.openAISessionId); + setHeaderIfAbsent(headers, "x-client-request-id", options.openAISessionId); } if (options.promptCacheSessionId && model.compat?.promptCacheSessionHeader) { - headers[model.compat.promptCacheSessionHeader] ??= options.promptCacheSessionId; + setHeaderIfAbsent(headers, model.compat.promptCacheSessionHeader, options.promptCacheSessionId); } if (options.defaultBaseUrl !== undefined) { @@ -368,7 +377,8 @@ export function calculateOpenAIUsageAccounting(accounting: OpenAIUsageAccounting }; } -export function normalizeOpenAIResponsesPromptCacheKey(sessionId: string | undefined): string | undefined { +/** Normalize a cache identity to the wire limit accepted by OpenAI-family providers. */ +export function normalizeOpenAIPromptCacheKey(sessionId: string | undefined): string | undefined { return normalizeOpenAIStableId(sessionId, 64, "pc_"); } @@ -376,20 +386,21 @@ export function normalizeOpenRouterResponsesSessionId(sessionId: string | undefi return normalizeOpenAIStableId(sessionId, 256, "session_"); } -export function getOpenAIResponsesPromptCacheKey(options: OpenAIResponsesCacheOptions | undefined): string | undefined { +/** Resolve a prompt-cache identity, falling back to the provider session unless caching is disabled. */ +export function getOpenAIPromptCacheKey(options: OpenAICacheOptions | undefined): string | undefined { if (resolveCacheRetention(options?.cacheRetention) === "none") return undefined; - return normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId); + return normalizeOpenAIPromptCacheKey(options?.promptCacheKey ?? options?.sessionId); } export function getOpenAIResponsesRoutingSessionId( - options: Pick | undefined, + options: Pick | undefined, ): string | undefined { if (resolveCacheRetention(options?.cacheRetention) === "none") return undefined; - return normalizeOpenAIResponsesPromptCacheKey(options?.sessionId); + return normalizeOpenAIPromptCacheKey(options?.sessionId); } export function getOpenRouterResponsesSessionId( - options: Pick | undefined, + options: Pick | undefined, ): string | undefined { if (resolveCacheRetention(options?.cacheRetention) === "none") return undefined; return normalizeOpenRouterResponsesSessionId(options?.sessionId); @@ -695,13 +706,19 @@ export interface OpenAICompatPolicy { }; } -function mapOpenAIReasoningEffort( +/** + * Map a user-facing effort to the provider wire value: explicit compat + * override first, then the model's baked `thinking.effortMap`, else identity. + * Shared by the chat-completions/Responses policy resolver and the Codex + * request transformer. + */ +export function mapOpenAIReasoningEffort( model: Pick, - compat: OpenAICompatPolicyCompat, + compat: { reasoningEffortMap?: Partial> } | undefined, effort: string, ): string { const level = effort as Effort; - return compat.reasoningEffortMap?.[level] ?? model.thinking?.effortMap?.[level] ?? effort; + return compat?.reasoningEffortMap?.[level] ?? model.thinking?.effortMap?.[level] ?? effort; } function isImplicitDisableWhenNotRequested(disableMode: OpenAIReasoningDisableMode): boolean { @@ -1059,6 +1076,7 @@ export const OPENAI_RESPONSES_PROGRESS_EVENT_TYPES: ReadonlySet = new Se "response.output_item.added", "response.reasoning_summary_part.added", "response.reasoning_summary_text.delta", + "response.reasoning_summary_text.done", "response.reasoning_summary_part.done", "response.reasoning_text.delta", "response.content_part.added", @@ -1684,6 +1702,41 @@ export function appendReasoningSummaryPart( item.summary.push(part); } +// Sequential-cutoff streams may repeat the full canonical summary as later parts. +function foldReasoningSummary(parts: ResponseReasoningItem["summary"] | undefined): string { + if (!parts) return ""; + let canonical = ""; + for (const part of parts) { + const text = part.text; + if (!text || text === canonical) continue; + const extendsCanonical = text.startsWith(canonical) && text[canonical.length] === "\n"; + canonical = !canonical || extendsCanonical ? text : `${canonical}\n\n${text}`; + } + return canonical; +} + +/** Chooses final reasoning text without making sequential-cutoff results disagree with emitted deltas. */ +export function finalizeReasoningThinking( + item: ResponseReasoningItem, + streamedThinking: string, + options: { cumulativeSummarySnapshots?: boolean } = {}, +): string { + const summaryThinking = options.cumulativeSummarySnapshots + ? foldReasoningSummary(item.summary) + : (item.summary?.map(part => part.text).join("\n\n") ?? ""); + if ( + options.cumulativeSummarySnapshots && + streamedThinking && + summaryThinking && + summaryThinking !== streamedThinking + ) { + return streamedThinking; + } + if (summaryThinking) return summaryThinking; + const contentThinking = item.content?.[0]?.type === "reasoning_text" ? (item.content[0].text ?? "") : ""; + return contentThinking || streamedThinking || ""; +} + export function appendReasoningSummaryTextDelta( item: ResponseReasoningItem, block: ThinkingContent, @@ -1715,6 +1768,36 @@ export function appendReasoningSummaryPartDone( stream.push({ type: "thinking_delta", contentIndex, delta: "\n\n", partial: output }); } +/** + * Applies an atomic `response.reasoning_summary_text.done` snapshot. + * + * Sequential-cutoff streams can replay an index or send the accumulated + * summary as a later part. Rebuild the canonical summary and emit only its + * append-only suffix. Divergent corrections stay buffered until finalization + * so delta consumers never receive suffixes based on unseen replacement text. + */ +export function applyReasoningSummaryDone( + item: ResponseReasoningItem, + block: ThinkingContent, + text: string, + summaryIndex: number, + stream: AssistantMessageEventStream, + output: AssistantMessage, + contentIndex: number, +): void { + item.summary = item.summary || []; + while (item.summary.length <= summaryIndex) { + item.summary.push({ type: "summary_text", text: "" }); + } + item.summary[summaryIndex].text = text; + const after = foldReasoningSummary(item.summary); + if (!after.startsWith(block.thinking)) return; + const delta = after.slice(block.thinking.length); + if (!delta) return; + block.thinking = after; + stream.push({ type: "thinking_delta", contentIndex, delta, partial: output }); +} + export function appendMessageContentPart( item: ResponseOutputMessage, part: ResponseContentPartAddedEvent["part"] | undefined, @@ -2208,12 +2291,6 @@ export async function processResponsesStream( ? lookupOpenItem({ output_index: event.output_index, item_id: item.id ?? item.call_id }) : lookupOpenItem({ output_index: event.output_index, item_id: item.id }); if (item.type === "reasoning") { - const thinking = - item.summary?.length > 0 - ? item.summary.map(part => part.text).join("\n\n") - : item.content?.[0]?.type === "reasoning_text" - ? (item.content[0].text ?? "") - : ""; // Prefer the routed entry; the bare itemId find misroutes when ids are // absent (`undefined === undefined` matches the FIRST thinking block) and // misses entirely when the done-event id drifts from the added-event id. @@ -2224,12 +2301,12 @@ export async function processResponsesStream( | ThinkingContent | undefined); if (reasoningBlock) { - reasoningBlock.thinking = thinking; + reasoningBlock.thinking = finalizeReasoningThinking(item, reasoningBlock.thinking); reasoningBlock.thinkingSignature = JSON.stringify(item); stream.push({ type: "thinking_end", contentIndex: contentIndexOf(reasoningBlock), - content: thinking, + content: reasoningBlock.thinking, partial: output, }); } diff --git a/packages/ai/src/providers/register-builtins.ts b/packages/ai/src/providers/register-builtins.ts index efd6b8dd6..8bb3a20f0 100644 --- a/packages/ai/src/providers/register-builtins.ts +++ b/packages/ai/src/providers/register-builtins.ts @@ -157,6 +157,7 @@ let openAICompletionsProviderModulePromise: Promise> | undefined; let ollamaProviderModulePromise: Promise> | undefined; let cursorProviderModulePromise: Promise> | undefined; +let cursorProviderModuleOverride: LazyProviderModule<"cursor-agent"> | undefined; let devinProviderModulePromise: Promise> | undefined; let bedrockProviderModuleOverride: LazyProviderModule<"bedrock-converse-stream"> | undefined; let bedrockProviderModulePromise: Promise> | undefined; @@ -167,6 +168,12 @@ export function setBedrockProviderModule(module: BedrockProviderModule): void { }; } +export function setCursorProviderModule(module: CursorProviderModule): void { + cursorProviderModuleOverride = { + stream: module.streamCursor, + }; +} + // --------------------------------------------------------------------------- // Stream forwarding / error helpers // --------------------------------------------------------------------------- @@ -245,6 +252,10 @@ function forwardStream( (limits?.openAIIdleEnvFloorsFirstEvent ? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, limits.defaultFirstEventTimeoutMs) : getStreamFirstEventTimeoutMs(idleTimeoutMs, limits?.defaultFirstEventTimeoutMs))); + // Providers with a server-driven local tool bridge (e.g. the Cursor + // exec channel) mark their stream busy while a local tool runs; the + // watchdog must not read that silence as a provider stall (#4593). + const localWorkSource = source instanceof EventStreamImpl ? source : undefined; const watchedSource = iterateWithIdleTimeout(source, { idleTimeoutMs, firstItemTimeoutMs, @@ -260,6 +271,7 @@ function forwardStream( // `idleTimeoutMs` while we're still legitimately waiting on the model's // first response (slow first-token from reasoning models, cold proxies, etc.). isProgressItem: event => (event as AssistantMessageEvent).type !== "start", + hasPendingLocalWork: localWorkSource ? () => localWorkSource.hasPendingLocalWork : undefined, }); for await (const event of watchedSource) { @@ -411,6 +423,9 @@ function loadOllamaProviderModule(): Promise> } function loadCursorProviderModule(): Promise> { + if (cursorProviderModuleOverride) { + return Promise.resolve(cursorProviderModuleOverride); + } cursorProviderModulePromise ||= import("./cursor").then(module => { const provider = module as CursorProviderModule; return { stream: provider.streamCursor }; diff --git a/packages/ai/src/registry/novita.ts b/packages/ai/src/registry/novita.ts new file mode 100644 index 000000000..a4c6abf23 --- /dev/null +++ b/packages/ai/src/registry/novita.ts @@ -0,0 +1,22 @@ +import { createApiKeyLogin } from "./api-key-login"; +import type { ProviderDefinition } from "./types"; + +export const loginNovita = createApiKeyLogin({ + providerLabel: "Novita", + authUrl: "https://novita.ai/settings/key-management", + instructions: "Create or copy your API key from the Novita dashboard", + promptMessage: "Paste your Novita API key", + placeholder: "sk_...", + validation: { + kind: "models-endpoint", + provider: "Novita", + modelsUrl: "https://api.novita.ai/openapi/v1/billing/balance/detail", + headers: { "Content-Type": "application/json" }, + }, +}); + +export const novitaProvider = { + id: "novita", + name: "Novita", + login: loginNovita, +} satisfies ProviderDefinition & { readonly id: "novita" }; diff --git a/packages/ai/src/registry/oauth/__tests__/xai-oauth.test.ts b/packages/ai/src/registry/oauth/__tests__/xai-oauth.test.ts index e80a6a544..ac5012ba0 100644 --- a/packages/ai/src/registry/oauth/__tests__/xai-oauth.test.ts +++ b/packages/ai/src/registry/oauth/__tests__/xai-oauth.test.ts @@ -1,5 +1,5 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; -import { isXAIAccessTokenExpiring, refreshXAIOAuthToken, validateXAIEndpoint, XAIOAuthFlow } from "../xai-oauth"; +import { isXAIAccessTokenExpiring, loginXAIOAuth, refreshXAIOAuthToken, validateXAIEndpoint } from "../xai-oauth"; afterEach(() => { vi.restoreAllMocks(); @@ -11,6 +11,76 @@ function jwtWithExp(exp: number): string { return `${header}.${payload}.sig`; } +const DISCOVERY_URL = "https://auth.x.ai/.well-known/openid-configuration"; +const DEVICE_CODE_URL = "https://auth.x.ai/oauth2/device/code"; +const TOKEN_ENDPOINT = "https://auth.x.ai/oauth2/token"; +const CLIENT_ID = "b1a00492-073a-47ea-816f-4c329264a828"; +const SCOPE = "openid profile email offline_access grok-cli:access api:access"; + +const DEVICE_AUTHORIZATION = { + device_code: "device-code-123", + user_code: "ABCD-EFGH", + verification_uri: "https://auth.x.ai/activate", + verification_uri_complete: "https://auth.x.ai/activate?user_code=ABCD-EFGH", + expires_in: 600, + interval: 1, +}; + +type RecordedRequest = { + url: string; + init: RequestInit | undefined; +}; + +type TokenResponse = { + body: unknown; + status?: number; +}; + +function jsonResponse(body: unknown, status: number = 200): Response { + return new Response(JSON.stringify(body), { + status, + headers: { "Content-Type": "application/json" }, + }); +} + +function createDeviceFlowFetch(tokenResponses: readonly TokenResponse[]) { + const requests: RecordedRequest[] = []; + let tokenResponseIndex = 0; + const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => { + const url = typeof input === "string" ? input : input instanceof Request ? input.url : input.toString(); + requests.push({ url, init }); + + if (url === DISCOVERY_URL) { + return jsonResponse({ token_endpoint: TOKEN_ENDPOINT }); + } + if (url === DEVICE_CODE_URL) { + return jsonResponse(DEVICE_AUTHORIZATION); + } + if (url === TOKEN_ENDPOINT) { + const tokenResponse = tokenResponses[tokenResponseIndex]; + tokenResponseIndex += 1; + if (!tokenResponse) { + throw new Error(`Unexpected xAI token poll ${tokenResponseIndex}`); + } + return jsonResponse(tokenResponse.body, tokenResponse.status); + } + throw new Error(`Unexpected xAI OAuth request: ${url}`); + }); + + return { + fetchMock: fetchMock as unknown as typeof fetch, + requests, + }; +} + +function requestForm(request: RecordedRequest | undefined): URLSearchParams { + const body = request?.init?.body; + if (!(body instanceof URLSearchParams)) { + throw new Error("Expected an application/x-www-form-urlencoded request body"); + } + return body; +} + describe("isXAIAccessTokenExpiring", () => { it("returns false for an empty string", () => { expect(isXAIAccessTokenExpiring("")).toBe(false); @@ -63,97 +133,138 @@ describe("refreshXAIOAuthToken", () => { }); }); -describe("XAIOAuthFlow", () => { - it("pins the redirect URI to xAI's allowlisted loopback port", () => { - const flow = new XAIOAuthFlow({}); - - expect(flow.redirectUri).toBe("http://127.0.0.1:56121/callback"); - }); - - it("uses pasted-code login without starting a callback server", async () => { - const serveSpy = vi.spyOn(Bun, "serve").mockImplementation(() => { - throw new Error("callback server should not start"); - }); - let authUrl = ""; - let tokenRequestBody = ""; - const progress: string[] = []; - const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => { - const url = typeof input === "string" ? input : input instanceof Request ? input.url : input.toString(); - if (url.includes("/.well-known/openid-configuration")) { - return new Response( - JSON.stringify({ - authorization_endpoint: "https://auth.x.ai/oauth/authorize", - token_endpoint: "https://auth.x.ai/oauth/token", - }), - { status: 200, headers: { "Content-Type": "application/json" } }, - ); - } - tokenRequestBody = init?.body instanceof URLSearchParams ? init.body.toString() : String(init?.body ?? ""); - return new Response( - JSON.stringify({ +describe("loginXAIOAuth", () => { + it("performs the RFC 8628 device flow and returns the issued credentials", async () => { + const now = 1_800_000_000_000; + vi.spyOn(Date, "now").mockReturnValue(now); + const { fetchMock, requests } = createDeviceFlowFetch([ + { + body: { access_token: "access-token", refresh_token: "refresh-token", expires_in: 3600, - }), - { status: 200, headers: { "Content-Type": "application/json" } }, - ); + }, + }, + ]); + const authEvents: Array<{ url: string; instructions?: string }> = []; + const progress: string[] = []; + const onAuth = vi.fn((info: { url: string; instructions?: string }) => { + authEvents.push(info); + }); + const onProgress = vi.fn((message: string) => { + progress.push(message); + }); + const onManualCodeInput = vi.fn(async () => { + throw new Error("device authorization must not request a pasted code"); }); - const flow = new XAIOAuthFlow({ - fetch: fetchMock as unknown as typeof fetch, - onAuth: info => { - authUrl = info.url; - }, - onManualCodeInput: async () => { - const parsed = new URL(authUrl); - const redirectUri = parsed.searchParams.get("redirect_uri") ?? ""; - const state = parsed.searchParams.get("state") ?? ""; - return `${redirectUri}?code=code-xyz&state=${encodeURIComponent(state)}`; - }, - onProgress: message => progress.push(message), + const credentials = await loginXAIOAuth({ + fetch: fetchMock, + onAuth, + onProgress, + onManualCodeInput, }); - const credentials = await flow.login(); - const authorizeUrl = new URL(authUrl); - const tokenParams = new URLSearchParams(tokenRequestBody); + expect(requests.map(request => request.url)).toEqual([DISCOVERY_URL, DEVICE_CODE_URL, TOKEN_ENDPOINT]); - expect(serveSpy).not.toHaveBeenCalled(); - expect(authorizeUrl.searchParams.get("redirect_uri")).toBe("http://127.0.0.1:56121/callback"); - expect(progress).toContain("Waiting for pasted authorization code..."); - expect(tokenParams.get("code")).toBe("code-xyz"); - expect(credentials.access).toBe("access-token"); - expect(credentials.refresh).toBe("refresh-token"); + const discoveryRequest = requests[0]; + expect(discoveryRequest?.init?.method).toBe("GET"); + expect(new Headers(discoveryRequest?.init?.headers).get("Accept")).toBe("application/json"); + + const deviceRequest = requests[1]; + expect(deviceRequest?.init?.method).toBe("POST"); + const deviceHeaders = new Headers(deviceRequest?.init?.headers); + expect(deviceHeaders.get("Content-Type")).toBe("application/x-www-form-urlencoded"); + expect(deviceHeaders.get("Accept")).toBe("application/json"); + const deviceForm = requestForm(deviceRequest); + expect([...deviceForm.keys()].sort()).toEqual(["client_id", "scope"]); + expect(Object.fromEntries(deviceForm)).toEqual({ + client_id: CLIENT_ID, + scope: SCOPE, + }); + + const tokenRequest = requests[2]; + expect(tokenRequest?.init?.method).toBe("POST"); + const tokenHeaders = new Headers(tokenRequest?.init?.headers); + expect(tokenHeaders.get("Content-Type")).toBe("application/x-www-form-urlencoded"); + expect(tokenHeaders.get("Accept")).toBe("application/json"); + const tokenForm = requestForm(tokenRequest); + expect([...tokenForm.keys()].sort()).toEqual(["client_id", "device_code", "grant_type"]); + expect(Object.fromEntries(tokenForm)).toEqual({ + grant_type: "urn:ietf:params:oauth:grant-type:device_code", + client_id: CLIENT_ID, + device_code: DEVICE_AUTHORIZATION.device_code, + }); + + expect(authEvents).toEqual([ + { + url: DEVICE_AUTHORIZATION.verification_uri_complete, + instructions: `Enter code: ${DEVICE_AUTHORIZATION.user_code}`, + }, + ]); + expect(authEvents[0]?.instructions).not.toMatch(/hermes/i); + expect(onManualCodeInput).not.toHaveBeenCalled(); + expect(progress).toEqual(["Waiting for xAI device authorization..."]); + expect(credentials).toEqual({ + access: "access-token", + refresh: "refresh-token", + expires: now + 3_300_000, + }); }); -}); -describe("XAIOAuthFlow.exchangeToken", () => { - it("rejects when the token-exchange response is missing access_token", async () => { - const fetchMock = vi.fn(async (input: string | URL) => { - const url = typeof input === "string" ? input : input.toString(); - if (url.includes("/.well-known/openid-configuration")) { - return new Response( - JSON.stringify({ - authorization_endpoint: "https://auth.x.ai/oauth/authorize", - token_endpoint: "https://auth.x.ai/oauth/token", - }), - { status: 200, headers: { "Content-Type": "application/json" } }, - ); - } - // Token-exchange response deliberately omits `access_token` to exercise - // the missing-token rejection path. The value of `refresh_token` here is - // a literal test marker, not a real secret — the test verifies - // exchangeToken throws before any token would be persisted. - return new Response(JSON.stringify({ refresh_token: "stub-refresh-token-for-test-only" }), { - status: 200, - headers: { "Content-Type": "application/json" }, - }); - }); + it("continues through authorization_pending and slow_down responses", async () => { + const sleepSpy = vi.spyOn(Bun, "sleep").mockResolvedValue(undefined); + const { fetchMock, requests } = createDeviceFlowFetch([ + { status: 400, body: { error: "authorization_pending" } }, + { status: 400, body: { error: "slow_down" } }, + { + body: { + access_token: "eventual-access-token", + refresh_token: "eventual-refresh-token", + expires_in: 3600, + }, + }, + ]); - const flow = new XAIOAuthFlow({ fetch: fetchMock as unknown as typeof fetch }); - await flow.generateAuthUrl("state-abc", "http://127.0.0.1:56121/callback"); + const credentials = await loginXAIOAuth({ fetch: fetchMock }); - await expect(flow.exchangeToken("code-xyz", "state-abc", "http://127.0.0.1:56121/callback")).rejects.toThrow( - /access_token/, + const tokenRequests = requests.filter(request => request.url === TOKEN_ENDPOINT); + expect(tokenRequests).toHaveLength(3); + expect(tokenRequests.map(request => Object.fromEntries(requestForm(request)))).toEqual([ + { + grant_type: "urn:ietf:params:oauth:grant-type:device_code", + client_id: CLIENT_ID, + device_code: DEVICE_AUTHORIZATION.device_code, + }, + { + grant_type: "urn:ietf:params:oauth:grant-type:device_code", + client_id: CLIENT_ID, + device_code: DEVICE_AUTHORIZATION.device_code, + }, + { + grant_type: "urn:ietf:params:oauth:grant-type:device_code", + client_id: CLIENT_ID, + device_code: DEVICE_AUTHORIZATION.device_code, + }, + ]); + expect(sleepSpy.mock.calls).toEqual([[1000], [6000]]); + expect(credentials.access).toBe("eventual-access-token"); + expect(credentials.refresh).toBe("eventual-refresh-token"); + }); + + it("rejects a token response that omits access_token", async () => { + const { fetchMock, requests } = createDeviceFlowFetch([ + { + body: { + refresh_token: "refresh-token", + expires_in: 3600, + }, + }, + ]); + + await expect(loginXAIOAuth({ fetch: fetchMock })).rejects.toThrow( + /xAI device-code token response missing access_token/, ); + expect(requests.filter(request => request.url === TOKEN_ENDPOINT)).toHaveLength(1); }); }); diff --git a/packages/ai/src/registry/oauth/device-code.ts b/packages/ai/src/registry/oauth/device-code.ts new file mode 100644 index 000000000..fc2dd723d --- /dev/null +++ b/packages/ai/src/registry/oauth/device-code.ts @@ -0,0 +1,92 @@ +import * as AIError from "../../error"; + +const DEVICE_FLOW_CANCEL_MESSAGE = "Login cancelled"; +const DEVICE_FLOW_TIMEOUT_MESSAGE = "Device flow timed out"; +const DEVICE_FLOW_SLOW_DOWN_TIMEOUT_MESSAGE = + "Device flow timed out after one or more slow_down responses. This is often caused by clock drift in WSL or VM environments. Please sync or restart the VM clock and try again."; +const MINIMUM_DEVICE_FLOW_INTERVAL_MS = 1000; +const DEFAULT_DEVICE_FLOW_INTERVAL_SECONDS = 5; +const SLOW_DOWN_INTERVAL_INCREMENT_MS = 5000; + +/** Result returned by one OAuth device-code polling attempt. */ +export type OAuthDeviceCodePollResult = + | { status: "complete"; value: T } + | { status: "pending" } + | { status: "slow_down" } + | { status: "failed"; message: string }; + +/** Options for polling an RFC 8628-style OAuth device-code flow. */ +export interface OAuthDeviceCodeFlowOptions { + /** Poll the provider once and classify the response. */ + poll(): OAuthDeviceCodePollResult | Promise>; + /** Provider-requested polling cadence; defaults to RFC 8628's five seconds. */ + intervalSeconds?: number; + /** Provider-issued expiry window for the device code. */ + expiresInSeconds?: number; + /** Cancels the flow with the legacy "Login cancelled" error. */ + signal?: AbortSignal; +} + +async function abortableDeviceFlowSleep(ms: number, signal: AbortSignal | undefined): Promise { + if (!signal) { + await Bun.sleep(ms); + return; + } + if (signal.aborted) { + throw new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE); + } + + const { promise, resolve, reject } = Promise.withResolvers(); + let timer: Timer | undefined; + const onAbort = () => { + clearTimeout(timer); + reject(new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE)); + }; + timer = setTimeout(() => { + signal.removeEventListener("abort", onAbort); + resolve(); + }, ms); + signal.addEventListener("abort", onAbort, { once: true }); + await promise; +} + +/** Poll an OAuth device-code flow until completion, provider failure, timeout, or cancellation. */ +export async function pollOAuthDeviceCodeFlow(options: OAuthDeviceCodeFlowOptions): Promise { + const deadline = + typeof options.expiresInSeconds === "number" + ? Date.now() + options.expiresInSeconds * 1000 + : Number.POSITIVE_INFINITY; + let intervalMs = Math.max( + MINIMUM_DEVICE_FLOW_INTERVAL_MS, + Math.floor((options.intervalSeconds ?? DEFAULT_DEVICE_FLOW_INTERVAL_SECONDS) * 1000), + ); + let slowDownResponses = 0; + + while (Date.now() < deadline) { + if (options.signal?.aborted) { + throw new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE); + } + const result = await options.poll(); + if (result.status === "complete") { + return result.value; + } + if (result.status === "failed") { + throw new AIError.OAuthError(result.message, { kind: "polling" }); + } + if (result.status === "slow_down") { + slowDownResponses += 1; + intervalMs = Math.max(MINIMUM_DEVICE_FLOW_INTERVAL_MS, intervalMs + SLOW_DOWN_INTERVAL_INCREMENT_MS); + } + + const remainingMs = deadline - Date.now(); + if (remainingMs <= 0) { + break; + } + await abortableDeviceFlowSleep(Math.min(intervalMs, remainingMs), options.signal); + } + + throw new AIError.OAuthError( + slowDownResponses > 0 ? DEVICE_FLOW_SLOW_DOWN_TIMEOUT_MESSAGE : DEVICE_FLOW_TIMEOUT_MESSAGE, + { kind: "timeout" }, + ); +} diff --git a/packages/ai/src/registry/oauth/index.ts b/packages/ai/src/registry/oauth/index.ts index 4cca3a778..31d402ce4 100644 --- a/packages/ai/src/registry/oauth/index.ts +++ b/packages/ai/src/registry/oauth/index.ts @@ -12,99 +12,9 @@ import type { OAuthProviderInterface, } from "./types"; +export * from "./device-code"; export type * from "./types"; -const DEVICE_FLOW_CANCEL_MESSAGE = "Login cancelled"; -const DEVICE_FLOW_TIMEOUT_MESSAGE = "Device flow timed out"; -const DEVICE_FLOW_SLOW_DOWN_TIMEOUT_MESSAGE = - "Device flow timed out after one or more slow_down responses. This is often caused by clock drift in WSL or VM environments. Please sync or restart the VM clock and try again."; -const MINIMUM_DEVICE_FLOW_INTERVAL_MS = 1000; -const DEFAULT_DEVICE_FLOW_INTERVAL_SECONDS = 5; -const SLOW_DOWN_INTERVAL_INCREMENT_MS = 5000; - -/** Result returned by one OAuth device-code polling attempt. */ -export type OAuthDeviceCodePollResult = - | { status: "complete"; value: T } - | { status: "pending" } - | { status: "slow_down" } - | { status: "failed"; message: string }; - -/** Options for polling an RFC 8628-style OAuth device-code flow. */ -export interface OAuthDeviceCodeFlowOptions { - /** Poll the provider once and classify the response. */ - poll(): OAuthDeviceCodePollResult | Promise>; - /** Provider-requested polling cadence; defaults to RFC 8628's five seconds. */ - intervalSeconds?: number; - /** Provider-issued expiry window for the device code. */ - expiresInSeconds?: number; - /** Cancels the flow with the legacy "Login cancelled" error. */ - signal?: AbortSignal; -} - -async function abortableDeviceFlowSleep(ms: number, signal: AbortSignal | undefined): Promise { - if (!signal) { - await Bun.sleep(ms); - return; - } - if (signal.aborted) { - throw new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE); - } - - const { promise, resolve, reject } = Promise.withResolvers(); - let timer: Timer | undefined; - const onAbort = () => { - if (timer) clearTimeout(timer); - reject(new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE)); - }; - timer = setTimeout(() => { - signal.removeEventListener("abort", onAbort); - resolve(); - }, ms); - signal.addEventListener("abort", onAbort, { once: true }); - await promise; -} - -/** Poll an OAuth device-code flow until completion, provider failure, timeout, or cancellation. */ -export async function pollOAuthDeviceCodeFlow(options: OAuthDeviceCodeFlowOptions): Promise { - const deadline = - typeof options.expiresInSeconds === "number" - ? Date.now() + options.expiresInSeconds * 1000 - : Number.POSITIVE_INFINITY; - let intervalMs = Math.max( - MINIMUM_DEVICE_FLOW_INTERVAL_MS, - Math.floor((options.intervalSeconds ?? DEFAULT_DEVICE_FLOW_INTERVAL_SECONDS) * 1000), - ); - let slowDownResponses = 0; - - while (Date.now() < deadline) { - if (options.signal?.aborted) { - throw new AIError.LoginCancelledError(DEVICE_FLOW_CANCEL_MESSAGE); - } - const result = await options.poll(); - if (result.status === "complete") { - return result.value; - } - if (result.status === "failed") { - throw new AIError.OAuthError(result.message, { kind: "polling" }); - } - if (result.status === "slow_down") { - slowDownResponses += 1; - intervalMs = Math.max(MINIMUM_DEVICE_FLOW_INTERVAL_MS, intervalMs + SLOW_DOWN_INTERVAL_INCREMENT_MS); - } - - const remainingMs = deadline - Date.now(); - if (remainingMs <= 0) { - break; - } - await abortableDeviceFlowSleep(Math.min(intervalMs, remainingMs), options.signal); - } - - throw new AIError.OAuthError( - slowDownResponses > 0 ? DEVICE_FLOW_SLOW_DOWN_TIMEOUT_MESSAGE : DEVICE_FLOW_TIMEOUT_MESSAGE, - { kind: "timeout" }, - ); -} - const builtInOAuthProviders: OAuthProviderInfo[] = PROVIDER_REGISTRY.filter( provider => provider.login && provider.showInLoginList !== false, ).map(provider => ({ diff --git a/packages/ai/src/registry/oauth/xai-oauth.ts b/packages/ai/src/registry/oauth/xai-oauth.ts index 8c55de347..d31da2750 100644 --- a/packages/ai/src/registry/oauth/xai-oauth.ts +++ b/packages/ai/src/registry/oauth/xai-oauth.ts @@ -1,31 +1,22 @@ -// Ported from NousResearch/hermes-agent (MIT) — hermes_cli/auth.py xAI sections (L93-111, L2979-3160, L5286-5469). +// Device authorization and token refresh adapted from NousResearch/hermes-agent (MIT). /** - * xAI Grok (SuperGrok or X Premium+) OAuth flow. + * xAI Grok OAuth device authorization flow. * - * Manual-code PKCE flow using `127.0.0.1:56121/callback` as the allowlisted - * redirect URI. One token unlocks Grok-4.x - * chat, Grok Imagine image generation, and Grok Voice TTS via subsequent - * commits. Endpoint discovery is hardened against MITM via - * {@link validateXAIEndpoint}: any non-HTTPS or non-`x.ai`/`*.x.ai` host is - * rejected on every call site, not just the first. + * Requests an RFC 8628 device code, opens xAI's verification page, and polls + * the discovered token endpoint until the user approves the login. */ import * as AIError from "../../error"; import type { FetchImpl } from "../../types"; -import { OAuthCallbackFlow, type OAuthCallbackFlowOptions } from "./callback-server"; -import { generatePKCE } from "./pkce"; +import { type OAuthDeviceCodePollResult, pollOAuthDeviceCodeFlow } from "./device-code"; import type { OAuthController, OAuthCredentials } from "./types"; -// Hermes hermes_cli/auth.py L93-111 const XAI_OAUTH_ISSUER = "https://auth.x.ai"; const XAI_OAUTH_DISCOVERY_URL = `${XAI_OAUTH_ISSUER}/.well-known/openid-configuration`; +const XAI_OAUTH_DEVICE_CODE_URL = `${XAI_OAUTH_ISSUER}/oauth2/device/code`; const XAI_OAUTH_CLIENT_ID = "b1a00492-073a-47ea-816f-4c329264a828"; const XAI_OAUTH_SCOPE = "openid profile email offline_access grok-cli:access api:access"; -const XAI_OAUTH_REDIRECT_HOST = "127.0.0.1"; -const XAI_OAUTH_REDIRECT_PORT = 56121; -const XAI_OAUTH_REDIRECT_PATH = "/callback"; -const XAI_OAUTH_DOCS_URL = "https://hermes-agent.nousresearch.com/docs/guides/xai-grok-oauth"; // Mirrors the 5-min skew used by anthropic.ts:160 — keeps every provider on the // same conservative client-side expiry window. @@ -35,18 +26,27 @@ const DISCOVERY_TIMEOUT_MS = 15_000; const TOKEN_REQUEST_TIMEOUT_MS = 20_000; interface XAIOAuthDiscovery { - authorization_endpoint: string; token_endpoint: string; } +interface XAIDeviceAuthorization { + deviceCode: string; + userCode: string; + verificationUriComplete: string; + expiresInSeconds: number; + intervalSeconds: number; +} + +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null; +} + /** - * Validate an xAI OIDC discovery endpoint against scheme + host. + * Validate an xAI OIDC endpoint against its scheme and host. * - * Hermes `_xai_validate_oauth_endpoint` L2997-3035. The discovery response is - * long-lived and cached in {@link OAuthCredentials}; a single MITM during - * initial login could substitute a malicious `token_endpoint` that would then - * receive every future refresh_token. Rejecting non-HTTPS or non-`x.ai` / - * `*.x.ai` hosts pins the cached endpoint to the xAI auth origin. + * The discovery response is long-lived and its token endpoint receives every + * future refresh token. Rejecting non-HTTPS or non-`x.ai` / `*.x.ai` hosts + * pins that endpoint to the xAI auth origin. * * @throws Error with message `Invalid xAI : ` when the URL fails * either scheme or host validation. @@ -68,11 +68,7 @@ export function validateXAIEndpoint(url: string, field: string): string { return url; } -/** - * Fetch xAI's OIDC discovery document and validate both endpoints. - * - * Hermes `_xai_oauth_discovery` L3038-3084. - */ +/** Fetch xAI's OIDC discovery document and validate the token endpoint. */ async function xaiOAuthDiscovery( timeoutMs: number = DISCOVERY_TIMEOUT_MS, fetchOverride?: FetchImpl, @@ -111,37 +107,29 @@ async function xaiOAuthDiscovery( { kind: "validation", provider: "xai", cause: error }, ); } - if (!payload || typeof payload !== "object") { + if (!isRecord(payload)) { throw new AIError.OAuthError("xAI OIDC discovery response was not a JSON object.", { kind: "validation", provider: "xai", }); } - const obj = payload as Record; - const authorizationEndpoint = - typeof obj.authorization_endpoint === "string" ? obj.authorization_endpoint.trim() : ""; - const tokenEndpoint = typeof obj.token_endpoint === "string" ? obj.token_endpoint.trim() : ""; - if (!authorizationEndpoint || !tokenEndpoint) { - throw new AIError.OAuthError("xAI OIDC discovery response was missing required endpoints.", { + const tokenEndpoint = typeof payload.token_endpoint === "string" ? payload.token_endpoint.trim() : ""; + if (!tokenEndpoint) { + throw new AIError.OAuthError("xAI OIDC discovery response was missing token_endpoint.", { kind: "validation", provider: "xai", }); } - validateXAIEndpoint(authorizationEndpoint, "authorization_endpoint"); validateXAIEndpoint(tokenEndpoint, "token_endpoint"); - return { - authorization_endpoint: authorizationEndpoint, - token_endpoint: tokenEndpoint, - }; + return { token_endpoint: tokenEndpoint }; } /** * Check whether a JWT access token is at or past its `exp` claim (with an * optional refresh-skew margin). * - * Hermes `_xai_access_token_is_expiring` L2979-2994. Returns `false` for any - * malformed input — this is a refresh-trigger check, not a validation, so - * non-JWTs ("no token in cache") must NOT trigger a spurious refresh. + * Returns `false` for malformed input because this is a refresh-trigger check, + * not token validation. */ export function isXAIAccessTokenExpiring(jwt: string, skewSeconds: number = 0): boolean { try { @@ -151,7 +139,8 @@ export function isXAIAccessTokenExpiring(jwt: string, skewSeconds: number = 0): const payloadPart = parts[1]; if (!payloadPart) return false; const decoded = Buffer.from(payloadPart, "base64url").toString("utf8"); - const payload = JSON.parse(decoded) as { exp?: unknown }; + const payload: unknown = JSON.parse(decoded); + if (!isRecord(payload)) return false; const exp = payload.exp; if (typeof exp !== "number" || !Number.isFinite(exp)) return false; const now = Math.floor(Date.now() / 1000); @@ -162,161 +151,232 @@ export function isXAIAccessTokenExpiring(jwt: string, skewSeconds: number = 0): } } -interface BuildXAIAuthorizeUrlOptions { - authorizationEndpoint: string; - redirectUri: string; - codeChallenge: string; - state: string; - nonce: string; -} - -/** - * Build the xAI authorization URL. - * - * Hermes `_xai_oauth_build_authorize_url` L5286-5312. `plan=generic` opts the - * consent screen into xAI's generic OAuth plan tier; without it, - * `accounts.x.ai` rejects loopback OAuth from non-allowlisted clients. - * `referrer=oh-my-pi` lets xAI attribute oh-my-pi-originated logins in their - * OAuth server logs (Hermes uses `referrer=hermes-agent`; oh-my-pi mirrors the - * pattern with its own attribution string). - */ -function buildXAIAuthorizeUrl(opts: BuildXAIAuthorizeUrlOptions): string { - const params = new URLSearchParams({ - response_type: "code", - client_id: XAI_OAUTH_CLIENT_ID, - redirect_uri: opts.redirectUri, - scope: XAI_OAUTH_SCOPE, - code_challenge: opts.codeChallenge, - code_challenge_method: "S256", - state: opts.state, - nonce: opts.nonce, - plan: "generic", - referrer: "oh-my-pi", - }); - return `${opts.authorizationEndpoint}?${params.toString()}`; -} - -/** - * xAI Grok OAuth code flow (Hermes `_xai_oauth_loopback_login` L5315-5469). - */ -export class XAIOAuthFlow extends OAuthCallbackFlow { - #verifier: string = ""; - #fetch: FetchImpl; - - constructor(ctrl: OAuthController) { - super(ctrl, { - preferredPort: XAI_OAUTH_REDIRECT_PORT, - callbackPath: XAI_OAUTH_REDIRECT_PATH, - callbackHostname: XAI_OAUTH_REDIRECT_HOST, - redirectUri: `http://${XAI_OAUTH_REDIRECT_HOST}:${XAI_OAUTH_REDIRECT_PORT}${XAI_OAUTH_REDIRECT_PATH}`, - manualInputOnly: true, - } satisfies OAuthCallbackFlowOptions); - this.#fetch = ctrl.fetch ?? fetch; +function parseXAIDeviceAuthorization(payload: unknown): XAIDeviceAuthorization { + if (!isRecord(payload)) { + throw new AIError.OAuthError("xAI device-code response was not a JSON object.", { + kind: "validation", + provider: "xai", + }); } - async generateAuthUrl(state: string, redirectUri: string): Promise<{ url: string; instructions?: string }> { - const pkce = await generatePKCE(); - this.#verifier = pkce.verifier; - const nonce = crypto.randomUUID().replace(/-/g, ""); - - const discovery = await xaiOAuthDiscovery(DISCOVERY_TIMEOUT_MS, this.#fetch); - const url = buildXAIAuthorizeUrl({ - authorizationEndpoint: discovery.authorization_endpoint, - redirectUri, - codeChallenge: pkce.challenge, - state, - nonce, + const deviceCode = typeof payload.device_code === "string" ? payload.device_code.trim() : ""; + const userCode = typeof payload.user_code === "string" ? payload.user_code.trim() : ""; + const verificationUri = typeof payload.verification_uri === "string" ? payload.verification_uri.trim() : ""; + const verificationUriComplete = + typeof payload.verification_uri_complete === "string" ? payload.verification_uri_complete.trim() : ""; + const expiresInSeconds = payload.expires_in; + const intervalSeconds = payload.interval; + if ( + !deviceCode || + !userCode || + !verificationUri || + !verificationUriComplete || + typeof expiresInSeconds !== "number" || + !Number.isFinite(expiresInSeconds) || + expiresInSeconds <= 0 || + typeof intervalSeconds !== "number" || + !Number.isFinite(intervalSeconds) || + intervalSeconds <= 0 + ) { + throw new AIError.OAuthError("xAI device-code response missing or invalid required fields.", { + kind: "validation", + provider: "xai", }); - - return { - url, - instructions: `Complete login in your browser for xAI Grok (SuperGrok or X Premium+). Docs: ${XAI_OAUTH_DOCS_URL}`, - }; } - async exchangeToken(code: string, _state: string, redirectUri: string): Promise { - const discovery = await xaiOAuthDiscovery(DISCOVERY_TIMEOUT_MS, this.#fetch); - const tokenEndpoint = validateXAIEndpoint(discovery.token_endpoint, "token_endpoint"); + validateXAIEndpoint(verificationUri, "verification_uri"); + validateXAIEndpoint(verificationUriComplete, "verification_uri_complete"); + return { + deviceCode, + userCode, + verificationUriComplete, + expiresInSeconds, + intervalSeconds, + }; +} - const body = new URLSearchParams({ - grant_type: "authorization_code", - client_id: XAI_OAUTH_CLIENT_ID, - code, - redirect_uri: redirectUri, - code_verifier: this.#verifier, +function parseXAITokenResponse(payload: unknown, label: string, refreshTokenFallback?: string): OAuthCredentials { + if (!isRecord(payload)) { + throw new AIError.OAuthError(`${label} was not a JSON object`, { + kind: "validation", + provider: "xai", }); + } + const accessToken = typeof payload.access_token === "string" ? payload.access_token : ""; + const responseRefreshToken = typeof payload.refresh_token === "string" ? payload.refresh_token : ""; + const refreshToken = responseRefreshToken || refreshTokenFallback || ""; + const expiresInSeconds = payload.expires_in; + if (!accessToken) { + throw new AIError.OAuthError(`${label} missing access_token`, { + kind: "validation", + provider: "xai", + }); + } + if (!refreshToken) { + throw new AIError.OAuthError(`${label} missing refresh_token`, { + kind: "validation", + provider: "xai", + }); + } + if (typeof expiresInSeconds !== "number" || !Number.isFinite(expiresInSeconds)) { + throw new AIError.OAuthError(`${label} missing expires_in`, { + kind: "validation", + provider: "xai", + }); + } + return { + access: accessToken, + refresh: refreshToken, + expires: Date.now() + expiresInSeconds * 1000 - ACCESS_TOKEN_CLIENT_SKEW_MS, + }; +} - const response = await this.#fetch(tokenEndpoint, { +async function requestXAIDeviceAuthorization( + fetchImpl: FetchImpl, + signal?: AbortSignal, +): Promise { + let response: Response; + try { + const timeoutSignal = AbortSignal.timeout(TOKEN_REQUEST_TIMEOUT_MS); + response = await fetchImpl(XAI_OAUTH_DEVICE_CODE_URL, { method: "POST", headers: { "Content-Type": "application/x-www-form-urlencoded", Accept: "application/json", }, - body, - signal: AbortSignal.timeout(TOKEN_REQUEST_TIMEOUT_MS), + body: new URLSearchParams({ + client_id: XAI_OAUTH_CLIENT_ID, + scope: XAI_OAUTH_SCOPE, + }), + signal: signal ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal, }); - - if (!response.ok) { - let detail = ""; - try { - detail = (await response.text()).trim(); - } catch { - // Ignore body-read failures; the status code is the diagnostic. - } - throw new AIError.OAuthError(`xAI token exchange failed: ${response.status}${detail ? ` ${detail}` : ""}`, { - kind: "token-exchange", - provider: "xai", - status: response.status, - }); - } - - let tokenData: { access_token?: unknown; refresh_token?: unknown; expires_in?: unknown }; - try { - tokenData = (await response.json()) as typeof tokenData; - } catch (error) { - throw new AIError.OAuthError( - `xAI token exchange returned invalid JSON: ${error instanceof Error ? error.message : String(error)}`, - { kind: "validation", provider: "xai", cause: error }, - ); - } - - if (typeof tokenData.access_token !== "string" || !tokenData.access_token) { - throw new AIError.OAuthError("xAI token exchange response missing access_token", { - kind: "validation", - provider: "xai", - }); - } - if (typeof tokenData.refresh_token !== "string" || !tokenData.refresh_token) { - throw new AIError.OAuthError("xAI token exchange response missing refresh_token", { - kind: "validation", - provider: "xai", - }); - } - if (typeof tokenData.expires_in !== "number" || !Number.isFinite(tokenData.expires_in)) { - throw new AIError.OAuthError("xAI token exchange response missing expires_in", { - kind: "validation", - provider: "xai", - }); - } - - return { - access: tokenData.access_token, - refresh: tokenData.refresh_token, - expires: Date.now() + tokenData.expires_in * 1000 - ACCESS_TOKEN_CLIENT_SKEW_MS, - }; + } catch (error) { + if (signal?.aborted) throw new AIError.LoginCancelledError(); + throw new AIError.OAuthError( + `xAI device-code request failed: ${error instanceof Error ? error.message : String(error)}`, + { kind: "device-auth", provider: "xai", cause: error }, + ); } + + if (!response.ok) { + let detail = ""; + try { + detail = (await response.text()).trim(); + } catch { + // Ignore body-read failures; the status code is the diagnostic. + } + throw new AIError.OAuthError(`xAI device-code request failed: ${response.status}${detail ? ` ${detail}` : ""}`, { + kind: "device-auth", + provider: "xai", + status: response.status, + }); + } + + let payload: unknown; + try { + payload = await response.json(); + } catch (error) { + throw new AIError.OAuthError( + `xAI device-code response returned invalid JSON: ${error instanceof Error ? error.message : String(error)}`, + { kind: "validation", provider: "xai", cause: error }, + ); + } + return parseXAIDeviceAuthorization(payload); } +async function pollXAIDeviceToken( + tokenEndpoint: string, + deviceCode: string, + fetchImpl: FetchImpl, + signal?: AbortSignal, +): Promise> { + let response: Response; + try { + const timeoutSignal = AbortSignal.timeout(TOKEN_REQUEST_TIMEOUT_MS); + response = await fetchImpl(tokenEndpoint, { + method: "POST", + headers: { + "Content-Type": "application/x-www-form-urlencoded", + Accept: "application/json", + }, + body: new URLSearchParams({ + grant_type: "urn:ietf:params:oauth:grant-type:device_code", + client_id: XAI_OAUTH_CLIENT_ID, + device_code: deviceCode, + }), + signal: signal ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal, + }); + } catch (error) { + if (signal?.aborted) throw new AIError.LoginCancelledError(); + throw new AIError.OAuthError( + `xAI device-code token polling failed: ${error instanceof Error ? error.message : String(error)}`, + { kind: "polling", provider: "xai", cause: error }, + ); + } + + let payload: unknown; + try { + payload = await response.json(); + } catch (error) { + throw new AIError.OAuthError( + `xAI device-code token polling returned invalid JSON: ${ + error instanceof Error ? error.message : String(error) + }`, + { kind: "polling", provider: "xai", status: response.status, cause: error }, + ); + } + + if (response.ok) { + return { + status: "complete", + value: parseXAITokenResponse(payload, "xAI device-code token response"), + }; + } + if (!isRecord(payload)) { + throw new AIError.OAuthError(`xAI device-code token polling failed: ${response.status}`, { + kind: "polling", + provider: "xai", + status: response.status, + }); + } + + const errorCode = typeof payload.error === "string" ? payload.error : ""; + if (errorCode === "authorization_pending") return { status: "pending" }; + if (errorCode === "slow_down") return { status: "slow_down" }; + + const errorDescription = typeof payload.error_description === "string" ? payload.error_description : ""; + const detail = errorDescription || errorCode || String(response.status); + throw new AIError.OAuthError(`xAI device-code token polling failed: ${detail}`, { + kind: "polling", + provider: "xai", + status: response.status, + }); +} + +/** Log in to xAI Grok with the RFC 8628 device authorization grant. */ export async function loginXAIOAuth(ctrl: OAuthController): Promise { - return new XAIOAuthFlow(ctrl).login(); + const fetchImpl = ctrl.fetch ?? fetch; + const discovery = await xaiOAuthDiscovery(DISCOVERY_TIMEOUT_MS, fetchImpl); + const device = await requestXAIDeviceAuthorization(fetchImpl, ctrl.signal); + ctrl.onAuth?.({ + url: device.verificationUriComplete, + instructions: `Enter code: ${device.userCode}`, + }); + ctrl.onProgress?.("Waiting for xAI device authorization..."); + + return pollOAuthDeviceCodeFlow({ + poll: () => pollXAIDeviceToken(discovery.token_endpoint, device.deviceCode, fetchImpl, ctrl.signal), + intervalSeconds: device.intervalSeconds, + expiresInSeconds: device.expiresInSeconds, + signal: ctrl.signal, + }); } /** * Refresh an xAI OAuth access token using a stored refresh_token. * - * Hermes `refresh_xai_oauth_pure` L3087-3160. Re-runs OIDC discovery and - * re-validates the cached `token_endpoint` on the refresh hot path so a - * cached-but-poisoned endpoint cannot silently leak a refresh_token. + * Re-runs OIDC discovery and re-validates the token endpoint before sending + * the stored refresh token. */ export async function refreshXAIOAuthToken(refreshToken: string, fetchOverride?: FetchImpl): Promise { const fetchImpl = fetchOverride ?? fetch; @@ -357,34 +417,14 @@ export async function refreshXAIOAuthToken(refreshToken: string, fetchOverride?: }); } - let data: { access_token?: unknown; refresh_token?: unknown; expires_in?: unknown }; + let payload: unknown; try { - data = (await response.json()) as typeof data; + payload = await response.json(); } catch (error) { throw new AIError.OAuthError( `xAI token refresh returned invalid JSON: ${error instanceof Error ? error.message : String(error)}`, { kind: "validation", provider: "xai", cause: error }, ); } - - if (typeof data.access_token !== "string" || !data.access_token) { - throw new AIError.OAuthError("xAI token refresh response missing access_token", { - kind: "validation", - provider: "xai", - }); - } - if (typeof data.expires_in !== "number" || !Number.isFinite(data.expires_in)) { - throw new AIError.OAuthError("xAI token refresh response missing expires_in", { - kind: "validation", - provider: "xai", - }); - } - - const newRefresh = typeof data.refresh_token === "string" && data.refresh_token ? data.refresh_token : refreshToken; - - return { - access: data.access_token, - refresh: newRefresh, - expires: Date.now() + data.expires_in * 1000 - ACCESS_TOKEN_CLIENT_SKEW_MS, - }; + return parseXAITokenResponse(payload, "xAI token refresh response", refreshToken); } diff --git a/packages/ai/src/registry/registry.ts b/packages/ai/src/registry/registry.ts index e6740996a..2d9420c49 100644 --- a/packages/ai/src/registry/registry.ts +++ b/packages/ai/src/registry/registry.ts @@ -34,6 +34,7 @@ import { minimaxCodeCnProvider } from "./minimax-code-cn"; import { mistralProvider } from "./mistral"; import { moonshotProvider } from "./moonshot"; import { nanogptProvider } from "./nanogpt"; +import { novitaProvider } from "./novita"; import { nvidiaProvider } from "./nvidia"; import { ollamaProvider } from "./ollama"; import { ollamaCloudProvider } from "./ollama-cloud"; @@ -110,6 +111,7 @@ const ALL = [ fireworksProvider, togetherProvider, nvidiaProvider, + novitaProvider, huggingfaceProvider, perplexityProvider, qianfanProvider, diff --git a/packages/ai/src/registry/xai-oauth.ts b/packages/ai/src/registry/xai-oauth.ts index ed1a22bd4..67f1b1bd9 100644 --- a/packages/ai/src/registry/xai-oauth.ts +++ b/packages/ai/src/registry/xai-oauth.ts @@ -14,5 +14,4 @@ export const xaiOauthProvider = { const { refreshXAIOAuthToken } = await import("./oauth/xai-oauth"); return refreshXAIOAuthToken(credentials.refresh); }, - pasteCodeFlow: true, } as const satisfies ProviderDefinition; diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 02d17ad2f..de922d926 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -1632,6 +1632,7 @@ function mapOptionsForApi( toolChoice: mapOpenAiToolChoice(options?.toolChoice), serviceTier: options?.serviceTier, preferWebsockets: options?.preferWebsockets, + codexCompaction: options?.codexCompaction, reasoningSummary: options?.hideThinkingSummary ? null : "detailed", textVerbosity: options?.textVerbosity, }); diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index 995fd454f..5b568de29 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -323,6 +323,30 @@ export interface RawSseEvent { raw: string[]; } +/** Lifecycle fields shared by every Codex compaction implementation. */ +export interface CodexCompactionContext { + /** Stable only for one logical compaction, including parallel summary calls. */ + operationId: string; + trigger: "manual" | "auto"; + reason: "user_requested" | "context_limit" | "model_downshift" | "comp_hash_changed"; + phase: "standalone_turn" | "pre_turn" | "mid_turn"; + strategy: "memento" | "prefix_compaction"; +} + +/** Canonical nested metadata serialized into the Codex turn envelope. */ +export interface CodexCompactionMetadata { + trigger: "manual" | "auto"; + reason: "user_requested" | "context_limit" | "model_downshift" | "comp_hash_changed"; + implementation: "responses" | "responses_compaction_v2" | "responses_compact"; + phase: "standalone_turn" | "pre_turn" | "mid_turn"; + strategy: "memento" | "prefix_compaction"; +} + +/** Dispatch context combining canonical metadata with its local operation identity. */ +export interface CodexCompactionRequestContext extends CodexCompactionMetadata { + operationId: string; +} + export interface StreamOptions { temperature?: number; topP?: number; @@ -388,9 +412,9 @@ export interface StreamOptions { */ sessionId?: string; /** - * Optional prompt-cache identity. When set, OpenAI Responses-compatible - * providers use this for `prompt_cache_key` while keeping `sessionId` for - * provider routing / conversation headers. + * Optional prompt-cache identity. OpenAI-family providers use this for + * `prompt_cache_key` payloads and cache-affinity headers such as + * `x-grok-conv-id`; when omitted, they fall back to `sessionId`. */ promptCacheKey?: string; /** @@ -398,6 +422,8 @@ export interface StreamOptions { * Providers can use this to persist transport/session state between turns. */ providerSessionState?: Map; + /** Canonical Codex compaction classification; ignored by other providers. */ + codexCompaction?: CodexCompactionRequestContext; /** * Force Gemini model-mode Interactions API transport for providers that support it. * When unset, those providers may still use Interactions to continue known diff --git a/packages/ai/src/utils/event-stream.ts b/packages/ai/src/utils/event-stream.ts index d7b75e863..6d1255e39 100644 --- a/packages/ai/src/utils/event-stream.ts +++ b/packages/ai/src/utils/event-stream.ts @@ -10,6 +10,14 @@ export class EventStream implements AsyncIterable { resultSettled = false; #failed = false; #error: unknown = undefined; + /** + * Consumer-side local operations currently in flight for this stream — a + * provider transport waiting on a server-requested local tool bridge + * (e.g. the Cursor exec channel) before it can send the result upstream. + * While non-zero, event silence is attributable to our own pending work, + * not a provider stall; idle watchdogs consult {@link hasPendingLocalWork}. + */ + #pendingLocalWork = 0; finalResultPromise: Promise; resolveFinalResult!: (result: R) => void; rejectFinalResult!: (err: unknown) => void; @@ -116,6 +124,24 @@ export class EventStream implements AsyncIterable { result(): Promise { return this.finalResultPromise; } + + /** True while local work tracked via {@link trackLocalWork} is pending. */ + get hasPendingLocalWork(): boolean { + return this.#pendingLocalWork > 0; + } + + /** + * Track a local-work promise so idle watchdogs on this stream do not treat + * the event silence while it is pending as a provider stall. + */ + async trackLocalWork(work: Promise): Promise { + this.#pendingLocalWork++; + try { + return await work; + } finally { + this.#pendingLocalWork--; + } + } } export class AssistantMessageEventStream extends EventStream { diff --git a/packages/ai/src/utils/idle-iterator.ts b/packages/ai/src/utils/idle-iterator.ts index 7cebaf64e..3accaf3c2 100644 --- a/packages/ai/src/utils/idle-iterator.ts +++ b/packages/ai/src/utils/idle-iterator.ts @@ -135,6 +135,16 @@ export interface IdleTimeoutIteratorOptions { * keepalive/no-op events from keeping a stalled tool call alive forever. */ isProgressItem?: (item: unknown) => boolean; + /** + * Reports consumer-side local work in flight for the stream: the provider + * transport is waiting on a server-requested local tool bridge (e.g. the + * Cursor exec channel) before anything can flow upstream again. While it + * returns true, an expired idle / first-item deadline slides forward + * instead of aborting — the silence is ours, not a provider stall. The + * watchdog re-arms with a full budget once the local work completes, so a + * provider that stalls afterwards is still caught. + */ + hasPendingLocalWork?: () => boolean; /** * Cancel iteration as soon as this signal aborts. Required for caller-driven * cancellation (ESC) when the underlying transport does not surface signal @@ -157,7 +167,7 @@ export async function* iterateWithIdleTimeout( options: IdleTimeoutIteratorOptions, ): AsyncGenerator { const firstItemTimeoutMs = options.firstItemTimeoutMs ?? options.idleTimeoutMs; - const firstItemDeadlineMs = + let firstItemDeadlineMs = firstItemTimeoutMs !== undefined && firstItemTimeoutMs > 0 ? Date.now() + firstItemTimeoutMs : undefined; const abortSignal = options.abortSignal; const iterator = iterable[Symbol.asyncIterator](); @@ -197,6 +207,28 @@ export async function* iterateWithIdleTimeout( }; let lastProgressAt = Date.now(); + const hasPendingLocalWork = (): boolean => { + if (!options.hasPendingLocalWork) return false; + try { + return options.hasPendingLocalWork(); + } catch { + return false; + } + }; + // Local work means the current gap is attributable to the consumer side, + // not the provider: slide the active deadline a full budget past now + // instead of aborting. Once the work completes the watchdog resumes from + // the last extension, so a provider that stalls afterwards is still caught. + const extendDeadlineForLocalWork = (): void => { + if (awaitingFirstItem) { + if (firstItemDeadlineMs !== undefined && firstItemTimeoutMs !== undefined) { + firstItemDeadlineMs = Date.now() + firstItemTimeoutMs; + } + } else { + lastProgressAt = Date.now(); + } + }; + const noTimeoutEnforced = (firstItemTimeoutMs === undefined || firstItemTimeoutMs <= 0) && (options.idleTimeoutMs === undefined || options.idleTimeoutMs <= 0); @@ -271,6 +303,12 @@ export async function* iterateWithIdleTimeout( timer = setTimeout(onTimerFire, Math.max(0, deadlineMs - Date.now())); }; + // The in-flight iterator.next() promise, persisted across loop iterations: + // a deadline extension for pending local work loops without consuming it, + // and issuing a second next() while one is outstanding would drop an item. + let pendingNext: + | Promise<{ kind: "next"; result: IteratorResult } | { kind: "error"; error: unknown }> + | undefined; try { let raceCount = 0; while (true) { @@ -291,21 +329,29 @@ export async function* iterateWithIdleTimeout( if (firstItemDeadlineMs !== undefined) { activeTimeoutMs = firstItemDeadlineMs - Date.now(); if (activeTimeoutMs <= 0) { - options.onFirstItemTimeout?.(); - closeIterator(); - throw new AIError.StreamTimeoutError(options.firstItemErrorMessage ?? options.errorMessage); + if (!hasPendingLocalWork()) { + options.onFirstItemTimeout?.(); + closeIterator(); + throw new AIError.StreamTimeoutError(options.firstItemErrorMessage ?? options.errorMessage); + } + extendDeadlineForLocalWork(); + activeTimeoutMs = firstItemDeadlineMs! - Date.now(); } } } else if (options.idleTimeoutMs !== undefined && options.idleTimeoutMs > 0) { activeTimeoutMs = options.idleTimeoutMs - (Date.now() - lastProgressAt); if (activeTimeoutMs <= 0) { - options.onIdle?.(); - closeIterator(); - throw new AIError.StreamTimeoutError(options.errorMessage); + if (!hasPendingLocalWork()) { + options.onIdle?.(); + closeIterator(); + throw new AIError.StreamTimeoutError(options.errorMessage); + } + extendDeadlineForLocalWork(); + activeTimeoutMs = options.idleTimeoutMs; } } - const nextResultPromise = withRacy(iterator.next()); + pendingNext ??= withRacy(iterator.next()); const racers: Array< Promise< @@ -314,7 +360,7 @@ export async function* iterateWithIdleTimeout( | { kind: "timeout" } | { kind: "abort" } > - > = [nextResultPromise]; + > = [pendingNext]; const enforceTimeout = !noTimeoutEnforced && activeTimeoutMs !== undefined && activeTimeoutMs > 0; if (enforceTimeout) { @@ -333,11 +379,21 @@ export async function* iterateWithIdleTimeout( let continuing = false; try { const outcome = await Promise.race(racers); + if (outcome.kind === "next" || outcome.kind === "error") { + pendingNext = undefined; + } if (outcome.kind === "abort") { closeIterator(); throw abortReason(abortSignal!); } if (outcome.kind === "timeout") { + if (hasPendingLocalWork()) { + // A local tool is still running; the provider cannot make + // progress until we hand its result back. Keep waiting. + extendDeadlineForLocalWork(); + continuing = true; + continue; + } if (!awaitingFirstItem) { options.onIdle?.(); } else { diff --git a/packages/ai/test/anthropic-ping-keepalive.test.ts b/packages/ai/test/anthropic-ping-keepalive.test.ts new file mode 100644 index 000000000..a7e309b9e --- /dev/null +++ b/packages/ai/test/anthropic-ping-keepalive.test.ts @@ -0,0 +1,233 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { streamAnthropic } from "../src/providers/anthropic"; +import type { AnthropicMessagesClientLike } from "../src/providers/anthropic-client"; +import type { Context, Model } from "../src/types"; +import { waitForDelayOrAbort } from "./helpers"; + +const model: Model<"anthropic-messages"> = buildModel({ + id: "claude-opus-4-8", + name: "Claude Opus 4.8", + api: "anthropic-messages", + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 8_192, +}); + +const context: Context = { + messages: [{ role: "user", content: "write a file", timestamp: Date.now() }], +}; + +type MockAnthropicEvent = Record; + +/** `{ waitMs, event }` script step; `waitMs` elapses (fake clock) before the event is yielded. */ +type ScriptStep = { waitMs: number; event: MockAnthropicEvent | "hang-with-pings" }; + +const writeToolCallOpening: MockAnthropicEvent[] = [ + { + type: "message_start", + message: { + id: "msg_ping_keepalive", + usage: { input_tokens: 10, output_tokens: 0, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 }, + }, + }, + { + type: "content_block_start", + index: 0, + content_block: { type: "tool_use", id: "toolu_ping_keepalive", name: "write", input: {} }, + }, + { + type: "content_block_delta", + index: 0, + delta: { type: "input_json_delta", partial_json: '{"path":"notes.md",' }, + }, +]; + +const writeToolCallClosing: MockAnthropicEvent[] = [ + { + type: "content_block_delta", + index: 0, + delta: { type: "input_json_delta", partial_json: '"content":"hello world"}' }, + }, + { type: "content_block_stop", index: 0 }, + { + type: "message_delta", + delta: { stop_reason: "tool_use" }, + usage: { input_tokens: 10, output_tokens: 6, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 }, + }, + { type: "message_stop" }, +]; + +function createScriptedClient( + script: ScriptStep[], + counters: { pings: number }, + onIteratorStart: () => void, +): AnthropicMessagesClientLike { + const create = ((_body: unknown, requestOptions?: { signal?: AbortSignal }) => { + const signal = requestOptions?.signal; + const response = new Response(null, { status: 200, headers: { "request-id": "req_ping_keepalive" } }); + const stream = { + async *[Symbol.asyncIterator]() { + onIteratorStart(); + for (const step of script) { + if (step.event === "hang-with-pings") { + // Wedged upstream: no semantic events ever again, but the edge + // keeps the SSE connection alive with keepalive pings. + while (true) { + await waitForDelayOrAbort(step.waitMs, signal); + counters.pings += 1; + yield { type: "ping" }; + } + } + if (step.waitMs > 0) { + await waitForDelayOrAbort(step.waitMs, signal); + } + if (step.event.type === "ping") counters.pings += 1; + yield step.event; + } + }, + }; + return { + async withResponse() { + return { data: stream, response, request_id: "req_ping_keepalive" }; + }, + } as never; + }) as unknown as AnthropicMessagesClientLike["messages"]["create"]; + return { messages: { create } } as AnthropicMessagesClientLike; +} + +async function drainMicrotasks(count: number): Promise { + for (let i = 0; i < count; i++) { + await Promise.resolve(); + } +} + +async function drainMicrotasksUntil(predicate: () => boolean, errorMessage: string): Promise { + for (let i = 0; i < 1000; i++) { + if (predicate()) return; + await Promise.resolve(); + } + throw new Error(errorMessage); +} + +afterEach(() => { + vi.useRealTimers(); + vi.restoreAllMocks(); +}); + +describe("anthropic ping keepalive idle cap", () => { + it("times out a stalled tool-call stream instead of letting pings extend it forever", async () => { + vi.useFakeTimers(); + const counters = { pings: 0 }; + let iteratorStarted = false; + const script: ScriptStep[] = [ + ...writeToolCallOpening.map(event => ({ waitMs: 0, event })), + { waitMs: 500, event: "hang-with-pings" as const }, + ]; + const client = createScriptedClient(script, counters, () => { + iteratorStarted = true; + }); + const providerRetryWait = vi.fn(async () => {}); + + let settled = false; + const resultPromise = streamAnthropic(model, context, { + client, + streamFirstEventTimeoutMs: 1_000, + streamIdleTimeoutMs: 1_000, + providerRetryWait, + }) + .result() + .then(message => { + settled = true; + return message; + }); + + await drainMicrotasksUntil(() => iteratorStarted, "Anthropic mock stream never started"); + await drainMicrotasks(30); + + // Pings arrive every 500 fake-ms while generation is wedged. Drive far + // past the bounded keepalive window (3x idle = 3_000ms) plus one idle + // budget; without the cap the idle deadline is reset by every ping and + // this loop ends with the result still pending (issue #4900's hang). + let stepsRun = 0; + for (let step = 0; step < 40 && !settled; step++) { + vi.advanceTimersByTime(500); + await drainMicrotasks(30); + stepsRun = step + 1; + } + + expect(settled).toBe(true); + // Cap (3_000ms) + idle budget (1_000ms) = fires at 3_500-4_000 fake ms. + expect(stepsRun).toBeLessThanOrEqual(9); + // Keepalives within the window were honored before the watchdog fired. + expect(counters.pings).toBeGreaterThanOrEqual(5); + + const result = await resultPromise; + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toBe("Anthropic stream stalled while waiting for the next event"); + // Mid-stream idle stalls are terminal for the provider loop (session-level + // auto-retry owns recovery); the provider must not silently re-request. + expect(providerRetryWait).not.toHaveBeenCalled(); + }); + + it("keeps a slow-but-alive stream open across ping-bridged gaps within the cap", async () => { + vi.useFakeTimers(); + const counters = { pings: 0 }; + let iteratorStarted = false; + // Silent generation gap of 1_800ms (> 1_000ms idle budget) bridged by + // pings at t=600 and t=1200, then semantic progress resumes and the + // tool call completes. Pings within the cap must count as liveness. + const script: ScriptStep[] = [ + ...writeToolCallOpening.map(event => ({ waitMs: 0, event })), + { waitMs: 600, event: { type: "ping" } }, + { waitMs: 600, event: { type: "ping" } }, + { waitMs: 600, event: writeToolCallClosing[0]! }, + ...writeToolCallClosing.slice(1).map(event => ({ waitMs: 0, event })), + ]; + const client = createScriptedClient(script, counters, () => { + iteratorStarted = true; + }); + const providerRetryWait = vi.fn(async () => {}); + + let settled = false; + const resultPromise = streamAnthropic(model, context, { + client, + streamFirstEventTimeoutMs: 1_000, + streamIdleTimeoutMs: 1_000, + providerRetryWait, + }) + .result() + .then(message => { + settled = true; + return message; + }); + + await drainMicrotasksUntil(() => iteratorStarted, "Anthropic mock stream never started"); + await drainMicrotasks(30); + + for (let step = 0; step < 30 && !settled; step++) { + vi.advanceTimersByTime(200); + await drainMicrotasks(30); + } + + expect(settled).toBe(true); + expect(counters.pings).toBe(2); + + const result = await resultPromise; + expect(result.errorMessage).toBeUndefined(); + expect(result.stopReason).toBe("toolUse"); + expect(providerRetryWait).not.toHaveBeenCalled(); + expect(JSON.parse(JSON.stringify(result.content))).toEqual([ + { + type: "toolCall", + id: "toolu_ping_keepalive", + name: "write", + arguments: { path: "notes.md", content: "hello world" }, + }, + ]); + }); +}); diff --git a/packages/ai/test/auth-broker-config-discovery.test.ts b/packages/ai/test/auth-broker-config-discovery.test.ts new file mode 100644 index 000000000..67d4033aa --- /dev/null +++ b/packages/ai/test/auth-broker-config-discovery.test.ts @@ -0,0 +1,59 @@ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { resolveAuthBrokerConfig } from "@oh-my-pi/pi-ai/auth-broker"; +import { removeWithRetries } from "../../utils/src/temp"; +import { withEnv } from "./helpers"; + +const SUPPRESS_AUTH_BROKER_ENV = { + OMP_AUTH_BROKER_URL: undefined, + OMP_AUTH_BROKER_TOKEN: undefined, +} as const; + +describe("resolveAuthBrokerConfig config discovery", () => { + let agentDir = ""; + + beforeEach(async () => { + agentDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-auth-broker-config-")); + }); + + afterEach(async () => { + if (agentDir) { + await removeWithRetries(agentDir); + agentDir = ""; + } + }); + + test("resolves broker URL and token from config.yaml when config.yml is absent", async () => { + await Bun.write( + path.join(agentDir, "config.yaml"), + "auth.broker.url: https://yaml-broker.example/v1\nauth.broker.token: yaml-token\n", + ); + + await withEnv(SUPPRESS_AUTH_BROKER_ENV, async () => { + await expect(resolveAuthBrokerConfig({ agentDir })).resolves.toEqual({ + url: "https://yaml-broker.example/v1", + token: "yaml-token", + }); + }); + }); + + test("prefers config.yml over config.yaml when both exist", async () => { + await Bun.write( + path.join(agentDir, "config.yaml"), + "auth.broker.url: https://yaml-broker.example/v1\nauth.broker.token: yaml-token\n", + ); + await Bun.write( + path.join(agentDir, "config.yml"), + "auth.broker.url: https://yml-broker.example/v1\nauth.broker.token: yml-token\n", + ); + + await withEnv(SUPPRESS_AUTH_BROKER_ENV, async () => { + await expect(resolveAuthBrokerConfig({ agentDir })).resolves.toEqual({ + url: "https://yml-broker.example/v1", + token: "yml-token", + }); + }); + }); +}); diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index 6bf0ff584..a0ff1e114 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -724,6 +724,157 @@ describe("AuthStorage codex oauth ranking", () => { expect(apiKey).toBe("api-acct-solo"); }); + test.each([ + ["gpt-5.6-sol", "free", "plus"], + ["gpt-5.6-luna", "go", "business"], + ["gpt-5.6-sol-pro", "free", "team"], + ])("%s routes away from a less-used %s account to an eligible %s account", async (modelId, freePlan, paidPlan) => { + if (!authStorage) throw new Error("test setup failed"); + + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-free", "free@example.com") }, + { type: "oauth", ...createCredential("acct-paid", "paid@example.com") }, + ]); + + usageByAccount.set( + "acct-free", + createCodexUsageReport({ + accountId: "acct-free", + primary: { usedFraction: 0.01, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.01, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: freePlan, email: "free@example.com" }, + }), + ); + usageByAccount.set( + "acct-paid", + createCodexUsageReport({ + accountId: "acct-paid", + primary: { usedFraction: 0.8, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.8, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: paidPlan, email: "paid@example.com" }, + }), + ); + + const apiKey = await authStorage.getApiKey("openai-codex", undefined, { modelId }); + expect(apiKey).toBe("api-acct-paid"); + }); + + test.each([ + ["gpt-5.6-terra", "free", "enterprise"], + ["gpt-5.6-terra-pro", "go", "pro"], + ])("%s keeps a less-used %s account in ordinary ranking ahead of %s", async (modelId, lowUsagePlan, highUsagePlan) => { + if (!authStorage) throw new Error("test setup failed"); + + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-low-usage", "low-usage@example.com") }, + { type: "oauth", ...createCredential("acct-high-usage", "high-usage@example.com") }, + ]); + + usageByAccount.set( + "acct-low-usage", + createCodexUsageReport({ + accountId: "acct-low-usage", + primary: { usedFraction: 0.01, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.01, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: lowUsagePlan, email: "low-usage@example.com" }, + }), + ); + usageByAccount.set( + "acct-high-usage", + createCodexUsageReport({ + accountId: "acct-high-usage", + primary: { usedFraction: 0.8, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.8, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: highUsagePlan, email: "high-usage@example.com" }, + }), + ); + + const apiKey = await authStorage.getApiKey("openai-codex", undefined, { modelId }); + expect(apiKey).toBe("api-acct-low-usage"); + }); + + test("reranks a Terra session on a Go account when it switches to Sol", async () => { + if (!authStorage) throw new Error("test setup failed"); + + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-go", "go@example.com") }, + { type: "oauth", ...createCredential("acct-business", "business@example.com") }, + ]); + + usageByAccount.set( + "acct-go", + createCodexUsageReport({ + accountId: "acct-go", + primary: { usedFraction: 0.01, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.01, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: "go", email: "go@example.com" }, + }), + ); + usageByAccount.set( + "acct-business", + createCodexUsageReport({ + accountId: "acct-business", + primary: { usedFraction: 0.8, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.8, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: "business", email: "business@example.com" }, + }), + ); + + let terraSession: string | undefined; + let terraApiKey: string | undefined; + for (let index = 0; index < 100; index += 1) { + const sessionId = `session-terra-to-sol-${index}`; + const apiKey = await authStorage.getApiKey("openai-codex", sessionId, { + modelId: "gpt-5.6-terra", + }); + if (apiKey === "api-acct-go") { + terraSession = sessionId; + terraApiKey = apiKey; + break; + } + } + expect(terraApiKey).toBe("api-acct-go"); + if (!terraSession) throw new Error("expected Terra to select the lower-usage Go account"); + + const solApiKey = await authStorage.getApiKey("openai-codex", terraSession, { + modelId: "gpt-5.6-sol", + }); + expect(solApiKey).toBe("api-acct-business"); + }); + + test("falls back by ordinary usage ranking for Sol when no account is confirmed paid", async () => { + if (!authStorage) throw new Error("test setup failed"); + + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-free", "free@example.com") }, + { type: "oauth", ...createCredential("acct-go", "go@example.com") }, + ]); + + usageByAccount.set( + "acct-free", + createCodexUsageReport({ + accountId: "acct-free", + primary: { usedFraction: 0.8, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.8, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: "free", email: "free@example.com" }, + }), + ); + usageByAccount.set( + "acct-go", + createCodexUsageReport({ + accountId: "acct-go", + primary: { usedFraction: 0.01, resetInMs: 30 * 60 * 1000 }, + secondary: { usedFraction: 0.01, resetInMs: 6 * 24 * 60 * 60 * 1000 }, + metadata: { planType: "go", email: "go@example.com" }, + }), + ); + + const apiKey = await authStorage.getApiKey("openai-codex", undefined, { + modelId: "gpt-5.6-sol", + }); + expect(apiKey).toBe("api-acct-go"); + }); + test("prefers Pro accounts for codex spark models over Plus accounts", async () => { if (!authStorage) throw new Error("test setup failed"); diff --git a/packages/ai/test/cursor-exec-handlers.test.ts b/packages/ai/test/cursor-exec-handlers.test.ts index e77bc484f..5063e81ec 100644 --- a/packages/ai/test/cursor-exec-handlers.test.ts +++ b/packages/ai/test/cursor-exec-handlers.test.ts @@ -1,14 +1,25 @@ import { describe, expect, it } from "bun:test"; +import { create } from "@bufbuild/protobuf"; import { + type BlockState, buildCursorHistoryForTest, buildCursorSystemPromptJsons, emptyGrepPatternRejection, + handleServerMessage, resolveExecHandler, streamCursor, + type ToolCallState, } from "@oh-my-pi/pi-ai/providers/cursor"; -import type { Context, Model } from "@oh-my-pi/pi-ai/types"; +import { streamCursor as lazyStreamCursor, setCursorProviderModule } from "@oh-my-pi/pi-ai/providers/register-builtins"; +import type { AssistantMessage, Context, CursorExecHandlers, Model, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; +import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; -import type { AgentRunRequest } from "@oh-my-pi/pi-catalog/discovery/cursor-gen/agent_pb"; +import { + type AgentRunRequest, + AgentServerMessageSchema, + ExecServerMessageSchema, + ReadArgsSchema, +} from "@oh-my-pi/pi-catalog/discovery/cursor-gen/agent_pb"; const cursorModel: Model<"cursor-agent"> = buildModel({ id: "cursor-composer-2.5", @@ -361,3 +372,188 @@ describe("Cursor grepArgs empty-pattern guard (issue #4574)", () => { expect(emptyGrepPatternRejection("\t\n", "src/**/*.ts")).toContain('"src/**/*.ts"'); }); }); + +function cursorAssistantMessage(): AssistantMessage { + return { + role: "assistant", + content: [], + api: "cursor-agent", + provider: "cursor", + model: "cursor-composer-2.5", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp: 0, + }; +} + +function newBlockState(): BlockState { + let textBlock: BlockState["currentTextBlock"] = null; + let thinkingBlock: BlockState["currentThinkingBlock"] = null; + let toolCall: ToolCallState | null = null; + return { + get currentTextBlock() { + return textBlock; + }, + get currentThinkingBlock() { + return thinkingBlock; + }, + get currentToolCall() { + return toolCall; + }, + firstTokenTime: undefined, + setTextBlock: b => { + textBlock = b; + }, + setThinkingBlock: b => { + thinkingBlock = b; + }, + setToolCall: t => { + toolCall = t; + }, + setFirstTokenTime: () => {}, + }; +} + +describe("Cursor exec local-work tracking (issue #4593)", () => { + it("marks the stream busy for the duration of a local exec handler", async () => { + const output = cursorAssistantMessage(); + const stream = new AssistantMessageEventStream(); + const state = newBlockState(); + const written: unknown[] = []; + const h2Request = { + write: (chunk: unknown) => { + written.push(chunk); + return true; + }, + } as unknown as Parameters[5]; + const handlerGate = Promise.withResolvers(); + const execHandlers: CursorExecHandlers = { + async read(args) { + await handlerGate.promise; + return { + role: "toolResult", + toolCallId: args.toolCallId, + toolName: "read", + content: [{ type: "text", text: "file contents" }], + isError: false, + timestamp: 1, + } satisfies ToolResultMessage; + }, + }; + const serverMsg = create(AgentServerMessageSchema, { + message: { + case: "execServerMessage", + value: create(ExecServerMessageSchema, { + id: 1, + execId: "exec-1", + message: { + case: "readArgs", + value: create(ReadArgsSchema, { path: "/tmp/slow-file", toolCallId: "call-read-1" }), + }, + }), + }, + }); + + expect(stream.hasPendingLocalWork).toBe(false); + const dispatch = handleServerMessage( + serverMsg, + output, + stream, + state, + new Map(), + h2Request, + execHandlers, + undefined, + { sawTokenDelta: false }, + [], + ); + + // The exec round-trip is in flight: the stream must advertise local + // work so the lazy idle watchdog defers instead of aborting. + expect(stream.hasPendingLocalWork).toBe(true); + + handlerGate.resolve(); + await dispatch; + + expect(stream.hasPendingLocalWork).toBe(false); + // The read result went back out on the exec channel. + expect(written.length).toBe(1); + }); + + it("survives a local exec tool outliving the lazy idle budget end to end", async () => { + const workDone = Promise.withResolvers(); + // The tracked work completes only once the lazy watchdog has consulted + // the stream's local-work state at two expired deadlines, proving the + // idle budget was truly exceeded while the exec tool ran. + class ProbedStream extends AssistantMessageEventStream { + probeCalls = 0; + override get hasPendingLocalWork(): boolean { + this.probeCalls++; + if (this.probeCalls >= 2) workDone.resolve(); + return super.hasPendingLocalWork; + } + } + const source = new ProbedStream(); + let providerSignal: AbortSignal | undefined; + setCursorProviderModule({ + streamCursor: (_model, _context, options) => { + providerSignal = options.signal; + void (async () => { + const partial = cursorAssistantMessage(); + source.push({ type: "start", partial }); + source.push({ type: "text_delta", contentIndex: 0, delta: "spawning local tool", partial }); + await source.trackLocalWork(workDone.promise); + const message = cursorAssistantMessage(); + source.push({ type: "done", reason: "stop", message }); + })(); + return source; + }, + }); + + const stream = lazyStreamCursor(cursorModel, { messages: [] }, { apiKey: "test", streamIdleTimeoutMs: 5 }); + const result = await stream.result(); + + expect(providerSignal?.aborted).toBe(false); + expect(source.probeCalls).toBeGreaterThanOrEqual(2); + expect(result.stopReason).toBe("stop"); + }); + + it("still aborts a silent cursor stream with no local work in flight", async () => { + const partial = cursorAssistantMessage(); + let providerSignal: AbortSignal | undefined; + const source = { + async *[Symbol.asyncIterator]() { + yield { type: "start", partial } as const; + yield { type: "text_delta", contentIndex: 0, delta: "hello", partial } as const; + const stalled = Promise.withResolvers(); + if (providerSignal?.aborted) { + stalled.reject(new Error("Request was aborted")); + } + providerSignal?.addEventListener("abort", () => stalled.reject(new Error("Request was aborted")), { + once: true, + }); + await stalled.promise; + }, + } as unknown as AssistantMessageEventStream; + setCursorProviderModule({ + streamCursor: (_model, _context, options) => { + providerSignal = options.signal; + return source; + }, + }); + + const stream = lazyStreamCursor(cursorModel, { messages: [] }, { apiKey: "test", streamIdleTimeoutMs: 10 }); + const result = await stream.result(); + + expect(providerSignal?.aborted).toBe(true); + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toBe("Provider stream stalled while waiting for the next event"); + }); +}); diff --git a/packages/ai/test/error-aierr.test.ts b/packages/ai/test/error-aierr.test.ts index 810a76f87..57617b60e 100644 --- a/packages/ai/test/error-aierr.test.ts +++ b/packages/ai/test/error-aierr.test.ts @@ -32,6 +32,11 @@ describe("AIError.classify — structural provider errors", () => { ).toBe(true); }); + it("classifies a typed AWS credential-resolution failure as authFailed", () => { + const id = AIError.classify(new AIError.AwsCredentialsError("opaque provider setup failure", "resolution")); + expect(AIError.is(id, AIError.Flag.AuthFailed)).toBe(true); + }); + it("maps the usage_limit_reached code to usageLimit on a 429", () => { const id = AIError.classify( new AIError.ProviderHttpError("Payment Required", 429, { code: "usage_limit_reached" }), diff --git a/packages/ai/test/helpers/index.ts b/packages/ai/test/helpers/index.ts index fe987fcf1..f8437e4f8 100644 --- a/packages/ai/test/helpers/index.ts +++ b/packages/ai/test/helpers/index.ts @@ -2,6 +2,7 @@ import * as os from "node:os"; import * as path from "node:path"; import type { Model } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; import { isEnoent } from "@oh-my-pi/pi-utils"; export async function withEnv( @@ -54,7 +55,10 @@ export async function waitForDelayOrAbort(delayMs: number, signal: AbortSignal | } } -export function createCodexModel(id: string): Model<"openai-codex-responses"> { +export function createCodexModel( + id: string, + spec?: Partial>, +): Model<"openai-codex-responses"> { return buildModel({ id, name: id, @@ -66,6 +70,7 @@ export function createCodexModel(id: string): Model<"openai-codex-responses"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 272000, maxTokens: 128000, + ...spec, }); } diff --git a/packages/ai/test/issue-1701-repro.test.ts b/packages/ai/test/issue-1701-repro.test.ts index 4016d573d..79318532c 100644 --- a/packages/ai/test/issue-1701-repro.test.ts +++ b/packages/ai/test/issue-1701-repro.test.ts @@ -1,12 +1,23 @@ -import { describe, expect, it } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import { streamAzureOpenAIResponses } from "@oh-my-pi/pi-ai/providers/azure-openai-responses"; import { streamOpenAICodexResponses } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses"; import type { Context, Model, Tool, ToolChoice } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import * as piUtils from "@oh-my-pi/pi-utils"; import { z } from "zod/v4"; +const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001"; + +beforeEach(() => { + vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID); +}); + +afterEach(() => { + vi.restoreAllMocks(); +}); + const completionsModel: Model<"openai-completions"> = buildModel({ id: "gpt-4o-mini-test", name: "GPT-4o Mini Test", diff --git a/packages/ai/test/issue-4593-repro.test.ts b/packages/ai/test/issue-4593-repro.test.ts new file mode 100644 index 000000000..e79f4f1e1 --- /dev/null +++ b/packages/ai/test/issue-4593-repro.test.ts @@ -0,0 +1,190 @@ +import { describe, expect, it } from "bun:test"; +import { setBedrockProviderModule, streamBedrock } from "@oh-my-pi/pi-ai/providers/register-builtins"; +import type { AssistantMessage, Context, Model } from "@oh-my-pi/pi-ai/types"; +import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; +import { iterateWithIdleTimeout } from "@oh-my-pi/pi-ai/utils/idle-iterator"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +// Issue #4593: the generic lazy stream watchdog treats "no AssistantMessageEvent" +// as "provider stalled". During a Cursor exec-channel round-trip the server is +// waiting on OUR local tool result and legitimately sends nothing, so a local +// tool outliving the idle budget aborted a healthy stream with "Provider stream +// stalled while waiting for the next event". Provider streams now advertise +// pending local work and the watchdog slides its deadline instead of aborting. +// +// These tests exercise the real watchdog timer against the platform clock (that +// timer IS the unit under test), but never guess durations: the simulated local +// work completes only once the watchdog has demonstrably reached an expired +// deadline and consulted the local-work probe, so the tests stay causal on a +// loaded machine. Budgets are a few milliseconds. + +function createModel(): Model<"bedrock-converse-stream"> { + return buildModel({ + id: "mock-bedrock", + name: "Mock Bedrock", + api: "bedrock-converse-stream", + provider: "amazon-bedrock", + baseUrl: "https://example.invalid", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 8192, + maxTokens: 2048, + }); +} + +function createAssistantMessage(): AssistantMessage { + return { + role: "assistant", + content: [{ type: "text", text: "ok" }], + api: "bedrock-converse-stream", + provider: "amazon-bedrock", + model: "mock-bedrock", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp: Date.now(), + }; +} + +const baseContext: Context = { messages: [] }; + +describe("idle watchdog local-work deferral (issue #4593)", () => { + it("slides the idle deadline while consumer-side local work is pending", async () => { + const workDone = Promise.withResolvers(); + let probeCalls = 0; + let busy = true; + async function* source() { + yield "first"; + // The "local tool": finishes only after the watchdog has hit an + // expired deadline twice and deferred both times. + await workDone.promise; + busy = false; + yield "second"; + } + let idleFired = false; + const items: string[] = []; + for await (const item of iterateWithIdleTimeout(source(), { + idleTimeoutMs: 5, + errorMessage: "stalled", + onIdle: () => { + idleFired = true; + }, + hasPendingLocalWork: () => { + probeCalls++; + if (probeCalls >= 2) workDone.resolve(); + return busy; + }, + })) { + items.push(item); + } + expect(items).toEqual(["first", "second"]); + expect(probeCalls).toBeGreaterThanOrEqual(2); + expect(idleFired).toBe(false); + }); + + it("still aborts a silent stream once local work has finished", async () => { + const workDone = Promise.withResolvers(); + let busy = true; + async function* source() { + yield "first"; + await workDone.promise; + busy = false; + // The provider genuinely stalls after the local work completed. + await new Promise(() => {}); + yield "never"; + } + const items: string[] = []; + let error: Error | undefined; + try { + for await (const item of iterateWithIdleTimeout(source(), { + idleTimeoutMs: 5, + errorMessage: "stalled", + hasPendingLocalWork: () => { + workDone.resolve(); + return busy; + }, + })) { + items.push(item); + } + } catch (err) { + error = err as Error; + } + expect(items).toEqual(["first"]); + expect(error?.message).toBe("stalled"); + }); + + it("slides the first-event deadline while local work is pending", async () => { + const workDone = Promise.withResolvers(); + let probeCalls = 0; + let busy = true; + async function* source() { + // Local bridge work before the model has produced any event. + await workDone.promise; + yield "first"; + } + const items: string[] = []; + for await (const item of iterateWithIdleTimeout(source(), { + idleTimeoutMs: 5, + firstItemTimeoutMs: 5, + errorMessage: "stalled", + firstItemErrorMessage: "first event timed out", + hasPendingLocalWork: () => { + probeCalls++; + if (probeCalls >= 2) workDone.resolve(); + return busy; + }, + })) { + items.push(item); + busy = false; + } + expect(items).toEqual(["first"]); + expect(probeCalls).toBeGreaterThanOrEqual(2); + }); + + it("does not abort a lazy provider stream while tracked local work outlives the idle budget", async () => { + const workDone = Promise.withResolvers(); + // Counts how often the lazy wrapper's watchdog consults the stream's + // local-work state at an expired deadline; the tracked work completes + // only after two deferrals, proving the budget was truly exceeded. + class ProbedStream extends AssistantMessageEventStream { + probeCalls = 0; + override get hasPendingLocalWork(): boolean { + this.probeCalls++; + if (this.probeCalls >= 2) workDone.resolve(); + return super.hasPendingLocalWork; + } + } + const source = new ProbedStream(); + let providerSignal: AbortSignal | undefined; + setBedrockProviderModule({ + streamBedrock: (_model, _context, options) => { + providerSignal = options.signal; + void (async () => { + const partial = createAssistantMessage(); + source.push({ type: "start", partial }); + source.push({ type: "text_delta", contentIndex: 0, delta: "running a local tool", partial }); + // Server-driven local tool run: no events flow while the + // tracked work is pending. + await source.trackLocalWork(workDone.promise); + source.push({ type: "done", reason: "stop", message: createAssistantMessage() }); + })(); + return source; + }, + }); + + const stream = streamBedrock(createModel(), baseContext, { streamIdleTimeoutMs: 5 }); + const result = await stream.result(); + + expect(providerSignal?.aborted).toBe(false); + expect(source.probeCalls).toBeGreaterThanOrEqual(2); + expect(result.stopReason).toBe("stop"); + expect(result.errorMessage).toBeUndefined(); + }); +}); diff --git a/packages/ai/test/novita-login.test.ts b/packages/ai/test/novita-login.test.ts new file mode 100644 index 000000000..810b27870 --- /dev/null +++ b/packages/ai/test/novita-login.test.ts @@ -0,0 +1,74 @@ +import { describe, expect, test, vi } from "bun:test"; +import { loginNovita } from "../src/registry/novita"; +import { getOAuthProviders } from "../src/registry/oauth"; +import type { FetchImpl } from "../src/types"; + +describe("Novita login", () => { + test("registers Novita as an available API-key provider", () => { + const provider = getOAuthProviders().find(item => item.id === "novita"); + expect(provider).toMatchObject({ id: "novita", name: "Novita", available: true }); + }); + + test("validates the pasted key against the authenticated balance endpoint", async () => { + const authEvents: Array<{ url: string; instructions?: string }> = []; + const prompts: Array<{ message: string; placeholder?: string }> = []; + const progress: string[] = []; + const requests: Array<{ + url: string; + method: string | undefined; + authorization: string | null; + contentType: string | null; + }> = []; + const fetchMock: FetchImpl = vi.fn(async (input: string | URL | Request, init?: RequestInit) => { + const headers = new Headers(init?.headers); + requests.push({ + url: String(input), + method: init?.method, + authorization: headers.get("authorization"), + contentType: headers.get("content-type"), + }); + return Response.json({ availableBalance: "0" }); + }); + + const apiKey = await loginNovita({ + onAuth: info => authEvents.push(info), + onPrompt: async prompt => { + prompts.push(prompt); + return " novita-test-key "; + }, + onProgress: message => progress.push(message), + fetch: fetchMock, + }); + + expect(apiKey).toBe("novita-test-key"); + expect(authEvents).toEqual([ + { + url: "https://novita.ai/settings/key-management", + instructions: "Create or copy your API key from the Novita dashboard", + }, + ]); + expect(prompts).toEqual([{ message: "Paste your Novita API key", placeholder: "sk_..." }]); + expect(progress).toEqual(["Validating API key..."]); + expect(requests).toEqual([ + { + url: "https://api.novita.ai/openapi/v1/billing/balance/detail", + method: "GET", + authorization: "Bearer novita-test-key", + contentType: "application/json", + }, + ]); + }); + + test("rejects a key rejected by Novita", async () => { + const fetchMock: FetchImpl = vi.fn(async () => + Response.json({ code: 401, reason: "UNAUTHORIZED", message: "key not found", metadata: {} }, { status: 401 }), + ); + + await expect( + loginNovita({ + onPrompt: async () => "invalid-novita-key", + fetch: fetchMock, + }), + ).rejects.toThrow("Novita API key validation failed (401)"); + }); +}); diff --git a/packages/ai/test/openai-codex-responses-lite.test.ts b/packages/ai/test/openai-codex-responses-lite.test.ts index c0bd32074..169f578a9 100644 --- a/packages/ai/test/openai-codex-responses-lite.test.ts +++ b/packages/ai/test/openai-codex-responses-lite.test.ts @@ -1,4 +1,4 @@ -import { describe, expect, it } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import { type InputItem, type RequestBody, @@ -7,12 +7,25 @@ import { import { buildTransformedCodexRequestBody, convertCodexResponsesMessages, + resetOpenAICodexHistoryAfterCompaction, streamOpenAICodexResponses, } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; -import type { Context, FetchImpl } from "@oh-my-pi/pi-ai/types"; +import { isOpenAIResponsesProgressEvent } from "@oh-my-pi/pi-ai/providers/openai-shared"; +import type { CodexCompactionRequestContext, Context, FetchImpl, ProviderSessionState } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import * as piUtils from "@oh-my-pi/pi-utils"; import { createCodexModel } from "./helpers"; +const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001"; + +beforeEach(() => { + vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID); +}); + +afterEach(() => { + vi.restoreAllMocks(); +}); + function createCodexTestToken(accountId = "acc_test"): string { const payload = Buffer.from( JSON.stringify({ "https://api.openai.com/auth": { chatgpt_account_id: accountId } }), @@ -63,6 +76,24 @@ interface CapturedCodexRequest { body: Record; } +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +function requireRecord(value: unknown, label: string): Record { + if (!isRecord(value)) { + throw new Error(`expected ${label} to be an object`); + } + return value; +} + +function parseTurnMetadata(clientMetadata: Record): Record { + const encoded = clientMetadata["x-codex-turn-metadata"]; + if (typeof encoded !== "string") throw new Error("expected x-codex-turn-metadata"); + const decoded: unknown = JSON.parse(encoded); + return requireRecord(decoded, "x-codex-turn-metadata"); +} + function createCodexFetchMock(sse: string, onRequest: (captured: CapturedCodexRequest) => void): FetchImpl { return (async (input: string | URL, init?: RequestInit) => { const url = typeof input === "string" ? input : input.toString(); @@ -191,7 +222,7 @@ describe("openai-codex reasoning.summary", () => { }); describe("openai-codex Responses Lite input shaping", () => { - it("keeps full Responses image details when a requested lite body contains images", async () => { + it("strips image detail and keeps lite when the input contains images", async () => { const model = createCodexModel("gpt-5.1-codex"); const makeInput = (): InputItem[] => [ { @@ -211,10 +242,11 @@ describe("openai-codex Responses Lite input shaping", () => { ]; const lite = await transformRequestBody({ model: model.id, input: makeInput() }, model, { responsesLite: true }); - const liteMessage = lite.input?.[0]?.content as Array>; - const liteOutput = lite.input?.[2]?.output as Array>; - expect(liteMessage[1]).toEqual({ type: "input_image", detail: "auto", image_url: "data:image/png;base64,AAAA" }); - expect(liteOutput[0]).toEqual({ type: "input_image", detail: "high", image_url: "data:image/png;base64,BBBB" }); + expect(lite.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools: [] }); + const liteMessage = lite.input?.[1]?.content as Array>; + const liteOutput = lite.input?.[3]?.output as Array>; + expect(liteMessage[1]).toEqual({ type: "input_image", image_url: "data:image/png;base64,AAAA" }); + expect(liteOutput[0]).toEqual({ type: "input_image", image_url: "data:image/png;base64,BBBB" }); const plain = await transformRequestBody({ model: model.id, input: makeInput() }, model, {}); const plainMessage = plain.input?.[0]?.content as Array>; @@ -253,7 +285,7 @@ describe("openai-codex Responses Lite input shaping", () => { }); }); - it("forces parallel_tool_calls off under lite when tools are present", async () => { + it("forces parallel_tool_calls off and moves tools into input under lite", async () => { const model = createCodexModel("gpt-5.1-codex"); const tools = [{ type: "function", name: "shot", parameters: { type: "object" } }]; @@ -261,12 +293,57 @@ describe("openai-codex Responses Lite input shaping", () => { responsesLite: true, }); expect(lite.parallel_tool_calls).toBe(false); + expect(lite.tools).toBeUndefined(); + expect(lite.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools }); const plain = await transformRequestBody({ model: model.id, tools, parallel_tool_calls: true }, model, {}); expect(plain.parallel_tool_calls).toBe(true); + expect(plain.tools).toEqual(tools); const noTools = await transformRequestBody({ model: model.id }, model, { responsesLite: true }); - expect(noTools.parallel_tool_calls).toBeUndefined(); + expect(noTools.parallel_tool_calls).toBe(false); + }); + + it("moves instructions and tools into input items under lite", async () => { + const model = createCodexModel("gpt-5.6-terra"); + const tools = [{ type: "function", name: "shot", parameters: { type: "object" } }]; + const body = await transformRequestBody( + { + model: model.id, + instructions: "test instructions", + tools, + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "hello" }] }], + }, + model, + { responsesLite: true }, + ); + + expect(body.instructions).toBeUndefined(); + expect(body.tools).toBeUndefined(); + expect(body.input?.[0]).toEqual({ type: "additional_tools", role: "developer", tools }); + expect(body.input?.[1]).toEqual({ + type: "message", + role: "developer", + content: [{ type: "input_text", text: "test instructions" }], + }); + expect(body.input?.[2]).toEqual({ + type: "message", + role: "user", + content: [{ type: "input_text", text: "hello" }], + }); + }); + + it("defaults lite from the model useResponsesLite flag and honors explicit opt-out", async () => { + const model = createCodexModel("gpt-5.6-terra", { useResponsesLite: true }); + const lite = await transformRequestBody({ model: model.id, instructions: "sys" }, model, {}); + expect(lite.instructions).toBeUndefined(); + expect(lite.input?.[0]?.type).toBe("additional_tools"); + + const optOut = await transformRequestBody({ model: model.id, instructions: "sys" }, model, { + responsesLite: false, + }); + expect(optOut.instructions).toBe("sys"); + expect(optOut.input?.some(item => item.type === "additional_tools")).toBe(false); }); }); @@ -323,15 +400,21 @@ describe("openai-codex fresh execution input shaping", () => { }); describe("openai-codex Responses Lite and client metadata wire format", () => { - it("sends the lite header and client_metadata body field over SSE", async () => { + it("sends canonical Codex metadata and protects reserved fields over SSE", async () => { const model = createCodexModel("gpt-5.1-codex"); - const clientMetadata = { "x-codex-turn-metadata": '{"thread_id":"thread_1","turn_id":"turn_1"}' }; + const context = createCodexTestContext(); + const clientMetadata = { + workspace_kind: "repo", + workspace_path: "東京/🚀", + session_id: "caller-session", + "x-codex-turn-metadata": '{"turn_id":"caller-turn"}', + }; let captured: CapturedCodexRequest | undefined; const fetchMock = createCodexFetchMock(createCodexSse(COMPLETED_CODEX_EVENTS), request => { captured = request; }); - const result = await streamOpenAICodexResponses(model, createCodexTestContext(), { + const result = await streamOpenAICodexResponses(model, context, { apiKey: createCodexTestToken(), fetch: fetchMock, responsesLite: true, @@ -339,10 +422,143 @@ describe("openai-codex Responses Lite and client metadata wire format", () => { }).result(); expect(result.stopReason).toBe("stop"); - expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); - expect(captured?.body.client_metadata).toEqual(clientMetadata); + if (!captured) throw new Error("expected a captured Codex request"); + expect(captured.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); + expect(captured.headers.get("x-codex-installation-id")).toBeNull(); + + const metadata = requireRecord(captured.body.client_metadata, "client_metadata"); + const turnMetadata = parseTurnMetadata(metadata); + expect(metadata.workspace_kind).toBeUndefined(); + expect(metadata.workspace_path).toBeUndefined(); + expect(metadata.session_id).not.toBe("caller-session"); + expect(turnMetadata.request_kind).toBe("turn"); + expect(turnMetadata.turn_started_at_unix_ms).toBe(context.messages[0]?.timestamp); + expect(turnMetadata.workspace_kind).toBe("repo"); + expect(turnMetadata.workspace_path).toBe("東京/🚀"); + expect(metadata["x-codex-installation-id"]).toMatch( + /^[0-9a-f]{8}-[0-9a-f]{4}-4[0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i, + ); + expect(metadata.session_id).toBe(turnMetadata.session_id); + expect(metadata.thread_id).toBe(turnMetadata.thread_id); + expect(metadata.turn_id).toBe(turnMetadata.turn_id); + expect(metadata["x-codex-window-id"]).toBe(turnMetadata.window_id); + expect(metadata.session_id).toBe(captured.headers.get("session-id")); + expect(metadata.thread_id).toBe(captured.headers.get("thread-id")); + expect(metadata["x-codex-window-id"]).toBe(captured.headers.get("x-codex-window-id")); + expect(metadata["x-codex-turn-metadata"]).toBe(captured.headers.get("x-codex-turn-metadata")); + const turnMetadataHeader = captured.headers.get("x-codex-turn-metadata"); + expect(turnMetadataHeader).toMatch(/^[\x20-\x7e]+$/); + const reparsedTurnMetadata: unknown = turnMetadataHeader ? JSON.parse(turnMetadataHeader) : undefined; + expect(requireRecord(reparsedTurnMetadata, "round-tripped turn metadata").workspace_path).toBe("東京/🚀"); }); - it("falls back to full Responses when a lite request contains images", async () => { + + it("keeps the installation identity stable across provider sessions", async () => { + const model = createCodexModel("gpt-5.1-codex"); + const captured: CapturedCodexRequest[] = []; + const fetchMock = createCodexFetchMock(createCodexSse(COMPLETED_CODEX_EVENTS), request => { + captured.push(request); + }); + + await streamOpenAICodexResponses(model, createCodexTestContext(), { + apiKey: createCodexTestToken(), + fetch: fetchMock, + sessionId: "metadata-session-one", + }).result(); + await streamOpenAICodexResponses(model, createCodexTestContext(), { + apiKey: createCodexTestToken(), + fetch: fetchMock, + sessionId: "metadata-session-two", + }).result(); + + const firstMetadata = requireRecord(captured[0]?.body.client_metadata, "first client_metadata"); + const secondMetadata = requireRecord(captured[1]?.body.client_metadata, "second client_metadata"); + expect(firstMetadata["x-codex-installation-id"]).toBe(secondMetadata["x-codex-installation-id"]); + expect(firstMetadata.session_id).toBe("metadata-session-one"); + expect(secondMetadata.session_id).toBe("metadata-session-two"); + expect(firstMetadata.thread_id).not.toBe(secondMetadata.thread_id); + }); + + it("rotates compaction turns by phase and reuses one operation across fan-out calls", async () => { + const model = createCodexModel("gpt-5.1-codex"); + const providerSessionState = new Map(); + const captured: CapturedCodexRequest[] = []; + const fetchMock = createCodexFetchMock(createCodexSse(COMPLETED_CODEX_EVENTS), request => { + captured.push(request); + }); + const send = async (codexCompaction?: CodexCompactionRequestContext): Promise => { + await streamOpenAICodexResponses(model, createCodexTestContext(), { + apiKey: createCodexTestToken(), + fetch: fetchMock, + sessionId: "compaction-lifecycle-session", + providerSessionState, + codexCompaction, + }).result(); + }; + const preTurn: CodexCompactionRequestContext = { + operationId: "pre-turn-operation", + trigger: "auto", + reason: "context_limit", + implementation: "responses", + phase: "pre_turn", + strategy: "memento", + }; + const midTurn: CodexCompactionRequestContext = { + ...preTurn, + operationId: "mid-turn-operation", + phase: "mid_turn", + }; + const standalone: CodexCompactionRequestContext = { + ...preTurn, + operationId: "standalone-operation", + trigger: "manual", + reason: "user_requested", + phase: "standalone_turn", + }; + + await send(); + await send(preTurn); + await send(preTurn); + resetOpenAICodexHistoryAfterCompaction({ + providerSessionState, + sessionId: "compaction-lifecycle-session", + compaction: preTurn, + }); + await send(); + await send(midTurn); + await send(standalone); + + const turns = captured.map((request, index) => + parseTurnMetadata(requireRecord(request.body.client_metadata, `client_metadata ${index}`)), + ); + expect(turns[0]?.request_kind).toBe("turn"); + expect(turns[1]?.turn_id).not.toBe(turns[0]?.turn_id); + expect(turns[2]?.turn_id).toBe(turns[1]?.turn_id); + expect(turns[2]?.turn_started_at_unix_ms).toBe(turns[1]?.turn_started_at_unix_ms); + expect(turns[3]?.request_kind).toBe("turn"); + expect(turns[3]?.turn_id).toBe(turns[1]?.turn_id); + expect(turns[3]?.window_id).not.toBe(turns[2]?.window_id); + expect(turns[4]?.turn_id).toBe(turns[1]?.turn_id); + expect(turns[5]?.turn_id).not.toBe(turns[4]?.turn_id); + expect(turns[1]?.thread_id).toBe(turns[5]?.thread_id); + expect(turns[1]?.compaction).toEqual({ + trigger: "auto", + reason: "context_limit", + implementation: "responses", + phase: "pre_turn", + strategy: "memento", + }); + const nestedCompaction = requireRecord(turns[1]?.compaction, "nested compaction metadata"); + expect(nestedCompaction.operationId).toBeUndefined(); + expect(nestedCompaction.operation_id).toBeUndefined(); + expect(turns[5]?.compaction).toEqual({ + trigger: "manual", + reason: "user_requested", + implementation: "responses", + phase: "standalone_turn", + strategy: "memento", + }); + }); + it("keeps lite and strips image detail when a lite request contains images", async () => { const model = buildModel({ id: "gpt-5.5", name: "GPT-5.5", @@ -382,19 +598,39 @@ describe("openai-codex Responses Lite and client metadata wire format", () => { ).result(); expect(result.stopReason).toBe("stop"); - expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBeNull(); + expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); expect(captured?.body.input).toEqual([ + { type: "additional_tools", role: "developer", tools: [] }, { role: "user", content: [ { type: "input_text", text: "read this image" }, - { type: "input_image", detail: "auto", image_url: "data:image/png;base64,AAAA" }, + { type: "input_image", image_url: "data:image/png;base64,AAAA" }, ], }, ]); }); - it("omits the lite header and client_metadata when not requested", async () => { + it("sends the lite header when the model defaults to Responses Lite", async () => { + const model = createCodexModel("gpt-5.6-terra", { useResponsesLite: true }); + let captured: CapturedCodexRequest | undefined; + const fetchMock = createCodexFetchMock(createCodexSse(COMPLETED_CODEX_EVENTS), request => { + captured = request; + }); + + const result = await streamOpenAICodexResponses(model, createCodexTestContext(), { + apiKey: createCodexTestToken(), + fetch: fetchMock, + }).result(); + + expect(result.stopReason).toBe("stop"); + expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBe("true"); + expect(captured?.body.instructions).toBeUndefined(); + expect(captured?.body.tools).toBeUndefined(); + expect((captured?.body.input as Array>)[0]?.type).toBe("additional_tools"); + }); + + it("omits the lite marker while retaining canonical client_metadata", async () => { const model = createCodexModel("gpt-5.1-codex"); let captured: CapturedCodexRequest | undefined; const fetchMock = createCodexFetchMock(createCodexSse(COMPLETED_CODEX_EVENTS), request => { @@ -408,7 +644,7 @@ describe("openai-codex Responses Lite and client metadata wire format", () => { expect(result.stopReason).toBe("stop"); expect(captured?.headers.get("x-openai-internal-codex-responses-lite")).toBeNull(); - expect(captured?.body.client_metadata).toBeUndefined(); + expect(captured?.body.client_metadata).toBeDefined(); }); }); @@ -467,3 +703,198 @@ describe("openai-codex websocket append with client metadata", () => { expect(transformed.client_metadata).toEqual({ "x-codex-turn-metadata": "{}" }); }); }); + +describe("openai-codex concurrent reasoning summaries", () => { + it("counts atomic summary dones as websocket watchdog progress", () => { + expect(isOpenAIResponsesProgressEvent({ type: "response.reasoning_summary_text.done" })).toBe(true); + }); + + it("sends stream_options only when a summary is requested and supported", async () => { + const terra = createCodexModel("gpt-5.6-terra"); + const withSummary = await transformRequestBody({ model: terra.id }, terra, { reasoningEffort: "medium" }); + expect(withSummary.stream_options).toEqual({ reasoning_summary_delivery: "sequential_cutoff" }); + + const suppressed = await transformRequestBody({ model: terra.id }, terra, { + reasoningEffort: "medium", + reasoningSummary: null, + }); + expect(suppressed.stream_options).toBeUndefined(); + + const noReasoning = await transformRequestBody({ model: terra.id }, terra, {}); + expect(noReasoning.stream_options).toBeUndefined(); + + const legacy = createCodexModel("gpt-5.1-codex"); + const unsupported = await transformRequestBody({ model: legacy.id }, legacy, { reasoningEffort: "medium" }); + expect(unsupported.stream_options).toBeUndefined(); + }); + + it("deduplicates cumulative atomic summaries and ignores legacy deltas under sequential cutoff", async () => { + const model = createCodexModel("gpt-5.6-terra"); + const events: Array> = [ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "reasoning", id: "reason_1", summary: [] }, + }, + { + type: "response.reasoning_summary_part.added", + item_id: "reason_1", + output_index: 0, + summary_index: 0, + part: { type: "summary_text", text: "" }, + }, + { + type: "response.reasoning_summary_text.delta", + item_id: "reason_1", + output_index: 0, + summary_index: 0, + delta: "IGNORED", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 0, + text: "Plan", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 1, + text: "Planning details", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 1, + text: "Planning details", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 2, + text: "Plan\n\nPlanning details\n\nInspect", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 2, + text: "Plan\n\nPlanning details\n\nInspect details", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 2, + text: "Plan\n\nPlanning details\n\nInspect details", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 3, + text: "Plan\n\nPlanning details\n\nInspect details", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 2, + text: "Plan\n\nPlanning details\n\nReview", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 2, + text: "Plan\n\nPlanning details\n\nReview output", + }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 3, + text: "Plan\n\nPlanning details\n\nReview output", + }, + { + type: "response.output_item.done", + output_index: 0, + item: { + type: "reasoning", + id: "reason_1", + summary: [ + { type: "summary_text", text: "Plan" }, + { type: "summary_text", text: "Planning details" }, + { type: "summary_text", text: "Plan\n\nPlanning details\n\nInspect details\n\nUnseen final" }, + { type: "summary_text", text: "Plan\n\nPlanning details\n\nInspect details\n\nUnseen final" }, + ], + }, + }, + { + type: "response.output_item.added", + output_index: 1, + item: { type: "message", id: "msg_1", role: "assistant", status: "in_progress", content: [] }, + }, + { type: "response.content_part.added", part: { type: "output_text", text: "" } }, + { type: "response.output_text.delta", item_id: "msg_1", output_index: 1, delta: "Hello" }, + { + type: "response.reasoning_summary_text.done", + item_id: "reason_1", + output_index: 0, + summary_index: 4, + text: "STALE", + }, + { + type: "response.output_item.done", + output_index: 1, + item: { + type: "message", + id: "msg_1", + role: "assistant", + status: "completed", + content: [{ type: "output_text", text: "Hello" }], + }, + }, + { + type: "response.completed", + response: { + status: "completed", + usage: { + input_tokens: 5, + output_tokens: 3, + total_tokens: 8, + input_tokens_details: { cached_tokens: 0 }, + }, + }, + }, + ]; + let captured: CapturedCodexRequest | undefined; + const fetchMock = createCodexFetchMock(createCodexSse(events), request => { + captured = request; + }); + + const stream = streamOpenAICodexResponses(model, createCodexTestContext(), { + apiKey: createCodexTestToken(), + fetch: fetchMock, + reasoning: "medium", + }); + const thinkingDeltas: string[] = []; + for await (const event of stream) { + if (event.type === "thinking_delta") thinkingDeltas.push(event.delta); + } + const result = await stream.result(); + + expect(captured?.body.stream_options).toEqual({ reasoning_summary_delivery: "sequential_cutoff" }); + expect(thinkingDeltas).toEqual(["Plan", "\n\nPlanning details", "\n\nInspect", " details"]); + expect(result.stopReason).toBe("stop"); + const thinking = result.content.find(block => block.type === "thinking"); + expect(thinking?.thinking).toBe("Plan\n\nPlanning details\n\nInspect details"); + expect(thinking?.thinking).toBe(thinkingDeltas.join("")); + const text = result.content.find(block => block.type === "text"); + expect(text?.text).toBe("Hello"); + }); +}); diff --git a/packages/ai/test/openai-codex-stream.test.ts b/packages/ai/test/openai-codex-stream.test.ts index e7fa92c2e..32b0b52ec 100644 --- a/packages/ai/test/openai-codex-stream.test.ts +++ b/packages/ai/test/openai-codex-stream.test.ts @@ -1,18 +1,29 @@ -import { afterEach, describe, expect, it, vi } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import { streamSimple } from "@oh-my-pi/pi-ai"; import { getOpenAICodexTransportDetails, getOpenAICodexWebSocketDebugStats, prewarmOpenAICodexResponses, + resetOpenAICodexHistoryAfterCompaction, streamOpenAICodexResponses, } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; -import type { Context, FetchImpl, Model, ModelSpec, ProviderSessionState } from "@oh-my-pi/pi-ai/types"; +import type { + CodexCompactionRequestContext, + Context, + FetchImpl, + Model, + ModelSpec, + ProviderSessionState, +} from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; -import { getAgentDir, setAgentDir, TempDir } from "@oh-my-pi/pi-utils"; +import * as piUtils from "@oh-my-pi/pi-utils"; + +const { getAgentDir, setAgentDir, TempDir } = piUtils; const originalAgentDir = getAgentDir(); const originalWebSocket = global.WebSocket; const originalCodexWebSocketV2 = Bun.env.PI_CODEX_WEBSOCKET_V2; +const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001"; function restoreEnv(name: string, value: string | undefined): void { if (value === undefined) { @@ -22,6 +33,10 @@ function restoreEnv(name: string, value: string | undefined): void { Bun.env[name] = value; } +beforeEach(() => { + vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID); +}); + afterEach(() => { global.WebSocket = originalWebSocket; setAgentDir(originalAgentDir); @@ -60,6 +75,22 @@ function createCodexTestContext(): Context { }; } +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +function requireRecord(value: unknown, label: string): Record { + if (!isRecord(value)) throw new Error(`expected ${label} to be an object`); + return value; +} + +function parseTurnMetadata(clientMetadata: Record): Record { + const encoded = clientMetadata["x-codex-turn-metadata"]; + if (typeof encoded !== "string") throw new Error("expected x-codex-turn-metadata"); + const decoded: unknown = JSON.parse(encoded); + return requireRecord(decoded, "x-codex-turn-metadata"); +} + function createCompletedCodexSse(text: string): string { return `${[ `data: ${JSON.stringify({ type: "response.content_part.added", part: { type: "output_text", text: "" } })}`, @@ -442,6 +473,129 @@ describe("openai-codex streaming", () => { expect(capturedText).toEqual({ verbosity: "low" }); }); + it("preserves streamed reasoning when the done item has no summary text", async () => { + const token = createCodexTestToken(); + const model = { ...createCodexTestModel("https://chatgpt.com/backend-api"), preferWebsockets: false }; + const events = [ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "reasoning", id: "rs_1", summary: [] }, + }, + { + type: "response.reasoning_summary_part.added", + output_index: 0, + item_id: "rs_1", + summary_index: 0, + part: { type: "summary_text", text: "" }, + }, + { + type: "response.reasoning_summary_text.delta", + output_index: 0, + item_id: "rs_1", + summary_index: 0, + delta: "streamed thinking", + }, + { + type: "response.output_item.done", + output_index: 0, + item: { type: "reasoning", id: "rs_1", summary: [] }, + }, + { + type: "response.output_item.added", + output_index: 1, + item: { type: "message", id: "msg_1", role: "assistant", status: "in_progress", content: [] }, + }, + { + type: "response.content_part.added", + output_index: 1, + item_id: "msg_1", + part: { type: "output_text", text: "" }, + }, + { type: "response.output_text.delta", output_index: 1, item_id: "msg_1", delta: "done" }, + { + type: "response.output_item.done", + output_index: 1, + item: { + type: "message", + id: "msg_1", + role: "assistant", + status: "completed", + content: [{ type: "output_text", text: "done" }], + }, + }, + { + type: "response.completed", + response: { + id: "resp_1", + status: "completed", + usage: { + input_tokens: 5, + output_tokens: 3, + total_tokens: 8, + input_tokens_details: { cached_tokens: 0 }, + }, + }, + }, + ]; + const sse = `${events.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`; + const fetchMock: FetchImpl = async () => + new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } }); + + const result = await streamOpenAICodexResponses(model, createCodexTestContext(), { + apiKey: token, + fetch: fetchMock, + }).result(); + + expect(result.content.find(block => block.type === "thinking")?.thinking).toBe("streamed thinking"); + }); + + it("streams raw reasoning text deltas into the final thinking block", async () => { + const token = createCodexTestToken(); + const model = { ...createCodexTestModel("https://chatgpt.com/backend-api"), preferWebsockets: false }; + const events = [ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "reasoning", id: "rs_raw", summary: [] }, + }, + { + type: "response.reasoning_text.delta", + output_index: 0, + item_id: "rs_raw", + delta: "raw streamed thinking", + }, + { + type: "response.output_item.done", + output_index: 0, + item: { type: "reasoning", id: "rs_raw", summary: [] }, + }, + { + type: "response.completed", + response: { + id: "resp_raw", + status: "completed", + usage: { + input_tokens: 5, + output_tokens: 3, + total_tokens: 8, + input_tokens_details: { cached_tokens: 0 }, + }, + }, + }, + ]; + const sse = `${events.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`; + const fetchMock: FetchImpl = async () => + new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } }); + + const result = await streamOpenAICodexResponses(model, createCodexTestContext(), { + apiKey: token, + fetch: fetchMock, + }).result(); + + expect(result.content.find(block => block.type === "thinking")?.thinking).toBe("raw streamed thinking"); + }); + it("maps end_turn=false on the terminal event to a pause_turn stop", async () => { const tempDir = TempDir.createSync("@pi-codex-stream-"); setAgentDir(tempDir.path()); @@ -1099,7 +1253,7 @@ describe("openai-codex streaming", () => { sessionId: "ws-lite-session", providerSessionState: new Map(), responsesLite: true, - clientMetadata: { "x-codex-turn-metadata": '{"thread_id":"t_1"}' }, + clientMetadata: { workspace_kind: "repo", "x-codex-turn-metadata": '{"thread_id":"caller"}' }, }, ).result(); @@ -1107,10 +1261,28 @@ describe("openai-codex streaming", () => { expect(capturedHeaders?.["x-openai-internal-codex-responses-lite"]).toBe("true"); expect(sentRequests).toHaveLength(1); expect(sentRequests[0]?.type).toBe("response.create"); - expect(sentRequests[0]?.client_metadata).toEqual({ - "x-codex-turn-metadata": '{"thread_id":"t_1"}', + const metadata = requireRecord(sentRequests[0]?.client_metadata, "client_metadata"); + const turnMetadata = parseTurnMetadata(metadata); + expect(metadata).toMatchObject({ + session_id: "ws-lite-session", ws_request_header_x_openai_internal_codex_responses_lite: "true", + "x-codex-installation-id": TEST_INSTALLATION_ID, }); + expect(metadata.workspace_kind).toBeUndefined(); + expect(turnMetadata).toMatchObject({ + installation_id: TEST_INSTALLATION_ID, + session_id: "ws-lite-session", + thread_id: metadata.thread_id, + turn_id: metadata.turn_id, + window_id: metadata["x-codex-window-id"], + request_kind: "turn", + workspace_kind: "repo", + }); + expect(capturedHeaders?.["x-codex-installation-id"]).toBeUndefined(); + expect(metadata.session_id).toBe(capturedHeaders?.["session-id"]); + expect(metadata.thread_id).toBe(capturedHeaders?.["thread-id"]); + expect(metadata["x-codex-window-id"]).toBe(capturedHeaders?.["x-codex-window-id"]); + expect(metadata["x-codex-turn-metadata"]).toBe(capturedHeaders?.["x-codex-turn-metadata"]); }); it("streams SSE responses into AssistantMessageEventStream", async () => { @@ -1987,7 +2159,7 @@ describe("openai-codex streaming", () => { expect(fallbackDetails.fallbackCount).toBe(1); }); - it("immediately falls back to SSE on fatal websocket connection errors", async () => { + it("carries fatal websocket fallback into isolated compaction transport", async () => { const tempDir = TempDir.createSync("@pi-codex-stream-"); setAgentDir(tempDir.path()); @@ -2050,6 +2222,23 @@ describe("openai-codex streaming", () => { expect(result.role).toBe("assistant"); expect(constructorCount).toBe(1); expect(fetchMock).toHaveBeenCalledTimes(1); + const compacted = await streamOpenAICodexResponses(model, context, { + fetch: fetchMock as FetchImpl, + apiKey: token, + sessionId: "ws-fatal-fallback-session", + providerSessionState, + codexCompaction: { + operationId: "fallback-compaction", + trigger: "auto", + reason: "context_limit", + implementation: "responses", + phase: "pre_turn", + strategy: "memento", + }, + }).result(); + expect(compacted.stopReason).toBe("stop"); + expect(constructorCount).toBe(1); + expect(fetchMock).toHaveBeenCalledTimes(2); const transportDetails = getOpenAICodexTransportDetails(model, { sessionId: "ws-fatal-fallback-session", providerSessionState, @@ -2059,7 +2248,7 @@ describe("openai-codex streaming", () => { expect(transportDetails.fallbackCount).toBe(1); }); - it("captures websocket handshake metadata and replays it on later SSE requests", async () => { + it("isolates compaction transport and preserves main mid-turn state", async () => { const tempDir = TempDir.createSync("@pi-codex-stream-"); setAgentDir(tempDir.path()); @@ -2076,12 +2265,21 @@ describe("openai-codex streaming", () => { `data: ${JSON.stringify({ type: "response.output_item.done", item: { type: "message", id: "msg_sse", role: "assistant", status: "completed", content: [{ type: "output_text", text: "Hello SSE" }] } })}`, `data: ${JSON.stringify({ type: "response.completed", response: { status: "completed", usage: { input_tokens: 5, output_tokens: 3, total_tokens: 8, input_tokens_details: { cached_tokens: 0 } } } })}`, ].join("\n\n")}\n\n`; + let firstRequest: Record | undefined; + let continuationRequest: Record | undefined; + let continuationHeaders: Headers | undefined; const fetchMock = vi.fn(async (_input: string | URL, init?: RequestInit) => { - const headers = init?.headers instanceof Headers ? init.headers : new Headers(init?.headers); - expect(headers.get("x-codex-turn-state")).toBe("ws-turn-state-1"); - expect(headers.get("x-models-etag")).toBe("models-etag-1"); + continuationHeaders = init?.headers instanceof Headers ? init.headers : new Headers(init?.headers); + expect(continuationHeaders.get("x-codex-turn-state")).toBe("ws-turn-state-1"); + expect(continuationHeaders.get("x-models-etag")).toBe("models-etag-1"); + if (typeof init?.body !== "string") throw new Error("expected an SSE request body"); + const body: unknown = JSON.parse(init.body); + continuationRequest = requireRecord(body, "SSE continuation request"); return new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } }); }); + let websocketRequestCount = 0; + let websocketConstructorCount = 0; + const websocketInstances: MockWebSocket[] = []; class HandshakeWebSocket extends MockWebSocket { handshakeHeaders = { @@ -2092,11 +2290,29 @@ describe("openai-codex streaming", () => { constructor(url: string, options?: { headers?: WsHeaders }) { super(url, options); + websocketConstructorCount += 1; + websocketInstances.push(this); this.scheduleOpen(); } - send(): void { - this.emitCodexResponse({ messageId: "msg_ws", responseId: "resp_ws", text: "Hello WS" }); + send(data: string): void { + websocketRequestCount += 1; + const body: unknown = JSON.parse(data); + if (websocketRequestCount === 1) { + firstRequest = requireRecord(body, "websocket request"); + } + if (websocketRequestCount === 3) { + this.sendJson({ + type: "response.failed", + response: { error: { code: "invalid_request_error", message: "isolated compaction failed" } }, + }); + return; + } + this.emitCodexResponse({ + messageId: `msg_ws_${websocketRequestCount}`, + responseId: `resp_ws_${websocketRequestCount}`, + text: "Hello WS", + }); } } @@ -2125,12 +2341,65 @@ describe("openai-codex streaming", () => { messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }], }; const providerSessionState = new Map(); + const midTurnCompaction: CodexCompactionRequestContext = { + operationId: "isolated-success", + trigger: "auto", + reason: "context_limit", + implementation: "responses", + phase: "mid_turn", + strategy: "memento", + }; const first = await streamOpenAICodexResponses(websocketModel, context, { fetch: fetchMock as FetchImpl, apiKey: token, sessionId: "ws-handshake-session", providerSessionState, }).result(); + expect(websocketInstances[0]?.readyState).toBe(MockWebSocket.OPEN); + const isolatedSuccess = await streamOpenAICodexResponses(websocketModel, createCodexTestContext(), { + fetch: fetchMock as FetchImpl, + apiKey: token, + sessionId: "ws-handshake-session", + providerSessionState, + codexCompaction: midTurnCompaction, + }).result(); + expect(isolatedSuccess.stopReason).toBe("stop"); + expect(websocketInstances[0]?.readyState).toBe(MockWebSocket.OPEN); + expect(websocketInstances[1]?.readyState).toBe(MockWebSocket.CLOSED); + expect(websocketInstances[1]?.options?.headers?.["x-codex-turn-state"]).toBe("ws-turn-state-1"); + expect(websocketInstances[1]?.options?.headers?.["x-models-etag"]).toBe("models-etag-1"); + resetOpenAICodexHistoryAfterCompaction({ + providerSessionState, + sessionId: "ws-handshake-session", + compaction: midTurnCompaction, + }); + const isolatedFailure = await streamOpenAICodexResponses(websocketModel, createCodexTestContext(), { + fetch: fetchMock as FetchImpl, + apiKey: token, + sessionId: "ws-handshake-session", + providerSessionState, + codexCompaction: { + operationId: "isolated-failure", + trigger: "auto", + reason: "context_limit", + implementation: "responses", + phase: "mid_turn", + strategy: "memento", + }, + }).result(); + expect(isolatedFailure.stopReason).toBe("error"); + expect(websocketInstances[0]?.readyState).toBe(MockWebSocket.OPEN); + expect(websocketInstances[2]?.readyState).toBe(MockWebSocket.CLOSED); + expect(websocketConstructorCount).toBe(3); + expect( + getOpenAICodexTransportDetails(websocketModel, { + sessionId: "ws-handshake-session", + providerSessionState, + }), + ).toMatchObject({ + websocketConnected: true, + hasTurnState: true, + }); // Turn-state is scoped to the current turn, so the SSE replay must be a // within-turn continuation (trailing tool result) to carry the header. const followUp: Context = { @@ -2162,6 +2431,155 @@ describe("openai-codex streaming", () => { providerSessionState, }).result(); expect(fetchMock).toHaveBeenCalledTimes(1); + if (!firstRequest || !continuationRequest || !continuationHeaders) { + throw new Error("expected both Codex transport requests"); + } + const firstMetadata = requireRecord(firstRequest.client_metadata, "first client_metadata"); + const continuationMetadata = requireRecord(continuationRequest.client_metadata, "continuation client_metadata"); + const firstTurnMetadata = parseTurnMetadata(firstMetadata); + const continuationTurnMetadata = parseTurnMetadata(continuationMetadata); + expect(continuationMetadata).toMatchObject({ + "x-codex-installation-id": TEST_INSTALLATION_ID, + session_id: firstMetadata.session_id, + thread_id: firstMetadata.thread_id, + turn_id: firstMetadata.turn_id, + }); + expect(continuationTurnMetadata).toMatchObject({ + installation_id: TEST_INSTALLATION_ID, + session_id: firstTurnMetadata.session_id, + thread_id: firstTurnMetadata.thread_id, + turn_id: firstTurnMetadata.turn_id, + window_id: continuationMetadata["x-codex-window-id"], + request_kind: "turn", + turn_started_at_unix_ms: context.messages[0]?.timestamp, + }); + expect(typeof continuationMetadata["x-codex-window-id"]).toBe("string"); + expect(continuationMetadata["x-codex-window-id"]).not.toBe(firstMetadata["x-codex-window-id"]); + expect(firstMetadata.session_id).toBe(continuationHeaders.get("session-id")); + expect(firstMetadata.thread_id).toBe(continuationHeaders.get("thread-id")); + expect(continuationMetadata["x-codex-window-id"]).toBe(continuationHeaders.get("x-codex-window-id")); + expect(continuationMetadata["x-codex-turn-metadata"]).toBe(continuationHeaders.get("x-codex-turn-metadata")); + }); + + it("clears stale main turn-state after pre-turn compaction", async () => { + const tempDir = TempDir.createSync("@pi-codex-stream-"); + setAgentDir(tempDir.path()); + const token = createCodexTestToken(); + const websocketInstances: MockWebSocket[] = []; + let websocketRequestCount = 0; + + class PreTurnCompactionWebSocket extends MockWebSocket { + handshakeHeaders = { + "x-codex-turn-state": "stale-main-turn-state", + "x-models-etag": "models-etag-1", + }; + + constructor(url: string, options?: { headers?: WsHeaders }) { + super(url, options); + websocketInstances.push(this); + queueMicrotask(() => { + this.readyState = MockWebSocket.OPEN; + this.emit("open", new Event("open")); + }); + } + + send(_data: string): void { + websocketRequestCount += 1; + this.emitCodexResponse({ + messageId: `msg_pre_turn_${websocketRequestCount}`, + responseId: `resp_pre_turn_${websocketRequestCount}`, + text: "Hello WS", + }); + } + } + + global.WebSocket = PreTurnCompactionWebSocket as unknown as typeof WebSocket; + const websocketModel = createCodexTestModel("https://chatgpt.com/backend-api"); + const sseModel: Model<"openai-codex-responses"> = buildModel({ + id: websocketModel.id, + name: websocketModel.name, + api: "openai-codex-responses", + provider: websocketModel.provider, + baseUrl: websocketModel.baseUrl, + reasoning: true, + preferWebsockets: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128000, + maxTokens: 128000, + }); + const providerSessionState = new Map(); + const sessionId = "pre-turn-reset-session"; + let sseHeaders: Headers | undefined; + const fetchMock = vi.fn(async (_input: string | URL, init?: RequestInit) => { + sseHeaders = init?.headers instanceof Headers ? init.headers : new Headers(init?.headers); + return new Response(createCompletedCodexSse("Hello SSE"), { + headers: { "content-type": "text/event-stream" }, + }); + }); + const compaction: CodexCompactionRequestContext = { + operationId: "pre-turn-reset-operation", + trigger: "auto", + reason: "context_limit", + implementation: "responses", + phase: "pre_turn", + strategy: "memento", + }; + + try { + await streamOpenAICodexResponses(websocketModel, createCodexTestContext(), { + apiKey: token, + fetch: fetchMock as FetchImpl, + sessionId, + providerSessionState, + }).result(); + await streamOpenAICodexResponses(websocketModel, createCodexTestContext(), { + apiKey: token, + fetch: fetchMock as FetchImpl, + sessionId, + providerSessionState, + codexCompaction: compaction, + }).result(); + expect(fetchMock).not.toHaveBeenCalled(); + expect(websocketInstances).toHaveLength(2); + expect(websocketInstances[0]?.readyState).toBe(MockWebSocket.OPEN); + expect(websocketInstances[1]?.readyState).toBe(MockWebSocket.CLOSED); + expect(websocketInstances[1]?.options?.headers?.["x-codex-turn-state"]).toBeUndefined(); + expect(websocketInstances[1]?.options?.headers?.["x-models-etag"]).toBe("models-etag-1"); + + resetOpenAICodexHistoryAfterCompaction({ + providerSessionState, + sessionId, + compaction, + }); + expect( + getOpenAICodexTransportDetails(websocketModel, { + sessionId, + providerSessionState, + }), + ).toMatchObject({ + websocketConnected: true, + hasTurnState: false, + }); + await streamOpenAICodexResponses( + sseModel, + { + systemPrompt: ["You are a helpful assistant."], + messages: [{ role: "user", content: "Continue after compaction", timestamp: Date.now() }], + }, + { + apiKey: token, + fetch: fetchMock as FetchImpl, + sessionId, + providerSessionState, + }, + ).result(); + expect(fetchMock).toHaveBeenCalledTimes(1); + expect(sseHeaders?.get("x-codex-turn-state")).toBeNull(); + } finally { + for (const state of providerSessionState.values()) state.close(); + providerSessionState.clear(); + } }); it("includes service_tier in websocket payloads when requested", async () => { @@ -2349,6 +2767,17 @@ describe("openai-codex streaming", () => { expect(deltaItems[0]?.role).toBe("user"); expect(JSON.stringify(deltaItems)).toContain("Second question"); expect(JSON.stringify(deltaItems)).not.toContain("First answer"); + const firstMetadata = requireRecord(sentRequests[0]?.client_metadata, "first client_metadata"); + const secondMetadata = requireRecord(sentRequests[1]?.client_metadata, "second client_metadata"); + expect(secondMetadata).toMatchObject({ + "x-codex-installation-id": firstMetadata["x-codex-installation-id"], + session_id: firstMetadata.session_id, + thread_id: firstMetadata.thread_id, + "x-codex-window-id": firstMetadata["x-codex-window-id"], + }); + expect(secondMetadata.turn_id).not.toBe(firstMetadata.turn_id); + expect(parseTurnMetadata(firstMetadata).turn_started_at_unix_ms).toBe(firstContext.messages[0]?.timestamp); + expect(parseTurnMetadata(secondMetadata).turn_started_at_unix_ms).toBe(secondContext.messages.at(-1)?.timestamp); const stats = getOpenAICodexWebSocketDebugStats(model, { sessionId: "ws-delta-session", @@ -3815,10 +4244,12 @@ describe("openai-codex streaming", () => { let constructorCount = 0; let sendCount = 0; + let prewarmHeaders: WsHeaders | undefined; class ReusableWebSocket extends MockWebSocket { constructor(url: string, options?: { headers?: WsHeaders }) { super(url, options); constructorCount += 1; + prewarmHeaders = options?.headers; this.scheduleOpen(); } @@ -3856,6 +4287,11 @@ describe("openai-codex streaming", () => { sessionId: "ws-reuse-session", providerSessionState, }); + expect(prewarmHeaders?.["session-id"]).toBe("ws-reuse-session"); + expect(prewarmHeaders?.["thread-id"]).toBeDefined(); + expect(prewarmHeaders?.["x-codex-window-id"]).toBeDefined(); + expect(prewarmHeaders?.["x-codex-turn-metadata"]).toBeUndefined(); + expect(prewarmHeaders?.["x-codex-installation-id"]).toBeUndefined(); const firstContext: Context = { systemPrompt: ["You are a helpful assistant."], @@ -3893,6 +4329,26 @@ describe("openai-codex streaming", () => { expect(transportDetails.websocketConnected).toBe(true); expect(transportDetails.prewarmed).toBe(true); expect(transportDetails.canAppend).toBe(true); + resetOpenAICodexHistoryAfterCompaction({ + providerSessionState, + sessionId: "ws-reuse-session", + compaction: { + operationId: "history-rewrite", + trigger: "auto", + reason: "context_limit", + phase: "pre_turn", + strategy: "memento", + }, + }); + expect( + getOpenAICodexTransportDetails(model, { + sessionId: "ws-reuse-session", + providerSessionState, + }), + ).toMatchObject({ + websocketConnected: true, + canAppend: false, + }); }); it("scopes x-codex-turn-state to the current turn on SSE requests", async () => { diff --git a/packages/ai/test/openai-codex.test.ts b/packages/ai/test/openai-codex.test.ts index e28633fd1..1c7492afc 100644 --- a/packages/ai/test/openai-codex.test.ts +++ b/packages/ai/test/openai-codex.test.ts @@ -314,6 +314,36 @@ describe("openai-codex reasoning effort validation", () => { }); }); +describe("openai-codex reasoning effort wire mapping", () => { + it("shifts gpt-5.6 user efforts one wire tier up via the baked effort map", async () => { + const model = createCodexModel("gpt-5.6-sol"); + const shifted = [ + ["minimal", "low"], + ["low", "medium"], + ["medium", "high"], + ["high", "xhigh"], + ["xhigh", "max"], + ] as const; + + for (const [requested, wire] of shifted) { + const transformed = await transformRequestBody({ model: model.id }, model, { + reasoningEffort: requested, + }); + expect(transformed.reasoning?.effort).toBe(wire); + } + }); + + it("keeps pre-5.6 efforts unshifted and passes none through unmapped", async () => { + const gpt55 = createCodexModel("gpt-5.5"); + const unshifted = await transformRequestBody({ model: gpt55.id }, gpt55, { reasoningEffort: "xhigh" }); + expect(unshifted.reasoning?.effort).toBe("xhigh"); + + const gpt56 = createCodexModel("gpt-5.6-sol"); + const none = await transformRequestBody({ model: gpt56.id }, gpt56, { reasoningEffort: "none" }); + expect(none.reasoning?.effort).toBe("none"); + }); +}); + describe("openai-codex error parsing", () => { it("produces friendly usage-limit messages and rate limits", async () => { const resetAt = Math.floor(Date.now() / 1000) + 600; diff --git a/packages/ai/test/openai-completions-cache-affinity.test.ts b/packages/ai/test/openai-completions-cache-affinity.test.ts new file mode 100644 index 000000000..564e22fae --- /dev/null +++ b/packages/ai/test/openai-completions-cache-affinity.test.ts @@ -0,0 +1,91 @@ +import { describe, expect, it } from "bun:test"; +import { type OpenAICompletionsOptions, streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; +import type { Context, FetchImpl } from "@oh-my-pi/pi-ai/types"; +import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; + +const model = getBundledModel<"openai-completions">("xai", "grok-code-fast-1"); +if (!model) throw new Error("Expected bundled xAI Grok model"); +if (model.api !== "openai-completions") throw new Error(`Expected Chat Completions model, received ${model.api}`); +const context: Context = { messages: [{ role: "user", content: "hello", timestamp: 0 }] }; + +function chatCompletionsSse(): Response { + const chunk = (delta: unknown, finishReason: string | null) => + JSON.stringify({ + id: "chatcmpl-affinity", + object: "chat.completion.chunk", + created: 0, + model: model.id, + choices: [{ index: 0, delta, finish_reason: finishReason }], + }); + + return new Response( + `data: ${chunk({ role: "assistant", content: "ok" }, null)}\n\ndata: ${chunk({}, "stop")}\n\ndata: [DONE]\n\n`, + { status: 200, headers: { "content-type": "text/event-stream" } }, + ); +} + +async function captureRequestHeaders(options: OpenAICompletionsOptions): Promise { + let requestHeaders: Headers | undefined; + const fetchMock: FetchImpl = async (input: string | URL | Request, init?: RequestInit) => { + const request = + input instanceof Request + ? new Request(input, init) + : new Request(input instanceof URL ? input.href : input, init); + requestHeaders = request.headers; + return chatCompletionsSse(); + }; + + await streamOpenAICompletions(model, context, { + apiKey: "test-key", + ...options, + fetch: fetchMock, + }).result(); + + if (!requestHeaders) throw new Error("Expected a serialized Chat Completions request"); + return requestHeaders; +} + +describe("openai-completions xAI cache affinity", () => { + const cases: Array<{ + name: string; + options: OpenAICompletionsOptions; + expectedHeader: string | null; + }> = [ + { + name: "uses sessionId when no prompt cache key is provided", + options: { sessionId: "session-fallback" }, + expectedHeader: "session-fallback", + }, + { + name: "keeps the prompt cache key stable across a distinct side-channel session", + options: { promptCacheKey: "stable-cache-key", sessionId: "side-channel-session" }, + expectedHeader: "stable-cache-key", + }, + { + name: "omits automatic affinity when caching is disabled", + options: { + promptCacheKey: "disabled-cache-key", + sessionId: "disabled-session", + cacheRetention: "none", + }, + expectedHeader: null, + }, + { + name: "preserves a caller-provided mixed-case affinity header", + options: { + promptCacheKey: "automatic-cache-key", + sessionId: "automatic-session", + headers: { "X-Grok-Conv-Id": "caller-affinity" }, + }, + expectedHeader: "caller-affinity", + }, + ]; + + for (const { name, options, expectedHeader } of cases) { + it(name, async () => { + const headers = await captureRequestHeaders(options); + + expect(headers.get("x-grok-conv-id")).toBe(expectedHeader); + }); + } +}); diff --git a/packages/ai/test/openai-responses-history-payload.test.ts b/packages/ai/test/openai-responses-history-payload.test.ts index afdedd481..9520fabf0 100644 --- a/packages/ai/test/openai-responses-history-payload.test.ts +++ b/packages/ai/test/openai-responses-history-payload.test.ts @@ -1,4 +1,4 @@ -import { describe, expect, it } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import { convertCodexResponsesMessages, streamOpenAICodexResponses, @@ -9,6 +9,17 @@ import type { Context, Model, ModelSpec, ProviderSessionState } from "@oh-my-pi/ import { createOpenAIResponsesHistoryPayload, truncateResponseItemId } from "@oh-my-pi/pi-ai/utils"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; +import * as piUtils from "@oh-my-pi/pi-utils"; + +const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001"; + +beforeEach(() => { + vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID); +}); + +afterEach(() => { + vi.restoreAllMocks(); +}); function createAbortedSignal(): AbortSignal { const controller = new AbortController(); diff --git a/packages/ai/test/openai-responses-stream-terminal.test.ts b/packages/ai/test/openai-responses-stream-terminal.test.ts index 95c501834..757a7f31f 100644 --- a/packages/ai/test/openai-responses-stream-terminal.test.ts +++ b/packages/ai/test/openai-responses-stream-terminal.test.ts @@ -288,7 +288,13 @@ describe("processResponsesStream: lost output_item.added recovery", () => { { type: "response.output_item.done", output_index: 0, - item: { type: "reasoning", summary: [{ type: "summary_text", text: "first" }] }, + item: { + type: "reasoning", + summary: [ + { type: "summary_text", text: "Plan" }, + { type: "summary_text", text: "Planning details" }, + ], + }, }, { type: "response.output_item.done", @@ -305,12 +311,55 @@ describe("processResponsesStream: lost output_item.added recovery", () => { expect(output.content).toHaveLength(2); const [first, second] = output.content; if (first?.type !== "thinking" || second?.type !== "thinking") throw new Error("expected thinking blocks"); - expect(first.thinking).toBe("first"); + expect(first.thinking).toBe("Plan\n\nPlanning details"); expect(second.thinking).toBe("second"); expect(first.thinkingSignature).toBeDefined(); expect(second.thinkingSignature).toBeDefined(); }); + test("preserves streamed reasoning when the done item has no summary text", async () => { + const output = makeOutput(); + const stream = { push: () => {}, end: () => {} } as never; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "reasoning", id: "rs_1", summary: [] }, + }, + { + type: "response.reasoning_summary_part.added", + output_index: 0, + item_id: "rs_1", + summary_index: 0, + part: { type: "summary_text", text: "" }, + }, + { + type: "response.reasoning_summary_text.delta", + output_index: 0, + item_id: "rs_1", + summary_index: 0, + delta: "streamed thinking", + }, + { + type: "response.output_item.done", + output_index: 0, + item: { type: "reasoning", id: "rs_1", summary: [] }, + }, + { type: "response.completed", response: { id: "resp_reasoning", status: "completed" } }, + ]), + output, + stream, + makeModel(), + ); + + const block = output.content[0]; + if (block?.type !== "thinking") throw new Error("expected a thinking block"); + expect(block.thinking).toBe("streamed thinking"); + expect(block.thinkingSignature).toBeDefined(); + }); + test("treats content_filter incomplete responses as errors, not length", async () => { const output = makeOutput(); const stream = { push: () => {}, end: () => {} } as never; diff --git a/packages/ai/test/provider-registry.test.ts b/packages/ai/test/provider-registry.test.ts index a72eb59c7..2ed3d88bd 100644 --- a/packages/ai/test/provider-registry.test.ts +++ b/packages/ai/test/provider-registry.test.ts @@ -81,7 +81,6 @@ describe("provider registry auth surface", () => { "google-antigravity", "google-gemini-cli", "openai-codex", - "xai-oauth", ].sort(), ); expect(PASTE_CODE_LOGIN_PROVIDERS.has("zenmux")).toBe(false); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 8bf50c3a7..2704465ef 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,62 @@ ## [Unreleased] +### Added + +- Added Grok 4.5 model family +- Added support for Dolphin Mistral 24b Venice Edition +- Added GLM5.2-Fast model +- Added Zenmux variants for GPT-5.6 (Luna, Sol, and Terra) +- Added Novita as a model provider with authoritative public catalog discovery and generated pricing, limits, modality, reasoning, and tool metadata ([#4917](https://github.com/can1357/oh-my-pi/pull/4917) by [@jason-wu-ai](https://github.com/jason-wu-ai)). + +- Added `useResponsesLite` to `Model`/`ModelSpec` and Codex discovery parsing of the upstream `use_responses_lite` flag; regenerated `models.json` marks the GPT-5.6 family (`sol`/`terra`/`luna` and their pro aliases) for the Responses Lite transport. Added the `x-openai-internal-codex-responses-lite` marker to `OPENAI_HEADERS`. + +### Changed + +- Updated costs and context windows for various models in the catalog + +## [16.3.15] - 2026-07-09 + +### Added + +- Added support for Grok 4.5 model +- Added `gpt-5.6` base models and `gpt-5.6-{luna,sol,terra}-pro` variants +- Added `meta/muse-spark-1.1` model support +- Added support for thinking modes on `poolside/laguna` models +- Added generated GPT-5.6 Pro aliases (`gpt-5.6-{luna,sol,terra}-pro`) on the `openai` and `openai-codex` providers: each alias sends the base model id on the wire (`requestModelId`) with the new `reasoningMode: "pro"` marker, and re-derives from the current base rows on every catalog regeneration. + +### Changed + +- Updated cache read costs for Grok models +- Reduced max token limit for Grok 4.3 model +- Enabled prompt cache affinity for Grok models via the x-grok-conv-id header in OpenAI compatible endpoints +- Enabled prompt cache affinity for Grok models via the x-grok-conv-id header +- Marked direct xAI Grok Chat Completions models for `x-grok-conv-id` prompt-cache affinity. + +## [16.3.14] - 2026-07-09 + +### Added + +- Added support for GPT-5.6 (Luna, Sol, Terra) model variants +- Enabled expanded five-tier reasoning effort scale (minimal to xhigh) for GPT-5.6 models +- Added GPT-5.6 (Terra/Luna/Sol) support for the new `max` reasoning tier: on wire-effort APIs (OpenAI Responses, Codex, Azure, openai-compat/OpenRouter models that advertise reasoning) user efforts shift up one notch — `xhigh` sends `max`, `high` sends `xhigh` — mirroring the Claude Fable/Opus 4.7+ five-tier mapping, and the exposed ladder becomes `minimal..xhigh` with `minimal` reaching the native `low` tier. Devin's per-tier GPT-5.6 sibling rows now collapse into `gpt-5-6-{luna,sol,terra}` logical models with the same shifted routing (`xhigh` → `-max`), plus `-fast` families that keep the direct `low..xhigh` `-priority` scale since Devin serves no `-max-priority` tier. + +## [16.3.13] - 2026-07-09 + +### Added + +- Added support for Grok 4.5 across multiple providers +- Added support for GPT-5.6 series models (Luna, Sol, Terra) +- Added Aion 3.0 and 3.0 Mini models +- Added Kuaishou KAT-Coder v2.5 models +- Added Nex-N2-Mini and SWE-1.7 series models +- Added Hy3 models and free variants + +### Changed + +- Updated cost and token configurations for various models across providers +- Renamed several models for consistency (e.g., MiniMax M3, Gemma 4 31B, Qwen variants) + ## [16.3.12] - 2026-07-08 ### Fixed diff --git a/packages/catalog/package.json b/packages/catalog/package.json index e2272c3d0..653aac2e6 100644 --- a/packages/catalog/package.json +++ b/packages/catalog/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-catalog", - "version": "16.3.12", + "version": "16.3.15", "description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/catalog/scripts/generate-models.ts b/packages/catalog/scripts/generate-models.ts index 09b9548a9..1d9df8b78 100644 --- a/packages/catalog/scripts/generate-models.ts +++ b/packages/catalog/scripts/generate-models.ts @@ -38,6 +38,7 @@ import { isKimiK27CodeModelId, MODELS_DEV_PROVIDER_DESCRIPTORS, mapModelsDevToModels, + projectOpenAIProReasoningAliases, SAKANA_FUGU_STATIC_MODELS, stripFireworksDeepSeekThinkingToggle, } from "../src/provider-models/openai-compat"; @@ -585,6 +586,10 @@ async function generateModels() { const name = cleanModelName(model.name); return name === model.name ? model : { ...model, name }; }); + // Re-derive the first-party gpt-5.6 pro-reasoning aliases from the current + // base rows (stale previous-snapshot aliases are dropped inside), before the + // policy re-bake so the aliases get the same baked thinking metadata. + allModels = projectOpenAIProReasoningAliases(allModels); applyGeneratedModelPolicies(allModels); linkOpenAIPromotionTargets(allModels); // Collapse effort-tier variants AFTER the policy re-bake: live-discovery diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index d5627ccd5..dcfcd3df9 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -541,7 +541,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv MINIMAX_PROVIDER_OR_ID_PATTERN.test(provider) || MINIMAX_PROVIDER_OR_ID_PATTERN.test(spec.id), emptyLengthFinishIsContextError: provider === "ollama", usesOpenAIToolCallIdLimit: provider === "openai", - promptCacheSessionHeader: undefined, + promptCacheSessionHeader: isGrok ? "x-grok-conv-id" : undefined, dropThinkingWhenReasoningEffort: provider === "fireworks", }; diff --git a/packages/catalog/src/discovery/codex.ts b/packages/catalog/src/discovery/codex.ts index e8a03e0ea..5784d7088 100644 --- a/packages/catalog/src/discovery/codex.ts +++ b/packages/catalog/src/discovery/codex.ts @@ -30,6 +30,7 @@ const codexModelEntrySchema = type({ "supported_in_api?": "unknown", "priority?": "unknown", "prefer_websockets?": "unknown", + "use_responses_lite?": "unknown", }); const codexModelsResponseSchema = type({ @@ -262,6 +263,7 @@ function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCo const reasoning = supportsReasoning(payload.default_reasoning_level, payload.supported_reasoning_levels); const input = normalizeInputModalities(payload.input_modalities); const preferWebsockets = toBoolean(payload.prefer_websockets) === true; + const useResponsesLite = toBoolean(payload.use_responses_lite) === true; const priority = toFiniteNumber(payload.priority) ?? Number.MAX_SAFE_INTEGER; return { @@ -279,6 +281,7 @@ function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCo contextWindow, maxTokens, ...(preferWebsockets ? { preferWebsockets: true } : {}), + ...(useResponsesLite ? { useResponsesLite: true } : {}), ...(priority !== Number.MAX_SAFE_INTEGER ? { priority } : {}), }, }; diff --git a/packages/catalog/src/discovery/cursor.ts b/packages/catalog/src/discovery/cursor.ts index 880aefec5..65d6ca645 100644 --- a/packages/catalog/src/discovery/cursor.ts +++ b/packages/catalog/src/discovery/cursor.ts @@ -13,6 +13,13 @@ const CURSOR_GET_USABLE_MODELS_PATH = "/agent.v1.AgentService/GetUsableModels"; const DEFAULT_CONTEXT_WINDOW = 200_000; const DEFAULT_MAX_TOKENS = 64_000; +/** + * Model-id families whose native catalogs (anthropic, openai/openai-codex, + * google) are multimodal. Cursor-only or text-only families (`composer-*`, + * `grok-code-*`) intentionally stay outside this pattern. + */ +const CURSOR_MULTIMODAL_ID_PATTERN = /claude|gemini|gpt-|codex/; + const OptionalDisplayNameSchema = type("unknown").pipe(raw => (typeof raw === "string" ? raw : undefined)); const CursorAliasesSchema = type("unknown").pipe(raw => { if (Array.isArray(raw)) { @@ -292,7 +299,7 @@ function normalizeCursorModel( provider: "cursor", baseUrl: baseUrlOverride ?? CURSOR_DEFAULT_BASE_URL, reasoning, - input: ["text"], + input: inferInputFromCursorId(id), cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: DEFAULT_CONTEXT_WINDOW, maxTokens: DEFAULT_MAX_TOKENS, @@ -312,3 +319,18 @@ function pickModelDisplayName(model: CursorModelDetailsValue, fallbackId: string } return fallbackId; } + +/** + * Infers input modalities for Cursor models without a bundled reference. + * + * `GetUsableModels` carries no per-model modality metadata, so classification + * falls back to the model family: families that are multimodal in OMP's own + * native catalogs accept images, everything else stays text-only. Mirrors + * `inferInputFromGeminiId` in ./gemini.ts. + */ +function inferInputFromCursorId(id: string): ("text" | "image")[] { + if (CURSOR_MULTIMODAL_ID_PATTERN.test(id.toLowerCase())) { + return ["text", "image"]; + } + return ["text"]; +} diff --git a/packages/catalog/src/identity/family.ts b/packages/catalog/src/identity/family.ts index b62bfe433..aa206cd9f 100644 --- a/packages/catalog/src/identity/family.ts +++ b/packages/catalog/src/identity/family.ts @@ -78,7 +78,7 @@ export const isMimoModelIdOrName = memo((value: string): boolean => { return value.toLowerCase().includes("mimo"); }); -const GROK_EFFORT_CAPABLE_PREFIXES = ["grok-3-mini", "grok-4.20-multi-agent", "grok-4.3"] as const; +const GROK_EFFORT_CAPABLE_PREFIXES = ["grok-3-mini", "grok-4.20-multi-agent", "grok-4.3", "grok-4.5"] as const; /** * Grok SKUs that expose the wire `reasoning.effort` dial. Other Grok reasoners diff --git a/packages/catalog/src/model-manager.ts b/packages/catalog/src/model-manager.ts index eccec451f..33fc8cf88 100644 --- a/packages/catalog/src/model-manager.ts +++ b/packages/catalog/src/model-manager.ts @@ -35,7 +35,7 @@ export interface ModelManagerOptions = { }; /** - * Effort → wire-value map for the 5-tier adaptive scale (Opus 4.7+ and - * Fable/Mythos 5 on the Messages API). User-facing efforts shift up one notch - * so the top tier reaches the genuine "max" and "high" lands on Anthropic's - * recommended "xhigh" coding/agentic default. + * Effort → wire-value map for a shifted five-tier scale (`low..max`): + * user-facing efforts shift up one notch so the top tier reaches the genuine + * "max" and "high" lands on the recommended "xhigh" coding/agentic default. + * Used by Anthropic adaptive models with a real xhigh tier (Opus 4.7+ and + * Fable/Mythos 5 on the Messages API) and by GPT-5.6+ wire-effort models, + * which expose the same genuine `max` tier above `xhigh`. */ -export const ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER: Readonly>> = { +export const SHIFTED_FIVE_TIER_EFFORT_MAP: Readonly>> = { [Effort.Minimal]: "low", [Effort.Low]: "medium", [Effort.Medium]: "high", @@ -295,6 +298,27 @@ function isOpenAICompatReasoningApi(api: Api): boolean { return api === "openai-completions" || api === "openrouter"; } +/** + * GPT-5.6+ addressed through a wire `reasoning.effort`/`reasoning_effort` + * field, where the shifted five-tier map applies. Devin (`devin-agent`) + * selects effort by routing to per-tier sibling model ids instead and must + * stay unmapped. + */ +function isGpt56PlusWireEffortModel(spec: ModelSpec): boolean { + switch (spec.api) { + case "openai-responses": + case "openai-codex-responses": + case "azure-openai-responses": + case "openai-completions": + case "openrouter": + break; + default: + return false; + } + const parsed = parseOpenAIModel(bareModelId(spec.id)); + return parsed !== null && semverGte(parsed.version, "5.6"); +} + function getModelDefinedEfforts( spec: ModelSpec, compat: CompatOf, @@ -313,6 +337,12 @@ function getModelDefinedEfforts( if (isSakanaFuguReasoningModel(spec)) { return FUGU_REASONING_EFFORTS; } + if (isGpt56PlusWireEffortModel(spec)) { + // Normalize stale baked/discovered `low..xhigh` surfaces to the full + // five-tier ladder so the shifted map keeps the native `low` tier + // reachable (user `minimal`). + return DEFAULT_REASONING_EFFORTS_WITH_XHIGH; + } return isOpenAICompatReasoningApi(spec.api) && (isMinimaxM2FamilyModelId(spec.id) || isOpenAIGptOssModelId(spec.id) || @@ -373,7 +403,7 @@ function inferDetectedEffortMap( return MINIMAX_ANTHROPIC_ADAPTIVE_EFFORT_MAP; } return anthropicModelHasRealXHighEffort(spec, parsedModel) - ? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER + ? SHIFTED_FIVE_TIER_EFFORT_MAP : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER; } // GLM-5.2 coding SKUs accept `reasoning_effort`, but the effort dialect is @@ -397,6 +427,9 @@ function inferDetectedEffortMap( if (isSakanaFuguReasoningModel(spec)) { return FUGU_REASONING_EFFORT_MAP; } + if (isGpt56PlusWireEffortModel(spec)) { + return SHIFTED_FIVE_TIER_EFFORT_MAP; + } if (!isOpenAICompatReasoningApi(spec.api)) { return undefined; } @@ -446,7 +479,7 @@ function getOpenRouterAnthropicReasoningEffortMap(modelId: string): EffortMap | if (!isAnthropicAdaptiveGenAtLeast(parsed, "4.6")) return undefined; const hasRealXHigh = isAnthropicAdaptiveGenAtLeast(parsed, "4.7"); - return hasRealXHigh ? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER; + return hasRealXHigh ? SHIFTED_FIVE_TIER_EFFORT_MAP : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER; } function inferSupportedEfforts( @@ -474,6 +507,11 @@ function inferOpenAISupportedEfforts(model: OpenAIModel): readonly Effort[] { if (model.variant === "codex-mini" && semverEqual(model.version, "5.1")) { return GPT_5_1_CODEX_MINI_EFFORTS; } + // 5.6+ exposes the full five-tier ladder: the shifted wire map spans + // low..max, with user `minimal` reaching the native `low` tier. + if (semverGte(model.version, "5.6")) { + return DEFAULT_REASONING_EFFORTS_WITH_XHIGH; + } if (semverGte(model.version, "5.2")) { return GPT_5_2_PLUS_EFFORTS; } diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 6493e482b..972fa5e15 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -4972,7 +4972,7 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", - "name": "MiniMax-M3", + "name": "MiniMax M3", "api": "openai-completions", "provider": "aimlapi", "baseUrl": "https://api.aimlapi.com/v1", @@ -14269,7 +14269,7 @@ "cost": { "input": 0.55, "output": 1.65, - "cacheRead": 0, + "cacheRead": 0.55, "cacheWrite": 0 }, "contextWindow": 161000, @@ -14286,13 +14286,13 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.14, + "output": 0.28, + "cacheRead": 0.07, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 384000, + "maxTokens": 1048576, "thinking": { "mode": "effort", "efforts": [ @@ -14322,13 +14322,13 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.74, + "output": 3.48, + "cacheRead": 0.14, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 393216, + "maxTokens": 1048576, "thinking": { "mode": "effort", "efforts": [ @@ -14349,7 +14349,7 @@ }, "google/gemma-4-31B-it": { "id": "google/gemma-4-31B-it", - "name": "Gemma 4 31B Instruct", + "name": "Gemma 4 31B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14359,13 +14359,13 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.12, + "output": 0.35, + "cacheRead": 0.09, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 131072, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -14388,9 +14388,9 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.05, + "output": 0.1, + "cacheRead": 0.05, "cacheWrite": 0 }, "contextWindow": 131072, @@ -14398,7 +14398,7 @@ }, "JetBrains/Mellum2-12B-A2.5B-Instruct": { "id": "JetBrains/Mellum2-12B-A2.5B-Instruct", - "name": "JetBrains/Mellum2-12B-A2.5B-Instruct", + "name": "Mellum2 12B A2.5B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14407,13 +14407,13 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.05, + "output": 0.1, + "cacheRead": 0.05, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 131072, + "maxTokens": 131072 }, "meta-llama/Llama-3.1-70B-Instruct": { "id": "meta-llama/Llama-3.1-70B-Instruct", @@ -14428,7 +14428,7 @@ "cost": { "input": 0.8, "output": 0.8, - "cacheRead": 0, + "cacheRead": 0.8, "cacheWrite": 0 }, "contextWindow": 128000, @@ -14436,7 +14436,7 @@ }, "meta-llama/Llama-3.1-8B-Instruct": { "id": "meta-llama/Llama-3.1-8B-Instruct", - "name": "Meta-Llama-3.1-8B-Instruct", + "name": "Llama 3.1 8B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14447,7 +14447,7 @@ "cost": { "input": 0.22, "output": 0.22, - "cacheRead": 0, + "cacheRead": 0.22, "cacheWrite": 0 }, "contextWindow": 128000, @@ -14455,7 +14455,7 @@ }, "meta-llama/Llama-3.3-70B-Instruct": { "id": "meta-llama/Llama-3.3-70B-Instruct", - "name": "Llama-3.3-70B-Instruct", + "name": "Llama 3.3 70B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14466,7 +14466,7 @@ "cost": { "input": 0.71, "output": 0.71, - "cacheRead": 0, + "cacheRead": 0.71, "cacheWrite": 0 }, "contextWindow": 128000, @@ -14494,7 +14494,7 @@ }, "microsoft/Phi-4-mini-instruct": { "id": "microsoft/Phi-4-mini-instruct", - "name": "Phi-4-mini-instruct", + "name": "Phi 4 Mini 3.8B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14505,7 +14505,7 @@ "cost": { "input": 0.08, "output": 0.35, - "cacheRead": 0, + "cacheRead": 0.08, "cacheWrite": 0 }, "contextWindow": 128000, @@ -14524,7 +14524,7 @@ "cost": { "input": 0.3, "output": 1.2, - "cacheRead": 0, + "cacheRead": 0.3, "cacheWrite": 0 }, "contextWindow": 196608, @@ -14551,9 +14551,9 @@ "image" ], "cost": { - "input": 0.5, - "output": 2.85, - "cacheRead": 0, + "input": 0.6, + "output": 3, + "cacheRead": 0.1, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14571,7 +14571,7 @@ }, "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", - "name": "Kimi-K2.6", + "name": "Kimi K2.6", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14581,9 +14581,9 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.95, + "output": 4, + "cacheRead": 0.16, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14611,9 +14611,9 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.94, + "output": 4, + "cacheRead": 0.19, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14631,7 +14631,7 @@ }, "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8": { "id": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8", - "name": "NVIDIA Nemotron 3 Super 120B", + "name": "Nemotron 3 Super", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14642,7 +14642,7 @@ "cost": { "input": 0.2, "output": 0.8, - "cacheRead": 0, + "cacheRead": 0.2, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14660,22 +14660,32 @@ }, "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": { "id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B", - "name": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B", + "name": "Nemotron 3 Ultra", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.75, + "output": 2.75, + "cacheRead": 0.15, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", @@ -14688,9 +14698,9 @@ "text" ], "cost": { - "input": 0.15, - "output": 0.6, - "cacheRead": 0, + "input": 0.04, + "output": 0.14, + "cacheRead": 0.04, "cacheWrite": 0 }, "contextWindow": 131072, @@ -14715,9 +14725,9 @@ "text" ], "cost": { - "input": 0.05, - "output": 0.2, - "cacheRead": 0, + "input": 0.03, + "output": 0.13, + "cacheRead": 0.03, "cacheWrite": 0 }, "contextWindow": 131072, @@ -14733,7 +14743,7 @@ }, "OpenPipe/Qwen3-14B-Instruct": { "id": "OpenPipe/Qwen3-14B-Instruct", - "name": "OpenPipe Qwen3 14B Instruct", + "name": "Qwen3 14B Instruct", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14744,7 +14754,7 @@ "cost": { "input": 0.05, "output": 0.22, - "cacheRead": 0, + "cacheRead": 0.05, "cacheWrite": 0 }, "contextWindow": 32768, @@ -14752,7 +14762,7 @@ }, "Qwen/Qwen3-235B-A22B-Instruct-2507": { "id": "Qwen/Qwen3-235B-A22B-Instruct-2507", - "name": "Qwen3 235B A22B Instruct 2507", + "name": "Qwen3 235B A22B-2507", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14763,7 +14773,7 @@ "cost": { "input": 0.1, "output": 0.1, - "cacheRead": 0, + "cacheRead": 0.1, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14771,7 +14781,7 @@ }, "Qwen/Qwen3-235B-A22B-Thinking-2507": { "id": "Qwen/Qwen3-235B-A22B-Thinking-2507", - "name": "Qwen3-235B-A22B-Thinking-2507", + "name": "Qwen3 235B A22B Thinking-2507", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14782,7 +14792,7 @@ "cost": { "input": 0.1, "output": 0.1, - "cacheRead": 0, + "cacheRead": 0.1, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14811,7 +14821,7 @@ "cost": { "input": 0.1, "output": 0.3, - "cacheRead": 0, + "cacheRead": 0.1, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14819,7 +14829,7 @@ }, "Qwen/Qwen3-Coder-480B-A35B-Instruct": { "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct", - "name": "Qwen3-Coder-480B-A35B-Instruct", + "name": "Qwen3 Coder 480B A35B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14830,7 +14840,7 @@ "cost": { "input": 1, "output": 1.5, - "cacheRead": 0, + "cacheRead": 1, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14838,7 +14848,7 @@ }, "Qwen/Qwen3.5-27B": { "id": "Qwen/Qwen3.5-27B", - "name": "Qwen3.5 27B", + "name": "Qwen3.5-27B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14848,13 +14858,13 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.39, + "output": 3.12, + "cacheRead": 0.08, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -14867,7 +14877,7 @@ }, "Qwen/Qwen3.5-35B-A3B": { "id": "Qwen/Qwen3.5-35B-A3B", - "name": "Qwen3.5 35B-A3B", + "name": "Qwen3.5-35B-A3B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14877,13 +14887,13 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.25, + "output": 1.25, + "cacheRead": 0.25, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -14906,13 +14916,13 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.6, + "output": 3.6, + "cacheRead": 0.12, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -14925,7 +14935,7 @@ }, "Qwen/Qwen3.6-35B-A3B": { "id": "Qwen/Qwen3.6-35B-A3B", - "name": "Qwen3.6 35B-A3B", + "name": "Qwen3.6 35B A3B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14935,13 +14945,13 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.25, + "output": 1.25, + "cacheRead": 0.25, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -14973,7 +14983,7 @@ }, "zai-org/GLM-5.1": { "id": "zai-org/GLM-5.1", - "name": "GLM-5.1", + "name": "GLM 5.1", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14987,8 +14997,8 @@ "cacheRead": 0.26, "cacheWrite": 0 }, - "contextWindow": 200000, - "maxTokens": 131072, + "contextWindow": 202752, + "maxTokens": 202752, "thinking": { "mode": "effort", "efforts": [ @@ -15002,7 +15012,7 @@ }, "zai-org/GLM-5.2": { "id": "zai-org/GLM-5.2", - "name": "GLM-5.2", + "name": "GLM 5.2", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -15011,13 +15021,13 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.39, + "output": 4.4, + "cacheRead": 0.26, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 164000, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -16764,26 +16774,6 @@ }, "requestModelId": "MODEL_GOOGLE_GEMINI_3_0_FLASH_MINIMAL" }, - "glm-5-1": { - "id": "glm-5-1", - "name": "GLM-5.1", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000 - }, "glm-5-2": { "id": "glm-5-2", "name": "GLM-5.2 High", @@ -17224,6 +17214,309 @@ }, "requestModelId": "gpt-5-5-none-priority" }, + "gpt-5-6-luna": { + "id": "gpt-5-6-luna", + "name": "GPT-5.6 Luna", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortRouting": { + "off": "gpt-5-6-luna-none", + "minimal": "gpt-5-6-luna-low", + "low": "gpt-5-6-luna-medium", + "medium": "gpt-5-6-luna-high", + "high": "gpt-5-6-luna-xhigh", + "xhigh": "gpt-5-6-luna-max" + } + }, + "requestModelId": "gpt-5-6-luna-none" + }, + "gpt-5-6-luna-fast": { + "id": "gpt-5-6-luna-fast", + "name": "GPT-5.6 Luna Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortRouting": { + "off": "gpt-5-6-luna-none-priority", + "minimal": "gpt-5-6-luna-low-priority", + "low": "gpt-5-6-luna-low-priority", + "medium": "gpt-5-6-luna-medium-priority", + "high": "gpt-5-6-luna-high-priority", + "xhigh": "gpt-5-6-luna-xhigh-priority" + } + }, + "requestModelId": "gpt-5-6-luna-none-priority" + }, + "gpt-5-6-sol": { + "id": "gpt-5-6-sol", + "name": "GPT-5.6 Sol", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortRouting": { + "off": "gpt-5-6-sol-none", + "minimal": "gpt-5-6-sol-low", + "low": "gpt-5-6-sol-medium", + "medium": "gpt-5-6-sol-high", + "high": "gpt-5-6-sol-xhigh", + "xhigh": "gpt-5-6-sol-max" + } + }, + "requestModelId": "gpt-5-6-sol-none" + }, + "gpt-5-6-sol-fast": { + "id": "gpt-5-6-sol-fast", + "name": "GPT-5.6 Sol Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortRouting": { + "off": "gpt-5-6-sol-none-priority", + "minimal": "gpt-5-6-sol-low-priority", + "low": "gpt-5-6-sol-low-priority", + "medium": "gpt-5-6-sol-medium-priority", + "high": "gpt-5-6-sol-high-priority", + "xhigh": "gpt-5-6-sol-xhigh-priority" + } + }, + "requestModelId": "gpt-5-6-sol-none-priority" + }, + "gpt-5-6-terra": { + "id": "gpt-5-6-terra", + "name": "GPT-5.6 Terra", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortRouting": { + "off": "gpt-5-6-terra-none", + "minimal": "gpt-5-6-terra-low", + "low": "gpt-5-6-terra-medium", + "medium": "gpt-5-6-terra-high", + "high": "gpt-5-6-terra-xhigh", + "xhigh": "gpt-5-6-terra-max" + } + }, + "requestModelId": "gpt-5-6-terra-none" + }, + "gpt-5-6-terra-fast": { + "id": "gpt-5-6-terra-fast", + "name": "GPT-5.6 Terra Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortRouting": { + "off": "gpt-5-6-terra-none-priority", + "minimal": "gpt-5-6-terra-low-priority", + "low": "gpt-5-6-terra-low-priority", + "medium": "gpt-5-6-terra-medium-priority", + "high": "gpt-5-6-terra-high-priority", + "xhigh": "gpt-5-6-terra-xhigh-priority" + } + }, + "requestModelId": "gpt-5-6-terra-none-priority" + }, + "grok-4-5-high": { + "id": "grok-4-5-high", + "name": "Grok 4.5 High", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 64000 + }, + "grok-4-5-low": { + "id": "grok-4-5-low", + "name": "Grok 4.5 Low", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 64000 + }, + "grok-4-5-medium": { + "id": "grok-4-5-medium", + "name": "Grok 4.5 Medium", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 64000 + }, "kimi-k2-6": { "id": "kimi-k2-6", "name": "Kimi K2.6", @@ -17454,6 +17747,46 @@ }, "contextWindow": 200000, "maxTokens": 64000 + }, + "swe-1-7": { + "id": "swe-1-7", + "name": "SWE-1.7", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262000, + "maxTokens": 64000 + }, + "swe-1-7-lightning": { + "id": "swe-1-7-lightning", + "name": "SWE-1.7 Lightning", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 202752, + "maxTokens": 64000 } }, "firepass": { @@ -23447,6 +23780,33 @@ ] } }, + "openai/gpt-oss-20b": { + "id": "openai/gpt-oss-20b", + "name": "GPT OSS 20B", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.1, + "output": 0.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, "Qwen/Qwen3-235B-A22B": { "id": "Qwen/Qwen3-235B-A22B", "name": "Qwen3 235B-A22B", @@ -24542,6 +24902,44 @@ "contextWindow": null, "maxTokens": null }, + "aion-labs/aion-3.0": { + "id": "aion-labs/aion-3.0", + "name": "Aion-3.0", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, + "aion-labs/aion-3.0-mini": { + "id": "aion-labs/aion-3.0-mini", + "name": "Aion-3.0-Mini", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "aion-labs/aion-rp-llama-3.1-8b": { "id": "aion-labs/aion-rp-llama-3.1-8b", "name": "Aion-RP 1.0 (8B)", @@ -25931,6 +26329,25 @@ "contextWindow": null, "maxTokens": null }, + "cognitivecomputations/dolphin-mistral-24b-venice-edition": { + "id": "cognitivecomputations/dolphin-mistral-24b-venice-edition", + "name": "Uncensored", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "cohere/command-a": { "id": "cohere/command-a", "name": "Command A", @@ -28532,7 +28949,7 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", - "name": "MiniMax-M3", + "name": "MiniMax M3", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -28542,13 +28959,13 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.3, + "output": 1.2, + "cacheRead": 0.06, "cacheWrite": 0 }, - "contextWindow": 512000, - "maxTokens": 128000, + "contextWindow": 1048576, + "maxTokens": 512000, "thinking": { "mode": "effort", "efforts": [ @@ -29470,6 +29887,25 @@ "contextWindow": 131072, "maxTokens": 163840 }, + "nex-agi/nex-n2-mini": { + "id": "nex-agi/nex-n2-mini", + "name": "Nex-N2-Mini", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144 + }, "nex-agi/nex-n2-pro": { "id": "nex-agi/nex-n2-pro", "name": "Nex-N2-Pro", @@ -29486,8 +29922,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 262144, + "maxTokens": 262144 }, "nex-agi/nex-n2-pro:free": { "id": "nex-agi/nex-n2-pro:free", @@ -29505,8 +29941,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 262144, + "maxTokens": 262144 }, "nousresearch/hermes-2-pro-llama-3-8b": { "id": "nousresearch/hermes-2-pro-llama-3-8b", @@ -31048,6 +31484,120 @@ }, "contextPromotionTarget": "kilo/openai/gpt-5.4" }, + "openai/gpt-5.6-luna": { + "id": "openai/gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 372000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-luna-pro": { + "id": "openai/gpt-5.6-luna-pro", + "name": "GPT-5.6 Luna Pro", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-sol": { + "id": "openai/gpt-5.6-sol", + "name": "GPT-5.6 Sol (new)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 372000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-sol-pro": { + "id": "openai/gpt-5.6-sol-pro", + "name": "GPT-5.6 Sol Pro", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-terra": { + "id": "openai/gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 372000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-terra-pro": { + "id": "openai/gpt-5.6-terra-pro", + "name": "GPT-5.6 Terra Pro", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, "openai/gpt-audio": { "id": "openai/gpt-audio", "name": "GPT Audio", @@ -31820,7 +32370,7 @@ "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -31831,7 +32381,17 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768 + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "poolside/laguna-m.1:free": { "id": "poolside/laguna-m.1:free", @@ -31906,7 +32466,7 @@ "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -31917,7 +32477,17 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768 + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "poolside/laguna-xs.2:free": { "id": "poolside/laguna-xs.2:free", @@ -33859,6 +34429,25 @@ "contextWindow": null, "maxTokens": null }, + "tencent/hy3": { + "id": "tencent/hy3", + "name": "Hy3", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": null + }, "tencent/hy3-preview": { "id": "tencent/hy3-preview", "name": "Hy3 Preview", @@ -33907,6 +34496,25 @@ "contextWindow": 262144, "maxTokens": 64000 }, + "tencent/hy3:free": { + "id": "tencent/hy3:free", + "name": "Hy3 (free)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": null + }, "thedrummer/cydonia-24b-v4.1": { "id": "thedrummer/cydonia-24b-v4.1", "name": "Cydonia 24B V4.1", @@ -37363,6 +37971,44 @@ "contextWindow": null, "maxTokens": null }, + "aion-labs/aion-3.0": { + "id": "aion-labs/aion-3.0", + "name": "aion-labs/aion-3.0", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, + "aion-labs/aion-3.0-mini": { + "id": "aion-labs/aion-3.0-mini", + "name": "aion-labs/aion-3.0-mini", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "aion-labs/aion-rp-llama-3.1-8b": { "id": "aion-labs/aion-rp-llama-3.1-8b", "name": "aion-labs/aion-rp-llama-3.1-8b", @@ -44952,6 +45598,25 @@ "contextWindow": null, "maxTokens": null }, + "mellum2-12b-a2-5b-instruct": { + "id": "mellum2-12b-a2-5b-instruct", + "name": "mellum2-12b-a2-5b-instruct", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "mercury-2": { "id": "mercury-2", "name": "Mercury 2", @@ -45116,6 +45781,25 @@ "contextWindow": 328000, "maxTokens": 65536 }, + "meta/muse-spark-1.1": { + "id": "meta/muse-spark-1.1", + "name": "meta/muse-spark-1.1", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 1048576 + }, "microsoft/MAI-DS-R1-FP8": { "id": "microsoft/MAI-DS-R1-FP8", "name": "microsoft/MAI-DS-R1-FP8", @@ -46617,6 +47301,25 @@ "contextWindow": 131072, "maxTokens": 163840 }, + "nex-agi/nex-n2-mini": { + "id": "nex-agi/nex-n2-mini", + "name": "nex-agi/nex-n2-mini", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144 + }, "nex-agi/nex-n2-pro": { "id": "nex-agi/nex-n2-pro", "name": "nex-agi/nex-n2-pro", @@ -46633,8 +47336,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 262144, + "maxTokens": 262144 }, "nothingiisreal/L3.1-70B-Celeste-V0.1-BF16": { "id": "nothingiisreal/L3.1-70B-Celeste-V0.1-BF16", @@ -47915,6 +48618,174 @@ }, "contextPromotionTarget": "nanogpt/openai/gpt-5.4" }, + "openai/gpt-5.6-luna": { + "id": "openai/gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.09999999999999999, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-luna-pro": { + "id": "openai/gpt-5.6-luna-pro", + "name": "GPT-5.6 Luna Pro", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-sol": { + "id": "openai/gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-sol-pro": { + "id": "openai/gpt-5.6-sol-pro", + "name": "GPT-5.6 Sol Pro", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-terra": { + "id": "openai/gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-terra-pro": { + "id": "openai/gpt-5.6-terra-pro", + "name": "GPT-5.6 Terra Pro", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, "openai/gpt-chat-latest": { "id": "openai/gpt-chat-latest", "name": "GPT Chat Latest", @@ -48523,41 +49394,61 @@ }, "poolside/laguna-m.1": { "id": "poolside/laguna-m.1", - "name": "poolside/laguna-m.1", + "name": "Laguna M.1", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], "cost": { - "input": 0, - "output": 0, + "input": 0.2, + "output": 0.4, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768 + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "poolside/laguna-xs.2": { "id": "poolside/laguna-xs.2", - "name": "poolside/laguna-xs.2", + "name": "Laguna XS.2", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], "cost": { - "input": 0, - "output": 0, + "input": 0.2, + "output": 0.4, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768 + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "qvq-max": { "id": "qvq-max", @@ -48807,7 +49698,7 @@ }, "Qwen/Qwen3-235B-A22B-Instruct-2507": { "id": "Qwen/Qwen3-235B-A22B-Instruct-2507", - "name": "Qwen3 235B A22B Instruct 2507", + "name": "Qwen3 235B A22B-2507", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", @@ -48864,7 +49755,7 @@ }, "Qwen/Qwen3-235B-A22B-Thinking-2507": { "id": "Qwen/Qwen3-235B-A22B-Thinking-2507", - "name": "Qwen3-235B-A22B-Thinking-2507", + "name": "Qwen3 235B A22B Thinking-2507", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", @@ -49250,7 +50141,7 @@ }, "Qwen/Qwen3.6-35B-A3B": { "id": "Qwen/Qwen3.6-35B-A3B", - "name": "Qwen3.6 35B-A3B", + "name": "Qwen3.6 35B A3B", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", @@ -52792,6 +53683,25 @@ ] } }, + "x-ai/grok-4.5": { + "id": "x-ai/grok-4.5", + "name": "x-ai/grok-4.5", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000 + }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", "name": "Grok Build 0.1", @@ -53901,6 +54811,2849 @@ } } }, + "novita": { + "baichuan/baichuan-m2-32b": { + "id": "baichuan/baichuan-m2-32b", + "name": "BaiChuan M2 32B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.07, + "output": 0.07, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 131072, + "supportsTools": false + }, + "baidu/cobuddy": { + "id": "baidu/cobuddy", + "name": "CoBuddy", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.28, + "output": 1.13, + "cacheRead": 0.07, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "baidu/ernie-4.5-21B-a3b": { + "id": "baidu/ernie-4.5-21B-a3b", + "name": "ERNIE 4.5 21B A3B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.07, + "output": 0.28, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 120000, + "maxTokens": 8000, + "supportsTools": true + }, + "baidu/ernie-4.5-vl-424b-a47b": { + "id": "baidu/ernie-4.5-vl-424b-a47b", + "name": "ERNIE 4.5 VL 424B A47B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.42, + "output": 1.25, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 123000, + "maxTokens": 16000, + "supportsTools": false, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "bunny": { + "id": "bunny", + "name": "Bunny", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "supportsTools": true + }, + "deepseek/deepseek_v3": { + "id": "deepseek/deepseek_v3", + "name": "DeepSeek V3", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.89, + "output": 0.89, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 16000, + "supportsTools": true + }, + "deepseek/deepseek-ocr": { + "id": "deepseek/deepseek-ocr", + "name": "DeepSeek-OCR", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.03, + "output": 0.03, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": false + }, + "deepseek/deepseek-ocr-2": { + "id": "deepseek/deepseek-ocr-2", + "name": "DeepSeek-OCR 2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.03, + "output": 0.03, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": false + }, + "deepseek/deepseek-r1": { + "id": "deepseek/deepseek-r1", + "name": "R1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 4, + "output": 4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 16000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-r1-0528": { + "id": "deepseek/deepseek-r1-0528", + "name": "R1 0528", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.7, + "output": 2.5, + "cacheRead": 0.35, + "cacheWrite": 0 + }, + "contextWindow": 163840, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-r1-0528-qwen3-8b": { + "id": "deepseek/deepseek-r1-0528-qwen3-8b", + "name": "DeepSeek R1 0528 Qwen3 8B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.06, + "output": 0.09, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 32000, + "supportsTools": false + }, + "deepseek/deepseek-r1-distill-llama-70b": { + "id": "deepseek/deepseek-r1-distill-llama-70b", + "name": "DeepSeek R1 Distill LLama 70B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.8, + "output": 0.8, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": false, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-r1-turbo": { + "id": "deepseek/deepseek-r1-turbo", + "name": "DeepSeek R1 (Turbo)", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.7, + "output": 2.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 16000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-r1/community": { + "id": "deepseek/deepseek-r1/community", + "name": "DeepSeek R1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 4, + "output": 4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 8000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v3-0324": { + "id": "deepseek/deepseek-v3-0324", + "name": "DeepSeek V3 0324", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.27, + "output": 1.12, + "cacheRead": 0.135, + "cacheWrite": 0 + }, + "contextWindow": 163840, + "maxTokens": 65536, + "supportsTools": true + }, + "deepseek/deepseek-v3-turbo": { + "id": "deepseek/deepseek-v3-turbo", + "name": "DeepSeek V3 (Turbo)", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.4, + "output": 1.3, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 16000, + "supportsTools": true + }, + "deepseek/deepseek-v3.1": { + "id": "deepseek/deepseek-v3.1", + "name": "DeepSeek V3.1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.27, + "output": 1, + "cacheRead": 0.135, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v3.1-terminus": { + "id": "deepseek/deepseek-v3.1-terminus", + "name": "DeepSeek V3.1 Terminus", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.27, + "output": 1, + "cacheRead": 0.135, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v3.2": { + "id": "deepseek/deepseek-v3.2", + "name": "DeepSeek V3.2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.269, + "output": 0.4, + "cacheRead": 0.1345, + "cacheWrite": 0 + }, + "contextWindow": 163840, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v3.2-exp": { + "id": "deepseek/deepseek-v3.2-exp", + "name": "DeepSeek V3.2 Exp", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.27, + "output": 0.41, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 163840, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v3/community": { + "id": "deepseek/deepseek-v3/community", + "name": "DeepSeek V3", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.89, + "output": 0.89, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 64000, + "maxTokens": 8000, + "supportsTools": true + }, + "deepseek/deepseek-v4-flash": { + "id": "deepseek/deepseek-v4-flash", + "name": "DeepSeek V4 Flash", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.14, + "output": 0.28, + "cacheRead": 0.028, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 393216, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "deepseek/deepseek-v4-pro": { + "id": "deepseek/deepseek-v4-pro", + "name": "DeepSeek V4 Pro", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.6, + "output": 3.2, + "cacheRead": 0.135, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 393216, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "dev/glm46": { + "id": "dev/glm46", + "name": "dev/glm46", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 256000, + "supportsTools": true + }, + "google/gemma-3-12b-it": { + "id": "google/gemma-3-12b-it", + "name": "Gemma3 12B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.05, + "output": 0.1, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 8192, + "supportsTools": false + }, + "google/gemma-3-27b-it": { + "id": "google/gemma-3-27b-it", + "name": "Gemma 3 27B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.119, + "output": 0.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 98304, + "maxTokens": 16384, + "supportsTools": false + }, + "google/gemma-4-26b-a4b-it": { + "id": "google/gemma-4-26b-a4b-it", + "name": "Gemma 4 26B A4B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.13, + "output": 0.4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "google/gemma-4-31b-it": { + "id": "google/gemma-4-31b-it", + "name": "Gemma 4 31B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.14, + "output": 0.4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gryphe/mythomax-l2-13b": { + "id": "gryphe/mythomax-l2-13b", + "name": "Mythomax L2 13B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.09, + "output": 0.09, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 4096, + "maxTokens": 3200, + "supportsTools": false + }, + "gt-4p": { + "id": "gt-4p", + "name": "gt-4p", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": 131072, + "supportsTools": true + }, + "inclusionai/ling-2.6-1t": { + "id": "inclusionai/ling-2.6-1t", + "name": "Ling-2.6-1T", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 2.5, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "supportsTools": true + }, + "inclusionai/ling-2.6-flash": { + "id": "inclusionai/ling-2.6-flash", + "name": "Ling-2.6 Flash", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.1, + "output": 0.3, + "cacheRead": 0.02, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "supportsTools": true + }, + "inclusionai/ring-2.6-1t": { + "id": "inclusionai/ring-2.6-1t", + "name": "Ring-2.6-1T", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 2.5, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "kwaipilot/kat-coder-pro": { + "id": "kwaipilot/kat-coder-pro", + "name": "Kat Coder Pro", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 128000, + "supportsTools": true + }, + "meta-llama/llama-3.1-8b-instruct": { + "id": "meta-llama/llama-3.1-8b-instruct", + "name": "Llama 3.1 8B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.02, + "output": 0.05, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 16384, + "maxTokens": 16384, + "supportsTools": false + }, + "meta-llama/llama-3.2-1b-instruct": { + "id": "meta-llama/llama-3.2-1b-instruct", + "name": "Llama 3.2 1B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.02, + "output": 0.02, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131000, + "maxTokens": 32000, + "supportsTools": false + }, + "meta-llama/llama-3.2-3b-instruct": { + "id": "meta-llama/llama-3.2-3b-instruct", + "name": "Llama 3.2 3B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.03, + "output": 0.05, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 32768, + "maxTokens": 32000, + "supportsTools": false + }, + "meta-llama/llama-3.3-70b-instruct": { + "id": "meta-llama/llama-3.3-70b-instruct", + "name": "Llama 3.3 70B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.135, + "output": 0.4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 6000, + "maxTokens": 120000, + "supportsTools": true + }, + "meta-llama/llama-4-maverick-17b-128e-instruct-fp8": { + "id": "meta-llama/llama-4-maverick-17b-128e-instruct-fp8", + "name": "Llama 4 Maverick Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.27, + "output": 0.85, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 8192, + "supportsTools": false + }, + "meta-llama/llama-4-scout-17b-16e-instruct": { + "id": "meta-llama/llama-4-scout-17b-16e-instruct", + "name": "Llama 4 Scout 17B 16E", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.18, + "output": 0.59, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 131072, + "supportsTools": false + }, + "microsoft/wizardlm-2-8x22b": { + "id": "microsoft/wizardlm-2-8x22b", + "name": "Wizardlm 2 8x22B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.62, + "output": 0.62, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 65535, + "maxTokens": 8000, + "supportsTools": false + }, + "minimax/minimax-m2": { + "id": "minimax/minimax-m2", + "name": "MiniMax M2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m2.1": { + "id": "minimax/minimax-m2.1", + "name": "MiniMax M2.1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m2.5": { + "id": "minimax/minimax-m2.5", + "name": "MiniMax M2.5", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131100, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m2.5-highspeed": { + "id": "minimax/minimax-m2.5-highspeed", + "name": "MiniMax M2.5-highspeed", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.4, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131100, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m2.7": { + "id": "minimax/minimax-m2.7", + "name": "MiniMax M2.7", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m2.7-highspeed": { + "id": "minimax/minimax-m2.7-highspeed", + "name": "MiniMax M2.7 highspeed", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.4, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "minimax/minimax-m3": { + "id": "minimax/minimax-m3", + "name": "MiniMax M3", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "minimaxai/minimax-m1-80k": { + "id": "minimaxai/minimax-m1-80k", + "name": "MiniMax M1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.55, + "output": 2.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 40000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "mistralai/mistral-nemo": { + "id": "mistralai/mistral-nemo", + "name": "Mistral Nemo", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.04, + "output": 0.17, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 60288, + "maxTokens": 16000, + "supportsTools": false + }, + "moonshotai/kimi-k2-0905": { + "id": "moonshotai/kimi-k2-0905", + "name": "Kimi K2 0905", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 100352, + "supportsTools": true + }, + "moonshotai/kimi-k2-instruct": { + "id": "moonshotai/kimi-k2-instruct", + "name": "Kimi K2 Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.57, + "output": 2.3, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 100352, + "supportsTools": true + }, + "moonshotai/kimi-k2-thinking": { + "id": "moonshotai/kimi-k2-thinking", + "name": "Kimi K2 Thinking", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.5, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 100352, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "requiresEffort": true + } + }, + "moonshotai/kimi-k2.5": { + "id": "moonshotai/kimi-k2.5", + "name": "Kimi K2.5", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.6, + "output": 3, + "cacheRead": 0.1, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "moonshotai/kimi-k2.6": { + "id": "moonshotai/kimi-k2.6", + "name": "Kimi K2.6", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.8, + "output": 3.4, + "cacheRead": 0.16, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "moonshotai/kimi-k2.7-code": { + "id": "moonshotai/kimi-k2.7-code", + "name": "Kimi K2.7 Code", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.95, + "output": 4, + "cacheRead": 0.19, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "nousresearch/hermes-2-pro-llama-3-8b": { + "id": "nousresearch/hermes-2-pro-llama-3-8b", + "name": "Hermes 2 Pro Llama 3 8B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.14, + "output": 0.14, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": false + }, + "nvidia/nemotron-3-nano-30b-a3b": { + "id": "nvidia/nemotron-3-nano-30b-a3b", + "name": "Nemotron 3 Nano 30B A3B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.05, + "output": 0.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai/gpt-oss-120b": { + "id": "openai/gpt-oss-120b", + "name": "GPT OSS 120B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.05, + "output": 0.25, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, + "openai/gpt-oss-20b": { + "id": "openai/gpt-oss-20b", + "name": "GPT OSS 20B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.04, + "output": 0.15, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": false, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, + "paddlepaddle/paddleocr-vl": { + "id": "paddlepaddle/paddleocr-vl", + "name": "PaddleOCR-VL", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.02, + "output": 0.02, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 16384, + "maxTokens": 16384, + "supportsTools": false + }, + "qwen/qwen-2.5-72b-instruct": { + "id": "qwen/qwen-2.5-72b-instruct", + "name": "Qwen2.5 72B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.38, + "output": 0.4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 32000, + "maxTokens": 8192, + "supportsTools": true + }, + "qwen/qwen-mt-plus": { + "id": "qwen/qwen-mt-plus", + "name": "Qwen MT Plus", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.25, + "output": 0.75, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 16384, + "maxTokens": 8192, + "supportsTools": false + }, + "qwen/qwen3-235b-a22b-fp8": { + "id": "qwen/qwen3-235b-a22b-fp8", + "name": "Qwen3 235B A22B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.2, + "output": 0.8, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 40960, + "maxTokens": 20000, + "supportsTools": false, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3-235b-a22b-instruct-2507": { + "id": "qwen/qwen3-235b-a22b-instruct-2507", + "name": "Qwen3 235B A22B Instruct 2507", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.09, + "output": 0.58, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 16384, + "supportsTools": true + }, + "qwen/qwen3-235b-a22b-thinking-2507": { + "id": "qwen/qwen3-235b-a22b-thinking-2507", + "name": "Qwen3 235B A22B Thinking 2507", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 3, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "qwen/qwen3-coder-30b-a3b-instruct": { + "id": "qwen/qwen3-coder-30b-a3b-instruct", + "name": "Qwen3 Coder 30B A3B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.07, + "output": 0.27, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 160000, + "maxTokens": 32768, + "supportsTools": true + }, + "qwen/qwen3-coder-480b-a35b-instruct": { + "id": "qwen/qwen3-coder-480b-a35b-instruct", + "name": "Qwen3 Coder 480B A35B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.38, + "output": 1.55, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true + }, + "qwen/qwen3-coder-next": { + "id": "qwen/qwen3-coder-next", + "name": "Qwen3 Coder Next", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.2, + "output": 1.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true + }, + "qwen/qwen3-max": { + "id": "qwen/qwen3-max", + "name": "Qwen3 Max", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 2.11, + "output": 8.45, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true + }, + "qwen/qwen3-next-80b-a3b-instruct": { + "id": "qwen/qwen3-next-80b-a3b-instruct", + "name": "Qwen3-Next-80B-A3B-Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.15, + "output": 1.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true + }, + "qwen/qwen3-omni-30b-a3b-instruct": { + "id": "qwen/qwen3-omni-30b-a3b-instruct", + "name": "Qwen3 Omni 30B A3B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 0.97, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 65536, + "maxTokens": 16384, + "supportsTools": true + }, + "qwen/qwen3-omni-30b-a3b-thinking": { + "id": "qwen/qwen3-omni-30b-a3b-thinking", + "name": "Qwen3 Omni 30B A3B Thinking", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 0.97, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 65536, + "maxTokens": 16384, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "qwen/qwen3-vl-235b-a22b-instruct": { + "id": "qwen/qwen3-vl-235b-a22b-instruct", + "name": "Qwen3 VL 235B A22B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 1.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true + }, + "qwen/qwen3-vl-235b-a22b-thinking": { + "id": "qwen/qwen3-vl-235b-a22b-thinking", + "name": "Qwen3 VL 235B A22B Thinking", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.98, + "output": 3.95, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "qwen/qwen3-vl-30b-a3b-instruct": { + "id": "qwen/qwen3-vl-30b-a3b-instruct", + "name": "Qwen3 VL 30B A3B Instruct", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.2, + "output": 0.7, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true + }, + "qwen/qwen3.5-122b-a10b": { + "id": "qwen/qwen3.5-122b-a10b", + "name": "Qwen3.5 122B-A10B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.4, + "output": 3.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.5-27b": { + "id": "qwen/qwen3.5-27b", + "name": "Qwen3.5-27B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 2.4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.5-35b-a3b": { + "id": "qwen/qwen3.5-35b-a3b", + "name": "Qwen3.5-35B-A3B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.5-397b-a17b": { + "id": "qwen/qwen3.5-397b-a17b", + "name": "Qwen3.5 397B A17B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.6, + "output": 3.6, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.5-plus": { + "id": "qwen/qwen3.5-plus", + "name": "Qwen3.5 Plus", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.6-27b": { + "id": "qwen/qwen3.6-27b", + "name": "Qwen3.6 27B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.6, + "output": 3.6, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.6-35b-a3b": { + "id": "qwen/qwen3.6-35b-a3b", + "name": "Qwen3.6-35B-A3B", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.248, + "output": 1.485, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.6-plus": { + "id": "qwen/qwen3.6-plus", + "name": "Qwen3.6 Plus", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen/qwen3.7-max": { + "id": "qwen/qwen3.7-max", + "name": "Qwen3.7 Max", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.25, + "output": 3.75, + "cacheRead": 0.25, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "sao10k/l3-70b-euryale-v2.1": { + "id": "sao10k/l3-70b-euryale-v2.1", + "name": "L3 70B Euryale V2.1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 1.48, + "output": 1.48, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": true + }, + "sao10k/l3-8b-lunaris": { + "id": "sao10k/l3-8b-lunaris", + "name": "Sao10k L3 8B Lunaris", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.05, + "output": 0.05, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": false + }, + "Sao10K/L3-8B-Stheno-v3.2": { + "id": "Sao10K/L3-8B-Stheno-v3.2", + "name": "L3 8B Stheno V3.2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.05, + "output": 0.05, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 32000, + "supportsTools": true + }, + "sao10k/l31-70b-euryale-v2.2": { + "id": "sao10k/l31-70b-euryale-v2.2", + "name": "L31 70B Euryale V2.2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.48, + "output": 1.48, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "stepfun/step-3.7-flash": { + "id": "stepfun/step-3.7-flash", + "name": "Step 3.7 Flash", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.2, + "output": 1.15, + "cacheRead": 0.04, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 256000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "tencent/hy3": { + "id": "tencent/hy3", + "name": "Hy3", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "thudm/glm-4-32b-0414": { + "id": "thudm/glm-4-32b-0414", + "name": "GLM-4-32B-0414", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.55, + "output": 1.66, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 32000, + "maxTokens": 32000, + "supportsTools": true + }, + "xiaomimimo/mimo-v2.5": { + "id": "xiaomimimo/mimo-v2.5", + "name": "XiaomiMiMo/MiMo-V2.5", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.168, + "output": 0.336, + "cacheRead": 0.0034, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, + "xiaomimimo/mimo-v2.5-pro": { + "id": "xiaomimimo/mimo-v2.5-pro", + "name": "XiaomiMiMo/MiMo-V2.5-Pro", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.522, + "output": 1.044, + "cacheRead": 0.0043, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, + "zai-org/autoglm-phone-9b-multilingual": { + "id": "zai-org/autoglm-phone-9b-multilingual", + "name": "AutoGLM-Phone-9B-Multilingual", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.035, + "output": 0.138, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 65536, + "maxTokens": 65536, + "supportsTools": false + }, + "zai-org/glm-4.5-air": { + "id": "zai-org/glm-4.5-air", + "name": "zai-org/glm-4.5-air", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.13, + "output": 0.85, + "cacheRead": 0.025, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 98304, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.5v": { + "id": "zai-org/glm-4.5v", + "name": "GLM 4.5V", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.6, + "output": 1.8, + "cacheRead": 0.11, + "cacheWrite": 0 + }, + "contextWindow": 65536, + "maxTokens": 16384, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.6": { + "id": "zai-org/glm-4.6", + "name": "GLM 4.6", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.55, + "output": 2.2, + "cacheRead": 0.11, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.6v": { + "id": "zai-org/glm-4.6v", + "name": "GLM 4.6V", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 0.9, + "cacheRead": 0.055, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.7": { + "id": "zai-org/glm-4.7", + "name": "GLM 4.7", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.2, + "cacheRead": 0.11, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.7-flash": { + "id": "zai-org/glm-4.7-flash", + "name": "GLM 4.7 Flash", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.07, + "output": 0.4, + "cacheRead": 0.01, + "cacheWrite": 0 + }, + "contextWindow": 200000, + "maxTokens": 128000, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-4.7-h": { + "id": "zai-org/glm-4.7-h", + "name": "GLM-4.7", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.2, + "cacheRead": 0.11, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-5": { + "id": "zai-org/glm-5", + "name": "GLM 5", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1, + "output": 3.2, + "cacheRead": 0.2, + "cacheWrite": 0 + }, + "contextWindow": 202800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-5-turbo": { + "id": "zai-org/glm-5-turbo", + "name": "GLM-5-Turbo", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.2, + "output": 4, + "cacheRead": 0.24, + "cacheWrite": 0 + }, + "contextWindow": 202800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-5.1": { + "id": "zai-org/glm-5.1", + "name": "GLM 5.1", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.38, + "output": 4.4, + "cacheRead": 0.26, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/glm-5.2": { + "id": "zai-org/glm-5.2", + "name": "GLM 5.2", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.4, + "output": 4.4, + "cacheRead": 0.26, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "xhigh": "max" + } + } + }, + "zai-org/glm-5v-turbo": { + "id": "zai-org/glm-5v-turbo", + "name": "GLM-5V-Turbo", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.2, + "output": 4, + "cacheRead": 0.24, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + } + }, "nvidia": { "01-ai/yi-large": { "id": "01-ai/yi-large", @@ -59495,6 +63248,278 @@ }, "contextPromotionTarget": "openai/gpt-5.4" }, + "gpt-5.6": { + "id": "gpt-5.6", + "name": "GPT-5.6", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "gpt-5.6-luna": { + "id": "gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 1.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "gpt-5.6-luna-pro": { + "id": "gpt-5.6-luna-pro", + "name": "GPT-5.6 Luna Pro", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 1.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "requestModelId": "gpt-5.6-luna", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "gpt-5.6-sol": { + "id": "gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "gpt-5.6-sol-pro": { + "id": "gpt-5.6-sol-pro", + "name": "GPT-5.6 Sol Pro", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "requestModelId": "gpt-5.6-sol", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "gpt-5.6-terra": { + "id": "gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "gpt-5.6-terra-pro": { + "id": "gpt-5.6-terra-pro", + "name": "GPT-5.6 Terra Pro", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "requestModelId": "gpt-5.6-terra", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "o1": { "id": "o1", "name": "o1", @@ -60273,6 +64298,288 @@ ] }, "contextPromotionTarget": "openai-codex/gpt-5.4" + }, + "gpt-5.6-luna": { + "id": "gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "openai-codex-responses", + "provider": "openai-codex", + "baseUrl": "https://chatgpt.com/backend-api", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 1.25 + }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, + "contextWindow": 372000, + "maxTokens": 128000, + "preferWebsockets": true, + "useResponsesLite": true, + "priority": 3, + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "gpt-5.6-luna-pro": { + "id": "gpt-5.6-luna-pro", + "name": "GPT-5.6 Luna Pro", + "api": "openai-codex-responses", + "provider": "openai-codex", + "baseUrl": "https://chatgpt.com/backend-api", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 1.25 + }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, + "contextWindow": 372000, + "maxTokens": 128000, + "preferWebsockets": true, + "useResponsesLite": true, + "priority": 3, + "requestModelId": "gpt-5.6-luna", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "gpt-5.6-sol": { + "id": "gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "openai-codex-responses", + "provider": "openai-codex", + "baseUrl": "https://chatgpt.com/backend-api", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, + "contextWindow": 372000, + "maxTokens": 128000, + "preferWebsockets": true, + "useResponsesLite": true, + "priority": 1, + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "gpt-5.6-sol-pro": { + "id": "gpt-5.6-sol-pro", + "name": "GPT-5.6 Sol Pro", + "api": "openai-codex-responses", + "provider": "openai-codex", + "baseUrl": "https://chatgpt.com/backend-api", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, + "contextWindow": 372000, + "maxTokens": 128000, + "preferWebsockets": true, + "useResponsesLite": true, + "priority": 1, + "requestModelId": "gpt-5.6-sol", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "gpt-5.6-terra": { + "id": "gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "openai-codex-responses", + "provider": "openai-codex", + "baseUrl": "https://chatgpt.com/backend-api", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, + "contextWindow": 372000, + "maxTokens": 128000, + "preferWebsockets": true, + "useResponsesLite": true, + "priority": 2, + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "gpt-5.6-terra-pro": { + "id": "gpt-5.6-terra-pro", + "name": "GPT-5.6 Terra Pro", + "api": "openai-codex-responses", + "provider": "openai-codex", + "baseUrl": "https://chatgpt.com/backend-api", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, + "contextWindow": 372000, + "maxTokens": 128000, + "preferWebsockets": true, + "useResponsesLite": true, + "priority": 2, + "requestModelId": "gpt-5.6-terra", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } } }, "opencode": { @@ -62311,6 +66618,36 @@ }, "contextPromotionTarget": "opencode-zen/gpt-5.4" }, + "grok-4.5": { + "id": "grok-4.5", + "name": "Grok 4.5", + "api": "openai-completions", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 6, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "grok-build-0.1": { "id": "grok-build-0.1", "name": "Grok Build 0.1", @@ -62341,6 +66678,35 @@ ] } }, + "hy3-free": { + "id": "hy3-free", + "name": "Hy3 Free", + "api": "openai-completions", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "hy3-preview-free": { "id": "hy3-preview-free", "name": "Hy3 preview Free", @@ -63216,7 +67582,7 @@ "cost": { "input": 0.66, "output": 3.41, - "cacheRead": 0.14, + "cacheRead": 0.15, "cacheWrite": 0 }, "contextWindow": 262144, @@ -63246,7 +67612,7 @@ "input": 5, "output": 30, "cacheRead": 0.5, - "cacheWrite": 0 + "cacheWrite": 6.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -63289,6 +67655,35 @@ ] } }, + "~x-ai/grok-latest": { + "id": "~x-ai/grok-latest", + "name": "Grok Latest", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 6, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": null, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "ai21/jamba-large-1.7": { "id": "ai21/jamba-large-1.7", "name": "Jamba Large 1.7", @@ -63308,6 +67703,90 @@ "contextWindow": 256000, "maxTokens": 4096 }, + "aion-labs/aion-2.0": { + "id": "aion-labs/aion-2.0", + "name": "Aion-2.0", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.7999999999999999, + "output": 1.5999999999999999, + "cacheRead": 0.19999999999999998, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "aion-labs/aion-3.0": { + "id": "aion-labs/aion-3.0", + "name": "Aion-3.0", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 3, + "output": 6, + "cacheRead": 0.75, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "aion-labs/aion-3.0-mini": { + "id": "aion-labs/aion-3.0-mini", + "name": "Aion-3.0-Mini", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.7, + "output": 1.4, + "cacheRead": 0.18, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "alibaba/tongyi-deepresearch-30b-a3b": { "id": "alibaba/tongyi-deepresearch-30b-a3b", "name": "Tongyi DeepResearch 30B A3B", @@ -64772,9 +69251,9 @@ "text" ], "cost": { - "input": 0.2288, - "output": 0.3432, - "cacheRead": 0.02288, + "input": 0.2145, + "output": 0.32175, + "cacheRead": 0.02145, "cacheWrite": 0 }, "contextWindow": 131072, @@ -64846,7 +69325,7 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 16384, + "maxTokens": 65536, "thinking": { "mode": "effort", "efforts": [ @@ -66146,8 +70625,8 @@ "text" ], "cost": { - "input": 0.12, - "output": 0.48, + "input": 0.15, + "output": 0.8999999999999999, "cacheRead": 0.049999999999999996, "cacheWrite": 0 }, @@ -66205,8 +70684,8 @@ "text" ], "cost": { - "input": 0.18, - "output": 0.72, + "input": 0.24, + "output": 0.96, "cacheRead": 0.049999999999999996, "cacheWrite": 0 }, @@ -66224,7 +70703,7 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", - "name": "MiniMax-M3", + "name": "MiniMax M3", "api": "openrouter", "provider": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", @@ -66240,7 +70719,7 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 512000, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -66875,7 +71354,7 @@ "cost": { "input": 0.375, "output": 2.025, - "cacheRead": 0.09, + "cacheRead": 0.203, "cacheWrite": 0 }, "contextWindow": 262144, @@ -66904,7 +71383,7 @@ "cost": { "input": 0.66, "output": 3.41, - "cacheRead": 0.14, + "cacheRead": 0.15, "cacheWrite": 0 }, "contextWindow": 262144, @@ -66960,13 +71439,13 @@ "image" ], "cost": { - "input": 0.74, - "output": 3.5, - "cacheRead": 0.15, + "input": 0.72, + "output": 3.49, + "cacheRead": 0.159, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 16384, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -66996,6 +71475,64 @@ "contextWindow": 131072, "maxTokens": 163840 }, + "nex-agi/nex-n2-mini": { + "id": "nex-agi/nex-n2-mini", + "name": "Nex-N2-Mini", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.024999999999999998, + "output": 0.09999999999999999, + "cacheRead": 0.0025, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "nex-agi/nex-n2-pro": { + "id": "nex-agi/nex-n2-pro", + "name": "Nex-N2-Pro", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 1, + "cacheRead": 0.024999999999999998, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "nex-agi/nex-n2-pro:free": { "id": "nex-agi/nex-n2-pro:free", "name": "Nex-N2-Pro (free)", @@ -68452,6 +72989,228 @@ }, "contextPromotionTarget": "openrouter/openai/gpt-5.4" }, + "openai/gpt-5.6-luna": { + "id": "openai/gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.09999999999999999, + "cacheWrite": 1.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-luna-pro": { + "id": "openai/gpt-5.6-luna-pro", + "name": "GPT-5.6 Luna Pro", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.09999999999999999, + "cacheWrite": 1.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-sol": { + "id": "openai/gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-sol-pro": { + "id": "openai/gpt-5.6-sol-pro", + "name": "GPT-5.6 Sol Pro", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-terra": { + "id": "openai/gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-terra-pro": { + "id": "openai/gpt-5.6-terra-pro", + "name": "GPT-5.6 Terra Pro", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "openai/gpt-audio": { "id": "openai/gpt-audio", "name": "GPT Audio", @@ -68530,8 +73289,8 @@ "text" ], "cost": { - "input": 0.03, - "output": 0.15, + "input": 0.036, + "output": 0.18, "cacheRead": 0, "cacheWrite": 0 }, @@ -70301,7 +75060,7 @@ "cost": { "input": 0.385, "output": 2.4499999999999997, - "cacheRead": 0.195, + "cacheRead": 0.111, "cacheWrite": 0 }, "contextWindow": 256000, @@ -70938,6 +75697,34 @@ "supportsToolChoice": false } }, + "tencent/hy3": { + "id": "tencent/hy3", + "name": "Hy3", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.14, + "output": 0.58, + "cacheRead": 0.035, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": null, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "tencent/hy3-preview": { "id": "tencent/hy3-preview", "name": "Hy3 Preview", @@ -70994,6 +75781,34 @@ ] } }, + "tencent/hy3:free": { + "id": "tencent/hy3:free", + "name": "Hy3 (free)", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "thedrummer/rocinante-12b": { "id": "thedrummer/rocinante-12b", "name": "Rocinante 12B", @@ -71418,6 +76233,35 @@ ] } }, + "x-ai/grok-4.5": { + "id": "x-ai/grok-4.5", + "name": "Grok 4.5", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 6, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", "name": "Grok Build 0.1", @@ -71571,7 +76415,7 @@ "cost": { "input": 0.105, "output": 0.28, - "cacheRead": 0.0028, + "cacheRead": 0.028, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -71971,13 +76815,13 @@ "text" ], "cost": { - "input": 0.9086, - "output": 2.8556, - "cacheRead": 0.16874, + "input": 0.84, + "output": 2.64, + "cacheRead": 0.156, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 131072, + "maxTokens": 128000, "thinking": { "mode": "effort", "efforts": [ @@ -73401,38 +78245,6 @@ "escapeBuiltinToolNames": true } }, - "umans-glm-5.2-nvfp4": { - "id": "umans-glm-5.2-nvfp4", - "name": "Umans GLM 5.2 NVFP4 (experimental, short test from Jun 29)", - "api": "anthropic-messages", - "provider": "umans", - "baseUrl": "https://api.code.umans.ai", - "reasoning": true, - "thinking": { - "mode": "anthropic-budget-effort", - "efforts": [ - "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } - }, - "input": [ - "text" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 405504, - "maxTokens": 131071, - "compat": { - "escapeBuiltinToolNames": true - } - }, "umans-kimi-k2.7": { "id": "umans-kimi-k2.7", "name": "Umans Kimi K2.7 Code", @@ -73524,6 +78336,64 @@ "supportsUsageInStreaming": false } }, + "aion-labs-aion-3-0": { + "id": "aion-labs-aion-3-0", + "name": "Aion 3.0", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 3.75, + "output": 7.5, + "cacheRead": 0.9375, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "aion-labs-aion-3-0-mini": { + "id": "aion-labs-aion-3-0-mini", + "name": "Aion 3.0 Mini", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.875, + "output": 1.75, + "cacheRead": 0.225, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "aion-labs.aion-2-0": { "id": "aion-labs.aion-2-0", "name": "aion-labs.aion-2-0", @@ -74080,6 +78950,28 @@ } } }, + "e2ee-deepseek-v4-flash": { + "id": "e2ee-deepseek-v4-flash", + "name": "e2ee-deepseek-v4-flash", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 1048576, + "compat": { + "supportsUsageInStreaming": false + } + }, "e2ee-gemma-3-27b-p": { "id": "e2ee-gemma-3-27b-p", "name": "e2ee-gemma-3-27b-p", @@ -74366,6 +79258,28 @@ "supportsUsageInStreaming": false } }, + "e2ee-qwen3-6-27b": { + "id": "e2ee-qwen3-6-27b", + "name": "e2ee-qwen3-6-27b", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 65536, + "compat": { + "supportsUsageInStreaming": false + } + }, "e2ee-qwen3-6-35b-a3b": { "id": "e2ee-qwen3-6-35b-a3b", "name": "e2ee-qwen3-6-35b-a3b", @@ -74845,6 +79759,36 @@ ] } }, + "grok-4-5": { + "id": "grok-4-5", + "name": "Grok 4.5", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.27, + "output": 6.8, + "cacheRead": 0.57, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 32000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "grok-41-fast": { "id": "grok-41-fast", "name": "Grok 4.1 Fast", @@ -75756,6 +80700,186 @@ ] } }, + "openai-gpt-56-luna": { + "id": "openai-gpt-56-luna", + "name": "GPT-5.6 Luna", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.25, + "output": 7.5, + "cacheRead": 0.125, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai-gpt-56-luna-pro": { + "id": "openai-gpt-56-luna-pro", + "name": "GPT-5.6 Luna Pro", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.25, + "output": 7.5, + "cacheRead": 0.125, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai-gpt-56-sol": { + "id": "openai-gpt-56-sol", + "name": "GPT-5.6 Sol", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 6.25, + "output": 37.5, + "cacheRead": 0.625, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai-gpt-56-sol-pro": { + "id": "openai-gpt-56-sol-pro", + "name": "GPT-5.6 Sol Pro", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 6.25, + "output": 37.5, + "cacheRead": 0.625, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai-gpt-56-terra": { + "id": "openai-gpt-56-terra", + "name": "GPT-5.6 Terra", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3.125, + "output": 18.75, + "cacheRead": 0.3125, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai-gpt-56-terra-pro": { + "id": "openai-gpt-56-terra-pro", + "name": "GPT-5.6 Terra Pro", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3.125, + "output": 18.75, + "cacheRead": 0.3125, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "openai-gpt-oss-120b": { "id": "openai-gpt-oss-120b", "name": "OpenAI GPT OSS 120B", @@ -78076,7 +83200,7 @@ "cost": { "input": 0.14, "output": 0.28, - "cacheRead": 0.0028, + "cacheRead": 0.028, "cacheWrite": 0 }, "contextWindow": 1000000, @@ -78865,6 +83989,36 @@ "contextWindow": 128000, "maxTokens": 8192 }, + "meta/muse-spark-1.1": { + "id": "meta/muse-spark-1.1", + "name": "Muse Spark 1.1", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.25, + "output": 4.25, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 1048576, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "minimax/minimax-m2": { "id": "minimax/minimax-m2", "name": "MiniMax M2", @@ -79099,7 +84253,7 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", - "name": "MiniMax-M3", + "name": "MiniMax M3", "api": "anthropic-messages", "provider": "vercel-ai-gateway", "baseUrl": "https://ai-gateway.vercel.sh", @@ -80667,6 +85821,96 @@ }, "contextPromotionTarget": "vercel-ai-gateway/openai/gpt-5.4" }, + "openai/gpt-5.6-luna": { + "id": "openai/gpt-5.6-luna", + "name": "GPT 5.6 Luna", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.09999999999999999, + "cacheWrite": 1.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai/gpt-5.6-sol": { + "id": "openai/gpt-5.6-sol", + "name": "GPT 5.6 Sol", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai/gpt-5.6-terra": { + "id": "openai/gpt-5.6-terra", + "name": "GPT 5.6 Terra", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", @@ -81557,6 +86801,36 @@ ] } }, + "xai/grok-4.5": { + "id": "xai/grok-4.5", + "name": "Grok 4.5", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 6, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "xai/grok-build-0.1": { "id": "xai/grok-build-0.1", "name": "Grok Build 0.1", @@ -82313,6 +87587,72 @@ "supportsDeveloperRole": false } }, + "glm5.2-fast": { + "id": "glm5.2-fast", + "name": "GLM5.2-Fast", + "api": "openai-completions", + "provider": "wafer-serverless", + "baseUrl": "https://pass.wafer.ai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 3.75, + "output": 12.8125, + "cacheRead": 0.625, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + }, + "compat": { + "reasoningContentField": "reasoning_content", + "supportsDeveloperRole": false + } + }, + "GLM5.2-Turbo": { + "id": "GLM5.2-Turbo", + "name": "GLM5.2-Turbo", + "api": "openai-completions", + "provider": "wafer-serverless", + "baseUrl": "https://pass.wafer.ai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 3.75, + "output": 12.8125, + "cacheRead": 0.625, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + }, + "compat": { + "reasoningContentField": "reasoning_content", + "supportsDeveloperRole": false + } + }, "Kimi-K2.6": { "id": "Kimi-K2.6", "name": "Kimi-K2.6", @@ -83094,6 +88434,35 @@ ] } }, + "grok-4.5": { + "id": "grok-4.5", + "name": "Grok 4.5", + "api": "openai-completions", + "provider": "xai", + "baseUrl": "https://api.x.ai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 6, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "grok-beta": { "id": "grok-beta", "name": "Grok Beta", @@ -83330,6 +88699,47 @@ "omitReasoningEffort": false } }, + "grok-4.5": { + "id": "grok-4.5", + "name": "Grok 4.5", + "api": "openai-responses", + "provider": "xai-oauth", + "baseUrl": "https://api.x.ai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low" + } + }, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "omitReasoningEffort": true + } + }, "grok-build": { "id": "grok-build", "name": "Grok Build", @@ -86308,6 +91718,25 @@ ] } }, + "kuaishou/kat-coder-air-v2.5": { + "id": "kuaishou/kat-coder-air-v2.5", + "name": "KAT-Coder-Air-V2.5", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.135, + "output": 0.54, + "cacheRead": 0.027, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": null + }, "kuaishou/kat-coder-pro-v1": { "id": "kuaishou/kat-coder-pro-v1", "name": "KAT-Coder-Pro-V1", @@ -86365,6 +91794,25 @@ "contextWindow": 256000, "maxTokens": 80000 }, + "kuaishou/kat-coder-pro-v2.5": { + "id": "kuaishou/kat-coder-pro-v2.5", + "name": "KAT-Coder-Pro-V2.5", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.444, + "output": 1.776, + "cacheRead": 0.09, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": null + }, "meituan/longcat-2.0": { "id": "meituan/longcat-2.0", "name": "LongCat-2.0", @@ -87648,6 +93096,117 @@ }, "contextPromotionTarget": "zenmux/openai/gpt-5.4" }, + "openai/gpt-5.6-luna": { + "id": "openai/gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 1.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-sol": { + "id": "openai/gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-terra": { + "id": "openai/gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "openai/gpt-image-1.5": { "id": "openai/gpt-image-1.5", "name": "GPT-Image-1.5", @@ -88482,9 +94041,9 @@ "text" ], "cost": { - "input": 0.134561595, - "output": 0.539161765, - "cacheRead": 0.033869245, + "input": 0.1323, + "output": 0.5301, + "cacheRead": 0.0333, "cacheWrite": 0 }, "contextWindow": 262144, @@ -88938,6 +94497,66 @@ ] } }, + "x-ai/grok-4.5": { + "id": "x-ai/grok-4.5", + "name": "Grok 4.5", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 6, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "x-ai/grok-4.5-free": { + "id": "x-ai/grok-4.5-free", + "name": "Grok 4.5 (Free)", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", "name": "Grok Build 0.1", @@ -89934,4 +95553,4 @@ } } } -} \ No newline at end of file +} diff --git a/packages/catalog/src/provider-models/descriptors.ts b/packages/catalog/src/provider-models/descriptors.ts index 29a80ecf4..57c169220 100644 --- a/packages/catalog/src/provider-models/descriptors.ts +++ b/packages/catalog/src/provider-models/descriptors.ts @@ -29,6 +29,7 @@ import { mistralModelManagerOptions, moonshotModelManagerOptions, nanoGptModelManagerOptions, + novitaModelManagerOptions, nvidiaModelManagerOptions, ollamaModelManagerOptions, openaiModelManagerOptions, @@ -271,6 +272,14 @@ export const CATALOG_PROVIDERS = [ createModelManagerOptions: (config: ModelManagerConfig) => nvidiaModelManagerOptions(config), catalogDiscovery: { label: "NVIDIA" }, }, + { + id: "novita", + defaultModel: "moonshotai/kimi-k2.7-code", + envVars: ["NOVITA_API_KEY"], + createModelManagerOptions: (config: ModelManagerConfig) => novitaModelManagerOptions(config), + dynamicModelsAuthoritative: true, + catalogDiscovery: { label: "Novita", allowUnauthenticated: true }, + }, { id: "ollama", defaultModel: "gpt-oss:20b", diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 83cd58136..0b785b6ca 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -821,6 +821,62 @@ export function openaiModelManagerOptions(config?: OpenAIModelManagerConfig): Mo }; } +/** First-party gpt-5.6 SKUs that accept `reasoning: { mode: "pro" }` on the Responses APIs. */ +const OPENAI_PRO_REASONING_BASE_IDS: Record = { + "gpt-5.6-luna": true, + "gpt-5.6-sol": true, + "gpt-5.6-terra": true, +}; +const OPENAI_PRO_REASONING_PROVIDERS: Record = { openai: true, "openai-codex": true }; + +/** + * A row this generator pass owns: one of the derived `gpt-5.6-*-pro` alias ids + * on `openai`/`openai-codex` that carries the generated `reasoningMode` marker. + * A real upstream model occupying the same id has no `reasoningMode` and is + * never touched. + */ +function isGeneratedOpenAIProReasoningAlias(model: ModelSpec): boolean { + return ( + OPENAI_PRO_REASONING_PROVIDERS[model.provider] === true && + model.reasoningMode !== undefined && + model.id.endsWith("-pro") && + OPENAI_PRO_REASONING_BASE_IDS[model.id.slice(0, -"-pro".length)] === true + ); +} + +/** + * Re-derive the generated pro-reasoning aliases (`gpt-5.6-*-pro`) for the + * first-party `openai`/`openai-codex` gpt-5.6 rows. Each alias inherits the + * base row's metadata, requests the base wire id via `requestModelId`, and + * sets `reasoningMode: "pro"` so Responses-family request builders emit + * `reasoning: { mode: "pro" }`. Called by the models.json generator after all + * sources merge: stale copies of the owned aliases (previous snapshot) are + * dropped and re-projected from the current base rows so alias metadata always + * tracks the base, while a real upstream model that occupies an alias id wins + * and suppresses the projection. + */ +export function projectOpenAIProReasoningAliases(models: readonly ModelSpec[]): ModelSpec[] { + const kept = models.filter(model => !isGeneratedOpenAIProReasoningAlias(model)); + const ids = new Set(kept.map(model => `${model.provider}/${model.id}`)); + const out = [...kept]; + for (const model of kept) { + if (!OPENAI_PRO_REASONING_PROVIDERS[model.provider]) continue; + if (!OPENAI_PRO_REASONING_BASE_IDS[model.id]) continue; + const aliasId = `${model.id}-pro`; + const aliasKey = `${model.provider}/${aliasId}`; + if (ids.has(aliasKey)) continue; + ids.add(aliasKey); + out.push({ + ...model, + id: aliasId, + name: `${model.name} Pro`, + requestModelId: model.id, + reasoningMode: "pro", + }); + } + return out; +} + // --------------------------------------------------------------------------- // 2. Groq // --------------------------------------------------------------------------- @@ -915,6 +971,102 @@ export function nvidiaModelManagerOptions( return createSimpleOpenAICompletionsOptions("nvidia", "https://integrate.api.nvidia.com/v1", config); } +// --------------------------------------------------------------------------- +// 5.5 Novita +// --------------------------------------------------------------------------- + +/** Novita OpenAI-compatible discovery configuration. */ +export interface NovitaModelManagerConfig { + apiKey?: string; + baseUrl?: string; + fetch?: FetchImpl; +} + +function novitaArrayIncludes(value: unknown, expected: string): boolean { + return Array.isArray(value) && value.some(item => item === expected); +} + +function isPublicNovitaModelId(id: string): boolean { + return !id.toLowerCase().startsWith("ai_infer_test"); +} + +// Novita reports token prices in 1/10,000 USD per million tokens. +function toNovitaCostPerMillion(value: unknown): number { + return toPositiveNumber(value, 0) / 10_000; +} + +function getNovitaCacheReadPricePerMillion(entry: OpenAICompatibleModelRecord): number { + const pricing = entry.pricing; + if (!isRecord(pricing)) { + return 0; + } + const cacheRead = pricing.input_cache_read; + if (!isRecord(cacheRead)) { + return 0; + } + return toNovitaCostPerMillion(cacheRead.price_per_m); +} + +function mapNovitaModel( + entry: OpenAICompatibleModelRecord, + defaults: ModelSpec<"openai-completions">, + reference: ModelSpec<"openai-completions"> | undefined, +): ModelSpec<"openai-completions"> { + const model = mapWithBundledReference( + { + ...entry, + name: entry.display_name ?? entry.title ?? entry.name, + }, + defaults, + reference, + ); + return { + ...model, + reasoning: novitaArrayIncludes(entry.features, "reasoning"), + supportsTools: novitaArrayIncludes(entry.features, "function-calling"), + input: toInputCapabilities(entry.input_modalities), + cost: { + input: toNovitaCostPerMillion(entry.input_token_price_per_m), + output: toNovitaCostPerMillion(entry.output_token_price_per_m), + cacheRead: getNovitaCacheReadPricePerMillion(entry), + cacheWrite: 0, + }, + contextWindow: toPositiveNumber(entry.context_size, model.contextWindow), + maxTokens: toPositiveNumber(entry.max_output_tokens, model.maxTokens), + }; +} + +/** Builds Novita's public model-discovery manager. */ +export function novitaModelManagerOptions( + config?: NovitaModelManagerConfig, +): ModelManagerOptions<"openai-completions"> { + const apiKey = config?.apiKey; + const baseUrl = config?.baseUrl ?? "https://api.novita.ai/openai/v1"; + const references = createBundledReferenceMap<"openai-completions">("novita"); + return { + providerId: "novita", + dynamicModelsAuthoritative: true, + fetchDynamicModels: async () => + fetchOpenAICompatibleModels({ + api: "openai-completions", + provider: "novita", + baseUrl, + apiKey, + mapModel: (entry, defaults) => mapNovitaModel(entry, defaults, references.get(defaults.id)), + filterModel: (entry, model) => { + const active = typeof entry.status !== "number" || entry.status === 1; + return ( + active && + isPublicNovitaModelId(model.id) && + novitaArrayIncludes(entry.endpoints, "chat/completions") && + toPositiveNumber(entry.max_output_tokens, 0) > 0 + ); + }, + fetch: config?.fetch, + }), + }; +} + // --------------------------------------------------------------------------- // 6. xAI // --------------------------------------------------------------------------- @@ -983,6 +1135,7 @@ export const XAI_OAUTH_CURATED_MODELS: readonly XAICuratedModel[] = [ input: ["text", "image"], }, { id: "grok-4.3", contextWindow: 1_000_000, name: "Grok 4.3", input: ["text", "image"] }, + { id: "grok-4.5", contextWindow: 500_000, name: "Grok 4.5", input: ["text", "image"] }, // grok-4.20-multi-agent-0309 is text-only per the bundled catalog; omit `input` for the default. { id: "grok-4.20-multi-agent-0309", contextWindow: 2_000_000, name: "Grok 4.20 (Multi-Agent)" }, { diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts index 103549d0f..0407a2050 100644 --- a/packages/catalog/src/types.ts +++ b/packages/catalog/src/types.ts @@ -691,6 +691,13 @@ export interface Model { * everything local (selection, caching, usage attribution) keys on `id`. */ requestModelId?: string; + /** + * `reasoning.mode` to send on OpenAI Responses-family requests. Set on + * generated pro aliases (`gpt-5.6-*-pro` on `openai`/`openai-codex`) that + * pair a base wire id (`requestModelId`) with OpenAI's pro reasoning + * serving path. Absent everywhere else; providers omit the wire field. + */ + reasoningMode?: "pro"; name: string; api: TApi; provider: Provider; @@ -751,6 +758,8 @@ export interface Model { transport?: "pi-native"; /** Hint that websocket transport should be preferred when supported by the provider implementation. */ preferWebsockets?: boolean; + /** Codex Responses Lite transport: send the lite marker and carry instructions/tools as input items (mirrors codex-rs `use_responses_lite`). */ + useResponsesLite?: boolean; /** Preferred model to switch to when context promotion is triggered (model id or provider/id). */ contextPromotionTarget?: string; /** Preferred model to use only for compaction (model id or provider/id); the active session model is unchanged. */ diff --git a/packages/catalog/src/variant-collapse.ts b/packages/catalog/src/variant-collapse.ts index b3042503d..46d1f9d52 100644 --- a/packages/catalog/src/variant-collapse.ts +++ b/packages/catalog/src/variant-collapse.ts @@ -113,6 +113,14 @@ function thinkingPair(baseId: string, name: string): EffortVariantFamily { type DevinTierRoutes = Partial>; +const DEVIN_FIVE_TIER_EFFORTS: readonly Effort[] = [ + Effort.Minimal, + Effort.Low, + Effort.Medium, + Effort.High, + Effort.XHigh, +]; + function devinTierFamily( id: string, name: string, @@ -160,6 +168,44 @@ function devinTierFamily( }; } +/** + * GPT-5.6 (Luna/Sol/Terra) adds a genuine `max` tier above `xhigh`, so the + * standard family shifts every user effort up one notch (`minimal` → `-low` + * … `xhigh` → `-max`), mirroring the Opus 4.7+ five-tier mapping. Devin + * serves no `-max-priority` sibling, so the fast family keeps the direct + * `low..xhigh` `-priority` scale. + */ +function devinGpt56Families(variant: "luna" | "sol" | "terra", name: string): readonly EffortVariantFamily[] { + const base = `gpt-5-6-${variant}`; + return [ + devinTierFamily( + base, + name, + { + off: `${base}-none`, + minimal: `${base}-low`, + low: `${base}-medium`, + medium: `${base}-high`, + high: `${base}-xhigh`, + xhigh: `${base}-max`, + }, + DEVIN_FIVE_TIER_EFFORTS, + ), + devinTierFamily( + `${base}-fast`, + `${name} Fast`, + { + off: `${base}-none-priority`, + low: `${base}-low-priority`, + medium: `${base}-medium-priority`, + high: `${base}-high-priority`, + xhigh: `${base}-xhigh-priority`, + }, + DEVIN_FIVE_TIER_EFFORTS, + ), + ]; +} + const GEMINI_3_FLASH_FAMILY_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]; const GEMINI_3_PRO_FAMILY_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High]; @@ -330,7 +376,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { }, thinking: { mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: DEVIN_FIVE_TIER_EFFORTS, requiresEffort: true, }, }, @@ -353,7 +399,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { }, thinking: { mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: DEVIN_FIVE_TIER_EFFORTS, requiresEffort: true, }, }, @@ -376,7 +422,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { }, thinking: { mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: DEVIN_FIVE_TIER_EFFORTS, requiresEffort: true, }, }, @@ -399,7 +445,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { }, thinking: { mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: DEVIN_FIVE_TIER_EFFORTS, requiresEffort: true, }, }, @@ -413,7 +459,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "MODEL_GPT_5_2_HIGH", xhigh: "MODEL_GPT_5_2_XHIGH", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), devinTierFamily( "gpt-5-3-codex", @@ -424,7 +470,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-3-codex-high", xhigh: "gpt-5-3-codex-xhigh", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), devinTierFamily( "gpt-5-3-codex-fast", @@ -435,7 +481,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-3-codex-high-priority", xhigh: "gpt-5-3-codex-xhigh-priority", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), devinTierFamily( "gpt-5-4", @@ -447,7 +493,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-4-high", xhigh: "gpt-5-4-xhigh", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), devinTierFamily( "gpt-5-4-fast", @@ -459,7 +505,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-4-high-priority", xhigh: "gpt-5-4-xhigh-priority", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), devinTierFamily( "gpt-5-4-mini", @@ -470,7 +516,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-4-mini-high", xhigh: "gpt-5-4-mini-xhigh", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), devinTierFamily( "gpt-5-5", @@ -482,7 +528,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-5-high", xhigh: "gpt-5-5-xhigh", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), devinTierFamily( "gpt-5-5-fast", @@ -494,8 +540,11 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-5-high-priority", xhigh: "gpt-5-5-xhigh-priority", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), + ...devinGpt56Families("luna", "GPT-5.6 Luna"), + ...devinGpt56Families("sol", "GPT-5.6 Sol"), + ...devinGpt56Families("terra", "GPT-5.6 Terra"), devinTierFamily( "gemini-3-1-pro", "Gemini 3.1 Pro", diff --git a/packages/catalog/src/wire/codex.ts b/packages/catalog/src/wire/codex.ts index 1d3d80700..329ace70f 100644 --- a/packages/catalog/src/wire/codex.ts +++ b/packages/catalog/src/wire/codex.ts @@ -10,6 +10,15 @@ export const OPENAI_HEADERS = { ORIGINATOR: "originator", SESSION_ID: "session_id", CONVERSATION_ID: "conversation_id", + SCOPED_SESSION_ID: "session-id", + THREAD_ID: "thread-id", + INSTALLATION_ID: "x-codex-installation-id", + WINDOW_ID: "x-codex-window-id", + TURN_METADATA: "x-codex-turn-metadata", + PARENT_THREAD_ID: "x-codex-parent-thread-id", + SUBAGENT: "x-openai-subagent", + /** Responses Lite transport marker (codex-rs `add_responses_lite_header`); value is always `"true"`. */ + RESPONSES_LITE: "x-openai-internal-codex-responses-lite", } as const; export const OPENAI_HEADER_VALUES = { diff --git a/packages/catalog/test/codex-discovery.test.ts b/packages/catalog/test/codex-discovery.test.ts index b6f87c79f..babff9c28 100644 --- a/packages/catalog/test/codex-discovery.test.ts +++ b/packages/catalog/test/codex-discovery.test.ts @@ -52,6 +52,50 @@ describe("Codex model discovery", () => { }); }); + it("carries use_responses_lite and prefer_websockets onto the model spec", async () => { + const fetchFn: typeof fetch = Object.assign( + async () => + new Response( + JSON.stringify({ + models: [ + { + slug: "gpt-5.6-terra", + display_name: "GPT-5.6-Terra", + context_window: 372_000, + default_reasoning_level: "medium", + supported_reasoning_levels: ["low", "medium", "high"], + input_modalities: ["text", "image"], + supported_in_api: true, + prefer_websockets: true, + use_responses_lite: true, + }, + { + slug: "gpt-5.5", + display_name: "GPT-5.5", + context_window: 272_000, + default_reasoning_level: "high", + supported_reasoning_levels: ["low", "high"], + input_modalities: ["text"], + supported_in_api: true, + }, + ], + }), + ), + { preconnect() {} }, + ); + const result = await fetchCodexModels({ + accessToken: "test-token", + baseUrl: "https://codex.example/backend-api", + clientVersion: "0.99.0", + fetchFn, + }); + + const terra = result?.models.find(model => model.id === "gpt-5.6-terra"); + expect(terra).toMatchObject({ preferWebsockets: true, useResponsesLite: true }); + const legacy = result?.models.find(model => model.id === "gpt-5.5"); + expect(legacy?.useResponsesLite).toBeUndefined(); + }); + it("ignores pre-V2 Codex discovery cache rows", async () => { const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-codex-v7-cache-")); const dbPath = path.join(tempDir, "models.db"); diff --git a/packages/catalog/test/cursor-discovery.test.ts b/packages/catalog/test/cursor-discovery.test.ts new file mode 100644 index 000000000..f62b7c37c --- /dev/null +++ b/packages/catalog/test/cursor-discovery.test.ts @@ -0,0 +1,96 @@ +import { afterAll, beforeAll, describe, expect, it } from "bun:test"; +import * as http2 from "node:http2"; +import { create, toBinary } from "@bufbuild/protobuf"; +// Import from source, not the package specifier: the workspace `node_modules` +// copy resolves to the primary checkout, not this worktree. +import { fetchCursorUsableModels } from "../src/discovery/cursor"; +import { GetUsableModelsResponseSchema, ModelDetailsSchema } from "../src/discovery/cursor-gen/agent_pb"; +import type { ModelSpec } from "../src/types"; + +const FIXTURE_MODEL_IDS = [ + // Reference-less ids from families whose native catalogs are multimodal. + "claude-opus-4-8-99999999", + "gpt-5.5-codex-20991231", + "gemini-4-pro-exp", + // Reference-less ids from text-only families. + "composer-3", + "grok-code-fast-2", + // Bundled-reference ids: the reference stays authoritative. + "claude-4.5-opus-high", + "claude-4.6-opus-high", + "composer-1", +]; + +let server: http2.Http2Server; +let baseUrl: string; + +beforeAll(async () => { + const response = create(GetUsableModelsResponseSchema, { + models: FIXTURE_MODEL_IDS.map(modelId => create(ModelDetailsSchema, { modelId })), + }); + const payload = Buffer.from(toBinary(GetUsableModelsResponseSchema, response)); + + server = http2.createServer(); + server.on("stream", (stream: http2.ServerHttp2Stream, headers: http2.IncomingHttpHeaders) => { + stream.on("data", () => {}); + stream.on("end", () => { + if (headers[":path"] !== "/agent.v1.AgentService/GetUsableModels") { + stream.respond({ ":status": 404 }); + stream.end(); + return; + } + stream.respond({ ":status": 200, "content-type": "application/proto" }); + stream.end(payload); + }); + }); + await new Promise(resolve => server.listen(0, "127.0.0.1", resolve)); + const address = server.address(); + if (!address || typeof address === "string") { + throw new Error("expected http2 fixture server to bind a tcp port"); + } + baseUrl = `http://127.0.0.1:${address.port}`; +}); + +afterAll(() => { + server?.close(); +}); + +async function discover(): Promise>> { + const models = await fetchCursorUsableModels({ apiKey: "test-key", baseUrl }); + expect(models).not.toBeNull(); + return new Map((models ?? []).map(model => [model.id, model])); +} + +describe("cursor discovery input modalities (issue #4726)", () => { + it("classifies reference-less multimodal-family models as text+image", async () => { + const byId = await discover(); + expect(byId.get("claude-opus-4-8-99999999")?.input).toEqual(["text", "image"]); + expect(byId.get("gpt-5.5-codex-20991231")?.input).toEqual(["text", "image"]); + expect(byId.get("gemini-4-pro-exp")?.input).toEqual(["text", "image"]); + }); + + it("keeps reference-less text-only families text-only", async () => { + const byId = await discover(); + expect(byId.get("composer-3")?.input).toEqual(["text"]); + expect(byId.get("grok-code-fast-2")?.input).toEqual(["text"]); + }); + + it("keeps bundled references authoritative for input modalities", async () => { + const byId = await discover(); + // Bundled cursor references carry their own input classification; the + // id-based inference must not override it in either direction. + expect(byId.get("claude-4.5-opus-high")?.input).toEqual(["text", "image"]); + expect(byId.get("claude-4.6-opus-high")?.input).toEqual(["text"]); + expect(byId.get("composer-1")?.input).toEqual(["text"]); + }); + + it("preserves fallback defaults for reference-less models", async () => { + const byId = await discover(); + const spec = byId.get("claude-opus-4-8-99999999"); + expect(spec?.provider).toBe("cursor"); + expect(spec?.api).toBe("cursor-agent"); + expect(spec?.contextWindow).toBe(200_000); + expect(spec?.maxTokens).toBe(64_000); + expect(spec?.cost).toEqual({ input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }); + }); +}); diff --git a/packages/catalog/test/identity-family.test.ts b/packages/catalog/test/identity-family.test.ts index 716731082..c0748fd8f 100644 --- a/packages/catalog/test/identity-family.test.ts +++ b/packages/catalog/test/identity-family.test.ts @@ -265,6 +265,7 @@ describe("isGrokReasoningEffortCapable", () => { expect(isGrokReasoningEffortCapable("grok-3-mini")).toBe(true); expect(isGrokReasoningEffortCapable("grok-4.20-multi-agent")).toBe(true); expect(isGrokReasoningEffortCapable("xai-oauth/grok-4.3")).toBe(true); + expect(isGrokReasoningEffortCapable("xai-oauth/grok-4.5")).toBe(true); expect(isGrokReasoningEffortCapable("openrouter/xai/grok-3-mini")).toBe(true); }); diff --git a/packages/catalog/test/model-thinking.test.ts b/packages/catalog/test/model-thinking.test.ts index 6d6019fe1..a21a323fc 100644 --- a/packages/catalog/test/model-thinking.test.ts +++ b/packages/catalog/test/model-thinking.test.ts @@ -567,6 +567,90 @@ describe("model thinking derivation", () => { expect(getSupportedEfforts(model)).toEqual([]); expect(clampThinkingLevelForModel(model, Effort.High)).toBeUndefined(); }); + + it("bakes the GPT-5.6 shifted five-tier effort map on wire-effort APIs", () => { + const codex = createModel({ + id: "gpt-5.6-sol", + api: "openai-codex-responses", + provider: "openai-codex", + }); + + expect(codex.thinking).toEqual({ + mode: "effort", + efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + effortMap: { + minimal: "low", + low: "medium", + medium: "high", + high: "xhigh", + xhigh: "max", + }, + }); + + // Stale baked four-tier metadata (caches/discovery) normalizes back to + // the five-tier ladder with the map attached — the wire-defaults + // backfill path — and namespaced OpenRouter ids parse. + const staleOpenRouter = createModel({ + id: "openai/gpt-5.6-terra", + api: "openrouter", + provider: "openrouter", + baseUrl: "https://openrouter.ai/api/v1", + thinking: { + mode: "effort", + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + }, + }); + + expect(staleOpenRouter.thinking).toEqual({ + mode: "effort", + efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + effortMap: { + minimal: "low", + low: "medium", + medium: "high", + high: "xhigh", + xhigh: "max", + }, + }); + }); + + it("keeps pre-5.6 and Devin-routed GPT models off the shifted effort map", () => { + const gpt55 = createModel({ + id: "gpt-5.5", + api: "openai-responses", + provider: "openai", + }); + + expect(gpt55.thinking).toEqual({ + mode: "effort", + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + }); + expect(gpt55.thinking?.effortMap).toBeUndefined(); + + // Devin selects effort by routing to per-tier sibling model ids, never + // via a wire reasoning.effort field — the shifted map must not attach. + const devin = createModel({ + id: "gpt-5-6-sol", + api: "devin-agent", + provider: "devin", + baseUrl: "https://server.codeium.com", + thinking: { + mode: "effort", + efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + effortRouting: { + off: "gpt-5-6-sol-none", + minimal: "gpt-5-6-sol-low", + low: "gpt-5-6-sol-medium", + medium: "gpt-5-6-sol-high", + high: "gpt-5-6-sol-xhigh", + xhigh: "gpt-5-6-sol-max", + }, + }, + }); + + expect(devin.thinking?.effortMap).toBeUndefined(); + expect(devin.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]); + }); }); describe("model thinking runtime helpers", () => { diff --git a/packages/catalog/test/novita-provider.test.ts b/packages/catalog/test/novita-provider.test.ts new file mode 100644 index 000000000..3bb351d68 --- /dev/null +++ b/packages/catalog/test/novita-provider.test.ts @@ -0,0 +1,94 @@ +import { describe, expect, test } from "bun:test"; +import { DEFAULT_MODEL_PER_PROVIDER, PROVIDER_DESCRIPTORS } from "@oh-my-pi/pi-catalog/provider-models/descriptors"; +import { novitaModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; + +describe("Novita built-in provider", () => { + test("registers catalog descriptor with NOVITA_API_KEY env discovery", () => { + const descriptor = PROVIDER_DESCRIPTORS.find(item => item.providerId === "novita"); + expect(descriptor).toBeDefined(); + expect(descriptor?.defaultModel).toBe("moonshotai/kimi-k2.7-code"); + expect(descriptor?.catalogDiscovery?.envVars).toContain("NOVITA_API_KEY"); + expect(descriptor?.catalogDiscovery?.allowUnauthenticated).toBe(true); + expect(descriptor?.dynamicModelsAuthoritative).toBe(true); + expect(DEFAULT_MODEL_PER_PROVIDER.novita).toBe("moonshotai/kimi-k2.7-code"); + }); + + test("maps Novita model catalog metadata from the public OpenAI-compatible endpoint", async () => { + const requests: string[] = []; + const fetchMock = async (input: string | URL | Request): Promise => { + requests.push(input.toString()); + return Response.json({ + data: [ + { + id: "moonshotai/kimi-k2.7-code", + display_name: "Kimi K2.7 Code", + status: 1, + context_size: 262144, + max_output_tokens: 131072, + input_token_price_per_m: 9500, + output_token_price_per_m: 40000, + pricing: { + input_cache_read: { + price_per_m: 1900, + }, + }, + features: ["serverless", "function-calling", "structured-outputs", "reasoning"], + endpoints: ["chat/completions", "anthropic"], + input_modalities: ["text", "image", "video"], + }, + { + id: "qwen/qwen3-8b-fp8", + status: 4, + context_size: 128000, + max_output_tokens: 20000, + endpoints: ["chat/completions"], + input_modalities: ["text"], + }, + { + id: "ai_infer_test_1", + status: 1, + context_size: 200000, + max_output_tokens: 200000, + features: ["function-calling"], + endpoints: ["chat/completions"], + input_modalities: ["text"], + }, + { + id: "minimax/m2-her", + status: 1, + context_size: 32000, + features: ["serverless"], + endpoints: ["chat/completions"], + input_modalities: ["text"], + }, + { + id: "test/zero-output", + status: 1, + context_size: 32000, + max_output_tokens: 0, + features: ["serverless"], + endpoints: ["chat/completions"], + input_modalities: ["text"], + }, + ], + }); + }; + + const options = novitaModelManagerOptions({ fetch: fetchMock }); + const models = await options.fetchDynamicModels?.(); + const model = models?.find(item => item.id === "moonshotai/kimi-k2.7-code"); + + expect(requests).toEqual(["https://api.novita.ai/openai/v1/models"]); + expect(options.dynamicModelsAuthoritative).toBe(true); + expect(models?.map(item => item.id)).toEqual(["moonshotai/kimi-k2.7-code"]); + expect(model?.provider).toBe("novita"); + expect(model?.baseUrl).toBe("https://api.novita.ai/openai/v1"); + expect(model?.name).toBe("Kimi K2.7 Code"); + expect(model?.reasoning).toBe(true); + expect(model?.supportsTools).toBe(true); + expect(model?.input).toEqual(["text", "image"]); + expect(model?.cost).toEqual({ input: 0.95, output: 4, cacheRead: 0.19, cacheWrite: 0 }); + expect(model?.contextWindow).toBe(262144); + expect(model?.maxTokens).toBe(131072); + }); +}); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index f9b8bb0ef..97a2a9e28 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,44 @@ ## [Unreleased] +### Fixed + +- Fixed compaction aborting instead of trying an authenticated fallback model when Amazon Bedrock credential resolution fails before a request is sent. ([#5030](https://github.com/can1357/oh-my-pi/pull/5030) by [@usr-bin-roygbiv](https://github.com/usr-bin-roygbiv)) +- Fixed full-context forks cold-missing OpenAI prompt caches by persisting an inherited provider prompt-cache key separately from the new OMP session id, adding `--prompt-cache-key` for explicit cache affinity, and dropping automatic inheritance when startup changes the model, thinking level, system prompt, or tool schema. ([#5035](https://github.com/can1357/oh-my-pi/issues/5035)) +- Fixed Codex advisor requests using local `-advisor` session labels as provider session IDs; advisors now use stable UUIDv7 provider identities while keeping labeled transcript names. ([#5040](https://github.com/can1357/oh-my-pi/issues/5040)) + +## [16.3.15] - 2026-07-09 + +### Changed + +- Integrated testing guidance directly into the main system prompt for improved workflow cohesion +- Moved testing guidance into the main system prompt and removed the bundled Tester subagent. + +## [16.3.14] - 2026-07-09 + +### Fixed + +- Fixed issue where unfinalized tool blocks could incorrectly pin the live-region scroll seam +- Improved rendering of raw thinking blocks by stripping empty HTML comment noise +- Fixed display of thinking blocks consisting entirely of hidden comment noise +- Fixed gpt-5.6 reasoning summaries rendering literal `` sentinel lines in thinking blocks; empty HTML comments (and the unterminated ``), streamed as a ``. Comments with actual content are left untouched. +const EMPTY_COMMENT_RE = /^$/; +const OPEN_COMMENT_RE = /^` + * sentinel lines outside code fences (see {@link isCommentNoise}); prose-only + * mode additionally elides fenced code down to a trailing ellipsis. + */ export function formatThinkingForDisplay(text: string, proseOnly: boolean): string { - if (!proseOnly || !text) return text; - if (text === formatCacheKey) return formatCacheValue; + if (!text) return text; + const hasComment = text.includes("` leaves no blank tail. + if (hasComment && isCommentNoise(line, i === lines.length - 1)) continue; + + const open = FENCE.exec(line); + if (open) { const marker = open[2]!; const ch = marker[0]!; // A backtick fence's info string may not contain a backtick. @@ -79,18 +116,25 @@ export function formatThinkingForDisplay(text: string, proseOnly: boolean): stri inFence = true; fenceChar = ch; fenceLen = marker.length; - appendEllipsis(); - } else { - resultLines.push(line); + if (proseOnly) { + appendEllipsis(); + } else { + resultLines.push(line); + } + continue; } - } else { - resultLines.push(line); } + resultLines.push(line); } const formatted = resultLines.join("\n"); - formatCacheKey = text; - formatCacheValue = formatted; + if (proseOnly) { + proseCacheKey = text; + proseCacheValue = formatted; + } else { + rawCacheKey = text; + rawCacheValue = formatted; + } return formatted; } @@ -99,9 +143,11 @@ export function hasDisplayableThinking( text: string | null | undefined, formattedText: string | null | undefined, ): boolean { - if (!text) return false; - if (!formattedText) return false; - return formattedText.length > 0 && canonicalizeMessage(text).length > 0; + if (!text || !formattedText) return false; + // Visibility keys off the formatted text: a block whose raw text is only + // comment noise (`\n`) formats to whitespace and stays hidden. The + // raw canonicalize check still hides dot/ellipsis-only placeholder blocks. + return formattedText.trim().length > 0 && canonicalizeMessage(text).length > 0; } /** Whether an assistant message contains thinking content the TUI can reveal. */ diff --git a/packages/coding-agent/test/acp-agent.test.ts b/packages/coding-agent/test/acp-agent.test.ts index 556e32813..7e9554119 100644 --- a/packages/coding-agent/test/acp-agent.test.ts +++ b/packages/coding-agent/test/acp-agent.test.ts @@ -124,6 +124,7 @@ class FakeAgentSession { } promptCalls: string[] = []; customMessages: Array<{ customType: string; content: string; details?: unknown }> = []; + customMessageOptions: Array<{ streamingBehavior?: "steer" | "followUp"; queueChipText?: string } | undefined> = []; skillsSettings = { enableSkillCommands: true }; skills: Array<{ name: string; description: string; filePath: string; baseDir: string; source: string }> = []; planModeState: PlanModeState | undefined; @@ -235,8 +236,12 @@ class FakeAgentSession { this.isStreaming = false; } - async promptCustomMessage(message: { customType: string; content: string; details?: unknown }): Promise { + async promptCustomMessage( + message: { customType: string; content: string; details?: unknown }, + options?: { streamingBehavior?: "steer" | "followUp"; queueChipText?: string }, + ): Promise { this.customMessages.push(message); + this.customMessageOptions.push(options); this.isStreaming = true; const assistantMessage = makeAssistantMessage("skill pong"); for (const listener of this.#listeners) { @@ -1013,6 +1018,115 @@ describe("ACP agent", () => { await Bun.sleep(0); }); + it("delivers the final visible answer when agent_end overtakes the assistant message_end (#4902)", async () => { + const harness = await createHarness(); + const created = await harness.agent.newSession({ cwd: harness.cwdA, mcpServers: [] }); + const session = harness.findSession(created.sessionId); + if (!session) throw new Error("session not registered"); + + // Live turn as observed through the prompt subscription when the + // fire-and-forget assistant message_end handler loses the race against + // the agent_end flush: thinking streams, then the turn ends. No + // text_delta and no message_end ever reach this subscriber — the final + // text exists only on the agent_end payload. + const assistantMessage = makeAssistantMessage("Final visible answer.", "Considering the greeting."); + session.prompt = async (text: string): Promise => { + session.promptCalls.push(text); + session.isStreaming = true; + for (const listener of session.listeners()) { + listener({ + type: "message_update", + message: assistantMessage, + assistantMessageEvent: { type: "thinking_delta", delta: "Considering the greeting." }, + } as AgentSessionEvent); + } + session.sessionManager.appendMessage(assistantMessage); + for (const listener of session.listeners()) { + listener({ type: "agent_end", messages: [assistantMessage] } as AgentSessionEvent); + } + session.isStreaming = false; + return true; + }; + + const response = await harness.agent.prompt({ + sessionId: created.sessionId, + prompt: [{ type: "text", text: "Say hello" }], + }); + expectAcpStructure(zPromptResponse, response); + expect(response.stopReason).toBe("end_turn"); + + const chunks = harness.updates.filter(update => update.sessionId === created.sessionId); + const thoughtChunks = chunks.filter(update => update.update.sessionUpdate === "agent_thought_chunk"); + const messageChunks = chunks.filter(update => update.update.sessionUpdate === "agent_message_chunk"); + expect(thoughtChunks).toHaveLength(1); + // The visible answer must reach the client exactly once even though the + // assistant message_end never arrived on this subscription. + expect(messageChunks).toHaveLength(1); + expect(messageChunks[0]?.update).toEqual( + expect.objectContaining({ + sessionUpdate: "agent_message_chunk", + content: { type: "text", text: "Final visible answer." }, + }), + ); + // Flushed answer belongs to the same live message as the thought chunk. + expect(getChunkMessageId(messageChunks[0]!)).toBe(getChunkMessageId(thoughtChunks[0]!)!); + expectAcpNotifications(harness.updates); + + harness.abortController.abort(); + await Bun.sleep(0); + }); + + it("does not duplicate the final answer when the assistant message_end arrives before agent_end", async () => { + // Companion to the #4902 regression: when message_end IS delivered, its + // fallback emission wins and the agent_end flush must stay silent. + const harness = await createHarness(); + const created = await harness.agent.newSession({ cwd: harness.cwdA, mcpServers: [] }); + const session = harness.findSession(created.sessionId); + if (!session) throw new Error("session not registered"); + + const assistantMessage = makeAssistantMessage("Composed offline.", "quiet planning"); + session.prompt = async (text: string): Promise => { + session.promptCalls.push(text); + session.isStreaming = true; + for (const listener of session.listeners()) { + listener({ + type: "message_update", + message: assistantMessage, + assistantMessageEvent: { type: "thinking_delta", delta: "quiet planning" }, + } as AgentSessionEvent); + } + for (const listener of session.listeners()) { + listener({ type: "message_end", message: assistantMessage } as AgentSessionEvent); + } + session.sessionManager.appendMessage(assistantMessage); + for (const listener of session.listeners()) { + listener({ type: "agent_end", messages: [assistantMessage] } as AgentSessionEvent); + } + session.isStreaming = false; + return true; + }; + + const response = await harness.agent.prompt({ + sessionId: created.sessionId, + prompt: [{ type: "text", text: "Say hello" }], + }); + expectAcpStructure(zPromptResponse, response); + + const messageChunks = harness.updates.filter( + update => update.sessionId === created.sessionId && update.update.sessionUpdate === "agent_message_chunk", + ); + expect(messageChunks).toHaveLength(1); + expect(messageChunks[0]?.update).toEqual( + expect.objectContaining({ + content: { type: "text", text: "Composed offline." }, + }), + ); + expectAcpNotifications(harness.updates); + + harness.abortController.abort(); + await Bun.sleep(0); + }); + it("replays assistant tool calls and matching results without duplicating the start", async () => { const harness = await createHarness(); const stored = new FakeAgentSession(harness.cwdA); @@ -1443,6 +1557,7 @@ describe("ACP agent", () => { expect(customMessage.content).toContain(`[Skill directory: ${skillDir}]`); expect(customMessage.content).toMatch(/[Rr]esolve any relative paths/); expect(customMessage.content).toContain("User: extra context"); + expect(session.customMessageOptions[0]).toEqual({ streamingBehavior: "steer" }); harness.abortController.abort(); await Bun.sleep(0); diff --git a/packages/coding-agent/test/agent-session-concurrent.test.ts b/packages/coding-agent/test/agent-session-concurrent.test.ts index 7b4b045d6..5d01d6b80 100644 --- a/packages/coding-agent/test/agent-session-concurrent.test.ts +++ b/packages/coding-agent/test/agent-session-concurrent.test.ts @@ -15,7 +15,7 @@ import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { AsyncJobManager } from "@oh-my-pi/pi-coding-agent/async"; import type { Rule } from "@oh-my-pi/pi-coding-agent/capability/rule"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; -import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { type SettingPath, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { TtsrManager } from "@oh-my-pi/pi-coding-agent/export/ttsr"; import type { ExtensionRunner } from "@oh-my-pi/pi-coding-agent/extensibility/extensions"; import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; @@ -68,7 +68,7 @@ describe("AgentSession concurrent prompt guard", () => { AsyncJobManager.resetForTests(); }); - async function createSession() { + async function createSession(settingsOverrides?: Partial>) { const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; let abortSignal: AbortSignal | undefined; @@ -100,7 +100,7 @@ describe("AgentSession concurrent prompt guard", () => { }); const sessionManager = SessionManager.inMemory(); - const settings = Settings.isolated(); + const settings = Settings.isolated(settingsOverrides); const authStorage = await AuthStorage.create(path.join(tempDir, "testauth.db")); authStorages.push(authStorage); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models.yml")); @@ -176,6 +176,86 @@ describe("AgentSession concurrent prompt guard", () => { await firstPrompt.catch(() => {}); }); + it("queues sendUserMessage as steer while streaming without AgentBusyError", async () => { + await createSession(); + + const firstPrompt = session.prompt("First message"); + await waitFor(() => session.isStreaming); + + // The first agent loop may dequeue a steer before the assertion runs, so + // observe agent.steer itself rather than the residual queue length. + const steered: AgentMessage[] = []; + const originalSteer = session.agent.steer.bind(session.agent); + session.agent.steer = (message: AgentMessage) => { + steered.push(message); + originalSteer(message); + }; + + // Extension path: no deliverAs while busy must queue, not throw. + await expect(session.sendUserMessage("hello from extension")).resolves.toBeUndefined(); + expect(steered).toHaveLength(1); + const queued = steered[0]; + expect(queued?.role).toBe("user"); + if (queued?.role === "user") { + expect(queued.content).toEqual([{ type: "text", text: "hello from extension" }]); + expect(queued.steering).toBe(true); + } + + session.agent.clearAllQueues(); + await session.abort(); + await firstPrompt.catch(() => {}); + }); + + it("sendUserMessage without deliverAs preserves prompt-flow keyword notices while streaming", async () => { + await createSession({ "magicKeywords.enabled": true, "magicKeywords.ultrathink": true }); + + const firstPrompt = session.prompt("First message"); + await waitFor(() => session.isStreaming); + + try { + await session.sendUserMessage("ultrathink fix via extension"); + const queuedShape = session.agent + .peekSteeringQueue() + .map(message => (message.role === "custom" ? message.customType : message.role)); + expect(queuedShape).toEqual(["ultrathink-notice", "user"]); + expect(session.getQueuedMessages()).toEqual({ + steering: ["ultrathink fix via extension"], + followUp: [], + }); + } finally { + session.agent.clearAllQueues(); + await session.abort(); + await firstPrompt.catch(() => {}); + } + }); + + it("sendUserMessage without deliverAs starts a normal prompt when idle", async () => { + await createSession(); + + let rejected: unknown; + let settled = false; + const turn = session + .sendUserMessage("Idle extension message") + .catch(error => { + rejected = error; + }) + .finally(() => { + settled = true; + }); + + try { + await waitFor(() => session.isStreaming || settled); + if (rejected) throw rejected; + + expect(session.isStreaming).toBe(true); + expect(settled).toBe(false); + expect(session.getQueuedMessages()).toEqual({ steering: [], followUp: [] }); + } finally { + await session.abort(); + await turn; + } + }); + it("delivers hidden nextTurn stop reactions through the next LLM call without exposing them in the visible queue", async () => { const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; let firstStream: AssistantMessageEventStream | undefined; diff --git a/packages/coding-agent/test/agent-session-eager-compaction.test.ts b/packages/coding-agent/test/agent-session-eager-compaction.test.ts index a4d0c2db2..cd8528ac9 100644 --- a/packages/coding-agent/test/agent-session-eager-compaction.test.ts +++ b/packages/coding-agent/test/agent-session-eager-compaction.test.ts @@ -2,7 +2,8 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import * as path from "node:path"; import { Agent, type AgentMessage, type AgentTool } from "@oh-my-pi/pi-agent-core"; import * as compactionModule from "@oh-my-pi/pi-agent-core/compaction"; -import type { TextContent } from "@oh-my-pi/pi-ai"; +import type { Model, TextContent } from "@oh-my-pi/pi-ai"; +import * as codexResponses from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; @@ -135,22 +136,23 @@ describe("AgentSession eager prelude re-injection after compaction", () => { async function createHarness( settingsOverride: Record = {}, - opts: { agentId?: string; agentKind?: "main" | "sub" } = {}, + opts: { agentId?: string; agentKind?: "main" | "sub"; model?: Model } = {}, ): Promise { const observedCalls: ObservedPromptCall[] = []; const waiters: Array<{ predicate: (call: ObservedPromptCall) => boolean; resolve: (call: ObservedPromptCall) => void; }> = []; - const bundled = getBundledModel("anthropic", "claude-sonnet-4-5"); - if (!bundled) throw new Error("Expected claude-sonnet-4-5 model to exist"); + const defaultModel = getBundledModel("anthropic", "claude-sonnet-4-5"); + if (!defaultModel) throw new Error("Expected claude-sonnet-4-5 model to exist"); + const selectedModel = opts.model ?? defaultModel; // Pin the window and output reservation: usage figures below trip the // context-full strategy at a 200k/64k threshold; catalog regeneration must // not shift the headroom math. - const model = { ...bundled, contextWindow: 200_000, maxTokens: 64_000 }; + const model = { ...selectedModel, contextWindow: 200_000, maxTokens: 64_000 }; const authStorage = await AuthStorage.create(path.join(tempDir.path(), `testauth-${cleanups.length}.db`)); - authStorage.setRuntimeApiKey("anthropic", "test-key"); + authStorage.setRuntimeApiKey(model.provider, "test-key"); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), `models-${cleanups.length}.yml`)); const settings = Settings.isolated({ "compaction.enabled": true, @@ -359,4 +361,25 @@ describe("AgentSession eager prelude re-injection after compaction", () => { expect(continuation.messageTexts.some(text => text.includes("Consider calling"))).toBe(false); expect(continuation.messageTexts.some(text => text.includes("You MUST call"))).toBe(false); }); + + it("resets Codex provider history after successful auto-compaction", async () => { + const model = getBundledModel("openai-codex", "gpt-5.6-terra"); + if (!model) throw new Error("Expected gpt-5.6-terra model to exist"); + const resetSpy = vi.spyOn(codexResponses, "resetOpenAICodexHistoryAfterCompaction"); + const { session, waitForCall } = await createHarness({}, { model }); + stubCompaction(); + + await runToContinuation(session, waitForCall); + + expect(resetSpy).toHaveBeenCalledTimes(1); + const reset = resetSpy.mock.calls[0]?.[0]; + if (!reset) throw new Error("Expected Codex compaction reset"); + expect(reset.providerSessionState).toBe(session.providerSessionState); + expect(reset.sessionId).toBe(session.sessionId); + expect(reset.compaction).toMatchObject({ + trigger: "auto", + reason: "context_limit", + phase: "pre_turn", + }); + }); }); diff --git a/packages/coding-agent/test/bundled-agent-parsing.test.ts b/packages/coding-agent/test/bundled-agent-parsing.test.ts new file mode 100644 index 000000000..a4da8e57d --- /dev/null +++ b/packages/coding-agent/test/bundled-agent-parsing.test.ts @@ -0,0 +1,63 @@ +import { describe, expect, it } from "bun:test"; +import { Effort } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { resolveAgentModelPatterns, resolveModelOverride } from "@oh-my-pi/pi-coding-agent/config/model-resolver"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { getBundledAgent } from "@oh-my-pi/pi-coding-agent/task/agents"; + +describe("bundled agent parsing", () => { + it("lets reviewer inherit thinking effort from its model role", () => { + const reviewer = getBundledAgent("reviewer"); + + expect(reviewer).toBeDefined(); + expect(reviewer?.source).toBe("bundled"); + expect(reviewer?.model).toEqual(["pi/slow"]); + expect(reviewer?.thinkingLevel).toBeUndefined(); + }); + + it("lets plan inherit thinking effort from its model role", () => { + const plan = getBundledAgent("plan"); + + expect(plan).toBeDefined(); + expect(plan?.source).toBe("bundled"); + expect(plan?.model).toEqual(["pi/plan", "pi/slow"]); + expect(plan?.thinkingLevel).toBeUndefined(); + }); + + // Issue #4761: with `modelRoles.slow: ...:xhigh`, the role's explicit effort + // suffix must survive agent-pattern expansion and model resolution for the + // bundled agents routed at that role. The executor picks + // `agent.thinkingLevel ?? resolvedThinkingLevel` (task/executor.ts), so a + // bundled frontmatter pin would mask the suffix — reviewer/plan declare none + // (asserted above) and the resolved level below is what the subagent runs at. + it("resolves the configured slow-role effort suffix for reviewer and plan", () => { + const gpt55 = buildModel({ + id: "gpt-5.5", + name: "GPT-5.5 Codex", + api: "openai-codex-responses", + provider: "openai-codex", + baseUrl: "https://chatgpt.com/backend-api/codex", + reasoning: true, + thinking: { mode: "effort", efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh] }, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 272000, + maxTokens: 128000, + }); + const settings = Settings.isolated({ + modelRoles: { slow: "openai-codex/gpt-5.5:xhigh", plan: "openai-codex/gpt-5.5:xhigh" }, + }); + const registry = { getAvailable: () => [gpt55] } as Parameters[1]; + + for (const name of ["reviewer", "plan"]) { + const agent = getBundledAgent(name); + expect(agent?.thinkingLevel).toBeUndefined(); + const patterns = resolveAgentModelPatterns({ agentModel: agent?.model, settings }); + const resolved = resolveModelOverride(patterns, registry, settings); + expect(resolved.model?.provider).toBe("openai-codex"); + expect(resolved.model?.id).toBe("gpt-5.5"); + expect(resolved.thinkingLevel).toBe(Effort.XHigh); + expect(resolved.explicitThinkingLevel).toBe(true); + } + }); +}); diff --git a/packages/coding-agent/test/compaction-prefer-current-model.test.ts b/packages/coding-agent/test/compaction-prefer-current-model.test.ts index 0bceded5f..9b6b5adc9 100644 --- a/packages/coding-agent/test/compaction-prefer-current-model.test.ts +++ b/packages/coding-agent/test/compaction-prefer-current-model.test.ts @@ -2,6 +2,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import * as path from "node:path"; import { Agent } from "@oh-my-pi/pi-agent-core"; import * as compactionModule from "@oh-my-pi/pi-agent-core/compaction"; +import * as AIError from "@oh-my-pi/pi-ai/error"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; @@ -99,6 +100,76 @@ describe("compaction prefers the current session model over modelRoles.default", expect(`${firstCandidate.provider}/${firstCandidate.id}`).toBe(`${currentModel.provider}/${currentModel.id}`); }); + it("falls back when the authenticated Bedrock candidate cannot resolve AWS credentials", async () => { + const currentModel = getBundledModel("amazon-bedrock", "global.anthropic.claude-opus-4-6-v1"); + const fallbackModel = getBundledModel("anthropic", "claude-sonnet-4-5"); + if (!currentModel || !fallbackModel) { + throw new Error("Expected bundled test models to exist"); + } + + const settings = Settings.isolated({ "compaction.keepRecentTokens": 1, "compaction.strategy": "context-full" }); + settings.setModelRole("smol", `${fallbackModel.provider}/${fallbackModel.id}`); + + const agent = new Agent({ + initialState: { + model: currentModel, + systemPrompt: ["Test"], + tools: [], + messages: [], + }, + }); + + authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db")); + authStorage.setRuntimeApiKey(currentModel.provider, "bedrock-credentials"); + authStorage.setRuntimeApiKey(fallbackModel.provider, "anthropic-token"); + modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); + + session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings, + modelRegistry, + }); + session.subscribe(() => {}); + + for (const [userText, assistantText] of [ + ["first question", "first answer"], + ["second question", "second answer"], + ] as const) { + const user = userMsg(userText); + const assistant = assistantMsg(assistantText); + session.agent.appendMessage(user); + session.sessionManager.appendMessage(user); + session.agent.appendMessage(assistant); + session.sessionManager.appendMessage(assistant); + } + + const compactSpy = vi.spyOn(compactionModule, "compact").mockImplementation(async (preparation, model) => { + if (model.provider === currentModel.provider && model.id === currentModel.id) { + throw new AIError.AwsCredentialsError("opaque provider setup failure", "resolution"); + } + if (model.provider !== fallbackModel.provider || model.id !== fallbackModel.id) { + throw new Error(`Unexpected compaction model ${model.provider}/${model.id}`); + } + return { + summary: "fallback summary", + shortSummary: "fallback short summary", + firstKeptEntryId: preparation.firstKeptEntryId, + tokensBefore: 42, + details: { provider: model.provider }, + }; + }); + + const result = await session.compact(); + + expect(result.summary).toBe("fallback summary"); + expect(compactSpy).toHaveBeenCalledTimes(2); + expect(compactSpy.mock.calls.map(([, model]) => `${model.provider}/${model.id}`)).toEqual([ + `${currentModel.provider}/${currentModel.id}`, + `${fallbackModel.provider}/${fallbackModel.id}`, + ]); + }); + it("uses compactionModel only for the summary call and leaves the active model unchanged", async () => { const baseCurrentModel = getBundledModel("anthropic", "claude-sonnet-4-5"); const compactionModel = getBundledModel("openai", "gpt-5"); diff --git a/packages/coding-agent/test/extensions-runner.test.ts b/packages/coding-agent/test/extensions-runner.test.ts index d24518737..0a2680515 100644 --- a/packages/coding-agent/test/extensions-runner.test.ts +++ b/packages/coding-agent/test/extensions-runner.test.ts @@ -1169,6 +1169,7 @@ describe("ExtensionRunner", () => { setEditorText: () => {}, getEditorText: () => "", editor: async () => undefined, + addAutocompleteProvider: () => {}, setEditorComponent: () => {}, get theme() { return {} as never; diff --git a/packages/coding-agent/test/input-controller-escape.test.ts b/packages/coding-agent/test/input-controller-escape.test.ts index e7ae3af47..d2828106d 100644 --- a/packages/coding-agent/test/input-controller-escape.test.ts +++ b/packages/coding-agent/test/input-controller-escape.test.ts @@ -269,6 +269,16 @@ function abortViewSession(ctx: InteractiveModeContext): AbortViewSession { // so property access is explicit. return ctx.viewSession as unknown as AbortViewSession; } + +type MutableSessionState = InteractiveModeContext["session"] & { + isStreaming: boolean; +}; + +function mutableSessionState(ctx: InteractiveModeContext): MutableSessionState { + // Test harness installs a mutable fake AgentSession; keep the unchecked cast named + // so state mutations are explicit. + return ctx.session as MutableSessionState; +} beforeEach(async () => { await Settings.init({ inMemory: true }); }); @@ -438,111 +448,37 @@ describe("InputController escape behavior", () => { expect(spies.abort).not.toHaveBeenCalled(); }); - it("requires a second Esc within two seconds to abort streaming", () => { - const now = vi.spyOn(Date, "now"); - now.mockReturnValue(1_000); + it("aborts an active streaming turn on the first Esc without asking for confirmation", () => { const { ctx, editor, spies } = createContext(); - (ctx.session as { isStreaming: boolean }).isStreaming = true; + mutableSessionState(ctx).isStreaming = true; const controller = new InputController(ctx); controller.setupKeyHandlers(); editor.onEscape?.(); + expect(spies.abort).toHaveBeenCalledTimes(1); + expect(spies.abort).toHaveBeenCalledWith({ reason: USER_INTERRUPT_LABEL }); + expect(spies.showStatus).not.toHaveBeenCalledWith("Press Esc again within 2s to cancel streaming."); + }); + + it("aborts the submitted turn on the first Esc once the main session starts streaming", async () => { + const { ctx, editor, spies } = createContext(); + const submission = createSubmission({ text: "fix issue #4921" }); + spies.startPendingSubmission.mockReturnValue(submission); + const controller = new InputController(ctx); + + controller.setupKeyHandlers(); + controller.setupEditorSubmitHandler(); + await editor.onSubmit?.("fix issue #4921"); + mutableSessionState(ctx).isStreaming = true; + ctx.loadingAnimation = undefined; + + editor.onEscape?.(); + expect(spies.cancelPendingSubmission).not.toHaveBeenCalled(); - expect(spies.clearQueue).not.toHaveBeenCalled(); - expect(spies.abort).not.toHaveBeenCalled(); - expect(spies.showStatus).toHaveBeenCalledWith("Press Esc again within 2s to cancel streaming."); - - now.mockReturnValue(2_500); - editor.onEscape?.(); - expect(spies.abort).toHaveBeenCalledTimes(1); expect(spies.abort).toHaveBeenCalledWith({ reason: USER_INTERRUPT_LABEL }); - }); - - it("expires the streaming Esc arm instead of aborting on a late second press", () => { - const now = vi.spyOn(Date, "now"); - now.mockReturnValue(1_000); - const { ctx, editor, spies } = createContext(); - (ctx.session as { isStreaming: boolean }).isStreaming = true; - const controller = new InputController(ctx); - - controller.setupKeyHandlers(); - editor.onEscape?.(); - now.mockReturnValue(3_001); - editor.onEscape?.(); - - expect(spies.abort).not.toHaveBeenCalled(); - expect(spies.showStatus).toHaveBeenCalledTimes(2); - }); - - it("preserves the streaming Esc arm when streamingComponent appears between presses", () => { - // Pre-`message_start`: first Esc arms on the per-turn sentinel. `message_start` - // then publishes `ctx.streamingComponent`; the second Esc must still abort the - // same live turn instead of re-arming on the new component reference. - const now = vi.spyOn(Date, "now"); - now.mockReturnValue(1_000); - const { ctx, editor, spies } = createContext(); - (ctx.session as { isStreaming: boolean }).isStreaming = true; - const controller = new InputController(ctx); - - controller.setupKeyHandlers(); - editor.onEscape?.(); - (ctx as unknown as { streamingComponent: object }).streamingComponent = {}; - now.mockReturnValue(1_500); - editor.onEscape?.(); - - expect(spies.abort).toHaveBeenCalledTimes(1); - expect(spies.abort).toHaveBeenCalledWith({ reason: USER_INTERRUPT_LABEL }); - }); - - it("aborts on the second Esc even when ctx.streamingMessage was replaced by a delta in between", () => { - // `EventController` replaces `ctx.streamingMessage` with a fresh immutable - // snapshot on every `message_update`; the per-turn sentinel is unaffected so - // swapping the message must not invalidate the armed token. - const now = vi.spyOn(Date, "now"); - now.mockReturnValue(1_000); - const { ctx, editor, spies } = createContext(); - (ctx.session as { isStreaming: boolean }).isStreaming = true; - (ctx as unknown as { streamingComponent: object }).streamingComponent = {}; - (ctx as unknown as { streamingMessage: object }).streamingMessage = { content: [] }; - const controller = new InputController(ctx); - - controller.setupKeyHandlers(); - editor.onEscape?.(); - (ctx as unknown as { streamingMessage: object }).streamingMessage = { content: ["delta"] }; - now.mockReturnValue(1_500); - editor.onEscape?.(); - - expect(spies.abort).toHaveBeenCalledTimes(1); - expect(spies.abort).toHaveBeenCalledWith({ reason: USER_INTERRUPT_LABEL }); - }); - - it("clears the streaming Esc arm when the current turn ends", () => { - const now = vi.spyOn(Date, "now"); - now.mockReturnValue(1_000); - const { ctx, editor, spies, sessionListeners } = createContext(); - (ctx.session as { isStreaming: boolean }).isStreaming = true; - const controller = new InputController(ctx); - - controller.setupKeyHandlers(); - // Fallback arm (no streamingMessage/streamingComponent yet — pre-message_start). - editor.onEscape?.(); - expect(sessionListeners).toHaveLength(1); - - // Turn 1 ends; a new turn starts. session.subscribe receives both transitions, - // either of which must invalidate the still-armed fallback token so it cannot - // fast-abort the new turn's first Esc. - for (const listener of sessionListeners) { - listener({ type: "agent_end" }); - listener({ type: "agent_start" }); - } - - now.mockReturnValue(1_500); - editor.onEscape?.(); - - expect(spies.abort).not.toHaveBeenCalled(); - expect(spies.showStatus).toHaveBeenCalledTimes(2); + expect(spies.showStatus).not.toHaveBeenCalledWith("Press Esc again within 2s to cancel streaming."); }); it("returns focused subagent view to main on Esc instead of aborting", () => { diff --git a/packages/coding-agent/test/internal-urls/memory-protocol.test.ts b/packages/coding-agent/test/internal-urls/memory-protocol.test.ts index b88eee52d..a0efc7bee 100644 --- a/packages/coding-agent/test/internal-urls/memory-protocol.test.ts +++ b/packages/coding-agent/test/internal-urls/memory-protocol.test.ts @@ -238,6 +238,59 @@ describe("MemoryProtocolHandler — mnemopi bridge (issue #4443)", () => { }); }); + it("resolves memory:// to a read-only fact row (issue #4725)", async () => { + await withMnemopiSession(async ({ state }) => { + const beam = state.memory.beam; + beam.db + .prepare( + "INSERT INTO facts (fact_id, session_id, subject, predicate, object, timestamp, confidence) VALUES (?, ?, ?, ?, ?, ?, ?)", + ) + .run( + "0473bbdb8da6df92", + beam.sessionId, + "Glab", + "works-without", + "mise prefix", + "2026-07-01T00:00:00.000Z", + 0.9, + ); + + const router = InternalUrlRouter.instance(); + const resource = await router.resolve("memory://0473bbdb8da6df92"); + + expect(resource.content).toContain("id: 0473bbdb8da6df92"); + expect(resource.content).toContain("store: fact"); + expect(resource.content).toContain("Glab works-without mise prefix"); + }); + }); + + it("reports not_editable (not not_found) for memory_edit ops on a fact id (issue #4725)", async () => { + await withMnemopiSession(async ({ state }) => { + const beam = state.memory.beam; + beam.db + .prepare( + "INSERT INTO facts (fact_id, session_id, subject, predicate, object, timestamp, confidence) VALUES (?, ?, ?, ?, ?, ?, ?)", + ) + .run("fact-readonly", beam.sessionId, "service", "uses", "postgres", "2026-07-01T00:00:00.000Z", 0.9); + + expect(state.editScopedMemory("update", "fact-readonly", { content: "x" })).toMatchObject({ + status: "not_editable", + store: "fact", + }); + expect(state.editScopedMemory("forget", "fact-readonly")).toMatchObject({ + status: "not_editable", + store: "fact", + }); + expect(state.editScopedMemory("invalidate", "fact-readonly")).toMatchObject({ + status: "not_editable", + store: "fact", + }); + + // The fact row itself is untouched by the rejected edits. + expect(beam.db.prepare("SELECT fact_id FROM facts WHERE fact_id = ?").get("fact-readonly")).not.toBeNull(); + }); + }); + it("routes memory://root to the file-backed summary even when mnemopi is active", async () => { await withMnemopiSession(async () => { const router = InternalUrlRouter.instance(); diff --git a/packages/coding-agent/test/issue-4919-extension-autocomplete-provider.test.ts b/packages/coding-agent/test/issue-4919-extension-autocomplete-provider.test.ts new file mode 100644 index 000000000..31fb9c42d --- /dev/null +++ b/packages/coding-agent/test/issue-4919-extension-autocomplete-provider.test.ts @@ -0,0 +1,256 @@ +/** + * Issue #4919: a pi extension calling `ctx.ui.addAutocompleteProvider(...)` in its + * `session_start` handler crashed at load under omp — the method was absent from + * `ExtensionUIContext`, so the call threw `TypeError: ... is not a function` and + * (for extensions that wrap init in try/catch, e.g. @ff-labs/pi-fff) aborted the + * extension's entire initialization. + * + * These tests pin the pi-compatible contract: + * - headless contexts accept the factory as a no-op instead of throwing, and + * - interactive mode stacks each factory on top of the built-in editor provider. + * + * NOTE: imports are relative (`../src/...`) so the tests exercise this checkout + * even when `node_modules/@oh-my-pi/pi-coding-agent` resolves elsewhere. + */ + +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { Agent, type AgentTool } from "@oh-my-pi/pi-agent-core"; +import { type Api, Effort, type Model } from "@oh-my-pi/pi-ai"; +import type { AutocompleteProvider } from "@oh-my-pi/pi-tui"; +import { logger, TempDir } from "@oh-my-pi/pi-utils"; +import { type } from "arktype"; +import { ModelRegistry } from "../src/config/model-registry"; +import { resetSettingsForTest, Settings } from "../src/config/settings"; +import { loadExtensions } from "../src/extensibility/extensions/loader"; +import { ExtensionRunner } from "../src/extensibility/extensions/runner"; +import { InteractiveMode } from "../src/modes/interactive-mode"; +import { initTheme } from "../src/modes/theme/theme"; +import { AgentSession } from "../src/session/agent-session"; +import { AuthStorage } from "../src/session/auth-storage"; +import { SessionManager } from "../src/session/session-manager"; + +function makeTool(name: string): AgentTool { + return { + name, + label: name, + description: `Fake ${name}`, + parameters: type({}), + async execute() { + return { content: [{ type: "text" as const, text: "ok" }] }; + }, + }; +} + +/** + * Wrap `current` the way a well-behaved pi extension does: contribute items for + * its own trigger prefix, delegate everything else to the wrapped provider. + */ +function makeWrappingFactory(tag: string): (current: AutocompleteProvider) => AutocompleteProvider { + return current => ({ + async getSuggestions(lines, cursorLine, cursorCol) { + const line = lines[cursorLine] ?? ""; + if (line.startsWith("##")) { + const base = await current.getSuggestions(lines, cursorLine, cursorCol); + return { + items: [...(base?.items ?? []), { value: tag, label: tag }], + prefix: base?.prefix ?? line.slice(0, cursorCol), + }; + } + return current.getSuggestions(lines, cursorLine, cursorCol); + }, + applyCompletion(lines, cursorLine, cursorCol, item, prefix) { + return current.applyCompletion(lines, cursorLine, cursorCol, item, prefix); + }, + }); +} + +describe("extension autocomplete provider API (#4919)", () => { + let tempDir: TempDir; + let authStorage: AuthStorage; + let registry: ModelRegistry; + let model: Model; + let tools: AgentTool[]; + let originalHome: string | undefined; + let mode: InteractiveMode | undefined; + let session: AgentSession | undefined; + + beforeAll(async () => { + initTheme(); + resetSettingsForTest(); + // One empty temp dir doubles as the project cwd and the (isolated) home + // directory, keeping `refreshSlashCommandState`'s capability scan off the + // real home dir (mirrors the prompt-template autocomplete harness). + tempDir = TempDir.createSync("@pi-ext-autocomplete-"); + originalHome = process.env.HOME; + process.env.HOME = tempDir.path(); + await Settings.init({ inMemory: true, cwd: tempDir.path() }); + Settings.instance.set("startup.quiet", true); + authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db")); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + registry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); + const resolved = registry.find("anthropic", "claude-sonnet-4-5"); + if (!resolved) throw new Error("Expected anthropic model claude-sonnet-4-5 to exist"); + model = resolved; + tools = [makeTool("read")]; + }); + + beforeEach(() => { + vi.spyOn(os, "homedir").mockReturnValue(tempDir.path()); + }); + + afterEach(async () => { + vi.restoreAllMocks(); + mode?.stop(); + await session?.dispose(); + mode = undefined; + session = undefined; + }); + + afterAll(() => { + authStorage?.close(); + if (originalHome === undefined) delete process.env.HOME; + else process.env.HOME = originalHome; + tempDir?.removeSync(); + resetSettingsForTest(); + }); + + function createHarness(): { mode: InteractiveMode; session: AgentSession } { + const manager = SessionManager.create(tempDir.path(), path.join(tempDir.path(), `active-${Bun.nanoseconds()}`)); + const created = new AgentSession({ + agent: new Agent({ + initialState: { + model, + systemPrompt: ["Test"], + tools, + messages: [], + thinkingLevel: Effort.Medium, + }, + }), + sessionManager: manager, + settings: Settings.isolated({ "compaction.enabled": false }), + modelRegistry: registry, + toolRegistry: new Map(tools.map(tool => [tool.name, tool])), + promptTemplates: [], + }); + const createdMode = new InteractiveMode(created, "test"); + session = created; + mode = createdMode; + return { mode: createdMode, session: created }; + } + + function captureAutocompleteProvider(target: InteractiveMode): { current: AutocompleteProvider | undefined } { + const slot: { current: AutocompleteProvider | undefined } = { current: undefined }; + vi.spyOn(target.editor, "setAutocompleteProvider").mockImplementation(provider => { + slot.current = provider; + }); + return slot; + } + + it("does not abort a session_start handler that registers a provider without UI", async () => { + // Mimics @ff-labs/pi-fff: registerAutocompleteProvider(ctx) runs first and + // unconditionally inside the try/catch that guards the whole init routine. + const extensionsDir = path.join(tempDir.path(), "runner-extensions"); + fs.mkdirSync(extensionsDir, { recursive: true }); + const markerPath = path.join(extensionsDir, "init-marker.txt"); + const extPath = path.join(extensionsDir, "fff-like.ts"); + fs.writeFileSync( + extPath, + `import * as fs from "node:fs"; +export default function (pi) { + pi.on("session_start", async (_event, ctx) => { + try { + ctx.ui.addAutocompleteProvider((current) => current); + // "Rest of init" — on baseline the call above throws and this never runs. + fs.writeFileSync(${JSON.stringify(markerPath)}, "initialized"); + } catch (error) { + fs.writeFileSync( + ${JSON.stringify(markerPath)}, + "failed: " + (error instanceof Error ? error.message : String(error)), + ); + } + }); +} +`, + ); + + const result = await loadExtensions([extPath], tempDir.path()); + expect(result.errors).toEqual([]); + const runner = new ExtensionRunner( + result.extensions, + result.runtime, + tempDir.path(), + SessionManager.inMemory(), + registry, + ); + const surfaced: string[] = []; + runner.onError(error => { + surfaced.push(error.error); + }); + + await runner.emit({ type: "session_start" }); + + expect(surfaced).toEqual([]); + expect(fs.readFileSync(markerPath, "utf8")).toBe("initialized"); + }); + + it("stacks extension factories on top of the built-in editor provider", async () => { + const created = createHarness(); + const slot = captureAutocompleteProvider(created.mode); + + // Registration before the first refresh (session_start fires before init's + // refreshSlashCommandState) must land once the base provider exists. + created.mode.addAutocompleteProvider(makeWrappingFactory("##fff-first")); + await created.mode.refreshSlashCommandState(tempDir.path()); + + const provider = slot.current; + expect(provider).toBeDefined(); + + // The extension's trigger prefix surfaces its items... + const extension = await provider!.getSuggestions(["##"], 0, 2); + expect(extension?.items.map(item => item.value)).toContain("##fff-first"); + + // ...while built-in slash completion still flows through the wrapper. + const slash = await provider!.getSuggestions(["/"], 0, 1); + expect(slash?.items.map(item => item.value)).toContain("model"); + + // Registration after the refresh re-applies immediately, preserving the chain. + created.mode.addAutocompleteProvider(makeWrappingFactory("##fff-second")); + const restacked = slot.current; + expect(restacked).toBeDefined(); + expect(restacked).not.toBe(provider); + + const chained = await restacked!.getSuggestions(["##"], 0, 2); + const values = chained?.items.map(item => item.value) ?? []; + expect(values).toContain("##fff-first"); + expect(values).toContain("##fff-second"); + }); + + it("skips broken factories without losing core autocomplete or healthy wrappers", async () => { + const created = createHarness(); + const slot = captureAutocompleteProvider(created.mode); + const warnSpy = vi.spyOn(logger, "warn").mockImplementation(() => {}); + + created.mode.addAutocompleteProvider(() => { + throw new Error("boom"); + }); + created.mode.addAutocompleteProvider(() => ({}) as AutocompleteProvider); + created.mode.addAutocompleteProvider(makeWrappingFactory("##healthy")); + await created.mode.refreshSlashCommandState(tempDir.path()); + + const provider = slot.current; + expect(provider).toBeDefined(); + + const slash = await provider!.getSuggestions(["/"], 0, 1); + expect(slash?.items.map(item => item.value)).toContain("model"); + + const extension = await provider!.getSuggestions(["##"], 0, 2); + expect(extension?.items.map(item => item.value)).toContain("##healthy"); + + expect(warnSpy.mock.calls.some(([message]) => String(message).includes("autocomplete provider factory"))).toBe( + true, + ); + }); +}); diff --git a/packages/coding-agent/test/keybindings-migration.test.ts b/packages/coding-agent/test/keybindings-migration.test.ts index ec113449f..bd54ff5a2 100644 --- a/packages/coding-agent/test/keybindings-migration.test.ts +++ b/packages/coding-agent/test/keybindings-migration.test.ts @@ -4,13 +4,32 @@ import * as os from "node:os"; import * as path from "node:path"; import { KeybindingsManager } from "@oh-my-pi/pi-coding-agent/config/keybindings"; import { matchesAppFollowUp } from "@oh-my-pi/pi-coding-agent/modes/utils/keybinding-matchers"; -import { setKeybindings } from "@oh-my-pi/pi-tui"; -import { removeWithRetries } from "@oh-my-pi/pi-utils"; +import { type KeybindingsConfig, setKeybindings } from "@oh-my-pi/pi-tui"; +import { + __resetDirsFromEnvForTests, + getAgentDir, + getProfileRootDir, + removeWithRetries, + setProfile, +} from "@oh-my-pi/pi-utils"; import { YAML } from "bun"; function ctrl(key: string): string { return String.fromCharCode(key.toLowerCase().charCodeAt(0) & 31); } + +async function writeKeybindingsYaml(agentDir: string, config: KeybindingsConfig): Promise { + await fs.mkdir(agentDir, { recursive: true }); + await Bun.write(path.join(agentDir, "keybindings.yml"), YAML.stringify(config, null, 2)); +} + +function restoreEnvValue(key: string, value: string | undefined): void { + if (value === undefined) { + delete process.env[key]; + } else { + process.env[key] = value; + } +} describe("KeybindingsManager.create", () => { beforeEach(() => { setKeybindings(KeybindingsManager.inMemory()); @@ -149,6 +168,117 @@ describe("KeybindingsManager.create", () => { } }); + it("inherits default user keybindings for a named profile without a profile keybindings file (#4867)", async () => { + const rootDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-keybindings-profile-")); + const defaultAgentDir = path.join(rootDir, "default", "agent"); + const profileAgentDir = path.join(rootDir, "profiles", "work", "agent"); + + await writeKeybindingsYaml(defaultAgentDir, { + "app.session.fork": "ctrl+f", + "tui.editor.deleteCharBackward": ["backspace", "ctrl+h"], + }); + + try { + const manager = KeybindingsManager.create(profileAgentDir, { inheritedAgentDir: defaultAgentDir }); + + expect(manager.getKeys("app.session.fork")).toEqual(["ctrl+f"]); + expect(manager.getKeys("tui.editor.deleteCharBackward")).toEqual(["backspace", "ctrl+h"]); + } finally { + await removeWithRetries(rootDir); + } + }); + + it("merges default user keybindings with profile overrides for a named profile (#4867)", async () => { + const rootDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-keybindings-profile-")); + const defaultAgentDir = path.join(rootDir, "default", "agent"); + const profileAgentDir = path.join(rootDir, "profiles", "work", "agent"); + + await writeKeybindingsYaml(defaultAgentDir, { + "app.session.fork": "ctrl+f", + "app.session.new": "ctrl+n", + }); + await writeKeybindingsYaml(profileAgentDir, { + "app.session.fork": "alt+f", + "app.clipboard.copyLine": "alt+l", + }); + + try { + const manager = KeybindingsManager.create(profileAgentDir, { inheritedAgentDir: defaultAgentDir }); + + expect(manager.getKeys("app.session.new")).toEqual(["ctrl+n"]); + expect(manager.getKeys("app.session.fork")).toEqual(["alt+f"]); + expect(manager.getKeys("app.clipboard.copyLine")).toEqual(["alt+l"]); + } finally { + await removeWithRetries(rootDir); + } + }); + + it("never writes migration output into the inherited default agent dir (#4867)", async () => { + const rootDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-keybindings-profile-")); + const defaultAgentDir = path.join(rootDir, "default", "agent"); + const profileAgentDir = path.join(rootDir, "profiles", "work", "agent"); + + // Legacy JSON in the default dir: loading it with a write-back path would + // materialize keybindings.yml there. The inherited load must stay read-only. + await fs.mkdir(defaultAgentDir, { recursive: true }); + await Bun.write( + path.join(defaultAgentDir, "keybindings.json"), + JSON.stringify({ "app.session.fork": "ctrl+f" }, null, 2), + ); + + try { + const manager = KeybindingsManager.create(profileAgentDir, { inheritedAgentDir: defaultAgentDir }); + + expect(manager.getKeys("app.session.fork")).toEqual(["ctrl+f"]); + expect(await Bun.file(path.join(defaultAgentDir, "keybindings.yml")).exists()).toBe(false); + } finally { + await removeWithRetries(rootDir); + } + }); + + it("merges default user keybindings when create uses the active profile with no arguments (#4867)", async () => { + const originalConfigDir = process.env.PI_CONFIG_DIR; + const originalAgentDirEnv = process.env.PI_CODING_AGENT_DIR; + const originalOmpProfile = process.env.OMP_PROFILE; + const originalPiProfile = process.env.PI_PROFILE; + const configRootDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-keybindings-active-profile-")); + + try { + process.env.PI_CONFIG_DIR = path.relative(os.homedir(), configRootDir); + restoreEnvValue("PI_CODING_AGENT_DIR", originalAgentDirEnv); + restoreEnvValue("OMP_PROFILE", originalOmpProfile); + restoreEnvValue("PI_PROFILE", originalPiProfile); + __resetDirsFromEnvForTests(); + + const defaultAgentDir = path.join(getProfileRootDir(undefined), "agent"); + const profileAgentDir = path.join(getProfileRootDir("work"), "agent"); + await writeKeybindingsYaml(defaultAgentDir, { + "app.session.fork": "ctrl+f", + "app.session.new": "ctrl+n", + }); + await writeKeybindingsYaml(profileAgentDir, { + "app.session.fork": "alt+f", + "app.clipboard.copyLine": "alt+l", + }); + + setProfile("work"); + + expect(getAgentDir()).toBe(profileAgentDir); + const manager = KeybindingsManager.create(); + + expect(manager.getKeys("app.session.new")).toEqual(["ctrl+n"]); + expect(manager.getKeys("app.session.fork")).toEqual(["alt+f"]); + expect(manager.getKeys("app.clipboard.copyLine")).toEqual(["alt+l"]); + } finally { + restoreEnvValue("PI_CONFIG_DIR", originalConfigDir); + restoreEnvValue("PI_CODING_AGENT_DIR", originalAgentDirEnv); + restoreEnvValue("OMP_PROFILE", originalOmpProfile); + restoreEnvValue("PI_PROFILE", originalPiProfile); + __resetDirsFromEnvForTests(); + await removeWithRetries(configRootDir); + } + }); + it("defaults model selection to Alt+M and display reset to Ctrl+L", () => { const manager = KeybindingsManager.inMemory(); diff --git a/packages/coding-agent/test/model-discovery.test.ts b/packages/coding-agent/test/model-discovery.test.ts index dd7d89913..393e60645 100644 --- a/packages/coding-agent/test/model-discovery.test.ts +++ b/packages/coding-agent/test/model-discovery.test.ts @@ -3,6 +3,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { Effort, type FetchImpl, type Model } from "@oh-my-pi/pi-ai"; +import type { OAuthCredentials } from "@oh-my-pi/pi-ai/oauth/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache"; import type { OpenAICompat } from "@oh-my-pi/pi-catalog/types"; @@ -19,15 +20,18 @@ describe("ModelRegistry runtime discovery", () => { let originalOllamaBaseUrl: string | undefined; let originalOllamaHost: string | undefined; let originalOllamaContextLength: string | undefined; + let originalAnthropicApiKey: string | undefined; beforeEach(async () => { resetSettingsForTest(); originalOllamaBaseUrl = Bun.env.OLLAMA_BASE_URL; originalOllamaHost = Bun.env.OLLAMA_HOST; originalOllamaContextLength = Bun.env.OLLAMA_CONTEXT_LENGTH; + originalAnthropicApiKey = Bun.env.ANTHROPIC_API_KEY; delete Bun.env.OLLAMA_BASE_URL; delete Bun.env.OLLAMA_HOST; delete Bun.env.OLLAMA_CONTEXT_LENGTH; + delete Bun.env.ANTHROPIC_API_KEY; tempDir = path.join(os.tmpdir(), `pi-test-model-registry-${Snowflake.next()}`); fs.mkdirSync(tempDir, { recursive: true }); modelsJsonPath = path.join(tempDir, "models.json"); @@ -55,6 +59,11 @@ describe("ModelRegistry runtime discovery", () => { } else { Bun.env.OLLAMA_CONTEXT_LENGTH = originalOllamaContextLength; } + if (originalAnthropicApiKey === undefined) { + delete Bun.env.ANTHROPIC_API_KEY; + } else { + Bun.env.ANTHROPIC_API_KEY = originalAnthropicApiKey; + } authStorage.close(); if (tempDir && fs.existsSync(tempDir)) { removeSyncWithRetries(tempDir); @@ -115,6 +124,197 @@ describe("ModelRegistry runtime discovery", () => { }; } + async function useAuthStorageWithRefreshTracker() { + authStorage.close(); + const refreshCalls: string[] = []; + authStorage = await AuthStorage.create(":memory:", { + refreshOAuthCredential: async (provider, _credentialId, credential): Promise => { + refreshCalls.push(provider); + return { + ...credential, + access: provider === "anthropic" ? "sk-ant-oat-fresh-anthropic" : `fresh-${provider}`, + expires: Date.now() + 3_600_000, + }; + }, + }); + return { refreshCalls }; + } + + type AnthropicDiscoveryCapture = { + modelListAuthorization?: string | null; + modelListXApiKey?: string | null; + modelListCalls: number; + }; + + function mockAnthropicModelsDiscovery(capture: AnthropicDiscoveryCapture): FetchImpl { + const endpointPrefix = "https://api.anthropic.com/"; + return async (input, init) => { + const url = String(input); + if (url === "https://models.dev/api.json") { + return Response.json({}); + } + if (url.startsWith(endpointPrefix) && url.endsWith("/models")) { + const headers = new Headers(init?.headers); + capture.modelListAuthorization = headers.get("authorization"); + capture.modelListXApiKey = headers.get("x-api-key"); + capture.modelListCalls++; + return Response.json({ + data: [{ id: "claude-regression-4893", display_name: "Claude Regression 4893" }], + }); + } + throw new Error(`Unexpected URL: ${url}`); + }; + } + + test("refreshProvider online refreshes expired anthropic OAuth before model discovery", async () => { + const { refreshCalls } = await useAuthStorageWithRefreshTracker(); + await authStorage.set("anthropic", { + type: "oauth", + access: "sk-ant-oat-expired-anthropic", + refresh: "refresh-anthropic", + expires: Date.now() - 60_000, + }); + const capture: AnthropicDiscoveryCapture = { modelListCalls: 0 }; + const registry = new ModelRegistry(authStorage, modelsJsonPath, { + fetch: mockAnthropicModelsDiscovery(capture), + }); + + await registry.refreshProvider("anthropic", "online"); + + expect(refreshCalls).toEqual(["anthropic"]); + expect(capture.modelListCalls).toBe(1); + expect(capture.modelListAuthorization).toBe("Bearer sk-ant-oat-fresh-anthropic"); + expect(capture.modelListXApiKey).toBeNull(); + expect(registry.find("anthropic", "claude-regression-4893")).toBeDefined(); + }); + + test("refreshProvider online does not refresh unrelated expired OAuth credentials", async () => { + const { refreshCalls } = await useAuthStorageWithRefreshTracker(); + await authStorage.set("anthropic", { + type: "oauth", + access: "sk-ant-oat-expired-anthropic", + refresh: "refresh-anthropic", + expires: Date.now() - 60_000, + }); + await authStorage.set("openai", { + type: "oauth", + access: "expired-openai", + refresh: "refresh-openai", + expires: Date.now() - 60_000, + }); + const capture: AnthropicDiscoveryCapture = { modelListCalls: 0 }; + const registry = new ModelRegistry(authStorage, modelsJsonPath, { + fetch: mockAnthropicModelsDiscovery(capture), + }); + + await registry.refreshProvider("anthropic", "online"); + + expect(refreshCalls).toEqual(["anthropic"]); + expect(authStorage.getOAuthCredential("openai")?.access).toBe("expired-openai"); + expect(capture.modelListCalls).toBe(1); + }); + + test("refreshProvider offline does not touch expired OAuth credentials", async () => { + const { refreshCalls } = await useAuthStorageWithRefreshTracker(); + await authStorage.set("anthropic", { + type: "oauth", + access: "sk-ant-oat-expired-anthropic", + refresh: "refresh-anthropic", + expires: Date.now() - 60_000, + }); + const registry = new ModelRegistry(authStorage, modelsJsonPath, { + fetch: async input => { + throw new Error(`Offline discovery should not fetch ${String(input)}`); + }, + }); + + await registry.refreshProvider("anthropic", "offline"); + + expect(refreshCalls).toEqual([]); + expect(authStorage.getOAuthCredential("anthropic")?.access).toBe("sk-ant-oat-expired-anthropic"); + }); + test("online-if-uncached refreshes expired OAuth when the discovery cache is stale for the model manager", async () => { + const { refreshCalls } = await useAuthStorageWithRefreshTracker(); + await authStorage.set("anthropic", { + type: "oauth", + access: "sk-ant-oat-expired-anthropic", + refresh: "refresh-anthropic", + expires: Date.now() - 60_000, + }); + // Older than the model manager's 2h default TTL: the manager WILL fetch, + // so the preflight must mint a fresh bearer first. + writeModelCache("anthropic", Date.now() - 3 * 60 * 60 * 1000, [], true, "", cacheDbPath); + const capture: AnthropicDiscoveryCapture = { modelListCalls: 0 }; + const registry = new ModelRegistry(authStorage, modelsJsonPath, { + fetch: mockAnthropicModelsDiscovery(capture), + }); + + await registry.refreshProvider("anthropic", "online-if-uncached"); + + expect(refreshCalls).toEqual(["anthropic"]); + expect(capture.modelListCalls).toBe(1); + expect(capture.modelListAuthorization).toBe("Bearer sk-ant-oat-fresh-anthropic"); + }); + + test("online-if-uncached leaves expired OAuth untouched when the discovery cache is fresh", async () => { + const { refreshCalls } = await useAuthStorageWithRefreshTracker(); + await authStorage.set("anthropic", { + type: "oauth", + access: "sk-ant-oat-expired-anthropic", + refresh: "refresh-anthropic", + expires: Date.now() - 60_000, + }); + // Fresh authoritative cache: the manager will not fetch, so opening a + // cached model selector must not rotate (or risk disabling) credentials. + writeModelCache("anthropic", Date.now() - 60_000, [], true, "", cacheDbPath); + const capture: AnthropicDiscoveryCapture = { modelListCalls: 0 }; + const registry = new ModelRegistry(authStorage, modelsJsonPath, { + fetch: mockAnthropicModelsDiscovery(capture), + }); + + await registry.refreshProvider("anthropic", "online-if-uncached"); + + expect(refreshCalls).toEqual([]); + expect(capture.modelListCalls).toBe(0); + expect(authStorage.getOAuthCredential("anthropic")?.access).toBe("sk-ant-oat-expired-anthropic"); + }); + + test("configured discovery suppresses built-in special OAuth discovery", async () => { + await authStorage.set("google-gemini-cli", { + type: "oauth", + access: "fresh-google-gemini-cli", + refresh: "refresh-google-gemini-cli", + expires: Date.now() + 3_600_000, + }); + writeRawModelsJson({ + "google-gemini-cli": { + baseUrl: "http://127.0.0.1:4893", + api: "openai-completions", + auth: "none", + discovery: { type: "openai-models-list" }, + }, + }); + const unexpectedUrls: string[] = []; + const fetchMock: FetchImpl = async input => { + const url = String(input); + if (url === "http://127.0.0.1:4893/v1/models") { + return Response.json({ + data: [{ id: "configured-gemini-cli-model", context_length: 65_536 }], + }); + } + unexpectedUrls.push(url); + throw new Error(`Unexpected URL: ${url}`); + }; + const registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: fetchMock }); + + await registry.refreshProvider("google-gemini-cli", "online"); + + expect(unexpectedUrls).toEqual([]); + const configuredModel = registry.find("google-gemini-cli", "configured-gemini-cli-model"); + expect(configuredModel?.baseUrl).toBe("http://127.0.0.1:4893"); + expect(configuredModel?.contextWindow).toBe(65_536); + }); + test("auto-discovers ollama models without provider config", async () => { const fetchMock = mockOllamaDiscovery(["phi4-mini"]); const registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: fetchMock }); diff --git a/packages/coding-agent/test/model-registry-runtime-provider.test.ts b/packages/coding-agent/test/model-registry-runtime-provider.test.ts index 467cc0ec8..977e8a665 100644 --- a/packages/coding-agent/test/model-registry-runtime-provider.test.ts +++ b/packages/coding-agent/test/model-registry-runtime-provider.test.ts @@ -261,6 +261,52 @@ describe("ModelRegistry runtime provider registration", () => { }); }); + test("configured discovery suppresses extension fetchDynamicModels for the same provider", async () => { + const providerName = "runtime-configured-provider"; + fs.writeFileSync( + modelsJsonPath, + JSON.stringify({ + providers: { + [providerName]: { + baseUrl: "http://127.0.0.1:4893", + api: "openai-completions", + auth: "none", + discovery: { type: "openai-models-list" }, + }, + }, + }), + ); + const configuredFetch: FetchImpl = async input => { + const url = String(input); + if (url === "http://127.0.0.1:4893/v1/models") { + return Response.json({ + data: [{ id: "shared-runtime-model", context_length: 32_768 }], + }); + } + throw new Error(`Unexpected URL: ${url}`); + }; + const configuredRegistry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: configuredFetch }); + let runtimeFetchCalls = 0; + configuredRegistry.registerProvider( + providerName, + { + baseUrl: "https://runtime.example.com/v1", + apiKey: "RUNTIME_KEY", + api: "openai-completions", + fetchDynamicModels: async () => { + runtimeFetchCalls++; + return [{ ...baseModel, id: "shared-runtime-model", contextWindow: 999_999 }]; + }, + }, + "ext://runtime", + ); + + await configuredRegistry.refreshProvider(providerName, "online"); + + expect(runtimeFetchCalls).toBe(0); + expect(configuredRegistry.find(providerName, "shared-runtime-model")?.contextWindow).toBe(32_768); + }); + test("refreshRuntimeProviders times out extension fetchDynamicModels that never resolves", async () => { vi.useFakeTimers(); const hangingFetch = Promise.withResolvers[number][]>(); diff --git a/packages/coding-agent/test/modes/components/transcript-container.test.ts b/packages/coding-agent/test/modes/components/transcript-container.test.ts index 73f6bbaa8..ae08beade 100644 --- a/packages/coding-agent/test/modes/components/transcript-container.test.ts +++ b/packages/coding-agent/test/modes/components/transcript-container.test.ts @@ -727,3 +727,163 @@ describe("TranscriptContainer renderViewportTail", () => { expect([...container.renderViewportTail(W, 0)]).toEqual([]); }); }); + +// A displaceable snapshot (todo/poll card): kept unfinalized only so a matching +// follow-up call can retract it. Mirrors ToolExecutionComponent.seal — sealing +// finalizes the block in place and it stops reporting displaceable. A pending +// tool starts non-displaceable and becomes a displaceable snapshot only when +// its successful result arrives (`makeDisplaceable`). +class DisplaceableBlock implements Component { + sealCount = 0; + #sealed = false; + #displaceable: boolean; + #lines: string[]; + constructor(lines: string[], displaceable = true) { + this.#lines = lines; + this.#displaceable = displaceable; + } + makeDisplaceable(): void { + this.#displaceable = true; + } + isTranscriptBlockFinalized(): boolean { + return this.#sealed; + } + isDisplaceableBlock(): boolean { + return this.#displaceable && !this.#sealed; + } + seal(): void { + this.sealCount++; + this.#sealed = true; + } + invalidate(): void {} + render(_width: number): string[] { + return [...this.#lines]; + } +} + +// Seal-on-commit: rows on the native-scrollback tape are immutable, so once the +// commit boundary covers any of a displaceable snapshot's rows the container +// must seal it in place — retracting it would strand an orphaned copy in +// terminal history, and left unfinalized it would pin the live-region seam +// open. setNativeScrollbackCommittedRows is a pure store; the seal walk runs at +// the start of the NEXT render, over the previous frame's segments (the +// geometry the committed count was computed against), before the seam scan so +// the seam unpins in that same frame. +describe("TranscriptContainer seal-on-commit", () => { + const W = 40; + + // history(0) | sep(1) | todo-header(2) | todo-body(3); the leading + // separator row belongs to the card's segment (segment.startRow = 1). + function cardAfterHistory(displaceable = true): { container: TranscriptContainer; card: DisplaceableBlock } { + const container = new TranscriptContainer(); + container.addChild(new MutableBlock(["history"])); + const card = new DisplaceableBlock(["todo-header", "todo-body"], displaceable); + container.addChild(card); + expect(container.render(W)).toEqual(["history", "", "todo-header", "todo-body"]); + return { container, card }; + } + + it("seals on the next render once the boundary covers the block's rows", () => { + const { container, card } = cardAfterHistory(); + // The unsealed card pins the live-region seam at its own rows. + expect(container.getNativeScrollbackLiveRegionStart()).toBe(2); + // Rows 0..2 (through the card's header) are immutable history now. + container.setNativeScrollbackCommittedRows(3); + // The setter is a pure store: sealing waits for the next compose. + expect(card.sealCount).toBe(0); + container.render(W); + expect(card.sealCount).toBe(1); + expect(container.isBlockUncommitted(card)).toBe(false); + // The seal pre-pass ran before the seam scan: the SAME render already + // reports the seam unpinned (no still-mutating block left). + expect(container.getNativeScrollbackLiveRegionStart()).toBeUndefined(); + }); + + it("does not seal while the boundary stays above the block", () => { + const { container, card } = cardAfterHistory(); + // Only "history" committed; the card's rows are all still retractable. + container.setNativeScrollbackCommittedRows(1); + container.render(W); + expect(card.sealCount).toBe(0); + expect(container.isBlockUncommitted(card)).toBe(true); + }); + + it("never seals across same-value or decreasing republishes above the block", () => { + const { container, card } = cardAfterHistory(); + // The engine republishes the committed count every frame (compose and + // post-emit): repeated same-value and decreasing stores above the + // card's rows never accumulate into a seal. + container.setNativeScrollbackCommittedRows(1); + container.render(W); + container.setNativeScrollbackCommittedRows(1); + container.render(W); + container.setNativeScrollbackCommittedRows(0); + container.render(W); + expect(card.sealCount).toBe(0); + expect(container.isBlockUncommitted(card)).toBe(true); + }); + + it("seals exactly once as the boundary sweeps past the block in stages", () => { + const container = new TranscriptContainer(); + container.addChild(new MutableBlock(["history"])); + const card = new DisplaceableBlock(["todo-header", "todo-body"]); + container.addChild(card); + container.addChild(new MutableBlock(["tail"])); + expect(container.render(W)).toEqual(["history", "", "todo-header", "todo-body", "", "tail"]); + // First crossing (through the header) seals; the sealed block stops + // reporting displaceable. + container.setNativeScrollbackCommittedRows(3); + container.render(W); + expect(card.sealCount).toBe(1); + // A later sweep past the whole block, and every subsequent render at + // that boundary, must not seal again. + container.setNativeScrollbackCommittedRows(6); + container.render(W); + container.render(W); + expect(card.sealCount).toBe(1); + }); + + it("never seals a displaceable block with an empty contribution", () => { + const container = new TranscriptContainer(); + container.addChild(new MutableBlock(["history"])); + const empty = new DisplaceableBlock([]); + container.addChild(empty); + container.addChild(new MutableBlock(["tail"])); + expect(container.render(W)).toEqual(["history", "", "tail"]); + // None of the block's rows are on the tape: nothing to seal, ever. + container.setNativeScrollbackCommittedRows(3); + container.render(W); + expect(empty.sealCount).toBe(0); + expect(container.isBlockUncommitted(empty)).toBe(true); + }); + + it("walks past blocks without the displaceable protocol", () => { + const container = new TranscriptContainer(); + const plain = new MutableBlock(["plain-block"]); + container.addChild(plain); + const card = new DisplaceableBlock(["todo-header"]); + container.addChild(card); + expect(container.render(W)).toEqual(["plain-block", "", "todo-header"]); + container.setNativeScrollbackCommittedRows(3); + // The pre-pass visits the plain block first (its rows also committed); + // absent duck-typed methods are a no-op and the card below still seals. + container.render(W); + expect(card.sealCount).toBe(1); + expect(container.isBlockUncommitted(plain)).toBe(false); + }); + + it("seals a block that became displaceable after its rows committed", () => { + // A pending tool's preview rows scroll into native scrollback before + // its successful result arrives; only then does the block become a + // displaceable snapshot. The walk runs every render, so the flip is + // caught on the next compose — not only when the boundary moves. + const { container, card } = cardAfterHistory(false); + container.setNativeScrollbackCommittedRows(3); + container.render(W); + // Rows committed while not displaceable: nothing to seal yet. + expect(card.sealCount).toBe(0); + card.makeDisplaceable(); + container.render(W); + expect(card.sealCount).toBe(1); + }); +}); diff --git a/packages/coding-agent/test/modes/controllers/event-controller-toolcall-finalize.test.ts b/packages/coding-agent/test/modes/controllers/event-controller-toolcall-finalize.test.ts index cd89df02b..01688768a 100644 --- a/packages/coding-agent/test/modes/controllers/event-controller-toolcall-finalize.test.ts +++ b/packages/coding-agent/test/modes/controllers/event-controller-toolcall-finalize.test.ts @@ -40,6 +40,10 @@ function makeStreamingMessage(content: AssistantMessage["content"]): AssistantMe }; } +// Components the controller mounts during a dispatch (pending tool previews). +// Sealed in afterEach so their spinner intervals never outlive the test file. +const mountedComponents: { seal?(): void }[] = []; + function createFixture(streamingMessage: AssistantMessage) { const markTranscriptBlockFinalized = vi.fn(); const streamingComponent = { @@ -49,14 +53,14 @@ function createFixture(streamingMessage: AssistantMessage) { const ctx = { isInitialized: true, init: vi.fn(async () => {}), - ui: { requestRender: vi.fn() }, + ui: { requestRender: vi.fn(), requestComponentRender: vi.fn() }, statusLine: { invalidate: vi.fn() }, updateEditorTopBorder: vi.fn(), streamingComponent, streamingMessage, pendingTools: new Map(), noteDisplayableThinkingContent: vi.fn(() => false), - chatContainer: { addChild: vi.fn() }, + chatContainer: { addChild: vi.fn((child: { seal?(): void }) => mountedComponents.push(child)) }, toolOutputExpanded: false, settings, session: { getToolByName: () => undefined }, @@ -84,6 +88,7 @@ async function dispatchUpdate(message: AssistantMessage) { describe("EventController finalizes assistant block when tool-call args stream", () => { afterEach(() => { + for (const component of mountedComponents.splice(0)) component.seal?.(); resetSettingsForTest(); vi.restoreAllMocks(); }); diff --git a/packages/coding-agent/test/read-tool-group.test.ts b/packages/coding-agent/test/read-tool-group.test.ts index e5b6c43da..f1d6da276 100644 --- a/packages/coding-agent/test/read-tool-group.test.ts +++ b/packages/coding-agent/test/read-tool-group.test.ts @@ -250,6 +250,45 @@ describe("ReadToolGroupComponent", () => { expect(extractLinkTexts(rendered)).not.toContain("src/example.ts:7-9"); }); + it("renders separate selector grouped summary paths while linking only the base path", () => { + settings.override("tui.hyperlinks", "always"); + const component = new ReadToolGroupComponent(); + const resolvedPath = path.resolve("/workspace/src/grouped.ts"); + component.updateArgs({ path: "src/grouped.ts", selector: "2-3" }, "read-split-selector"); + component.updateResult( + { + content: [{ type: "text", text: "line 2" }], + details: { meta: { source: { type: "path", value: resolvedPath } } }, + }, + false, + "read-split-selector", + ); + + const rendered = component.render(120).join("\n"); + + const groupedUri = new URL(url.pathToFileURL(path.resolve(resolvedPath)).href); + groupedUri.searchParams.set("line", "2"); + expect(Bun.stripANSI(rendered)).toContain("Read src/grouped.ts:2-3"); + expect(extractLinkUris(rendered)).toContain(groupedUri.href); + expect(extractLinkTexts(rendered)).toContain("src/grouped.ts"); + expect(extractLinkTexts(rendered)).not.toContain("src/grouped.ts:2-3"); + }); + + it("ignores non-string selectors from malformed runtime args", () => { + const component = new ReadToolGroupComponent(); + const malformedArgs = { path: "src/example.ts", selector: 10 } as unknown as { + path: string; + selector: string; + }; + + expect(() => component.updateArgs(malformedArgs, "read-malformed-selector")).not.toThrow(); + + const plain = Bun.stripANSI(component.render(120).join("\n")); + + expect(plain).toContain("Read src/example.ts"); + expect(plain).not.toContain("src/example.ts:10"); + }); + it("links inline preview titles when the summary row is suppressed", () => { settings.override("tui.hyperlinks", "always"); const component = new ReadToolGroupComponent({ showContentPreview: true }); diff --git a/packages/coding-agent/test/rpc-skill-command.test.ts b/packages/coding-agent/test/rpc-skill-command.test.ts index 068b55bfb..1100ace66 100644 --- a/packages/coding-agent/test/rpc-skill-command.test.ts +++ b/packages/coding-agent/test/rpc-skill-command.test.ts @@ -16,6 +16,7 @@ describe("tryRunRpcSkillCommand", () => { ); let message: Pick | undefined; + let options: { streamingBehavior?: "steer" | "followUp" } | undefined; const handled = await tryRunRpcSkillCommand( { @@ -23,8 +24,9 @@ describe("tryRunRpcSkillCommand", () => { skills: [ { name: "reviewer", description: "Review code", filePath: skillPath, baseDir: dir, source: "project" }, ], - async promptCustomMessage(nextMessage: typeof message) { + async promptCustomMessage(nextMessage: typeof message, nextOptions?: typeof options) { message = nextMessage; + options = nextOptions; }, }, "/skill:reviewer focus on risks", @@ -39,10 +41,49 @@ describe("tryRunRpcSkillCommand", () => { expect(message?.content).toContain("User: focus on risks"); expect(message?.display).toBe(true); expect(message?.attribution).toBe("user"); + expect(options).toEqual({ streamingBehavior: "steer" }); await removeWithRetries(dir); }); + test("honors the RPC prompt streaming behavior for registered /skill commands", async () => { + const dir = await fs.mkdtemp(path.join(os.tmpdir(), `omp-rpc-skill-${Snowflake.next()}-`)); + const skillPath = path.join(dir, "SKILL.md"); + await Bun.write( + skillPath, + "---\nname: reviewer\ndescription: Review code\n---\n\nReview the supplied code carefully.\n", + ); + + let options: { streamingBehavior?: "steer" | "followUp" } | undefined; + try { + const handled = await tryRunRpcSkillCommand( + { + skillsSettings: { enableSkillCommands: true }, + skills: [ + { + name: "reviewer", + description: "Review code", + filePath: skillPath, + baseDir: dir, + source: "project", + }, + ], + async promptCustomMessage(nextMessage, nextOptions) { + expect(nextMessage.customType).toBe(SKILL_PROMPT_MESSAGE_TYPE); + options = nextOptions; + }, + }, + "/skill:reviewer wait for the current turn", + "followUp", + ); + + expect(handled).toEqual({ agentInvoked: true }); + expect(options?.streamingBehavior).toBe("followUp"); + } finally { + await removeWithRetries(dir); + } + }); + test("ignores unknown skill commands so normal prompt handling can continue", async () => { const handled = await tryRunRpcSkillCommand( { diff --git a/packages/coding-agent/test/session-fork-prompt-cache-key.test.ts b/packages/coding-agent/test/session-fork-prompt-cache-key.test.ts new file mode 100644 index 000000000..1c64a15e4 --- /dev/null +++ b/packages/coding-agent/test/session-fork-prompt-cache-key.test.ts @@ -0,0 +1,233 @@ +import { describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as path from "node:path"; +import { ThinkingLevel } from "@oh-my-pi/pi-agent-core"; +import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; +import { type Args, parseArgs } from "@oh-my-pi/pi-coding-agent/cli/args"; +import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; +import type { ScopedModel } from "@oh-my-pi/pi-coding-agent/config/model-resolver"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { buildSessionOptions } from "@oh-my-pi/pi-coding-agent/main"; +import { type CreateAgentSessionOptions, createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk"; +import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; +import { CURRENT_SESSION_VERSION, type SessionHeader } from "@oh-my-pi/pi-coding-agent/session/session-entries"; +import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; +import { TempDir } from "@oh-my-pi/pi-utils"; + +const OPENAI_TEST_MODEL = getBundledModel("openai", "gpt-4o-mini"); + +interface ArgsWithPromptCacheKey extends Args { + providerPromptCacheKey?: string; +} + +interface SourceSessionFixture { + cwd: string; + sourceFile: string; + sourceHeader: SessionHeader; + forkSessionDir: string; +} + +async function createSourceSessionFixture(tempDir: TempDir, parentId: string): Promise { + const cwd = tempDir.join("project"); + const sourceDir = tempDir.join("source-sessions"); + const forkSessionDir = tempDir.join("forked-sessions"); + await fs.mkdir(cwd, { recursive: true }); + await fs.mkdir(sourceDir, { recursive: true }); + await fs.mkdir(forkSessionDir, { recursive: true }); + const sourceFile = path.join(sourceDir, `${parentId}.jsonl`); + const sourceHeader: SessionHeader = { + type: "session", + version: CURRENT_SESSION_VERSION, + id: parentId, + timestamp: new Date().toISOString(), + cwd, + }; + await Bun.write(sourceFile, `${JSON.stringify(sourceHeader)}\n`); + return { cwd, sourceFile, sourceHeader, forkSessionDir }; +} + +async function createMinimalSession( + tempDir: TempDir, + options: CreateAgentSessionOptions, +): Promise<{ session: AgentSession; authStorage: AuthStorage }> { + const authStorage = await AuthStorage.create(tempDir.join("sdk-auth.db")); + authStorage.setRuntimeApiKey("openai", "test-key"); + const shouldSupplyModel = options.sessionManager?.getHeader()?.parentSession === undefined; + const result = await createAgentSession({ + ...options, + cwd: options.cwd ?? tempDir.path(), + agentDir: tempDir.path(), + authStorage, + modelRegistry: undefined, + model: shouldSupplyModel ? (options.model ?? OPENAI_TEST_MODEL) : options.model, + settings: Settings.isolated({ + "async.enabled": false, + "marketplace.autoUpdate": "off", + }), + disableExtensionDiscovery: true, + preloadedExtensions: undefined, + skills: [], + contextFiles: [], + promptTemplates: [], + slashCommands: [], + workspaceTree: { + rootPath: options.cwd ?? tempDir.path(), + rendered: "", + truncated: false, + totalLines: 0, + agentsMdFiles: [], + }, + enableMCP: false, + enableLsp: false, + ...(options.toolNames !== undefined ? { toolNames: options.toolNames } : {}), + }); + return { session: result.session, authStorage }; +} + +describe("provider prompt-cache key session affinity", () => { + it("parses --prompt-cache-key without folding it into provider session id or prompt text", () => { + const parsed = parseArgs([ + "--provider-session-id", + "provider-lineage", + "--prompt-cache-key", + "cache-affinity", + "hello", + ]); + const promptCacheArgs: ArgsWithPromptCacheKey = parsed; + + expect(parsed.providerSessionId).toBe("provider-lineage"); + expect(promptCacheArgs.providerPromptCacheKey).toBe("cache-affinity"); + expect(parsed.messages).toEqual(["hello"]); + expect(parsed.unrecognizedFlags).toEqual([]); + }); + + it("creates an agent whose prompt-cache key can differ from provider request lineage", async () => { + using tempDir = TempDir.createSync("@omp-prompt-cache-sdk-"); + let session: AgentSession | undefined; + let authStorage: AuthStorage | undefined; + try { + const created = await createMinimalSession(tempDir, { + providerSessionId: "provider-lineage", + providerPromptCacheKey: "cache-affinity", + sessionManager: SessionManager.inMemory(tempDir.path()), + }); + session = created.session; + authStorage = created.authStorage; + + expect(session.agent.sessionId).toBe("provider-lineage"); + expect(session.agent.promptCacheKey).toBe("cache-affinity"); + expect(session.agent.promptCacheKey).not.toBe(session.agent.sessionId); + } finally { + await session?.dispose(); + authStorage?.close(); + } + }); + + it("initializes a full fork with child request lineage and parent prompt-cache affinity", async () => { + using tempDir = TempDir.createSync("@omp-prompt-cache-fork-"); + const source = await createSourceSessionFixture(tempDir, "parent-cache-session"); + const forkedManager = await SessionManager.forkFrom(source.sourceFile, source.cwd, source.forkSessionDir); + let session: AgentSession | undefined; + let authStorage: AuthStorage | undefined; + try { + const created = await createMinimalSession(tempDir, { + cwd: source.cwd, + sessionManager: forkedManager, + }); + session = created.session; + authStorage = created.authStorage; + const childSessionId = forkedManager.getSessionId(); + + expect(forkedManager.getHeader()?.parentSession).toBe(source.sourceHeader.id); + expect(childSessionId).toBeString(); + expect(childSessionId).not.toBe(source.sourceHeader.id); + expect(session.agent.sessionId).toBe(childSessionId); + expect(session.agent.promptCacheKey).toBe(source.sourceHeader.id); + expect(session.agent.promptCacheKey).not.toBe(session.agent.sessionId); + } finally { + await session?.dispose(); + authStorage?.close(); + } + }); + + it("does not auto-inherit parent prompt-cache affinity when fork startup changes request-shaping inputs", async () => { + const cases: Array<{ name: string; options: CreateAgentSessionOptions }> = [ + { + name: "model", + options: { model: OPENAI_TEST_MODEL }, + }, + { + name: "thinking", + options: { thinkingLevel: ThinkingLevel.High }, + }, + { + name: "system", + options: { customSystemPrompt: "Use a different provider prompt." }, + }, + { + name: "tools", + options: { toolNames: ["read"] }, + }, + ]; + + for (const entry of cases) { + using tempDir = TempDir.createSync(`@omp-prompt-cache-fork-${entry.name}-`); + const source = await createSourceSessionFixture(tempDir, `parent-cache-session-${entry.name}`); + const forkedManager = await SessionManager.forkFrom(source.sourceFile, source.cwd, source.forkSessionDir); + let session: AgentSession | undefined; + let authStorage: AuthStorage | undefined; + try { + const created = await createMinimalSession(tempDir, { + ...entry.options, + cwd: source.cwd, + sessionManager: forkedManager, + }); + session = created.session; + authStorage = created.authStorage; + + expect(forkedManager.getHeader()?.parentSession).toBe(source.sourceHeader.id); + expect(session.agent.promptCacheKey, entry.name).toBeUndefined(); + } finally { + await session?.dispose(); + authStorage?.close(); + } + } + }); + + it("does not pre-pin parent prompt-cache affinity when a scoped model selects the startup route", async () => { + using tempDir = TempDir.createSync("@omp-prompt-cache-scoped-model-"); + const source = await createSourceSessionFixture(tempDir, "parent-cache-session-scoped"); + const forkedManager = await SessionManager.forkFrom(source.sourceFile, source.cwd, source.forkSessionDir); + const authStorage = await AuthStorage.create(tempDir.join("scoped-auth.db")); + authStorage.setRuntimeApiKey(OPENAI_TEST_MODEL.provider, "test-key"); + try { + const modelRegistry = new ModelRegistry(authStorage, tempDir.join("models.yml")); + const parsed = parseArgs([ + "--cwd", + source.cwd, + "--models", + `${OPENAI_TEST_MODEL.provider}/${OPENAI_TEST_MODEL.id}`, + ]); + const scopedModels: ScopedModel[] = [ + { + model: OPENAI_TEST_MODEL, + explicitThinkingLevel: false, + }, + ]; + + const options = await buildSessionOptions( + parsed, + scopedModels, + forkedManager, + modelRegistry, + Settings.isolated({ "marketplace.autoUpdate": "off" }), + ); + + expect(options.model).toBe(OPENAI_TEST_MODEL); + expect(options.providerPromptCacheKey).toBeUndefined(); + } finally { + authStorage.close(); + } + }); +}); diff --git a/packages/coding-agent/test/settings-manager.test.ts b/packages/coding-agent/test/settings-manager.test.ts index a8bf84f85..daf6e15a0 100644 --- a/packages/coding-agent/test/settings-manager.test.ts +++ b/packages/coding-agent/test/settings-manager.test.ts @@ -70,6 +70,51 @@ describe("Settings", () => { await Bun.sleep(0); await tempDir?.remove(); }); + + describe("main config file selection", () => { + it("loads and updates an existing config.yaml without creating config.yml", async () => { + const yamlConfigPath = path.join(agentDir, "config.yaml"); + await Bun.write(yamlConfigPath, YAML.stringify({ setupVersion: 1 }, null, 2)); + + const settings = await Settings.init({ cwd: projectDir, agentDir }); + expect(settings.get("setupVersion")).toBe(1); + + settings.set("setupVersion", 2); + await settings.flush(); + + const savedSettings = YAML.parse(await Bun.file(yamlConfigPath).text()) as Record; + expect(savedSettings.setupVersion).toBe(2); + expect(await Bun.file(getConfigPath()).exists()).toBe(false); + }); + + it("clones the selected config.yaml path for persisted settings", async () => { + const yamlConfigPath = path.join(agentDir, "config.yaml"); + await Bun.write(yamlConfigPath, YAML.stringify({ setupVersion: 1 }, null, 2)); + + const settings = await Settings.init({ cwd: projectDir, agentDir }); + const cloned = await settings.cloneForCwd(tempDir.join("other-project")); + + cloned.set("setupVersion", 2); + await cloned.flush(); + + const savedSettings = YAML.parse(await Bun.file(yamlConfigPath).text()) as Record; + expect(savedSettings.setupVersion).toBe(2); + expect(await Bun.file(getConfigPath()).exists()).toBe(false); + }); + + it("creates config.yml for new persisted settings when no main config exists", async () => { + const yamlConfigPath = path.join(agentDir, "config.yaml"); + + const settings = await Settings.init({ cwd: projectDir, agentDir }); + settings.set("setupVersion", 1); + await settings.flush(); + + expect(await Bun.file(getConfigPath()).exists()).toBe(true); + expect(await Bun.file(yamlConfigPath).exists()).toBe(false); + expect((await readSettings()).setupVersion).toBe(1); + }); + }); + describe("defaults", () => { it("keeps eight inline images live by default", async () => { const settings = await Settings.init({ cwd: projectDir, agentDir }); diff --git a/packages/coding-agent/test/tool-execution-write-repaint.test.ts b/packages/coding-agent/test/tool-execution-write-repaint.test.ts new file mode 100644 index 000000000..d011987e5 --- /dev/null +++ b/packages/coding-agent/test/tool-execution-write-repaint.test.ts @@ -0,0 +1,140 @@ +import { afterEach, beforeAll, describe, expect, it, vi } from "bun:test"; +import { ToolExecutionComponent } from "@oh-my-pi/pi-coding-agent/modes/components/tool-execution"; +import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import { type Component, TUI } from "@oh-my-pi/pi-tui"; +import { StressRenderScheduler } from "../../tui/test/render-stress-scheduler"; +import { VirtualTerminal } from "../../tui/test/virtual-terminal"; + +function writeArgs(lineCount: number) { + return { + path: "notes.txt", + content: Array.from({ length: lineCount }, (_, i) => `line ${i + 1}`).join("\n"), + }; +} + +function partialWriteResult(text = "Writing notes.txt...") { + return { content: [{ type: "text", text }] }; +} + +class Footer implements Component { + constructor(readonly rows: number) {} + invalidate(): void {} + render(_width: number): string[] { + return Array.from({ length: this.rows }, (_, i) => `editor-${i}`); + } +} + +function plainBuffer(term: VirtualTerminal): string[] { + return term + .getScrollBuffer() + .map(row => Bun.stripANSI(row).trimEnd()) + .filter(Boolean); +} + +describe("ToolExecutionComponent write repaint seam", () => { + const components: ToolExecutionComponent[] = []; + + beforeAll(async () => { + await initTheme(); + }); + + afterEach(() => { + for (const component of components) component.stopAnimation(); + components.length = 0; + vi.restoreAllMocks(); + }); + + function makeComponent(args: unknown) { + const resetDisplay = vi.fn(); + const ui = { requestRender() {}, requestComponentRender() {}, resetDisplay } as unknown as TUI; + const component = new ToolExecutionComponent("write", args, {}, undefined, ui); + components.push(component); + resetDisplay.mockClear(); + return { component, resetDisplay }; + } + + it("forces a viewport repaint when a painted collapsed tail window receives its first result", () => { + // 20 lines > WRITE_STREAMING_PREVIEW_LINES (12): the pending preview is a + // tail window the first-result render re-anchors to the top of the file. + const { component, resetDisplay } = makeComponent(writeArgs(20)); + component.render(80); + + component.updateResult(partialWriteResult(), true); + + expect(resetDisplay).toHaveBeenCalledTimes(1); + }); + + it("does not repaint when the pending tail window never reaches the terminal", () => { + const { component, resetDisplay } = makeComponent(writeArgs(20)); + // No render() before the result: a resetDisplay here would wipe native + // scrollback for a shape the user never saw. + + component.updateResult(partialWriteResult(), true); + + expect(resetDisplay).not.toHaveBeenCalled(); + }); + + it("does not repaint a collapsed preview that fits the streaming window", () => { + // 12 lines render top-anchored without a tail window, so the first result + // does not re-anchor the frame; wiping scrollback would be gratuitous. + const { component, resetDisplay } = makeComponent(writeArgs(12)); + component.render(80); + + component.updateResult(partialWriteResult(), true); + + expect(resetDisplay).not.toHaveBeenCalled(); + }); + + it("does not repaint an expanded pending preview", () => { + // Expanded previews show the whole file top-anchored — no tail window to + // re-anchor. + const { component, resetDisplay } = makeComponent(writeArgs(20)); + component.setExpanded(true); + component.render(80); + + component.updateResult(partialWriteResult(), true); + + expect(resetDisplay).not.toHaveBeenCalled(); + }); + + it("removes stale pending tail rows from the terminal buffer when the first partial result arrives", async () => { + const term = new VirtualTerminal(80, 8, 1_000); + const scheduler = new StressRenderScheduler(); + const tui = new TUI(term, undefined, { renderScheduler: scheduler }); + const component = new ToolExecutionComponent("write", writeArgs(20), {}, undefined, tui); + components.push(component); + tui.addChild(component); + tui.addChild(new Footer(5)); + + try { + tui.start(); + await scheduler.drain(term); + const pendingRows = plainBuffer(term); + expect(pendingRows.some(row => row.includes("… (8 earlier lines)"))).toBe(true); + expect(pendingRows.some(row => row.includes("… (streaming)"))).toBe(true); + expect(pendingRows.some(row => row.includes("20 line 20"))).toBe(true); + + component.setArgsComplete(); + tui.requestRender(); + await scheduler.drain(term); + + component.updateResult(partialWriteResult(), true); + tui.requestRender(); + await scheduler.drain(term); + + const rows = plainBuffer(term); + // The stale pending tail window must not survive above the new frame. + expect(rows.some(row => row.includes("… (streaming)"))).toBe(false); + expect(rows.some(row => row.includes("earlier lines"))).toBe(false); + expect(rows.some(row => row.includes("20 line 20"))).toBe(false); + // The first partial-result frame is what remains: progress line plus the + // top-anchored preview. + expect(rows.some(row => row.includes("Writing notes.txt..."))).toBe(true); + expect(rows.some(row => row.includes(" 1 line 1"))).toBe(true); + expect(rows.some(row => row.includes("… 14 more lines"))).toBe(true); + } finally { + tui.stop(); + await term.flush(); + } + }); +}); diff --git a/packages/coding-agent/test/tools.test.ts b/packages/coding-agent/test/tools.test.ts index a3baf9bbd..90c001c02 100644 --- a/packages/coding-agent/test/tools.test.ts +++ b/packages/coding-agent/test/tools.test.ts @@ -330,6 +330,34 @@ describe("Coding Agent Tools", () => { expect(result.details?.truncation).toBeUndefined(); }); + it("treats empty optional selector as omitted for read", async () => { + const testFile = path.join(testDir, "read-empty-selector.txt"); + const content = "alpha\nselector target\nomega"; + fs.writeFileSync(testFile, content); + + const omitted = getTextOutput(await readTool.execute("test-read-empty-selector-omitted", { path: testFile })); + expect(omitted).toContain("alpha"); + expect(omitted).toContain("selector target"); + expect(omitted).toContain("omega"); + + for (const { name, selector } of [ + { name: "empty", selector: "" }, + { name: "whitespace", selector: " \t\n " }, + ]) { + const withOptionalSelector = getTextOutput( + await readTool.execute(`test-read-empty-selector-${name}`, { + path: testFile, + selector, + }), + ); + expect(withOptionalSelector).toBe(omitted); + } + + await expect( + readTool.execute("test-read-empty-selector-malformed", { path: testFile, selector: "-100" }), + ).rejects.toThrow(/Invalid selector/); + }); + it("truncates lines wider than the read column cap, leaving narrow lines untouched", async () => { const wideLine = "x".repeat(1500); const testFile = path.join(testDir, "wide.txt"); @@ -1649,6 +1677,42 @@ function b() { expect(output).toMatch(/\*2\|match line/); }); + it("treats empty optional selector as omitted for search", async () => { + const testFile = path.join(testDir, "grep-empty-selector.txt"); + fs.writeFileSync(testFile, "before\nneedle empty selector\nbetween\nneedle whitespace selector\nafter"); + + const omitted = getTextOutput( + await searchTool.execute("test-search-empty-selector-omitted", { + pattern: "needle", + path: testFile, + }), + ); + expect(omitted).toMatch(/\*2\|needle empty selector/); + expect(omitted).toMatch(/\*4\|needle whitespace selector/); + + for (const { name, selector } of [ + { name: "empty", selector: "" }, + { name: "whitespace", selector: " \t\n " }, + ]) { + const withOptionalSelector = getTextOutput( + await searchTool.execute(`test-search-empty-selector-${name}`, { + pattern: "needle", + path: testFile, + selector, + }), + ); + expect(withOptionalSelector).toBe(omitted); + } + + await expect( + searchTool.execute("test-search-empty-selector-malformed", { + pattern: "needle", + path: testFile, + selector: "not-a-range", + }), + ).rejects.toThrow(/selector "not-a-range" is invalid/); + }); + it("flags a zero-match search as contextually useless", async () => { fs.writeFileSync(path.join(testDir, "plain.txt"), "nothing interesting here\n"); diff --git a/packages/coding-agent/test/tools/grep-directory-selector.test.ts b/packages/coding-agent/test/tools/grep-directory-selector.test.ts new file mode 100644 index 000000000..f164fc20b --- /dev/null +++ b/packages/coding-agent/test/tools/grep-directory-selector.test.ts @@ -0,0 +1,99 @@ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; +import { GrepTool } from "@oh-my-pi/pi-coding-agent/tools/grep"; +import { removeWithRetries } from "@oh-my-pi/pi-utils"; + +function resultText(result: { content: Array<{ type: string; text?: string }> }): string { + return result.content + .filter(entry => entry.type === "text") + .map(entry => entry.text ?? "") + .join("\n"); +} + +describe("grep explicit line selector on directory searches", () => { + let testDir: string; + + beforeEach(async () => { + testDir = await fs.mkdtemp(path.join(os.tmpdir(), "grep-directory-selector-")); + }); + + afterEach(async () => { + await removeWithRetries(testDir); + }); + + function createSession(): ToolSession { + return { + cwd: testDir, + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => "*", + settings: Settings.isolated({ "grep.contextBefore": 0, "grep.contextAfter": 0 }), + }; + } + + it("filters matches by per-file line number instead of rejecting the directory", async () => { + const appDir = path.join(testDir, "scripts", "app"); + await fs.mkdir(appDir, { recursive: true }); + await Bun.write(path.join(appDir, "one.ts"), "outside one\ninside one\noutside one again\n"); + await Bun.write(path.join(appDir, "two.ts"), "outside two\ninside two\noutside two again\n"); + + const result = await new GrepTool(createSession()).execute("grep-directory-selector", { + pattern: "inside|outside", + path: "scripts/app", + selector: "2-2", + }); + + const text = resultText(result); + expect(text).toContain("inside one"); + expect(text).toContain("inside two"); + expect(text).not.toContain("outside one"); + expect(text).not.toContain("outside two"); + expect(text).not.toContain("Line-range selector requires a single file"); + }); + + it("fetches enough directory matches before applying an explicit later line selector", async () => { + const appDir = path.join(testDir, "scripts", "hot"); + await fs.mkdir(appDir, { recursive: true }); + const content = `${Array.from( + { length: 30 }, + (_, index) => `cap-needle line ${String(index + 1).padStart(2, "0")}`, + ).join("\n")}\n`; + await Bun.write(path.join(appDir, "many.ts"), content); + + const result = await new GrepTool(createSession()).execute("grep-directory-selector-cap", { + pattern: "cap-needle", + path: "scripts/hot", + selector: "25-25", + }); + + const text = resultText(result); + expect(text).toContain("cap-needle line 25"); + expect(text).not.toContain("cap-needle line 24"); + expect(text).not.toContain("cap-needle line 26"); + }); + + it("supports open-ended selectors on directories with a finite fetch budget", async () => { + const appDir = path.join(testDir, "scripts", "tail"); + await fs.mkdir(appDir, { recursive: true }); + const content = `${Array.from( + { length: 60 }, + (_, index) => `open-needle line ${String(index + 1).padStart(2, "0")}`, + ).join("\n")}\n`; + await Bun.write(path.join(appDir, "long.ts"), content); + + const result = await new GrepTool(createSession()).execute("grep-directory-selector-open", { + pattern: "open-needle", + path: "scripts/tail", + selector: "35-", + }); + + const text = resultText(result); + expect(text).toContain("open-needle line 35"); + expect(text).not.toContain("open-needle line 34"); + expect(text).not.toContain("Line-range selector requires a single file"); + }); +}); diff --git a/packages/coding-agent/test/tools/read-renderer.test.ts b/packages/coding-agent/test/tools/read-renderer.test.ts index babab0ba0..2a128bd34 100644 --- a/packages/coding-agent/test/tools/read-renderer.test.ts +++ b/packages/coding-agent/test/tools/read-renderer.test.ts @@ -83,6 +83,46 @@ describe("readToolRenderer hyperlinks", () => { expect(extractLinkTexts(rendered)).not.toContain(`${examplePath}:10-12`); }); + it("renders separate selector read call paths while linking only the base path", async () => { + settings.override("tui.hyperlinks", "always"); + const theme = await getThemeByName("dark"); + expect(theme).toBeDefined(); + + const examplePath = path.resolve("/tmp/omp-read/separate-selector.ts"); + const component = readToolRenderer.renderCall( + { path: examplePath, selector: "10-12" }, + { expanded: false, isPartial: false }, + theme!, + ); + + const rendered = component.render(200).join("\n"); + expect(Bun.stripANSI(rendered)).toContain(`${examplePath}:10-12`); + const exampleUri = new URL(url.pathToFileURL(path.resolve(examplePath)).href); + exampleUri.searchParams.set("line", "10"); + expect(extractLinkUris(rendered)).toContain(exampleUri.href); + expect(extractLinkTexts(rendered)).toContain(examplePath); + expect(extractLinkTexts(rendered)).not.toContain(`${examplePath}:10-12`); + }); + + it("renders separate raw read selectors while linking only the base path", async () => { + settings.override("tui.hyperlinks", "always"); + const theme = await getThemeByName("dark"); + expect(theme).toBeDefined(); + + const examplePath = path.resolve("/tmp/omp-read/raw-selector.ts"); + const component = readToolRenderer.renderCall( + { path: examplePath, selector: "raw" }, + { expanded: false, isPartial: false }, + theme!, + ); + + const rendered = component.render(200).join("\n"); + expect(Bun.stripANSI(rendered)).toContain(`${examplePath}:raw`); + expect(extractLinkUris(rendered)).toContain(url.pathToFileURL(path.resolve(examplePath)).href); + expect(extractLinkTexts(rendered)).toContain(examplePath); + expect(extractLinkTexts(rendered)).not.toContain(`${examplePath}:raw`); + }); + it("links HTTP read result headers to the final URL", async () => { settings.override("tui.hyperlinks", "always"); const theme = await getThemeByName("dark"); diff --git a/packages/coding-agent/test/utils/clipboard.test.ts b/packages/coding-agent/test/utils/clipboard.test.ts index 5e7b5f53d..b9598bce0 100644 --- a/packages/coding-agent/test/utils/clipboard.test.ts +++ b/packages/coding-agent/test/utils/clipboard.test.ts @@ -166,6 +166,23 @@ describe("readImageFromClipboard dispatch", () => { expect(calls[0]?.cmd).toContain("-Sta"); }); + it("falls back to PowerShell when native Windows image conversion fails", async () => { + setPlatform("win32"); + const calls: SpawnCall[] = []; + spyPowershell(calls, RED_1X1_PNG_BASE64); + vi.spyOn(native, "readImageFromClipboard").mockRejectedValue( + new Error("The clipboard image could not be converted to the appropriate format."), + ); + + const image = await readImageFromClipboard(); + + expect(calls).toHaveLength(1); + expect(calls[0]?.cmd[0]).toBe("powershell.exe"); + expect(calls[0]?.cmd).toContain("-Sta"); + expect(image?.mimeType).toBe("image/png"); + expect(Array.from(image!.data.subarray(0, 8))).toEqual([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a]); + }); + it("delegates straight to the native bridge on non-WSL linux with a display", async () => { setPlatform("linux"); process.env.DISPLAY = ":0"; diff --git a/packages/hashline/package.json b/packages/hashline/package.json index 742cc1670..73c31c71a 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/hashline", - "version": "16.3.12", + "version": "16.3.15", "description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/mnemopi/package.json b/packages/mnemopi/package.json index 2f37b76b6..b5a62f06e 100644 --- a/packages/mnemopi/package.json +++ b/packages/mnemopi/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-mnemopi", - "version": "16.3.12", + "version": "16.3.15", "description": "Local SQLite memory engine for Oh My Pi agents", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/mnemopi/src/core/beam/store.ts b/packages/mnemopi/src/core/beam/store.ts index 5d0d2f9dd..225452fc5 100644 --- a/packages/mnemopi/src/core/beam/store.ts +++ b/packages/mnemopi/src/core/beam/store.ts @@ -665,7 +665,46 @@ export function get(beam: BeamMemoryState, memoryId: string): Row | null { WHERE id = ? AND (session_id = ? OR scope = 'global') `) .get(memoryId, beam.sessionId) as Row | null | undefined; - return episodic == null ? null : { ...episodic, metadata: episodic.metadata_json, memory_store: "episodic" }; + if (episodic != null) return { ...episodic, metadata: episodic.metadata_json, memory_store: "episodic" }; + + return getFact(beam, memoryId); +} + +/** + * Read-only resolution for ids minted from the `facts` table. `recall` + * surfaces `facts.fact_id` as a result id (`factRecall`), so `get` must + * resolve those ids too — otherwise every surfaced fact id is a dead end + * for the read path (issue #4725). Visibility mirrors `factRecall`: + * same-session facts plus explicitly global ones (`scope` is an optional + * column on `facts`; `SELECT *` tolerates banks without it, in which case + * only same-session facts resolve). The row is shaped like the + * working/episodic hits with the full triple as content; + * `memory_store: "fact"` marks it read-only — no update/forget/invalidate + * path mutates `facts`. + */ +function getFact(beam: BeamMemoryState, memoryId: string): Row | null { + const fact = beam.db.prepare("SELECT * FROM facts WHERE fact_id = ?").get(memoryId) as Row | null | undefined; + if (fact == null) return null; + if (fact.session_id !== beam.sessionId && fact.scope !== "global") return null; + const subject = typeof fact.subject === "string" ? fact.subject : ""; + const predicate = typeof fact.predicate === "string" ? fact.predicate : ""; + const object = typeof fact.object === "string" ? fact.object : ""; + return { + id: fact.fact_id, + content: [subject, predicate, object].filter(part => part.length > 0).join(" "), + source: "facts", + timestamp: fact.timestamp ?? null, + session_id: fact.session_id ?? null, + importance: fact.confidence ?? null, + metadata: JSON.stringify({ + subject, + predicate, + object, + source_msg_id: fact.source_msg_id ?? null, + }), + created_at: fact.created_at ?? null, + memory_store: "fact", + }; } export function forgetWorking(beam: BeamMemoryState, memoryId: string): boolean { diff --git a/packages/mnemopi/test/beam-store.test.ts b/packages/mnemopi/test/beam-store.test.ts index f9af3a8b5..fb09c7a62 100644 --- a/packages/mnemopi/test/beam-store.test.ts +++ b/packages/mnemopi/test/beam-store.test.ts @@ -1,4 +1,5 @@ import { afterEach, describe, expect, it } from "bun:test"; +import { recallEnhanced } from "@oh-my-pi/pi-mnemopi/core/beam/recall"; import { initBeam } from "@oh-my-pi/pi-mnemopi/core/beam/schema"; import { exportToDict, @@ -217,3 +218,70 @@ describe("beam store free functions", () => { expect(scratchpadRead(dest).map(row => row.content)).toEqual([]); }); }); + +describe("fact-id read path (issue #4725)", () => { + function insertFact( + beam: BeamMemoryState, + factId: string, + sessionId: string, + subject: string, + predicate: string, + object: string, + confidence = 0.9, + ): void { + beam.db + .prepare( + "INSERT INTO facts (fact_id, session_id, subject, predicate, object, timestamp, confidence) VALUES (?, ?, ?, ?, ?, ?, ?)", + ) + .run(factId, sessionId, subject, predicate, object, "2026-05-30T00:00:00.000Z", confidence); + } + + it("resolves an id surfaced by fact recall to a read-only fact row", async () => { + const beam = makeState(); + insertFact(beam, "fact-postgres", beam.sessionId, "service", "uses", "postgres database", 0.91); + + const results = await recallEnhanced(beam, "postgres", 5, { includeFacts: true }); + const surfaced = results.find(result => result.source === "facts"); + expect(surfaced?.id).toBe("fact-postgres"); + + // memory:// reads and memory_edit both resolve ids via get(); a + // surfaced fact id must not be a dead end. + const row = get(beam, "fact-postgres"); + expect(row).toMatchObject({ + id: "fact-postgres", + content: "service uses postgres database", + source: "facts", + importance: 0.91, + session_id: beam.sessionId, + memory_store: "fact", + }); + expect(JSON.parse(String(row?.metadata))).toMatchObject({ + subject: "service", + predicate: "uses", + object: "postgres database", + }); + }); + + it("keeps fact reads session-scoped like fact recall, honoring explicit global scope", () => { + const beam = makeState(); + insertFact(beam, "fact-other", "session-other", "service", "uses", "postgres database"); + expect(get(beam, "fact-other")).toBeNull(); + + beam.db.run("ALTER TABLE facts ADD COLUMN scope TEXT DEFAULT 'session'"); + beam.db.run("UPDATE facts SET scope = 'global' WHERE fact_id = 'fact-other'"); + expect(get(beam, "fact-other")?.memory_store).toBe("fact"); + }); + + it("keeps working rows first on id collision and never deletes facts via forgetWorking", () => { + const beam = makeState(); + insertFact(beam, "shared-id", beam.sessionId, "service", "uses", "postgres database"); + const workingId = remember(beam, "working row shadowing a fact id"); + beam.db.prepare("UPDATE working_memory SET id = ? WHERE id = ?").run("shared-id", workingId); + + expect(get(beam, "shared-id")?.memory_store).toBe("working"); + + expect(forgetWorking(beam, "fact-missing")).toBe(false); + expect(forgetWorking(beam, "shared-id")).toBe(true); + expect(get(beam, "shared-id")?.memory_store).toBe("fact"); + }); +}); diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index f16aff66a..fb1e08bea 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -2,6 +2,14 @@ ## [Unreleased] +## [16.3.13] - 2026-07-09 + +### Fixed + +- Fixed unbounded memory growth in the native bash output bridge when a command produces output faster than the JS event loop consumes it: the shell streaming path now uses a bounded chunk queue with real backpressure (pipe readers park until the JS callback catches up, parking the child on its pipe) instead of buffering the entire surplus in memory. No output is dropped — the rolling tail view, `[raw output: artifact://…]` lossless capture, and byte accounting are unaffected ([#4078](https://github.com/can1357/oh-my-pi/issues/4078)). +- Fixed `readImageFromClipboard` on Windows failing with "could not be converted to the appropriate format" for screenshots taken by Qt-based tools such as PixPin and Snipaste. arboard hands their `CF_DIBV5` payload (`BI_RGB` plus an alpha mask, rewritten to `BI_BITFIELDS`) to a header-less BMP decode that mis-places the pixel offset for V4/V5 bitfield headers; the native reader now falls back to decoding the raw `CF_DIB` clipboard bytes directly, so image paste no longer depends on the PowerShell bridge. ([#3426](https://github.com/can1357/oh-my-pi/issues/3426)) +- Fixed OMP being killed outright (OOM on memory-capped hosts such as WSL) when an output-heavy bash command hit its timeout: the unbounded output-bridge backlog could grow by gigabytes before cancellation and starve the JS event loop far past the deadline; with the bounded backpressured bridge the run resolves at its deadline with flat memory ([#4866](https://github.com/can1357/oh-my-pi/issues/4866)). + ## [16.3.12] - 2026-07-08 ### Fixed diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index 6649dc486..bb54f9bf5 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -170,7 +170,7 @@ export declare function __ompInstallTokioRuntime(): void * `packages/natives/native/index.js` (which derives the name from * `package.json#version`). */ -export declare function __piNativesV16_3_12(): void +export declare function __piNativesV16_3_15(): void /** * Apply conservative pre-execution rewrites to a bash command. diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index d7314c43d..f439ff2c7 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -24,7 +24,7 @@ export const Shell = nativeBindings.Shell; // functions export const __ompInstallTokioRuntime = nativeBindings.__ompInstallTokioRuntime; -export const __piNativesV16_3_12 = nativeBindings.__piNativesV16_3_12; +export const __piNativesV16_3_15 = nativeBindings.__piNativesV16_3_15; export const applyBashFixups = nativeBindings.applyBashFixups; export const astEdit = nativeBindings.astEdit; export const astGrep = nativeBindings.astGrep; diff --git a/packages/natives/package.json b/packages/natives/package.json index c0f0e2add..183a3876d 100644 --- a/packages/natives/package.json +++ b/packages/natives/package.json @@ -1,6 +1,6 @@ { "name": "@oh-my-pi/pi-natives", - "version": "16.3.12", + "version": "16.3.15", "description": "Native Rust bindings for grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API", "type": "module", "homepage": "https://omp.sh", diff --git a/packages/snapcompact/package.json b/packages/snapcompact/package.json index 3e52a7a2c..88b2c10be 100644 --- a/packages/snapcompact/package.json +++ b/packages/snapcompact/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/snapcompact", - "version": "16.3.12", + "version": "16.3.15", "description": "Bitmap-frame context compression for vision-capable LLMs", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/stats/package.json b/packages/stats/package.json index 10c8e2e97..329574cb8 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/omp-stats", - "version": "16.3.12", + "version": "16.3.15", "description": "Local observability dashboard for pi AI usage statistics", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/swarm-extension/package.json b/packages/swarm-extension/package.json index ccc809a1b..0e71287c1 100644 --- a/packages/swarm-extension/package.json +++ b/packages/swarm-extension/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/swarm-extension", - "version": "16.3.12", + "version": "16.3.15", "description": "Swarm orchestration extension for omp", "homepage": "https://omp.sh", "author": "Derek Rynd", diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index e93024ea9..cb2b07009 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,19 @@ ## [Unreleased] +## [16.3.14] - 2026-07-09 + +### Fixed + +- Fixed race condition where scrollback rows could be incorrectly retracted between render frames + +## [16.3.13] - 2026-07-09 + +### Fixed + +- Fixed late terminal appearance subscribers missing the already-detected OSC 11 light/dark result, so theme auto-detection picks up the terminal appearance even when the response arrives before the UI subscribes ([#4731](https://github.com/can1357/oh-my-pi/issues/4731)). +- Fixed slash command Tab completion reopening the file autocomplete drawer after accepting no-argument commands ([#4808](https://github.com/can1357/oh-my-pi/issues/4808)). + ## [16.3.12] - 2026-07-08 ### Fixed diff --git a/packages/tui/package.json b/packages/tui/package.json index 682e2b532..fea5e808d 100644 --- a/packages/tui/package.json +++ b/packages/tui/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-tui", - "version": "16.3.12", + "version": "16.3.15", "description": "Terminal User Interface library with differential rendering for efficient text-based applications", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/tui/src/autocomplete.ts b/packages/tui/src/autocomplete.ts index df6b34b1e..54298a60d 100644 --- a/packages/tui/src/autocomplete.ts +++ b/packages/tui/src/autocomplete.ts @@ -437,6 +437,9 @@ export class CombinedAutocompleteProvider implements AutocompleteProvider { const argumentText = commandText.slice(spaceIndex + 1); // Text after space const command = this.#commands.find(cmd => commandMatchesNameOrAlias(cmd, commandName)); + if (command && "allowArgs" in command && command.allowArgs === false && !/\S/.test(argumentText)) { + return null; + } if (command && (!("allowArgs" in command) || command.allowArgs !== false)) { if (!("getArgumentCompletions" in command) || !command.getArgumentCompletions) { return null; // No argument completion for this command diff --git a/packages/tui/src/terminal.ts b/packages/tui/src/terminal.ts index ed9210af9..b35dfa1f5 100644 --- a/packages/tui/src/terminal.ts +++ b/packages/tui/src/terminal.ts @@ -382,6 +382,8 @@ export interface Terminal { * Register a callback for terminal appearance (dark/light) changes. * Detection uses OSC 11 background color query with Mode 2031 as a change trigger. * Fires when the detected appearance changes, including the initial detection. + * Subscribers registered after detection are invoked immediately with the + * already-detected appearance so late subscribers never miss it. */ onAppearanceChange(callback: (appearance: TerminalAppearance) => void): void; /** The last detected terminal appearance, or undefined if not yet known. */ @@ -516,6 +518,17 @@ export class ProcessTerminal implements Terminal { onAppearanceChange(callback: (appearance: TerminalAppearance) => void): void { this.#appearanceCallbacks.push(callback); + // Replay an already-detected appearance: the startup OSC 11 response can + // arrive before consumers (e.g. the theme bridge) subscribe, and the + // dedup in #handleOsc11Response would otherwise suppress the value for + // them forever (#4731). + if (this.#appearance) { + try { + callback(this.#appearance); + } catch { + /* ignore callback errors */ + } + } } onPrivateModeReport(callback: (mode: number, supported: boolean) => void): void { @@ -704,7 +717,10 @@ export class ProcessTerminal implements Terminal { const decrpmResponsePattern = /^\x1b\[\?(\d+);(\d+)\$y$/; // In-band resize report (DEC mode 2048): \x1b[48;rows;cols;yPixels;xPixels t - const inBandResizePattern = /^\x1b\[48;(\d+);(\d+);(\d+);(\d+)t$/; + // Any field may carry `:`-separated subparameters, which clients MUST + // ignore per spec (#4748): capture the leading digits of each field and + // skip the subparameter tail instead of dropping the whole report. + const inBandResizePattern = /^\x1b\[48;(\d+)(?::[\d:]*)?;(\d+)(?::[\d:]*)?;(\d+)(?::[\d:]*)?;(\d+)(?::[\d:]*)?t$/; this.#stdinBuffer.on("data", (sequence: string) => { // Fast path for plain-text bytes: every escape-probe regex below @@ -776,7 +792,7 @@ export class ProcessTerminal implements Terminal { // reassembled sequence that turns out not to be a resize report (e.g. a // split kitty `\x1b[48;…u` for a digit key) is forwarded to the input // handler rather than dropped. - const inBandResizePartialPattern = /^\x1b\[4[\d;]*$/; + const inBandResizePartialPattern = /^\x1b\[4[\d;:]*$/; const isInBandResizePartial = this.#inBandResizeActive && inBandResizePartialPattern.test(sequence); if (this.#inBandResizeBuffer && sequence.startsWith("\x1b")) { // A new escape interrupted the partial; the stale partial is @@ -1185,9 +1201,9 @@ export class ProcessTerminal implements Terminal { * `rows` before the `resize` event fires, so they are authoritative for the * new cell geometry. A cached DEC 2048 report can be stale: the matching * post-resize report may be dropped (split across stdin reads past the flush - * window) or carry `:`-subparameters the parser skips, leaving the getters - * pinned to the old size — which freezes the rendered width because the - * renderer reflows against {@link columns}/{@link rows}, not the live OS + * window, or interrupted by another escape mid-reassembly), leaving the + * getters pinned to the old size — which freezes the rendered width because + * the renderer reflows against {@link columns}/{@link rows}, not the live OS * value. Drop a cached dimension that disagrees with the live OS value; the * terminal's next valid in-band report re-seeds pixel sizing. */ diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index 0d44beb4a..a916af1d5 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -1137,8 +1137,14 @@ export class TUI extends Container { // Feed the engine's committed-row claim (from the previous frame's // emit) before rendering so the child can skip re-deriving blocks // that already live in immutable native scrollback. Reused segments - // skip this: they never call render(), so the signal is moot. - setNativeScrollbackCommittedRows(child, Math.max(0, this.#committedRows - offset)); + // skip this: they never call render(), so the signal is moot. The + // claim is in the previous frame's coordinates and never exceeds + // the rows the child actually contributed there — history that + // advanced into LATER root children must not read as this child's + // own future rows being pre-committed. + const prevRows = previous !== undefined && previous.component === child ? previous.rowCount : 0; + const prevStart = previous !== undefined && previous.component === child ? previous.start : offset; + setNativeScrollbackCommittedRows(child, Math.min(prevRows, Math.max(0, this.#committedRows - prevStart))); childLines = child.render(width); const liveRegionStart = getNativeScrollbackLiveRegionStart(child); if (liveRegionStart !== undefined) { @@ -2802,6 +2808,7 @@ export class TUI extends Container { this.#committedPrefixAuditRows = Math.min(chunkTo, finalBoundary); this.#clearScrollbackOnNextRender = false; this.#hasEverRendered = true; + this.#publishCommittedRows(); if (!firstPaint && frameLength > height) this.#armPostFullPaintSettle(); return; } @@ -2829,6 +2836,7 @@ export class TUI extends Container { } else { this.#committedPrefixAuditRows = Math.min(preAuditRows, this.#committedRows); } + this.#publishCommittedRows(); } /** @@ -2853,6 +2861,25 @@ export class TUI extends Container { } } + /** + * Push the post-emit committed-row count to root children that implement + * {@link NativeScrollbackCommittedRows}. Compose feeds the same signal + * before each child render (see {@link render}), but guards that run + * BETWEEN frames — e.g. a controller consulting the transcript's + * committed boundary to decide whether a displaceable block may still be + * retracted — would otherwise observe a count one frame stale and retract + * rows that just entered immutable native scrollback, stranding an + * orphaned copy above the repainted block. + */ + #publishCommittedRows(): void { + for (const segment of this.#frameSegments) { + setNativeScrollbackCommittedRows( + segment.component, + Math.min(segment.rowCount, Math.max(0, this.#committedRows - segment.start)), + ); + } + } + /** * Prepare the composed frame for emission, in place. Rows below * `#preparedValidRows` are already prepared against the current frame (the diff --git a/packages/tui/test/autocomplete.test.ts b/packages/tui/test/autocomplete.test.ts index 5f2733d07..f3f21ce4c 100644 --- a/packages/tui/test/autocomplete.test.ts +++ b/packages/tui/test/autocomplete.test.ts @@ -119,6 +119,23 @@ describe("CombinedAutocompleteProvider", () => { expect(result?.items.map(item => item.value)).toContain("/tmp/"); }); + it("does not treat whitespace-only no-arg slash command arguments as file prefixes", async () => { + const baseDir = fs.mkdtempSync(path.join(os.tmpdir(), "autocomplete-quit-whitespace-")); + try { + fs.writeFileSync(path.join(baseDir, "copy-target.ts"), "export {};\n"); + const provider = new CombinedAutocompleteProvider( + [{ name: "quit", description: "Quit", allowArgs: false }], + baseDir, + ); + const line = "/quit "; + const result = await provider.getSuggestions([line], 0, line.length); + + expect(result).toBeNull(); + } finally { + fs.rmSync(baseDir, { recursive: true, force: true }); + } + }); + it("treats @ file-reference tokens as literal text inside slash command arguments without completions", async () => { const baseDir = fs.mkdtempSync(path.join(os.tmpdir(), "autocomplete-rename-args-")); try { diff --git a/packages/tui/test/editor.test.ts b/packages/tui/test/editor.test.ts index 1e0b27dfb..a4eb03ded 100644 --- a/packages/tui/test/editor.test.ts +++ b/packages/tui/test/editor.test.ts @@ -1,4 +1,7 @@ -import { afterEach, describe, expect, it } from "bun:test"; +import { afterEach, describe, expect, it, vi } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; import { stripVTControlCharacters } from "node:util"; import { CURSOR_MARKER } from "@oh-my-pi/pi-tui"; import { CombinedAutocompleteProvider } from "@oh-my-pi/pi-tui/autocomplete"; @@ -424,6 +427,46 @@ describe("Editor component", () => { expect(editor.getText()).toBe("/help "); expect(editor.isShowingAutocomplete()).toBe(false); }); + + it("does not open file autocomplete after tab-completing no-arg slash commands", async () => { + vi.useFakeTimers(); + const baseDir = await fs.mkdtemp(path.join(os.tmpdir(), "slash-tab-no-arg-")); + try { + await Bun.write(path.join(baseDir, "visible-file.ts"), "export {};\n"); + const editor = new Editor(defaultEditorTheme); + editor.setAutocompleteProvider( + new CombinedAutocompleteProvider([{ name: "quit", description: "Quit", allowArgs: false }], baseDir), + ); + + let nextUpdate = Promise.withResolvers(); + editor.onAutocompleteUpdate = () => nextUpdate.resolve(); + editor.handleInput("/"); + await nextUpdate.promise; + + nextUpdate = Promise.withResolvers(); + editor.onAutocompleteUpdate = () => nextUpdate.resolve(); + editor.handleInput("q"); + vi.advanceTimersByTime(100); + await nextUpdate.promise; + + const chainedUpdates = Promise.withResolvers(); + let updateCount = 0; + editor.onAutocompleteUpdate = () => { + updateCount += 1; + if (updateCount === 2) { + chainedUpdates.resolve(); + } + }; + editor.handleInput(" "); + await chainedUpdates.promise; + + expect(editor.getText()).toBe("/quit "); + expect(editor.isShowingAutocomplete()).toBe(false); + } finally { + vi.useRealTimers(); + await fs.rm(baseDir, { recursive: true, force: true }); + } + }); }); describe("Unicode text editing behavior", () => { diff --git a/packages/tui/test/process-terminal-render.test.ts b/packages/tui/test/process-terminal-render.test.ts index d95bf6243..d9af2d757 100644 --- a/packages/tui/test/process-terminal-render.test.ts +++ b/packages/tui/test/process-terminal-render.test.ts @@ -71,4 +71,21 @@ describe("ProcessTerminal geometry reflow through the renderer", () => { expect(harness.terminal.columns).toBe(160); expect(harness.probe.last).toBe(160); }); + + it("recovers the full height when the grow-back report carries colon subparameters (#4748)", async () => { + // iOS soft keyboard under tmux-over-SSH: an in-band shrink lands (keyboard + // up), then the keyboard is dismissed and the grow-back report arrives with + // a spec-permitted `:`-subparameter and no accompanying OS resize. The + // parser must ignore the subparameter — dropping the report leaves the + // viewport pinned at the keyboard-present height. + harness = createProcessTerminalRenderHarness(100, 30); + await harness.feed("\x1b[?2048;1$y"); + await harness.inBand(15, 100, 300, 1000); // keyboard appears: 30 -> 15 rows + expect(harness.terminal.rows).toBe(15); + + await harness.feed("\x1b[48;30;100;600;1000:0t"); // keyboard dismissed: grow back + + expect(harness.terminal.rows).toBe(30); + expect(harness.terminal.columns).toBe(100); + }); }); diff --git a/packages/tui/test/streaming-scrollback-defer.test.ts b/packages/tui/test/streaming-scrollback-defer.test.ts index 6dfd7427a..7098aceaf 100644 --- a/packages/tui/test/streaming-scrollback-defer.test.ts +++ b/packages/tui/test/streaming-scrollback-defer.test.ts @@ -80,6 +80,23 @@ class CommittedRowsProbe extends SeamLineList implements NativeScrollbackCommitt } } +/** + * Extends the compose-time probe with the raw wire: every value the engine + * pushes through `setNativeScrollbackCommittedRows`, in arrival order — + * including the post-emit publish that lands *between* frames. Guards that + * run between frames (a controller deciding whether a displaceable block may + * still be retracted) read exactly this last value; if it lags the emit by + * one frame they retract rows that already entered immutable history. + */ +class CommittedRowsWireProbe extends CommittedRowsProbe { + received: number[] = []; + + override setNativeScrollbackCommittedRows(rows: number): void { + this.received.push(rows); + super.setNativeScrollbackCommittedRows(rows); + } +} + async function settle(term: VirtualTerminal): Promise { const nextTick = Promise.withResolvers(); process.nextTick(nextTick.resolve); @@ -604,6 +621,130 @@ describe("streaming scrollback — visual record", () => { } }); + it("publishes the post-emit committed count between frames — never one frame stale", async () => { + if (process.platform === "win32") return; + const term = new VirtualTerminal(40, 8); + overrideProbe(term, undefined); + const tui = new TUI(term); + const probe = new CommittedRowsWireProbe([]); + + try { + tui.addChild(probe); + tui.start(); + await settle(term); + + // Nothing has scrolled: the between-frames claim is 0 — no phantom rows. + expect(probe.received.at(-1)).toBe(0); + + // One frame grows past the viewport; its emit scrolls rows into + // native scrollback. No further render is requested — whatever the + // probe last received IS the claim a between-frames guard consults. + probe.setLines(rows("hist-", 20)); + tui.requestRender(); + await settle(term); + + // Compose ran before the emit advanced the boundary, so this frame's + // render() saw the pre-emit count. The emit must then push the fresh + // count: with compose-only propagation the last received value would + // still equal the stale compose view, and a guard would retract rows + // that just became immutable — stranding an orphaned copy in history. + const composeView = probe.committedRowsAtRender.at(-1)!; + const betweenFrames = probe.received.at(-1)!; + expect(betweenFrames).toBeGreaterThan(composeView); + // The fresh claim is the truth: exactly the rows above the window + // (tape = committed history rows + the 8-row grid). + expect(betweenFrames).toBe(tape(term).length - 8); + + // A frame that commits nothing must restate the boundary verbatim on + // every push — compose feed and post-emit publish alike. No regress, + // no phantom advance. + const wireLength = probe.received.length; + tui.requestRender(); + await settle(term); + expect(probe.received.length).toBeGreaterThan(wireLength); + for (const value of probe.received.slice(wireLength)) { + expect(value).toBe(betweenFrames); + } + } finally { + tui.stop(); + } + }); + + it("publishes the post-emit committed count on the full-paint replay path too", async () => { + if (process.platform === "win32") return; + const term = new VirtualTerminal(40, 8); + overrideProbe(term, undefined); + const tui = new TUI(term); + // Content taller than the viewport before the first paint: the initial + // frame takes the full-paint path, whose replay commits (frame - height) + // rows in one shot on a separate exit from the ordinary update emit. + const probe = new CommittedRowsWireProbe(rows("hist-", 20)); + + try { + tui.addChild(probe); + tui.start(); + await settle(term); + + // Compose fed the pre-emit count (0); the replay committed 12 rows. + // The full-paint return must publish the fresh count too — leaving + // it stale until the next compose is the same one-frame lag. + const composeView = probe.committedRowsAtRender.at(-1)!; + const betweenFrames = probe.received.at(-1)!; + expect(betweenFrames).toBeGreaterThan(composeView); + expect(betweenFrames).toBe(tape(term).length - 8); + } finally { + tui.stop(); + } + }); + + it("clamps each child's committed-count feed to its own extent", async () => { + if (process.platform === "win32") return; + const term = new VirtualTerminal(40, 8); + overrideProbe(term, undefined); + const tui = new TUI(term); + // A short header above a tall overflowing body: the engine's committed + // boundary sails past the header's 2-row extent. Both feeds are in the + // child's own coordinates and must saturate at what the child actually + // contributed — an unclamped count would make rows the header appends + // LATER read as already-committed, exempting them from ever painting. + const header = new CommittedRowsWireProbe(rows("hdr-", 2)); + const body = new CommittedRowsWireProbe([]); + + try { + tui.addChild(header); + tui.addChild(body); + tui.start(); + await settle(term); + + body.setLines(rows("body-", 20)); + tui.requestRender(); + await settle(term); + + // 22-row frame in an 8-row window: 14 rows committed, the boundary + // 12 rows past the header. Post-emit publish: the header's claim + // saturates at its own extent; the body receives the remainder in + // its own coordinates (boundary minus its start offset). + expect(tape(term).length).toBe(22); + expect(header.received.at(-1)).toBe(2); + expect(Math.max(...header.received)).toBe(2); + expect(body.received.at(-1)).toBe(12); + // Post-emit freshness holds per child in the multi-child layout: + // the body's compose view was still pre-emit, the publish delivered + // the advanced count. + expect(body.received.at(-1)!).toBeGreaterThan(body.committedRowsAtRender.at(-1)!); + + // The compose-time feed clamps identically: an idle frame restates + // each child's saturated claim on every push — never more. + tui.requestRender(); + await settle(term); + expect(header.received.at(-1)).toBe(2); + expect(Math.max(...header.received)).toBe(2); + expect(body.received.at(-1)).toBe(12); + } finally { + tui.stop(); + } + }); + it("never re-anchors a re-laying-out live block mid-run, repairs once at finalize", async () => { if (process.platform === "win32") return; const term = new VirtualTerminal(20, 4); diff --git a/packages/tui/test/terminal-appearance.test.ts b/packages/tui/test/terminal-appearance.test.ts index 88438c841..ea1067c1b 100644 --- a/packages/tui/test/terminal-appearance.test.ts +++ b/packages/tui/test/terminal-appearance.test.ts @@ -144,6 +144,24 @@ describe("ProcessTerminal OSC 11 appearance detection", () => { terminal.stop(); }); + it("replays already detected OSC 11 appearance to late subscribers", () => { + const { terminal } = setupTerminal(); + + process.stdin.emit("data", "\x1b]11;rgb:ffff/ffff/ffff\x07"); + process.stdin.emit("data", "\x1b[?1;2c"); + + const appearances: string[] = []; + terminal.onAppearanceChange(a => appearances.push(a)); + const detected = terminal.appearance; + + // Stop before asserting: a failing expect must not leak a live terminal + // (stdin listeners, kitty push) into subsequent tests. + terminal.stop(); + + expect(detected).toBe("light"); + expect(appearances).toEqual(["light"]); + }); + it("2-digit hex OSC 11 response is correctly normalized", () => { const { terminal } = setupTerminal(); @@ -582,6 +600,30 @@ describe("ProcessTerminal DECRQM + in-band resize (DEC 2026/2048)", () => { terminal.stop(); }); + it("applies a grow-back report whose fields carry colon subparameters (#4748)", () => { + // iOS soft keyboard dismissed under tmux-over-SSH: the pane grows back and + // the terminal reports the restored geometry in-band with a spec-permitted + // `:`-subparameter appended to a field. Mode 2048 allows subparameters on + // any field and requires clients to IGNORE them — dropping the whole + // report instead pins `rows` at the keyboard-present height, because no + // OS resize event accompanies the report to reconcile cached geometry. + Object.defineProperty(process.stdout, "columns", { value: 100, configurable: true }); + Object.defineProperty(process.stdout, "rows", { value: 40, configurable: true }); + const { terminal, received, resizeCount } = setup(); + process.stdin.emit("data", "\x1b[?2048;1$y"); // in-band active + process.stdin.emit("data", "\x1b[48;20;100;400;1000t"); // keyboard appears: shrink + expect(terminal.rows).toBe(20); + expect(resizeCount()).toBe(1); + + process.stdin.emit("data", "\x1b[48;40;100;800;1000:0t"); // keyboard dismissed: grow back + + expect(terminal.rows).toBe(40); + expect(terminal.columns).toBe(100); + expect(resizeCount()).toBe(2); + expect(received).toEqual([]); + terminal.stop(); + }); + it("tracks OS geometry on resize when the post-resize in-band report is missed", () => { // Real terminals always fire SIGWINCH (process.stdout dims refresh first), // but the matching DEC 2048 report can be dropped or arrive malformed. The @@ -658,6 +700,29 @@ describe("ProcessTerminal DECRQM + in-band resize (DEC 2026/2048)", () => { terminal.stop(); }); + it("reassembles a split grow-back report with colon subparameters without dropping or leaking it", () => { + // Same grow-back report, fragmented by the StdinBuffer flush window right + // after the subparameter colon. The partial pattern must accept `:`, or + // the prefix is rejected as garbage, the report never applies, and the + // `0t` tail leaks into the editor as literal keystrokes. + vi.useFakeTimers(); + Object.defineProperty(process.stdout, "columns", { value: 100, configurable: true }); + Object.defineProperty(process.stdout, "rows", { value: 40, configurable: true }); + const { terminal, received, resizeCount } = setup(); + process.stdin.emit("data", "\x1b[?2048;1$y"); // in-band active + process.stdin.emit("data", "\x1b[48;20;100;400;1000t"); // keyboard appears: shrink + expect(terminal.rows).toBe(20); + + process.stdin.emit("data", "\x1b[48;40;100;800;1000:"); + vi.advanceTimersByTime(50); // flush window elapses mid-report + process.stdin.emit("data", "0t"); + + expect(received).toEqual([]); + expect(terminal.rows).toBe(40); + expect(resizeCount()).toBe(2); + terminal.stop(); + }); + it("forwards a split report fragment as one escape sequence instead of leaking bare characters", () => { // The reported symptom: a fragment like `8;125;1156;1125t` (the tail of // `\x1b[48;125;1156;1125t`, missing a field) appeared as literal text in the diff --git a/packages/utils/package.json b/packages/utils/package.json index 9a805d384..2e186b008 100644 --- a/packages/utils/package.json +++ b/packages/utils/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-utils", - "version": "16.3.12", + "version": "16.3.15", "description": "Shared utilities for pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/utils/src/dirs.ts b/packages/utils/src/dirs.ts index c6fb9d9df..d0b63af69 100644 --- a/packages/utils/src/dirs.ts +++ b/packages/utils/src/dirs.ts @@ -22,6 +22,9 @@ export const APP_NAME: string = "omp"; /** Config directory name (e.g. ".omp") */ export const CONFIG_DIR_NAME: string = ".omp"; +/** Ordered main settings filenames: canonical write target first, legacy-compatible YAML fallback second. */ +export const MAIN_CONFIG_FILENAMES = ["config.yml", "config.yaml"] as const; + /** Version (e.g. "1.0.0") */ export const VERSION: string = version; diff --git a/packages/wire/package.json b/packages/wire/package.json index 94fda90cf..51c90f13f 100644 --- a/packages/wire/package.json +++ b/packages/wire/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-wire", - "version": "16.3.12", + "version": "16.3.15", "description": "Shared wire protocol types for Oh My Pi packages", "homepage": "https://omp.sh", "author": "Can Boluk",