Merge branch 'main' into pr-8052

This commit is contained in:
can1357
2026-08-17 11:00:58 +03:00
317 changed files with 15976 additions and 3215 deletions
+13
View File
@@ -10,6 +10,19 @@ common --enable_platform_specific_config
# PATH through crate annotations in MODULE.bazel.
build --incompatible_strict_action_env
# Third-party crate resolution (rules_rust crate_universe) shells out to
# `cargo-bazel splice`, which runs `cargo metadata` + fetches every git/registry
# dep. Two knobs matter on cold caches:
# 1. CARGO_BAZEL_ISOLATED=0 — reuse the host's ~/.cargo registry/index instead
# of building a throwaway cargo home per invocation. Without this, every
# splice re-clones the entire crates.io index (~1 GiB, minutes on cold DNS).
# 2. CARGO_BAZEL_TIMEOUT=1800 — cargo-bazel honors this to raise Bazel's
# default `repository_ctx.execute` 600s cap. Fresh machines / ephemeral
# runners with git-hosted forks (brush, uutils) blow past 600s and abort
# with `Timed out` (rules_rust common_utils.bzl:61) before we can help.
common --repo_env=CARGO_BAZEL_ISOLATED=0
common --repo_env=CARGO_BAZEL_TIMEOUT=1800
# NOTE: rust pipelined_compilation stays OFF: handing dependents rmeta-only
# crates breaks `rust_test(crate = ...)` harness compiles whose deps export
# macro_rules! ("can't find crate" at macro expansion).
+1 -1
View File
@@ -319,7 +319,7 @@ jobs:
# Not `test:scripts`: scripts/musl-release.test.ts fails on main
# (its install.sh smoke-check executes a fake binary), so running
# the whole group here would red this job on an unrelated break.
bun test scripts/release.test.ts
bun test scripts/ci-test-ts.test.ts scripts/release.test.ts
test_coding_agent_singleton:
name: Test coding-agent singleton/global-state (TS)
Generated
+21 -21
View File
@@ -2029,9 +2029,9 @@ dependencies = [
[[package]]
name = "error-code"
version = "3.3.2"
version = "3.4.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "dea2df4cf52843e0452895c455a1a2cfbb842a1e7329671acf418fdc53ed4c59"
checksum = "0b5343afd4a8365a643ac588dab4cf234a190c7f6c88c9f6dd6ffe00837661b7"
[[package]]
name = "event-listener"
@@ -2146,9 +2146,9 @@ dependencies = [
[[package]]
name = "find-msvc-tools"
version = "0.1.10"
version = "0.1.11"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "26b73573e6edcd2af0cdf47bd6cb58f0b3839491263c314eaad1ccf24430e1de"
checksum = "d45db016d36b838f563236e9193d0ee6ce38f3f68b6c94e914b4929c96bbb890"
[[package]]
name = "fixed_decimal"
@@ -3261,9 +3261,9 @@ dependencies = [
[[package]]
name = "inotify"
version = "0.11.4"
version = "0.11.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "153be1941a183ec9ccd095ddbe17a8b8d435ef6c76e9e02451b933c3999af2c8"
checksum = "4cc00ea907cab49550b7da656f80ebb97be1b997d931fbcd28d39734e17ce592"
dependencies = [
"bitflags 2.13.1",
"inotify-sys",
@@ -3590,9 +3590,9 @@ checksum = "b6d2cec3eae94f9f509c767b45932f1ada8350c4bdb85af2fcab4a3c14807981"
[[package]]
name = "libredox"
version = "0.1.19"
version = "0.1.20"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2026a5056764a10b2bf5d56488cba40da507f5493a6a429340e2004d9ed085fa"
checksum = "28d0a00925a9f930d679b6789b721e3a7f9ed110f41b86d2497caa780c3a070a"
dependencies = [
"libc",
]
@@ -4880,7 +4880,7 @@ dependencies = [
[[package]]
name = "pi-ast"
version = "17.3.4"
version = "17.3.5"
dependencies = [
"anyhow",
"ast-grep-core",
@@ -4949,7 +4949,7 @@ dependencies = [
[[package]]
name = "pi-builtins"
version = "17.3.4"
version = "17.3.5"
dependencies = [
"ansi-width",
"anyhow",
@@ -5034,7 +5034,7 @@ dependencies = [
[[package]]
name = "pi-iso"
version = "17.3.4"
version = "17.3.5"
dependencies = [
"async-trait",
"libc",
@@ -5046,7 +5046,7 @@ dependencies = [
[[package]]
name = "pi-natives"
version = "17.3.4"
version = "17.3.5"
dependencies = [
"anyhow",
"arboard",
@@ -5118,7 +5118,7 @@ dependencies = [
[[package]]
name = "pi-shell"
version = "17.3.4"
version = "17.3.5"
dependencies = [
"anyhow",
"brush-core",
@@ -5146,7 +5146,7 @@ dependencies = [
[[package]]
name = "pi-voice"
version = "17.3.4"
version = "17.3.5"
dependencies = [
"audiopus_sys",
"bytes",
@@ -5161,7 +5161,7 @@ dependencies = [
[[package]]
name = "pi-walker"
version = "17.3.4"
version = "17.3.5"
dependencies = [
"dashmap",
"globset",
@@ -5937,9 +5937,9 @@ checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f"
[[package]]
name = "safe_arch"
version = "1.1.0"
version = "1.2.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3a52ec151f024d703f9fd65abb7cbe81e7cdb39f18917a3a37e3014470dc7c59"
checksum = "42c6efa15875e6ecb39ca61fb0b0c1a40b84fac5a5ffe71eef7d1000c8eb3f5f"
dependencies = [
"bytemuck",
]
@@ -7673,9 +7673,9 @@ dependencies = [
[[package]]
name = "uuid"
version = "1.24.0"
version = "1.24.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bf3923a6f5c4c6382e0b653c4117f48d631ea17f38ed86e2a828e6f7412f5239"
checksum = "2cefc03fd367c0c6d4305de1b312cf00248c4114f4a0418ce6a6af769e3b0bd9"
dependencies = [
"getrandom 0.4.3",
"js-sys",
@@ -7819,9 +7819,9 @@ dependencies = [
[[package]]
name = "wayland-backend"
version = "0.3.16"
version = "0.3.17"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "016ccf01d1c58b6f8999612813e17c9b2390f7d70671428869913310f83f54b8"
checksum = "38a91b4eaddff87b1cd1074985e3713da4af2c49742d1b356b2c01670a67a078"
dependencies = [
"cc",
"downcast-rs",
+1 -1
View File
@@ -16,7 +16,7 @@ members = [
resolver = "3"
[workspace.package]
version = "17.3.4"
version = "17.3.5"
edition = "2024"
license = "MIT"
authors = ["Can Boluk"]
+47 -47
View File
@@ -21,7 +21,7 @@
},
"packages/agent": {
"name": "@oh-my-pi/pi-agent-core",
"version": "17.3.4",
"version": "17.3.5",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
@@ -40,7 +40,7 @@
},
"packages/ai": {
"name": "@oh-my-pi/pi-ai",
"version": "17.3.4",
"version": "17.3.5",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/omptype": "catalog:",
@@ -63,7 +63,7 @@
},
"packages/catalog": {
"name": "@oh-my-pi/pi-catalog",
"version": "17.3.4",
"version": "17.3.5",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/omptype": "catalog:",
@@ -76,7 +76,7 @@
},
"packages/coding-agent": {
"name": "@oh-my-pi/pi-coding-agent",
"version": "17.3.4",
"version": "17.3.5",
"bin": {
"omp": "src/cli.ts",
},
@@ -133,7 +133,7 @@
},
"packages/hashline": {
"name": "@oh-my-pi/hashline",
"version": "17.3.4",
"version": "17.3.5",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
@@ -177,7 +177,7 @@
},
"packages/mnemopi": {
"name": "@oh-my-pi/pi-mnemopi",
"version": "17.3.4",
"version": "17.3.5",
"bin": {
"mnemopi": "src/cli.ts",
},
@@ -203,7 +203,7 @@
},
"packages/natives": {
"name": "@oh-my-pi/pi-natives",
"version": "17.3.4",
"version": "17.3.5",
"devDependencies": {
"@napi-rs/cli": "catalog:",
"@types/bun": "catalog:",
@@ -211,7 +211,7 @@
},
"packages/omptype": {
"name": "@oh-my-pi/omptype",
"version": "17.3.4",
"version": "17.3.5",
"devDependencies": {
"@ark/attest": "0.56.3",
"@ark/schema": "0.56.2",
@@ -224,7 +224,7 @@
},
"packages/snapcompact": {
"name": "@oh-my-pi/snapcompact",
"version": "17.3.4",
"version": "17.3.5",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
@@ -238,7 +238,7 @@
},
"packages/stats": {
"name": "@oh-my-pi/omp-stats",
"version": "17.3.4",
"version": "17.3.5",
"bin": {
"omp-stats": "./src/index.ts",
},
@@ -263,7 +263,7 @@
},
"packages/tui": {
"name": "@oh-my-pi/pi-tui",
"version": "17.3.4",
"version": "17.3.5",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
@@ -299,7 +299,7 @@
},
"packages/utils": {
"name": "@oh-my-pi/pi-utils",
"version": "17.3.4",
"version": "17.3.5",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
},
@@ -309,7 +309,7 @@
},
"packages/wire": {
"name": "@oh-my-pi/pi-wire",
"version": "17.3.4",
"version": "17.3.5",
"devDependencies": {
"@types/bun": "catalog:",
},
@@ -347,19 +347,19 @@
"@bufbuild/protoc-gen-es": "^2.12.1",
"@huggingface/transformers": "^4.2.0",
"@napi-rs/cli": "3.7.2",
"@oh-my-pi/hashline": "17.3.4",
"@oh-my-pi/omp-stats": "17.3.4",
"@oh-my-pi/omptype": "17.3.4",
"@oh-my-pi/pi-agent-core": "17.3.4",
"@oh-my-pi/pi-ai": "17.3.4",
"@oh-my-pi/pi-catalog": "17.3.4",
"@oh-my-pi/pi-coding-agent": "17.3.4",
"@oh-my-pi/pi-mnemopi": "17.3.4",
"@oh-my-pi/pi-natives": "17.3.4",
"@oh-my-pi/pi-tui": "17.3.4",
"@oh-my-pi/pi-utils": "17.3.4",
"@oh-my-pi/pi-wire": "17.3.4",
"@oh-my-pi/snapcompact": "17.3.4",
"@oh-my-pi/hashline": "17.3.5",
"@oh-my-pi/omp-stats": "17.3.5",
"@oh-my-pi/omptype": "17.3.5",
"@oh-my-pi/pi-agent-core": "17.3.5",
"@oh-my-pi/pi-ai": "17.3.5",
"@oh-my-pi/pi-catalog": "17.3.5",
"@oh-my-pi/pi-coding-agent": "17.3.5",
"@oh-my-pi/pi-mnemopi": "17.3.5",
"@oh-my-pi/pi-natives": "17.3.5",
"@oh-my-pi/pi-tui": "17.3.5",
"@oh-my-pi/pi-utils": "17.3.5",
"@oh-my-pi/pi-wire": "17.3.5",
"@oh-my-pi/snapcompact": "17.3.5",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/api-logs": "^0.220.0",
"@opentelemetry/context-async-hooks": "^2.9.0",
@@ -489,7 +489,7 @@
"@huggingface/jinja": ["@huggingface/jinja@0.5.9", "", {}, "sha512-uWTG+l3VJRsl7EXxYizuL3P+cCPoc3cRqbWWRcQN0FhejRfbdq0RNhCmbY/YDtnTcz9icdLYuLDjsnz4d8JMuw=="],
"@huggingface/tasks": ["@huggingface/tasks@0.21.33", "", {}, "sha512-efQa8g+WjPwzlwk7Wl4eDgbM1Jq+WooUw/b1+eKl2LVUdPMhnbwsFNMJJxhEywaLJvJBy93+jjna6855U9GePw=="],
"@huggingface/tasks": ["@huggingface/tasks@0.21.34", "", {}, "sha512-NRc1vw2Q/nZQPKjlzEkXxi0YNXstI1hXoaVruvJbgeznpXtEmPD0pey/ujIWTnhmsBQIVexr/QJJ5pyIJBPrMw=="],
"@huggingface/tokenizers": ["@huggingface/tokenizers@0.1.3", "", {}, "sha512-8rF/RRT10u+kn7YuUbUg0OF30K8rjTc78aHpxT+qJ1uWSqxT1MHi8+9ltwYfkFYJzT/oS+qw3JVfHtNMGAdqyA=="],
@@ -667,7 +667,7 @@
"@napi-rs/tar-win32-x64-msvc": ["@napi-rs/tar-win32-x64-msvc@1.1.1", "", { "os": "win32", "cpu": "x64" }, "sha512-yJsB2IsrODQVLKbm2Fg1nHiVRbEj49mSPbj4x7JPZWJI0jGVPjohE2Sif0FBbx8OxsVoUODvS0BwksZZ8jl/OA=="],
"@napi-rs/wasm-runtime": ["@napi-rs/wasm-runtime@1.2.2", "", { "dependencies": { "@tybys/wasm-util": "^0.10.3" }, "peerDependencies": { "@emnapi/core": "^1.7.1 || ^2.0.0-alpha.3", "@emnapi/runtime": "^1.7.1 || ^2.0.0-alpha.3" } }, "sha512-JfB4kuJQjaoHuCTseIINHtHWeJnvgEcxjwA5t/Y00ZgaOO1Crz3fjT/p8kT28zA/Caz7oiUMn3d6H2yOVCVwuw=="],
"@napi-rs/wasm-runtime": ["@napi-rs/wasm-runtime@1.2.3", "", { "dependencies": { "@tybys/wasm-util": "^0.10.3" }, "peerDependencies": { "@emnapi/core": "^1.7.1 || ^2.0.0-alpha.4", "@emnapi/runtime": "^1.7.1 || ^2.0.0-alpha.4" } }, "sha512-UMduMbqO5s5zF2NkNacMT/yK5Y5QiKvWr2+50bzIIxFDwVJ2h49b+oyjaCGPhJxd2/gC2x39EHv/gHVuu36x2Q=="],
"@napi-rs/wasm-tools": ["@napi-rs/wasm-tools@1.1.0", "", { "optionalDependencies": { "@napi-rs/wasm-tools-android-arm-eabi": "1.1.0", "@napi-rs/wasm-tools-android-arm64": "1.1.0", "@napi-rs/wasm-tools-darwin-arm64": "1.1.0", "@napi-rs/wasm-tools-darwin-x64": "1.1.0", "@napi-rs/wasm-tools-freebsd-x64": "1.1.0", "@napi-rs/wasm-tools-linux-arm64-gnu": "1.1.0", "@napi-rs/wasm-tools-linux-arm64-musl": "1.1.0", "@napi-rs/wasm-tools-linux-x64-gnu": "1.1.0", "@napi-rs/wasm-tools-linux-x64-musl": "1.1.0", "@napi-rs/wasm-tools-wasm32-wasi": "1.1.0", "@napi-rs/wasm-tools-win32-arm64-msvc": "1.1.0", "@napi-rs/wasm-tools-win32-ia32-msvc": "1.1.0", "@napi-rs/wasm-tools-win32-x64-msvc": "1.1.0" } }, "sha512-VjHyKEqXAwYZK+HY7iJctYvRm3TFEbaQxeZwvAG1QRkoo1a39phMY8J6x9tUEqJI03W6MysB8F2jacI6wvcx+w=="],
@@ -789,7 +789,7 @@
"@opentelemetry/semantic-conventions": ["@opentelemetry/semantic-conventions@1.43.0", "", {}, "sha512-eSYWTm620tTk45EKSedaUL8MFYI8hW164hIXsgIHyxu3VobUB3fFCu5t0hQby6OoWRPsG1KkKUG2M5UadiLiVg=="],
"@oxc-project/types": ["@oxc-project/types@0.143.0", "", {}, "sha512-u6JZdLBTLotrNC9Vd6vPssINdzcCzleKAH6EJKImQb7GtYvX5keN2dxkoK44stCc4tffE6QQRtZTXVSzsLUlWA=="],
"@oxc-project/types": ["@oxc-project/types@0.144.0", "", {}, "sha512-nuhZIOLuI6TFQ32I/WnUx+SCPY7SdSKwgnFHydAuoS1+Z4BRcaP+RRJmGzl9lw+0OFF7UmaESf7KQRXaNLHypg=="],
"@prettier/sync": ["@prettier/sync@0.6.1", "", { "dependencies": { "make-synchronized": "^0.8.0" }, "peerDependencies": { "prettier": "*" } }, "sha512-yF9G8vK/LYUTF3Cijd7VC9La3b20F20/J/fgoR4H0B8JGOWnZVZX6+I6+vODPosjmMcpdlUV+gUqJQZp3kLOcw=="],
@@ -813,33 +813,33 @@
"@puppeteer/browsers": ["@puppeteer/browsers@3.0.6", "", { "dependencies": { "modern-tar": "^0.7.6", "yargs": "^18.0.0" }, "peerDependencies": { "proxy-agent": ">=8.0.1", "yauzl": "^2.10.0 || ^3.4.0" }, "optionalPeers": ["proxy-agent", "yauzl"], "bin": { "browsers": "lib/main-cli.js" } }, "sha512-B/gKoqlFkzhvzsI6jo9K1cZz9o5ypviVv/xu8CwA4grZzyVwN+XfkT+tu8T1zrauuEXv6VhS2oGX+6NL95WcKA=="],
"@rolldown/binding-android-arm64": ["@rolldown/binding-android-arm64@1.2.3", "", { "os": "android", "cpu": "arm64" }, "sha512-zrJtHDcaZJ1Fp7xf4hNl+7seH9Cn/N5TwLYkhgXREtBwAd/jaqW3uqeHxpDugJLVICWg4eW44kOQEGJ1r6jCGw=="],
"@rolldown/binding-android-arm64": ["@rolldown/binding-android-arm64@1.2.4", "", { "os": "android", "cpu": "arm64" }, "sha512-jHC2cnyKz5xU2fhECtFl8OZ83cYNt13GZQD+0uMJ/X3o+ijmd56okHhTUwxVSHPx1IRVIJEZ1/1pPzeLCU6XKA=="],
"@rolldown/binding-darwin-arm64": ["@rolldown/binding-darwin-arm64@1.2.3", "", { "os": "darwin", "cpu": "arm64" }, "sha512-ieIiibVCp0tX7TLu2cafoNPv8wJyYi01ekXpbf8q2j7F4rGAhhXb/eQh7ge9DRBY78GwmRQtvjZDux7EDbA8kA=="],
"@rolldown/binding-darwin-arm64": ["@rolldown/binding-darwin-arm64@1.2.4", "", { "os": "darwin", "cpu": "arm64" }, "sha512-Dc5mPD8F5F/FS8i01syd7FTF6yB2fVthH/TRkjwJkzUK6EpoxHtqvZQP5Zwq80/5z19TWYHIg1KOHboCgVx/aQ=="],
"@rolldown/binding-darwin-x64": ["@rolldown/binding-darwin-x64@1.2.3", "", { "os": "darwin", "cpu": "x64" }, "sha512-Zh9tCon19eDXJoihx0rqKhMUlMYqzwj3aPsSuHmI4RWZh62dWUL+DJN4C5YQya5TcQBJU/Fe8+rY0jhXTQITqA=="],
"@rolldown/binding-darwin-x64": ["@rolldown/binding-darwin-x64@1.2.4", "", { "os": "darwin", "cpu": "x64" }, "sha512-fpDm4oBo6SqLvWUYCmFhdde3U9KH2fRNNMeAnAPAIwxRL345xutL0EtEUcuoxsoazdJGv/MuDBQHlCDrtbvqOg=="],
"@rolldown/binding-freebsd-x64": ["@rolldown/binding-freebsd-x64@1.2.3", "", { "os": "freebsd", "cpu": "x64" }, "sha512-nGbJWewA1wrXXZiQhjAT5rhibGfns5ZNkDVqxsO6zJ3f3YvpoDNNmGMSbbhLuXKjNScaBJVOAboztAWVespQMg=="],
"@rolldown/binding-freebsd-x64": ["@rolldown/binding-freebsd-x64@1.2.4", "", { "os": "freebsd", "cpu": "x64" }, "sha512-rSJoreDE/HoIzoaib6MTp5jQtCTdMHKIvItAKT/ImS6Y6Ww76oUaeMyp4Vc/fAgd/ehji068IxetHXAnqUwN9A=="],
"@rolldown/binding-linux-arm-gnueabihf": ["@rolldown/binding-linux-arm-gnueabihf@1.2.3", "", { "os": "linux", "cpu": "arm" }, "sha512-QNniJr5Kml0kDEB98jiDOJjXNroxIIi0IXIbdYzY26Xt1pVbeP62+KnoIZLwirOymX/0jDk/2gI/bNUv7A7OIw=="],
"@rolldown/binding-linux-arm-gnueabihf": ["@rolldown/binding-linux-arm-gnueabihf@1.2.4", "", { "os": "linux", "cpu": "arm" }, "sha512-/jm8OGHgn7oGaJu3i/qZI9spUGcJ+y/lk43ttQ/iO1tOd9NissG6o97bighBCiL+BKRngmcDuR6ikfwYdJmVuQ=="],
"@rolldown/binding-linux-arm64-gnu": ["@rolldown/binding-linux-arm64-gnu@1.2.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-TkqEAcmmvH3I/q4114NB4RVt6241Dao48pF45uLcFGrwAaIn0iITgTAKP/dLjbN0R4buJjGb91+UHSoFmpgIWw=="],
"@rolldown/binding-linux-arm64-gnu": ["@rolldown/binding-linux-arm64-gnu@1.2.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-tIP06BeD9EqvECBrPZ+sqdPlYrT+aYaAiu1wYziVx5elRK/ftm33JxVDy2bXGbr6J0CrtirCkR87/X5a2euEng=="],
"@rolldown/binding-linux-arm64-musl": ["@rolldown/binding-linux-arm64-musl@1.2.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-NHqjnxpsndf4MPymxteFAWHHfkTL8HjWh1KB7z23ofZ6QO2euONuxDXjat69dKZRALnGypg8k8SsK8vZJoXv1Q=="],
"@rolldown/binding-linux-arm64-musl": ["@rolldown/binding-linux-arm64-musl@1.2.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-Ql1Q0EQqVThvn9VAVlwNzsUvbSFtCMGjLpRRi4pk5i7NZZ4n5ISiLMjHYtus4VQ2PvkSw24zyaCVsiS+sXPj1w=="],
"@rolldown/binding-linux-ppc64-gnu": ["@rolldown/binding-linux-ppc64-gnu@1.2.3", "", { "os": "linux", "cpu": "ppc64" }, "sha512-6tbrbwfz5GB9DQ4Jwo6hy9v+vR31xZlvzZ6n5Xut6Hhx5PvrA9q/HsK8KMaYQp063iqZGXwNvZtYNLD7EM/x0w=="],
"@rolldown/binding-linux-ppc64-gnu": ["@rolldown/binding-linux-ppc64-gnu@1.2.4", "", { "os": "linux", "cpu": "ppc64" }, "sha512-GjbjXD4XXfN19D0LZNbmiCBUoDiRACsYHr0yaIbbn8aFsXjHZifcYqu/W5Er5X2X990WjHXFrxarn5chzItorQ=="],
"@rolldown/binding-linux-s390x-gnu": ["@rolldown/binding-linux-s390x-gnu@1.2.3", "", { "os": "linux", "cpu": "s390x" }, "sha512-oyuXxXmoZHjXC917IAPFAAv4wWAa0cM9afk8nx1+9/jNNOX1uPf8yDA6p7G0RypOfw/X0PQt5IfoquY1um+zSg=="],
"@rolldown/binding-linux-s390x-gnu": ["@rolldown/binding-linux-s390x-gnu@1.2.4", "", { "os": "linux", "cpu": "s390x" }, "sha512-p5WR0NOwaRmJ/B1b6IjEFLLivwEsf3PrdBIhRbhTCQisbo2SvHHpG4ELB/+FgQNnB88LTOF86upmJmbvZdQ2lw=="],
"@rolldown/binding-linux-x64-gnu": ["@rolldown/binding-linux-x64-gnu@1.2.3", "", { "os": "linux", "cpu": "x64" }, "sha512-TytMwF2KVGqP2tgd0I1OY0PAv78dZRAYcF5ssDzjM34SUXCED3uXvSd5+lHoC0bTD6eEdFz7LdQNCO1y0oVk9w=="],
"@rolldown/binding-linux-x64-gnu": ["@rolldown/binding-linux-x64-gnu@1.2.4", "", { "os": "linux", "cpu": "x64" }, "sha512-4/GyVjmhR+Tc6HLJvwc1sOhPqAZtySiSMesOZyX6JQ5XBxoTDEMKQzvo07NIK6nTon/SivlZqvhzvuVBNQhObQ=="],
"@rolldown/binding-linux-x64-musl": ["@rolldown/binding-linux-x64-musl@1.2.3", "", { "os": "linux", "cpu": "x64" }, "sha512-/E9m3qstrJFVPoULV25mVQblSNExY2+kBsYe4sy0Tn0yOOgJ8wZbZt3KnRbF/XeU2Gl1STKUQnDNTqhIE5MD4A=="],
"@rolldown/binding-linux-x64-musl": ["@rolldown/binding-linux-x64-musl@1.2.4", "", { "os": "linux", "cpu": "x64" }, "sha512-l9eeLsCNvPpmSXUej0etw/J1eqV0Jj1D5G/xG6YTijmE6dkv6E2QezgWbTfQk63v952DPqrjOCoiqxq7Bw0YUQ=="],
"@rolldown/binding-openharmony-arm64": ["@rolldown/binding-openharmony-arm64@1.2.3", "", { "os": "none", "cpu": "arm64" }, "sha512-Kr0OcsoQI816i6HOl3vFHpd1K0eZyh76zgfj4c1nTyaTsd5r2Mj1lwM4R90y/qaCfmTn9eHy0SKwi98eitRxug=="],
"@rolldown/binding-openharmony-arm64": ["@rolldown/binding-openharmony-arm64@1.2.4", "", { "os": "none", "cpu": "arm64" }, "sha512-e0F355MSTMm3+UOqtV3L24gFUp2N5m1f8L/7d56deik6va+AXdrt9F8LbzGpeWGWRbZEDq4m8NVnJDeBtf9DZg=="],
"@rolldown/binding-win32-arm64-msvc": ["@rolldown/binding-win32-arm64-msvc@1.2.3", "", { "os": "win32", "cpu": "arm64" }, "sha512-hOtMwTqnME+/gJcH/PCZ0wn0zPUjiWOgkHpxbSJpfGKMezHltx1S7/k1SitzVa7Ww2cqrDDaFbZEhcJZO8o+Jw=="],
"@rolldown/binding-win32-arm64-msvc": ["@rolldown/binding-win32-arm64-msvc@1.2.4", "", { "os": "win32", "cpu": "arm64" }, "sha512-AWLi0uBRYh6QlE7OKhiz+phZC0qwtij2QZmhmOdsLdFn64m7oMpooE9ICE3lhm9xMb4SpDo2WbHcxX1iFLFtqw=="],
"@rolldown/binding-win32-x64-msvc": ["@rolldown/binding-win32-x64-msvc@1.2.3", "", { "os": "win32", "cpu": "x64" }, "sha512-ekcqMMkI2PlhYnfzQnB/cEdYUVVJViWvoUyLrbzgDoi3Snfc1mVBwdnc306ufA5ejy8JSPjT2RlW1nQSjW7efg=="],
"@rolldown/binding-win32-x64-msvc": ["@rolldown/binding-win32-x64-msvc@1.2.4", "", { "os": "win32", "cpu": "x64" }, "sha512-UwSDJOg3dqCAejWdxclJjCsh3Qq4vLYMDxmyHqo1btz3stK2VqgwNd3mm5tuIwzSlGIQ/1H9Hr+Zn09mrezNqQ=="],
"@rolldown/pluginutils": ["@rolldown/pluginutils@1.0.1", "", {}, "sha512-2j9bGt5Jh8hj+vPtgzPtl72j0yRxHAyumoo6TNfAjsLB04UtpSvPbPcDcBMxz7n+9CYB0c1GxQFxYRg2jimqGw=="],
@@ -965,7 +965,7 @@
"adm-zip": ["adm-zip@0.5.18", "", {}, "sha512-ufJnssQGbxzLNS1Ho9bCtX4rQKCCvoVuDLHoJyc3F9dOGDB4BkWs2Ci0kv53lqocAEQ/Cbi+I2XCsNYGqVYqng=="],
"ansi-regex": ["ansi-regex@6.2.2", "", {}, "sha512-Bq3SmSpyFHaWjPk8If9yc6svM8c56dB5BAtW4Qbw5jHTwwXXcTLoRMkpDJp6VL0XzlWaCHTXrkFURMYmD0sLqg=="],
"ansi-regex": ["ansi-regex@6.3.0", "", {}, "sha512-WpDfL7NO6j7tH88IDBNVdUJxDh9nmCteAVW9dsep846XdwF4naCBK+/tGLX3KJgcpgMRXCFlTM2hKGoK9FsdrQ=="],
"ansi-styles": ["ansi-styles@4.3.0", "", { "dependencies": { "color-convert": "^2.0.1" } }, "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg=="],
@@ -1061,7 +1061,7 @@
"diff": ["diff@9.0.0", "", {}, "sha512-svtcdpS8CgJyqAjEQIXdb3OjhFVVYjzGAPO8WGCmRbrml64SPw/jJD4GoE98aR7r25A0XcgrK3F02yw9R/vhQw=="],
"electron-to-chromium": ["electron-to-chromium@1.5.402", "", {}, "sha512-/oOpMaPT6Yg+6/1XQhyIPlzgj7Ye9zf+nNM2Uh6OcE2G2oNptWazFa+qB2Pdqqbsc9KnIDzgAntoYN0dbwOXwA=="],
"electron-to-chromium": ["electron-to-chromium@1.5.405", "", {}, "sha512-bNglH7lPH5l+yHOes7Zr4VqxhOy4BQ9ZBUX4VdoFgxMpzJk7W1ZoO3Vgd9Pxa9PyjQ76sfm2aKH/nzEcCNRlew=="],
"emnapi": ["emnapi@1.11.3", "", { "peerDependencies": { "node-addon-api": ">= 6.1.0" }, "optionalPeers": ["node-addon-api"] }, "sha512-+/ZS90YK/rYfVOHtGLHkGffVsnmD/MAKaBHio+Y4XAtg75RLr4cveV/w0jTkUdLM1CcAlaRgG76mpIemWAlk0A=="],
@@ -1273,7 +1273,7 @@
"robomp-web": ["robomp-web@workspace:python/robomp/web"],
"rolldown": ["rolldown@1.2.3", "", { "dependencies": { "@oxc-project/types": "=0.143.0", "@rolldown/pluginutils": "^1.0.0" }, "optionalDependencies": { "@rolldown/binding-android-arm64": "1.2.3", "@rolldown/binding-darwin-arm64": "1.2.3", "@rolldown/binding-darwin-x64": "1.2.3", "@rolldown/binding-freebsd-x64": "1.2.3", "@rolldown/binding-linux-arm-gnueabihf": "1.2.3", "@rolldown/binding-linux-arm64-gnu": "1.2.3", "@rolldown/binding-linux-arm64-musl": "1.2.3", "@rolldown/binding-linux-ppc64-gnu": "1.2.3", "@rolldown/binding-linux-s390x-gnu": "1.2.3", "@rolldown/binding-linux-x64-gnu": "1.2.3", "@rolldown/binding-linux-x64-musl": "1.2.3", "@rolldown/binding-openharmony-arm64": "1.2.3", "@rolldown/binding-win32-arm64-msvc": "1.2.3", "@rolldown/binding-win32-x64-msvc": "1.2.3" }, "bin": { "rolldown": "./bin/cli.mjs" } }, "sha512-rn9wpmxplLf7NLNyCk9FyWh3FM43DbY8jOzCdEPzH7uflhTftRbCEpqi6Ly2osgoU8OwObtmavMbWLaWy4LX7A=="],
"rolldown": ["rolldown@1.2.4", "", { "dependencies": { "@oxc-project/types": "=0.144.0", "@rolldown/pluginutils": "^1.0.0" }, "optionalDependencies": { "@rolldown/binding-android-arm64": "1.2.4", "@rolldown/binding-darwin-arm64": "1.2.4", "@rolldown/binding-darwin-x64": "1.2.4", "@rolldown/binding-freebsd-x64": "1.2.4", "@rolldown/binding-linux-arm-gnueabihf": "1.2.4", "@rolldown/binding-linux-arm64-gnu": "1.2.4", "@rolldown/binding-linux-arm64-musl": "1.2.4", "@rolldown/binding-linux-ppc64-gnu": "1.2.4", "@rolldown/binding-linux-s390x-gnu": "1.2.4", "@rolldown/binding-linux-x64-gnu": "1.2.4", "@rolldown/binding-linux-x64-musl": "1.2.4", "@rolldown/binding-openharmony-arm64": "1.2.4", "@rolldown/binding-win32-arm64-msvc": "1.2.4", "@rolldown/binding-win32-x64-msvc": "1.2.4" }, "bin": { "rolldown": "./bin/cli.mjs" } }, "sha512-rSr7irW0K7QRWzjdJXqZowkcRdDtjRduh43rBltnVKd0VFq839l1lJoDvGJb6gl7+4rTTCrPWu+YfujUL8Ug7w=="],
"safe-buffer": ["safe-buffer@5.2.1", "", {}, "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ=="],
@@ -1459,7 +1459,7 @@
"@tailwindcss/oxide-wasm32-wasi/@emnapi/wasi-threads": ["@emnapi/wasi-threads@1.2.3", "", { "dependencies": { "tslib": "^2.4.0" }, "bundled": true }, "sha512-ELEBe8PsLvvJ6QMr0zLt8ffvOHW/dc1m3CEzNMg7aJUv3bMaoDtw2TXyDAwkYBuroxxuHEwhRTLJSe5sya547g=="],
"@tailwindcss/oxide-wasm32-wasi/@napi-rs/wasm-runtime": ["@napi-rs/wasm-runtime@1.2.2", "", { "dependencies": { "@tybys/wasm-util": "^0.10.3" }, "peerDependencies": { "@emnapi/core": "^1.7.1 || ^2.0.0-alpha.3", "@emnapi/runtime": "^1.7.1 || ^2.0.0-alpha.3" }, "bundled": true }, "sha512-JfB4kuJQjaoHuCTseIINHtHWeJnvgEcxjwA5t/Y00ZgaOO1Crz3fjT/p8kT28zA/Caz7oiUMn3d6H2yOVCVwuw=="],
"@tailwindcss/oxide-wasm32-wasi/@napi-rs/wasm-runtime": ["@napi-rs/wasm-runtime@1.2.3", "", { "dependencies": { "@tybys/wasm-util": "^0.10.3" }, "peerDependencies": { "@emnapi/core": "^1.7.1 || ^2.0.0-alpha.4", "@emnapi/runtime": "^1.7.1 || ^2.0.0-alpha.4" }, "bundled": true }, "sha512-UMduMbqO5s5zF2NkNacMT/yK5Y5QiKvWr2+50bzIIxFDwVJ2h49b+oyjaCGPhJxd2/gC2x39EHv/gHVuu36x2Q=="],
"@tailwindcss/oxide-wasm32-wasi/@tybys/wasm-util": ["@tybys/wasm-util@0.10.3", "", { "dependencies": { "tslib": "^2.4.0" }, "bundled": true }, "sha512-F3fo1MYrRJYL3zER0OUOmkutjr1Vp23m7OsSgp7nq4SP6OqX6C/56XFIPAl5bt3zaBRjmW7SGz3u/6LwFpYcOg=="],
+7
View File
@@ -15,6 +15,11 @@ saveTextLockfile = true
[test]
# bun test does NOT honor .gitignore; prune robomp's repo clones and
# scratch dirs so a root-level `bun test` doesn't walk into them.
#
# `bazel-*` are Bazel's convenience symlinks, and `bazel-oh-my-pi` points at
# the workspace root itself: without this the scanner recurses through that
# loop, exhausts file descriptors, and every test that pipes a child process's
# stdio fails with empty output or EPIPE.
pathIgnorePatterns = [
"**/node_modules/**",
"**/.git/**",
@@ -26,6 +31,8 @@ pathIgnorePatterns = [
"python/robomp/data/**",
".wt/**",
".worktrees/**",
"bazel-*/**",
"**/bazel-out/**",
]
[run]
+7 -1
View File
@@ -741,7 +741,7 @@ fn process_input(
return Ok(result);
}
if !options.no_run_if_empty || have_pending_command {
if have_pending_command || (!options.no_run_if_empty && builder_options.replace.is_none()) {
result.combine(current_builder.execute(host)?);
}
@@ -1468,6 +1468,12 @@ mod tests {
assert_eq!(out, "\n");
}
#[test]
fn replace_mode_skips_empty_input_without_r() {
let result = run_simple(&["-I", "{}", "echo", "{}"], "");
assert_eq!(result, (0, String::new(), String::new()));
}
#[test]
fn verbose_echoes_command_line_to_stderr() {
let (code, out, err) = run_simple(&["-t", "echo", "a"], "b\n");
+1 -1
View File
@@ -257,7 +257,7 @@ fn create_windows_napi_tokio_runtime() -> Option<tokio::runtime::Runtime> {
/// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in
/// `packages/natives/native/index.js` (which derives the name from
/// `package.json#version`).
#[napi(js_name = "__piNativesV17_3_4")]
#[napi(js_name = "__piNativesV17_3_5")]
pub const fn pi_natives_version_sentinel() {}
/// Native module entry point: install crash diagnostics before any tool can
+41 -1
View File
@@ -1000,7 +1000,19 @@ fn split_into_tokens_with_ansi(line: &[u16]) -> SmallVec<[Vec<u16>; 4]> {
if line[i] == ESC
&& let Some(seq_len) = ansi_seq_len_u16(line, i)
{
pending_ansi.extend_from_slice(&line[i..i + seq_len]);
let seq = &line[i..i + seq_len];
// A sequence that follows visible content closes it (color reset,
// underline off) and must ride along with that token so the closer
// cannot migrate into whitespace discarded at a soft wrap (#8582).
// A sequence after whitespace opens the *next* token's style, so it
// waits for it: gluing it to the whitespace token would keep that
// space alive past the wrap point and open the style on the line
// being broken instead of the one carrying the styled word.
if current.is_empty() || in_whitespace {
pending_ansi.extend_from_slice(seq);
} else {
current.extend_from_slice(seq);
}
i += seq_len;
continue;
}
@@ -2003,6 +2015,34 @@ mod tests {
assert!(second.contains("world"));
}
#[test]
fn test_wrap_text_with_ansi_keeps_trailing_color_reset_before_soft_wrap() {
let data = to_u16("plain \x1b[33mcode\x1b[39m next");
let lines = wrap_text_with_ansi_impl(&data, 10, DEFAULT_TAB_WIDTH);
let actual: Vec<String> = lines
.iter()
.map(|line| String::from_utf16_lossy(line))
.collect();
assert_eq!(actual, ["plain \x1b[33mcode\x1b[39m", "next"]);
}
#[test]
fn test_wrap_text_with_ansi_defers_style_open_after_trailing_space() {
let data = to_u16("read this thread \x1b[4mhttps://example.com/very/long/path\x1b[24m");
let lines = wrap_text_with_ansi_impl(&data, 40, DEFAULT_TAB_WIDTH);
let actual: Vec<String> = lines
.iter()
.map(|line| String::from_utf16_lossy(line))
.collect();
// The underline opens on the wrapped line that carries the URL: neither
// the discarded space nor a stray `4m`/`24m` pair stays on the head line.
assert_eq!(actual[0], "read this thread");
assert!(actual[1].starts_with("\x1b[4m"));
assert!(actual[1].contains("https://"));
}
#[test]
fn test_wrap_text_with_ansi_resets_strike_without_resetting_colors() {
let data =
+1 -1
View File
@@ -42,7 +42,7 @@ Filtering behavior:
- `enableProjectConfig: false` removes project-level entries (`_source.level === "project"`).
- `enabled: false` entries are suppressed unless the active-profile user `enabledServers` allowlist names them; the user `disabledServers` denylist always suppresses a same-named entry.
- Exa servers are filtered out by default and API keys are extracted for native Exa tool integration; browser automation MCP servers are filtered when `filterBrowser` is true.
- Exa servers are filtered out by default and API keys are extracted for native Exa tool integration, unless the config explicitly requests Exa tools the native integration does not provide (`web_fetch_exa`, `web_search_advanced_exa`); browser automation MCP servers are filtered when `filterBrowser` is true.
Result includes both `configs` and `sources` (metadata used later for provider labeling).
+1 -1
View File
@@ -54,7 +54,7 @@ Current retryable categories include:
The normalized classifier recognizes the transient categories above from structured flags/status and provider-aware text patterns. Classifier refusals remain a separate typed `stopDetails` decision.
Beyond `isRetryableError(...)`, empty generic aborts may enter the same retry engine when no user, dispose, or streaming-edit-guard abort is in progress. An interrupted turn whose tool calls already have matching results can also be continued safely: the failed assistant/tool-result sequence is preserved so completed side effects are not replayed. Resolved stream stalls use the same preserve-and-continue path.
Beyond `isRetryableError(...)`, empty generic aborts may enter the same retry engine when no user, dispose, or streaming-edit-guard abort is in progress. An interrupted turn whose tool calls already have matching results can also be continued safely: the failed assistant/tool-result sequence is preserved so completed side effects are not replayed. Resolved stream stalls and HTTP/2 stream resets (`NGHTTP2_INTERNAL_ERROR`, `NGHTTP2_REFUSED_STREAM`, `HTTP2StreamReset`) use the same preserve-and-continue path. Cursor idle-stall recovery still requires the exec-resolved marker (the Connect stream may still be open); an HTTP/2 RST does not, because the stream is already dead.
Retry state is owned by `TurnRecovery`:
+8 -5
View File
@@ -235,13 +235,16 @@ Reasoning fields are not interchangeable.
- Compat needs both policies: disable reasoning for any tool choice, and disable
reasoning only for forced tool choice.
### xAI Grok through Responses/SuperGrok
### xAI Grok through Responses (`xai` and `xai-oauth`)
Keep these independent:
Both the paid API-key provider (`xai` / `XAI_API_KEY`) and SuperGrok OAuth
(`xai-oauth`) chat over `https://api.x.ai/v1/responses`. Keep these independent:
- omit `reasoning.effort`
- include or drop encrypted reasoning replay
- filter reasoning-history wrappers
- omit `reasoning.effort` unless the model is on the Grok effort-capable allowlist
- omit `reasoning.summary` (the host rejects it; do not fall back to `"auto"`)
- omit presence/frequency penalties (`/v1/responses` rejects them for every Grok model)
- include `reasoning.encrypted_content` on the request
- replay encrypted reasoning items on later turns
Some models reject only one of those fields; do not collapse them into one
"Grok mode" branch.
+2 -2
View File
@@ -152,7 +152,7 @@ The Anthropic provider (`packages/ai/src/providers/anthropic.ts`) implements the
- **Claude Code Fingerprint Headers & Betas**: Default headers include `anthropic-version: 2023-06-01`, `anthropic-dangerous-direct-browser-access: true`, `x-app: cli`, and `User-Agent: claude-cli/2.1.220 (external, claude-desktop)` (`coworkUserAgent`). Active beta flags (`buildCoworkBetas`) include `claude-code-20250219`, `interleaved-thinking-2025-05-14`, `thinking-token-count-2026-05-13`, `context-management-2025-06-27`, `prompt-caching-scope-2026-01-05`, `mid-conversation-system-2026-04-07`, `advanced-tool-use-2025-11-20`, `effort-2025-11-24`, and `fallback-credit-2026-06-01` (`context-1m-2025-08-07` is omitted to avoid 429 credit errors on subscription tokens, #7238). Fingerprint metadata (`generateClaudeCloakingUserId`, `deriveClaudeDeviceId`, `generateClaudeJsonUserId`) generates device/session IDs. Billing attestation headers (`createClaudeBillingHeader`, `wrapFetchForCch`, `patchCch`) embed `cch=00000` XXHash64 hashes into `system[0]`.
- **System-Prompt Injection**: `buildAnthropicSystemBlocks` (`packages/ai/src/providers/anthropic.ts`) automatically prepends `claudeCodeSystemInstruction` ("You are a Claude agent, built on Anthropic's Claude Agent SDK.") as `system[0]` for OAuth credentials. Mid-conversation system messages in turn history are enabled for Opus 4.8+ / Sonnet 5+ via `mid-conversation-system-2026-04-07`.
- **Thinking Signatures & Redacted Thinking**: Replaying modified or unsigned thinking blocks causes Anthropic API errors (`invalid signature in thinking block`). `convertAnthropicMessages` converts `ThinkingContent` and `RedactedThinkingContent` (`type: "redacted_thinking"`, `data`). `maybeAddReplayUnsignedThinkingHint` attaches recovery hints on signature errors, while `unwrapAnthropicThinkingEnvelope` strips legacy `<thinking>` XML wrappers.
- **Tool Use Replay & Prefixes**: `encodeAnthropicToolName` / `decodeAnthropicToolName` (`packages/ai/src/providers/anthropic.ts`) prefixes custom tool names with `_` (`claudeToolPrefix`) when using OAuth to prevent collisions with built-in tools (`web_search`, `code_execution`, `text_editor`, `computer`). Server-executed web searches (`ServerToolUseBlockParam`, `WebSearchToolResultBlockParam` in `anthropic-wire.ts`) are detected via `isAnthropicWebSearchHistoryBlock` for turn replay. Empty tool errors are filled by `ensureErrorToolResultWireContent`.
- **Tool Use Replay & Prefixes**: `encodeAnthropicToolName` / `decodeAnthropicToolName` (`packages/ai/src/providers/anthropic.ts`) prefixes custom tool names with `_` (`claudeToolPrefix`) when using OAuth to prevent collisions with built-in tools (`web_search`, `code_execution`, `text_editor`, `computer`). Server-executed web searches and tool searches (`AnthropicServerToolHistoryBlockParam` in `anthropic-wire.ts`) are detected via `isAnthropicServerToolHistoryBlock` for turn replay. Empty tool errors are filled by `ensureErrorToolResultWireContent`.
- **Strict-Tool Schema Normalization & Fallback**: `normalizeAnthropicToolSchema` and `normalizeAnthropicStrictSchema` strip unsupported JSON schema keywords (e.g. `minItems`/`maxItems` on objects) for the `structured-outputs-2025-12-15` beta. If a strict tool schema causes HTTP 400, `streamAnthropicOnce` calls `dropAnthropicStrictTools` and automatically retries without strict mode.
- **Adaptive vs Budget Thinking**: `ThinkingConfigParam` (`anthropic-wire.ts`) supports budget thinking (`{ type: "enabled", budget_tokens: N }` enforced by `ensureMaxTokensForThinking`) and adaptive thinking (`{ type: "adaptive" }` paired with `output_config: { effort: level }` via `effort-2025-11-24` beta). Forced tool choices (`disableThinkingIfToolChoiceForced`) automatically disable thinking.
- **Prompt Cache Breakpoints**: `applyPromptCaching` (`packages/ai/src/providers/anthropic.ts`) attaches `{ type: "ephemeral", scope: "global" }` breakpoints to system prompts (`cacheSystemPrefixBreakpoints`), tool definitions, and historical user turns. `enforceCacheControlLimit` caps total breakpoints to 4 per request.
@@ -1456,7 +1456,7 @@ Umans AI Coding Plan is a proxy service for AI coding models, operating via the
### Auth & usage
- **Auth**: Uses `UMANS_AI_CODING_PLAN_API_KEY` environment variable or `/login umans` key prompt (`packages/ai/src/registry/umans.ts`, `packages/ai/src/registry/registry.ts`). Key validation executes a lightweight Anthropic messages call (`max_tokens: 1`) to `https://api.code.umans.ai/v1/messages`.
- **Usage endpoint**: Fetches quota and rate limit status from `GET /v1/usage` (`packages/ai/src/usage/umans.ts`) using `Authorization: Bearer <key>`.
- **Limits surfaced**: Returns a rolling 5-hour request limit (`umans:requests`) and an instantaneous session concurrency limit (`umans:concurrency`). Also surfaces low-priority status notes when rate-limit bursts occur.
- **Limits surfaced**: Returns a rolling 5-hour request split into a model-weighted soft cap (`umans:requests:soft`, the "effective requests" contract) and a raw burst ceiling (`umans:requests:hard`, `hard_cap`), plus an instantaneous session concurrency limit (`umans:concurrency`). The soft cap only ever warns — `exhausted` is reserved for the burst ceiling, where throttling actually starts. Payloads without a reported burst ceiling (`hard_cap`) collapse to a single weighted `umans:requests` row that can exhaust at the effective-request limit, so request exhaustion is never unreportable; legacy payloads without weighted counters fall back to a single raw `umans:requests` row. In both single-row shapes the weighted counter (when present) stays authoritative — raw burst traffic above the limit never fabricates an exhausted state. Also surfaces low-priority status notes when rate-limit bursts occur.
### Catalog model handling
- **Descriptor & discovery**: Registered as `umans` with default model `umans-coder` (`packages/catalog/src/provider-models/descriptors.ts`). Dynamic discovery fetches model details from `GET /v1/models/info` (`packages/catalog/src/provider-models/openai-compat.ts`).
+4 -2
View File
@@ -47,7 +47,9 @@ The custom template keeps these generated surfaces:
- always-apply rules and the rulebook listing;
- secret-redaction guidance when enabled.
The separate project/environment footer remains and carries workstation data, deeper-directory context pointers, optional workspace information, current date/cwd, and the final completion requirements. Optional extra system blocks, such as computer-tool safety and active nested-repository context, also remain when applicable.
The separate project/environment footer remains and carries workstation data, deeper-directory context pointers, optional workspace information, and the final completion requirements. Optional extra system blocks, such as computer-tool safety and active nested-repository context, also remain when applicable.
The current date and working directory no longer live in the footer: they are emitted as a `<system-reminder>` block on the first user turn of each provider request (`date-cwd-reminder.md`). Keeping per-request bytes out of the system prompt lets open-weight providers (DeepSeek, Qwen, GLM, …) that render tool schemas after the system content keep their prefix cache, and lets a session crossing midnight refresh the date without rebuilding the prompt (#7404).
What disappears is the content unique to the default instruction template: its built-in role/personality text, tool inventory and general tool policy, internal-URL catalog, exploration/delegation/workflow rules, and `xd://` protocol guidance. Generated skills and rules are **not** lost; the custom template renders them explicitly.
@@ -79,7 +81,7 @@ on
{{#if hasMemoryRoot}}Memory enabled.{{/if}}
```
those characters reach the model literally. Internal values such as `cwd`, `date`, `skills`, `rules`, and `toolRefs` are private template implementation details, not a user templating API.
those characters reach the model literally. Internal values such as `cwd`, `skills`, `rules`, and `toolRefs` are private template implementation details, not a user templating API. The calendar date is deliberately not exposed as a template value anymore — it rides the per-request first-turn reminder instead (see above).
## Recipes
+5 -5
View File
@@ -65,8 +65,8 @@
| Field | Type | Required | Description |
| --- | --- | --- | --- |
| `all` | `boolean` | No | Close every known tab. Omitted closes only `name`. |
| `kill` | `boolean` | No | When a tab release drops a spawned-app browser handle to refcount 0, also terminate its process tree. Has no effect on headless shutdown and only disconnects connected CDP browsers. |
| `all` | `boolean` | No | Release every known managed tab. Omitted releases only `name`. Tool-owned headless pages and owned cmux surfaces close; spawned, connected, and relay pages remain open unless `kill: true` terminates a spawned browser. |
| `kill` | `boolean` | No | When a tab release drops a spawned-app browser handle to refcount 0, also terminate its process tree. Has no effect on headless shutdown; connected and relay browsers are only disconnected. |
### `action: "run"`
@@ -160,7 +160,7 @@ The tool returns one result per call; no streaming partial output is emitted fro
18. `tab.click()` uses a custom retry loop for `text/...` selectors to find an actionable visible match; other selectors use `page.locator(...).click()`. Interactive actions (`click`/`fill`/`type`/`press`/`scroll`/`drag`/`scrollIntoView`/`select`/`uploadFile`) and the `waitFor*` helpers run under a per-op deadline (`min(cellBudget − slack, ceiling)`) threaded into both the puppeteer `signal` and `.setTimeout()`, so a stalled helper aborts the CDP action and rejects with a named `tab.<op> timed out after <ms>ms` that leaves cell budget — never the opaque whole-cell timeout. `goto`/`evaluate` stay uncapped.
19. `tab.screenshot()` captures the page or selected element as PNG, resizes a model copy, saves under `browser.screenshotDir` or the OS temp directory, returns that path, records metadata, and optionally emits text plus image content.
20. `display()` calls accumulate in an array. After code finishes, the worker posts `{ displays, returnValue, screenshots }`; `BrowserTool.#run()` appends the return value as trailing text content when not `undefined`.
21. `close` releases one tab or all tabs via `releaseTab()` / `releaseAllTabs()`. Each tab aborts pending runs, asks the worker to close, waits up to `750` ms for a `closed` ack, terminates the worker, decrements browser refcount, and disposes the browser handle when refcount reaches zero.
21. `close` releases one managed tab handle or all handles via `releaseTab()` / `releaseAllTabs()`. Each tab aborts pending runs, asks the worker to clean up, waits up to `750` ms for a `closed` ack, terminates the worker, decrements browser refcount, and disposes the browser handle when refcount reaches zero. Headless workers close their tool-owned page; attach workers disconnect without closing spawned, connected, or relay pages.
## Modes / Variants
- **Action dispatch**
@@ -251,14 +251,14 @@ The tool returns one result per call; no streaming partial output is emitted fro
- Use `read` for static URLs; use `browser` when JavaScript execution, authentication, or interaction is required. A tab must be opened before `run`, and named tabs persist until closed.
- `run` code has full Node/Bun and session-tool access; it is not sandboxed.
- `loadPuppeteer()` and `loadPuppeteerInWorker()` temporarily redirect `cwd` to a safe Puppeteer directory before importing `puppeteer-core`, because Puppeteer probes the current working directory during module load.
- Headless launch prefers a detected system Chrome/Chromium, then `PUPPETEER_EXECUTABLE_PATH`, and only then downloads Chromium.
- Headless launch resolves its executable in this order: `PUPPETEER_EXECUTABLE_PATH` always wins; otherwise, on macOS the isolated Chrome for Testing binary (`com.google.chrome.for.testing`) is preferred over a detected system Chrome and downloaded on first use, falling back to system Chrome only when Chrome for Testing cannot be obtained (a headless daemon launched from a system `Google Chrome.app` bundle shares its `com.google.Chrome` LaunchServices identity, so macOS can route the user's link clicks to the daemon — #8673). On other platforms a detected system Chrome/Chromium is preferred, then a downloaded Chrome for Testing.
- Headless launch always passes `--no-sandbox`, `--disable-setuid-sandbox`, `--disable-blink-features=AutomationControlled`, and a `--window-size=...` matching the initial viewport. It also ignores Puppeteer default args `--disable-extensions`, `--disable-default-apps`, and `--disable-component-extensions-with-background-pages`.
- Proxy-related env vars only affect headless launch argv (shared and local): `PUPPETEER_PROXY`, `PUPPETEER_PROXY_BYPASS_LOOPBACK`, and `PUPPETEER_PROXY_IGNORE_CERT_ERRORS`. For the shared daemon they are baked in at first launch and take effect again after the daemon's next cold start.
- Stealth patches are applied only in headless mode. Spawned or externally connected browsers are intentionally left untouched.
- Relay mode drives an existing user browser and receives no stealth patches. Anything that can reach the relay endpoint can drive logged-in tabs; the built-in server binds loopback, and an optional shared token gates the extension connection.
- `applyStealthPatches()` also strips Puppeteer's `//# sourceURL=__puppeteer_evaluation_script__` suffix from CDP `Runtime.evaluate` / `Runtime.callFunctionOn` payloads.
- `tab.extract()` reads `page.content()`, runs Readability first, then falls back to the first non-empty of `[data-pagefind-body]`/`main article`/`article`/`main`/`[role='main']`/`body`, and returns `null` if neither extraction path yields content.
- `close(all: true, kill: false)` disconnects from spawned, connected, and relay browsers when the last tab closes but leaves spawned app processes and the user's Chrome running.
- `close(all: true, kill: false)` disconnects from spawned, connected, and relay browsers when the last managed tab is released but leaves their pages, spawned app processes, and the user's Chrome running. `kill: true` additionally terminates spawned-app processes; it never closes or kills connected or relay browsers.
- Headless orphan cleanup is best-effort: if a worker dies before closing its page, the supervisor searches browser targets by `targetId` and closes that page.
- Console methods inside `run` do not appear in tool output; they are forwarded as debug/warn/error logs through the worker transport.
- Raw page request interception is run-scoped. At run end the worker removes user `request` handlers, disables interception, and releases held requests; cleanup failure marks the tab for recovery.
+1 -1
View File
@@ -211,7 +211,7 @@ Uses the same location normalization and output shape as `definition`, but sends
**Execution**
- Workspace mode first invalidates the per-cwd configuration cache, reloads configuration from disk, and then reloads every newly configured non-custom LSP server.
- Single-file mode keeps the cached configuration and reloads the primary server for that file.
- Both modes clear matching recent initialization failures before starting a server. `reloadServer()` then tries the `rust-analyzer/reloadWorkspace` request, falls back to a `workspace/didChangeConfiguration` notification with `{ settings: {} }`, and finally tears down the client so the next request cold-starts it. For a shared-mux client, teardown first sends the mux restart notification so the shared server—not only this session's link—is replaced.
- Both modes clear matching recent initialization failures before starting a server. For rust-analyzer servers, `reloadServer()` first tries the `rust-analyzer/reloadWorkspace` request (only rust-analyzer implements it; sending it to other servers such as Roslyn can crash them, so it is gated on the server binary/name). Every server then falls back to a `workspace/didChangeConfiguration` notification with `{ settings: {} }`, and finally tears down the client so the next request cold-starts it. For a shared-mux client, teardown first sends the mux restart notification so the shared server—not only this session's link—is replaced.
**Output text**
- One line per server: `Reloaded <server>`, `Restarted <server>`, or `Failed to reload <server>: ...`.
+60 -60
View File
@@ -217,9 +217,9 @@
url = "https://registry.npmjs.org/@huggingface/jinja/-/jinja-0.5.9.tgz";
hash = "sha512-uWTG+l3VJRsl7EXxYizuL3P+cCPoc3cRqbWWRcQN0FhejRfbdq0RNhCmbY/YDtnTcz9icdLYuLDjsnz4d8JMuw==";
};
"@huggingface/tasks@0.21.33" = fetchurl {
url = "https://registry.npmjs.org/@huggingface/tasks/-/tasks-0.21.33.tgz";
hash = "sha512-efQa8g+WjPwzlwk7Wl4eDgbM1Jq+WooUw/b1+eKl2LVUdPMhnbwsFNMJJxhEywaLJvJBy93+jjna6855U9GePw==";
"@huggingface/tasks@0.21.34" = fetchurl {
url = "https://registry.npmjs.org/@huggingface/tasks/-/tasks-0.21.34.tgz";
hash = "sha512-NRc1vw2Q/nZQPKjlzEkXxi0YNXstI1hXoaVruvJbgeznpXtEmPD0pey/ujIWTnhmsBQIVexr/QJJ5pyIJBPrMw==";
};
"@huggingface/tokenizers@0.1.3" = fetchurl {
url = "https://registry.npmjs.org/@huggingface/tokenizers/-/tokenizers-0.1.3.tgz";
@@ -573,9 +573,9 @@
url = "https://registry.npmjs.org/@napi-rs/tar/-/tar-1.1.1.tgz";
hash = "sha512-p6q2HhUc5vwH1CNwfOcrhLoxfgn8ust8Sqlfx+sA4VzAcp1cMbvbkl99tZZlDqOjCHgQNSiTfk/yWPjl/D42qA==";
};
"@napi-rs/wasm-runtime@1.2.2" = fetchurl {
url = "https://registry.npmjs.org/@napi-rs/wasm-runtime/-/wasm-runtime-1.2.2.tgz";
hash = "sha512-JfB4kuJQjaoHuCTseIINHtHWeJnvgEcxjwA5t/Y00ZgaOO1Crz3fjT/p8kT28zA/Caz7oiUMn3d6H2yOVCVwuw==";
"@napi-rs/wasm-runtime@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/@napi-rs/wasm-runtime/-/wasm-runtime-1.2.3.tgz";
hash = "sha512-UMduMbqO5s5zF2NkNacMT/yK5Y5QiKvWr2+50bzIIxFDwVJ2h49b+oyjaCGPhJxd2/gC2x39EHv/gHVuu36x2Q==";
};
"@napi-rs/wasm-tools-android-arm-eabi@1.1.0" = fetchurl {
url = "https://registry.npmjs.org/@napi-rs/wasm-tools-android-arm-eabi/-/wasm-tools-android-arm-eabi-1.1.0.tgz";
@@ -790,9 +790,9 @@
url = "https://registry.npmjs.org/@opentelemetry/semantic-conventions/-/semantic-conventions-1.43.0.tgz";
hash = "sha512-eSYWTm620tTk45EKSedaUL8MFYI8hW164hIXsgIHyxu3VobUB3fFCu5t0hQby6OoWRPsG1KkKUG2M5UadiLiVg==";
};
"@oxc-project/types@0.143.0" = fetchurl {
url = "https://registry.npmjs.org/@oxc-project/types/-/types-0.143.0.tgz";
hash = "sha512-u6JZdLBTLotrNC9Vd6vPssINdzcCzleKAH6EJKImQb7GtYvX5keN2dxkoK44stCc4tffE6QQRtZTXVSzsLUlWA==";
"@oxc-project/types@0.144.0" = fetchurl {
url = "https://registry.npmjs.org/@oxc-project/types/-/types-0.144.0.tgz";
hash = "sha512-nuhZIOLuI6TFQ32I/WnUx+SCPY7SdSKwgnFHydAuoS1+Z4BRcaP+RRJmGzl9lw+0OFF7UmaESf7KQRXaNLHypg==";
};
"@prettier/sync@0.6.1" = fetchurl {
url = "https://registry.npmjs.org/@prettier/sync/-/sync-0.6.1.tgz";
@@ -838,61 +838,61 @@
url = "https://registry.npmjs.org/@puppeteer/browsers/-/browsers-3.0.6.tgz";
hash = "sha512-B/gKoqlFkzhvzsI6jo9K1cZz9o5ypviVv/xu8CwA4grZzyVwN+XfkT+tu8T1zrauuEXv6VhS2oGX+6NL95WcKA==";
};
"@rolldown/binding-android-arm64@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-android-arm64/-/binding-android-arm64-1.2.3.tgz";
hash = "sha512-zrJtHDcaZJ1Fp7xf4hNl+7seH9Cn/N5TwLYkhgXREtBwAd/jaqW3uqeHxpDugJLVICWg4eW44kOQEGJ1r6jCGw==";
"@rolldown/binding-android-arm64@1.2.4" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-android-arm64/-/binding-android-arm64-1.2.4.tgz";
hash = "sha512-jHC2cnyKz5xU2fhECtFl8OZ83cYNt13GZQD+0uMJ/X3o+ijmd56okHhTUwxVSHPx1IRVIJEZ1/1pPzeLCU6XKA==";
};
"@rolldown/binding-darwin-arm64@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-darwin-arm64/-/binding-darwin-arm64-1.2.3.tgz";
hash = "sha512-ieIiibVCp0tX7TLu2cafoNPv8wJyYi01ekXpbf8q2j7F4rGAhhXb/eQh7ge9DRBY78GwmRQtvjZDux7EDbA8kA==";
"@rolldown/binding-darwin-arm64@1.2.4" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-darwin-arm64/-/binding-darwin-arm64-1.2.4.tgz";
hash = "sha512-Dc5mPD8F5F/FS8i01syd7FTF6yB2fVthH/TRkjwJkzUK6EpoxHtqvZQP5Zwq80/5z19TWYHIg1KOHboCgVx/aQ==";
};
"@rolldown/binding-darwin-x64@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-darwin-x64/-/binding-darwin-x64-1.2.3.tgz";
hash = "sha512-Zh9tCon19eDXJoihx0rqKhMUlMYqzwj3aPsSuHmI4RWZh62dWUL+DJN4C5YQya5TcQBJU/Fe8+rY0jhXTQITqA==";
"@rolldown/binding-darwin-x64@1.2.4" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-darwin-x64/-/binding-darwin-x64-1.2.4.tgz";
hash = "sha512-fpDm4oBo6SqLvWUYCmFhdde3U9KH2fRNNMeAnAPAIwxRL345xutL0EtEUcuoxsoazdJGv/MuDBQHlCDrtbvqOg==";
};
"@rolldown/binding-freebsd-x64@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-freebsd-x64/-/binding-freebsd-x64-1.2.3.tgz";
hash = "sha512-nGbJWewA1wrXXZiQhjAT5rhibGfns5ZNkDVqxsO6zJ3f3YvpoDNNmGMSbbhLuXKjNScaBJVOAboztAWVespQMg==";
"@rolldown/binding-freebsd-x64@1.2.4" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-freebsd-x64/-/binding-freebsd-x64-1.2.4.tgz";
hash = "sha512-rSJoreDE/HoIzoaib6MTp5jQtCTdMHKIvItAKT/ImS6Y6Ww76oUaeMyp4Vc/fAgd/ehji068IxetHXAnqUwN9A==";
};
"@rolldown/binding-linux-arm-gnueabihf@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-linux-arm-gnueabihf/-/binding-linux-arm-gnueabihf-1.2.3.tgz";
hash = "sha512-QNniJr5Kml0kDEB98jiDOJjXNroxIIi0IXIbdYzY26Xt1pVbeP62+KnoIZLwirOymX/0jDk/2gI/bNUv7A7OIw==";
"@rolldown/binding-linux-arm-gnueabihf@1.2.4" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-linux-arm-gnueabihf/-/binding-linux-arm-gnueabihf-1.2.4.tgz";
hash = "sha512-/jm8OGHgn7oGaJu3i/qZI9spUGcJ+y/lk43ttQ/iO1tOd9NissG6o97bighBCiL+BKRngmcDuR6ikfwYdJmVuQ==";
};
"@rolldown/binding-linux-arm64-gnu@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-linux-arm64-gnu/-/binding-linux-arm64-gnu-1.2.3.tgz";
hash = "sha512-TkqEAcmmvH3I/q4114NB4RVt6241Dao48pF45uLcFGrwAaIn0iITgTAKP/dLjbN0R4buJjGb91+UHSoFmpgIWw==";
"@rolldown/binding-linux-arm64-gnu@1.2.4" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-linux-arm64-gnu/-/binding-linux-arm64-gnu-1.2.4.tgz";
hash = "sha512-tIP06BeD9EqvECBrPZ+sqdPlYrT+aYaAiu1wYziVx5elRK/ftm33JxVDy2bXGbr6J0CrtirCkR87/X5a2euEng==";
};
"@rolldown/binding-linux-arm64-musl@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-linux-arm64-musl/-/binding-linux-arm64-musl-1.2.3.tgz";
hash = "sha512-NHqjnxpsndf4MPymxteFAWHHfkTL8HjWh1KB7z23ofZ6QO2euONuxDXjat69dKZRALnGypg8k8SsK8vZJoXv1Q==";
"@rolldown/binding-linux-arm64-musl@1.2.4" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-linux-arm64-musl/-/binding-linux-arm64-musl-1.2.4.tgz";
hash = "sha512-Ql1Q0EQqVThvn9VAVlwNzsUvbSFtCMGjLpRRi4pk5i7NZZ4n5ISiLMjHYtus4VQ2PvkSw24zyaCVsiS+sXPj1w==";
};
"@rolldown/binding-linux-ppc64-gnu@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-linux-ppc64-gnu/-/binding-linux-ppc64-gnu-1.2.3.tgz";
hash = "sha512-6tbrbwfz5GB9DQ4Jwo6hy9v+vR31xZlvzZ6n5Xut6Hhx5PvrA9q/HsK8KMaYQp063iqZGXwNvZtYNLD7EM/x0w==";
"@rolldown/binding-linux-ppc64-gnu@1.2.4" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-linux-ppc64-gnu/-/binding-linux-ppc64-gnu-1.2.4.tgz";
hash = "sha512-GjbjXD4XXfN19D0LZNbmiCBUoDiRACsYHr0yaIbbn8aFsXjHZifcYqu/W5Er5X2X990WjHXFrxarn5chzItorQ==";
};
"@rolldown/binding-linux-s390x-gnu@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-linux-s390x-gnu/-/binding-linux-s390x-gnu-1.2.3.tgz";
hash = "sha512-oyuXxXmoZHjXC917IAPFAAv4wWAa0cM9afk8nx1+9/jNNOX1uPf8yDA6p7G0RypOfw/X0PQt5IfoquY1um+zSg==";
"@rolldown/binding-linux-s390x-gnu@1.2.4" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-linux-s390x-gnu/-/binding-linux-s390x-gnu-1.2.4.tgz";
hash = "sha512-p5WR0NOwaRmJ/B1b6IjEFLLivwEsf3PrdBIhRbhTCQisbo2SvHHpG4ELB/+FgQNnB88LTOF86upmJmbvZdQ2lw==";
};
"@rolldown/binding-linux-x64-gnu@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-linux-x64-gnu/-/binding-linux-x64-gnu-1.2.3.tgz";
hash = "sha512-TytMwF2KVGqP2tgd0I1OY0PAv78dZRAYcF5ssDzjM34SUXCED3uXvSd5+lHoC0bTD6eEdFz7LdQNCO1y0oVk9w==";
"@rolldown/binding-linux-x64-gnu@1.2.4" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-linux-x64-gnu/-/binding-linux-x64-gnu-1.2.4.tgz";
hash = "sha512-4/GyVjmhR+Tc6HLJvwc1sOhPqAZtySiSMesOZyX6JQ5XBxoTDEMKQzvo07NIK6nTon/SivlZqvhzvuVBNQhObQ==";
};
"@rolldown/binding-linux-x64-musl@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-linux-x64-musl/-/binding-linux-x64-musl-1.2.3.tgz";
hash = "sha512-/E9m3qstrJFVPoULV25mVQblSNExY2+kBsYe4sy0Tn0yOOgJ8wZbZt3KnRbF/XeU2Gl1STKUQnDNTqhIE5MD4A==";
"@rolldown/binding-linux-x64-musl@1.2.4" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-linux-x64-musl/-/binding-linux-x64-musl-1.2.4.tgz";
hash = "sha512-l9eeLsCNvPpmSXUej0etw/J1eqV0Jj1D5G/xG6YTijmE6dkv6E2QezgWbTfQk63v952DPqrjOCoiqxq7Bw0YUQ==";
};
"@rolldown/binding-openharmony-arm64@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-openharmony-arm64/-/binding-openharmony-arm64-1.2.3.tgz";
hash = "sha512-Kr0OcsoQI816i6HOl3vFHpd1K0eZyh76zgfj4c1nTyaTsd5r2Mj1lwM4R90y/qaCfmTn9eHy0SKwi98eitRxug==";
"@rolldown/binding-openharmony-arm64@1.2.4" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-openharmony-arm64/-/binding-openharmony-arm64-1.2.4.tgz";
hash = "sha512-e0F355MSTMm3+UOqtV3L24gFUp2N5m1f8L/7d56deik6va+AXdrt9F8LbzGpeWGWRbZEDq4m8NVnJDeBtf9DZg==";
};
"@rolldown/binding-win32-arm64-msvc@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-win32-arm64-msvc/-/binding-win32-arm64-msvc-1.2.3.tgz";
hash = "sha512-hOtMwTqnME+/gJcH/PCZ0wn0zPUjiWOgkHpxbSJpfGKMezHltx1S7/k1SitzVa7Ww2cqrDDaFbZEhcJZO8o+Jw==";
"@rolldown/binding-win32-arm64-msvc@1.2.4" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-win32-arm64-msvc/-/binding-win32-arm64-msvc-1.2.4.tgz";
hash = "sha512-AWLi0uBRYh6QlE7OKhiz+phZC0qwtij2QZmhmOdsLdFn64m7oMpooE9ICE3lhm9xMb4SpDo2WbHcxX1iFLFtqw==";
};
"@rolldown/binding-win32-x64-msvc@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-win32-x64-msvc/-/binding-win32-x64-msvc-1.2.3.tgz";
hash = "sha512-ekcqMMkI2PlhYnfzQnB/cEdYUVVJViWvoUyLrbzgDoi3Snfc1mVBwdnc306ufA5ejy8JSPjT2RlW1nQSjW7efg==";
"@rolldown/binding-win32-x64-msvc@1.2.4" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/binding-win32-x64-msvc/-/binding-win32-x64-msvc-1.2.4.tgz";
hash = "sha512-UwSDJOg3dqCAejWdxclJjCsh3Qq4vLYMDxmyHqo1btz3stK2VqgwNd3mm5tuIwzSlGIQ/1H9Hr+Zn09mrezNqQ==";
};
"@rolldown/pluginutils@1.0.1" = fetchurl {
url = "https://registry.npmjs.org/@rolldown/pluginutils/-/pluginutils-1.0.1.tgz";
@@ -1150,9 +1150,9 @@
url = "https://registry.npmjs.org/ansi-regex/-/ansi-regex-5.0.1.tgz";
hash = "sha512-quJQXlTSUGL2LH9SUXo8VwsY4soanhgo6LNSm84E1LBcE8s3O0wpdiRzyR9z/ZZJMlMWv37qOOb9pdJlMUEKFQ==";
};
"ansi-regex@6.2.2" = fetchurl {
url = "https://registry.npmjs.org/ansi-regex/-/ansi-regex-6.2.2.tgz";
hash = "sha512-Bq3SmSpyFHaWjPk8If9yc6svM8c56dB5BAtW4Qbw5jHTwwXXcTLoRMkpDJp6VL0XzlWaCHTXrkFURMYmD0sLqg==";
"ansi-regex@6.3.0" = fetchurl {
url = "https://registry.npmjs.org/ansi-regex/-/ansi-regex-6.3.0.tgz";
hash = "sha512-WpDfL7NO6j7tH88IDBNVdUJxDh9nmCteAVW9dsep846XdwF4naCBK+/tGLX3KJgcpgMRXCFlTM2hKGoK9FsdrQ==";
};
"ansi-styles@4.3.0" = fetchurl {
url = "https://registry.npmjs.org/ansi-styles/-/ansi-styles-4.3.0.tgz";
@@ -1354,9 +1354,9 @@
url = "https://registry.npmjs.org/diff/-/diff-9.0.0.tgz";
hash = "sha512-svtcdpS8CgJyqAjEQIXdb3OjhFVVYjzGAPO8WGCmRbrml64SPw/jJD4GoE98aR7r25A0XcgrK3F02yw9R/vhQw==";
};
"electron-to-chromium@1.5.402" = fetchurl {
url = "https://registry.npmjs.org/electron-to-chromium/-/electron-to-chromium-1.5.402.tgz";
hash = "sha512-/oOpMaPT6Yg+6/1XQhyIPlzgj7Ye9zf+nNM2Uh6OcE2G2oNptWazFa+qB2Pdqqbsc9KnIDzgAntoYN0dbwOXwA==";
"electron-to-chromium@1.5.405" = fetchurl {
url = "https://registry.npmjs.org/electron-to-chromium/-/electron-to-chromium-1.5.405.tgz";
hash = "sha512-bNglH7lPH5l+yHOes7Zr4VqxhOy4BQ9ZBUX4VdoFgxMpzJk7W1ZoO3Vgd9Pxa9PyjQ76sfm2aKH/nzEcCNRlew==";
};
"emnapi@1.11.3" = fetchurl {
url = "https://registry.npmjs.org/emnapi/-/emnapi-1.11.3.tgz";
@@ -1871,9 +1871,9 @@
hash = "sha512-CHhPh+UNHD2GTXNYhPWLnU8ONHdI+5DI+4EYIAOaiD63rHeYlZvyh8P+in5999TTSFgUYuKUAjzRI4mdh/p+2A==";
};
"robomp-web" = copyPathToStore ../python/robomp/web;
"rolldown@1.2.3" = fetchurl {
url = "https://registry.npmjs.org/rolldown/-/rolldown-1.2.3.tgz";
hash = "sha512-rn9wpmxplLf7NLNyCk9FyWh3FM43DbY8jOzCdEPzH7uflhTftRbCEpqi6Ly2osgoU8OwObtmavMbWLaWy4LX7A==";
"rolldown@1.2.4" = fetchurl {
url = "https://registry.npmjs.org/rolldown/-/rolldown-1.2.4.tgz";
hash = "sha512-rSr7irW0K7QRWzjdJXqZowkcRdDtjRduh43rBltnVKd0VFq839l1lJoDvGJb6gl7+4rTTCrPWu+YfujUL8Ug7w==";
};
"safe-buffer@5.2.1" = fetchurl {
url = "https://registry.npmjs.org/safe-buffer/-/safe-buffer-5.2.1.tgz";
+11
View File
@@ -11,6 +11,7 @@
ninja,
pipewire,
pkg-config,
removeReferencesTo,
rustPlatform,
rustToolchain,
source,
@@ -85,6 +86,7 @@ stdenv.mkDerivation {
cmake
ninja
pkg-config
removeReferencesTo
rustPlatform.bindgenHook
rustPlatform.cargoSetupHook
rustToolchain
@@ -186,6 +188,15 @@ stdenv.mkDerivation {
runHook postInstall
'';
# Bun serializes the build interpreter path into the bundled entrypoint's
# inert shebang. Remove its hash before Nix scans output references; this
# runs before Darwin's binary-signing fixup hook.
preFixup = ''
remove-references-to -t ${bun} "$out/bin/omp"
'';
disallowedReferences = [ bun ];
doInstallCheck = true;
installCheckPhase = ''
runHook preInstallCheck
+15 -15
View File
@@ -23,19 +23,19 @@
"@bufbuild/protoc-gen-es": "^2.12.1",
"@huggingface/transformers": "^4.2.0",
"@napi-rs/cli": "3.7.2",
"@oh-my-pi/hashline": "17.3.4",
"@oh-my-pi/omp-stats": "17.3.4",
"@oh-my-pi/omptype": "17.3.4",
"@oh-my-pi/pi-agent-core": "17.3.4",
"@oh-my-pi/pi-ai": "17.3.4",
"@oh-my-pi/pi-catalog": "17.3.4",
"@oh-my-pi/pi-coding-agent": "17.3.4",
"@oh-my-pi/pi-mnemopi": "17.3.4",
"@oh-my-pi/pi-natives": "17.3.4",
"@oh-my-pi/pi-tui": "17.3.4",
"@oh-my-pi/pi-utils": "17.3.4",
"@oh-my-pi/pi-wire": "17.3.4",
"@oh-my-pi/snapcompact": "17.3.4",
"@oh-my-pi/hashline": "17.3.5",
"@oh-my-pi/omp-stats": "17.3.5",
"@oh-my-pi/omptype": "17.3.5",
"@oh-my-pi/pi-agent-core": "17.3.5",
"@oh-my-pi/pi-ai": "17.3.5",
"@oh-my-pi/pi-catalog": "17.3.5",
"@oh-my-pi/pi-coding-agent": "17.3.5",
"@oh-my-pi/pi-mnemopi": "17.3.5",
"@oh-my-pi/pi-natives": "17.3.5",
"@oh-my-pi/pi-tui": "17.3.5",
"@oh-my-pi/pi-utils": "17.3.5",
"@oh-my-pi/pi-wire": "17.3.5",
"@oh-my-pi/snapcompact": "17.3.5",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/api-logs": "^0.220.0",
"@opentelemetry/context-async-hooks": "^2.9.0",
@@ -81,7 +81,7 @@
"@ark/schema": "0.56.2"
},
"scripts": {
"setup": "bun install && bun run build:native && bun --cwd=packages/coding-agent link && sh scripts/link-omp.sh",
"setup": "bun scripts/setup.ts",
"dev": "bun --cwd=packages/coding-agent src/cli.ts",
"dev:timing": "PI_TIMING=x bun --cwd=packages/coding-agent --preload ../utils/src/module-timer.ts src/cli.ts",
"stats": "bun --cwd=packages/coding-agent src/cli.ts stats",
@@ -95,7 +95,7 @@
"build:native": "bun --cwd=packages/natives run build",
"test": "bun scripts/ci-test-ts.ts local",
"test:ts": "bun scripts/ci-test-ts.ts local-ts",
"test:scripts": "bun test scripts/ci-release-build-binaries.test.ts scripts/musl-release.test.ts scripts/ci-release-publish.test.ts scripts/release.test.ts",
"test:scripts": "bun test scripts/ci-test-ts.test.ts scripts/ci-release-build-binaries.test.ts scripts/musl-release.test.ts scripts/ci-release-publish.test.ts scripts/release.test.ts",
"test:rs": "bun scripts/run-rs-task.ts test:rs",
"check": "bun run --parallel check:ts check:rs",
"check:ts": "bun run check:tools && bun run --workspaces --if-present check",
+10
View File
@@ -2,6 +2,16 @@
## [Unreleased]
## [17.3.5] - 2026-08-16
### Added
- Added automatic retry support for transient provider failures during one-shot completions, allowing callers such as compaction to opt in to resilient request handling.
### Fixed
- Fixed /handoff, branch summarization, and manual /compact failing outright on transient provider errors (e.g. Anthropic overloaded/429/529 responses); these operations now retry automatically instead of leaving the user's context full.
## [17.3.4] - 2026-08-14
### Fixed
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-agent-core",
"version": "17.3.4",
"version": "17.3.5",
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
"homepage": "https://omp.sh",
"author": "Can Boluk",
@@ -334,12 +334,18 @@ export async function generateBranchSummary(
];
// Call LLM for summarization
const response = await instrumentedCompleteSimple(
let response: AssistantMessage;
try {
response = await instrumentedCompleteSimple(
model,
{ systemPrompt: [SUMMARIZATION_SYSTEM_PROMPT], messages: summarizationMessages },
{ apiKey, signal, maxTokens: 2048, metadata },
{ telemetry: options.telemetry, oneshotKind: "branch_summary", completeImpl: options.completeImpl },
{ telemetry: options.telemetry, oneshotKind: "branch_summary", completeImpl: options.completeImpl, retry: {} },
);
} catch (error) {
if (signal.aborted) return { aborted: true };
throw error;
}
// Check if aborted or errored
if (response.stopReason === "aborted") {
+48 -6
View File
@@ -16,6 +16,7 @@ import {
type Message,
type MessageAttribution,
type Model,
type OneshotRetryOptions,
type ProviderSessionState,
type SimpleStreamOptions,
type Tool,
@@ -470,8 +471,8 @@ function computeMessageTokens(message: AgentMessage, options?: { excludeEncrypte
if (!options?.excludeEncryptedReasoning) fragments.push(block.data);
} else if (block.type === "anthropicServerTool") {
// Native Anthropic server-tool call/result replayed verbatim on the
// wire (server_tool_use input, web_search_tool_result
// encrypted_content). Opaque provider-replay state the provider still
// wire (server_tool_use input and opaque result content). This opaque
// provider-replay state the provider still
// bills for on same-provider replay; excluded from the compaction
// floor like other encrypted reasoning because its local byte size
// diverges from provider billing.
@@ -831,6 +832,31 @@ export interface SummaryOptions {
ctx: Context,
options: SimpleStreamOptions,
) => Promise<AssistantMessage>;
/**
* Transient-failure retry for the summarization oneshots (`generateSummary`,
* `generateShortSummary`, `generateTurnPrefixSummary`).
*
* Defaults to enabled, which is what a one-shot caller such as manual
* `/compact` needs: a single Anthropic `overloaded_error` / 429 / 529 should
* not abort compaction and leave the context full.
*
* Pass `false` when the CALLER already owns a retry loop around the whole
* compaction attempt — auto-compaction does — otherwise the two budgets
* multiply (10 outer attempts x 3 inner = 30 requests) and each outer wait
* stacks on top of the inner backoff.
*/
oneshotRetry?: OneshotRetryOptions | false;
}
/**
* Resolve the oneshot retry policy for a summarization call. Enabled by default
* so a lone transient blip cannot abort compaction; `false` opts out for callers
* that already retry the whole attempt (see `SummaryOptions.oneshotRetry`).
*/
function summaryOneshotRetry(options: SummaryOptions | undefined): OneshotRetryOptions | undefined {
const configured = options?.oneshotRetry;
if (configured === false) return undefined;
return configured ?? {};
}
function localCodexCompaction(options: SummaryOptions | undefined) {
@@ -935,7 +961,12 @@ export async function generateSummary(
providerSessionState: options?.providerSessionState,
codexCompaction: localCodexCompaction(options),
},
{ telemetry: options?.telemetry, oneshotKind: "compaction_summary", completeImpl: options?.completeImpl },
{
telemetry: options?.telemetry,
oneshotKind: "compaction_summary",
completeImpl: options?.completeImpl,
retry: summaryOneshotRetry(options),
},
);
if (response.stopReason === "error") {
@@ -1034,13 +1065,14 @@ export async function generateHandoffFromContext(
telemetry: options.telemetry,
oneshotKind: "handoff",
completeImpl: options.completeImpl,
retry: {},
});
if (response.stopReason === "error" && shouldRetryHandoffWithAutoToolChoice(response)) {
response = await instrumentedCompleteSimple(
model,
context,
{ ...requestOptions, toolChoice: "auto" },
{ telemetry: options.telemetry, oneshotKind: "handoff", completeImpl: options.completeImpl },
{ telemetry: options.telemetry, oneshotKind: "handoff", completeImpl: options.completeImpl, retry: {} },
);
}
@@ -1143,7 +1175,12 @@ async function generateShortSummary(
providerSessionState: options?.providerSessionState,
codexCompaction: localCodexCompaction(options),
},
{ telemetry: options?.telemetry, oneshotKind: "compaction_short_summary", completeImpl: options?.completeImpl },
{
telemetry: options?.telemetry,
oneshotKind: "compaction_short_summary",
completeImpl: options?.completeImpl,
retry: summaryOneshotRetry(options),
},
);
if (response.stopReason === "error") {
@@ -1719,7 +1756,12 @@ async function generateTurnPrefixSummary(
providerSessionState: options?.providerSessionState,
codexCompaction: localCodexCompaction(options),
},
{ telemetry: options?.telemetry, oneshotKind: "compaction_turn_prefix", completeImpl: options?.completeImpl },
{
telemetry: options?.telemetry,
oneshotKind: "compaction_turn_prefix",
completeImpl: options?.completeImpl,
retry: summaryOneshotRetry(options),
},
);
if (response.stopReason === "error") {
+39 -4
View File
@@ -30,6 +30,8 @@ import {
completeSimple,
type Message,
type Model,
type OneshotRetryOptions,
retryTransientCompletion,
type ServiceTier,
type SimpleStreamOptions,
type StopReason,
@@ -1663,6 +1665,23 @@ export interface InstrumentedChatSpanOptions {
ctx: Context,
options: SimpleStreamOptions,
) => Promise<AssistantMessage>;
/**
* Opt in to transient-failure retry for this oneshot (Anthropic
* `overloaded_error`, `rate_limit_error`, 429/500/502/503/529). Omitted or
* `undefined` means **no retry** — the failure is surfaced exactly as before.
*
* Deliberately opt-in rather than default-on: `oneshotKind` is free-form and
* callers may pass arbitrary `ctx.tools` / `options.toolChoice`, so this
* funnel cannot itself prove a given request is replay-safe. Re-issuing is
* only safe when the call performs no side effect and nothing consumed
* partial output — true for summaries, titles, handoffs and image
* descriptions, which parse a complete response after it resolves. Enable it
* per call site, as a reviewed decision.
*
* Pass `{}` to accept the {@link retryTransientCompletion} defaults
* (3 attempts, 500ms base backoff, `retry-after` honored).
*/
readonly retry?: OneshotRetryOptions;
}
/**
@@ -1723,10 +1742,26 @@ export async function instrumentedCompleteSimple<TApi extends Api>(
try {
return await runInActiveSpan(chatSpan, async () => {
const complete = span.completeImpl ?? completeSimple;
const message = await complete(model, ctx, {
...options,
onResponse: captureOnResponse,
});
// Opt-in only (see `retry` on InstrumentedChatSpanOptions): each attempt
// re-issues the whole request, which is safe only for replay-safe
// oneshots. `getResponseHeaders` hands the failed attempt's headers to
// the retry layer — an AssistantMessage carries none, so this is what
// makes `retry-after` on a real 429/529 actually honored.
const runOnce = () => {
// Clear first so a previous attempt's `retry-after` can never be
// reused for a later failure that arrived without headers.
capturedHeaders = undefined;
return complete(model, ctx, { ...options, onResponse: captureOnResponse });
};
const message = span.retry
? await retryTransientCompletion(runOnce, {
...span.retry,
// Framework-owned: the caller must not be able to detach the
// abort signal or the header source by passing them itself.
signal: options.signal,
getResponseHeaders: () => capturedHeaders,
})
: await runOnce();
await finishChatSpan(telemetry, chatSpan, message, {
stepNumber,
serviceTier: options.serviceTier,
@@ -230,4 +230,59 @@ describe("branch summarization", () => {
expect(messages.some(m => m.role === "toolResult")).toBe(true);
});
test("returns an aborted result when cancelled during transient retry backoff", async () => {
const reason = new Error("user cancelled branch summary");
let aborted = false;
const signal = {
get aborted() {
return aborted;
},
get reason() {
return aborted ? reason : undefined;
},
addEventListener(type: string, listener: ((event: Event) => void) | { handleEvent(event: Event): void }) {
if (type !== "abort") return;
aborted = true;
const event = new Event("abort");
if (typeof listener === "function") listener(event);
else listener.handleEvent(event);
},
removeEventListener() {},
} as unknown as AbortSignal;
const entries: SessionEntry[] = [
{
type: "message",
id: "user-1",
parentId: null,
timestamp: new Date(0).toISOString(),
message: { role: "user", content: "Summarize this branch.", timestamp: 0 },
},
];
let calls = 0;
const result = await generateBranchSummary(entries, {
model: MODEL,
apiKey: "test-api-key",
signal,
completeImpl: async () => {
calls += 1;
return {
role: "assistant",
content: [],
api: "mock",
provider: "mock",
model: "mock-model",
usage: ZERO_USAGE,
stopReason: "error",
errorStatus: 529,
errorMessage: "overloaded_error: Overloaded",
timestamp: 1,
};
},
});
expect(calls).toBe(1);
expect(result).toEqual({ aborted: true });
});
});
@@ -0,0 +1,102 @@
import { describe, expect, it } from "bun:test";
import { generateSummary } from "@oh-my-pi/pi-agent-core/compaction";
import type { AgentMessage } from "@oh-my-pi/pi-agent-core/types";
import type { AssistantMessage, Model, Usage } from "@oh-my-pi/pi-ai/types";
/**
* Defends `SummaryOptions.oneshotRetry`, the split that lets manual `/compact`
* survive a transient provider blip without inflating auto-compaction's budget.
*
* Both paths call the same `generateSummary`, so the policy cannot be a constant
* inside it. Auto-compaction wraps the whole attempt in its own retry loop
* (`session-maintenance.ts`), and a nested inner loop would multiply requests
* (10 outer x 3 inner) while stacking each outer wait on an inner backoff.
* Manual `/compact` has no outer loop: without retry, one `overloaded_error`
* aborts compaction and leaves the user's context full.
*
* A transient failure arrives as a **resolved** `AssistantMessage` with
* `stopReason: "error"`, which is why `completeImpl` returning that value —
* rather than throwing — is the realistic stub here.
*/
const emptyUsage = (): Usage =>
({
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
}) as unknown as Usage;
const model = {
id: "claude-sonnet-4-6",
provider: "anthropic",
api: "anthropic-messages",
baseUrl: "https://api.anthropic.com",
maxTokens: 8192,
} as unknown as Model;
// `ApiKey` is `string | ApiKeyResolver`; the stubbed `completeImpl` never uses it.
const apiKey = "test-key";
const messages = [{ role: "user", content: "summarize this session", timestamp: 0 }] as unknown as AgentMessage[];
function overloaded(): AssistantMessage {
return {
role: "assistant",
content: [{ type: "text", text: "" }],
provider: "anthropic",
model: "claude-sonnet-4-6",
usage: emptyUsage(),
stopReason: "error",
errorMessage: "overloaded_error: Overloaded",
errorStatus: 529,
timestamp: 0,
} as unknown as AssistantMessage;
}
function summary(text: string): AssistantMessage {
return {
role: "assistant",
content: [{ type: "text", text }],
provider: "anthropic",
model: "claude-sonnet-4-6",
usage: emptyUsage(),
stopReason: "stop",
timestamp: 0,
} as unknown as AssistantMessage;
}
describe("SummaryOptions.oneshotRetry", () => {
it("retries a transient failure by default, so manual /compact survives a blip", async () => {
let calls = 0;
const text = await generateSummary(messages, model, 10_000, apiKey, undefined, undefined, undefined, {
// No `oneshotRetry`: the manual `/compact` shape. The one real backoff
// wait (default 500ms) is the price of asserting the DEFAULT rather than
// a value this test picked for itself.
completeImpl: () => {
calls += 1;
return Promise.resolve(calls === 1 ? overloaded() : summary("recovered summary"));
},
});
expect(calls).toBe(2);
expect(text).toContain("recovered summary");
});
it("makes exactly one attempt when the caller owns the retry loop", async () => {
let calls = 0;
const attempt = generateSummary(messages, model, 10_000, apiKey, undefined, undefined, undefined, {
// What auto-compaction passes: its own loop re-runs the whole attempt.
oneshotRetry: false,
completeImpl: () => {
calls += 1;
return Promise.resolve(overloaded());
},
});
// The failure must surface for the outer loop to classify and retry.
await expect(attempt).rejects.toThrow();
expect(calls).toBe(1);
});
});
@@ -0,0 +1,135 @@
import { describe, expect, it } from "bun:test";
import { instrumentedCompleteSimple } from "@oh-my-pi/pi-agent-core/telemetry";
import type { Api, AssistantMessage, Context, Model, SimpleStreamOptions, Usage } from "@oh-my-pi/pi-ai/types";
/**
* Defends the opt-in contract of the oneshot retry funnel.
*
* `instrumentedCompleteSimple` is the single entry point for every oneshot LLM
* call in the agent (compaction summaries, handoff, branch summary, image
* inspection). A transient Anthropic failure surfaces as a **resolved**
* `AssistantMessage` with `stopReason: "error"`, so retry has to be decided
* here rather than by a try/catch anywhere above.
*
* Retry is deliberately opt-in: `oneshotKind` is free-form and callers may pass
* arbitrary `ctx.tools`, so the funnel cannot prove an arbitrary request is
* replay-safe. These tests pin both halves of that contract — omitted means one
* attempt, `{}` means the transient failure is re-issued — plus the
* attempt-local header reset that keeps a stale `retry-after` from leaking into
* a later attempt.
*/
const emptyUsage = (): Usage =>
({
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
}) as unknown as Usage;
const model = {
id: "claude-sonnet-4-6",
provider: "anthropic",
api: "anthropic-messages",
baseUrl: "https://api.anthropic.com",
} as unknown as Model<Api>;
const ctx = { systemPrompt: "s", messages: [] } as unknown as Context;
function reply(overrides: Partial<AssistantMessage> = {}): AssistantMessage {
return {
role: "assistant",
content: [],
api: "anthropic-messages",
provider: "anthropic",
model: "claude-sonnet-4-6",
usage: emptyUsage(),
stopReason: "stop",
timestamp: 0,
...overrides,
} as AssistantMessage;
}
const overloaded = (): AssistantMessage =>
reply({
stopReason: "error",
errorStatus: 529,
errorMessage: "Anthropic stream error (overloaded_error): Overloaded",
});
describe("instrumentedCompleteSimple transient retry", () => {
it("does not retry when `retry` is omitted", async () => {
let calls = 0;
const result = await instrumentedCompleteSimple(model, ctx, {} as SimpleStreamOptions, {
telemetry: undefined,
oneshotKind: "test_no_retry",
completeImpl: () => {
calls += 1;
return Promise.resolve(overloaded());
},
});
expect(calls).toBe(1);
expect(result.stopReason).toBe("error");
});
it("re-issues a transient failure when `retry: {}` is set", async () => {
let calls = 0;
const result = await instrumentedCompleteSimple(model, ctx, {} as SimpleStreamOptions, {
telemetry: undefined,
oneshotKind: "test_retry",
retry: { baseDelayMs: 1 },
completeImpl: () => {
calls += 1;
return Promise.resolve(calls === 1 ? overloaded() : reply());
},
});
expect(calls).toBe(2);
expect(result.stopReason).toBe("stop");
});
it("keeps the caller's failure contract once attempts are exhausted", async () => {
let calls = 0;
const result = await instrumentedCompleteSimple(model, ctx, {} as SimpleStreamOptions, {
telemetry: undefined,
oneshotKind: "test_exhausted",
retry: { baseDelayMs: 1, maxAttempts: 2 },
completeImpl: () => {
calls += 1;
return Promise.resolve(overloaded());
},
});
expect(calls).toBe(2);
expect(result.stopReason).toBe("error");
expect(result.errorMessage).toContain("overloaded_error");
});
it("does not reuse a previous attempt's retry-after header", async () => {
// Attempt 1 fails WITH a retry-after header; attempt 2 fails WITHOUT one.
// If headers were not cleared per attempt, the stale hint would be applied
// again and the second wait would jump from ~1ms backoff back to 300ms.
const waits: number[] = [];
let calls = 0;
await instrumentedCompleteSimple(model, ctx, {} as SimpleStreamOptions, {
telemetry: undefined,
oneshotKind: "test_header_reset",
retry: { baseDelayMs: 1, maxAttempts: 3, onRetry: info => waits.push(info.delayMs) },
completeImpl: (_model, _ctx, options) => {
calls += 1;
if (calls === 1) {
options.onResponse?.({ status: 429, headers: { "retry-after-ms": "300" } }, undefined as never);
}
return Promise.resolve(calls < 3 ? overloaded() : reply());
},
});
expect(calls).toBe(3);
expect(waits).toHaveLength(2);
expect(waits[0]).toBe(300);
// Second failure carried no header, so it must fall back to plain backoff.
expect(waits[1]).toBeLessThan(50);
});
});
+24
View File
@@ -2,6 +2,30 @@
## [Unreleased]
## [17.3.5] - 2026-08-16
### Added
- Added retryable oneshot completion support (`retryTransientCompletion`) so non-agent LLM calls correctly retry on transient provider failures (Anthropic overload/rate-limit errors, HTTP 429/500/502/503/529), honoring provider-supplied retry-after timing before giving up.
### Fixed
- Fixed xAI availability detection so paid-key-only setups correctly default to `xai/grok-4.5` instead of the free SuperGrok catalog; explicit `xai-oauth/…` selectors still work as before.
- Fixed xAI Responses requests sending unsupported parameters (reasoning summary, presence/frequency penalties) that some models rejected.
- Fixed Umans usage reporting incorrectly marking quota as exhausted based on raw request counts instead of actual weighted usage, and improved the usage display to show both a soft-cap warning and a hard exhaustion limit with an accurate countdown to reset.
- Fixed `omp usage invalidate` to fully clear stale usage data and force a fresh refresh, so upgraded subscriptions no longer show outdated quota information.
- Improved session recovery to correctly treat certain Cursor HTTP/2 connection errors as transient instead of ending the session.
- Fixed OpenAI-compatible streams (e.g. DeepSeek) that are cut off mid-generation being silently treated as a completed response instead of being retried.
- Fixed DeepSeek resource-exhaustion interruptions not being automatically retried.
- Fixed tool-call IDs being lost during same-model replay, which could break correlation with custom gateways.
- Fixed Kimi Code multi-account routing to prefer accounts with more available quota, respect usage-limit cooldowns, and keep consistent usage history across token refreshes.
- Fixed Anthropic custom signing-proxy conversations losing tool-search results and thinking content during replay.
- Fixed rare runaway response loops across model providers so they now fail gracefully instead of repeating indefinitely.
- Fixed xAI rejecting entire turns due to certain MCP tool schema shapes, restoring compatibility while isolating any remaining incompatible tools rather than failing the whole request.
- Fixed Alibaba DashScope/Bailian transient per-minute rate limits being misclassified as full quota exhaustion, causing unnecessary long backoffs instead of quick retries.
- Fixed Anthropic-compatible streams dropping thinking content, which broke replay of prior reasoning.
- Updated the Alibaba Coding Plan China login flow to point to the current Bailian API-key management console.
## [17.3.4] - 2026-08-14
### Fixed
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-ai",
"version": "17.3.4",
"version": "17.3.5",
"description": "Unified LLM API with automatic model discovery and provider configuration",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+36 -5
View File
@@ -53,7 +53,7 @@ import { cursorUsageProvider } from "./usage/cursor";
import { googleGeminiCliUsageProvider } from "./usage/gemini";
import { githubCopilotUsageProvider } from "./usage/github-copilot";
import { antigravityRankingStrategy, antigravityUsageProvider } from "./usage/google-antigravity";
import { kimiUsageProvider } from "./usage/kimi";
import { kimiRankingStrategy, kimiUsageProvider } from "./usage/kimi";
import { minimaxCodeUsageProvider } from "./usage/minimax-code";
import { ollamaCloudUsageProvider, ollamaUsageProvider } from "./usage/ollama";
import { codexRankingStrategy, openaiCodexUsageProvider } from "./usage/openai-codex";
@@ -1082,6 +1082,7 @@ const DEFAULT_RANKING_STRATEGIES = new Map<Provider, CredentialRankingStrategy>(
["openai-codex", codexRankingStrategy],
["anthropic", claudeRankingStrategy],
["google-antigravity", antigravityRankingStrategy],
["kimi-code", kimiRankingStrategy],
["zai", zaiRankingStrategy],
["opencode-go", opencodeGoRankingStrategy],
]);
@@ -2680,18 +2681,32 @@ export class AuthStorage {
}
/**
* Check if any form of auth is configured for a provider.
* Unlike getApiKey(), this doesn't refresh OAuth tokens.
* Dedicated auth for default-model availability (picker / `getAvailable`).
* Unlike {@link getApiKey}, this does not refresh OAuth tokens, and unlike
* {@link hasResolvableAuth} it ignores cross-provider env aliases so
* `XAI_API_KEY` does not auto-select SuperGrok (`xai-oauth`).
*/
hasAuth(provider: string): boolean {
if (this.#runtimeOverrides.has(provider)) return true;
if (this.#configOverrides.has(provider)) return true;
if (this.#getCredentialsForProvider(provider).length > 0) return true;
if (getEnvApiKey(provider)) return true;
if (this.#hasDedicatedEnvAuth(provider)) return true;
if (this.#fallbackResolver?.(provider)) return true;
return false;
}
/**
* Whether a request could resolve a key for this provider, including
* cross-provider env aliases (`xai-oauth` borrowing `XAI_API_KEY`).
* Use this for explicit model preflight (`xai-oauth/grok-4.5`); use
* {@link hasAuth} for auto-availability so the default picker stays on
* paid `xai` when only `XAI_API_KEY` is set.
*/
hasResolvableAuth(provider: string): boolean {
if (this.hasAuth(provider)) return true;
return Boolean(getEnvApiKey(provider));
}
/**
* True iff a dedicated, non-env credential source is configured for this
* provider — i.e. anything in the cascade EXCEPT `getEnvApiKey(provider)`.
@@ -2711,6 +2726,22 @@ export class AuthStorage {
return false;
}
/**
* Env auth that belongs to this provider, not a cross-provider alias.
*
* `getEnvApiKey("xai-oauth")` also accepts `XAI_API_KEY` so an explicit
* `xai-oauth/…` stream can still borrow the paid key. Availability and
* origin must not: otherwise an API-key-only setup marks SuperGrok as
* signed in and `pickDefaultAvailableModel` prefers `xai-oauth/grok-4.5`
* over paid `xai/grok-4.5`.
*/
#hasDedicatedEnvAuth(provider: string): boolean {
if (provider === "xai-oauth") {
return Boolean($env.XAI_OAUTH_TOKEN?.trim());
}
return Boolean(getEnvApiKey(provider));
}
/**
* Classify where a provider's auth comes from, following the same precedence
* as {@link AuthStorage.getApiKey}: runtime override → config override →
@@ -2727,7 +2758,7 @@ export class AuthStorage {
if (stored.some(credential => credential.type === "api_key" && credential.source === "login")) {
return { kind: "api_key" };
}
if (getEnvApiKey(provider)) return { kind: "env", envVar: getEnvApiKeyName(provider) };
if (this.#hasDedicatedEnvAuth(provider)) return { kind: "env", envVar: getEnvApiKeyName(provider) };
if (stored.some(credential => credential.type === "api_key")) return { kind: "api_key" };
if (this.#fallbackResolver?.(provider)) return { kind: "fallback" };
return undefined;
+6 -2
View File
@@ -9,6 +9,7 @@ import {
} from "./classes";
import {
isAccountScopedCapText,
isDashScopeTokenLimitText,
isOpaqueStatusBody,
isUsageLimitStatus,
matchesUsageLimitText,
@@ -106,7 +107,7 @@ const TRANSIENT_ENVELOPE_PATTERN = /anthropic stream envelope error:/i;
const TRANSIENT_ENVELOPE_BEFORE_START_PATTERN = /before message_start/i;
export const STREAM_READ_ERROR_PATTERN = /stream[_ -]?read[_ -]?error/i;
export const TRANSIENT_TRANSPORT_PATTERN =
/\b(?:no[_ -]?capacity|(?:high|peak)[ _-]?demand|(?:at|over|insufficient)[ _-]?capacity|capacity[ _-]?(?:exceeded|exhausted)|peak[ _-]?load)\b|overloaded|provider.?returned.?error|rate.?limit|too many requests|429|500|502|503|504|service.?unavailable|server.?error|internal.?error|retry your request|network.?error|connection.?error|connection.?refused|unable.?to.?connect\.\s*is the computer able to access the url\?|other side closed|fetch failed|upstream.?connect|upstream.?request.?failed|reset before headers|socket hang up|timed? out|timeout|terminated|retry delay|stream stall|no error details in response|HTTP2(?:StreamReset|RefusedStream|EnhanceYourCalm)|malformed.?function.?call/i;
/\b(?:no[_ -]?capacity|(?:high|peak)[ _-]?demand|(?:at|over|insufficient)[ _-]?capacity|capacity[ _-]?(?:exceeded|exhausted)|peak[ _-]?load)\b|overloaded|provider.?returned.?error|rate.?limit|too many requests|429|500|502|503|504|service.?unavailable|server.?error|internal.?error|retry your request|network.?error|connection.?error|connection.?refused|unable.?to.?connect\.\s*is the computer able to access the url\?|other side closed|fetch failed|upstream.?connect|upstream.?request.?failed|reset before headers|socket hang up|timed? out|timeout|terminated|retry delay|stream stall|no error details in response|HTTP2(?:StreamReset|RefusedStream|EnhanceYourCalm)|nghttp2_(?:internal_error|refused_stream)|stream closed with error code nghttp2_(?:internal_error|refused_stream)|malformed.?function.?call/i;
const AUTH_FAILURE_PATTERN =
/\b(?:401|403|unauthorized|forbidden|authentication|auth[_ ]?unavailable|no auth available|(?:invalid|no)[_ ]?api[_ ]?key)\b/i;
const MALFORMED_FUNCTION_CALL_PATTERN = /\bmalformed.?function.?call\b/i;
@@ -441,7 +442,10 @@ export function classify(error: unknown, api?: Api): number {
} else if (link instanceof ProviderHttpError) {
let linkKinds = 0;
const { status: codeStatus, code } = link;
if (code === "usage_limit_reached" || code === "insufficient_quota") {
if (
code === "usage_limit_reached" ||
(code === "insufficient_quota" && !isDashScopeTokenLimitText(link.message))
) {
linkKinds |= Flag.UsageLimit;
}
if (code === "overloaded_error" || code === "rate_limit_error") {
+31 -2
View File
@@ -69,6 +69,26 @@ const CN_TRANSIENT_CAP_PATTERN =
// isOpaqueStatusBody so CN transients stay in the provider backoff lane instead
// of rotating through the opaque-429 fallback.
const CN_THROTTLE_PATTERN = /速率(?:限制|过快)|频率(?:过高|过快)|过于频繁|稍后[重再]试/;
// DashScope / Bailian (Alibaba Model Studio) reports its per-minute token
// throttle (429 Throttling.AllocationQuota, type `insufficient_quota`) with
// OpenAI-compatible billing wording — "You exceeded your current quota,
// please check your plan and billing details. … (type=insufficient_quota
// param=insufficient_quota)" — and links the error-code doc's `token-limit`
// anchor. Per that doc section the error is a transient TPM/TPS cap that
// clears within the minute window, not an account-local quota exhaustion.
// The same doc anchor also covers permanent errors such as "Free allocated
// quota exceeded", so require both the anchor and the exact throttle wording.
// The identical wording WITHOUT the anchor stays quota-exhausted (OpenAI's
// real account-quota error uses the same sentence).
const DASHSCOPE_TOKEN_LIMIT_DOC_PATTERN = /error-code[^()\s]*#token-limit/i;
const DASHSCOPE_TOKEN_LIMIT_MESSAGE_PATTERN =
/\byou exceeded your current quota, please check your plan and billing details\b/i;
/** True for DashScope/Bailian's documented OpenAI-compatible TPM/TPS throttle. */
export function isDashScopeTokenLimitText(errorMessage: string): boolean {
return (
DASHSCOPE_TOKEN_LIMIT_DOC_PATTERN.test(errorMessage) && DASHSCOPE_TOKEN_LIMIT_MESSAGE_PATTERN.test(errorMessage)
);
}
const GOOGLE_RPC_ERROR_INFO_TYPE = "type.googleapis.com/google.rpc.ErrorInfo";
const LONG_RATE_LIMIT_DELAY_MS = 5 * 60 * 1000;
@@ -133,8 +153,9 @@ function isQuotaExhaustedReason(reason: RateLimitReason): boolean {
/**
* Classify a rate-limit error message into a reason category.
* Priority order: explicit details in a resource-exhausted error > QUOTA
* (Antigravity "quota will reset") > CONCURRENT_LIMIT > MODEL_CAPACITY >
* QUOTA (account) > RATE_LIMIT > QUOTA (generic) > SERVER_ERROR > bare resource-exhausted > UNKNOWN.
* (Antigravity "quota will reset") > CN quota > DASHSCOPE_TOKEN_LIMIT (TPM/TPS
* throttle) > CONCURRENT_LIMIT > MODEL_CAPACITY > QUOTA (account) > RATE_LIMIT >
* QUOTA (generic) > SERVER_ERROR > bare resource-exhausted > UNKNOWN.
*
* Bare "resource exhausted" / "resource_exhausted" maps to MODEL_CAPACITY (transient, short wait).
* Explicit details such as "quota exceeded" retain their normal classification.
@@ -162,6 +183,13 @@ export function parseRateLimitReason(errorMessage: string): RateLimitReason {
return "QUOTA_EXHAUSTED";
}
// DashScope/Bailian TPM/TPS throttle: billing-worded like OpenAI's account
// quota, but the doc anchor marks it a per-minute token cap that clears on
// its own — short backoff on the same credential, never rotation/block.
if (isDashScopeTokenLimitText(errorMessage)) {
return "RATE_LIMIT_EXCEEDED";
}
if (CONCURRENT_LIMIT_PATTERN.test(errorMessage)) {
return "CONCURRENT_LIMIT";
}
@@ -342,6 +370,7 @@ export function isOpaqueStatusBody(message: string): boolean {
export function matchesUsageLimitText(errorMessage: string): boolean {
const structuredReason = parseGoogleRpcRateLimitReason(errorMessage);
if (structuredReason !== undefined) return isQuotaExhaustedReason(structuredReason);
if (isDashScopeTokenLimitText(errorMessage)) return false;
return (
USAGE_LIMIT_PATTERN.test(errorMessage) ||
(CN_QUOTA_EXHAUSTED_PATTERN.test(errorMessage) && !CN_TRANSIENT_CAP_PATTERN.test(errorMessage)) ||
+1
View File
@@ -6,6 +6,7 @@ export * from "./auth-gateway/types";
export * from "./auth-retry";
export * from "./auth-storage";
export * from "./error/rate-limit";
export * from "./oneshot-retry";
export * from "./provider-details";
export * from "./providers/anthropic";
export * from "./providers/anthropic-client";
+230
View File
@@ -0,0 +1,230 @@
import { extractRetryHint } from "@oh-my-pi/pi-utils";
import * as AIError from "./error";
import type { AssistantMessage } from "./types";
import { getHeadersFromError, getRetryAfterMsFromHeaders, type HeadersLike } from "./utils/retry-after";
/**
* Transient-failure retry for **oneshot** (non-agent-loop) completions.
*
* Why this exists: `streamSimple`/`completeSimple` retry *auth* failures
* (credential rotation) but deliberately surface *transient* provider failures
* — Anthropic `overloaded_error`, `rate_limit_error`, HTTP 429/500/502/503/529
* — as a **resolved** `AssistantMessage` with `stopReason: "error"`. For the
* main agent turn that is correct: `TurnRecovery` owns recovery there, and it
* must refuse to replay once tool calls or visible text have streamed.
*
* Oneshots have no such hazard. A summary, title, handoff, or image
* description produces no side effects, so re-issuing the whole request is
* safe and is almost always what the caller wants. Before this helper every
* oneshot call site had to re-implement that decision, and most did not —
* failing on the first blip, or swallowing it into `null` so a transient
* overload was indistinguishable from a legitimate empty result.
*
* Classification reuses the existing provider predicates (`AIError`), so the
* set of retryable Anthropic failures stays defined in exactly one place.
* Usage limits are included: unlike the provider loop — which excludes them so
* credential rotation can own them — a oneshot has no rotation layer above it,
* and the retry hint the provider supplies (`retry-after`, "try again in ~5m")
* is honored, so waiting is the correct response.
*/
export interface OneshotRetryOptions {
/** Total attempts, including the first. Default 3. Values < 1 are treated as 1. */
maxAttempts?: number;
/** First backoff step in ms; doubles per attempt. Default 500. */
baseDelayMs?: number;
/**
* Upper bound for a single wait. Default 30_000. A provider retry hint
* longer than this aborts the retry instead of parking the caller — the
* error surfaces so higher-level recovery (or the user) can decide.
*/
maxDelayMs?: number;
/**
* Stops further attempts. Two distinct paths, both preserving the caller's
* intent: an abort already visible when an attempt settles surfaces that
* attempt's own result (`completeSimple` reports `stopReason: "aborted"`),
* while an abort that lands during the backoff wait rejects with the abort
* reason — a user cancel stays a cancel and is never relabelled as the
* provider failure we happened to be waiting on.
*
* This helper does NOT pass the signal into `run` — cancelling the in-flight
* request is the closure's job, because a per-attempt deadline must be
* rebuilt on every attempt. Construct it inside `run`
* (`signal: AbortSignal.timeout(MS)`, or `AbortSignal.any([outer, perAttempt])`);
* a deadline captured outside would fire once and then abort every retry,
* silently turning this helper into a single attempt.
*/
signal?: AbortSignal;
/**
* Headers of the attempt that just failed, used to honor `retry-after`.
*
* Load-bearing: a transient Anthropic failure arrives as a **resolved**
* `AssistantMessage`, and `AssistantMessage` carries no headers — so without
* this the real `retry-after` / `x-ratelimit-reset` values on a 429/529 are
* invisible and only the (usually hint-free) error text is available.
* Callers that already capture headers via `SimpleStreamOptions.onResponse`
* should return the latest capture here; it is read once per failed attempt.
* Thrown errors need no wiring — headers are recovered from the error itself.
*/
getResponseHeaders?: () => HeadersLike;
/** Observability hook. Fires immediately before sleeping. */
onRetry?: (info: OneshotRetryInfo) => void;
}
export interface OneshotRetryInfo {
/** 1-based index of the attempt that just failed. */
attempt: number;
maxAttempts: number;
delayMs: number;
/** True when `delayMs` came from a provider retry hint rather than backoff. */
fromRetryHint: boolean;
errorMessage: string;
/** `AIError` classification bits of the failure. */
errorId: number;
}
const DEFAULT_MAX_ATTEMPTS = 3;
const DEFAULT_BASE_DELAY_MS = 500;
const DEFAULT_MAX_DELAY_MS = 30_000;
/** Cap on pure backoff growth. A provider hint may still exceed this, up to `maxDelayMs`. */
const BACKOFF_CEILING_MS = 8_000;
const RETRY_AFTER_MS_SUFFIX = /(?:^|\s)retry-after-ms=([0-9]+(?:\.[0-9]+)?)(?=\s|$)/i;
function backoffDelayMs(attempt: number, baseDelayMs: number): number {
const growth = Math.min(baseDelayMs * 2 ** (attempt - 1), BACKOFF_CEILING_MS);
// 75-100% jitter, matching the provider loop and TurnRecovery, so a fleet of
// concurrent oneshots does not re-converge on the same instant.
return Math.round(growth * (0.75 + Math.random() * 0.25));
}
/** Retryable when the provider says transient, or when it says "wait, then retry". */
function isRetryableOneshotFailure(errorId: number, errorStatus: number | undefined, errorMessage: string): boolean {
// llama.cpp reports deterministic tool-call JSON parse failures as HTTP 500.
// Replaying the same prompt produces the same malformed output.
if (AIError.LLAMA_CPP_TOOL_CALL_PARSE_PATTERN.test(errorMessage)) return false;
if (AIError.is(errorId, AIError.Flag.ContentBlocked)) return false;
return (
AIError.isTransientStatus(errorStatus) ||
AIError.is(errorId, AIError.Flag.Transient) ||
AIError.is(errorId, AIError.Flag.UsageLimit) ||
AIError.retriable(errorId)
);
}
function sleep(delayMs: number, signal?: AbortSignal): Promise<void> {
if (delayMs <= 0) return Promise.resolve();
const { promise, resolve, reject } = Promise.withResolvers<void>();
const timer = setTimeout(() => {
signal?.removeEventListener("abort", onAbort);
resolve();
}, delayMs);
const onAbort = () => {
clearTimeout(timer);
reject(signal?.reason ?? new AIError.AbortError("oneshot retry aborted"));
};
if (signal) {
if (signal.aborted) {
clearTimeout(timer);
return Promise.reject(signal.reason ?? new AIError.AbortError("oneshot retry aborted"));
}
signal.addEventListener("abort", onAbort, { once: true });
}
return promise;
}
/**
* Run a oneshot completion, retrying transient provider failures.
*
* Handles both failure shapes: a resolved `AssistantMessage` carrying
* `stopReason: "error"` (what `completeSimple` produces) and a thrown error
* (what the raw HTTP helpers produce). A non-retryable failure is returned or
* rethrown unchanged, so existing caller error handling keeps working — this
* only removes the *first-blip* failure mode.
*/
export async function retryTransientCompletion(
run: (attempt: number) => Promise<AssistantMessage>,
options?: OneshotRetryOptions,
): Promise<AssistantMessage> {
const maxAttempts = Math.max(1, options?.maxAttempts ?? DEFAULT_MAX_ATTEMPTS);
const baseDelayMs = options?.baseDelayMs ?? DEFAULT_BASE_DELAY_MS;
const maxDelayMs = options?.maxDelayMs ?? DEFAULT_MAX_DELAY_MS;
const signal = options?.signal;
for (let attempt = 1; ; attempt++) {
let message: AssistantMessage | undefined;
let thrown: unknown;
try {
message = await run(attempt);
if (message.stopReason !== "error") return message;
} catch (error) {
thrown = error;
}
// A caller abort is never a transient failure — surface it immediately so
// cancellation stays responsive.
if (signal?.aborted) {
if (thrown !== undefined) throw thrown;
return message as AssistantMessage;
}
const errorId =
thrown !== undefined ? AIError.classify(thrown) : AIError.classifyMessage(message as AssistantMessage);
if (AIError.is(errorId, AIError.Flag.Abort) || AIError.is(errorId, AIError.Flag.UserInterrupt)) {
if (thrown !== undefined) throw thrown;
return message as AssistantMessage;
}
const errorMessage =
thrown !== undefined
? thrown instanceof Error
? thrown.message
: String(thrown)
: ((message as AssistantMessage).errorMessage ?? "unknown error");
const errorStatus = thrown !== undefined ? AIError.status(thrown) : (message as AssistantMessage).errorStatus;
const lastAttempt = attempt >= maxAttempts;
if (lastAttempt || !isRetryableOneshotFailure(errorId, errorStatus, errorMessage)) {
if (thrown !== undefined) throw thrown;
return message as AssistantMessage;
}
// Headers first: a real Anthropic 429/529 carries `retry-after` /
// `x-ratelimit-reset*` in the response, and the resolved AssistantMessage
// has none — the caller supplies them via `getResponseHeaders`. Thrown
// errors (e.g. AnthropicApiError) carry their own headers.
const headers: HeadersLike = thrown !== undefined ? getHeadersFromError(thrown) : options?.getResponseHeaders?.();
const headerHintMs = getRetryAfterMsFromHeaders(headers);
const extractedTextHintMs = extractRetryHint(undefined, errorMessage);
const suffixValue = RETRY_AFTER_MS_SUFFIX.exec(errorMessage)?.[1];
const parsedSuffixMs = suffixValue === undefined ? undefined : Number(suffixValue);
const suffixHintMs =
parsedSuffixMs !== undefined && Number.isFinite(parsedSuffixMs) && parsedSuffixMs > 0
? Math.ceil(parsedSuffixMs)
: undefined;
const textHintMs =
extractedTextHintMs === undefined && suffixHintMs === undefined
? undefined
: Math.max(extractedTextHintMs ?? 0, suffixHintMs ?? 0);
const hintMs =
headerHintMs === undefined && textHintMs === undefined
? undefined
: Math.max(headerHintMs ?? 0, textHintMs ?? 0);
// An over-cap hint means "come back much later"; parking a oneshot that
// long is worse than surfacing the failure to the caller.
if (hintMs !== undefined && hintMs > maxDelayMs) {
if (thrown !== undefined) throw thrown;
return message as AssistantMessage;
}
const backoff = backoffDelayMs(attempt, baseDelayMs);
const delayMs = Math.min(Math.max(hintMs ?? 0, backoff), maxDelayMs);
options?.onRetry?.({
attempt,
maxAttempts,
delayMs,
fromRetryHint: hintMs !== undefined && hintMs >= backoff,
errorMessage,
errorId,
});
// Aborting mid-backoff rejects with the caller's abort reason: a user
// cancel must stay a cancel, not get relabelled as the provider failure we
// happened to be waiting on.
await sleep(delayMs, signal);
}
}
@@ -27,7 +27,7 @@ import {
type AnthropicUserContentBlock,
anthropicMessagesRequestSchema,
} from "./anthropic-messages-server-schema";
import { isAnthropicWebSearchHistoryBlock } from "./anthropic-wire";
import { isAnthropicServerToolHistoryBlock } from "./anthropic-wire";
/**
* Anthropic Messages API (https://docs.anthropic.com/en/api/messages) ↔ pi-ai
@@ -216,10 +216,10 @@ function walkAssistantContent(
break;
case "server_tool_use":
case "web_search_tool_result":
if (isAnthropicWebSearchHistoryBlock(block)) {
// Native web-search call/result. Anthropic requires these
// replayed verbatim (encrypted_content included), so retain
// the block instead of flattening it to text.
case "tool_search_tool_result":
if (isAnthropicServerToolHistoryBlock(block)) {
// Anthropic requires supported server-tool call/results replayed
// verbatim, so retain each opaque block instead of flattening it.
out.push({ type: "anthropicServerTool", block: { ...block } });
} else {
// Other server tools use distinct result block types that omp
+31 -5
View File
@@ -77,6 +77,11 @@ export type ServerToolUseBlockParam = {
/** Web-search server-tool call whose matching result is replayable by omp. */
export type WebSearchServerToolUseBlockParam = ServerToolUseBlockParam & { name: "web_search" };
/** Tool-search server-tool call whose matching result is replayable by omp. */
export type ToolSearchServerToolUseBlockParam = ServerToolUseBlockParam & {
name: "tool_search_tool_regex" | "tool_search_tool_bm25";
};
/** Native web-search result replayed inside an assistant turn. */
export type WebSearchToolResultBlockParam = {
type: "web_search_tool_result";
@@ -85,18 +90,37 @@ export type WebSearchToolResultBlockParam = {
[key: string]: unknown;
};
/** True for the complete native web-search history variants omp can replay. */
export function isAnthropicWebSearchHistoryBlock(block: {
/** Native tool-search result replayed inside an assistant turn. */
export type ToolSearchToolResultBlockParam = {
type: "tool_search_tool_result";
tool_use_id: string;
content: unknown;
[key: string]: unknown;
};
/** Anthropic server-tool history variants omp can replay atomically. */
export type AnthropicServerToolHistoryBlockParam =
| WebSearchServerToolUseBlockParam
| WebSearchToolResultBlockParam
| ToolSearchServerToolUseBlockParam
| ToolSearchToolResultBlockParam;
/** True when a block is complete Anthropic server-tool history omp can replay. */
export function isAnthropicServerToolHistoryBlock(block: {
type: string;
name?: unknown;
id?: unknown;
tool_use_id?: unknown;
content?: unknown;
}): block is WebSearchServerToolUseBlockParam | WebSearchToolResultBlockParam {
}): block is AnthropicServerToolHistoryBlockParam {
if (block.type === "server_tool_use") {
return block.name === "web_search" && typeof block.id === "string" && block.id.length > 0;
const supportedName =
block.name === "web_search" ||
block.name === "tool_search_tool_regex" ||
block.name === "tool_search_tool_bm25";
return supportedName && typeof block.id === "string" && block.id.length > 0;
}
if (block.type === "web_search_tool_result") {
if (block.type === "web_search_tool_result" || block.type === "tool_search_tool_result") {
return typeof block.tool_use_id === "string" && block.tool_use_id.length > 0 && Object.hasOwn(block, "content");
}
return false;
@@ -132,6 +156,7 @@ export type ContentBlockParam =
| ToolResultBlockParam
| ServerToolUseBlockParam
| WebSearchToolResultBlockParam
| ToolSearchToolResultBlockParam
| ThinkingBlockParam
| RedactedThinkingBlockParam
| FallbackBlockParam;
@@ -319,6 +344,7 @@ export type ResponseContentBlock =
| { type: "tool_use"; id: string; name: string; input?: Record<string, unknown> | null }
| ServerToolUseBlockParam
| WebSearchToolResultBlockParam
| ToolSearchToolResultBlockParam
| { type: "fallback"; from: { model: string }; to: { model: string } };
export type ContentBlockDelta =
+15 -4
View File
@@ -79,7 +79,7 @@ import {
type Usage as AnthropicWireUsage,
type ContentBlockParam,
type FallbackParam,
isAnthropicWebSearchHistoryBlock,
isAnthropicServerToolHistoryBlock,
type MessageCreateParams,
type MessageCreateParamsStreaming,
type MessageParam,
@@ -2404,7 +2404,7 @@ const streamAnthropicOnce = (
streamedReplayUnsafeContent = true;
const block: Block = {
type: "thinking",
thinking: "",
thinking: event.content_block.thinking ?? "",
thinkingSignature: "",
[kStreamingBlockIndex]: event.index,
};
@@ -2416,6 +2416,14 @@ const streamAnthropicOnce = (
contentIndex,
partial: output,
});
if (block.thinking) {
stream.push({
type: "thinking_delta",
contentIndex,
delta: block.thinking,
partial: output,
});
}
} else if (event.content_block.type === "redacted_thinking") {
streamedReplayUnsafeContent = true;
const block: Block = {
@@ -2429,8 +2437,11 @@ const streamAnthropicOnce = (
kind: "redactedThinking",
});
} else if (
isAnthropicWebSearchHistoryBlock(event.content_block) &&
umansGatewayWebSearchHeader === undefined
isAnthropicServerToolHistoryBlock(event.content_block) &&
(umansGatewayWebSearchHeader === undefined ||
(event.content_block.type === "server_tool_use"
? event.content_block.name !== "web_search"
: event.content_block.type !== "web_search_tool_result"))
) {
streamedReplayUnsafeContent = true;
const block: Block = {
+44 -4
View File
@@ -293,11 +293,23 @@ const CURSOR_PROXY_TUNNEL_TIMEOUT_MS = 30_000;
* model reads it and should route around the capability, not retry the call.
*/
const NOT_IMPLEMENTED_SUFFIX = "not implemented by this client";
/** Bare gRPC `resource_exhausted` end-streams (also inside a Connect error message). */
const RESOURCE_EXHAUSTED_PATTERN = /resource.?exhausted/i;
const NOT_IMPLEMENTED = `Not implemented by this client`;
const conversationStateCache = new Map<string, ConversationStateStructure>();
const conversationBlobStores = new Map<string, Map<string, Uint8Array>>();
const warnedCursorKimiK3ReplayMessages = new Set<string>();
/**
* Base conversation id → rotated wire id (#8345). Cursor's backend can pin a
* per-conversation rejection (bare `resource_exhausted`, zero tokens) to one
* conversationId forever; the session is then unusable until /fork mints a
* new id. On the first such failure the id is rotated once and the cached
* state migrates, so the retry loop's next attempt starts a fresh
* conversation — the same recovery /fork performs. Keyed by the base id the
* caller derived, so a failed rotation is never repeated.
*/
const rotatedConversationIds = new Map<string, string>();
export interface CursorOptions extends StreamOptions {
customSystemPrompt?: string;
@@ -587,13 +599,19 @@ export const streamCursor: StreamFunction<"cursor-agent"> = (
h2Completion.resolve();
};
// Hoisted out of the try block: the #8345 rotation in the catch path
// needs both ids, and the catch block cannot see try-scoped consts.
let baseConversationId: string | undefined;
let conversationId: string | undefined;
let usageState: UsageState | undefined;
try {
const apiKey = options?.apiKey;
if (!apiKey) {
throw new AIError.MissingApiKeyError(undefined, "Cursor API key (access token) is required");
}
const conversationId = options?.conversationId ?? options?.sessionId ?? crypto.randomUUID();
baseConversationId = options?.conversationId ?? options?.sessionId ?? crypto.randomUUID();
conversationId = rotatedConversationIds.get(baseConversationId) ?? baseConversationId;
const blobStore = conversationBlobStores.get(conversationId) ?? new Map<string, Uint8Array>();
conversationBlobStores.set(conversationId, blobStore);
const cachedState = conversationStateCache.get(conversationId);
@@ -667,7 +685,7 @@ export const streamCursor: StreamFunction<"cursor-agent"> = (
let currentThinkingBlock: (ThinkingContent & { [kStreamingBlockIndex]: number }) | null = null;
let currentToolCall: ToolCallState | null = null;
const resolvedMcpToolCallIds = new Set<string>();
const usageState: UsageState = { sawTokenDelta: false };
usageState = { sawTokenDelta: false };
const state: BlockState = {
get currentTextBlock() {
@@ -702,7 +720,7 @@ export const streamCursor: StreamFunction<"cursor-agent"> = (
openBlockState = state;
const onConversationCheckpoint = (checkpoint: ConversationStateStructure) => {
conversationStateCache.set(conversationId, checkpoint);
conversationStateCache.set(conversationId!, checkpoint);
};
h2Request.on("response", headers => {
@@ -759,7 +777,7 @@ export const streamCursor: StreamFunction<"cursor-agent"> = (
h2Request!,
options?.execHandlers,
options?.onToolResult,
usageState,
usageState!,
requestContextTools,
onConversationCheckpoint,
).catch(error => {
@@ -873,6 +891,28 @@ export const streamCursor: StreamFunction<"cursor-agent"> = (
flushOpenToolCalls(output, stream, openBlockState);
}
const result = await AIError.finalize(error, { api: model.api, signal: options?.signal });
// #8345: a server-side per-conversation rejection surfaces as a bare
// resource_exhausted with zero tokens — the conversation is poisoned,
// not the account (sibling conversations keep working). Rotate the
// wire id once and migrate the cached state so the next attempt (the
// caller's retry loop) starts a fresh conversation, exactly like
// /fork. Only the first failure rotates; repeated failures keep the
// rotated id so a genuine account-level exhaustion is not hidden.
if (
conversationId !== undefined &&
baseConversationId !== undefined &&
usageState !== undefined &&
!usageState.sawTokenDelta &&
RESOURCE_EXHAUSTED_PATTERN.test(result.message) &&
!rotatedConversationIds.has(baseConversationId)
) {
const rotated = crypto.randomUUID();
rotatedConversationIds.set(baseConversationId, rotated);
const state = conversationStateCache.get(conversationId);
if (state) conversationStateCache.set(rotated, state);
const blobs = conversationBlobStores.get(conversationId);
if (blobs) conversationBlobStores.set(rotated, blobs);
}
output.stopReason = result.stopReason;
output.errorStatus = result.status;
output.errorId = result.id;
+78 -34
View File
@@ -3,7 +3,7 @@ import { isKimiModelId } from "@oh-my-pi/pi-catalog/identity";
import { resolveWireModelId } from "@oh-my-pi/pi-catalog/model-thinking";
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types";
import { $env, parseStreamingJson, parseStreamingJsonThrottled } from "@oh-my-pi/pi-utils";
import { $env, logger, parseStreamingJson, parseStreamingJsonThrottled } from "@oh-my-pi/pi-utils";
import { renderDemotedThinking } from "../dialect/demotion";
import * as AIError from "../error";
import { getKimiCommonHeaders } from "../registry/oauth/kimi";
@@ -44,6 +44,8 @@ import { notifyProviderResponse } from "../utils/provider-response";
import { callWithCopilotModelRetry } from "../utils/retry";
import {
adaptSchemaForStrict,
findStrictToolSchemaViolation,
flattenExclusiveRequiredRootUnion,
NO_STRICT,
normalizeSchemaForMoonshot,
sanitizeSchemaForGrammar,
@@ -1299,6 +1301,16 @@ const streamOpenAICompletionsOnce = (
flushDeepseekStripBuffer(true);
}
// Detect premature stream closure before the normal block-finalization
// sweep. Throwing after that sweep would make the error handler emit a
// second text_end/thinking_end for the same partial block.
if (streamFinishedAt === undefined && output.content.length > 0) {
throw new AIError.ProviderResponseError(
"OpenAI completions stream closed before a finish_reason was received",
{ provider: model.provider, kind: "incomplete-stream" },
);
}
if (currentBlock?.type === "toolCall") {
finishPendingToolCallBlocks();
} else {
@@ -1575,6 +1587,7 @@ function buildParams(
if (options?.minP !== undefined) {
params.min_p = options.minP;
}
if (initialCompat.supportsPenaltyAndStopParams) {
if (options?.presencePenalty !== undefined) {
params.presence_penalty = options.presencePenalty;
}
@@ -1585,14 +1598,15 @@ function buildParams(
params.frequency_penalty = options.frequencyPenalty;
}
}
if (options?.stopSequences?.length) {
}
if (options?.stopSequences?.length && initialCompat.supportsPenaltyAndStopParams) {
const seqs = options.stopSequences;
params.stop = seqs.length === 1 ? seqs[0] : seqs.slice(0, 4);
}
applyOpenAIServiceTier(params, options?.serviceTier, model);
if (context.tools?.length) {
const builtTools = convertTools(context.tools, initialCompat, toolStrictModeOverride);
const builtTools = convertTools(context.tools, initialCompat, toolStrictModeOverride, model.provider);
params.tools = builtTools.tools;
toolStrictMode = builtTools.toolStrictMode;
strictToolsApplied = builtTools.strictToolsApplied;
@@ -1653,7 +1667,10 @@ function buildParams(
params.tool_choice = "auto";
}
if (params.tool_choice === "none" && (!Array.isArray(params.tools) || params.tools.length === 0)) {
if (
(!Array.isArray(params.tools) || params.tools.length === 0) &&
(params.tool_choice === "none" || isForcedToolChoice(params.tool_choice))
) {
// `tool_choice: "none"` with no tools to gate is redundant and also
// trips LiteLLM → Bedrock: the proxy serializes the directive into a
// `toolConfig` block, and Bedrock requires `toolConfig.tools` to be
@@ -1662,6 +1679,8 @@ function buildParams(
// Side-channel turns hit this: `/btw` and IRC background replies route
// through `AgentSession.runEphemeralTurn`, which sets `context.tools = []`
// and `toolChoice: "none"` (see packages/coding-agent/src/session/agent-session.ts).
// The same empty-tools case applies after leftover-union quarantine: a
// leftover `"required"` / named force would 400 just like the bad schema.
delete params.tool_choice;
}
@@ -1808,16 +1827,19 @@ export function convertMessages(
? 40
: undefined;
const duplicateToolCallIdSuffixPrefix = compat.requiresMistralToolIds ? "dup" : undefined;
const normalizeToolCallId = (id: string): string => {
const normalizeToolCallId = (id: string, source?: AssistantMessage): string => {
if (compat.requiresMistralToolIds) return normalizeMistralToolId(id, true);
// Handle pipe-separated IDs from OpenAI Responses API
// Format: {call_id}|{id} where {id} can be 400+ chars with special chars (+, /, =)
// These come from providers like github-copilot, openai-codex, opencode
// Extract just the call_id part and normalize it
if (id.includes("|")) {
const isSameModelSource =
source !== undefined &&
source.provider === model.provider &&
source.api === model.api &&
source.model === model.id;
// Cross-model replay converts OpenAI Responses composite IDs from
// `{call_id}|{item_id}` to the Chat Completions `call_id`. Same-model
// Chat Completions IDs are provider-issued opaque correlation tokens.
if (!isSameModelSource && id.includes("|")) {
const [callId] = id.split("|");
// Sanitize to allowed chars and truncate to 40 chars (OpenAI limit)
return callId.replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 40);
}
@@ -1827,7 +1849,7 @@ export function convertMessages(
const transformedMessages = transformMessages(
context.messages,
model,
id => normalizeToolCallId(id),
(id, _target, source) => normalizeToolCallId(id, source),
maxNormalizedToolCallIdLength,
duplicateToolCallIdSuffixPrefix,
compat,
@@ -1859,8 +1881,8 @@ export function convertMessages(
return nextId;
};
const ensureToolCallId = (rawId: string, seed: string): string => {
const normalized = normalizeToolCallId(rawId);
const ensureToolCallId = (rawId: string, seed: string, source?: AssistantMessage): string => {
const normalized = normalizeToolCallId(rawId, source);
if (normalized.trim().length > 0) return normalized;
return generateFallbackToolCallId(seed);
};
@@ -2142,7 +2164,7 @@ export function convertMessages(
}
if (toolCalls.length > 0) {
assistantMsg.tool_calls = toolCalls.map((tc, toolCallIndex) => {
const toolCallId = ensureToolCallId(tc.id, `${i}:${toolCallIndex}:${tc.name}`);
const toolCallId = ensureToolCallId(tc.id, `${i}:${toolCallIndex}:${tc.name}`, msg);
rememberToolCallId(tc.id, toolCallId);
return {
id: normalizeMistralToolId(toolCallId, compat.requiresMistralToolIds),
@@ -2285,10 +2307,14 @@ function convertTools(
tools: Tool[],
compat: ResolvedOpenAICompat,
toolStrictModeOverride?: ToolStrictModeOverride,
provider?: string,
): BuiltOpenAICompletionTools {
const rejectXaiRootObjectUnion = provider === "xai" || provider === "xai-oauth";
const adaptedTools = tools.map(tool => {
const strict = !NO_STRICT && compat.supportsStrictMode !== false && tool.strict !== false;
const baseParameters = toolWireSchema(tool);
const baseParameters = rejectXaiRootObjectUnion
? flattenExclusiveRequiredRootUnion(toolWireSchema(tool))
: toolWireSchema(tool);
const adapted = adaptSchemaForStrict(baseParameters, strict);
return {
tool,
@@ -2308,8 +2334,9 @@ function convertTools(
: "none"
: "mixed";
return {
tools: adaptedTools.map(({ tool, baseParameters, parameters, strict }) => {
const wireTools: ChatCompletionTool[] = [];
let anyStrictEmitted = false;
for (const { tool, baseParameters, parameters, strict } of adaptedTools) {
const includeStrict = toolStrictMode === "all_strict" || (toolStrictMode === "mixed" && strict);
// `strict: false` is semantically distinct from omitted `strict` on some
// backends: with it absent, optional properties may be over-filled with
@@ -2318,37 +2345,44 @@ function convertTools(
// field — the `all_strict → none` collapse and `supportsStrictMode:
// false` paths deliberately keep the wire flag uniformly absent.
const includeExplicitFalse =
!includeStrict &&
tool.strict === false &&
toolStrictMode === "mixed" &&
compat.supportsStrictMode !== false;
!includeStrict && tool.strict === false && toolStrictMode === "mixed" && compat.supportsStrictMode !== false;
const wireParameters = includeStrict ? parameters : baseParameters;
return {
type: "function",
function: {
name: tool.name,
description: tool.description || "",
// Moonshot/Kimi native hosts validate against the stricter MFJS subset
// (const→enum, typed enums, no validators) and 400 otherwise.
// Grammar-constrained local backends (llama.cpp, LM Studio, vLLM)
// build a GBNF grammar from the schema and 400 with
// `Unrecognized schema: true` on the bare boolean subschema
// `toolWireSchema` emits for open fields (issue #5914).
parameters:
const emittedParameters =
compat.toolSchemaFlavor === "moonshot-mfjs"
? (normalizeSchemaForMoonshot(wireParameters) as Record<string, unknown>)
: compat.toolSchemaFlavor === "grammar"
? sanitizeSchemaForGrammar(wireParameters)
: wireParameters,
: wireParameters;
const violation = findStrictToolSchemaViolation(emittedParameters, "#", { rejectXaiRootObjectUnion });
if (violation) {
logger.warn(
`Tool "${tool.name}" omitted from the openai-completions request: its parameter schema is invalid for this provider at ${violation} (an enum/const value cannot match its declared type, or leftover xAI object-root union). Other tools are unaffected.`,
);
continue;
}
if (includeStrict) anyStrictEmitted = true;
wireTools.push({
type: "function",
function: {
name: tool.name,
description: tool.description || "",
parameters: emittedParameters,
// Only include strict if provider supports it. Some reject unknown fields.
...(includeStrict ? { strict: true } : includeExplicitFalse ? { strict: false } : {}),
},
};
}),
});
}
return {
tools: wireTools,
toolStrictMode,
strictToolsApplied:
tools.length > 0 &&
(toolStrictMode === "all_strict" || (toolStrictMode === "mixed" && adaptedTools.some(tool => tool.strict))),
strictToolsApplied: wireTools.length > 0 && anyStrictEmitted,
};
}
@@ -2380,6 +2414,16 @@ function mapStopReason(reason: ChatCompletionChunk.Choice["finish_reason"] | str
// the message to match the session retry classifier's transient-transport
// pattern (`provider.?returned.?error`) and get the turn auto-retried.
return { stopReason: "error", errorMessage: "Provider returned error finish_reason" };
case "insufficient_system_resource":
// DeepSeek kills the generation mid-stream when its inference system runs
// out of resources (docs: "the request is interrupted due to insufficient
// resource of the inference system"). Server-side capacity failure — like
// the bare `error` case, word the message to match the transient-transport
// retry pattern so the turn is auto-retried instead of pinned as an error.
return {
stopReason: "error",
errorMessage: "Provider returned error finish_reason: insufficient_system_resource",
};
default:
return {
stopReason: "error",
+12 -9
View File
@@ -39,6 +39,7 @@ import { callWithCopilotModelRetry } from "../utils/retry";
import {
adaptSchemaForStrict,
findStrictToolSchemaViolation,
flattenExclusiveRequiredRootUnion,
NO_STRICT,
normalizeSchemaForMoonshot,
sanitizeSchemaForOpenAIResponses,
@@ -1285,12 +1286,11 @@ export function buildParams(
filterReasoningHistory: options?.filterReasoningHistory,
omitReasoningEffort: options?.omitReasoningEffort,
});
const reasoningSummary =
model.provider === "xai-oauth"
? options?.reasoning === undefined
const reasoningSummary = model.compat.supportsReasoningSummary
? options?.reasoningSummary
: options?.reasoning === undefined
? undefined
: null
: options?.reasoningSummary;
: null;
applyResponsesCompatPolicy(params, reasoningPolicy, {
reasoningSummary,
forceReasoningOff: options?.forceReasoningOff,
@@ -1388,6 +1388,7 @@ export function convertTools(
),
): OpenAITool[] {
const allowFreeform = supportsFreeformApplyPatch(model);
const rejectXaiRootObjectUnion = model.provider === "xai" || model.provider === "xai-oauth";
const out: OpenAITool[] = [];
for (const tool of tools) {
if (tool.native?.type === "computer" && model.supportsComputerUse === true) {
@@ -1420,16 +1421,18 @@ export function convertTools(
// subschemas ("property schema … must be an object"), so the Moonshot
// pass re-coerces them last.
const sanitized = sanitizeSchemaForOpenAIResponses(baseParameters);
const providerParameters = rejectXaiRootObjectUnion ? flattenExclusiveRequiredRootUnion(sanitized) : sanitized;
const responseParameters =
model.compat.toolSchemaFlavor === "moonshot-mfjs"
? (normalizeSchemaForMoonshot(sanitized) as Record<string, unknown>)
: sanitized;
? (normalizeSchemaForMoonshot(providerParameters) as Record<string, unknown>)
: providerParameters;
const { schema: parameters, strict: effectiveStrict } = adaptSchemaForStrict(responseParameters, strict);
// Quarantine a tool whose emitted schema carries a provider-rejecting
// enum/const-vs-type contradiction: dropping just that tool keeps the rest
// of the request valid instead of letting one bad MCP schema 400 the whole
// turn (#2652). Other tools and built-ins are unaffected.
const violation = findStrictToolSchemaViolation(parameters);
// turn (#2652). Other tools and built-ins are unaffected. Leftover
// object-root unions are an xAI-only 400; OpenAI/Azure/Codex keep them.
const violation = findStrictToolSchemaViolation(parameters, "#", { rejectXaiRootObjectUnion });
if (violation) {
onQuarantine(tool.name, violation);
continue;
+3 -1
View File
@@ -3308,7 +3308,7 @@ export function applyCommonResponsesSamplingParams<P extends CommonResponsesPara
params: P,
options: CommonSamplingOptions | undefined,
model: Pick<Model, "provider" | "api" | "id" | "omitMaxOutputTokens" | "maxTokens"> & {
compat: Pick<ResolvedOpenAISharedCompat, "supportsSamplingParams">;
compat: Pick<ResolvedOpenAISharedCompat, "supportsSamplingParams" | "supportsPenaltyAndStopParams">;
},
): void {
if (options?.maxTokens && !model.omitMaxOutputTokens) {
@@ -3325,9 +3325,11 @@ export function applyCommonResponsesSamplingParams<P extends CommonResponsesPara
if (options?.topP !== undefined) params.top_p = options.topP;
if (options?.topK !== undefined) params.top_k = options.topK;
if (options?.minP !== undefined) params.min_p = options.minP;
if (model.compat.supportsPenaltyAndStopParams) {
if (options?.presencePenalty !== undefined) params.presence_penalty = options.presencePenalty;
if (options?.repetitionPenalty !== undefined) params.repetition_penalty = options.repetitionPenalty;
}
}
applyOpenAIServiceTier(params, options?.serviceTier, model);
}
@@ -4,7 +4,7 @@ import type { OAuthController, OAuthCredentials, OAuthLoginCallbacks } from "./o
import type { ProviderDefinition } from "./types";
const DEFAULT_AUTH_URL = "https://modelstudio.console.alibabacloud.com/";
const CHINA_AUTH_URL = "https://dashscope.console.aliyun.com/";
const CHINA_AUTH_URL = "https://bailian.console.aliyun.com/?tab=model#/api-key";
const DEFAULT_API_BASE_URL = "https://coding-intl.dashscope.aliyuncs.com/v1";
const CHINA_API_BASE_URL = "https://coding.dashscope.aliyuncs.com/v1";
const VALIDATION_MODEL = "qwen3.5-plus";
@@ -32,7 +32,7 @@ export async function loginAlibabaCodingPlan(options: OAuthController): Promise<
if (choice === "2") {
baseUrl = CHINA_API_BASE_URL;
authUrl = CHINA_AUTH_URL;
instructions = "Copy your API key from the Alibaba Cloud DashScope console (China mainland)";
instructions = "Copy your API key from the Alibaba Cloud Bailian console (China mainland)";
} else if (choice === "3") {
const customUrl = await options.onPrompt({
message: "Enter custom base URL",
+20
View File
@@ -10,6 +10,7 @@ import { scheduler } from "node:timers/promises";
import { $env, getAgentDir } from "@oh-my-pi/pi-utils";
import packageJson from "../../../package.json" with { type: "json" };
import * as AIError from "../../error";
import { isRecord } from "../../utils";
import type { OAuthController, OAuthCredentials } from "./types";
const CLIENT_ID = "17e5f671-d194-4dfb-9706-5516cb48c098";
@@ -172,10 +173,29 @@ function parseTokenPayload(payload: TokenResponse, refreshTokenFallback?: string
});
}
let accountId: string | undefined;
const tokenParts = payload.access_token.split(".");
const jwtPayloadPart = tokenParts.length === 3 ? tokenParts[1] : undefined;
if (jwtPayloadPart) {
try {
const jwtPayload: unknown = JSON.parse(
new TextDecoder("utf-8").decode(Uint8Array.fromBase64(jwtPayloadPart, { alphabet: "base64url" })),
);
if (isRecord(jwtPayload)) {
const userId = typeof jwtPayload.user_id === "string" ? jwtPayload.user_id.trim() : "";
const subject = typeof jwtPayload.sub === "string" ? jwtPayload.sub.trim() : "";
accountId = userId || subject || undefined;
}
} catch {
// Opaque access tokens remain valid credentials without account metadata.
}
}
return {
access: payload.access_token,
refresh,
expires: Date.now() + payload.expires_in * 1000 - OAUTH_EXPIRY_SKEW_MS,
accountId,
};
}
+46 -44
View File
@@ -1022,45 +1022,47 @@ function streamDispatch<TApi extends Api>(
}
}
/** Thinking-loop re-samples spent before {@link resolveWithThinkingLoopCook} cooks. */
const THINKING_LOOP_MAX_ABORTS = 3;
/** Maximum guarded attempts for a detected thinking loop. */
const THINKING_LOOP_MAX_ATTEMPTS = 3;
const THINKING_LOOP_RETRY_BASE_DELAY_MS = 500;
const THINKING_LOOP_RETRY_MAX_DELAY_MS = 8_000;
function isRetryableThinkingLoop(message: AssistantMessage): boolean {
return (
message.stopReason === "error" &&
message.content.length === 0 &&
AIError.is(message.errorId, AIError.Flag.ThinkingLoop)
);
}
/**
* Resolve a completion, re-sampling a thinking-loop stall up to
* {@link THINKING_LOOP_MAX_ABORTS} times before letting it cook. The loop guard
* raises an empty `stopReason: "error"` stall on each guarded attempt; this
* result-path consumer re-dispatches a fresh request per stall and, once the abort
* budget is spent, runs one final pass with the guard disabled so a stubborn loop
* returns the model's raw output instead of a fatal stall. Non-stall results —
* including genuine errors — return immediately; a caller abort during backoff
* propagates so cancellation surfaces as an abort, never a stale stall result.
* Resolve a completion, re-sampling a thinking-loop stall for at most
* {@link THINKING_LOOP_MAX_ATTEMPTS} guarded attempts. The loop guard raises an
* empty `stopReason: "error"` stall; after the budget is spent that error is
* returned unchanged. Detection is never disabled as a fallback, because an
* unguarded retry can consume the remaining output budget and persist runaway
* content. Non-stall results, including genuine errors, return immediately. A
* caller abort during backoff propagates so cancellation surfaces as an abort,
* never a stale stall result.
*/
async function resolveWithThinkingLoopCook(
async function resolveWithThinkingLoopRetries(
signal: AbortSignal | undefined,
dispatch: () => AssistantMessageEventStream,
cook: () => AssistantMessageEventStream,
): Promise<AssistantMessage> {
let message = await dispatch().result();
let thinkingLoopRetry = AIError.is(message.errorId, AIError.Flag.ThinkingLoop);
for (let attempt = 0; thinkingLoopRetry && attempt < THINKING_LOOP_MAX_ABORTS - 1; attempt += 1) {
let thinkingLoopRetry = isRetryableThinkingLoop(message);
for (let attempt = 1; thinkingLoopRetry && attempt < THINKING_LOOP_MAX_ATTEMPTS; attempt += 1) {
// A caller abort surfaces as a thrown abort (never the stall, which would
// misclassify as a 502): throwIfAborted before backoff, and scheduler.wait
// rejects if the abort lands mid-delay.
signal?.throwIfAborted();
const delay = Math.min(THINKING_LOOP_RETRY_BASE_DELAY_MS * 2 ** attempt, THINKING_LOOP_RETRY_MAX_DELAY_MS);
const delay = Math.min(THINKING_LOOP_RETRY_BASE_DELAY_MS * 2 ** (attempt - 1), THINKING_LOOP_RETRY_MAX_DELAY_MS);
await scheduler.wait(delay, { signal });
message = await dispatch().result();
thinkingLoopRetry =
message.stopReason === "error" &&
message.content.length === 0 &&
AIError.is(message.errorId, AIError.Flag.ThinkingLoop);
thinkingLoopRetry = isRetryableThinkingLoop(message);
}
if (!thinkingLoopRetry) return message;
signal?.throwIfAborted();
// Abort budget spent and still looping: let it cook with the guard disabled.
return cook().result();
if (thinkingLoopRetry) signal?.throwIfAborted();
return message;
}
export async function complete<TApi extends Api>(
@@ -1068,11 +1070,7 @@ export async function complete<TApi extends Api>(
context: Context,
options?: OptionsForApi<TApi>,
): Promise<AssistantMessage> {
return resolveWithThinkingLoopCook(
options?.signal,
() => stream(model, context, options),
() => stream(model, context, { ...options, loopGuard: { ...options?.loopGuard, enabled: false } }),
);
return resolveWithThinkingLoopRetries(options?.signal, () => stream(model, context, options));
}
type AuthRetryFailure = {
@@ -1578,23 +1576,27 @@ function streamSimpleRequest<TApi extends Api>(
// GitLab Duo - wraps Anthropic/OpenAI behind GitLab AI Gateway direct access tokens
if (isGitLabDuoModel(model)) {
return withProviderInFlightLimit(model, requestOptions, () =>
return withThinkingLoopGuard(model, requestOptions, opts =>
withProviderInFlightLimit(model, opts, () =>
streamGitLabDuo(model, context, {
...requestOptions,
...opts,
apiKey,
}),
),
);
}
// GitLab Duo Workflow - IDE workflow protocol + WebSocket action bridge
if (model.api === "gitlab-duo-agent") {
// Does not route through withProviderInFlightLimit, so heal explicitly.
return healLeakedThinking(
return withThinkingLoopGuard(model, requestOptions, opts =>
healLeakedThinking(
model,
streamGitLabDuoWorkflow(model as Model<"gitlab-duo-agent">, context, {
...requestOptions,
...opts,
apiKey,
}),
),
);
}
@@ -1606,24 +1608,28 @@ function streamSimpleRequest<TApi extends Api>(
// thinking, so clamp disabled requests to the lowest supported effort
// (mirrors the mapOptionsForApi path every other provider takes).
const kimiOptions = normalizeMandatoryReasoningOptions(model, requestOptions);
return withProviderInFlightLimit(model, kimiOptions, () =>
return withThinkingLoopGuard(model, kimiOptions, opts =>
withProviderInFlightLimit(model, opts, () =>
streamKimi(model as Model<"openai-completions">, context, {
...kimiOptions,
...opts,
apiKey,
format: kimiOptions?.kimiApiFormat,
format: opts?.kimiApiFormat,
}),
),
);
}
// Synthetic - route to dedicated handler that wraps OpenAI or Anthropic API
if (isSyntheticModel(model)) {
// Pass raw SimpleStreamOptions - streamSynthetic handles mapping internally
return withProviderInFlightLimit(model, requestOptions, () =>
// Pass raw SimpleStreamOptions - streamSynthetic handles mapping internally.
return withThinkingLoopGuard(model, requestOptions, opts =>
withProviderInFlightLimit(model, opts, () =>
streamSynthetic(model as Model<"openai-completions">, context, {
...requestOptions,
...opts,
apiKey,
format: requestOptions?.syntheticApiFormat ?? "openai", // Default to OpenAI format
format: opts?.syntheticApiFormat ?? "openai",
}),
),
);
}
const providerOptions = mapOptionsForApi(model, requestOptions, apiKey);
@@ -1635,11 +1641,7 @@ export async function completeSimple<TApi extends Api>(
context: Context,
options?: SimpleStreamOptions,
): Promise<AssistantMessage> {
return resolveWithThinkingLoopCook(
options?.signal,
() => streamSimple(model, context, options),
() => streamSimple(model, context, { ...options, loopGuard: { ...options?.loopGuard, enabled: false } }),
);
return resolveWithThinkingLoopRetries(options?.signal, () => streamSimple(model, context, options));
}
const MIN_OUTPUT_TOKENS = 1024;
+5 -4
View File
@@ -716,8 +716,9 @@ export interface AnthropicFallbackContent {
}
/**
* Verbatim Anthropic web-search call/result retained for same-provider
* history replay. Other providers discard it in `transformMessages`.
* Verbatim Anthropic web-search or tool-search call/result retained for
* same-provider history replay. Other providers discard it in
* `transformMessages`.
*/
export interface AnthropicServerToolContent {
type: "anthropicServerTool";
@@ -725,12 +726,12 @@ export interface AnthropicServerToolContent {
| {
type: "server_tool_use";
id: string;
name: "web_search";
name: "web_search" | "tool_search_tool_regex" | "tool_search_tool_bm25";
input?: Record<string, unknown> | null;
[key: string]: unknown;
}
| {
type: "web_search_tool_result";
type: "web_search_tool_result" | "tool_search_tool_result";
tool_use_id: string;
content: unknown;
[key: string]: unknown;
+15
View File
@@ -2,6 +2,7 @@ import { toNumber } from "@oh-my-pi/pi-catalog/utils";
import { $env } from "@oh-my-pi/pi-utils";
import { getKimiCommonHeaders } from "../registry/oauth/kimi";
import type {
CredentialRankingStrategy,
UsageAmount,
UsageFetchContext,
UsageFetchParams,
@@ -285,6 +286,7 @@ export const kimiUsageProvider: UsageProvider = {
fetchedAt: nowMs,
limits,
metadata: {
accountId: credential.accountId,
endpoint: url,
},
raw: parsed.raw,
@@ -293,3 +295,16 @@ export const kimiUsageProvider: UsageProvider = {
return report;
},
};
/** Ranks Kimi OAuth accounts by the canonical 5-hour and 7-day quota windows. */
export const kimiRankingStrategy: CredentialRankingStrategy = {
findWindowLimits: report => ({
primary: report.limits.find(limit => limit.window?.id === "5h"),
secondary: report.limits.find(limit => limit.window?.id === "7d"),
}),
scopeLimits: report => report.limits.filter(limit => limit.window?.id === "5h" || limit.window?.id === "7d"),
windowDefaults: {
primaryMs: 5 * HOUR_MS,
secondaryMs: 7 * DAY_MS,
},
};
+87 -14
View File
@@ -23,9 +23,14 @@ interface UmansUsagePayload {
requests?: { limit?: number; hard_cap?: number | null; window_seconds?: number };
concurrency?: { limit?: number; hard_cap?: number | null };
};
/** Rolling 5h window metadata; `resets_at` anchors the status-line countdown. */
window?: { started_at?: string; resets_at?: string; remaining_minutes?: number };
usage?: {
requests_in_window?: number;
remaining_requests?: number;
/** Model-weighted "effective requests" (umans-flash counts 0.5). */
weighted_in_window?: number;
weighted_remaining_requests?: number;
concurrent_sessions?: number;
tokens_in?: number;
tokens_out?: number;
@@ -55,6 +60,18 @@ function resolveStatus(usedFraction: number | undefined): UsageStatus | undefine
return "ok";
}
/**
* Soft-cap status never reaches `exhausted`: hitting the effective-request
* limit only means burst headroom is being consumed — Umans throttles (429)
* only near the burst ceiling, which the hard row tracks. `exhausted` must
* stay off this row or the usage-aware fallback demotes a healthy account.
*/
function softCapStatus(usedFraction: number | undefined): UsageStatus | undefined {
if (usedFraction === undefined) return undefined;
if (usedFraction >= 0.9) return "warning";
return "ok";
}
function buildAmount(args: {
used: number | undefined;
limit: number | undefined;
@@ -75,30 +92,88 @@ function buildAmount(args: {
};
}
function buildRequestsLimit(payload: UmansUsagePayload, provider: string): UsageLimit | null {
function buildRequestsLimits(payload: UmansUsagePayload, provider: string): UsageLimit[] {
const limit = toFiniteNumber(payload.limits?.requests?.limit);
const hardCap = toFiniteNumber(payload.limits?.requests?.hard_cap);
const windowSeconds = toFiniteNumber(payload.limits?.requests?.window_seconds);
const used = toFiniteNumber(payload.usage?.requests_in_window);
const remaining = toFiniteNumber(payload.usage?.remaining_requests);
if (limit === undefined && used === undefined) return null;
const amount = buildAmount({ used, limit, remaining, unit: "requests" });
// Rolling window: each request ages out `window_seconds` after it fired, so
// there is no single reset timestamp. Surface the window size + label only.
// `window.id` is `"5h"` to match the status-line usage segment's window-id
// contract (it only recognizes `"5h"`/`"7d"`); `label` stays human-readable.
const rawUsed = toFiniteNumber(payload.usage?.requests_in_window);
const rawRemaining = toFiniteNumber(payload.usage?.remaining_requests);
const weightedUsed = toFiniteNumber(payload.usage?.weighted_in_window);
const weightedRemaining = toFiniteNumber(payload.usage?.weighted_remaining_requests);
if (limit === undefined && rawUsed === undefined && weightedUsed === undefined) return [];
// The 5h window is rolling (FIFO: each request ages out five hours after it
// fired), but the payload still reports an absolute `resets_at` for the
// current window epoch — surface it as an incremental countdown (`tick`)
// rather than a hard reset. `window.id` is `"5h"` to match the status-line
// usage segment's window-id contract (it only recognizes `"5h"`/`"7d"`).
let resetsAt: number | undefined;
if (payload.window?.resets_at) {
const parsed = Date.parse(payload.window.resets_at);
resetsAt = Number.isNaN(parsed) ? undefined : parsed;
}
const window: UsageWindow = {
id: "5h",
label: "rolling 5h",
durationMs: windowSeconds ? windowSeconds * 1000 : 5 * HOUR_MS,
...(resetsAt !== undefined ? { resetsAt, resetLabel: "tick" } : {}),
};
return {
// Single row: either payloads without weighted counters (legacy) or payloads
// that report weighted usage but no burst ceiling (`hard_cap`). Without a
// burst ceiling there is no hard row to defer exhaustion to, so the
// authoritative counter — weighted when available, else raw — drives the
// single row and CAN exhaust at the limit. Raw burst traffic above the
// limit still never drives exhaustion on its own: weighted headroom stays
// decisive (https://github.com/can1357/oh-my-pi/issues/7858).
if (weightedUsed === undefined || hardCap === undefined) {
const amount = buildAmount({
used: weightedUsed ?? rawUsed,
limit,
remaining: weightedUsed !== undefined ? weightedRemaining : rawRemaining,
unit: "requests",
});
return [
{
id: "umans:requests",
label: "Requests (rolling 5h)",
scope: { provider, windowId: window.id, shared: true },
window,
amount,
status: resolveStatus(amount.usedFraction),
};
},
];
}
// Umans weights requests by model ("effective requests": umans-flash counts
// 0.5), so the weighted counters are the authoritative utilization against
// the soft `limit`; the raw counters include burst/superseded traffic and
// read as exhausted mid-window while the account still has weighted
// headroom (https://github.com/can1357/oh-my-pi/issues/7858). Soft cap hits
// warn; only the burst ceiling (`hard_cap`, raw counts) can exhaust.
const softAmount = buildAmount({ used: weightedUsed, limit, remaining: weightedRemaining, unit: "requests" });
const limits: UsageLimit[] = [
{
id: "umans:requests:soft",
label: "Requests (soft cap)",
scope: { provider, windowId: window.id, shared: true },
window,
amount: softAmount,
status: softCapStatus(softAmount.usedFraction),
},
];
if (hardCap !== undefined && rawUsed !== undefined) {
const hardAmount = buildAmount({ used: rawUsed, limit: hardCap, remaining: undefined, unit: "requests" });
limits.push({
id: "umans:requests:hard",
label: "Requests (burst ceiling)",
scope: { provider, windowId: window.id, shared: true },
window,
amount: hardAmount,
status: resolveStatus(hardAmount.usedFraction),
});
}
return limits;
}
function buildConcurrencyLimit(payload: UmansUsagePayload, provider: string): UsageLimit | null {
@@ -157,9 +232,7 @@ async function fetchUmansUsage(params: UsageFetchParams, ctx: UsageFetchContext)
return null;
}
const limits: UsageLimit[] = [];
const requests = buildRequestsLimit(payload, params.provider);
if (requests) limits.push(requests);
const limits: UsageLimit[] = [...buildRequestsLimits(payload, params.provider)];
const concurrency = buildConcurrencyLimit(payload, params.provider);
if (concurrency) limits.push(concurrency);
if (limits.length === 0) return null;
@@ -25,7 +25,7 @@
* events are forwarded verbatim.
*/
import { isAnthropicWebSearchHistoryBlock } from "../providers/anthropic-wire";
import { isAnthropicServerToolHistoryBlock } from "../providers/anthropic-wire";
import type {
AnthropicServerToolContent,
AssistantMessage,
@@ -430,7 +430,7 @@ class LeakedThinkingProjector {
const pairedIndexes = new Set<number>();
for (let srcIndex = 0; srcIndex < message.content.length; srcIndex++) {
const content = message.content[srcIndex];
if (content?.type !== "anthropicServerTool" || !isAnthropicWebSearchHistoryBlock(content.block)) continue;
if (content?.type !== "anthropicServerTool" || !isAnthropicServerToolHistoryBlock(content.block)) continue;
if (content.block.type === "server_tool_use") {
pendingCalls.set(content.block.id, srcIndex);
continue;
@@ -7,8 +7,18 @@
* copied onto a `type: "null"` branch, or an `enum` placed on an `array`
* schema instead of its `items`). One such tool 400s the entire turn, so
* callers quarantine just the offending tool. See issue #2652.
*
* xAI additionally rejects a leftover *root* `anyOf`/`oneOf` whose branches
* are not objects ("tool parameter root must be an object type"). That class
* is opt-in via {@link FindStrictToolSchemaViolationOptions.rejectXaiRootObjectUnion}
* so OpenAI/Azure/Codex keep valid object-root unions.
*/
export interface FindStrictToolSchemaViolationOptions {
/** xAI (paid + OAuth) only: leftover object-root unions 400 the whole turn. */
rejectXaiRootObjectUnion?: boolean;
}
type JsonRecord = Record<string, unknown>;
const SCHEMA_TYPE_NAMES: Record<string, true> = {
@@ -70,10 +80,14 @@ const CHILD_ARRAY_KEYS = ["anyOf", "oneOf", "allOf", "prefixItems"] as const;
* contradictions. Returns a JSON-pointer-ish path to the first offending node,
* or `null` when the schema is safe to emit.
*/
export function findStrictToolSchemaViolation(schema: unknown, path = "#"): string | null {
export function findStrictToolSchemaViolation(
schema: unknown,
path = "#",
options?: FindStrictToolSchemaViolationOptions,
): string | null {
if (Array.isArray(schema)) {
for (let i = 0; i < schema.length; i++) {
const hit = findStrictToolSchemaViolation(schema[i], `${path}/${i}`);
const hit = findStrictToolSchemaViolation(schema[i], `${path}/${i}`, options);
if (hit) return hit;
}
return null;
@@ -91,25 +105,45 @@ export function findStrictToolSchemaViolation(schema: unknown, path = "#"): stri
}
}
// xAI rejects the whole request when the *root* schema is typed as object
// (or has properties) AND still carries an anyOf/oneOf with a typeless or
// non-object branch. Nested unions and pure root unions are not this error.
if (
options?.rejectXaiRootObjectUnion &&
path === "#" &&
(types.includes("object") || (node.properties !== undefined && typeof node.properties === "object"))
) {
for (const key of ["anyOf", "oneOf"] as const) {
const arr = node[key];
if (!Array.isArray(arr) || arr.length === 0) continue;
const hasNonObjectBranch = arr.some(branch => {
if (typeof branch !== "object" || branch === null || Array.isArray(branch)) return true;
const branchTypes = declaredTypes(branch as JsonRecord);
return !branchTypes.includes("object");
});
if (hasNonObjectBranch) return `${path}/${key}`;
}
}
for (const key of CHILD_MAP_KEYS) {
const sub = node[key];
if (sub && typeof sub === "object" && !Array.isArray(sub)) {
for (const k of Object.keys(sub as JsonRecord)) {
const hit = findStrictToolSchemaViolation((sub as JsonRecord)[k], `${path}/${key}/${k}`);
const hit = findStrictToolSchemaViolation((sub as JsonRecord)[k], `${path}/${key}/${k}`, options);
if (hit) return hit;
}
}
}
for (const key of CHILD_SCHEMA_KEYS) {
if (key in node) {
const hit = findStrictToolSchemaViolation(node[key], `${path}/${key}`);
const hit = findStrictToolSchemaViolation(node[key], `${path}/${key}`, options);
if (hit) return hit;
}
}
for (const key of CHILD_ARRAY_KEYS) {
const arr = node[key];
if (Array.isArray(arr)) {
const hit = findStrictToolSchemaViolation(arr, `${path}/${key}`);
const hit = findStrictToolSchemaViolation(arr, `${path}/${key}`, options);
if (hit) return hit;
}
}
+31
View File
@@ -204,6 +204,37 @@ function rewriteNullableScalarAnyOf(schema: Record<string, unknown>): void {
schema.type = [scalarType, "null"];
}
function isExclusiveRequiredBranch(branch: unknown): boolean {
if (!isSchemaRecord(branch)) return false;
if (Object.hasOwn(branch, "type")) return false;
if (!Array.isArray(branch.required) || branch.required.length === 0) return false;
if (!branch.required.every(name => typeof name === "string" && name.length > 0)) return false;
for (const key in branch) {
if (!Object.hasOwn(branch, key)) continue;
if (key === "required" || key === "description" || key === "title") continue;
return false;
}
return true;
}
/**
* Return an xAI-compatible copy of an object-root schema whose union consists
* only of typeless required-key fragments. Other providers must retain the
* union because it is a real model-facing constraint.
*/
export function flattenExclusiveRequiredRootUnion(schema: Record<string, unknown>): Record<string, unknown> {
const unionKey = Array.isArray(schema.anyOf) ? "anyOf" : Array.isArray(schema.oneOf) ? "oneOf" : undefined;
if (!unionKey) return schema;
const union = schema[unionKey];
if (!Array.isArray(union) || union.length === 0) return schema;
const typedObject = schema.type === "object" || (Array.isArray(schema.type) && schema.type.includes("object"));
if (!typedObject && !isSchemaRecord(schema.properties)) return schema;
if (!union.every(isExclusiveRequiredBranch)) return schema;
const flattened = { ...schema };
delete flattened[unionKey];
return flattened;
}
/** Keys whose values are a single JSON Schema (not an array or map). */
const SCHEMA_VALUE_KEYS = [
"additionalProperties",
+95 -62
View File
@@ -8,15 +8,17 @@
* or answering. The runaway is *not* byte-identical, so a cheap verbatim
* tail-repeat check alone misses it.
*
* This guard watches the streamed `thinking` deltas and, on a match, terminates
* the stream with a synthetic `error` {@link AssistantMessage} that carries
* **no observable content**. An empty-content `stopReason: "error"` message tagged
* with `AIError.Flag.ThinkingLoop` lets result consumers and `AgentSession` discard
* the runaway and re-sample instead of committing garbage transcript.
* This guard watches streamed deltas and, on a match, terminates the stream with
* a synthetic `error` {@link AssistantMessage} whose terminal content is empty.
* Deltas emitted before enough evidence accumulates may already be observable to
* a live streaming consumer; the empty terminal prevents the failed attempt from
* being committed or replayed. Tagged with `AIError.Flag.ThinkingLoop`, the
* result lets `AgentSession` discard the runaway and re-sample.
*
* Three failure shapes are detected:
* 1. **Verbatim tail repetition** — a short unit repeated back-to-back (e.g.
* "🌊 🌊 🌊 …"). Caught from a rolling 250-char tail.
* Four failure shapes are detected:
* 1. **Exact suffix cycles** — a byte-identical unit repeated back-to-back,
* including long cycles such as the observed 311-character Kiro runaway.
* This bounded detector applies to every model.
* 2. **Near-duplicate segments** — paragraphs that normalize to the same
* word-trigram fingerprint. Caught with a Jaccard window over recent
* paragraphs. Thresholds were calibrated on a real loop transcript plus
@@ -28,13 +30,16 @@
* vocabulary and name nothing concrete. Caught by a run of low-novelty,
* anchor-free segments; a segment naming a path/identifier resets the run, so
* genuine but vocabulary-repetitive work (per-file templates) is spared.
* 4. **Gemini summary-header runaway** — handled separately by
* {@link GeminiHeaderRunDetector}.
*
* Scope is narrow: guarded Gemini, DeepSeek, and Grok family streams before any tool call. Native
* thinking is checked first; assistant text can also be checked for providers
* that surface reasoning as visible prose. On a hit the failed turn is emitted as
* an empty retryable stream-stall error; result-awaiting callers (`complete`,
* `completeSimple`) re-sample it a few times and then let a stubborn loop cook
* through one unguarded pass. Disable detection with `PI_NO_THINKING_LOOP_GUARD=1`.
* Scope: exact cycles are guarded for every model; semantic heuristics remain
* limited to Gemini, DeepSeek, and Grok family streams before any tool call.
* Native thinking is checked first; assistant text can also be checked for
* providers that surface reasoning as visible prose. On a hit the failed turn is
* emitted as an empty retryable stream-stall error; result-awaiting callers
* (`complete`, `completeSimple`) re-sample at most three guarded attempts and
* then fail closed. Disable detection with `PI_NO_THINKING_LOOP_GUARD=1`.
*/
import { modelFamilyToken } from "@oh-my-pi/pi-catalog/identity";
import { logger } from "@oh-my-pi/pi-utils";
@@ -47,12 +52,17 @@ import { AssistantMessageEventStream } from "./event-stream";
* classifiers treat it as a transient (retryable) stop without bespoke rules. */
export const THINKING_LOOP_ERROR_MARKER = "Thinking loop detected";
/** Rolling tail (chars) inspected for verbatim back-to-back repetition. */
const VERBATIM_TAIL_WINDOW = 250;
/** Minimum total repeated chars before a verbatim run counts as a loop. */
const VERBATIM_MIN_REPEATED_CHARS = 180;
/** Longest unit length probed for a verbatim repeat. */
const VERBATIM_MAX_UNIT = 60;
/** Rolling tail retained for exact suffix-cycle detection. */
const EXACT_TAIL_WINDOW = 4096;
/** Longest exact cycle length considered. */
const EXACT_MAX_UNIT = 1024;
/** New characters between scans. Large deltas are scanned immediately. */
const EXACT_CHECK_STRIDE = 128;
/** Short cycles need four repeats covering at least this many characters. */
const EXACT_SHORT_MAX_UNIT = 60;
const EXACT_SHORT_MIN_REPEATED_CHARS = 180;
/** Long cycles need at least three repeats covering at least this many chars. */
const EXACT_LONG_MIN_REPEATED_CHARS = 1024;
/** Char cap for an unterminated segment; forces a flush so a wall-of-text loop
* (no blank lines / headings) still segments. */
@@ -96,11 +106,12 @@ const CONCRETE_ANCHOR =
/`[^`]+`|\b\w{2,}\.[a-zA-Z]\w{0,4}\b|[\w-]+(?:\/[\w-]+){2,}|\b\w+_\w+\b|\b[a-z]+[A-Z]\w*\b|\b[A-Z][a-z]+[A-Z]\w*\b/g;
/**
* True when `model.id` belongs to a family guarded for thinking/response loops:
* Gemini, DeepSeek, or Grok.
* True when `model.id` belongs to a family guarded by the semantic loop
* heuristics: Gemini, DeepSeek, or Grok. Exact suffix-cycle detection applies to
* every enabled model independently of this predicate.
*
* Model identity is derived only from its id; provider and compatibility metadata
* do not opt opaque aliases into the guard.
* do not opt opaque aliases into semantic detection.
*/
export function isLoopGuardedModel(model: Model<Api>, options?: StreamOptions): boolean {
if (options?.loopGuard?.enabled === false) return false;
@@ -120,8 +131,10 @@ export function isLoopGuardedModel(model: Model<Api>, options?: StreamOptions):
* is responsible for stopping after the first hit.
*/
export class ThinkingLoopDetector {
/** Rolling char tail for verbatim repeat detection. */
/** Rolling char tail for exact suffix-cycle detection. */
#tail = "";
/** Total characters received when the exact detector last scanned. */
#exactScannedAt = 0;
/** Pending thinking text not yet split into completed segments. */
#pending = "";
/** Fingerprints of the most recent substantial segments (≤ SEGMENT_WINDOW). */
@@ -138,17 +151,26 @@ export class ThinkingLoopDetector {
* path/identifier every paragraph is still caught. */
#anchorWindow: Set<string>[] = [];
constructor(private readonly semanticHeuristics = true) {}
push(delta: string): string | null {
if (!delta) return null;
// 1. Verbatim back-to-back repetition over the rolling tail.
// 1. Exact suffix cycles. Scan at a bounded cadence rather than doing
// quadratic work for every token-sized delta.
this.#tail += delta;
if (this.#tail.length > VERBATIM_TAIL_WINDOW) this.#tail = this.#tail.slice(-VERBATIM_TAIL_WINDOW);
const verbatim = detectVerbatimRepetition(this.#tail);
if (verbatim) {
const [unit, times] = verbatim;
return `repeated "${unit.trim()}" ${times}× back-to-back`;
if (this.#tail.length > EXACT_TAIL_WINDOW) this.#tail = this.#tail.slice(-EXACT_TAIL_WINDOW);
this.#exactScannedAt += delta.length;
if (this.#exactScannedAt >= EXACT_CHECK_STRIDE || delta.length >= EXACT_CHECK_STRIDE) {
this.#exactScannedAt = 0;
const exact = detectExactSuffixCycle(this.#tail);
if (exact) {
const [unit, times] = exact;
return `repeated an exact ${unit.length}-character cycle ${times}× back-to-back`;
}
}
if (!this.semanticHeuristics) return null;
// 2. Near-duplicate paragraph loop. Append, then drain completed segments.
this.#pending += delta;
@@ -179,7 +201,14 @@ export class ThinkingLoopDetector {
* terminator). Called when the thinking block ends so the final segment —
* which may be the one that completes a duplicate cluster — is not dropped. */
flush(): string | null {
if (!this.#pending) return null;
// A stream can end before the next cadence boundary. Force one final exact
// check even when semantic heuristics are disabled and #pending is empty.
const exact = detectExactSuffixCycle(this.#tail);
if (exact) {
const [unit, times] = exact;
return `repeated an exact ${unit.length}-character cycle ${times}× back-to-back`;
}
if (!this.semanticHeuristics || !this.#pending) return null;
let rest = this.#pending;
this.#pending = "";
while (rest.length > 0) {
@@ -349,8 +378,9 @@ export function guardThinkingLoopStream(
options?: StreamOptions,
): AssistantMessageEventStream {
const outer = new AssistantMessageEventStream();
const thinkingDetector = new ThinkingLoopDetector();
const textDetector = new ThinkingLoopDetector();
const semanticHeuristics = isLoopGuardedModel(model, options);
const thinkingDetector = new ThinkingLoopDetector(semanticHeuristics);
const textDetector = new ThinkingLoopDetector(semanticHeuristics);
const checkAssistantContent = options?.loopGuard?.checkAssistantContent !== false;
void (async () => {
@@ -415,12 +445,12 @@ export function guardThinkingLoopStream(
}
/**
* Apply the loop guard around a provider dispatch. For non-guarded models
* (or when disabled) this is a transparent pass-through. For guarded models it injects a
* guard abort signal into the provider call so a detected loop tears down the
* upstream, then wraps the returned stream. The guard only raises the retryable
* stall; bounding the re-samples and the final cook pass lives in the
* result-awaiting caller.
* Apply the loop guard around a provider dispatch. Unless explicitly disabled,
* every model gets exact suffix-cycle detection; Gemini, DeepSeek, and Grok also
* get the semantic heuristics selected by {@link isLoopGuardedModel}. The guard
* injects an abort signal into the provider call so a detected loop tears down
* the upstream, then wraps the returned stream. Bounding result-path re-samples
* lives in the result-awaiting caller.
*/
export function withThinkingLoopGuard<
O extends { signal?: AbortSignal; loopGuard?: { enabled?: boolean; checkAssistantContent?: boolean } },
@@ -429,7 +459,7 @@ export function withThinkingLoopGuard<
options: O | undefined,
dispatch: (options: O | undefined) => AssistantMessageEventStream,
): AssistantMessageEventStream {
if (process.env.PI_NO_THINKING_LOOP_GUARD === "1" || !isLoopGuardedModel(model, options)) {
if (process.env.PI_NO_THINKING_LOOP_GUARD === "1" || options?.loopGuard?.enabled === false) {
return dispatch(options);
}
const controller = new AbortController();
@@ -467,31 +497,34 @@ function buildThinkingLoopError(model: Model<Api>, detail: string): AssistantMes
}
/**
* Detect a short unit repeated back-to-back at the tail (verbatim loop). Only a
* unit carrying a letter or pictographic emoji counts — runs of digits,
* whitespace or punctuation are legitimate in tabular / hex / numeric output.
* Detect an exact cycle at the text suffix. A Z-array over the reversed tail
* finds every possible suffix period in linear time without substring churn.
* Short cycles retain the original 180-character/four-repeat sensitivity; long
* cycles require at least three repeats and 1024 repeated characters.
*/
function detectVerbatimRepetition(text: string): [unit: string, count: number] | null {
if (text.length < VERBATIM_MIN_REPEATED_CHARS) return null;
const windowSize = Math.min(text.length, VERBATIM_TAIL_WINDOW);
const searchSpace = text.slice(-windowSize);
for (let len = 2; len <= VERBATIM_MAX_UNIT; len++) {
if (searchSpace.length < len * 4) continue;
const unit = searchSpace.slice(-len);
if (!/[\p{L}\p{Extended_Pictographic}]/u.test(unit)) continue;
let count = 0;
let pos = searchSpace.length;
while (pos >= len) {
if (searchSpace.slice(pos - len, pos) === unit) {
count++;
pos -= len;
} else {
break;
function detectExactSuffixCycle(text: string): [unit: string, count: number] | null {
if (text.length < EXACT_SHORT_MIN_REPEATED_CHARS) return null;
const reversed = text.split("").reverse().join("");
const z = new Uint16Array(reversed.length);
let left = 0;
let right = 0;
for (let i = 1; i < reversed.length; i++) {
if (i <= right) z[i] = Math.min(right - i + 1, z[i - left]);
while (i + z[i] < reversed.length && reversed[z[i]] === reversed[i + z[i]]) z[i]++;
if (i + z[i] - 1 > right) {
left = i;
right = i + z[i] - 1;
}
}
if (count >= 4 && len * count >= VERBATIM_MIN_REPEATED_CHARS) return [unit, count];
const maxUnit = Math.min(EXACT_MAX_UNIT, Math.floor(reversed.length / 3));
for (let len = 2; len <= maxUnit; len++) {
const count = 1 + Math.floor(z[len] / len);
const minCount = len <= EXACT_SHORT_MAX_UNIT ? 4 : 3;
const minChars = len <= EXACT_SHORT_MAX_UNIT ? EXACT_SHORT_MIN_REPEATED_CHARS : EXACT_LONG_MIN_REPEATED_CHARS;
if (count < minCount || len * count < minChars) continue;
const unit = text.slice(-len);
if (/\p{L}|\p{Extended_Pictographic}/u.test(unit)) return [unit, count];
}
return null;
}
@@ -69,8 +69,10 @@ describe("alibaba-coding-plan endpoint selection", () => {
expect(result.access).toBe("sk-cn-key");
expect(result.refresh).toBe("sk-cn-key");
expect(result.enterpriseUrl).toBe("https://coding.dashscope.aliyuncs.com/v1");
expect(capturedAuth?.url).toBe("https://dashscope.console.aliyun.com/");
expect(capturedAuth?.instructions).toContain("China mainland");
expect(capturedAuth?.url).toBe("https://bailian.console.aliyun.com/?tab=model#/api-key");
expect(capturedAuth?.instructions).toBe(
"Copy your API key from the Alibaba Cloud Bailian console (China mainland)",
);
expect(validateSpy).toHaveBeenCalledWith({
provider: "Alibaba Coding Plan",
@@ -6,7 +6,10 @@ import {
type AnthropicMessagesClientLike,
type AnthropicRequestOptions,
} from "@oh-my-pi/pi-ai/providers/anthropic-client";
import type { WebSearchToolResultBlockParam } from "@oh-my-pi/pi-ai/providers/anthropic-wire";
import type {
ToolSearchToolResultBlockParam,
WebSearchToolResultBlockParam,
} from "@oh-my-pi/pi-ai/providers/anthropic-wire";
import type { AssistantMessageEvent, Context, Model, ModelSpec, ProviderSessionState } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { structuredCloneJSON } from "@oh-my-pi/pi-utils";
@@ -676,6 +679,117 @@ describe("anthropic stream envelope handling", () => {
]);
});
it("replays tool-search server blocks between signed thinking and a client tool call", async () => {
const toolSearchResult: ToolSearchToolResultBlockParam = {
type: "tool_search_tool_result",
tool_use_id: "srvtoolu_search",
content: {
type: "tool_search_tool_search_result",
tool_references: [{ type: "tool_reference", tool_name: "_read" }],
},
};
vi.spyOn(AnthropicMessages.prototype, "create").mockImplementation(
() =>
createMockRequest([
{
type: "message_start",
message: {
id: "msg_tool_search",
usage: {
input_tokens: 12,
output_tokens: 0,
cache_read_input_tokens: 0,
cache_creation_input_tokens: 0,
},
},
},
{ type: "content_block_start", index: 0, content_block: { type: "thinking", thinking: "" } },
{ type: "content_block_delta", index: 0, delta: { type: "thinking_delta", thinking: "Find read." } },
{ type: "content_block_delta", index: 0, delta: { type: "signature_delta", signature: "sig-1" } },
{ type: "content_block_stop", index: 0 },
{
type: "content_block_start",
index: 1,
content_block: {
type: "server_tool_use",
id: "srvtoolu_search",
name: "tool_search_tool_regex",
},
},
{
type: "content_block_delta",
index: 1,
delta: { type: "input_json_delta", partial_json: '{"pattern":"read"}' },
},
{ type: "content_block_stop", index: 1 },
{ type: "content_block_start", index: 2, content_block: toolSearchResult },
{ type: "content_block_stop", index: 2 },
{ type: "content_block_start", index: 3, content_block: { type: "thinking", thinking: "" } },
{ type: "content_block_delta", index: 3, delta: { type: "thinking_delta", thinking: "Use read." } },
{ type: "content_block_delta", index: 3, delta: { type: "signature_delta", signature: "sig-2" } },
{ type: "content_block_stop", index: 3 },
{
type: "content_block_start",
index: 4,
content_block: { type: "tool_use", id: "tool_1", name: "_read", input: {} },
},
{
type: "content_block_delta",
index: 4,
delta: { type: "input_json_delta", partial_json: '{"path":"notes.txt"}' },
},
{ type: "content_block_stop", index: 4 },
{
type: "message_delta",
delta: { stop_reason: "tool_use" },
usage: {
input_tokens: 12,
output_tokens: 20,
cache_read_input_tokens: 0,
cache_creation_input_tokens: 0,
},
},
{ type: "message_stop" },
]) as never,
);
const stream = streamAnthropic(model, context, { apiKey: "sk-ant-test" });
for await (const _ of stream) {
// drain stream
}
const result = structuredCloneJSON(await stream.result());
const replay = convertAnthropicMessages(
[
context.messages[0],
result,
{
role: "toolResult",
toolCallId: "tool_1",
toolName: "_read",
content: [{ type: "text", text: "notes" }],
isError: false,
timestamp: 2,
},
],
model,
false,
);
const assistant = replay.find(message => message.role === "assistant");
expect(assistant?.content).toEqual([
{ type: "thinking", thinking: "Find read.", signature: "sig-1" },
{
type: "server_tool_use",
id: "srvtoolu_search",
name: "tool_search_tool_regex",
input: { pattern: "read" },
},
toolSearchResult,
{ type: "thinking", thinking: "Use read.", signature: "sig-2" },
{ type: "tool_use", id: "tool_1", name: "_read", input: { path: "notes.txt" } },
]);
});
it("does not persist a code-execution call without its unsupported result block", async () => {
vi.spyOn(AnthropicMessages.prototype, "create").mockImplementation(
() =>
@@ -1,6 +1,8 @@
import { describe, expect, it } from "bun:test";
import { encodeResponse, encodeStream, parseRequest } from "@oh-my-pi/pi-ai/providers/anthropic-messages-server";
import type {
ToolSearchServerToolUseBlockParam,
ToolSearchToolResultBlockParam,
WebSearchServerToolUseBlockParam,
WebSearchToolResultBlockParam,
} from "@oh-my-pi/pi-ai/providers/anthropic-wire";
@@ -344,6 +346,47 @@ describe("anthropic-messages parseRequest", () => {
]);
});
it("preserves inbound assistant tool-search call/result blocks verbatim", () => {
const serverToolUse: ToolSearchServerToolUseBlockParam = {
type: "server_tool_use",
id: "srvtoolu_search",
name: "tool_search_tool_regex",
input: { pattern: "read" },
};
const searchResult: ToolSearchToolResultBlockParam = {
type: "tool_search_tool_result",
tool_use_id: "srvtoolu_search",
content: {
type: "tool_search_tool_search_result",
tool_references: [{ type: "tool_reference", tool_name: "_read" }],
},
};
const parsed = parseRequest({
model: "claude-opus-4-7",
max_tokens: 8,
messages: [
{ role: "user", content: "read notes" },
{
role: "assistant",
content: [
{ type: "thinking", thinking: "find read", signature: "sig-1" },
serverToolUse,
searchResult,
{ type: "text", text: "tool loaded" },
],
},
],
});
const assistant = parsed.context.messages.find(message => message.role === "assistant");
expect(assistant?.content).toEqual([
{ type: "thinking", thinking: "find read", thinkingSignature: "sig-1" },
{ type: "anthropicServerTool", block: serverToolUse },
{ type: "anthropicServerTool", block: searchResult },
{ type: "text", text: "tool loaded" },
]);
});
it("flattens malformed web-search history blocks instead of preserving invalid replay state", () => {
const parsed = parseRequest({
model: "claude-opus-4-7",
@@ -26,17 +26,15 @@ afterEach(() => {
clearCustomApis();
});
describe("auth-gateway non-streaming thinking-loop cook", () => {
it("returns 200 with cooked output instead of a 502 when the model loops", async () => {
describe("auth-gateway non-streaming thinking-loop retries", () => {
it("returns an error after three guarded looping attempts", async () => {
registerMockApi();
const dir = await fs.mkdtemp(path.join(os.tmpdir(), "gw-thinking-loop-"));
const storage = await AuthStorage.create(path.join(dir, "auth.db"));
storage.setRuntimeApiKey("openrouter", "test-key");
const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" });
// Three guarded attempts stall on the thinking loop; the fourth (cook) pass
// runs with the guard disabled and returns the visible answer.
for (let i = 0; i < 4; i++) {
mock.push({ content: [{ type: "thinking", thinking: loopThinking() }, "Final answer after cooking."] });
mock.push({ content: [{ type: "thinking", thinking: loopThinking() }, "Unreachable cooked answer."] });
}
const waitSpy = spyOn(scheduler, "wait").mockResolvedValue(undefined);
const handle = startAuthGateway({
@@ -56,18 +54,12 @@ describe("auth-gateway non-streaming thinking-loop cook", () => {
stream: false,
}),
});
const body = (await res.json()) as {
error?: unknown;
choices?: Array<{ message?: { content?: string | null } }>;
};
const body = (await res.json()) as { error?: unknown };
expect(res.status).toBe(200);
expect(body.error).toBeUndefined();
expect(body.choices?.[0]?.message?.content).toContain("Final answer after cooking.");
// Three guarded stalls + one unguarded cook pass.
expect(mock.calls).toHaveLength(4);
expect(mock.calls[0]?.options?.loopGuard?.enabled).toBeUndefined();
expect(mock.calls[3]?.options?.loopGuard?.enabled).toBe(false);
expect(res.status).toBe(502);
expect(body.error).toBeDefined();
expect(mock.calls).toHaveLength(3);
expect(mock.calls.every(call => call.options?.loopGuard?.enabled !== false)).toBe(true);
} finally {
waitSpy.mockRestore();
await handle.close();
@@ -101,7 +93,7 @@ describe("auth-gateway non-streaming thinking-loop cook", () => {
}),
});
// A genuine error is never a loop stall, so the cook fallback must not mask it.
// A genuine error is never a loop stall, so loop retry handling must not mask it.
expect(res.status).toBe(502);
expect(mock.calls).toHaveLength(1);
expect(THINKING_LOOP_ERROR_MARKER.length).toBeGreaterThan(0);
@@ -0,0 +1,202 @@
import { afterEach, describe, expect, it } from "bun:test";
import * as http2 from "node:http2";
import { create, fromBinary, toBinary } from "@bufbuild/protobuf";
import { streamCursor } from "@oh-my-pi/pi-ai/providers/cursor";
import type { Context, Model } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import {
AgentClientMessageSchema,
AgentServerMessageSchema,
InteractionUpdateSchema,
TextDeltaUpdateSchema,
TurnEndedUpdateSchema,
} from "@oh-my-pi/pi-catalog/discovery/cursor-gen/agent_pb";
// #8345: a server-side per-conversation rejection (bare resource_exhausted,
// zero tokens) poisons the wire conversationId; the next attempt must rotate
// to a fresh id and succeed, instead of failing forever until /fork.
let server: http2.Http2Server | undefined;
const sessions = new Set<http2.Http2Session>();
function frameConnectMessage(data: Uint8Array, flags = 0): Buffer {
const frame = Buffer.alloc(5 + data.length);
frame[0] = flags;
frame.writeUInt32BE(data.length, 1);
frame.set(data, 5);
return frame;
}
function textDeltaFrame(text: string): Buffer {
const message = create(AgentServerMessageSchema, {
message: {
case: "interactionUpdate",
value: create(InteractionUpdateSchema, {
message: { case: "textDelta", value: create(TextDeltaUpdateSchema, { text }) },
}),
},
});
return frameConnectMessage(toBinary(AgentServerMessageSchema, message));
}
function turnEndedFrame(): Buffer {
const message = create(AgentServerMessageSchema, {
message: {
case: "interactionUpdate",
value: create(InteractionUpdateSchema, {
message: { case: "turnEnded", value: create(TurnEndedUpdateSchema, {}) },
}),
},
});
return frameConnectMessage(toBinary(AgentServerMessageSchema, message));
}
/** Decode the wire conversationId from the first client frame of a request. */
function decodeConversationId(chunk: Buffer): string | undefined {
const msg = fromBinary(AgentClientMessageSchema, chunk.subarray(5));
if (msg.message.case !== "runRequest") return undefined;
return msg.message.value.conversationId;
}
/** First request ends with a bare resource_exhausted; later ones turn normally. */
async function startServer(seenConversationIds: string[]): Promise<string> {
server = http2.createServer();
server.on("session", session => {
sessions.add(session);
session.on("close", () => sessions.delete(session));
});
let requestCount = 0;
server.on("stream", (stream: http2.ServerHttp2Stream) => {
stream.on("data", (chunk: Buffer) => {
const conversationId = decodeConversationId(chunk);
if (conversationId !== undefined) seenConversationIds.push(conversationId);
requestCount++;
if (requestCount === 1) {
stream.respond({ ":status": 200, "content-type": "application/connect+proto" }, { waitForTrailers: true });
stream.once("wantTrailers", () => {
stream.sendTrailers({ "grpc-status": "8", "grpc-message": "resource_exhausted" });
});
stream.end();
} else {
stream.respond({ ":status": 200, "content-type": "application/connect+proto" });
stream.write(textDeltaFrame("recovered"));
stream.write(turnEndedFrame());
stream.end();
}
});
});
const listening = Promise.withResolvers<void>();
server.once("error", listening.reject);
server.listen(0, "127.0.0.1", listening.resolve);
await listening.promise;
const address = server.address();
if (!address || typeof address === "string") throw new Error("expected the fixture server to bind a tcp port");
return `http://127.0.0.1:${address.port}`;
}
async function stopServer(): Promise<void> {
for (const session of sessions) session.destroy();
sessions.clear();
if (!server) return;
const closing = server;
server = undefined;
const closed = Promise.withResolvers<void>();
closing.close(error => (error ? closed.reject(error) : closed.resolve()));
await closed.promise;
}
function makeModel(baseUrl: string): Model<"cursor-agent"> {
return buildModel({
id: "cursor-rotation-fixture",
name: "Cursor rotation fixture",
api: "cursor-agent",
provider: "cursor",
baseUrl,
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 1,
maxTokens: 1,
});
}
const context: Context = { messages: [{ role: "user", content: "hello", timestamp: 1 }] };
/** Drain a stream and return its terminal event (done / error). */
async function runToEnd(baseUrl: string, sessionId: string): Promise<{ type: "done" | "error"; message?: string }> {
const stream = streamCursor(makeModel(baseUrl), context, { apiKey: "test-token", sessionId });
let terminal: { type: "done" | "error"; message?: string } = { type: "done" };
for await (const event of stream) {
if (event.type === "error") {
terminal = { type: "error", message: event.error.errorMessage };
}
}
await stream.result().catch(() => {});
return terminal;
}
afterEach(async () => {
await stopServer();
});
describe("Cursor conversationId rotation (issue #8345)", () => {
it("rotates the poisoned conversationId and recovers on the next attempt", async () => {
const seenConversationIds: string[] = [];
const baseUrl = await startServer(seenConversationIds);
const first = await runToEnd(baseUrl, "sess-poisoned");
expect(first.type).toBe("error");
expect(first.message).toMatch(/resource.?exhausted/i);
const second = await runToEnd(baseUrl, "sess-poisoned");
expect(second.type).toBe("done");
expect(seenConversationIds).toHaveLength(2);
expect(seenConversationIds[0]).toBe("sess-poisoned");
expect(seenConversationIds[1]).not.toBe(seenConversationIds[0]);
});
it("keeps the rotated id when the new conversation is also rejected", async () => {
const seenConversationIds: string[] = [];
// Fail every request: rotation must happen exactly once.
server = http2.createServer();
server.on("session", session => {
sessions.add(session);
session.on("close", () => sessions.delete(session));
});
server.on("stream", (stream: http2.ServerHttp2Stream) => {
stream.on("data", (chunk: Buffer) => {
const conversationId = decodeConversationId(chunk);
if (conversationId !== undefined) seenConversationIds.push(conversationId);
stream.respond({ ":status": 200, "content-type": "application/connect+proto" }, { waitForTrailers: true });
stream.once("wantTrailers", () => {
stream.sendTrailers({ "grpc-status": "8", "grpc-message": "resource_exhausted" });
});
stream.end();
});
});
const listening = Promise.withResolvers<void>();
server.once("error", listening.reject);
server.listen(0, "127.0.0.1", listening.resolve);
await listening.promise;
const address = server.address();
if (!address || typeof address === "string") throw new Error("expected the fixture server to bind a tcp port");
const baseUrl = `http://127.0.0.1:${address.port}`;
const r1 = await runToEnd(baseUrl, "sess-sticky");
const r2 = await runToEnd(baseUrl, "sess-sticky");
const r3 = await runToEnd(baseUrl, "sess-sticky");
console.log(
"[test] results:",
JSON.stringify([r1.type, r1.message, r2.type, r2.message, r3.type, r3.message]),
"seen:",
JSON.stringify(seenConversationIds),
);
expect(seenConversationIds).toHaveLength(3);
expect(seenConversationIds[0]).toBe("sess-sticky");
expect(seenConversationIds[1]).toBe(seenConversationIds[2]);
expect(seenConversationIds[1]).not.toBe("sess-sticky");
});
});
+18
View File
@@ -130,6 +130,24 @@ describe("error-id classification", () => {
expect(assistant.errorId).toBe(id);
});
it("classifies Cursor NGHTTP2 stream resets as transient", () => {
for (const errorMessage of [
"Stream closed with error code NGHTTP2_INTERNAL_ERROR",
"Stream closed with error code NGHTTP2_REFUSED_STREAM",
"Connect error failed_precondition: Error: Stream closed with error code NGHTTP2_REFUSED_STREAM",
]) {
const assistant = message({
api: "cursor-agent",
provider: "cursor",
model: "composer-2.5",
errorMessage,
});
const id = AIError.classifyMessage(assistant);
expect(AIError.is(id, AIError.Flag.Transient)).toBe(true);
expect(AIError.retriable(id)).toBe(true);
}
});
it("merges existing cause-chain kinds with finalized error text kinds", () => {
const assistant = message({
errorId: AIError.create(AIError.Flag.ThinkingLoop),
@@ -0,0 +1,101 @@
import { describe, expect, it } from "bun:test";
import { Effort, type FetchImpl } from "@oh-my-pi/pi-ai";
import { streamSimple } from "@oh-my-pi/pi-ai/stream";
import type { Context, Model } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import type { ModelSpec } from "@oh-my-pi/pi-catalog/types";
// GLM-5.3 replaces GLM-5.2's host-specific reasoning_effort dialects with a
// single uniform wire-exact low/high/max ladder on every host, and thinking can
// no longer be disabled (thinking.type must always be "enabled"). These tests
// pin both contracts so a future change cannot regress to the GLM-5.2 shape.
const context: Context = {
messages: [{ role: "user", content: "hello", timestamp: Date.now() }],
};
function glm53OnFireworks(): Model<"openai-completions"> {
return buildModel({
id: "glm-5.3",
name: "GLM-5.3",
api: "openai-completions",
provider: "fireworks",
baseUrl: "https://api.fireworks.ai/inference/v1",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 1_000_000,
maxTokens: 131_072,
} satisfies ModelSpec<"openai-completions">);
}
function glm53OnZaiAnthropic(): Model<"anthropic-messages"> {
return buildModel({
id: "glm-5.3",
name: "GLM-5.3",
api: "anthropic-messages",
provider: "zai",
baseUrl: "https://api.z.ai/api/anthropic",
reasoning: true,
input: ["text"],
cost: { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
contextWindow: 1_000_000,
maxTokens: 131_072,
} satisfies ModelSpec<"anthropic-messages">);
}
async function captureChatBody(
model: Model<"openai-completions">,
options: { reasoning?: Effort; disableReasoning?: boolean },
): Promise<{ reasoning_effort?: string; thinking?: { type?: string } }> {
let requestBody: string | undefined;
const fetchMock: FetchImpl = (_input, init) => {
requestBody = typeof init?.body === "string" ? init.body : undefined;
return Promise.resolve(
new Response(
'data: {"choices":[{"delta":{"content":"ok"}}]}\ndata: {"choices":[{"finish_reason":"stop"}]}\ndata: [DONE]\n',
{ status: 200, headers: { "content-type": "text/event-stream" } },
),
);
};
const stream = streamSimple(model, context, { apiKey: "k", fetch: fetchMock, ...options });
await stream.result();
if (!requestBody) throw new Error("request body was not captured");
return JSON.parse(requestBody);
}
describe("GLM-5.3 reasoning effort wire mapping", () => {
it("derives the uniform low/high/max ladder on a direct GLM host (not the GLM-5.2 host-specific shape)", () => {
const model = glm53OnFireworks();
expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.High, Effort.Max]);
expect(model.thinking?.requiresEffort).toBe(true);
expect(model.thinking?.defaultLevel).toBe(Effort.Max);
});
it("sends wire-exact low/high/max reasoning_effort on a direct GLM host", async () => {
const model = glm53OnFireworks();
expect((await captureChatBody(model, { reasoning: Effort.Low })).reasoning_effort).toBe("low");
expect((await captureChatBody(model, { reasoning: Effort.High })).reasoning_effort).toBe("high");
expect((await captureChatBody(model, { reasoning: Effort.Max })).reasoning_effort).toBe("max");
});
it("clamps thinking-off to the lowest effort instead of disabling (GLM-5.3 cannot disable thinking)", async () => {
const model = glm53OnFireworks();
const body = await captureChatBody(model, { disableReasoning: true });
expect(body.reasoning_effort).toBe("low");
expect(body.thinking).toBeUndefined();
});
it("clamps omitted reasoning to the lowest effort", async () => {
const model = glm53OnFireworks();
const body = await captureChatBody(model, {});
expect(body.reasoning_effort).toBe("low");
});
it("derives mandatory reasoning on the zai Anthropic endpoint too", () => {
const model = glm53OnZaiAnthropic();
expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.High, Effort.Max]);
expect(model.thinking?.requiresEffort).toBe(true);
expect(model.thinking?.defaultLevel).toBe(Effort.Max);
expect(model.thinking?.mode).toBe("anthropic-budget-effort");
});
});
+5 -2
View File
@@ -1,6 +1,7 @@
import { describe, expect, it } from "bun:test";
import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
import type { Context, Model, OpenAICompat } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { Effort } from "@oh-my-pi/pi-catalog/effort";
const testContext: Context = {
@@ -29,7 +30,9 @@ function captureResponsesPayload(
}
function customResponsesModel(compat: OpenAICompat): Model<"openai-responses"> {
return {
// Resolve compat through the production constructor: sparse user overrides on
// top of host detection, exactly as a configured custom model is built.
return buildModel<"openai-responses">({
id: "deepseek-v4-flash:cloud",
name: "deepseek-v4-flash:cloud",
api: "openai-responses",
@@ -46,7 +49,7 @@ function customResponsesModel(compat: OpenAICompat): Model<"openai-responses"> {
effortMap: compat.reasoningEffortMap,
},
compat,
} as unknown as Model<"openai-responses">;
});
}
describe("issue #931 — openai-responses reasoning effort mapping", () => {
+27 -2
View File
@@ -1,6 +1,7 @@
import { describe, expect, it } from "bun:test";
import { type } from "@oh-my-pi/omptype";
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import { type OpenAICompletionsOptions, streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import { type OpenAIResponsesOptions, streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
import type { Context, Model, Tool } from "@oh-my-pi/pi-ai/types";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
@@ -23,7 +24,7 @@ function abortedSignal(): AbortSignal {
async function capturePayload(
model: Model<"openai-completions">,
opts: Parameters<typeof streamOpenAICompletions>[2],
opts: OpenAICompletionsOptions,
): Promise<Record<string, unknown>> {
const { promise, resolve } = Promise.withResolvers<unknown>();
streamOpenAICompletions(model, context, {
@@ -35,12 +36,36 @@ async function capturePayload(
return (await promise) as Record<string, unknown>;
}
async function captureResponsesPayload(
model: Model<"openai-responses">,
opts: OpenAIResponsesOptions,
): Promise<Record<string, unknown>> {
const { promise, resolve } = Promise.withResolvers<unknown>();
streamOpenAIResponses(model, context, {
...opts,
apiKey: "test-key",
signal: abortedSignal(),
onPayload: payload => resolve(payload),
});
return (await promise) as Record<string, unknown>;
}
describe("OpenCode Go tool_choice compatibility", () => {
it("marks deepseek-v4-pro as not supporting tool_choice via compat override", () => {
const model = getBundledModel("opencode-go", "deepseek-v4-pro") as Model<"openai-completions">;
expect(model.compat?.supportsToolChoice).toBe(false);
});
it("omits forced tool_choice from DeepSeek Flash Responses payloads while preserving tools", async () => {
const model = getBundledModel("opencode-go", "deepseek-v4-flash") as Model<"openai-responses">;
expect(model.compat.supportsToolChoice).toBe(false);
const body = await captureResponsesPayload(model, {
toolChoice: { type: "tool", name: "echo" },
});
expect(body.tools).toEqual([expect.objectContaining({ type: "function", name: "echo" })]);
expect(body.tool_choice).toBeUndefined();
});
it("marks mimo-v2.5-pro as not supporting tool_choice via compat override", () => {
const model = getBundledModel("opencode-go", "mimo-v2.5-pro") as Model<"openai-completions">;
expect(model.compat?.supportsToolChoice).toBe(false);
@@ -28,6 +28,7 @@ const compat: ResolvedOpenAICompat = {
supportsReasoningEffort: true,
supportsReasoningParams: true,
supportsSamplingParams: true,
supportsPenaltyAndStopParams: true,
alwaysSendMaxTokens: false,
isOpenRouterHost: false,
isVercelGatewayHost: false,
+163
View File
@@ -0,0 +1,163 @@
import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import { type AuthCredentialStore, AuthStorage, SqliteAuthCredentialStore } from "@oh-my-pi/pi-ai/auth-storage";
import * as oauthUtils from "@oh-my-pi/pi-ai/registry/oauth";
import * as kimiOauth from "@oh-my-pi/pi-ai/registry/oauth/kimi";
import type { OAuthCredentials } from "@oh-my-pi/pi-ai/registry/oauth/types";
import type { UsageLimit, UsageProvider, UsageReport } from "@oh-my-pi/pi-ai/usage";
import { removeWithRetries } from "../../utils/src/temp";
const HOUR_MS = 60 * 60 * 1000;
const WEEK_MS = 7 * 24 * HOUR_MS;
function createCredential(accountId: string): OAuthCredentials {
return {
access: `access-${accountId}`,
refresh: `refresh-${accountId}`,
expires: Date.now() + HOUR_MS,
accountId,
};
}
function createUsageReport(accountId: string, fiveHourUsed: number, weeklyUsed: number): UsageReport {
const now = Date.now();
const limits: UsageLimit[] = [
{
id: "kimi-code:7d",
label: "7d limit",
scope: { provider: "kimi-code", accountId, windowId: "7d", shared: true },
window: { id: "7d", label: "7d limit", durationMs: WEEK_MS, resetsAt: now + WEEK_MS },
amount: {
unit: "percent",
usedFraction: weeklyUsed,
remainingFraction: 1 - weeklyUsed,
},
status: weeklyUsed >= 1 ? "exhausted" : weeklyUsed >= 0.9 ? "warning" : "ok",
},
{
id: "kimi-code:5h",
label: "5h limit",
scope: { provider: "kimi-code", accountId, windowId: "5h", shared: true },
window: { id: "5h", label: "5h limit", durationMs: 5 * HOUR_MS, resetsAt: now + 5 * HOUR_MS },
amount: {
unit: "percent",
usedFraction: fiveHourUsed,
remainingFraction: 1 - fiveHourUsed,
},
status: fiveHourUsed >= 1 ? "exhausted" : fiveHourUsed >= 0.9 ? "warning" : "ok",
},
];
return { provider: "kimi-code", fetchedAt: now, limits, metadata: { accountId } };
}
describe("AuthStorage Kimi OAuth ranking", () => {
let tempDir = "";
let store: AuthCredentialStore | null = null;
let authStorage: AuthStorage | null = null;
const usageByAccount = new Map<string, UsageReport>();
const usageProvider: UsageProvider = {
id: "kimi-code",
async fetchUsage(params) {
const accountId = params.credential.accountId;
return accountId ? (usageByAccount.get(accountId) ?? null) : null;
},
};
beforeEach(async () => {
tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-auth-kimi-selection-"));
store = await SqliteAuthCredentialStore.open(path.join(tempDir, "agent.db"));
authStorage = new AuthStorage(store, {
usageProviderResolver: provider => (provider === "kimi-code" ? usageProvider : undefined),
});
usageByAccount.clear();
vi.spyOn(oauthUtils, "getOAuthApiKey").mockImplementation(async (_provider, credentials) => {
const credential = credentials["kimi-code"] as OAuthCredentials | undefined;
if (!credential) return null;
return { apiKey: credential.access, newCredentials: credential };
});
});
afterEach(async () => {
vi.restoreAllMocks();
store?.close();
store = null;
authStorage = null;
if (tempDir) {
await removeWithRetries(tempDir);
tempDir = "";
}
});
test("new sessions choose the Kimi account with more 5h and 7d headroom", async () => {
if (!authStorage) throw new Error("test setup failed");
await authStorage.set("kimi-code", [
{ type: "oauth", ...createCredential("loaded") },
{ type: "oauth", ...createCredential("fresh") },
]);
usageByAccount.set("loaded", createUsageReport("loaded", 0.92, 0.8));
usageByAccount.set("fresh", createUsageReport("fresh", 0.01, 0.02));
const selected = new Set<string>();
for (let index = 0; index < 20; index += 1) {
const apiKey = await authStorage.getApiKey("kimi-code", `kimi-ranking-${index}`);
if (apiKey) selected.add(apiKey);
}
expect(selected).toEqual(new Set(["access-fresh"]));
});
test("usage-limit blocks last until the exhausted Kimi window resets", async () => {
if (!authStorage || !store?.getCredentialBlock) throw new Error("test setup failed");
await authStorage.set("kimi-code", [
{ type: "oauth", ...createCredential("exhausted") },
{ type: "oauth", ...createCredential("sibling") },
]);
const exhaustedReport = createUsageReport("exhausted", 1, 0.7);
usageByAccount.set("exhausted", exhaustedReport);
usageByAccount.set("sibling", createUsageReport("sibling", 0, 0));
const exhaustedRow = store.listAuthCredentials("kimi-code").find(row => {
const credential = row.credential;
return credential.type === "oauth" && credential.accountId === "exhausted";
});
if (!exhaustedRow) throw new Error("expected exhausted Kimi credential");
const result = await authStorage.markUsageLimitReached("kimi-code", undefined, {
credentialId: exhaustedRow.id,
});
expect(result.switched).toBe(true);
const resetAt = exhaustedReport.limits.find(limit => limit.window?.id === "5h")?.window?.resetsAt;
expect(store.getCredentialBlock(exhaustedRow.id, "kimi-code:oauth", "")).toBe(resetAt);
});
});
describe("Kimi OAuth account identity", () => {
afterEach(() => {
vi.restoreAllMocks();
});
test("keeps the JWT user id across token refresh", async () => {
const accessToken = `header.${Buffer.from(JSON.stringify({ user_id: "kimi-user-42", sub: "kimi-user-42" })).toString("base64url")}.signature`;
vi.spyOn(globalThis, "fetch").mockImplementation(
Object.assign(
async () =>
new Response(
JSON.stringify({
access_token: accessToken,
refresh_token: "refresh-1",
expires_in: 60 * 60,
}),
{ status: 200, headers: { "Content-Type": "application/json" } },
),
{ preconnect: fetch.preconnect },
),
);
const refreshed = await kimiOauth.refreshKimiToken("refresh-0");
expect(refreshed.accountId).toBe("kimi-user-42");
});
});
+11 -1
View File
@@ -3,10 +3,11 @@ import type { FetchImpl } from "@oh-my-pi/pi-ai/types";
import type { UsageFetchContext, UsageFetchParams } from "@oh-my-pi/pi-ai/usage";
import { kimiUsageProvider } from "@oh-my-pi/pi-ai/usage/kimi";
function makeCredential(): UsageFetchParams["credential"] {
function makeCredential(accountId?: string): UsageFetchParams["credential"] {
return {
type: "oauth",
accessToken: "kimi-test-token",
accountId,
};
}
@@ -102,4 +103,13 @@ describe("kimi usage provider", () => {
expect(report!.limits[0]!.window?.id).toBe("7d");
expect(report!.limits[1]!.window?.id).toBe("90m");
});
it("attaches the credential account id used by stable usage labels", async () => {
const report = await kimiUsageProvider.fetchUsage!(
{ provider: "kimi-code", credential: makeCredential("kimi-user-42"), signal: undefined },
makeCtx({ usage: { limit: "100", used: "28", remaining: "72" } }),
);
expect(report?.metadata?.accountId).toBe("kimi-user-42");
});
});
@@ -419,6 +419,58 @@ describe("wrapLeakedThinkingStream", () => {
expect(result.content.slice(1, 3)).toEqual(serverBlocks);
});
it("preserves complete Anthropic tool-search history through the custom-endpoint projector", async () => {
const firstThinking: ThinkingContent = {
type: "thinking",
thinking: "find the deferred tool",
thinkingSignature: "sig-1",
};
const serverBlocks: AnthropicServerToolContent[] = [
{
type: "anthropicServerTool",
block: {
type: "server_tool_use",
id: "srvtoolu_search",
name: "tool_search_tool_regex",
input: { pattern: "read" },
},
},
{
type: "anthropicServerTool",
block: {
type: "tool_search_tool_result",
tool_use_id: "srvtoolu_search",
content: {
type: "tool_search_tool_search_result",
tool_references: [{ type: "tool_reference", tool_name: "_read" }],
},
},
},
];
const secondThinking: ThinkingContent = {
type: "thinking",
thinking: "use the discovered tool",
thinkingSignature: "sig-2",
};
const call: ToolCall = {
type: "toolCall",
id: "toolu_read",
name: "_read",
arguments: { path: "notes.txt" },
};
const content: AssistantMessage["content"] = [firstThinking, ...serverBlocks, secondThinking, call];
const terminal = msg({ content, stopReason: "toolUse" });
const { result } = await runWrapper(inner => {
inner.push({ type: "start", partial: msg() });
inner.push({ type: "toolcall_start", contentIndex: 4, partial: terminal });
inner.push({ type: "toolcall_end", contentIndex: 4, toolCall: call, partial: terminal });
inner.push({ type: "done", reason: "toolUse", message: terminal });
});
expect(result.content).toEqual(content);
});
it("drops incomplete Anthropic web-search history instead of replaying orphan blocks", async () => {
const content: AssistantMessage["content"] = [
{
@@ -659,6 +711,43 @@ describe("leaked thinking healing through stream()", () => {
return Object.assign(fn, { preconnect: fetch.preconnect });
}
function anthropicThinkingFetch(): FetchImpl {
const body = [
sseFrame("message_start", {
type: "message_start",
message: { id: "msg_thinking_prefix", usage: { input_tokens: 5, output_tokens: 0 } },
}),
sseFrame("content_block_start", {
type: "content_block_start",
index: 0,
content_block: { type: "thinking", thinking: "Summary prefix" },
}),
sseFrame("content_block_delta", {
type: "content_block_delta",
index: 0,
delta: { type: "thinking_delta", thinking: " summary tail" },
}),
sseFrame("content_block_delta", {
type: "content_block_delta",
index: 0,
delta: { type: "signature_delta", signature: "sig_thinking" },
}),
sseFrame("content_block_stop", { type: "content_block_stop", index: 0 }),
sseFrame("message_delta", {
type: "message_delta",
delta: { stop_reason: "end_turn" },
usage: { input_tokens: 5, output_tokens: 4 },
}),
sseFrame("message_stop", { type: "message_stop" }),
].join("");
const fn = async (_input: string | URL | Request, _init?: RequestInit): Promise<Response> =>
new Response(body, {
status: 200,
headers: { "content-type": "text/event-stream", "request-id": "req_thinking_prefix" },
});
return Object.assign(fn, { preconnect: fetch.preconnect });
}
function anthropicModel(overrides: Partial<Model<"anthropic-messages">> = {}): Model<"anthropic-messages"> {
return buildModel({
id: "claude-sonnet-4-5",
@@ -711,6 +800,25 @@ describe("leaked thinking healing through stream()", () => {
expect(texts(result).join("").trim()).toBe("Final answer.");
});
it("preserves thinking bytes from content_block_start through a non-official endpoint", async () => {
const result = await stream(
anthropicModel({ provider: "zai", baseUrl: "https://api.z.ai/api/anthropic" }),
context,
{
apiKey: "test",
fetch: anthropicThinkingFetch(),
},
).result();
expect(thinks(result)).toEqual([
{
type: "thinking",
thinking: "Summary prefix summary tail",
thinkingSignature: "sig_thinking",
},
]);
});
it("replays native web-search history on a custom Anthropic continuation", async () => {
const searchResult = {
type: "web_search_tool_result",
+377
View File
@@ -0,0 +1,377 @@
import { describe, expect, it } from "bun:test";
import { retryTransientCompletion } from "@oh-my-pi/pi-ai/oneshot-retry";
import type { AssistantMessage, Usage } from "@oh-my-pi/pi-ai/types";
/**
* Defends the contract every oneshot LLM call site now depends on:
* `completeSimple` reports a transient provider failure by RESOLVING with
* `stopReason: "error"` rather than throwing, so a retry layer that only
* catches exceptions silently never fires. Before this helper, an Anthropic
* `overloaded_error` / 429 / 529 on a summary, title, handoff or image
* description failed on the first blip — or was swallowed into `null`, making a
* transient overload indistinguishable from a legitimate empty result.
*
* These tests pin the four properties the call sites rely on: transient
* error-stops are re-issued, non-transient ones are not, the final failure is
* handed back unchanged (so existing `null`/throw fallbacks still work), and a
* caller abort wins immediately.
*/
const emptyUsage = (): Usage =>
({
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
}) as unknown as Usage;
function message(overrides: Partial<AssistantMessage> = {}): AssistantMessage {
return {
role: "assistant",
content: [],
api: "anthropic-messages",
provider: "anthropic",
model: "claude-sonnet-4-6",
usage: emptyUsage(),
stopReason: "stop",
timestamp: 0,
...overrides,
} as AssistantMessage;
}
const overloaded = (): AssistantMessage =>
message({
stopReason: "error",
errorStatus: 529,
errorMessage: "Anthropic stream error (overloaded_error): Overloaded",
});
const rateLimited = (): AssistantMessage =>
message({ stopReason: "error", errorStatus: 429, errorMessage: "rate_limit_error: too many requests" });
// Keep the suite fast: the helper's real backoff floor is 500ms.
const fast = { baseDelayMs: 1, maxAttempts: 3 } as const;
describe("retryTransientCompletion", () => {
it("re-issues an Anthropic 529 error-stop and returns the eventual success", async () => {
const results = [overloaded(), overloaded(), message({ stopReason: "stop" })];
let calls = 0;
const final = await retryTransientCompletion(() => {
calls += 1;
return Promise.resolve(results.shift()!);
}, fast);
expect(calls).toBe(3);
expect(final.stopReason).toBe("stop");
});
it("re-issues a 429 rate-limit error-stop", async () => {
let calls = 0;
const final = await retryTransientCompletion(() => {
calls += 1;
return Promise.resolve(calls === 1 ? rateLimited() : message());
}, fast);
expect(calls).toBe(2);
expect(final.stopReason).toBe("stop");
});
it("re-issues a status-only 503 error-stop", async () => {
let calls = 0;
const final = await retryTransientCompletion(() => {
calls += 1;
return Promise.resolve(
calls === 1
? message({ stopReason: "error", errorStatus: 503, errorMessage: "request failed" })
: message(),
);
}, fast);
expect(calls).toBe(2);
expect(final.stopReason).toBe("stop");
});
it("returns the failing message unchanged once attempts are exhausted, so caller fallbacks still apply", async () => {
let calls = 0;
const final = await retryTransientCompletion(() => {
calls += 1;
return Promise.resolve(overloaded());
}, fast);
expect(calls).toBe(3);
expect(final.stopReason).toBe("error");
expect(final.errorMessage).toContain("overloaded_error");
});
it("does not retry a non-transient provider error", async () => {
let calls = 0;
const final = await retryTransientCompletion(
() => {
calls += 1;
return Promise.resolve(
message({
stopReason: "error",
errorStatus: 400,
errorMessage: "invalid_request_error: messages: at least one message is required",
}),
);
},
{ ...fast, maxAttempts: 5 },
);
expect(calls).toBe(1);
expect(final.stopReason).toBe("error");
});
it("does not retry a deterministic llama.cpp tool-call parse failure reported as 500", async () => {
let calls = 0;
const final = await retryTransientCompletion(
() => {
calls += 1;
return Promise.resolve(
message({
stopReason: "error",
errorStatus: 500,
errorMessage: "failed to parse tool call arguments as JSON",
}),
);
},
{ ...fast, maxAttempts: 5 },
);
expect(calls).toBe(1);
expect(final.stopReason).toBe("error");
});
it("retries a thrown transient error and rethrows the last one when exhausted", async () => {
let calls = 0;
const attempt = retryTransientCompletion(() => {
calls += 1;
const error = new Error("529 overloaded_error: Overloaded") as Error & { status?: number };
error.status = 529;
throw error;
}, fast);
await expect(attempt).rejects.toThrow(/overloaded_error/);
expect(calls).toBe(3);
});
it("retries a thrown status-only 503 error", async () => {
let calls = 0;
const final = await retryTransientCompletion(() => {
calls += 1;
if (calls === 1) {
const error = new Error("request failed") as Error & { status: number };
error.status = 503;
throw error;
}
return Promise.resolve(message());
}, fast);
expect(calls).toBe(2);
expect(final.stopReason).toBe("stop");
});
it("does not retry a thrown non-transient error", async () => {
let calls = 0;
const attempt = retryTransientCompletion(() => {
calls += 1;
throw new Error("invalid_request_error: bad tool schema");
}, fast);
await expect(attempt).rejects.toThrow(/invalid_request_error/);
expect(calls).toBe(1);
});
it("stops immediately when the caller aborts", async () => {
const controller = new AbortController();
let calls = 0;
const final = await retryTransientCompletion(
() => {
calls += 1;
controller.abort();
return Promise.resolve(overloaded());
},
{ ...fast, signal: controller.signal },
);
expect(calls).toBe(1);
expect(final.stopReason).toBe("error");
});
it("rejects with the abort reason when the caller cancels during backoff", async () => {
// The cancel lands while we are waiting, not while an attempt is in flight:
// it must stay a cancellation rather than being reported as the provider
// failure we happened to be sleeping on.
const controller = new AbortController();
const reason = new Error("user pressed escape");
let calls = 0;
const attempt = retryTransientCompletion(
() => {
calls += 1;
setTimeout(() => controller.abort(reason), 5);
return Promise.resolve(overloaded());
},
{ maxAttempts: 3, baseDelayMs: 200, signal: controller.signal },
);
await expect(attempt).rejects.toThrow("user pressed escape");
expect(calls).toBe(1);
});
it("reports each retry through onRetry so callers can log the wait", async () => {
const seen: number[] = [];
let calls = 0;
await retryTransientCompletion(
() => {
calls += 1;
return Promise.resolve(calls === 1 ? overloaded() : message());
},
{ ...fast, onRetry: info => seen.push(info.attempt) },
);
expect(seen).toEqual([1]);
});
it("surfaces the failure instead of parking when the provider asks for longer than maxDelayMs", async () => {
let calls = 0;
const final = await retryTransientCompletion(
() => {
calls += 1;
return Promise.resolve(
message({
stopReason: "error",
errorStatus: 429,
errorMessage: "rate_limit_error: please retry in 600s",
}),
);
},
{ ...fast, maxDelayMs: 1_000 },
);
expect(calls).toBe(1);
expect(final.stopReason).toBe("error");
});
it("honors a retry-after-ms response header over the backoff floor", async () => {
// The header is the only place a real Anthropic 429 carries its wait: the
// resolved AssistantMessage has no headers, so a helper that reads only the
// error text would silently fall back to plain backoff.
let calls = 0;
let observedDelay = -1;
await retryTransientCompletion(
() => {
calls += 1;
return Promise.resolve(calls === 1 ? rateLimited() : message());
},
{
maxAttempts: 2,
baseDelayMs: 1,
getResponseHeaders: () => ({ "retry-after-ms": "120" }),
onRetry: info => {
observedDelay = info.delayMs;
},
},
);
expect(calls).toBe(2);
expect(observedDelay).toBe(120);
});
it("honors the canonical retry-after-ms error-message suffix", async () => {
let calls = 0;
let observedDelay = -1;
await retryTransientCompletion(
() => {
calls += 1;
return Promise.resolve(
calls === 1
? message({
stopReason: "error",
errorStatus: 429,
errorMessage: "rate_limit_error: too many requests retry-after-ms=5",
})
: message(),
);
},
{
maxAttempts: 2,
baseDelayMs: 1,
onRetry: info => {
observedDelay = info.delayMs;
},
},
);
expect(calls).toBe(2);
expect(observedDelay).toBe(5);
});
it("surfaces a canonical retry-after-ms suffix above maxDelayMs", async () => {
let calls = 0;
const final = await retryTransientCompletion(
() => {
calls += 1;
return Promise.resolve(
message({
stopReason: "error",
errorStatus: 429,
errorMessage: "rate_limit_error: too many requests retry-after-ms=12000",
}),
);
},
{ ...fast, maxDelayMs: 1_000 },
);
expect(calls).toBe(1);
expect(final.stopReason).toBe("error");
});
it("surfaces the failure when a retry-after header exceeds maxDelayMs", async () => {
let calls = 0;
const final = await retryTransientCompletion(
() => {
calls += 1;
return Promise.resolve(rateLimited());
},
{
maxAttempts: 3,
baseDelayMs: 1,
maxDelayMs: 1_000,
getResponseHeaders: () => ({ "retry-after": "300" }),
},
);
expect(calls).toBe(1);
expect(final.stopReason).toBe("error");
});
it("recovers retry-after from a thrown provider error's own headers", async () => {
let calls = 0;
let observedDelay = -1;
const attempt = retryTransientCompletion(
() => {
calls += 1;
const error = new Error("529 overloaded_error: Overloaded") as Error & {
status?: number;
headers?: Record<string, string>;
};
error.status = 529;
error.headers = { "retry-after-ms": "90" };
throw error;
},
{
maxAttempts: 2,
baseDelayMs: 1,
onRetry: info => {
observedDelay = info.delayMs;
},
},
);
await expect(attempt).rejects.toThrow(/overloaded_error/);
expect(calls).toBe(2);
expect(observedDelay).toBe(90);
});
});
@@ -5,9 +5,6 @@ import type { AssistantMessage, Context, FetchImpl, Model, SimpleStreamOptions,
import { buildOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
const model = getBundledModel<"openai-completions">("xai", "grok-code-fast-1");
if (!model) throw new Error("Expected bundled xAI Grok model");
if (model.api !== "openai-completions") throw new Error(`Expected Chat Completions model, received ${model.api}`);
const context: Context = { messages: [{ role: "user", content: "hello", timestamp: 0 }] };
const openAI56ResponsesModel = getBundledModel<"openai-responses">("openai", "gpt-5.6");
@@ -44,7 +41,7 @@ function chatCompletionsSse(): Response {
id: "chatcmpl-affinity",
object: "chat.completion.chunk",
created: 0,
model: model.id,
model: openAI56CompletionsModel.id,
choices: [{ index: 0, delta, finish_reason: finishReason }],
});
@@ -56,7 +53,7 @@ function chatCompletionsSse(): Response {
async function captureRequest(
options: OpenAICompletionsOptions,
requestModel: Model<"openai-completions"> = model,
requestModel: Model<"openai-completions"> = openAI56CompletionsModel,
requestContext: Context = context,
): Promise<{ headers: Headers; body: Record<string, unknown> }> {
let requestHeaders: Headers | undefined;
@@ -83,7 +80,7 @@ async function captureRequest(
async function captureSimpleRequest(
options: SimpleStreamOptions,
requestModel: Model<"openai-completions"> = model,
requestModel: Model<"openai-completions"> = openAI56CompletionsModel,
requestContext: Context = context,
): Promise<{ headers: Headers; body: Record<string, unknown> }> {
let requestHeaders: Headers | undefined;
@@ -104,51 +101,6 @@ async function captureSimpleRequest(
return { headers: requestHeaders, body };
}
describe("openai-completions xAI cache affinity", () => {
const cases: Array<{
name: string;
options: OpenAICompletionsOptions;
expectedHeader: string | null;
}> = [
{
name: "uses sessionId when no prompt cache key is provided",
options: { sessionId: "session-fallback" },
expectedHeader: "session-fallback",
},
{
name: "keeps the prompt cache key stable across a distinct side-channel session",
options: { promptCacheKey: "stable-cache-key", sessionId: "side-channel-session" },
expectedHeader: "stable-cache-key",
},
{
name: "omits automatic affinity when caching is disabled",
options: {
promptCacheKey: "disabled-cache-key",
sessionId: "disabled-session",
cacheRetention: "none",
},
expectedHeader: null,
},
{
name: "preserves a caller-provided mixed-case affinity header",
options: {
promptCacheKey: "automatic-cache-key",
sessionId: "automatic-session",
headers: { "X-Grok-Conv-Id": "caller-affinity" },
},
expectedHeader: "caller-affinity",
},
];
for (const { name, options, expectedHeader } of cases) {
it(name, async () => {
const { headers } = await captureRequest(options);
expect(headers.get("x-grok-conv-id")).toBe(expectedHeader);
});
}
});
describe("OpenAI Chat Completions explicit prompt cache policy", () => {
const historicalContext: Context = {
messages: [
@@ -210,6 +210,7 @@ describe("openai-completions compatibility", () => {
toolStrictMode: "none",
supportsReasoningParams: true,
supportsSamplingParams: true,
supportsPenaltyAndStopParams: true,
alwaysSendMaxTokens: false,
isOpenRouterHost: false,
isVercelGatewayHost: false,
@@ -628,6 +629,79 @@ describe("openai-completions compatibility", () => {
expect(result.usage.totalTokens).toBe(15);
});
it("preserves opaque tool-call IDs when replaying a custom Chat Completions turn", async () => {
const model: Model<"openai-completions"> = buildModel({
id: "gateway-model",
name: "Gateway Model",
api: "openai-completions",
provider: "custom-gateway",
baseUrl: "https://gateway.example/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128_000,
maxTokens: 8_192,
} satisfies ModelSpec<"openai-completions">);
const toolCallId = "call_abc||gateway_state||opaque";
const assistant = await streamOpenAICompletions(model, baseContext(), {
apiKey: "test-key",
fetch: createMockFetch([
{
id: "chatcmpl-opaque-tool-id",
object: "chat.completion.chunk",
created: 0,
model: model.id,
choices: [
{
index: 0,
delta: {
tool_calls: [
{
index: 0,
id: toolCallId,
type: "function",
function: { name: "read", arguments: '{"path":"README.md"}' },
},
],
},
},
],
},
{
id: "chatcmpl-opaque-tool-id",
object: "chat.completion.chunk",
created: 0,
model: model.id,
choices: [{ index: 0, delta: {}, finish_reason: "tool_calls" }],
},
"[DONE]",
]),
}).result();
const streamedToolCall = assistant.content.find(content => content.type === "toolCall");
expect(streamedToolCall?.id).toBe(toolCallId);
const payload = await captureOpenAICompletionsPayload(model, {
messages: [
{ role: "user", content: "Read README", timestamp: 1 },
assistant,
{
role: "toolResult",
toolCallId,
toolName: "read",
content: [{ type: "text", text: "done" }],
isError: false,
timestamp: 2,
},
],
});
const replayMessages = getPayloadMessages(payload);
const assistantPayload = replayMessages.find(message => message.role === "assistant");
const toolCalls = assistantPayload?.tool_calls;
if (!Array.isArray(toolCalls)) throw new Error("assistant tool_calls missing");
expect(toObject(toolCalls[0])?.id).toBe(toolCallId);
expect(replayMessages.find(message => message.role === "tool")?.tool_call_id).toBe(toolCallId);
});
it("keeps unindexed batched tool-call arguments isolated", async () => {
const model: Model<"openai-completions"> = buildModel({
...gpt4oMiniSpec,
@@ -107,3 +107,93 @@ describe("finish_reason: error", () => {
expect(result.errorMessage).toMatch(RETRYABLE_PATTERN);
}, 10_000);
});
describe("finish_reason: insufficient_system_resource", () => {
// DeepSeek interrupts the generation mid-stream when its inference system
// runs out of resources; the terminal chunk carries
// `finish_reason: "insufficient_system_resource"`. It must surface as a
// retryable provider error — never as a clean `stop`.
it("maps to a retryable error message", async () => {
const fetchMock = createSseFetch([
completionChunk({ choices: [{ index: 0, delta: { role: "assistant", content: "Hel" } }] }),
completionChunk({ choices: [{ index: 0, delta: {}, finish_reason: "insufficient_system_resource" }] }),
"[DONE]",
]);
const result = await streamOpenAICompletions(completionsModel, baseContext(), {
apiKey: "test-key",
fetch: fetchMock,
}).result();
expect(result.stopReason).toBe("error");
expect(result.errorMessage).toMatch(RETRYABLE_PATTERN);
}, 10_000);
});
describe("premature stream closure", () => {
// The connection dies mid-generation without any `finish_reason` chunk
// (DeepSeek insufficient-system-resource interruption, flaky gateway).
// Before the guard, the partial message finalized as a clean `stop` and
// the agent loop treated the truncated turn as complete — the silent
// mid-sentence halt. Now it must surface as an error turn.
it("fails the turn instead of silently stopping", async () => {
const fetchMock = createSseFetch([
completionChunk({ choices: [{ index: 0, delta: { role: "assistant", content: "Hel" } }] }),
completionChunk({ choices: [{ index: 0, delta: { content: "lo" } }] }),
]);
const eventTypes: string[] = [];
let errorMessage: string | undefined;
for await (const event of streamOpenAICompletions(completionsModel, baseContext(), {
apiKey: "test-key",
fetch: fetchMock,
})) {
eventTypes.push(event.type);
if (event.type === "error") errorMessage = event.error.errorMessage;
}
expect(eventTypes).toEqual(["start", "text_start", "text_delta", "text_delta", "text_end", "error"]);
expect(errorMessage).toContain("finish_reason");
}, 10_000);
it("still retries a genuinely empty close via the empty-completion path", async () => {
// Zero content + no finish_reason is the flaky-gateway empty completion:
// it stays a clean `stop` so withEmptyCompletionRetry can re-sample
// instead of failing outright.
let attempts = 0;
async function fetchMock(_input: string | URL | Request, _init?: RequestInit): Promise<Response> {
attempts++;
const events =
attempts === 1
? []
: [
completionChunk({ choices: [{ index: 0, delta: { role: "assistant", content: "Hi" } }] }),
completionChunk({ choices: [{ index: 0, delta: {}, finish_reason: "stop" }] }),
"[DONE]",
];
const encoder = new TextEncoder();
const stream = new ReadableStream<Uint8Array>({
start(controller) {
for (const event of events) {
const data = typeof event === "string" ? event : JSON.stringify(event);
controller.enqueue(encoder.encode(`data: ${data}\n\n`));
}
controller.close();
},
});
return new Response(stream, {
status: 200,
headers: { "content-type": "text/event-stream" },
});
}
const result = await streamOpenAICompletions(completionsModel, baseContext(), {
apiKey: "test-key",
fetch: fetchMock as typeof fetch,
}).result();
expect(attempts).toBeGreaterThan(1);
expect(result.stopReason).toBe("stop");
expect(result.content).toEqual([{ type: "text", text: "Hi" }]);
}, 10_000);
});
@@ -51,6 +51,7 @@ const compat: ResolvedOpenAICompat = {
toolStrictMode: "none",
supportsReasoningParams: true,
supportsSamplingParams: true,
supportsPenaltyAndStopParams: true,
alwaysSendMaxTokens: false,
isOpenRouterHost: false,
isVercelGatewayHost: false,
@@ -227,7 +228,7 @@ describe("openai-completions convertMessages", () => {
const assistantMessage: AssistantMessage = {
role: "assistant",
content: [{ type: "toolCall", id: emptyNormalizingId, name: "read", arguments: { path: "README.md" } }],
api: model.api,
api: "openai-responses",
provider: model.provider,
model: model.id,
usage: emptyUsage,
@@ -0,0 +1,122 @@
import { describe, expect, it } from "bun:test";
import { type } from "@oh-my-pi/omptype";
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import type { Context, Model, ModelSpec, Tool, ToolChoice } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
interface ChatCompletionsPayload {
tool_choice?: unknown;
tools?: Array<{ type?: string; function?: { name?: string; parameters?: { anyOf?: unknown } } }>;
}
const coverageTool: Tool = {
name: "mcp__codebase_memory_check_index_coverage",
description: "coverage",
parameters: {
type: "object",
properties: {
project: { type: "string" },
paths: { type: "array", items: { type: "string" } },
scopes: { type: "array", items: { type: "string" } },
},
required: ["project"],
anyOf: [{ required: ["paths"] }, { required: ["scopes"] }],
} as unknown as Tool["parameters"],
};
const leftoverTool: Tool = {
name: "mcp__leftover_union",
description: "union",
parameters: {
type: "object",
properties: { kind: { type: "string" } },
anyOf: [
{ required: ["kind"], minProperties: 1 },
{ required: ["kind"], minProperties: 2 },
],
} as unknown as Tool["parameters"],
};
const goodTool: Tool = {
name: "read_file",
description: "read a file",
parameters: type({ path: type("string") }),
};
function makeModel(provider: "openai" | "xai"): Model<"openai-completions"> {
return buildModel({
id: provider === "xai" ? "grok-4" : "gpt-4o-mini",
name: provider === "xai" ? "Grok 4" : "GPT-4o Mini",
api: "openai-completions",
provider,
baseUrl: provider === "xai" ? "https://api.x.ai/v1" : "https://api.openai.com/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128000,
maxTokens: 4096,
} as ModelSpec<"openai-completions">);
}
function abortedSignal(): AbortSignal {
const controller = new AbortController();
controller.abort();
return controller.signal;
}
function capturePayload(
provider: "openai" | "xai",
tools: Tool[],
toolChoice?: ToolChoice,
): Promise<ChatCompletionsPayload> {
const { promise, resolve } = Promise.withResolvers<ChatCompletionsPayload>();
const context: Context = {
messages: [{ role: "user", content: "check coverage", timestamp: 0 }],
tools,
};
streamOpenAICompletions(makeModel(provider), context, {
apiKey: "test-key",
toolChoice,
signal: abortedSignal(),
onPayload: payload => resolve(payload as ChatCompletionsPayload),
});
return promise;
}
function toolNames(payload: ChatCompletionsPayload): Array<string | undefined> {
return payload.tools?.map(tool => tool.function?.name) ?? [];
}
describe("openai-completions xAI leftover-union quarantine", () => {
it("keeps an exclusive-required MCP tool after xAI flatten on paid xAI", async () => {
const payload = await capturePayload("xai", [coverageTool, goodTool]);
expect(toolNames(payload)).toEqual(["mcp__codebase_memory_check_index_coverage", "read_file"]);
expect(payload.tools?.[0]?.function?.parameters?.anyOf).toBeUndefined();
});
it("preserves an exclusive-required MCP tool on OpenAI Completions", async () => {
const payload = await capturePayload("openai", [coverageTool, goodTool]);
expect(toolNames(payload)).toEqual(["mcp__codebase_memory_check_index_coverage", "read_file"]);
expect(payload.tools?.[0]?.function?.parameters?.anyOf).toHaveLength(2);
});
it("keeps a leftover object-root union on OpenAI Completions", async () => {
const payload = await capturePayload("openai", [leftoverTool, goodTool]);
expect(toolNames(payload)).toEqual(["mcp__leftover_union", "read_file"]);
expect(payload.tools?.[0]?.function?.parameters?.anyOf).toHaveLength(2);
});
it("quarantines a leftover object-root union on paid xAI only", async () => {
const payload = await capturePayload("xai", [leftoverTool, goodTool]);
expect(toolNames(payload)).toEqual(["read_file"]);
});
it("drops a forced tool_choice when the leftover-union tool was quarantined", async () => {
const payload = await capturePayload("xai", [leftoverTool, goodTool], {
type: "tool",
name: "mcp__leftover_union",
});
expect(toolNames(payload)).toEqual(["read_file"]);
expect(payload.tool_choice).toBeUndefined();
});
});
@@ -57,6 +57,20 @@ const xaiOAuthResponsesModel: Model<"openai-responses"> = {
reasoning: true,
}),
};
const xaiApiKeyResponsesModel: Model<"openai-responses"> = {
...model,
id: "grok-code-fast-1",
name: "Grok Code Fast 1",
provider: "xai",
baseUrl: "https://api.x.ai/v1",
compat: buildOpenAIResponsesCompat({
id: "grok-code-fast-1",
name: "Grok Code Fast 1",
provider: "xai",
baseUrl: "https://api.x.ai/v1",
reasoning: true,
}),
};
const openAI56ResponsesModel: Model<"openai-responses"> = {
...model,
@@ -675,6 +689,16 @@ describe("openai-responses cache affinity", () => {
}
});
it("sets x-grok-conv-id cache affinity for paid xai Responses requests", async () => {
const captured = await captureDispatchedOpenAIResponseHeaders(
{ sessionId: "session-fallback" },
xaiApiKeyResponsesModel,
);
expect(getHeader(captured.headers, "x-grok-conv-id")).toBe("session-fallback");
expect(captured.body?.prompt_cache_key).toBe("session-fallback");
});
it("sets OpenRouter Responses session_id from sessionId in the body", async () => {
const captured = await captureOpenAIResponseHeaders(
{ sessionId: "workflow-123", promptCacheKey: "cache-key-123" },
@@ -5,13 +5,13 @@ import type { Context, Model, ModelSpec, Tool } from "@oh-my-pi/pi-ai/types";
import { findStrictToolSchemaViolation } from "@oh-my-pi/pi-ai/utils/schema";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
function makeModel(): Model<"openai-responses"> {
function makeModel(provider: "openai" | "xai-oauth" = "openai"): Model<"openai-responses"> {
return buildModel({
id: "gpt-5",
name: "GPT-5",
id: provider === "xai-oauth" ? "grok-4" : "gpt-5",
name: provider === "xai-oauth" ? "Grok 4" : "GPT-5",
api: "openai-responses",
provider: "openai",
baseUrl: "https://api.openai.com/v1",
provider,
baseUrl: provider === "xai-oauth" ? "https://api.x.ai/v1" : "https://api.openai.com/v1",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
@@ -20,6 +20,15 @@ function makeModel(): Model<"openai-responses"> {
} as ModelSpec<"openai-responses">);
}
const leftoverRootUnion = {
type: "object",
properties: { kind: { type: "string" } },
anyOf: [
{ required: ["kind"], minProperties: 1 },
{ required: ["kind"], minProperties: 2 },
],
} as const;
describe("findStrictToolSchemaViolation (#2652)", () => {
test("flags a non-null enum on a null-typed node (nullable-enum shape)", () => {
expect(findStrictToolSchemaViolation({ enum: ["A", "B"], type: "null" })).toBe("#/enum");
@@ -54,6 +63,26 @@ describe("findStrictToolSchemaViolation (#2652)", () => {
// enum without a declared type cannot contradict anything.
expect(findStrictToolSchemaViolation({ enum: ["x"] })).toBeNull();
});
test("flags a leftover xAI root anyOf only when the xAI option is on", () => {
expect(findStrictToolSchemaViolation(leftoverRootUnion)).toBeNull();
expect(findStrictToolSchemaViolation(leftoverRootUnion, "#", { rejectXaiRootObjectUnion: true })).toBe("#/anyOf");
});
test("accepts a root anyOf of typed object branches even for xAI", () => {
expect(
findStrictToolSchemaViolation(
{
anyOf: [
{ type: "object", properties: { a: { type: "string" } } },
{ type: "object", properties: { b: { type: "number" } } },
],
},
"#",
{ rejectXaiRootObjectUnion: true },
),
).toBeNull();
});
});
const badTool: Tool = {
@@ -66,6 +95,20 @@ const badTool: Tool = {
additionalProperties: false,
} as unknown as Tool["parameters"],
};
const coverageTool: Tool = {
name: "mcp__codebase_memory_check_index_coverage",
description: "coverage",
parameters: {
type: "object",
properties: {
project: { type: "string" },
paths: { type: "array", items: { type: "string" } },
scopes: { type: "array", items: { type: "string" } },
},
required: ["project"],
anyOf: [{ required: ["paths"] }, { required: ["scopes"] }],
} as unknown as Tool["parameters"],
};
const goodTool: Tool = {
name: "read_file",
description: "read a file",
@@ -91,6 +134,61 @@ describe("convertTools quarantine (#2652)", () => {
expect(convertTools([goodTool], true, makeModel())).toHaveLength(1);
});
test("flattens an exclusive-required MCP tool on xAI Responses", () => {
const out = convertTools([coverageTool, goodTool], true, makeModel("xai-oauth")) as Array<{
name: string;
parameters: { anyOf?: unknown };
}>;
expect(out.map(t => t.name)).toEqual(["mcp__codebase_memory_check_index_coverage", "read_file"]);
expect(out[0]?.parameters.anyOf).toBeUndefined();
});
test("preserves an exclusive-required MCP tool on OpenAI Responses", () => {
const out = convertTools([coverageTool, goodTool], true, makeModel()) as Array<{
name: string;
parameters: { anyOf?: unknown };
}>;
expect(out.map(t => t.name)).toEqual(["mcp__codebase_memory_check_index_coverage", "read_file"]);
expect(out[0]?.parameters.anyOf).toHaveLength(2);
});
test("keeps a leftover object-root union on OpenAI Responses", () => {
const leftoverTool: Tool = {
name: "mcp__leftover_union",
description: "union",
parameters: {
type: "object",
properties: { kind: { type: "string" } },
anyOf: [
{ required: ["kind"], minProperties: 1 },
{ required: ["kind"], minProperties: 2 },
],
} as unknown as Tool["parameters"],
};
const out = convertTools([leftoverTool, goodTool], true, makeModel()) as Array<{
name: string;
parameters: { anyOf?: unknown };
}>;
expect(out.map(t => t.name)).toEqual(["mcp__leftover_union", "read_file"]);
expect(out[0]?.parameters.anyOf).toHaveLength(2);
});
test("quarantines a leftover object-root union on xAI Responses only", () => {
const leftoverTool: Tool = {
name: "mcp__leftover_union",
description: "union",
parameters: {
type: "object",
properties: { kind: { type: "string" } },
anyOf: [
{ required: ["kind"], minProperties: 1 },
{ required: ["kind"], minProperties: 2 },
],
} as unknown as Tool["parameters"],
};
const out = convertTools([leftoverTool, goodTool], true, makeModel("xai-oauth")) as Array<{ name: string }>;
expect(out.map(t => t.name)).toEqual(["read_file"]);
});
test("reports the hidden tool name and the offending schema path", () => {
const dropped: Array<{ name: string; path: string }> = [];
convertTools([badTool], true, makeModel(), (name, path) => dropped.push({ name, path }));
+32
View File
@@ -154,6 +154,38 @@ describe("parseRateLimitReason", () => {
expect(parseRateLimitReason("API 使用频率已达上限")).toBe("UNKNOWN");
});
it("keeps DashScope/Bailian TPM throttle in the transient lane", () => {
// Bailian reports its per-minute token throttle (429
// Throttling.AllocationQuota) with OpenAI-compatible billing wording,
// but links the error-code doc's #token-limit anchor, which documents
// the error as a transient TPM/TPS cap (clears within the minute
// window). Must retry on the same credential with a short backoff —
// previously classified QUOTA_EXHAUSTED, blocking the credential for
// 30 minutes and stalling the session.
const throttle =
"429 You exceeded your current quota, please check your plan and billing details. For details, see: https://help.aliyun.com/zh/model-studio/error-code#token-limit\nYou exceeded your current quota, please check your plan and billing details. For details, see: https://help.aliyun.com/zh/model-studio/error-code#token-limit (type=insufficient_quota param=insufficient_quota)";
expect(parseRateLimitReason(throttle)).toBe("RATE_LIMIT_EXCEEDED");
expect(isUsageLimit(throttle)).toBe(false);
expect(isUsageLimit(Object.assign(new Error(throttle), { status: 429 }))).toBe(false);
expect(isUsageLimit(new ProviderHttpError(throttle, 429, { code: "insufficient_quota" }))).toBe(false);
expect(isUsageLimitOutcome(429, throttle)).toBe(false);
// The identical wording WITHOUT the doc anchor is OpenAI's real
// account-quota error and stays quota-exhausted.
const openaiQuota =
"429 You exceeded your current quota, please check your plan and billing details. For details, see: https://platform.openai.com/account/usage (type=insufficient_quota)";
expect(parseRateLimitReason(openaiQuota)).toBe("QUOTA_EXHAUSTED");
expect(isUsageLimitOutcome(429, openaiQuota)).toBe(true);
// The same DashScope doc anchor also covers permanent free-quota
// exhaustion. The anchor alone must not turn that into a retry loop.
const freeQuota =
"429 Free allocated quota exceeded. For details, see: https://help.aliyun.com/zh/model-studio/error-code#token-limit (type=insufficient_quota)";
expect(parseRateLimitReason(freeQuota)).toBe("QUOTA_EXHAUSTED");
expect(isUsageLimit(new ProviderHttpError(freeQuota, 429, { code: "insufficient_quota" }))).toBe(true);
expect(isUsageLimitOutcome(429, freeQuota)).toBe(true);
});
it("classifies Codex usage limit error as QUOTA_EXHAUSTED", () => {
expect(
parseRateLimitReason("Codex error event: The usage limit has been reached (code=usage_limit_reached)"),
@@ -606,6 +606,63 @@ describe("sanitizeSchemaForOpenAIResponses", () => {
expect(properties.self).toBe(sanitized as unknown as object);
expect((sanitized as { type: unknown }).type).toBe("object");
});
it("preserves exclusive-required anyOf for provider-specific handling", () => {
const schema = {
type: "object",
properties: {
project: { type: "string" },
paths: { type: "array", items: { type: "string" } },
scopes: { type: "array", items: { type: "string" } },
},
required: ["project"],
anyOf: [{ required: ["paths"] }, { required: ["scopes"] }],
};
expect(sanitizeSchemaForOpenAIResponses(schema)).toEqual({
type: "object",
properties: {
project: { type: "string" },
paths: { type: "array", items: { type: "string" } },
scopes: { type: "array", items: { type: "string" } },
},
required: ["project"],
anyOf: [{ required: ["paths"] }, { required: ["scopes"] }],
});
});
it("does not flatten nested exclusive-required anyOf (xAI only rejects the tool root)", () => {
const schema = {
type: "object",
properties: {
outputSchema: {
type: "object",
properties: {
paths: { type: "array", items: { type: "string" } },
scopes: { type: "array", items: { type: "string" } },
},
anyOf: [{ required: ["paths"] }, { required: ["scopes"] }],
},
},
required: ["outputSchema"],
};
const sanitized = sanitizeSchemaForOpenAIResponses(schema);
expect(sanitized.anyOf).toBeUndefined();
const outputSchema = (sanitized.properties as Record<string, unknown>).outputSchema as Record<string, unknown>;
expect(outputSchema.anyOf).toEqual([{ required: ["paths"] }, { required: ["scopes"] }]);
});
it("does not flatten a root union that constrains existing properties", () => {
const schema = {
type: "object",
properties: { kind: { type: "string" } },
anyOf: [{ properties: { kind: { const: "a" } } }, { properties: { kind: { const: "b" } } }],
};
expect(sanitizeSchemaForOpenAIResponses(schema).anyOf).toEqual([
{ properties: { kind: { const: "a" } } },
{ properties: { kind: { const: "b" } } },
]);
});
});
// ---------------------------------------------------------------------------
+57
View File
@@ -87,6 +87,63 @@ describe("toolWireSchema — raw JSON Schema normalization", () => {
});
});
it("preserves exclusive-required anyOf for provider-specific handling", () => {
const wire = toolWireSchema(
jsonTool({
type: "object",
properties: {
project: { type: "string" },
paths: { type: "array", items: { type: "string" } },
scopes: { type: "array", items: { type: "string" } },
},
required: ["project"],
anyOf: [{ required: ["paths"] }, { required: ["scopes"] }],
}),
);
expect(wire.anyOf).toEqual([{ required: ["paths"] }, { required: ["scopes"] }]);
expect(wire.type).toBe("object");
expect(wire.required).toEqual(["project"]);
});
it("does not flatten nested exclusive-required anyOf (only the tool root 400s xAI)", () => {
const wire = toolWireSchema(
jsonTool({
type: "object",
properties: {
outputSchema: {
type: "object",
properties: {
paths: { type: "array", items: { type: "string" } },
scopes: { type: "array", items: { type: "string" } },
},
anyOf: [{ required: ["paths"] }, { required: ["scopes"] }],
},
},
required: ["outputSchema"],
}),
);
expect(wire.anyOf).toBeUndefined();
const outputSchema = (wire.properties as Record<string, unknown>).outputSchema as Record<string, unknown>;
expect(outputSchema.anyOf).toEqual([{ required: ["paths"] }, { required: ["scopes"] }]);
});
it("does not flatten a root union that constrains existing properties", () => {
const wire = toolWireSchema(
jsonTool({
type: "object",
properties: { kind: { type: "string" } },
anyOf: [{ properties: { kind: { const: "a" } } }, { properties: { kind: { const: "b" } } }],
}),
);
expect(wire.anyOf).toEqual([{ properties: { kind: { const: "a" } } }, { properties: { kind: { const: "b" } } }]);
const properties = wire.properties;
expect(
properties && typeof properties === "object" && "kind" in properties ? properties.kind : undefined,
).toEqual({
type: "string",
});
});
it("preserves raw JSON Schema required defaults and safe-integer bounds", () => {
const wire = toolWireSchema(
jsonTool({
+1 -1
View File
@@ -1000,7 +1000,7 @@ describe("Generate E2E Tests", () => {
);
});
describe.skipIf(!e2eApiKey("XAI_API_KEY"))("xAI Provider (grok-code-fast-1 via OpenAI Completions)", () => {
describe.skipIf(!e2eApiKey("XAI_API_KEY"))("xAI Provider (grok-code-fast-1 via OpenAI Responses)", () => {
const llm = getBundledModel("xai", "grok-code-fast-1");
it(
+158 -29
View File
@@ -1,6 +1,6 @@
import { describe, expect, spyOn, test } from "bun:test";
import { scheduler } from "node:timers/promises";
import { clearCustomApis } from "@oh-my-pi/pi-ai/api-registry";
import { clearCustomApis, registerCustomApi } from "@oh-my-pi/pi-ai/api-registry";
import * as AIError from "@oh-my-pi/pi-ai/error";
import { createMockModel, type MockContent, registerMockApi } from "@oh-my-pi/pi-ai/providers/mock";
import { complete, completeSimple, stream, streamSimple } from "@oh-my-pi/pi-ai/stream";
@@ -42,6 +42,11 @@ function nearDuplicateLoop(paragraphs: number): string {
return out.join("\n\n\n");
}
/** The exact 311-character cycle observed from Kiro gpt-5-6-sol on 2026-08-14.
* The persisted assistant message repeated this cycle 58 times before abort. */
const OBSERVED_KIRO_CYCLE =
"% shipped. 100% delivered. 100% verified. 100% validated. 100% approved. 100% accepted. 100% merged. 100% deployed. 100% live. 100% operational. 100% successful. 100% excellent. 100% perfect. 100% final. 100% absolute. 100% total. 100% whole. 100% full. 100% entire. 100% complete. 100% done. 100% finished. 100";
/** Genuinely distinct reasoning paragraphs — must never trip the detector. */
function distinctReasoning(): string {
return [
@@ -335,7 +340,7 @@ describe("thinking-loop guard (stream wrapper)", () => {
});
}
test("emits no observable thinking/text content before the error terminal", async () => {
test("drops the failed attempt from the terminal after a loop is detected", async () => {
registerMockApi();
try {
const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" });
@@ -344,9 +349,11 @@ describe("thinking-loop guard (stream wrapper)", () => {
const events = await collect(stream(mock.model, context()));
const terminal = events.at(-1);
expect(terminal?.type).toBe("error");
// The guard must not forward the looping thinking_end / done.
// A prefix can stream before detection, but the failed terminal must not
// carry replayable content or forward the normal completion boundary.
expect(events.some(e => e.type === "thinking_end")).toBe(false);
expect(events.some(e => e.type === "done")).toBe(false);
if (terminal?.type === "error") expect(terminal.error.content).toEqual([]);
} finally {
clearCustomApis();
}
@@ -367,6 +374,75 @@ describe("thinking-loop guard (stream wrapper)", () => {
}
});
test("terminates the observed long-cycle Kiro text runaway", async () => {
registerMockApi();
try {
const mock = createMockModel({ provider: "kiro", id: "gpt-5-6-sol" });
mock.push({ content: [`Healthy lead sentence. ${OBSERVED_KIRO_CYCLE.repeat(6)}`] });
const result = await stream(mock.model, context()).result();
expect(result.stopReason).toBe("error");
expect(result.content).toEqual([]);
expect(result.errorMessage).toContain(THINKING_LOOP_ERROR_MARKER);
expect(AIError.is(result.errorId, AIError.Flag.ThinkingLoop)).toBe(true);
} finally {
clearCustomApis();
}
});
test("detects the Kiro cycle across token-sized synthetic-provider deltas", async () => {
const model = {
api: "openai-completions",
provider: "synthetic",
id: "gpt-5-6-sol",
name: "GPT 5.6 Sol",
baseUrl: "https://unused.example.com",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 32_768,
} as Model<"openai-completions">;
const text = `Healthy lead sentence. ${OBSERVED_KIRO_CYCLE.repeat(6)}`;
const fetch = async (): Promise<Response> => {
const events: string[] = [];
for (let i = 0; i < text.length; i += 23) {
events.push(
JSON.stringify({
id: "cycle",
object: "chat.completion.chunk",
created: 0,
model: model.id,
choices: [{ index: 0, delta: { content: text.slice(i, i + 23) } }],
}),
);
}
events.push(
JSON.stringify({
id: "cycle",
object: "chat.completion.chunk",
created: 0,
model: model.id,
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
}),
"[DONE]",
);
return new Response(`${events.map(event => `data: ${event}`).join("\n\n")}\n\n`, {
headers: { "content-type": "text/event-stream" },
});
};
const events = await collect(streamSimple(model, context(), { apiKey: "test", fetch }));
const terminal = events.at(-1);
expect(events.some(event => event.type === "text_delta")).toBe(true);
expect(terminal?.type).toBe("error");
if (terminal?.type !== "error") throw new Error("expected loop error terminal");
expect(terminal.error.content).toEqual([]);
expect(AIError.is(terminal.error.errorId, AIError.Flag.ThinkingLoop)).toBe(true);
});
test("does not trip on a healthy gemini turn that reasons then answers", async () => {
registerMockApi();
try {
@@ -636,12 +712,12 @@ describe("GeminiHeaderRunDetector", () => {
});
});
describe("thinking-loop cook fallback (result path)", () => {
describe("thinking-loop retry budget (result path)", () => {
function loopResponse(): { content: MockContent[] } {
return { content: [{ type: "thinking", thinking: nearDuplicateLoop(12) }] };
}
test("completeSimple re-samples a loop then cooks through with the guard disabled", async () => {
test("completeSimple fails closed after three guarded attempts", async () => {
registerMockApi();
const waitSpy = spyOn(scheduler, "wait").mockResolvedValue(undefined);
try {
@@ -650,21 +726,18 @@ describe("thinking-loop cook fallback (result path)", () => {
const result = await completeSimple(mock.model, context());
// Three guarded attempts raise the stall; the fourth (guard disabled) cooks through.
expect(mock.calls).toHaveLength(4);
expect(result.stopReason).toBe("stop");
expect(result.content.some(block => block.type === "thinking")).toBe(true);
expect(result.errorMessage).toBeUndefined();
// First three dispatches are guarded; only the final cook pass disables it.
expect(mock.calls[0]?.options?.loopGuard?.enabled).toBeUndefined();
expect(mock.calls[3]?.options?.loopGuard?.enabled).toBe(false);
expect(mock.calls).toHaveLength(3);
expect(result.stopReason).toBe("error");
expect(result.content).toEqual([]);
expect(result.errorMessage).toContain(THINKING_LOOP_ERROR_MARKER);
expect(mock.calls.every(call => call.options?.loopGuard?.enabled !== false)).toBe(true);
} finally {
waitSpy.mockRestore();
clearCustomApis();
}
});
test("complete (non-simple) also cooks through after the abort budget", async () => {
test("complete (non-simple) also fails closed after three guarded attempts", async () => {
registerMockApi();
const waitSpy = spyOn(scheduler, "wait").mockResolvedValue(undefined);
try {
@@ -673,11 +746,11 @@ describe("thinking-loop cook fallback (result path)", () => {
const result = await complete(mock.model, context());
expect(mock.calls).toHaveLength(4);
expect(result.stopReason).toBe("stop");
expect(result.errorMessage).toBeUndefined();
expect(mock.calls[0]?.options?.loopGuard?.enabled).toBeUndefined();
expect(mock.calls[3]?.options?.loopGuard?.enabled).toBe(false);
expect(mock.calls).toHaveLength(3);
expect(result.stopReason).toBe("error");
expect(result.content).toEqual([]);
expect(result.errorMessage).toContain(THINKING_LOOP_ERROR_MARKER);
expect(mock.calls.every(call => call.options?.loopGuard?.enabled !== false)).toBe(true);
} finally {
waitSpy.mockRestore();
clearCustomApis();
@@ -706,23 +779,79 @@ describe("thinking-loop cook fallback (result path)", () => {
}
});
test("does not retry a contentful marker error (replay-unsafe output)", async () => {
test("a caller abort after the third guarded result supersedes the loop error", async () => {
registerMockApi();
const controller = new AbortController();
const waitSpy = spyOn(scheduler, "wait").mockResolvedValue(undefined);
try {
const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" });
mock.push({
content: ["Looping visible reasoning garbage."],
stopReason: "error",
errorMessage: `${THINKING_LOOP_ERROR_MARKER}: already streamed, non-retryable`,
let attempts = 0;
const mock = createMockModel({
provider: "openrouter",
id: "google/gemini-3.5-flash",
handler: () => {
if (++attempts === 3) controller.abort(new Error("cancelled after final result"));
return loopResponse();
},
});
const result = await completeSimple(mock.model, context());
await expect(completeSimple(mock.model, context(), { signal: controller.signal })).rejects.toThrow(
"cancelled after final result",
);
expect(mock.calls).toHaveLength(3);
} finally {
waitSpy.mockRestore();
clearCustomApis();
}
});
// Visible content already escaped: the marker error is returned as-is, never re-sampled.
expect(mock.calls).toHaveLength(1);
test("does not retry a contentful ThinkingLoop error (replay-unsafe output)", async () => {
const api = "contentful-loop-test";
let calls = 0;
const model = {
api,
provider: "test",
id: "test-model",
name: "Test model",
baseUrl: "test://",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 1_000,
maxTokens: 100,
} as unknown as Model<Api>;
registerCustomApi(api, () => {
calls++;
const inner = new AssistantMessageEventStream();
const error: AssistantMessage = {
role: "assistant",
content: [{ type: "text", text: "Looping visible reasoning garbage." }],
api,
provider: model.provider,
model: model.id,
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "error",
errorMessage: `${THINKING_LOOP_ERROR_MARKER}: already streamed, non-retryable`,
errorId: AIError.create(AIError.Flag.ThinkingLoop),
timestamp: Date.now(),
};
inner.push({ type: "error", reason: "error", error });
return inner;
});
const waitSpy = spyOn(scheduler, "wait").mockResolvedValue(undefined);
try {
const result = await completeSimple(model, context());
expect(calls).toBe(1);
expect(result.stopReason).toBe("error");
expect(result.errorMessage).toContain(THINKING_LOOP_ERROR_MARKER);
expect(result.content).toHaveLength(1);
expect(AIError.is(result.errorId, AIError.Flag.ThinkingLoop)).toBe(true);
} finally {
waitSpy.mockRestore();
clearCustomApis();
+224 -7
View File
@@ -4,6 +4,8 @@ import { umansUsageProvider } from "../src/usage/umans";
const DEFAULT_BASE_URL = "https://api.code.umans.ai";
const RESETS_AT = "2026-08-06T21:52:21.202174+00:00";
function umansPayload(overrides: Record<string, unknown> = {}): Record<string, unknown> {
return {
plan: { display_name: "Code Max" },
@@ -11,9 +13,16 @@ function umansPayload(overrides: Record<string, unknown> = {}): Record<string, u
requests: { limit: 200, hard_cap: 400, burst_pct: 1.0, window_seconds: 18000 },
concurrency: { limit: 4, hard_cap: 8, burst_pct: 1.0 },
},
window: {
started_at: "2026-08-06T16:52:21.202174+00:00",
resets_at: RESETS_AT,
remaining_minutes: 9,
},
usage: {
requests_in_window: 48,
remaining_requests: 152,
weighted_in_window: 96,
weighted_remaining_requests: 104,
concurrent_sessions: 1,
tokens_in: 1_200_000,
tokens_out: 340_000,
@@ -49,9 +58,8 @@ function fetchRecorder(
};
return fn as unknown as typeof fetch;
}
describe("umans usage provider", () => {
it("parses the rolling 5h request window into a UsageLimit with used/remaining/fraction", async () => {
it("splits requests into soft-cap (weighted) and burst-ceiling (raw) limits", async () => {
const report = await umansUsageProvider.fetchUsage(
{
provider: "umans",
@@ -60,20 +68,229 @@ describe("umans usage provider", () => {
{ fetch: fakeFetch(umansPayload()) },
);
expect(report).not.toBeNull();
const soft = report?.limits.find(l => l.id === "umans:requests:soft");
expect(soft).toBeDefined();
// Weighted "effective requests" are authoritative against the soft cap:
// 96 effective used of 200.
expect(soft?.amount.used).toBe(96);
expect(soft?.amount.limit).toBe(200);
expect(soft?.amount.remaining).toBe(104);
expect(soft?.amount.usedFraction).toBeCloseTo(0.48, 5);
expect(soft?.amount.remainingFraction).toBeCloseTo(0.52, 5);
expect(soft?.amount.unit).toBe("requests");
expect(soft?.status).toBe("ok");
// The rolling 5h window still exposes its absolute `resets_at` as an
// incremental countdown for the status line.
expect(soft?.window?.resetsAt).toBe(Date.parse(RESETS_AT));
expect(soft?.window?.resetLabel).toBe("tick");
expect(soft?.window?.durationMs).toBe(18000_000);
expect(soft?.window?.label).toBe("rolling 5h");
// Raw counts against the burst ceiling are a separate row.
const hard = report?.limits.find(l => l.id === "umans:requests:hard");
expect(hard).toBeDefined();
expect(hard?.amount.used).toBe(48);
expect(hard?.amount.limit).toBe(400);
expect(hard?.amount.usedFraction).toBeCloseTo(0.12, 5);
expect(hard?.status).toBe("ok");
});
it("falls back to the legacy raw row when weighted fields are absent", async () => {
const report = await umansUsageProvider.fetchUsage(
{
provider: "umans",
credential: { type: "api_key", apiKey: "sk-test" },
},
{
fetch: fakeFetch(
umansPayload({
window: undefined,
usage: {
requests_in_window: 48,
remaining_requests: 152,
concurrent_sessions: 1,
tokens_in: 0,
tokens_out: 0,
priority: { low: false },
},
}),
),
},
);
const requests = report?.limits.find(l => l.id === "umans:requests");
expect(requests).toBeDefined();
expect(report?.limits.some(l => l.id.startsWith("umans:requests:"))).toBe(false);
expect(requests?.amount.used).toBe(48);
expect(requests?.amount.limit).toBe(200);
expect(requests?.amount.remaining).toBe(152);
expect(requests?.amount.usedFraction).toBeCloseTo(0.24, 5);
expect(requests?.amount.remainingFraction).toBeCloseTo(0.76, 5);
expect(requests?.amount.unit).toBe("requests");
// Rolling window: no fabricated reset timestamp.
expect(requests?.window?.resetsAt).toBeUndefined();
expect(requests?.window?.durationMs).toBe(18000_000);
expect(requests?.window?.label).toBe("rolling 5h");
});
it("does not report exhausted when raw requests exceed the soft cap but weighted headroom remains (#7858)", async () => {
// Real payload from https://github.com/can1357/oh-my-pi/issues/7858:
// raw 838 exceeds the 500 soft cap (previously clamped to 1.0 → false
// exhausted), while weighted "effective requests" are 207/500 with 293
// remaining — the account continues normally. Raw traffic only reaches
// the burst ceiling (1000) before throttling applies.
const report = await umansUsageProvider.fetchUsage(
{
provider: "umans",
credential: { type: "api_key", apiKey: "sk-test" },
},
{
fetch: fakeFetch(
umansPayload({
limits: {
requests: { limit: 500, hard_cap: 1000, burst_pct: 1.0, window_seconds: 18000 },
concurrency: { limit: 4, hard_cap: 8, burst_pct: 1.0 },
},
usage: {
requests_in_window: 838,
remaining_requests: 0,
weighted_in_window: 207,
weighted_remaining_requests: 293,
concurrent_sessions: 0,
tokens_in: 3_557_477,
tokens_out: 723_550,
priority: { low: false, boxed_until: null, reason: null },
},
}),
),
},
);
const soft = report?.limits.find(l => l.id === "umans:requests:soft");
expect(soft).toBeDefined();
expect(soft?.amount.used).toBe(207);
expect(soft?.amount.remaining).toBe(293);
expect(soft?.amount.usedFraction).toBeCloseTo(0.414, 3);
expect(soft?.status).toBe("ok");
expect(soft?.window?.resetsAt).toBe(Date.parse(RESETS_AT));
const hard = report?.limits.find(l => l.id === "umans:requests:hard");
expect(hard).toBeDefined();
expect(hard?.amount.used).toBe(838);
expect(hard?.amount.limit).toBe(1000);
expect(hard?.amount.usedFraction).toBeCloseTo(0.838, 3);
expect(hard?.status).toBe("ok");
expect(report?.limits.some(l => l.status === "exhausted")).toBe(false);
});
it("collapses to a single weighted requests row that can exhaust when no burst ceiling is reported", async () => {
// Weighted counters present but `hard_cap` absent: without a burst
// ceiling there is no hard row to defer exhaustion to, so the weighted
// effective-request budget is the operative ceiling — the single row
// must be able to report `exhausted` or a spent account could never
// trigger the usage-aware fallback.
const report = await umansUsageProvider.fetchUsage(
{
provider: "umans",
credential: { type: "api_key", apiKey: "sk-test" },
},
{
fetch: fakeFetch(
umansPayload({
limits: {
requests: { limit: 200, window_seconds: 18000 },
concurrency: { limit: 4, hard_cap: 8, burst_pct: 1.0 },
},
usage: {
requests_in_window: 400,
remaining_requests: 0,
weighted_in_window: 200,
weighted_remaining_requests: 0,
concurrent_sessions: 0,
tokens_in: 0,
tokens_out: 0,
priority: { low: false },
},
}),
),
},
);
const requests = report?.limits.find(l => l.id === "umans:requests");
expect(requests).toBeDefined();
// No soft/hard split without a reported burst ceiling.
expect(report?.limits.some(l => l.id.startsWith("umans:requests:"))).toBe(false);
// Weighted effective requests are authoritative: raw 400 overshoots the
// 200 limit, but it is the weighted 200/200 that reports exhausted.
expect(requests?.amount.used).toBe(200);
expect(requests?.amount.limit).toBe(200);
expect(requests?.amount.usedFraction).toBe(1);
expect(requests?.status).toBe("exhausted");
});
it("keeps weighted headroom decisive when no burst ceiling is reported", async () => {
// Same #7858 shape (raw usage over the soft limit, weighted headroom
// remaining) but with no `hard_cap` in the payload: the weighted counter
// must still decide, so raw burst traffic cannot fabricate an exhausted
// state even when there is no hard row to buffer it.
const report = await umansUsageProvider.fetchUsage(
{
provider: "umans",
credential: { type: "api_key", apiKey: "sk-test" },
},
{
fetch: fakeFetch(
umansPayload({
limits: {
requests: { limit: 200, window_seconds: 18000 },
concurrency: { limit: 4, hard_cap: 8, burst_pct: 1.0 },
},
usage: {
requests_in_window: 300,
remaining_requests: 0,
weighted_in_window: 100,
weighted_remaining_requests: 100,
concurrent_sessions: 0,
tokens_in: 0,
tokens_out: 0,
priority: { low: false },
},
}),
),
},
);
const requests = report?.limits.find(l => l.id === "umans:requests");
expect(requests).toBeDefined();
expect(requests?.amount.used).toBe(100);
expect(requests?.amount.remaining).toBe(100);
expect(requests?.amount.usedFraction).toBeCloseTo(0.5, 5);
expect(requests?.status).toBe("ok");
expect(report?.limits.some(l => l.status === "exhausted")).toBe(false);
});
it("reserves exhausted for the burst ceiling and warns at the soft cap", async () => {
const report = await umansUsageProvider.fetchUsage(
{
provider: "umans",
credential: { type: "api_key", apiKey: "sk-test" },
},
{
fetch: fakeFetch(
umansPayload({
usage: {
requests_in_window: 1000,
remaining_requests: 0,
weighted_in_window: 500,
weighted_remaining_requests: 0,
concurrent_sessions: 0,
tokens_in: 0,
tokens_out: 0,
priority: { low: false },
},
}),
),
},
);
const soft = report?.limits.find(l => l.id === "umans:requests:soft");
expect(soft?.amount.usedFraction).toBe(1);
// Soft cap hit = burst headroom in use; warn, never exhaust.
expect(soft?.status).toBe("warning");
const hard = report?.limits.find(l => l.id === "umans:requests:hard");
expect(hard?.amount.usedFraction).toBe(1);
// Only the raw burst ceiling can exhaust (that's where 429s start).
expect(hard?.status).toBe("exhausted");
});
it("emits a concurrency limit from limits.concurrency", async () => {
const report = await umansUsageProvider.fetchUsage(
{
+46
View File
@@ -29,6 +29,52 @@ describe("xAI API login wiring", () => {
expect(getEnvApiKey("xai")).toBe("xai-env-key");
});
test("XAI_API_KEY alone does not mark SuperGrok as available", async () => {
const originalOauthToken = Bun.env.XAI_OAUTH_TOKEN;
Bun.env.XAI_API_KEY = "xai-env-key";
delete Bun.env.XAI_OAUTH_TOKEN;
const store = new SqliteAuthCredentialStore(new Database(":memory:"));
const storage = new AuthStorage(store);
await storage.reload();
try {
expect(storage.hasAuth("xai")).toBe(true);
expect(storage.hasAuth("xai-oauth")).toBe(false);
expect(storage.hasResolvableAuth("xai")).toBe(true);
expect(storage.hasResolvableAuth("xai-oauth")).toBe(true);
expect(getEnvApiKey("xai-oauth")).toBe("xai-env-key");
expect(storage.getCredentialOrigin("xai")).toEqual({ kind: "env", envVar: "XAI_API_KEY" });
expect(storage.getCredentialOrigin("xai-oauth")).toBeUndefined();
} finally {
if (originalOauthToken === undefined) {
delete Bun.env.XAI_OAUTH_TOKEN;
} else {
Bun.env.XAI_OAUTH_TOKEN = originalOauthToken;
}
store.close();
}
});
test("XAI_OAUTH_TOKEN marks SuperGrok available without a paid API key", async () => {
const originalOauthToken = Bun.env.XAI_OAUTH_TOKEN;
delete Bun.env.XAI_API_KEY;
Bun.env.XAI_OAUTH_TOKEN = "xai-oauth-env";
const store = new SqliteAuthCredentialStore(new Database(":memory:"));
const storage = new AuthStorage(store);
await storage.reload();
try {
expect(storage.hasAuth("xai")).toBe(false);
expect(storage.hasAuth("xai-oauth")).toBe(true);
expect(storage.getCredentialOrigin("xai-oauth")).toEqual({ kind: "env" });
} finally {
if (originalOauthToken === undefined) {
delete Bun.env.XAI_OAUTH_TOKEN;
} else {
Bun.env.XAI_OAUTH_TOKEN = originalOauthToken;
}
store.close();
}
});
test("AuthStorage.login('xai') validates against /models and stores the pasted key", async () => {
const fetchCalls: Array<{ url: string; init: RequestInit | undefined }> = [];
const fetchMock: FetchImpl = vi.fn(async (input: string | URL | Request, init?: RequestInit) => {
+156 -1
View File
@@ -1,6 +1,6 @@
import { describe, expect, test } from "bun:test";
import { buildParams } from "@oh-my-pi/pi-ai/providers/openai-responses";
import type { Context } from "@oh-my-pi/pi-ai/types";
import type { AssistantMessage, Context, Model } from "@oh-my-pi/pi-ai/types";
import { Effort } from "@oh-my-pi/pi-catalog/effort";
import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
@@ -38,6 +38,23 @@ describe("effort-dial-less reasoner encoding (regression)", () => {
expect(grokR.thinking).toBeUndefined();
});
test("paid xai/grok-code-fast-1 reasons but carries no thinking config", () => {
const grokCodeFast = getBundledModel("xai", "grok-code-fast-1");
if (!grokCodeFast) throw new Error("xai/grok-code-fast-1 must be in bundled models.json");
expect(grokCodeFast.api).toBe("openai-responses");
expect(grokCodeFast.reasoning).toBe(true);
expect(grokCodeFast.thinking).toBeUndefined();
expect(getSupportedEfforts(grokCodeFast)).toEqual([]);
});
test("paid xai/grok-4.3 keeps its effort dial", () => {
const grok43 = getBundledModel("xai", "grok-4.3");
if (!grok43) throw new Error("xai/grok-4.3 must be in bundled models.json");
expect(grok43.api).toBe("openai-responses");
expect(grok43.thinking).toBeDefined();
expect(getSupportedEfforts(grok43).length).toBeGreaterThan(0);
});
test("the no-dial encoding stays scoped to openai-responses*", () => {
const claude = getBundledModel("anthropic", "claude-sonnet-4-6");
if (!claude) throw new Error("anthropic/claude-sonnet-4-6 must be in bundled models.json");
@@ -57,6 +74,7 @@ describe("xAI OAuth Responses reasoning payload (regression)", () => {
const { params } = buildParams(grok45, singleUserContext, undefined, undefined);
expect(params.reasoning).toBeUndefined();
expect(params.include).toContain("reasoning.encrypted_content");
});
test("xai-oauth/grok-4.5 omits unsupported reasoning summary", () => {
@@ -66,5 +84,142 @@ describe("xAI OAuth Responses reasoning payload (regression)", () => {
const { params } = buildParams(grok45, singleUserContext, { reasoning: Effort.High }, undefined);
expect(params.reasoning).toEqual({ effort: "high" });
expect(params.include).toContain("reasoning.encrypted_content");
});
test("paid xai/grok-4.5 omits unsupported reasoning summary", () => {
const grok45 = getBundledModel<"openai-responses">("xai", "grok-4.5");
if (!grok45) throw new Error("xai/grok-4.5 must be in bundled models.json");
const { params } = buildParams(grok45, singleUserContext, { reasoning: Effort.High }, undefined);
expect(params.reasoning).toEqual({ effort: "high" });
expect(params.include).toContain("reasoning.encrypted_content");
});
test("paid xai/grok-4.5 requests encrypted reasoning content", () => {
const grok45 = getBundledModel<"openai-responses">("xai", "grok-4.5");
if (!grok45) throw new Error("xai/grok-4.5 must be in bundled models.json");
const { params } = buildParams(grok45, singleUserContext, { reasoning: Effort.High }, undefined);
expect(params.include).toContain("reasoning.encrypted_content");
});
test("paid xai/grok-4.5 omits presence_penalty on reasoning models", () => {
const grok45 = getBundledModel<"openai-responses">("xai", "grok-4.5");
if (!grok45) throw new Error("xai/grok-4.5 must be in bundled models.json");
const { params } = buildParams(
grok45,
singleUserContext,
{ reasoning: Effort.High, presencePenalty: 0.4, temperature: 0.2 },
undefined,
);
expect(params).not.toHaveProperty("presence_penalty");
expect(params.temperature).toBe(0.2);
});
test("paid xai/grok-2 omits presence_penalty on non-reasoning Responses models", () => {
const grok2 = getBundledModel<"openai-responses">("xai", "grok-2");
if (!grok2) throw new Error("xai/grok-2 must be in bundled models.json");
const { params } = buildParams(grok2, singleUserContext, { presencePenalty: 0.4, temperature: 0.2 }, undefined);
expect(params).not.toHaveProperty("presence_penalty");
expect(params.temperature).toBe(0.2);
});
test("paid xai/grok-4.5 clamps minimal reasoning effort to low", () => {
const grok45 = getBundledModel<"openai-responses">("xai", "grok-4.5");
if (!grok45) throw new Error("xai/grok-4.5 must be in bundled models.json");
const { params } = buildParams(grok45, singleUserContext, { reasoning: Effort.Minimal }, undefined);
expect(params.reasoning).toEqual({ effort: "low" });
});
test("xai-oauth/grok-4.5 clamps minimal reasoning effort to low", () => {
const grok45 = getBundledModel<"openai-responses">("xai-oauth", "grok-4.5");
if (!grok45) throw new Error("xai-oauth/grok-4.5 must be in bundled models.json");
const { params } = buildParams(grok45, singleUserContext, { reasoning: Effort.Minimal }, undefined);
expect(params.reasoning).toEqual({ effort: "low" });
});
test("xai-oauth/grok-4.5 replays encrypted reasoning on the next turn", () => {
const grok45 = getBundledModel<"openai-responses">("xai-oauth", "grok-4.5");
if (!grok45) throw new Error("xai-oauth/grok-4.5 must be in bundled models.json");
const { params } = buildParams(grok45, followUpContextWithEncryptedReasoning(grok45), undefined, undefined);
expect(params.include).toContain("reasoning.encrypted_content");
expect(findEncryptedReasoning(params.input)).toEqual({
type: "reasoning",
id: "rs_xai_next_turn",
encrypted_content: "enc_next_turn",
});
});
test("paid xai/grok-4.5 replays encrypted reasoning on the next turn", () => {
const grok45 = getBundledModel<"openai-responses">("xai", "grok-4.5");
if (!grok45) throw new Error("xai/grok-4.5 must be in bundled models.json");
const { params } = buildParams(grok45, followUpContextWithEncryptedReasoning(grok45), undefined, undefined);
expect(findEncryptedReasoning(params.input)).toEqual({
type: "reasoning",
id: "rs_xai_next_turn",
encrypted_content: "enc_next_turn",
});
});
});
function followUpContextWithEncryptedReasoning(model: Model<"openai-responses">): Context {
const assistant: AssistantMessage = {
role: "assistant",
content: [
{
type: "thinking",
thinking: "internal plan",
thinkingSignature: JSON.stringify({
type: "reasoning",
id: "rs_xai_next_turn",
encrypted_content: "enc_next_turn",
}),
},
{ type: "text", text: "done" },
],
api: "openai-responses",
provider: model.provider,
model: model.id,
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "stop",
timestamp: 1,
};
return {
messages: [
{ role: "user", content: "first", timestamp: 0 },
assistant,
{ role: "user", content: "continue", timestamp: 2 },
],
};
}
function findEncryptedReasoning(input: unknown): Record<string, unknown> | undefined {
if (!Array.isArray(input)) return undefined;
return input.find(item => {
if (!item || typeof item !== "object") return false;
const candidate = item as { type?: unknown; encrypted_content?: unknown };
return candidate.type === "reasoning" && typeof candidate.encrypted_content === "string";
}) as Record<string, unknown> | undefined;
}
+23
View File
@@ -1,6 +1,29 @@
# Changelog
## [Unreleased]
### Fixed
- Raised the GPT-5.6 Sol/Terra/Luna context window on the Codex transport (openai-codex) from 372K to 1M tokens: OpenAI enabled the 1M window for subscription Codex on 2026-08-16, but the Codex model registry still reports the stale 272,000, so discovery now floors these SKUs at 1,000,000 instead of trusting the reported value ([openai/codex#38917](https://github.com/openai/codex/issues/38917)).
## [17.3.5] - 2026-08-16
### Added
- Added support for GLM-5.3 on the z.AI provider, featuring a unified low/high/max reasoning-effort ladder across all hosts, mandatory thinking mode, 1M context, and default-model status for the z.AI provider.
### Changed
- Switched the paid xAI provider (xai / XAI_API_KEY) from Chat Completions to the OpenAI Responses API, aligning it with SuperGrok (xai-oauth) for prompt-cache affinity, reasoning-effort handling, and encrypted-reasoning replay.
- Changed the paid xAI (XAI_API_KEY) default model to grok-4.5.
- Changed the SuperGrok (xai-oauth) default model to grok-4.5.
- Improved reasoning continuity for xAI models by requesting and replaying encrypted reasoning content across multi-turn Responses API calls.
### Fixed
- Fixed Codex Daybreak Blue and Red model discovery reporting zero token prices, which incorrectly labeled the models as free in the model picker.
- Fixed Baseten's moonshotai/Kimi-K3 catalog metadata so its low/high/max thinking levels are available.
- Fixed opencode-go/deepseek-v4-flash Responses requests sending forced named tool_choice selectors that are rejected while thinking mode is active.
## [17.3.4] - 2026-08-14
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-catalog",
"version": "17.3.4",
"version": "17.3.5",
"description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+20 -1
View File
@@ -20,6 +20,7 @@ import { ANTIGRAVITY_PRIMARY_ENDPOINT, fetchAntigravityDiscoveryModels } from ".
import { buildGitLabDuoWorkflowFallbackModel } from "../src/discovery/gitlab-duo-workflow";
import { createModelManager } from "../src/model-manager";
import prevModelsJson from "../src/models.json" with { type: "json" };
import { resolveOpenAIDaybreakStandardCost } from "../src/openai-pricing";
import { toModelSpec } from "../src/provider-models/bundled-references";
import {
allowsUnauthenticatedCatalogDiscovery,
@@ -285,7 +286,7 @@ function applyCodexPricingFallback(models: readonly ModelSpec[]): ModelSpec[] {
return model;
}
const openAICost = openAIModels.get(model.id);
const openAICost = openAIModels.get(model.id) ?? resolveOpenAIDaybreakStandardCost(model.id);
if (!openAICost) {
return model;
}
@@ -555,6 +556,24 @@ async function generateModels() {
// Mythos 5). Deduped behind upstream entries; metadata is pinned in
// applyAnthropicCatalogPolicy.
allModels.push(...ANTHROPIC_CURATED_FALLBACK_MODELS);
// Seed GLM-5.3 on the z.AI provider. GLM-5.3 is live on the Anthropic and
// coding endpoints but not yet advertised in `/v1/models` (which still tops
// out at glm-5.2), so endpoint discovery misses it. The zai provider is not
// authoritative, so the seed survives regeneration; thinking metadata
// (low/high/max uniform ladder, mandatory reasoning, defaultLevel=max) is
// derived by rebakeModelThinking from the identity classifiers.
allModels.push({
id: "glm-5.3",
name: "GLM-5.3",
api: "anthropic-messages",
provider: "zai",
baseUrl: "https://api.z.ai/api/anthropic",
reasoning: true,
input: ["text"],
cost: { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
contextWindow: 1_000_000,
maxTokens: 131_072,
} as ModelSpec<"anthropic-messages">);
// Seed Meta's documented Muse model so first-run selection does not depend on
// credentials or live discovery.
allModels.push(...META_MUSE_STATIC_MODELS);
+20 -11
View File
@@ -21,6 +21,7 @@ import { resolveModelThinking } from "../src/model-thinking";
import { isOllamaCloudOutputCapped, OLLAMA_CLOUD_MAX_OUTPUT_TOKENS } from "../src/provider-models/ollama";
import {
ALIBABA_TOKEN_PLAN_STATIC_MODELS,
applyXaiResponsesThinkingPolicy,
OPENAI_GPT_56_LONG_CONTEXT_COSTS,
resolveWaferServerlessThinkingFormat,
} from "../src/provider-models/openai-compat";
@@ -144,7 +145,7 @@ const CODEX_GPT_5_4_PRIORITY_BY_VARIANT: Partial<Record<OpenAIVariant, number>>
nano: 2,
};
const CODEX_GPT_5_6_372K_MODEL_IDS: Record<string, true> = {
const CODEX_GPT_5_6_1M_MODEL_IDS: Record<string, true> = {
"gpt-5.6-luna": true,
"gpt-5.6-sol": true,
"gpt-5.6-terra": true,
@@ -353,6 +354,10 @@ export function applyOllamaCloudOutputCap(models: ModelSpec<Api>[]): void {
}
function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
if ((model.provider === "xai" || model.provider === "xai-oauth") && model.api === "openai-responses") {
const updated = applyXaiResponsesThinkingPolicy(model as ModelSpec<"openai-responses">);
model.compat = updated.compat;
}
const copilotLimits = model.provider === "github-copilot" ? COPILOT_GENERATED_LIMITS[model.id] : undefined;
if (copilotLimits) {
model.contextWindow = copilotLimits.contextWindow;
@@ -367,9 +372,13 @@ function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
model.omitMaxOutputTokens = true;
}
// GLM Coding Plan: GLM-5.2 is the selectable 1M served id; pin it so
// GLM Coding Plan: the selectable 1M-context served ids; pin them so
// endpoint discovery or older bundled fallbacks cannot regress to 200k.
if ((model.provider === "zai" || model.provider === "zhipu-coding-plan") && model.id === "glm-5.2") {
// GLM-5.3 succeeds GLM-5.2 with the same 1M context window.
if (
(model.provider === "zai" || model.provider === "zhipu-coding-plan") &&
(model.id === "glm-5.2" || model.id === "glm-5.3")
) {
model.contextWindow = 1_000_000;
model.maxTokens = 131_072;
}
@@ -425,7 +434,7 @@ function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
};
}
if (
model.api === "openai-completions" &&
(model.api === "openai-completions" || model.api === "openai-responses") &&
model.provider === "opencode-go" &&
(model.id === "deepseek-v4-flash" || model.id === "deepseek-v4-pro")
) {
@@ -527,12 +536,12 @@ function applyOpenAICatalogPolicy(model: ModelSpec<Api>, parsedModel: OpenAIMode
model.contextWindow = 272000;
}
}
// GPT-5.6 luna/sol/terra on the Codex transport: OpenAI's Codex model
// registry declares context_window = max_context_window = 372000, but Codex
// discovery omits `context_window` for these SKUs and falls back to
// DEFAULT_CONTEXT_WINDOW (272000, src/discovery/codex.ts), which regressed
// the bundled hard capacity (#5705). Pin the true 372K input window.
if (model.api === "openai-codex-responses" && CODEX_GPT_5_6_372K_MODEL_IDS[model.id]) {
model.contextWindow = 372000;
// GPT-5.6 luna/sol/terra on the Codex transport: OpenAI enabled a 1M-token
// window for subscription Codex (2026-08-16), but the Codex model registry
// still reports the stale 272000 (openai/codex#38917), so floor the bundled
// window at 1,000,000. Daybreak aliases are excluded — the registry actively
// reports their true window.
if (model.api === "openai-codex-responses" && CODEX_GPT_5_6_1M_MODEL_IDS[model.id]) {
model.contextWindow = Math.max(model.contextWindow ?? 0, 1_000_000);
}
}
+54 -11
View File
@@ -16,6 +16,7 @@ import {
isDeepseekModelIdOrName,
isGlm52ReasoningEffortModelId,
isGrokReasoningEffortCapable,
isGrokXHighEffortCapable,
isKimiK3ModelId,
isKimiK26ModelId,
isKimiModelId,
@@ -177,6 +178,22 @@ const MIMO_REASONING_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffortMap"]>
xhigh: "high",
};
/** Shared `minimal → low` clamp. xhigh-capable Grok keeps `xhigh` unmapped. */
const XAI_RESPONSES_MINIMAL_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffortMap"]> = {
minimal: "low",
};
/** Grok 4.5 / 4.3 / 3-mini: leftover `xhigh`/`max` clamp to `high`. */
const XAI_RESPONSES_CLAMPED_EFFORT_MAP: NonNullable<OpenAICompat["reasoningEffortMap"]> = {
minimal: "low",
xhigh: "high",
max: "high",
};
/** Wire effort remap for first-party xAI Responses. */
export function xaiResponsesReasoningEffortMap(modelId: string): NonNullable<OpenAICompat["reasoningEffortMap"]> {
return isGrokXHighEffortCapable(modelId) ? XAI_RESPONSES_MINIMAL_EFFORT_MAP : XAI_RESPONSES_CLAMPED_EFFORT_MAP;
}
function mergeModelReasoningEffortMap(
compat: ResolvedOpenAISharedCompat,
modelId: string,
@@ -471,6 +488,8 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
// OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit
// temperature/top_p/… with a 400 on every serving host (#5606).
supportsSamplingParams: !isOpenAISamplingRestrictedModelId(spec.id),
// xAI reasoning models 400 on presence/frequency penalties and stop.
supportsPenaltyAndStopParams: !(isGrok && Boolean(spec.reasoning)),
reasoningEffortMap: {},
supportsUsageInStreaming: !isCerebras,
// Kimi (including via OpenRouter and Fireworks router-form IDs such as
@@ -684,34 +703,46 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
const isLocalServingBackend =
(!PROXY_OPENAI_COMPAT_PROVIDERS.has(spec.provider) && LOCAL_OPENAI_COMPAT_PROVIDERS.has(spec.provider)) ||
hasLocalLoopbackBaseUrl(baseUrl);
const isXaiHost = modelMatchesHost({ provider: spec.provider, baseUrl }, "xai");
const compat: ResolvedOpenAIResponsesCompat = {
supportsDeveloperRole: isAzure || isOpenAIUrl || hostMatchesUrl(baseUrl, "githubCopilot"),
supportsStrictMode: isAzure || detectStrictModeSupport(spec.provider, baseUrl),
supportsReasoningEffort: spec.provider !== "xai-oauth" || isGrokReasoningEffortCapable(id),
// Paid `xai` and SuperGrok `xai-oauth` share api.x.ai `/v1/responses`.
// Only the Grok effort-capable allowlist accepts `reasoning.effort`;
// other reasoners (grok-build, grok-code-fast-1, …) 400 if it is sent.
supportsReasoningEffort: !isXaiHost || isGrokReasoningEffortCapable(id),
supportsLongPromptCacheRetention: isOpenAIUrl,
supportsPromptCacheBreakpoints,
promptCacheBreakpointTtl: supportsPromptCacheBreakpoints ? "30m" : undefined,
// Azure OpenAI and GitHub Copilot Responses paths require tool results
// to strictly match prior tool calls when building Responses inputs.
strictResponsesPairing: isAzure || spec.provider === "github-copilot",
// GitHub Copilot and xAI OAuth reject `detail: "original"` (400 / 422).
// Every other host preserves native-resolution frames (snapcompact relies
// on `original`). Detect Copilot by provider id or base-URL host so a
// model pointed at the Copilot host under a different provider id still
// clamps; xai-oauth is provider-id only (same host family as paid `xai`).
// GitHub Copilot and first-party xAI `/v1/responses` reject
// `detail: "original"` (400 / 422). Every other host preserves
// native-resolution frames (snapcompact relies on `original`). Detect
// Copilot by provider id or base-URL host so a model pointed at the
// Copilot host under a different provider id still clamps.
supportsImageDetailOriginal:
spec.provider !== "xai-oauth" && !modelMatchesHost({ provider: spec.provider, baseUrl }, "githubCopilot"),
reasoningEffortMap: {},
!isXaiHost && !modelMatchesHost({ provider: spec.provider, baseUrl }, "githubCopilot"),
// api.x.ai rejects `reasoning.summary` (SuperGrok and paid key alike).
supportsReasoningSummary: !isXaiHost,
reasoningEffortMap: isXaiHost ? { ...xaiResponsesReasoningEffortMap(id) } : {},
supportsReasoningParams: true,
// OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit
// temperature/top_p/… with a 400 on every serving host (#5606).
supportsSamplingParams: !isOpenAISamplingRestrictedModelId(id),
// xAI `/v1/responses` rejects presence/frequency penalties for every
// model, not only reasoners (https://docs.x.ai/developers/rest-api-reference/inference/chat).
supportsPenaltyAndStopParams: !isXaiHost,
thinkingFormat,
reasoningDisableMode: resolveReasoningDisableMode(thinkingFormat),
omitReasoningEffort: false,
includeEncryptedReasoning: spec.provider !== "xai-oauth",
filterReasoningHistory: spec.provider === "xai-oauth" || (isOpenRouter && isAnthropicModel),
// Ask xAI `/v1/responses` for `reasoning.encrypted_content` and replay
// those items on later turns. OpenRouter Anthropic still filters
// reasoning wrappers independently.
includeEncryptedReasoning: true,
filterReasoningHistory: isOpenRouter && isAnthropicModel,
disableReasoningOnForcedToolChoice: isKimiModel,
disableReasoningOnToolChoice: isDeepseekFamily && reasoningCapable && !isOpenRouter,
supportsToolChoice: true,
@@ -752,13 +783,24 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
MINIMAX_PROVIDER_OR_ID_PATTERN.test(spec.provider) || (id ? MINIMAX_PROVIDER_OR_ID_PATTERN.test(id) : false),
emptyLengthFinishIsContextError: spec.provider === "ollama",
usesOpenAIToolCallIdLimit: spec.provider === "openai",
promptCacheSessionHeader: spec.provider === "xai-oauth" ? "x-grok-conv-id" : undefined,
promptCacheSessionHeader: isXaiHost ? "x-grok-conv-id" : undefined,
streamFirstEventTimeoutMs: isLocalServingBackend ? 0 : spec.compat?.streamFirstEventTimeoutMs,
streamIdleTimeoutMs: isLocalServingBackend
? LOCAL_OPENAI_COMPAT_STREAM_IDLE_TIMEOUT_MS
: spec.compat?.streamIdleTimeoutMs,
};
applyCompatOverrides(compat, spec.compat);
if (isXaiHost) {
const canonical = xaiResponsesReasoningEffortMap(id);
compat.reasoningEffortMap = { ...compat.reasoningEffortMap, ...canonical };
// xhigh-capable Grok advertises unmapped `xhigh`; drop a stale clamp
// from previous snapshots so 4.6 / 16-agent mode is not rewritten to `high`.
for (const key of ["xhigh", "max"] as const) {
if (!(key in canonical)) {
delete compat.reasoningEffortMap[key];
}
}
}
if (spec.compat?.reasoningDisableMode === undefined) {
compat.reasoningDisableMode = resolveReasoningDisableMode(compat.thinkingFormat);
}
@@ -776,6 +818,7 @@ function pickResponsesOnly(compat: ResolvedOpenAIResponsesCompat): ResponsesOnly
strictResponsesPairing: compat.strictResponsesPairing,
supportsImageDetailOriginal: compat.supportsImageDetailOriginal,
supportsObfuscationOptOut: compat.supportsObfuscationOptOut,
supportsReasoningSummary: compat.supportsReasoningSummary,
isVercelGatewayHost: compat.isVercelGatewayHost,
} satisfies ResponsesOnlyCompat;
}
+21 -9
View File
@@ -1,5 +1,6 @@
import { type } from "@oh-my-pi/omptype";
import { parseKnownModel, semverEqual } from "../identity/classify";
import { resolveOpenAIDaybreakStandardCost } from "../openai-pricing";
import type { FetchImpl, ModelSpec } from "../types";
import { discoveryFetch } from "../utils";
import { CODEX_BASE_URL, CODEX_CLIENT_VERSION, OPENAI_HEADER_VALUES, OPENAI_HEADERS } from "../wire/codex";
@@ -8,13 +9,19 @@ const DEFAULT_MODEL_LIST_PATHS = ["/codex/models", "/models"] as const;
const DEFAULT_CONTEXT_WINDOW = 272_000;
const DEFAULT_MAX_TOKENS = 128_000;
/**
* GPT-5.6 luna/sol/terra hard context capacity. Codex discovery omits
* `context_window` for these SKUs, so the generic {@link DEFAULT_CONTEXT_WINDOW}
* (272000) would understate the real window — OpenAI's Codex model registry
* declares context_window = max_context_window = 372000 (#5705). Used as the
* fallback only when upstream reports no value.
* Fallback for GPT-5.6-family SKUs when upstream omits `context_window`: the
* generic {@link DEFAULT_CONTEXT_WINDOW} (272000) understates the registry's
* former 372000 hard capacity (#5705).
*/
const GPT_5_6_CONTEXT_WINDOW = 372_000;
/**
* OpenAI enabled a 1M-token window for subscription Codex on GPT-5.6
* luna/sol/terra (2026-08-16), but the Codex model registry still reports the
* stale 272000 — so the reported value must be floored, not just defaulted
* (openai/codex#38917; Codex CLI override `model_context_window = 1000000`).
*/
const GPT_5_6_1M_CONTEXT_WINDOW = 1_000_000;
const CODEX_GPT_5_6_1M_SLUGS: ReadonlySet<string> = new Set(["gpt-5.6-luna", "gpt-5.6-sol", "gpt-5.6-terra"]);
const CODEX_REMOTE_COMPACTION = {
enabled: true,
api: "openai-codex-responses",
@@ -223,20 +230,25 @@ function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCo
}
const name = toNonEmptyString(payload.display_name) ?? slug;
// Codex discovery omits `context_window` for GPT-5.6 luna/sol/terra; the
// generic 272000 fallback understates their real 372000 window (#5705).
// Codex discovery historically omitted `context_window` for GPT-5.6-family
// SKUs (#5705); luna/sol/terra additionally floor the reported value because
// the registry still declares the pre-1M 272000 window.
const parsed = parseKnownModel(slug);
const fallbackContextWindow =
parsed.family === "openai" && semverEqual(parsed.version, "5.6")
? GPT_5_6_CONTEXT_WINDOW
: DEFAULT_CONTEXT_WINDOW;
const contextWindow = toPositiveInt(payload.context_window) ?? fallbackContextWindow;
const reportedContextWindow = toPositiveInt(payload.context_window) ?? fallbackContextWindow;
const contextWindow = CODEX_GPT_5_6_1M_SLUGS.has(slug)
? Math.max(reportedContextWindow, GPT_5_6_1M_CONTEXT_WINDOW)
: reportedContextWindow;
const maxTokens = Math.min(DEFAULT_MAX_TOKENS, contextWindow);
const reasoning = supportsReasoning(payload.default_reasoning_level, payload.supported_reasoning_levels);
const input = normalizeInputModalities(payload.input_modalities);
const preferWebsockets = toBoolean(payload.prefer_websockets) === true;
const useResponsesLite = toBoolean(payload.use_responses_lite) === true;
const priority = toFiniteNumber(payload.priority) ?? Number.MAX_SAFE_INTEGER;
const daybreakCost = resolveOpenAIDaybreakStandardCost(slug);
return {
priority,
@@ -248,7 +260,7 @@ function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCo
baseUrl,
reasoning,
input,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
cost: daybreakCost ? { ...daybreakCost } : { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
remoteCompaction: CODEX_REMOTE_COMPACTION,
contextWindow,
maxTokens,
+1 -1
View File
@@ -47,7 +47,7 @@ export const KNOWN_HOSTS = {
},
umans: { providers: ["umans"], urlMarkers: ["api.code.umans.ai"] },
xiaomi: { providers: ["xiaomi"], providerPrefixes: ["xiaomi-token-plan-"], urlMarkers: ["xiaomimimo.com"] },
xai: { providers: ["xai"], urlMarkers: ["api.x.ai"] },
xai: { providers: ["xai", "xai-oauth"], urlMarkers: ["api.x.ai"] },
mistral: { providers: ["mistral"], urlMarkers: ["mistral.ai"] },
together: { providers: ["together"], urlMarkers: ["api.together.xyz"] },
baseten: { providers: ["baseten"], urlMarkers: ["baseten.co"] },
+47 -1
View File
@@ -110,7 +110,13 @@ export const isGrokModelId = memo((modelId: string): boolean => {
return /(?:^|[./_-])grok(?:[-.]|$)/i.test(modelId);
});
const GROK_EFFORT_CAPABLE_PREFIXES = ["grok-3-mini", "grok-4.20-multi-agent", "grok-4.3", "grok-4.5"] as const;
const GROK_EFFORT_CAPABLE_PREFIXES = [
"grok-3-mini",
"grok-4.20-multi-agent",
"grok-4.3",
"grok-4.5",
"grok-4.6",
] as const;
/**
* Grok SKUs that expose the wire `reasoning.effort` dial. Other Grok reasoners
@@ -123,6 +129,27 @@ export const isGrokReasoningEffortCapable = memo((modelId: string): boolean => {
return GROK_EFFORT_CAPABLE_PREFIXES.some(prefix => bare.startsWith(prefix));
});
/**
* `grok-4.20-multi-agent*` uses `reasoning.effort` to pick agent count
* (`xhigh` is the 16-agent mode). Other first-party Grok effort SKUs stay on
* `low|medium|high` unless {@link isGrokXHighEffortCapable} (currently
* `grok-4.6*` plus multi-agent).
* https://docs.x.ai/developers/model-capabilities/text/reasoning
*/
export const isGrokMultiAgentModelId = memo((modelId: string): boolean => {
return bareModelId(modelId).trim().toLowerCase().startsWith("grok-4.20-multi-agent");
});
/**
* First-party Grok SKUs whose Responses wire accepts `reasoning.effort: "xhigh"`.
* `grok-4.6*` documents xhigh as a reasoning depth; multi-agent uses it as
* 16-agent mode. `grok-4.5` / `grok-4.3` / `grok-3-mini` do not.
*/
export const isGrokXHighEffortCapable = memo((modelId: string): boolean => {
if (isGrokMultiAgentModelId(modelId)) return true;
return bareModelId(modelId).trim().toLowerCase().startsWith("grok-4.6");
});
/**
* MiniMax M2-generation family (M2, M2.1, M2.5, M2.7, including `-highspeed`/
* `-lightning`/`-her`/`-turbo` variants, dotless aliases like `minimax-m21`,
@@ -251,6 +278,25 @@ export const isGlm52ReasoningEffortModelId = memo((modelId: string): boolean =>
return semverGte(glm.version, "5.2");
});
/**
* GLM-5.3+ coding SKUs. Unlike GLM-5.2 (whose reasoning_effort dialect is
* host-specific), GLM-5.3+ exposes a uniform wire-exact `low`/`high`/`max`
* ladder on every host, and thinking can no longer be disabled —
* `thinking.type` must always be `enabled`. Matching the family keeps future
* bumps (`glm-5.4`, `glm-6`, …) covered while excluding the vision (`…v`)
* shape and the non-reasoning `-flash`/`-flashx`/`-preview` variants.
*/
export const isGlm53ReasoningEffortModelId = memo((modelId: string): boolean => {
const glm = parseGlmModel(bareModelId(modelId));
if (!glm || glm.vision) {
return false;
}
if (glm.variant !== "base" && glm.variant !== "air" && glm.variant !== "turbo") {
return false;
}
return semverGte(glm.version, "5.3");
});
/** GLM vision SKUs — the `v` that attaches to the version (`glm-4v`, `glm-4.5v`). */
export const isGlmVisionModelId = memo((modelId: string): boolean => {
return parseGlmModel(bareModelId(modelId))?.vision === true;
+20 -2
View File
@@ -26,6 +26,8 @@ import {
isDeepseekModelIdOrName,
isDeepseekV4FlashModelId,
isGlm52ReasoningEffortModelId,
isGlm53ReasoningEffortModelId,
isGrokXHighEffortCapable,
isKimiK3ModelId,
isMimoModelIdOrName,
isMinimaxM2FamilyModelId,
@@ -178,7 +180,8 @@ function fillThinkingWireDefaults<TApi extends Api>(
(spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") &&
supportsAdaptiveThinkingDisplay(spec.id);
const needsRequiresEffort = thinking.requiresEffort === undefined && impliesMandatoryReasoning(parsed, spec.id);
const needsDefaultLevel = thinking.defaultLevel === undefined && isKimiK3ModelId(spec.id);
const needsDefaultLevel =
thinking.defaultLevel === undefined && (isKimiK3ModelId(spec.id) || isGlm53ReasoningEffortModelId(spec.id));
if (!effortsChanged && !shouldReplaceEffortMap && !needsDisplay && !needsRequiresEffort && !needsDefaultLevel) {
return thinking;
}
@@ -216,7 +219,7 @@ export function deriveThinking<TApi extends Api>(spec: ModelSpec<TApi>, compat:
mode: inferThinkingControlMode(spec, parsed),
efforts,
};
if (isKimiK3ModelId(spec.id)) {
if (isKimiK3ModelId(spec.id) || isGlm53ReasoningEffortModelId(spec.id)) {
config.defaultLevel = Effort.Max;
}
const effortMap = inferEffortMap(spec, compat, config.mode, config.efforts);
@@ -312,6 +315,13 @@ function getModelDefinedEfforts<TApi extends Api>(
spec: ModelSpec<TApi>,
compat: CompatOf<TApi>,
): readonly Effort[] | undefined {
if (isGlm53ReasoningEffortModelId(spec.id)) {
// GLM-5.3+ exposes a uniform wire-exact low/high/max ladder on every
// host — unlike GLM-5.2, whose reasoning_effort dialect is
// host-specific. Thinking can no longer be disabled (handled by
// impliesMandatoryReasoning), and the default effort is `max`.
return LOW_HIGH_MAX_REASONING_EFFORTS;
}
if (isGlm52ReasoningEffortModelId(spec.id)) {
// GLM-5.2's reasoning_effort dialect is host-specific (verified against
// live endpoints):
@@ -395,6 +405,11 @@ function getModelDefinedEfforts<TApi extends Api>(
// Baseten's gpt-oss router mirrors its GLM route: high/max only.
return HIGH_MAX_REASONING_EFFORTS;
}
// First-party Grok: `grok-4.6*` and `grok-4.20-multi-agent*` advertise
// `xhigh`. Other effort-capable SKUs stay on `minimal/low/medium/high`.
if (modelMatchesHost({ provider: spec.provider, baseUrl: spec.baseUrl ?? "" }, "xai")) {
return isGrokXHighEffortCapable(spec.id) ? DEFAULT_REASONING_EFFORTS_WITH_XHIGH : DEFAULT_REASONING_EFFORTS;
}
return isOpenAICompatReasoningApi(spec.api) &&
(isMinimaxM2FamilyModelId(spec.id) ||
isOpenAIGptOssModelId(spec.id) ||
@@ -579,6 +594,9 @@ function impliesMandatoryReasoning(parsed: ParsedModel, modelId: string): boolea
if (parsed.kind === "pro" && semverGte(parsed.version, "2.5")) return true;
}
if (isKimiK3ModelId(modelId)) return true;
// GLM-5.3+ no longer supports disabling thinking — thinking.type must
// always be "enabled". Floor thinking-off requests to the lowest effort.
if (isGlm53ReasoningEffortModelId(modelId)) return true;
if (isMinimaxM2FamilyModelId(modelId)) return true;
if (OPENAI_O_SERIES_RE.test(bareModelId(modelId))) return true;
return findThinkingVariantToken(modelId) !== undefined;
File diff suppressed because it is too large Load Diff
+29
View File
@@ -0,0 +1,29 @@
import type { TokenCost } from "./types";
/** Standard GPT-5.6 Sol rates used by the Daybreak Blue aliases. */
export const OPENAI_GPT_56_SOL_STANDARD_COST = {
input: 5,
output: 30,
cacheRead: 0.5,
cacheWrite: 6.25,
} as const satisfies TokenCost;
/** Standard GPT-5.6 Cyber rates used by the Daybreak Red aliases. */
export const OPENAI_GPT_56_CYBER_STANDARD_COST = {
input: 12.5,
output: 75,
cacheRead: 1.25,
cacheWrite: 15.625,
} as const satisfies TokenCost;
/** Resolve standard rates for Codex-prefixed Daybreak aliases. */
export function resolveOpenAIDaybreakStandardCost(modelId: string): TokenCost | undefined {
switch (modelId) {
case "gpt-daybreak-blue-latest":
return OPENAI_GPT_56_SOL_STANDARD_COST;
case "gpt-daybreak-red-latest":
return OPENAI_GPT_56_CYBER_STANDARD_COST;
default:
return undefined;
}
}
@@ -478,13 +478,13 @@ export const CATALOG_PROVIDERS = [
},
{
id: "xai",
defaultModel: "grok-4-fast-non-reasoning",
defaultModel: "grok-4.5",
envVars: ["XAI_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => xaiModelManagerOptions(config),
},
{
id: "xai-oauth",
defaultModel: "grok-4.3",
defaultModel: "grok-4.5",
envVars: ["XAI_OAUTH_TOKEN", "XAI_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => xaiOAuthModelManagerOptions(config),
catalogDiscovery: {
@@ -522,7 +522,7 @@ export const CATALOG_PROVIDERS = [
},
{
id: "zai",
defaultModel: "glm-5.2",
defaultModel: "glm-5.3",
envVars: ["ZAI_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => zaiModelManagerOptions(config),
catalogDiscovery: { label: "zAI" },
@@ -1,5 +1,6 @@
import { USER_AGENT } from "@oh-my-pi/pi-utils";
import * as logger from "@oh-my-pi/pi-utils/logger";
import { xaiResponsesReasoningEffortMap } from "../compat/openai";
import {
DEFAULT_OPENAI_COMPATIBLE_DISCOVERY_TIMEOUT_MS,
fetchOpenAICompatibleModels,
@@ -20,7 +21,17 @@ import {
import { resolveModelReference } from "../identity/reference";
import type { ModelManagerOptions } from "../model-manager";
import { type GeneratedProvider, getBundledModels } from "../models";
import type { Api, FetchImpl, Model, ModelSpec, OpenAICompat, Provider, ThinkingConfig } from "../types";
import { OPENAI_GPT_56_CYBER_STANDARD_COST, OPENAI_GPT_56_SOL_STANDARD_COST } from "../openai-pricing";
import type {
Api,
FetchImpl,
LongContextTokenCost,
Model,
ModelSpec,
OpenAICompat,
Provider,
ThinkingConfig,
} from "../types";
import { discoveryFetch, isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils";
import { ALIBABA_TOKEN_PLAN_BASE_URL, parseAlibabaTokenPlanCredential } from "../wire/alibaba-token-plan";
import { coreWeaveProjectHeaders } from "../wire/coreweave";
@@ -869,6 +880,7 @@ export function umansModelManagerOptions(config?: UmansModelManagerConfig): Mode
// ---------------------------------------------------------------------------
const OPENAI_API_BASE_URL = "https://api.openai.com/v1";
/** GPT-5.6 rates applied when a first-party request exceeds 272K input tokens. */
export const OPENAI_GPT_56_LONG_CONTEXT_COSTS = {
luna: {
inputThreshold: 272_000,
@@ -891,20 +903,7 @@ export const OPENAI_GPT_56_LONG_CONTEXT_COSTS = {
cacheRead: 0.4,
cacheWrite: 5,
},
} as const;
const OPENAI_GPT_56_SOL_STANDARD_COST = {
input: 5,
output: 30,
cacheRead: 0.5,
cacheWrite: 6.25,
longContext: OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol,
} as const;
const OPENAI_GPT_56_CYBER_STANDARD_COST = {
input: 12.5,
output: 75,
cacheRead: 1.25,
cacheWrite: 15.625,
} as const;
} as const satisfies Readonly<Record<"luna" | "sol" | "terra", LongContextTokenCost>>;
export interface OpenAIModelManagerConfig {
apiKey?: string;
@@ -938,7 +937,10 @@ export const OPENAI_DAYBREAK_CURATED_FALLBACK_MODELS: readonly ModelSpec<"openai
baseUrl: OPENAI_API_BASE_URL,
reasoning: true,
input: ["text", "image"],
cost: OPENAI_GPT_56_SOL_STANDARD_COST,
cost: {
...OPENAI_GPT_56_SOL_STANDARD_COST,
longContext: OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol,
},
contextWindow: 1_050_000,
maxTokens: 128_000,
},
@@ -1257,8 +1259,23 @@ export interface XaiModelManagerConfig {
fetch?: FetchImpl;
}
export function xaiModelManagerOptions(config?: XaiModelManagerConfig): ModelManagerOptions<"openai-completions"> {
return createSimpleOpenAICompletionsOptions("xai", "https://api.x.ai/v1", config);
export function xaiModelManagerOptions(config?: XaiModelManagerConfig): ModelManagerOptions<"openai-responses"> {
return {
...createOpenAICompatibleModelManagerOptions({
api: "openai-responses",
providerId: "xai",
defaultBaseUrl: "https://api.x.ai/v1",
config,
requireApiKey: true,
mapModel: mapWithBundledReference,
}),
// Completions → Responses migration: a fresh authoritative cache written
// by the old resolver stores `api: "openai-completions"` for these ids.
// Without a drop list, `online-if-uncached` skips the network and
// `mergeDynamicModel` lets the cached api win over the new static
// Responses entries until TTL expiry.
dropCachedModelIdsOnStaticMismatch: getBundledModels("xai").map(model => model.id),
};
}
export interface XaiOAuthModelManagerConfig {
@@ -1316,6 +1333,7 @@ export const XAI_OAUTH_CURATED_MODELS: readonly XAICuratedModel[] = [
},
{ id: "grok-4.3", contextWindow: 1_000_000, name: "Grok 4.3", input: ["text", "image"] },
{ id: "grok-4.5", contextWindow: 500_000, name: "Grok 4.5", input: ["text", "image"] },
{ id: "grok-4.6", contextWindow: 500_000, name: "Grok 4.6", input: ["text", "image"] },
// grok-4.20-multi-agent-0309 is text-only per the bundled catalog; omit `input` for the default.
{ id: "grok-4.20-multi-agent-0309", contextWindow: 2_000_000, name: "Grok 4.20 (Multi-Agent)" },
{
@@ -1352,21 +1370,47 @@ const XAI_NON_CHAT_PREFIXES = ["grok-imagine-", "grok-stt-", "grok-voice-"] as c
function withXaiOAuthCompatDefaults(model: ModelSpec<"openai-responses">): ModelSpec<"openai-responses"> {
const compat = {
...(model.compat ?? {}),
includeEncryptedReasoning: model.compat?.includeEncryptedReasoning ?? false,
filterReasoningHistory: model.compat?.filterReasoningHistory ?? true,
includeEncryptedReasoning: model.compat?.includeEncryptedReasoning ?? true,
filterReasoningHistory: model.compat?.filterReasoningHistory ?? false,
supportsImageDetailOriginal: model.compat?.supportsImageDetailOriginal ?? false,
omitReasoningEffort: model.compat?.omitReasoningEffort ?? !isGrokReasoningEffortCapable(model.id),
};
return { ...model, compat };
}
// Hermes-agent parity: only the `minimal -> low` clamp is applied (see
// hermes-agent/agent/transports/codex.py:92 `_effort_clamp = {"minimal":
// "low"}`). Hermes sends `xhigh` to xAI verbatim and we match that contract
// — let xAI decide if the level is valid for the specific Grok model.
// `resolveModelThinking` folds this into `model.thinking.effortMap`, downstream
// of the omitReasoningEffort gate in pi-ai's stream.ts.
const XAI_REASONING_EFFORT_MAP = { minimal: "low" } as const;
// Hermes-agent parity for `minimal -> low` (see hermes-agent/agent/transports/
// codex.py:92). Multi-agent Grok keeps `xhigh` unmapped (agent-count mode);
// other first-party SKUs clamp leftover `xhigh`/`max` to `high`.
// `resolveModelThinking` folds this into `model.thinking.effortMap`.
/**
* Bake first-party xAI Responses effort-dial metadata onto a catalog spec.
*
* models.dev marks many Grok SKUs as reasoners and the thinking rebake would
* otherwise emit a default `minimal/low/medium/high` dial. api.x.ai only
* accepts `reasoning.effort` for {@link isGrokReasoningEffortCapable} ids —
* off-allowlist reasoners (`grok-code-fast-1`, `grok-build-0.1`,
* `grok-4.20-0309-reasoning`, …) 400 if the param is sent. SuperGrok
* (`xai-oauth`) already curates this via {@link mergeCuratedIntoModel}; paid
* `xai` rows come from stencil.so and need the same wire facts in the exported
* `models.json` so direct catalog readers do not present an unsupported dial.
*
* Explicit `compat.supportsReasoningEffort` / `omitReasoningEffort` win.
*/
export function applyXaiResponsesThinkingPolicy(model: ModelSpec<"openai-responses">): ModelSpec<"openai-responses"> {
const effortCapable = model.compat?.supportsReasoningEffort ?? isGrokReasoningEffortCapable(model.id);
const compat = {
...(model.compat ?? {}),
supportsReasoningEffort: effortCapable,
omitReasoningEffort: model.compat?.omitReasoningEffort ?? !effortCapable,
};
if (effortCapable) {
compat.reasoningEffortMap = { ...xaiResponsesReasoningEffortMap(model.id) };
} else {
delete compat.reasoningEffortMap;
}
return { ...model, compat };
}
// xai-oauth's /v1/models exposes no per-request output limit on the OAuth
// (Grok Build / SuperGrok) surface, so the curated catalog owns `maxTokens`
@@ -1381,9 +1425,9 @@ const XAI_REASONING_EFFORT_MAP = { minimal: "low" } as const;
// reasoning metadata and fetchOpenAICompatibleModels defaults reasoning to
// false). Caller supplies a `base` Model (either a freshly synthesised seed
// or a dynamic-fetched entry); the helper layers curated fields on top.
// The `minimal -> low` effort clamp (XAI_REASONING_EFFORT_MAP) is always
// merged in so dynamic-fetched models — which arrive without curated
// compat keys — still get the clamp applyResponsesReasoningParams expects.
// The effort remap from {@link xaiResponsesReasoningEffortMap} is merged
// only onto effort-capable rows. Off-allowlist reasoners omit the wire
// param, so a map on those specs is dead weight.
// The effort-dial pair (`supportsReasoningEffort`/`omitReasoningEffort`) is
// authoritative: a stale flag on `base` (previous snapshot or dynamic fetch)
// must not outlive an allowlist change in identity/family.ts.
@@ -1394,13 +1438,17 @@ function mergeCuratedIntoModel(
const effortCapable = curated.supportsReasoningEffort ?? isGrokReasoningEffortCapable(curated.id);
const compat = {
...(base.compat ?? {}),
reasoningEffortMap: { ...XAI_REASONING_EFFORT_MAP, ...(base.compat?.reasoningEffortMap ?? {}) },
includeEncryptedReasoning: base.compat?.includeEncryptedReasoning ?? false,
filterReasoningHistory: base.compat?.filterReasoningHistory ?? true,
includeEncryptedReasoning: base.compat?.includeEncryptedReasoning ?? true,
filterReasoningHistory: false,
supportsImageDetailOriginal: base.compat?.supportsImageDetailOriginal ?? false,
omitReasoningEffort: !effortCapable,
supportsReasoningEffort: effortCapable,
};
if (effortCapable) {
compat.reasoningEffortMap = { ...xaiResponsesReasoningEffortMap(curated.id) };
} else {
delete compat.reasoningEffortMap;
}
return {
...base,
contextWindow: curated.contextWindow,
@@ -1502,7 +1550,7 @@ export function buildXaiOAuthStaticSeed(baseUrl?: string): ModelSpec<"openai-res
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: curated.contextWindow,
maxTokens: curated.contextWindow,
compat: { reasoningEffortMap: XAI_REASONING_EFFORT_MAP },
compat: { reasoningEffortMap: xaiResponsesReasoningEffortMap(curated.id) },
};
return mergeCuratedIntoModel(base, curated);
});
@@ -3634,15 +3682,19 @@ export function basetenModelManagerOptions(
const features = Array.isArray(raw.supported_features) ? raw.supported_features : [];
const modalities = Array.isArray(raw.input_modalities) ? raw.input_modalities : [];
// Baseten's reasoning router accepts only the high/max
// effort tiers for its GLM-5.2 and gpt-oss routes.
const isEffortReasoning =
// Baseten's discovery flags are not enough to enable OMP reasoning for every
// model. Only models with a verified Baseten reasoning policy are enabled
// here; an unknown model may use a different reasoning wire shape or effort
// vocabulary, which OMP must not guess.
const isSupportedBasetenReasoningModel =
isKimiK3ModelId(defaults.id) ||
defaults.id === "openai/gpt-oss-120b" ||
defaults.id === "deepseek-ai/DeepSeek-V4-Pro" ||
defaults.id === "zai-org/GLM-5.2" ||
defaults.id === "zai-org/GLM-5.2-Fast";
const isBasetenNativeReasoning = isEffortReasoning || defaults.id === "deepseek-ai/DeepSeek-V4-Pro";
const reasoning =
isBasetenNativeReasoning && (features.includes("reasoning") || features.includes("reasoning_effort"));
isSupportedBasetenReasoningModel &&
(features.includes("reasoning") || features.includes("reasoning_effort"));
const supportsTools = features.includes("tools") ? undefined : false;
const vision = modalities.includes("image") || (reference?.input.includes("image") ?? false);
@@ -3656,14 +3708,7 @@ export function basetenModelManagerOptions(
const contextWindow = toPositiveNumber(raw.context_length, reference?.contextWindow ?? defaults.contextWindow);
const maxTokens = toPositiveNumber(raw.max_completion_tokens, reference?.maxTokens ?? defaults.maxTokens);
const baseModel = mapWithBundledReference(entry, defaults, reference);
const thinking = isEffortReasoning
? {
mode: "effort" as const,
efforts: [Effort.High, Effort.Max],
}
: undefined;
return {
...baseModel,
@@ -3672,7 +3717,6 @@ export function basetenModelManagerOptions(
cost,
contextWindow,
maxTokens,
...(thinking ? { thinking } : {}),
...(supportsTools === false ? { supportsTools } : {}),
};
},
@@ -5713,6 +5757,15 @@ function openAiCompletionsDescriptor(
return simpleModelsDevDescriptor(modelsDevKey, providerId, "openai-completions", baseUrl, options);
}
function openAiResponsesDescriptor(
modelsDevKey: string,
providerId: string,
baseUrl: string,
options: Omit<ModelsDevProviderDescriptor, "modelsDevKey" | "providerId" | "api" | "baseUrl"> = {},
): ModelsDevProviderDescriptor {
return simpleModelsDevDescriptor(modelsDevKey, providerId, "openai-responses", baseUrl, options);
}
function anthropicMessagesDescriptor(
modelsDevKey: string,
providerId: string,
@@ -5837,7 +5890,9 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
defaultContextWindow: 131072,
}),
// --- xAI ---
openAiCompletionsDescriptor("xai", "xai", "https://api.x.ai/v1"),
openAiResponsesDescriptor("xai", "xai", "https://api.x.ai/v1", {
transformModel: model => applyXaiResponsesThinkingPolicy(model as ModelSpec<"openai-responses">),
}),
// --- DeepSeek ---
openAiCompletionsDescriptor("deepseek", "deepseek", "https://api.deepseek.com", {
// Only ship the v4 family as built-ins; older deepseek-chat / deepseek-reasoner
+15
View File
@@ -363,6 +363,13 @@ export interface OpenAICompat {
* model id. Default: true. Issue #5606.
*/
supportsSamplingParams?: boolean;
/**
* Whether presence/frequency penalties and stop sequences may be sent.
* First-party xAI `/v1/responses` rejects penalty fields for every model.
* xAI reasoning models also reject them (and `stop`) on chat completions.
* When unset, auto-detected. Default: true.
*/
supportsPenaltyAndStopParams?: boolean;
/** Always send a max-token field when the caller did not provide one. Default: auto-detected (Kimi-family models derive TPM limits from max_tokens). */
alwaysSendMaxTokens?: boolean;
/** Whether Responses-API tool-call/result history must be strictly paired. Default: auto-detected (Azure OpenAI, GitHub Copilot). */
@@ -578,6 +585,7 @@ export interface ResolvedOpenAISharedCompat {
reasoningEffortMap: Partial<Record<Effort, string>>;
supportsReasoningParams: boolean;
supportsSamplingParams: boolean;
supportsPenaltyAndStopParams: boolean;
thinkingFormat: OpenAIReasoningFormat;
/** Kimi Code transport selected by live per-model protocol metadata. */
kimiApiFormat?: OpenAICompat["kimiApiFormat"];
@@ -643,6 +651,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
| "reasoningEffortMap"
| "supportsReasoningParams"
| "supportsSamplingParams"
| "supportsPenaltyAndStopParams"
| "thinkingFormat"
| "kimiApiFormat"
| "reasoningDisableMode"
@@ -711,6 +720,12 @@ export interface ResolvedOpenAIResponsesCompat extends ResolvedOpenAISharedCompa
strictResponsesPairing: boolean;
supportsImageDetailOriginal: boolean;
supportsObfuscationOptOut: boolean;
/**
* Whether `reasoning.summary` may be sent. First-party xAI `/v1/responses`
* rejects the field; handlers pass `null` so the wire omits it instead of
* filling `"auto"`.
*/
supportsReasoningSummary: boolean;
streamIdleTimeoutMs?: number;
vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"];
/** The model sits behind Vercel AI Gateway's Responses endpoint. */
+26 -6
View File
@@ -1,4 +1,5 @@
import { describe, expect, test } from "bun:test";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { basetenModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
import type { FetchImpl } from "@oh-my-pi/pi-catalog/types";
@@ -28,6 +29,20 @@ describe("Baseten provider discovery", () => {
input_cache_read: "0.00000016",
},
},
{
id: "moonshotai/Kimi-K3",
object: "model",
name: "Kimi K3",
context_length: 1048576,
max_completion_tokens: 262144,
supported_features: ["tools", "json_mode", "structured_outputs", "reasoning_effort"],
input_modalities: ["text", "image"],
pricing: {
prompt: "0.000003",
completion: "0.000015",
input_cache_read: "0.0000003",
},
},
{
id: "deepseek-ai/DeepSeek-V4-Pro",
object: "model",
@@ -90,6 +105,15 @@ describe("Baseten provider discovery", () => {
},
});
const kimiK3 = models?.find(model => model.id === "moonshotai/Kimi-K3");
if (!kimiK3) throw new Error("Baseten Kimi K3 was not discovered");
expect(kimiK3.reasoning).toBe(true);
expect(buildModel(kimiK3).thinking).toMatchObject({
mode: "effort",
efforts: ["low", "high", "max"],
defaultLevel: "max",
});
const deepseek = models?.find(model => model.id === "deepseek-ai/DeepSeek-V4-Pro");
expect(deepseek).toBeDefined();
expect(deepseek).toMatchObject({
@@ -110,14 +134,10 @@ describe("Baseten provider discovery", () => {
const glmFast = models?.find(model => model.id === "zai-org/GLM-5.2-Fast");
expect(glmFast).toBeDefined();
expect(glmFast).toMatchObject({
provider: "baseten",
api: "openai-completions",
reasoning: true,
thinking: {
if (!glmFast) throw new Error("Baseten GLM-5.2 Fast was not discovered");
expect(buildModel(glmFast).thinking).toMatchObject({
mode: "effort",
efforts: ["high", "max"],
},
});
});
});
+48 -4
View File
@@ -223,12 +223,15 @@ describe("buildModel", () => {
});
});
describe("xAI-OAuth Responses reasoning-effort suppression", () => {
const grokResponsesSpec = (id: string): ModelSpec<"openai-responses"> => ({
describe("xAI Responses reasoning-effort suppression", () => {
const grokResponsesSpec = (
id: string,
provider: "xai" | "xai-oauth" = "xai-oauth",
): ModelSpec<"openai-responses"> => ({
id,
name: id,
api: "openai-responses",
provider: "xai-oauth",
provider,
baseUrl: "https://api.x.ai/v1",
reasoning: true,
input: ["text"],
@@ -248,6 +251,47 @@ describe("xAI-OAuth Responses reasoning-effort suppression", () => {
expect(buildOpenAIResponsesCompat(grokResponsesSpec("grok-4.3")).supportsReasoningEffort).toBe(true);
});
it("applies the same Responses dialect to paid xai and xai-oauth", () => {
const paid = buildOpenAIResponsesCompat(grokResponsesSpec("grok-4.3", "xai"));
const oauth = buildOpenAIResponsesCompat(grokResponsesSpec("grok-4.3", "xai-oauth"));
expect(paid.promptCacheSessionHeader).toBe("x-grok-conv-id");
expect(oauth.promptCacheSessionHeader).toBe("x-grok-conv-id");
expect(paid.includeEncryptedReasoning).toBe(true);
expect(oauth.includeEncryptedReasoning).toBe(true);
expect(paid.filterReasoningHistory).toBe(false);
expect(oauth.filterReasoningHistory).toBe(false);
expect(paid.supportsImageDetailOriginal).toBe(false);
expect(oauth.supportsImageDetailOriginal).toBe(false);
expect(paid.supportsReasoningEffort).toBe(true);
expect(oauth.supportsReasoningEffort).toBe(true);
expect(paid.reasoningEffortMap).toEqual({ minimal: "low", xhigh: "high", max: "high" });
expect(oauth.reasoningEffortMap).toEqual({ minimal: "low", xhigh: "high", max: "high" });
expect(
buildOpenAIResponsesCompat(grokResponsesSpec("grok-4.20-multi-agent-0309", "xai")).reasoningEffortMap,
).toEqual({ minimal: "low" });
expect(paid.supportsPenaltyAndStopParams).toBe(false);
expect(oauth.supportsPenaltyAndStopParams).toBe(false);
expect(paid.supportsReasoningSummary).toBe(false);
expect(oauth.supportsReasoningSummary).toBe(false);
});
it("suppresses penalty params on every first-party xAI Responses model", () => {
const reasoning = buildOpenAIResponsesCompat(grokResponsesSpec("grok-4.5", "xai"));
const nonReasoning = buildOpenAIResponsesCompat({
...grokResponsesSpec("grok-2", "xai"),
reasoning: false,
});
expect(reasoning.supportsPenaltyAndStopParams).toBe(false);
expect(nonReasoning.supportsPenaltyAndStopParams).toBe(false);
});
it("omits effort for paid xai models off the Grok allowlist", () => {
const compat = buildOpenAIResponsesCompat(grokResponsesSpec("grok-code-fast-1", "xai"));
expect(compat.supportsReasoningEffort).toBe(false);
expect(compat.omitReasoningEffort).toBe(true);
expect(buildModel(grokResponsesSpec("grok-code-fast-1", "xai")).thinking).toBeUndefined();
});
it("lets an explicit compat.supportsReasoningEffort override the allowlist default", () => {
const compat = buildOpenAIResponsesCompat({
...grokResponsesSpec("grok-build"),
@@ -256,7 +300,7 @@ describe("xAI-OAuth Responses reasoning-effort suppression", () => {
expect(compat.supportsReasoningEffort).toBe(true);
});
it("does not suppress effort for a non-xai-oauth provider with a grok-like id", () => {
it("does not suppress effort for a non-xAI provider with a grok-like id", () => {
const compat = buildOpenAIResponsesCompat({
...grokResponsesSpec("grok-build"),
provider: "openai",
+37 -9
View File
@@ -103,7 +103,7 @@ describe("Codex model discovery", () => {
expect(legacy?.useResponsesLite).toBeUndefined();
});
it("falls back to the 372K window for GPT-5.6 SKUs when upstream omits context_window (#5705)", async () => {
it("floors GPT-5.6 luna/sol/terra at the 1M window when upstream omits context_window (#5705)", async () => {
const fetchFn: typeof fetch = Object.assign(
async () =>
new Response(
@@ -138,12 +138,12 @@ describe("Codex model discovery", () => {
});
const sol = result?.models.find(model => model.id === "gpt-5.6-sol");
expect(sol?.contextWindow).toBe(372_000);
expect(sol?.contextWindow).toBe(1_000_000);
const legacy = result?.models.find(model => model.id === "gpt-5.5");
expect(legacy?.contextWindow).toBe(272_000);
});
it("normalizes Codex Daybreak aliases to the GPT-5.6 window and effort ladder", async () => {
it("normalizes Codex Daybreak aliases to GPT-5.6 capabilities and pricing", async () => {
const fetchFn: typeof fetch = Object.assign(
async () =>
new Response(
@@ -157,6 +157,15 @@ describe("Codex model discovery", () => {
input_modalities: ["text", "image"],
supported_in_api: true,
},
{
slug: "gpt-daybreak-red-latest",
display_name: "Daybreak Red",
context_window: 400_000,
default_reasoning_level: "high",
supported_reasoning_levels: ["minimal", "low", "medium", "high", "xhigh"],
input_modalities: ["text", "image"],
supported_in_api: true,
},
],
}),
),
@@ -168,20 +177,25 @@ describe("Codex model discovery", () => {
clientVersion: "0.99.0",
fetchFn,
});
const spec = result?.models.find(model => model.id === "gpt-daybreak-blue-latest");
if (!spec) throw new Error("Expected discovered Daybreak model");
const blue = result?.models.find(model => model.id === "gpt-daybreak-blue-latest");
if (!blue) throw new Error("Expected discovered Daybreak Blue model");
const red = result?.models.find(model => model.id === "gpt-daybreak-red-latest");
if (!red) throw new Error("Expected discovered Daybreak Red model");
expect(spec.contextWindow).toBe(372_000);
expect(getSupportedEfforts(buildModel(spec))).toEqual([
expect(blue.contextWindow).toBe(372_000);
expect(getSupportedEfforts(buildModel(blue))).toEqual([
Effort.Low,
Effort.Medium,
Effort.High,
Effort.XHigh,
Effort.Max,
]);
expect(blue.cost).toEqual({ input: 5, output: 30, cacheRead: 0.5, cacheWrite: 6.25 });
expect(red.contextWindow).toBe(400_000);
expect(red.cost).toEqual({ input: 12.5, output: 75, cacheRead: 1.25, cacheWrite: 15.625 });
});
it("honors context_window when upstream actively reports it for GPT-5.6 SKUs", async () => {
it("floors stale reported windows for GPT-5.6 luna/sol/terra and honors reports above the floor", async () => {
const fetchFn: typeof fetch = Object.assign(
async () =>
new Response(
@@ -196,6 +210,15 @@ describe("Codex model discovery", () => {
input_modalities: ["text", "image"],
supported_in_api: true,
},
{
slug: "gpt-5.6-terra",
display_name: "GPT-5.6-Terra",
context_window: 1_050_000,
default_reasoning_level: "medium",
supported_reasoning_levels: ["low", "medium", "high"],
input_modalities: ["text", "image"],
supported_in_api: true,
},
{
slug: "gpt-5.5",
display_name: "GPT-5.5",
@@ -217,8 +240,13 @@ describe("Codex model discovery", () => {
fetchFn,
});
// Registry still reports the pre-1M 272000 for sol; the floor must win.
const sol = result?.models.find(model => model.id === "gpt-5.6-sol");
expect(sol?.contextWindow).toBe(272_000);
expect(sol?.contextWindow).toBe(1_000_000);
// Reports above the floor are honored as-is.
const terra = result?.models.find(model => model.id === "gpt-5.6-terra");
expect(terra?.contextWindow).toBe(1_050_000);
// Non-floored SKUs keep the actively reported value.
const legacy = result?.models.find(model => model.id === "gpt-5.5");
expect(legacy?.contextWindow).toBe(272_000);
});
@@ -120,9 +120,9 @@ describe("generated model policies", () => {
expect(models[4]?.cost.longContext).toBeUndefined();
});
it("pins GPT-5.6 Codex-transport context window to the 372K hard capacity (#5705)", () => {
it("floors GPT-5.6 Codex-transport context windows at 1M (openai/codex#38917)", () => {
const models: ModelSpec<Api>[] = [
// Codex discovery underreports these via DEFAULT_CONTEXT_WINDOW=272000.
// Codex discovery/registry still reports the stale 272000 for these.
createSpec({
id: "gpt-5.6-luna",
api: "openai-codex-responses",
@@ -155,9 +155,9 @@ describe("generated model policies", () => {
applyGeneratedModelPolicies(models);
expect(models[0]?.contextWindow).toBe(372000);
expect(models[1]?.contextWindow).toBe(372000);
expect(models[2]?.contextWindow).toBe(372000);
expect(models[0]?.contextWindow).toBe(1_000_000);
expect(models[1]?.contextWindow).toBe(1_000_000);
expect(models[2]?.contextWindow).toBe(1_000_000);
expect(models[3]?.contextWindow).toBe(1050000);
expect(models[4]?.contextWindow).toBe(272000);
});
@@ -240,6 +240,40 @@ describe("generated model policies", () => {
expect(models[0]?.maxTokens).toBe(131_072);
});
it("pins zai glm-5.3 to 1M context and derives uniform low/high/max thinking with mandatory reasoning", () => {
const models = [
createSpec({
id: "glm-5.3",
api: "anthropic-messages",
provider: "zai",
contextWindow: 200_000,
maxTokens: 8192,
}),
createSpec({
id: "glm-5.3",
api: "openai-completions",
provider: "zhipu-coding-plan",
contextWindow: 200_000,
maxTokens: 8192,
}),
];
applyGeneratedModelPolicies(models);
// Context pinning — same 1M tier as glm-5.2 on both GLM coding-plan hosts.
for (const model of models) {
expect(model.contextWindow).toBe(1_000_000);
expect(model.maxTokens).toBe(131_072);
// Uniform wire-exact low/high/max ladder (NOT the host-specific
// high/max scale GLM-5.2 uses on zai/zhipu).
expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.High, Effort.Max]);
// Thinking can no longer be disabled.
expect(model.thinking?.requiresEffort).toBe(true);
// Default effort is `max` per the GLM-5.3 API spec.
expect(model.thinking?.defaultLevel).toBe(Effort.Max);
}
});
it("pins MiniMax-M3 long-context providers to 1M context", () => {
const models = [
createSpec({
@@ -357,11 +391,11 @@ describe("generated model policies", () => {
expect(models[0]?.compat?.supportsToolChoice).toBe(false);
});
it("sets OpenCode Go DeepSeek V4 tool-call request compat", () => {
const models: ModelSpec<"openai-completions">[] = [
it("sets OpenCode Go DeepSeek V4 tool-call request compat for both OpenAI APIs", () => {
const models: ModelSpec<Api>[] = [
createSpec({
id: "deepseek-v4-flash",
api: "openai-completions",
api: "openai-responses",
provider: "opencode-go",
}),
createSpec({
@@ -467,6 +501,43 @@ describe("generated model policies", () => {
expect(models[2]?.applyPatchToolType).toBeUndefined();
expect(models[3]?.applyPatchToolType).toBeUndefined();
});
it("strips paid xAI Responses effort dials for off-allowlist reasoners", () => {
const models: ModelSpec<"openai-responses">[] = [
createSpec({
id: "grok-code-fast-1",
api: "openai-responses",
provider: "xai",
thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
}),
createSpec({
id: "grok-4.5",
api: "openai-responses",
provider: "xai",
thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
}),
createSpec({
id: "grok-code-fast-1",
api: "openai-responses",
provider: "openrouter",
thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
}),
];
applyGeneratedModelPolicies(models);
expect(models[0]?.thinking).toBeUndefined();
expect(models[0]?.compat).toMatchObject({
supportsReasoningEffort: false,
omitReasoningEffort: true,
});
expect(models[0]?.compat).not.toHaveProperty("reasoningEffortMap");
expect(models[1]?.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]);
expect(models[1]?.compat?.supportsReasoningEffort).toBe(true);
// Non-xAI hosts are outside this policy — no baked no-dial compat.
expect(models[2]?.thinking).toBeDefined();
expect(models[2]?.compat?.supportsReasoningEffort).toBeUndefined();
});
});
describe("applyOllamaCloudOutputCap", () => {
@@ -5,7 +5,9 @@ import {
isGeminiModelId,
isGlmVisionModelId,
isGrokModelId,
isGrokMultiAgentModelId,
isGrokReasoningEffortCapable,
isGrokXHighEffortCapable,
isKimiK26ModelId,
isKimiModelId,
isMinimaxM2FamilyModelId,
@@ -339,6 +341,7 @@ describe("isGrokReasoningEffortCapable", () => {
expect(isGrokReasoningEffortCapable("grok-4.20-multi-agent")).toBe(true);
expect(isGrokReasoningEffortCapable("xai-oauth/grok-4.3")).toBe(true);
expect(isGrokReasoningEffortCapable("xai-oauth/grok-4.5")).toBe(true);
expect(isGrokReasoningEffortCapable("xai-oauth/grok-4.6")).toBe(true);
expect(isGrokReasoningEffortCapable("openrouter/xai/grok-3-mini")).toBe(true);
});
@@ -349,3 +352,35 @@ describe("isGrokReasoningEffortCapable", () => {
expect(isGrokReasoningEffortCapable("")).toBe(false);
});
});
describe("isGrokMultiAgentModelId", () => {
test("matches grok-4.20-multi-agent SKUs across namespaces", () => {
expect(isGrokMultiAgentModelId("grok-4.20-multi-agent")).toBe(true);
expect(isGrokMultiAgentModelId("grok-4.20-multi-agent-0309")).toBe(true);
expect(isGrokMultiAgentModelId("xai/grok-4.20-multi-agent-beta-latest")).toBe(true);
});
test("rejects other Grok ids", () => {
expect(isGrokMultiAgentModelId("grok-4.5")).toBe(false);
expect(isGrokMultiAgentModelId("grok-4.6")).toBe(false);
expect(isGrokMultiAgentModelId("grok-4.20-0309-reasoning")).toBe(false);
expect(isGrokMultiAgentModelId("")).toBe(false);
});
});
describe("isGrokXHighEffortCapable", () => {
test("matches grok-4.6 and multi-agent SKUs across namespaces", () => {
expect(isGrokXHighEffortCapable("grok-4.6")).toBe(true);
expect(isGrokXHighEffortCapable("xai/grok-4.6")).toBe(true);
expect(isGrokXHighEffortCapable("xai-oauth/grok-4.6")).toBe(true);
expect(isGrokXHighEffortCapable("grok-4.20-multi-agent-0309")).toBe(true);
});
test("rejects Grok SKUs that clamp leftover xhigh to high", () => {
expect(isGrokXHighEffortCapable("grok-4.5")).toBe(false);
expect(isGrokXHighEffortCapable("grok-4.3")).toBe(false);
expect(isGrokXHighEffortCapable("grok-3-mini")).toBe(false);
expect(isGrokXHighEffortCapable("grok-build")).toBe(false);
expect(isGrokXHighEffortCapable("")).toBe(false);
});
});
@@ -1,65 +0,0 @@
/**
* Repro for #887 — OpenCode Go: Minimax M2.7 (and Qwen3.5/3.6 Plus) return 404
* because the resolver routes them to anthropic-messages /v1/messages while
* the OpenCode Go gateway only serves them at /v1/chat/completions.
*
* stencil.so declares these ids with `provider.npm = "@ai-sdk/anthropic"`,
* which by default would resolve to anthropic-messages on opencode-go. The
* descriptor must override these specific ids to openai-completions so that
* regenerated models.json keeps the correct routing.
*/
import { describe, expect, test } from "bun:test";
import {
MODELS_DEV_PROVIDER_DESCRIPTORS,
type ModelsDevModel,
opencodeGoModelManagerOptions,
} from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
const OPENCODE_GO_BASE = "https://opencode.ai/zen/go/v1";
describe("opencode-go resolver routes 404-ing ids to openai-completions (issue #887)", () => {
const descriptor = MODELS_DEV_PROVIDER_DESCRIPTORS.find(d => d.providerId === "opencode-go");
// Per upstream stencil.so (verified 2026-05-02 against
// https://stencil.so/api.json["opencode-go"].models), these three ids carry
// `provider.npm = "@ai-sdk/anthropic"`. The naive @ai-sdk/anthropic rule
// would route them to /v1/messages on opencode.ai/zen/go which 404s.
const npmAnthropic: ModelsDevModel = { provider: { npm: "@ai-sdk/anthropic" }, tool_call: true };
test.each([["minimax-m2.7"], ["qwen3.5-plus"], ["qwen3.6-plus"]])(
"%s resolves to openai-completions on /v1/chat/completions",
modelId => {
const resolved = descriptor?.resolveApi?.(modelId, npmAnthropic);
expect(resolved).toEqual({ api: "openai-completions", baseUrl: OPENCODE_GO_BASE });
},
);
test("minimax-m2.5 (control: works empirically) also resolves to openai-completions", () => {
// stencil.so currently lists minimax-m2.5 without an explicit provider.npm,
// so it falls through to the default openai-completions resolution.
const m25: ModelsDevModel = { tool_call: true };
const resolved = descriptor?.resolveApi?.("minimax-m2.5", m25);
expect(resolved).toEqual({ api: "openai-completions", baseUrl: OPENCODE_GO_BASE });
});
test("runtime /v1/models refresh preserves qwen3.7-max Anthropic transport", async () => {
let requestedUrl = "";
const fetchMock = (async (input: string | Request | URL): Promise<Response> => {
requestedUrl = input instanceof Request ? input.url : String(input);
return new Response(
JSON.stringify({
data: [{ id: "qwen3.7-max", name: "Qwen3.7 Max", context_length: 1000000 }],
}),
{ headers: { "content-type": "application/json" } },
);
}) as typeof fetch;
const options = opencodeGoModelManagerOptions({ apiKey: "opencode-test-key", fetch: fetchMock });
const models = await options.fetchDynamicModels?.();
const qwenMax = models?.find(model => model.id === "qwen3.7-max");
expect(requestedUrl).toBe("https://opencode.ai/zen/go/v1/models");
expect(qwenMax?.api).toBe("anthropic-messages");
expect(qwenMax?.baseUrl).toBe("https://opencode.ai/zen/go");
});
});

Some files were not shown because too many files have changed in this diff Show More