diff --git a/Cargo.lock b/Cargo.lock index e5bcf3376..a2f4c89f2 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -238,9 +238,9 @@ checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a" [[package]] name = "bitflags" -version = "2.12.1" +version = "2.13.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "84d7ced0ae9557296835c32bf1b1e02b44c746701f898460fb000d7eaa84f00a" +checksum = "b4388bee8683e3d04af747c73422af53102d2bd24d9eadb6cbc100baef4b43f8" [[package]] name = "bitvec" @@ -872,7 +872,7 @@ version = "0.3.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1e0e367e4e7da84520dedcac1901e4da967309406d1e51017ae1abfb97adbd38" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "objc2", ] @@ -1830,7 +1830,7 @@ version = "3.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f1d395473824516f38dd1071a1a37bc57daa7be65b293ebba4ead5f7abb017a2" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "ctor", "futures", "napi-build", @@ -1894,7 +1894,7 @@ version = "0.28.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ab2156c4fce2f8df6c499cc1c763e4394b7482525bf2a9701c9d79d215f519e4" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "cfg-if", "cfg_aliases 0.1.1", "libc", @@ -1906,7 +1906,7 @@ version = "0.31.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cf20d2fde8ff38632c426f1165ed7436270b44f199fc55284c38276f9db47c3d" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "cfg-if", "cfg_aliases 0.2.1", "libc", @@ -1997,7 +1997,7 @@ version = "0.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d49e936b501e5c5bf01fda3a9452ff86dc3ea98ad5f283e1455153142d97518c" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "objc2", "objc2-core-graphics", "objc2-foundation", @@ -2009,7 +2009,7 @@ version = "0.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2a180dd8642fa45cdb7dd721cd4c11b1cadd4929ce112ebd8b9f5803cc79d536" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "dispatch2", "objc2", ] @@ -2020,7 +2020,7 @@ version = "0.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e022c9d066895efa1345f8e33e584b9f958da2fd4cd116792e15e07e4720a807" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "dispatch2", "objc2", "objc2-core-foundation", @@ -2039,7 +2039,7 @@ version = "0.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e3e0adef53c21f888deb4fa59fc59f7eb17404926ee8a6f59f5df0fd7f9f3272" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "objc2", "objc2-core-foundation", ] @@ -2050,7 +2050,7 @@ version = "0.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "180788110936d59bab6bd83b6060ffdfffb3b922ba1396b312ae795e1de9d81d" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "objc2", "objc2-core-foundation", ] @@ -2331,7 +2331,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "15.9.4" +version = "15.10.1" dependencies = [ "anyhow", "ast-grep-core", @@ -2399,7 +2399,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "15.9.4" +version = "15.10.1" dependencies = [ "async-trait", "libc", @@ -2411,7 +2411,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "15.9.4" +version = "15.10.1" dependencies = [ "anyhow", "arboard", @@ -2457,7 +2457,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "15.9.4" +version = "15.10.1" dependencies = [ "anyhow", "brush-builtins", @@ -2497,7 +2497,7 @@ version = "0.18.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "60769b8b31b2a9f263dae2776c37b1b28ae246943cf719eb6946a1db05128a61" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "crc32fast", "fdeflate", "flate2", @@ -2567,7 +2567,7 @@ version = "0.18.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "25485360a54d6861439d60facef26de713b1e126bf015ec8f98239467a2b82f7" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "chrono", "flate2", "procfs-core", @@ -2580,7 +2580,7 @@ version = "0.18.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e6401bf7b6af22f78b563665d15a22e9aef27775b79b149a66ca022468a4e405" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "chrono", "hex", ] @@ -2735,7 +2735,7 @@ version = "0.5.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", ] [[package]] @@ -2833,7 +2833,7 @@ version = "1.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "errno", "libc", "linux-raw-sys", @@ -4279,7 +4279,7 @@ version = "0.244.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "47b807c72e1bac69382b3a6fb3dbe8ea4c0ed87ff5629b8685ae6b9a611028fe" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "hashbrown 0.15.5", "indexmap", "semver", @@ -4304,7 +4304,7 @@ version = "0.31.14" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "645c7c96bb74690c3189b5c9cb4ca1627062bb23693a4fad9d8c3de958260144" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "rustix", "wayland-backend", "wayland-scanner", @@ -4316,7 +4316,7 @@ version = "0.32.12" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "563a85523cade2429938e790815fd7319062103b9f4a2dc806e9b53b95982d8f" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "wayland-backend", "wayland-client", "wayland-scanner", @@ -4328,7 +4328,7 @@ version = "0.3.12" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "eb04e52f7836d7c7976c78ca0250d61e33873c34156a2a1fc9474828ec268234" dependencies = [ - "bitflags 2.12.1", + "bitflags 2.13.0", "wayland-backend", "wayland-client", "wayland-protocols", @@ -4836,7 +4836,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9d66ea20e9553b30172b5e831994e35fbde2d165325bec84fc43dbf6f4eb9cb2" dependencies = [ "anyhow", - "bitflags 2.12.1", + "bitflags 2.13.0", "indexmap", "log", "serde", diff --git a/Cargo.toml b/Cargo.toml index 49e6ebf38..f6301942c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"] resolver = "3" [workspace.package] -version = "15.9.4" +version = "15.10.1" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index d2646c93e..3cff53436 100644 --- a/bun.lock +++ b/bun.lock @@ -15,7 +15,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "15.9.4", + "version": "15.10.1", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -30,7 +30,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "15.9.4", + "version": "15.10.1", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -44,7 +44,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "15.9.4", + "version": "15.10.1", "bin": { "omp": "src/cli.ts", }, @@ -90,7 +90,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "15.9.4", + "version": "15.10.1", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -101,7 +101,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "15.9.4", + "version": "15.10.1", "bin": { "mnemopi": "src/cli.ts", }, @@ -118,7 +118,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "15.9.4", + "version": "15.10.1", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -126,7 +126,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "15.9.4", + "version": "15.10.1", "bin": { "omp-stats": "./src/index.ts", }, @@ -151,7 +151,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "15.9.4", + "version": "15.10.1", "bin": { "omp-swarm": "src/cli.ts", }, @@ -167,7 +167,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "15.9.4", + "version": "15.10.1", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -208,7 +208,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "15.9.4", + "version": "15.10.1", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "beautiful-mermaid": "catalog:", @@ -248,15 +248,15 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "15.9.4", - "@oh-my-pi/omp-stats": "15.9.4", - "@oh-my-pi/pi-agent-core": "15.9.4", - "@oh-my-pi/pi-ai": "15.9.4", - "@oh-my-pi/pi-coding-agent": "15.9.4", - "@oh-my-pi/pi-mnemopi": "15.9.4", - "@oh-my-pi/pi-natives": "15.9.4", - "@oh-my-pi/pi-tui": "15.9.4", - "@oh-my-pi/pi-utils": "15.9.4", + "@oh-my-pi/hashline": "15.10.1", + "@oh-my-pi/omp-stats": "15.10.1", + "@oh-my-pi/pi-agent-core": "15.10.1", + "@oh-my-pi/pi-ai": "15.10.1", + "@oh-my-pi/pi-coding-agent": "15.10.1", + "@oh-my-pi/pi-mnemopi": "15.10.1", + "@oh-my-pi/pi-natives": "15.10.1", + "@oh-my-pi/pi-tui": "15.10.1", + "@oh-my-pi/pi-utils": "15.10.1", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -395,7 +395,7 @@ "@huggingface/jinja": ["@huggingface/jinja@0.5.9", "", {}, "sha512-uWTG+l3VJRsl7EXxYizuL3P+cCPoc3cRqbWWRcQN0FhejRfbdq0RNhCmbY/YDtnTcz9icdLYuLDjsnz4d8JMuw=="], - "@huggingface/tasks": ["@huggingface/tasks@0.21.2", "", {}, "sha512-e8dw3tZ7mbZ/mytr9zIFmsr67tMpd9rm/pURCi5ciFqlVvLvdH9FjnoZxVO4KfRXyqzPeV0shXv9z5s2M8Msmw=="], + "@huggingface/tasks": ["@huggingface/tasks@0.21.7", "", {}, "sha512-GuEXszIkir4j/Oywp4hXP+wfwojo/SKWA/omroNkzWWgqUGiOQ5p6HuyXcDOcinYnLQW1WsO8fwdEvtLTZbA4w=="], "@huggingface/tokenizers": ["@huggingface/tokenizers@0.1.3", "", {}, "sha512-8rF/RRT10u+kn7YuUbUg0OF30K8rjTc78aHpxT+qJ1uWSqxT1MHi8+9ltwYfkFYJzT/oS+qw3JVfHtNMGAdqyA=="], @@ -1159,7 +1159,7 @@ "onnxruntime-web": ["onnxruntime-web@1.26.0-dev.20260416-b7804b056c", "", { "dependencies": { "flatbuffers": "^25.1.24", "guid-typescript": "^1.0.9", "long": "^5.2.3", "onnxruntime-common": "1.24.0-dev.20251116-b39e144322", "platform": "^1.3.6", "protobufjs": "^7.2.4" } }, "sha512-MD6Ss4GSpQBo6zqoJzyT9LRbKYs7x/JVN23FT24EcEvlqF4VuzPOeH6X38orZPKHQDbprn7K+SBpu0/mj2CQiw=="], - "openai": ["openai@6.39.1", "", { "peerDependencies": { "ws": "^8.18.0", "zod": "^3.25 || ^4.0" }, "optionalPeers": ["ws", "zod"], "bin": { "openai": "bin/cli" } }, "sha512-z3dO9fEWOXBzlXynVb/xZ/tujzUjFWQWn3C0n0mw6Vo0zJTbEkaN4b2cLWjhJ6haJQx8LlREoafHRl+Gu/Hl+A=="], + "openai": ["openai@6.42.0", "", { "peerDependencies": { "ws": "^8.18.0", "zod": "^3.25 || ^4.0" }, "optionalPeers": ["ws", "zod"] }, "sha512-1WFEt/uXMXOLhYRNkgJWo08Y2YNvNwpVU72K7ibrWgWpNOXd4VojXLbe6SQ4bLiUQ3Y8jz4IiyVkylJCL1DtZg=="], "option": ["option@0.2.4", "", {}, "sha512-pkEqbDyl8ou5cpq+VsnQbe/WlEy5qS7xPzMS1U55OCG9KPvwFD46zDbxQIj3egJSFc3D+XhYOPUzz49zQAVy7A=="], @@ -1259,7 +1259,7 @@ "string-width": ["string-width@4.2.3", "", { "dependencies": { "emoji-regex": "^8.0.0", "is-fullwidth-code-point": "^3.0.0", "strip-ansi": "^6.0.1" } }, "sha512-wKyQRQpjJ0sIp62ErSZdGsjMJWsap5oRNihHhu6G7JVO/9jIB6UyevL+tXuOqrng8j/cxKTWyWUwvSTriiZz/g=="], - "string_decoder": ["string_decoder@1.3.0", "", { "dependencies": { "safe-buffer": "~5.2.0" } }, "sha512-hkRX8U1WjJFd8LsDJ2yQ/wWWxaopEsABU1XfkM8A+j0+85JAGppt16cr1Whg6KIbb4okU6Mql6BOj+uup/wKeA=="], + "string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="], "strip-ansi": ["strip-ansi@7.2.0", "", { "dependencies": { "ansi-regex": "^6.2.2" } }, "sha512-yDPMNjp4WyfYBkHnjIRLfca1i6KMyGCtsVgoKe/z1+6vukgaENdgGBZt+ZmKPc4gavvEZ5OgHfHdrazhgNyG7w=="], @@ -1413,8 +1413,6 @@ "string-width/strip-ansi": ["strip-ansi@6.0.1", "", { "dependencies": { "ansi-regex": "^5.0.1" } }, "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A=="], - "string_decoder/safe-buffer": ["safe-buffer@5.2.1", "", {}, "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ=="], - "wrap-ansi/string-width": ["string-width@8.2.1", "", { "dependencies": { "get-east-asian-width": "^1.5.0", "strip-ansi": "^7.1.2" } }, "sha512-IIaP0g3iy9Cyy18w3M9YcaDudujEAVHKt3a3QJg1+sr/oX96TbaGUubG0hJyCjCBThFH+tFpcIyoUHUn1ogaLA=="], "xml2js/xmlbuilder": ["xmlbuilder@11.0.1", "", {}, "sha512-fDlsI/kFEx7gLvbecc0/ohLG50fugQp8ryHzMTuW9vSa1GJ0XYWKnhsUx7oie3G98+r56aTQIUB4kht42R3JvA=="], @@ -1429,8 +1427,6 @@ "fastembed/onnxruntime-node/tar": ["tar@7.5.16", "", { "dependencies": { "@isaacs/fs-minipass": "^4.0.0", "chownr": "^3.0.0", "minipass": "^7.1.2", "minizlib": "^3.1.0", "yallist": "^5.0.0" } }, "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w=="], - "jszip/readable-stream/string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="], - "log-update/slice-ansi/is-fullwidth-code-point": ["is-fullwidth-code-point@5.1.0", "", { "dependencies": { "get-east-asian-width": "^1.3.1" } }, "sha512-5XHYaSyiqADb4RnZ1Bdad6cPp8Toise4TzEjcOYDHZkTCbKgiUl7WTUCpNWHuxmDt91wnsZBc9xinNzopv3JMQ=="], "log-update/wrap-ansi/string-width": ["string-width@7.2.0", "", { "dependencies": { "emoji-regex": "^10.3.0", "get-east-asian-width": "^1.0.0", "strip-ansi": "^7.1.0" } }, "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ=="], diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index ae74dbadd..6b9c17c74 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -68,5 +68,5 @@ use napi_derive::napi; /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV15_9_4")] +#[napi(js_name = "__piNativesV15_10_1")] pub const fn pi_natives_version_sentinel() {} diff --git a/crates/pi-shell/src/fixup.rs b/crates/pi-shell/src/fixup.rs index 86b6974d4..f9fe9d70a 100644 --- a/crates/pi-shell/src/fixup.rs +++ b/crates/pi-shell/src/fixup.rs @@ -154,7 +154,12 @@ fn try_strip_head_tail( // span under-reports when its suffix contains unlocated `IoRedirect`s // (e.g. the synthetic `2>&1` inserted by `|&`). let bytes = cmd.as_bytes(); - let last_start = last_loc.start.index; + let Some(last_start) = byte_offset(cmd, last_loc.start.index) else { + return default; + }; + let Some(last_end) = byte_offset(cmd, last_loc.end.index) else { + return default; + }; let Some(head) = cmd.get(..last_start) else { return default; }; @@ -172,7 +177,7 @@ fn try_strip_head_tail( // Reported text starts at the pipe and is right-trimmed. The deletion // range walks back through any leading whitespace so the rewrite is // contiguous. - let stripped_text = cmd[pipe_pos..last_loc.end.index].trim_end().to_owned(); + let stripped_text = cmd[pipe_pos..last_end].trim_end().to_owned(); if stripped_text.is_empty() { return default; } @@ -180,7 +185,7 @@ fn try_strip_head_tail( while delete_start > 0 && matches!(bytes[delete_start - 1], b' ' | b'\t') { delete_start -= 1; } - ranges.push((delete_start, last_loc.end.index)); + ranges.push((delete_start, last_end)); stripped.push(stripped_text); HeadTailOutcome { stripped: true, last_idx: n - 2 } } @@ -253,10 +258,14 @@ fn try_strip_2to1( let Some(name_loc) = name_word.loc.as_ref() else { return; }; - let mut anchor = name_loc.end.index; + let Some(mut anchor) = byte_offset(cmd, name_loc.end.index) else { + return; + }; for item in &suffix.0 { - if let Some(loc) = item.location() { - anchor = anchor.max(loc.end.index); + if let Some(loc) = item.location() + && let Some(end) = byte_offset(cmd, loc.end.index) + { + anchor = anchor.max(end); } } let bytes = cmd.as_bytes(); @@ -283,6 +292,28 @@ fn try_strip_2to1( stripped.push("2>&1".to_owned()); } +/// Translate a `brush-parser` source-position index into a byte offset in +/// `cmd`. The parser counts positions in Unicode scalars (one increment per +/// `char`; see `tokenizer::next_char`), but we slice `cmd` — a `&str` — by +/// byte index, so the two diverge as soon as the command contains any +/// multi-byte UTF-8 (e.g. a `✓`/`×` literal inside a `grep` pattern). Without +/// this conversion the head/tail and `2>&1` strips cut at the wrong place, +/// corrupting the command (notably orphaning a closing quote). +/// +/// Returns `None` only when `char_idx` is past the end of the input. The +/// end-of-input position (`char_idx == cmd.chars().count()`) maps to +/// `cmd.len()`. +fn byte_offset(cmd: &str, char_idx: usize) -> Option { + let mut count = 0usize; + for (byte, _) in cmd.char_indices() { + if count == char_idx { + return Some(byte); + } + count += 1; + } + (count == char_idx).then_some(cmd.len()) +} + fn is_stderr_to_stdout(io: &IoRedirect) -> bool { let IoRedirect::File(Some(2), IoFileRedirectKind::DuplicateOutput, target) = io else { return false; @@ -380,6 +411,45 @@ mod tests { } } + #[test] + fn strips_with_multibyte_content() { + // brush-parser reports char-indexed positions; multi-byte UTF-8 before + // the trailing `| tail` must not shift the byte-level cut. Regression + // for a corrupted command that orphaned the grep pattern's closing + // quote (`… |✓|×-80`) and broke later re-parsing. + let cases: &[(&str, &str, &[&str])] = &[ + // Mirrors the real bug: `2>&1` sits mid-pipeline (grep becomes the + // effective tail) so only `| tail -80` is stripped, leaving the + // quoted grep pattern intact. + ( + "xcodebuild 2>&1 | grep -E \"a|✓|×|b\" | tail -80", + "xcodebuild 2>&1 | grep -E \"a|✓|×|b\"", + &["| tail -80"], + ), + ("echo ✓ | head -3", "echo ✓", &["| head -3"]), + ("printf '日本語' | tail -n 5", "printf '日本語'", &["| tail -n 5"]), + ]; + for (input, want_cmd, want_stripped) in cases { + let (cmd, stripped) = run(input); + assert_eq!(cmd, *want_cmd, "input: {input:?}"); + assert_eq!(stripped, *want_stripped, "input: {input:?}"); + // The rewrite must remain valid shell (no orphaned quote). + let options = ParserOptions::default(); + let source_info = SourceInfo::default(); + let mut reader = BufReader::new(cmd.as_bytes()); + let mut parser = Parser::new(&mut reader, &options, &source_info); + assert!(parser.parse_program().is_ok(), "rewrite not re-parseable: {cmd:?}"); + } + } + + #[test] + fn strips_2to1_after_multibyte() { + // Multi-byte before a trailing `2>&1` must not misplace the strip. + let (cmd, stripped) = run("echo ✓ × 2>&1"); + assert_eq!(cmd, "echo ✓ ×"); + assert_eq!(stripped, vec!["2>&1"]); + } + #[test] fn strips_redundant_2to1() { let cases: &[(&str, &str, &[&str])] = &[ diff --git a/docs/extensions.md b/docs/extensions.md index 119d0f2cb..faedb4c81 100644 --- a/docs/extensions.md +++ b/docs/extensions.md @@ -382,7 +382,7 @@ Provide `renderCall` / `renderResult` on `registerTool` definitions for custom t - Runtime actions are unavailable during extension load. - `tool_call` errors block execution (fail-closed). - Command name conflicts with built-ins are skipped with diagnostics. -- Reserved shortcuts are ignored (`ctrl+c`, `ctrl+d`, `ctrl+z`, `ctrl+k`, `ctrl+p`, `ctrl+l`, `ctrl+o`, `ctrl+t`, `ctrl+g`, `shift+tab`, `shift+ctrl+p`, `alt+enter`, `escape`, `enter`). +- Reserved shortcuts are ignored (`ctrl+c`, `ctrl+d`, `ctrl+z`, `ctrl+k`, `ctrl+p`, `ctrl+l`, `ctrl+o`, `ctrl+t`, `ctrl+g`, `ctrl+q`, `alt+m`, `shift+tab`, `shift+ctrl+p`, `alt+enter`, `escape`, `enter`). - Treat `ctx.reload()` as terminal for the current command handler frame. ## Extensions vs hooks vs custom-tools diff --git a/docs/keybindings.md b/docs/keybindings.md index 8e8158bbf..dfc881bbe 100644 --- a/docs/keybindings.md +++ b/docs/keybindings.md @@ -27,20 +27,23 @@ app.stt.toggle: [] | `app.model.cycleForward` | `Ctrl+P` | Cycle role models forward | | `app.model.cycleBackward` | `Shift+Ctrl+P` | Cycle role models in temporary mode | | `app.model.selectTemporary` | `Alt+P` | Pick a model temporarily for this session | -| `app.model.select` | `Ctrl+L` | Open the model selector and set roles | +| `app.model.select` | `Alt+M` | Open the model selector and set roles | | `app.plan.toggle` | `Alt+Shift+P` | Toggle plan mode | | `app.history.search` | `Ctrl+R` | Search prompt history | | `app.tools.expand` | `Ctrl+O` | Toggle tool-output expansion | | `app.thinking.toggle` | `Ctrl+T` | Toggle thinking-block visibility | | `app.thinking.cycle` | `Shift+Tab` | Cycle thinking level | | `app.editor.external` | `Ctrl+G` | Edit the draft in `$VISUAL` / `$EDITOR` | -| `app.message.followUp` | `Ctrl+Enter` | Queue a follow-up message | +| `app.message.followUp` | `Ctrl+Q`, `Ctrl+Enter` | Queue a follow-up message | | `app.message.dequeue` | `Alt+Up` | Dequeue a queued message back into the editor | +| `app.display.reset` | `Ctrl+L` | Reset terminal display | | `app.clipboard.copyLine` | `Alt+Shift+L` | Copy the current line | | `app.clipboard.copyPrompt` | `Alt+Shift+C` | Copy the whole prompt | | `app.clipboard.pasteImage` | `Ctrl+V` (`Alt+V` fallback on Windows) | Paste an image from the clipboard | | `app.stt.toggle` | `Alt+H` | Toggle speech-to-text recording | -On Windows Terminal, `Ctrl+V` may be handled by the terminal paste command before `omp` sees it; use the `Alt+V` fallback when clipboard image paste appears to do nothing. +On Windows Terminal, `Ctrl+V` may be handled by the terminal paste command before `omp` sees it; use the `Alt+V` fallback when clipboard image paste appears to do nothing. Windows Terminal also swallows `Ctrl+Enter`, so the follow-up shortcut also binds `Ctrl+Q` — the same chord GitHub Copilot CLI uses. If your existing `keybindings.yml` already assigns `Ctrl+Q` to another action, that user remap wins and follow-up keeps `Ctrl+Enter` unless you explicitly bind `app.message.followUp`. + +Terminals that implement OSC 5522 enhanced paste can send clipboard MIME data directly to `omp`; image pastes are attached as `[Image #N]`, while text/plain paste events keep normal paste behavior. When OSC 5522 is unavailable, bracketed paste still handles text, and a pasted single image-file path is loaded as an image when the file is readable from the `omp` host. Older unqualified action names are migrated when `keybindings.yml` is loaded, but new docs and new configs should use the namespaced action IDs above. Existing `keybindings.json` files are still accepted and migrated to `keybindings.yml`; `keybindings.yaml` is also accepted. diff --git a/docs/models.md b/docs/models.md index daa9e7a2a..a1ca6d89f 100644 --- a/docs/models.md +++ b/docs/models.md @@ -557,9 +557,9 @@ For `anthropic-messages` models the runtime uses a separate `AnthropicCompat` sh ### Strict tool schemas (`disableStrictTools`) -Anthropic's API supports a `strict` field on tool definitions that forces the model to always follow the provided schema exactly. This is enabled by default for all `anthropic-messages` providers because it guarantees schema conformance in agentic systems. +Anthropic's API supports a `strict` field on tool definitions that forces the model to always follow the provided schema exactly. OMP enables it by default for a small allowlist of high-frequency built-in `anthropic-messages` tools (`bash`, `python`, `edit`, and `find`) whose schemas fit Anthropic's strict grammar limits; other tools still send normalized schemas but omit `strict`. -Third-party providers that front the Anthropic API (AWS Bedrock, Azure, self-hosted proxies) do not always implement this field and will reject requests that include it. Set `disableStrictTools: true` at the provider level to opt out: +Third-party providers that front the Anthropic API (AWS Bedrock, Azure, self-hosted proxies) do not always implement this field and will reject requests that include it. Set `disableStrictTools: true` at the provider level to opt out of strict mode for the allowlisted tools: ```yaml providers: @@ -581,7 +581,7 @@ providers: cacheWrite: 3.75 ``` -`disableStrictTools` is a provider-level flag that applies to all models in the provider. +`disableStrictTools` is a provider-level flag that applies to all models in the provider. It disables the Anthropic `strict` marker only for tools that OMP would otherwise mark strict; it does not change runtime tool argument validation. OMP can automatically retry without strict tools after Anthropic reports a strict-grammar-too-large error before the first streamed token, but proxies that reject the `strict` field for other reasons should set this flag explicitly. Tool schemas going on the wire are normalized by the unified flow in `packages/ai/src/utils/schema/normalize.ts` (Google/CCA/MCP dispatchers diff --git a/docs/session-operations-export-share-fork-resume.md b/docs/session-operations-export-share-fork-resume.md index 0173137c1..0d6de4f70 100644 --- a/docs/session-operations-export-share-fork-resume.md +++ b/docs/session-operations-export-share-fork-resume.md @@ -1,4 +1,4 @@ -# Session Operations: export, dump, share, fork, resume/continue +# Session Operations: export, dump, share, fresh, fork, resume/continue This document describes operator-visible behavior for session export/share/fork/resume operations as currently implemented. @@ -19,6 +19,7 @@ This document describes operator-visible behavior for session export/share/fork/ | `/export [path]` | Interactive slash command | No | No | HTML file | | `--export [outputPath]` | CLI startup fast-path | No runtime session mutation | No active session; reads target file | HTML file | | `/share` | Interactive slash command | No | No | Temp HTML + share URL/gist | +| `/fresh` | Interactive slash command | Yes (provider-facing in-memory id/state only) | No; keeps current session file/header | None | | `/fork` | Interactive slash command | Yes (active session identity changes) | Creates new session file and switches current session to it (persistent mode only) | Copies artifact directory to new session namespace when present | | `--fork ` | CLI startup | Yes after session creation | Creates a new session fork from the selected source into current cwd/session dir | None | | `/resume` | Interactive slash command | Yes (active in-memory state replaced) | Switches to selected existing session file | None | diff --git a/docs/skills/authoring-extensions.md b/docs/skills/authoring-extensions.md index 6419f153d..5d6c8367b 100644 --- a/docs/skills/authoring-extensions.md +++ b/docs/skills/authoring-extensions.md @@ -242,7 +242,7 @@ The derived name is the filename stem (or directory name for `index.ts`-style en - **Do not call runtime actions during load.** Methods like `pi.sendMessage()` throw `ExtensionRuntimeNotInitializedError` if called synchronously during module evaluation (before a session is active). Register handlers/tools/commands during load; perform runtime actions only from event handlers, tools, or commands. - **`tool_call` errors are fail-closed.** If a `tool_call` handler throws, the tool is blocked. - **Command names must not clash with built-ins.** Conflicts are skipped with a diagnostic log. -- **Reserved shortcuts are ignored** (`ctrl+c`, `ctrl+d`, `ctrl+z`, `ctrl+k`, `ctrl+p`, `ctrl+l`, `ctrl+o`, `ctrl+t`, `ctrl+g`, `shift+tab`, `shift+ctrl+p`, `alt+enter`, `escape`, `enter`). +- **Reserved shortcuts are ignored** (`ctrl+c`, `ctrl+d`, `ctrl+z`, `ctrl+k`, `ctrl+p`, `ctrl+l`, `ctrl+o`, `ctrl+t`, `ctrl+g`, `ctrl+q`, `alt+m`, `shift+tab`, `shift+ctrl+p`, `alt+enter`, `escape`, `enter`). ## Further reading diff --git a/docs/theme.md b/docs/theme.md index 738ab3b5b..36be98c4e 100644 --- a/docs/theme.md +++ b/docs/theme.md @@ -85,7 +85,7 @@ If omitted, export code derives defaults from resolved theme colors. - `symbols.preset` sets a theme-level default symbol set. - `symbols.overrides` can override individual `SymbolKey` values. -- `symbols.spinnerFrames` overrides the loading spinner frames. Accepts either a flat `string[]` (applied to both spinner types) or an object `{ "status"?: string[], "activity"?: string[] }` to override each type independently. Any type not specified falls back to the symbol preset's default frames. `status` drives the ~12.5fps spinner used by loaders and tool-execution indicators; `activity` drives the ~60fps spinner used by markdown progress bars and similar high-frequency UI. +- `symbols.spinnerFrames` overrides the loading spinner frames. Accepts either a flat `string[]` (applied to both spinner types) or an object `{ "status"?: string[], "activity"?: string[] }` to override each type independently. Any type not specified falls back to the symbol preset's default frames. `status` drives the ~12.5fps spinner used by loaders and tool-execution indicators; `activity` drives the ~30fps spinner used by markdown progress bars and similar high-frequency UI. Runtime precedence: diff --git a/docs/tools/search_tool_bm25.md b/docs/tools/search_tool_bm25.md index 1de189867..ab362da7e 100644 --- a/docs/tools/search_tool_bm25.md +++ b/docs/tools/search_tool_bm25.md @@ -42,7 +42,7 @@ - The renderer shows a status line plus up to 5 collapsed tree items by default (`COLLAPSED_MATCH_LIMIT`), each with label, optional server name, score to 3 decimals, and truncated description. The ranked match list is not serialized into `content`. ## Flow -1. `SearchToolBm25Tool.createIf()` in `packages/coding-agent/src/tools/search-tool-bm25.ts` exposes the tool only when `tools.discoveryMode` is set to a non-`"off"` value or legacy `mcp.discoveryMode === true`, and only if the session implements the discovery hooks. +1. `SearchToolBm25Tool.createIf()` in `packages/coding-agent/src/tools/search-tool-bm25.ts` exposes the tool for explicit discovery modes (`"mcp-only"` / `"all"`) or legacy `mcp.discoveryMode === true`. The default `"auto"` mode is resolved later by `createAgentSession()` after MCP/extension tools are registered. 2. `description` is rendered from `packages/coding-agent/src/prompts/tools/search-tool-bm25.md` via `renderSearchToolBm25Description()`, using the current discoverable-tool list plus per-server summary/count. 3. `execute()` re-checks capability and settings: - missing discovery hooks -> `ToolError("Tool discovery is unavailable in this session.")` @@ -59,9 +59,10 @@ ## Modes / Variants - Discovery-mode gating: + - `tools.discoveryMode = "auto"` (default): when the registered tool set has more than 40 tools, searches hidden MCP tools only; otherwise discovery stays off. - `tools.discoveryMode = "all"`: searches hidden discoverable built-ins plus hidden MCP tools. - `tools.discoveryMode = "mcp-only"`: searches hidden MCP tools only. - - legacy `mcp.discoveryMode = true` with `tools.discoveryMode = "off"`: same as MCP-only. + - legacy `mcp.discoveryMode = true`: same as MCP-only. - Search-index source: - generic cached discoverable index from the session - legacy cached MCP index, cast to the generic shape @@ -113,6 +114,6 @@ - Built-in entries appear only in `"all"` mode and only for registry tools whose `loadMode === "discoverable"` and are not currently active. - Hidden/internal built-ins are intentionally excluded from the built-in corpus: `resolve`, `yield`, `report_finding`, `report_tool_issue` are called out in the `#collectDiscoverableBuiltinTools()` comment. - `DiscoverableToolSource` includes `"extension"` and `"custom"`, but `AgentSession.getDiscoverableTools()` currently assembles only built-in and MCP sources. -- On startup, `packages/coding-agent/src/sdk.ts` hides non-essential discoverable built-ins in `tools.discoveryMode = "all"`; defaults are `read`, `bash`, and `edit` unless `tools.essentialOverride` changes them. +- On startup, `packages/coding-agent/src/sdk.ts` resolves `"auto"` after the full registry exists and injects `search_tool_bm25` when the count exceeds 40. It hides non-essential discoverable built-ins only in `tools.discoveryMode = "all"`; defaults are `read`, `bash`, and `edit` unless `tools.essentialOverride` changes them. - Query tokenization is simple and deterministic: Unicode is NFKD-normalized, combining marks are dropped, acronym/camelCase and digit-to-capital boundaries are split, non-letter/non-number characters become spaces, tokens are lowercased, and only non-empty tokens survive. - Scores are rounded differently by surface: `details.tools[].score` keeps 6 decimals; the TUI line renders 3. diff --git a/docs/tools/web_search.md b/docs/tools/web_search.md index 5b3aa380b..c62df385d 100644 --- a/docs/tools/web_search.md +++ b/docs/tools/web_search.md @@ -161,9 +161,9 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - Output: `sources`, `requestId`. - **Kagi** — `packages/coding-agent/src/web/search/providers/kagi.ts`, `packages/coding-agent/src/web/kagi.ts` - Availability: env or `agent.db` credential for `kagi`. - - Querying: GET `https://kagi.com/api/v0/search?q=&limit=` with `Authorization: Bot `. + - Querying: POST `https://kagi.com/api/v1/search` with `Authorization: Bearer ` and JSON body `{ query, workflow: "search", limit, filters?: { after } }`. `recency` maps to `filters.after` as a UTC `YYYY-MM-DD` string (`day`/`week`/`month`/`year`). - `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..40`, default `10`. - - Output: `sources`, `relatedQuestions`, `requestId`. + - Output: `sources` (concatenated `data.search` + `data.video` + `data.news` + `data.infobox`, with video/news/infobox results tagged in the title), `relatedQuestions` (`data.adjacent_question` + `data.related_search` `props.question`), `answer` (`data.direct_answer[0].snippet ?? title`), `requestId` (`meta.trace`). - **Synthetic** — `packages/coding-agent/src/web/search/providers/synthetic.ts` - Availability: env or `agent.db` credential for `synthetic`. - Querying: POST `https://api.synthetic.new/v2/search` with `{ query }`. diff --git a/docs/tui-core-renderer.md b/docs/tui-core-renderer.md new file mode 100644 index 000000000..ba9705e8c --- /dev/null +++ b/docs/tui-core-renderer.md @@ -0,0 +1,389 @@ +# TUI core renderer — invariants & failure modes + +What you are dealing with before you touch the rendering engine. This is the +companion to [`tui-runtime-internals.md`](./tui-runtime-internals.md): that doc +maps the *flow* (input → component tree → render); this doc explains what +**does not work, why it keeps breaking, and the invariants you must not +violate**. Scope is the core engine only: + +- [`packages/tui/src/tui.ts`](../packages/tui/src/tui.ts) — render planner, intent emitters, native-scrollback bookkeeping, cursor placement. +- [`packages/tui/src/terminal.ts`](../packages/tui/src/terminal.ts) — `ProcessTerminal`, capability probes, private-CSI reassembly. +- [`packages/tui/src/terminal-capabilities.ts`](../packages/tui/src/terminal-capabilities.ts) — `TERMINAL` profile, ED3 risk / sync-output / DECCARA / image detection. +- [`packages/tui/src/stdin-buffer.ts`](../packages/tui/src/stdin-buffer.ts) — escape-sequence reassembly. +- [`packages/tui/src/utils.ts`](../packages/tui/src/utils.ts) — width/slice/wrap (the width model). +- [`packages/tui/src/kitty-graphics.ts`](../packages/tui/src/kitty-graphics.ts) + [`components/image.ts`](../packages/tui/src/components/image.ts) — inline images. +- [`packages/tui/src/deccara.ts`](../packages/tui/src/deccara.ts) — rectangular-fill optimizer. + +Application-layer renderers (transcript, tool calls, session tree, editor, +widgets) are **out of scope** — they live in `packages/coding-agent`. + +--- + +## 1. The one thing to understand first + +> **The renderer cannot observe the terminal's scroll position on most hosts it +> runs on.** Every decision about rewriting native scrollback is therefore a +> *guess*, and the guess has two opposite failure modes that cannot both be +> avoided by a single policy. + +We keep our transcript on the **normal screen**. We deliberately have not moved +the engine to the alternate screen: alt-screen would make the terminal handle +viewport isolation, but the transcript/resume affordances would disappear with +the alternate buffer. Keeping the normal screen means +*we* own native scrollback, which means we must decide, per frame, whether it is +safe to rebuild it. To rebuild history we emit xterm **ED3** (`CSI 3 J`, erase +saved lines). Deciding when ED3 is safe requires knowing whether the user has +scrolled up — and we usually can't: + +- **ConPTY hosts** (Windows Terminal, Tabby, Hyper, VS Code, conhost): the + pseudo-console buffer is pinned to the visible grid, so any "am I at the + bottom?" console query answers "yes" even when the reader scrolled up. The + probe *lies*. +- **POSIX terminals**: there is no scroll-position API at all. The probe is + *absent*. + +So `Terminal.isNativeViewportAtBottom()` returns `true` / `false` / **`undefined`**, +and `undefined` ("unknown") is the common case. The whole renderer is built +around not trusting `undefined`. + +### The two-way bind + +| If you guess… | …and you're wrong | Symptom | +|---|---|---| +| **Eager** (rebuild now → emit `CSI 3 J`) | reader was scrolled up | **YANK** to top + **FLASH** on terminals that snap scroll on ED3 | +| **Defer** (emit nothing, reconcile later) | viewport really was at the bottom | **CORRUPTION** (stale/duplicated rows) + **invisible-until-resize** | + +Yank, flash, and buffer corruption are **the same bug wearing three masks.** +Historically, every fix that suppressed one mask for one terminal class +re-enabled the opposite mask for a neighbouring class, and the follow-on +complaint landed within a day. If you "fix flashing" by making rebuilds more +eager, you will reintroduce yank. If you "fix yank" by deferring more, you will +reintroduce corruption / invisibility. **Do not move this lever without the +fidelity harness (§9) green.** + +--- + +## 2. The render-intent planner (what you are editing) + +`#doRender` is split into a **planner** (`#planRender`) that classifies a frame +into exactly one `RenderIntent`, and one `#emit*` method per intent that owns +the bytes written and the state update. All state flows through a single +`#commit` checkpoint at the end of every emitter. The intent union +(`tui.ts`, search `type RenderIntent`): + +| Intent | Emits | When | +|---|---|---| +| `noop` | cursor only | nothing visible changed | +| `initial` | clear viewport, paint transcript, **keep** prior shell scrollback | first paint after `start()` | +| `sessionReplace` | clear viewport **+ ED3** (outside multiplexers) | caller forced `{ clearScrollback: true }` (switch/branch/reload/resume) | +| `historyRebuild` | clear viewport **+ ED3** (outside multiplexers) | geometry change rewrapped history, or a proven-at-tail rebuild | +| `overlayRebuild` | rebuild viewport with overlay composite | overlay visibility changed | +| `liveRegionPinned` | relative moves + per-row rewrite/suffix-clear + `\r\n` | foreground streaming on an ED3-risk host, commit-as-you-go | +| `viewportRepaint` | rewrite the visible viewport in place (optional `appendFrom` tail first) | safe non-destructive repaint | +| `deferredShrink` | padded viewport repaint, history left dirty | bottom-anchored shrink, viewport unobservable | +| `deferredMutation` | **zero bytes**, history left dirty | row-reindexing edit while possibly scrolled | +| `shrink` / `diff` | trailing-row clear / changed-line diff | ordinary in-place updates | + +**ED3 (`CSI 3 J`) is emitted in exactly one place** — `#emitFullPaint` when +`clearScrollback: true` (`\x1b[2J\x1b[H\x1b[3J`). The ordinary clear is +**non-destructive**: `\x1b[22J` (copy-screen-to-scrollback, only when +`TERMINAL.supportsScreenToScrollback`) then `\x1b[2J\x1b[H`, **no `3J`**. ED3 is +reached only by `sessionReplace`/`historyRebuild`/`overlayRebuild`, and those +suppress the scrollback clear inside multiplexers (`isMultiplexerSession()` = +`TMUX || STY || ZELLIJ`). + +### The predicate gates + +Three private predicates encode the guessing policy. Do not "simplify" them — +each branch is load-bearing: + +- `#canReplayNativeScrollbackAtCheckpoint(atBottom)` → `atBottom === true`. A + rebuild at a **keystroke checkpoint** (prompt submit) is allowed only with a + *positive* at-tail proof. A prompt submit is **no longer** treated as implicit + proof for an unobservable host. +- `#canRebuildNativeScrollbackLive(atBottom, allowUnknown)` → `true` iff + `atBottom === true`, **or** (`atBottom === undefined && allowUnknown && + platform !== "win32"`). i.e. live ED3 during streaming requires either proof + or an explicit direct-user-input opt-in, and **never** on win32. +- `#nativeViewportIsScrolled(atBottom, allowUnknown)` → `true` if + `atBottom === false`, or (`undefined && win32 && !allowUnknown`). Used to + decide deferral. + +`allowUnknownViewportMutation` is the **direct-user-input opt-in** (autocomplete +/ IME / a keystroke the user just typed). A keystroke pins the host viewport to +the bottom, so it is safe to repaint live then. It is **not** set by passive +streaming. `setEagerNativeScrollbackRebuild(true)` is the streaming opt-in; on +ED3-risk hosts it is downgraded so it never promotes to a live ED3 clear. + +### Deferral + checkpoint discipline + +When the viewport is unobservable during **passive streaming**, the planner +defers (`deferredMutation`/`deferredShrink`/`viewportRepaint`) and marks native +scrollback dirty (`#markNativeScrollbackDirty()`). Reconciliation happens later +at a checkpoint via `refreshNativeScrollbackIfDirty()` — and only if +`#canReplayNativeScrollbackAtCheckpoint` proves at-tail. The streaming-defer + +live-region-pin seam (`NativeScrollbackLiveRegion`, +`getNativeScrollbackLiveRegionStart` / `getNativeScrollbackCommitSafeEnd`) is the +**actively-churning** part of the engine; if you change how transient rows are +committed, every structural-mutation branch (shrink **and** grow/offscreen-edit) +must defer **symmetrically**, or you reopen the corruption family. + +--- + +## 3. The five fault families + +### YANK — viewport snapped to top — NOT fully converged +- **Mechanism:** a live `historyRebuild` fires `CSI 3 J` while the reader is + scrolled up; ED3-snap terminals reset the visible viewport to the top of the + (now-erased) scrollback. +- **Trigger to avoid:** treating an unobservable probe as "at bottom" during + *passive* streaming, or OR-ing an eager-streaming flag into the live ED3 path. +- **Current stance:** never emit ED3 on an unobservable host during passive + streaming; defer and reconcile at a keystroke checkpoint. ConPTY/win32 never + trust the probe at all. + +### CORRUPTION — duplicated / stale rows — NOT fully converged +- **Mechanism:** the flip side of the yank fix. A deferred/repainted frame + leaves rows already committed to native scrollback out of sync with the live + viewport; the scrollback↔viewport seam duplicates (e.g. a 2-row dup, a + streaming-tail dup, or an async-expansion dup). +- **Trigger to avoid:** repainting the viewport over scrollback that still holds + the old copy; a frozen/deferred block whose snapshot no longer matches after + the region above it reflowed; one mutation branch deferring while its mirror + branch repaints. +- **Current stance:** commit only the **stable prefix** line-count to native + history; keep unstable rows out; reconcile drift at the checkpoint; park the + hardware cursor at real content bottom, not padded bottom. + +### FLASH (and invisible-until-resize) — NOT fully converged +- **Two distinct causes, one symptom:** + - *Flash* = eager ED3 rebuild wrapped in DEC 2026 BSU/ESU fired per streaming + frame on a terminal that clamps scroll on ED3 (VTE/GNOME family). + - *Invisible-until-resize* = the defer fix over-firing, so a structural frame + emits **zero bytes** (`deferredMutation` returns nothing) until a resize + forces a repaint. +- **Trigger to avoid:** env-detection that misses a flashing terminal (SSH + strips `VTE_VERSION`; some hosts set no distinguishing var); collapsing an + `undefined` probe into a definite scrolled/at-bottom verdict. +- **Current stance:** confine ED3 to the destructive path; auto-disable DEC 2026 + at runtime when the terminal reports it unsupported (DECRQM), with + `PI_NO_SYNC_OUTPUT` as a manual hatch; keep autowrap discipline regardless. + +### WIDTH — measurement crashes / fidelity — crash class dead, accuracy unproven +- **Mechanism:** the measured column width of a line disagreed with the + terminal's painted cells (emoji, wide graphemes, combining marks, Hangul + jamo), and the old render loop **threw** on any mismatch — a 1-cell cosmetic + error became a fatal whole-agent crash. +- **Current stance:** **never throw in the render hot path — clamp.** The loop + truncates over-wide lines with `truncateToWidth`/`sliceByColumn` and logs + (under debug) instead of dying. Width is owned end-to-end by one native UAX#11 + engine shared by measure/slice/wrap (see §6). Accuracy across all scripts + (e.g. RTL/combining marks) is still not proven by a green gate. + +### PROBE — stray bytes injected as keystrokes — RESOLVED +- **Mechanism:** a private-CSI probe reply (DA1 / kitty / mode 2031) split + across a stdin flush; the unmatched prefix was dropped and the continuation + bytes were forwarded as keystrokes. +- **Current stance:** buffer-and-reassemble partial CSI responses; give each + probe a typed sentinel owner. This is the **one cleanly-closed family** — + because its contract is *bounded and observable* (bytes in = bytes out), + unlike the unobservable-viewport families. See §7. + +--- + +## 4. Invariants — MUST / NEVER + +These are the rules the recurrence taught us. Treat them as load-bearing. + +1. **NEVER add a new `CSI 3 J` (ED3) callsite.** ED3 must flow only through + `#emitFullPaint({ clearScrollback: true })`, for the existing destructive + intents (`sessionReplace`, proven/safe `historyRebuild`, `overlayRebuild`). + Ordinary redraws use the non-destructive `\x1b[22J` + `\x1b[2J\x1b[H` clear. +2. **NEVER trust an unobservable viewport probe (`undefined`) for *passive* + streaming.** Only a positive at-tail proof, or a direct-user-input opt-in + (`allowUnknownViewportMutation`), authorizes a live rebuild — and never on + win32/ConPTY. +3. **NEVER throw in the render hot path.** Clamp over-wide lines; a width + mismatch is cosmetic, not fatal. +4. **NEVER let a defer path emit a structurally-changed frame as zero bytes + while at the bottom** — that is invisible-until-resize. `deferredMutation`/ + `deferredShrink` are only safe when the viewport is (or may be) scrolled. +5. **Defer symmetrically.** If one structural-mutation branch (shrink) defers on + an unobservable ED3-risk host, the mirror branch (grow / offscreen-edit) must + too. Asymmetry reopens corruption. +6. **Commit only the stable prefix to native history.** Transient/unsettled rows + stay out of scrollback until a checkpoint; reconcile drift at the checkpoint. +7. **Park the hardware cursor at real content bottom**, not the padded viewport + bottom, or height shrinks scroll live rows into scrollback and duplicate them + per resize step. +8. **Cursor writes live *inside* the synchronized-output frame**, before ESU — + never as a second frame after it (that teleports/blinks the caret). +9. **Detect terminal *risk*, not terminal *brand*, and default unknown to + risky.** Env sniffing is necessarily incomplete (see §5); never assume an + un-enumerated host is safe. +10. **Multiplexers (tmux/screen/zellij) get no destructive scrollback clear and + no viewport probe.** ED3 is a no-op there and a full replay duplicates the + transcript; repaint in place and rely on the pinned/commit-as-you-go path. +11. **Any change to the eager/defer lever, the predicates, or the live-region + seam must be validated by the render-stress fidelity harness (§9)** across + `{win32, POSIX} × {unknown, scrolled, at-bottom}`, not by a single-terminal + smoke test. + +--- + +## 5. Terminal capability detection (and why it is fragile) + +`TERMINAL` (`terminal-capabilities.ts`) is resolved once at import from +`TERMINAL_ID` plus environment sniffing. The detection helpers are pure and +parameterized over `(env, platform)` so they are unit-testable: + +- `detectTerminalEagerEraseScrollbackRisk(env, platform)` → is a live ED3 + rebuild unsafe here? Current policy: `false` on win32 (dedicated ConPTY + deferral paths handle it) and when `PI_TUI_ED3_SAFE=1`; otherwise **`true`** + for `WT_SESSION` (WT fronting WSL), SSH/tmux/screen/zellij, known + ED3-snap/scrollback-clearing terminals (WezTerm, kitty, ghostty, alacritty, + VTE, iTerm2, Apple Terminal, GNOME Terminal, Ptyxis, xfce4-terminal), Linux + truecolor, **and every other unknown POSIX terminal**. The default is *risky* + on purpose. +- `shouldEnableSynchronizedOutputByDefault(env, id)` → DEC 2026 default. Precedence: + user opt-out (`PI_NO_SYNC_OUTPUT`/`PI_TUI_SYNC_OUTPUT=0`) → user force-on + (`PI_FORCE_SYNC_OUTPUT=1`/`PI_TUI_SYNC_OUTPUT=1`) → `TERM_FEATURES` advertises + `Sy` → `WT_SESSION` (WT/WSL) → known direct terminals + (kitty/ghostty/wezterm/iterm2/alacritty/vscode; SSH passes through) → off for + risky multiplexers and everything else (VTE-family, GNU screen, Apple Terminal, + legacy conhost, unknown). Reconciled at runtime by the DECRQM mode-2026 report: + a positive report **enables** sync (upgrading default-off muxes like + zellij/tmux-master), a negative one disables it; a user override still wins. + `synchronizedOutputUserOverride(env)` is the shared opt-out/force resolver. +- `detectRectangularSgrSupport(id, env)` → DECCARA fills: **kitty only** + (ghostty does not implement the SGR-background extension), off in multiplexers + and under `PI_NO_DECCARA`. + +**Why this keeps leaking:** terminal class is inferred from env vars that are +**not durable**. `VTE_VERSION` is stripped by `sshd` (default `AcceptEnv`); +`COLORTERM` is also not in default `AcceptEnv`; some hosts (Tabby) set no +distinguishing var; WSL-fronting-WT is neither pure win32 nor pure POSIX. Every +missed env var is a missed terminal class is a new complaint. The mitigations +are: (a) **default unknown to risky** rather than safe, and (b) detect by +*behavior/handshake* (DECRQM) where possible rather than a host allow-list. When +you add a terminal, add it to the pure detector and add the **SSH-stripped env +shape** to the test, not just the env-present shape. + +--- + +## 6. Width model + +`visibleWidth` / `truncateToWidth` / `sliceByColumn` / `wrapTextWithAnsi` +(`utils.ts`) all route through **one native UAX#11 engine** (`@oh-my-pi/pi-natives`, +Rust `unicode-width`). We deliberately dropped `Bun.stringWidth` because it +disagreed with the engine on combining marks and jamo, and mixing two width +models in measure-vs-slice produced the crashes. + +- Fast path: printable ASCII is one cell per code unit. +- ZWJ pictographic emoji take the `visibleWidthByGrapheme` override (ANSI spans + excised first, then `Intl.Segmenter`), because the native scanner double-counts + SGR bytes when a sequence is split by the segmenter. +- OSC 66 sized text (`\x1b]66;…`) takes the native path. + +**Rule:** if you add a code path that measures width, route it through these +helpers. Never reintroduce `Bun.stringWidth` or a parallel width table — the +measure model and the slice/wrap model must agree, or you get over-wide lines +that the hot-path clamp silently truncates (cosmetic loss) or, worse, seam +duplication. + +--- + +## 7. Capability probes & stdin reassembly + +`ProcessTerminal` fuses capability queries with a bare DA1 (`CSI c`) sentinel so +a non-answering terminal is detected when DA1 returns first. Replies can arrive +**split across a stdin flush**, so: + +- `#privateCsiResponseBuffer` accumulates `\x1b[?…` partials while a sentinel is + outstanding, rejoins on the terminator byte (0x40–0x7e), then runs the + DA1/kitty/mode-2031 handlers on the **complete** reply. A new `\x1b` + mid-reassembly or >256 bytes abandons the partial so real keys (e.g. arrow + `\x1b[A`) still reach input. +- `#da1SentinelOwners` is a **typed FIFO** discriminated by `kind` (`keyboard`, + `osc11`, `privateMode`, `kittyGraphicsProbe`, `osc99Probe`) so a keyboard DA1 + cannot be mistaken for an OSC 11 / DECRQM / graphics-probe sentinel. +- DECRQM probes (`#queryPrivateMode(2026/2048/2031)`) record support via DECRPM + and drive runtime feature gating (e.g. auto-disabling DEC 2026 sync output). + +**Rule:** any new probe must own a typed sentinel and survive a split reply. The +contract is bytes-in = bytes-out; it is testable, so test it (feed the reply +byte-by-byte and assert nothing leaks to the input handler). + +--- + +## 8. Inline images & memory + +Kitty images are **transmit-once, place-many** (`kitty-graphics.ts`): +`encodeKittyTransmit` (`a=t`, keyed by a stable `i=`) writes the base64 a single +time; repaints emit only `encodeKittyPlacement` (`a=p`). Text clears +(`CSI 2 J` / `CSI 3 J`) do **not** purge the terminal's image store — only +`encodeKittyDeleteImage` (`a=d,d=I`) does. `ImageBudget` (`components/image.ts`) +keeps only the most-recent N images live; demoted images render their text +fallback and are explicitly purged. + +**Rule:** never re-emit full base64 per frame (it pegged RAM and pinned the UI +thread). Kitty Unicode placeholders are default-on only for kitty/ghostty +(`PI_NO_KITTY_PLACEHOLDERS` / `PI_KITTY_PLACEHOLDERS`); other Kitty-protocol +hosts render placeholder cells as literal PUA glyphs, so they fall back to +direct `a=p` placement. + +--- + +## 9. The fidelity gate (use it) + +`packages/tui/test/render-stress-harness.ts` renders the renderer's **real emitted ANSI** into +a ghostty-web `VirtualTerminal` and asserts viewport fidelity (a scrolled reader +stays put), background-column fidelity, and scrollback-buffer fidelity, across +parameterized terminal shapes and randomized op sequences. + +This harness is the structural fix for the whole recurrence: every guess-flip and +sniffing-gap regression historically **shipped blind and was caught by a user**, +because no automated "a scrolled-up reader stays pinned across kitty/WT/WSL/ +ConPTY" assertion gated CI. **Before you change the eager/defer lever, a +predicate, the live-region seam, or width math, run the stress harness and the +targeted repro tests** (`packages/tui/test/render-regressions.test.ts`, +`packages/tui/test/streaming-scrollback-defer.test.ts`, the `issue-*-repro.test.ts` files). +A change that passes one terminal and one seed is not verified. + +--- + +## 10. Escape hatches (env vars) + +| Var | Effect | +|---|---| +| `PI_NO_SYNC_OUTPUT=1` | Disable DEC 2026 BSU/ESU wrappers (autowrap discipline stays on). For terminals that advertise but mishandle mode 2026. | +| `PI_TUI_SYNC_OUTPUT=0\|1` / `PI_FORCE_SYNC_OUTPUT=1` | Force sync output off / on. | +| `PI_TUI_ED3_SAFE=1` | Declare the terminal safe for live ED3 (disables `eagerEraseScrollbackRisk`). | +| `PI_NO_DECCARA` | Disable Kitty DECCARA rectangular-fill optimization (force padded-string fills). | +| `PI_FORCE_IMAGE_PROTOCOL=kitty\|iterm2\|sixel\|off` | Override image protocol detection. | +| `PI_NO_KITTY_PLACEHOLDERS=1` / `PI_KITTY_PLACEHOLDERS=1` | Force Kitty Unicode placeholders off / on. | +| `PI_CLEAR_ON_SHRINK=1` | Clear empty rows when content shrinks (default off). | +| `PI_HARDWARE_CURSOR=1` | Show the real hardware cursor instead of a rendered one. | +| `PI_NOTIFICATIONS=off\|0\|false` | Suppress terminal notifications. | +| `PI_DEBUG_REDRAW=1` | Log the chosen render intent per frame to the debug log. | +| `PI_TUI_DEBUG=1` | Dump per-render diff state under `/tmp/tui`. | + +--- + +## 11. Before you touch the render core — checklist + +- [ ] Are you about to emit `CSI 3 J` anywhere other than the destructive + `clearScrollback` path? **Stop.** +- [ ] Does your change trust `isNativeViewportAtBottom() === undefined` as + "at bottom" during passive streaming? **Stop.** +- [ ] Did you change one structural-mutation branch without mirroring its + sibling (shrink ↔ grow)? **Defer symmetrically.** +- [ ] Could any frame now emit zero bytes while the viewport is at the bottom? + That's invisible-until-resize. +- [ ] Did you add a terminal by brand instead of by behavior, or skip the + SSH-stripped env shape in the test? +- [ ] Did you run `packages/tui/test/render-stress-harness.ts` + the repro suite across + win32/POSIX × unknown/scrolled/at-bottom — not just one terminal? +- [ ] New probe? Typed sentinel owner + split-reply test. +- [ ] New width path? Routed through the shared native engine, clamped (never + thrown) in the hot path. diff --git a/docs/tui-runtime-internals.md b/docs/tui-runtime-internals.md index e5ef9e32d..79df9186d 100644 --- a/docs/tui-runtime-internals.md +++ b/docs/tui-runtime-internals.md @@ -2,6 +2,12 @@ This document maps the non-theme runtime path from terminal input to rendered output in interactive mode. It focuses on behavior in `packages/tui` and its integration from `packages/coding-agent` controllers. +> **Editing the rendering engine itself?** Read +> [`tui-core-renderer.md`](./tui-core-renderer.md) first — it documents the +> failure modes (yank / corruption / flash / width crashes) and the invariants +> the render planner, native-scrollback bookkeeping, and capability detection +> must not violate. + ## Runtime layers and ownership - **`packages/tui` engine**: terminal lifecycle, stdin normalization, focus routing, render scheduling, differential painting, overlay composition, hardware cursor placement. @@ -11,44 +17,49 @@ Boundary rule: the TUI engine is message-agnostic. It only knows `Component.rend ## Implementation files -- [`../src/modes/interactive-mode.ts`](../packages/coding-agent/src/modes/interactive-mode.ts) -- [`../src/modes/controllers/event-controller.ts`](../packages/coding-agent/src/modes/controllers/event-controller.ts) -- [`../src/modes/controllers/input-controller.ts`](../packages/coding-agent/src/modes/controllers/input-controller.ts) -- [`../src/modes/components/custom-editor.ts`](../packages/coding-agent/src/modes/components/custom-editor.ts) -- [`../../tui/src/tui.ts`](../packages/tui/src/tui.ts) -- [`../../tui/src/terminal.ts`](../packages/tui/src/terminal.ts) -- [`../../tui/src/editor-component.ts`](../packages/tui/src/editor-component.ts) -- [`../../tui/src/stdin-buffer.ts`](../packages/tui/src/stdin-buffer.ts) -- [`../../tui/src/components/loader.ts`](../packages/tui/src/components/loader.ts) +- [`packages/coding-agent/src/modes/interactive-mode.ts`](../packages/coding-agent/src/modes/interactive-mode.ts) +- [`packages/coding-agent/src/modes/controllers/event-controller.ts`](../packages/coding-agent/src/modes/controllers/event-controller.ts) +- [`packages/coding-agent/src/modes/controllers/input-controller.ts`](../packages/coding-agent/src/modes/controllers/input-controller.ts) +- [`packages/coding-agent/src/modes/components/custom-editor.ts`](../packages/coding-agent/src/modes/components/custom-editor.ts) +- [`packages/tui/src/tui.ts`](../packages/tui/src/tui.ts) +- [`packages/tui/src/terminal.ts`](../packages/tui/src/terminal.ts) +- [`packages/tui/src/editor-component.ts`](../packages/tui/src/editor-component.ts) +- [`packages/tui/src/stdin-buffer.ts`](../packages/tui/src/stdin-buffer.ts) +- [`packages/tui/src/components/loader.ts`](../packages/tui/src/components/loader.ts) ## Boot and component tree assembly -`InteractiveMode` constructs `TUI(new ProcessTerminal(), settings.get("showHardwareCursor"))`, applies `settings.get("clearOnShrink")`, and creates persistent containers: +`InteractiveMode` constructs `TUI(new ProcessTerminal(), settings.get("showHardwareCursor"))`, applies `clearOnShrink`, `tui.maxInlineImages`, and Kitty text-sizing settings, then creates persistent containers: - `chatContainer` - `pendingMessagesContainer` - `statusContainer` - `todoContainer` - `btwContainer` +- `omfgContainer` +- `errorBannerContainer` - `statusLine` - `hookWidgetContainerAbove` - `editorContainer` (holds `CustomEditor`) - `hookWidgetContainerBelow` -`init()` wires the tree in that order, focuses the editor, registers input handlers via `InputController`, subscribes terminal appearance changes into theme auto-detection, starts TUI, and requests a forced render. -A forced render (`requestRender(true)`) resets previous-line caches and cursor bookkeeping before repainting. +`init()` wires the tree in that order after any startup warnings/welcome/changelog, focuses the editor, registers input handlers via `InputController`, starts TUI, pushes terminal title state, updates the editor border, and requests a forced render. +A forced render (`requestRender(true)`) queues a viewport repaint or explicit session replacement; it does **not** throw away previous-line history by default. ## Terminal lifecycle and stdin normalization `ProcessTerminal.start()`: 1. Enables raw mode and bracketed paste. -2. Attaches resize handler. -3. Creates a `StdinBuffer` to split partial escape chunks into complete sequences. -4. Queries Kitty keyboard protocol support (`CSI ? u`), then enables protocol flags if supported; otherwise enables modifyOtherKeys fallback after a short timeout. -5. Queries OSC 11 background color and enables Mode 2031 appearance notifications for dark/light theme detection. -6. On Windows, attempts VT input enablement via `kernel32` mode flags. - `StdinBuffer` behavior: +2. Attaches resize handler and refreshes dimensions. +3. Enables Windows VT input mode when running on win32. +4. Creates a `StdinBuffer` to split partial escape chunks into complete sequences. +5. Queries Kitty keyboard protocol support (`CSI ? u`), then enables protocol flags if supported; otherwise enables modifyOtherKeys fallback after a short timeout. +6. Queries OSC 11 background color and Mode 2031 appearance notifications for dark/light theme detection. +7. Queries OSC 99 notification capabilities. +8. Starts periodic OSC 11 polling only where safe, then probes DEC private modes 2026/2048/2031 via DECRQM. + +`StdinBuffer` behavior: - Buffers fragmented escape sequences (CSI/OSC/DCS/APC/SS3). - Emits `data` only when a sequence is complete or timeout-flushed. @@ -90,7 +101,7 @@ This keeps key parsing/editor mechanics in `packages/tui` and mode semantics in `TUI.requestRender()` coalesces render requests and rate-limits ordinary frames: -- forced renders (`requestRender(true, ...)`) reset cached frame/viewport state and run on `process.nextTick` +- forced renders (`requestRender(true, ...)`) schedule an immediate frame and set `#forceViewportRepaintOnNextRender`; with `clearScrollback`, they also queue `sessionReplace` - ordinary renders schedule through `#scheduleRender()` and respect `TUI.#MIN_RENDER_INTERVAL_MS` - repeated requests while a render is pending collapse into the same scheduled frame @@ -110,7 +121,7 @@ This keeps key parsing/editor mechanics in `packages/tui` and mode semantics in - noop 6. Emit only the bytes required by the intent and commit cached frame/cursor/viewport state. -Render writes use synchronized output mode (`CSI ? 2026 h/l`) to reduce flicker/tearing. +Render writes use synchronized output mode (`CSI ? 2026 h/l`) when enabled; capability detection, DECRQM, or `PI_NO_SYNC_OUTPUT` can disable the wrappers while leaving autowrap discipline on. ## Render safety constraints @@ -123,14 +134,19 @@ Critical safety checks in `TUI`: These constraints are runtime guards plus component conventions; renderers should still return width-safe lines rather than rely on truncation. +The deeper reasons these guards exist — why the renderer cannot observe scroll +position, why ED3 (`CSI 3 J`) is confined to one path, and why the hot path +clamps instead of throwing — are documented in +[`tui-core-renderer.md`](./tui-core-renderer.md). + ## Resize handling Resize events are event-driven from `ProcessTerminal` to `TUI.requestRender()`. Effects: -- Width changes repaint or rebuild because wrapping semantics change. -- Height-only changes repaint the viewport when needed, but skip repaint in Termux and terminal multiplexers where replays are scrollback-hostile. +- Width or height changes repaint or rebuild because terminal reflow invalidates wrapping, viewport, and cursor anchors. +- Inside terminal multiplexers, resize uses viewport repaint instead of destructive native-scrollback replay; pane history cannot be erased safely and a full replay duplicates transcript rows. - Viewport/top tracking (`#viewportTopRow`, `#maxLinesRendered`, scrollback high-water state) avoids invalid relative cursor math and defers destructive native scrollback rewrites while the user is scrolled into history. - Overlay visibility can depend on terminal dimensions (`OverlayOptions.visible`); focus is corrected when overlays become non-visible after resize. @@ -183,18 +199,6 @@ Escape exits inactive mode by clearing editor text and restoring border color; w 2. Stops TUI before suspend. 3. Sends `SIGTSTP` to process group. -### Background mode (`/background` or `/bg`) - -`handleBackgroundCommand()`: - -- Rejects when idle. -- Switches tool UI context to non-interactive (`hasUI=false`) so interactive UI tools fail fast. -- Stops loaders/status line and unsubscribes foreground event handler. -- Subscribes background event handler (primarily waits for `agent_end`). -- Stops TUI and sends `SIGTSTP` (POSIX job control path). - -On `agent_end` in background with no queued work, controller sends completion notification and shuts down. - ## Cancellation paths Primary cancellation inputs: diff --git a/docs/tui.md b/docs/tui.md index a62b5f0fe..cba829fbe 100644 --- a/docs/tui.md +++ b/docs/tui.md @@ -28,7 +28,7 @@ export interface Component { render(width: number): string[]; handleInput?(data: string): void; wantsKeyRelease?: boolean; - invalidate(): void; + invalidate?(): void; } ``` diff --git a/package.json b/package.json index 7f85dbc86..435a43242 100644 --- a/package.json +++ b/package.json @@ -20,15 +20,15 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "15.9.4", - "@oh-my-pi/omp-stats": "15.9.4", - "@oh-my-pi/pi-agent-core": "15.9.4", - "@oh-my-pi/pi-ai": "15.9.4", - "@oh-my-pi/pi-coding-agent": "15.9.4", - "@oh-my-pi/pi-mnemopi": "15.9.4", - "@oh-my-pi/pi-natives": "15.9.4", - "@oh-my-pi/pi-tui": "15.9.4", - "@oh-my-pi/pi-utils": "15.9.4", + "@oh-my-pi/hashline": "15.10.1", + "@oh-my-pi/omp-stats": "15.10.1", + "@oh-my-pi/pi-agent-core": "15.10.1", + "@oh-my-pi/pi-ai": "15.10.1", + "@oh-my-pi/pi-coding-agent": "15.10.1", + "@oh-my-pi/pi-mnemopi": "15.10.1", + "@oh-my-pi/pi-natives": "15.10.1", + "@oh-my-pi/pi-tui": "15.10.1", + "@oh-my-pi/pi-utils": "15.10.1", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -85,14 +85,15 @@ }, "overrides": {}, "scripts": { - "install:dev": "bun install && bun --cwd=packages/coding-agent link && bun --cwd=packages/ai link", + "install:dev": "bun install && bun --cwd=packages/coding-agent link && ln -sfn \"$(pwd)/packages/coding-agent/scripts/dev-launch\" \"$(bun pm -g bin)/omp\"", "dev": "bun --cwd=packages/coding-agent src/cli.ts", + "dev:timing": "PI_TIMING=x bun --cwd=packages/coding-agent --preload ../utils/src/module-timer.ts src/cli.ts", "stats": "bun --cwd=packages/coding-agent src/cli.ts stats", "claude:trace": "bun scripts/claude-trace.ts", "build": "bun run --workspaces --if-present build", "build:native": "bun --cwd=packages/natives run build", "test": "bun run --parallel test:ts test:rs", - "test:ts": "GITHUB_ACTIONS=0 bun run --workspaces --if-present test -- --only-failures", + "test:ts": "GITHUB_ACTIONS= bun run --workspaces --if-present test -- --only-failures", "test:rs": "bun scripts/run-rs-task.ts test:rs", "check": "bun run --parallel check:ts check:rs", "check:ts": "bun run check:tools && bun run --workspaces --if-present check", diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 22d853150..5e4dd0b4f 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,26 @@ ## [Unreleased] +## [15.10.1] - 2026-06-07 + +### Added + +- Added optional `promptCacheKey` support to `AgentOptions` and `Agent` via a new `promptCacheKey` property so providers can receive a caller-provided prompt cache key +- Added optional `ApiKeyResolveContext` parameter to `getApiKey` in `AgentOptions` and `AgentLoopConfig` so key resolvers can receive retry context + +### Changed + +- Enabled streaming API calls to re-resolve credentials through the `getApiKey` callback when retries occur after authentication-related errors +- `Agent.abort(reason?)` now forwards `reason` to the underlying `AbortController`, and the synthesized aborted assistant message carries that reason on `errorMessage` (string or non-`AbortError` `Error` message) instead of always defaulting to `"Request was aborted"`. Bare `abort()` is unchanged. + +### Fixed + +- Fixed handling of short-lived API keys so that expired tokens are retried with a refreshed value during 401/usage-limit failures +- Ensured fallback API key resolution uses the initially configured static `apiKey` when `getApiKey` is present +- Wrapped oneshot LLM completions (`instrumentedCompleteSimple`: handoff, compaction/branch summaries) in an `EventLoopKeepalive`. These run outside the agent `#runLoop`, so without the keepalive Bun's event loop stopped servicing timers while parked on the completion promise — freezing host spinners (e.g. the `/handoff` loader) until an unrelated terminal resize poked the loop into rendering again. + +## [15.9.5] - 2026-06-05 + ### Fixed - Surfaced Anthropic stream failures whose message starts with `Output blocked by conten` as normal assistant error lifecycle events, so interactive clients render content-filter blocks instead of silently dropping the streaming bubble at `agent_end`. diff --git a/packages/agent/package.json b/packages/agent/package.json index cb7560996..6d256bd40 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "15.9.4", + "version": "15.10.1", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index 565420527..c4bddb3cd 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -3,6 +3,7 @@ * Transforms to Message[] only at the LLM call boundary. */ import { + type ApiKeyResolveContext, type AssistantMessage, type AssistantMessageEvent, type Context, @@ -727,8 +728,9 @@ async function streamAssistantResponse( // Resolve API key (important for expiring tokens) — do this before resolving // metadata so that the session-sticky credential recorded by getApiKey is // visible to metadataResolver (e.g. for the correct account_uuid in metadata.user_id). + const staticApiKey = typeof config.apiKey === "string" ? config.apiKey : undefined; const resolvedApiKey = - (config.getApiKey ? await config.getApiKey(config.model.provider) : undefined) || config.apiKey; + (config.getApiKey ? await config.getApiKey(config.model.provider) : undefined) || staticApiKey; // Re-resolve metadata after credential selection so the per-request value // reflects the credential actually used, not the snapshot from AgentLoopConfig construction. @@ -798,7 +800,19 @@ async function streamAssistantResponse( return await runInActiveSpan(chatSpan, async () => { const response = await streamFunction(config.model, llmContext, { ...config, - apiKey: resolvedApiKey, + // Hand streamSimple a resolver so its central auth-retry policy can + // re-resolve on 401 / usage-limit: the initial step reuses the key + // already resolved above (which set the session-sticky credential + // feeding metadataResolver), and retry steps forward the a/b/c ctx + // to config.getApiKey (force-refresh, then rotate). With no + // getApiKey hook the caller's own apiKey (string or resolver) flows + // through unchanged. + apiKey: config.getApiKey + ? (ctx: ApiKeyResolveContext) => + ctx.error === undefined + ? resolvedApiKey + : Promise.resolve(config.getApiKey!(config.model.provider, ctx)) + : config.apiKey, metadata: resolvedMetadata, toolChoice: effectiveToolChoice, reasoning: effectiveReasoning, @@ -839,7 +853,14 @@ async function streamAssistantResponse( let detachAbortListener: (() => void) | undefined; if (requestSignal) { if (requestSignal.aborted) { - const aborted = emitAbortedAssistantMessage(partialMessage, addedPartial, context, config, stream); + const aborted = emitAbortedAssistantMessage( + partialMessage, + addedPartial, + context, + config, + stream, + requestSignal, + ); await finishChat(aborted); return aborted; } @@ -861,7 +882,14 @@ async function streamAssistantResponse( if (capped) return capped; } responseIterator.return?.()?.catch(() => {}); - const aborted = emitAbortedAssistantMessage(partialMessage, addedPartial, context, config, stream); + const aborted = emitAbortedAssistantMessage( + partialMessage, + addedPartial, + context, + config, + stream, + requestSignal, + ); await finishChat(aborted); return aborted; } @@ -874,7 +902,14 @@ async function streamAssistantResponse( const capped = await finishCappedAssistantMessage(); if (capped) return capped; } - const aborted = emitAbortedAssistantMessage(partialMessage, addedPartial, context, config, stream); + const aborted = emitAbortedAssistantMessage( + partialMessage, + addedPartial, + context, + config, + stream, + requestSignal, + ); await finishChat(aborted); return aborted; } @@ -982,14 +1017,30 @@ async function streamAssistantResponse( } } +/** Resolve the human-readable reason an abort carried. A caller that aborts via + * `AbortController.abort(reason)` with a string or a non-`AbortError` `Error` + * (e.g. the coding agent's user-interrupt label) gets that text surfaced on the + * synthesized assistant message's `errorMessage`; a bare `abort()` (whose + * `signal.reason` is the default `AbortError` `DOMException`) falls back to the + * generic sentinel that downstream renderers treat as "no specific reason". */ +export function abortReasonText(signal: AbortSignal | undefined): string { + const reason = signal?.reason; + if (typeof reason === "string" && reason.trim().length > 0) return reason; + if (reason instanceof Error && reason.name !== "AbortError" && reason.message.trim().length > 0) { + return reason.message; + } + return "Request was aborted"; +} + function emitAbortedAssistantMessage( partialMessage: AssistantMessage | null, addedPartial: boolean, context: AgentContext, config: AgentLoopConfig, stream: EventStream, + requestSignal: AbortSignal | undefined, ): AssistantMessage { - const errorMessage = "Request was aborted"; + const errorMessage = abortReasonText(requestSignal); const abortedMessage: AssistantMessage = partialMessage ? { ...partialMessage, stopReason: "aborted", errorMessage } : { diff --git a/packages/agent/src/agent.ts b/packages/agent/src/agent.ts index 1b3c57f46..99496de45 100644 --- a/packages/agent/src/agent.ts +++ b/packages/agent/src/agent.ts @@ -3,6 +3,7 @@ */ import { isPromise } from "node:util/types"; import { + type ApiKeyResolveContext, type AssistantMessage, type AssistantMessageEvent, type CursorExecHandlers, @@ -21,7 +22,7 @@ import { type ToolChoice, type ToolResultMessage, } from "@oh-my-pi/pi-ai"; -import { agentLoop, agentLoopContinue } from "./agent-loop"; +import { abortReasonText, agentLoop, agentLoopContinue } from "./agent-loop"; import type { AppendOnlyContextManager } from "./append-only-context"; import type { HarmonyAuditEvent } from "./harmony-leak"; import type { @@ -132,6 +133,11 @@ export interface AgentOptions { * Used by providers that support session-based caching (e.g., OpenAI Codex). */ sessionId?: string; + /** + * Optional prompt cache key forwarded to LLM providers. + * When omitted, providers may fall back to sessionId. + */ + promptCacheKey?: string; /** * Shared provider state map for session-scoped transport/session caches. */ @@ -141,7 +147,7 @@ export interface AgentOptions { * Resolves an API key dynamically for each LLM call. * Useful for expiring tokens (e.g., GitHub Copilot OAuth). */ - getApiKey?: (provider: string) => Promise | string | undefined; + getApiKey?: (provider: string, ctx?: ApiKeyResolveContext) => Promise | string | undefined; /** * Inspect or replace provider payloads before they are sent. @@ -283,6 +289,7 @@ export class Agent { #interruptMode: "immediate" | "wait"; #maxToolCallsPerTurn?: number; #sessionId?: string; + #promptCacheKey?: string; #metadata?: Record; #metadataResolver?: (provider: string) => Record | undefined; #providerSessionState?: Map; @@ -319,7 +326,7 @@ export class Agent { #cursorToolResultBuffer: CursorToolResultEntry[] = []; streamFn: StreamFn; - getApiKey?: (provider: string) => Promise | string | undefined; + getApiKey?: (provider: string, ctx?: ApiKeyResolveContext) => Promise | string | undefined; /** * Hook invoked after tool arguments are validated and before execution. * Reassign at any time to swap the implementation (e.g. on extension reload). @@ -344,6 +351,7 @@ export class Agent { this.#maxToolCallsPerTurn = opts.maxToolCallsPerTurn; this.streamFn = opts.streamFn || streamSimple; this.#sessionId = opts.sessionId; + this.#promptCacheKey = opts.promptCacheKey; this.#providerSessionState = opts.providerSessionState; this.#thinkingBudgets = opts.thinkingBudgets; this.#temperature = opts.temperature; @@ -390,6 +398,20 @@ export class Agent { this.#sessionId = value; } + /** + * Get the prompt cache key forwarded to providers. + */ + get promptCacheKey(): string | undefined { + return this.#promptCacheKey; + } + + /** + * Set the prompt cache key forwarded to providers. + */ + set promptCacheKey(value: string | undefined) { + this.#promptCacheKey = value; + } + /** * Static metadata forwarded to every API request when no resolver is installed * (e.g. `metadata.user_id` for Anthropic session attribution). Setting this @@ -768,8 +790,8 @@ export class Agent { this.#state.messages.length = 0; } - abort() { - this.#abortController?.abort(); + abort(reason?: unknown) { + this.#abortController?.abort(reason); } waitForIdle(): Promise { @@ -936,6 +958,7 @@ export class Agent { interruptMode: this.#interruptMode, maxToolCallsPerTurn: this.#maxToolCallsPerTurn, sessionId: this.#sessionId, + promptCacheKey: this.#promptCacheKey, metadata: this.#metadataResolver ? undefined : this.#metadata, metadataResolver: this.#metadataResolver, providerSessionState: this.#providerSessionState, @@ -1053,8 +1076,12 @@ export class Agent { } } } catch (err) { - const errorMessage = err instanceof Error ? err.message : String(err); const stoppedForAbort = this.#abortController?.signal.aborted === true; + const errorMessage = stoppedForAbort + ? abortReasonText(this.#abortController?.signal) + : err instanceof Error + ? err.message + : String(err); const shouldEmitVisibleOutputBlockedError = !stoppedForAbort && isAnthropicOutputBlockedError(errorMessage); const assistantPartial = partial?.role === "assistant" ? partial : undefined; const hadAssistantStart = assistantPartial !== undefined; diff --git a/packages/agent/src/telemetry.ts b/packages/agent/src/telemetry.ts index 5cd99a135..8655ccb6c 100644 --- a/packages/agent/src/telemetry.ts +++ b/packages/agent/src/telemetry.ts @@ -50,6 +50,7 @@ import { } from "@opentelemetry/api"; import { AgentRunCollector, type AgentRunCoverage, type AgentRunSummary, type ToolStatus } from "./run-collector"; import type { AgentTool } from "./types"; +import { EventLoopKeepalive } from "./utils/yield"; /** Default tracer name. Override via {@link AgentTelemetryConfig.tracerName}. */ export const DEFAULT_TRACER_NAME = "@oh-my-pi/pi-agent-core"; @@ -1629,6 +1630,13 @@ export async function instrumentedCompleteSimple( options: SimpleStreamOptions, span: InstrumentedChatSpanOptions, ): Promise { + // Oneshot LLM calls (handoff, compaction/branch summaries) run outside the + // agent `#runLoop`, which is where the EventLoopKeepalive normally lives. + // Without it, Bun's JSC loop stops servicing timers while parked on the + // long-lived completion promise, freezing any host spinner (e.g. the + // `/handoff` Loader) until an unrelated I/O event (a terminal resize) + // pokes the loop. Keep the loop healthy for the duration of the call. + using _keepalive = new EventLoopKeepalive(); const { telemetry, parent, oneshotKind } = span; const stepNumber = span.stepNumber ?? -1; const reasoning = options.reasoning; diff --git a/packages/agent/src/types.ts b/packages/agent/src/types.ts index 7ca6a37ec..8ecc1458c 100644 --- a/packages/agent/src/types.ts +++ b/packages/agent/src/types.ts @@ -1,4 +1,5 @@ import type { + ApiKeyResolveContext, AssistantMessage, AssistantMessageEvent, AssistantMessageEventStream, @@ -112,7 +113,7 @@ export interface AgentLoopConfig extends SimpleStreamOptions { * Useful for short-lived OAuth tokens (e.g., GitHub Copilot) that may expire * during long-running tool execution phases. */ - getApiKey?: (provider: string) => Promise | string | undefined; + getApiKey?: (provider: string, ctx?: ApiKeyResolveContext) => Promise | string | undefined; /** * Returns steering messages to inject into the conversation mid-run. diff --git a/packages/agent/test/agent-loop.test.ts b/packages/agent/test/agent-loop.test.ts index d0913c609..55a9ac7a8 100644 --- a/packages/agent/test/agent-loop.test.ts +++ b/packages/agent/test/agent-loop.test.ts @@ -157,6 +157,35 @@ describe("agentLoop with AgentMessage", () => { expect(events.map(event => event.type)).toContain("agent_end"); }); + it("surfaces a custom abort reason on the synthesized aborted message", async () => { + const context: AgentContext = { + systemPrompt: ["You are helpful."], + messages: [], + tools: [], + }; + const mock = createMockModel(); + const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter, maxToolCallsPerTurn: 8 }; + const controller = new AbortController(); + const streamFn = () => new AssistantMessageEventStream(); + + const stream = agentLoop([createUserMessage("Hello")], context, config, controller.signal, streamFn); + // Abort with a reason (as the coding agent does for a user Esc interrupt). + queueMicrotask(() => controller.abort("Interrupted by user")); + + for await (const _event of stream) { + // drain + } + + const messages = await stream.result(); + const finalMessage = messages[messages.length - 1]; + expect(finalMessage.role).toBe("assistant"); + if (finalMessage.role !== "assistant") throw new Error("Expected assistant message"); + expect(finalMessage.stopReason).toBe("aborted"); + // The reason rides AbortController.abort(reason) onto the message verbatim, + // instead of the generic "Request was aborted" default. + expect(finalMessage.errorMessage).toBe("Interrupted by user"); + }); + it("should handle custom message types via convertToLlm", async () => { // Create a custom message type interface CustomNotification { diff --git a/packages/agent/test/agent.test.ts b/packages/agent/test/agent.test.ts index c54f9366a..db95e43df 100644 --- a/packages/agent/test/agent.test.ts +++ b/packages/agent/test/agent.test.ts @@ -354,6 +354,21 @@ describe("Agent", () => { expect(reasoningPerCall).toEqual([ThinkingLevel.Low, ThinkingLevel.High]); }); + it("forwards distinct provider session id and prompt cache key to the stream", async () => { + const mock = createMockModel({ responses: [{ content: ["ok"] }] }); + const agent = new Agent({ + initialState: { model: mock.model, messages: [] }, + streamFn: mock.stream, + sessionId: "provider-lineage", + promptCacheKey: "parent-cache", + }); + + await agent.prompt("run"); + + expect(mock.calls[0]?.options?.sessionId).toBe("provider-lineage"); + expect(mock.calls[0]?.options?.promptCacheKey).toBe("parent-cache"); + }); + it("returns static metadata via the plain setter", () => { const agent = new Agent(); expect(agent.metadata).toBeUndefined(); diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 83afdf41e..344d76b9a 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,56 @@ ## [Unreleased] +## [15.10.1] - 2026-06-07 + +### Breaking Changes + +- Removed the `onAuthError` option from stream request options and shifted auth retry handling to resolver-based `apiKey` behavior, requiring callers using custom auth-retry hooks to migrate + +### Added + +- Added `ApiKeyResolver` and `ApiKey` auth helpers, including `isApiKeyResolver`, `isAuthRetryableError`, `resolveApiKeyOnce`, and `withAuth`, and exported them from the package root +- Added support for a function-valued `apiKey` in `SimpleStreamOptions` so a single stream request can refresh or rotate credentials during retry +- Added `forceRefresh` credential option to `AuthStorage.getApiKey` and `rotateSessionCredential` support for session-level credential rotation after auth failures +- Added `AuthStorage.resolver(provider, options)` method that builds an `ApiKeyResolver` implementing the a/b/c auth-retry policy directly on the storage instance + +### Changed + +- Changed gateway and stream auth flows to share the a/b/c retry policy, refreshing the same session credential first and then switching to a sibling credential on repeated auth failures + +### Fixed + +- Fixed streaming auth retries to handle `401` and usage-limit errors before replay-unsafe content is emitted, including failures surfaced only via `errorStatus` +- Fixed tool argument validation to coerce singleton non-string values into arrays when the schema expects an array, preventing Anthropic-compatible models that emit `todo.ops` as an object from getting stuck in repeated validation-error loops. ([#2026](https://github.com/can1357/oh-my-pi/issues/2026)) +- Fixed streaming retries to buffer and suppress partial `start` events from failed auth attempts so only clean retried events are delivered +- Fixed the HTTP 400 raw-request dumper (`appendRawHttpRequestDumpFor400`) littering the real `~/.omp/logs/http-400-requests` directory during tests. Provider suites exercise the 400 error path with mocked `fetch` responses, which the dumper could not distinguish from genuine failures; it now skips persistence under the Bun test runner (`isBunTestRuntime()`). +- Fixed Anthropic Opus requests unnecessarily forcing `tool_choice.disable_parallel_tool_use`, allowing Claude Opus to use the provider's default parallel tool-calling behavior again. +- Fixed parallel `function_call` items losing arguments against llama.cpp's OpenAI Responses endpoint (`/v1/responses`), where every call but the last finalized with `{}` and the agent rejected them with `path: Invalid input: expected string, received undefined`. llama.cpp's `to_json_oaicompat_resp` emits `output_item.added` with only `item.call_id` (no `item.id`, no `output_index`) while the matching `function_call_arguments.delta` carries `item_id: "fc_"`. `processResponsesStream` now registers function-call and custom-tool-call items under `item.call_id` as a secondary lookup key (alongside `item.id`/`output_index`) so identifier-deviant hosts route deltas and done events to the right block. ([#2015](https://github.com/can1357/oh-my-pi/issues/2015)) +- Fixed `PI_REQ_DEBUG` response recording truncating the captured body when a streamed response was cancelled mid-flight. The response tee in `wrapResponse` could call `FileRequestDebugResponseLog.close()` from both the `cancel` callback and the resumed `pull` (which observes `done` once the source reader is cancelled); the second caller saw the handle already nulled and returned before the first caller's pending write flushed, so the `.res.log` lost the already-buffered chunk. `close()` now memoizes its flush-and-close promise so every caller awaits the same completion. + +## [15.10.0] - 2026-06-06 + +### Added + +- Added a dependency-free `@oh-my-pi/pi-ai/effort` module exporting the `Effort` enum and `THINKING_EFFORTS`, split out of `model-thinking` so hot-path consumers can import the thinking levels without pulling in `model-thinking` and its provider-compat dependency graph. The package barrel still re-exports both names, so existing imports are unaffected. + +### Fixed + +- Fixed Antigravity usage provider emitting one bar per model instead of deduplicating by tier — a single account's 15+ model entries now collapse to one bar per tier, matching the shared-quota reality of the upstream API. +- Fixed Antigravity usage reports missing `email` and `accountId` in metadata, so the `/usage` display and the deduplicator can associate reports with their credentials. +- Fixed usage-report dedup ignoring `projectId` for Google Cloud providers, preventing duplicate credential entries from being recognized as the same account. + +- Fixed Cloud Code Assist (Antigravity / Gemini CLI) rejecting the `github` tool with HTTP 400 when the `pr` parameter schema contained `anyOf: [string, array]`. The CCA mixed-type combiner collapse picked the first non-null type (`string`) but indiscriminately copied type-specific keys from variant branches — `items` from the array variant leaked onto the string-typed result, producing `{type: "string", items: {...}}` which Google's API rejects as invalid. The collapse now filters merged variant fields against the winning type's allowed key set. ([#2002](https://github.com/can1357/oh-my-pi/pull/2002)) +- Fixed OpenAI Responses-family providers (Codex, OpenAI Responses, Azure Responses) rejecting requests with `400 No tool output found for function call …` after the user branched/navigated the session tree to a node that ends on a tool call (the tool-result child is dropped from the reconstructed history) or after a turn was aborted/crashed between the call streaming and its result persisting. The converters now synthesize a placeholder `function_call_output`/`custom_tool_call_output` immediately after any unpaired `function_call`/`custom_tool_call`, symmetric to the existing orphan-output repair, so the model still sees the call and can recover instead of the whole request 400ing. +- Fixed Anthropic-compatible reasoning endpoints losing prior-turn reasoning on continuation requests when they emit unsigned `thinking` blocks. `convertAnthropicMessages` treated unknown endpoints as signature-enforcing and demoted unsigned reasoning to `type: "text"`, which destabilized tool-call argument serialization on the next turn — the upstream symptom behind the `args?.ops?.map is not a function` crash reported against the `todo` tool. Official `api.anthropic.com` keeps the conservative text fallback; non-official `anthropic-messages` reasoning models now replay unsigned reasoning as native `type: "thinking"` ([#2005](https://github.com/can1357/oh-my-pi/issues/2005)). + +## [15.9.67] - 2026-06-06 + +### Fixed + +- Fixed llama.cpp/OpenAI Responses parallel tool calls losing arguments when `function_call_arguments.done` events omit `output_index` and `item_id`, by routing those identifierless final-argument events through the open function calls in item order. ([#1970](https://github.com/can1357/oh-my-pi/issues/1970)) +- Fixed local Ollama (`openai-responses`) turns failing with HTTP 400 `invalid reasoning value: "minimal"` when a discovered model ran with `minimal` (or `xhigh`) thinking. Ollama's OpenAI-compatible `reasoning.effort` only accepts `high|medium|low|max|none`, so discovered reasoning-capable Ollama models now carry a `compat.reasoningEffortMap` remapping `minimal → low` and `xhigh → max`; non-reasoning models are left untouched. + ## [15.9.2] - 2026-06-05 ### Added diff --git a/packages/ai/package.json b/packages/ai/package.json index 708eba2c3..7a108d2e6 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "15.9.4", + "version": "15.10.1", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/scripts/generate-models.ts b/packages/ai/scripts/generate-models.ts index 6431ade6d..497fb7c25 100644 --- a/packages/ai/scripts/generate-models.ts +++ b/packages/ai/scripts/generate-models.ts @@ -32,6 +32,7 @@ import { isFireworksKimiK2ModelId, MODELS_DEV_PROVIDER_DESCRIPTORS, mapModelsDevToModels, + stripFireworksDeepSeekThinkingToggle, UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS, } from "../src/provider-models/openai-compat"; @@ -243,6 +244,18 @@ function applyFireworksKimiMaxTokensCap(models: readonly Model[]): Model[] { }); } +/** + * Fireworks' DeepSeek V4 endpoint accepts the user's effort through + * `reasoning_effort` and rejects the DeepSeek-native binary `thinking` toggle + * when both are present. Strip stale reference metadata from generated fallbacks. + */ +function applyFireworksDeepSeekReasoningShape(models: readonly Model[]): Model[] { + return models.map(model => { + if (model.provider !== "fireworks" || model.api !== "openai-completions") return model; + return stripFireworksDeepSeekThinkingToggle(model, model.id); + }); +} + const ANTIGRAVITY_ENDPOINT = "https://daily-cloudcode-pa.sandbox.googleapis.com"; async function getOAuthAccessFromStorage(provider: OAuthProvider): Promise { @@ -416,6 +429,7 @@ async function generateModels() { allModels = applyPremiumMultiplierOverrides(allModels); allModels = applyCodexPricingFallback(allModels); allModels = applyFireworksKimiMaxTokensCap(allModels); + allModels = applyFireworksDeepSeekReasoningShape(allModels); applyGeneratedModelPolicies(allModels); linkOpenAIPromotionTargets(allModels); diff --git a/packages/ai/src/auth-gateway/server.ts b/packages/ai/src/auth-gateway/server.ts index dad2aa394..83a3e4338 100644 --- a/packages/ai/src/auth-gateway/server.ts +++ b/packages/ai/src/auth-gateway/server.ts @@ -18,8 +18,9 @@ * POST /v1/responses → OpenAI Responses in/out */ import { extractRetryHint, logger } from "@oh-my-pi/pi-utils"; +import type { ApiKeyResolver } from "../auth-retry"; import type { AuthStorage } from "../auth-storage"; -import { Effort } from "../model-thinking"; +import { Effort } from "../effort"; import * as anthropicMessages from "../providers/anthropic-messages-server"; import * as openaiChat from "../providers/openai-chat-server"; import * as openaiResponses from "../providers/openai-responses-server"; @@ -340,6 +341,60 @@ async function refreshGatewayApiKeyAfterAuthError( return storage.getApiKey(provider, sessionId, { modelId: model.id, signal }); } +/** + * Build the {@link ApiKeyResolver} handed to `streamSimple` for a gateway + * request. Drives the central a/b/c auth-retry policy server-side: + * + * - initial resolve → the credential already resolved for this request. + * - step (b) `!lastChance` → force-refresh the SAME session-sticky credential + * (a peer/broker may have rotated its token out from under our cached copy). + * - step (c) `lastChance` → {@link refreshGatewayApiKeyAfterAuthError} switches + * to a sibling (usage-limit block vs credential invalidation by error class). + * + * `lastKey` tracks the most recent bearer so the switch step invalidates the + * credential that actually failed. + */ +function buildGatewayApiKeyResolver( + storage: AuthStorage, + model: Model, + sessionId: string, + initialKey: string, + requestSignal: AbortSignal, + format: string, + peer: string, +): ApiKeyResolver { + let lastKey = initialKey; + return async ({ lastChance, error, signal }) => { + const sig = signal ?? requestSignal; + if (error === undefined) { + lastKey = initialKey; + return initialKey; + } + if (!lastChance) { + const refreshed = await storage.getApiKey(model.provider, sessionId, { + modelId: model.id, + signal: sig, + forceRefresh: true, + }); + lastKey = refreshed ?? lastKey; + return refreshed; + } + const next = await refreshGatewayApiKeyAfterAuthError( + storage, + model, + sessionId, + model.provider, + lastKey, + error, + sig, + format, + peer, + ); + lastKey = next ?? lastKey; + return next; + }; +} + function clientClosedResponse(route: { module: FormatModule }): Response { return route.module.formatError(499, "request_aborted", "client closed request"); } @@ -447,19 +502,15 @@ async function handleFormatEndpoint( } const streamOpts = buildStreamOptions(parsed, model.api, controller.signal); - streamOpts.apiKey = apiKey; - streamOpts.onAuthError = (provider, oldKey, error) => - refreshGatewayApiKeyAfterAuthError( - bootOpts.storage, - model, - sessionId, - provider, - oldKey, - error, - controller.signal, - route.label, - peer, - ); + streamOpts.apiKey = buildGatewayApiKeyResolver( + bootOpts.storage, + model, + sessionId, + apiKey, + controller.signal, + route.label, + peer, + ); logger.info("auth-gateway request", { format: route.label, @@ -604,18 +655,15 @@ async function handlePiNative(bootOpts: AuthGatewayBootOptions, req: Request, pe // only inject server-controlled fields. The codex temperature/topP strip // matches `buildStreamOptions` — Codex rejects them with a 400. const streamOpts: SimpleStreamOptions = { ...parsed.options, apiKey, signal: controller.signal }; - streamOpts.onAuthError = (provider, oldKey, error) => - refreshGatewayApiKeyAfterAuthError( - bootOpts.storage, - model, - sessionId, - provider, - oldKey, - error, - controller.signal, - "pi-native", - peer, - ); + streamOpts.apiKey = buildGatewayApiKeyResolver( + bootOpts.storage, + model, + sessionId, + apiKey, + controller.signal, + "pi-native", + peer, + ); if (model.api === "openai-codex-responses") { delete streamOpts.temperature; delete streamOpts.topP; diff --git a/packages/ai/src/auth-gateway/types.ts b/packages/ai/src/auth-gateway/types.ts index 0390759c0..bdb563e3b 100644 --- a/packages/ai/src/auth-gateway/types.ts +++ b/packages/ai/src/auth-gateway/types.ts @@ -1,4 +1,4 @@ -import type { Effort } from "../model-thinking"; +import type { Effort } from "../effort"; import type { AssistantMessage, AssistantMessageEventStream, diff --git a/packages/ai/src/auth-retry.ts b/packages/ai/src/auth-retry.ts new file mode 100644 index 000000000..e53567f01 --- /dev/null +++ b/packages/ai/src/auth-retry.ts @@ -0,0 +1,141 @@ +import { extractHttpStatusFromError } from "@oh-my-pi/pi-utils"; +import { isUsageLimitError } from "./rate-limit-utils"; + +/** + * Context passed to an {@link ApiKeyResolver} on each resolution attempt. + * + * The `error`/`lastChance` pair drives the central a/b/c retry policy shared by + * the streaming ({@link streamSimple}) and non-streaming ({@link withAuth}) + * drivers: + * - `error === undefined` → **initial resolve** (no force-refresh; cheap, may + * return a locally-cached not-yet-expired token). + * - `error !== undefined && !lastChance` → **step (b): refresh the SAME + * account** (force a token re-mint / await an in-flight broker refresh). + * - `error !== undefined && lastChance` → **step (c): switch account** + * (invalidate/usage-limit the current credential and rotate to a sibling). + * + * The resolver returns the bearer to send, or `undefined` to stop retrying and + * surface the last error to the caller. + */ +export interface ApiKeyResolveContext { + /** True on the final retry step — the resolver should rotate to a sibling credential. */ + lastChance: boolean; + /** The auth error that triggered this re-resolution, or `undefined` on the initial resolve. */ + error: unknown; + /** Caller cancel signal, threaded into any credential refresh / rotation work. */ + signal?: AbortSignal; +} + +/** + * Resolves the API key to send for a request, retried through the a/b/c policy + * described on {@link ApiKeyResolveContext}. + */ +export type ApiKeyResolver = (ctx: ApiKeyResolveContext) => Promise | string | undefined; + +/** A static bearer string, or a {@link ApiKeyResolver} that mints/rotates one. */ +export type ApiKey = string | ApiKeyResolver; + +/** Narrows {@link ApiKey} to its resolver form. */ +export function isApiKeyResolver(key: ApiKey | undefined): key is ApiKeyResolver { + return typeof key === "function"; +} + +/** + * Performs the initial resolve of an {@link ApiKey} (`error: undefined`, + * `lastChance: false`). Static keys pass through unchanged. + */ +export async function resolveApiKeyOnce(key: ApiKey | undefined, signal?: AbortSignal): Promise { + if (key === undefined) return undefined; + if (isApiKeyResolver(key)) return (await key({ lastChance: false, error: undefined, signal })) || undefined; + return key; +} + +/** + * Classifies whether an error should trigger a credential refresh/rotation + * retry: a hard `401`, or a rotatable usage-limit ("usage_limit_reached", + * Codex's "you have hit your ChatGPT usage limit", etc.). + */ +export function isAuthRetryableError(error: unknown): boolean { + if (extractHttpStatusFromError(error) === 401) return true; + const message = error instanceof Error ? error.message : typeof error === "string" ? error : undefined; + if (!message) return false; + if (extractHttpStatusFromError({ message }) === 401) return true; + return isUsageLimitError(message); +} + +/** + * The ordered `lastChance` values for the retry steps after the initial + * attempt fails: `false` → step (b) refresh-same, `true` → step (c) switch. + * Shared by {@link withAuth} and the streaming retry driver so both run the + * same policy. + */ +export const AUTH_RETRY_STEPS: readonly boolean[] = [false, true]; + +/** Resolve a single retry step, swallowing resolver failures into `undefined`. */ +export async function resolveRetryKey( + resolver: ApiKeyResolver, + lastChance: boolean, + error: unknown, + signal?: AbortSignal, +): Promise { + try { + return (await resolver({ lastChance, error, signal })) || undefined; + } catch { + return undefined; + } +} + +/** + * Runs an auth-protected operation through the central a/b/c retry policy. + * + * - A static string key (or any non-resolver) → a single `attempt` with no + * retry (identical to the legacy static-key path). + * - A resolver → initial `attempt`, then on a retryable auth error up to two + * more attempts (refresh-same, then switch). A step is skipped when the + * resolver returns the same key it just tried or `undefined`; non-auth errors + * propagate immediately. + * + * Used by non-streaming consumers (image generation, web search, completion + * helpers). The streaming driver in `stream.ts` implements the same policy with + * its replay-safe buffering machinery. + */ +export async function withAuth( + key: ApiKey | undefined, + attempt: (key: string) => Promise, + opts?: { isAuthError?: (error: unknown) => boolean; signal?: AbortSignal; missingKeyMessage?: string }, +): Promise { + const isAuthError = opts?.isAuthError ?? isAuthRetryableError; + const missingKey = (): Error => new Error(opts?.missingKeyMessage ?? "No API key available"); + + if (!isApiKeyResolver(key)) { + if (key === undefined) throw missingKey(); + return attempt(key); + } + + const resolver = key; + const signal = opts?.signal; + let lastKey = await resolveRetryKey(resolver, false, undefined, signal); + if (lastKey === undefined) throw missingKey(); + + let lastError: unknown; + try { + return await attempt(lastKey); + } catch (error) { + if (!isAuthError(error)) throw error; + lastError = error; + } + + for (let i = 0; i < AUTH_RETRY_STEPS.length; i++) { + const nextKey = await resolveRetryKey(resolver, AUTH_RETRY_STEPS[i]!, lastError, signal); + if (nextKey === undefined || nextKey === lastKey) continue; + lastKey = nextKey; + try { + return await attempt(nextKey); + } catch (error) { + if (!isAuthError(error)) throw error; + lastError = error; + } + } + + throw lastError; +} diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index f03f7b810..b76c2bbd6 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -11,6 +11,8 @@ import { Database, type Statement } from "bun:sqlite"; import * as fs from "node:fs/promises"; import * as path from "node:path"; import { getAgentDbPath, logger } from "@oh-my-pi/pi-utils"; +import type { ApiKeyResolver } from "./auth-retry"; +import { isUsageLimitError } from "./rate-limit-utils"; import { getEnvApiKey } from "./stream"; import type { Provider } from "./types"; import type { @@ -36,6 +38,8 @@ import { loginOpenAICodexDevice } from "./utils/oauth/openai-codex"; import type { OAuthController, OAuthCredentials, OAuthProvider, OAuthProviderId } from "./utils/oauth/types"; import { loginXiaomi, loginXiaomiTokenPlan } from "./utils/oauth/xiaomi"; +const USAGE_RANKING_METRIC_EPSILON = 1e-9; + // ───────────────────────────────────────────────────────────────────────────── // Credential Types // ───────────────────────────────────────────────────────────────────────────── @@ -544,6 +548,13 @@ type AuthApiKeyOptions = { * stranding the caller for `timeoutMs * (maxRetries + 1)`. */ signal?: AbortSignal; + /** + * Force a re-mint of the session-preferred OAuth credential's access token, + * bypassing the not-yet-expired short-circuit. Powers step (b) of the + * auth-retry policy ("refresh the SAME account") so a locally-cached token + * that a peer/broker rotated out from under us is replaced before retrying. + */ + forceRefresh?: boolean; }; type OAuthResolutionResult = { apiKey: string; credential: OAuthCredential }; @@ -606,6 +617,14 @@ function hasOpenAICodexProPlan(report: UsageReport | null): boolean { return getUsagePlanType(report)?.includes("pro") === true; } +function compareUsageRankingMetric(left: number, right: number): number { + if (left === right) return 0; + if (!Number.isFinite(left) || !Number.isFinite(right)) return left < right ? -1 : 1; + const delta = left - right; + const tolerance = Math.max(USAGE_RANKING_METRIC_EPSILON, Math.max(Math.abs(left), Math.abs(right)) * 0.000001); + return Math.abs(delta) <= tolerance ? 0 : delta; +} + function resolveDefaultUsageProvider(provider: Provider): UsageProvider | undefined { return DEFAULT_USAGE_PROVIDER_MAP.get(provider); } @@ -2288,6 +2307,16 @@ export class AuthStorage { return undefined; } + #getUsageReportScopeProjectId(report: UsageReport): string | undefined { + const ids = new Set(); + for (const limit of report.limits) { + const projectId = limit.scope.projectId?.trim(); + if (projectId) ids.add(projectId); + } + if (ids.size === 1) return [...ids][0]; + return undefined; + } + #getUsageReportIdentifiers(report: UsageReport): string[] { const identifiers: string[] = []; const email = this.#getUsageReportMetadataValue(report, "email"); @@ -2295,6 +2324,11 @@ export class AuthStorage { if (report.provider === "openai-codex" || report.provider === "anthropic") { return identifiers.map(identifier => `${report.provider}:${identifier.toLowerCase()}`); } + const projectId = + this.#getUsageReportMetadataValue(report, "projectId") ?? this.#getUsageReportScopeProjectId(report); + // Only add project as a fallback when no email is available — two users + // with different emails on the same GCP project must not merge. + if (projectId && !email) identifiers.push(`project:${projectId}`); const accountId = this.#getUsageReportMetadataValue(report, "accountId"); if (accountId) identifiers.push(`account:${accountId}`); const account = this.#getUsageReportMetadataValue(report, "account"); @@ -2781,12 +2815,14 @@ export class AuthStorage { return left.planPriority - right.planPriority; } if (left.hasPriorityBoost !== right.hasPriorityBoost) return left.hasPriorityBoost ? -1 : 1; - if (left.secondaryDrainRate !== right.secondaryDrainRate) { - return left.secondaryDrainRate - right.secondaryDrainRate; - } - if (left.secondaryUsed !== right.secondaryUsed) return left.secondaryUsed - right.secondaryUsed; - if (left.primaryDrainRate !== right.primaryDrainRate) return left.primaryDrainRate - right.primaryDrainRate; - if (left.primaryUsed !== right.primaryUsed) return left.primaryUsed - right.primaryUsed; + let metric = compareUsageRankingMetric(left.secondaryDrainRate, right.secondaryDrainRate); + if (metric !== 0) return metric; + metric = compareUsageRankingMetric(left.secondaryUsed, right.secondaryUsed); + if (metric !== 0) return metric; + metric = compareUsageRankingMetric(left.primaryDrainRate, right.primaryDrainRate); + if (metric !== 0) return metric; + metric = compareUsageRankingMetric(left.primaryUsed, right.primaryUsed); + if (metric !== 0) return metric; return 0; } @@ -3019,19 +3055,38 @@ export class AuthStorage { candidates.unshift(preferred); } } + // Step (b) of the auth-retry policy: when `forceRefresh` is set, re-mint + // the session-preferred credential (or the first candidate when no + // session preference exists yet) even if its cached token still looks + // valid — a peer/broker may have rotated it out from under us. + const forceRefreshIndex = options?.forceRefresh + ? (sessionPreferredIndex ?? candidates[0]?.selection.index) + : undefined; await Promise.all( candidates.map(async candidate => { - if (Date.now() + OAUTH_REFRESH_SKEW_MS < candidate.selection.credential.expires) return; + const force = forceRefreshIndex !== undefined && candidate.selection.index === forceRefreshIndex; + if (!force && Date.now() + OAUTH_REFRESH_SKEW_MS < candidate.selection.credential.expires) return; const latestCredential = this.#getCredentialsForProvider(provider)[candidate.selection.index]; - if (latestCredential?.type === "oauth" && Date.now() + OAUTH_REFRESH_SKEW_MS < latestCredential.expires) { + if ( + !force && + latestCredential?.type === "oauth" && + Date.now() + OAUTH_REFRESH_SKEW_MS < latestCredential.expires + ) { candidate.selection.credential = latestCredential; return; } try { const credentialId = this.#getStoredCredentials(provider)[candidate.selection.index]?.id; + // Hand #refreshOAuthCredential a stale clone (expires:0) so its + // not-yet-expired short-circuit doesn't suppress the forced + // re-mint; an in-flight peer refresh is still awaited via the + // per-credential single-flight. + const refreshTarget = force + ? { ...candidate.selection.credential, expires: 0 } + : candidate.selection.credential; const refreshedCredentials = await this.#refreshOAuthCredential( provider, - candidate.selection.credential, + refreshTarget, credentialId, options?.signal, ); @@ -3643,6 +3698,90 @@ export class AuthStorage { return true; } + /** + * Rotate away from the session's current credential after a retryable auth + * error — step (c) of the auth-retry policy. Stateless: looks up the + * session-sticky credential (no API-key matching needed), applies the + * storage action for the error class, then clears the sticky so the next + * {@link AuthStorage.getApiKey} for this session picks a sibling. + * + * - usage-limit / account-rate-limit error → {@link AuthStorage.markUsageLimitReached} + * (temporary block via its own backoff — default plus server usage-report + * reset; sticky left intact so the next resolve re-ranks around the block). + * - otherwise (hard 401 / auth failure) → mark the credential suspect (or + * reload when no broker hook is wired) and block it, then drop the sticky. + * + * Returns whether another usable credential of the same type remains. + */ + async rotateSessionCredential( + provider: string, + sessionId: string | undefined, + options?: { error?: unknown; signal?: AbortSignal }, + ): Promise { + const sessionCredential = this.#getSessionCredential(provider, sessionId); + if (!sessionCredential) return false; + + const error = options?.error; + const message = error instanceof Error ? error.message : typeof error === "string" ? error : undefined; + if (message && isUsageLimitError(message)) { + return this.markUsageLimitReached(provider, sessionId, { signal: options?.signal }); + } + + const providerKey = this.#getProviderTypeKey(provider, sessionCredential.type); + // Snapshot sibling availability before mutating so a soft-deleting + // suspect hook can't reindex the answer out from under us. + const hasSibling = this.#getCredentialsForProvider(provider).some( + (credential, index) => + credential.type === sessionCredential.type && + index !== sessionCredential.index && + !this.#isCredentialBlocked(providerKey, index), + ); + const target = this.#getStoredCredentials(provider)[sessionCredential.index]; + this.#clearSessionCredential(provider, sessionId); + this.#markCredentialBlocked(providerKey, sessionCredential.index, Date.now() + AuthStorage.#defaultBackoffMs); + + if (target) { + const markSuspect = this.#store.markCredentialSuspect?.bind(this.#store); + if (markSuspect) { + await markSuspect(target.id, { signal: options?.signal }); + } else { + await this.reload(); + } + const latestRows = this.#store.listAuthCredentials(provider); + this.#setStoredCredentials( + provider, + latestRows.map(row => ({ id: row.id, credential: row.credential })), + ); + } + + return hasSibling; + } + + /** + * Build an {@link ApiKeyResolver} backed by this storage, implementing the + * central a/b/c auth-retry policy: + * + * - initial (`error: undefined`) → resolve the session credential. + * - step (b) `!lastChance` → force-refresh the SAME session-sticky credential. + * - step (c) `lastChance` → rotate to a sibling credential, then re-resolve. + * + * Used by web-search providers and other consumers that hold an AuthStorage + * directly (no ModelRegistry in scope). + */ + resolver(provider: string, options?: { sessionId?: string; baseUrl?: string; modelId?: string }): ApiKeyResolver { + const { sessionId, baseUrl, modelId } = options ?? {}; + return async ({ lastChance, error, signal }) => { + if (error === undefined) { + return this.getApiKey(provider, sessionId, { baseUrl, modelId, signal }); + } + if (lastChance) { + await this.rotateSessionCredential(provider, sessionId, { error, signal }); + return this.getApiKey(provider, sessionId, { baseUrl, modelId, signal }); + } + return this.getApiKey(provider, sessionId, { baseUrl, modelId, forceRefresh: true, signal }); + }; + } + // ─── Auth Broker integration ──────────────────────────────────────────── /** @@ -3932,6 +4071,8 @@ function resolveProviderCredentialIdentityKey(provider: string, identifiers: str const accountIdentifier = identifiers.find(identifier => identifier.startsWith("account:")); if (accountIdentifier) return accountIdentifier; if (emailIdentifier) return emailIdentifier; + const projectIdentifier = identifiers.find(identifier => identifier.startsWith("project:")); + if (projectIdentifier) return projectIdentifier; return null; } @@ -3967,6 +4108,8 @@ function extractOAuthCredentialIdentifiers(credential: OAuthCredential): string[ if (accountId) identifiers.add(`account:${accountId}`); const email = normalizeStoredEmail(credential.email); if (email) identifiers.add(`email:${email}`); + const projectId = normalizeStoredAccountId(credential.projectId); + if (projectId) identifiers.add(`project:${projectId}`); const accessIdentifiers = extractOAuthTokenIdentifiers(credential.access) ?? []; for (const identifier of accessIdentifiers) { identifiers.add(identifier); diff --git a/packages/ai/src/effort.ts b/packages/ai/src/effort.ts new file mode 100644 index 000000000..831a13ede --- /dev/null +++ b/packages/ai/src/effort.ts @@ -0,0 +1,16 @@ +/** User-facing thinking levels, ordered least to most intensive. */ +export const enum Effort { + Minimal = "minimal", + Low = "low", + Medium = "medium", + High = "high", + XHigh = "xhigh", +} + +export const THINKING_EFFORTS: readonly Effort[] = [ + Effort.Minimal, + Effort.Low, + Effort.Medium, + Effort.High, + Effort.XHigh, +]; diff --git a/packages/ai/src/index.ts b/packages/ai/src/index.ts index ecdce180e..a9d3c724a 100644 --- a/packages/ai/src/index.ts +++ b/packages/ai/src/index.ts @@ -3,7 +3,9 @@ export * from "./api-registry"; export * from "./auth-broker"; export { type AuthGatewayBootOptions, type ModelResolver, startAuthGateway } from "./auth-gateway/server"; export * from "./auth-gateway/types"; +export * from "./auth-retry"; export * from "./auth-storage"; +export * from "./effort"; export * from "./model-cache"; export * from "./model-manager"; export * from "./model-thinking"; diff --git a/packages/ai/src/model-thinking.ts b/packages/ai/src/model-thinking.ts index 75e930315..3758c6d6c 100644 --- a/packages/ai/src/model-thinking.ts +++ b/packages/ai/src/model-thinking.ts @@ -1,23 +1,7 @@ +import { Effort, THINKING_EFFORTS } from "./effort"; import { resolveOpenAICompat } from "./providers/openai-completions-compat"; import type { Api, Model as ApiModel, ThinkingConfig } from "./types"; -/** User-facing thinking levels, ordered least to most intensive. */ -export const enum Effort { - Minimal = "minimal", - Low = "low", - Medium = "medium", - High = "high", - XHigh = "xhigh", -} - -export const THINKING_EFFORTS: readonly Effort[] = [ - Effort.Minimal, - Effort.Low, - Effort.Medium, - Effort.High, - Effort.XHigh, -]; - const DEFAULT_REASONING_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]; const DEFAULT_REASONING_EFFORTS_WITH_XHIGH: readonly Effort[] = [ Effort.Minimal, @@ -379,18 +363,6 @@ export function supportsMidConversationSystemMessages(modelId: string): boolean return parsed.kind === "opus" && semverGte(parsed.version, "4.8"); } -/** - * Claude Opus 4.8 must emit at most one tool call per turn: the Anthropic - * Messages provider sends `tool_choice.disable_parallel_tool_use = true` for - * this model. Scoped to exactly 4.8 — earlier and later Opus versions keep - * Anthropic's default parallel tool-calling. - */ -export function disablesParallelToolUse(modelId: string): boolean { - const parsed = parseAnthropicModel(getCanonicalModelId(modelId)); - if (!parsed) return false; - return parsed.kind === "opus" && semverEqual(parsed.version, "4.8"); -} - function anthropicModelHasRealXHighEffort(model: ApiModel): boolean { if (model.api !== "anthropic-messages") return false; const parsedModel = parseKnownModel(model.id); diff --git a/packages/ai/src/models.json b/packages/ai/src/models.json index e98a047fc..d3cd30bf1 100644 --- a/packages/ai/src/models.json +++ b/packages/ai/src/models.json @@ -1932,6 +1932,75 @@ "maxLevel": "high" } }, + "openai.gpt-5.4": { + "id": "openai.gpt-5.4", + "name": "GPT-5.4", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.75, + "output": 16.5, + "cacheRead": 0.275, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "minLevel": "low", + "maxLevel": "xhigh" + } + }, + "openai.gpt-5.5": { + "id": "openai.gpt-5.5", + "name": "GPT-5.5", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5.5, + "output": 33, + "cacheRead": 0.55, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "minLevel": "low", + "maxLevel": "xhigh" + } + }, + "openai.gpt-oss-120b": { + "id": "openai.gpt-oss-120b", + "name": "gpt-oss-120b", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.15, + "output": 0.6, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 16384 + }, "openai.gpt-oss-120b-1:0": { "id": "openai.gpt-oss-120b-1:0", "name": "gpt-oss-120b", @@ -1951,6 +2020,25 @@ "contextWindow": 128000, "maxTokens": 16384 }, + "openai.gpt-oss-20b": { + "id": "openai.gpt-oss-20b", + "name": "gpt-oss-20b", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.07, + "output": 0.3, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 16384 + }, "openai.gpt-oss-20b-1:0": { "id": "openai.gpt-oss-20b-1:0", "name": "gpt-oss-20b", @@ -5364,11 +5452,6 @@ }, "maxTokensField": "max_tokens", "supportsToolChoice": false, - "extraBody": { - "thinking": { - "type": "enabled" - } - }, "reasoningContentField": "reasoning_content", "requiresReasoningContentForToolCalls": true, "requiresAssistantContentForToolCalls": true @@ -11197,7 +11280,7 @@ }, "deepseek/deepseek-r1-distill-llama-70b": { "id": "deepseek/deepseek-r1-distill-llama-70b", - "name": "DeepSeek: R1 Distill Llama 70B", + "name": "DeepSeek: R1 Distill Llama 70B (retires Jun 11)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -12648,7 +12731,7 @@ }, "meta-llama/llama-3-70b-instruct": { "id": "meta-llama/llama-3-70b-instruct", - "name": "Meta: Llama 3 70B Instruct", + "name": "Meta: Llama 3 70B Instruct (retires Jun 19)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -13187,6 +13270,25 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", + "name": "MiniMax: MiniMax M3 (new)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "minimax/minimax-m3:discounted": { + "id": "minimax/minimax-m3:discounted", "name": "MiniMax: MiniMax M3 (50% off through 2026-06-07)", "api": "openai-completions", "provider": "kilo", @@ -14262,6 +14364,63 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "nvidia/nemotron-3-ultra-550b-a55b": { + "id": "nvidia/nemotron-3-ultra-550b-a55b", + "name": "NVIDIA: Nemotron 3 Ultra", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "nvidia/nemotron-3-ultra-550b-a55b:free": { + "id": "nvidia/nemotron-3-ultra-550b-a55b:free", + "name": "NVIDIA: Nemotron 3 Ultra (free)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "nvidia/nemotron-3.5-content-safety:free": { + "id": "nvidia/nemotron-3.5-content-safety:free", + "name": "NVIDIA: Nemotron 3.5 Content Safety (free)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "nvidia/nemotron-nano-12b-v2-vl": { "id": "nvidia/nemotron-nano-12b-v2-vl", "name": "NVIDIA: Nemotron Nano 12B 2 VL", @@ -14283,7 +14442,7 @@ }, "nvidia/nemotron-nano-9b-v2": { "id": "nvidia/nemotron-nano-9b-v2", - "name": "NVIDIA: Nemotron Nano 9B V2", + "name": "NVIDIA: Nemotron Nano 9B V2 (retires Jun 11)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -16371,7 +16530,7 @@ }, "qwen/qwen3-30b-a3b": { "id": "qwen/qwen3-30b-a3b", - "name": "Qwen: Qwen3 30B A3B (retires Jun 5)", + "name": "Qwen: Qwen3 30B A3B", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -19993,24 +20152,19 @@ "api": "openai-completions", "provider": "mistral", "baseUrl": "https://api.mistral.ai/v1", - "reasoning": true, + "reasoning": false, "input": [ "text", "image" ], "cost": { - "input": 1.5, - "output": 7.5, + "input": 0.4, + "output": 2, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, - "thinking": { - "mode": "effort", - "minLevel": "minimal", - "maxLevel": "xhigh" - } + "maxTokens": 262144 }, "mistral-nemo": { "id": "mistral-nemo", @@ -25743,6 +25897,82 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "linkup-research-high": { + "id": "linkup-research-high", + "name": "linkup-research-high", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "linkup-research-low": { + "id": "linkup-research-low", + "name": "linkup-research-low", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "linkup-research-medium": { + "id": "linkup-research-medium", + "name": "linkup-research-medium", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "linkup-research-xhigh": { + "id": "linkup-research-xhigh", + "name": "linkup-research-xhigh", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "liquid/lfm-2-24b-a2b": { "id": "liquid/lfm-2-24b-a2b", "name": "liquid/lfm-2-24b-a2b", @@ -28343,6 +28573,30 @@ "maxLevel": "xhigh" } }, + "nvidia/nemotron-3-ultra-550b-a55b": { + "id": "nvidia/nemotron-3-ultra-550b-a55b", + "name": "nvidia/nemotron-3-ultra-550b-a55b", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "nvidia/nvidia-nemotron-nano-9b-v2": { "id": "nvidia/nvidia-nemotron-nano-9b-v2", "name": "nvidia-nemotron-nano-9b-v2", @@ -38214,7 +38468,7 @@ "cacheRead": 0.05, "cacheWrite": 0.625 }, - "contextWindow": 262144, + "contextWindow": 1000000, "maxTokens": 65536, "thinking": { "mode": "effort", @@ -38263,7 +38517,7 @@ "cacheRead": 0.04, "cacheWrite": 0.5 }, - "contextWindow": 262144, + "contextWindow": 1000000, "maxTokens": 65536, "thinking": { "mode": "budget", @@ -39625,6 +39879,30 @@ "maxLevel": "xhigh" } }, + "nemotron-3-ultra-free": { + "id": "nemotron-3-ultra-free", + "name": "Nemotron 3 Ultra Free", + "api": "openai-completions", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "qwen3.5-plus": { "id": "qwen3.5-plus", "name": "Qwen3.5 Plus", @@ -41767,12 +42045,12 @@ ], "cost": { "input": 0.12, - "output": 0.37, - "cacheRead": 0.019999999499999997, + "output": 0.36, + "cacheRead": 0.09, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 16384, + "maxTokens": 8192, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -42135,7 +42413,7 @@ ], "cost": { "input": 0.02, - "output": 0.049999999999999996, + "output": 0.03, "cacheRead": 0, "cacheWrite": 0 }, @@ -42360,7 +42638,7 @@ "cacheWrite": 0 }, "contextWindow": 204800, - "maxTokens": 131072, + "maxTokens": 196608, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -43295,6 +43573,54 @@ "maxLevel": "high" } }, + "nvidia/nemotron-3-ultra-550b-a55b": { + "id": "nvidia/nemotron-3-ultra-550b-a55b", + "name": "NVIDIA: Nemotron 3 Ultra", + "api": "openai-completions", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.5, + "output": 2.5, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "nvidia/nemotron-3-ultra-550b-a55b:free": { + "id": "nvidia/nemotron-3-ultra-550b-a55b:free", + "name": "NVIDIA: Nemotron 3 Ultra (free)", + "api": "openai-completions", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } + }, "nvidia/nemotron-nano-12b-v2-vl:free": { "id": "nvidia/nemotron-nano-12b-v2-vl:free", "name": "NVIDIA: Nemotron Nano 12B 2 VL (free)", @@ -45244,7 +45570,7 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 20000, + "maxTokens": 16384, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -46000,13 +46326,13 @@ "image" ], "cost": { - "input": 0.29, - "output": 3.1999999999999997, + "input": 0.28900000000000003, + "output": 2.4, "cacheRead": 0.15, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262140, + "maxTokens": 131072, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -49634,6 +49960,28 @@ "supportsUsageInStreaming": false } }, + "nvidia-nemotron-3-ultra-550b-a55b": { + "id": "nvidia-nemotron-3-ultra-550b-a55b", + "name": "nvidia-nemotron-3-ultra-550b-a55b", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888, + "compat": { + "supportsUsageInStreaming": false + } + }, "nvidia-nemotron-cascade-2-30b-a3b": { "id": "nvidia-nemotron-cascade-2-30b-a3b", "name": "nvidia-nemotron-cascade-2-30b-a3b", @@ -52853,6 +53201,30 @@ "maxLevel": "xhigh" } }, + "nvidia/nemotron-3-ultra-550b-a55b": { + "id": "nvidia/nemotron-3-ultra-550b-a55b", + "name": "Nemotron 3 Ultra", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.4, + "cacheRead": 0.12, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "nvidia/nemotron-nano-12b-v2-vl": { "id": "nvidia/nemotron-nano-12b-v2-vl", "name": "Nvidia Nemotron Nano 12B V2 VL", @@ -57962,7 +58334,7 @@ "cost": { "input": 0.5, "output": 1.5, - "cacheRead": 0, + "cacheRead": 0.05, "cacheWrite": 0 }, "contextWindow": 256000, diff --git a/packages/ai/src/provider-models/ollama.ts b/packages/ai/src/provider-models/ollama.ts index c71fbf617..ed539d924 100644 --- a/packages/ai/src/provider-models/ollama.ts +++ b/packages/ai/src/provider-models/ollama.ts @@ -1,6 +1,6 @@ import { fetchWithRetry } from "@oh-my-pi/pi-utils"; +import { Effort } from "../effort"; import type { ModelManagerOptions } from "../model-manager"; -import { Effort } from "../model-thinking"; import type { ThinkingConfig } from "../types"; import { createBundledReferenceMap, createReferenceResolver } from "./bundled-references"; diff --git a/packages/ai/src/provider-models/openai-compat.ts b/packages/ai/src/provider-models/openai-compat.ts index 7004521cf..c529fb135 100644 --- a/packages/ai/src/provider-models/openai-compat.ts +++ b/packages/ai/src/provider-models/openai-compat.ts @@ -1,5 +1,5 @@ +import { Effort } from "../effort"; import type { ModelManagerOptions } from "../model-manager"; -import { Effort } from "../model-thinking"; import { getBundledModels } from "../models"; import type { Api, Model, Provider, ThinkingConfig } from "../types"; import { isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils"; @@ -242,6 +242,22 @@ async function fetchOllamaNativeModels( const OLLAMA_FALLBACK_CONTEXT_WINDOW = 128_000; /** Cap max output tokens at a value that matches OMP's other openai-responses defaults. */ const OLLAMA_DEFAULT_MAX_TOKENS = 8192; +/** + * Ollama's OpenAI-compatible `reasoning.effort` only accepts + * `high|medium|low|max|none`; passing OMP's `minimal`/`xhigh` levels verbatim + * makes the server reject the turn with HTTP 400 `invalid reasoning value`. + * Map the two unsupported levels onto the closest accepted ones (`low`/`max`). + */ +const OLLAMA_REASONING_EFFORT_MAP = { minimal: "low", xhigh: "max" } as const; + +/** Stamp the Ollama reasoning-effort map onto a reasoning-capable model. */ +function applyOllamaReasoningCompat(model: Model<"openai-responses">): void { + if (!model.reasoning) return; + model.compat = { + ...model.compat, + reasoningEffortMap: { ...OLLAMA_REASONING_EFFORT_MAP, ...model.compat?.reasoningEffortMap }, + }; +} interface OllamaResolvedMetadata { contextWindow: number; @@ -955,6 +971,29 @@ export function clampFireworksKimiMaxTokens(modelId: string, candidate: number): return isFireworksKimiK2ModelId(modelId) ? Math.min(candidate, FIREWORKS_KIMI_MAX_TOKENS) : candidate; } +/** + * Fireworks DeepSeek V4 accepts effort via `reasoning_effort` but rejects the + * DeepSeek-native binary `thinking` toggle when both are present. + */ +export function stripFireworksDeepSeekThinkingToggle( + model: Model<"openai-completions">, + publicModelId: string, +): Model<"openai-completions"> { + if (!publicModelId.startsWith("deepseek-v4")) return model; + const compat = model.compat; + if (!compat?.extraBody || !("thinking" in compat.extraBody)) return model; + + const extraBody = { ...compat.extraBody }; + delete extraBody.thinking; + if (Object.keys(extraBody).length > 0) { + return { ...model, compat: { ...compat, extraBody } }; + } + + const nextCompat = { ...compat }; + delete nextCompat.extraBody; + return { ...model, compat: nextCompat }; +} + export interface FireworksModelManagerConfig { apiKey?: string; baseUrl?: string; @@ -1024,7 +1063,10 @@ export function fireworksModelManagerOptions( mapModel: (entry, defaults) => { const publicModelId = toFireworksPublicModelId(defaults.id); const reference = modelsDevReferences.get(publicModelId) ?? bundledReferences(publicModelId); - const model = mapWithBundledReference(entry, defaults, reference); + const model = stripFireworksDeepSeekThinkingToggle( + mapWithBundledReference(entry, defaults, reference), + publicModelId, + ); return { ...model, id: publicModelId, @@ -1357,12 +1399,14 @@ export function ollamaModelManagerOptions(config?: OllamaModelManagerConfig): Mo if (metadata.input) { model.input = metadata.input; } + applyOllamaReasoningCompat(model); }), ); return openAiCompatible; } const nativeFallback = await fetchOllamaNativeModels(baseUrl, resolveMetadata); if (nativeFallback && nativeFallback.length > 0) { + for (const model of nativeFallback) applyOllamaReasoningCompat(model); return nativeFallback; } return openAiCompatible; diff --git a/packages/ai/src/providers/amazon-bedrock.ts b/packages/ai/src/providers/amazon-bedrock.ts index 7f161f32c..49b1523aa 100644 --- a/packages/ai/src/providers/amazon-bedrock.ts +++ b/packages/ai/src/providers/amazon-bedrock.ts @@ -8,7 +8,7 @@ */ import { $env, $flag, extractHttpStatusFromError, fetchWithRetry } from "@oh-my-pi/pi-utils"; -import type { Effort } from "../model-thinking"; +import type { Effort } from "../effort"; import { mapEffortToAnthropicAdaptiveEffort, requireSupportedEffort } from "../model-thinking"; import { calculateCost } from "../models"; import type { diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 52c0c901e..c8e9f3c35 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -13,7 +13,6 @@ import { readSseEvents, } from "@oh-my-pi/pi-utils"; import { - disablesParallelToolUse, hasOpus47ApiRestrictions, mapEffortToAnthropicAdaptiveEffort, supportsMidConversationSystemMessages, @@ -1778,16 +1777,12 @@ type SystemBlockOptions = { cacheControl?: AnthropicCacheControl; }; -function withGlobalCacheScope(cacheControl: AnthropicCacheControl): AnthropicCacheControl { - return { ...cacheControl, scope: "global" }; -} - function applyClaudeCodeSystemCache( blocks: AnthropicSystemBlock[], cacheControl: AnthropicCacheControl | undefined, ): number { if (!cacheControl || blocks.length <= 2) return 0; - blocks[2] = { ...blocks[2], cache_control: withGlobalCacheScope(cacheControl) }; + blocks[2] = { ...blocks[2], cache_control: cacheControl }; if (blocks.length === 3) return 1; const lastIndex = blocks.length - 1; blocks[lastIndex] = { ...blocks[lastIndex], cache_control: cacheControl }; @@ -2375,21 +2370,6 @@ function buildParams( } } - // Claude Opus 4.8 must emit at most one tool call per turn. Force - // `disable_parallel_tool_use` onto the outgoing tool_choice (synthesizing an - // `auto` choice when none is set). Gated on tools being present: Anthropic - // rejects `tool_choice` without `tools`, and parallelism is moot otherwise. - // `none` rejects the field, so leave it untouched. A fresh object is built - // rather than mutated so the caller's `options.toolChoice` is never aliased. - if (disablesParallelToolUse(model.id) && params.tools && params.tools.length > 0) { - const current = params.tool_choice; - if (!current) { - params.tool_choice = { type: "auto", disable_parallel_tool_use: true }; - } else if (current.type !== "none") { - params.tool_choice = { ...current, disable_parallel_tool_use: true }; - } - } - const shouldInjectClaudeCodeInstruction = isOAuthToken && !model.id.startsWith("claude-3-5-haiku"); const firstUserMessageText = shouldInjectClaudeCodeInstruction ? extractClaudeCodeFirstUserMessageText(context.messages) @@ -2431,22 +2411,31 @@ function isZaiAnthropicEndpoint(model: Model<"anthropic-messages">): boolean { } /** - * Returns true for providers whose Anthropic-compatible endpoints do NOT - * implement signature-based thinking-chain integrity (DeepSeek, Z.AI, etc.). - * For these providers, unsigned thinking blocks must be preserved as - * `type: "thinking"` instead of being degraded to text. + * Returns true when unsigned `thinking` blocks from prior assistant turns should + * be replayed as Anthropic-native thinking instead of demoted to text. + * + * Official Anthropic (matched via `isAnthropicApiBaseUrl`, which intentionally + * treats a missing baseUrl as official since `resolveAnthropicBaseUrl` routes + * it to `https://api.anthropic.com`) enforces signature-based thinking-chain + * integrity, so unsigned blocks must remain text there. Anthropic-compatible + * reasoning endpoints commonly emit unsigned thinking blocks while still + * expecting them back as `type: "thinking"` on continuation; demoting them + * loses the model's reasoning chain and can destabilize the next tool-call + * arguments (#2005). Known non-signing hosts are also preserved for + * compatibility. */ -function isNonSigningAnthropicEndpoint(model: Model<"anthropic-messages">): boolean { - // Known non-signing providers +function shouldReplayUnsignedThinking(model: Model<"anthropic-messages">): boolean { if (model.provider === "zai" || model.provider === "deepseek") return true; const baseUrl = model.baseUrl; - if (!baseUrl) return false; - try { - const hostname = new URL(baseUrl).hostname.toLowerCase(); - return hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com"); - } catch { - return false; + if (baseUrl) { + try { + const hostname = new URL(baseUrl).hostname.toLowerCase(); + if (hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com")) return true; + } catch { + // Fall through to the protocol-level reasoning rule below. + } } + return model.reasoning && !isAnthropicApiBaseUrl(baseUrl); } function buildToolResultBlock(model: Model<"anthropic-messages">, msg: ToolResultMessage): ContentBlockParam { @@ -2537,7 +2526,7 @@ export function convertAnthropicMessages( } if (block.thinking.trim().length === 0) continue; if (!block.thinkingSignature || block.thinkingSignature.trim().length === 0) { - if (isNonSigningAnthropicEndpoint(model)) { + if (shouldReplayUnsignedThinking(model)) { blocks.push({ type: "thinking", thinking: block.thinking.toWellFormed(), diff --git a/packages/ai/src/providers/azure-openai-responses.ts b/packages/ai/src/providers/azure-openai-responses.ts index 04027d02a..26b3f0a16 100644 --- a/packages/ai/src/providers/azure-openai-responses.ts +++ b/packages/ai/src/providers/azure-openai-responses.ts @@ -40,6 +40,7 @@ import { isOpenAIResponsesProgressEvent, normalizeResponsesToolCallIdForTransform, processResponsesStream, + repairOrphanResponsesToolCalls, } from "./openai-responses-shared"; import { transformMessages } from "./transform-messages"; @@ -347,7 +348,7 @@ function convertMessages( msgIndex++; } - return messages; + return repairOrphanResponsesToolCalls(messages); } function convertTools(tools: Tool[]): OpenAITool[] { diff --git a/packages/ai/src/providers/gitlab-duo.ts b/packages/ai/src/providers/gitlab-duo.ts index 50c43b372..382ce56c9 100644 --- a/packages/ai/src/providers/gitlab-duo.ts +++ b/packages/ai/src/providers/gitlab-duo.ts @@ -234,7 +234,8 @@ export function streamGitLabDuo( (async () => { try { - if (!options?.apiKey) { + const apiKey = typeof options?.apiKey === "string" ? options.apiKey : undefined; + if (!apiKey || !options) { throw new Error("Missing GitLab access token. Run /login gitlab-duo or set GITLAB_TOKEN."); } @@ -243,7 +244,7 @@ export function streamGitLabDuo( throw new Error(`Unsupported GitLab Duo model: ${model.id}`); } - const directAccess = await getDirectAccessToken(options.apiKey, options.fetch); + const directAccess = await getDirectAccessToken(apiKey, options.fetch); const headers = { ...directAccess.headers, ...options.headers, diff --git a/packages/ai/src/providers/openai-anthropic-shim.ts b/packages/ai/src/providers/openai-anthropic-shim.ts index f2dad56d7..a4f9b8fac 100644 --- a/packages/ai/src/providers/openai-anthropic-shim.ts +++ b/packages/ai/src/providers/openai-anthropic-shim.ts @@ -44,6 +44,9 @@ export function streamOpenAIAnthropicShim( ): AssistantMessageEventStream { const stream = new AssistantMessageEventStream(); const format = options?.format ?? config.defaultFormat; + // The resolver form of `apiKey` is resolved upstream in `streamSimple`; + // this shim only ever receives a static bearer string. + const apiKey = typeof options?.apiKey === "string" ? options.apiKey : undefined; (async () => { try { @@ -74,7 +77,7 @@ export function streamOpenAIAnthropicShim( : undefined; const innerStream = streamAnthropic(anthropicModel, context, { - apiKey: options?.apiKey, + apiKey, temperature: options?.temperature, topP: options?.topP, topK: options?.topK, @@ -103,7 +106,7 @@ export function streamOpenAIAnthropicShim( const reasoningEffort = options?.reasoning; const innerStream = streamOpenAICompletions(openaiModel, context, { - apiKey: options?.apiKey, + apiKey, temperature: options?.temperature, topP: options?.topP, topK: options?.topK, diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index f271bbc81..91abe9dc5 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -1,4 +1,4 @@ -import type { Effort } from "../../model-thinking"; +import type { Effort } from "../../effort"; import { requireSupportedEffort } from "../../model-thinking"; import type { Api, Model } from "../../types"; @@ -76,6 +76,83 @@ function filterInput(input: InputItem[] | undefined): InputItem[] | undefined { }); } +const CODEX_ORPHAN_OUTPUT_LIMIT = 16_000; +/** Placeholder output for a tool call whose result never landed in the input. */ +const CODEX_INTERRUPTED_TOOL_OUTPUT = + "[No tool output recorded: the tool call was interrupted before it produced a result.]"; + +function orphanFunctionOutputToMessage(item: InputItem, callId: string): InputItem { + const itemRecord = item as unknown as Record; + const toolName = typeof itemRecord.name === "string" ? itemRecord.name : "tool"; + let text = ""; + try { + const output = itemRecord.output; + text = typeof output === "string" ? output : JSON.stringify(output); + } catch { + text = String(itemRecord.output ?? ""); + } + if (text.length > CODEX_ORPHAN_OUTPUT_LIMIT) { + text = `${text.slice(0, CODEX_ORPHAN_OUTPUT_LIMIT)}\n...[truncated]`; + } + return { + type: "message", + role: "assistant", + content: `[Previous ${toolName} result; call_id=${callId}]: ${text}`, + } as InputItem; +} + +/** + * Repair both halves of unpaired tool exchanges so the Responses input grammar + * stays valid — the API rejects either orphan with a 400: + * + * - `function_call_output` with no matching `function_call` → folded into an + * assistant message (`400 No tool call found for function call output …`). + * Regression of #472 / #1351. + * - `function_call` / `custom_tool_call` with no matching `*_output` → a + * placeholder output is synthesized immediately after the call + * (`400 No tool output found for function call …`). Hit when the user + * branches/navigates the session tree to a node that ends on a tool call (the + * tool-result child is dropped from the reconstructed history) or when a turn + * is aborted/crashes after the call streamed but before its result persisted. + */ +function repairToolCallPairs(input: InputItem[]): InputItem[] { + const callIds = new Set(); + const outputCallIds = new Set(); + for (const item of input) { + const callId = typeof item.call_id === "string" ? item.call_id : undefined; + if (callId === undefined) continue; + if (item.type === "function_call" || item.type === "custom_tool_call") callIds.add(callId); + else if (item.type === "function_call_output" || item.type === "custom_tool_call_output") { + outputCallIds.add(callId); + } + } + + const repaired: InputItem[] = []; + for (const item of input) { + const callId = typeof item.call_id === "string" ? item.call_id : undefined; + + if (item.type === "function_call_output" && callId !== undefined && !callIds.has(callId)) { + repaired.push(orphanFunctionOutputToMessage(item, callId)); + continue; + } + + repaired.push(item); + + if ( + (item.type === "function_call" || item.type === "custom_tool_call") && + callId !== undefined && + !outputCallIds.has(callId) + ) { + repaired.push({ + type: item.type === "custom_tool_call" ? "custom_tool_call_output" : "function_call_output", + call_id: callId, + output: CODEX_INTERRUPTED_TOOL_OUTPUT, + } as InputItem); + } + } + return repaired; +} + export async function transformRequestBody( body: RequestBody, model: Model, @@ -87,39 +164,8 @@ export async function transformRequestBody( if (body.input && Array.isArray(body.input)) { body.input = filterInput(body.input); - if (body.input) { - const functionCallIds = new Set( - body.input - .filter(item => item.type === "function_call" && typeof item.call_id === "string") - .map(item => item.call_id as string), - ); - - body.input = body.input.map(item => { - if (item.type === "function_call_output" && typeof item.call_id === "string") { - const callId = item.call_id as string; - if (!functionCallIds.has(callId)) { - const itemRecord = item as unknown as Record; - const toolName = typeof itemRecord.name === "string" ? itemRecord.name : "tool"; - let text = ""; - try { - const output = itemRecord.output; - text = typeof output === "string" ? output : JSON.stringify(output); - } catch { - text = String(itemRecord.output ?? ""); - } - if (text.length > 16000) { - text = `${text.slice(0, 16000)}\n...[truncated]`; - } - return { - type: "message", - role: "assistant", - content: `[Previous ${toolName} result; call_id=${callId}]: ${text}`, - } as InputItem; - } - } - return item; - }); + body.input = repairToolCallPairs(body.input); } } diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index b8b5e591c..e58e4b8fd 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -10,7 +10,8 @@ import type { ChatCompletionToolMessageParam, } from "openai/resources/chat/completions"; import packageJson from "../../package.json" with { type: "json" }; -import { type Effort, getSupportedEfforts } from "../model-thinking"; +import type { Effort } from "../effort"; +import { getSupportedEfforts } from "../model-thinking"; import { calculateCost } from "../models"; import { getEnvApiKey } from "../stream"; import { @@ -1344,6 +1345,11 @@ function buildParams( if (compat.extraBody) { Object.assign(params, compat.extraBody); + if (model.provider === "fireworks" && params.reasoning_effort !== undefined) { + // Fireworks rejects simultaneous DeepSeek-style `thinking` toggles and + // OpenAI-style `reasoning_effort`; the effort field carries the user's level. + delete params.thinking; + } } return { params, toolStrictMode }; diff --git a/packages/ai/src/providers/openai-responses-shared.ts b/packages/ai/src/providers/openai-responses-shared.ts index 135f92589..28532dcc8 100644 --- a/packages/ai/src/providers/openai-responses-shared.ts +++ b/packages/ai/src/providers/openai-responses-shared.ts @@ -212,6 +212,59 @@ export function repairOrphanResponsesToolOutputs(input: ResponseInput): Response }); } +/** Placeholder output for a tool call whose result is absent from the input. */ +const ORPHAN_TOOL_CALL_PLACEHOLDER = + "[No tool output recorded: the tool call was interrupted before it produced a result.]"; + +/** + * Synthesize a placeholder `function_call_output` / `custom_tool_call_output` + * for every `function_call` / `custom_tool_call` whose `call_id` has no matching + * output later in the same input. The Responses API rejects an unpaired call + * with `400 No tool output found for function call …`. + * + * Orphan calls surface when the user branches/navigates the session tree to a + * node that ends on a tool call (the tool-result child is excluded from the + * reconstructed history) or when a turn is aborted/crashes after the call + * streamed but before its result persisted. Dropping the call would erase the + * assistant's action; a placeholder output keeps the call visible so the model + * can recover (e.g. re-issue the call). Symmetric to + * {@link repairOrphanResponsesToolOutputs}. + */ +export function repairOrphanResponsesToolCalls(input: ResponseInput): ResponseInput { + const outputCallIds = new Set(); + for (const item of input) { + const t = (item as { type?: string }).type; + if (t !== "function_call_output" && t !== "custom_tool_call_output") continue; + const callId = (item as { call_id?: unknown }).call_id; + if (typeof callId === "string") outputCallIds.add(callId); + } + let hasOrphan = false; + for (const item of input) { + const t = (item as { type?: string }).type; + if (t !== "function_call" && t !== "custom_tool_call") continue; + const callId = (item as { call_id?: unknown }).call_id; + if (typeof callId === "string" && !outputCallIds.has(callId)) { + hasOrphan = true; + break; + } + } + if (!hasOrphan) return input; + const repaired: ResponseInput = []; + for (const item of input) { + repaired.push(item); + const t = (item as { type?: string }).type; + if (t !== "function_call" && t !== "custom_tool_call") continue; + const callId = (item as { call_id?: unknown }).call_id; + if (typeof callId !== "string" || outputCallIds.has(callId)) continue; + repaired.push({ + type: t === "custom_tool_call" ? "custom_tool_call_output" : "function_call_output", + call_id: callId, + output: ORPHAN_TOOL_CALL_PLACEHOLDER, + } as ResponseInput[number]); + } + return repaired; +} + export function convertResponsesInputContent( content: string | Array, supportsImages: boolean, @@ -395,7 +448,7 @@ export async function processResponsesStream( model: Model, options?: ProcessResponsesStreamOptions, ): Promise { - type StreamingToolCallBlock = ToolCall & { partialJson: string; lastParseLen?: number }; + type StreamingToolCallBlock = ToolCall & { partialJson: string; lastParseLen?: number; argumentsDone?: boolean }; interface StreamingItem { item: ResponseReasoningItem | ResponseOutputMessage | ResponseFunctionToolCall | ResponseCustomToolCall; block: ThinkingContent | TextContent | StreamingToolCallBlock; @@ -406,17 +459,28 @@ export async function processResponsesStream( // see https://github.com/can1357/oh-my-pi/issues/1880 — llama.cpp emits parallel // function_call deltas interleaved, and a singleton `current` reference would // fold them into the wrong block and drop arguments on every call but the last. + // + // llama.cpp's `to_json_oaicompat_resp` (issue #2015) compounds this: `output_item.added` + // for function_call/custom_tool_call carries `item.call_id` but no `item.id` and no + // `output_index`, while the matching `function_call_arguments.delta` carries + // `item_id = "fc_"`. Registering function-call items by `call_id` as a + // secondary key lets the delta lookup find the right block on hosts that emit one + // identifier but not the other. const openItemsByOutputIndex = new Map(); const openItemsByItemId = new Map(); let lastOpenItem: StreamingItem | null = null; + const openItemsInOrder: StreamingItem[] = []; const registerOpenItem = ( outputIndex: number | undefined, itemId: string | undefined, entry: StreamingItem, + alternateItemKey?: string, ): void => { if (typeof outputIndex === "number") openItemsByOutputIndex.set(outputIndex, entry); if (itemId) openItemsByItemId.set(itemId, entry); + if (alternateItemKey && alternateItemKey !== itemId) openItemsByItemId.set(alternateItemKey, entry); + openItemsInOrder.push(entry); lastOpenItem = entry; }; const lookupOpenItem = (event: { output_index?: number; item_id?: string }): StreamingItem | undefined => { @@ -431,13 +495,37 @@ export async function processResponsesStream( // Fallback for tests / mock providers that omit identifiers on stream events. return lastOpenItem ?? undefined; }; + const hasOpenItemKey = (event: { output_index?: number; item_id?: string }): boolean => + typeof event.output_index === "number" || event.item_id !== undefined; + const lookupOpenFunctionCallItem = (event: { + output_index?: number; + item_id?: string; + }): StreamingItem | undefined => { + if (hasOpenItemKey(event)) return lookupOpenItem(event); + for (const candidate of openItemsInOrder) { + if ( + candidate.item.type === "function_call" && + candidate.block.type === "toolCall" && + !candidate.block.argumentsDone + ) { + return candidate; + } + } + return lastOpenItem?.item.type === "function_call" ? lastOpenItem : undefined; + }; const closeOpenItem = ( outputIndex: number | undefined, itemId: string | undefined, entry: StreamingItem | undefined, + alternateItemKey?: string, ): void => { if (typeof outputIndex === "number") openItemsByOutputIndex.delete(outputIndex); if (itemId) openItemsByItemId.delete(itemId); + if (alternateItemKey && alternateItemKey !== itemId) openItemsByItemId.delete(alternateItemKey); + if (entry) { + const index = openItemsInOrder.indexOf(entry); + if (index >= 0) openItemsInOrder.splice(index, 1); + } if (entry && lastOpenItem === entry) lastOpenItem = null; }; const contentIndexOf = (block: ThinkingContent | TextContent | StreamingToolCallBlock): number => @@ -473,7 +561,7 @@ export async function processResponsesStream( partialJson: item.arguments || "", }; output.content.push(block); - registerOpenItem(event.output_index, item.id, { item, block }); + registerOpenItem(event.output_index, item.id, { item, block }, item.call_id); stream.push({ type: "toolcall_start", contentIndex: contentIndexOf(block), partial: output }); } else if (item.type === "custom_tool_call") { const block: StreamingToolCallBlock = { @@ -491,7 +579,7 @@ export async function processResponsesStream( partialJson: item.input ?? "", }; output.content.push(block); - registerOpenItem(event.output_index, item.id, { item, block }); + registerOpenItem(event.output_index, item.id, { item, block }, item.call_id); stream.push({ type: "toolcall_start", contentIndex: contentIndexOf(block), partial: output }); } } else if (event.type === "response.reasoning_summary_part.added") { @@ -584,7 +672,7 @@ export async function processResponsesStream( } } } else if (event.type === "response.function_call_arguments.delta") { - const entry = lookupOpenItem(event); + const entry = lookupOpenFunctionCallItem(event); if (entry?.item.type === "function_call" && entry.block.type === "toolCall") { const block = entry.block; block.partialJson += event.delta; @@ -601,11 +689,12 @@ export async function processResponsesStream( }); } } else if (event.type === "response.function_call_arguments.done") { - const entry = lookupOpenItem(event); + const entry = lookupOpenFunctionCallItem(event); if (entry?.item.type === "function_call" && entry.block.type === "toolCall") { const block = entry.block; block.partialJson = event.arguments; block.arguments = parseStreamingJson(block.partialJson); + block.argumentsDone = true; delete (block as { partialJson?: string }).partialJson; delete (block as { lastParseLen?: number }).lastParseLen; } @@ -631,7 +720,10 @@ export async function processResponsesStream( } else if (event.type === "response.output_item.done") { const item = structuredCloneJSON(event.item); options?.onOutputItemDone?.(item); - const entry = lookupOpenItem({ output_index: event.output_index, item_id: item.id }); + const entry = + item.type === "function_call" || item.type === "custom_tool_call" + ? lookupOpenItem({ output_index: event.output_index, item_id: item.id ?? item.call_id }) + : lookupOpenItem({ output_index: event.output_index, item_id: item.id }); if (item.type === "reasoning") { const thinking = item.summary?.length > 0 @@ -668,9 +760,11 @@ export async function processResponsesStream( closeOpenItem(event.output_index, item.id, entry); } else if (item.type === "function_call") { const block = entry?.block.type === "toolCall" ? entry.block : undefined; - const args = block?.partialJson - ? parseStreamingJson(block.partialJson) - : parseStreamingJson(item.arguments || "{}"); + const args = block?.argumentsDone + ? block.arguments + : block?.partialJson + ? parseStreamingJson(block.partialJson) + : parseStreamingJson(item.arguments || "{}"); const toolCall: ToolCall = { type: "toolCall", id: encodeResponsesToolCallId(item.call_id, item.id), @@ -685,9 +779,10 @@ export async function processResponsesStream( block.arguments = args; delete (block as { partialJson?: string }).partialJson; delete (block as { lastParseLen?: number }).lastParseLen; + delete (block as { argumentsDone?: boolean }).argumentsDone; } const contentIndex = block ? contentIndexOf(block) : output.content.length - 1; - closeOpenItem(event.output_index, item.id, entry); + closeOpenItem(event.output_index, item.id, entry, item.call_id); stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); } else if (item.type === "custom_tool_call") { const block = entry?.block.type === "toolCall" ? entry.block : undefined; @@ -700,7 +795,7 @@ export async function processResponsesStream( customWireName: item.name, }; const contentIndex = block ? contentIndexOf(block) : output.content.length - 1; - closeOpenItem(event.output_index, item.id, entry); + closeOpenItem(event.output_index, item.id, entry, item.call_id); stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); } } else if (event.type === "response.completed") { diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index ac1684b43..ed6de0481 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -62,6 +62,7 @@ import { isOpenAIResponsesProgressEvent, normalizeResponsesToolCallIdForTransform, processResponsesStream, + repairOrphanResponsesToolCalls, repairOrphanResponsesToolOutputs, } from "./openai-responses-shared"; import { transformMessages } from "./transform-messages"; @@ -614,7 +615,7 @@ function convertConversationMessages( msgIndex++; } - return repairOrphanResponsesToolOutputs(messages); + return repairOrphanResponsesToolCalls(repairOrphanResponsesToolOutputs(messages)); } /** diff --git a/packages/ai/src/providers/pi-native-client.ts b/packages/ai/src/providers/pi-native-client.ts index b5df79636..68f479d2a 100644 --- a/packages/ai/src/providers/pi-native-client.ts +++ b/packages/ai/src/providers/pi-native-client.ts @@ -149,7 +149,10 @@ export function streamPiNative( try { const url = resolveStreamUrl(model as Model); const fetchImpl = options?.fetch ?? globalThis.fetch; - const headers = buildHeaders(model as Model, options?.apiKey); + const headers = buildHeaders( + model as Model, + typeof options?.apiKey === "string" ? options.apiKey : undefined, + ); const body = JSON.stringify({ modelId: model.id, context, diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 437a49344..f592dcd89 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -3,7 +3,8 @@ import * as os from "node:os"; import * as path from "node:path"; import { $env, $pickenv, extractHttpStatusFromError } from "@oh-my-pi/pi-utils"; import { getCustomApi } from "./api-registry"; -import type { Effort } from "./model-thinking"; +import { AUTH_RETRY_STEPS, isApiKeyResolver, resolveRetryKey } from "./auth-retry"; +import type { Effort } from "./effort"; import { mapEffortToAnthropicAdaptiveEffort, mapEffortToGoogleThinkingLevel, @@ -443,12 +444,15 @@ export function streamSimple( options?: SimpleStreamOptions, ): AssistantMessageEventStream { const requestOptions = withRequestDebugFetch(options); - const retryApiKey = requestOptions?.onAuthError - ? (requestOptions.apiKey ?? getEnvApiKey(model.provider)) - : undefined; - if (retryApiKey) { + const apiKeyResolver = isApiKeyResolver(requestOptions?.apiKey) ? requestOptions.apiKey : undefined; + if (apiKeyResolver) { const outer = new AssistantMessageEventStream(); - const onAuthError = requestOptions!.onAuthError!; + const signal = requestOptions?.signal; + // One inner attempt against a resolved string key. When + // `captureAuthFailure` is set, a retryable auth error that arrives before + // any replay-unsafe event is buffered and returned (so the caller can + // retry with a fresh key) instead of surfaced. The terminal attempt + // clears the flag and emits whatever it gets. const runAttempt = async (apiKey: string, captureAuthFailure: boolean): Promise => { const bufferedEvents: AssistantMessageEvent[] = []; let emittedReplayUnsafeEvent = false; @@ -458,7 +462,7 @@ export function streamSimple( }; try { - const inner = streamSimple(model, context, { ...requestOptions, apiKey, onAuthError: undefined }); + const inner = streamSimple(model, context, { ...requestOptions, apiKey }); for await (const event of inner) { if (!emittedReplayUnsafeEvent && event.type === "start") { bufferedEvents.push(event); @@ -510,19 +514,32 @@ export function streamSimple( }; void (async () => { - const failure = await runAttempt(retryApiKey, true); - if (!failure) return; - let nextKey: string | undefined; + let lastKey: string | undefined; try { - nextKey = await onAuthError(model.provider, retryApiKey, failure.error); + lastKey = (await apiKeyResolver({ lastChance: false, error: undefined, signal })) || undefined; } catch { - nextKey = undefined; + lastKey = undefined; } - if (!nextKey || nextKey === retryApiKey) { - emitFailure(failure); + if (lastKey === undefined) { + outer.fail(new Error(`No API key for provider: ${model.provider}`)); return; } - await runAttempt(nextKey, false); + let failure = await runAttempt(lastKey, true); + if (!failure) return; + // a/b/c policy: refresh the same account (lastChance=false), then + // switch to a sibling (lastChance=true). A step is skipped when the + // resolver yields the same key it just tried or `undefined`; the + // final step's attempt clears the capture flag so it emits directly. + for (let step = 0; step < AUTH_RETRY_STEPS.length; step++) { + const nextKey = await resolveRetryKey(apiKeyResolver, AUTH_RETRY_STEPS[step]!, failure.error, signal); + if (nextKey === undefined || nextKey === lastKey) continue; + lastKey = nextKey; + const isLastStep = step === AUTH_RETRY_STEPS.length - 1; + const next = await runAttempt(nextKey, !isLastStep); + if (!next) return; + failure = next; + } + emitFailure(failure); })(); return outer; } @@ -553,7 +570,10 @@ export function streamSimple( return stream(model, context, providerOptions); } - const apiKey = requestOptions?.apiKey || getEnvApiKey(model.provider); + // The resolver form is handled by the wrapper above; only a static string + // key reaches this point. + const apiKey = + (typeof requestOptions?.apiKey === "string" ? requestOptions.apiKey : undefined) || getEnvApiKey(model.provider); if (!apiKey) { throw new Error(`No API key for provider: ${model.provider}`); } @@ -724,7 +744,7 @@ function mapOptionsForApi( repetitionPenalty: options?.repetitionPenalty, maxTokens: options?.maxTokens ?? model.maxTokens, signal: options?.signal, - apiKey: apiKey || options?.apiKey, + apiKey: apiKey ?? (typeof options?.apiKey === "string" ? options.apiKey : undefined), cacheRetention: options?.cacheRetention, headers: options?.headers, initiatorOverride: options?.initiatorOverride, diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index 9b03d99d8..566574be4 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -1,4 +1,5 @@ import type { ZodType, z } from "zod/v4"; +import type { ApiKey } from "./auth-retry"; import type { BedrockOptions } from "./providers/amazon-bedrock"; import type { AnthropicOptions } from "./providers/anthropic"; import type { AzureOpenAIResponsesOptions } from "./providers/azure-openai-responses"; @@ -151,7 +152,7 @@ export type KnownProvider = | "lm-studio"; export type Provider = KnownProvider | string; -import type { Effort } from "./model-thinking"; +import type { Effort } from "./effort"; /** Token budgets for each thinking level (token-based providers only) */ export type ThinkingBudgets = { [key in Effort]?: number }; @@ -298,12 +299,6 @@ export interface StreamOptions { maxTokens?: number; signal?: AbortSignal; apiKey?: string; - /** - * Called when a provider returns 401 before any replay-unsafe assistant - * event has been emitted. Returning a different key retries the provider - * request once. - */ - onAuthError?: (provider: string, apiKey: string, error: unknown) => Promise; cacheRetention?: CacheRetention; /** * Additional headers to include in provider requests. @@ -415,7 +410,15 @@ export interface StreamOptions { } // Unified options with reasoning passed to streamSimple() and completeSimple() -export interface SimpleStreamOptions extends StreamOptions { +export interface SimpleStreamOptions extends Omit { + /** + * API key for the request: either a static bearer string, or an + * {@link ApiKeyResolver} that mints/rotates the key across the central + * a/b/c auth-retry policy. `streamSimple`/`completeSimple` resolve a + * resolver to a string before per-provider dispatch, so providers only + * ever see the resolved {@link StreamOptions.apiKey} string. + */ + apiKey?: ApiKey; reasoning?: Effort; /** * Force-disable reasoning for the request even when the model supports it. diff --git a/packages/ai/src/usage/google-antigravity.ts b/packages/ai/src/usage/google-antigravity.ts index 26963d7b5..54e2e11cf 100644 --- a/packages/ai/src/usage/google-antigravity.ts +++ b/packages/ai/src/usage/google-antigravity.ts @@ -148,10 +148,17 @@ async function fetchAntigravityUsage(params: UsageFetchParams, ctx: UsageFetchCo } const data = (await response.json()) as AntigravityUsageResponse; - const limits: UsageLimit[] = []; + + // The API returns per-model quota entries, but quota is shared across + // models within the same tier. Deduplicate by (tier, windowId) so one + // account doesn't produce 15 redundant bars. + const deduped = new Map< + string, + { amount: UsageAmount; window: UsageWindow | undefined; tier: string | undefined } + >(); let earliestReset: number | undefined; - for (const [modelId, modelInfo] of Object.entries(data.models ?? {})) { + for (const [_modelId, modelInfo] of Object.entries(data.models ?? {})) { const quotaInfos = normalizeQuotaInfos(modelInfo); for (const quotaInfo of quotaInfos) { const amount = buildAmount(quotaInfo); @@ -159,35 +166,81 @@ async function fetchAntigravityUsage(params: UsageFetchParams, ctx: UsageFetchCo if (window?.resetsAt) { earliestReset = earliestReset ? Math.min(earliestReset, window.resetsAt) : window.resetsAt; } - const labelBase = modelInfo.displayName || modelId; - const label = quotaInfo.tier ? `${labelBase} (${quotaInfo.tier})` : labelBase; - const windowId = window?.id ?? "default"; - limits.push({ - id: `${modelId}:${quotaInfo.tier ?? "default"}:${windowId}`, - label, - scope: { - provider: params.provider, - accountId: credential.accountId, - projectId: credential.projectId, - modelId, - tier: quotaInfo.tier, - windowId, - }, - window, - amount, - status: getUsageStatus(amount.remainingFraction), - }); + const tier = (quotaInfo.tier ?? "default").toLowerCase(); + // Use quotaInfo.windowId even when parseWindow returns undefined + // (no resetTime) — separate windows must not collapse to "default". + const windowId = quotaInfo.windowId ?? window?.id ?? "default"; + const key = `${tier}|${windowId}`; + const existing = deduped.get(key); + if (!existing) { + deduped.set(key, { amount, window, tier: quotaInfo.tier }); + continue; + } + // Merge: keep the entry with fraction data for the bar, but + // also keep any window with a reset time so "resets in…" survives. + const eFrac = existing.amount.remainingFraction; + const cFrac = amount.remainingFraction; + const eHasFrac = eFrac !== undefined; + const cHasFrac = cFrac !== undefined; + + let bestAmount = existing.amount; + let bestWindow = existing.window?.resetsAt ? existing.window : (window ?? existing.window); + let bestTier = existing.tier ?? quotaInfo.tier; + + if (!eHasFrac && cHasFrac) { + bestAmount = amount; + bestTier = quotaInfo.tier ?? existing.tier; + } else if (eHasFrac && cHasFrac && cFrac! < eFrac!) { + bestAmount = amount; + bestTier = quotaInfo.tier ?? existing.tier; + } + // Always merge in window with reset time if the current + // best doesn't have one. + if (!bestWindow?.resetsAt && window?.resetsAt) { + bestWindow = window; + } + deduped.set(key, { amount: bestAmount, window: bestWindow, tier: bestTier }); } } + const limits: UsageLimit[] = []; + for (const [key, entry] of deduped) { + const [tier, windowId] = key.split("|") as [string, string]; + const label = "Usage"; + limits.push({ + id: `${params.provider}:${tier}:${windowId}`, + label, + scope: { + provider: params.provider, + accountId: credential.accountId, + projectId: credential.projectId, + tier: entry.tier, + windowId, + }, + window: entry.window, + amount: entry.amount, + status: getUsageStatus(entry.amount.remainingFraction), + }); + } + + limits.sort((a, b) => { + const aFraction = a.amount.remainingFraction ?? 1; + const bFraction = b.amount.remainingFraction ?? 1; + return aFraction - bFraction; + }); + + const metadata: UsageReport["metadata"] = { + endpoint: url, + projectId: credential.projectId, + }; + if (credential.email) metadata.email = credential.email; + if (credential.accountId) metadata.accountId = credential.accountId; + const report: UsageReport = { provider: params.provider, fetchedAt: nowMs, limits, - metadata: { - endpoint: url, - projectId: credential.projectId, - }, + metadata, raw: data, }; diff --git a/packages/ai/src/utils/http-inspector.ts b/packages/ai/src/utils/http-inspector.ts index 31a6ccb06..df738c3d8 100644 --- a/packages/ai/src/utils/http-inspector.ts +++ b/packages/ai/src/utils/http-inspector.ts @@ -1,5 +1,5 @@ import * as path from "node:path"; -import { extractHttpStatusFromError, getLogsDir } from "@oh-my-pi/pi-utils"; +import { extractHttpStatusFromError, getLogsDir, isBunTestRuntime } from "@oh-my-pi/pi-utils"; import { isCopilotTransientModelError } from "./retry.js"; import { formatErrorMessageWithRetryAfter } from "./retry-after.js"; @@ -31,7 +31,9 @@ export async function appendRawHttpRequestDumpFor400( error: unknown, dump: RawHttpRequestDump | undefined, ): Promise { - if (!dump || extractHttpStatusFromError(error) !== 400) { + // Never persist dumps under the test runner: providers exercise the 400 path + // with mocked fetch responses, which would otherwise litter the real ~/.omp logs. + if (!dump || isBunTestRuntime() || extractHttpStatusFromError(error) !== 400) { return message; } diff --git a/packages/ai/src/utils/request-debug.ts b/packages/ai/src/utils/request-debug.ts index 193f51be6..daa6831c4 100644 --- a/packages/ai/src/utils/request-debug.ts +++ b/packages/ai/src/utils/request-debug.ts @@ -170,6 +170,7 @@ class FileRequestDebugSession implements RequestDebugSession { class FileRequestDebugResponseLog implements RequestDebugResponseLog { #handle: fs.FileHandle | undefined; #pending: Promise = Promise.resolve(); + #closed: Promise | undefined; constructor(handle: fs.FileHandle) { this.#handle = handle; @@ -184,15 +185,19 @@ class FileRequestDebugResponseLog implements RequestDebugResponseLog { }); } - async close(): Promise { + close(): Promise { + if (this.#closed) return this.#closed; const handle = this.#handle; - if (!handle) return; + if (!handle) return Promise.resolve(); this.#handle = undefined; - try { - await this.#pending; - } finally { - await handle.close(); - } + this.#closed = (async () => { + try { + await this.#pending; + } finally { + await handle.close(); + } + })(); + return this.#closed; } } diff --git a/packages/ai/src/utils/schema/fields.ts b/packages/ai/src/utils/schema/fields.ts index 41e9aacf1..b25006248 100644 --- a/packages/ai/src/utils/schema/fields.ts +++ b/packages/ai/src/utils/schema/fields.ts @@ -154,6 +154,22 @@ export const CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS: Record = buildAllCcaTypeSpecificKeys(); + +function buildAllCcaTypeSpecificKeys(): Record { + const all: Record = {}; + for (const typeKeys of Object.values(CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS)) { + for (const key in typeKeys) { + all[key] = true; + } + } + return all; +} + /** * Cloud Code Assist shared schema keys allowed on any type. * Used alongside CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS for CCA combiner collapsing. diff --git a/packages/ai/src/utils/schema/json-schema-validator.ts b/packages/ai/src/utils/schema/json-schema-validator.ts index 93a69cf15..79a93052e 100644 --- a/packages/ai/src/utils/schema/json-schema-validator.ts +++ b/packages/ai/src/utils/schema/json-schema-validator.ts @@ -20,6 +20,14 @@ export interface JsonSchemaValidationIssue { message: string; expectedTypes?: string[]; keyword?: string; + /** + * Marks issues that originate inside a failed `anyOf` / `oneOf` branch. + * Consumers such as the tool-argument coercion layer use this to avoid + * applying type repairs (e.g. singleton-array wrapping) that would be + * authoritative outside of a combinator but are only one candidate + * branch's expectation here. + */ + fromUnionBranch?: boolean; } export interface JsonSchemaValidationResult { @@ -242,7 +250,17 @@ function validateSchemaNode( const branchValid = keyword === "anyOf" ? matches > 0 : matches === 1; if (!branchValid) { if (matches === 0 && firstIssues && firstIssues.length > 0) { - issues.push(...firstIssues); + // Only tag issues that sit at the combinator's own path as + // union-branch; deeper issues describe a specific field within + // the failed branch and should remain individually repairable. + const unionDepth = path.length; + for (const branchIssue of firstIssues) { + if (branchIssue.path.length === unionDepth) { + issues.push({ ...branchIssue, fromUnionBranch: true }); + } else { + issues.push(branchIssue); + } + } } else { pushIssue( issues, diff --git a/packages/ai/src/utils/schema/normalize.ts b/packages/ai/src/utils/schema/normalize.ts index 1b21afd67..ac50eccb7 100644 --- a/packages/ai/src/utils/schema/normalize.ts +++ b/packages/ai/src/utils/schema/normalize.ts @@ -11,6 +11,7 @@ import { dereferenceJsonSchema } from "./dereference"; import { upgradeJsonSchemaTo202012 } from "./draft"; import { areJsonValuesEqual, mergePropertySchemas } from "./equality"; import { + ALL_CCA_TYPE_SPECIFIC_KEYS, CLOUD_CODE_ASSIST_SHARED_SCHEMA_KEYS, CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS, COMBINATOR_KEYS, @@ -501,12 +502,32 @@ function collapseMixedTypeCombinerVariants(schema: JsonObject, combiner: "anyOf" if (variantTypes.length < 2 || variantTypes.every(type => type === "object")) { return schema; } - const nextSchema = copySchemaWithout(schema, combiner); const nonNullTypes = variantTypes.filter(t => t !== "null"); - nextSchema.type = nonNullTypes[0] ?? variantTypes[0]; + const chosenType: string = nonNullTypes[0] ?? variantTypes[0]; + nextSchema.type = chosenType; + const chosenTypeAllowedKeys = CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS[chosenType] ?? {}; + + // Strip sibling keys that were copied from the parent and belong to a + // different type (e.g. `items` sibling on a now-string-typed schema). + for (const key in nextSchema) { + if (!Object.hasOwn(nextSchema, key)) continue; + if (key === "type") continue; + if ( + Object.hasOwn(ALL_CCA_TYPE_SPECIFIC_KEYS, key) && + !Object.hasOwn(chosenTypeAllowedKeys, key) && + !Object.hasOwn(CLOUD_CODE_ASSIST_SHARED_SCHEMA_KEYS, key) + ) { + delete nextSchema[key]; + } + } + for (const key in mergedVariantFields) { if (!Object.hasOwn(mergedVariantFields, key)) continue; + // Drop type-specific keys that don't belong to the chosen type + if (!Object.hasOwn(chosenTypeAllowedKeys, key) && !Object.hasOwn(CLOUD_CODE_ASSIST_SHARED_SCHEMA_KEYS, key)) { + continue; + } const value = mergedVariantFields[key]; const existingValue = nextSchema[key]; if (existingValue !== undefined && !areJsonValuesEqual(existingValue, value)) { diff --git a/packages/ai/src/utils/validation.ts b/packages/ai/src/utils/validation.ts index 7f82626ec..506c72978 100644 --- a/packages/ai/src/utils/validation.ts +++ b/packages/ai/src/utils/validation.ts @@ -724,6 +724,7 @@ interface FlatIssue { keyword: "type" | "unrecognized" | "other"; instancePath: string; expectedTypes: string[]; + unionBranch: boolean; } /** @@ -759,12 +760,12 @@ function mapZodExpectedToJsonSchemaType(expected: unknown): string | null { */ function flattenIssues(issues: ReadonlyArray): FlatIssue[] { const out: FlatIssue[] = []; - const walk = (issue: ZodIssue, prefix: ReadonlyArray): void => { + const walk = (issue: ZodIssue, prefix: ReadonlyArray, unionBranch: boolean): void => { const fullPath = prefix.length === 0 ? issue.path : [...prefix, ...issue.path]; if (issue.code === "invalid_type") { const mapped = mapZodExpectedToJsonSchemaType((issue as { expected?: unknown }).expected); if (mapped) { - out.push({ keyword: "type", instancePath: pathToPointer(fullPath), expectedTypes: [mapped] }); + out.push({ keyword: "type", instancePath: pathToPointer(fullPath), expectedTypes: [mapped], unionBranch }); return; } } @@ -775,6 +776,7 @@ function flattenIssues(issues: ReadonlyArray): FlatIssue[] { keyword: "unrecognized", instancePath: pathToPointer([...fullPath, key]), expectedTypes: [], + unionBranch, }); } return; @@ -782,17 +784,21 @@ function flattenIssues(issues: ReadonlyArray): FlatIssue[] { if (issue.code === "invalid_union") { const inner = (issue as unknown as { errors?: ReadonlyArray> }).errors; if (inner) { + // A union-branch issue only competes with a sibling branch when it + // sits at the union node's own path. Issues whose own path is + // non-empty live on a deeper field that an already-identified + // branch owns, so the singleton-array repair should still apply. for (const branch of inner) { for (const child of branch) { - walk(child, fullPath); + walk(child, fullPath, child.path.length === 0); } } } return; } - out.push({ keyword: "other", instancePath: pathToPointer(fullPath), expectedTypes: [] }); + out.push({ keyword: "other", instancePath: pathToPointer(fullPath), expectedTypes: [], unionBranch }); }; - for (const issue of issues) walk(issue, []); + for (const issue of issues) walk(issue, [], false); return out; } @@ -801,7 +807,9 @@ function flattenIssues(issues: ReadonlyArray): FlatIssue[] { * * Two kinds of repair are applied: * - **type**: when a value is a JSON-encoded string and the schema wants - * something else, parse it and substitute the parsed value. + * something else, parse it and substitute the parsed value. When a + * non-union schema wants an array but receives a singleton value, wrap that + * value in a one-element array. * - **unrecognized**: when a strict object received an extra key (Zod's * `unrecognized_keys` or JSON Schema's `additionalProperties: false`), * drop that key so re-validation succeeds. This effectively coerces every @@ -811,6 +819,7 @@ function flattenIssues(issues: ReadonlyArray): FlatIssue[] { * The function is safe and conservative: * - Only processes "type" and "unrecognized" issues * - Only attempts JSON coercion on string values + * - Only wraps singleton array values for non-union type expectations * - Only accepts parsed results that match the expected type * - Clones the args object before mutation (copy-on-write) */ @@ -836,12 +845,16 @@ function coerceArgsFromIssues(args: unknown, issues: FlatIssue[]): { value: unkn if (issue.expectedTypes.length === 0) continue; const currentValue = getValueAtPointer(nextArgs, issue.instancePath); - if (typeof currentValue !== "string") continue; - - const result = tryParseJsonForTypes(currentValue, issue.expectedTypes); + const result = + typeof currentValue === "string" + ? tryParseJsonForTypes(currentValue, issue.expectedTypes) + : { value: currentValue, changed: false }; const coercedValue = result.changed ? result.value - : issue.expectedTypes.includes("array") + : issue.expectedTypes.includes("array") && + !issue.unionBranch && + currentValue !== undefined && + !Array.isArray(currentValue) ? [currentValue] : undefined; if (coercedValue === undefined) continue; @@ -905,17 +918,20 @@ function preserveUnknownRootFields(input: unknown, parsed: unknown): unknown { function flattenJsonSchemaIssues(issues: ReadonlyArray): FlatIssue[] { return issues.map(issue => { + const unionBranch = issue.fromUnionBranch === true; if (issue.keyword === "additionalProperties") { return { keyword: "unrecognized", instancePath: pathToPointer(issue.path), expectedTypes: [], + unionBranch, }; } return { keyword: issue.keyword === "type" ? "type" : "other", instancePath: pathToPointer(issue.path), expectedTypes: issue.expectedTypes ?? [], + unionBranch, }; }); } diff --git a/packages/ai/test/anthropic-alignment.test.ts b/packages/ai/test/anthropic-alignment.test.ts index b40a1b8b9..dd4878a7c 100644 --- a/packages/ai/test/anthropic-alignment.test.ts +++ b/packages/ai/test/anthropic-alignment.test.ts @@ -203,6 +203,11 @@ describe("Anthropic request fingerprint alignment", () => { }); it("matches CC system-block layout: billing and instruction uncached, context cached in order", () => { + // We mimic Claude Code's billing+instruction system layout but do NOT emit + // the `scope: "global"` field that CC attaches to its middle breakpoint — + // `prompt-caching-scope-2026-01-05` only works against canonical + // `api.anthropic.com`, and third-party Anthropic-compatible proxies + // (z.ai, openrouter, g0i, …) reject the unknown field outright. const blocks = buildAnthropicSystemBlocks(["Stay concise."], { includeClaudeCodeInstruction: true, extraInstructions: ["Use citations when possible"], @@ -217,7 +222,7 @@ describe("Anthropic request fingerprint alignment", () => { expect(blocks?.[2]).toEqual({ type: "text", text: "Use citations when possible", - cache_control: { type: "ephemeral", scope: "global" }, + cache_control: { type: "ephemeral" }, }); expect(blocks?.[3]).toEqual({ type: "text", @@ -239,7 +244,7 @@ describe("Anthropic request fingerprint alignment", () => { expect(payload.system?.[0]?.cache_control).toBeUndefined(); expect(payload.system?.[1]?.text).toBe(claudeCodeSystemInstruction); expect(payload.system?.[1]?.cache_control).toBeUndefined(); - expect(payload.system?.[2]?.cache_control).toEqual({ type: "ephemeral", ttl: "1h", scope: "global" }); + expect(payload.system?.[2]?.cache_control).toEqual({ type: "ephemeral", ttl: "1h" }); const content = payload.messages?.[0]?.content; expect(Array.isArray(content)).toBe(true); expect(Array.isArray(content) ? content[0]?.cache_control : undefined).toEqual({ diff --git a/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts new file mode 100644 index 000000000..9b3f8dcbc --- /dev/null +++ b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts @@ -0,0 +1,164 @@ +import { describe, expect, it } from "bun:test"; +import { convertAnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic"; +import type { AssistantMessage, Message, Model, ToolResultMessage, UserMessage } from "@oh-my-pi/pi-ai/types"; + +/** + * Regression: Anthropic-compatible reasoning endpoints often emit `thinking` + * blocks without a first-party Anthropic signature, but still expect those + * blocks back as native `type: "thinking"` on continuation. Demoting unsigned + * thinking to text strips the reasoning chain and can destabilize follow-up + * tool-call argument serialization (the upstream cause behind #2005's `todo` + * renderer crash). + * + * Official Anthropic remains conservative: unsigned thinking is demoted to text + * there because the first-party API enforces signature-based integrity. + */ +function makeModel(overrides: Partial> = {}): Model<"anthropic-messages"> { + return { + api: "anthropic-messages", + provider: "custom-anthropic", + id: "reasoning-model", + name: "Reasoning Anthropic-Compatible Model", + baseUrl: "https://llm.example.com/anthropic", + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + maxTokens: 8_192, + contextWindow: 200_000, + reasoning: true, + ...overrides, + }; +} + +function makeUser(text = "continue"): UserMessage { + return { role: "user", content: text, timestamp: 0 }; +} + +function makeAssistantThinking(thinking: string, tail: AssistantMessage["content"][number][] = []): AssistantMessage { + return { + role: "assistant", + content: [{ type: "thinking", thinking, thinkingSignature: "" }, ...tail], + api: "anthropic-messages", + provider: "custom-anthropic", + model: "reasoning-model", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "toolUse", + timestamp: 0, + }; +} + +interface WireThinkingBlock { + type: "thinking"; + thinking: string; + signature: string; +} +interface WireTextBlock { + type: "text"; + text: string; +} +interface WireToolUseBlock { + type: "tool_use"; + id: string; + name: string; + input: Record; +} +type WireBlock = WireThinkingBlock | WireTextBlock | WireToolUseBlock | { type: string; [key: string]: unknown }; + +function assistantWireBlocks(messages: Message[], model: Model<"anthropic-messages">): WireBlock[] { + const params = convertAnthropicMessages(messages, model, false); + const assistant = params.find(p => p.role === "assistant"); + return (assistant?.content as WireBlock[] | undefined) ?? []; +} + +describe("Anthropic-compatible unsigned thinking replay (#2005)", () => { + it("preserves unsigned thinking for non-official reasoning endpoints", () => { + const blocks = assistantWireBlocks( + [ + makeUser("solve x"), + makeAssistantThinking("plan: read the file, then edit", [{ type: "text", text: "Sure." }]), + ], + makeModel(), + ); + expect(blocks[0]).toEqual({ + type: "thinking", + thinking: "plan: read the file, then edit", + signature: "", + }); + expect(blocks[1]).toEqual({ type: "text", text: "Sure." }); + }); + + it("covers the Xiaomi MiMo Anthropic-compatible reporter configuration without provider allowlists", () => { + const model = makeModel({ + provider: "user-custom", + id: "mimo-v2.5-pro", + name: "MiMo V2.5 Pro (Singapore)", + baseUrl: "https://token-plan-sgp.xiaomimimo.com/anthropic", + maxTokens: 131_072, + contextWindow: 1_048_576, + }); + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("hidden reasoning")], model); + expect(blocks[0]).toEqual({ type: "thinking", thinking: "hidden reasoning", signature: "" }); + }); + + it("preserves legacy known non-signing endpoints even if model.reasoning is false", () => { + const model = makeModel({ provider: "custom", baseUrl: "https://api.deepseek.com/v1", reasoning: false }); + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("deepseek reasoning")], model); + expect(blocks[0]?.type).toBe("thinking"); + }); + + it("still degrades unsigned thinking to text for official Anthropic", () => { + const model = makeModel({ provider: "anthropic", baseUrl: "https://api.anthropic.com" }); + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model); + expect(blocks[0]?.type).toBe("text"); + expect((blocks[0] as WireTextBlock).text).toBe("internal scratch"); + }); + + it("treats a missing baseUrl as official Anthropic (resolveAnthropicBaseUrl default)", () => { + // `isAnthropicApiBaseUrl(undefined) === true` because the actual HTTP + // dispatch falls back to https://api.anthropic.com. Same-id custom + // overrides that only tweak model metadata (no baseUrl override) must + // not regress to native-thinking replay against the first-party API. + const model = { ...makeModel(), provider: "anthropic", baseUrl: "" }; + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model); + expect(blocks[0]?.type).toBe("text"); + expect((blocks[0] as WireTextBlock).text).toBe("internal scratch"); + }); + + it("still degrades unsigned thinking to text for non-reasoning unknown endpoints", () => { + const model = makeModel({ reasoning: false, baseUrl: "https://plain.example.com/anthropic" }); + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("scratch")], model); + expect(blocks[0]?.type).toBe("text"); + expect((blocks[0] as WireTextBlock).text).toBe("scratch"); + }); + + it("keeps thinking → tool_use pairing intact across continuation conversion", () => { + const toolResult: ToolResultMessage = { + role: "toolResult", + toolCallId: "toolu_reasoning_1", + toolName: "read", + content: [{ type: "text", text: "file body" }], + isError: false, + timestamp: 0, + }; + const model = makeModel(); + const messages: Message[] = [ + makeUser("read README"), + makeAssistantThinking("I need to call the read tool", [ + { type: "toolCall", id: "toolu_reasoning_1", name: "read", arguments: { path: "README.md" } }, + ]), + toolResult, + ]; + const params = convertAnthropicMessages(messages, model, false); + expect(params.map(p => p.role)).toEqual(["user", "assistant", "user"]); + const assistantBlocks = params[1].content as WireBlock[]; + expect(assistantBlocks[0]?.type).toBe("thinking"); + expect(assistantBlocks[1]?.type).toBe("tool_use"); + expect((assistantBlocks[1] as WireToolUseBlock).id).toBe("toolu_reasoning_1"); + }); +}); diff --git a/packages/ai/test/auth-gateway-openai-responses.test.ts b/packages/ai/test/auth-gateway-openai-responses.test.ts index fe3a21801..c6cbd8c9b 100644 --- a/packages/ai/test/auth-gateway-openai-responses.test.ts +++ b/packages/ai/test/auth-gateway-openai-responses.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { encodeResponse, encodeStream, parseRequest } from "../src/providers/openai-responses-server"; import type { AssistantMessage } from "../src/types"; import { AssistantMessageEventStream } from "../src/utils/event-stream"; diff --git a/packages/ai/test/auth-gateway-pi-native.test.ts b/packages/ai/test/auth-gateway-pi-native.test.ts index 7c10a65a7..5f3a77a76 100644 --- a/packages/ai/test/auth-gateway-pi-native.test.ts +++ b/packages/ai/test/auth-gateway-pi-native.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { encodeStream, formatError, parseRequest } from "../src/providers/pi-native-server"; import type { AssistantMessage, diff --git a/packages/ai/test/auth-retry.test.ts b/packages/ai/test/auth-retry.test.ts new file mode 100644 index 000000000..d992a33c1 --- /dev/null +++ b/packages/ai/test/auth-retry.test.ts @@ -0,0 +1,162 @@ +import { describe, expect, it } from "bun:test"; +import type { ApiKeyResolveContext } from "@oh-my-pi/pi-ai"; +import { isApiKeyResolver, isAuthRetryableError, resolveApiKeyOnce, withAuth } from "@oh-my-pi/pi-ai"; + +function authError(status = 401): Error & { status: number } { + return Object.assign(new Error(`${status} authentication_error`), { status }); +} + +function usageLimitError(): Error & { status: number } { + return Object.assign(new Error("You have hit your ChatGPT usage limit (pro plan). Try again in ~158 min."), { + status: 429, + }); +} + +describe("isApiKeyResolver / resolveApiKeyOnce", () => { + it("narrows resolver vs static key and resolves the initial value", async () => { + expect(isApiKeyResolver("static")).toBe(false); + expect(isApiKeyResolver(undefined)).toBe(false); + expect(isApiKeyResolver(() => "k")).toBe(true); + + expect(await resolveApiKeyOnce("static")).toBe("static"); + expect(await resolveApiKeyOnce(undefined)).toBeUndefined(); + + let seen: ApiKeyResolveContext | undefined; + const resolved = await resolveApiKeyOnce(ctx => { + seen = ctx; + return "minted"; + }); + expect(resolved).toBe("minted"); + // Initial resolve must look like an initial resolve, not a retry. + expect(seen).toEqual({ lastChance: false, error: undefined, signal: undefined }); + }); +}); + +describe("isAuthRetryableError", () => { + it("treats 401 and usage-limit phrasing as retryable, everything else as not", () => { + expect(isAuthRetryableError(authError(401))).toBe(true); + expect(isAuthRetryableError(usageLimitError())).toBe(true); + // A 429 whose body names the *account's* rate limit is rotatable (switch + // account), even though it isn't a 401 and isn't phrased "usage limit". + expect( + isAuthRetryableError( + Object.assign( + new Error( + '429 {"type":"error","error":{"type":"rate_limit_error","message":"This request would exceed your account\'s rate limit. Please try again later."}} retry-after-ms=9779000', + ), + { status: 429 }, + ), + ), + ).toBe(true); + // A generic (non-account) 429 rate limit is NOT rotatable — switching + // credentials won't help an org/global limit. + expect(isAuthRetryableError(Object.assign(new Error("429 too many requests"), { status: 429 }))).toBe(false); + expect(isAuthRetryableError("Error: 401 unauthorized")).toBe(true); + expect(isAuthRetryableError(authError(403))).toBe(false); + expect(isAuthRetryableError(authError(500))).toBe(false); + expect(isAuthRetryableError(new Error("network blip"))).toBe(false); + expect(isAuthRetryableError(undefined)).toBe(false); + }); +}); + +describe("withAuth", () => { + it("runs a single attempt for a static string key (no retry)", async () => { + const keys: Array = []; + const result = await withAuth("static-key", async key => { + keys.push(key); + return `ok:${key}`; + }); + expect(result).toBe("ok:static-key"); + expect(keys).toEqual(["static-key"]); + }); + + it("throws when a static key is missing", async () => { + await expect(withAuth(undefined, async () => "never", { missingKeyMessage: "no key for foo" })).rejects.toThrow( + "no key for foo", + ); + }); + + it("refreshes the same account, then switches, in order", async () => { + const keys: string[] = []; + const contexts: ApiKeyResolveContext[] = []; + const result = await withAuth( + ctx => { + contexts.push(ctx); + return ctx.error === undefined ? "k0" : ctx.lastChance ? "k2" : "k1"; + }, + async key => { + keys.push(key); + if (key === "k2") return "success"; + throw authError(); + }, + ); + expect(result).toBe("success"); + expect(keys).toEqual(["k0", "k1", "k2"]); + expect(contexts.map(ctx => ({ lastChance: ctx.lastChance, hasError: ctx.error !== undefined }))).toEqual([ + { lastChance: false, hasError: false }, + { lastChance: false, hasError: true }, + { lastChance: true, hasError: true }, + ]); + }); + + it("stops retrying when the resolver returns undefined", async () => { + const keys: string[] = []; + const original = authError(); + await expect( + withAuth( + ctx => (ctx.error === undefined ? "k0" : undefined), + async key => { + keys.push(key); + throw original; + }, + ), + ).rejects.toBe(original); + expect(keys).toEqual(["k0"]); + }); + + it("does not re-attempt when the re-resolved key is unchanged", async () => { + const keys: string[] = []; + const original = authError(); + // refresh-same returns the same key (skip), switch returns the same key (skip). + await expect( + withAuth( + () => "same", + async key => { + keys.push(key); + throw original; + }, + ), + ).rejects.toBe(original); + expect(keys).toEqual(["same"]); + }); + + it("propagates non-auth errors without retrying", async () => { + const keys: string[] = []; + const boom = new Error("network blip"); + await expect( + withAuth( + ctx => (ctx.error === undefined ? "k0" : "k1"), + async key => { + keys.push(key); + throw boom; + }, + ), + ).rejects.toBe(boom); + expect(keys).toEqual(["k0"]); + }); + + it("honors a custom isAuthError classifier", async () => { + const keys: string[] = []; + const result = await withAuth( + ctx => (ctx.error === undefined ? "k0" : "k1"), + async key => { + keys.push(key); + if (key === "k0") throw new Error("CUSTOM_RETRY"); + return "ok"; + }, + { isAuthError: error => error instanceof Error && error.message === "CUSTOM_RETRY" }, + ); + expect(result).toBe("ok"); + expect(keys).toEqual(["k0", "k1"]); + }); +}); diff --git a/packages/ai/test/auth-storage-force-refresh-rotate.test.ts b/packages/ai/test/auth-storage-force-refresh-rotate.test.ts new file mode 100644 index 000000000..fc7d0bb01 --- /dev/null +++ b/packages/ai/test/auth-storage-force-refresh-rotate.test.ts @@ -0,0 +1,163 @@ +import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { type AuthCredentialStore, AuthStorage, SqliteAuthCredentialStore } from "../src/auth-storage"; +import { registerOAuthProvider, unregisterOAuthProviders } from "../src/utils/oauth"; + +const PROVIDER = "unit-rotate-oauth"; +const SOURCE = "auth-storage-force-refresh-rotate-test"; + +function farExpiry(): number { + return Date.now() + 60 * 60_000; +} + +function authError(): Error & { status: number } { + return Object.assign(new Error("401 authentication_error"), { status: 401 }); +} + +function usageLimitError(): Error & { status: number } { + return Object.assign(new Error("You have hit your ChatGPT usage limit (pro plan). Try again in ~158 min."), { + status: 429, + }); +} + +describe("AuthStorage forceRefresh + rotateSessionCredential", () => { + let tempDir = ""; + let store: AuthCredentialStore | undefined; + let authStorage: AuthStorage | undefined; + + beforeEach(async () => { + tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-rotate-")); + store = await SqliteAuthCredentialStore.open(path.join(tempDir, "agent.db")); + authStorage = new AuthStorage(store); + }); + + afterEach(async () => { + vi.restoreAllMocks(); + unregisterOAuthProviders(SOURCE); + store?.close(); + store = undefined; + authStorage = undefined; + if (tempDir) { + await fs.rm(tempDir, { recursive: true, force: true }); + tempDir = ""; + } + }); + + function registerProvider(onRefresh?: () => void): void { + registerOAuthProvider({ + id: PROVIDER, + name: "Rotate Unit", + sourceId: SOURCE, + async login() { + return { access: "login", refresh: "login", expires: farExpiry() }; + }, + async refreshToken(credentials) { + onRefresh?.(); + return { + ...credentials, + access: "minted-access", + refresh: "minted-refresh", + expires: farExpiry(), + }; + }, + getApiKey(credentials) { + return credentials.access; + }, + }); + } + + test("forceRefresh re-mints a not-yet-expired token; a normal resolve uses the cached token", async () => { + if (!authStorage) throw new Error("test setup failed"); + let refreshCalls = 0; + registerProvider(() => { + refreshCalls += 1; + }); + await authStorage.set(PROVIDER, [ + { type: "oauth", access: "cached-access", refresh: "cached-refresh", expires: farExpiry() }, + ]); + + const cached = await authStorage.getApiKey(PROVIDER, "s-control"); + expect(cached).toBe("cached-access"); + expect(refreshCalls).toBe(0); + + const forced = await authStorage.getApiKey(PROVIDER, "s-force", { forceRefresh: true }); + expect(forced).toBe("minted-access"); + expect(refreshCalls).toBe(1); + + // The re-minted credential is persisted, so the next plain resolve sees it. + const after = await authStorage.getApiKey(PROVIDER, "s-after"); + expect(after).toBe("minted-access"); + }); + + test("rotateSessionCredential(401) blocks + clears the sticky and rotates to a sibling", async () => { + if (!authStorage) throw new Error("test setup failed"); + registerProvider(); + await authStorage.set(PROVIDER, [ + { type: "oauth", access: "acc-A", refresh: "ref-A", expires: farExpiry() }, + { type: "oauth", access: "acc-B", refresh: "ref-B", expires: farExpiry() }, + ]); + + const first = await authStorage.getApiKey(PROVIDER, "sess"); + expect(["acc-A", "acc-B"]).toContain(first ?? ""); + + const usageLimitSpy = vi.spyOn(authStorage, "markUsageLimitReached"); + const rotated = await authStorage.rotateSessionCredential(PROVIDER, "sess", { error: authError() }); + + expect(rotated).toBe(true); + // A hard 401 must NOT take the usage-limit code path. + expect(usageLimitSpy).not.toHaveBeenCalled(); + + const second = await authStorage.getApiKey(PROVIDER, "sess"); + expect(["acc-A", "acc-B"]).toContain(second ?? ""); + expect(second).not.toBe(first); + }); + + test("rotateSessionCredential(usage-limit) delegates to markUsageLimitReached", async () => { + if (!authStorage) throw new Error("test setup failed"); + registerProvider(); + await authStorage.set(PROVIDER, [ + { type: "oauth", access: "acc-A", refresh: "ref-A", expires: farExpiry() }, + { type: "oauth", access: "acc-B", refresh: "ref-B", expires: farExpiry() }, + ]); + + const first = await authStorage.getApiKey(PROVIDER, "sess"); + const usageLimitSpy = vi.spyOn(authStorage, "markUsageLimitReached"); + + const rotated = await authStorage.rotateSessionCredential(PROVIDER, "sess", { + error: usageLimitError(), + }); + + expect(rotated).toBe(true); + // Usage / account-rate-limit errors route to markUsageLimitReached, which + // owns the block duration (default + server usage-report reset) — the + // resolver never parses retry-after itself. + expect(usageLimitSpy).toHaveBeenCalledTimes(1); + expect(usageLimitSpy.mock.calls[0]?.[0]).toBe(PROVIDER); + expect(usageLimitSpy.mock.calls[0]?.[1]).toBe("sess"); + + const second = await authStorage.getApiKey(PROVIDER, "sess"); + expect(second).not.toBe(first); + }); + + test("rotateSessionCredential reports no sibling for a single-credential setup", async () => { + if (!authStorage) throw new Error("test setup failed"); + registerProvider(); + await authStorage.set(PROVIDER, [ + { type: "oauth", access: "only-access", refresh: "only-refresh", expires: farExpiry() }, + ]); + + await authStorage.getApiKey(PROVIDER, "sess"); + expect(await authStorage.rotateSessionCredential(PROVIDER, "sess", { error: authError() })).toBe(false); + }); + + test("rotateSessionCredential returns false when the session has no sticky credential", async () => { + if (!authStorage) throw new Error("test setup failed"); + registerProvider(); + await authStorage.set(PROVIDER, [{ type: "oauth", access: "acc-A", refresh: "ref-A", expires: farExpiry() }]); + + // Never resolved a key for this session → nothing to rotate away from. + expect(await authStorage.rotateSessionCredential(PROVIDER, "untouched", { error: authError() })).toBe(false); + }); +}); diff --git a/packages/ai/test/github-copilot-model-limits.test.ts b/packages/ai/test/github-copilot-model-limits.test.ts index c46f14487..07d76ee87 100644 --- a/packages/ai/test/github-copilot-model-limits.test.ts +++ b/packages/ai/test/github-copilot-model-limits.test.ts @@ -2,8 +2,8 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; +import { Effort } from "../src/effort"; import { createModelManager } from "../src/model-manager"; -import { Effort } from "../src/model-thinking"; import { getBundledModel } from "../src/models"; import { githubCopilotModelManagerOptions } from "../src/provider-models/openai-compat"; diff --git a/packages/ai/test/github-copilot-reasoning.test.ts b/packages/ai/test/github-copilot-reasoning.test.ts index 32a88c006..28746c85d 100644 --- a/packages/ai/test/github-copilot-reasoning.test.ts +++ b/packages/ai/test/github-copilot-reasoning.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { getBundledModel } from "../src/models"; import { streamAnthropic } from "../src/providers/anthropic"; import { streamOpenAIResponses } from "../src/providers/openai-responses"; diff --git a/packages/ai/test/google-antigravity-usage.test.ts b/packages/ai/test/google-antigravity-usage.test.ts new file mode 100644 index 000000000..4fdec1ae0 --- /dev/null +++ b/packages/ai/test/google-antigravity-usage.test.ts @@ -0,0 +1,196 @@ +/** + * Antigravity usage provider contract tests. The merge logic + * deduplicates per-model quota entries by (tier, windowId), + * preserves reset times when bar data and window data come from + * different model entries, and handles mixed-case tier names. + */ +import { describe, expect, it } from "bun:test"; +import type { UsageFetchContext, UsageFetchParams } from "../src/usage"; +import { antigravityUsageProvider } from "../src/usage/google-antigravity"; + +const accessTokenFixture = (() => { + const header = Buffer.from(JSON.stringify({ alg: "none", typ: "JWT" })).toString("base64url"); + const body = Buffer.from(JSON.stringify({ sub: "user-fixture" })).toString("base64url"); + return `${header}.${body}.sig`; +})(); + +function makeCredential(overrides?: Partial) { + return { + type: "oauth" as const, + accessToken: accessTokenFixture, + refreshToken: "refresh-fixture", + expiresAt: Date.now() + 3600_000, + projectId: "test-project", + email: "test@example.com", + accountId: "acct-1", + ...overrides, + } satisfies UsageFetchParams["credential"]; +} + +function fakeFetch(json: unknown): typeof fetch { + const fn = async () => + new Response(JSON.stringify(json), { + status: 200, + headers: { "content-type": "application/json" }, + }); + return fn as unknown as typeof fetch; +} + +function makeCtx(fetchImpl?: typeof fetch): UsageFetchContext { + return { fetch: fetchImpl ?? fakeFetch({}) }; +} + +// ── helpers ────────────────────────────────────────────────────────── + +function makeApiModel( + displayName: string, + quota: { remainingFraction?: number; resetTime?: string; tier?: string; windowId?: string }, +) { + return { + displayName, + quotaInfo: { + remainingFraction: quota.remainingFraction, + resetTime: quota.resetTime, + tier: quota.tier, + windowId: quota.windowId, + }, + }; +} + +// ── tests ──────────────────────────────────────────────────────────── + +describe("antigravity usage provider", () => { + it("merges two models with same tier into one limit", async () => { + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "premium" }), + modelB: makeApiModel("Model B", { remainingFraction: 0.5, tier: "premium" }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report).not.toBeNull(); + expect(report!.limits.length).toBe(1); + }); + + it("keeps the worst remainingFraction when merging same tier", async () => { + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.1, tier: "premium" }), + modelB: makeApiModel("Model B", { remainingFraction: 0.8, tier: "premium" }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.limits.length).toBe(1); + expect(report!.limits[0]!.amount.remainingFraction).toBe(0.1); + }); + + it("merges mixed-case tier names under lowercased key", async () => { + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "Default" }), + modelB: makeApiModel("Model B", { remainingFraction: 0.6, tier: "default" }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.limits.length).toBe(1); + }); + + it("preserves reset time from an entry even when bar data comes from another", async () => { + const now = Date.now(); + const resetTime = new Date(now + 4 * 3600_000).toISOString(); + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "default" }), + modelB: makeApiModel("Model B", { remainingFraction: undefined, tier: "default", resetTime }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.limits.length).toBe(1); + expect(report!.limits[0]!.amount.remainingFraction).toBe(0.3); + expect(report!.limits[0]!.window).toBeDefined(); + expect(report!.limits[0]!.window!.resetsAt).toBeGreaterThan(now); + }); + + it("separates models with different windowIds in the same tier", async () => { + const now = Date.now(); + const t1 = new Date(now + 5 * 3600_000).toISOString(); + const t2 = new Date(now + 24 * 3600_000).toISOString(); + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "premium", windowId: "5h", resetTime: t1 }), + modelB: makeApiModel("Model B", { + remainingFraction: 0.7, + tier: "premium", + windowId: "daily", + resetTime: t2, + }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.limits.length).toBe(2); + }); + + it("includes email and projectId in report metadata", async () => { + const payload = { models: { m: makeApiModel("M", { remainingFraction: 1 }) } }; + const report = await antigravityUsageProvider.fetchUsage!( + { + provider: "google-antigravity", + credential: makeCredential({ email: "user@example.com", projectId: "proj-1" }), + signal: undefined, + }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.metadata?.email).toBe("user@example.com"); + expect(report!.metadata?.projectId).toBe("proj-1"); + }); + + it("does not include email when credential has none", async () => { + const payload = { models: { m: makeApiModel("M", { remainingFraction: 1 }) } }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential({ email: undefined }), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.metadata?.email).toBeUndefined(); + }); + + it("sorts limits by remainingFraction ascending (worst first)", async () => { + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.9, tier: "high" }), + modelB: makeApiModel("Model B", { remainingFraction: 0.2, tier: "low" }), + modelC: makeApiModel("Model C", { remainingFraction: 0.5, tier: "mid" }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.limits.length).toBe(3); + expect(report!.limits[0]!.amount.remainingFraction).toBe(0.2); + expect(report!.limits[1]!.amount.remainingFraction).toBe(0.5); + expect(report!.limits[2]!.amount.remainingFraction).toBe(0.9); + }); + + it("returns null when credential has no projectId", async () => { + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential({ projectId: undefined }), signal: undefined }, + makeCtx(), + ); + expect(report).toBeNull(); + }); +}); diff --git a/packages/ai/test/issue-1207-repro.test.ts b/packages/ai/test/issue-1207-repro.test.ts index 85e31868d..d71c91b3d 100644 --- a/packages/ai/test/issue-1207-repro.test.ts +++ b/packages/ai/test/issue-1207-repro.test.ts @@ -91,6 +91,19 @@ describe("issue #1207 — DeepSeek V4 keeps reasoning with tools", () => { expect(body.max_completion_tokens).toBeUndefined(); }); + it("does not mix Fireworks DeepSeek effort with the native thinking toggle", async () => { + const model = getBundledModel("fireworks", "deepseek-v4-pro") as Model<"openai-completions">; + const compat = resolveOpenAICompat(model); + const body = await capturePayload(model); + + expect(compat.extraBody).toBeUndefined(); + expect(body.tools).toBeDefined(); + expect(body.tool_choice).toBeUndefined(); + expect(body.reasoning_effort).toBe("high"); + expect(body.thinking).toBeUndefined(); + expect(body.max_tokens).toBe(123); + }); + it("preserves OpenRouter reasoning when tool_choice auto is present", async () => { const model = getBundledModel("openrouter", "deepseek/deepseek-v4-flash") as Model<"openai-completions">; const compat = detectOpenAICompat(model); diff --git a/packages/ai/test/issue-1373-repro.test.ts b/packages/ai/test/issue-1373-repro.test.ts index 986acbf3c..88e855f4e 100644 --- a/packages/ai/test/issue-1373-repro.test.ts +++ b/packages/ai/test/issue-1373-repro.test.ts @@ -1,5 +1,5 @@ import { afterAll, beforeAll, describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { streamBedrock } from "../src/providers/amazon-bedrock"; import type { Context, Model } from "../src/types"; diff --git a/packages/ai/test/issue-826-repro.test.ts b/packages/ai/test/issue-826-repro.test.ts index efcf9ccea..dd78aeda0 100644 --- a/packages/ai/test/issue-826-repro.test.ts +++ b/packages/ai/test/issue-826-repro.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { streamAnthropic } from "../src/providers/anthropic"; import type { Context, Model, Tool } from "../src/types"; diff --git a/packages/ai/test/issue-969-repro.test.ts b/packages/ai/test/issue-969-repro.test.ts index 1b347470e..9f42a85bb 100644 --- a/packages/ai/test/issue-969-repro.test.ts +++ b/packages/ai/test/issue-969-repro.test.ts @@ -1,5 +1,6 @@ import { afterEach, describe, expect, it } from "bun:test"; -import { Effort, getSupportedEfforts } from "../src/model-thinking"; +import { Effort } from "../src/effort"; +import { getSupportedEfforts } from "../src/model-thinking"; import { streamOpenAICompletions } from "../src/providers/openai-completions"; import type { Context, Model } from "../src/types"; diff --git a/packages/ai/test/model-thinking.test.ts b/packages/ai/test/model-thinking.test.ts index 89b90da8e..e8475974c 100644 --- a/packages/ai/test/model-thinking.test.ts +++ b/packages/ai/test/model-thinking.test.ts @@ -1,8 +1,8 @@ import { describe, expect, it } from "bun:test"; +import { Effort } from "@oh-my-pi/pi-ai/effort"; import { applyGeneratedModelPolicies, clampThinkingLevelForModel, - Effort, enrichModelThinking, linkOpenAIPromotionTargets, mapEffortToAnthropicAdaptiveEffort, diff --git a/packages/ai/test/nanogpt-model-limits.test.ts b/packages/ai/test/nanogpt-model-limits.test.ts index 03d7b8bab..0270b18ff 100644 --- a/packages/ai/test/nanogpt-model-limits.test.ts +++ b/packages/ai/test/nanogpt-model-limits.test.ts @@ -1,5 +1,5 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { nanoGptModelManagerOptions } from "../src/provider-models/openai-compat"; const originalFetch = global.fetch; diff --git a/packages/ai/test/ollama-provider.test.ts b/packages/ai/test/ollama-provider.test.ts index 164d17812..4fa4866c2 100644 --- a/packages/ai/test/ollama-provider.test.ts +++ b/packages/ai/test/ollama-provider.test.ts @@ -1,5 +1,5 @@ import { afterEach, describe, expect, test, vi } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { ollamaModelManagerOptions } from "../src/provider-models/openai-compat"; import { streamOllama } from "../src/providers/ollama"; import type { Context, Model, Tool } from "../src/types"; @@ -53,6 +53,48 @@ describe("ollama local provider discovery", () => { expect(model?.thinking).toEqual({ mode: "effort", minLevel: Effort.Minimal, maxLevel: Effort.High }); expect(model?.input).toEqual(["text", "image"]); }); + + test("remaps Ollama's unsupported reasoning levels and skips non-reasoning models", async () => { + global.fetch = vi.fn(async (input, init) => { + const url = String(input); + if (url === "http://127.0.0.1:11434/v1/models") { + return new Response( + JSON.stringify({ + object: "list", + data: [ + { id: "gemma4:e4b", object: "model" }, + { id: "llama-plain:latest", object: "model" }, + ], + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + } + if (url === "http://127.0.0.1:11434/api/show") { + const body = JSON.parse(String(init?.body ?? "{}")) as { model?: string }; + const thinking = body.model === "gemma4:e4b"; + return new Response( + JSON.stringify({ + capabilities: thinking ? ["completion", "tools", "thinking"] : ["completion", "tools"], + model_info: {}, + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + } + throw new Error(`Unexpected URL: ${url}`); + }) as unknown as typeof fetch; + + const models = await ollamaModelManagerOptions().fetchDynamicModels?.(); + const reasoningModel = models?.find(candidate => candidate.id === "gemma4:e4b"); + const plainModel = models?.find(candidate => candidate.id === "llama-plain:latest"); + + // Ollama's OpenAI-compatible endpoint rejects "minimal"/"xhigh" with HTTP 400; + // reasoning models must remap them onto accepted levels (low/max). + expect(reasoningModel?.reasoning).toBe(true); + expect(reasoningModel?.compat?.reasoningEffortMap).toMatchObject({ minimal: "low", xhigh: "max" }); + // Non-reasoning models never send an effort, so they carry no remap. + expect(plainModel?.reasoning).toBe(false); + expect(plainModel?.compat?.reasoningEffortMap).toBeUndefined(); + }); }); describe("ollama tool forcing", () => { diff --git a/packages/ai/test/openai-codex.test.ts b/packages/ai/test/openai-codex.test.ts index 4dbc616a7..ed81d7c78 100644 --- a/packages/ai/test/openai-codex.test.ts +++ b/packages/ai/test/openai-codex.test.ts @@ -71,6 +71,62 @@ describe("openai-codex request transformer", () => { }); }); +describe("openai-codex orphan tool-call repair", () => { + it("synthesizes a function_call_output for a function_call with no result", async () => { + const body: RequestBody = { + model: "gpt-5.1-codex", + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }, + { type: "function_call", call_id: "call_orphan", name: "read", arguments: "{}" }, + { type: "message", role: "user", content: [{ type: "input_text", text: "next" }] }, + ], + }; + + const transformed = await transformRequestBody(body, createCodexModel(body.model), {}); + const input = transformed.input || []; + + const callIndex = input.findIndex(item => item.type === "function_call" && item.call_id === "call_orphan"); + expect(callIndex).toBeGreaterThanOrEqual(0); + // The synthesized output sits immediately after the orphan call. + const output = input[callIndex + 1]; + expect(output?.type).toBe("function_call_output"); + expect(output?.call_id).toBe("call_orphan"); + expect(typeof output?.output).toBe("string"); + expect(output?.output as string).toMatch(/interrupted/i); + }); + + it("leaves a paired function_call untouched", async () => { + const body: RequestBody = { + model: "gpt-5.1-codex", + input: [ + { type: "function_call", call_id: "call_paired", name: "read", arguments: "{}" }, + { type: "function_call_output", call_id: "call_paired", output: "real result" }, + ], + }; + + const transformed = await transformRequestBody(body, createCodexModel(body.model), {}); + const input = transformed.input || []; + + const outputs = input.filter(item => item.type === "function_call_output" && item.call_id === "call_paired"); + expect(outputs).toHaveLength(1); + expect(outputs[0]?.output).toBe("real result"); + }); + + it("synthesizes a custom_tool_call_output for an orphan custom_tool_call", async () => { + const body: RequestBody = { + model: "gpt-5.1-codex", + input: [{ type: "custom_tool_call", call_id: "call_custom", name: "apply_patch" }], + }; + + const transformed = await transformRequestBody(body, createCodexModel(body.model), {}); + const input = transformed.input || []; + + const output = input.find(item => item.type === "custom_tool_call_output" && item.call_id === "call_custom"); + expect(output).toBeDefined(); + expect(output?.output as string).toMatch(/interrupted/i); + }); +}); + describe("openai-codex reasoning effort validation", () => { it("rejects gpt-5.1 xhigh when metadata does not list it", async () => { const body: RequestBody = { model: "gpt-5.1", input: [] }; diff --git a/packages/ai/test/openai-completions-disable-reasoning.test.ts b/packages/ai/test/openai-completions-disable-reasoning.test.ts index f2d9107f6..629a74887 100644 --- a/packages/ai/test/openai-completions-disable-reasoning.test.ts +++ b/packages/ai/test/openai-completions-disable-reasoning.test.ts @@ -1,5 +1,5 @@ import { afterEach, describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { streamOpenAICompletions } from "../src/providers/openai-completions"; import type { Context, Model } from "../src/types"; diff --git a/packages/ai/test/openai-responses-orphan-repair.test.ts b/packages/ai/test/openai-responses-orphan-repair.test.ts new file mode 100644 index 000000000..014d6988f --- /dev/null +++ b/packages/ai/test/openai-responses-orphan-repair.test.ts @@ -0,0 +1,69 @@ +import { describe, expect, it } from "bun:test"; +import { + repairOrphanResponsesToolCalls, + repairOrphanResponsesToolOutputs, +} from "@oh-my-pi/pi-ai/providers/openai-responses-shared"; +import type { ResponseInput } from "openai/resources/responses/responses"; + +describe("repairOrphanResponsesToolCalls", () => { + it("appends a synthetic function_call_output after a call with no result", () => { + const input: ResponseInput = [ + { type: "function_call", call_id: "call_a", name: "read", arguments: "{}" }, + { role: "user", content: [{ type: "input_text", text: "continue" }] }, + ]; + + const repaired = repairOrphanResponsesToolCalls(input); + const callIndex = repaired.findIndex( + item => + (item as { type?: string }).type === "function_call" && (item as { call_id?: string }).call_id === "call_a", + ); + const output = repaired[callIndex + 1] as { type?: string; call_id?: string; output?: unknown }; + expect(output.type).toBe("function_call_output"); + expect(output.call_id).toBe("call_a"); + expect(output.output).toMatch(/interrupted/i); + }); + + it("uses custom_tool_call_output for an orphan custom_tool_call", () => { + const input: ResponseInput = [ + { type: "custom_tool_call", call_id: "call_c", name: "apply_patch", input: "patch" } as ResponseInput[number], + ]; + + const repaired = repairOrphanResponsesToolCalls(input); + const output = repaired.find(item => (item as { type?: string }).type === "custom_tool_call_output") as + | { call_id?: string } + | undefined; + expect(output?.call_id).toBe("call_c"); + }); + + it("returns the input unchanged when every call is paired", () => { + const input: ResponseInput = [ + { type: "function_call", call_id: "call_a", name: "read", arguments: "{}" }, + { type: "function_call_output", call_id: "call_a", output: "ok" } as ResponseInput[number], + ]; + + const repaired = repairOrphanResponsesToolCalls(input); + expect(repaired).toBe(input); + }); + + it("composes with output repair so a tree-branch snapshot stays API-valid", () => { + // Branching to a node that ends on a tool call drops the result child: + // the assistant turn keeps the call, but no matching output remains. + const input: ResponseInput = [ + { role: "user", content: [{ type: "input_text", text: "do it" }] }, + { type: "function_call", call_id: "call_x", name: "bash", arguments: "{}" }, + ]; + + const repaired = repairOrphanResponsesToolCalls(repairOrphanResponsesToolOutputs(input)); + const callIds = new Set( + repaired + .filter(i => (i as { type?: string }).type === "function_call") + .map(i => (i as { call_id: string }).call_id), + ); + const outputIds = new Set( + repaired + .filter(i => (i as { type?: string }).type === "function_call_output") + .map(i => (i as { call_id: string }).call_id), + ); + for (const id of callIds) expect(outputIds.has(id)).toBe(true); + }); +}); diff --git a/packages/ai/test/openai-responses-parallel-tool-calls.test.ts b/packages/ai/test/openai-responses-parallel-tool-calls.test.ts index 56e076e6d..dbff9a657 100644 --- a/packages/ai/test/openai-responses-parallel-tool-calls.test.ts +++ b/packages/ai/test/openai-responses-parallel-tool-calls.test.ts @@ -202,4 +202,137 @@ describe("processResponsesStream: parallel function_call items", () => { expect(byCallId.get("call_a")?.toolCall.arguments).toEqual({ path: "test.txt" }); expect(byCallId.get("call_b")?.toolCall.arguments).toEqual({ path: "test.md" }); }); + + test("routes identifierless final argument events in item order", async () => { + const output = makeOutput(); + const emitted: EmittedEvent[] = []; + const stream = { push: (e: unknown) => emitted.push(e as EmittedEvent), end: () => {} } as never; + + const argsA = JSON.stringify({ command: "printf a" }); + const argsB = JSON.stringify({ command: "printf b" }); + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "function_call", id: "fc_a", call_id: "call_a", name: "bash", arguments: "" }, + }, + { + type: "response.output_item.added", + output_index: 1, + item: { type: "function_call", id: "fc_b", call_id: "call_b", name: "bash", arguments: "" }, + }, + { + type: "response.function_call_arguments.done", + arguments: argsA, + }, + { + type: "response.function_call_arguments.done", + arguments: argsB, + }, + { + type: "response.output_item.done", + output_index: 0, + item: { type: "function_call", id: "fc_a", call_id: "call_a", name: "bash", arguments: "" }, + }, + { + type: "response.output_item.done", + output_index: 1, + item: { type: "function_call", id: "fc_b", call_id: "call_b", name: "bash", arguments: "" }, + }, + ]), + output, + stream, + makeModel(), + ); + + const [blockA, blockB] = output.content; + if (blockA?.type !== "toolCall" || blockB?.type !== "toolCall") throw new Error("expected toolCalls"); + expect(blockA.arguments).toEqual({ command: "printf a" }); + expect(blockB.arguments).toEqual({ command: "printf b" }); + + const ends = emitted.filter(e => e.type === "toolcall_end") as Array<{ + toolCall: { id: string; arguments: Record }; + }>; + expect(ends).toHaveLength(2); + const byCallId = new Map(ends.map(e => [e.toolCall.id.split("|")[0], e])); + expect(byCallId.get("call_a")?.toolCall.arguments).toEqual({ command: "printf a" }); + expect(byCallId.get("call_b")?.toolCall.arguments).toEqual({ command: "printf b" }); + }); + + test("routes deltas by item.call_id when llama.cpp omits item.id and output_index (issue #2015)", async () => { + // llama.cpp's `to_json_oaicompat_resp` (tools/server/server-task.cpp) emits a + // function_call's `output_item.added` with only `item.call_id` — no `item.id`, + // no `output_index`. The matching `function_call_arguments.delta` then carries + // `item_id: "fc_"` and again no `output_index`. Without secondary + // indexing on `call_id`, `processResponsesStream`'s lookup map stays empty and + // every delta lands on the trailing block, leaving earlier calls with empty + // arguments (= `{}`) — the read tool then rejects them with + // `path: Invalid input: expected string, received undefined`. + const output = makeOutput(); + const emitted: EmittedEvent[] = []; + const stream = { push: (e: unknown) => emitted.push(e as EmittedEvent), end: () => {} } as never; + + const argsA = JSON.stringify({ path: "a.txt" }); + const argsB = JSON.stringify({ path: "b.txt" }); + const argsC = JSON.stringify({ path: "c.txt" }); + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + item: { type: "function_call", call_id: "fc_a", name: "read", arguments: "" }, + }, + { + type: "response.output_item.added", + item: { type: "function_call", call_id: "fc_b", name: "read", arguments: "" }, + }, + { + type: "response.output_item.added", + item: { type: "function_call", call_id: "fc_c", name: "read", arguments: "" }, + }, + { type: "response.function_call_arguments.delta", item_id: "fc_a", delta: argsA }, + { type: "response.function_call_arguments.delta", item_id: "fc_b", delta: argsB }, + { type: "response.function_call_arguments.delta", item_id: "fc_c", delta: argsC }, + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "fc_a", name: "read", arguments: argsA }, + }, + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "fc_b", name: "read", arguments: argsB }, + }, + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "fc_c", name: "read", arguments: argsC }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.content).toHaveLength(3); + const [a, b, c] = output.content; + if (a?.type !== "toolCall" || b?.type !== "toolCall" || c?.type !== "toolCall") { + throw new Error("expected toolCalls"); + } + expect(a.arguments).toEqual({ path: "a.txt" }); + expect(b.arguments).toEqual({ path: "b.txt" }); + expect(c.arguments).toEqual({ path: "c.txt" }); + + const ends = emitted.filter(e => e.type === "toolcall_end") as Array<{ + toolCall: { id: string; arguments: Record }; + contentIndex: number; + }>; + expect(ends).toHaveLength(3); + const byCallId = new Map(ends.map(e => [e.toolCall.id.split("|")[0], e])); + expect(byCallId.get("fc_a")?.toolCall.arguments).toEqual({ path: "a.txt" }); + expect(byCallId.get("fc_b")?.toolCall.arguments).toEqual({ path: "b.txt" }); + expect(byCallId.get("fc_c")?.toolCall.arguments).toEqual({ path: "c.txt" }); + expect(byCallId.get("fc_a")?.contentIndex).toBe(0); + expect(byCallId.get("fc_b")?.contentIndex).toBe(1); + expect(byCallId.get("fc_c")?.contentIndex).toBe(2); + }); }); diff --git a/packages/ai/test/schema-normalization.test.ts b/packages/ai/test/schema-normalization.test.ts index 1f6ba7688..22efcb2cd 100644 --- a/packages/ai/test/schema-normalization.test.ts +++ b/packages/ai/test/schema-normalization.test.ts @@ -728,6 +728,18 @@ describe("stripResidualCombiners", () => { expect(normalized.anyOf).toBeUndefined(); expect(normalized.oneOf).toBeUndefined(); }); + + it("drops array-only keys when mixed-type collapse picks string from anyOf fixpoint", () => { + const stripped = stripResidualCombiners({ + anyOf: [{ type: "string" }, { type: "array", items: { type: "string" } }], + description: "pr number, url, or branch", + }) as Record; + + expect(stripped.type).toBe("string"); + expect(stripped.items).toBeUndefined(); + expect(stripped.anyOf).toBeUndefined(); + expect(stripped.description).toBe("pr number, url, or branch"); + }); }); // --------------------------------------------------------------------------- @@ -952,6 +964,36 @@ describe("normalizeSchemaForCCA", () => { properties: {}, }); }); + + it("strips array-only keys when mixed-type collapse picks a non-array type", () => { + // Regression: anyOf [{type:"string"}, {type:"array", items:{type:"string"}}] + // collapsed to {type:"string", items:{type:"string"}} which is invalid. + // The fix filters mergedVariantFields against the chosen type's allowed keys. + const normalized = normalizeSchemaForCCA({ + anyOf: [{ type: "string" }, { type: "array", items: { type: "string" } }], + description: "pr number, url, or branch", + }); + + expect(normalized).toEqual({ + type: "string", + description: "pr number, url, or branch", + }); + }); + + it("strips sibling type-specific keys copied from parent when mixed-type collapse picks opposing type", () => { + // Edge case: parent has a sibling `items` outside the anyOf, + // and the chosen type is string. The sibling must be stripped. + const normalized = normalizeSchemaForCCA({ + anyOf: [{ type: "string" }, { type: "array", items: { type: "number" } }], + items: { type: "string" }, + description: "pr number, url, or branch", + }); + + expect(normalized).toEqual({ + type: "string", + description: "pr number, url, or branch", + }); + }); }); // --------------------------------------------------------------------------- diff --git a/packages/ai/test/stream-auth-retry.test.ts b/packages/ai/test/stream-auth-retry.test.ts index d019e57ce..7192cb8c8 100644 --- a/packages/ai/test/stream-auth-retry.test.ts +++ b/packages/ai/test/stream-auth-retry.test.ts @@ -1,4 +1,5 @@ import { afterEach, describe, expect, it } from "bun:test"; +import type { ApiKeyResolveContext } from "@oh-my-pi/pi-ai"; import { registerCustomApi, unregisterCustomApis } from "@oh-my-pi/pi-ai"; import { streamSimple } from "@oh-my-pi/pi-ai/stream"; import type { Api, AssistantMessage, Context, Model, SimpleStreamOptions, Usage } from "@oh-my-pi/pi-ai/types"; @@ -25,9 +26,9 @@ function assistant(content: string[] = []): AssistantMessage { api: API, provider: "test-provider", model: "test-model", - usage: usage(), + timestamp: 1, stopReason: "stop", - timestamp: Date.now(), + usage: usage(), }; } @@ -39,19 +40,21 @@ function authError(): Error & { status: number } { return Object.assign(new Error("401 authentication_error"), { status: 401 }); } +function usageLimitError(): Error & { status: number } { + return Object.assign(new Error("You have hit your ChatGPT usage limit (pro plan). Try again in ~158 min."), { + status: 429, + }); +} + function model(): Model { return { id: "test-model", - name: "test-model", + name: "Test Model", api: API, provider: "test-provider", - baseUrl: "mock://", - reasoning: false, - input: ["text"], - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, - contextWindow: 1024, - maxTokens: 1024, - }; + contextWindow: 1000, + maxTokens: 100, + } as Model; } const context: Context = { @@ -59,61 +62,64 @@ const context: Context = { messages: [{ role: "user", content: "hello", timestamp: 1 }], }; -describe("streamSimple auth retry", () => { +/** Records the static key each inner attempt actually received. */ +function pushKey(keys: unknown[], options?: SimpleStreamOptions): void { + keys.push(options?.apiKey); +} + +function ok(stream: AssistantMessageEventStream): void { + const message = assistant(["ok"]); + stream.push({ type: "start", partial: message }); + stream.push({ type: "done", reason: "stop", message }); +} + +describe("streamSimple resolver auth retry", () => { afterEach(() => { unregisterCustomApis(SOURCE_ID); }); - it("retries once with a fresh key when 401 happens before the first event", async () => { - const keys: Array = []; - let authCalls = 0; + it("retries with a refreshed key when a 401 is thrown before the first event", async () => { + const keys: unknown[] = []; + const contexts: ApiKeyResolveContext[] = []; registerCustomApi( API, (_model: Model, _context: Context, options?: SimpleStreamOptions) => { - keys.push(options?.apiKey); + pushKey(keys, options); const stream = new AssistantMessageEventStream(); - queueMicrotask(() => { - if (keys.length === 1) { - stream.fail(authError()); - return; - } - const message = assistant(["ok"]); - stream.push({ type: "start", partial: message }); - stream.push({ type: "done", reason: "stop", message }); - }); + queueMicrotask(() => (keys.length === 1 ? stream.fail(authError()) : ok(stream))); return stream; }, SOURCE_ID, ); const stream = streamSimple(model(), context, { - apiKey: "old-key", - onAuthError: async (provider, oldKey, error) => { - authCalls += 1; - expect(provider).toBe("test-provider"); - expect(oldKey).toBe("old-key"); - expect((error as { status?: number }).status).toBe(401); - return "new-key"; + apiKey: async ctx => { + contexts.push(ctx); + return ctx.error === undefined ? "old-key" : ctx.lastChance ? "switch-key" : "refresh-key"; }, }); - for await (const _event of stream) { // drain } expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); - expect(keys).toEqual(["old-key", "new-key"]); - expect(authCalls).toBe(1); + // Initial resolve, then step (b) refresh-same — the switch step is never reached. + expect(keys).toEqual(["old-key", "refresh-key"]); + expect(keys.every(key => typeof key === "string")).toBe(true); + expect(contexts.map(ctx => ({ lastChance: ctx.lastChance, hasError: ctx.error !== undefined }))).toEqual([ + { lastChance: false, hasError: false }, + { lastChance: false, hasError: true }, + ]); + expect((contexts[1]?.error as { status?: number }).status).toBe(401); }); - it("retries when a provider emits start then a 401 error event before content", async () => { - const keys: Array = []; + it("buffers the start event and retries on a 401 error event before content", async () => { + const keys: unknown[] = []; const eventTypes: string[] = []; - let authCalls = 0; registerCustomApi( API, (_model: Model, _context: Context, options?: SimpleStreamOptions) => { - keys.push(options?.apiKey); + pushKey(keys, options); const stream = new AssistantMessageEventStream(); queueMicrotask(() => { if (keys.length === 1) { @@ -127,9 +133,7 @@ describe("streamSimple auth retry", () => { }); return; } - const message = assistant(["ok"]); - stream.push({ type: "start", partial: message }); - stream.push({ type: "done", reason: "stop", message }); + ok(stream); }); return stream; }, @@ -137,28 +141,59 @@ describe("streamSimple auth retry", () => { ); const stream = streamSimple(model(), context, { - apiKey: "old-key", - onAuthError: async (provider, oldKey, error) => { - authCalls += 1; - expect(provider).toBe("test-provider"); - expect(oldKey).toBe("old-key"); - expect((error as { status?: number }).status).toBe(401); - return "new-key"; - }, + apiKey: async ctx => (ctx.error === undefined ? "old-key" : "new-key"), }); - for await (const event of stream) { eventTypes.push(event.type); } expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); expect(keys).toEqual(["old-key", "new-key"]); + // The buffered `start` of the failed attempt must not leak — the user + // sees exactly one clean start/done pair. expect(eventTypes).toEqual(["start", "done"]); - expect(authCalls).toBe(1); + }); + + it("retries on a 401 carried only via errorStatus", async () => { + const keys: unknown[] = []; + registerCustomApi( + API, + (_model: Model, _context: Context, options?: SimpleStreamOptions) => { + pushKey(keys, options); + const stream = new AssistantMessageEventStream(); + queueMicrotask(() => { + if (keys.length === 1) { + stream.push({ type: "start", partial: assistant() }); + stream.push({ + type: "error", + reason: "error", + error: assistantError( + '{"type":"error","error":{"type":"authentication_error","message":"Invalid authentication credentials"}}', + 401, + ), + }); + return; + } + ok(stream); + }); + return stream; + }, + SOURCE_ID, + ); + + const stream = streamSimple(model(), context, { + apiKey: async ctx => (ctx.error === undefined ? "old-key" : "new-key"), + }); + for await (const _event of stream) { + // drain + } + + expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); + expect(keys).toEqual(["old-key", "new-key"]); }); it("does not retry after replay-unsafe content has been emitted", async () => { - let authCalls = 0; + let retryResolves = 0; const failure = authError(); registerCustomApi( API, @@ -175,10 +210,9 @@ describe("streamSimple auth retry", () => { ); const stream = streamSimple(model(), context, { - apiKey: "old-key", - onAuthError: async () => { - authCalls += 1; - return "new-key"; + apiKey: async ctx => { + if (ctx.error !== undefined) retryResolves += 1; + return ctx.error === undefined ? "old-key" : "new-key"; }, }); @@ -192,94 +226,79 @@ describe("streamSimple auth retry", () => { } expect(caught).toBe(failure); - expect(authCalls).toBe(0); + // The resolver is never asked for a retry key once a replay-unsafe event shipped. + expect(retryResolves).toBe(0); }); - it("retries on 401 carried via errorStatus when the message has no parseable status", async () => { - const keys: Array = []; - let authCalls = 0; + it("escalates refresh-same then switch in order (2-retry ordering)", async () => { + const keys: unknown[] = []; registerCustomApi( API, (_model: Model, _context: Context, options?: SimpleStreamOptions) => { - keys.push(options?.apiKey); + pushKey(keys, options); const stream = new AssistantMessageEventStream(); - queueMicrotask(() => { - if (keys.length === 1) { - stream.push({ type: "start", partial: assistant() }); - stream.push({ - type: "error", - reason: "error", - // Realistic Anthropic SDK shape: message begins with ` ` - // and the regex fallback in extractHttpStatusFromError cannot find 401 - // inside the JSON body. Only `errorStatus` carries the signal. - error: assistantError( - '{"type":"error","error":{"type":"authentication_error","message":"Invalid authentication credentials"}}', - 401, - ), - }); - return; - } - const message = assistant(["ok"]); - stream.push({ type: "start", partial: message }); - stream.push({ type: "done", reason: "stop", message }); - }); + queueMicrotask(() => (options?.apiKey === "switch-key" ? ok(stream) : stream.fail(authError()))); return stream; }, SOURCE_ID, ); const stream = streamSimple(model(), context, { - apiKey: "old-key", - onAuthError: async () => { - authCalls += 1; - return "new-key"; - }, + apiKey: async ctx => (ctx.error === undefined ? "old-key" : ctx.lastChance ? "switch-key" : "refresh-key"), }); - for await (const _event of stream) { // drain } expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); - expect(keys).toEqual(["old-key", "new-key"]); - expect(authCalls).toBe(1); + expect(keys).toEqual(["old-key", "refresh-key", "switch-key"]); }); - it("retries on a thrown usage_limit_reached error before any event has been emitted", async () => { - const keys: Array = []; - const errors: unknown[] = []; + it("skips the refresh-same step when the resolver returns an unchanged key", async () => { + const keys: unknown[] = []; registerCustomApi( API, (_model: Model, _context: Context, options?: SimpleStreamOptions) => { - keys.push(options?.apiKey); + pushKey(keys, options); const stream = new AssistantMessageEventStream(); - queueMicrotask(() => { - if (keys.length === 1) { - stream.fail( - Object.assign( - new Error("You have hit your ChatGPT usage limit (pro plan). Try again in ~158 min."), - { status: 429 }, - ), - ); - return; - } - const message = assistant(["ok"]); - stream.push({ type: "start", partial: message }); - stream.push({ type: "done", reason: "stop", message }); - }); + queueMicrotask(() => (options?.apiKey === "switch-key" ? ok(stream) : stream.fail(authError()))); return stream; }, SOURCE_ID, ); const stream = streamSimple(model(), context, { - apiKey: "old-key", - onAuthError: async (_provider, _oldKey, error) => { - errors.push(error); - return "new-key"; + // refresh-same yields the same failing key → that attempt is skipped. + apiKey: async ctx => (ctx.error === undefined ? "old-key" : ctx.lastChance ? "switch-key" : "old-key"), + }); + for await (const _event of stream) { + // drain + } + + expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); + expect(keys).toEqual(["old-key", "switch-key"]); + }); + + it("retries a thrown usage-limit error and passes the cause to the resolver", async () => { + const keys: unknown[] = []; + const errors: unknown[] = []; + registerCustomApi( + API, + (_model: Model, _context: Context, options?: SimpleStreamOptions) => { + pushKey(keys, options); + const stream = new AssistantMessageEventStream(); + queueMicrotask(() => (keys.length === 1 ? stream.fail(usageLimitError()) : ok(stream))); + return stream; + }, + SOURCE_ID, + ); + + const stream = streamSimple(model(), context, { + apiKey: async ctx => { + if (ctx.error !== undefined) errors.push(ctx.error); + return ctx.error === undefined ? "old-key" : "new-key"; }, }); - for await (const _event of stream) { // drain } @@ -287,18 +306,17 @@ describe("streamSimple auth retry", () => { expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); expect(keys).toEqual(["old-key", "new-key"]); expect(errors).toHaveLength(1); - // The error surfaced to onAuthError carries the original 429 status so - // the gateway's refresh hook can branch on it. + // The cause carries the original 429 so the resolver can branch usage-limit vs 401. expect((errors[0] as { status?: number }).status).toBe(429); expect((errors[0] as Error).message).toMatch(/usage limit/i); }); - it("retries when a provider emits a usage_limit_reached error event before content", async () => { - const keys: Array = []; + it("retries a usage-limit error event before content", async () => { + const keys: unknown[] = []; registerCustomApi( API, (_model: Model, _context: Context, options?: SimpleStreamOptions) => { - keys.push(options?.apiKey); + pushKey(keys, options); const stream = new AssistantMessageEventStream(); queueMicrotask(() => { if (keys.length === 1) { @@ -306,16 +324,11 @@ describe("streamSimple auth retry", () => { stream.push({ type: "error", reason: "error", - // errorStatus deliberately omitted: matches how the codex - // provider serializes the message-only path that used to - // 502 in the gateway. error: assistantError("You have hit your ChatGPT usage limit (pro plan). Try again in ~158 min."), }); return; } - const message = assistant(["ok"]); - stream.push({ type: "start", partial: message }); - stream.push({ type: "done", reason: "stop", message }); + ok(stream); }); return stream; }, @@ -323,10 +336,8 @@ describe("streamSimple auth retry", () => { ); const stream = streamSimple(model(), context, { - apiKey: "old-key", - onAuthError: async () => "new-key", + apiKey: async ctx => (ctx.error === undefined ? "old-key" : "new-key"), }); - for await (const _event of stream) { // drain } @@ -335,13 +346,13 @@ describe("streamSimple auth retry", () => { expect(keys).toEqual(["old-key", "new-key"]); }); - it("surfaces the original usage_limit error when the retry callback declines", async () => { - const keys: Array = []; - const original = Object.assign(new Error("You have hit your ChatGPT usage limit (pro plan)."), { status: 429 }); + it("surfaces the original error when the resolver declines every retry", async () => { + const keys: unknown[] = []; + const original = usageLimitError(); registerCustomApi( API, (_model: Model, _context: Context, options?: SimpleStreamOptions) => { - keys.push(options?.apiKey); + pushKey(keys, options); const stream = new AssistantMessageEventStream(); queueMicrotask(() => stream.fail(original)); return stream; @@ -349,12 +360,9 @@ describe("streamSimple auth retry", () => { SOURCE_ID, ); - // Callback returns undefined → no sibling credential to rotate to. - // The original failure must reach the caller untouched so the client - // can decide what to do (back off, surface to user, …). const stream = streamSimple(model(), context, { - apiKey: "old-key", - onAuthError: async () => undefined, + // Decline all retries: no sibling credential to rotate to. + apiKey: async ctx => (ctx.error === undefined ? "old-key" : undefined), }); let caught: unknown; @@ -367,8 +375,32 @@ describe("streamSimple auth retry", () => { } expect(caught).toBe(original); - // Single attempt — the inner stream is only re-invoked when a new key - // is provided. expect(keys).toEqual(["old-key"]); }); + + it("fails the stream when the initial resolve yields no key", async () => { + let attempts = 0; + registerCustomApi( + API, + () => { + attempts += 1; + return new AssistantMessageEventStream(); + }, + SOURCE_ID, + ); + + const stream = streamSimple(model(), context, { apiKey: async () => undefined }); + + let caught: unknown; + try { + for await (const _event of stream) { + // drain + } + } catch (error) { + caught = error; + } + + expect((caught as Error).message).toMatch(/No API key for provider/); + expect(attempts).toBe(0); + }); }); diff --git a/packages/ai/test/tool-argument-coercion.test.ts b/packages/ai/test/tool-argument-coercion.test.ts index 224159d5b..e0ca64378 100644 --- a/packages/ai/test/tool-argument-coercion.test.ts +++ b/packages/ai/test/tool-argument-coercion.test.ts @@ -78,6 +78,178 @@ describe("Tool argument coercion", () => { expect(result.paths).toEqual(["src/**/*.ts"]); }); + it("wraps a singleton object in an array when schema expects object array", () => { + const tool: Tool = { + name: "todo_like", + description: "", + parameters: z.object({ + ops: z.array( + z.object({ + op: z.literal("init"), + list: z.array( + z.object({ + phase: z.string(), + items: z.array(z.string()), + }), + ), + }), + ), + }), + }; + + const result = validateToolArguments(tool, { + type: "toolCall", + id: "call-singleton-object-array", + name: "todo_like", + arguments: { + ops: { + op: "init", + list: [{ phase: "Repro", items: ["capture"] }], + }, + }, + }); + + expect(result).toEqual({ + ops: [{ op: "init", list: [{ phase: "Repro", items: ["capture"] }] }], + }); + }); + + it("wraps a singleton number in an array when schema expects number array", () => { + const tool: Tool = { + name: "numeric_list", + description: "", + parameters: z.object({ values: z.array(z.number()) }), + }; + + const result = validateToolArguments(tool, { + type: "toolCall", + id: "call-singleton-number-array", + name: "numeric_list", + arguments: { values: 7 }, + }); + + expect(result).toEqual({ values: [7] }); + }); + + it("does not wrap singleton values for array expectations from failed union branches", () => { + const entry = z.object({ id: z.number() }); + const tool: Tool = { + name: "union_shape", + description: "", + parameters: z.union([z.array(entry), entry]), + }; + + const result = validateToolArguments(tool, { + type: "toolCall", + id: "call-union-shape", + name: "union_shape", + arguments: { id: "1" }, + }); + + expect(result).toEqual({ id: 1 }); + }); + + it("does not wrap singleton values for JSON Schema anyOf array branches", () => { + const tool: Tool = { + name: "json_schema_union", + description: "", + parameters: { + type: "object", + properties: { + target: { + anyOf: [ + { + type: "array", + items: { + type: "object", + properties: { a: { type: "boolean" } }, + required: ["a"], + additionalProperties: false, + }, + }, + { + type: "object", + properties: { a: { type: "boolean" } }, + required: ["a"], + additionalProperties: false, + }, + ], + }, + }, + required: ["target"], + additionalProperties: false, + }, + }; + + // The bug would silently coerce `{ a: "true" }` into `[{ a: true }]` by + // wrapping the object to satisfy the failed `anyOf` array branch and + // then coercing the inner string into a boolean. Branch tracking keeps + // the wrap from firing so the wrong shape never makes it through. + expect(() => + validateToolArguments(tool, { + type: "toolCall", + id: "call-jsonschema-union", + name: "json_schema_union", + arguments: { target: { a: "true" } }, + }), + ).toThrow("Validation failed"); + }); + + it("still wraps nested array fields inside a tag-selected Zod union branch", () => { + const tool: Tool = { + name: "tagged_union", + description: "", + parameters: z.union([ + z.object({ type: z.literal("indices"), indices: z.array(z.number()) }), + z.object({ type: z.literal("all") }), + ]), + }; + + const result = validateToolArguments(tool, { + type: "toolCall", + id: "call-tagged-union", + name: "tagged_union", + arguments: { type: "indices", indices: 1 }, + }); + + expect(result).toEqual({ type: "indices", indices: [1] }); + }); + + it("still wraps nested array fields inside a tag-selected JSON Schema anyOf branch", () => { + const tool: Tool = { + name: "tagged_json_union", + description: "", + parameters: { + anyOf: [ + { + type: "object", + properties: { + type: { const: "indices" }, + indices: { type: "array", items: { type: "number" } }, + }, + required: ["type", "indices"], + additionalProperties: false, + }, + { + type: "object", + properties: { type: { const: "all" } }, + required: ["type"], + additionalProperties: false, + }, + ], + }, + }; + + const result = validateToolArguments(tool, { + type: "toolCall", + id: "call-tagged-json-union", + name: "tagged_json_union", + arguments: { type: "indices", indices: 1 }, + }); + + expect(result).toEqual({ type: "indices", indices: [1] }); + }); + it("parses JSON objects in string values when schema expects object", () => { const tool: Tool = { name: "t4", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 95733447a..0243ed841 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,178 @@ # Changelog ## [Unreleased] + +## [15.10.1] - 2026-06-07 + +### Added + +- Added `display.smoothStreaming` setting (default `true`) to let users enable or disable smooth assistant-stream text reveal +- Added `/tan ` slash command to fork the current conversation into a background agent so tangential work can continue asynchronously while your main session stays active +- Added a background `/tan` dispatch message that records the handoff in the transcript and marks the delegated work as non-blocking +- Added `providerPromptCacheKey` support to `CreateAgentSessionOptions` so `/tan` background sessions can reuse the parent session’s prompt-cache lineage +- Added session cloning for `/tan` runs with copied artifacts and shared MCP proxy tools +- Added `SessionManager.forkFrom`’s optional `suppressBreadcrumb` mode to avoid breadcrumb updates when forking background `/tan` sessions +- Added OSC 5522 enhanced paste handling in `InputController`, so terminal clipboard events are decoded as image or text payloads and inserted without passing raw paste sequences to the editor +- Added bracketed image-path paste support in `CustomEditor` so a single pasted image file path (PNG/JPEG/GIF/WEBP) is loaded from disk and inserted as an image candidate +- Added direct support for `Image #N` insertion from pasted local image paths by routing successful image-path pastes through the same image normalization and resize flow as clipboard image pastes +- Added `/fresh` to rotate the provider-facing session id and clear in-memory provider stream/cache state without changing the local session file. +- Added a `ChatBlock` transcript primitive (`modes/components/chat-block.ts`) and a single `ctx.present(...)` sink (with `ctx.resetTranscript()`) so chat output is mounted in one place instead of the repeated `chatContainer.addChild(...)` + `ui.requestRender()` pattern scattered across controllers. `ChatBlock` carries a React/Svelte-style lifecycle — `onMount` starts effects, `onCleanup` registers teardown, `finish()` self-completes (stops timers and freezes the block at its final content), and `dispose()`/`resetTranscript()` tears everything down — so animated blocks own their own resources instead of leaking `setInterval`/`requestRender` bookkeeping into callers. The MCP "Connecting…" spinner is now such a block. +- Added a `framedBlock` output-block helper (`tui/output-block.ts`) plus a `borderColor` override and `applyBg: false` (no background fill) on output blocks, a `renderStatusLine` `iconOverride`, and an `icon.search` (magnifier) theme symbol — so tool renderers can draw self-contained muted-outline frames and search-family tools can show a magnifier instead of a checkmark. + +### Changed + +- Changed the bash tool frame to use a plain top rule instead of repeating "Bash" in the title bar, and folded minimizer raw-output artifact links into the status footer as `Artifact: `. + +- Changed grouped `read` output to use a white filled-circle mark for the group/single-read success state and omit duplicate per-file success marks inside multi-read groups. + +- Changed assistant streaming output to reveal text incrementally at 30 FPS with grapheme-safe adaptive catch-up, instead of replacing the whole message chunk-by-chunk +- Changed shimmer-driven TUI animations (working text, pending bash/eval borders, and theme activity-spinner documentation) to render at 30fps instead of 60fps. +- Changed running `task` tool agent rows to use a static `•` marker and shimmer only the subagent name, leaving descriptions, stats, and nested tool detail text solid while removing the rotating status glyph from those rows. +- Changed settings singleton method access to reuse bound methods for the active instance instead of allocating a new bound function on every `settings.get` lookup. +- Changed plan-mode approval to keep the drafted `local://-plan.md` file at its original name as the canonical plan path, so approved plans are no longer renamed when leaving plan mode +- Changed plan-mode write enforcement so only `local://` artifact files are writable during planning, blocking working-tree edits and allowing scratch or draft plan files in the local artifact area +- Changed the `todo` tool result renderer to stop redrawing every phase's full task list on each update: when a multi-phase list is rendered collapsed (the default, not manually expanded), only phases the latest update touched — the phase holding the in_progress task, any phase with a just-completed task, and phases named by the ops that ran (`init` counts as touching all) — render their tasks; untouched phases collapse to a one-line `N. Name done/total` summary. When call args are unavailable (e.g. transcript rebuilds) it falls back to the in_progress/completed-transition signals, and the manual expand toggle still shows every task. Also dropped the blank separator line previously inserted between phases. +- Changed non-agent API operations (title and commit-message generation, image generation, web search, eval `llm()`, auto-thinking classifier, memory consolidation) to use session-aware API key resolution with auth retries via `registry.resolver()` / `authStorage.resolver()`, refreshing the active credential before rotating to another account +- Changed image generation to wrap every provider fetch branch in `withAuth`, so 401 / usage-limit errors trigger credential force-refresh and rotation for authStorage-backed providers (OpenAI-hosted, antigravity, xai-oauth) while env-only providers (openrouter, gemini) stay single-attempt +- Changed web-search providers using `authStorage.getApiKey` (anthropic, exa, tavily, parallel, synthetic, zai, kimi) to wrap HTTP calls in `withAuth` for automatic credential rotation on 401 / usage-limit errors +- Changed the directory grouping for `find`, `search`, `ast_grep`, `ast_edit`, and `lsp` diagnostics from a single flat `# dir/` heading per immediate directory to a multi-level tree that folds the common path prefix into one heading. Previously every group repeated the full directory path — so results rooted outside cwd printed the absolute prefix (e.g. `/Users/me/proj/`) on every heading and nested directories were never collapsed. Now a single-child directory chain folds into one heading (`# packages/pkg/src/`, including an absolute root for out-of-cwd results), subdirectories nest one `#` deeper (`## nested/` → `### child.ts`), and each directory's own files are listed before its subdirectories. TUI hyperlink reconstruction tracks the nested directory stack across the whole output so file and code-frame links keep resolving to the correct absolute paths. +- Changed the plan-mode approval surface from an inline transcript block plus a separate bottom selector into a single fullscreen overlay (like `/copy`) and overhauled its navigation. The overlay now renders the plan per-section through `ScrollView` (line-level ↑/↓ scroll, Shift+↑/↓ to scroll faster, PgUp/PgDn, g/G) with no stray per-line `…`, and — when the terminal is wide enough and the plan has ≥2 headings — shows a compact VS Code-style section sidebar (the redundant plan-title heading and any "Contents" label are omitted). Focus moves between regions with Tab/Shift+Tab (and flows at the edges: Down past the last section or the bottom of the body drops into the approval options; Up steps back), while the sidebar glows to track the scrolled section. The sidebar can fast-jump between sections, delete a section (with `u` undo), and annotate sections with feedback (`a`); deletions and annotations are collected into refinement feedback that is submitted back to the model when the operator picks "Refine plan". Mouse works too: clicking an approval option activates it, clicking a sidebar section jumps to it, and the wheel scrolls the plan. ←/→ always drive the model-tier slider, Enter confirms, the external-editor key opens the plan, and Esc cancels. The overlay borrows the terminal's alternate screen buffer for its lifetime (`fullscreen` overlay), so the transcript stays put on the normal screen instead of bleeding through scrollback behind the modal. +- Changed the interactive controllers (command, MCP, selector, extension-UI, event), debug panels, and the status/error/warning helpers to render chat output through `ctx.present(...)` instead of appending to `chatContainer` and calling `ui.requestRender()` directly; transcript rebuilds dispose live blocks via `ctx.resetTranscript()` so animated blocks' timers stop on reset. +- Changed tool-execution block rendering so the container (`ToolExecutionComponent`) is a transparent passthrough — it no longer inserts a top/bottom blank line, adds left/right padding, or paints a state-colored background behind tool output. Tools with substantial body now self-frame with a muted outline and the tool title in the frame's top bar (`edit`/`apply_patch`, `write`, `ask`, `todo`, `github`, `goal`, `inspect_image`, `search_tool_bm25`, `task`), matching the already-framed `bash`/`read`/`eval`/`debug`/`web_search`/`lsp` blocks, while streaming/in-progress and trivial results collapse to a clean status line. The search-family list tools (`find`, `search`, `ast_grep`) and `job` render frameless/minimal; `find`/`search`/`ast_grep` show a magnifier on success instead of a checkmark, and `job` drops its `Job:` label prefix (the per-job rows are self-describing). The `search_tool_bm25`, `github`, and `inspect_image` frames draw with no background fill, and `inspect_image`'s label was shortened to `Inspect`. +- Changed the plan-mode active prompt (`prompts/system/plan-mode-active.md`) to make plans decision-complete and cut filler. Added an Objective framing ("another engineer can execute end-to-end without making a single design decision"), a shared "Resolving Unknowns" section (explore discoverable facts before asking; reserve `ask` for non-derivable preferences/tradeoffs with 2–4 options + a recommended default), and a single shared "The Plan" structure (Context / Approach grouped by behavior not file-by-file / ≤5 Critical files / Verification / Assumptions) that replaces the per-branch structure guidance previously duplicated across the iterative and parallel workflows. Added explicit prohibitions on sections that decide nothing (Non-Goals, Out of Scope, Alternatives Considered, Risks/Mitigations boilerplate, Future Work), on enumerating every file/line, and on inventing schema/validation/precedence policy the request never established. +- Changed completion notifications (`completion.notify`) to fire whenever the agent yields its turn, including in the foreground. The `agent_end` notification was previously gated behind background mode (`isBackgrounded`), so an ordinary foreground turn never emitted one; the gate is gone and the desktop toast now fires on every normal turn completion (still skipped for aborted/error turns and when `completion.notify` is `off`). +- Changed the in-progress `task` tool block to keep the shared `context` brief (`# Goal` / `# Constraints` background) visible after the first progress snapshot arrives, instead of dropping it the moment the streaming call view was replaced by the result frame, and to stop animating a spinner/clock next to the `Task` frame header while running — the per-agent body lines already carry their own running spinner, so the header now shows a static state icon (matching the completed/failed header icons). The context is rendered through a shared `buildContextSection` helper that also undoes per-field double-encoding, so the brief reads cleanly in the result frame even though `renderResult` receives the raw (un-repaired) tool args. +- Changed the messaging shown when you press Esc to interrupt a streaming turn from the ambiguous `Operation aborted` / `Tool execution was aborted: Request was aborted` to `Interrupted by user`, so a deliberate user interrupt no longer reads like an internal failure. Every Esc/flush interrupt path (`onEscape` while streaming, the queued-message restore-and-abort path, and the empty-submit queue flush) threads the reason through `AgentSession.abort({ reason })` → `Agent.abort(reason)` so it rides the `AbortController` onto the aborted assistant message's `errorMessage`; the turn label renders it verbatim on both the live and replay paths, and the synthetic placeholder results paired with in-flight tool calls now read `Tool execution was aborted: Interrupted by user`. Aborts that carry no reason still fall back to the retry-aware `Operation aborted` generic. Transcript label resolution is centralized in `resolveAbortLabel` (`session/messages.ts`). + +### Removed + +- Removed the `/background` (and `/bg`) slash command and the background-mode subsystem it was the sole entry point for — `InteractiveMode.isBackgrounded`, `createBackgroundUiContext`, `handleBackgroundEvent`, and every `isBackgrounded` guard across the input/event/extension-UI controllers and UI helpers. The command suspended the whole process group via `SIGTSTP` (a leftover testing shortcut) instead of detaching the running agent, which is not the expected workflow — use terminal panes or a multiplexer instead. + +### Fixed + +- Fixed inline `find` and `search` result blocks to align with grouped `read` output and render their success headers with the normal tool-title color instead of accent blue. + +- Fixed the working-status shimmer to opt into the loader's 30fps animated-message repaint path while keeping both the status spinner and pending bash/eval tool spinners on their normal 80 ms glyph cadence. +- Fixed consecutive `read` tool calls failing to collapse into a single grouped block when a reasoning model emits one read per completion (`[thinking, read]`). The read group was reset on every assistant `message_start`, so each read rendered as its own one-entry `Read …` line; now a read run accretes across completions and is broken only by a rendered non-empty text/thinking block, a non-read tool, or a user/IRC message — matching the transcript-rebuild path. `ReadToolGroupComponent` now reports its live/finalized state so the growing `Read (N)` header repaints correctly on native-scrollback (risk) terminals. +- Fixed the `task` tool shared-context brief rendering raw Markdown headings (`# Goal`, `# Constraints`) inside framed call/result blocks instead of using the normal Markdown renderer. +- Fixed the animated pending border on `bash`/`eval` blocks leaving a frozen dark "bar" segment behind after a backgrounded command finalized through the async update path. Once a command is auto-backgrounded (`details.async.state === "running"`) the block stays "partial" in the TUI until the async job-manager delivers the final result, but it also gets committed to native scrollback — so a mid-sweep shimmer frame baked a stray darkened border segment into the committed copy. The border now stops animating (and the 60fps redraw loop stops) the moment a block enters the backgrounded state, so the committed frame is a clean static border. +- Fixed cold `omp` launch to clear native terminal history on the first paint, avoiding a once-per-launch duplicate welcome/transcript copy before the normal session replay. +- Fixed plan approval resolution so `resolve` with `action: "apply"` can still find the plan file when `extra.title` is missing or stale by falling back to the current plan path and most-recent local plan artifacts +- Fixed the search-family tool magnifier glyph (`find`, `search`, `ast_grep`, `search_tool_bm25`) to use the `accent` title color instead of `success` green, so the icon matches the tool title in the status header instead of standing out +- Fixed TTSR stream interrupts to pass the matched rule name through the abort reason, so aborted in-flight tool placeholders say why they were stopped instead of `Request was aborted`. +- Fixed URL reads for binary/special payloads to reuse local readers: remote archives list their root entries, SQLite databases show their table overview, notebooks render as editable cells, and unrenderable binary returns a metadata notice instead of decoded byte garbage. +- Fixed pasted image-file paths that cannot be loaded to fall back to normal text paste with status feedback instead of disappearing. +- Fixed tool-output file paths not being clickable OSC 8 `file://` hyperlinks in several renderers. `read` titles for plain text and image files (the common case) emitted no link at all because the renderer only linked when a `resolvedPath` was recorded — which the ordinary file/image read paths never set, keeping the absolute path only in `meta.source`; the renderer now falls back to that source path. `write` headers were never wrapped in a hyperlink and now link to the absolute path written (file, archive entry, SQLite, and conflict resolutions). `edit`/`apply_patch` headers wrapped the model-supplied (often cwd-relative) argument path, producing a root-anchored `file:///rel/path` URI; they now link the absolute `details.path` instead. Finally, `search`, `ast_grep`, and `ast_edit` produced doubled link targets (`/proj/src/src/file.ts`) for searches scoped to a subdirectory, because the renderer resolved the cwd-relative display paths against the scope directory rather than cwd — the scoped-search base is now the session cwd (with the scoped file's absolute path still seeding single-file body lines). +- Fixed `omp dry-balance --bench` to recover from 401 token failures by re-minting the failing OAuth credential in place before switching accounts +- Fixed the bash tool corrupting commands that embed multi-byte UTF-8 (e.g. `✓`/`×` inside a `grep -E` pattern) ahead of a trailing `| head`/`| tail`. The `bash.stripTrailingHeadTail` rewrite cut at char-offset positions reported by `brush-parser` while slicing the command by byte offset, so the trailing-pipe strip landed mid-pattern and dropped the closing quote — turning `… |✓|×|XCTAssert" | tail -80` into `… |✓|×-80` and making execution fail with `pi-natives:command: unterminated double quote`. Fixed in `pi_shell::fixup` (`@oh-my-pi/pi-natives`). +- Fixed `omp dry-balance --bench` to recover from 401 token failures by re-minting the failing OAuth credential in place before switching accounts +- Fixed duplicate file entries in grouped outputs for `find`, `search`, `ast_grep`, `ast_edit`, and `lsp` diagnostics when the same path appeared multiple times +- Fixed search, grep, and edit output rendering so repeated directory group blank-line boundaries no longer break nested path/link reconstruction +- Fixed `omp dry-balance --bench` flooding the terminal with staircased, duplicated spinner/status lines (and an indented summary) when the tty has ONLCR/OPOST disabled (raw mode). The interactive progress region separated rows with a bare LF and repositioned with a column-preserving `\x1b[A` cursor-up, both of which only land at column 0 when the terminal translates LF→CRLF; with that translation off, every 80 ms redraw cascaded down and to the right into scrollback. The live region now carriage-returns before every cleared row, terminates each row with CRLF, and caps each row to the terminal width so a wrapped line cannot desync the cursor-up from the logical line count. +- Fixed inconsistent vertical spacing between transcript blocks: some blocks (tool results from `search`/`find` and other renderer-backed tools) rendered with a doubled gap (a leading `Spacer` plus the content box's own `paddingY`), while others (the grouped `read` card, file-mention lists, IRC cards) rendered with no gap at all. Vertical spacing is now owned entirely by the chat renderer: `TranscriptContainer` strips each block's plain-blank top/bottom edges and inserts exactly one blank line between consecutive blocks, so every block is separated by a single consistent gap regardless of which component produced it. Individual components (assistant/user/tool/read-group/bash/eval/skill/custom/hook/compaction/branch/todo-reminder/plan-review messages) no longer emit their own leading `Spacer`/`paddingY` for separation, and multi-row groups (IRC cards, file-mention lists, completed-job batches, and the bordered command/`/changelog`/`/context`/version/OAuth/debug panels) are wrapped as single `TranscriptBlock` children so the renderer spaces them as one unit. Background-colored box padding is preserved as block-internal design. +- Fixed `resolve` with `action: "discard"` surfacing a hard `isError` "No pending action to resolve" failure to the model when the agent asked to cancel a staged action (e.g. an `ast_edit` preview) but nothing was pending. A discard is a request to reach the "no staged change" end-state, which already holds in that case, so it is now honored as a successful cancellation (`"Nothing to discard; no pending action remains."` with `details.action: "discard"`) instead of an error. `action: "apply"` with no pending action still errors. +- Fixed the collapsed tool-output expand hint rendering double brackets (e.g. `((Ctrl+O for more))`) — the `EXPAND_HINT` text already carried its own parentheses and then `formatExpandHint` wrapped it again with the theme's bracket glyphs. The hint now resolves the key actually bound to `app.tools.expand` at render time and reads `⟨: Expand⟩` (e.g. `⟨Ctrl+O: Expand⟩`), so a single bracket pair surrounds it and a user remap of the expand keybinding is reflected instead of a hard-coded `Ctrl+O`. +- Fixed the `edit`/`apply_patch` tool dropping its outlined frame while streaming/in-progress (only the final result was framed); the in-progress diff preview now renders inside the same muted frame as the completed result. +- Fixed the `todo` and `job` tools rendering a success icon and success styling on a failed/error result; error results now show the error icon and a red frame border. +- Fixed `debug` tool refusing every `dlv` launch on Go modules. The launch handler ran `validateLaunchProgram` before adapter selection and rejected any directory program with `launch program resolves to a directory`, while dlv's default `mode=debug` requires a Go package path (a directory or `.go` source file). Adapter resolution now precedes validation, directory programs prefer adapters that advertise `acceptsDirectoryProgram` before falling back to native extensionless debuggers, the rejection only fires when the resolved adapter does not advertise that flag (set on `dlv` in `dap/defaults.json`), and dlv's `mode` is derived from the program shape — directories and `.go` files launch as `mode=debug`, other files as `mode=exec` — so `omp` can debug both Go packages and pre-built binaries ([#2020](https://github.com/can1357/oh-my-pi/issues/2020)). + +## [15.10.0] - 2026-06-06 + +### Breaking Changes + +- Replaced the `providers.parallelFetch` boolean setting with the `providers.fetch` enum (`auto` / `native` / `trafilatura` / `lynx` / `parallel` / `jina`) that selects the URL reader-backend priority for the `read`/`fetch` tool, mirroring `providers.image`/`providers.webSearch`. Existing configs are migrated automatically: the legacy key is dropped and the new `auto` default applies. + +### Added + +- Added a GitHub Actions read handler to the `read`/web-fetch GitHub scraper. Fetching `github.com/{owner}/{repo}/actions/runs/{id}` renders the run metadata plus a per-job breakdown (steps listed for any job that did not succeed), and `…/actions/runs/{id}/job/{id}` (also the API-style `…/jobs/{id}`) renders a single job's metadata, step table, and full plain-text logs. Logs are fetched via the `actions/jobs/{id}/logs` redirect using `GITHUB_TOKEN`/`GH_TOKEN` when present, with the per-line ISO timestamp prefix and leading BOM stripped; the section degrades to an explicit notice when logs are unavailable (no token, private repo, or expired/unfinalized run). + +### Changed + +- Changed eval `agent()` subagents so they are never subject to the `task.maxRuntimeMs` wall-clock cap. The parent cell's idle watchdog is already suspended for the entire bridge call (`withBridgeTimeoutPause`), so a long-running fan-out/recovery workflow must not be killed by a per-subagent runtime limit. `runEvalAgent` now passes `maxRuntimeMs: 0` to `runSubprocess`, which honors an explicit `ExecutorOptions.maxRuntimeMs` override over the inherited setting. +- Changed interactive timing behavior so `PI_TIMING=x pi` preloads the module timer before the CLI graph loads and includes the `(modules)` report. `PI_TIMING=full` now also exits after printing, matching `PI_TIMING=x`, so full module reports are usable for cold-start measurement without launching the TUI. Added the root `dev:timing` script for the same profiled startup path. +- Changed coding-agent startup imports so normal TUI launch imports `InteractiveMode` directly, keeps print/RPC/ACP runners on their branch-only paths, and moves marketplace auto-update work behind a lightweight deferred starter. +- Changed cold-launch setup gating so the full setup wizard (every scene plus the overlay and their TUI/OAuth/web-search/theme dependencies) is no longer statically imported by `main.ts`. The current setup version now lives in a tiny dependency-free `modes/setup-version` module, and the wizard barrel is lazy-loaded only when the stored setup version is stale or the wizard is forced — the common up-to-date launch skips loading it entirely. +- Changed cold-launch startup imports so the hot-path CLI files no longer pull the full `@oh-my-pi/pi-ai` barrel: `commands/launch.ts` and `cli/args.ts` import `THINKING_EFFORTS`/`Effort` from the tiny `@oh-my-pi/pi-ai/effort` module, and `config/model-registry.ts` now imports its ~20 symbols from narrow subpaths (`api-registry`, `model-cache`, `model-manager`, `model-thinking`, `models`, `provider-models`, `types`, `utils/event-stream`) instead of the barrel — so launching no longer eagerly loads every provider, auth, OAuth, and usage module re-exported by the barrel. +- Changed the `read`/`fetch` HTML reader-backend priority to `native > trafilatura > lynx > parallel > jina` (was `parallel > jina > trafilatura > lynx > native`). The in-process native `htmlToMarkdown` runs first — instant, no network, full-fidelity — so the common case no longer depends on a remote service, and a stalled remote backend can no longer mask it. Selecting a specific backend via `providers.fetch` tries it first, then the rest fall back. The low-quality gate (`>100` chars and not `isLowQualityOutput`) now applies uniformly to every backend; when none clears it, the highest-priority substantial-but-low-quality output is still surfaced so the `llms.txt` / document-extraction fallbacks keep running. + +### Fixed + +- Fixed eval `agent()` failures surfacing as an opaque `RuntimeError: bridge call '__agent__' failed` with no reason. When a subagent aborted, `runEvalAgent` built its failure message with `result.error ?? result.stderr ?? result.abortReason ?? …`, but `result.stderr` is the empty string on a clean abort (and `result.error` is gated on a non-empty `stderr`), so the nullish chain stopped at `""` and never reached `abortReason`. The empty string propagated through the loopback bridge and the Python prelude's `RuntimeError(msg or "bridge call … failed")`, discarding the real reason. The chain now uses `||` so an empty `stderr` falls through to `abortReason`. +- Fixed subagent aborts being mislabeled as the generic "Cancelled by caller" when the abort originated inside the subagent's own turn (`stopReason: "aborted"` with no caller signal and no runtime-limit timer). `runSubprocess` now prefers the aborted assistant message's `errorMessage` (e.g. "Request was aborted" or a specific stream error) for that case, while a real caller signal or wall-clock abort still reports its precise reason. +- Fixed a long streaming tool preview that alone overflows the viewport dropping its scrolled-off head on ED3-risk terminals (ghostty/kitty/iTerm2/…). When expanded with `Ctrl+O`, a streaming `write` (content streaming in) and a streaming `eval` (stdout streaming below its fixed code cell) render top-anchored and grow append-only, but the tool block never reported itself append-only to the transcript, so the renderer's commit-as-you-go boundary stopped at the block start and the earlier rows that scrolled above the viewport were committed nowhere — they vanished, leaving the preview looking like a viewport-tall circular buffer. `ToolExecutionComponent` now implements `isTranscriptBlockAppendOnly()` (gated on `isTranscriptBlockFinalized()`, so it also covers partial-result streams like `eval`), delegating to a renderer-declared `isStreamingPreviewAppendOnly` predicate so the expanded stream commits its head exactly like a streamed assistant reply. Collapsed previews (bounded sliding tail windows) and finalized/result previews (which can collapse to a capped view) stay deferred. +- Fixed `read`/`fetch` silently dropping whole list sections on pages with malformed list markup — stray ``, text, or `