diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 09245ef0a..fad426fb6 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -16,6 +16,9 @@ concurrency: group: ${{ github.workflow }}-${{ github.ref }} cancel-in-progress: true +env: + FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true + jobs: # scripts/release.ts pushes the version-bump commit and its `v*` tag # atomically (`git push --atomic origin main refs/tags/v*`), so a release diff --git a/Cargo.lock b/Cargo.lock index b88543c78..a2f4c89f2 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2331,7 +2331,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "15.9.67" +version = "15.10.1" dependencies = [ "anyhow", "ast-grep-core", @@ -2399,7 +2399,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "15.9.67" +version = "15.10.1" dependencies = [ "async-trait", "libc", @@ -2411,7 +2411,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "15.9.67" +version = "15.10.1" dependencies = [ "anyhow", "arboard", @@ -2457,7 +2457,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "15.9.67" +version = "15.10.1" dependencies = [ "anyhow", "brush-builtins", diff --git a/Cargo.toml b/Cargo.toml index 2cf4d2996..f6301942c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"] resolver = "3" [workspace.package] -version = "15.9.67" +version = "15.10.1" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index bff8b3963..3cff53436 100644 --- a/bun.lock +++ b/bun.lock @@ -15,7 +15,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "15.9.67", + "version": "15.10.1", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -30,7 +30,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "15.9.67", + "version": "15.10.1", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -44,7 +44,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "15.9.67", + "version": "15.10.1", "bin": { "omp": "src/cli.ts", }, @@ -90,7 +90,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "15.9.67", + "version": "15.10.1", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -101,7 +101,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "15.9.67", + "version": "15.10.1", "bin": { "mnemopi": "src/cli.ts", }, @@ -118,7 +118,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "15.9.67", + "version": "15.10.1", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -126,7 +126,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "15.9.67", + "version": "15.10.1", "bin": { "omp-stats": "./src/index.ts", }, @@ -151,7 +151,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "15.9.67", + "version": "15.10.1", "bin": { "omp-swarm": "src/cli.ts", }, @@ -167,7 +167,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "15.9.67", + "version": "15.10.1", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -208,7 +208,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "15.9.67", + "version": "15.10.1", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "beautiful-mermaid": "catalog:", @@ -248,15 +248,15 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "15.9.67", - "@oh-my-pi/omp-stats": "15.9.67", - "@oh-my-pi/pi-agent-core": "15.9.67", - "@oh-my-pi/pi-ai": "15.9.67", - "@oh-my-pi/pi-coding-agent": "15.9.67", - "@oh-my-pi/pi-mnemopi": "15.9.67", - "@oh-my-pi/pi-natives": "15.9.67", - "@oh-my-pi/pi-tui": "15.9.67", - "@oh-my-pi/pi-utils": "15.9.67", + "@oh-my-pi/hashline": "15.10.1", + "@oh-my-pi/omp-stats": "15.10.1", + "@oh-my-pi/pi-agent-core": "15.10.1", + "@oh-my-pi/pi-ai": "15.10.1", + "@oh-my-pi/pi-coding-agent": "15.10.1", + "@oh-my-pi/pi-mnemopi": "15.10.1", + "@oh-my-pi/pi-natives": "15.10.1", + "@oh-my-pi/pi-tui": "15.10.1", + "@oh-my-pi/pi-utils": "15.10.1", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -395,7 +395,7 @@ "@huggingface/jinja": ["@huggingface/jinja@0.5.9", "", {}, "sha512-uWTG+l3VJRsl7EXxYizuL3P+cCPoc3cRqbWWRcQN0FhejRfbdq0RNhCmbY/YDtnTcz9icdLYuLDjsnz4d8JMuw=="], - "@huggingface/tasks": ["@huggingface/tasks@0.21.6", "", {}, "sha512-XfLE2clF0uHw7kMb6HHMkpyJ+bmu2T0EZ8O1WxbJXIqdRwKRB0RM7Y639Ph3aYj2GjFLJhNo1Lz8y0jn9k10LQ=="], + "@huggingface/tasks": ["@huggingface/tasks@0.21.7", "", {}, "sha512-GuEXszIkir4j/Oywp4hXP+wfwojo/SKWA/omroNkzWWgqUGiOQ5p6HuyXcDOcinYnLQW1WsO8fwdEvtLTZbA4w=="], "@huggingface/tokenizers": ["@huggingface/tokenizers@0.1.3", "", {}, "sha512-8rF/RRT10u+kn7YuUbUg0OF30K8rjTc78aHpxT+qJ1uWSqxT1MHi8+9ltwYfkFYJzT/oS+qw3JVfHtNMGAdqyA=="], @@ -1159,7 +1159,7 @@ "onnxruntime-web": ["onnxruntime-web@1.26.0-dev.20260416-b7804b056c", "", { "dependencies": { "flatbuffers": "^25.1.24", "guid-typescript": "^1.0.9", "long": "^5.2.3", "onnxruntime-common": "1.24.0-dev.20251116-b39e144322", "platform": "^1.3.6", "protobufjs": "^7.2.4" } }, "sha512-MD6Ss4GSpQBo6zqoJzyT9LRbKYs7x/JVN23FT24EcEvlqF4VuzPOeH6X38orZPKHQDbprn7K+SBpu0/mj2CQiw=="], - "openai": ["openai@6.39.1", "", { "peerDependencies": { "ws": "^8.18.0", "zod": "^3.25 || ^4.0" }, "optionalPeers": ["ws", "zod"], "bin": { "openai": "bin/cli" } }, "sha512-z3dO9fEWOXBzlXynVb/xZ/tujzUjFWQWn3C0n0mw6Vo0zJTbEkaN4b2cLWjhJ6haJQx8LlREoafHRl+Gu/Hl+A=="], + "openai": ["openai@6.42.0", "", { "peerDependencies": { "ws": "^8.18.0", "zod": "^3.25 || ^4.0" }, "optionalPeers": ["ws", "zod"] }, "sha512-1WFEt/uXMXOLhYRNkgJWo08Y2YNvNwpVU72K7ibrWgWpNOXd4VojXLbe6SQ4bLiUQ3Y8jz4IiyVkylJCL1DtZg=="], "option": ["option@0.2.4", "", {}, "sha512-pkEqbDyl8ou5cpq+VsnQbe/WlEy5qS7xPzMS1U55OCG9KPvwFD46zDbxQIj3egJSFc3D+XhYOPUzz49zQAVy7A=="], diff --git a/crates/pi-natives/src/keys.rs b/crates/pi-natives/src/keys.rs index ccdb9a935..d132d70e8 100644 --- a/crates/pi-natives/src/keys.rs +++ b/crates/pi-natives/src/keys.rs @@ -70,6 +70,7 @@ const CP_KP_EQUALS: i32 = 57415; const MOD_SHIFT: u32 = 1; const MOD_ALT: u32 = 2; const MOD_CTRL: u32 = 4; +const MOD_SUPER: u32 = 8; const MOD_NUM_LOCK: u32 = 128; /// Event types from Kitty keyboard protocol (flag 2). @@ -457,6 +458,10 @@ fn parse_key_id(key_id: &str) -> Option> { modifier |= MOD_SHIFT; continue; }, + b's' | b'S' if p.eq_ignore_ascii_case("super") => { + modifier |= MOD_SUPER; + continue; + }, b'a' | b'A' if p.eq_ignore_ascii_case("alt") => { modifier |= MOD_ALT; continue; @@ -1378,7 +1383,7 @@ fn parse_functional(bytes: &[u8]) -> Option { fn format_kitty_key(parsed: &ParsedKittySequence) -> Option> { let effective_mod = parsed.modifier & !LOCK_MASK; - if effective_mod & !(MOD_SHIFT | MOD_CTRL | MOD_ALT) != 0 { + if effective_mod & !(MOD_SHIFT | MOD_CTRL | MOD_ALT | MOD_SUPER) != 0 { return None; } let effective_codepoint = @@ -1480,6 +1485,9 @@ fn format_with_mods(mods: u32, key_name: &str) -> String { if mods & MOD_ALT != 0 { result.push_str("alt+"); } + if mods & MOD_SUPER != 0 { + result.push_str("super+"); + } result.push_str(key_name); result } @@ -1585,7 +1593,11 @@ mod tests { #[test] fn parse_key_ignores_kitty_sequences_with_unsupported_modifiers() { - assert_eq!(parse_key_inner(b"\x1b[99;9u", true).as_deref(), None); + // Hyper (16) and meta (32) are kitty modifier bits we do not surface + // because nothing in the editor binds them. Wire mod 17 = mask 16 = hyper. + assert_eq!(parse_key_inner(b"\x1b[99;17u", true).as_deref(), None); + // Wire mod 33 = mask 32 = meta. + assert_eq!(parse_key_inner(b"\x1b[99;33u", true).as_deref(), None); } #[test] @@ -1675,4 +1687,32 @@ mod tests { assert!(matches_key_inner(b"\x1b[109;7u", "ctrl+alt+m", true)); assert!(matches_key_inner(b"\x1b[27;7;109~", "ctrl+alt+m", false)); } + + #[test] + fn super_alt_backspace_matches_ghostty_default() { + // Issue #2064: Ghostty on macOS reports Option+Backspace as kitty + // modifier 11 (wire) = 10 (mask) = super(8)|alt(2). Before super + // support landed, the matcher rejected this entirely. + assert!(matches_key_inner(b"\x1b[127;11u", "super+alt+backspace", true)); + assert!(matches_key_inner(b"\x1b[127;11u", "alt+super+backspace", true)); + assert_eq!(parse_key_inner(b"\x1b[127;11u", true).as_deref(), Some("alt+super+backspace")); + // Plain alt+backspace must still NOT match — the modifier really is super|alt. + assert!(!matches_key_inner(b"\x1b[127;11u", "alt+backspace", true)); + // And plain backspace (mod 0) must still not match a super+alt-modified press. + assert!(!matches_key_inner(b"\x1b[127;11u", "backspace", true)); + // Release events stay ignored: super+alt+backspace release must not match a press. + assert!(!matches_key_inner(b"\x1b[127;11:3u", "super+alt+backspace", true)); + assert_eq!(parse_key_inner(b"\x1b[127;11:3u", true).as_deref(), None); + } + + #[test] + fn super_modifier_parses_for_arbitrary_keys() { + // Cmd+letter on macOS under kitty flag=1+: super(8)+'a'(97) → wire mod 9. + assert!(matches_key_inner(b"\x1b[97;9u", "super+a", true)); + assert_eq!(parse_key_inner(b"\x1b[97;9u", true).as_deref(), Some("super+a")); + // Cmd+Shift+letter: super(8)|shift(1) = 9 mask, wire 10. + assert!(matches_key_inner(b"\x1b[97;10u", "super+shift+a", true)); + assert!(matches_key_inner(b"\x1b[97;10u", "shift+super+a", true)); + assert_eq!(parse_key_inner(b"\x1b[97;10u", true).as_deref(), Some("shift+super+a")); + } } diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index 5017e2ddd..6b9c17c74 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -68,5 +68,5 @@ use napi_derive::napi; /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV15_9_67")] +#[napi(js_name = "__piNativesV15_10_1")] pub const fn pi_natives_version_sentinel() {} diff --git a/crates/pi-shell/src/fixup.rs b/crates/pi-shell/src/fixup.rs index 86b6974d4..f9fe9d70a 100644 --- a/crates/pi-shell/src/fixup.rs +++ b/crates/pi-shell/src/fixup.rs @@ -154,7 +154,12 @@ fn try_strip_head_tail( // span under-reports when its suffix contains unlocated `IoRedirect`s // (e.g. the synthetic `2>&1` inserted by `|&`). let bytes = cmd.as_bytes(); - let last_start = last_loc.start.index; + let Some(last_start) = byte_offset(cmd, last_loc.start.index) else { + return default; + }; + let Some(last_end) = byte_offset(cmd, last_loc.end.index) else { + return default; + }; let Some(head) = cmd.get(..last_start) else { return default; }; @@ -172,7 +177,7 @@ fn try_strip_head_tail( // Reported text starts at the pipe and is right-trimmed. The deletion // range walks back through any leading whitespace so the rewrite is // contiguous. - let stripped_text = cmd[pipe_pos..last_loc.end.index].trim_end().to_owned(); + let stripped_text = cmd[pipe_pos..last_end].trim_end().to_owned(); if stripped_text.is_empty() { return default; } @@ -180,7 +185,7 @@ fn try_strip_head_tail( while delete_start > 0 && matches!(bytes[delete_start - 1], b' ' | b'\t') { delete_start -= 1; } - ranges.push((delete_start, last_loc.end.index)); + ranges.push((delete_start, last_end)); stripped.push(stripped_text); HeadTailOutcome { stripped: true, last_idx: n - 2 } } @@ -253,10 +258,14 @@ fn try_strip_2to1( let Some(name_loc) = name_word.loc.as_ref() else { return; }; - let mut anchor = name_loc.end.index; + let Some(mut anchor) = byte_offset(cmd, name_loc.end.index) else { + return; + }; for item in &suffix.0 { - if let Some(loc) = item.location() { - anchor = anchor.max(loc.end.index); + if let Some(loc) = item.location() + && let Some(end) = byte_offset(cmd, loc.end.index) + { + anchor = anchor.max(end); } } let bytes = cmd.as_bytes(); @@ -283,6 +292,28 @@ fn try_strip_2to1( stripped.push("2>&1".to_owned()); } +/// Translate a `brush-parser` source-position index into a byte offset in +/// `cmd`. The parser counts positions in Unicode scalars (one increment per +/// `char`; see `tokenizer::next_char`), but we slice `cmd` — a `&str` — by +/// byte index, so the two diverge as soon as the command contains any +/// multi-byte UTF-8 (e.g. a `✓`/`×` literal inside a `grep` pattern). Without +/// this conversion the head/tail and `2>&1` strips cut at the wrong place, +/// corrupting the command (notably orphaning a closing quote). +/// +/// Returns `None` only when `char_idx` is past the end of the input. The +/// end-of-input position (`char_idx == cmd.chars().count()`) maps to +/// `cmd.len()`. +fn byte_offset(cmd: &str, char_idx: usize) -> Option { + let mut count = 0usize; + for (byte, _) in cmd.char_indices() { + if count == char_idx { + return Some(byte); + } + count += 1; + } + (count == char_idx).then_some(cmd.len()) +} + fn is_stderr_to_stdout(io: &IoRedirect) -> bool { let IoRedirect::File(Some(2), IoFileRedirectKind::DuplicateOutput, target) = io else { return false; @@ -380,6 +411,45 @@ mod tests { } } + #[test] + fn strips_with_multibyte_content() { + // brush-parser reports char-indexed positions; multi-byte UTF-8 before + // the trailing `| tail` must not shift the byte-level cut. Regression + // for a corrupted command that orphaned the grep pattern's closing + // quote (`… |✓|×-80`) and broke later re-parsing. + let cases: &[(&str, &str, &[&str])] = &[ + // Mirrors the real bug: `2>&1` sits mid-pipeline (grep becomes the + // effective tail) so only `| tail -80` is stripped, leaving the + // quoted grep pattern intact. + ( + "xcodebuild 2>&1 | grep -E \"a|✓|×|b\" | tail -80", + "xcodebuild 2>&1 | grep -E \"a|✓|×|b\"", + &["| tail -80"], + ), + ("echo ✓ | head -3", "echo ✓", &["| head -3"]), + ("printf '日本語' | tail -n 5", "printf '日本語'", &["| tail -n 5"]), + ]; + for (input, want_cmd, want_stripped) in cases { + let (cmd, stripped) = run(input); + assert_eq!(cmd, *want_cmd, "input: {input:?}"); + assert_eq!(stripped, *want_stripped, "input: {input:?}"); + // The rewrite must remain valid shell (no orphaned quote). + let options = ParserOptions::default(); + let source_info = SourceInfo::default(); + let mut reader = BufReader::new(cmd.as_bytes()); + let mut parser = Parser::new(&mut reader, &options, &source_info); + assert!(parser.parse_program().is_ok(), "rewrite not re-parseable: {cmd:?}"); + } + } + + #[test] + fn strips_2to1_after_multibyte() { + // Multi-byte before a trailing `2>&1` must not misplace the strip. + let (cmd, stripped) = run("echo ✓ × 2>&1"); + assert_eq!(cmd, "echo ✓ ×"); + assert_eq!(stripped, vec!["2>&1"]); + } + #[test] fn strips_redundant_2to1() { let cases: &[(&str, &str, &[&str])] = &[ diff --git a/docs/environment-variables.md b/docs/environment-variables.md index 60b1da7aa..7d2bc9b75 100644 --- a/docs/environment-variables.md +++ b/docs/environment-variables.md @@ -81,13 +81,13 @@ These are consumed via `getEnvApiKey()` (`packages/ai/src/stream.ts`) unless not | `WAFER_SERVERLESS_API_KEY` | Wafer Serverless auth | Using `wafer-serverless` provider | Pay-as-you-go Wafer SKU; validated against `https://pass.wafer.ai/v1/models` | | `GITLAB_TOKEN` | GitLab Duo auth | Using `gitlab-duo` provider | | -### GitHub/Copilot token chains +### GitHub/Copilot tokens -| Variable | Used for | Chain | -| ---------------------- | ------------------------------------------------ | ---------------------------------------------------- | -| `COPILOT_GITHUB_TOKEN` | GitHub Copilot provider auth | `COPILOT_GITHUB_TOKEN` → `GH_TOKEN` → `GITHUB_TOKEN` | -| `GH_TOKEN` | Copilot fallback; GitHub API auth in web scraper | In web scraper: `GITHUB_TOKEN` → `GH_TOKEN` | -| `GITHUB_TOKEN` | Copilot fallback; GitHub API auth in web scraper | In web scraper: checked before `GH_TOKEN` | +| Variable | Used for | Notes | +| ---------------------- | ------------------------------------------------ | ------------------------------------------ | +| `COPILOT_GITHUB_TOKEN` | GitHub Copilot provider auth | Generic GitHub tokens are not used here | +| `GH_TOKEN` | GitHub API auth in web scraper | Web scraper fallback after `GITHUB_TOKEN` | +| `GITHUB_TOKEN` | GitHub API auth in web scraper | Web scraper checks this before `GH_TOKEN` | ### Auth broker / auth gateway (remote credential vault) diff --git a/docs/keybindings.md b/docs/keybindings.md index 9fa872698..dfc881bbe 100644 --- a/docs/keybindings.md +++ b/docs/keybindings.md @@ -44,4 +44,6 @@ app.stt.toggle: [] On Windows Terminal, `Ctrl+V` may be handled by the terminal paste command before `omp` sees it; use the `Alt+V` fallback when clipboard image paste appears to do nothing. Windows Terminal also swallows `Ctrl+Enter`, so the follow-up shortcut also binds `Ctrl+Q` — the same chord GitHub Copilot CLI uses. If your existing `keybindings.yml` already assigns `Ctrl+Q` to another action, that user remap wins and follow-up keeps `Ctrl+Enter` unless you explicitly bind `app.message.followUp`. +Terminals that implement OSC 5522 enhanced paste can send clipboard MIME data directly to `omp`; image pastes are attached as `[Image #N]`, while text/plain paste events keep normal paste behavior. When OSC 5522 is unavailable, bracketed paste still handles text, and a pasted single image-file path is loaded as an image when the file is readable from the `omp` host. + Older unqualified action names are migrated when `keybindings.yml` is loaded, but new docs and new configs should use the namespaced action IDs above. Existing `keybindings.json` files are still accepted and migrated to `keybindings.yml`; `keybindings.yaml` is also accepted. diff --git a/docs/models.md b/docs/models.md index daa9e7a2a..a1ca6d89f 100644 --- a/docs/models.md +++ b/docs/models.md @@ -557,9 +557,9 @@ For `anthropic-messages` models the runtime uses a separate `AnthropicCompat` sh ### Strict tool schemas (`disableStrictTools`) -Anthropic's API supports a `strict` field on tool definitions that forces the model to always follow the provided schema exactly. This is enabled by default for all `anthropic-messages` providers because it guarantees schema conformance in agentic systems. +Anthropic's API supports a `strict` field on tool definitions that forces the model to always follow the provided schema exactly. OMP enables it by default for a small allowlist of high-frequency built-in `anthropic-messages` tools (`bash`, `python`, `edit`, and `find`) whose schemas fit Anthropic's strict grammar limits; other tools still send normalized schemas but omit `strict`. -Third-party providers that front the Anthropic API (AWS Bedrock, Azure, self-hosted proxies) do not always implement this field and will reject requests that include it. Set `disableStrictTools: true` at the provider level to opt out: +Third-party providers that front the Anthropic API (AWS Bedrock, Azure, self-hosted proxies) do not always implement this field and will reject requests that include it. Set `disableStrictTools: true` at the provider level to opt out of strict mode for the allowlisted tools: ```yaml providers: @@ -581,7 +581,7 @@ providers: cacheWrite: 3.75 ``` -`disableStrictTools` is a provider-level flag that applies to all models in the provider. +`disableStrictTools` is a provider-level flag that applies to all models in the provider. It disables the Anthropic `strict` marker only for tools that OMP would otherwise mark strict; it does not change runtime tool argument validation. OMP can automatically retry without strict tools after Anthropic reports a strict-grammar-too-large error before the first streamed token, but proxies that reject the `strict` field for other reasons should set this flag explicitly. Tool schemas going on the wire are normalized by the unified flow in `packages/ai/src/utils/schema/normalize.ts` (Google/CCA/MCP dispatchers diff --git a/docs/porting-from-pi-mono.md b/docs/porting-from-pi-mono.md index 3bfb97bc1..23e4dfcd1 100644 --- a/docs/porting-from-pi-mono.md +++ b/docs/porting-from-pi-mono.md @@ -47,6 +47,9 @@ Upstream uses different package scopes. Replace them consistently. - `@mariozechner/pi-agent-core` → `@oh-my-pi/pi-agent-core` - `@mariozechner/pi-tui` → `@oh-my-pi/pi-tui` - `@mariozechner/pi-ai` → `@oh-my-pi/pi-ai` + - `@mariozechner/pi-utils` → `@oh-my-pi/pi-utils` +- Some upstream packages publish under the `@earendil-works/*` scope instead of `@mariozechner/*`. Map it the same way (`@earendil-works/pi-coding-agent` → `@oh-my-pi/pi-coding-agent`, and so on). +- The bare `typebox` package is not an `@oh-my-pi/*` scope; do not rewrite it as one. See the Extensions divergence in section 15 for how tool-parameter schemas map. ## 4) Use Bun APIs where they improve on Node @@ -353,10 +356,13 @@ Our fork has architectural decisions that differ from upstream. **Do not port th ### Extensions -| Upstream | Our Fork | -| ----------------------------- | ------------------------------------------------- | -| `jiti` for TypeScript loading | Native Bun `import()` | -| `pkg.pi` manifest field | `pkg.omp` preferred; fallback to `pkg.pi` remains | +| Upstream | Our Fork | +| ---------------------------------------------------------------- | ------------------------------------------------------------------------------------------------- | +| `jiti` for TypeScript loading | Native Bun `import()` | +| `pkg.pi` manifest field | `pkg.omp` preferred; fallback to `pkg.pi` remains | +| `StringEnum` from `pi-ai` | `Type.Enum` from the `pi.typebox` shim (or author the schema with `pi.zod`); `pi-ai` no longer exports `StringEnum` | +| `formatSize` from `pi-coding-agent` | `formatBytes` from `@oh-my-pi/pi-utils` | +| `DefaultResourceLoader` / `DefaultPackageManager` / `SettingsManager` / `createEventBus` | Capability-based discovery (`loadCapability(...)`) plus the `Settings` singleton and `EventBus` | ### Skip These Upstream Features diff --git a/docs/session-operations-export-share-fork-resume.md b/docs/session-operations-export-share-fork-resume.md index 0173137c1..a82ef075d 100644 --- a/docs/session-operations-export-share-fork-resume.md +++ b/docs/session-operations-export-share-fork-resume.md @@ -1,4 +1,4 @@ -# Session Operations: export, dump, share, fork, resume/continue +# Session Operations: export, dump, share, fresh, fork, resume/continue This document describes operator-visible behavior for session export/share/fork/resume operations as currently implemented. @@ -19,12 +19,13 @@ This document describes operator-visible behavior for session export/share/fork/ | `/export [path]` | Interactive slash command | No | No | HTML file | | `--export [outputPath]` | CLI startup fast-path | No runtime session mutation | No active session; reads target file | HTML file | | `/share` | Interactive slash command | No | No | Temp HTML + share URL/gist | +| `/fresh` | Interactive slash command | Yes (provider-facing in-memory id/state only) | No; keeps current session file/header | None | | `/fork` | Interactive slash command | Yes (active session identity changes) | Creates new session file and switches current session to it (persistent mode only) | Copies artifact directory to new session namespace when present | | `--fork ` | CLI startup | Yes after session creation | Creates a new session fork from the selected source into current cwd/session dir | None | | `/resume` | Interactive slash command | Yes (active in-memory state replaced) | Switches to selected existing session file | None | | `--resume` | CLI startup picker | Yes after session creation | Opens selected existing session file | None | -| `--resume ` | CLI startup | Yes after session creation | Opens existing session; global cross-project match can fork into current project | None | -| `--continue` | CLI startup | Yes after session creation | Opens terminal breadcrumb or most-recent session; creates new one if none exists | None | +| `--resume ` | CLI startup | Yes after session creation | Opens existing session; global cross-project match re-roots (moved dir) or forks into current project | None | +| `--continue` | CLI startup | Yes after session creation | Opens terminal breadcrumb (re-roots it if its dir was moved) or most-recent session; creates new one if none exists | None | ## Export and dump @@ -214,19 +215,23 @@ Notes: Cross-project id match behavior: -- If matched session cwd differs from current cwd, CLI asks: - - `Session found in different project ... Fork into current directory? [y/N]` -- On yes: `SessionManager.forkFrom(match.path, cwd, sessionDir)` creates a new local forked file. -- On no/non-TTY default: command errors. +- If matched session cwd differs from current cwd, behavior depends on whether the matched session's recorded directory still exists: + - **Directory gone (moved/renamed, e.g. `git worktree move`)**: CLI asks `Session's directory no longer exists (...). Move (re-root) it into the current directory? [Y/n]`. + - On yes (default): `SessionManager.open(match.path)` then `manager.moveTo(cwd)` re-roots the existing session into the current directory (no duplicate file). + - On no: command cancels (returns no session). On non-TTY: command errors. + - **Directory still exists (genuinely different project)**: CLI asks `Session found in different project ... Fork into current directory? [y/N]`. + - On yes: `SessionManager.forkFrom(match.path, cwd, sessionDir)` creates a new local forked file. + - On no: command cancels. On non-TTY: command errors. ## CLI `--continue` `SessionManager.continueRecent(cwd, sessionDir)`: 1. Resolves session dir for current cwd. -2. Reads terminal-scoped breadcrumb first. -3. Falls back to most recently modified session file. -4. Opens found session; if none exists, creates new session. +2. Reads the terminal-scoped breadcrumb. +3. If the breadcrumb points at a session recorded under a different cwd whose directory no longer exists (moved/renamed) **and** the current directory has no sessions of its own, re-roots that session into the current directory via `moveTo` instead of starting fresh. +4. Otherwise, if the breadcrumb's cwd matches the current cwd, uses the breadcrumb session; else falls back to the most recently modified session file. +5. Opens the found session; if none exists, creates a new session. This is startup-only behavior; there is no interactive `/continue` slash command. diff --git a/docs/theme.md b/docs/theme.md index 738ab3b5b..36be98c4e 100644 --- a/docs/theme.md +++ b/docs/theme.md @@ -85,7 +85,7 @@ If omitted, export code derives defaults from resolved theme colors. - `symbols.preset` sets a theme-level default symbol set. - `symbols.overrides` can override individual `SymbolKey` values. -- `symbols.spinnerFrames` overrides the loading spinner frames. Accepts either a flat `string[]` (applied to both spinner types) or an object `{ "status"?: string[], "activity"?: string[] }` to override each type independently. Any type not specified falls back to the symbol preset's default frames. `status` drives the ~12.5fps spinner used by loaders and tool-execution indicators; `activity` drives the ~60fps spinner used by markdown progress bars and similar high-frequency UI. +- `symbols.spinnerFrames` overrides the loading spinner frames. Accepts either a flat `string[]` (applied to both spinner types) or an object `{ "status"?: string[], "activity"?: string[] }` to override each type independently. Any type not specified falls back to the symbol preset's default frames. `status` drives the ~12.5fps spinner used by loaders and tool-execution indicators; `activity` drives the ~30fps spinner used by markdown progress bars and similar high-frequency UI. Runtime precedence: diff --git a/docs/tools/web_search.md b/docs/tools/web_search.md index 5b3aa380b..c62df385d 100644 --- a/docs/tools/web_search.md +++ b/docs/tools/web_search.md @@ -161,9 +161,9 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - Output: `sources`, `requestId`. - **Kagi** — `packages/coding-agent/src/web/search/providers/kagi.ts`, `packages/coding-agent/src/web/kagi.ts` - Availability: env or `agent.db` credential for `kagi`. - - Querying: GET `https://kagi.com/api/v0/search?q=&limit=` with `Authorization: Bot `. + - Querying: POST `https://kagi.com/api/v1/search` with `Authorization: Bearer ` and JSON body `{ query, workflow: "search", limit, filters?: { after } }`. `recency` maps to `filters.after` as a UTC `YYYY-MM-DD` string (`day`/`week`/`month`/`year`). - `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..40`, default `10`. - - Output: `sources`, `relatedQuestions`, `requestId`. + - Output: `sources` (concatenated `data.search` + `data.video` + `data.news` + `data.infobox`, with video/news/infobox results tagged in the title), `relatedQuestions` (`data.adjacent_question` + `data.related_search` `props.question`), `answer` (`data.direct_answer[0].snippet ?? title`), `requestId` (`meta.trace`). - **Synthetic** — `packages/coding-agent/src/web/search/providers/synthetic.ts` - Availability: env or `agent.db` credential for `synthetic`. - Querying: POST `https://api.synthetic.new/v2/search` with `{ query }`. diff --git a/docs/tui-core-renderer.md b/docs/tui-core-renderer.md index dd984a830..ba9705e8c 100644 --- a/docs/tui-core-renderer.md +++ b/docs/tui-core-renderer.md @@ -78,7 +78,7 @@ the bytes written and the state update. All state flows through a single | `sessionReplace` | clear viewport **+ ED3** (outside multiplexers) | caller forced `{ clearScrollback: true }` (switch/branch/reload/resume) | | `historyRebuild` | clear viewport **+ ED3** (outside multiplexers) | geometry change rewrapped history, or a proven-at-tail rebuild | | `overlayRebuild` | rebuild viewport with overlay composite | overlay visibility changed | -| `liveRegionPinned` | relative moves + per-line `\x1b[2K` + `\r\n` | foreground streaming on an ED3-risk host, commit-as-you-go | +| `liveRegionPinned` | relative moves + per-row rewrite/suffix-clear + `\r\n` | foreground streaming on an ED3-risk host, commit-as-you-go | | `viewportRepaint` | rewrite the visible viewport in place (optional `appendFrom` tail first) | safe non-destructive repaint | | `deferredShrink` | padded viewport repaint, history left dirty | bottom-anchored shrink, viewport unobservable | | `deferredMutation` | **zero bytes**, history left dirty | row-reindexing edit while possibly scrolled | @@ -245,9 +245,16 @@ parameterized over `(env, platform)` so they are unit-testable: VTE, iTerm2, Apple Terminal, GNOME Terminal, Ptyxis, xfce4-terminal), Linux truecolor, **and every other unknown POSIX terminal**. The default is *risky* on purpose. -- `shouldEnableSynchronizedOutputByDefault(env, platform, id)` → DEC 2026 on by - default only for kitty/ghostty/wezterm/iterm2, off for win32 / SSH / - multiplexers / VTE-family. Layered with a runtime DECRQM auto-disable. +- `shouldEnableSynchronizedOutputByDefault(env, id)` → DEC 2026 default. Precedence: + user opt-out (`PI_NO_SYNC_OUTPUT`/`PI_TUI_SYNC_OUTPUT=0`) → user force-on + (`PI_FORCE_SYNC_OUTPUT=1`/`PI_TUI_SYNC_OUTPUT=1`) → `TERM_FEATURES` advertises + `Sy` → `WT_SESSION` (WT/WSL) → known direct terminals + (kitty/ghostty/wezterm/iterm2/alacritty/vscode; SSH passes through) → off for + risky multiplexers and everything else (VTE-family, GNU screen, Apple Terminal, + legacy conhost, unknown). Reconciled at runtime by the DECRQM mode-2026 report: + a positive report **enables** sync (upgrading default-off muxes like + zellij/tmux-master), a negative one disables it; a user override still wins. + `synchronizedOutputUserOverride(env)` is the shared opt-out/force resolver. - `detectRectangularSgrSupport(id, env)` → DECCARA fills: **kitty only** (ghostty does not implement the SGR-background extension), off in multiplexers and under `PI_NO_DECCARA`. diff --git a/docs/tui-runtime-internals.md b/docs/tui-runtime-internals.md index c4fb02030..79df9186d 100644 --- a/docs/tui-runtime-internals.md +++ b/docs/tui-runtime-internals.md @@ -56,7 +56,7 @@ A forced render (`requestRender(true)`) queues a viewport repaint or explicit se 4. Creates a `StdinBuffer` to split partial escape chunks into complete sequences. 5. Queries Kitty keyboard protocol support (`CSI ? u`), then enables protocol flags if supported; otherwise enables modifyOtherKeys fallback after a short timeout. 6. Queries OSC 11 background color and Mode 2031 appearance notifications for dark/light theme detection. -7. Queries OSC 99 notification capabilities and Kitty temp-file graphics support. +7. Queries OSC 99 notification capabilities. 8. Starts periodic OSC 11 polling only where safe, then probes DEC private modes 2026/2048/2031 via DECRQM. `StdinBuffer` behavior: @@ -199,18 +199,6 @@ Escape exits inactive mode by clearing editor text and restoring border color; w 2. Stops TUI before suspend. 3. Sends `SIGTSTP` to process group. -### Background mode (`/background` or `/bg`) - -`handleBackgroundCommand()`: - -- Rejects when idle. -- Switches tool UI context to non-interactive (`hasUI=false`) so interactive UI tools fail fast. -- Stops loaders/status line and unsubscribes foreground event handler. -- Subscribes background event handler (primarily waits for `agent_end`). -- Stops TUI and sends `SIGTSTP` (POSIX job control path). - -On `agent_end` in background with no queued work, controller sends completion notification and shuts down. - ## Cancellation paths Primary cancellation inputs: diff --git a/docs/tui.md b/docs/tui.md index a62b5f0fe..cba829fbe 100644 --- a/docs/tui.md +++ b/docs/tui.md @@ -28,7 +28,7 @@ export interface Component { render(width: number): string[]; handleInput?(data: string): void; wantsKeyRelease?: boolean; - invalidate(): void; + invalidate?(): void; } ``` diff --git a/package.json b/package.json index a936a6642..435a43242 100644 --- a/package.json +++ b/package.json @@ -20,15 +20,15 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "15.9.67", - "@oh-my-pi/omp-stats": "15.9.67", - "@oh-my-pi/pi-agent-core": "15.9.67", - "@oh-my-pi/pi-ai": "15.9.67", - "@oh-my-pi/pi-coding-agent": "15.9.67", - "@oh-my-pi/pi-mnemopi": "15.9.67", - "@oh-my-pi/pi-natives": "15.9.67", - "@oh-my-pi/pi-tui": "15.9.67", - "@oh-my-pi/pi-utils": "15.9.67", + "@oh-my-pi/hashline": "15.10.1", + "@oh-my-pi/omp-stats": "15.10.1", + "@oh-my-pi/pi-agent-core": "15.10.1", + "@oh-my-pi/pi-ai": "15.10.1", + "@oh-my-pi/pi-coding-agent": "15.10.1", + "@oh-my-pi/pi-mnemopi": "15.10.1", + "@oh-my-pi/pi-natives": "15.10.1", + "@oh-my-pi/pi-tui": "15.10.1", + "@oh-my-pi/pi-utils": "15.10.1", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -85,8 +85,9 @@ }, "overrides": {}, "scripts": { - "install:dev": "bun install && bun --cwd=packages/coding-agent link && bun --cwd=packages/ai link", + "install:dev": "bun install && bun --cwd=packages/coding-agent link && ln -sfn \"$(pwd)/packages/coding-agent/scripts/dev-launch\" \"$(bun pm -g bin)/omp\"", "dev": "bun --cwd=packages/coding-agent src/cli.ts", + "dev:timing": "PI_TIMING=x bun --cwd=packages/coding-agent --preload ../utils/src/module-timer.ts src/cli.ts", "stats": "bun --cwd=packages/coding-agent src/cli.ts stats", "claude:trace": "bun scripts/claude-trace.ts", "build": "bun run --workspaces --if-present build", diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 324aac24b..5e4dd0b4f 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,24 @@ ## [Unreleased] +## [15.10.1] - 2026-06-07 + +### Added + +- Added optional `promptCacheKey` support to `AgentOptions` and `Agent` via a new `promptCacheKey` property so providers can receive a caller-provided prompt cache key +- Added optional `ApiKeyResolveContext` parameter to `getApiKey` in `AgentOptions` and `AgentLoopConfig` so key resolvers can receive retry context + +### Changed + +- Enabled streaming API calls to re-resolve credentials through the `getApiKey` callback when retries occur after authentication-related errors +- `Agent.abort(reason?)` now forwards `reason` to the underlying `AbortController`, and the synthesized aborted assistant message carries that reason on `errorMessage` (string or non-`AbortError` `Error` message) instead of always defaulting to `"Request was aborted"`. Bare `abort()` is unchanged. + +### Fixed + +- Fixed handling of short-lived API keys so that expired tokens are retried with a refreshed value during 401/usage-limit failures +- Ensured fallback API key resolution uses the initially configured static `apiKey` when `getApiKey` is present +- Wrapped oneshot LLM completions (`instrumentedCompleteSimple`: handoff, compaction/branch summaries) in an `EventLoopKeepalive`. These run outside the agent `#runLoop`, so without the keepalive Bun's event loop stopped servicing timers while parked on the completion promise — freezing host spinners (e.g. the `/handoff` loader) until an unrelated terminal resize poked the loop into rendering again. + ## [15.9.5] - 2026-06-05 ### Fixed diff --git a/packages/agent/package.json b/packages/agent/package.json index eccf73aa8..6d256bd40 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "15.9.67", + "version": "15.10.1", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index 565420527..c4bddb3cd 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -3,6 +3,7 @@ * Transforms to Message[] only at the LLM call boundary. */ import { + type ApiKeyResolveContext, type AssistantMessage, type AssistantMessageEvent, type Context, @@ -727,8 +728,9 @@ async function streamAssistantResponse( // Resolve API key (important for expiring tokens) — do this before resolving // metadata so that the session-sticky credential recorded by getApiKey is // visible to metadataResolver (e.g. for the correct account_uuid in metadata.user_id). + const staticApiKey = typeof config.apiKey === "string" ? config.apiKey : undefined; const resolvedApiKey = - (config.getApiKey ? await config.getApiKey(config.model.provider) : undefined) || config.apiKey; + (config.getApiKey ? await config.getApiKey(config.model.provider) : undefined) || staticApiKey; // Re-resolve metadata after credential selection so the per-request value // reflects the credential actually used, not the snapshot from AgentLoopConfig construction. @@ -798,7 +800,19 @@ async function streamAssistantResponse( return await runInActiveSpan(chatSpan, async () => { const response = await streamFunction(config.model, llmContext, { ...config, - apiKey: resolvedApiKey, + // Hand streamSimple a resolver so its central auth-retry policy can + // re-resolve on 401 / usage-limit: the initial step reuses the key + // already resolved above (which set the session-sticky credential + // feeding metadataResolver), and retry steps forward the a/b/c ctx + // to config.getApiKey (force-refresh, then rotate). With no + // getApiKey hook the caller's own apiKey (string or resolver) flows + // through unchanged. + apiKey: config.getApiKey + ? (ctx: ApiKeyResolveContext) => + ctx.error === undefined + ? resolvedApiKey + : Promise.resolve(config.getApiKey!(config.model.provider, ctx)) + : config.apiKey, metadata: resolvedMetadata, toolChoice: effectiveToolChoice, reasoning: effectiveReasoning, @@ -839,7 +853,14 @@ async function streamAssistantResponse( let detachAbortListener: (() => void) | undefined; if (requestSignal) { if (requestSignal.aborted) { - const aborted = emitAbortedAssistantMessage(partialMessage, addedPartial, context, config, stream); + const aborted = emitAbortedAssistantMessage( + partialMessage, + addedPartial, + context, + config, + stream, + requestSignal, + ); await finishChat(aborted); return aborted; } @@ -861,7 +882,14 @@ async function streamAssistantResponse( if (capped) return capped; } responseIterator.return?.()?.catch(() => {}); - const aborted = emitAbortedAssistantMessage(partialMessage, addedPartial, context, config, stream); + const aborted = emitAbortedAssistantMessage( + partialMessage, + addedPartial, + context, + config, + stream, + requestSignal, + ); await finishChat(aborted); return aborted; } @@ -874,7 +902,14 @@ async function streamAssistantResponse( const capped = await finishCappedAssistantMessage(); if (capped) return capped; } - const aborted = emitAbortedAssistantMessage(partialMessage, addedPartial, context, config, stream); + const aborted = emitAbortedAssistantMessage( + partialMessage, + addedPartial, + context, + config, + stream, + requestSignal, + ); await finishChat(aborted); return aborted; } @@ -982,14 +1017,30 @@ async function streamAssistantResponse( } } +/** Resolve the human-readable reason an abort carried. A caller that aborts via + * `AbortController.abort(reason)` with a string or a non-`AbortError` `Error` + * (e.g. the coding agent's user-interrupt label) gets that text surfaced on the + * synthesized assistant message's `errorMessage`; a bare `abort()` (whose + * `signal.reason` is the default `AbortError` `DOMException`) falls back to the + * generic sentinel that downstream renderers treat as "no specific reason". */ +export function abortReasonText(signal: AbortSignal | undefined): string { + const reason = signal?.reason; + if (typeof reason === "string" && reason.trim().length > 0) return reason; + if (reason instanceof Error && reason.name !== "AbortError" && reason.message.trim().length > 0) { + return reason.message; + } + return "Request was aborted"; +} + function emitAbortedAssistantMessage( partialMessage: AssistantMessage | null, addedPartial: boolean, context: AgentContext, config: AgentLoopConfig, stream: EventStream, + requestSignal: AbortSignal | undefined, ): AssistantMessage { - const errorMessage = "Request was aborted"; + const errorMessage = abortReasonText(requestSignal); const abortedMessage: AssistantMessage = partialMessage ? { ...partialMessage, stopReason: "aborted", errorMessage } : { diff --git a/packages/agent/src/agent.ts b/packages/agent/src/agent.ts index 1b3c57f46..99496de45 100644 --- a/packages/agent/src/agent.ts +++ b/packages/agent/src/agent.ts @@ -3,6 +3,7 @@ */ import { isPromise } from "node:util/types"; import { + type ApiKeyResolveContext, type AssistantMessage, type AssistantMessageEvent, type CursorExecHandlers, @@ -21,7 +22,7 @@ import { type ToolChoice, type ToolResultMessage, } from "@oh-my-pi/pi-ai"; -import { agentLoop, agentLoopContinue } from "./agent-loop"; +import { abortReasonText, agentLoop, agentLoopContinue } from "./agent-loop"; import type { AppendOnlyContextManager } from "./append-only-context"; import type { HarmonyAuditEvent } from "./harmony-leak"; import type { @@ -132,6 +133,11 @@ export interface AgentOptions { * Used by providers that support session-based caching (e.g., OpenAI Codex). */ sessionId?: string; + /** + * Optional prompt cache key forwarded to LLM providers. + * When omitted, providers may fall back to sessionId. + */ + promptCacheKey?: string; /** * Shared provider state map for session-scoped transport/session caches. */ @@ -141,7 +147,7 @@ export interface AgentOptions { * Resolves an API key dynamically for each LLM call. * Useful for expiring tokens (e.g., GitHub Copilot OAuth). */ - getApiKey?: (provider: string) => Promise | string | undefined; + getApiKey?: (provider: string, ctx?: ApiKeyResolveContext) => Promise | string | undefined; /** * Inspect or replace provider payloads before they are sent. @@ -283,6 +289,7 @@ export class Agent { #interruptMode: "immediate" | "wait"; #maxToolCallsPerTurn?: number; #sessionId?: string; + #promptCacheKey?: string; #metadata?: Record; #metadataResolver?: (provider: string) => Record | undefined; #providerSessionState?: Map; @@ -319,7 +326,7 @@ export class Agent { #cursorToolResultBuffer: CursorToolResultEntry[] = []; streamFn: StreamFn; - getApiKey?: (provider: string) => Promise | string | undefined; + getApiKey?: (provider: string, ctx?: ApiKeyResolveContext) => Promise | string | undefined; /** * Hook invoked after tool arguments are validated and before execution. * Reassign at any time to swap the implementation (e.g. on extension reload). @@ -344,6 +351,7 @@ export class Agent { this.#maxToolCallsPerTurn = opts.maxToolCallsPerTurn; this.streamFn = opts.streamFn || streamSimple; this.#sessionId = opts.sessionId; + this.#promptCacheKey = opts.promptCacheKey; this.#providerSessionState = opts.providerSessionState; this.#thinkingBudgets = opts.thinkingBudgets; this.#temperature = opts.temperature; @@ -390,6 +398,20 @@ export class Agent { this.#sessionId = value; } + /** + * Get the prompt cache key forwarded to providers. + */ + get promptCacheKey(): string | undefined { + return this.#promptCacheKey; + } + + /** + * Set the prompt cache key forwarded to providers. + */ + set promptCacheKey(value: string | undefined) { + this.#promptCacheKey = value; + } + /** * Static metadata forwarded to every API request when no resolver is installed * (e.g. `metadata.user_id` for Anthropic session attribution). Setting this @@ -768,8 +790,8 @@ export class Agent { this.#state.messages.length = 0; } - abort() { - this.#abortController?.abort(); + abort(reason?: unknown) { + this.#abortController?.abort(reason); } waitForIdle(): Promise { @@ -936,6 +958,7 @@ export class Agent { interruptMode: this.#interruptMode, maxToolCallsPerTurn: this.#maxToolCallsPerTurn, sessionId: this.#sessionId, + promptCacheKey: this.#promptCacheKey, metadata: this.#metadataResolver ? undefined : this.#metadata, metadataResolver: this.#metadataResolver, providerSessionState: this.#providerSessionState, @@ -1053,8 +1076,12 @@ export class Agent { } } } catch (err) { - const errorMessage = err instanceof Error ? err.message : String(err); const stoppedForAbort = this.#abortController?.signal.aborted === true; + const errorMessage = stoppedForAbort + ? abortReasonText(this.#abortController?.signal) + : err instanceof Error + ? err.message + : String(err); const shouldEmitVisibleOutputBlockedError = !stoppedForAbort && isAnthropicOutputBlockedError(errorMessage); const assistantPartial = partial?.role === "assistant" ? partial : undefined; const hadAssistantStart = assistantPartial !== undefined; diff --git a/packages/agent/src/telemetry.ts b/packages/agent/src/telemetry.ts index 5cd99a135..8655ccb6c 100644 --- a/packages/agent/src/telemetry.ts +++ b/packages/agent/src/telemetry.ts @@ -50,6 +50,7 @@ import { } from "@opentelemetry/api"; import { AgentRunCollector, type AgentRunCoverage, type AgentRunSummary, type ToolStatus } from "./run-collector"; import type { AgentTool } from "./types"; +import { EventLoopKeepalive } from "./utils/yield"; /** Default tracer name. Override via {@link AgentTelemetryConfig.tracerName}. */ export const DEFAULT_TRACER_NAME = "@oh-my-pi/pi-agent-core"; @@ -1629,6 +1630,13 @@ export async function instrumentedCompleteSimple( options: SimpleStreamOptions, span: InstrumentedChatSpanOptions, ): Promise { + // Oneshot LLM calls (handoff, compaction/branch summaries) run outside the + // agent `#runLoop`, which is where the EventLoopKeepalive normally lives. + // Without it, Bun's JSC loop stops servicing timers while parked on the + // long-lived completion promise, freezing any host spinner (e.g. the + // `/handoff` Loader) until an unrelated I/O event (a terminal resize) + // pokes the loop. Keep the loop healthy for the duration of the call. + using _keepalive = new EventLoopKeepalive(); const { telemetry, parent, oneshotKind } = span; const stepNumber = span.stepNumber ?? -1; const reasoning = options.reasoning; diff --git a/packages/agent/src/types.ts b/packages/agent/src/types.ts index 7ca6a37ec..8ecc1458c 100644 --- a/packages/agent/src/types.ts +++ b/packages/agent/src/types.ts @@ -1,4 +1,5 @@ import type { + ApiKeyResolveContext, AssistantMessage, AssistantMessageEvent, AssistantMessageEventStream, @@ -112,7 +113,7 @@ export interface AgentLoopConfig extends SimpleStreamOptions { * Useful for short-lived OAuth tokens (e.g., GitHub Copilot) that may expire * during long-running tool execution phases. */ - getApiKey?: (provider: string) => Promise | string | undefined; + getApiKey?: (provider: string, ctx?: ApiKeyResolveContext) => Promise | string | undefined; /** * Returns steering messages to inject into the conversation mid-run. diff --git a/packages/agent/test/agent-loop.test.ts b/packages/agent/test/agent-loop.test.ts index d0913c609..55a9ac7a8 100644 --- a/packages/agent/test/agent-loop.test.ts +++ b/packages/agent/test/agent-loop.test.ts @@ -157,6 +157,35 @@ describe("agentLoop with AgentMessage", () => { expect(events.map(event => event.type)).toContain("agent_end"); }); + it("surfaces a custom abort reason on the synthesized aborted message", async () => { + const context: AgentContext = { + systemPrompt: ["You are helpful."], + messages: [], + tools: [], + }; + const mock = createMockModel(); + const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter, maxToolCallsPerTurn: 8 }; + const controller = new AbortController(); + const streamFn = () => new AssistantMessageEventStream(); + + const stream = agentLoop([createUserMessage("Hello")], context, config, controller.signal, streamFn); + // Abort with a reason (as the coding agent does for a user Esc interrupt). + queueMicrotask(() => controller.abort("Interrupted by user")); + + for await (const _event of stream) { + // drain + } + + const messages = await stream.result(); + const finalMessage = messages[messages.length - 1]; + expect(finalMessage.role).toBe("assistant"); + if (finalMessage.role !== "assistant") throw new Error("Expected assistant message"); + expect(finalMessage.stopReason).toBe("aborted"); + // The reason rides AbortController.abort(reason) onto the message verbatim, + // instead of the generic "Request was aborted" default. + expect(finalMessage.errorMessage).toBe("Interrupted by user"); + }); + it("should handle custom message types via convertToLlm", async () => { // Create a custom message type interface CustomNotification { diff --git a/packages/agent/test/agent.test.ts b/packages/agent/test/agent.test.ts index c54f9366a..db95e43df 100644 --- a/packages/agent/test/agent.test.ts +++ b/packages/agent/test/agent.test.ts @@ -354,6 +354,21 @@ describe("Agent", () => { expect(reasoningPerCall).toEqual([ThinkingLevel.Low, ThinkingLevel.High]); }); + it("forwards distinct provider session id and prompt cache key to the stream", async () => { + const mock = createMockModel({ responses: [{ content: ["ok"] }] }); + const agent = new Agent({ + initialState: { model: mock.model, messages: [] }, + streamFn: mock.stream, + sessionId: "provider-lineage", + promptCacheKey: "parent-cache", + }); + + await agent.prompt("run"); + + expect(mock.calls[0]?.options?.sessionId).toBe("provider-lineage"); + expect(mock.calls[0]?.options?.promptCacheKey).toBe("parent-cache"); + }); + it("returns static metadata via the plain setter", () => { const agent = new Agent(); expect(agent.metadata).toBeUndefined(); diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 2f8dea9f7..321ae1a0b 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -1,7 +1,55 @@ # Changelog - ## [Unreleased] +### Added + +- Added support for `impersonated_service_account` Application Default Credentials (ADC) in Vertex AI to enable chained impersonation without failing via 401 `invalid_client`. + + +## [15.10.1] - 2026-06-07 + +### Breaking Changes + +- Removed the `onAuthError` option from stream request options and shifted auth retry handling to resolver-based `apiKey` behavior, requiring callers using custom auth-retry hooks to migrate + +### Added + +- Added `ApiKeyResolver` and `ApiKey` auth helpers, including `isApiKeyResolver`, `isAuthRetryableError`, `resolveApiKeyOnce`, and `withAuth`, and exported them from the package root +- Added support for a function-valued `apiKey` in `SimpleStreamOptions` so a single stream request can refresh or rotate credentials during retry +- Added `forceRefresh` credential option to `AuthStorage.getApiKey` and `rotateSessionCredential` support for session-level credential rotation after auth failures +- Added `AuthStorage.resolver(provider, options)` method that builds an `ApiKeyResolver` implementing the a/b/c auth-retry policy directly on the storage instance + +### Changed + +- Changed gateway and stream auth flows to share the a/b/c retry policy, refreshing the same session credential first and then switching to a sibling credential on repeated auth failures + +### Fixed + +- Fixed duplicate upstream `tool_call_id` values collapsing distinct tool calls during message transformation, preserving one call/result pairing per emitted tool call before provider replay. ([#2055](https://github.com/can1357/oh-my-pi/issues/2055)) +- Fixed streaming auth retries to handle `401` and usage-limit errors before replay-unsafe content is emitted, including failures surfaced only via `errorStatus` +- Fixed tool argument validation to coerce singleton non-string values into arrays when the schema expects an array, preventing Anthropic-compatible models that emit `todo.ops` as an object from getting stuck in repeated validation-error loops. ([#2026](https://github.com/can1357/oh-my-pi/issues/2026)) +- Fixed streaming retries to buffer and suppress partial `start` events from failed auth attempts so only clean retried events are delivered +- Fixed the HTTP 400 raw-request dumper (`appendRawHttpRequestDumpFor400`) littering the real `~/.omp/logs/http-400-requests` directory during tests. Provider suites exercise the 400 error path with mocked `fetch` responses, which the dumper could not distinguish from genuine failures; it now skips persistence under the Bun test runner (`isBunTestRuntime()`). +- Fixed Anthropic Opus requests unnecessarily forcing `tool_choice.disable_parallel_tool_use`, allowing Claude Opus to use the provider's default parallel tool-calling behavior again. +- Fixed parallel `function_call` items losing arguments against llama.cpp's OpenAI Responses endpoint (`/v1/responses`), where every call but the last finalized with `{}` and the agent rejected them with `path: Invalid input: expected string, received undefined`. llama.cpp's `to_json_oaicompat_resp` emits `output_item.added` with only `item.call_id` (no `item.id`, no `output_index`) while the matching `function_call_arguments.delta` carries `item_id: "fc_"`. `processResponsesStream` now registers function-call and custom-tool-call items under `item.call_id` as a secondary lookup key (alongside `item.id`/`output_index`) so identifier-deviant hosts route deltas and done events to the right block. ([#2015](https://github.com/can1357/oh-my-pi/issues/2015)) +- Fixed `PI_REQ_DEBUG` response recording truncating the captured body when a streamed response was cancelled mid-flight. The response tee in `wrapResponse` could call `FileRequestDebugResponseLog.close()` from both the `cancel` callback and the resumed `pull` (which observes `done` once the source reader is cancelled); the second caller saw the handle already nulled and returned before the first caller's pending write flushed, so the `.res.log` lost the already-buffered chunk. `close()` now memoizes its flush-and-close promise so every caller awaits the same completion. + +## [15.10.0] - 2026-06-06 + +### Added + +- Added a dependency-free `@oh-my-pi/pi-ai/effort` module exporting the `Effort` enum and `THINKING_EFFORTS`, split out of `model-thinking` so hot-path consumers can import the thinking levels without pulling in `model-thinking` and its provider-compat dependency graph. The package barrel still re-exports both names, so existing imports are unaffected. + +### Fixed + +- Fixed Antigravity usage provider emitting one bar per model instead of deduplicating by tier — a single account's 15+ model entries now collapse to one bar per tier, matching the shared-quota reality of the upstream API. +- Fixed Antigravity usage reports missing `email` and `accountId` in metadata, so the `/usage` display and the deduplicator can associate reports with their credentials. +- Fixed usage-report dedup ignoring `projectId` for Google Cloud providers, preventing duplicate credential entries from being recognized as the same account. + +- Fixed Cloud Code Assist (Antigravity / Gemini CLI) rejecting the `github` tool with HTTP 400 when the `pr` parameter schema contained `anyOf: [string, array]`. The CCA mixed-type combiner collapse picked the first non-null type (`string`) but indiscriminately copied type-specific keys from variant branches — `items` from the array variant leaked onto the string-typed result, producing `{type: "string", items: {...}}` which Google's API rejects as invalid. The collapse now filters merged variant fields against the winning type's allowed key set. ([#2002](https://github.com/can1357/oh-my-pi/pull/2002)) +- Fixed OpenAI Responses-family providers (Codex, OpenAI Responses, Azure Responses) rejecting requests with `400 No tool output found for function call …` after the user branched/navigated the session tree to a node that ends on a tool call (the tool-result child is dropped from the reconstructed history) or after a turn was aborted/crashed between the call streaming and its result persisting. The converters now synthesize a placeholder `function_call_output`/`custom_tool_call_output` immediately after any unpaired `function_call`/`custom_tool_call`, symmetric to the existing orphan-output repair, so the model still sees the call and can recover instead of the whole request 400ing. +- Fixed Anthropic-compatible reasoning endpoints losing prior-turn reasoning on continuation requests when they emit unsigned `thinking` blocks. `convertAnthropicMessages` treated unknown endpoints as signature-enforcing and demoted unsigned reasoning to `type: "text"`, which destabilized tool-call argument serialization on the next turn — the upstream symptom behind the `args?.ops?.map is not a function` crash reported against the `todo` tool. Official `api.anthropic.com` keeps the conservative text fallback; non-official `anthropic-messages` reasoning models now replay unsigned reasoning as native `type: "thinking"` ([#2005](https://github.com/can1357/oh-my-pi/issues/2005)). + ## [15.9.67] - 2026-06-06 ### Fixed diff --git a/packages/ai/package.json b/packages/ai/package.json index 5c05d9e97..7a108d2e6 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "15.9.67", + "version": "15.10.1", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/scripts/generate-models.ts b/packages/ai/scripts/generate-models.ts index 6431ade6d..497fb7c25 100644 --- a/packages/ai/scripts/generate-models.ts +++ b/packages/ai/scripts/generate-models.ts @@ -32,6 +32,7 @@ import { isFireworksKimiK2ModelId, MODELS_DEV_PROVIDER_DESCRIPTORS, mapModelsDevToModels, + stripFireworksDeepSeekThinkingToggle, UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS, } from "../src/provider-models/openai-compat"; @@ -243,6 +244,18 @@ function applyFireworksKimiMaxTokensCap(models: readonly Model[]): Model[] { }); } +/** + * Fireworks' DeepSeek V4 endpoint accepts the user's effort through + * `reasoning_effort` and rejects the DeepSeek-native binary `thinking` toggle + * when both are present. Strip stale reference metadata from generated fallbacks. + */ +function applyFireworksDeepSeekReasoningShape(models: readonly Model[]): Model[] { + return models.map(model => { + if (model.provider !== "fireworks" || model.api !== "openai-completions") return model; + return stripFireworksDeepSeekThinkingToggle(model, model.id); + }); +} + const ANTIGRAVITY_ENDPOINT = "https://daily-cloudcode-pa.sandbox.googleapis.com"; async function getOAuthAccessFromStorage(provider: OAuthProvider): Promise { @@ -416,6 +429,7 @@ async function generateModels() { allModels = applyPremiumMultiplierOverrides(allModels); allModels = applyCodexPricingFallback(allModels); allModels = applyFireworksKimiMaxTokensCap(allModels); + allModels = applyFireworksDeepSeekReasoningShape(allModels); applyGeneratedModelPolicies(allModels); linkOpenAIPromotionTargets(allModels); diff --git a/packages/ai/src/auth-gateway/server.ts b/packages/ai/src/auth-gateway/server.ts index dad2aa394..83a3e4338 100644 --- a/packages/ai/src/auth-gateway/server.ts +++ b/packages/ai/src/auth-gateway/server.ts @@ -18,8 +18,9 @@ * POST /v1/responses → OpenAI Responses in/out */ import { extractRetryHint, logger } from "@oh-my-pi/pi-utils"; +import type { ApiKeyResolver } from "../auth-retry"; import type { AuthStorage } from "../auth-storage"; -import { Effort } from "../model-thinking"; +import { Effort } from "../effort"; import * as anthropicMessages from "../providers/anthropic-messages-server"; import * as openaiChat from "../providers/openai-chat-server"; import * as openaiResponses from "../providers/openai-responses-server"; @@ -340,6 +341,60 @@ async function refreshGatewayApiKeyAfterAuthError( return storage.getApiKey(provider, sessionId, { modelId: model.id, signal }); } +/** + * Build the {@link ApiKeyResolver} handed to `streamSimple` for a gateway + * request. Drives the central a/b/c auth-retry policy server-side: + * + * - initial resolve → the credential already resolved for this request. + * - step (b) `!lastChance` → force-refresh the SAME session-sticky credential + * (a peer/broker may have rotated its token out from under our cached copy). + * - step (c) `lastChance` → {@link refreshGatewayApiKeyAfterAuthError} switches + * to a sibling (usage-limit block vs credential invalidation by error class). + * + * `lastKey` tracks the most recent bearer so the switch step invalidates the + * credential that actually failed. + */ +function buildGatewayApiKeyResolver( + storage: AuthStorage, + model: Model, + sessionId: string, + initialKey: string, + requestSignal: AbortSignal, + format: string, + peer: string, +): ApiKeyResolver { + let lastKey = initialKey; + return async ({ lastChance, error, signal }) => { + const sig = signal ?? requestSignal; + if (error === undefined) { + lastKey = initialKey; + return initialKey; + } + if (!lastChance) { + const refreshed = await storage.getApiKey(model.provider, sessionId, { + modelId: model.id, + signal: sig, + forceRefresh: true, + }); + lastKey = refreshed ?? lastKey; + return refreshed; + } + const next = await refreshGatewayApiKeyAfterAuthError( + storage, + model, + sessionId, + model.provider, + lastKey, + error, + sig, + format, + peer, + ); + lastKey = next ?? lastKey; + return next; + }; +} + function clientClosedResponse(route: { module: FormatModule }): Response { return route.module.formatError(499, "request_aborted", "client closed request"); } @@ -447,19 +502,15 @@ async function handleFormatEndpoint( } const streamOpts = buildStreamOptions(parsed, model.api, controller.signal); - streamOpts.apiKey = apiKey; - streamOpts.onAuthError = (provider, oldKey, error) => - refreshGatewayApiKeyAfterAuthError( - bootOpts.storage, - model, - sessionId, - provider, - oldKey, - error, - controller.signal, - route.label, - peer, - ); + streamOpts.apiKey = buildGatewayApiKeyResolver( + bootOpts.storage, + model, + sessionId, + apiKey, + controller.signal, + route.label, + peer, + ); logger.info("auth-gateway request", { format: route.label, @@ -604,18 +655,15 @@ async function handlePiNative(bootOpts: AuthGatewayBootOptions, req: Request, pe // only inject server-controlled fields. The codex temperature/topP strip // matches `buildStreamOptions` — Codex rejects them with a 400. const streamOpts: SimpleStreamOptions = { ...parsed.options, apiKey, signal: controller.signal }; - streamOpts.onAuthError = (provider, oldKey, error) => - refreshGatewayApiKeyAfterAuthError( - bootOpts.storage, - model, - sessionId, - provider, - oldKey, - error, - controller.signal, - "pi-native", - peer, - ); + streamOpts.apiKey = buildGatewayApiKeyResolver( + bootOpts.storage, + model, + sessionId, + apiKey, + controller.signal, + "pi-native", + peer, + ); if (model.api === "openai-codex-responses") { delete streamOpts.temperature; delete streamOpts.topP; diff --git a/packages/ai/src/auth-gateway/types.ts b/packages/ai/src/auth-gateway/types.ts index 0390759c0..bdb563e3b 100644 --- a/packages/ai/src/auth-gateway/types.ts +++ b/packages/ai/src/auth-gateway/types.ts @@ -1,4 +1,4 @@ -import type { Effort } from "../model-thinking"; +import type { Effort } from "../effort"; import type { AssistantMessage, AssistantMessageEventStream, diff --git a/packages/ai/src/auth-retry.ts b/packages/ai/src/auth-retry.ts new file mode 100644 index 000000000..e53567f01 --- /dev/null +++ b/packages/ai/src/auth-retry.ts @@ -0,0 +1,141 @@ +import { extractHttpStatusFromError } from "@oh-my-pi/pi-utils"; +import { isUsageLimitError } from "./rate-limit-utils"; + +/** + * Context passed to an {@link ApiKeyResolver} on each resolution attempt. + * + * The `error`/`lastChance` pair drives the central a/b/c retry policy shared by + * the streaming ({@link streamSimple}) and non-streaming ({@link withAuth}) + * drivers: + * - `error === undefined` → **initial resolve** (no force-refresh; cheap, may + * return a locally-cached not-yet-expired token). + * - `error !== undefined && !lastChance` → **step (b): refresh the SAME + * account** (force a token re-mint / await an in-flight broker refresh). + * - `error !== undefined && lastChance` → **step (c): switch account** + * (invalidate/usage-limit the current credential and rotate to a sibling). + * + * The resolver returns the bearer to send, or `undefined` to stop retrying and + * surface the last error to the caller. + */ +export interface ApiKeyResolveContext { + /** True on the final retry step — the resolver should rotate to a sibling credential. */ + lastChance: boolean; + /** The auth error that triggered this re-resolution, or `undefined` on the initial resolve. */ + error: unknown; + /** Caller cancel signal, threaded into any credential refresh / rotation work. */ + signal?: AbortSignal; +} + +/** + * Resolves the API key to send for a request, retried through the a/b/c policy + * described on {@link ApiKeyResolveContext}. + */ +export type ApiKeyResolver = (ctx: ApiKeyResolveContext) => Promise | string | undefined; + +/** A static bearer string, or a {@link ApiKeyResolver} that mints/rotates one. */ +export type ApiKey = string | ApiKeyResolver; + +/** Narrows {@link ApiKey} to its resolver form. */ +export function isApiKeyResolver(key: ApiKey | undefined): key is ApiKeyResolver { + return typeof key === "function"; +} + +/** + * Performs the initial resolve of an {@link ApiKey} (`error: undefined`, + * `lastChance: false`). Static keys pass through unchanged. + */ +export async function resolveApiKeyOnce(key: ApiKey | undefined, signal?: AbortSignal): Promise { + if (key === undefined) return undefined; + if (isApiKeyResolver(key)) return (await key({ lastChance: false, error: undefined, signal })) || undefined; + return key; +} + +/** + * Classifies whether an error should trigger a credential refresh/rotation + * retry: a hard `401`, or a rotatable usage-limit ("usage_limit_reached", + * Codex's "you have hit your ChatGPT usage limit", etc.). + */ +export function isAuthRetryableError(error: unknown): boolean { + if (extractHttpStatusFromError(error) === 401) return true; + const message = error instanceof Error ? error.message : typeof error === "string" ? error : undefined; + if (!message) return false; + if (extractHttpStatusFromError({ message }) === 401) return true; + return isUsageLimitError(message); +} + +/** + * The ordered `lastChance` values for the retry steps after the initial + * attempt fails: `false` → step (b) refresh-same, `true` → step (c) switch. + * Shared by {@link withAuth} and the streaming retry driver so both run the + * same policy. + */ +export const AUTH_RETRY_STEPS: readonly boolean[] = [false, true]; + +/** Resolve a single retry step, swallowing resolver failures into `undefined`. */ +export async function resolveRetryKey( + resolver: ApiKeyResolver, + lastChance: boolean, + error: unknown, + signal?: AbortSignal, +): Promise { + try { + return (await resolver({ lastChance, error, signal })) || undefined; + } catch { + return undefined; + } +} + +/** + * Runs an auth-protected operation through the central a/b/c retry policy. + * + * - A static string key (or any non-resolver) → a single `attempt` with no + * retry (identical to the legacy static-key path). + * - A resolver → initial `attempt`, then on a retryable auth error up to two + * more attempts (refresh-same, then switch). A step is skipped when the + * resolver returns the same key it just tried or `undefined`; non-auth errors + * propagate immediately. + * + * Used by non-streaming consumers (image generation, web search, completion + * helpers). The streaming driver in `stream.ts` implements the same policy with + * its replay-safe buffering machinery. + */ +export async function withAuth( + key: ApiKey | undefined, + attempt: (key: string) => Promise, + opts?: { isAuthError?: (error: unknown) => boolean; signal?: AbortSignal; missingKeyMessage?: string }, +): Promise { + const isAuthError = opts?.isAuthError ?? isAuthRetryableError; + const missingKey = (): Error => new Error(opts?.missingKeyMessage ?? "No API key available"); + + if (!isApiKeyResolver(key)) { + if (key === undefined) throw missingKey(); + return attempt(key); + } + + const resolver = key; + const signal = opts?.signal; + let lastKey = await resolveRetryKey(resolver, false, undefined, signal); + if (lastKey === undefined) throw missingKey(); + + let lastError: unknown; + try { + return await attempt(lastKey); + } catch (error) { + if (!isAuthError(error)) throw error; + lastError = error; + } + + for (let i = 0; i < AUTH_RETRY_STEPS.length; i++) { + const nextKey = await resolveRetryKey(resolver, AUTH_RETRY_STEPS[i]!, lastError, signal); + if (nextKey === undefined || nextKey === lastKey) continue; + lastKey = nextKey; + try { + return await attempt(nextKey); + } catch (error) { + if (!isAuthError(error)) throw error; + lastError = error; + } + } + + throw lastError; +} diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index f03f7b810..b76c2bbd6 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -11,6 +11,8 @@ import { Database, type Statement } from "bun:sqlite"; import * as fs from "node:fs/promises"; import * as path from "node:path"; import { getAgentDbPath, logger } from "@oh-my-pi/pi-utils"; +import type { ApiKeyResolver } from "./auth-retry"; +import { isUsageLimitError } from "./rate-limit-utils"; import { getEnvApiKey } from "./stream"; import type { Provider } from "./types"; import type { @@ -36,6 +38,8 @@ import { loginOpenAICodexDevice } from "./utils/oauth/openai-codex"; import type { OAuthController, OAuthCredentials, OAuthProvider, OAuthProviderId } from "./utils/oauth/types"; import { loginXiaomi, loginXiaomiTokenPlan } from "./utils/oauth/xiaomi"; +const USAGE_RANKING_METRIC_EPSILON = 1e-9; + // ───────────────────────────────────────────────────────────────────────────── // Credential Types // ───────────────────────────────────────────────────────────────────────────── @@ -544,6 +548,13 @@ type AuthApiKeyOptions = { * stranding the caller for `timeoutMs * (maxRetries + 1)`. */ signal?: AbortSignal; + /** + * Force a re-mint of the session-preferred OAuth credential's access token, + * bypassing the not-yet-expired short-circuit. Powers step (b) of the + * auth-retry policy ("refresh the SAME account") so a locally-cached token + * that a peer/broker rotated out from under us is replaced before retrying. + */ + forceRefresh?: boolean; }; type OAuthResolutionResult = { apiKey: string; credential: OAuthCredential }; @@ -606,6 +617,14 @@ function hasOpenAICodexProPlan(report: UsageReport | null): boolean { return getUsagePlanType(report)?.includes("pro") === true; } +function compareUsageRankingMetric(left: number, right: number): number { + if (left === right) return 0; + if (!Number.isFinite(left) || !Number.isFinite(right)) return left < right ? -1 : 1; + const delta = left - right; + const tolerance = Math.max(USAGE_RANKING_METRIC_EPSILON, Math.max(Math.abs(left), Math.abs(right)) * 0.000001); + return Math.abs(delta) <= tolerance ? 0 : delta; +} + function resolveDefaultUsageProvider(provider: Provider): UsageProvider | undefined { return DEFAULT_USAGE_PROVIDER_MAP.get(provider); } @@ -2288,6 +2307,16 @@ export class AuthStorage { return undefined; } + #getUsageReportScopeProjectId(report: UsageReport): string | undefined { + const ids = new Set(); + for (const limit of report.limits) { + const projectId = limit.scope.projectId?.trim(); + if (projectId) ids.add(projectId); + } + if (ids.size === 1) return [...ids][0]; + return undefined; + } + #getUsageReportIdentifiers(report: UsageReport): string[] { const identifiers: string[] = []; const email = this.#getUsageReportMetadataValue(report, "email"); @@ -2295,6 +2324,11 @@ export class AuthStorage { if (report.provider === "openai-codex" || report.provider === "anthropic") { return identifiers.map(identifier => `${report.provider}:${identifier.toLowerCase()}`); } + const projectId = + this.#getUsageReportMetadataValue(report, "projectId") ?? this.#getUsageReportScopeProjectId(report); + // Only add project as a fallback when no email is available — two users + // with different emails on the same GCP project must not merge. + if (projectId && !email) identifiers.push(`project:${projectId}`); const accountId = this.#getUsageReportMetadataValue(report, "accountId"); if (accountId) identifiers.push(`account:${accountId}`); const account = this.#getUsageReportMetadataValue(report, "account"); @@ -2781,12 +2815,14 @@ export class AuthStorage { return left.planPriority - right.planPriority; } if (left.hasPriorityBoost !== right.hasPriorityBoost) return left.hasPriorityBoost ? -1 : 1; - if (left.secondaryDrainRate !== right.secondaryDrainRate) { - return left.secondaryDrainRate - right.secondaryDrainRate; - } - if (left.secondaryUsed !== right.secondaryUsed) return left.secondaryUsed - right.secondaryUsed; - if (left.primaryDrainRate !== right.primaryDrainRate) return left.primaryDrainRate - right.primaryDrainRate; - if (left.primaryUsed !== right.primaryUsed) return left.primaryUsed - right.primaryUsed; + let metric = compareUsageRankingMetric(left.secondaryDrainRate, right.secondaryDrainRate); + if (metric !== 0) return metric; + metric = compareUsageRankingMetric(left.secondaryUsed, right.secondaryUsed); + if (metric !== 0) return metric; + metric = compareUsageRankingMetric(left.primaryDrainRate, right.primaryDrainRate); + if (metric !== 0) return metric; + metric = compareUsageRankingMetric(left.primaryUsed, right.primaryUsed); + if (metric !== 0) return metric; return 0; } @@ -3019,19 +3055,38 @@ export class AuthStorage { candidates.unshift(preferred); } } + // Step (b) of the auth-retry policy: when `forceRefresh` is set, re-mint + // the session-preferred credential (or the first candidate when no + // session preference exists yet) even if its cached token still looks + // valid — a peer/broker may have rotated it out from under us. + const forceRefreshIndex = options?.forceRefresh + ? (sessionPreferredIndex ?? candidates[0]?.selection.index) + : undefined; await Promise.all( candidates.map(async candidate => { - if (Date.now() + OAUTH_REFRESH_SKEW_MS < candidate.selection.credential.expires) return; + const force = forceRefreshIndex !== undefined && candidate.selection.index === forceRefreshIndex; + if (!force && Date.now() + OAUTH_REFRESH_SKEW_MS < candidate.selection.credential.expires) return; const latestCredential = this.#getCredentialsForProvider(provider)[candidate.selection.index]; - if (latestCredential?.type === "oauth" && Date.now() + OAUTH_REFRESH_SKEW_MS < latestCredential.expires) { + if ( + !force && + latestCredential?.type === "oauth" && + Date.now() + OAUTH_REFRESH_SKEW_MS < latestCredential.expires + ) { candidate.selection.credential = latestCredential; return; } try { const credentialId = this.#getStoredCredentials(provider)[candidate.selection.index]?.id; + // Hand #refreshOAuthCredential a stale clone (expires:0) so its + // not-yet-expired short-circuit doesn't suppress the forced + // re-mint; an in-flight peer refresh is still awaited via the + // per-credential single-flight. + const refreshTarget = force + ? { ...candidate.selection.credential, expires: 0 } + : candidate.selection.credential; const refreshedCredentials = await this.#refreshOAuthCredential( provider, - candidate.selection.credential, + refreshTarget, credentialId, options?.signal, ); @@ -3643,6 +3698,90 @@ export class AuthStorage { return true; } + /** + * Rotate away from the session's current credential after a retryable auth + * error — step (c) of the auth-retry policy. Stateless: looks up the + * session-sticky credential (no API-key matching needed), applies the + * storage action for the error class, then clears the sticky so the next + * {@link AuthStorage.getApiKey} for this session picks a sibling. + * + * - usage-limit / account-rate-limit error → {@link AuthStorage.markUsageLimitReached} + * (temporary block via its own backoff — default plus server usage-report + * reset; sticky left intact so the next resolve re-ranks around the block). + * - otherwise (hard 401 / auth failure) → mark the credential suspect (or + * reload when no broker hook is wired) and block it, then drop the sticky. + * + * Returns whether another usable credential of the same type remains. + */ + async rotateSessionCredential( + provider: string, + sessionId: string | undefined, + options?: { error?: unknown; signal?: AbortSignal }, + ): Promise { + const sessionCredential = this.#getSessionCredential(provider, sessionId); + if (!sessionCredential) return false; + + const error = options?.error; + const message = error instanceof Error ? error.message : typeof error === "string" ? error : undefined; + if (message && isUsageLimitError(message)) { + return this.markUsageLimitReached(provider, sessionId, { signal: options?.signal }); + } + + const providerKey = this.#getProviderTypeKey(provider, sessionCredential.type); + // Snapshot sibling availability before mutating so a soft-deleting + // suspect hook can't reindex the answer out from under us. + const hasSibling = this.#getCredentialsForProvider(provider).some( + (credential, index) => + credential.type === sessionCredential.type && + index !== sessionCredential.index && + !this.#isCredentialBlocked(providerKey, index), + ); + const target = this.#getStoredCredentials(provider)[sessionCredential.index]; + this.#clearSessionCredential(provider, sessionId); + this.#markCredentialBlocked(providerKey, sessionCredential.index, Date.now() + AuthStorage.#defaultBackoffMs); + + if (target) { + const markSuspect = this.#store.markCredentialSuspect?.bind(this.#store); + if (markSuspect) { + await markSuspect(target.id, { signal: options?.signal }); + } else { + await this.reload(); + } + const latestRows = this.#store.listAuthCredentials(provider); + this.#setStoredCredentials( + provider, + latestRows.map(row => ({ id: row.id, credential: row.credential })), + ); + } + + return hasSibling; + } + + /** + * Build an {@link ApiKeyResolver} backed by this storage, implementing the + * central a/b/c auth-retry policy: + * + * - initial (`error: undefined`) → resolve the session credential. + * - step (b) `!lastChance` → force-refresh the SAME session-sticky credential. + * - step (c) `lastChance` → rotate to a sibling credential, then re-resolve. + * + * Used by web-search providers and other consumers that hold an AuthStorage + * directly (no ModelRegistry in scope). + */ + resolver(provider: string, options?: { sessionId?: string; baseUrl?: string; modelId?: string }): ApiKeyResolver { + const { sessionId, baseUrl, modelId } = options ?? {}; + return async ({ lastChance, error, signal }) => { + if (error === undefined) { + return this.getApiKey(provider, sessionId, { baseUrl, modelId, signal }); + } + if (lastChance) { + await this.rotateSessionCredential(provider, sessionId, { error, signal }); + return this.getApiKey(provider, sessionId, { baseUrl, modelId, signal }); + } + return this.getApiKey(provider, sessionId, { baseUrl, modelId, forceRefresh: true, signal }); + }; + } + // ─── Auth Broker integration ──────────────────────────────────────────── /** @@ -3932,6 +4071,8 @@ function resolveProviderCredentialIdentityKey(provider: string, identifiers: str const accountIdentifier = identifiers.find(identifier => identifier.startsWith("account:")); if (accountIdentifier) return accountIdentifier; if (emailIdentifier) return emailIdentifier; + const projectIdentifier = identifiers.find(identifier => identifier.startsWith("project:")); + if (projectIdentifier) return projectIdentifier; return null; } @@ -3967,6 +4108,8 @@ function extractOAuthCredentialIdentifiers(credential: OAuthCredential): string[ if (accountId) identifiers.add(`account:${accountId}`); const email = normalizeStoredEmail(credential.email); if (email) identifiers.add(`email:${email}`); + const projectId = normalizeStoredAccountId(credential.projectId); + if (projectId) identifiers.add(`project:${projectId}`); const accessIdentifiers = extractOAuthTokenIdentifiers(credential.access) ?? []; for (const identifier of accessIdentifiers) { identifiers.add(identifier); diff --git a/packages/ai/src/effort.ts b/packages/ai/src/effort.ts new file mode 100644 index 000000000..831a13ede --- /dev/null +++ b/packages/ai/src/effort.ts @@ -0,0 +1,16 @@ +/** User-facing thinking levels, ordered least to most intensive. */ +export const enum Effort { + Minimal = "minimal", + Low = "low", + Medium = "medium", + High = "high", + XHigh = "xhigh", +} + +export const THINKING_EFFORTS: readonly Effort[] = [ + Effort.Minimal, + Effort.Low, + Effort.Medium, + Effort.High, + Effort.XHigh, +]; diff --git a/packages/ai/src/index.ts b/packages/ai/src/index.ts index ecdce180e..a9d3c724a 100644 --- a/packages/ai/src/index.ts +++ b/packages/ai/src/index.ts @@ -3,7 +3,9 @@ export * from "./api-registry"; export * from "./auth-broker"; export { type AuthGatewayBootOptions, type ModelResolver, startAuthGateway } from "./auth-gateway/server"; export * from "./auth-gateway/types"; +export * from "./auth-retry"; export * from "./auth-storage"; +export * from "./effort"; export * from "./model-cache"; export * from "./model-manager"; export * from "./model-thinking"; diff --git a/packages/ai/src/model-thinking.ts b/packages/ai/src/model-thinking.ts index 75e930315..3758c6d6c 100644 --- a/packages/ai/src/model-thinking.ts +++ b/packages/ai/src/model-thinking.ts @@ -1,23 +1,7 @@ +import { Effort, THINKING_EFFORTS } from "./effort"; import { resolveOpenAICompat } from "./providers/openai-completions-compat"; import type { Api, Model as ApiModel, ThinkingConfig } from "./types"; -/** User-facing thinking levels, ordered least to most intensive. */ -export const enum Effort { - Minimal = "minimal", - Low = "low", - Medium = "medium", - High = "high", - XHigh = "xhigh", -} - -export const THINKING_EFFORTS: readonly Effort[] = [ - Effort.Minimal, - Effort.Low, - Effort.Medium, - Effort.High, - Effort.XHigh, -]; - const DEFAULT_REASONING_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]; const DEFAULT_REASONING_EFFORTS_WITH_XHIGH: readonly Effort[] = [ Effort.Minimal, @@ -379,18 +363,6 @@ export function supportsMidConversationSystemMessages(modelId: string): boolean return parsed.kind === "opus" && semverGte(parsed.version, "4.8"); } -/** - * Claude Opus 4.8 must emit at most one tool call per turn: the Anthropic - * Messages provider sends `tool_choice.disable_parallel_tool_use = true` for - * this model. Scoped to exactly 4.8 — earlier and later Opus versions keep - * Anthropic's default parallel tool-calling. - */ -export function disablesParallelToolUse(modelId: string): boolean { - const parsed = parseAnthropicModel(getCanonicalModelId(modelId)); - if (!parsed) return false; - return parsed.kind === "opus" && semverEqual(parsed.version, "4.8"); -} - function anthropicModelHasRealXHighEffort(model: ApiModel): boolean { if (model.api !== "anthropic-messages") return false; const parsedModel = parseKnownModel(model.id); diff --git a/packages/ai/src/models.json b/packages/ai/src/models.json index e98a047fc..d3cd30bf1 100644 --- a/packages/ai/src/models.json +++ b/packages/ai/src/models.json @@ -1932,6 +1932,75 @@ "maxLevel": "high" } }, + "openai.gpt-5.4": { + "id": "openai.gpt-5.4", + "name": "GPT-5.4", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.75, + "output": 16.5, + "cacheRead": 0.275, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "minLevel": "low", + "maxLevel": "xhigh" + } + }, + "openai.gpt-5.5": { + "id": "openai.gpt-5.5", + "name": "GPT-5.5", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5.5, + "output": 33, + "cacheRead": 0.55, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "minLevel": "low", + "maxLevel": "xhigh" + } + }, + "openai.gpt-oss-120b": { + "id": "openai.gpt-oss-120b", + "name": "gpt-oss-120b", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.15, + "output": 0.6, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 16384 + }, "openai.gpt-oss-120b-1:0": { "id": "openai.gpt-oss-120b-1:0", "name": "gpt-oss-120b", @@ -1951,6 +2020,25 @@ "contextWindow": 128000, "maxTokens": 16384 }, + "openai.gpt-oss-20b": { + "id": "openai.gpt-oss-20b", + "name": "gpt-oss-20b", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.07, + "output": 0.3, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 16384 + }, "openai.gpt-oss-20b-1:0": { "id": "openai.gpt-oss-20b-1:0", "name": "gpt-oss-20b", @@ -5364,11 +5452,6 @@ }, "maxTokensField": "max_tokens", "supportsToolChoice": false, - "extraBody": { - "thinking": { - "type": "enabled" - } - }, "reasoningContentField": "reasoning_content", "requiresReasoningContentForToolCalls": true, "requiresAssistantContentForToolCalls": true @@ -11197,7 +11280,7 @@ }, "deepseek/deepseek-r1-distill-llama-70b": { "id": "deepseek/deepseek-r1-distill-llama-70b", - "name": "DeepSeek: R1 Distill Llama 70B", + "name": "DeepSeek: R1 Distill Llama 70B (retires Jun 11)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -12648,7 +12731,7 @@ }, "meta-llama/llama-3-70b-instruct": { "id": "meta-llama/llama-3-70b-instruct", - "name": "Meta: Llama 3 70B Instruct", + "name": "Meta: Llama 3 70B Instruct (retires Jun 19)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -13187,6 +13270,25 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", + "name": "MiniMax: MiniMax M3 (new)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "minimax/minimax-m3:discounted": { + "id": "minimax/minimax-m3:discounted", "name": "MiniMax: MiniMax M3 (50% off through 2026-06-07)", "api": "openai-completions", "provider": "kilo", @@ -14262,6 +14364,63 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "nvidia/nemotron-3-ultra-550b-a55b": { + "id": "nvidia/nemotron-3-ultra-550b-a55b", + "name": "NVIDIA: Nemotron 3 Ultra", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "nvidia/nemotron-3-ultra-550b-a55b:free": { + "id": "nvidia/nemotron-3-ultra-550b-a55b:free", + "name": "NVIDIA: Nemotron 3 Ultra (free)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "nvidia/nemotron-3.5-content-safety:free": { + "id": "nvidia/nemotron-3.5-content-safety:free", + "name": "NVIDIA: Nemotron 3.5 Content Safety (free)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "nvidia/nemotron-nano-12b-v2-vl": { "id": "nvidia/nemotron-nano-12b-v2-vl", "name": "NVIDIA: Nemotron Nano 12B 2 VL", @@ -14283,7 +14442,7 @@ }, "nvidia/nemotron-nano-9b-v2": { "id": "nvidia/nemotron-nano-9b-v2", - "name": "NVIDIA: Nemotron Nano 9B V2", + "name": "NVIDIA: Nemotron Nano 9B V2 (retires Jun 11)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -16371,7 +16530,7 @@ }, "qwen/qwen3-30b-a3b": { "id": "qwen/qwen3-30b-a3b", - "name": "Qwen: Qwen3 30B A3B (retires Jun 5)", + "name": "Qwen: Qwen3 30B A3B", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -19993,24 +20152,19 @@ "api": "openai-completions", "provider": "mistral", "baseUrl": "https://api.mistral.ai/v1", - "reasoning": true, + "reasoning": false, "input": [ "text", "image" ], "cost": { - "input": 1.5, - "output": 7.5, + "input": 0.4, + "output": 2, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, - "thinking": { - "mode": "effort", - "minLevel": "minimal", - "maxLevel": "xhigh" - } + "maxTokens": 262144 }, "mistral-nemo": { "id": "mistral-nemo", @@ -25743,6 +25897,82 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "linkup-research-high": { + "id": "linkup-research-high", + "name": "linkup-research-high", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "linkup-research-low": { + "id": "linkup-research-low", + "name": "linkup-research-low", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "linkup-research-medium": { + "id": "linkup-research-medium", + "name": "linkup-research-medium", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "linkup-research-xhigh": { + "id": "linkup-research-xhigh", + "name": "linkup-research-xhigh", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "liquid/lfm-2-24b-a2b": { "id": "liquid/lfm-2-24b-a2b", "name": "liquid/lfm-2-24b-a2b", @@ -28343,6 +28573,30 @@ "maxLevel": "xhigh" } }, + "nvidia/nemotron-3-ultra-550b-a55b": { + "id": "nvidia/nemotron-3-ultra-550b-a55b", + "name": "nvidia/nemotron-3-ultra-550b-a55b", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "nvidia/nvidia-nemotron-nano-9b-v2": { "id": "nvidia/nvidia-nemotron-nano-9b-v2", "name": "nvidia-nemotron-nano-9b-v2", @@ -38214,7 +38468,7 @@ "cacheRead": 0.05, "cacheWrite": 0.625 }, - "contextWindow": 262144, + "contextWindow": 1000000, "maxTokens": 65536, "thinking": { "mode": "effort", @@ -38263,7 +38517,7 @@ "cacheRead": 0.04, "cacheWrite": 0.5 }, - "contextWindow": 262144, + "contextWindow": 1000000, "maxTokens": 65536, "thinking": { "mode": "budget", @@ -39625,6 +39879,30 @@ "maxLevel": "xhigh" } }, + "nemotron-3-ultra-free": { + "id": "nemotron-3-ultra-free", + "name": "Nemotron 3 Ultra Free", + "api": "openai-completions", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "qwen3.5-plus": { "id": "qwen3.5-plus", "name": "Qwen3.5 Plus", @@ -41767,12 +42045,12 @@ ], "cost": { "input": 0.12, - "output": 0.37, - "cacheRead": 0.019999999499999997, + "output": 0.36, + "cacheRead": 0.09, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 16384, + "maxTokens": 8192, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -42135,7 +42413,7 @@ ], "cost": { "input": 0.02, - "output": 0.049999999999999996, + "output": 0.03, "cacheRead": 0, "cacheWrite": 0 }, @@ -42360,7 +42638,7 @@ "cacheWrite": 0 }, "contextWindow": 204800, - "maxTokens": 131072, + "maxTokens": 196608, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -43295,6 +43573,54 @@ "maxLevel": "high" } }, + "nvidia/nemotron-3-ultra-550b-a55b": { + "id": "nvidia/nemotron-3-ultra-550b-a55b", + "name": "NVIDIA: Nemotron 3 Ultra", + "api": "openai-completions", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.5, + "output": 2.5, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "nvidia/nemotron-3-ultra-550b-a55b:free": { + "id": "nvidia/nemotron-3-ultra-550b-a55b:free", + "name": "NVIDIA: Nemotron 3 Ultra (free)", + "api": "openai-completions", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } + }, "nvidia/nemotron-nano-12b-v2-vl:free": { "id": "nvidia/nemotron-nano-12b-v2-vl:free", "name": "NVIDIA: Nemotron Nano 12B 2 VL (free)", @@ -45244,7 +45570,7 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 20000, + "maxTokens": 16384, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -46000,13 +46326,13 @@ "image" ], "cost": { - "input": 0.29, - "output": 3.1999999999999997, + "input": 0.28900000000000003, + "output": 2.4, "cacheRead": 0.15, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262140, + "maxTokens": 131072, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -49634,6 +49960,28 @@ "supportsUsageInStreaming": false } }, + "nvidia-nemotron-3-ultra-550b-a55b": { + "id": "nvidia-nemotron-3-ultra-550b-a55b", + "name": "nvidia-nemotron-3-ultra-550b-a55b", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888, + "compat": { + "supportsUsageInStreaming": false + } + }, "nvidia-nemotron-cascade-2-30b-a3b": { "id": "nvidia-nemotron-cascade-2-30b-a3b", "name": "nvidia-nemotron-cascade-2-30b-a3b", @@ -52853,6 +53201,30 @@ "maxLevel": "xhigh" } }, + "nvidia/nemotron-3-ultra-550b-a55b": { + "id": "nvidia/nemotron-3-ultra-550b-a55b", + "name": "Nemotron 3 Ultra", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.4, + "cacheRead": 0.12, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "nvidia/nemotron-nano-12b-v2-vl": { "id": "nvidia/nemotron-nano-12b-v2-vl", "name": "Nvidia Nemotron Nano 12B V2 VL", @@ -57962,7 +58334,7 @@ "cost": { "input": 0.5, "output": 1.5, - "cacheRead": 0, + "cacheRead": 0.05, "cacheWrite": 0 }, "contextWindow": 256000, diff --git a/packages/ai/src/provider-models/ollama.ts b/packages/ai/src/provider-models/ollama.ts index c71fbf617..ed539d924 100644 --- a/packages/ai/src/provider-models/ollama.ts +++ b/packages/ai/src/provider-models/ollama.ts @@ -1,6 +1,6 @@ import { fetchWithRetry } from "@oh-my-pi/pi-utils"; +import { Effort } from "../effort"; import type { ModelManagerOptions } from "../model-manager"; -import { Effort } from "../model-thinking"; import type { ThinkingConfig } from "../types"; import { createBundledReferenceMap, createReferenceResolver } from "./bundled-references"; diff --git a/packages/ai/src/provider-models/openai-compat.ts b/packages/ai/src/provider-models/openai-compat.ts index d774bfd9f..c529fb135 100644 --- a/packages/ai/src/provider-models/openai-compat.ts +++ b/packages/ai/src/provider-models/openai-compat.ts @@ -1,5 +1,5 @@ +import { Effort } from "../effort"; import type { ModelManagerOptions } from "../model-manager"; -import { Effort } from "../model-thinking"; import { getBundledModels } from "../models"; import type { Api, Model, Provider, ThinkingConfig } from "../types"; import { isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils"; @@ -971,6 +971,29 @@ export function clampFireworksKimiMaxTokens(modelId: string, candidate: number): return isFireworksKimiK2ModelId(modelId) ? Math.min(candidate, FIREWORKS_KIMI_MAX_TOKENS) : candidate; } +/** + * Fireworks DeepSeek V4 accepts effort via `reasoning_effort` but rejects the + * DeepSeek-native binary `thinking` toggle when both are present. + */ +export function stripFireworksDeepSeekThinkingToggle( + model: Model<"openai-completions">, + publicModelId: string, +): Model<"openai-completions"> { + if (!publicModelId.startsWith("deepseek-v4")) return model; + const compat = model.compat; + if (!compat?.extraBody || !("thinking" in compat.extraBody)) return model; + + const extraBody = { ...compat.extraBody }; + delete extraBody.thinking; + if (Object.keys(extraBody).length > 0) { + return { ...model, compat: { ...compat, extraBody } }; + } + + const nextCompat = { ...compat }; + delete nextCompat.extraBody; + return { ...model, compat: nextCompat }; +} + export interface FireworksModelManagerConfig { apiKey?: string; baseUrl?: string; @@ -1040,7 +1063,10 @@ export function fireworksModelManagerOptions( mapModel: (entry, defaults) => { const publicModelId = toFireworksPublicModelId(defaults.id); const reference = modelsDevReferences.get(publicModelId) ?? bundledReferences(publicModelId); - const model = mapWithBundledReference(entry, defaults, reference); + const model = stripFireworksDeepSeekThinkingToggle( + mapWithBundledReference(entry, defaults, reference), + publicModelId, + ); return { ...model, id: publicModelId, diff --git a/packages/ai/src/providers/__tests__/google-auth.test.ts b/packages/ai/src/providers/__tests__/google-auth.test.ts new file mode 100644 index 000000000..8b72ffcc4 --- /dev/null +++ b/packages/ai/src/providers/__tests__/google-auth.test.ts @@ -0,0 +1,144 @@ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { Buffer } from "node:buffer"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import type { FetchImpl } from "../../types"; +import { __resetVertexTokenCache, getVertexAccessToken } from "../google-auth"; + +const CLOUD_PLATFORM_SCOPE = "https://www.googleapis.com/auth/cloud-platform"; +const JWT_BEARER_GRANT = "urn:ietf:params:oauth:grant-type:jwt-bearer"; + +/** Generate a real RS256 private key so signJwtRs256 / pemToPkcs8 run for real. */ +async function generateServiceAccountPem(): Promise { + const keyPair = (await globalThis.crypto.subtle.generateKey( + { name: "RSASSA-PKCS1-v1_5", modulusLength: 2048, publicExponent: new Uint8Array([1, 0, 1]), hash: "SHA-256" }, + true, + ["sign", "verify"], + )) as CryptoKeyPair; + const pkcs8 = new Uint8Array(await globalThis.crypto.subtle.exportKey("pkcs8", keyPair.privateKey)); + const body = ( + Buffer.from(pkcs8) + .toString("base64") + .match(/.{1,64}/g) ?? [] + ).join("\n"); + return `-----BEGIN PRIVATE KEY-----\n${body}\n-----END PRIVATE KEY-----\n`; +} + +function urlOf(input: string | URL | Request): string { + if (typeof input === "string") return input; + if (input instanceof URL) return input.toString(); + return input.url; +} + +describe("getVertexAccessToken impersonated_service_account ADC", () => { + let tmpDir: string; + let originalGac: string | undefined; + + beforeEach(async () => { + __resetVertexTokenCache(); + originalGac = Bun.env.GOOGLE_APPLICATION_CREDENTIALS; + tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-vertex-adc-")); + }); + + afterEach(async () => { + __resetVertexTokenCache(); + if (originalGac === undefined) delete Bun.env.GOOGLE_APPLICATION_CREDENTIALS; + else Bun.env.GOOGLE_APPLICATION_CREDENTIALS = originalGac; + await fs.rm(tmpDir, { recursive: true, force: true }); + }); + + it("rejects a malformed service_account_impersonation_url before any network call", async () => { + const adcPath = path.join(tmpDir, "impersonated-bad-url.json"); + await Bun.write( + adcPath, + JSON.stringify({ + type: "impersonated_service_account", + // Missing the trailing ":generateAccessToken" the principal parser requires. + service_account_impersonation_url: + "https://iamcredentials.googleapis.com/v1/projects/-/serviceAccounts/target@project.iam.gserviceaccount.com", + source_credentials: { + type: "authorized_user", + client_id: "client-id", + client_secret: "client-secret", + refresh_token: "refresh-token", + }, + }), + ); + Bun.env.GOOGLE_APPLICATION_CREDENTIALS = adcPath; + + const calls: string[] = []; + const fetchImpl: FetchImpl = async input => { + calls.push(urlOf(input)); + return new Response("{}"); + }; + + // The principal is parsed before the source exchange, so a bad URL must fail + // up front rather than after burning a source-token round trip. + await expect(getVertexAccessToken({ fetch: fetchImpl })).rejects.toBeInstanceOf(RangeError); + expect(calls).toEqual([]); + }); + + it("signs an RS256 JWT for a service_account source and reconstructs the IAM URL", async () => { + const pem = await generateServiceAccountPem(); + const adcPath = path.join(tmpDir, "impersonated-sa.json"); + await Bun.write( + adcPath, + JSON.stringify({ + type: "impersonated_service_account", + // Non-canonical project segment proves the request URL is rebuilt, not echoed. + service_account_impersonation_url: + "https://iamcredentials.googleapis.com/v1/projects/explicit-proj/serviceAccounts/target@project.iam.gserviceaccount.com:generateAccessToken", + source_credentials: { + type: "service_account", + client_email: "source@project.iam.gserviceaccount.com", + private_key: pem, + private_key_id: "key-1", + }, + // delegates intentionally omitted — the IAM body must default to []. + }), + ); + Bun.env.GOOGLE_APPLICATION_CREDENTIALS = adcPath; + + const calls: { url: string; init?: RequestInit }[] = []; + const fetchImpl: FetchImpl = async (input, init) => { + const url = urlOf(input); + calls.push({ url, init }); + if (url === "https://oauth2.googleapis.com/token") { + return new Response(JSON.stringify({ access_token: "sa-source-token", expires_in: 3600 })); + } + if (url.startsWith("https://iamcredentials.googleapis.com/")) { + return new Response( + JSON.stringify({ + accessToken: "impersonated-token", + expireTime: new Date(Date.now() + 3_600_000).toISOString(), + }), + ); + } + return new Response("unexpected", { status: 404 }); + }; + + const token = await getVertexAccessToken({ fetch: fetchImpl }); + expect(token).toBe("impersonated-token"); + + // Source JWT exchange happens first, then the impersonation exchange. + expect(calls.map(c => c.url)).toEqual([ + "https://oauth2.googleapis.com/token", + "https://iamcredentials.googleapis.com/v1/projects/-/serviceAccounts/target@project.iam.gserviceaccount.com:generateAccessToken", + ]); + + // Source credential is exchanged via a signed JWT bearer assertion, not a refresh grant. + const sourceBody = new URLSearchParams(String(calls[0].init?.body)); + expect(sourceBody.get("grant_type")).toBe(JWT_BEARER_GRANT); + expect((sourceBody.get("assertion") ?? "").split(".")).toHaveLength(3); + + // The IAM call carries the source-derived bearer token and defaults delegates to []. + const iamHeaders = calls[1].init?.headers as Record; + expect(iamHeaders.Authorization).toBe("Bearer sa-source-token"); + expect(JSON.parse(String(calls[1].init?.body))).toEqual({ + delegates: [], + scope: [CLOUD_PLATFORM_SCOPE], + lifetime: "3600s", + }); + }); +}); diff --git a/packages/ai/src/providers/amazon-bedrock.ts b/packages/ai/src/providers/amazon-bedrock.ts index 7f161f32c..49b1523aa 100644 --- a/packages/ai/src/providers/amazon-bedrock.ts +++ b/packages/ai/src/providers/amazon-bedrock.ts @@ -8,7 +8,7 @@ */ import { $env, $flag, extractHttpStatusFromError, fetchWithRetry } from "@oh-my-pi/pi-utils"; -import type { Effort } from "../model-thinking"; +import type { Effort } from "../effort"; import { mapEffortToAnthropicAdaptiveEffort, requireSupportedEffort } from "../model-thinking"; import { calculateCost } from "../models"; import type { diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 6716952fc..c8e9f3c35 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -13,7 +13,6 @@ import { readSseEvents, } from "@oh-my-pi/pi-utils"; import { - disablesParallelToolUse, hasOpus47ApiRestrictions, mapEffortToAnthropicAdaptiveEffort, supportsMidConversationSystemMessages, @@ -2371,21 +2370,6 @@ function buildParams( } } - // Claude Opus 4.8 must emit at most one tool call per turn. Force - // `disable_parallel_tool_use` onto the outgoing tool_choice (synthesizing an - // `auto` choice when none is set). Gated on tools being present: Anthropic - // rejects `tool_choice` without `tools`, and parallelism is moot otherwise. - // `none` rejects the field, so leave it untouched. A fresh object is built - // rather than mutated so the caller's `options.toolChoice` is never aliased. - if (disablesParallelToolUse(model.id) && params.tools && params.tools.length > 0) { - const current = params.tool_choice; - if (!current) { - params.tool_choice = { type: "auto", disable_parallel_tool_use: true }; - } else if (current.type !== "none") { - params.tool_choice = { ...current, disable_parallel_tool_use: true }; - } - } - const shouldInjectClaudeCodeInstruction = isOAuthToken && !model.id.startsWith("claude-3-5-haiku"); const firstUserMessageText = shouldInjectClaudeCodeInstruction ? extractClaudeCodeFirstUserMessageText(context.messages) @@ -2427,22 +2411,31 @@ function isZaiAnthropicEndpoint(model: Model<"anthropic-messages">): boolean { } /** - * Returns true for providers whose Anthropic-compatible endpoints do NOT - * implement signature-based thinking-chain integrity (DeepSeek, Z.AI, etc.). - * For these providers, unsigned thinking blocks must be preserved as - * `type: "thinking"` instead of being degraded to text. + * Returns true when unsigned `thinking` blocks from prior assistant turns should + * be replayed as Anthropic-native thinking instead of demoted to text. + * + * Official Anthropic (matched via `isAnthropicApiBaseUrl`, which intentionally + * treats a missing baseUrl as official since `resolveAnthropicBaseUrl` routes + * it to `https://api.anthropic.com`) enforces signature-based thinking-chain + * integrity, so unsigned blocks must remain text there. Anthropic-compatible + * reasoning endpoints commonly emit unsigned thinking blocks while still + * expecting them back as `type: "thinking"` on continuation; demoting them + * loses the model's reasoning chain and can destabilize the next tool-call + * arguments (#2005). Known non-signing hosts are also preserved for + * compatibility. */ -function isNonSigningAnthropicEndpoint(model: Model<"anthropic-messages">): boolean { - // Known non-signing providers +function shouldReplayUnsignedThinking(model: Model<"anthropic-messages">): boolean { if (model.provider === "zai" || model.provider === "deepseek") return true; const baseUrl = model.baseUrl; - if (!baseUrl) return false; - try { - const hostname = new URL(baseUrl).hostname.toLowerCase(); - return hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com"); - } catch { - return false; + if (baseUrl) { + try { + const hostname = new URL(baseUrl).hostname.toLowerCase(); + if (hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com")) return true; + } catch { + // Fall through to the protocol-level reasoning rule below. + } } + return model.reasoning && !isAnthropicApiBaseUrl(baseUrl); } function buildToolResultBlock(model: Model<"anthropic-messages">, msg: ToolResultMessage): ContentBlockParam { @@ -2533,7 +2526,7 @@ export function convertAnthropicMessages( } if (block.thinking.trim().length === 0) continue; if (!block.thinkingSignature || block.thinkingSignature.trim().length === 0) { - if (isNonSigningAnthropicEndpoint(model)) { + if (shouldReplayUnsignedThinking(model)) { blocks.push({ type: "thinking", thinking: block.thinking.toWellFormed(), diff --git a/packages/ai/src/providers/azure-openai-responses.ts b/packages/ai/src/providers/azure-openai-responses.ts index 04027d02a..26b3f0a16 100644 --- a/packages/ai/src/providers/azure-openai-responses.ts +++ b/packages/ai/src/providers/azure-openai-responses.ts @@ -40,6 +40,7 @@ import { isOpenAIResponsesProgressEvent, normalizeResponsesToolCallIdForTransform, processResponsesStream, + repairOrphanResponsesToolCalls, } from "./openai-responses-shared"; import { transformMessages } from "./transform-messages"; @@ -347,7 +348,7 @@ function convertMessages( msgIndex++; } - return messages; + return repairOrphanResponsesToolCalls(messages); } function convertTools(tools: Tool[]): OpenAITool[] { diff --git a/packages/ai/src/providers/gitlab-duo.ts b/packages/ai/src/providers/gitlab-duo.ts index 50c43b372..382ce56c9 100644 --- a/packages/ai/src/providers/gitlab-duo.ts +++ b/packages/ai/src/providers/gitlab-duo.ts @@ -234,7 +234,8 @@ export function streamGitLabDuo( (async () => { try { - if (!options?.apiKey) { + const apiKey = typeof options?.apiKey === "string" ? options.apiKey : undefined; + if (!apiKey || !options) { throw new Error("Missing GitLab access token. Run /login gitlab-duo or set GITLAB_TOKEN."); } @@ -243,7 +244,7 @@ export function streamGitLabDuo( throw new Error(`Unsupported GitLab Duo model: ${model.id}`); } - const directAccess = await getDirectAccessToken(options.apiKey, options.fetch); + const directAccess = await getDirectAccessToken(apiKey, options.fetch); const headers = { ...directAccess.headers, ...options.headers, diff --git a/packages/ai/src/providers/google-auth.ts b/packages/ai/src/providers/google-auth.ts index a8004508f..0af7ddea8 100644 --- a/packages/ai/src/providers/google-auth.ts +++ b/packages/ai/src/providers/google-auth.ts @@ -42,7 +42,14 @@ interface AuthorizedUserCredentials { refresh_token: string; } -type AdcFileCredentials = ServiceAccountCredentials | AuthorizedUserCredentials; +interface ImpersonatedServiceAccountCredentials { + type: "impersonated_service_account"; + service_account_impersonation_url: string; + source_credentials: AuthorizedUserCredentials | ServiceAccountCredentials; + delegates?: string[]; +} + +type AdcFileCredentials = ServiceAccountCredentials | AuthorizedUserCredentials | ImpersonatedServiceAccountCredentials; interface TokenResponse { access_token: string; @@ -196,10 +203,52 @@ async function resolveAccessTokenUncached( ): Promise<{ source: string; token: TokenResponse }> { const adc = await loadAdcCredentials(); if (adc) { - const token = - adc.creds.type === "service_account" - ? await exchangeJwtForToken(adc.creds, signal, fetchImpl) - : await exchangeRefreshToken(adc.creds, signal, fetchImpl); + const creds = adc.creds; + let token: TokenResponse; + + if (creds.type === "impersonated_service_account") { + const targetPrincipalMatch = /(?[^/]+):(generateAccessToken|generateIdToken)$/.exec( + creds.service_account_impersonation_url, + ); + const targetPrincipal = targetPrincipalMatch?.groups?.target; + if (!targetPrincipal) { + throw new RangeError(`Cannot extract target principal from ${creds.service_account_impersonation_url}`); + } + + const sourceToken = + creds.source_credentials.type === "service_account" + ? await exchangeJwtForToken(creds.source_credentials, signal, fetchImpl) + : await exchangeRefreshToken(creds.source_credentials, signal, fetchImpl); + + const response = await fetchImpl( + `https://iamcredentials.googleapis.com/v1/projects/-/serviceAccounts/${targetPrincipal}:generateAccessToken`, + { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${sourceToken.access_token}`, + }, + body: JSON.stringify({ + delegates: creds.delegates ?? [], + scope: [CLOUD_PLATFORM_SCOPE], + lifetime: "3600s", + }), + signal, + }, + ); + if (!response.ok) { + const detail = await response.text().catch(() => ""); + throw new Error(`Google Impersonation token exchange failed (${response.status}): ${detail}`); + } + const data = (await response.json()) as { accessToken: string; expireTime: string }; + const expiresIn = Math.max(0, Math.floor((new Date(data.expireTime).getTime() - Date.now()) / 1000)); + token = { access_token: data.accessToken, expires_in: expiresIn, token_type: "Bearer" }; + } else { + token = + creds.type === "service_account" + ? await exchangeJwtForToken(creds, signal, fetchImpl) + : await exchangeRefreshToken(creds, signal, fetchImpl); + } return { source: adc.source, token }; } const metadata = await fetchMetadataToken(signal, fetchImpl); diff --git a/packages/ai/src/providers/openai-anthropic-shim.ts b/packages/ai/src/providers/openai-anthropic-shim.ts index f2dad56d7..a4f9b8fac 100644 --- a/packages/ai/src/providers/openai-anthropic-shim.ts +++ b/packages/ai/src/providers/openai-anthropic-shim.ts @@ -44,6 +44,9 @@ export function streamOpenAIAnthropicShim( ): AssistantMessageEventStream { const stream = new AssistantMessageEventStream(); const format = options?.format ?? config.defaultFormat; + // The resolver form of `apiKey` is resolved upstream in `streamSimple`; + // this shim only ever receives a static bearer string. + const apiKey = typeof options?.apiKey === "string" ? options.apiKey : undefined; (async () => { try { @@ -74,7 +77,7 @@ export function streamOpenAIAnthropicShim( : undefined; const innerStream = streamAnthropic(anthropicModel, context, { - apiKey: options?.apiKey, + apiKey, temperature: options?.temperature, topP: options?.topP, topK: options?.topK, @@ -103,7 +106,7 @@ export function streamOpenAIAnthropicShim( const reasoningEffort = options?.reasoning; const innerStream = streamOpenAICompletions(openaiModel, context, { - apiKey: options?.apiKey, + apiKey, temperature: options?.temperature, topP: options?.topP, topK: options?.topK, diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index f271bbc81..91abe9dc5 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -1,4 +1,4 @@ -import type { Effort } from "../../model-thinking"; +import type { Effort } from "../../effort"; import { requireSupportedEffort } from "../../model-thinking"; import type { Api, Model } from "../../types"; @@ -76,6 +76,83 @@ function filterInput(input: InputItem[] | undefined): InputItem[] | undefined { }); } +const CODEX_ORPHAN_OUTPUT_LIMIT = 16_000; +/** Placeholder output for a tool call whose result never landed in the input. */ +const CODEX_INTERRUPTED_TOOL_OUTPUT = + "[No tool output recorded: the tool call was interrupted before it produced a result.]"; + +function orphanFunctionOutputToMessage(item: InputItem, callId: string): InputItem { + const itemRecord = item as unknown as Record; + const toolName = typeof itemRecord.name === "string" ? itemRecord.name : "tool"; + let text = ""; + try { + const output = itemRecord.output; + text = typeof output === "string" ? output : JSON.stringify(output); + } catch { + text = String(itemRecord.output ?? ""); + } + if (text.length > CODEX_ORPHAN_OUTPUT_LIMIT) { + text = `${text.slice(0, CODEX_ORPHAN_OUTPUT_LIMIT)}\n...[truncated]`; + } + return { + type: "message", + role: "assistant", + content: `[Previous ${toolName} result; call_id=${callId}]: ${text}`, + } as InputItem; +} + +/** + * Repair both halves of unpaired tool exchanges so the Responses input grammar + * stays valid — the API rejects either orphan with a 400: + * + * - `function_call_output` with no matching `function_call` → folded into an + * assistant message (`400 No tool call found for function call output …`). + * Regression of #472 / #1351. + * - `function_call` / `custom_tool_call` with no matching `*_output` → a + * placeholder output is synthesized immediately after the call + * (`400 No tool output found for function call …`). Hit when the user + * branches/navigates the session tree to a node that ends on a tool call (the + * tool-result child is dropped from the reconstructed history) or when a turn + * is aborted/crashes after the call streamed but before its result persisted. + */ +function repairToolCallPairs(input: InputItem[]): InputItem[] { + const callIds = new Set(); + const outputCallIds = new Set(); + for (const item of input) { + const callId = typeof item.call_id === "string" ? item.call_id : undefined; + if (callId === undefined) continue; + if (item.type === "function_call" || item.type === "custom_tool_call") callIds.add(callId); + else if (item.type === "function_call_output" || item.type === "custom_tool_call_output") { + outputCallIds.add(callId); + } + } + + const repaired: InputItem[] = []; + for (const item of input) { + const callId = typeof item.call_id === "string" ? item.call_id : undefined; + + if (item.type === "function_call_output" && callId !== undefined && !callIds.has(callId)) { + repaired.push(orphanFunctionOutputToMessage(item, callId)); + continue; + } + + repaired.push(item); + + if ( + (item.type === "function_call" || item.type === "custom_tool_call") && + callId !== undefined && + !outputCallIds.has(callId) + ) { + repaired.push({ + type: item.type === "custom_tool_call" ? "custom_tool_call_output" : "function_call_output", + call_id: callId, + output: CODEX_INTERRUPTED_TOOL_OUTPUT, + } as InputItem); + } + } + return repaired; +} + export async function transformRequestBody( body: RequestBody, model: Model, @@ -87,39 +164,8 @@ export async function transformRequestBody( if (body.input && Array.isArray(body.input)) { body.input = filterInput(body.input); - if (body.input) { - const functionCallIds = new Set( - body.input - .filter(item => item.type === "function_call" && typeof item.call_id === "string") - .map(item => item.call_id as string), - ); - - body.input = body.input.map(item => { - if (item.type === "function_call_output" && typeof item.call_id === "string") { - const callId = item.call_id as string; - if (!functionCallIds.has(callId)) { - const itemRecord = item as unknown as Record; - const toolName = typeof itemRecord.name === "string" ? itemRecord.name : "tool"; - let text = ""; - try { - const output = itemRecord.output; - text = typeof output === "string" ? output : JSON.stringify(output); - } catch { - text = String(itemRecord.output ?? ""); - } - if (text.length > 16000) { - text = `${text.slice(0, 16000)}\n...[truncated]`; - } - return { - type: "message", - role: "assistant", - content: `[Previous ${toolName} result; call_id=${callId}]: ${text}`, - } as InputItem; - } - } - return item; - }); + body.input = repairToolCallPairs(body.input); } } diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index b8b5e591c..4816e492b 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -10,7 +10,8 @@ import type { ChatCompletionToolMessageParam, } from "openai/resources/chat/completions"; import packageJson from "../../package.json" with { type: "json" }; -import { type Effort, getSupportedEfforts } from "../model-thinking"; +import type { Effort } from "../effort"; +import { getSupportedEfforts } from "../model-thinking"; import { calculateCost } from "../models"; import { getEnvApiKey } from "../stream"; import { @@ -536,7 +537,6 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = ( } stream.push({ type: "start", partial: output }); - const parseMiniMaxThinkTags = model.provider === "minimax-code" || model.provider === "minimax-code-cn"; // Some OpenAI-compatible DeepSeek hosts (including NVIDIA NIM and DeepSeek's // native API) leak chat-template tool-call markers in `delta.content` even // though tool calls are also surfaced structurally. Strip the leaked markers @@ -677,9 +677,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = ( } }; - const streamMarkupHealingPattern = getStreamMarkupHealingPattern(model.provider, model.id, { - parseThinkingTags: parseMiniMaxThinkTags, - }); + const streamMarkupHealingPattern = getStreamMarkupHealingPattern(model.provider, model.id); const streamMarkupHealing = streamMarkupHealingPattern ? new StreamMarkupHealing({ pattern: streamMarkupHealingPattern }) : undefined; @@ -1344,6 +1342,11 @@ function buildParams( if (compat.extraBody) { Object.assign(params, compat.extraBody); + if (model.provider === "fireworks" && params.reasoning_effort !== undefined) { + // Fireworks rejects simultaneous DeepSeek-style `thinking` toggles and + // OpenAI-style `reasoning_effort`; the effort field carries the user's level. + delete params.thinking; + } } return { params, toolStrictMode }; diff --git a/packages/ai/src/providers/openai-responses-shared.ts b/packages/ai/src/providers/openai-responses-shared.ts index 2b0f5f2b3..28532dcc8 100644 --- a/packages/ai/src/providers/openai-responses-shared.ts +++ b/packages/ai/src/providers/openai-responses-shared.ts @@ -212,6 +212,59 @@ export function repairOrphanResponsesToolOutputs(input: ResponseInput): Response }); } +/** Placeholder output for a tool call whose result is absent from the input. */ +const ORPHAN_TOOL_CALL_PLACEHOLDER = + "[No tool output recorded: the tool call was interrupted before it produced a result.]"; + +/** + * Synthesize a placeholder `function_call_output` / `custom_tool_call_output` + * for every `function_call` / `custom_tool_call` whose `call_id` has no matching + * output later in the same input. The Responses API rejects an unpaired call + * with `400 No tool output found for function call …`. + * + * Orphan calls surface when the user branches/navigates the session tree to a + * node that ends on a tool call (the tool-result child is excluded from the + * reconstructed history) or when a turn is aborted/crashes after the call + * streamed but before its result persisted. Dropping the call would erase the + * assistant's action; a placeholder output keeps the call visible so the model + * can recover (e.g. re-issue the call). Symmetric to + * {@link repairOrphanResponsesToolOutputs}. + */ +export function repairOrphanResponsesToolCalls(input: ResponseInput): ResponseInput { + const outputCallIds = new Set(); + for (const item of input) { + const t = (item as { type?: string }).type; + if (t !== "function_call_output" && t !== "custom_tool_call_output") continue; + const callId = (item as { call_id?: unknown }).call_id; + if (typeof callId === "string") outputCallIds.add(callId); + } + let hasOrphan = false; + for (const item of input) { + const t = (item as { type?: string }).type; + if (t !== "function_call" && t !== "custom_tool_call") continue; + const callId = (item as { call_id?: unknown }).call_id; + if (typeof callId === "string" && !outputCallIds.has(callId)) { + hasOrphan = true; + break; + } + } + if (!hasOrphan) return input; + const repaired: ResponseInput = []; + for (const item of input) { + repaired.push(item); + const t = (item as { type?: string }).type; + if (t !== "function_call" && t !== "custom_tool_call") continue; + const callId = (item as { call_id?: unknown }).call_id; + if (typeof callId !== "string" || outputCallIds.has(callId)) continue; + repaired.push({ + type: t === "custom_tool_call" ? "custom_tool_call_output" : "function_call_output", + call_id: callId, + output: ORPHAN_TOOL_CALL_PLACEHOLDER, + } as ResponseInput[number]); + } + return repaired; +} + export function convertResponsesInputContent( content: string | Array, supportsImages: boolean, @@ -406,6 +459,13 @@ export async function processResponsesStream( // see https://github.com/can1357/oh-my-pi/issues/1880 — llama.cpp emits parallel // function_call deltas interleaved, and a singleton `current` reference would // fold them into the wrong block and drop arguments on every call but the last. + // + // llama.cpp's `to_json_oaicompat_resp` (issue #2015) compounds this: `output_item.added` + // for function_call/custom_tool_call carries `item.call_id` but no `item.id` and no + // `output_index`, while the matching `function_call_arguments.delta` carries + // `item_id = "fc_"`. Registering function-call items by `call_id` as a + // secondary key lets the delta lookup find the right block on hosts that emit one + // identifier but not the other. const openItemsByOutputIndex = new Map(); const openItemsByItemId = new Map(); let lastOpenItem: StreamingItem | null = null; @@ -415,9 +475,11 @@ export async function processResponsesStream( outputIndex: number | undefined, itemId: string | undefined, entry: StreamingItem, + alternateItemKey?: string, ): void => { if (typeof outputIndex === "number") openItemsByOutputIndex.set(outputIndex, entry); if (itemId) openItemsByItemId.set(itemId, entry); + if (alternateItemKey && alternateItemKey !== itemId) openItemsByItemId.set(alternateItemKey, entry); openItemsInOrder.push(entry); lastOpenItem = entry; }; @@ -455,9 +517,11 @@ export async function processResponsesStream( outputIndex: number | undefined, itemId: string | undefined, entry: StreamingItem | undefined, + alternateItemKey?: string, ): void => { if (typeof outputIndex === "number") openItemsByOutputIndex.delete(outputIndex); if (itemId) openItemsByItemId.delete(itemId); + if (alternateItemKey && alternateItemKey !== itemId) openItemsByItemId.delete(alternateItemKey); if (entry) { const index = openItemsInOrder.indexOf(entry); if (index >= 0) openItemsInOrder.splice(index, 1); @@ -497,7 +561,7 @@ export async function processResponsesStream( partialJson: item.arguments || "", }; output.content.push(block); - registerOpenItem(event.output_index, item.id, { item, block }); + registerOpenItem(event.output_index, item.id, { item, block }, item.call_id); stream.push({ type: "toolcall_start", contentIndex: contentIndexOf(block), partial: output }); } else if (item.type === "custom_tool_call") { const block: StreamingToolCallBlock = { @@ -515,7 +579,7 @@ export async function processResponsesStream( partialJson: item.input ?? "", }; output.content.push(block); - registerOpenItem(event.output_index, item.id, { item, block }); + registerOpenItem(event.output_index, item.id, { item, block }, item.call_id); stream.push({ type: "toolcall_start", contentIndex: contentIndexOf(block), partial: output }); } } else if (event.type === "response.reasoning_summary_part.added") { @@ -656,7 +720,10 @@ export async function processResponsesStream( } else if (event.type === "response.output_item.done") { const item = structuredCloneJSON(event.item); options?.onOutputItemDone?.(item); - const entry = lookupOpenItem({ output_index: event.output_index, item_id: item.id }); + const entry = + item.type === "function_call" || item.type === "custom_tool_call" + ? lookupOpenItem({ output_index: event.output_index, item_id: item.id ?? item.call_id }) + : lookupOpenItem({ output_index: event.output_index, item_id: item.id }); if (item.type === "reasoning") { const thinking = item.summary?.length > 0 @@ -715,7 +782,7 @@ export async function processResponsesStream( delete (block as { argumentsDone?: boolean }).argumentsDone; } const contentIndex = block ? contentIndexOf(block) : output.content.length - 1; - closeOpenItem(event.output_index, item.id, entry); + closeOpenItem(event.output_index, item.id, entry, item.call_id); stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); } else if (item.type === "custom_tool_call") { const block = entry?.block.type === "toolCall" ? entry.block : undefined; @@ -728,7 +795,7 @@ export async function processResponsesStream( customWireName: item.name, }; const contentIndex = block ? contentIndexOf(block) : output.content.length - 1; - closeOpenItem(event.output_index, item.id, entry); + closeOpenItem(event.output_index, item.id, entry, item.call_id); stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); } } else if (event.type === "response.completed") { diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index ac1684b43..ed6de0481 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -62,6 +62,7 @@ import { isOpenAIResponsesProgressEvent, normalizeResponsesToolCallIdForTransform, processResponsesStream, + repairOrphanResponsesToolCalls, repairOrphanResponsesToolOutputs, } from "./openai-responses-shared"; import { transformMessages } from "./transform-messages"; @@ -614,7 +615,7 @@ function convertConversationMessages( msgIndex++; } - return repairOrphanResponsesToolOutputs(messages); + return repairOrphanResponsesToolCalls(repairOrphanResponsesToolOutputs(messages)); } /** diff --git a/packages/ai/src/providers/pi-native-client.ts b/packages/ai/src/providers/pi-native-client.ts index b5df79636..68f479d2a 100644 --- a/packages/ai/src/providers/pi-native-client.ts +++ b/packages/ai/src/providers/pi-native-client.ts @@ -149,7 +149,10 @@ export function streamPiNative( try { const url = resolveStreamUrl(model as Model); const fetchImpl = options?.fetch ?? globalThis.fetch; - const headers = buildHeaders(model as Model, options?.apiKey); + const headers = buildHeaders( + model as Model, + typeof options?.apiKey === "string" ? options.apiKey : undefined, + ); const body = JSON.stringify({ modelId: model.id, context, diff --git a/packages/ai/src/providers/transform-messages.ts b/packages/ai/src/providers/transform-messages.ts index 7959f7828..096b454e7 100644 --- a/packages/ai/src/providers/transform-messages.ts +++ b/packages/ai/src/providers/transform-messages.ts @@ -17,6 +17,96 @@ const enum ToolCallStatus { Aborted = 2, } +/** + * Maximum tool-call id length the strictest replay provider accepts. + * + * Anthropic requires `^[a-zA-Z0-9_-]+$` with a 64-char cap; Google and Codex + * `normalizeToolCallId` implementations cap individual id segments to the same + * 64-char ceiling. Replacement ids minted here flow back through + * `convertAnthropicMessages` (and friends) unchanged, so the `_dupN` suffix + * MUST not push a normalized id past this bound. + */ +const MAX_TOOL_CALL_ID_LENGTH = 64; + +function appendDuplicateSuffix(originalId: string, suffix: string): string { + if (originalId.length + suffix.length <= MAX_TOOL_CALL_ID_LENGTH) return `${originalId}${suffix}`; + const prefixBudget = Math.max(0, MAX_TOOL_CALL_ID_LENGTH - suffix.length); + return `${originalId.slice(0, prefixBudget)}${suffix}`; +} + +type PendingToolResultRewrite = { replacementId: string } | undefined; + +function deduplicateToolCallIds(messages: Message[]): Message[] { + const seenToolCallIds = new Map(); + const pendingToolResultRewrites = new Map(); + + return messages.map(msg => { + if (msg.role === "toolResult") { + const rewrites = pendingToolResultRewrites.get(msg.toolCallId); + if (!rewrites || rewrites.length === 0) return msg; + + const rewrite = rewrites.shift(); + if (rewrites.length === 0) pendingToolResultRewrites.delete(msg.toolCallId); + if (rewrite) return { ...msg, toolCallId: rewrite.replacementId }; + return msg; + } + + if (msg.role !== "assistant") return msg; + + const enqueueToolResultRewrite = (id: string, rewrite: PendingToolResultRewrite): void => { + const rewrites = pendingToolResultRewrites.get(id); + if (rewrites) { + rewrites.push(rewrite); + return; + } + pendingToolResultRewrites.set(id, [rewrite]); + }; + + // Ids this turn has already touched; used to scope the "drop carried-over + // pending rewrites" semantics to the FIRST occurrence per turn so multiple + // blocks of the same id within one turn still accumulate as duplicates. + const idsTouchedInTurn = new Set(); + let contentChanged = false; + const content = msg.content.map(block => { + if (block.type !== "toolCall") return block; + + // Drop any pending rewrites carried over from a prior assistant turn + // for this id on its first appearance this turn. When a later turn + // re-emits the same id, the older duplicate call's expected result + // never landed in time — the second pass synthesizes + // "No result provided" for it, and the upcoming real result(id) must + // route to one of THIS turn's calls. Without this guard the older + // `_dup` id would steal the next result. + if (!idsTouchedInTurn.has(block.id)) { + pendingToolResultRewrites.delete(block.id); + idsTouchedInTurn.add(block.id); + } + + const previousCount = seenToolCallIds.get(block.id) ?? 0; + if (previousCount === 0) { + seenToolCallIds.set(block.id, 1); + enqueueToolResultRewrite(block.id, undefined); + return block; + } + + let duplicateIndex = previousCount; + let replacementId = appendDuplicateSuffix(block.id, `_dup${duplicateIndex}`); + while (seenToolCallIds.has(replacementId)) { + duplicateIndex += 1; + replacementId = appendDuplicateSuffix(block.id, `_dup${duplicateIndex}`); + } + seenToolCallIds.set(block.id, duplicateIndex + 1); + seenToolCallIds.set(replacementId, 1); + enqueueToolResultRewrite(block.id, { replacementId }); + contentChanged = true; + return { ...block, id: replacementId }; + }); + + if (!contentChanged) return msg; + return { ...msg, content }; + }); +} + function shouldDropTruncatedThinkingOnlyAssistant(msg: AssistantMessage): boolean { const isTruncatedStop = msg.stopReason === "length" || msg.stopReason === "error" || msg.stopReason === "aborted"; return isTruncatedStop && !msg.content.some(block => block.type === "toolCall" || block.type === "text"); @@ -52,116 +142,120 @@ export function transformMessages( const latestSurvivingAssistantIndex = getLatestSurvivingAssistantIndex(messages); // First pass: transform messages (thinking blocks, tool call ID normalization) - const transformed = messages.map((msg, index) => { - // User and developer messages pass through unchanged - if (msg.role === "user" || msg.role === "developer") { - return msg; - } + const transformed = deduplicateToolCallIds( + messages.map((msg, index) => { + // User and developer messages pass through unchanged + if (msg.role === "user" || msg.role === "developer") { + return msg; + } - // Handle toolResult messages - normalize toolCallId if we have a mapping - if (msg.role === "toolResult") { - const normalizedId = toolCallIdMap.get(msg.toolCallId); - if (normalizedId && normalizedId !== msg.toolCallId) { - return { ...msg, toolCallId: normalizedId }; + // Handle toolResult messages - normalize toolCallId if we have a mapping + if (msg.role === "toolResult") { + const normalizedId = toolCallIdMap.get(msg.toolCallId); + if (normalizedId && normalizedId !== msg.toolCallId) { + return { ...msg, toolCallId: normalizedId }; + } + return msg; + } + + // Assistant messages need transformation check + if (msg.role === "assistant") { + const assistantMsg = msg as AssistantMessage; + const isSameModel = + assistantMsg.provider === model.provider && + assistantMsg.api === model.api && + assistantMsg.model === model.id; + + const mustPreserveLatestAnthropicThinking = + index === latestSurvivingAssistantIndex && + model.api === "anthropic-messages" && + assistantMsg.api === "anthropic-messages"; + // Aborted/errored messages may have partially-streamed thinking signatures. + // A partial signature is invalid and will be rejected by the API, so we must + // strip signatures from thinking blocks in these messages. + // + // Abandoned tool-use turns get the same treatment once they are no longer + // the latest assistant message. When a turn carries toolCall blocks but did + // NOT request tool execution (stopReason !== "toolUse" — e.g. + // adaptive-thinking Opus emitting tool calls and then ending the turn on + // `end_turn`/`stop`), the agent loop pairs those calls with placeholder + // tool_results to keep the tool_use/tool_result contract valid. Historical + // abandoned turns cannot safely replay their end_turn-bound signatures in + // that continuation, so stripping downgrades them to plain text downstream. + // Latest abandoned turns are exempt because Anthropic requires thinking + // blocks from its most recent response to remain byte-for-byte unmodified. + const invalidStopReason = assistantMsg.stopReason === "aborted" || assistantMsg.stopReason === "error"; + const abandonedToolUse = + !invalidStopReason && + assistantMsg.stopReason !== "toolUse" && + assistantMsg.content.some(b => b.type === "toolCall"); + const hasInvalidSignatures = invalidStopReason || abandonedToolUse; + + const transformedContent = assistantMsg.content.flatMap(block => { + if (block.type === "thinking") { + // Strip untrustworthy signatures so the encoder can downgrade to text. + const sanitized = + hasInvalidSignatures && block.thinkingSignature + ? { ...block, thinkingSignature: undefined } + : block; + if (mustPreserveLatestAnthropicThinking) return abandonedToolUse ? block : sanitized; + // For same model: keep thinking blocks with signatures (needed for replay) + // even if the thinking text is empty (OpenAI encrypted reasoning) + if (isSameModel && sanitized.thinkingSignature) return sanitized; + // Skip empty thinking blocks, convert others to plain text + if (!sanitized.thinking || sanitized.thinking.trim() === "") return []; + if (isSameModel) return sanitized; + return { + type: "text" as const, + text: sanitized.thinking, + }; + } + + if (block.type === "redactedThinking") { + if (mustPreserveLatestAnthropicThinking) return block; + if (isSameModel) return block; + return []; + } + + if (block.type === "text") { + if (isSameModel) return block; + return { + type: "text" as const, + text: block.text, + }; + } + + if (block.type === "toolCall") { + const toolCall = block as ToolCall; + let normalizedToolCall: ToolCall = toolCall; + + if (!isSameModel && toolCall.thoughtSignature) { + normalizedToolCall = { ...toolCall }; + delete (normalizedToolCall as { thoughtSignature?: string }).thoughtSignature; + } + + if (!isSameModel && normalizeToolCallId) { + const normalizedId = normalizeToolCallId(toolCall.id, model, assistantMsg); + if (normalizedId !== toolCall.id) { + toolCallIdMap.set(toolCall.id, normalizedId); + normalizedToolCall = { ...normalizedToolCall, id: normalizedId }; + } + } + + return normalizedToolCall; + } + + return block; + }); + + return { + ...assistantMsg, + content: transformedContent, + }; } return msg; - } - - // Assistant messages need transformation check - if (msg.role === "assistant") { - const assistantMsg = msg as AssistantMessage; - const isSameModel = - assistantMsg.provider === model.provider && - assistantMsg.api === model.api && - assistantMsg.model === model.id; - - const mustPreserveLatestAnthropicThinking = - index === latestSurvivingAssistantIndex && - model.api === "anthropic-messages" && - assistantMsg.api === "anthropic-messages"; - // Aborted/errored messages may have partially-streamed thinking signatures. - // A partial signature is invalid and will be rejected by the API, so we must - // strip signatures from thinking blocks in these messages. - // - // Abandoned tool-use turns get the same treatment once they are no longer - // the latest assistant message. When a turn carries toolCall blocks but did - // NOT request tool execution (stopReason !== "toolUse" — e.g. - // adaptive-thinking Opus emitting tool calls and then ending the turn on - // `end_turn`/`stop`), the agent loop pairs those calls with placeholder - // tool_results to keep the tool_use/tool_result contract valid. Historical - // abandoned turns cannot safely replay their end_turn-bound signatures in - // that continuation, so stripping downgrades them to plain text downstream. - // Latest abandoned turns are exempt because Anthropic requires thinking - // blocks from its most recent response to remain byte-for-byte unmodified. - const invalidStopReason = assistantMsg.stopReason === "aborted" || assistantMsg.stopReason === "error"; - const abandonedToolUse = - !invalidStopReason && - assistantMsg.stopReason !== "toolUse" && - assistantMsg.content.some(b => b.type === "toolCall"); - const hasInvalidSignatures = invalidStopReason || abandonedToolUse; - - const transformedContent = assistantMsg.content.flatMap(block => { - if (block.type === "thinking") { - // Strip untrustworthy signatures so the encoder can downgrade to text. - const sanitized = - hasInvalidSignatures && block.thinkingSignature ? { ...block, thinkingSignature: undefined } : block; - if (mustPreserveLatestAnthropicThinking) return abandonedToolUse ? block : sanitized; - // For same model: keep thinking blocks with signatures (needed for replay) - // even if the thinking text is empty (OpenAI encrypted reasoning) - if (isSameModel && sanitized.thinkingSignature) return sanitized; - // Skip empty thinking blocks, convert others to plain text - if (!sanitized.thinking || sanitized.thinking.trim() === "") return []; - if (isSameModel) return sanitized; - return { - type: "text" as const, - text: sanitized.thinking, - }; - } - - if (block.type === "redactedThinking") { - if (mustPreserveLatestAnthropicThinking) return block; - if (isSameModel) return block; - return []; - } - - if (block.type === "text") { - if (isSameModel) return block; - return { - type: "text" as const, - text: block.text, - }; - } - - if (block.type === "toolCall") { - const toolCall = block as ToolCall; - let normalizedToolCall: ToolCall = toolCall; - - if (!isSameModel && toolCall.thoughtSignature) { - normalizedToolCall = { ...toolCall }; - delete (normalizedToolCall as { thoughtSignature?: string }).thoughtSignature; - } - - if (!isSameModel && normalizeToolCallId) { - const normalizedId = normalizeToolCallId(toolCall.id, model, assistantMsg); - if (normalizedId !== toolCall.id) { - toolCallIdMap.set(toolCall.id, normalizedId); - normalizedToolCall = { ...normalizedToolCall, id: normalizedId }; - } - } - - return normalizedToolCall; - } - - return block; - }); - - return { - ...assistantMsg, - content: transformedContent, - }; - } - return msg; - }); + }), + ); const realToolResultsById = new Map(); for (const msg of transformed) { if (msg.role === "toolResult" && !realToolResultsById.has(msg.toolCallId)) { diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 437a49344..f44c0a5b4 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -3,7 +3,8 @@ import * as os from "node:os"; import * as path from "node:path"; import { $env, $pickenv, extractHttpStatusFromError } from "@oh-my-pi/pi-utils"; import { getCustomApi } from "./api-registry"; -import type { Effort } from "./model-thinking"; +import { AUTH_RETRY_STEPS, isApiKeyResolver, resolveRetryKey } from "./auth-retry"; +import type { Effort } from "./effort"; import { mapEffortToAnthropicAdaptiveEffort, mapEffortToGoogleThinkingLevel, @@ -208,8 +209,7 @@ const serviceProviderMap: Record = { tavily: "TAVILY_API_KEY", parallel: "PARALLEL_API_KEY", kagi: "KAGI_API_KEY", - // GitHub Copilot uses GitHub personal access token - "github-copilot": () => $pickenv("COPILOT_GITHUB_TOKEN", "GH_TOKEN", "GITHUB_TOKEN"), + "github-copilot": "COPILOT_GITHUB_TOKEN", // Foundry mode optionally switches Anthropic auth to enterprise gateway credentials. anthropic: () => isFoundryEnabled() @@ -443,12 +443,15 @@ export function streamSimple( options?: SimpleStreamOptions, ): AssistantMessageEventStream { const requestOptions = withRequestDebugFetch(options); - const retryApiKey = requestOptions?.onAuthError - ? (requestOptions.apiKey ?? getEnvApiKey(model.provider)) - : undefined; - if (retryApiKey) { + const apiKeyResolver = isApiKeyResolver(requestOptions?.apiKey) ? requestOptions.apiKey : undefined; + if (apiKeyResolver) { const outer = new AssistantMessageEventStream(); - const onAuthError = requestOptions!.onAuthError!; + const signal = requestOptions?.signal; + // One inner attempt against a resolved string key. When + // `captureAuthFailure` is set, a retryable auth error that arrives before + // any replay-unsafe event is buffered and returned (so the caller can + // retry with a fresh key) instead of surfaced. The terminal attempt + // clears the flag and emits whatever it gets. const runAttempt = async (apiKey: string, captureAuthFailure: boolean): Promise => { const bufferedEvents: AssistantMessageEvent[] = []; let emittedReplayUnsafeEvent = false; @@ -458,7 +461,7 @@ export function streamSimple( }; try { - const inner = streamSimple(model, context, { ...requestOptions, apiKey, onAuthError: undefined }); + const inner = streamSimple(model, context, { ...requestOptions, apiKey }); for await (const event of inner) { if (!emittedReplayUnsafeEvent && event.type === "start") { bufferedEvents.push(event); @@ -510,19 +513,32 @@ export function streamSimple( }; void (async () => { - const failure = await runAttempt(retryApiKey, true); - if (!failure) return; - let nextKey: string | undefined; + let lastKey: string | undefined; try { - nextKey = await onAuthError(model.provider, retryApiKey, failure.error); + lastKey = (await apiKeyResolver({ lastChance: false, error: undefined, signal })) || undefined; } catch { - nextKey = undefined; + lastKey = undefined; } - if (!nextKey || nextKey === retryApiKey) { - emitFailure(failure); + if (lastKey === undefined) { + outer.fail(new Error(`No API key for provider: ${model.provider}`)); return; } - await runAttempt(nextKey, false); + let failure = await runAttempt(lastKey, true); + if (!failure) return; + // a/b/c policy: refresh the same account (lastChance=false), then + // switch to a sibling (lastChance=true). A step is skipped when the + // resolver yields the same key it just tried or `undefined`; the + // final step's attempt clears the capture flag so it emits directly. + for (let step = 0; step < AUTH_RETRY_STEPS.length; step++) { + const nextKey = await resolveRetryKey(apiKeyResolver, AUTH_RETRY_STEPS[step]!, failure.error, signal); + if (nextKey === undefined || nextKey === lastKey) continue; + lastKey = nextKey; + const isLastStep = step === AUTH_RETRY_STEPS.length - 1; + const next = await runAttempt(nextKey, !isLastStep); + if (!next) return; + failure = next; + } + emitFailure(failure); })(); return outer; } @@ -553,7 +569,10 @@ export function streamSimple( return stream(model, context, providerOptions); } - const apiKey = requestOptions?.apiKey || getEnvApiKey(model.provider); + // The resolver form is handled by the wrapper above; only a static string + // key reaches this point. + const apiKey = + (typeof requestOptions?.apiKey === "string" ? requestOptions.apiKey : undefined) || getEnvApiKey(model.provider); if (!apiKey) { throw new Error(`No API key for provider: ${model.provider}`); } @@ -724,7 +743,7 @@ function mapOptionsForApi( repetitionPenalty: options?.repetitionPenalty, maxTokens: options?.maxTokens ?? model.maxTokens, signal: options?.signal, - apiKey: apiKey || options?.apiKey, + apiKey: apiKey ?? (typeof options?.apiKey === "string" ? options.apiKey : undefined), cacheRetention: options?.cacheRetention, headers: options?.headers, initiatorOverride: options?.initiatorOverride, diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index 9b03d99d8..566574be4 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -1,4 +1,5 @@ import type { ZodType, z } from "zod/v4"; +import type { ApiKey } from "./auth-retry"; import type { BedrockOptions } from "./providers/amazon-bedrock"; import type { AnthropicOptions } from "./providers/anthropic"; import type { AzureOpenAIResponsesOptions } from "./providers/azure-openai-responses"; @@ -151,7 +152,7 @@ export type KnownProvider = | "lm-studio"; export type Provider = KnownProvider | string; -import type { Effort } from "./model-thinking"; +import type { Effort } from "./effort"; /** Token budgets for each thinking level (token-based providers only) */ export type ThinkingBudgets = { [key in Effort]?: number }; @@ -298,12 +299,6 @@ export interface StreamOptions { maxTokens?: number; signal?: AbortSignal; apiKey?: string; - /** - * Called when a provider returns 401 before any replay-unsafe assistant - * event has been emitted. Returning a different key retries the provider - * request once. - */ - onAuthError?: (provider: string, apiKey: string, error: unknown) => Promise; cacheRetention?: CacheRetention; /** * Additional headers to include in provider requests. @@ -415,7 +410,15 @@ export interface StreamOptions { } // Unified options with reasoning passed to streamSimple() and completeSimple() -export interface SimpleStreamOptions extends StreamOptions { +export interface SimpleStreamOptions extends Omit { + /** + * API key for the request: either a static bearer string, or an + * {@link ApiKeyResolver} that mints/rotates the key across the central + * a/b/c auth-retry policy. `streamSimple`/`completeSimple` resolve a + * resolver to a string before per-provider dispatch, so providers only + * ever see the resolved {@link StreamOptions.apiKey} string. + */ + apiKey?: ApiKey; reasoning?: Effort; /** * Force-disable reasoning for the request even when the model supports it. diff --git a/packages/ai/src/usage/google-antigravity.ts b/packages/ai/src/usage/google-antigravity.ts index 26963d7b5..54e2e11cf 100644 --- a/packages/ai/src/usage/google-antigravity.ts +++ b/packages/ai/src/usage/google-antigravity.ts @@ -148,10 +148,17 @@ async function fetchAntigravityUsage(params: UsageFetchParams, ctx: UsageFetchCo } const data = (await response.json()) as AntigravityUsageResponse; - const limits: UsageLimit[] = []; + + // The API returns per-model quota entries, but quota is shared across + // models within the same tier. Deduplicate by (tier, windowId) so one + // account doesn't produce 15 redundant bars. + const deduped = new Map< + string, + { amount: UsageAmount; window: UsageWindow | undefined; tier: string | undefined } + >(); let earliestReset: number | undefined; - for (const [modelId, modelInfo] of Object.entries(data.models ?? {})) { + for (const [_modelId, modelInfo] of Object.entries(data.models ?? {})) { const quotaInfos = normalizeQuotaInfos(modelInfo); for (const quotaInfo of quotaInfos) { const amount = buildAmount(quotaInfo); @@ -159,35 +166,81 @@ async function fetchAntigravityUsage(params: UsageFetchParams, ctx: UsageFetchCo if (window?.resetsAt) { earliestReset = earliestReset ? Math.min(earliestReset, window.resetsAt) : window.resetsAt; } - const labelBase = modelInfo.displayName || modelId; - const label = quotaInfo.tier ? `${labelBase} (${quotaInfo.tier})` : labelBase; - const windowId = window?.id ?? "default"; - limits.push({ - id: `${modelId}:${quotaInfo.tier ?? "default"}:${windowId}`, - label, - scope: { - provider: params.provider, - accountId: credential.accountId, - projectId: credential.projectId, - modelId, - tier: quotaInfo.tier, - windowId, - }, - window, - amount, - status: getUsageStatus(amount.remainingFraction), - }); + const tier = (quotaInfo.tier ?? "default").toLowerCase(); + // Use quotaInfo.windowId even when parseWindow returns undefined + // (no resetTime) — separate windows must not collapse to "default". + const windowId = quotaInfo.windowId ?? window?.id ?? "default"; + const key = `${tier}|${windowId}`; + const existing = deduped.get(key); + if (!existing) { + deduped.set(key, { amount, window, tier: quotaInfo.tier }); + continue; + } + // Merge: keep the entry with fraction data for the bar, but + // also keep any window with a reset time so "resets in…" survives. + const eFrac = existing.amount.remainingFraction; + const cFrac = amount.remainingFraction; + const eHasFrac = eFrac !== undefined; + const cHasFrac = cFrac !== undefined; + + let bestAmount = existing.amount; + let bestWindow = existing.window?.resetsAt ? existing.window : (window ?? existing.window); + let bestTier = existing.tier ?? quotaInfo.tier; + + if (!eHasFrac && cHasFrac) { + bestAmount = amount; + bestTier = quotaInfo.tier ?? existing.tier; + } else if (eHasFrac && cHasFrac && cFrac! < eFrac!) { + bestAmount = amount; + bestTier = quotaInfo.tier ?? existing.tier; + } + // Always merge in window with reset time if the current + // best doesn't have one. + if (!bestWindow?.resetsAt && window?.resetsAt) { + bestWindow = window; + } + deduped.set(key, { amount: bestAmount, window: bestWindow, tier: bestTier }); } } + const limits: UsageLimit[] = []; + for (const [key, entry] of deduped) { + const [tier, windowId] = key.split("|") as [string, string]; + const label = "Usage"; + limits.push({ + id: `${params.provider}:${tier}:${windowId}`, + label, + scope: { + provider: params.provider, + accountId: credential.accountId, + projectId: credential.projectId, + tier: entry.tier, + windowId, + }, + window: entry.window, + amount: entry.amount, + status: getUsageStatus(entry.amount.remainingFraction), + }); + } + + limits.sort((a, b) => { + const aFraction = a.amount.remainingFraction ?? 1; + const bFraction = b.amount.remainingFraction ?? 1; + return aFraction - bFraction; + }); + + const metadata: UsageReport["metadata"] = { + endpoint: url, + projectId: credential.projectId, + }; + if (credential.email) metadata.email = credential.email; + if (credential.accountId) metadata.accountId = credential.accountId; + const report: UsageReport = { provider: params.provider, fetchedAt: nowMs, limits, - metadata: { - endpoint: url, - projectId: credential.projectId, - }, + metadata, raw: data, }; diff --git a/packages/ai/src/utils/http-inspector.ts b/packages/ai/src/utils/http-inspector.ts index 31a6ccb06..df738c3d8 100644 --- a/packages/ai/src/utils/http-inspector.ts +++ b/packages/ai/src/utils/http-inspector.ts @@ -1,5 +1,5 @@ import * as path from "node:path"; -import { extractHttpStatusFromError, getLogsDir } from "@oh-my-pi/pi-utils"; +import { extractHttpStatusFromError, getLogsDir, isBunTestRuntime } from "@oh-my-pi/pi-utils"; import { isCopilotTransientModelError } from "./retry.js"; import { formatErrorMessageWithRetryAfter } from "./retry-after.js"; @@ -31,7 +31,9 @@ export async function appendRawHttpRequestDumpFor400( error: unknown, dump: RawHttpRequestDump | undefined, ): Promise { - if (!dump || extractHttpStatusFromError(error) !== 400) { + // Never persist dumps under the test runner: providers exercise the 400 path + // with mocked fetch responses, which would otherwise litter the real ~/.omp logs. + if (!dump || isBunTestRuntime() || extractHttpStatusFromError(error) !== 400) { return message; } diff --git a/packages/ai/src/utils/request-debug.ts b/packages/ai/src/utils/request-debug.ts index 193f51be6..daa6831c4 100644 --- a/packages/ai/src/utils/request-debug.ts +++ b/packages/ai/src/utils/request-debug.ts @@ -170,6 +170,7 @@ class FileRequestDebugSession implements RequestDebugSession { class FileRequestDebugResponseLog implements RequestDebugResponseLog { #handle: fs.FileHandle | undefined; #pending: Promise = Promise.resolve(); + #closed: Promise | undefined; constructor(handle: fs.FileHandle) { this.#handle = handle; @@ -184,15 +185,19 @@ class FileRequestDebugResponseLog implements RequestDebugResponseLog { }); } - async close(): Promise { + close(): Promise { + if (this.#closed) return this.#closed; const handle = this.#handle; - if (!handle) return; + if (!handle) return Promise.resolve(); this.#handle = undefined; - try { - await this.#pending; - } finally { - await handle.close(); - } + this.#closed = (async () => { + try { + await this.#pending; + } finally { + await handle.close(); + } + })(); + return this.#closed; } } diff --git a/packages/ai/src/utils/schema/fields.ts b/packages/ai/src/utils/schema/fields.ts index 41e9aacf1..b25006248 100644 --- a/packages/ai/src/utils/schema/fields.ts +++ b/packages/ai/src/utils/schema/fields.ts @@ -154,6 +154,22 @@ export const CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS: Record = buildAllCcaTypeSpecificKeys(); + +function buildAllCcaTypeSpecificKeys(): Record { + const all: Record = {}; + for (const typeKeys of Object.values(CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS)) { + for (const key in typeKeys) { + all[key] = true; + } + } + return all; +} + /** * Cloud Code Assist shared schema keys allowed on any type. * Used alongside CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS for CCA combiner collapsing. diff --git a/packages/ai/src/utils/schema/json-schema-validator.ts b/packages/ai/src/utils/schema/json-schema-validator.ts index 93a69cf15..79a93052e 100644 --- a/packages/ai/src/utils/schema/json-schema-validator.ts +++ b/packages/ai/src/utils/schema/json-schema-validator.ts @@ -20,6 +20,14 @@ export interface JsonSchemaValidationIssue { message: string; expectedTypes?: string[]; keyword?: string; + /** + * Marks issues that originate inside a failed `anyOf` / `oneOf` branch. + * Consumers such as the tool-argument coercion layer use this to avoid + * applying type repairs (e.g. singleton-array wrapping) that would be + * authoritative outside of a combinator but are only one candidate + * branch's expectation here. + */ + fromUnionBranch?: boolean; } export interface JsonSchemaValidationResult { @@ -242,7 +250,17 @@ function validateSchemaNode( const branchValid = keyword === "anyOf" ? matches > 0 : matches === 1; if (!branchValid) { if (matches === 0 && firstIssues && firstIssues.length > 0) { - issues.push(...firstIssues); + // Only tag issues that sit at the combinator's own path as + // union-branch; deeper issues describe a specific field within + // the failed branch and should remain individually repairable. + const unionDepth = path.length; + for (const branchIssue of firstIssues) { + if (branchIssue.path.length === unionDepth) { + issues.push({ ...branchIssue, fromUnionBranch: true }); + } else { + issues.push(branchIssue); + } + } } else { pushIssue( issues, diff --git a/packages/ai/src/utils/schema/normalize.ts b/packages/ai/src/utils/schema/normalize.ts index 1b21afd67..ac50eccb7 100644 --- a/packages/ai/src/utils/schema/normalize.ts +++ b/packages/ai/src/utils/schema/normalize.ts @@ -11,6 +11,7 @@ import { dereferenceJsonSchema } from "./dereference"; import { upgradeJsonSchemaTo202012 } from "./draft"; import { areJsonValuesEqual, mergePropertySchemas } from "./equality"; import { + ALL_CCA_TYPE_SPECIFIC_KEYS, CLOUD_CODE_ASSIST_SHARED_SCHEMA_KEYS, CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS, COMBINATOR_KEYS, @@ -501,12 +502,32 @@ function collapseMixedTypeCombinerVariants(schema: JsonObject, combiner: "anyOf" if (variantTypes.length < 2 || variantTypes.every(type => type === "object")) { return schema; } - const nextSchema = copySchemaWithout(schema, combiner); const nonNullTypes = variantTypes.filter(t => t !== "null"); - nextSchema.type = nonNullTypes[0] ?? variantTypes[0]; + const chosenType: string = nonNullTypes[0] ?? variantTypes[0]; + nextSchema.type = chosenType; + const chosenTypeAllowedKeys = CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS[chosenType] ?? {}; + + // Strip sibling keys that were copied from the parent and belong to a + // different type (e.g. `items` sibling on a now-string-typed schema). + for (const key in nextSchema) { + if (!Object.hasOwn(nextSchema, key)) continue; + if (key === "type") continue; + if ( + Object.hasOwn(ALL_CCA_TYPE_SPECIFIC_KEYS, key) && + !Object.hasOwn(chosenTypeAllowedKeys, key) && + !Object.hasOwn(CLOUD_CODE_ASSIST_SHARED_SCHEMA_KEYS, key) + ) { + delete nextSchema[key]; + } + } + for (const key in mergedVariantFields) { if (!Object.hasOwn(mergedVariantFields, key)) continue; + // Drop type-specific keys that don't belong to the chosen type + if (!Object.hasOwn(chosenTypeAllowedKeys, key) && !Object.hasOwn(CLOUD_CODE_ASSIST_SHARED_SCHEMA_KEYS, key)) { + continue; + } const value = mergedVariantFields[key]; const existingValue = nextSchema[key]; if (existingValue !== undefined && !areJsonValuesEqual(existingValue, value)) { diff --git a/packages/ai/src/utils/stream-markup-healing.ts b/packages/ai/src/utils/stream-markup-healing.ts index 6cb1c6531..7114b9019 100644 --- a/packages/ai/src/utils/stream-markup-healing.ts +++ b/packages/ai/src/utils/stream-markup-healing.ts @@ -600,12 +600,17 @@ export function modelMayLeakDsmlToolCalls(provider: string, modelId: string): bo ); } +/** Cheap model/provider gate for MiniMax plain thinking tag leaks. */ +export function modelMayLeakThinkingTags(provider: string, modelId: string): boolean { + return /minimax/i.test(provider) || /minimax/i.test(modelId); +} + export function getStreamMarkupHealingPattern( provider: string, modelId: string, options?: { readonly parseThinkingTags?: boolean }, ): StreamMarkupHealingPattern | undefined { - if (options?.parseThinkingTags) return "thinking"; + if (options?.parseThinkingTags || modelMayLeakThinkingTags(provider, modelId)) return "thinking"; if (modelMayLeakKimiToolCalls(provider, modelId)) return "kimi"; if (modelMayLeakDsmlToolCalls(provider, modelId)) return "dsml"; return undefined; diff --git a/packages/ai/src/utils/validation.ts b/packages/ai/src/utils/validation.ts index 7f82626ec..506c72978 100644 --- a/packages/ai/src/utils/validation.ts +++ b/packages/ai/src/utils/validation.ts @@ -724,6 +724,7 @@ interface FlatIssue { keyword: "type" | "unrecognized" | "other"; instancePath: string; expectedTypes: string[]; + unionBranch: boolean; } /** @@ -759,12 +760,12 @@ function mapZodExpectedToJsonSchemaType(expected: unknown): string | null { */ function flattenIssues(issues: ReadonlyArray): FlatIssue[] { const out: FlatIssue[] = []; - const walk = (issue: ZodIssue, prefix: ReadonlyArray): void => { + const walk = (issue: ZodIssue, prefix: ReadonlyArray, unionBranch: boolean): void => { const fullPath = prefix.length === 0 ? issue.path : [...prefix, ...issue.path]; if (issue.code === "invalid_type") { const mapped = mapZodExpectedToJsonSchemaType((issue as { expected?: unknown }).expected); if (mapped) { - out.push({ keyword: "type", instancePath: pathToPointer(fullPath), expectedTypes: [mapped] }); + out.push({ keyword: "type", instancePath: pathToPointer(fullPath), expectedTypes: [mapped], unionBranch }); return; } } @@ -775,6 +776,7 @@ function flattenIssues(issues: ReadonlyArray): FlatIssue[] { keyword: "unrecognized", instancePath: pathToPointer([...fullPath, key]), expectedTypes: [], + unionBranch, }); } return; @@ -782,17 +784,21 @@ function flattenIssues(issues: ReadonlyArray): FlatIssue[] { if (issue.code === "invalid_union") { const inner = (issue as unknown as { errors?: ReadonlyArray> }).errors; if (inner) { + // A union-branch issue only competes with a sibling branch when it + // sits at the union node's own path. Issues whose own path is + // non-empty live on a deeper field that an already-identified + // branch owns, so the singleton-array repair should still apply. for (const branch of inner) { for (const child of branch) { - walk(child, fullPath); + walk(child, fullPath, child.path.length === 0); } } } return; } - out.push({ keyword: "other", instancePath: pathToPointer(fullPath), expectedTypes: [] }); + out.push({ keyword: "other", instancePath: pathToPointer(fullPath), expectedTypes: [], unionBranch }); }; - for (const issue of issues) walk(issue, []); + for (const issue of issues) walk(issue, [], false); return out; } @@ -801,7 +807,9 @@ function flattenIssues(issues: ReadonlyArray): FlatIssue[] { * * Two kinds of repair are applied: * - **type**: when a value is a JSON-encoded string and the schema wants - * something else, parse it and substitute the parsed value. + * something else, parse it and substitute the parsed value. When a + * non-union schema wants an array but receives a singleton value, wrap that + * value in a one-element array. * - **unrecognized**: when a strict object received an extra key (Zod's * `unrecognized_keys` or JSON Schema's `additionalProperties: false`), * drop that key so re-validation succeeds. This effectively coerces every @@ -811,6 +819,7 @@ function flattenIssues(issues: ReadonlyArray): FlatIssue[] { * The function is safe and conservative: * - Only processes "type" and "unrecognized" issues * - Only attempts JSON coercion on string values + * - Only wraps singleton array values for non-union type expectations * - Only accepts parsed results that match the expected type * - Clones the args object before mutation (copy-on-write) */ @@ -836,12 +845,16 @@ function coerceArgsFromIssues(args: unknown, issues: FlatIssue[]): { value: unkn if (issue.expectedTypes.length === 0) continue; const currentValue = getValueAtPointer(nextArgs, issue.instancePath); - if (typeof currentValue !== "string") continue; - - const result = tryParseJsonForTypes(currentValue, issue.expectedTypes); + const result = + typeof currentValue === "string" + ? tryParseJsonForTypes(currentValue, issue.expectedTypes) + : { value: currentValue, changed: false }; const coercedValue = result.changed ? result.value - : issue.expectedTypes.includes("array") + : issue.expectedTypes.includes("array") && + !issue.unionBranch && + currentValue !== undefined && + !Array.isArray(currentValue) ? [currentValue] : undefined; if (coercedValue === undefined) continue; @@ -905,17 +918,20 @@ function preserveUnknownRootFields(input: unknown, parsed: unknown): unknown { function flattenJsonSchemaIssues(issues: ReadonlyArray): FlatIssue[] { return issues.map(issue => { + const unionBranch = issue.fromUnionBranch === true; if (issue.keyword === "additionalProperties") { return { keyword: "unrecognized", instancePath: pathToPointer(issue.path), expectedTypes: [], + unionBranch, }; } return { keyword: issue.keyword === "type" ? "type" : "other", instancePath: pathToPointer(issue.path), expectedTypes: issue.expectedTypes ?? [], + unionBranch, }; }); } diff --git a/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts new file mode 100644 index 000000000..9b3f8dcbc --- /dev/null +++ b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts @@ -0,0 +1,164 @@ +import { describe, expect, it } from "bun:test"; +import { convertAnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic"; +import type { AssistantMessage, Message, Model, ToolResultMessage, UserMessage } from "@oh-my-pi/pi-ai/types"; + +/** + * Regression: Anthropic-compatible reasoning endpoints often emit `thinking` + * blocks without a first-party Anthropic signature, but still expect those + * blocks back as native `type: "thinking"` on continuation. Demoting unsigned + * thinking to text strips the reasoning chain and can destabilize follow-up + * tool-call argument serialization (the upstream cause behind #2005's `todo` + * renderer crash). + * + * Official Anthropic remains conservative: unsigned thinking is demoted to text + * there because the first-party API enforces signature-based integrity. + */ +function makeModel(overrides: Partial> = {}): Model<"anthropic-messages"> { + return { + api: "anthropic-messages", + provider: "custom-anthropic", + id: "reasoning-model", + name: "Reasoning Anthropic-Compatible Model", + baseUrl: "https://llm.example.com/anthropic", + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + maxTokens: 8_192, + contextWindow: 200_000, + reasoning: true, + ...overrides, + }; +} + +function makeUser(text = "continue"): UserMessage { + return { role: "user", content: text, timestamp: 0 }; +} + +function makeAssistantThinking(thinking: string, tail: AssistantMessage["content"][number][] = []): AssistantMessage { + return { + role: "assistant", + content: [{ type: "thinking", thinking, thinkingSignature: "" }, ...tail], + api: "anthropic-messages", + provider: "custom-anthropic", + model: "reasoning-model", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "toolUse", + timestamp: 0, + }; +} + +interface WireThinkingBlock { + type: "thinking"; + thinking: string; + signature: string; +} +interface WireTextBlock { + type: "text"; + text: string; +} +interface WireToolUseBlock { + type: "tool_use"; + id: string; + name: string; + input: Record; +} +type WireBlock = WireThinkingBlock | WireTextBlock | WireToolUseBlock | { type: string; [key: string]: unknown }; + +function assistantWireBlocks(messages: Message[], model: Model<"anthropic-messages">): WireBlock[] { + const params = convertAnthropicMessages(messages, model, false); + const assistant = params.find(p => p.role === "assistant"); + return (assistant?.content as WireBlock[] | undefined) ?? []; +} + +describe("Anthropic-compatible unsigned thinking replay (#2005)", () => { + it("preserves unsigned thinking for non-official reasoning endpoints", () => { + const blocks = assistantWireBlocks( + [ + makeUser("solve x"), + makeAssistantThinking("plan: read the file, then edit", [{ type: "text", text: "Sure." }]), + ], + makeModel(), + ); + expect(blocks[0]).toEqual({ + type: "thinking", + thinking: "plan: read the file, then edit", + signature: "", + }); + expect(blocks[1]).toEqual({ type: "text", text: "Sure." }); + }); + + it("covers the Xiaomi MiMo Anthropic-compatible reporter configuration without provider allowlists", () => { + const model = makeModel({ + provider: "user-custom", + id: "mimo-v2.5-pro", + name: "MiMo V2.5 Pro (Singapore)", + baseUrl: "https://token-plan-sgp.xiaomimimo.com/anthropic", + maxTokens: 131_072, + contextWindow: 1_048_576, + }); + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("hidden reasoning")], model); + expect(blocks[0]).toEqual({ type: "thinking", thinking: "hidden reasoning", signature: "" }); + }); + + it("preserves legacy known non-signing endpoints even if model.reasoning is false", () => { + const model = makeModel({ provider: "custom", baseUrl: "https://api.deepseek.com/v1", reasoning: false }); + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("deepseek reasoning")], model); + expect(blocks[0]?.type).toBe("thinking"); + }); + + it("still degrades unsigned thinking to text for official Anthropic", () => { + const model = makeModel({ provider: "anthropic", baseUrl: "https://api.anthropic.com" }); + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model); + expect(blocks[0]?.type).toBe("text"); + expect((blocks[0] as WireTextBlock).text).toBe("internal scratch"); + }); + + it("treats a missing baseUrl as official Anthropic (resolveAnthropicBaseUrl default)", () => { + // `isAnthropicApiBaseUrl(undefined) === true` because the actual HTTP + // dispatch falls back to https://api.anthropic.com. Same-id custom + // overrides that only tweak model metadata (no baseUrl override) must + // not regress to native-thinking replay against the first-party API. + const model = { ...makeModel(), provider: "anthropic", baseUrl: "" }; + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model); + expect(blocks[0]?.type).toBe("text"); + expect((blocks[0] as WireTextBlock).text).toBe("internal scratch"); + }); + + it("still degrades unsigned thinking to text for non-reasoning unknown endpoints", () => { + const model = makeModel({ reasoning: false, baseUrl: "https://plain.example.com/anthropic" }); + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("scratch")], model); + expect(blocks[0]?.type).toBe("text"); + expect((blocks[0] as WireTextBlock).text).toBe("scratch"); + }); + + it("keeps thinking → tool_use pairing intact across continuation conversion", () => { + const toolResult: ToolResultMessage = { + role: "toolResult", + toolCallId: "toolu_reasoning_1", + toolName: "read", + content: [{ type: "text", text: "file body" }], + isError: false, + timestamp: 0, + }; + const model = makeModel(); + const messages: Message[] = [ + makeUser("read README"), + makeAssistantThinking("I need to call the read tool", [ + { type: "toolCall", id: "toolu_reasoning_1", name: "read", arguments: { path: "README.md" } }, + ]), + toolResult, + ]; + const params = convertAnthropicMessages(messages, model, false); + expect(params.map(p => p.role)).toEqual(["user", "assistant", "user"]); + const assistantBlocks = params[1].content as WireBlock[]; + expect(assistantBlocks[0]?.type).toBe("thinking"); + expect(assistantBlocks[1]?.type).toBe("tool_use"); + expect((assistantBlocks[1] as WireToolUseBlock).id).toBe("toolu_reasoning_1"); + }); +}); diff --git a/packages/ai/test/auth-gateway-openai-responses.test.ts b/packages/ai/test/auth-gateway-openai-responses.test.ts index fe3a21801..c6cbd8c9b 100644 --- a/packages/ai/test/auth-gateway-openai-responses.test.ts +++ b/packages/ai/test/auth-gateway-openai-responses.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { encodeResponse, encodeStream, parseRequest } from "../src/providers/openai-responses-server"; import type { AssistantMessage } from "../src/types"; import { AssistantMessageEventStream } from "../src/utils/event-stream"; diff --git a/packages/ai/test/auth-gateway-pi-native.test.ts b/packages/ai/test/auth-gateway-pi-native.test.ts index 7c10a65a7..5f3a77a76 100644 --- a/packages/ai/test/auth-gateway-pi-native.test.ts +++ b/packages/ai/test/auth-gateway-pi-native.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { encodeStream, formatError, parseRequest } from "../src/providers/pi-native-server"; import type { AssistantMessage, diff --git a/packages/ai/test/auth-retry.test.ts b/packages/ai/test/auth-retry.test.ts new file mode 100644 index 000000000..d992a33c1 --- /dev/null +++ b/packages/ai/test/auth-retry.test.ts @@ -0,0 +1,162 @@ +import { describe, expect, it } from "bun:test"; +import type { ApiKeyResolveContext } from "@oh-my-pi/pi-ai"; +import { isApiKeyResolver, isAuthRetryableError, resolveApiKeyOnce, withAuth } from "@oh-my-pi/pi-ai"; + +function authError(status = 401): Error & { status: number } { + return Object.assign(new Error(`${status} authentication_error`), { status }); +} + +function usageLimitError(): Error & { status: number } { + return Object.assign(new Error("You have hit your ChatGPT usage limit (pro plan). Try again in ~158 min."), { + status: 429, + }); +} + +describe("isApiKeyResolver / resolveApiKeyOnce", () => { + it("narrows resolver vs static key and resolves the initial value", async () => { + expect(isApiKeyResolver("static")).toBe(false); + expect(isApiKeyResolver(undefined)).toBe(false); + expect(isApiKeyResolver(() => "k")).toBe(true); + + expect(await resolveApiKeyOnce("static")).toBe("static"); + expect(await resolveApiKeyOnce(undefined)).toBeUndefined(); + + let seen: ApiKeyResolveContext | undefined; + const resolved = await resolveApiKeyOnce(ctx => { + seen = ctx; + return "minted"; + }); + expect(resolved).toBe("minted"); + // Initial resolve must look like an initial resolve, not a retry. + expect(seen).toEqual({ lastChance: false, error: undefined, signal: undefined }); + }); +}); + +describe("isAuthRetryableError", () => { + it("treats 401 and usage-limit phrasing as retryable, everything else as not", () => { + expect(isAuthRetryableError(authError(401))).toBe(true); + expect(isAuthRetryableError(usageLimitError())).toBe(true); + // A 429 whose body names the *account's* rate limit is rotatable (switch + // account), even though it isn't a 401 and isn't phrased "usage limit". + expect( + isAuthRetryableError( + Object.assign( + new Error( + '429 {"type":"error","error":{"type":"rate_limit_error","message":"This request would exceed your account\'s rate limit. Please try again later."}} retry-after-ms=9779000', + ), + { status: 429 }, + ), + ), + ).toBe(true); + // A generic (non-account) 429 rate limit is NOT rotatable — switching + // credentials won't help an org/global limit. + expect(isAuthRetryableError(Object.assign(new Error("429 too many requests"), { status: 429 }))).toBe(false); + expect(isAuthRetryableError("Error: 401 unauthorized")).toBe(true); + expect(isAuthRetryableError(authError(403))).toBe(false); + expect(isAuthRetryableError(authError(500))).toBe(false); + expect(isAuthRetryableError(new Error("network blip"))).toBe(false); + expect(isAuthRetryableError(undefined)).toBe(false); + }); +}); + +describe("withAuth", () => { + it("runs a single attempt for a static string key (no retry)", async () => { + const keys: Array = []; + const result = await withAuth("static-key", async key => { + keys.push(key); + return `ok:${key}`; + }); + expect(result).toBe("ok:static-key"); + expect(keys).toEqual(["static-key"]); + }); + + it("throws when a static key is missing", async () => { + await expect(withAuth(undefined, async () => "never", { missingKeyMessage: "no key for foo" })).rejects.toThrow( + "no key for foo", + ); + }); + + it("refreshes the same account, then switches, in order", async () => { + const keys: string[] = []; + const contexts: ApiKeyResolveContext[] = []; + const result = await withAuth( + ctx => { + contexts.push(ctx); + return ctx.error === undefined ? "k0" : ctx.lastChance ? "k2" : "k1"; + }, + async key => { + keys.push(key); + if (key === "k2") return "success"; + throw authError(); + }, + ); + expect(result).toBe("success"); + expect(keys).toEqual(["k0", "k1", "k2"]); + expect(contexts.map(ctx => ({ lastChance: ctx.lastChance, hasError: ctx.error !== undefined }))).toEqual([ + { lastChance: false, hasError: false }, + { lastChance: false, hasError: true }, + { lastChance: true, hasError: true }, + ]); + }); + + it("stops retrying when the resolver returns undefined", async () => { + const keys: string[] = []; + const original = authError(); + await expect( + withAuth( + ctx => (ctx.error === undefined ? "k0" : undefined), + async key => { + keys.push(key); + throw original; + }, + ), + ).rejects.toBe(original); + expect(keys).toEqual(["k0"]); + }); + + it("does not re-attempt when the re-resolved key is unchanged", async () => { + const keys: string[] = []; + const original = authError(); + // refresh-same returns the same key (skip), switch returns the same key (skip). + await expect( + withAuth( + () => "same", + async key => { + keys.push(key); + throw original; + }, + ), + ).rejects.toBe(original); + expect(keys).toEqual(["same"]); + }); + + it("propagates non-auth errors without retrying", async () => { + const keys: string[] = []; + const boom = new Error("network blip"); + await expect( + withAuth( + ctx => (ctx.error === undefined ? "k0" : "k1"), + async key => { + keys.push(key); + throw boom; + }, + ), + ).rejects.toBe(boom); + expect(keys).toEqual(["k0"]); + }); + + it("honors a custom isAuthError classifier", async () => { + const keys: string[] = []; + const result = await withAuth( + ctx => (ctx.error === undefined ? "k0" : "k1"), + async key => { + keys.push(key); + if (key === "k0") throw new Error("CUSTOM_RETRY"); + return "ok"; + }, + { isAuthError: error => error instanceof Error && error.message === "CUSTOM_RETRY" }, + ); + expect(result).toBe("ok"); + expect(keys).toEqual(["k0", "k1"]); + }); +}); diff --git a/packages/ai/test/auth-storage-force-refresh-rotate.test.ts b/packages/ai/test/auth-storage-force-refresh-rotate.test.ts new file mode 100644 index 000000000..fc7d0bb01 --- /dev/null +++ b/packages/ai/test/auth-storage-force-refresh-rotate.test.ts @@ -0,0 +1,163 @@ +import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { type AuthCredentialStore, AuthStorage, SqliteAuthCredentialStore } from "../src/auth-storage"; +import { registerOAuthProvider, unregisterOAuthProviders } from "../src/utils/oauth"; + +const PROVIDER = "unit-rotate-oauth"; +const SOURCE = "auth-storage-force-refresh-rotate-test"; + +function farExpiry(): number { + return Date.now() + 60 * 60_000; +} + +function authError(): Error & { status: number } { + return Object.assign(new Error("401 authentication_error"), { status: 401 }); +} + +function usageLimitError(): Error & { status: number } { + return Object.assign(new Error("You have hit your ChatGPT usage limit (pro plan). Try again in ~158 min."), { + status: 429, + }); +} + +describe("AuthStorage forceRefresh + rotateSessionCredential", () => { + let tempDir = ""; + let store: AuthCredentialStore | undefined; + let authStorage: AuthStorage | undefined; + + beforeEach(async () => { + tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-rotate-")); + store = await SqliteAuthCredentialStore.open(path.join(tempDir, "agent.db")); + authStorage = new AuthStorage(store); + }); + + afterEach(async () => { + vi.restoreAllMocks(); + unregisterOAuthProviders(SOURCE); + store?.close(); + store = undefined; + authStorage = undefined; + if (tempDir) { + await fs.rm(tempDir, { recursive: true, force: true }); + tempDir = ""; + } + }); + + function registerProvider(onRefresh?: () => void): void { + registerOAuthProvider({ + id: PROVIDER, + name: "Rotate Unit", + sourceId: SOURCE, + async login() { + return { access: "login", refresh: "login", expires: farExpiry() }; + }, + async refreshToken(credentials) { + onRefresh?.(); + return { + ...credentials, + access: "minted-access", + refresh: "minted-refresh", + expires: farExpiry(), + }; + }, + getApiKey(credentials) { + return credentials.access; + }, + }); + } + + test("forceRefresh re-mints a not-yet-expired token; a normal resolve uses the cached token", async () => { + if (!authStorage) throw new Error("test setup failed"); + let refreshCalls = 0; + registerProvider(() => { + refreshCalls += 1; + }); + await authStorage.set(PROVIDER, [ + { type: "oauth", access: "cached-access", refresh: "cached-refresh", expires: farExpiry() }, + ]); + + const cached = await authStorage.getApiKey(PROVIDER, "s-control"); + expect(cached).toBe("cached-access"); + expect(refreshCalls).toBe(0); + + const forced = await authStorage.getApiKey(PROVIDER, "s-force", { forceRefresh: true }); + expect(forced).toBe("minted-access"); + expect(refreshCalls).toBe(1); + + // The re-minted credential is persisted, so the next plain resolve sees it. + const after = await authStorage.getApiKey(PROVIDER, "s-after"); + expect(after).toBe("minted-access"); + }); + + test("rotateSessionCredential(401) blocks + clears the sticky and rotates to a sibling", async () => { + if (!authStorage) throw new Error("test setup failed"); + registerProvider(); + await authStorage.set(PROVIDER, [ + { type: "oauth", access: "acc-A", refresh: "ref-A", expires: farExpiry() }, + { type: "oauth", access: "acc-B", refresh: "ref-B", expires: farExpiry() }, + ]); + + const first = await authStorage.getApiKey(PROVIDER, "sess"); + expect(["acc-A", "acc-B"]).toContain(first ?? ""); + + const usageLimitSpy = vi.spyOn(authStorage, "markUsageLimitReached"); + const rotated = await authStorage.rotateSessionCredential(PROVIDER, "sess", { error: authError() }); + + expect(rotated).toBe(true); + // A hard 401 must NOT take the usage-limit code path. + expect(usageLimitSpy).not.toHaveBeenCalled(); + + const second = await authStorage.getApiKey(PROVIDER, "sess"); + expect(["acc-A", "acc-B"]).toContain(second ?? ""); + expect(second).not.toBe(first); + }); + + test("rotateSessionCredential(usage-limit) delegates to markUsageLimitReached", async () => { + if (!authStorage) throw new Error("test setup failed"); + registerProvider(); + await authStorage.set(PROVIDER, [ + { type: "oauth", access: "acc-A", refresh: "ref-A", expires: farExpiry() }, + { type: "oauth", access: "acc-B", refresh: "ref-B", expires: farExpiry() }, + ]); + + const first = await authStorage.getApiKey(PROVIDER, "sess"); + const usageLimitSpy = vi.spyOn(authStorage, "markUsageLimitReached"); + + const rotated = await authStorage.rotateSessionCredential(PROVIDER, "sess", { + error: usageLimitError(), + }); + + expect(rotated).toBe(true); + // Usage / account-rate-limit errors route to markUsageLimitReached, which + // owns the block duration (default + server usage-report reset) — the + // resolver never parses retry-after itself. + expect(usageLimitSpy).toHaveBeenCalledTimes(1); + expect(usageLimitSpy.mock.calls[0]?.[0]).toBe(PROVIDER); + expect(usageLimitSpy.mock.calls[0]?.[1]).toBe("sess"); + + const second = await authStorage.getApiKey(PROVIDER, "sess"); + expect(second).not.toBe(first); + }); + + test("rotateSessionCredential reports no sibling for a single-credential setup", async () => { + if (!authStorage) throw new Error("test setup failed"); + registerProvider(); + await authStorage.set(PROVIDER, [ + { type: "oauth", access: "only-access", refresh: "only-refresh", expires: farExpiry() }, + ]); + + await authStorage.getApiKey(PROVIDER, "sess"); + expect(await authStorage.rotateSessionCredential(PROVIDER, "sess", { error: authError() })).toBe(false); + }); + + test("rotateSessionCredential returns false when the session has no sticky credential", async () => { + if (!authStorage) throw new Error("test setup failed"); + registerProvider(); + await authStorage.set(PROVIDER, [{ type: "oauth", access: "acc-A", refresh: "ref-A", expires: farExpiry() }]); + + // Never resolved a key for this session → nothing to rotate away from. + expect(await authStorage.rotateSessionCredential(PROVIDER, "untouched", { error: authError() })).toBe(false); + }); +}); diff --git a/packages/ai/test/duplicate-tool-results.test.ts b/packages/ai/test/duplicate-tool-results.test.ts index 1be33d66b..46f34ee2c 100644 --- a/packages/ai/test/duplicate-tool-results.test.ts +++ b/packages/ai/test/duplicate-tool-results.test.ts @@ -32,6 +32,43 @@ describe("Duplicate Tool Results Regression", () => { reasoning: true, }; + const makeEvalAssistantMessage = (id: string, timestamp: number): AssistantMessage => ({ + role: "assistant", + content: [{ type: "toolCall", id, name: "eval", arguments: {} }], + api: "anthropic-messages", + provider: "anthropic", + model: "claude-3-5-sonnet-20241022", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "toolUse", + timestamp, + }); + + const makeEvalToolResult = (id: string, text: string, timestamp: number): ToolResultMessage => ({ + role: "toolResult", + toolCallId: id, + toolName: "eval", + content: [{ type: "text", text }], + isError: false, + timestamp, + }); + + const getAssistantToolIds = (messages: Message[]): string[] => + messages.flatMap(message => + message.role === "assistant" + ? message.content.filter((block): block is ToolCall => block.type === "toolCall").map(block => block.id) + : [], + ); + + const getToolResults = (messages: Message[]): ToolResultMessage[] => + messages.filter((message): message is ToolResultMessage => message.role === "toolResult"); + it("should not duplicate tool results for errored messages when results already exist", () => { const toolCallId = "toolu_019xqMTvqWZiTDy8XxmjxrTo"; @@ -316,6 +353,144 @@ describe("Duplicate Tool Results Regression", () => { expect(result2.length).toBe(1); expect(result3.length).toBe(1); }); + + it("deduplicates repeated tool call ids and preserves call/result pairing", () => { + const duplicateId = "functions.eval:301"; + const distinctId = "functions.eval:302"; + + const messages: Message[] = [ + makeEvalAssistantMessage(duplicateId, 1), + makeEvalToolResult(duplicateId, "first", 2), + makeEvalAssistantMessage(duplicateId, 3), + makeEvalToolResult(duplicateId, "second", 4), + makeEvalAssistantMessage(duplicateId, 5), + makeEvalAssistantMessage(distinctId, 6), + makeEvalToolResult(distinctId, "third", 7), + ]; + + const transformed = transformMessages(messages, model); + const assistantToolIds = getAssistantToolIds(transformed); + const toolResults = getToolResults(transformed); + + expect(assistantToolIds).toEqual([duplicateId, `${duplicateId}_dup1`, `${duplicateId}_dup2`, distinctId]); + expect(toolResults.map(result => result.toolCallId)).toEqual([ + duplicateId, + `${duplicateId}_dup1`, + `${duplicateId}_dup2`, + distinctId, + ]); + expect(toolResults.find(result => result.toolCallId === `${duplicateId}_dup1`)?.content).toEqual([ + { type: "text", text: "second" }, + ]); + expect(toolResults.find(result => result.toolCallId === `${duplicateId}_dup2`)?.content).toEqual([ + { type: "text", text: "No result provided" }, + ]); + }); + + it("deduplicates repeated ids without colliding with existing generated-looking ids", () => { + const duplicateId = "functions.eval:301"; + const generatedLookingId = `${duplicateId}_dup1`; + const messages: Message[] = [ + makeEvalAssistantMessage(duplicateId, 1), + makeEvalToolResult(duplicateId, "first", 2), + makeEvalAssistantMessage(generatedLookingId, 3), + makeEvalToolResult(generatedLookingId, "already-used", 4), + makeEvalAssistantMessage(duplicateId, 5), + makeEvalToolResult(duplicateId, "second", 6), + ]; + + const transformed = transformMessages(messages, model); + const assistantToolIds = getAssistantToolIds(transformed); + const toolResults = getToolResults(transformed); + + expect(assistantToolIds).toEqual([duplicateId, generatedLookingId, `${duplicateId}_dup2`]); + expect(toolResults.map(result => result.toolCallId)).toEqual([ + duplicateId, + generatedLookingId, + `${duplicateId}_dup2`, + ]); + expect(toolResults.find(result => result.toolCallId === `${duplicateId}_dup2`)?.content).toEqual([ + { type: "text", text: "second" }, + ]); + }); + + it("preserves delayed duplicate tool results across message gaps", () => { + const duplicateId = "functions.eval:301"; + const developerMessage: DeveloperMessage = { role: "developer", content: "handoff summary", timestamp: 4 }; + const messages: Message[] = [ + makeEvalAssistantMessage(duplicateId, 1), + makeEvalToolResult(duplicateId, "first", 2), + makeEvalAssistantMessage(duplicateId, 3), + developerMessage, + makeEvalToolResult(duplicateId, "second", 5), + ]; + + const transformed = transformMessages(messages, model); + const toolResults = getToolResults(transformed); + + expect(getAssistantToolIds(transformed)).toEqual([duplicateId, `${duplicateId}_dup1`]); + expect(toolResults.map(result => result.toolCallId)).toEqual([duplicateId, `${duplicateId}_dup1`]); + expect(toolResults.find(result => result.toolCallId === `${duplicateId}_dup1`)?.content).toEqual([ + { type: "text", text: "second" }, + ]); + }); + + it("routes the late result to the most recent duplicate call when a new turn re-emits the id across a gap", () => { + const duplicateId = "functions.eval:301"; + const developerMessage: DeveloperMessage = { role: "developer", content: "handoff summary", timestamp: 4 }; + const messages: Message[] = [ + makeEvalAssistantMessage(duplicateId, 1), + makeEvalToolResult(duplicateId, "first", 2), + makeEvalAssistantMessage(duplicateId, 3), + developerMessage, + makeEvalAssistantMessage(duplicateId, 5), + makeEvalToolResult(duplicateId, "second", 6), + ]; + + const transformed = transformMessages(messages, model); + const toolResults = getToolResults(transformed); + + expect(getAssistantToolIds(transformed)).toEqual([duplicateId, `${duplicateId}_dup1`, `${duplicateId}_dup2`]); + expect(toolResults.map(result => result.toolCallId)).toEqual([ + duplicateId, + `${duplicateId}_dup1`, + `${duplicateId}_dup2`, + ]); + expect(toolResults.find(result => result.toolCallId === `${duplicateId}_dup1`)?.content).toEqual([ + { type: "text", text: "No result provided" }, + ]); + expect(toolResults.find(result => result.toolCallId === `${duplicateId}_dup2`)?.content).toEqual([ + { type: "text", text: "second" }, + ]); + }); + + it("keeps duplicate-id rewrites within the 64-char tool-call id limit", () => { + const baseId = `toolu_${"a".repeat(58)}`; + expect(baseId.length).toBe(64); + const messages: Message[] = [ + makeEvalAssistantMessage(baseId, 1), + makeEvalToolResult(baseId, "first", 2), + makeEvalAssistantMessage(baseId, 3), + makeEvalToolResult(baseId, "second", 4), + ]; + + const transformed = transformMessages(messages, model); + const assistantToolIds = getAssistantToolIds(transformed); + const toolResults = getToolResults(transformed); + + expect(assistantToolIds).toHaveLength(2); + for (const id of assistantToolIds) { + expect(id.length).toBeLessThanOrEqual(64); + expect(id).toMatch(/^[A-Za-z0-9_-]+$/); + } + const rewrittenId = assistantToolIds[1]; + expect(rewrittenId).not.toBe(baseId); + expect(rewrittenId.endsWith("_dup1")).toBe(true); + expect(toolResults.map(result => result.toolCallId)).toEqual([baseId, rewrittenId]); + expect(toolResults.find(result => result.toolCallId === rewrittenId)?.content).toEqual([ + { type: "text", text: "second" }, + ]); + }); }); /** diff --git a/packages/ai/test/github-copilot-model-limits.test.ts b/packages/ai/test/github-copilot-model-limits.test.ts index c46f14487..07d76ee87 100644 --- a/packages/ai/test/github-copilot-model-limits.test.ts +++ b/packages/ai/test/github-copilot-model-limits.test.ts @@ -2,8 +2,8 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; +import { Effort } from "../src/effort"; import { createModelManager } from "../src/model-manager"; -import { Effort } from "../src/model-thinking"; import { getBundledModel } from "../src/models"; import { githubCopilotModelManagerOptions } from "../src/provider-models/openai-compat"; diff --git a/packages/ai/test/github-copilot-reasoning.test.ts b/packages/ai/test/github-copilot-reasoning.test.ts index 32a88c006..28746c85d 100644 --- a/packages/ai/test/github-copilot-reasoning.test.ts +++ b/packages/ai/test/github-copilot-reasoning.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { getBundledModel } from "../src/models"; import { streamAnthropic } from "../src/providers/anthropic"; import { streamOpenAIResponses } from "../src/providers/openai-responses"; diff --git a/packages/ai/test/google-antigravity-usage.test.ts b/packages/ai/test/google-antigravity-usage.test.ts new file mode 100644 index 000000000..4fdec1ae0 --- /dev/null +++ b/packages/ai/test/google-antigravity-usage.test.ts @@ -0,0 +1,196 @@ +/** + * Antigravity usage provider contract tests. The merge logic + * deduplicates per-model quota entries by (tier, windowId), + * preserves reset times when bar data and window data come from + * different model entries, and handles mixed-case tier names. + */ +import { describe, expect, it } from "bun:test"; +import type { UsageFetchContext, UsageFetchParams } from "../src/usage"; +import { antigravityUsageProvider } from "../src/usage/google-antigravity"; + +const accessTokenFixture = (() => { + const header = Buffer.from(JSON.stringify({ alg: "none", typ: "JWT" })).toString("base64url"); + const body = Buffer.from(JSON.stringify({ sub: "user-fixture" })).toString("base64url"); + return `${header}.${body}.sig`; +})(); + +function makeCredential(overrides?: Partial) { + return { + type: "oauth" as const, + accessToken: accessTokenFixture, + refreshToken: "refresh-fixture", + expiresAt: Date.now() + 3600_000, + projectId: "test-project", + email: "test@example.com", + accountId: "acct-1", + ...overrides, + } satisfies UsageFetchParams["credential"]; +} + +function fakeFetch(json: unknown): typeof fetch { + const fn = async () => + new Response(JSON.stringify(json), { + status: 200, + headers: { "content-type": "application/json" }, + }); + return fn as unknown as typeof fetch; +} + +function makeCtx(fetchImpl?: typeof fetch): UsageFetchContext { + return { fetch: fetchImpl ?? fakeFetch({}) }; +} + +// ── helpers ────────────────────────────────────────────────────────── + +function makeApiModel( + displayName: string, + quota: { remainingFraction?: number; resetTime?: string; tier?: string; windowId?: string }, +) { + return { + displayName, + quotaInfo: { + remainingFraction: quota.remainingFraction, + resetTime: quota.resetTime, + tier: quota.tier, + windowId: quota.windowId, + }, + }; +} + +// ── tests ──────────────────────────────────────────────────────────── + +describe("antigravity usage provider", () => { + it("merges two models with same tier into one limit", async () => { + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "premium" }), + modelB: makeApiModel("Model B", { remainingFraction: 0.5, tier: "premium" }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report).not.toBeNull(); + expect(report!.limits.length).toBe(1); + }); + + it("keeps the worst remainingFraction when merging same tier", async () => { + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.1, tier: "premium" }), + modelB: makeApiModel("Model B", { remainingFraction: 0.8, tier: "premium" }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.limits.length).toBe(1); + expect(report!.limits[0]!.amount.remainingFraction).toBe(0.1); + }); + + it("merges mixed-case tier names under lowercased key", async () => { + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "Default" }), + modelB: makeApiModel("Model B", { remainingFraction: 0.6, tier: "default" }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.limits.length).toBe(1); + }); + + it("preserves reset time from an entry even when bar data comes from another", async () => { + const now = Date.now(); + const resetTime = new Date(now + 4 * 3600_000).toISOString(); + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "default" }), + modelB: makeApiModel("Model B", { remainingFraction: undefined, tier: "default", resetTime }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.limits.length).toBe(1); + expect(report!.limits[0]!.amount.remainingFraction).toBe(0.3); + expect(report!.limits[0]!.window).toBeDefined(); + expect(report!.limits[0]!.window!.resetsAt).toBeGreaterThan(now); + }); + + it("separates models with different windowIds in the same tier", async () => { + const now = Date.now(); + const t1 = new Date(now + 5 * 3600_000).toISOString(); + const t2 = new Date(now + 24 * 3600_000).toISOString(); + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "premium", windowId: "5h", resetTime: t1 }), + modelB: makeApiModel("Model B", { + remainingFraction: 0.7, + tier: "premium", + windowId: "daily", + resetTime: t2, + }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.limits.length).toBe(2); + }); + + it("includes email and projectId in report metadata", async () => { + const payload = { models: { m: makeApiModel("M", { remainingFraction: 1 }) } }; + const report = await antigravityUsageProvider.fetchUsage!( + { + provider: "google-antigravity", + credential: makeCredential({ email: "user@example.com", projectId: "proj-1" }), + signal: undefined, + }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.metadata?.email).toBe("user@example.com"); + expect(report!.metadata?.projectId).toBe("proj-1"); + }); + + it("does not include email when credential has none", async () => { + const payload = { models: { m: makeApiModel("M", { remainingFraction: 1 }) } }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential({ email: undefined }), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.metadata?.email).toBeUndefined(); + }); + + it("sorts limits by remainingFraction ascending (worst first)", async () => { + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.9, tier: "high" }), + modelB: makeApiModel("Model B", { remainingFraction: 0.2, tier: "low" }), + modelC: makeApiModel("Model C", { remainingFraction: 0.5, tier: "mid" }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.limits.length).toBe(3); + expect(report!.limits[0]!.amount.remainingFraction).toBe(0.2); + expect(report!.limits[1]!.amount.remainingFraction).toBe(0.5); + expect(report!.limits[2]!.amount.remainingFraction).toBe(0.9); + }); + + it("returns null when credential has no projectId", async () => { + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential({ projectId: undefined }), signal: undefined }, + makeCtx(), + ); + expect(report).toBeNull(); + }); +}); diff --git a/packages/ai/test/issue-1207-repro.test.ts b/packages/ai/test/issue-1207-repro.test.ts index 85e31868d..d71c91b3d 100644 --- a/packages/ai/test/issue-1207-repro.test.ts +++ b/packages/ai/test/issue-1207-repro.test.ts @@ -91,6 +91,19 @@ describe("issue #1207 — DeepSeek V4 keeps reasoning with tools", () => { expect(body.max_completion_tokens).toBeUndefined(); }); + it("does not mix Fireworks DeepSeek effort with the native thinking toggle", async () => { + const model = getBundledModel("fireworks", "deepseek-v4-pro") as Model<"openai-completions">; + const compat = resolveOpenAICompat(model); + const body = await capturePayload(model); + + expect(compat.extraBody).toBeUndefined(); + expect(body.tools).toBeDefined(); + expect(body.tool_choice).toBeUndefined(); + expect(body.reasoning_effort).toBe("high"); + expect(body.thinking).toBeUndefined(); + expect(body.max_tokens).toBe(123); + }); + it("preserves OpenRouter reasoning when tool_choice auto is present", async () => { const model = getBundledModel("openrouter", "deepseek/deepseek-v4-flash") as Model<"openai-completions">; const compat = detectOpenAICompat(model); diff --git a/packages/ai/test/issue-1373-repro.test.ts b/packages/ai/test/issue-1373-repro.test.ts index 986acbf3c..88e855f4e 100644 --- a/packages/ai/test/issue-1373-repro.test.ts +++ b/packages/ai/test/issue-1373-repro.test.ts @@ -1,5 +1,5 @@ import { afterAll, beforeAll, describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { streamBedrock } from "../src/providers/amazon-bedrock"; import type { Context, Model } from "../src/types"; diff --git a/packages/ai/test/issue-826-repro.test.ts b/packages/ai/test/issue-826-repro.test.ts index efcf9ccea..dd78aeda0 100644 --- a/packages/ai/test/issue-826-repro.test.ts +++ b/packages/ai/test/issue-826-repro.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { streamAnthropic } from "../src/providers/anthropic"; import type { Context, Model, Tool } from "../src/types"; diff --git a/packages/ai/test/issue-969-repro.test.ts b/packages/ai/test/issue-969-repro.test.ts index 1b347470e..9f42a85bb 100644 --- a/packages/ai/test/issue-969-repro.test.ts +++ b/packages/ai/test/issue-969-repro.test.ts @@ -1,5 +1,6 @@ import { afterEach, describe, expect, it } from "bun:test"; -import { Effort, getSupportedEfforts } from "../src/model-thinking"; +import { Effort } from "../src/effort"; +import { getSupportedEfforts } from "../src/model-thinking"; import { streamOpenAICompletions } from "../src/providers/openai-completions"; import type { Context, Model } from "../src/types"; diff --git a/packages/ai/test/model-thinking.test.ts b/packages/ai/test/model-thinking.test.ts index 89b90da8e..e8475974c 100644 --- a/packages/ai/test/model-thinking.test.ts +++ b/packages/ai/test/model-thinking.test.ts @@ -1,8 +1,8 @@ import { describe, expect, it } from "bun:test"; +import { Effort } from "@oh-my-pi/pi-ai/effort"; import { applyGeneratedModelPolicies, clampThinkingLevelForModel, - Effort, enrichModelThinking, linkOpenAIPromotionTargets, mapEffortToAnthropicAdaptiveEffort, diff --git a/packages/ai/test/nanogpt-model-limits.test.ts b/packages/ai/test/nanogpt-model-limits.test.ts index 03d7b8bab..0270b18ff 100644 --- a/packages/ai/test/nanogpt-model-limits.test.ts +++ b/packages/ai/test/nanogpt-model-limits.test.ts @@ -1,5 +1,5 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { nanoGptModelManagerOptions } from "../src/provider-models/openai-compat"; const originalFetch = global.fetch; diff --git a/packages/ai/test/ollama-provider.test.ts b/packages/ai/test/ollama-provider.test.ts index 1a2e2aadb..4fa4866c2 100644 --- a/packages/ai/test/ollama-provider.test.ts +++ b/packages/ai/test/ollama-provider.test.ts @@ -1,5 +1,5 @@ import { afterEach, describe, expect, test, vi } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { ollamaModelManagerOptions } from "../src/provider-models/openai-compat"; import { streamOllama } from "../src/providers/ollama"; import type { Context, Model, Tool } from "../src/types"; diff --git a/packages/ai/test/openai-codex.test.ts b/packages/ai/test/openai-codex.test.ts index 4dbc616a7..ed81d7c78 100644 --- a/packages/ai/test/openai-codex.test.ts +++ b/packages/ai/test/openai-codex.test.ts @@ -71,6 +71,62 @@ describe("openai-codex request transformer", () => { }); }); +describe("openai-codex orphan tool-call repair", () => { + it("synthesizes a function_call_output for a function_call with no result", async () => { + const body: RequestBody = { + model: "gpt-5.1-codex", + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }, + { type: "function_call", call_id: "call_orphan", name: "read", arguments: "{}" }, + { type: "message", role: "user", content: [{ type: "input_text", text: "next" }] }, + ], + }; + + const transformed = await transformRequestBody(body, createCodexModel(body.model), {}); + const input = transformed.input || []; + + const callIndex = input.findIndex(item => item.type === "function_call" && item.call_id === "call_orphan"); + expect(callIndex).toBeGreaterThanOrEqual(0); + // The synthesized output sits immediately after the orphan call. + const output = input[callIndex + 1]; + expect(output?.type).toBe("function_call_output"); + expect(output?.call_id).toBe("call_orphan"); + expect(typeof output?.output).toBe("string"); + expect(output?.output as string).toMatch(/interrupted/i); + }); + + it("leaves a paired function_call untouched", async () => { + const body: RequestBody = { + model: "gpt-5.1-codex", + input: [ + { type: "function_call", call_id: "call_paired", name: "read", arguments: "{}" }, + { type: "function_call_output", call_id: "call_paired", output: "real result" }, + ], + }; + + const transformed = await transformRequestBody(body, createCodexModel(body.model), {}); + const input = transformed.input || []; + + const outputs = input.filter(item => item.type === "function_call_output" && item.call_id === "call_paired"); + expect(outputs).toHaveLength(1); + expect(outputs[0]?.output).toBe("real result"); + }); + + it("synthesizes a custom_tool_call_output for an orphan custom_tool_call", async () => { + const body: RequestBody = { + model: "gpt-5.1-codex", + input: [{ type: "custom_tool_call", call_id: "call_custom", name: "apply_patch" }], + }; + + const transformed = await transformRequestBody(body, createCodexModel(body.model), {}); + const input = transformed.input || []; + + const output = input.find(item => item.type === "custom_tool_call_output" && item.call_id === "call_custom"); + expect(output).toBeDefined(); + expect(output?.output as string).toMatch(/interrupted/i); + }); +}); + describe("openai-codex reasoning effort validation", () => { it("rejects gpt-5.1 xhigh when metadata does not list it", async () => { const body: RequestBody = { model: "gpt-5.1", input: [] }; diff --git a/packages/ai/test/openai-completions-disable-reasoning.test.ts b/packages/ai/test/openai-completions-disable-reasoning.test.ts index f2d9107f6..629a74887 100644 --- a/packages/ai/test/openai-completions-disable-reasoning.test.ts +++ b/packages/ai/test/openai-completions-disable-reasoning.test.ts @@ -1,5 +1,5 @@ import { afterEach, describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { streamOpenAICompletions } from "../src/providers/openai-completions"; import type { Context, Model } from "../src/types"; diff --git a/packages/ai/test/openai-responses-orphan-repair.test.ts b/packages/ai/test/openai-responses-orphan-repair.test.ts new file mode 100644 index 000000000..014d6988f --- /dev/null +++ b/packages/ai/test/openai-responses-orphan-repair.test.ts @@ -0,0 +1,69 @@ +import { describe, expect, it } from "bun:test"; +import { + repairOrphanResponsesToolCalls, + repairOrphanResponsesToolOutputs, +} from "@oh-my-pi/pi-ai/providers/openai-responses-shared"; +import type { ResponseInput } from "openai/resources/responses/responses"; + +describe("repairOrphanResponsesToolCalls", () => { + it("appends a synthetic function_call_output after a call with no result", () => { + const input: ResponseInput = [ + { type: "function_call", call_id: "call_a", name: "read", arguments: "{}" }, + { role: "user", content: [{ type: "input_text", text: "continue" }] }, + ]; + + const repaired = repairOrphanResponsesToolCalls(input); + const callIndex = repaired.findIndex( + item => + (item as { type?: string }).type === "function_call" && (item as { call_id?: string }).call_id === "call_a", + ); + const output = repaired[callIndex + 1] as { type?: string; call_id?: string; output?: unknown }; + expect(output.type).toBe("function_call_output"); + expect(output.call_id).toBe("call_a"); + expect(output.output).toMatch(/interrupted/i); + }); + + it("uses custom_tool_call_output for an orphan custom_tool_call", () => { + const input: ResponseInput = [ + { type: "custom_tool_call", call_id: "call_c", name: "apply_patch", input: "patch" } as ResponseInput[number], + ]; + + const repaired = repairOrphanResponsesToolCalls(input); + const output = repaired.find(item => (item as { type?: string }).type === "custom_tool_call_output") as + | { call_id?: string } + | undefined; + expect(output?.call_id).toBe("call_c"); + }); + + it("returns the input unchanged when every call is paired", () => { + const input: ResponseInput = [ + { type: "function_call", call_id: "call_a", name: "read", arguments: "{}" }, + { type: "function_call_output", call_id: "call_a", output: "ok" } as ResponseInput[number], + ]; + + const repaired = repairOrphanResponsesToolCalls(input); + expect(repaired).toBe(input); + }); + + it("composes with output repair so a tree-branch snapshot stays API-valid", () => { + // Branching to a node that ends on a tool call drops the result child: + // the assistant turn keeps the call, but no matching output remains. + const input: ResponseInput = [ + { role: "user", content: [{ type: "input_text", text: "do it" }] }, + { type: "function_call", call_id: "call_x", name: "bash", arguments: "{}" }, + ]; + + const repaired = repairOrphanResponsesToolCalls(repairOrphanResponsesToolOutputs(input)); + const callIds = new Set( + repaired + .filter(i => (i as { type?: string }).type === "function_call") + .map(i => (i as { call_id: string }).call_id), + ); + const outputIds = new Set( + repaired + .filter(i => (i as { type?: string }).type === "function_call_output") + .map(i => (i as { call_id: string }).call_id), + ); + for (const id of callIds) expect(outputIds.has(id)).toBe(true); + }); +}); diff --git a/packages/ai/test/openai-responses-parallel-tool-calls.test.ts b/packages/ai/test/openai-responses-parallel-tool-calls.test.ts index 9523d13db..dbff9a657 100644 --- a/packages/ai/test/openai-responses-parallel-tool-calls.test.ts +++ b/packages/ai/test/openai-responses-parallel-tool-calls.test.ts @@ -260,4 +260,79 @@ describe("processResponsesStream: parallel function_call items", () => { expect(byCallId.get("call_a")?.toolCall.arguments).toEqual({ command: "printf a" }); expect(byCallId.get("call_b")?.toolCall.arguments).toEqual({ command: "printf b" }); }); + + test("routes deltas by item.call_id when llama.cpp omits item.id and output_index (issue #2015)", async () => { + // llama.cpp's `to_json_oaicompat_resp` (tools/server/server-task.cpp) emits a + // function_call's `output_item.added` with only `item.call_id` — no `item.id`, + // no `output_index`. The matching `function_call_arguments.delta` then carries + // `item_id: "fc_"` and again no `output_index`. Without secondary + // indexing on `call_id`, `processResponsesStream`'s lookup map stays empty and + // every delta lands on the trailing block, leaving earlier calls with empty + // arguments (= `{}`) — the read tool then rejects them with + // `path: Invalid input: expected string, received undefined`. + const output = makeOutput(); + const emitted: EmittedEvent[] = []; + const stream = { push: (e: unknown) => emitted.push(e as EmittedEvent), end: () => {} } as never; + + const argsA = JSON.stringify({ path: "a.txt" }); + const argsB = JSON.stringify({ path: "b.txt" }); + const argsC = JSON.stringify({ path: "c.txt" }); + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + item: { type: "function_call", call_id: "fc_a", name: "read", arguments: "" }, + }, + { + type: "response.output_item.added", + item: { type: "function_call", call_id: "fc_b", name: "read", arguments: "" }, + }, + { + type: "response.output_item.added", + item: { type: "function_call", call_id: "fc_c", name: "read", arguments: "" }, + }, + { type: "response.function_call_arguments.delta", item_id: "fc_a", delta: argsA }, + { type: "response.function_call_arguments.delta", item_id: "fc_b", delta: argsB }, + { type: "response.function_call_arguments.delta", item_id: "fc_c", delta: argsC }, + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "fc_a", name: "read", arguments: argsA }, + }, + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "fc_b", name: "read", arguments: argsB }, + }, + { + type: "response.output_item.done", + item: { type: "function_call", call_id: "fc_c", name: "read", arguments: argsC }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.content).toHaveLength(3); + const [a, b, c] = output.content; + if (a?.type !== "toolCall" || b?.type !== "toolCall" || c?.type !== "toolCall") { + throw new Error("expected toolCalls"); + } + expect(a.arguments).toEqual({ path: "a.txt" }); + expect(b.arguments).toEqual({ path: "b.txt" }); + expect(c.arguments).toEqual({ path: "c.txt" }); + + const ends = emitted.filter(e => e.type === "toolcall_end") as Array<{ + toolCall: { id: string; arguments: Record }; + contentIndex: number; + }>; + expect(ends).toHaveLength(3); + const byCallId = new Map(ends.map(e => [e.toolCall.id.split("|")[0], e])); + expect(byCallId.get("fc_a")?.toolCall.arguments).toEqual({ path: "a.txt" }); + expect(byCallId.get("fc_b")?.toolCall.arguments).toEqual({ path: "b.txt" }); + expect(byCallId.get("fc_c")?.toolCall.arguments).toEqual({ path: "c.txt" }); + expect(byCallId.get("fc_a")?.contentIndex).toBe(0); + expect(byCallId.get("fc_b")?.contentIndex).toBe(1); + expect(byCallId.get("fc_c")?.contentIndex).toBe(2); + }); }); diff --git a/packages/ai/test/schema-normalization.test.ts b/packages/ai/test/schema-normalization.test.ts index 1f6ba7688..22efcb2cd 100644 --- a/packages/ai/test/schema-normalization.test.ts +++ b/packages/ai/test/schema-normalization.test.ts @@ -728,6 +728,18 @@ describe("stripResidualCombiners", () => { expect(normalized.anyOf).toBeUndefined(); expect(normalized.oneOf).toBeUndefined(); }); + + it("drops array-only keys when mixed-type collapse picks string from anyOf fixpoint", () => { + const stripped = stripResidualCombiners({ + anyOf: [{ type: "string" }, { type: "array", items: { type: "string" } }], + description: "pr number, url, or branch", + }) as Record; + + expect(stripped.type).toBe("string"); + expect(stripped.items).toBeUndefined(); + expect(stripped.anyOf).toBeUndefined(); + expect(stripped.description).toBe("pr number, url, or branch"); + }); }); // --------------------------------------------------------------------------- @@ -952,6 +964,36 @@ describe("normalizeSchemaForCCA", () => { properties: {}, }); }); + + it("strips array-only keys when mixed-type collapse picks a non-array type", () => { + // Regression: anyOf [{type:"string"}, {type:"array", items:{type:"string"}}] + // collapsed to {type:"string", items:{type:"string"}} which is invalid. + // The fix filters mergedVariantFields against the chosen type's allowed keys. + const normalized = normalizeSchemaForCCA({ + anyOf: [{ type: "string" }, { type: "array", items: { type: "string" } }], + description: "pr number, url, or branch", + }); + + expect(normalized).toEqual({ + type: "string", + description: "pr number, url, or branch", + }); + }); + + it("strips sibling type-specific keys copied from parent when mixed-type collapse picks opposing type", () => { + // Edge case: parent has a sibling `items` outside the anyOf, + // and the chosen type is string. The sibling must be stripped. + const normalized = normalizeSchemaForCCA({ + anyOf: [{ type: "string" }, { type: "array", items: { type: "number" } }], + items: { type: "string" }, + description: "pr number, url, or branch", + }); + + expect(normalized).toEqual({ + type: "string", + description: "pr number, url, or branch", + }); + }); }); // --------------------------------------------------------------------------- diff --git a/packages/ai/test/stream-auth-retry.test.ts b/packages/ai/test/stream-auth-retry.test.ts index d019e57ce..7192cb8c8 100644 --- a/packages/ai/test/stream-auth-retry.test.ts +++ b/packages/ai/test/stream-auth-retry.test.ts @@ -1,4 +1,5 @@ import { afterEach, describe, expect, it } from "bun:test"; +import type { ApiKeyResolveContext } from "@oh-my-pi/pi-ai"; import { registerCustomApi, unregisterCustomApis } from "@oh-my-pi/pi-ai"; import { streamSimple } from "@oh-my-pi/pi-ai/stream"; import type { Api, AssistantMessage, Context, Model, SimpleStreamOptions, Usage } from "@oh-my-pi/pi-ai/types"; @@ -25,9 +26,9 @@ function assistant(content: string[] = []): AssistantMessage { api: API, provider: "test-provider", model: "test-model", - usage: usage(), + timestamp: 1, stopReason: "stop", - timestamp: Date.now(), + usage: usage(), }; } @@ -39,19 +40,21 @@ function authError(): Error & { status: number } { return Object.assign(new Error("401 authentication_error"), { status: 401 }); } +function usageLimitError(): Error & { status: number } { + return Object.assign(new Error("You have hit your ChatGPT usage limit (pro plan). Try again in ~158 min."), { + status: 429, + }); +} + function model(): Model { return { id: "test-model", - name: "test-model", + name: "Test Model", api: API, provider: "test-provider", - baseUrl: "mock://", - reasoning: false, - input: ["text"], - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, - contextWindow: 1024, - maxTokens: 1024, - }; + contextWindow: 1000, + maxTokens: 100, + } as Model; } const context: Context = { @@ -59,61 +62,64 @@ const context: Context = { messages: [{ role: "user", content: "hello", timestamp: 1 }], }; -describe("streamSimple auth retry", () => { +/** Records the static key each inner attempt actually received. */ +function pushKey(keys: unknown[], options?: SimpleStreamOptions): void { + keys.push(options?.apiKey); +} + +function ok(stream: AssistantMessageEventStream): void { + const message = assistant(["ok"]); + stream.push({ type: "start", partial: message }); + stream.push({ type: "done", reason: "stop", message }); +} + +describe("streamSimple resolver auth retry", () => { afterEach(() => { unregisterCustomApis(SOURCE_ID); }); - it("retries once with a fresh key when 401 happens before the first event", async () => { - const keys: Array = []; - let authCalls = 0; + it("retries with a refreshed key when a 401 is thrown before the first event", async () => { + const keys: unknown[] = []; + const contexts: ApiKeyResolveContext[] = []; registerCustomApi( API, (_model: Model, _context: Context, options?: SimpleStreamOptions) => { - keys.push(options?.apiKey); + pushKey(keys, options); const stream = new AssistantMessageEventStream(); - queueMicrotask(() => { - if (keys.length === 1) { - stream.fail(authError()); - return; - } - const message = assistant(["ok"]); - stream.push({ type: "start", partial: message }); - stream.push({ type: "done", reason: "stop", message }); - }); + queueMicrotask(() => (keys.length === 1 ? stream.fail(authError()) : ok(stream))); return stream; }, SOURCE_ID, ); const stream = streamSimple(model(), context, { - apiKey: "old-key", - onAuthError: async (provider, oldKey, error) => { - authCalls += 1; - expect(provider).toBe("test-provider"); - expect(oldKey).toBe("old-key"); - expect((error as { status?: number }).status).toBe(401); - return "new-key"; + apiKey: async ctx => { + contexts.push(ctx); + return ctx.error === undefined ? "old-key" : ctx.lastChance ? "switch-key" : "refresh-key"; }, }); - for await (const _event of stream) { // drain } expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); - expect(keys).toEqual(["old-key", "new-key"]); - expect(authCalls).toBe(1); + // Initial resolve, then step (b) refresh-same — the switch step is never reached. + expect(keys).toEqual(["old-key", "refresh-key"]); + expect(keys.every(key => typeof key === "string")).toBe(true); + expect(contexts.map(ctx => ({ lastChance: ctx.lastChance, hasError: ctx.error !== undefined }))).toEqual([ + { lastChance: false, hasError: false }, + { lastChance: false, hasError: true }, + ]); + expect((contexts[1]?.error as { status?: number }).status).toBe(401); }); - it("retries when a provider emits start then a 401 error event before content", async () => { - const keys: Array = []; + it("buffers the start event and retries on a 401 error event before content", async () => { + const keys: unknown[] = []; const eventTypes: string[] = []; - let authCalls = 0; registerCustomApi( API, (_model: Model, _context: Context, options?: SimpleStreamOptions) => { - keys.push(options?.apiKey); + pushKey(keys, options); const stream = new AssistantMessageEventStream(); queueMicrotask(() => { if (keys.length === 1) { @@ -127,9 +133,7 @@ describe("streamSimple auth retry", () => { }); return; } - const message = assistant(["ok"]); - stream.push({ type: "start", partial: message }); - stream.push({ type: "done", reason: "stop", message }); + ok(stream); }); return stream; }, @@ -137,28 +141,59 @@ describe("streamSimple auth retry", () => { ); const stream = streamSimple(model(), context, { - apiKey: "old-key", - onAuthError: async (provider, oldKey, error) => { - authCalls += 1; - expect(provider).toBe("test-provider"); - expect(oldKey).toBe("old-key"); - expect((error as { status?: number }).status).toBe(401); - return "new-key"; - }, + apiKey: async ctx => (ctx.error === undefined ? "old-key" : "new-key"), }); - for await (const event of stream) { eventTypes.push(event.type); } expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); expect(keys).toEqual(["old-key", "new-key"]); + // The buffered `start` of the failed attempt must not leak — the user + // sees exactly one clean start/done pair. expect(eventTypes).toEqual(["start", "done"]); - expect(authCalls).toBe(1); + }); + + it("retries on a 401 carried only via errorStatus", async () => { + const keys: unknown[] = []; + registerCustomApi( + API, + (_model: Model, _context: Context, options?: SimpleStreamOptions) => { + pushKey(keys, options); + const stream = new AssistantMessageEventStream(); + queueMicrotask(() => { + if (keys.length === 1) { + stream.push({ type: "start", partial: assistant() }); + stream.push({ + type: "error", + reason: "error", + error: assistantError( + '{"type":"error","error":{"type":"authentication_error","message":"Invalid authentication credentials"}}', + 401, + ), + }); + return; + } + ok(stream); + }); + return stream; + }, + SOURCE_ID, + ); + + const stream = streamSimple(model(), context, { + apiKey: async ctx => (ctx.error === undefined ? "old-key" : "new-key"), + }); + for await (const _event of stream) { + // drain + } + + expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); + expect(keys).toEqual(["old-key", "new-key"]); }); it("does not retry after replay-unsafe content has been emitted", async () => { - let authCalls = 0; + let retryResolves = 0; const failure = authError(); registerCustomApi( API, @@ -175,10 +210,9 @@ describe("streamSimple auth retry", () => { ); const stream = streamSimple(model(), context, { - apiKey: "old-key", - onAuthError: async () => { - authCalls += 1; - return "new-key"; + apiKey: async ctx => { + if (ctx.error !== undefined) retryResolves += 1; + return ctx.error === undefined ? "old-key" : "new-key"; }, }); @@ -192,94 +226,79 @@ describe("streamSimple auth retry", () => { } expect(caught).toBe(failure); - expect(authCalls).toBe(0); + // The resolver is never asked for a retry key once a replay-unsafe event shipped. + expect(retryResolves).toBe(0); }); - it("retries on 401 carried via errorStatus when the message has no parseable status", async () => { - const keys: Array = []; - let authCalls = 0; + it("escalates refresh-same then switch in order (2-retry ordering)", async () => { + const keys: unknown[] = []; registerCustomApi( API, (_model: Model, _context: Context, options?: SimpleStreamOptions) => { - keys.push(options?.apiKey); + pushKey(keys, options); const stream = new AssistantMessageEventStream(); - queueMicrotask(() => { - if (keys.length === 1) { - stream.push({ type: "start", partial: assistant() }); - stream.push({ - type: "error", - reason: "error", - // Realistic Anthropic SDK shape: message begins with ` ` - // and the regex fallback in extractHttpStatusFromError cannot find 401 - // inside the JSON body. Only `errorStatus` carries the signal. - error: assistantError( - '{"type":"error","error":{"type":"authentication_error","message":"Invalid authentication credentials"}}', - 401, - ), - }); - return; - } - const message = assistant(["ok"]); - stream.push({ type: "start", partial: message }); - stream.push({ type: "done", reason: "stop", message }); - }); + queueMicrotask(() => (options?.apiKey === "switch-key" ? ok(stream) : stream.fail(authError()))); return stream; }, SOURCE_ID, ); const stream = streamSimple(model(), context, { - apiKey: "old-key", - onAuthError: async () => { - authCalls += 1; - return "new-key"; - }, + apiKey: async ctx => (ctx.error === undefined ? "old-key" : ctx.lastChance ? "switch-key" : "refresh-key"), }); - for await (const _event of stream) { // drain } expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); - expect(keys).toEqual(["old-key", "new-key"]); - expect(authCalls).toBe(1); + expect(keys).toEqual(["old-key", "refresh-key", "switch-key"]); }); - it("retries on a thrown usage_limit_reached error before any event has been emitted", async () => { - const keys: Array = []; - const errors: unknown[] = []; + it("skips the refresh-same step when the resolver returns an unchanged key", async () => { + const keys: unknown[] = []; registerCustomApi( API, (_model: Model, _context: Context, options?: SimpleStreamOptions) => { - keys.push(options?.apiKey); + pushKey(keys, options); const stream = new AssistantMessageEventStream(); - queueMicrotask(() => { - if (keys.length === 1) { - stream.fail( - Object.assign( - new Error("You have hit your ChatGPT usage limit (pro plan). Try again in ~158 min."), - { status: 429 }, - ), - ); - return; - } - const message = assistant(["ok"]); - stream.push({ type: "start", partial: message }); - stream.push({ type: "done", reason: "stop", message }); - }); + queueMicrotask(() => (options?.apiKey === "switch-key" ? ok(stream) : stream.fail(authError()))); return stream; }, SOURCE_ID, ); const stream = streamSimple(model(), context, { - apiKey: "old-key", - onAuthError: async (_provider, _oldKey, error) => { - errors.push(error); - return "new-key"; + // refresh-same yields the same failing key → that attempt is skipped. + apiKey: async ctx => (ctx.error === undefined ? "old-key" : ctx.lastChance ? "switch-key" : "old-key"), + }); + for await (const _event of stream) { + // drain + } + + expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); + expect(keys).toEqual(["old-key", "switch-key"]); + }); + + it("retries a thrown usage-limit error and passes the cause to the resolver", async () => { + const keys: unknown[] = []; + const errors: unknown[] = []; + registerCustomApi( + API, + (_model: Model, _context: Context, options?: SimpleStreamOptions) => { + pushKey(keys, options); + const stream = new AssistantMessageEventStream(); + queueMicrotask(() => (keys.length === 1 ? stream.fail(usageLimitError()) : ok(stream))); + return stream; + }, + SOURCE_ID, + ); + + const stream = streamSimple(model(), context, { + apiKey: async ctx => { + if (ctx.error !== undefined) errors.push(ctx.error); + return ctx.error === undefined ? "old-key" : "new-key"; }, }); - for await (const _event of stream) { // drain } @@ -287,18 +306,17 @@ describe("streamSimple auth retry", () => { expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); expect(keys).toEqual(["old-key", "new-key"]); expect(errors).toHaveLength(1); - // The error surfaced to onAuthError carries the original 429 status so - // the gateway's refresh hook can branch on it. + // The cause carries the original 429 so the resolver can branch usage-limit vs 401. expect((errors[0] as { status?: number }).status).toBe(429); expect((errors[0] as Error).message).toMatch(/usage limit/i); }); - it("retries when a provider emits a usage_limit_reached error event before content", async () => { - const keys: Array = []; + it("retries a usage-limit error event before content", async () => { + const keys: unknown[] = []; registerCustomApi( API, (_model: Model, _context: Context, options?: SimpleStreamOptions) => { - keys.push(options?.apiKey); + pushKey(keys, options); const stream = new AssistantMessageEventStream(); queueMicrotask(() => { if (keys.length === 1) { @@ -306,16 +324,11 @@ describe("streamSimple auth retry", () => { stream.push({ type: "error", reason: "error", - // errorStatus deliberately omitted: matches how the codex - // provider serializes the message-only path that used to - // 502 in the gateway. error: assistantError("You have hit your ChatGPT usage limit (pro plan). Try again in ~158 min."), }); return; } - const message = assistant(["ok"]); - stream.push({ type: "start", partial: message }); - stream.push({ type: "done", reason: "stop", message }); + ok(stream); }); return stream; }, @@ -323,10 +336,8 @@ describe("streamSimple auth retry", () => { ); const stream = streamSimple(model(), context, { - apiKey: "old-key", - onAuthError: async () => "new-key", + apiKey: async ctx => (ctx.error === undefined ? "old-key" : "new-key"), }); - for await (const _event of stream) { // drain } @@ -335,13 +346,13 @@ describe("streamSimple auth retry", () => { expect(keys).toEqual(["old-key", "new-key"]); }); - it("surfaces the original usage_limit error when the retry callback declines", async () => { - const keys: Array = []; - const original = Object.assign(new Error("You have hit your ChatGPT usage limit (pro plan)."), { status: 429 }); + it("surfaces the original error when the resolver declines every retry", async () => { + const keys: unknown[] = []; + const original = usageLimitError(); registerCustomApi( API, (_model: Model, _context: Context, options?: SimpleStreamOptions) => { - keys.push(options?.apiKey); + pushKey(keys, options); const stream = new AssistantMessageEventStream(); queueMicrotask(() => stream.fail(original)); return stream; @@ -349,12 +360,9 @@ describe("streamSimple auth retry", () => { SOURCE_ID, ); - // Callback returns undefined → no sibling credential to rotate to. - // The original failure must reach the caller untouched so the client - // can decide what to do (back off, surface to user, …). const stream = streamSimple(model(), context, { - apiKey: "old-key", - onAuthError: async () => undefined, + // Decline all retries: no sibling credential to rotate to. + apiKey: async ctx => (ctx.error === undefined ? "old-key" : undefined), }); let caught: unknown; @@ -367,8 +375,32 @@ describe("streamSimple auth retry", () => { } expect(caught).toBe(original); - // Single attempt — the inner stream is only re-invoked when a new key - // is provided. expect(keys).toEqual(["old-key"]); }); + + it("fails the stream when the initial resolve yields no key", async () => { + let attempts = 0; + registerCustomApi( + API, + () => { + attempts += 1; + return new AssistantMessageEventStream(); + }, + SOURCE_ID, + ); + + const stream = streamSimple(model(), context, { apiKey: async () => undefined }); + + let caught: unknown; + try { + for await (const _event of stream) { + // drain + } + } catch (error) { + caught = error; + } + + expect((caught as Error).message).toMatch(/No API key for provider/); + expect(attempts).toBe(0); + }); }); diff --git a/packages/ai/test/stream-markup-healing.test.ts b/packages/ai/test/stream-markup-healing.test.ts index 935eb9481..79b6b74b5 100644 --- a/packages/ai/test/stream-markup-healing.test.ts +++ b/packages/ai/test/stream-markup-healing.test.ts @@ -148,6 +148,7 @@ describe("StreamMarkupHealing pattern selection", () => { expect(getStreamMarkupHealingPattern("minimax-code", "MiniMax-M2.5", { parseThinkingTags: true })).toBe( "thinking", ); + expect(getStreamMarkupHealingPattern("opencode-zen", "minimax-m3")).toBe("thinking"); expect(getStreamMarkupHealingPattern("nanogpt", "deepseek/deepseek-v4-pro")).toBe("dsml"); expect(getStreamMarkupHealingPattern("ollama-cloud", "gpt-oss:120b")).toBeUndefined(); expect(getStreamMarkupHealingPattern("openai", "deepseek-v4-pro")).toBeUndefined(); @@ -582,6 +583,39 @@ describe("Ollama provider DSML envelope healing", () => { }); }); +describe("OpenAI completions MiniMax thinking healing", () => { + it("parses OpenCode Zen MiniMax think tags into a thinking block", async () => { + const model: Model<"openai-completions"> = { + id: "minimax-m3", + name: "MiniMax M3", + api: "openai-completions", + provider: "opencode-zen", + baseUrl: "https://opencode.ai/zen/v1", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 8_192, + }; + global.fetch = mockFetch([ + chunk(model.id, { content: "visible hidden reasoning" }), + chunk(model.id, { content: " answer" }), + chunk(model.id, {}, "stop"), + "[DONE]", + ]); + + const result = await streamOpenAICompletions(model, baseContext(), { apiKey: "test-key" }).result(); + + expect(result.content).toEqual([ + { type: "text", text: "visible " }, + { type: "thinking", thinking: "hidden reasoning", thinkingSignature: undefined }, + { type: "text", text: " answer" }, + ]); + }); +}); + describe("OpenAI completions provider DSML envelope healing", () => { it("heals the envelope into a structured tool call and suppresses leaked text", async () => { const model: Model<"openai-completions"> = { diff --git a/packages/ai/test/stream.test.ts b/packages/ai/test/stream.test.ts index aaf0c6070..3d5325d42 100644 --- a/packages/ai/test/stream.test.ts +++ b/packages/ai/test/stream.test.ts @@ -652,6 +652,138 @@ describe("Generate E2E Tests", () => { else Bun.env.GOOGLE_APPLICATION_CREDENTIALS = originalGac; } }); + + it("routes impersonated_service_account ADC through IAM to the Vertex request", async () => { + const originalProject = Bun.env.GOOGLE_CLOUD_PROJECT; + const originalGcpProject = Bun.env.GCP_PROJECT; + const originalGcloudProject = Bun.env.GCLOUD_PROJECT; + const originalVertexLocation = Bun.env.GOOGLE_VERTEX_LOCATION; + const originalCloudLocation = Bun.env.GOOGLE_CLOUD_LOCATION; + const originalLocation = Bun.env.VERTEX_LOCATION; + const originalApiKey = Bun.env.GOOGLE_CLOUD_API_KEY; + const originalGac = Bun.env.GOOGLE_APPLICATION_CREDENTIALS; + const tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-vertex-impersonation-")); + const adcPath = path.join(tmpDir, "impersonated-adc.json"); + await Bun.write( + adcPath, + JSON.stringify({ + type: "impersonated_service_account", + service_account_impersonation_url: + "https://iamcredentials.googleapis.com/v1/projects/-/serviceAccounts/target@project.iam.gserviceaccount.com:generateAccessToken", + source_credentials: { + type: "authorized_user", + client_id: "client-id", + client_secret: "client-secret", + refresh_token: "refresh-token", + }, + delegates: ["projects/-/serviceAccounts/delegate@project.iam.gserviceaccount.com"], + }), + ); + const model: Model<"anthropic-messages"> = { + id: "claude-sonnet-4@20250514", + name: "Claude Sonnet 4", + api: "anthropic-messages", + provider: "google-vertex", + baseUrl: + "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/publishers/anthropic/models/claude-sonnet-4@20250514:streamRawPredict", + reasoning: true, + input: ["text", "image"], + cost: { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 }, + contextWindow: 200_000, + maxTokens: 64_000, + }; + const callOrder: string[] = []; + let iamRequest: { url: string; authorization: string | null; body: unknown } | undefined; + const captured = Promise.withResolvers<{ url: string; authorization: string | null }>(); + + try { + __resetVertexTokenCache(); + Bun.env.GOOGLE_CLOUD_PROJECT = "vertex-project"; + Bun.env.GOOGLE_VERTEX_LOCATION = "global"; + delete Bun.env.GCP_PROJECT; + delete Bun.env.GCLOUD_PROJECT; + delete Bun.env.GOOGLE_CLOUD_LOCATION; + delete Bun.env.VERTEX_LOCATION; + delete Bun.env.GOOGLE_CLOUD_API_KEY; + Bun.env.GOOGLE_APPLICATION_CREDENTIALS = adcPath; + + const events = stream( + model, + { messages: [{ role: "user", content: "Hello", timestamp: Date.now() }] }, + { + apiKey: "", + fetch: async (input, init) => { + const url = input instanceof Request ? input.url : input.toString(); + const headers = input instanceof Request ? input.headers : new Headers(init?.headers); + if (url === "https://oauth2.googleapis.com/token") { + callOrder.push("source"); + return new Response(JSON.stringify({ access_token: "source-token", expires_in: 3600 })); + } + if (url.startsWith("https://iamcredentials.googleapis.com/")) { + callOrder.push("iam"); + const bodyText = + input instanceof Request ? await input.clone().text() : String(init?.body ?? ""); + iamRequest = { url, authorization: headers.get("authorization"), body: JSON.parse(bodyText) }; + return new Response( + JSON.stringify({ + accessToken: "impersonated-token", + expireTime: new Date(Date.now() + 3_600_000).toISOString(), + }), + ); + } + callOrder.push("vertex"); + captured.resolve({ url, authorization: headers.get("authorization") }); + return new Response(JSON.stringify({ error: { message: "stop after capture" } }), { status: 400 }); + }, + }, + ); + + for await (const _event of events) { + } + + const request = await captured.promise; + + // Source refresh, then IAM generateAccessToken, then the actual Vertex call. + expect(callOrder).toEqual(["source", "iam", "vertex"]); + + // IAM exchange is authorized by the freshly minted source token, posts the + // reconstructed canonical URL, and forwards the configured delegates verbatim. + expect(iamRequest?.url).toBe( + "https://iamcredentials.googleapis.com/v1/projects/-/serviceAccounts/target@project.iam.gserviceaccount.com:generateAccessToken", + ); + expect(iamRequest?.authorization).toBe("Bearer source-token"); + expect(iamRequest?.body).toEqual({ + delegates: ["projects/-/serviceAccounts/delegate@project.iam.gserviceaccount.com"], + scope: ["https://www.googleapis.com/auth/cloud-platform"], + lifetime: "3600s", + }); + + // The impersonated token (not the source token) authorizes the Vertex request. + expect(request.url).toBe( + "https://aiplatform.googleapis.com/v1/projects/vertex-project/locations/global/publishers/anthropic/models/claude-sonnet-4@20250514:streamRawPredict", + ); + expect(request.authorization).toBe("Bearer impersonated-token"); + } finally { + __resetVertexTokenCache(); + await fs.rm(tmpDir, { recursive: true, force: true }); + if (originalProject === undefined) delete Bun.env.GOOGLE_CLOUD_PROJECT; + else Bun.env.GOOGLE_CLOUD_PROJECT = originalProject; + if (originalGcpProject === undefined) delete Bun.env.GCP_PROJECT; + else Bun.env.GCP_PROJECT = originalGcpProject; + if (originalGcloudProject === undefined) delete Bun.env.GCLOUD_PROJECT; + else Bun.env.GCLOUD_PROJECT = originalGcloudProject; + if (originalVertexLocation === undefined) delete Bun.env.GOOGLE_VERTEX_LOCATION; + else Bun.env.GOOGLE_VERTEX_LOCATION = originalVertexLocation; + if (originalCloudLocation === undefined) delete Bun.env.GOOGLE_CLOUD_LOCATION; + else Bun.env.GOOGLE_CLOUD_LOCATION = originalCloudLocation; + if (originalLocation === undefined) delete Bun.env.VERTEX_LOCATION; + else Bun.env.VERTEX_LOCATION = originalLocation; + if (originalApiKey === undefined) delete Bun.env.GOOGLE_CLOUD_API_KEY; + else Bun.env.GOOGLE_CLOUD_API_KEY = originalApiKey; + if (originalGac === undefined) delete Bun.env.GOOGLE_APPLICATION_CREDENTIALS; + else Bun.env.GOOGLE_APPLICATION_CREDENTIALS = originalGac; + } + }); }); describe("Google Vertex Provider (gemini-3-flash-preview)", () => { diff --git a/packages/ai/test/tool-argument-coercion.test.ts b/packages/ai/test/tool-argument-coercion.test.ts index 224159d5b..e0ca64378 100644 --- a/packages/ai/test/tool-argument-coercion.test.ts +++ b/packages/ai/test/tool-argument-coercion.test.ts @@ -78,6 +78,178 @@ describe("Tool argument coercion", () => { expect(result.paths).toEqual(["src/**/*.ts"]); }); + it("wraps a singleton object in an array when schema expects object array", () => { + const tool: Tool = { + name: "todo_like", + description: "", + parameters: z.object({ + ops: z.array( + z.object({ + op: z.literal("init"), + list: z.array( + z.object({ + phase: z.string(), + items: z.array(z.string()), + }), + ), + }), + ), + }), + }; + + const result = validateToolArguments(tool, { + type: "toolCall", + id: "call-singleton-object-array", + name: "todo_like", + arguments: { + ops: { + op: "init", + list: [{ phase: "Repro", items: ["capture"] }], + }, + }, + }); + + expect(result).toEqual({ + ops: [{ op: "init", list: [{ phase: "Repro", items: ["capture"] }] }], + }); + }); + + it("wraps a singleton number in an array when schema expects number array", () => { + const tool: Tool = { + name: "numeric_list", + description: "", + parameters: z.object({ values: z.array(z.number()) }), + }; + + const result = validateToolArguments(tool, { + type: "toolCall", + id: "call-singleton-number-array", + name: "numeric_list", + arguments: { values: 7 }, + }); + + expect(result).toEqual({ values: [7] }); + }); + + it("does not wrap singleton values for array expectations from failed union branches", () => { + const entry = z.object({ id: z.number() }); + const tool: Tool = { + name: "union_shape", + description: "", + parameters: z.union([z.array(entry), entry]), + }; + + const result = validateToolArguments(tool, { + type: "toolCall", + id: "call-union-shape", + name: "union_shape", + arguments: { id: "1" }, + }); + + expect(result).toEqual({ id: 1 }); + }); + + it("does not wrap singleton values for JSON Schema anyOf array branches", () => { + const tool: Tool = { + name: "json_schema_union", + description: "", + parameters: { + type: "object", + properties: { + target: { + anyOf: [ + { + type: "array", + items: { + type: "object", + properties: { a: { type: "boolean" } }, + required: ["a"], + additionalProperties: false, + }, + }, + { + type: "object", + properties: { a: { type: "boolean" } }, + required: ["a"], + additionalProperties: false, + }, + ], + }, + }, + required: ["target"], + additionalProperties: false, + }, + }; + + // The bug would silently coerce `{ a: "true" }` into `[{ a: true }]` by + // wrapping the object to satisfy the failed `anyOf` array branch and + // then coercing the inner string into a boolean. Branch tracking keeps + // the wrap from firing so the wrong shape never makes it through. + expect(() => + validateToolArguments(tool, { + type: "toolCall", + id: "call-jsonschema-union", + name: "json_schema_union", + arguments: { target: { a: "true" } }, + }), + ).toThrow("Validation failed"); + }); + + it("still wraps nested array fields inside a tag-selected Zod union branch", () => { + const tool: Tool = { + name: "tagged_union", + description: "", + parameters: z.union([ + z.object({ type: z.literal("indices"), indices: z.array(z.number()) }), + z.object({ type: z.literal("all") }), + ]), + }; + + const result = validateToolArguments(tool, { + type: "toolCall", + id: "call-tagged-union", + name: "tagged_union", + arguments: { type: "indices", indices: 1 }, + }); + + expect(result).toEqual({ type: "indices", indices: [1] }); + }); + + it("still wraps nested array fields inside a tag-selected JSON Schema anyOf branch", () => { + const tool: Tool = { + name: "tagged_json_union", + description: "", + parameters: { + anyOf: [ + { + type: "object", + properties: { + type: { const: "indices" }, + indices: { type: "array", items: { type: "number" } }, + }, + required: ["type", "indices"], + additionalProperties: false, + }, + { + type: "object", + properties: { type: { const: "all" } }, + required: ["type"], + additionalProperties: false, + }, + ], + }, + }; + + const result = validateToolArguments(tool, { + type: "toolCall", + id: "call-tagged-json-union", + name: "tagged_json_union", + arguments: { type: "indices", indices: 1 }, + }); + + expect(result).toEqual({ type: "indices", indices: [1] }); + }); + it("parses JSON objects in string values when schema expects object", () => { const tool: Tool = { name: "t4", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 5d24d55f3..f35498894 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,23 +1,162 @@ # Changelog ## [Unreleased] +### Fixed + +- Fixed the working spinner appearing to ignore Esc for 2-3 seconds when an interrupt lands mid-tool. Esc fires the abort synchronously, but the agent loop only stops the loader at `agent_end`, which it cannot reach until every in-flight tool settles in `executeToolCalls`' `await Promise.allSettled(...)` — and process/subagent/kernel-owning tools tear down gracefully (SIGTERM, 2-3s grace, SIGKILL), so the loader kept showing the unchanged "Working…/" line and read as a dropped keypress. The loader now switches to "Interrupting…" the instant Esc requests the abort and freezes intent-driven label updates until the turn unwinds (`EventController.notifyInterrupting`), so the interrupt is acknowledged immediately even while teardown completes. +- Fixed a flaky JS eval worker startup that intermittently failed unrelated CI runs. The worker-ready wait reused Bun's 5s default per-test timeout as its floor, so a slow cold-start under `--isolate` + high concurrency was aborted mid-init; terminating a still-initializing Bun worker is the documented SIGILL/SIGTRAP crash trigger, which took down the whole test file. Worker init now floors at a fixed 15s infrastructure budget (independent of, and still dominated by, a larger per-cell `timeout`), and the JS eval test suites set a 20s file-local timeout so cold starts complete instead of being torn down. +- Fixed reviewer-style subagent yields crashing the calling eval cell when a caller-supplied output schema declares `additionalProperties: false` without a `findings` property. `normalizeCompleteData` now consults the active validator before splicing collected `report_finding` entries onto the yielded payload, so injection is suppressed when the schema would reject it — keeping the executor's post-mortem validation in lockstep with the in-tool `yield` validation that already accepted the same raw payload ([#2070](https://github.com/can1357/oh-my-pi/issues/2070)) +- Fixed Anthropic empty `toolUse` stops without tool calls corrupting session history by retrying them and removing orphaned turns even at the retry cap. +- Fixed MCP tools hanging in non-yolo modes by declaring `approval = "write"` on `MCPTool` and `DeferredMCPTool`, and propagating the `approval` property through `customToolToDefinition()` in `sdk.ts` +- Fixed session resumption after a working directory is moved/renamed (e.g. `git worktree move`): `--continue` now re-roots the terminal's last session into the new directory when its original directory no longer exists, instead of silently starting a fresh empty session; cross-project `--resume ` offers to move (re-root) the session rather than only forking a duplicate copy when the source directory is gone +- Fixed Kitty OSC 5522 paste rejecting plain text as "no supported text or image data": the listing parser now decodes the `mime="."` DATA payload (whitespace-separated MIME list) Kitty actually sends, in addition to the per-type DATA packets described by the ancillary 5522-mode spec ([#2051](https://github.com/can1357/oh-my-pi/issues/2051)) +- Fixed follow-up shortcut submission of builtin slash commands so `/goal set ...` applies goal mode instead of queueing as plain text. +- Fixed Ctrl+Z crashing the agent on Windows with `TypeError: Unknown signal: SIGTSTP`. `InputController.handleCtrlZ` called `process.kill(0, "SIGTSTP")` unconditionally, but `SIGTSTP` is POSIX job-control and Bun/Node on Windows rejects the signal name from the JS side; the throw propagated out of the TUI input dispatcher as an uncaught exception. The handler now no-ops with a "Suspend (Ctrl+Z) is not supported on this platform" status on Windows, and on POSIX wraps `process.kill` in a try/catch that detaches the registered SIGCONT resume hook and re-`start()`s the TUI on failure so a rejected signal can never leave the UI stranded with a leaked listener ([#2036](https://github.com/can1357/oh-my-pi/issues/2036)). + +### Fixed + +- Fixed the `--cwd` launch flag so it is parsed and can override the startup directory instead of always falling back to the current process directory or home auto-switch target. + +## [15.10.1] - 2026-06-07 ### Added +- Added `display.smoothStreaming` setting (default `true`) to let users enable or disable smooth assistant-stream text reveal +- Added `/tan ` slash command to fork the current conversation into a background agent so tangential work can continue asynchronously while your main session stays active +- Added a background `/tan` dispatch message that records the handoff in the transcript and marks the delegated work as non-blocking +- Added `providerPromptCacheKey` support to `CreateAgentSessionOptions` so `/tan` background sessions can reuse the parent session’s prompt-cache lineage +- Added session cloning for `/tan` runs with copied artifacts and shared MCP proxy tools +- Added `SessionManager.forkFrom`’s optional `suppressBreadcrumb` mode to avoid breadcrumb updates when forking background `/tan` sessions +- Added OSC 5522 enhanced paste handling in `InputController`, so terminal clipboard events are decoded as image or text payloads and inserted without passing raw paste sequences to the editor +- Added bracketed image-path paste support in `CustomEditor` so a single pasted image file path (PNG/JPEG/GIF/WEBP) is loaded from disk and inserted as an image candidate +- Added direct support for `Image #N` insertion from pasted local image paths by routing successful image-path pastes through the same image normalization and resize flow as clipboard image pastes +- Added `/fresh` to rotate the provider-facing session id and clear in-memory provider stream/cache state without changing the local session file. +- Added a `ChatBlock` transcript primitive (`modes/components/chat-block.ts`) and a single `ctx.present(...)` sink (with `ctx.resetTranscript()`) so chat output is mounted in one place instead of the repeated `chatContainer.addChild(...)` + `ui.requestRender()` pattern scattered across controllers. `ChatBlock` carries a React/Svelte-style lifecycle — `onMount` starts effects, `onCleanup` registers teardown, `finish()` self-completes (stops timers and freezes the block at its final content), and `dispose()`/`resetTranscript()` tears everything down — so animated blocks own their own resources instead of leaking `setInterval`/`requestRender` bookkeeping into callers. The MCP "Connecting…" spinner is now such a block. +- Added a `framedBlock` output-block helper (`tui/output-block.ts`) plus a `borderColor` override and `applyBg: false` (no background fill) on output blocks, a `renderStatusLine` `iconOverride`, and an `icon.search` (magnifier) theme symbol — so tool renderers can draw self-contained muted-outline frames and search-family tools can show a magnifier instead of a checkmark. + +### Changed + +- Changed the bash tool frame to use a plain top rule instead of repeating "Bash" in the title bar, and folded minimizer raw-output artifact links into the status footer as `Artifact: `. + +- Changed grouped `read` output to use a white filled-circle mark for the group/single-read success state and omit duplicate per-file success marks inside multi-read groups. + +- Changed assistant streaming output to reveal text incrementally at 30 FPS with grapheme-safe adaptive catch-up, instead of replacing the whole message chunk-by-chunk +- Changed shimmer-driven TUI animations (working text, pending bash/eval borders, and theme activity-spinner documentation) to render at 30fps instead of 60fps. +- Changed running `task` tool agent rows to use a static `•` marker and shimmer only the subagent name, leaving descriptions, stats, and nested tool detail text solid while removing the rotating status glyph from those rows. +- Changed settings singleton method access to reuse bound methods for the active instance instead of allocating a new bound function on every `settings.get` lookup. +- Changed plan-mode approval to keep the drafted `local://-plan.md` file at its original name as the canonical plan path, so approved plans are no longer renamed when leaving plan mode +- Changed plan-mode write enforcement so only `local://` artifact files are writable during planning, blocking working-tree edits and allowing scratch or draft plan files in the local artifact area +- Changed the `todo` tool result renderer to stop redrawing every phase's full task list on each update: when a multi-phase list is rendered collapsed (the default, not manually expanded), only phases the latest update touched — the phase holding the in_progress task, any phase with a just-completed task, and phases named by the ops that ran (`init` counts as touching all) — render their tasks; untouched phases collapse to a one-line `N. Name done/total` summary. When call args are unavailable (e.g. transcript rebuilds) it falls back to the in_progress/completed-transition signals, and the manual expand toggle still shows every task. Also dropped the blank separator line previously inserted between phases. +- Changed non-agent API operations (title and commit-message generation, image generation, web search, eval `llm()`, auto-thinking classifier, memory consolidation) to use session-aware API key resolution with auth retries via `registry.resolver()` / `authStorage.resolver()`, refreshing the active credential before rotating to another account +- Changed image generation to wrap every provider fetch branch in `withAuth`, so 401 / usage-limit errors trigger credential force-refresh and rotation for authStorage-backed providers (OpenAI-hosted, antigravity, xai-oauth) while env-only providers (openrouter, gemini) stay single-attempt +- Changed web-search providers using `authStorage.getApiKey` (anthropic, exa, tavily, parallel, synthetic, zai, kimi) to wrap HTTP calls in `withAuth` for automatic credential rotation on 401 / usage-limit errors +- Changed the directory grouping for `find`, `search`, `ast_grep`, `ast_edit`, and `lsp` diagnostics from a single flat `# dir/` heading per immediate directory to a multi-level tree that folds the common path prefix into one heading. Previously every group repeated the full directory path — so results rooted outside cwd printed the absolute prefix (e.g. `/Users/me/proj/`) on every heading and nested directories were never collapsed. Now a single-child directory chain folds into one heading (`# packages/pkg/src/`, including an absolute root for out-of-cwd results), subdirectories nest one `#` deeper (`## nested/` → `### child.ts`), and each directory's own files are listed before its subdirectories. TUI hyperlink reconstruction tracks the nested directory stack across the whole output so file and code-frame links keep resolving to the correct absolute paths. +- Changed the plan-mode approval surface from an inline transcript block plus a separate bottom selector into a single fullscreen overlay (like `/copy`) and overhauled its navigation. The overlay now renders the plan per-section through `ScrollView` (line-level ↑/↓ scroll, Shift+↑/↓ to scroll faster, PgUp/PgDn, g/G) with no stray per-line `…`, and — when the terminal is wide enough and the plan has ≥2 headings — shows a compact VS Code-style section sidebar (the redundant plan-title heading and any "Contents" label are omitted). Focus moves between regions with Tab/Shift+Tab (and flows at the edges: Down past the last section or the bottom of the body drops into the approval options; Up steps back), while the sidebar glows to track the scrolled section. The sidebar can fast-jump between sections, delete a section (with `u` undo), and annotate sections with feedback (`a`); deletions and annotations are collected into refinement feedback that is submitted back to the model when the operator picks "Refine plan". Mouse works too: clicking an approval option activates it, clicking a sidebar section jumps to it, and the wheel scrolls the plan. ←/→ always drive the model-tier slider, Enter confirms, the external-editor key opens the plan, and Esc cancels. The overlay borrows the terminal's alternate screen buffer for its lifetime (`fullscreen` overlay), so the transcript stays put on the normal screen instead of bleeding through scrollback behind the modal. +- Changed the interactive controllers (command, MCP, selector, extension-UI, event), debug panels, and the status/error/warning helpers to render chat output through `ctx.present(...)` instead of appending to `chatContainer` and calling `ui.requestRender()` directly; transcript rebuilds dispose live blocks via `ctx.resetTranscript()` so animated blocks' timers stop on reset. +- Changed tool-execution block rendering so the container (`ToolExecutionComponent`) is a transparent passthrough — it no longer inserts a top/bottom blank line, adds left/right padding, or paints a state-colored background behind tool output. Tools with substantial body now self-frame with a muted outline and the tool title in the frame's top bar (`edit`/`apply_patch`, `write`, `ask`, `todo`, `github`, `goal`, `inspect_image`, `search_tool_bm25`, `task`), matching the already-framed `bash`/`read`/`eval`/`debug`/`web_search`/`lsp` blocks, while streaming/in-progress and trivial results collapse to a clean status line. The search-family list tools (`find`, `search`, `ast_grep`) and `job` render frameless/minimal; `find`/`search`/`ast_grep` show a magnifier on success instead of a checkmark, and `job` drops its `Job:` label prefix (the per-job rows are self-describing). The `search_tool_bm25`, `github`, and `inspect_image` frames draw with no background fill, and `inspect_image`'s label was shortened to `Inspect`. +- Changed the plan-mode active prompt (`prompts/system/plan-mode-active.md`) to make plans decision-complete and cut filler. Added an Objective framing ("another engineer can execute end-to-end without making a single design decision"), a shared "Resolving Unknowns" section (explore discoverable facts before asking; reserve `ask` for non-derivable preferences/tradeoffs with 2–4 options + a recommended default), and a single shared "The Plan" structure (Context / Approach grouped by behavior not file-by-file / ≤5 Critical files / Verification / Assumptions) that replaces the per-branch structure guidance previously duplicated across the iterative and parallel workflows. Added explicit prohibitions on sections that decide nothing (Non-Goals, Out of Scope, Alternatives Considered, Risks/Mitigations boilerplate, Future Work), on enumerating every file/line, and on inventing schema/validation/precedence policy the request never established. +- Changed completion notifications (`completion.notify`) to fire whenever the agent yields its turn, including in the foreground. The `agent_end` notification was previously gated behind background mode (`isBackgrounded`), so an ordinary foreground turn never emitted one; the gate is gone and the desktop toast now fires on every normal turn completion (still skipped for aborted/error turns and when `completion.notify` is `off`). +- Changed the in-progress `task` tool block to keep the shared `context` brief (`# Goal` / `# Constraints` background) visible after the first progress snapshot arrives, instead of dropping it the moment the streaming call view was replaced by the result frame, and to stop animating a spinner/clock next to the `Task` frame header while running — the per-agent body lines already carry their own running spinner, so the header now shows a static state icon (matching the completed/failed header icons). The context is rendered through a shared `buildContextSection` helper that also undoes per-field double-encoding, so the brief reads cleanly in the result frame even though `renderResult` receives the raw (un-repaired) tool args. +- Changed the messaging shown when you press Esc to interrupt a streaming turn from the ambiguous `Operation aborted` / `Tool execution was aborted: Request was aborted` to `Interrupted by user`, so a deliberate user interrupt no longer reads like an internal failure. Every Esc/flush interrupt path (`onEscape` while streaming, the queued-message restore-and-abort path, and the empty-submit queue flush) threads the reason through `AgentSession.abort({ reason })` → `Agent.abort(reason)` so it rides the `AbortController` onto the aborted assistant message's `errorMessage`; the turn label renders it verbatim on both the live and replay paths, and the synthetic placeholder results paired with in-flight tool calls now read `Tool execution was aborted: Interrupted by user`. Aborts that carry no reason still fall back to the retry-aware `Operation aborted` generic. Transcript label resolution is centralized in `resolveAbortLabel` (`session/messages.ts`). + +### Removed + +- Removed the `/background` (and `/bg`) slash command and the background-mode subsystem it was the sole entry point for — `InteractiveMode.isBackgrounded`, `createBackgroundUiContext`, `handleBackgroundEvent`, and every `isBackgrounded` guard across the input/event/extension-UI controllers and UI helpers. The command suspended the whole process group via `SIGTSTP` (a leftover testing shortcut) instead of detaching the running agent, which is not the expected workflow — use terminal panes or a multiplexer instead. + +### Fixed + +- Fixed session auto-retry for generic `upstream_error: Upstream request failed` gateway failures. + +- Fixed inline `find` and `search` result blocks to align with grouped `read` output and render their success headers with the normal tool-title color instead of accent blue. + +- Fixed the working-status shimmer to opt into the loader's 30fps animated-message repaint path while keeping both the status spinner and pending bash/eval tool spinners on their normal 80 ms glyph cadence. +- Fixed consecutive `read` tool calls failing to collapse into a single grouped block when a reasoning model emits one read per completion (`[thinking, read]`). The read group was reset on every assistant `message_start`, so each read rendered as its own one-entry `Read …` line; now a read run accretes across completions and is broken only by a rendered non-empty text/thinking block, a non-read tool, or a user/IRC message — matching the transcript-rebuild path. `ReadToolGroupComponent` now reports its live/finalized state so the growing `Read (N)` header repaints correctly on native-scrollback (risk) terminals. +- Fixed the `task` tool shared-context brief rendering raw Markdown headings (`# Goal`, `# Constraints`) inside framed call/result blocks instead of using the normal Markdown renderer. +- Fixed the animated pending border on `bash`/`eval` blocks leaving a frozen dark "bar" segment behind after a backgrounded command finalized through the async update path. Once a command is auto-backgrounded (`details.async.state === "running"`) the block stays "partial" in the TUI until the async job-manager delivers the final result, but it also gets committed to native scrollback — so a mid-sweep shimmer frame baked a stray darkened border segment into the committed copy. The border now stops animating (and the 60fps redraw loop stops) the moment a block enters the backgrounded state, so the committed frame is a clean static border. +- Fixed cold `omp` launch to clear native terminal history on the first paint, avoiding a once-per-launch duplicate welcome/transcript copy before the normal session replay. +- Fixed plan approval resolution so `resolve` with `action: "apply"` can still find the plan file when `extra.title` is missing or stale by falling back to the current plan path and most-recent local plan artifacts +- Fixed the search-family tool magnifier glyph (`find`, `search`, `ast_grep`, `search_tool_bm25`) to use the `accent` title color instead of `success` green, so the icon matches the tool title in the status header instead of standing out +- Fixed TTSR stream interrupts to pass the matched rule name through the abort reason, so aborted in-flight tool placeholders say why they were stopped instead of `Request was aborted`. +- Fixed URL reads for binary/special payloads to reuse local readers: remote archives list their root entries, SQLite databases show their table overview, notebooks render as editable cells, and unrenderable binary returns a metadata notice instead of decoded byte garbage. +- Fixed pasted image-file paths that cannot be loaded to fall back to normal text paste with status feedback instead of disappearing. +- Fixed tool-output file paths not being clickable OSC 8 `file://` hyperlinks in several renderers. `read` titles for plain text and image files (the common case) emitted no link at all because the renderer only linked when a `resolvedPath` was recorded — which the ordinary file/image read paths never set, keeping the absolute path only in `meta.source`; the renderer now falls back to that source path. `write` headers were never wrapped in a hyperlink and now link to the absolute path written (file, archive entry, SQLite, and conflict resolutions). `edit`/`apply_patch` headers wrapped the model-supplied (often cwd-relative) argument path, producing a root-anchored `file:///rel/path` URI; they now link the absolute `details.path` instead. Finally, `search`, `ast_grep`, and `ast_edit` produced doubled link targets (`/proj/src/src/file.ts`) for searches scoped to a subdirectory, because the renderer resolved the cwd-relative display paths against the scope directory rather than cwd — the scoped-search base is now the session cwd (with the scoped file's absolute path still seeding single-file body lines). +- Fixed `omp dry-balance --bench` to recover from 401 token failures by re-minting the failing OAuth credential in place before switching accounts +- Fixed the bash tool corrupting commands that embed multi-byte UTF-8 (e.g. `✓`/`×` inside a `grep -E` pattern) ahead of a trailing `| head`/`| tail`. The `bash.stripTrailingHeadTail` rewrite cut at char-offset positions reported by `brush-parser` while slicing the command by byte offset, so the trailing-pipe strip landed mid-pattern and dropped the closing quote — turning `… |✓|×|XCTAssert" | tail -80` into `… |✓|×-80` and making execution fail with `pi-natives:command: unterminated double quote`. Fixed in `pi_shell::fixup` (`@oh-my-pi/pi-natives`). +- Fixed `omp dry-balance --bench` to recover from 401 token failures by re-minting the failing OAuth credential in place before switching accounts +- Fixed duplicate file entries in grouped outputs for `find`, `search`, `ast_grep`, `ast_edit`, and `lsp` diagnostics when the same path appeared multiple times +- Fixed search, grep, and edit output rendering so repeated directory group blank-line boundaries no longer break nested path/link reconstruction +- Fixed `omp dry-balance --bench` flooding the terminal with staircased, duplicated spinner/status lines (and an indented summary) when the tty has ONLCR/OPOST disabled (raw mode). The interactive progress region separated rows with a bare LF and repositioned with a column-preserving `\x1b[A` cursor-up, both of which only land at column 0 when the terminal translates LF→CRLF; with that translation off, every 80 ms redraw cascaded down and to the right into scrollback. The live region now carriage-returns before every cleared row, terminates each row with CRLF, and caps each row to the terminal width so a wrapped line cannot desync the cursor-up from the logical line count. +- Fixed inconsistent vertical spacing between transcript blocks: some blocks (tool results from `search`/`find` and other renderer-backed tools) rendered with a doubled gap (a leading `Spacer` plus the content box's own `paddingY`), while others (the grouped `read` card, file-mention lists, IRC cards) rendered with no gap at all. Vertical spacing is now owned entirely by the chat renderer: `TranscriptContainer` strips each block's plain-blank top/bottom edges and inserts exactly one blank line between consecutive blocks, so every block is separated by a single consistent gap regardless of which component produced it. Individual components (assistant/user/tool/read-group/bash/eval/skill/custom/hook/compaction/branch/todo-reminder/plan-review messages) no longer emit their own leading `Spacer`/`paddingY` for separation, and multi-row groups (IRC cards, file-mention lists, completed-job batches, and the bordered command/`/changelog`/`/context`/version/OAuth/debug panels) are wrapped as single `TranscriptBlock` children so the renderer spaces them as one unit. Background-colored box padding is preserved as block-internal design. +- Fixed `resolve` with `action: "discard"` surfacing a hard `isError` "No pending action to resolve" failure to the model when the agent asked to cancel a staged action (e.g. an `ast_edit` preview) but nothing was pending. A discard is a request to reach the "no staged change" end-state, which already holds in that case, so it is now honored as a successful cancellation (`"Nothing to discard; no pending action remains."` with `details.action: "discard"`) instead of an error. `action: "apply"` with no pending action still errors. +- Fixed the collapsed tool-output expand hint rendering double brackets (e.g. `((Ctrl+O for more))`) — the `EXPAND_HINT` text already carried its own parentheses and then `formatExpandHint` wrapped it again with the theme's bracket glyphs. The hint now resolves the key actually bound to `app.tools.expand` at render time and reads `⟨: Expand⟩` (e.g. `⟨Ctrl+O: Expand⟩`), so a single bracket pair surrounds it and a user remap of the expand keybinding is reflected instead of a hard-coded `Ctrl+O`. +- Fixed the `edit`/`apply_patch` tool dropping its outlined frame while streaming/in-progress (only the final result was framed); the in-progress diff preview now renders inside the same muted frame as the completed result. +- Fixed the `todo` and `job` tools rendering a success icon and success styling on a failed/error result; error results now show the error icon and a red frame border. +- Fixed `debug` tool refusing every `dlv` launch on Go modules. The launch handler ran `validateLaunchProgram` before adapter selection and rejected any directory program with `launch program resolves to a directory`, while dlv's default `mode=debug` requires a Go package path (a directory or `.go` source file). Adapter resolution now precedes validation, directory programs prefer adapters that advertise `acceptsDirectoryProgram` before falling back to native extensionless debuggers, the rejection only fires when the resolved adapter does not advertise that flag (set on `dlv` in `dap/defaults.json`), and dlv's `mode` is derived from the program shape — directories and `.go` files launch as `mode=debug`, other files as `mode=exec` — so `omp` can debug both Go packages and pre-built binaries ([#2020](https://github.com/can1357/oh-my-pi/issues/2020)). + +## [15.10.0] - 2026-06-06 + +### Breaking Changes + +- Replaced the `providers.parallelFetch` boolean setting with the `providers.fetch` enum (`auto` / `native` / `trafilatura` / `lynx` / `parallel` / `jina`) that selects the URL reader-backend priority for the `read`/`fetch` tool, mirroring `providers.image`/`providers.webSearch`. Existing configs are migrated automatically: the legacy key is dropped and the new `auto` default applies. + +### Added + +- Added a GitHub Actions read handler to the `read`/web-fetch GitHub scraper. Fetching `github.com/{owner}/{repo}/actions/runs/{id}` renders the run metadata plus a per-job breakdown (steps listed for any job that did not succeed), and `…/actions/runs/{id}/job/{id}` (also the API-style `…/jobs/{id}`) renders a single job's metadata, step table, and full plain-text logs. Logs are fetched via the `actions/jobs/{id}/logs` redirect using `GITHUB_TOKEN`/`GH_TOKEN` when present, with the per-line ISO timestamp prefix and leading BOM stripped; the section degrades to an explicit notice when logs are unavailable (no token, private repo, or expired/unfinalized run). + +### Changed + +- Changed eval `agent()` subagents so they are never subject to the `task.maxRuntimeMs` wall-clock cap. The parent cell's idle watchdog is already suspended for the entire bridge call (`withBridgeTimeoutPause`), so a long-running fan-out/recovery workflow must not be killed by a per-subagent runtime limit. `runEvalAgent` now passes `maxRuntimeMs: 0` to `runSubprocess`, which honors an explicit `ExecutorOptions.maxRuntimeMs` override over the inherited setting. +- Changed interactive timing behavior so `PI_TIMING=x pi` preloads the module timer before the CLI graph loads and includes the `(modules)` report. `PI_TIMING=full` now also exits after printing, matching `PI_TIMING=x`, so full module reports are usable for cold-start measurement without launching the TUI. Added the root `dev:timing` script for the same profiled startup path. +- Changed coding-agent startup imports so normal TUI launch imports `InteractiveMode` directly, keeps print/RPC/ACP runners on their branch-only paths, and moves marketplace auto-update work behind a lightweight deferred starter. +- Changed cold-launch setup gating so the full setup wizard (every scene plus the overlay and their TUI/OAuth/web-search/theme dependencies) is no longer statically imported by `main.ts`. The current setup version now lives in a tiny dependency-free `modes/setup-version` module, and the wizard barrel is lazy-loaded only when the stored setup version is stale or the wizard is forced — the common up-to-date launch skips loading it entirely. +- Changed cold-launch startup imports so the hot-path CLI files no longer pull the full `@oh-my-pi/pi-ai` barrel: `commands/launch.ts` and `cli/args.ts` import `THINKING_EFFORTS`/`Effort` from the tiny `@oh-my-pi/pi-ai/effort` module, and `config/model-registry.ts` now imports its ~20 symbols from narrow subpaths (`api-registry`, `model-cache`, `model-manager`, `model-thinking`, `models`, `provider-models`, `types`, `utils/event-stream`) instead of the barrel — so launching no longer eagerly loads every provider, auth, OAuth, and usage module re-exported by the barrel. +- Changed the `read`/`fetch` HTML reader-backend priority to `native > trafilatura > lynx > parallel > jina` (was `parallel > jina > trafilatura > lynx > native`). The in-process native `htmlToMarkdown` runs first — instant, no network, full-fidelity — so the common case no longer depends on a remote service, and a stalled remote backend can no longer mask it. Selecting a specific backend via `providers.fetch` tries it first, then the rest fall back. The low-quality gate (`>100` chars and not `isLowQualityOutput`) now applies uniformly to every backend; when none clears it, the highest-priority substantial-but-low-quality output is still surfaced so the `llms.txt` / document-extraction fallbacks keep running. + +### Fixed + +- Fixed eval `agent()` failures surfacing as an opaque `RuntimeError: bridge call '__agent__' failed` with no reason. When a subagent aborted, `runEvalAgent` built its failure message with `result.error ?? result.stderr ?? result.abortReason ?? …`, but `result.stderr` is the empty string on a clean abort (and `result.error` is gated on a non-empty `stderr`), so the nullish chain stopped at `""` and never reached `abortReason`. The empty string propagated through the loopback bridge and the Python prelude's `RuntimeError(msg or "bridge call … failed")`, discarding the real reason. The chain now uses `||` so an empty `stderr` falls through to `abortReason`. +- Fixed subagent aborts being mislabeled as the generic "Cancelled by caller" when the abort originated inside the subagent's own turn (`stopReason: "aborted"` with no caller signal and no runtime-limit timer). `runSubprocess` now prefers the aborted assistant message's `errorMessage` (e.g. "Request was aborted" or a specific stream error) for that case, while a real caller signal or wall-clock abort still reports its precise reason. +- Fixed a long streaming tool preview that alone overflows the viewport dropping its scrolled-off head on ED3-risk terminals (ghostty/kitty/iTerm2/…). When expanded with `Ctrl+O`, a streaming `write` (content streaming in) and a streaming `eval` (stdout streaming below its fixed code cell) render top-anchored and grow append-only, but the tool block never reported itself append-only to the transcript, so the renderer's commit-as-you-go boundary stopped at the block start and the earlier rows that scrolled above the viewport were committed nowhere — they vanished, leaving the preview looking like a viewport-tall circular buffer. `ToolExecutionComponent` now implements `isTranscriptBlockAppendOnly()` (gated on `isTranscriptBlockFinalized()`, so it also covers partial-result streams like `eval`), delegating to a renderer-declared `isStreamingPreviewAppendOnly` predicate so the expanded stream commits its head exactly like a streamed assistant reply. Collapsed previews (bounded sliding tail windows) and finalized/result previews (which can collapse to a capped view) stay deferred. +- Fixed `read`/`fetch` silently dropping whole list sections on pages with malformed list markup — stray ``, text, or `