Merge branch 'main' into fix/1492-hashline-bare-body-prefix
This commit is contained in:
@@ -16,6 +16,9 @@ concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true
|
||||
|
||||
jobs:
|
||||
# scripts/release.ts pushes the version-bump commit and its `v*` tag
|
||||
# atomically (`git push --atomic origin main refs/tags/v*`), so a release
|
||||
|
||||
Generated
+4
-4
@@ -2331,7 +2331,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pi-ast"
|
||||
version = "15.9.67"
|
||||
version = "15.10.1"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"ast-grep-core",
|
||||
@@ -2399,7 +2399,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pi-iso"
|
||||
version = "15.9.67"
|
||||
version = "15.10.1"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"libc",
|
||||
@@ -2411,7 +2411,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pi-natives"
|
||||
version = "15.9.67"
|
||||
version = "15.10.1"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"arboard",
|
||||
@@ -2457,7 +2457,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pi-shell"
|
||||
version = "15.9.67"
|
||||
version = "15.10.1"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"brush-builtins",
|
||||
|
||||
+1
-1
@@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"]
|
||||
resolver = "3"
|
||||
|
||||
[workspace.package]
|
||||
version = "15.9.67"
|
||||
version = "15.10.1"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
authors = ["Can Boluk"]
|
||||
|
||||
@@ -15,7 +15,7 @@
|
||||
},
|
||||
"packages/agent": {
|
||||
"name": "@oh-my-pi/pi-agent-core",
|
||||
"version": "15.9.67",
|
||||
"version": "15.10.1",
|
||||
"dependencies": {
|
||||
"@oh-my-pi/pi-ai": "catalog:",
|
||||
"@oh-my-pi/pi-natives": "catalog:",
|
||||
@@ -30,7 +30,7 @@
|
||||
},
|
||||
"packages/ai": {
|
||||
"name": "@oh-my-pi/pi-ai",
|
||||
"version": "15.9.67",
|
||||
"version": "15.10.1",
|
||||
"dependencies": {
|
||||
"@bufbuild/protobuf": "catalog:",
|
||||
"@oh-my-pi/pi-utils": "catalog:",
|
||||
@@ -44,7 +44,7 @@
|
||||
},
|
||||
"packages/coding-agent": {
|
||||
"name": "@oh-my-pi/pi-coding-agent",
|
||||
"version": "15.9.67",
|
||||
"version": "15.10.1",
|
||||
"bin": {
|
||||
"omp": "src/cli.ts",
|
||||
},
|
||||
@@ -90,7 +90,7 @@
|
||||
},
|
||||
"packages/hashline": {
|
||||
"name": "@oh-my-pi/hashline",
|
||||
"version": "15.9.67",
|
||||
"version": "15.10.1",
|
||||
"dependencies": {
|
||||
"diff": "catalog:",
|
||||
"lru-cache": "catalog:",
|
||||
@@ -101,7 +101,7 @@
|
||||
},
|
||||
"packages/mnemopi": {
|
||||
"name": "@oh-my-pi/pi-mnemopi",
|
||||
"version": "15.9.67",
|
||||
"version": "15.10.1",
|
||||
"bin": {
|
||||
"mnemopi": "src/cli.ts",
|
||||
},
|
||||
@@ -118,7 +118,7 @@
|
||||
},
|
||||
"packages/natives": {
|
||||
"name": "@oh-my-pi/pi-natives",
|
||||
"version": "15.9.67",
|
||||
"version": "15.10.1",
|
||||
"devDependencies": {
|
||||
"@napi-rs/cli": "catalog:",
|
||||
"@types/bun": "catalog:",
|
||||
@@ -126,7 +126,7 @@
|
||||
},
|
||||
"packages/stats": {
|
||||
"name": "@oh-my-pi/omp-stats",
|
||||
"version": "15.9.67",
|
||||
"version": "15.10.1",
|
||||
"bin": {
|
||||
"omp-stats": "./src/index.ts",
|
||||
},
|
||||
@@ -151,7 +151,7 @@
|
||||
},
|
||||
"packages/swarm-extension": {
|
||||
"name": "@oh-my-pi/swarm-extension",
|
||||
"version": "15.9.67",
|
||||
"version": "15.10.1",
|
||||
"bin": {
|
||||
"omp-swarm": "src/cli.ts",
|
||||
},
|
||||
@@ -167,7 +167,7 @@
|
||||
},
|
||||
"packages/tui": {
|
||||
"name": "@oh-my-pi/pi-tui",
|
||||
"version": "15.9.67",
|
||||
"version": "15.10.1",
|
||||
"dependencies": {
|
||||
"@oh-my-pi/pi-natives": "catalog:",
|
||||
"@oh-my-pi/pi-utils": "catalog:",
|
||||
@@ -208,7 +208,7 @@
|
||||
},
|
||||
"packages/utils": {
|
||||
"name": "@oh-my-pi/pi-utils",
|
||||
"version": "15.9.67",
|
||||
"version": "15.10.1",
|
||||
"dependencies": {
|
||||
"@oh-my-pi/pi-natives": "catalog:",
|
||||
"beautiful-mermaid": "catalog:",
|
||||
@@ -248,15 +248,15 @@
|
||||
"@huggingface/transformers": "^4.2.0",
|
||||
"@mozilla/readability": "^0.6.0",
|
||||
"@napi-rs/cli": "3.7.0",
|
||||
"@oh-my-pi/hashline": "15.9.67",
|
||||
"@oh-my-pi/omp-stats": "15.9.67",
|
||||
"@oh-my-pi/pi-agent-core": "15.9.67",
|
||||
"@oh-my-pi/pi-ai": "15.9.67",
|
||||
"@oh-my-pi/pi-coding-agent": "15.9.67",
|
||||
"@oh-my-pi/pi-mnemopi": "15.9.67",
|
||||
"@oh-my-pi/pi-natives": "15.9.67",
|
||||
"@oh-my-pi/pi-tui": "15.9.67",
|
||||
"@oh-my-pi/pi-utils": "15.9.67",
|
||||
"@oh-my-pi/hashline": "15.10.1",
|
||||
"@oh-my-pi/omp-stats": "15.10.1",
|
||||
"@oh-my-pi/pi-agent-core": "15.10.1",
|
||||
"@oh-my-pi/pi-ai": "15.10.1",
|
||||
"@oh-my-pi/pi-coding-agent": "15.10.1",
|
||||
"@oh-my-pi/pi-mnemopi": "15.10.1",
|
||||
"@oh-my-pi/pi-natives": "15.10.1",
|
||||
"@oh-my-pi/pi-tui": "15.10.1",
|
||||
"@oh-my-pi/pi-utils": "15.10.1",
|
||||
"@opentelemetry/api": "^1.9.1",
|
||||
"@opentelemetry/context-async-hooks": "^2.7.1",
|
||||
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
|
||||
@@ -395,7 +395,7 @@
|
||||
|
||||
"@huggingface/jinja": ["@huggingface/jinja@0.5.9", "", {}, "sha512-uWTG+l3VJRsl7EXxYizuL3P+cCPoc3cRqbWWRcQN0FhejRfbdq0RNhCmbY/YDtnTcz9icdLYuLDjsnz4d8JMuw=="],
|
||||
|
||||
"@huggingface/tasks": ["@huggingface/tasks@0.21.6", "", {}, "sha512-XfLE2clF0uHw7kMb6HHMkpyJ+bmu2T0EZ8O1WxbJXIqdRwKRB0RM7Y639Ph3aYj2GjFLJhNo1Lz8y0jn9k10LQ=="],
|
||||
"@huggingface/tasks": ["@huggingface/tasks@0.21.7", "", {}, "sha512-GuEXszIkir4j/Oywp4hXP+wfwojo/SKWA/omroNkzWWgqUGiOQ5p6HuyXcDOcinYnLQW1WsO8fwdEvtLTZbA4w=="],
|
||||
|
||||
"@huggingface/tokenizers": ["@huggingface/tokenizers@0.1.3", "", {}, "sha512-8rF/RRT10u+kn7YuUbUg0OF30K8rjTc78aHpxT+qJ1uWSqxT1MHi8+9ltwYfkFYJzT/oS+qw3JVfHtNMGAdqyA=="],
|
||||
|
||||
@@ -1159,7 +1159,7 @@
|
||||
|
||||
"onnxruntime-web": ["onnxruntime-web@1.26.0-dev.20260416-b7804b056c", "", { "dependencies": { "flatbuffers": "^25.1.24", "guid-typescript": "^1.0.9", "long": "^5.2.3", "onnxruntime-common": "1.24.0-dev.20251116-b39e144322", "platform": "^1.3.6", "protobufjs": "^7.2.4" } }, "sha512-MD6Ss4GSpQBo6zqoJzyT9LRbKYs7x/JVN23FT24EcEvlqF4VuzPOeH6X38orZPKHQDbprn7K+SBpu0/mj2CQiw=="],
|
||||
|
||||
"openai": ["openai@6.39.1", "", { "peerDependencies": { "ws": "^8.18.0", "zod": "^3.25 || ^4.0" }, "optionalPeers": ["ws", "zod"], "bin": { "openai": "bin/cli" } }, "sha512-z3dO9fEWOXBzlXynVb/xZ/tujzUjFWQWn3C0n0mw6Vo0zJTbEkaN4b2cLWjhJ6haJQx8LlREoafHRl+Gu/Hl+A=="],
|
||||
"openai": ["openai@6.42.0", "", { "peerDependencies": { "ws": "^8.18.0", "zod": "^3.25 || ^4.0" }, "optionalPeers": ["ws", "zod"] }, "sha512-1WFEt/uXMXOLhYRNkgJWo08Y2YNvNwpVU72K7ibrWgWpNOXd4VojXLbe6SQ4bLiUQ3Y8jz4IiyVkylJCL1DtZg=="],
|
||||
|
||||
"option": ["option@0.2.4", "", {}, "sha512-pkEqbDyl8ou5cpq+VsnQbe/WlEy5qS7xPzMS1U55OCG9KPvwFD46zDbxQIj3egJSFc3D+XhYOPUzz49zQAVy7A=="],
|
||||
|
||||
|
||||
@@ -70,6 +70,7 @@ const CP_KP_EQUALS: i32 = 57415;
|
||||
const MOD_SHIFT: u32 = 1;
|
||||
const MOD_ALT: u32 = 2;
|
||||
const MOD_CTRL: u32 = 4;
|
||||
const MOD_SUPER: u32 = 8;
|
||||
const MOD_NUM_LOCK: u32 = 128;
|
||||
|
||||
/// Event types from Kitty keyboard protocol (flag 2).
|
||||
@@ -457,6 +458,10 @@ fn parse_key_id(key_id: &str) -> Option<ParsedKeyId<'_>> {
|
||||
modifier |= MOD_SHIFT;
|
||||
continue;
|
||||
},
|
||||
b's' | b'S' if p.eq_ignore_ascii_case("super") => {
|
||||
modifier |= MOD_SUPER;
|
||||
continue;
|
||||
},
|
||||
b'a' | b'A' if p.eq_ignore_ascii_case("alt") => {
|
||||
modifier |= MOD_ALT;
|
||||
continue;
|
||||
@@ -1378,7 +1383,7 @@ fn parse_functional(bytes: &[u8]) -> Option<ParsedKittySequence> {
|
||||
|
||||
fn format_kitty_key(parsed: &ParsedKittySequence) -> Option<Cow<'static, str>> {
|
||||
let effective_mod = parsed.modifier & !LOCK_MASK;
|
||||
if effective_mod & !(MOD_SHIFT | MOD_CTRL | MOD_ALT) != 0 {
|
||||
if effective_mod & !(MOD_SHIFT | MOD_CTRL | MOD_ALT | MOD_SUPER) != 0 {
|
||||
return None;
|
||||
}
|
||||
let effective_codepoint =
|
||||
@@ -1480,6 +1485,9 @@ fn format_with_mods(mods: u32, key_name: &str) -> String {
|
||||
if mods & MOD_ALT != 0 {
|
||||
result.push_str("alt+");
|
||||
}
|
||||
if mods & MOD_SUPER != 0 {
|
||||
result.push_str("super+");
|
||||
}
|
||||
result.push_str(key_name);
|
||||
result
|
||||
}
|
||||
@@ -1585,7 +1593,11 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn parse_key_ignores_kitty_sequences_with_unsupported_modifiers() {
|
||||
assert_eq!(parse_key_inner(b"\x1b[99;9u", true).as_deref(), None);
|
||||
// Hyper (16) and meta (32) are kitty modifier bits we do not surface
|
||||
// because nothing in the editor binds them. Wire mod 17 = mask 16 = hyper.
|
||||
assert_eq!(parse_key_inner(b"\x1b[99;17u", true).as_deref(), None);
|
||||
// Wire mod 33 = mask 32 = meta.
|
||||
assert_eq!(parse_key_inner(b"\x1b[99;33u", true).as_deref(), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -1675,4 +1687,32 @@ mod tests {
|
||||
assert!(matches_key_inner(b"\x1b[109;7u", "ctrl+alt+m", true));
|
||||
assert!(matches_key_inner(b"\x1b[27;7;109~", "ctrl+alt+m", false));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn super_alt_backspace_matches_ghostty_default() {
|
||||
// Issue #2064: Ghostty on macOS reports Option+Backspace as kitty
|
||||
// modifier 11 (wire) = 10 (mask) = super(8)|alt(2). Before super
|
||||
// support landed, the matcher rejected this entirely.
|
||||
assert!(matches_key_inner(b"\x1b[127;11u", "super+alt+backspace", true));
|
||||
assert!(matches_key_inner(b"\x1b[127;11u", "alt+super+backspace", true));
|
||||
assert_eq!(parse_key_inner(b"\x1b[127;11u", true).as_deref(), Some("alt+super+backspace"));
|
||||
// Plain alt+backspace must still NOT match — the modifier really is super|alt.
|
||||
assert!(!matches_key_inner(b"\x1b[127;11u", "alt+backspace", true));
|
||||
// And plain backspace (mod 0) must still not match a super+alt-modified press.
|
||||
assert!(!matches_key_inner(b"\x1b[127;11u", "backspace", true));
|
||||
// Release events stay ignored: super+alt+backspace release must not match a press.
|
||||
assert!(!matches_key_inner(b"\x1b[127;11:3u", "super+alt+backspace", true));
|
||||
assert_eq!(parse_key_inner(b"\x1b[127;11:3u", true).as_deref(), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn super_modifier_parses_for_arbitrary_keys() {
|
||||
// Cmd+letter on macOS under kitty flag=1+: super(8)+'a'(97) → wire mod 9.
|
||||
assert!(matches_key_inner(b"\x1b[97;9u", "super+a", true));
|
||||
assert_eq!(parse_key_inner(b"\x1b[97;9u", true).as_deref(), Some("super+a"));
|
||||
// Cmd+Shift+letter: super(8)|shift(1) = 9 mask, wire 10.
|
||||
assert!(matches_key_inner(b"\x1b[97;10u", "super+shift+a", true));
|
||||
assert!(matches_key_inner(b"\x1b[97;10u", "shift+super+a", true));
|
||||
assert_eq!(parse_key_inner(b"\x1b[97;10u", true).as_deref(), Some("shift+super+a"));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -68,5 +68,5 @@ use napi_derive::napi;
|
||||
/// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in
|
||||
/// `packages/natives/native/index.js` (which derives the name from
|
||||
/// `package.json#version`).
|
||||
#[napi(js_name = "__piNativesV15_9_67")]
|
||||
#[napi(js_name = "__piNativesV15_10_1")]
|
||||
pub const fn pi_natives_version_sentinel() {}
|
||||
|
||||
@@ -154,7 +154,12 @@ fn try_strip_head_tail(
|
||||
// span under-reports when its suffix contains unlocated `IoRedirect`s
|
||||
// (e.g. the synthetic `2>&1` inserted by `|&`).
|
||||
let bytes = cmd.as_bytes();
|
||||
let last_start = last_loc.start.index;
|
||||
let Some(last_start) = byte_offset(cmd, last_loc.start.index) else {
|
||||
return default;
|
||||
};
|
||||
let Some(last_end) = byte_offset(cmd, last_loc.end.index) else {
|
||||
return default;
|
||||
};
|
||||
let Some(head) = cmd.get(..last_start) else {
|
||||
return default;
|
||||
};
|
||||
@@ -172,7 +177,7 @@ fn try_strip_head_tail(
|
||||
// Reported text starts at the pipe and is right-trimmed. The deletion
|
||||
// range walks back through any leading whitespace so the rewrite is
|
||||
// contiguous.
|
||||
let stripped_text = cmd[pipe_pos..last_loc.end.index].trim_end().to_owned();
|
||||
let stripped_text = cmd[pipe_pos..last_end].trim_end().to_owned();
|
||||
if stripped_text.is_empty() {
|
||||
return default;
|
||||
}
|
||||
@@ -180,7 +185,7 @@ fn try_strip_head_tail(
|
||||
while delete_start > 0 && matches!(bytes[delete_start - 1], b' ' | b'\t') {
|
||||
delete_start -= 1;
|
||||
}
|
||||
ranges.push((delete_start, last_loc.end.index));
|
||||
ranges.push((delete_start, last_end));
|
||||
stripped.push(stripped_text);
|
||||
HeadTailOutcome { stripped: true, last_idx: n - 2 }
|
||||
}
|
||||
@@ -253,10 +258,14 @@ fn try_strip_2to1(
|
||||
let Some(name_loc) = name_word.loc.as_ref() else {
|
||||
return;
|
||||
};
|
||||
let mut anchor = name_loc.end.index;
|
||||
let Some(mut anchor) = byte_offset(cmd, name_loc.end.index) else {
|
||||
return;
|
||||
};
|
||||
for item in &suffix.0 {
|
||||
if let Some(loc) = item.location() {
|
||||
anchor = anchor.max(loc.end.index);
|
||||
if let Some(loc) = item.location()
|
||||
&& let Some(end) = byte_offset(cmd, loc.end.index)
|
||||
{
|
||||
anchor = anchor.max(end);
|
||||
}
|
||||
}
|
||||
let bytes = cmd.as_bytes();
|
||||
@@ -283,6 +292,28 @@ fn try_strip_2to1(
|
||||
stripped.push("2>&1".to_owned());
|
||||
}
|
||||
|
||||
/// Translate a `brush-parser` source-position index into a byte offset in
|
||||
/// `cmd`. The parser counts positions in Unicode scalars (one increment per
|
||||
/// `char`; see `tokenizer::next_char`), but we slice `cmd` — a `&str` — by
|
||||
/// byte index, so the two diverge as soon as the command contains any
|
||||
/// multi-byte UTF-8 (e.g. a `✓`/`×` literal inside a `grep` pattern). Without
|
||||
/// this conversion the head/tail and `2>&1` strips cut at the wrong place,
|
||||
/// corrupting the command (notably orphaning a closing quote).
|
||||
///
|
||||
/// Returns `None` only when `char_idx` is past the end of the input. The
|
||||
/// end-of-input position (`char_idx == cmd.chars().count()`) maps to
|
||||
/// `cmd.len()`.
|
||||
fn byte_offset(cmd: &str, char_idx: usize) -> Option<usize> {
|
||||
let mut count = 0usize;
|
||||
for (byte, _) in cmd.char_indices() {
|
||||
if count == char_idx {
|
||||
return Some(byte);
|
||||
}
|
||||
count += 1;
|
||||
}
|
||||
(count == char_idx).then_some(cmd.len())
|
||||
}
|
||||
|
||||
fn is_stderr_to_stdout(io: &IoRedirect) -> bool {
|
||||
let IoRedirect::File(Some(2), IoFileRedirectKind::DuplicateOutput, target) = io else {
|
||||
return false;
|
||||
@@ -380,6 +411,45 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strips_with_multibyte_content() {
|
||||
// brush-parser reports char-indexed positions; multi-byte UTF-8 before
|
||||
// the trailing `| tail` must not shift the byte-level cut. Regression
|
||||
// for a corrupted command that orphaned the grep pattern's closing
|
||||
// quote (`… |✓|×-80`) and broke later re-parsing.
|
||||
let cases: &[(&str, &str, &[&str])] = &[
|
||||
// Mirrors the real bug: `2>&1` sits mid-pipeline (grep becomes the
|
||||
// effective tail) so only `| tail -80` is stripped, leaving the
|
||||
// quoted grep pattern intact.
|
||||
(
|
||||
"xcodebuild 2>&1 | grep -E \"a|✓|×|b\" | tail -80",
|
||||
"xcodebuild 2>&1 | grep -E \"a|✓|×|b\"",
|
||||
&["| tail -80"],
|
||||
),
|
||||
("echo ✓ | head -3", "echo ✓", &["| head -3"]),
|
||||
("printf '日本語' | tail -n 5", "printf '日本語'", &["| tail -n 5"]),
|
||||
];
|
||||
for (input, want_cmd, want_stripped) in cases {
|
||||
let (cmd, stripped) = run(input);
|
||||
assert_eq!(cmd, *want_cmd, "input: {input:?}");
|
||||
assert_eq!(stripped, *want_stripped, "input: {input:?}");
|
||||
// The rewrite must remain valid shell (no orphaned quote).
|
||||
let options = ParserOptions::default();
|
||||
let source_info = SourceInfo::default();
|
||||
let mut reader = BufReader::new(cmd.as_bytes());
|
||||
let mut parser = Parser::new(&mut reader, &options, &source_info);
|
||||
assert!(parser.parse_program().is_ok(), "rewrite not re-parseable: {cmd:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strips_2to1_after_multibyte() {
|
||||
// Multi-byte before a trailing `2>&1` must not misplace the strip.
|
||||
let (cmd, stripped) = run("echo ✓ × 2>&1");
|
||||
assert_eq!(cmd, "echo ✓ ×");
|
||||
assert_eq!(stripped, vec!["2>&1"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strips_redundant_2to1() {
|
||||
let cases: &[(&str, &str, &[&str])] = &[
|
||||
|
||||
@@ -81,13 +81,13 @@ These are consumed via `getEnvApiKey()` (`packages/ai/src/stream.ts`) unless not
|
||||
| `WAFER_SERVERLESS_API_KEY` | Wafer Serverless auth | Using `wafer-serverless` provider | Pay-as-you-go Wafer SKU; validated against `https://pass.wafer.ai/v1/models` |
|
||||
| `GITLAB_TOKEN` | GitLab Duo auth | Using `gitlab-duo` provider | |
|
||||
|
||||
### GitHub/Copilot token chains
|
||||
### GitHub/Copilot tokens
|
||||
|
||||
| Variable | Used for | Chain |
|
||||
| ---------------------- | ------------------------------------------------ | ---------------------------------------------------- |
|
||||
| `COPILOT_GITHUB_TOKEN` | GitHub Copilot provider auth | `COPILOT_GITHUB_TOKEN` → `GH_TOKEN` → `GITHUB_TOKEN` |
|
||||
| `GH_TOKEN` | Copilot fallback; GitHub API auth in web scraper | In web scraper: `GITHUB_TOKEN` → `GH_TOKEN` |
|
||||
| `GITHUB_TOKEN` | Copilot fallback; GitHub API auth in web scraper | In web scraper: checked before `GH_TOKEN` |
|
||||
| Variable | Used for | Notes |
|
||||
| ---------------------- | ------------------------------------------------ | ------------------------------------------ |
|
||||
| `COPILOT_GITHUB_TOKEN` | GitHub Copilot provider auth | Generic GitHub tokens are not used here |
|
||||
| `GH_TOKEN` | GitHub API auth in web scraper | Web scraper fallback after `GITHUB_TOKEN` |
|
||||
| `GITHUB_TOKEN` | GitHub API auth in web scraper | Web scraper checks this before `GH_TOKEN` |
|
||||
|
||||
### Auth broker / auth gateway (remote credential vault)
|
||||
|
||||
|
||||
@@ -44,4 +44,6 @@ app.stt.toggle: []
|
||||
|
||||
On Windows Terminal, `Ctrl+V` may be handled by the terminal paste command before `omp` sees it; use the `Alt+V` fallback when clipboard image paste appears to do nothing. Windows Terminal also swallows `Ctrl+Enter`, so the follow-up shortcut also binds `Ctrl+Q` — the same chord GitHub Copilot CLI uses. If your existing `keybindings.yml` already assigns `Ctrl+Q` to another action, that user remap wins and follow-up keeps `Ctrl+Enter` unless you explicitly bind `app.message.followUp`.
|
||||
|
||||
Terminals that implement OSC 5522 enhanced paste can send clipboard MIME data directly to `omp`; image pastes are attached as `[Image #N]`, while text/plain paste events keep normal paste behavior. When OSC 5522 is unavailable, bracketed paste still handles text, and a pasted single image-file path is loaded as an image when the file is readable from the `omp` host.
|
||||
|
||||
Older unqualified action names are migrated when `keybindings.yml` is loaded, but new docs and new configs should use the namespaced action IDs above. Existing `keybindings.json` files are still accepted and migrated to `keybindings.yml`; `keybindings.yaml` is also accepted.
|
||||
|
||||
+3
-3
@@ -557,9 +557,9 @@ For `anthropic-messages` models the runtime uses a separate `AnthropicCompat` sh
|
||||
|
||||
### Strict tool schemas (`disableStrictTools`)
|
||||
|
||||
Anthropic's API supports a `strict` field on tool definitions that forces the model to always follow the provided schema exactly. This is enabled by default for all `anthropic-messages` providers because it guarantees schema conformance in agentic systems.
|
||||
Anthropic's API supports a `strict` field on tool definitions that forces the model to always follow the provided schema exactly. OMP enables it by default for a small allowlist of high-frequency built-in `anthropic-messages` tools (`bash`, `python`, `edit`, and `find`) whose schemas fit Anthropic's strict grammar limits; other tools still send normalized schemas but omit `strict`.
|
||||
|
||||
Third-party providers that front the Anthropic API (AWS Bedrock, Azure, self-hosted proxies) do not always implement this field and will reject requests that include it. Set `disableStrictTools: true` at the provider level to opt out:
|
||||
Third-party providers that front the Anthropic API (AWS Bedrock, Azure, self-hosted proxies) do not always implement this field and will reject requests that include it. Set `disableStrictTools: true` at the provider level to opt out of strict mode for the allowlisted tools:
|
||||
|
||||
```yaml
|
||||
providers:
|
||||
@@ -581,7 +581,7 @@ providers:
|
||||
cacheWrite: 3.75
|
||||
```
|
||||
|
||||
`disableStrictTools` is a provider-level flag that applies to all models in the provider.
|
||||
`disableStrictTools` is a provider-level flag that applies to all models in the provider. It disables the Anthropic `strict` marker only for tools that OMP would otherwise mark strict; it does not change runtime tool argument validation. OMP can automatically retry without strict tools after Anthropic reports a strict-grammar-too-large error before the first streamed token, but proxies that reject the `strict` field for other reasons should set this flag explicitly.
|
||||
|
||||
Tool schemas going on the wire are normalized by the unified flow in
|
||||
`packages/ai/src/utils/schema/normalize.ts` (Google/CCA/MCP dispatchers
|
||||
|
||||
@@ -47,6 +47,9 @@ Upstream uses different package scopes. Replace them consistently.
|
||||
- `@mariozechner/pi-agent-core` → `@oh-my-pi/pi-agent-core`
|
||||
- `@mariozechner/pi-tui` → `@oh-my-pi/pi-tui`
|
||||
- `@mariozechner/pi-ai` → `@oh-my-pi/pi-ai`
|
||||
- `@mariozechner/pi-utils` → `@oh-my-pi/pi-utils`
|
||||
- Some upstream packages publish under the `@earendil-works/*` scope instead of `@mariozechner/*`. Map it the same way (`@earendil-works/pi-coding-agent` → `@oh-my-pi/pi-coding-agent`, and so on).
|
||||
- The bare `typebox` package is not an `@oh-my-pi/*` scope; do not rewrite it as one. See the Extensions divergence in section 15 for how tool-parameter schemas map.
|
||||
|
||||
## 4) Use Bun APIs where they improve on Node
|
||||
|
||||
@@ -353,10 +356,13 @@ Our fork has architectural decisions that differ from upstream. **Do not port th
|
||||
|
||||
### Extensions
|
||||
|
||||
| Upstream | Our Fork |
|
||||
| ----------------------------- | ------------------------------------------------- |
|
||||
| `jiti` for TypeScript loading | Native Bun `import()` |
|
||||
| `pkg.pi` manifest field | `pkg.omp` preferred; fallback to `pkg.pi` remains |
|
||||
| Upstream | Our Fork |
|
||||
| ---------------------------------------------------------------- | ------------------------------------------------------------------------------------------------- |
|
||||
| `jiti` for TypeScript loading | Native Bun `import()` |
|
||||
| `pkg.pi` manifest field | `pkg.omp` preferred; fallback to `pkg.pi` remains |
|
||||
| `StringEnum` from `pi-ai` | `Type.Enum` from the `pi.typebox` shim (or author the schema with `pi.zod`); `pi-ai` no longer exports `StringEnum` |
|
||||
| `formatSize` from `pi-coding-agent` | `formatBytes` from `@oh-my-pi/pi-utils` |
|
||||
| `DefaultResourceLoader` / `DefaultPackageManager` / `SettingsManager` / `createEventBus` | Capability-based discovery (`loadCapability(...)`) plus the `Settings` singleton and `EventBus` |
|
||||
|
||||
### Skip These Upstream Features
|
||||
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# Session Operations: export, dump, share, fork, resume/continue
|
||||
# Session Operations: export, dump, share, fresh, fork, resume/continue
|
||||
|
||||
This document describes operator-visible behavior for session export/share/fork/resume operations as currently implemented.
|
||||
|
||||
@@ -19,12 +19,13 @@ This document describes operator-visible behavior for session export/share/fork/
|
||||
| `/export [path]` | Interactive slash command | No | No | HTML file |
|
||||
| `--export <session.jsonl> [outputPath]` | CLI startup fast-path | No runtime session mutation | No active session; reads target file | HTML file |
|
||||
| `/share` | Interactive slash command | No | No | Temp HTML + share URL/gist |
|
||||
| `/fresh` | Interactive slash command | Yes (provider-facing in-memory id/state only) | No; keeps current session file/header | None |
|
||||
| `/fork` | Interactive slash command | Yes (active session identity changes) | Creates new session file and switches current session to it (persistent mode only) | Copies artifact directory to new session namespace when present |
|
||||
| `--fork <id\|path>` | CLI startup | Yes after session creation | Creates a new session fork from the selected source into current cwd/session dir | None |
|
||||
| `/resume` | Interactive slash command | Yes (active in-memory state replaced) | Switches to selected existing session file | None |
|
||||
| `--resume` | CLI startup picker | Yes after session creation | Opens selected existing session file | None |
|
||||
| `--resume <id\|path>` | CLI startup | Yes after session creation | Opens existing session; global cross-project match can fork into current project | None |
|
||||
| `--continue` | CLI startup | Yes after session creation | Opens terminal breadcrumb or most-recent session; creates new one if none exists | None |
|
||||
| `--resume <id\|path>` | CLI startup | Yes after session creation | Opens existing session; global cross-project match re-roots (moved dir) or forks into current project | None |
|
||||
| `--continue` | CLI startup | Yes after session creation | Opens terminal breadcrumb (re-roots it if its dir was moved) or most-recent session; creates new one if none exists | None |
|
||||
|
||||
## Export and dump
|
||||
|
||||
@@ -214,19 +215,23 @@ Notes:
|
||||
|
||||
Cross-project id match behavior:
|
||||
|
||||
- If matched session cwd differs from current cwd, CLI asks:
|
||||
- `Session found in different project ... Fork into current directory? [y/N]`
|
||||
- On yes: `SessionManager.forkFrom(match.path, cwd, sessionDir)` creates a new local forked file.
|
||||
- On no/non-TTY default: command errors.
|
||||
- If matched session cwd differs from current cwd, behavior depends on whether the matched session's recorded directory still exists:
|
||||
- **Directory gone (moved/renamed, e.g. `git worktree move`)**: CLI asks `Session's directory no longer exists (...). Move (re-root) it into the current directory? [Y/n]`.
|
||||
- On yes (default): `SessionManager.open(match.path)` then `manager.moveTo(cwd)` re-roots the existing session into the current directory (no duplicate file).
|
||||
- On no: command cancels (returns no session). On non-TTY: command errors.
|
||||
- **Directory still exists (genuinely different project)**: CLI asks `Session found in different project ... Fork into current directory? [y/N]`.
|
||||
- On yes: `SessionManager.forkFrom(match.path, cwd, sessionDir)` creates a new local forked file.
|
||||
- On no: command cancels. On non-TTY: command errors.
|
||||
|
||||
## CLI `--continue`
|
||||
|
||||
`SessionManager.continueRecent(cwd, sessionDir)`:
|
||||
|
||||
1. Resolves session dir for current cwd.
|
||||
2. Reads terminal-scoped breadcrumb first.
|
||||
3. Falls back to most recently modified session file.
|
||||
4. Opens found session; if none exists, creates new session.
|
||||
2. Reads the terminal-scoped breadcrumb.
|
||||
3. If the breadcrumb points at a session recorded under a different cwd whose directory no longer exists (moved/renamed) **and** the current directory has no sessions of its own, re-roots that session into the current directory via `moveTo` instead of starting fresh.
|
||||
4. Otherwise, if the breadcrumb's cwd matches the current cwd, uses the breadcrumb session; else falls back to the most recently modified session file.
|
||||
5. Opens the found session; if none exists, creates a new session.
|
||||
|
||||
This is startup-only behavior; there is no interactive `/continue` slash command.
|
||||
|
||||
|
||||
+1
-1
@@ -85,7 +85,7 @@ If omitted, export code derives defaults from resolved theme colors.
|
||||
|
||||
- `symbols.preset` sets a theme-level default symbol set.
|
||||
- `symbols.overrides` can override individual `SymbolKey` values.
|
||||
- `symbols.spinnerFrames` overrides the loading spinner frames. Accepts either a flat `string[]` (applied to both spinner types) or an object `{ "status"?: string[], "activity"?: string[] }` to override each type independently. Any type not specified falls back to the symbol preset's default frames. `status` drives the ~12.5fps spinner used by loaders and tool-execution indicators; `activity` drives the ~60fps spinner used by markdown progress bars and similar high-frequency UI.
|
||||
- `symbols.spinnerFrames` overrides the loading spinner frames. Accepts either a flat `string[]` (applied to both spinner types) or an object `{ "status"?: string[], "activity"?: string[] }` to override each type independently. Any type not specified falls back to the symbol preset's default frames. `status` drives the ~12.5fps spinner used by loaders and tool-execution indicators; `activity` drives the ~30fps spinner used by markdown progress bars and similar high-frequency UI.
|
||||
|
||||
Runtime precedence:
|
||||
|
||||
|
||||
@@ -161,9 +161,9 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec
|
||||
- Output: `sources`, `requestId`.
|
||||
- **Kagi** — `packages/coding-agent/src/web/search/providers/kagi.ts`, `packages/coding-agent/src/web/kagi.ts`
|
||||
- Availability: env or `agent.db` credential for `kagi`.
|
||||
- Querying: GET `https://kagi.com/api/v0/search?q=<query>&limit=<n>` with `Authorization: Bot <key>`.
|
||||
- Querying: POST `https://kagi.com/api/v1/search` with `Authorization: Bearer <key>` and JSON body `{ query, workflow: "search", limit, filters?: { after } }`. `recency` maps to `filters.after` as a UTC `YYYY-MM-DD` string (`day`/`week`/`month`/`year`).
|
||||
- `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..40`, default `10`.
|
||||
- Output: `sources`, `relatedQuestions`, `requestId`.
|
||||
- Output: `sources` (concatenated `data.search` + `data.video` + `data.news` + `data.infobox`, with video/news/infobox results tagged in the title), `relatedQuestions` (`data.adjacent_question` + `data.related_search` `props.question`), `answer` (`data.direct_answer[0].snippet ?? title`), `requestId` (`meta.trace`).
|
||||
- **Synthetic** — `packages/coding-agent/src/web/search/providers/synthetic.ts`
|
||||
- Availability: env or `agent.db` credential for `synthetic`.
|
||||
- Querying: POST `https://api.synthetic.new/v2/search` with `{ query }`.
|
||||
|
||||
@@ -78,7 +78,7 @@ the bytes written and the state update. All state flows through a single
|
||||
| `sessionReplace` | clear viewport **+ ED3** (outside multiplexers) | caller forced `{ clearScrollback: true }` (switch/branch/reload/resume) |
|
||||
| `historyRebuild` | clear viewport **+ ED3** (outside multiplexers) | geometry change rewrapped history, or a proven-at-tail rebuild |
|
||||
| `overlayRebuild` | rebuild viewport with overlay composite | overlay visibility changed |
|
||||
| `liveRegionPinned` | relative moves + per-line `\x1b[2K` + `\r\n` | foreground streaming on an ED3-risk host, commit-as-you-go |
|
||||
| `liveRegionPinned` | relative moves + per-row rewrite/suffix-clear + `\r\n` | foreground streaming on an ED3-risk host, commit-as-you-go |
|
||||
| `viewportRepaint` | rewrite the visible viewport in place (optional `appendFrom` tail first) | safe non-destructive repaint |
|
||||
| `deferredShrink` | padded viewport repaint, history left dirty | bottom-anchored shrink, viewport unobservable |
|
||||
| `deferredMutation` | **zero bytes**, history left dirty | row-reindexing edit while possibly scrolled |
|
||||
@@ -245,9 +245,16 @@ parameterized over `(env, platform)` so they are unit-testable:
|
||||
VTE, iTerm2, Apple Terminal, GNOME Terminal, Ptyxis, xfce4-terminal), Linux
|
||||
truecolor, **and every other unknown POSIX terminal**. The default is *risky*
|
||||
on purpose.
|
||||
- `shouldEnableSynchronizedOutputByDefault(env, platform, id)` → DEC 2026 on by
|
||||
default only for kitty/ghostty/wezterm/iterm2, off for win32 / SSH /
|
||||
multiplexers / VTE-family. Layered with a runtime DECRQM auto-disable.
|
||||
- `shouldEnableSynchronizedOutputByDefault(env, id)` → DEC 2026 default. Precedence:
|
||||
user opt-out (`PI_NO_SYNC_OUTPUT`/`PI_TUI_SYNC_OUTPUT=0`) → user force-on
|
||||
(`PI_FORCE_SYNC_OUTPUT=1`/`PI_TUI_SYNC_OUTPUT=1`) → `TERM_FEATURES` advertises
|
||||
`Sy` → `WT_SESSION` (WT/WSL) → known direct terminals
|
||||
(kitty/ghostty/wezterm/iterm2/alacritty/vscode; SSH passes through) → off for
|
||||
risky multiplexers and everything else (VTE-family, GNU screen, Apple Terminal,
|
||||
legacy conhost, unknown). Reconciled at runtime by the DECRQM mode-2026 report:
|
||||
a positive report **enables** sync (upgrading default-off muxes like
|
||||
zellij/tmux-master), a negative one disables it; a user override still wins.
|
||||
`synchronizedOutputUserOverride(env)` is the shared opt-out/force resolver.
|
||||
- `detectRectangularSgrSupport(id, env)` → DECCARA fills: **kitty only**
|
||||
(ghostty does not implement the SGR-background extension), off in multiplexers
|
||||
and under `PI_NO_DECCARA`.
|
||||
|
||||
@@ -56,7 +56,7 @@ A forced render (`requestRender(true)`) queues a viewport repaint or explicit se
|
||||
4. Creates a `StdinBuffer` to split partial escape chunks into complete sequences.
|
||||
5. Queries Kitty keyboard protocol support (`CSI ? u`), then enables protocol flags if supported; otherwise enables modifyOtherKeys fallback after a short timeout.
|
||||
6. Queries OSC 11 background color and Mode 2031 appearance notifications for dark/light theme detection.
|
||||
7. Queries OSC 99 notification capabilities and Kitty temp-file graphics support.
|
||||
7. Queries OSC 99 notification capabilities.
|
||||
8. Starts periodic OSC 11 polling only where safe, then probes DEC private modes 2026/2048/2031 via DECRQM.
|
||||
|
||||
`StdinBuffer` behavior:
|
||||
@@ -199,18 +199,6 @@ Escape exits inactive mode by clearing editor text and restoring border color; w
|
||||
2. Stops TUI before suspend.
|
||||
3. Sends `SIGTSTP` to process group.
|
||||
|
||||
### Background mode (`/background` or `/bg`)
|
||||
|
||||
`handleBackgroundCommand()`:
|
||||
|
||||
- Rejects when idle.
|
||||
- Switches tool UI context to non-interactive (`hasUI=false`) so interactive UI tools fail fast.
|
||||
- Stops loaders/status line and unsubscribes foreground event handler.
|
||||
- Subscribes background event handler (primarily waits for `agent_end`).
|
||||
- Stops TUI and sends `SIGTSTP` (POSIX job control path).
|
||||
|
||||
On `agent_end` in background with no queued work, controller sends completion notification and shuts down.
|
||||
|
||||
## Cancellation paths
|
||||
|
||||
Primary cancellation inputs:
|
||||
|
||||
+1
-1
@@ -28,7 +28,7 @@ export interface Component {
|
||||
render(width: number): string[];
|
||||
handleInput?(data: string): void;
|
||||
wantsKeyRelease?: boolean;
|
||||
invalidate(): void;
|
||||
invalidate?(): void;
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
+11
-10
@@ -20,15 +20,15 @@
|
||||
"@huggingface/transformers": "^4.2.0",
|
||||
"@mozilla/readability": "^0.6.0",
|
||||
"@napi-rs/cli": "3.7.0",
|
||||
"@oh-my-pi/hashline": "15.9.67",
|
||||
"@oh-my-pi/omp-stats": "15.9.67",
|
||||
"@oh-my-pi/pi-agent-core": "15.9.67",
|
||||
"@oh-my-pi/pi-ai": "15.9.67",
|
||||
"@oh-my-pi/pi-coding-agent": "15.9.67",
|
||||
"@oh-my-pi/pi-mnemopi": "15.9.67",
|
||||
"@oh-my-pi/pi-natives": "15.9.67",
|
||||
"@oh-my-pi/pi-tui": "15.9.67",
|
||||
"@oh-my-pi/pi-utils": "15.9.67",
|
||||
"@oh-my-pi/hashline": "15.10.1",
|
||||
"@oh-my-pi/omp-stats": "15.10.1",
|
||||
"@oh-my-pi/pi-agent-core": "15.10.1",
|
||||
"@oh-my-pi/pi-ai": "15.10.1",
|
||||
"@oh-my-pi/pi-coding-agent": "15.10.1",
|
||||
"@oh-my-pi/pi-mnemopi": "15.10.1",
|
||||
"@oh-my-pi/pi-natives": "15.10.1",
|
||||
"@oh-my-pi/pi-tui": "15.10.1",
|
||||
"@oh-my-pi/pi-utils": "15.10.1",
|
||||
"@opentelemetry/api": "^1.9.1",
|
||||
"@opentelemetry/context-async-hooks": "^2.7.1",
|
||||
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
|
||||
@@ -85,8 +85,9 @@
|
||||
},
|
||||
"overrides": {},
|
||||
"scripts": {
|
||||
"install:dev": "bun install && bun --cwd=packages/coding-agent link && bun --cwd=packages/ai link",
|
||||
"install:dev": "bun install && bun --cwd=packages/coding-agent link && ln -sfn \"$(pwd)/packages/coding-agent/scripts/dev-launch\" \"$(bun pm -g bin)/omp\"",
|
||||
"dev": "bun --cwd=packages/coding-agent src/cli.ts",
|
||||
"dev:timing": "PI_TIMING=x bun --cwd=packages/coding-agent --preload ../utils/src/module-timer.ts src/cli.ts",
|
||||
"stats": "bun --cwd=packages/coding-agent src/cli.ts stats",
|
||||
"claude:trace": "bun scripts/claude-trace.ts",
|
||||
"build": "bun run --workspaces --if-present build",
|
||||
|
||||
@@ -2,6 +2,24 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
## [15.10.1] - 2026-06-07
|
||||
|
||||
### Added
|
||||
|
||||
- Added optional `promptCacheKey` support to `AgentOptions` and `Agent` via a new `promptCacheKey` property so providers can receive a caller-provided prompt cache key
|
||||
- Added optional `ApiKeyResolveContext` parameter to `getApiKey` in `AgentOptions` and `AgentLoopConfig` so key resolvers can receive retry context
|
||||
|
||||
### Changed
|
||||
|
||||
- Enabled streaming API calls to re-resolve credentials through the `getApiKey` callback when retries occur after authentication-related errors
|
||||
- `Agent.abort(reason?)` now forwards `reason` to the underlying `AbortController`, and the synthesized aborted assistant message carries that reason on `errorMessage` (string or non-`AbortError` `Error` message) instead of always defaulting to `"Request was aborted"`. Bare `abort()` is unchanged.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed handling of short-lived API keys so that expired tokens are retried with a refreshed value during 401/usage-limit failures
|
||||
- Ensured fallback API key resolution uses the initially configured static `apiKey` when `getApiKey` is present
|
||||
- Wrapped oneshot LLM completions (`instrumentedCompleteSimple`: handoff, compaction/branch summaries) in an `EventLoopKeepalive`. These run outside the agent `#runLoop`, so without the keepalive Bun's event loop stopped servicing timers while parked on the completion promise — freezing host spinners (e.g. the `/handoff` loader) until an unrelated terminal resize poked the loop into rendering again.
|
||||
|
||||
## [15.9.5] - 2026-06-05
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"type": "module",
|
||||
"name": "@oh-my-pi/pi-agent-core",
|
||||
"version": "15.9.67",
|
||||
"version": "15.10.1",
|
||||
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
|
||||
"homepage": "https://omp.sh",
|
||||
"author": "Can Boluk",
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
* Transforms to Message[] only at the LLM call boundary.
|
||||
*/
|
||||
import {
|
||||
type ApiKeyResolveContext,
|
||||
type AssistantMessage,
|
||||
type AssistantMessageEvent,
|
||||
type Context,
|
||||
@@ -727,8 +728,9 @@ async function streamAssistantResponse(
|
||||
// Resolve API key (important for expiring tokens) — do this before resolving
|
||||
// metadata so that the session-sticky credential recorded by getApiKey is
|
||||
// visible to metadataResolver (e.g. for the correct account_uuid in metadata.user_id).
|
||||
const staticApiKey = typeof config.apiKey === "string" ? config.apiKey : undefined;
|
||||
const resolvedApiKey =
|
||||
(config.getApiKey ? await config.getApiKey(config.model.provider) : undefined) || config.apiKey;
|
||||
(config.getApiKey ? await config.getApiKey(config.model.provider) : undefined) || staticApiKey;
|
||||
|
||||
// Re-resolve metadata after credential selection so the per-request value
|
||||
// reflects the credential actually used, not the snapshot from AgentLoopConfig construction.
|
||||
@@ -798,7 +800,19 @@ async function streamAssistantResponse(
|
||||
return await runInActiveSpan(chatSpan, async () => {
|
||||
const response = await streamFunction(config.model, llmContext, {
|
||||
...config,
|
||||
apiKey: resolvedApiKey,
|
||||
// Hand streamSimple a resolver so its central auth-retry policy can
|
||||
// re-resolve on 401 / usage-limit: the initial step reuses the key
|
||||
// already resolved above (which set the session-sticky credential
|
||||
// feeding metadataResolver), and retry steps forward the a/b/c ctx
|
||||
// to config.getApiKey (force-refresh, then rotate). With no
|
||||
// getApiKey hook the caller's own apiKey (string or resolver) flows
|
||||
// through unchanged.
|
||||
apiKey: config.getApiKey
|
||||
? (ctx: ApiKeyResolveContext) =>
|
||||
ctx.error === undefined
|
||||
? resolvedApiKey
|
||||
: Promise.resolve(config.getApiKey!(config.model.provider, ctx))
|
||||
: config.apiKey,
|
||||
metadata: resolvedMetadata,
|
||||
toolChoice: effectiveToolChoice,
|
||||
reasoning: effectiveReasoning,
|
||||
@@ -839,7 +853,14 @@ async function streamAssistantResponse(
|
||||
let detachAbortListener: (() => void) | undefined;
|
||||
if (requestSignal) {
|
||||
if (requestSignal.aborted) {
|
||||
const aborted = emitAbortedAssistantMessage(partialMessage, addedPartial, context, config, stream);
|
||||
const aborted = emitAbortedAssistantMessage(
|
||||
partialMessage,
|
||||
addedPartial,
|
||||
context,
|
||||
config,
|
||||
stream,
|
||||
requestSignal,
|
||||
);
|
||||
await finishChat(aborted);
|
||||
return aborted;
|
||||
}
|
||||
@@ -861,7 +882,14 @@ async function streamAssistantResponse(
|
||||
if (capped) return capped;
|
||||
}
|
||||
responseIterator.return?.()?.catch(() => {});
|
||||
const aborted = emitAbortedAssistantMessage(partialMessage, addedPartial, context, config, stream);
|
||||
const aborted = emitAbortedAssistantMessage(
|
||||
partialMessage,
|
||||
addedPartial,
|
||||
context,
|
||||
config,
|
||||
stream,
|
||||
requestSignal,
|
||||
);
|
||||
await finishChat(aborted);
|
||||
return aborted;
|
||||
}
|
||||
@@ -874,7 +902,14 @@ async function streamAssistantResponse(
|
||||
const capped = await finishCappedAssistantMessage();
|
||||
if (capped) return capped;
|
||||
}
|
||||
const aborted = emitAbortedAssistantMessage(partialMessage, addedPartial, context, config, stream);
|
||||
const aborted = emitAbortedAssistantMessage(
|
||||
partialMessage,
|
||||
addedPartial,
|
||||
context,
|
||||
config,
|
||||
stream,
|
||||
requestSignal,
|
||||
);
|
||||
await finishChat(aborted);
|
||||
return aborted;
|
||||
}
|
||||
@@ -982,14 +1017,30 @@ async function streamAssistantResponse(
|
||||
}
|
||||
}
|
||||
|
||||
/** Resolve the human-readable reason an abort carried. A caller that aborts via
|
||||
* `AbortController.abort(reason)` with a string or a non-`AbortError` `Error`
|
||||
* (e.g. the coding agent's user-interrupt label) gets that text surfaced on the
|
||||
* synthesized assistant message's `errorMessage`; a bare `abort()` (whose
|
||||
* `signal.reason` is the default `AbortError` `DOMException`) falls back to the
|
||||
* generic sentinel that downstream renderers treat as "no specific reason". */
|
||||
export function abortReasonText(signal: AbortSignal | undefined): string {
|
||||
const reason = signal?.reason;
|
||||
if (typeof reason === "string" && reason.trim().length > 0) return reason;
|
||||
if (reason instanceof Error && reason.name !== "AbortError" && reason.message.trim().length > 0) {
|
||||
return reason.message;
|
||||
}
|
||||
return "Request was aborted";
|
||||
}
|
||||
|
||||
function emitAbortedAssistantMessage(
|
||||
partialMessage: AssistantMessage | null,
|
||||
addedPartial: boolean,
|
||||
context: AgentContext,
|
||||
config: AgentLoopConfig,
|
||||
stream: EventStream<AgentEvent, AgentMessage[]>,
|
||||
requestSignal: AbortSignal | undefined,
|
||||
): AssistantMessage {
|
||||
const errorMessage = "Request was aborted";
|
||||
const errorMessage = abortReasonText(requestSignal);
|
||||
const abortedMessage: AssistantMessage = partialMessage
|
||||
? { ...partialMessage, stopReason: "aborted", errorMessage }
|
||||
: {
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
*/
|
||||
import { isPromise } from "node:util/types";
|
||||
import {
|
||||
type ApiKeyResolveContext,
|
||||
type AssistantMessage,
|
||||
type AssistantMessageEvent,
|
||||
type CursorExecHandlers,
|
||||
@@ -21,7 +22,7 @@ import {
|
||||
type ToolChoice,
|
||||
type ToolResultMessage,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
import { agentLoop, agentLoopContinue } from "./agent-loop";
|
||||
import { abortReasonText, agentLoop, agentLoopContinue } from "./agent-loop";
|
||||
import type { AppendOnlyContextManager } from "./append-only-context";
|
||||
import type { HarmonyAuditEvent } from "./harmony-leak";
|
||||
import type {
|
||||
@@ -132,6 +133,11 @@ export interface AgentOptions {
|
||||
* Used by providers that support session-based caching (e.g., OpenAI Codex).
|
||||
*/
|
||||
sessionId?: string;
|
||||
/**
|
||||
* Optional prompt cache key forwarded to LLM providers.
|
||||
* When omitted, providers may fall back to sessionId.
|
||||
*/
|
||||
promptCacheKey?: string;
|
||||
/**
|
||||
* Shared provider state map for session-scoped transport/session caches.
|
||||
*/
|
||||
@@ -141,7 +147,7 @@ export interface AgentOptions {
|
||||
* Resolves an API key dynamically for each LLM call.
|
||||
* Useful for expiring tokens (e.g., GitHub Copilot OAuth).
|
||||
*/
|
||||
getApiKey?: (provider: string) => Promise<string | undefined> | string | undefined;
|
||||
getApiKey?: (provider: string, ctx?: ApiKeyResolveContext) => Promise<string | undefined> | string | undefined;
|
||||
|
||||
/**
|
||||
* Inspect or replace provider payloads before they are sent.
|
||||
@@ -283,6 +289,7 @@ export class Agent {
|
||||
#interruptMode: "immediate" | "wait";
|
||||
#maxToolCallsPerTurn?: number;
|
||||
#sessionId?: string;
|
||||
#promptCacheKey?: string;
|
||||
#metadata?: Record<string, unknown>;
|
||||
#metadataResolver?: (provider: string) => Record<string, unknown> | undefined;
|
||||
#providerSessionState?: Map<string, ProviderSessionState>;
|
||||
@@ -319,7 +326,7 @@ export class Agent {
|
||||
#cursorToolResultBuffer: CursorToolResultEntry[] = [];
|
||||
|
||||
streamFn: StreamFn;
|
||||
getApiKey?: (provider: string) => Promise<string | undefined> | string | undefined;
|
||||
getApiKey?: (provider: string, ctx?: ApiKeyResolveContext) => Promise<string | undefined> | string | undefined;
|
||||
/**
|
||||
* Hook invoked after tool arguments are validated and before execution.
|
||||
* Reassign at any time to swap the implementation (e.g. on extension reload).
|
||||
@@ -344,6 +351,7 @@ export class Agent {
|
||||
this.#maxToolCallsPerTurn = opts.maxToolCallsPerTurn;
|
||||
this.streamFn = opts.streamFn || streamSimple;
|
||||
this.#sessionId = opts.sessionId;
|
||||
this.#promptCacheKey = opts.promptCacheKey;
|
||||
this.#providerSessionState = opts.providerSessionState;
|
||||
this.#thinkingBudgets = opts.thinkingBudgets;
|
||||
this.#temperature = opts.temperature;
|
||||
@@ -390,6 +398,20 @@ export class Agent {
|
||||
this.#sessionId = value;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get the prompt cache key forwarded to providers.
|
||||
*/
|
||||
get promptCacheKey(): string | undefined {
|
||||
return this.#promptCacheKey;
|
||||
}
|
||||
|
||||
/**
|
||||
* Set the prompt cache key forwarded to providers.
|
||||
*/
|
||||
set promptCacheKey(value: string | undefined) {
|
||||
this.#promptCacheKey = value;
|
||||
}
|
||||
|
||||
/**
|
||||
* Static metadata forwarded to every API request when no resolver is installed
|
||||
* (e.g. `metadata.user_id` for Anthropic session attribution). Setting this
|
||||
@@ -768,8 +790,8 @@ export class Agent {
|
||||
this.#state.messages.length = 0;
|
||||
}
|
||||
|
||||
abort() {
|
||||
this.#abortController?.abort();
|
||||
abort(reason?: unknown) {
|
||||
this.#abortController?.abort(reason);
|
||||
}
|
||||
|
||||
waitForIdle(): Promise<void> {
|
||||
@@ -936,6 +958,7 @@ export class Agent {
|
||||
interruptMode: this.#interruptMode,
|
||||
maxToolCallsPerTurn: this.#maxToolCallsPerTurn,
|
||||
sessionId: this.#sessionId,
|
||||
promptCacheKey: this.#promptCacheKey,
|
||||
metadata: this.#metadataResolver ? undefined : this.#metadata,
|
||||
metadataResolver: this.#metadataResolver,
|
||||
providerSessionState: this.#providerSessionState,
|
||||
@@ -1053,8 +1076,12 @@ export class Agent {
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
const errorMessage = err instanceof Error ? err.message : String(err);
|
||||
const stoppedForAbort = this.#abortController?.signal.aborted === true;
|
||||
const errorMessage = stoppedForAbort
|
||||
? abortReasonText(this.#abortController?.signal)
|
||||
: err instanceof Error
|
||||
? err.message
|
||||
: String(err);
|
||||
const shouldEmitVisibleOutputBlockedError = !stoppedForAbort && isAnthropicOutputBlockedError(errorMessage);
|
||||
const assistantPartial = partial?.role === "assistant" ? partial : undefined;
|
||||
const hadAssistantStart = assistantPartial !== undefined;
|
||||
|
||||
@@ -50,6 +50,7 @@ import {
|
||||
} from "@opentelemetry/api";
|
||||
import { AgentRunCollector, type AgentRunCoverage, type AgentRunSummary, type ToolStatus } from "./run-collector";
|
||||
import type { AgentTool } from "./types";
|
||||
import { EventLoopKeepalive } from "./utils/yield";
|
||||
|
||||
/** Default tracer name. Override via {@link AgentTelemetryConfig.tracerName}. */
|
||||
export const DEFAULT_TRACER_NAME = "@oh-my-pi/pi-agent-core";
|
||||
@@ -1629,6 +1630,13 @@ export async function instrumentedCompleteSimple<TApi extends Api>(
|
||||
options: SimpleStreamOptions,
|
||||
span: InstrumentedChatSpanOptions,
|
||||
): Promise<AssistantMessage> {
|
||||
// Oneshot LLM calls (handoff, compaction/branch summaries) run outside the
|
||||
// agent `#runLoop`, which is where the EventLoopKeepalive normally lives.
|
||||
// Without it, Bun's JSC loop stops servicing timers while parked on the
|
||||
// long-lived completion promise, freezing any host spinner (e.g. the
|
||||
// `/handoff` Loader) until an unrelated I/O event (a terminal resize)
|
||||
// pokes the loop. Keep the loop healthy for the duration of the call.
|
||||
using _keepalive = new EventLoopKeepalive();
|
||||
const { telemetry, parent, oneshotKind } = span;
|
||||
const stepNumber = span.stepNumber ?? -1;
|
||||
const reasoning = options.reasoning;
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import type {
|
||||
ApiKeyResolveContext,
|
||||
AssistantMessage,
|
||||
AssistantMessageEvent,
|
||||
AssistantMessageEventStream,
|
||||
@@ -112,7 +113,7 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
|
||||
* Useful for short-lived OAuth tokens (e.g., GitHub Copilot) that may expire
|
||||
* during long-running tool execution phases.
|
||||
*/
|
||||
getApiKey?: (provider: string) => Promise<string | undefined> | string | undefined;
|
||||
getApiKey?: (provider: string, ctx?: ApiKeyResolveContext) => Promise<string | undefined> | string | undefined;
|
||||
|
||||
/**
|
||||
* Returns steering messages to inject into the conversation mid-run.
|
||||
|
||||
@@ -157,6 +157,35 @@ describe("agentLoop with AgentMessage", () => {
|
||||
expect(events.map(event => event.type)).toContain("agent_end");
|
||||
});
|
||||
|
||||
it("surfaces a custom abort reason on the synthesized aborted message", async () => {
|
||||
const context: AgentContext = {
|
||||
systemPrompt: ["You are helpful."],
|
||||
messages: [],
|
||||
tools: [],
|
||||
};
|
||||
const mock = createMockModel();
|
||||
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter, maxToolCallsPerTurn: 8 };
|
||||
const controller = new AbortController();
|
||||
const streamFn = () => new AssistantMessageEventStream();
|
||||
|
||||
const stream = agentLoop([createUserMessage("Hello")], context, config, controller.signal, streamFn);
|
||||
// Abort with a reason (as the coding agent does for a user Esc interrupt).
|
||||
queueMicrotask(() => controller.abort("Interrupted by user"));
|
||||
|
||||
for await (const _event of stream) {
|
||||
// drain
|
||||
}
|
||||
|
||||
const messages = await stream.result();
|
||||
const finalMessage = messages[messages.length - 1];
|
||||
expect(finalMessage.role).toBe("assistant");
|
||||
if (finalMessage.role !== "assistant") throw new Error("Expected assistant message");
|
||||
expect(finalMessage.stopReason).toBe("aborted");
|
||||
// The reason rides AbortController.abort(reason) onto the message verbatim,
|
||||
// instead of the generic "Request was aborted" default.
|
||||
expect(finalMessage.errorMessage).toBe("Interrupted by user");
|
||||
});
|
||||
|
||||
it("should handle custom message types via convertToLlm", async () => {
|
||||
// Create a custom message type
|
||||
interface CustomNotification {
|
||||
|
||||
@@ -354,6 +354,21 @@ describe("Agent", () => {
|
||||
expect(reasoningPerCall).toEqual([ThinkingLevel.Low, ThinkingLevel.High]);
|
||||
});
|
||||
|
||||
it("forwards distinct provider session id and prompt cache key to the stream", async () => {
|
||||
const mock = createMockModel({ responses: [{ content: ["ok"] }] });
|
||||
const agent = new Agent({
|
||||
initialState: { model: mock.model, messages: [] },
|
||||
streamFn: mock.stream,
|
||||
sessionId: "provider-lineage",
|
||||
promptCacheKey: "parent-cache",
|
||||
});
|
||||
|
||||
await agent.prompt("run");
|
||||
|
||||
expect(mock.calls[0]?.options?.sessionId).toBe("provider-lineage");
|
||||
expect(mock.calls[0]?.options?.promptCacheKey).toBe("parent-cache");
|
||||
});
|
||||
|
||||
it("returns static metadata via the plain setter", () => {
|
||||
const agent = new Agent();
|
||||
expect(agent.metadata).toBeUndefined();
|
||||
|
||||
@@ -1,7 +1,55 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added support for `impersonated_service_account` Application Default Credentials (ADC) in Vertex AI to enable chained impersonation without failing via 401 `invalid_client`.
|
||||
|
||||
|
||||
## [15.10.1] - 2026-06-07
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
- Removed the `onAuthError` option from stream request options and shifted auth retry handling to resolver-based `apiKey` behavior, requiring callers using custom auth-retry hooks to migrate
|
||||
|
||||
### Added
|
||||
|
||||
- Added `ApiKeyResolver` and `ApiKey` auth helpers, including `isApiKeyResolver`, `isAuthRetryableError`, `resolveApiKeyOnce`, and `withAuth`, and exported them from the package root
|
||||
- Added support for a function-valued `apiKey` in `SimpleStreamOptions` so a single stream request can refresh or rotate credentials during retry
|
||||
- Added `forceRefresh` credential option to `AuthStorage.getApiKey` and `rotateSessionCredential` support for session-level credential rotation after auth failures
|
||||
- Added `AuthStorage.resolver(provider, options)` method that builds an `ApiKeyResolver` implementing the a/b/c auth-retry policy directly on the storage instance
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed gateway and stream auth flows to share the a/b/c retry policy, refreshing the same session credential first and then switching to a sibling credential on repeated auth failures
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed duplicate upstream `tool_call_id` values collapsing distinct tool calls during message transformation, preserving one call/result pairing per emitted tool call before provider replay. ([#2055](https://github.com/can1357/oh-my-pi/issues/2055))
|
||||
- Fixed streaming auth retries to handle `401` and usage-limit errors before replay-unsafe content is emitted, including failures surfaced only via `errorStatus`
|
||||
- Fixed tool argument validation to coerce singleton non-string values into arrays when the schema expects an array, preventing Anthropic-compatible models that emit `todo.ops` as an object from getting stuck in repeated validation-error loops. ([#2026](https://github.com/can1357/oh-my-pi/issues/2026))
|
||||
- Fixed streaming retries to buffer and suppress partial `start` events from failed auth attempts so only clean retried events are delivered
|
||||
- Fixed the HTTP 400 raw-request dumper (`appendRawHttpRequestDumpFor400`) littering the real `~/.omp/logs/http-400-requests` directory during tests. Provider suites exercise the 400 error path with mocked `fetch` responses, which the dumper could not distinguish from genuine failures; it now skips persistence under the Bun test runner (`isBunTestRuntime()`).
|
||||
- Fixed Anthropic Opus requests unnecessarily forcing `tool_choice.disable_parallel_tool_use`, allowing Claude Opus to use the provider's default parallel tool-calling behavior again.
|
||||
- Fixed parallel `function_call` items losing arguments against llama.cpp's OpenAI Responses endpoint (`/v1/responses`), where every call but the last finalized with `{}` and the agent rejected them with `path: Invalid input: expected string, received undefined`. llama.cpp's `to_json_oaicompat_resp` emits `output_item.added` with only `item.call_id` (no `item.id`, no `output_index`) while the matching `function_call_arguments.delta` carries `item_id: "fc_<call_id>"`. `processResponsesStream` now registers function-call and custom-tool-call items under `item.call_id` as a secondary lookup key (alongside `item.id`/`output_index`) so identifier-deviant hosts route deltas and done events to the right block. ([#2015](https://github.com/can1357/oh-my-pi/issues/2015))
|
||||
- Fixed `PI_REQ_DEBUG` response recording truncating the captured body when a streamed response was cancelled mid-flight. The response tee in `wrapResponse` could call `FileRequestDebugResponseLog.close()` from both the `cancel` callback and the resumed `pull` (which observes `done` once the source reader is cancelled); the second caller saw the handle already nulled and returned before the first caller's pending write flushed, so the `.res.log` lost the already-buffered chunk. `close()` now memoizes its flush-and-close promise so every caller awaits the same completion.
|
||||
|
||||
## [15.10.0] - 2026-06-06
|
||||
|
||||
### Added
|
||||
|
||||
- Added a dependency-free `@oh-my-pi/pi-ai/effort` module exporting the `Effort` enum and `THINKING_EFFORTS`, split out of `model-thinking` so hot-path consumers can import the thinking levels without pulling in `model-thinking` and its provider-compat dependency graph. The package barrel still re-exports both names, so existing imports are unaffected.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Antigravity usage provider emitting one bar per model instead of deduplicating by tier — a single account's 15+ model entries now collapse to one bar per tier, matching the shared-quota reality of the upstream API.
|
||||
- Fixed Antigravity usage reports missing `email` and `accountId` in metadata, so the `/usage` display and the deduplicator can associate reports with their credentials.
|
||||
- Fixed usage-report dedup ignoring `projectId` for Google Cloud providers, preventing duplicate credential entries from being recognized as the same account.
|
||||
|
||||
- Fixed Cloud Code Assist (Antigravity / Gemini CLI) rejecting the `github` tool with HTTP 400 when the `pr` parameter schema contained `anyOf: [string, array]`. The CCA mixed-type combiner collapse picked the first non-null type (`string`) but indiscriminately copied type-specific keys from variant branches — `items` from the array variant leaked onto the string-typed result, producing `{type: "string", items: {...}}` which Google's API rejects as invalid. The collapse now filters merged variant fields against the winning type's allowed key set. ([#2002](https://github.com/can1357/oh-my-pi/pull/2002))
|
||||
- Fixed OpenAI Responses-family providers (Codex, OpenAI Responses, Azure Responses) rejecting requests with `400 No tool output found for function call …` after the user branched/navigated the session tree to a node that ends on a tool call (the tool-result child is dropped from the reconstructed history) or after a turn was aborted/crashed between the call streaming and its result persisting. The converters now synthesize a placeholder `function_call_output`/`custom_tool_call_output` immediately after any unpaired `function_call`/`custom_tool_call`, symmetric to the existing orphan-output repair, so the model still sees the call and can recover instead of the whole request 400ing.
|
||||
- Fixed Anthropic-compatible reasoning endpoints losing prior-turn reasoning on continuation requests when they emit unsigned `thinking` blocks. `convertAnthropicMessages` treated unknown endpoints as signature-enforcing and demoted unsigned reasoning to `type: "text"`, which destabilized tool-call argument serialization on the next turn — the upstream symptom behind the `args?.ops?.map is not a function` crash reported against the `todo` tool. Official `api.anthropic.com` keeps the conservative text fallback; non-official `anthropic-messages` reasoning models now replay unsigned reasoning as native `type: "thinking"` ([#2005](https://github.com/can1357/oh-my-pi/issues/2005)).
|
||||
|
||||
## [15.9.67] - 2026-06-06
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"type": "module",
|
||||
"name": "@oh-my-pi/pi-ai",
|
||||
"version": "15.9.67",
|
||||
"version": "15.10.1",
|
||||
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
||||
"homepage": "https://omp.sh",
|
||||
"author": "Can Boluk",
|
||||
|
||||
@@ -32,6 +32,7 @@ import {
|
||||
isFireworksKimiK2ModelId,
|
||||
MODELS_DEV_PROVIDER_DESCRIPTORS,
|
||||
mapModelsDevToModels,
|
||||
stripFireworksDeepSeekThinkingToggle,
|
||||
UNK_CONTEXT_WINDOW,
|
||||
UNK_MAX_TOKENS,
|
||||
} from "../src/provider-models/openai-compat";
|
||||
@@ -243,6 +244,18 @@ function applyFireworksKimiMaxTokensCap(models: readonly Model[]): Model[] {
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Fireworks' DeepSeek V4 endpoint accepts the user's effort through
|
||||
* `reasoning_effort` and rejects the DeepSeek-native binary `thinking` toggle
|
||||
* when both are present. Strip stale reference metadata from generated fallbacks.
|
||||
*/
|
||||
function applyFireworksDeepSeekReasoningShape(models: readonly Model[]): Model[] {
|
||||
return models.map(model => {
|
||||
if (model.provider !== "fireworks" || model.api !== "openai-completions") return model;
|
||||
return stripFireworksDeepSeekThinkingToggle(model, model.id);
|
||||
});
|
||||
}
|
||||
|
||||
const ANTIGRAVITY_ENDPOINT = "https://daily-cloudcode-pa.sandbox.googleapis.com";
|
||||
|
||||
async function getOAuthAccessFromStorage(provider: OAuthProvider): Promise<OAuthAccess | null> {
|
||||
@@ -416,6 +429,7 @@ async function generateModels() {
|
||||
allModels = applyPremiumMultiplierOverrides(allModels);
|
||||
allModels = applyCodexPricingFallback(allModels);
|
||||
allModels = applyFireworksKimiMaxTokensCap(allModels);
|
||||
allModels = applyFireworksDeepSeekReasoningShape(allModels);
|
||||
applyGeneratedModelPolicies(allModels);
|
||||
linkOpenAIPromotionTargets(allModels);
|
||||
|
||||
|
||||
@@ -18,8 +18,9 @@
|
||||
* POST /v1/responses → OpenAI Responses in/out
|
||||
*/
|
||||
import { extractRetryHint, logger } from "@oh-my-pi/pi-utils";
|
||||
import type { ApiKeyResolver } from "../auth-retry";
|
||||
import type { AuthStorage } from "../auth-storage";
|
||||
import { Effort } from "../model-thinking";
|
||||
import { Effort } from "../effort";
|
||||
import * as anthropicMessages from "../providers/anthropic-messages-server";
|
||||
import * as openaiChat from "../providers/openai-chat-server";
|
||||
import * as openaiResponses from "../providers/openai-responses-server";
|
||||
@@ -340,6 +341,60 @@ async function refreshGatewayApiKeyAfterAuthError(
|
||||
return storage.getApiKey(provider, sessionId, { modelId: model.id, signal });
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the {@link ApiKeyResolver} handed to `streamSimple` for a gateway
|
||||
* request. Drives the central a/b/c auth-retry policy server-side:
|
||||
*
|
||||
* - initial resolve → the credential already resolved for this request.
|
||||
* - step (b) `!lastChance` → force-refresh the SAME session-sticky credential
|
||||
* (a peer/broker may have rotated its token out from under our cached copy).
|
||||
* - step (c) `lastChance` → {@link refreshGatewayApiKeyAfterAuthError} switches
|
||||
* to a sibling (usage-limit block vs credential invalidation by error class).
|
||||
*
|
||||
* `lastKey` tracks the most recent bearer so the switch step invalidates the
|
||||
* credential that actually failed.
|
||||
*/
|
||||
function buildGatewayApiKeyResolver(
|
||||
storage: AuthStorage,
|
||||
model: Model<Api>,
|
||||
sessionId: string,
|
||||
initialKey: string,
|
||||
requestSignal: AbortSignal,
|
||||
format: string,
|
||||
peer: string,
|
||||
): ApiKeyResolver {
|
||||
let lastKey = initialKey;
|
||||
return async ({ lastChance, error, signal }) => {
|
||||
const sig = signal ?? requestSignal;
|
||||
if (error === undefined) {
|
||||
lastKey = initialKey;
|
||||
return initialKey;
|
||||
}
|
||||
if (!lastChance) {
|
||||
const refreshed = await storage.getApiKey(model.provider, sessionId, {
|
||||
modelId: model.id,
|
||||
signal: sig,
|
||||
forceRefresh: true,
|
||||
});
|
||||
lastKey = refreshed ?? lastKey;
|
||||
return refreshed;
|
||||
}
|
||||
const next = await refreshGatewayApiKeyAfterAuthError(
|
||||
storage,
|
||||
model,
|
||||
sessionId,
|
||||
model.provider,
|
||||
lastKey,
|
||||
error,
|
||||
sig,
|
||||
format,
|
||||
peer,
|
||||
);
|
||||
lastKey = next ?? lastKey;
|
||||
return next;
|
||||
};
|
||||
}
|
||||
|
||||
function clientClosedResponse(route: { module: FormatModule }): Response {
|
||||
return route.module.formatError(499, "request_aborted", "client closed request");
|
||||
}
|
||||
@@ -447,19 +502,15 @@ async function handleFormatEndpoint(
|
||||
}
|
||||
|
||||
const streamOpts = buildStreamOptions(parsed, model.api, controller.signal);
|
||||
streamOpts.apiKey = apiKey;
|
||||
streamOpts.onAuthError = (provider, oldKey, error) =>
|
||||
refreshGatewayApiKeyAfterAuthError(
|
||||
bootOpts.storage,
|
||||
model,
|
||||
sessionId,
|
||||
provider,
|
||||
oldKey,
|
||||
error,
|
||||
controller.signal,
|
||||
route.label,
|
||||
peer,
|
||||
);
|
||||
streamOpts.apiKey = buildGatewayApiKeyResolver(
|
||||
bootOpts.storage,
|
||||
model,
|
||||
sessionId,
|
||||
apiKey,
|
||||
controller.signal,
|
||||
route.label,
|
||||
peer,
|
||||
);
|
||||
|
||||
logger.info("auth-gateway request", {
|
||||
format: route.label,
|
||||
@@ -604,18 +655,15 @@ async function handlePiNative(bootOpts: AuthGatewayBootOptions, req: Request, pe
|
||||
// only inject server-controlled fields. The codex temperature/topP strip
|
||||
// matches `buildStreamOptions` — Codex rejects them with a 400.
|
||||
const streamOpts: SimpleStreamOptions = { ...parsed.options, apiKey, signal: controller.signal };
|
||||
streamOpts.onAuthError = (provider, oldKey, error) =>
|
||||
refreshGatewayApiKeyAfterAuthError(
|
||||
bootOpts.storage,
|
||||
model,
|
||||
sessionId,
|
||||
provider,
|
||||
oldKey,
|
||||
error,
|
||||
controller.signal,
|
||||
"pi-native",
|
||||
peer,
|
||||
);
|
||||
streamOpts.apiKey = buildGatewayApiKeyResolver(
|
||||
bootOpts.storage,
|
||||
model,
|
||||
sessionId,
|
||||
apiKey,
|
||||
controller.signal,
|
||||
"pi-native",
|
||||
peer,
|
||||
);
|
||||
if (model.api === "openai-codex-responses") {
|
||||
delete streamOpts.temperature;
|
||||
delete streamOpts.topP;
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import type { Effort } from "../model-thinking";
|
||||
import type { Effort } from "../effort";
|
||||
import type {
|
||||
AssistantMessage,
|
||||
AssistantMessageEventStream,
|
||||
|
||||
@@ -0,0 +1,141 @@
|
||||
import { extractHttpStatusFromError } from "@oh-my-pi/pi-utils";
|
||||
import { isUsageLimitError } from "./rate-limit-utils";
|
||||
|
||||
/**
|
||||
* Context passed to an {@link ApiKeyResolver} on each resolution attempt.
|
||||
*
|
||||
* The `error`/`lastChance` pair drives the central a/b/c retry policy shared by
|
||||
* the streaming ({@link streamSimple}) and non-streaming ({@link withAuth})
|
||||
* drivers:
|
||||
* - `error === undefined` → **initial resolve** (no force-refresh; cheap, may
|
||||
* return a locally-cached not-yet-expired token).
|
||||
* - `error !== undefined && !lastChance` → **step (b): refresh the SAME
|
||||
* account** (force a token re-mint / await an in-flight broker refresh).
|
||||
* - `error !== undefined && lastChance` → **step (c): switch account**
|
||||
* (invalidate/usage-limit the current credential and rotate to a sibling).
|
||||
*
|
||||
* The resolver returns the bearer to send, or `undefined` to stop retrying and
|
||||
* surface the last error to the caller.
|
||||
*/
|
||||
export interface ApiKeyResolveContext {
|
||||
/** True on the final retry step — the resolver should rotate to a sibling credential. */
|
||||
lastChance: boolean;
|
||||
/** The auth error that triggered this re-resolution, or `undefined` on the initial resolve. */
|
||||
error: unknown;
|
||||
/** Caller cancel signal, threaded into any credential refresh / rotation work. */
|
||||
signal?: AbortSignal;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolves the API key to send for a request, retried through the a/b/c policy
|
||||
* described on {@link ApiKeyResolveContext}.
|
||||
*/
|
||||
export type ApiKeyResolver = (ctx: ApiKeyResolveContext) => Promise<string | undefined> | string | undefined;
|
||||
|
||||
/** A static bearer string, or a {@link ApiKeyResolver} that mints/rotates one. */
|
||||
export type ApiKey = string | ApiKeyResolver;
|
||||
|
||||
/** Narrows {@link ApiKey} to its resolver form. */
|
||||
export function isApiKeyResolver(key: ApiKey | undefined): key is ApiKeyResolver {
|
||||
return typeof key === "function";
|
||||
}
|
||||
|
||||
/**
|
||||
* Performs the initial resolve of an {@link ApiKey} (`error: undefined`,
|
||||
* `lastChance: false`). Static keys pass through unchanged.
|
||||
*/
|
||||
export async function resolveApiKeyOnce(key: ApiKey | undefined, signal?: AbortSignal): Promise<string | undefined> {
|
||||
if (key === undefined) return undefined;
|
||||
if (isApiKeyResolver(key)) return (await key({ lastChance: false, error: undefined, signal })) || undefined;
|
||||
return key;
|
||||
}
|
||||
|
||||
/**
|
||||
* Classifies whether an error should trigger a credential refresh/rotation
|
||||
* retry: a hard `401`, or a rotatable usage-limit ("usage_limit_reached",
|
||||
* Codex's "you have hit your ChatGPT usage limit", etc.).
|
||||
*/
|
||||
export function isAuthRetryableError(error: unknown): boolean {
|
||||
if (extractHttpStatusFromError(error) === 401) return true;
|
||||
const message = error instanceof Error ? error.message : typeof error === "string" ? error : undefined;
|
||||
if (!message) return false;
|
||||
if (extractHttpStatusFromError({ message }) === 401) return true;
|
||||
return isUsageLimitError(message);
|
||||
}
|
||||
|
||||
/**
|
||||
* The ordered `lastChance` values for the retry steps after the initial
|
||||
* attempt fails: `false` → step (b) refresh-same, `true` → step (c) switch.
|
||||
* Shared by {@link withAuth} and the streaming retry driver so both run the
|
||||
* same policy.
|
||||
*/
|
||||
export const AUTH_RETRY_STEPS: readonly boolean[] = [false, true];
|
||||
|
||||
/** Resolve a single retry step, swallowing resolver failures into `undefined`. */
|
||||
export async function resolveRetryKey(
|
||||
resolver: ApiKeyResolver,
|
||||
lastChance: boolean,
|
||||
error: unknown,
|
||||
signal?: AbortSignal,
|
||||
): Promise<string | undefined> {
|
||||
try {
|
||||
return (await resolver({ lastChance, error, signal })) || undefined;
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Runs an auth-protected operation through the central a/b/c retry policy.
|
||||
*
|
||||
* - A static string key (or any non-resolver) → a single `attempt` with no
|
||||
* retry (identical to the legacy static-key path).
|
||||
* - A resolver → initial `attempt`, then on a retryable auth error up to two
|
||||
* more attempts (refresh-same, then switch). A step is skipped when the
|
||||
* resolver returns the same key it just tried or `undefined`; non-auth errors
|
||||
* propagate immediately.
|
||||
*
|
||||
* Used by non-streaming consumers (image generation, web search, completion
|
||||
* helpers). The streaming driver in `stream.ts` implements the same policy with
|
||||
* its replay-safe buffering machinery.
|
||||
*/
|
||||
export async function withAuth<T>(
|
||||
key: ApiKey | undefined,
|
||||
attempt: (key: string) => Promise<T>,
|
||||
opts?: { isAuthError?: (error: unknown) => boolean; signal?: AbortSignal; missingKeyMessage?: string },
|
||||
): Promise<T> {
|
||||
const isAuthError = opts?.isAuthError ?? isAuthRetryableError;
|
||||
const missingKey = (): Error => new Error(opts?.missingKeyMessage ?? "No API key available");
|
||||
|
||||
if (!isApiKeyResolver(key)) {
|
||||
if (key === undefined) throw missingKey();
|
||||
return attempt(key);
|
||||
}
|
||||
|
||||
const resolver = key;
|
||||
const signal = opts?.signal;
|
||||
let lastKey = await resolveRetryKey(resolver, false, undefined, signal);
|
||||
if (lastKey === undefined) throw missingKey();
|
||||
|
||||
let lastError: unknown;
|
||||
try {
|
||||
return await attempt(lastKey);
|
||||
} catch (error) {
|
||||
if (!isAuthError(error)) throw error;
|
||||
lastError = error;
|
||||
}
|
||||
|
||||
for (let i = 0; i < AUTH_RETRY_STEPS.length; i++) {
|
||||
const nextKey = await resolveRetryKey(resolver, AUTH_RETRY_STEPS[i]!, lastError, signal);
|
||||
if (nextKey === undefined || nextKey === lastKey) continue;
|
||||
lastKey = nextKey;
|
||||
try {
|
||||
return await attempt(nextKey);
|
||||
} catch (error) {
|
||||
if (!isAuthError(error)) throw error;
|
||||
lastError = error;
|
||||
}
|
||||
}
|
||||
|
||||
throw lastError;
|
||||
}
|
||||
@@ -11,6 +11,8 @@ import { Database, type Statement } from "bun:sqlite";
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as path from "node:path";
|
||||
import { getAgentDbPath, logger } from "@oh-my-pi/pi-utils";
|
||||
import type { ApiKeyResolver } from "./auth-retry";
|
||||
import { isUsageLimitError } from "./rate-limit-utils";
|
||||
import { getEnvApiKey } from "./stream";
|
||||
import type { Provider } from "./types";
|
||||
import type {
|
||||
@@ -36,6 +38,8 @@ import { loginOpenAICodexDevice } from "./utils/oauth/openai-codex";
|
||||
import type { OAuthController, OAuthCredentials, OAuthProvider, OAuthProviderId } from "./utils/oauth/types";
|
||||
import { loginXiaomi, loginXiaomiTokenPlan } from "./utils/oauth/xiaomi";
|
||||
|
||||
const USAGE_RANKING_METRIC_EPSILON = 1e-9;
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────────────
|
||||
// Credential Types
|
||||
// ─────────────────────────────────────────────────────────────────────────────
|
||||
@@ -544,6 +548,13 @@ type AuthApiKeyOptions = {
|
||||
* stranding the caller for `timeoutMs * (maxRetries + 1)`.
|
||||
*/
|
||||
signal?: AbortSignal;
|
||||
/**
|
||||
* Force a re-mint of the session-preferred OAuth credential's access token,
|
||||
* bypassing the not-yet-expired short-circuit. Powers step (b) of the
|
||||
* auth-retry policy ("refresh the SAME account") so a locally-cached token
|
||||
* that a peer/broker rotated out from under us is replaced before retrying.
|
||||
*/
|
||||
forceRefresh?: boolean;
|
||||
};
|
||||
type OAuthResolutionResult = { apiKey: string; credential: OAuthCredential };
|
||||
|
||||
@@ -606,6 +617,14 @@ function hasOpenAICodexProPlan(report: UsageReport | null): boolean {
|
||||
return getUsagePlanType(report)?.includes("pro") === true;
|
||||
}
|
||||
|
||||
function compareUsageRankingMetric(left: number, right: number): number {
|
||||
if (left === right) return 0;
|
||||
if (!Number.isFinite(left) || !Number.isFinite(right)) return left < right ? -1 : 1;
|
||||
const delta = left - right;
|
||||
const tolerance = Math.max(USAGE_RANKING_METRIC_EPSILON, Math.max(Math.abs(left), Math.abs(right)) * 0.000001);
|
||||
return Math.abs(delta) <= tolerance ? 0 : delta;
|
||||
}
|
||||
|
||||
function resolveDefaultUsageProvider(provider: Provider): UsageProvider | undefined {
|
||||
return DEFAULT_USAGE_PROVIDER_MAP.get(provider);
|
||||
}
|
||||
@@ -2288,6 +2307,16 @@ export class AuthStorage {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
#getUsageReportScopeProjectId(report: UsageReport): string | undefined {
|
||||
const ids = new Set<string>();
|
||||
for (const limit of report.limits) {
|
||||
const projectId = limit.scope.projectId?.trim();
|
||||
if (projectId) ids.add(projectId);
|
||||
}
|
||||
if (ids.size === 1) return [...ids][0];
|
||||
return undefined;
|
||||
}
|
||||
|
||||
#getUsageReportIdentifiers(report: UsageReport): string[] {
|
||||
const identifiers: string[] = [];
|
||||
const email = this.#getUsageReportMetadataValue(report, "email");
|
||||
@@ -2295,6 +2324,11 @@ export class AuthStorage {
|
||||
if (report.provider === "openai-codex" || report.provider === "anthropic") {
|
||||
return identifiers.map(identifier => `${report.provider}:${identifier.toLowerCase()}`);
|
||||
}
|
||||
const projectId =
|
||||
this.#getUsageReportMetadataValue(report, "projectId") ?? this.#getUsageReportScopeProjectId(report);
|
||||
// Only add project as a fallback when no email is available — two users
|
||||
// with different emails on the same GCP project must not merge.
|
||||
if (projectId && !email) identifiers.push(`project:${projectId}`);
|
||||
const accountId = this.#getUsageReportMetadataValue(report, "accountId");
|
||||
if (accountId) identifiers.push(`account:${accountId}`);
|
||||
const account = this.#getUsageReportMetadataValue(report, "account");
|
||||
@@ -2781,12 +2815,14 @@ export class AuthStorage {
|
||||
return left.planPriority - right.planPriority;
|
||||
}
|
||||
if (left.hasPriorityBoost !== right.hasPriorityBoost) return left.hasPriorityBoost ? -1 : 1;
|
||||
if (left.secondaryDrainRate !== right.secondaryDrainRate) {
|
||||
return left.secondaryDrainRate - right.secondaryDrainRate;
|
||||
}
|
||||
if (left.secondaryUsed !== right.secondaryUsed) return left.secondaryUsed - right.secondaryUsed;
|
||||
if (left.primaryDrainRate !== right.primaryDrainRate) return left.primaryDrainRate - right.primaryDrainRate;
|
||||
if (left.primaryUsed !== right.primaryUsed) return left.primaryUsed - right.primaryUsed;
|
||||
let metric = compareUsageRankingMetric(left.secondaryDrainRate, right.secondaryDrainRate);
|
||||
if (metric !== 0) return metric;
|
||||
metric = compareUsageRankingMetric(left.secondaryUsed, right.secondaryUsed);
|
||||
if (metric !== 0) return metric;
|
||||
metric = compareUsageRankingMetric(left.primaryDrainRate, right.primaryDrainRate);
|
||||
if (metric !== 0) return metric;
|
||||
metric = compareUsageRankingMetric(left.primaryUsed, right.primaryUsed);
|
||||
if (metric !== 0) return metric;
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -3019,19 +3055,38 @@ export class AuthStorage {
|
||||
candidates.unshift(preferred);
|
||||
}
|
||||
}
|
||||
// Step (b) of the auth-retry policy: when `forceRefresh` is set, re-mint
|
||||
// the session-preferred credential (or the first candidate when no
|
||||
// session preference exists yet) even if its cached token still looks
|
||||
// valid — a peer/broker may have rotated it out from under us.
|
||||
const forceRefreshIndex = options?.forceRefresh
|
||||
? (sessionPreferredIndex ?? candidates[0]?.selection.index)
|
||||
: undefined;
|
||||
await Promise.all(
|
||||
candidates.map(async candidate => {
|
||||
if (Date.now() + OAUTH_REFRESH_SKEW_MS < candidate.selection.credential.expires) return;
|
||||
const force = forceRefreshIndex !== undefined && candidate.selection.index === forceRefreshIndex;
|
||||
if (!force && Date.now() + OAUTH_REFRESH_SKEW_MS < candidate.selection.credential.expires) return;
|
||||
const latestCredential = this.#getCredentialsForProvider(provider)[candidate.selection.index];
|
||||
if (latestCredential?.type === "oauth" && Date.now() + OAUTH_REFRESH_SKEW_MS < latestCredential.expires) {
|
||||
if (
|
||||
!force &&
|
||||
latestCredential?.type === "oauth" &&
|
||||
Date.now() + OAUTH_REFRESH_SKEW_MS < latestCredential.expires
|
||||
) {
|
||||
candidate.selection.credential = latestCredential;
|
||||
return;
|
||||
}
|
||||
try {
|
||||
const credentialId = this.#getStoredCredentials(provider)[candidate.selection.index]?.id;
|
||||
// Hand #refreshOAuthCredential a stale clone (expires:0) so its
|
||||
// not-yet-expired short-circuit doesn't suppress the forced
|
||||
// re-mint; an in-flight peer refresh is still awaited via the
|
||||
// per-credential single-flight.
|
||||
const refreshTarget = force
|
||||
? { ...candidate.selection.credential, expires: 0 }
|
||||
: candidate.selection.credential;
|
||||
const refreshedCredentials = await this.#refreshOAuthCredential(
|
||||
provider,
|
||||
candidate.selection.credential,
|
||||
refreshTarget,
|
||||
credentialId,
|
||||
options?.signal,
|
||||
);
|
||||
@@ -3643,6 +3698,90 @@ export class AuthStorage {
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Rotate away from the session's current credential after a retryable auth
|
||||
* error — step (c) of the auth-retry policy. Stateless: looks up the
|
||||
* session-sticky credential (no API-key matching needed), applies the
|
||||
* storage action for the error class, then clears the sticky so the next
|
||||
* {@link AuthStorage.getApiKey} for this session picks a sibling.
|
||||
*
|
||||
* - usage-limit / account-rate-limit error → {@link AuthStorage.markUsageLimitReached}
|
||||
* (temporary block via its own backoff — default plus server usage-report
|
||||
* reset; sticky left intact so the next resolve re-ranks around the block).
|
||||
* - otherwise (hard 401 / auth failure) → mark the credential suspect (or
|
||||
* reload when no broker hook is wired) and block it, then drop the sticky.
|
||||
*
|
||||
* Returns whether another usable credential of the same type remains.
|
||||
*/
|
||||
async rotateSessionCredential(
|
||||
provider: string,
|
||||
sessionId: string | undefined,
|
||||
options?: { error?: unknown; signal?: AbortSignal },
|
||||
): Promise<boolean> {
|
||||
const sessionCredential = this.#getSessionCredential(provider, sessionId);
|
||||
if (!sessionCredential) return false;
|
||||
|
||||
const error = options?.error;
|
||||
const message = error instanceof Error ? error.message : typeof error === "string" ? error : undefined;
|
||||
if (message && isUsageLimitError(message)) {
|
||||
return this.markUsageLimitReached(provider, sessionId, { signal: options?.signal });
|
||||
}
|
||||
|
||||
const providerKey = this.#getProviderTypeKey(provider, sessionCredential.type);
|
||||
// Snapshot sibling availability before mutating so a soft-deleting
|
||||
// suspect hook can't reindex the answer out from under us.
|
||||
const hasSibling = this.#getCredentialsForProvider(provider).some(
|
||||
(credential, index) =>
|
||||
credential.type === sessionCredential.type &&
|
||||
index !== sessionCredential.index &&
|
||||
!this.#isCredentialBlocked(providerKey, index),
|
||||
);
|
||||
const target = this.#getStoredCredentials(provider)[sessionCredential.index];
|
||||
this.#clearSessionCredential(provider, sessionId);
|
||||
this.#markCredentialBlocked(providerKey, sessionCredential.index, Date.now() + AuthStorage.#defaultBackoffMs);
|
||||
|
||||
if (target) {
|
||||
const markSuspect = this.#store.markCredentialSuspect?.bind(this.#store);
|
||||
if (markSuspect) {
|
||||
await markSuspect(target.id, { signal: options?.signal });
|
||||
} else {
|
||||
await this.reload();
|
||||
}
|
||||
const latestRows = this.#store.listAuthCredentials(provider);
|
||||
this.#setStoredCredentials(
|
||||
provider,
|
||||
latestRows.map(row => ({ id: row.id, credential: row.credential })),
|
||||
);
|
||||
}
|
||||
|
||||
return hasSibling;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build an {@link ApiKeyResolver} backed by this storage, implementing the
|
||||
* central a/b/c auth-retry policy:
|
||||
*
|
||||
* - initial (`error: undefined`) → resolve the session credential.
|
||||
* - step (b) `!lastChance` → force-refresh the SAME session-sticky credential.
|
||||
* - step (c) `lastChance` → rotate to a sibling credential, then re-resolve.
|
||||
*
|
||||
* Used by web-search providers and other consumers that hold an AuthStorage
|
||||
* directly (no ModelRegistry in scope).
|
||||
*/
|
||||
resolver(provider: string, options?: { sessionId?: string; baseUrl?: string; modelId?: string }): ApiKeyResolver {
|
||||
const { sessionId, baseUrl, modelId } = options ?? {};
|
||||
return async ({ lastChance, error, signal }) => {
|
||||
if (error === undefined) {
|
||||
return this.getApiKey(provider, sessionId, { baseUrl, modelId, signal });
|
||||
}
|
||||
if (lastChance) {
|
||||
await this.rotateSessionCredential(provider, sessionId, { error, signal });
|
||||
return this.getApiKey(provider, sessionId, { baseUrl, modelId, signal });
|
||||
}
|
||||
return this.getApiKey(provider, sessionId, { baseUrl, modelId, forceRefresh: true, signal });
|
||||
};
|
||||
}
|
||||
|
||||
// ─── Auth Broker integration ────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
@@ -3932,6 +4071,8 @@ function resolveProviderCredentialIdentityKey(provider: string, identifiers: str
|
||||
const accountIdentifier = identifiers.find(identifier => identifier.startsWith("account:"));
|
||||
if (accountIdentifier) return accountIdentifier;
|
||||
if (emailIdentifier) return emailIdentifier;
|
||||
const projectIdentifier = identifiers.find(identifier => identifier.startsWith("project:"));
|
||||
if (projectIdentifier) return projectIdentifier;
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -3967,6 +4108,8 @@ function extractOAuthCredentialIdentifiers(credential: OAuthCredential): string[
|
||||
if (accountId) identifiers.add(`account:${accountId}`);
|
||||
const email = normalizeStoredEmail(credential.email);
|
||||
if (email) identifiers.add(`email:${email}`);
|
||||
const projectId = normalizeStoredAccountId(credential.projectId);
|
||||
if (projectId) identifiers.add(`project:${projectId}`);
|
||||
const accessIdentifiers = extractOAuthTokenIdentifiers(credential.access) ?? [];
|
||||
for (const identifier of accessIdentifiers) {
|
||||
identifiers.add(identifier);
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
/** User-facing thinking levels, ordered least to most intensive. */
|
||||
export const enum Effort {
|
||||
Minimal = "minimal",
|
||||
Low = "low",
|
||||
Medium = "medium",
|
||||
High = "high",
|
||||
XHigh = "xhigh",
|
||||
}
|
||||
|
||||
export const THINKING_EFFORTS: readonly Effort[] = [
|
||||
Effort.Minimal,
|
||||
Effort.Low,
|
||||
Effort.Medium,
|
||||
Effort.High,
|
||||
Effort.XHigh,
|
||||
];
|
||||
@@ -3,7 +3,9 @@ export * from "./api-registry";
|
||||
export * from "./auth-broker";
|
||||
export { type AuthGatewayBootOptions, type ModelResolver, startAuthGateway } from "./auth-gateway/server";
|
||||
export * from "./auth-gateway/types";
|
||||
export * from "./auth-retry";
|
||||
export * from "./auth-storage";
|
||||
export * from "./effort";
|
||||
export * from "./model-cache";
|
||||
export * from "./model-manager";
|
||||
export * from "./model-thinking";
|
||||
|
||||
@@ -1,23 +1,7 @@
|
||||
import { Effort, THINKING_EFFORTS } from "./effort";
|
||||
import { resolveOpenAICompat } from "./providers/openai-completions-compat";
|
||||
import type { Api, Model as ApiModel, ThinkingConfig } from "./types";
|
||||
|
||||
/** User-facing thinking levels, ordered least to most intensive. */
|
||||
export const enum Effort {
|
||||
Minimal = "minimal",
|
||||
Low = "low",
|
||||
Medium = "medium",
|
||||
High = "high",
|
||||
XHigh = "xhigh",
|
||||
}
|
||||
|
||||
export const THINKING_EFFORTS: readonly Effort[] = [
|
||||
Effort.Minimal,
|
||||
Effort.Low,
|
||||
Effort.Medium,
|
||||
Effort.High,
|
||||
Effort.XHigh,
|
||||
];
|
||||
|
||||
const DEFAULT_REASONING_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High];
|
||||
const DEFAULT_REASONING_EFFORTS_WITH_XHIGH: readonly Effort[] = [
|
||||
Effort.Minimal,
|
||||
@@ -379,18 +363,6 @@ export function supportsMidConversationSystemMessages(modelId: string): boolean
|
||||
return parsed.kind === "opus" && semverGte(parsed.version, "4.8");
|
||||
}
|
||||
|
||||
/**
|
||||
* Claude Opus 4.8 must emit at most one tool call per turn: the Anthropic
|
||||
* Messages provider sends `tool_choice.disable_parallel_tool_use = true` for
|
||||
* this model. Scoped to exactly 4.8 — earlier and later Opus versions keep
|
||||
* Anthropic's default parallel tool-calling.
|
||||
*/
|
||||
export function disablesParallelToolUse(modelId: string): boolean {
|
||||
const parsed = parseAnthropicModel(getCanonicalModelId(modelId));
|
||||
if (!parsed) return false;
|
||||
return parsed.kind === "opus" && semverEqual(parsed.version, "4.8");
|
||||
}
|
||||
|
||||
function anthropicModelHasRealXHighEffort<TApi extends Api>(model: ApiModel<TApi>): boolean {
|
||||
if (model.api !== "anthropic-messages") return false;
|
||||
const parsedModel = parseKnownModel(model.id);
|
||||
|
||||
+402
-30
@@ -1932,6 +1932,75 @@
|
||||
"maxLevel": "high"
|
||||
}
|
||||
},
|
||||
"openai.gpt-5.4": {
|
||||
"id": "openai.gpt-5.4",
|
||||
"name": "GPT-5.4",
|
||||
"api": "bedrock-converse-stream",
|
||||
"provider": "amazon-bedrock",
|
||||
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 2.75,
|
||||
"output": 16.5,
|
||||
"cacheRead": 0.275,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 272000,
|
||||
"maxTokens": 128000,
|
||||
"thinking": {
|
||||
"mode": "budget",
|
||||
"minLevel": "low",
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"openai.gpt-5.5": {
|
||||
"id": "openai.gpt-5.5",
|
||||
"name": "GPT-5.5",
|
||||
"api": "bedrock-converse-stream",
|
||||
"provider": "amazon-bedrock",
|
||||
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 5.5,
|
||||
"output": 33,
|
||||
"cacheRead": 0.55,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 272000,
|
||||
"maxTokens": 128000,
|
||||
"thinking": {
|
||||
"mode": "budget",
|
||||
"minLevel": "low",
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"openai.gpt-oss-120b": {
|
||||
"id": "openai.gpt-oss-120b",
|
||||
"name": "gpt-oss-120b",
|
||||
"api": "bedrock-converse-stream",
|
||||
"provider": "amazon-bedrock",
|
||||
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
|
||||
"reasoning": false,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.15,
|
||||
"output": 0.6,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 128000,
|
||||
"maxTokens": 16384
|
||||
},
|
||||
"openai.gpt-oss-120b-1:0": {
|
||||
"id": "openai.gpt-oss-120b-1:0",
|
||||
"name": "gpt-oss-120b",
|
||||
@@ -1951,6 +2020,25 @@
|
||||
"contextWindow": 128000,
|
||||
"maxTokens": 16384
|
||||
},
|
||||
"openai.gpt-oss-20b": {
|
||||
"id": "openai.gpt-oss-20b",
|
||||
"name": "gpt-oss-20b",
|
||||
"api": "bedrock-converse-stream",
|
||||
"provider": "amazon-bedrock",
|
||||
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
|
||||
"reasoning": false,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.07,
|
||||
"output": 0.3,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 128000,
|
||||
"maxTokens": 16384
|
||||
},
|
||||
"openai.gpt-oss-20b-1:0": {
|
||||
"id": "openai.gpt-oss-20b-1:0",
|
||||
"name": "gpt-oss-20b",
|
||||
@@ -5364,11 +5452,6 @@
|
||||
},
|
||||
"maxTokensField": "max_tokens",
|
||||
"supportsToolChoice": false,
|
||||
"extraBody": {
|
||||
"thinking": {
|
||||
"type": "enabled"
|
||||
}
|
||||
},
|
||||
"reasoningContentField": "reasoning_content",
|
||||
"requiresReasoningContentForToolCalls": true,
|
||||
"requiresAssistantContentForToolCalls": true
|
||||
@@ -11197,7 +11280,7 @@
|
||||
},
|
||||
"deepseek/deepseek-r1-distill-llama-70b": {
|
||||
"id": "deepseek/deepseek-r1-distill-llama-70b",
|
||||
"name": "DeepSeek: R1 Distill Llama 70B",
|
||||
"name": "DeepSeek: R1 Distill Llama 70B (retires Jun 11)",
|
||||
"api": "openai-completions",
|
||||
"provider": "kilo",
|
||||
"baseUrl": "https://api.kilo.ai/api/gateway",
|
||||
@@ -12648,7 +12731,7 @@
|
||||
},
|
||||
"meta-llama/llama-3-70b-instruct": {
|
||||
"id": "meta-llama/llama-3-70b-instruct",
|
||||
"name": "Meta: Llama 3 70B Instruct",
|
||||
"name": "Meta: Llama 3 70B Instruct (retires Jun 19)",
|
||||
"api": "openai-completions",
|
||||
"provider": "kilo",
|
||||
"baseUrl": "https://api.kilo.ai/api/gateway",
|
||||
@@ -13187,6 +13270,25 @@
|
||||
},
|
||||
"minimax/minimax-m3": {
|
||||
"id": "minimax/minimax-m3",
|
||||
"name": "MiniMax: MiniMax M3 (new)",
|
||||
"api": "openai-completions",
|
||||
"provider": "kilo",
|
||||
"baseUrl": "https://api.kilo.ai/api/gateway",
|
||||
"reasoning": false,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 222222,
|
||||
"maxTokens": 8888
|
||||
},
|
||||
"minimax/minimax-m3:discounted": {
|
||||
"id": "minimax/minimax-m3:discounted",
|
||||
"name": "MiniMax: MiniMax M3 (50% off through 2026-06-07)",
|
||||
"api": "openai-completions",
|
||||
"provider": "kilo",
|
||||
@@ -14262,6 +14364,63 @@
|
||||
"contextWindow": 222222,
|
||||
"maxTokens": 8888
|
||||
},
|
||||
"nvidia/nemotron-3-ultra-550b-a55b": {
|
||||
"id": "nvidia/nemotron-3-ultra-550b-a55b",
|
||||
"name": "NVIDIA: Nemotron 3 Ultra",
|
||||
"api": "openai-completions",
|
||||
"provider": "kilo",
|
||||
"baseUrl": "https://api.kilo.ai/api/gateway",
|
||||
"reasoning": false,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 222222,
|
||||
"maxTokens": 8888
|
||||
},
|
||||
"nvidia/nemotron-3-ultra-550b-a55b:free": {
|
||||
"id": "nvidia/nemotron-3-ultra-550b-a55b:free",
|
||||
"name": "NVIDIA: Nemotron 3 Ultra (free)",
|
||||
"api": "openai-completions",
|
||||
"provider": "kilo",
|
||||
"baseUrl": "https://api.kilo.ai/api/gateway",
|
||||
"reasoning": false,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 222222,
|
||||
"maxTokens": 8888
|
||||
},
|
||||
"nvidia/nemotron-3.5-content-safety:free": {
|
||||
"id": "nvidia/nemotron-3.5-content-safety:free",
|
||||
"name": "NVIDIA: Nemotron 3.5 Content Safety (free)",
|
||||
"api": "openai-completions",
|
||||
"provider": "kilo",
|
||||
"baseUrl": "https://api.kilo.ai/api/gateway",
|
||||
"reasoning": false,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 222222,
|
||||
"maxTokens": 8888
|
||||
},
|
||||
"nvidia/nemotron-nano-12b-v2-vl": {
|
||||
"id": "nvidia/nemotron-nano-12b-v2-vl",
|
||||
"name": "NVIDIA: Nemotron Nano 12B 2 VL",
|
||||
@@ -14283,7 +14442,7 @@
|
||||
},
|
||||
"nvidia/nemotron-nano-9b-v2": {
|
||||
"id": "nvidia/nemotron-nano-9b-v2",
|
||||
"name": "NVIDIA: Nemotron Nano 9B V2",
|
||||
"name": "NVIDIA: Nemotron Nano 9B V2 (retires Jun 11)",
|
||||
"api": "openai-completions",
|
||||
"provider": "kilo",
|
||||
"baseUrl": "https://api.kilo.ai/api/gateway",
|
||||
@@ -16371,7 +16530,7 @@
|
||||
},
|
||||
"qwen/qwen3-30b-a3b": {
|
||||
"id": "qwen/qwen3-30b-a3b",
|
||||
"name": "Qwen: Qwen3 30B A3B (retires Jun 5)",
|
||||
"name": "Qwen: Qwen3 30B A3B",
|
||||
"api": "openai-completions",
|
||||
"provider": "kilo",
|
||||
"baseUrl": "https://api.kilo.ai/api/gateway",
|
||||
@@ -19993,24 +20152,19 @@
|
||||
"api": "openai-completions",
|
||||
"provider": "mistral",
|
||||
"baseUrl": "https://api.mistral.ai/v1",
|
||||
"reasoning": true,
|
||||
"reasoning": false,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 1.5,
|
||||
"output": 7.5,
|
||||
"input": 0.4,
|
||||
"output": 2,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 262144,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
"maxTokens": 262144
|
||||
},
|
||||
"mistral-nemo": {
|
||||
"id": "mistral-nemo",
|
||||
@@ -25743,6 +25897,82 @@
|
||||
"contextWindow": 222222,
|
||||
"maxTokens": 8888
|
||||
},
|
||||
"linkup-research-high": {
|
||||
"id": "linkup-research-high",
|
||||
"name": "linkup-research-high",
|
||||
"api": "openai-completions",
|
||||
"provider": "nanogpt",
|
||||
"baseUrl": "https://nano-gpt.com/api/v1",
|
||||
"reasoning": false,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 222222,
|
||||
"maxTokens": 8888
|
||||
},
|
||||
"linkup-research-low": {
|
||||
"id": "linkup-research-low",
|
||||
"name": "linkup-research-low",
|
||||
"api": "openai-completions",
|
||||
"provider": "nanogpt",
|
||||
"baseUrl": "https://nano-gpt.com/api/v1",
|
||||
"reasoning": false,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 222222,
|
||||
"maxTokens": 8888
|
||||
},
|
||||
"linkup-research-medium": {
|
||||
"id": "linkup-research-medium",
|
||||
"name": "linkup-research-medium",
|
||||
"api": "openai-completions",
|
||||
"provider": "nanogpt",
|
||||
"baseUrl": "https://nano-gpt.com/api/v1",
|
||||
"reasoning": false,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 222222,
|
||||
"maxTokens": 8888
|
||||
},
|
||||
"linkup-research-xhigh": {
|
||||
"id": "linkup-research-xhigh",
|
||||
"name": "linkup-research-xhigh",
|
||||
"api": "openai-completions",
|
||||
"provider": "nanogpt",
|
||||
"baseUrl": "https://nano-gpt.com/api/v1",
|
||||
"reasoning": false,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 222222,
|
||||
"maxTokens": 8888
|
||||
},
|
||||
"liquid/lfm-2-24b-a2b": {
|
||||
"id": "liquid/lfm-2-24b-a2b",
|
||||
"name": "liquid/lfm-2-24b-a2b",
|
||||
@@ -28343,6 +28573,30 @@
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"nvidia/nemotron-3-ultra-550b-a55b": {
|
||||
"id": "nvidia/nemotron-3-ultra-550b-a55b",
|
||||
"name": "nvidia/nemotron-3-ultra-550b-a55b",
|
||||
"api": "openai-completions",
|
||||
"provider": "nanogpt",
|
||||
"baseUrl": "https://nano-gpt.com/api/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 222222,
|
||||
"maxTokens": 8888,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"nvidia/nvidia-nemotron-nano-9b-v2": {
|
||||
"id": "nvidia/nvidia-nemotron-nano-9b-v2",
|
||||
"name": "nvidia-nemotron-nano-9b-v2",
|
||||
@@ -38214,7 +38468,7 @@
|
||||
"cacheRead": 0.05,
|
||||
"cacheWrite": 0.625
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": 65536,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
@@ -38263,7 +38517,7 @@
|
||||
"cacheRead": 0.04,
|
||||
"cacheWrite": 0.5
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": 65536,
|
||||
"thinking": {
|
||||
"mode": "budget",
|
||||
@@ -39625,6 +39879,30 @@
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"nemotron-3-ultra-free": {
|
||||
"id": "nemotron-3-ultra-free",
|
||||
"name": "Nemotron 3 Ultra Free",
|
||||
"api": "openai-completions",
|
||||
"provider": "opencode-zen",
|
||||
"baseUrl": "https://opencode.ai/zen/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": 128000,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"qwen3.5-plus": {
|
||||
"id": "qwen3.5-plus",
|
||||
"name": "Qwen3.5 Plus",
|
||||
@@ -41767,12 +42045,12 @@
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.12,
|
||||
"output": 0.37,
|
||||
"cacheRead": 0.019999999499999997,
|
||||
"output": 0.36,
|
||||
"cacheRead": 0.09,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 16384,
|
||||
"maxTokens": 8192,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
@@ -42135,7 +42413,7 @@
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.02,
|
||||
"output": 0.049999999999999996,
|
||||
"output": 0.03,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
@@ -42360,7 +42638,7 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 204800,
|
||||
"maxTokens": 131072,
|
||||
"maxTokens": 196608,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
@@ -43295,6 +43573,54 @@
|
||||
"maxLevel": "high"
|
||||
}
|
||||
},
|
||||
"nvidia/nemotron-3-ultra-550b-a55b": {
|
||||
"id": "nvidia/nemotron-3-ultra-550b-a55b",
|
||||
"name": "NVIDIA: Nemotron 3 Ultra",
|
||||
"api": "openai-completions",
|
||||
"provider": "openrouter",
|
||||
"baseUrl": "https://openrouter.ai/api/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.5,
|
||||
"output": 2.5,
|
||||
"cacheRead": 0.15,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": 16384,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "high"
|
||||
}
|
||||
},
|
||||
"nvidia/nemotron-3-ultra-550b-a55b:free": {
|
||||
"id": "nvidia/nemotron-3-ultra-550b-a55b:free",
|
||||
"name": "NVIDIA: Nemotron 3 Ultra (free)",
|
||||
"api": "openai-completions",
|
||||
"provider": "openrouter",
|
||||
"baseUrl": "https://openrouter.ai/api/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": 65536,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "high"
|
||||
}
|
||||
},
|
||||
"nvidia/nemotron-nano-12b-v2-vl:free": {
|
||||
"id": "nvidia/nemotron-nano-12b-v2-vl:free",
|
||||
"name": "NVIDIA: Nemotron Nano 12B 2 VL (free)",
|
||||
@@ -45244,7 +45570,7 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 131072,
|
||||
"maxTokens": 20000,
|
||||
"maxTokens": 16384,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
@@ -46000,13 +46326,13 @@
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.29,
|
||||
"output": 3.1999999999999997,
|
||||
"input": 0.28900000000000003,
|
||||
"output": 2.4,
|
||||
"cacheRead": 0.15,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 262140,
|
||||
"maxTokens": 131072,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
@@ -49634,6 +49960,28 @@
|
||||
"supportsUsageInStreaming": false
|
||||
}
|
||||
},
|
||||
"nvidia-nemotron-3-ultra-550b-a55b": {
|
||||
"id": "nvidia-nemotron-3-ultra-550b-a55b",
|
||||
"name": "nvidia-nemotron-3-ultra-550b-a55b",
|
||||
"api": "openai-completions",
|
||||
"provider": "venice",
|
||||
"baseUrl": "https://api.venice.ai/api/v1",
|
||||
"reasoning": false,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 222222,
|
||||
"maxTokens": 8888,
|
||||
"compat": {
|
||||
"supportsUsageInStreaming": false
|
||||
}
|
||||
},
|
||||
"nvidia-nemotron-cascade-2-30b-a3b": {
|
||||
"id": "nvidia-nemotron-cascade-2-30b-a3b",
|
||||
"name": "nvidia-nemotron-cascade-2-30b-a3b",
|
||||
@@ -52853,6 +53201,30 @@
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"nvidia/nemotron-3-ultra-550b-a55b": {
|
||||
"id": "nvidia/nemotron-3-ultra-550b-a55b",
|
||||
"name": "Nemotron 3 Ultra",
|
||||
"api": "anthropic-messages",
|
||||
"provider": "vercel-ai-gateway",
|
||||
"baseUrl": "https://ai-gateway.vercel.sh",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.6,
|
||||
"output": 2.4,
|
||||
"cacheRead": 0.12,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": 65000,
|
||||
"thinking": {
|
||||
"mode": "budget",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"nvidia/nemotron-nano-12b-v2-vl": {
|
||||
"id": "nvidia/nemotron-nano-12b-v2-vl",
|
||||
"name": "Nvidia Nemotron Nano 12B V2 VL",
|
||||
@@ -57962,7 +58334,7 @@
|
||||
"cost": {
|
||||
"input": 0.5,
|
||||
"output": 1.5,
|
||||
"cacheRead": 0,
|
||||
"cacheRead": 0.05,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 256000,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import { fetchWithRetry } from "@oh-my-pi/pi-utils";
|
||||
import { Effort } from "../effort";
|
||||
import type { ModelManagerOptions } from "../model-manager";
|
||||
import { Effort } from "../model-thinking";
|
||||
import type { ThinkingConfig } from "../types";
|
||||
import { createBundledReferenceMap, createReferenceResolver } from "./bundled-references";
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { Effort } from "../effort";
|
||||
import type { ModelManagerOptions } from "../model-manager";
|
||||
import { Effort } from "../model-thinking";
|
||||
import { getBundledModels } from "../models";
|
||||
import type { Api, Model, Provider, ThinkingConfig } from "../types";
|
||||
import { isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils";
|
||||
@@ -971,6 +971,29 @@ export function clampFireworksKimiMaxTokens(modelId: string, candidate: number):
|
||||
return isFireworksKimiK2ModelId(modelId) ? Math.min(candidate, FIREWORKS_KIMI_MAX_TOKENS) : candidate;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fireworks DeepSeek V4 accepts effort via `reasoning_effort` but rejects the
|
||||
* DeepSeek-native binary `thinking` toggle when both are present.
|
||||
*/
|
||||
export function stripFireworksDeepSeekThinkingToggle(
|
||||
model: Model<"openai-completions">,
|
||||
publicModelId: string,
|
||||
): Model<"openai-completions"> {
|
||||
if (!publicModelId.startsWith("deepseek-v4")) return model;
|
||||
const compat = model.compat;
|
||||
if (!compat?.extraBody || !("thinking" in compat.extraBody)) return model;
|
||||
|
||||
const extraBody = { ...compat.extraBody };
|
||||
delete extraBody.thinking;
|
||||
if (Object.keys(extraBody).length > 0) {
|
||||
return { ...model, compat: { ...compat, extraBody } };
|
||||
}
|
||||
|
||||
const nextCompat = { ...compat };
|
||||
delete nextCompat.extraBody;
|
||||
return { ...model, compat: nextCompat };
|
||||
}
|
||||
|
||||
export interface FireworksModelManagerConfig {
|
||||
apiKey?: string;
|
||||
baseUrl?: string;
|
||||
@@ -1040,7 +1063,10 @@ export function fireworksModelManagerOptions(
|
||||
mapModel: (entry, defaults) => {
|
||||
const publicModelId = toFireworksPublicModelId(defaults.id);
|
||||
const reference = modelsDevReferences.get(publicModelId) ?? bundledReferences(publicModelId);
|
||||
const model = mapWithBundledReference(entry, defaults, reference);
|
||||
const model = stripFireworksDeepSeekThinkingToggle(
|
||||
mapWithBundledReference(entry, defaults, reference),
|
||||
publicModelId,
|
||||
);
|
||||
return {
|
||||
...model,
|
||||
id: publicModelId,
|
||||
|
||||
@@ -0,0 +1,144 @@
|
||||
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
|
||||
import { Buffer } from "node:buffer";
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import type { FetchImpl } from "../../types";
|
||||
import { __resetVertexTokenCache, getVertexAccessToken } from "../google-auth";
|
||||
|
||||
const CLOUD_PLATFORM_SCOPE = "https://www.googleapis.com/auth/cloud-platform";
|
||||
const JWT_BEARER_GRANT = "urn:ietf:params:oauth:grant-type:jwt-bearer";
|
||||
|
||||
/** Generate a real RS256 private key so signJwtRs256 / pemToPkcs8 run for real. */
|
||||
async function generateServiceAccountPem(): Promise<string> {
|
||||
const keyPair = (await globalThis.crypto.subtle.generateKey(
|
||||
{ name: "RSASSA-PKCS1-v1_5", modulusLength: 2048, publicExponent: new Uint8Array([1, 0, 1]), hash: "SHA-256" },
|
||||
true,
|
||||
["sign", "verify"],
|
||||
)) as CryptoKeyPair;
|
||||
const pkcs8 = new Uint8Array(await globalThis.crypto.subtle.exportKey("pkcs8", keyPair.privateKey));
|
||||
const body = (
|
||||
Buffer.from(pkcs8)
|
||||
.toString("base64")
|
||||
.match(/.{1,64}/g) ?? []
|
||||
).join("\n");
|
||||
return `-----BEGIN PRIVATE KEY-----\n${body}\n-----END PRIVATE KEY-----\n`;
|
||||
}
|
||||
|
||||
function urlOf(input: string | URL | Request): string {
|
||||
if (typeof input === "string") return input;
|
||||
if (input instanceof URL) return input.toString();
|
||||
return input.url;
|
||||
}
|
||||
|
||||
describe("getVertexAccessToken impersonated_service_account ADC", () => {
|
||||
let tmpDir: string;
|
||||
let originalGac: string | undefined;
|
||||
|
||||
beforeEach(async () => {
|
||||
__resetVertexTokenCache();
|
||||
originalGac = Bun.env.GOOGLE_APPLICATION_CREDENTIALS;
|
||||
tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-vertex-adc-"));
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
__resetVertexTokenCache();
|
||||
if (originalGac === undefined) delete Bun.env.GOOGLE_APPLICATION_CREDENTIALS;
|
||||
else Bun.env.GOOGLE_APPLICATION_CREDENTIALS = originalGac;
|
||||
await fs.rm(tmpDir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
it("rejects a malformed service_account_impersonation_url before any network call", async () => {
|
||||
const adcPath = path.join(tmpDir, "impersonated-bad-url.json");
|
||||
await Bun.write(
|
||||
adcPath,
|
||||
JSON.stringify({
|
||||
type: "impersonated_service_account",
|
||||
// Missing the trailing ":generateAccessToken" the principal parser requires.
|
||||
service_account_impersonation_url:
|
||||
"https://iamcredentials.googleapis.com/v1/projects/-/serviceAccounts/target@project.iam.gserviceaccount.com",
|
||||
source_credentials: {
|
||||
type: "authorized_user",
|
||||
client_id: "client-id",
|
||||
client_secret: "client-secret",
|
||||
refresh_token: "refresh-token",
|
||||
},
|
||||
}),
|
||||
);
|
||||
Bun.env.GOOGLE_APPLICATION_CREDENTIALS = adcPath;
|
||||
|
||||
const calls: string[] = [];
|
||||
const fetchImpl: FetchImpl = async input => {
|
||||
calls.push(urlOf(input));
|
||||
return new Response("{}");
|
||||
};
|
||||
|
||||
// The principal is parsed before the source exchange, so a bad URL must fail
|
||||
// up front rather than after burning a source-token round trip.
|
||||
await expect(getVertexAccessToken({ fetch: fetchImpl })).rejects.toBeInstanceOf(RangeError);
|
||||
expect(calls).toEqual([]);
|
||||
});
|
||||
|
||||
it("signs an RS256 JWT for a service_account source and reconstructs the IAM URL", async () => {
|
||||
const pem = await generateServiceAccountPem();
|
||||
const adcPath = path.join(tmpDir, "impersonated-sa.json");
|
||||
await Bun.write(
|
||||
adcPath,
|
||||
JSON.stringify({
|
||||
type: "impersonated_service_account",
|
||||
// Non-canonical project segment proves the request URL is rebuilt, not echoed.
|
||||
service_account_impersonation_url:
|
||||
"https://iamcredentials.googleapis.com/v1/projects/explicit-proj/serviceAccounts/target@project.iam.gserviceaccount.com:generateAccessToken",
|
||||
source_credentials: {
|
||||
type: "service_account",
|
||||
client_email: "source@project.iam.gserviceaccount.com",
|
||||
private_key: pem,
|
||||
private_key_id: "key-1",
|
||||
},
|
||||
// delegates intentionally omitted — the IAM body must default to [].
|
||||
}),
|
||||
);
|
||||
Bun.env.GOOGLE_APPLICATION_CREDENTIALS = adcPath;
|
||||
|
||||
const calls: { url: string; init?: RequestInit }[] = [];
|
||||
const fetchImpl: FetchImpl = async (input, init) => {
|
||||
const url = urlOf(input);
|
||||
calls.push({ url, init });
|
||||
if (url === "https://oauth2.googleapis.com/token") {
|
||||
return new Response(JSON.stringify({ access_token: "sa-source-token", expires_in: 3600 }));
|
||||
}
|
||||
if (url.startsWith("https://iamcredentials.googleapis.com/")) {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
accessToken: "impersonated-token",
|
||||
expireTime: new Date(Date.now() + 3_600_000).toISOString(),
|
||||
}),
|
||||
);
|
||||
}
|
||||
return new Response("unexpected", { status: 404 });
|
||||
};
|
||||
|
||||
const token = await getVertexAccessToken({ fetch: fetchImpl });
|
||||
expect(token).toBe("impersonated-token");
|
||||
|
||||
// Source JWT exchange happens first, then the impersonation exchange.
|
||||
expect(calls.map(c => c.url)).toEqual([
|
||||
"https://oauth2.googleapis.com/token",
|
||||
"https://iamcredentials.googleapis.com/v1/projects/-/serviceAccounts/target@project.iam.gserviceaccount.com:generateAccessToken",
|
||||
]);
|
||||
|
||||
// Source credential is exchanged via a signed JWT bearer assertion, not a refresh grant.
|
||||
const sourceBody = new URLSearchParams(String(calls[0].init?.body));
|
||||
expect(sourceBody.get("grant_type")).toBe(JWT_BEARER_GRANT);
|
||||
expect((sourceBody.get("assertion") ?? "").split(".")).toHaveLength(3);
|
||||
|
||||
// The IAM call carries the source-derived bearer token and defaults delegates to [].
|
||||
const iamHeaders = calls[1].init?.headers as Record<string, string>;
|
||||
expect(iamHeaders.Authorization).toBe("Bearer sa-source-token");
|
||||
expect(JSON.parse(String(calls[1].init?.body))).toEqual({
|
||||
delegates: [],
|
||||
scope: [CLOUD_PLATFORM_SCOPE],
|
||||
lifetime: "3600s",
|
||||
});
|
||||
});
|
||||
});
|
||||
@@ -8,7 +8,7 @@
|
||||
*/
|
||||
|
||||
import { $env, $flag, extractHttpStatusFromError, fetchWithRetry } from "@oh-my-pi/pi-utils";
|
||||
import type { Effort } from "../model-thinking";
|
||||
import type { Effort } from "../effort";
|
||||
import { mapEffortToAnthropicAdaptiveEffort, requireSupportedEffort } from "../model-thinking";
|
||||
import { calculateCost } from "../models";
|
||||
import type {
|
||||
|
||||
@@ -13,7 +13,6 @@ import {
|
||||
readSseEvents,
|
||||
} from "@oh-my-pi/pi-utils";
|
||||
import {
|
||||
disablesParallelToolUse,
|
||||
hasOpus47ApiRestrictions,
|
||||
mapEffortToAnthropicAdaptiveEffort,
|
||||
supportsMidConversationSystemMessages,
|
||||
@@ -2371,21 +2370,6 @@ function buildParams(
|
||||
}
|
||||
}
|
||||
|
||||
// Claude Opus 4.8 must emit at most one tool call per turn. Force
|
||||
// `disable_parallel_tool_use` onto the outgoing tool_choice (synthesizing an
|
||||
// `auto` choice when none is set). Gated on tools being present: Anthropic
|
||||
// rejects `tool_choice` without `tools`, and parallelism is moot otherwise.
|
||||
// `none` rejects the field, so leave it untouched. A fresh object is built
|
||||
// rather than mutated so the caller's `options.toolChoice` is never aliased.
|
||||
if (disablesParallelToolUse(model.id) && params.tools && params.tools.length > 0) {
|
||||
const current = params.tool_choice;
|
||||
if (!current) {
|
||||
params.tool_choice = { type: "auto", disable_parallel_tool_use: true };
|
||||
} else if (current.type !== "none") {
|
||||
params.tool_choice = { ...current, disable_parallel_tool_use: true };
|
||||
}
|
||||
}
|
||||
|
||||
const shouldInjectClaudeCodeInstruction = isOAuthToken && !model.id.startsWith("claude-3-5-haiku");
|
||||
const firstUserMessageText = shouldInjectClaudeCodeInstruction
|
||||
? extractClaudeCodeFirstUserMessageText(context.messages)
|
||||
@@ -2427,22 +2411,31 @@ function isZaiAnthropicEndpoint(model: Model<"anthropic-messages">): boolean {
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns true for providers whose Anthropic-compatible endpoints do NOT
|
||||
* implement signature-based thinking-chain integrity (DeepSeek, Z.AI, etc.).
|
||||
* For these providers, unsigned thinking blocks must be preserved as
|
||||
* `type: "thinking"` instead of being degraded to text.
|
||||
* Returns true when unsigned `thinking` blocks from prior assistant turns should
|
||||
* be replayed as Anthropic-native thinking instead of demoted to text.
|
||||
*
|
||||
* Official Anthropic (matched via `isAnthropicApiBaseUrl`, which intentionally
|
||||
* treats a missing baseUrl as official since `resolveAnthropicBaseUrl` routes
|
||||
* it to `https://api.anthropic.com`) enforces signature-based thinking-chain
|
||||
* integrity, so unsigned blocks must remain text there. Anthropic-compatible
|
||||
* reasoning endpoints commonly emit unsigned thinking blocks while still
|
||||
* expecting them back as `type: "thinking"` on continuation; demoting them
|
||||
* loses the model's reasoning chain and can destabilize the next tool-call
|
||||
* arguments (#2005). Known non-signing hosts are also preserved for
|
||||
* compatibility.
|
||||
*/
|
||||
function isNonSigningAnthropicEndpoint(model: Model<"anthropic-messages">): boolean {
|
||||
// Known non-signing providers
|
||||
function shouldReplayUnsignedThinking(model: Model<"anthropic-messages">): boolean {
|
||||
if (model.provider === "zai" || model.provider === "deepseek") return true;
|
||||
const baseUrl = model.baseUrl;
|
||||
if (!baseUrl) return false;
|
||||
try {
|
||||
const hostname = new URL(baseUrl).hostname.toLowerCase();
|
||||
return hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com");
|
||||
} catch {
|
||||
return false;
|
||||
if (baseUrl) {
|
||||
try {
|
||||
const hostname = new URL(baseUrl).hostname.toLowerCase();
|
||||
if (hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com")) return true;
|
||||
} catch {
|
||||
// Fall through to the protocol-level reasoning rule below.
|
||||
}
|
||||
}
|
||||
return model.reasoning && !isAnthropicApiBaseUrl(baseUrl);
|
||||
}
|
||||
|
||||
function buildToolResultBlock(model: Model<"anthropic-messages">, msg: ToolResultMessage): ContentBlockParam {
|
||||
@@ -2533,7 +2526,7 @@ export function convertAnthropicMessages(
|
||||
}
|
||||
if (block.thinking.trim().length === 0) continue;
|
||||
if (!block.thinkingSignature || block.thinkingSignature.trim().length === 0) {
|
||||
if (isNonSigningAnthropicEndpoint(model)) {
|
||||
if (shouldReplayUnsignedThinking(model)) {
|
||||
blocks.push({
|
||||
type: "thinking",
|
||||
thinking: block.thinking.toWellFormed(),
|
||||
|
||||
@@ -40,6 +40,7 @@ import {
|
||||
isOpenAIResponsesProgressEvent,
|
||||
normalizeResponsesToolCallIdForTransform,
|
||||
processResponsesStream,
|
||||
repairOrphanResponsesToolCalls,
|
||||
} from "./openai-responses-shared";
|
||||
import { transformMessages } from "./transform-messages";
|
||||
|
||||
@@ -347,7 +348,7 @@ function convertMessages(
|
||||
msgIndex++;
|
||||
}
|
||||
|
||||
return messages;
|
||||
return repairOrphanResponsesToolCalls(messages);
|
||||
}
|
||||
|
||||
function convertTools(tools: Tool[]): OpenAITool[] {
|
||||
|
||||
@@ -234,7 +234,8 @@ export function streamGitLabDuo(
|
||||
|
||||
(async () => {
|
||||
try {
|
||||
if (!options?.apiKey) {
|
||||
const apiKey = typeof options?.apiKey === "string" ? options.apiKey : undefined;
|
||||
if (!apiKey || !options) {
|
||||
throw new Error("Missing GitLab access token. Run /login gitlab-duo or set GITLAB_TOKEN.");
|
||||
}
|
||||
|
||||
@@ -243,7 +244,7 @@ export function streamGitLabDuo(
|
||||
throw new Error(`Unsupported GitLab Duo model: ${model.id}`);
|
||||
}
|
||||
|
||||
const directAccess = await getDirectAccessToken(options.apiKey, options.fetch);
|
||||
const directAccess = await getDirectAccessToken(apiKey, options.fetch);
|
||||
const headers = {
|
||||
...directAccess.headers,
|
||||
...options.headers,
|
||||
|
||||
@@ -42,7 +42,14 @@ interface AuthorizedUserCredentials {
|
||||
refresh_token: string;
|
||||
}
|
||||
|
||||
type AdcFileCredentials = ServiceAccountCredentials | AuthorizedUserCredentials;
|
||||
interface ImpersonatedServiceAccountCredentials {
|
||||
type: "impersonated_service_account";
|
||||
service_account_impersonation_url: string;
|
||||
source_credentials: AuthorizedUserCredentials | ServiceAccountCredentials;
|
||||
delegates?: string[];
|
||||
}
|
||||
|
||||
type AdcFileCredentials = ServiceAccountCredentials | AuthorizedUserCredentials | ImpersonatedServiceAccountCredentials;
|
||||
|
||||
interface TokenResponse {
|
||||
access_token: string;
|
||||
@@ -196,10 +203,52 @@ async function resolveAccessTokenUncached(
|
||||
): Promise<{ source: string; token: TokenResponse }> {
|
||||
const adc = await loadAdcCredentials();
|
||||
if (adc) {
|
||||
const token =
|
||||
adc.creds.type === "service_account"
|
||||
? await exchangeJwtForToken(adc.creds, signal, fetchImpl)
|
||||
: await exchangeRefreshToken(adc.creds, signal, fetchImpl);
|
||||
const creds = adc.creds;
|
||||
let token: TokenResponse;
|
||||
|
||||
if (creds.type === "impersonated_service_account") {
|
||||
const targetPrincipalMatch = /(?<target>[^/]+):(generateAccessToken|generateIdToken)$/.exec(
|
||||
creds.service_account_impersonation_url,
|
||||
);
|
||||
const targetPrincipal = targetPrincipalMatch?.groups?.target;
|
||||
if (!targetPrincipal) {
|
||||
throw new RangeError(`Cannot extract target principal from ${creds.service_account_impersonation_url}`);
|
||||
}
|
||||
|
||||
const sourceToken =
|
||||
creds.source_credentials.type === "service_account"
|
||||
? await exchangeJwtForToken(creds.source_credentials, signal, fetchImpl)
|
||||
: await exchangeRefreshToken(creds.source_credentials, signal, fetchImpl);
|
||||
|
||||
const response = await fetchImpl(
|
||||
`https://iamcredentials.googleapis.com/v1/projects/-/serviceAccounts/${targetPrincipal}:generateAccessToken`,
|
||||
{
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
Authorization: `Bearer ${sourceToken.access_token}`,
|
||||
},
|
||||
body: JSON.stringify({
|
||||
delegates: creds.delegates ?? [],
|
||||
scope: [CLOUD_PLATFORM_SCOPE],
|
||||
lifetime: "3600s",
|
||||
}),
|
||||
signal,
|
||||
},
|
||||
);
|
||||
if (!response.ok) {
|
||||
const detail = await response.text().catch(() => "");
|
||||
throw new Error(`Google Impersonation token exchange failed (${response.status}): ${detail}`);
|
||||
}
|
||||
const data = (await response.json()) as { accessToken: string; expireTime: string };
|
||||
const expiresIn = Math.max(0, Math.floor((new Date(data.expireTime).getTime() - Date.now()) / 1000));
|
||||
token = { access_token: data.accessToken, expires_in: expiresIn, token_type: "Bearer" };
|
||||
} else {
|
||||
token =
|
||||
creds.type === "service_account"
|
||||
? await exchangeJwtForToken(creds, signal, fetchImpl)
|
||||
: await exchangeRefreshToken(creds, signal, fetchImpl);
|
||||
}
|
||||
return { source: adc.source, token };
|
||||
}
|
||||
const metadata = await fetchMetadataToken(signal, fetchImpl);
|
||||
|
||||
@@ -44,6 +44,9 @@ export function streamOpenAIAnthropicShim(
|
||||
): AssistantMessageEventStream {
|
||||
const stream = new AssistantMessageEventStream();
|
||||
const format = options?.format ?? config.defaultFormat;
|
||||
// The resolver form of `apiKey` is resolved upstream in `streamSimple`;
|
||||
// this shim only ever receives a static bearer string.
|
||||
const apiKey = typeof options?.apiKey === "string" ? options.apiKey : undefined;
|
||||
|
||||
(async () => {
|
||||
try {
|
||||
@@ -74,7 +77,7 @@ export function streamOpenAIAnthropicShim(
|
||||
: undefined;
|
||||
|
||||
const innerStream = streamAnthropic(anthropicModel, context, {
|
||||
apiKey: options?.apiKey,
|
||||
apiKey,
|
||||
temperature: options?.temperature,
|
||||
topP: options?.topP,
|
||||
topK: options?.topK,
|
||||
@@ -103,7 +106,7 @@ export function streamOpenAIAnthropicShim(
|
||||
|
||||
const reasoningEffort = options?.reasoning;
|
||||
const innerStream = streamOpenAICompletions(openaiModel, context, {
|
||||
apiKey: options?.apiKey,
|
||||
apiKey,
|
||||
temperature: options?.temperature,
|
||||
topP: options?.topP,
|
||||
topK: options?.topK,
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import type { Effort } from "../../model-thinking";
|
||||
import type { Effort } from "../../effort";
|
||||
import { requireSupportedEffort } from "../../model-thinking";
|
||||
import type { Api, Model } from "../../types";
|
||||
|
||||
@@ -76,6 +76,83 @@ function filterInput(input: InputItem[] | undefined): InputItem[] | undefined {
|
||||
});
|
||||
}
|
||||
|
||||
const CODEX_ORPHAN_OUTPUT_LIMIT = 16_000;
|
||||
/** Placeholder output for a tool call whose result never landed in the input. */
|
||||
const CODEX_INTERRUPTED_TOOL_OUTPUT =
|
||||
"[No tool output recorded: the tool call was interrupted before it produced a result.]";
|
||||
|
||||
function orphanFunctionOutputToMessage(item: InputItem, callId: string): InputItem {
|
||||
const itemRecord = item as unknown as Record<string, unknown>;
|
||||
const toolName = typeof itemRecord.name === "string" ? itemRecord.name : "tool";
|
||||
let text = "";
|
||||
try {
|
||||
const output = itemRecord.output;
|
||||
text = typeof output === "string" ? output : JSON.stringify(output);
|
||||
} catch {
|
||||
text = String(itemRecord.output ?? "");
|
||||
}
|
||||
if (text.length > CODEX_ORPHAN_OUTPUT_LIMIT) {
|
||||
text = `${text.slice(0, CODEX_ORPHAN_OUTPUT_LIMIT)}\n...[truncated]`;
|
||||
}
|
||||
return {
|
||||
type: "message",
|
||||
role: "assistant",
|
||||
content: `[Previous ${toolName} result; call_id=${callId}]: ${text}`,
|
||||
} as InputItem;
|
||||
}
|
||||
|
||||
/**
|
||||
* Repair both halves of unpaired tool exchanges so the Responses input grammar
|
||||
* stays valid — the API rejects either orphan with a 400:
|
||||
*
|
||||
* - `function_call_output` with no matching `function_call` → folded into an
|
||||
* assistant message (`400 No tool call found for function call output …`).
|
||||
* Regression of #472 / #1351.
|
||||
* - `function_call` / `custom_tool_call` with no matching `*_output` → a
|
||||
* placeholder output is synthesized immediately after the call
|
||||
* (`400 No tool output found for function call …`). Hit when the user
|
||||
* branches/navigates the session tree to a node that ends on a tool call (the
|
||||
* tool-result child is dropped from the reconstructed history) or when a turn
|
||||
* is aborted/crashes after the call streamed but before its result persisted.
|
||||
*/
|
||||
function repairToolCallPairs(input: InputItem[]): InputItem[] {
|
||||
const callIds = new Set<string>();
|
||||
const outputCallIds = new Set<string>();
|
||||
for (const item of input) {
|
||||
const callId = typeof item.call_id === "string" ? item.call_id : undefined;
|
||||
if (callId === undefined) continue;
|
||||
if (item.type === "function_call" || item.type === "custom_tool_call") callIds.add(callId);
|
||||
else if (item.type === "function_call_output" || item.type === "custom_tool_call_output") {
|
||||
outputCallIds.add(callId);
|
||||
}
|
||||
}
|
||||
|
||||
const repaired: InputItem[] = [];
|
||||
for (const item of input) {
|
||||
const callId = typeof item.call_id === "string" ? item.call_id : undefined;
|
||||
|
||||
if (item.type === "function_call_output" && callId !== undefined && !callIds.has(callId)) {
|
||||
repaired.push(orphanFunctionOutputToMessage(item, callId));
|
||||
continue;
|
||||
}
|
||||
|
||||
repaired.push(item);
|
||||
|
||||
if (
|
||||
(item.type === "function_call" || item.type === "custom_tool_call") &&
|
||||
callId !== undefined &&
|
||||
!outputCallIds.has(callId)
|
||||
) {
|
||||
repaired.push({
|
||||
type: item.type === "custom_tool_call" ? "custom_tool_call_output" : "function_call_output",
|
||||
call_id: callId,
|
||||
output: CODEX_INTERRUPTED_TOOL_OUTPUT,
|
||||
} as InputItem);
|
||||
}
|
||||
}
|
||||
return repaired;
|
||||
}
|
||||
|
||||
export async function transformRequestBody(
|
||||
body: RequestBody,
|
||||
model: Model<Api>,
|
||||
@@ -87,39 +164,8 @@ export async function transformRequestBody(
|
||||
|
||||
if (body.input && Array.isArray(body.input)) {
|
||||
body.input = filterInput(body.input);
|
||||
|
||||
if (body.input) {
|
||||
const functionCallIds = new Set(
|
||||
body.input
|
||||
.filter(item => item.type === "function_call" && typeof item.call_id === "string")
|
||||
.map(item => item.call_id as string),
|
||||
);
|
||||
|
||||
body.input = body.input.map(item => {
|
||||
if (item.type === "function_call_output" && typeof item.call_id === "string") {
|
||||
const callId = item.call_id as string;
|
||||
if (!functionCallIds.has(callId)) {
|
||||
const itemRecord = item as unknown as Record<string, unknown>;
|
||||
const toolName = typeof itemRecord.name === "string" ? itemRecord.name : "tool";
|
||||
let text = "";
|
||||
try {
|
||||
const output = itemRecord.output;
|
||||
text = typeof output === "string" ? output : JSON.stringify(output);
|
||||
} catch {
|
||||
text = String(itemRecord.output ?? "");
|
||||
}
|
||||
if (text.length > 16000) {
|
||||
text = `${text.slice(0, 16000)}\n...[truncated]`;
|
||||
}
|
||||
return {
|
||||
type: "message",
|
||||
role: "assistant",
|
||||
content: `[Previous ${toolName} result; call_id=${callId}]: ${text}`,
|
||||
} as InputItem;
|
||||
}
|
||||
}
|
||||
return item;
|
||||
});
|
||||
body.input = repairToolCallPairs(body.input);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -10,7 +10,8 @@ import type {
|
||||
ChatCompletionToolMessageParam,
|
||||
} from "openai/resources/chat/completions";
|
||||
import packageJson from "../../package.json" with { type: "json" };
|
||||
import { type Effort, getSupportedEfforts } from "../model-thinking";
|
||||
import type { Effort } from "../effort";
|
||||
import { getSupportedEfforts } from "../model-thinking";
|
||||
import { calculateCost } from "../models";
|
||||
import { getEnvApiKey } from "../stream";
|
||||
import {
|
||||
@@ -536,7 +537,6 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
||||
}
|
||||
stream.push({ type: "start", partial: output });
|
||||
|
||||
const parseMiniMaxThinkTags = model.provider === "minimax-code" || model.provider === "minimax-code-cn";
|
||||
// Some OpenAI-compatible DeepSeek hosts (including NVIDIA NIM and DeepSeek's
|
||||
// native API) leak chat-template tool-call markers in `delta.content` even
|
||||
// though tool calls are also surfaced structurally. Strip the leaked markers
|
||||
@@ -677,9 +677,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
||||
}
|
||||
};
|
||||
|
||||
const streamMarkupHealingPattern = getStreamMarkupHealingPattern(model.provider, model.id, {
|
||||
parseThinkingTags: parseMiniMaxThinkTags,
|
||||
});
|
||||
const streamMarkupHealingPattern = getStreamMarkupHealingPattern(model.provider, model.id);
|
||||
const streamMarkupHealing = streamMarkupHealingPattern
|
||||
? new StreamMarkupHealing({ pattern: streamMarkupHealingPattern })
|
||||
: undefined;
|
||||
@@ -1344,6 +1342,11 @@ function buildParams(
|
||||
|
||||
if (compat.extraBody) {
|
||||
Object.assign(params, compat.extraBody);
|
||||
if (model.provider === "fireworks" && params.reasoning_effort !== undefined) {
|
||||
// Fireworks rejects simultaneous DeepSeek-style `thinking` toggles and
|
||||
// OpenAI-style `reasoning_effort`; the effort field carries the user's level.
|
||||
delete params.thinking;
|
||||
}
|
||||
}
|
||||
|
||||
return { params, toolStrictMode };
|
||||
|
||||
@@ -212,6 +212,59 @@ export function repairOrphanResponsesToolOutputs(input: ResponseInput): Response
|
||||
});
|
||||
}
|
||||
|
||||
/** Placeholder output for a tool call whose result is absent from the input. */
|
||||
const ORPHAN_TOOL_CALL_PLACEHOLDER =
|
||||
"[No tool output recorded: the tool call was interrupted before it produced a result.]";
|
||||
|
||||
/**
|
||||
* Synthesize a placeholder `function_call_output` / `custom_tool_call_output`
|
||||
* for every `function_call` / `custom_tool_call` whose `call_id` has no matching
|
||||
* output later in the same input. The Responses API rejects an unpaired call
|
||||
* with `400 No tool output found for function call …`.
|
||||
*
|
||||
* Orphan calls surface when the user branches/navigates the session tree to a
|
||||
* node that ends on a tool call (the tool-result child is excluded from the
|
||||
* reconstructed history) or when a turn is aborted/crashes after the call
|
||||
* streamed but before its result persisted. Dropping the call would erase the
|
||||
* assistant's action; a placeholder output keeps the call visible so the model
|
||||
* can recover (e.g. re-issue the call). Symmetric to
|
||||
* {@link repairOrphanResponsesToolOutputs}.
|
||||
*/
|
||||
export function repairOrphanResponsesToolCalls(input: ResponseInput): ResponseInput {
|
||||
const outputCallIds = new Set<string>();
|
||||
for (const item of input) {
|
||||
const t = (item as { type?: string }).type;
|
||||
if (t !== "function_call_output" && t !== "custom_tool_call_output") continue;
|
||||
const callId = (item as { call_id?: unknown }).call_id;
|
||||
if (typeof callId === "string") outputCallIds.add(callId);
|
||||
}
|
||||
let hasOrphan = false;
|
||||
for (const item of input) {
|
||||
const t = (item as { type?: string }).type;
|
||||
if (t !== "function_call" && t !== "custom_tool_call") continue;
|
||||
const callId = (item as { call_id?: unknown }).call_id;
|
||||
if (typeof callId === "string" && !outputCallIds.has(callId)) {
|
||||
hasOrphan = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!hasOrphan) return input;
|
||||
const repaired: ResponseInput = [];
|
||||
for (const item of input) {
|
||||
repaired.push(item);
|
||||
const t = (item as { type?: string }).type;
|
||||
if (t !== "function_call" && t !== "custom_tool_call") continue;
|
||||
const callId = (item as { call_id?: unknown }).call_id;
|
||||
if (typeof callId !== "string" || outputCallIds.has(callId)) continue;
|
||||
repaired.push({
|
||||
type: t === "custom_tool_call" ? "custom_tool_call_output" : "function_call_output",
|
||||
call_id: callId,
|
||||
output: ORPHAN_TOOL_CALL_PLACEHOLDER,
|
||||
} as ResponseInput[number]);
|
||||
}
|
||||
return repaired;
|
||||
}
|
||||
|
||||
export function convertResponsesInputContent(
|
||||
content: string | Array<TextContent | ImageContent>,
|
||||
supportsImages: boolean,
|
||||
@@ -406,6 +459,13 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
// see https://github.com/can1357/oh-my-pi/issues/1880 — llama.cpp emits parallel
|
||||
// function_call deltas interleaved, and a singleton `current` reference would
|
||||
// fold them into the wrong block and drop arguments on every call but the last.
|
||||
//
|
||||
// llama.cpp's `to_json_oaicompat_resp` (issue #2015) compounds this: `output_item.added`
|
||||
// for function_call/custom_tool_call carries `item.call_id` but no `item.id` and no
|
||||
// `output_index`, while the matching `function_call_arguments.delta` carries
|
||||
// `item_id = "fc_<call_id>"`. Registering function-call items by `call_id` as a
|
||||
// secondary key lets the delta lookup find the right block on hosts that emit one
|
||||
// identifier but not the other.
|
||||
const openItemsByOutputIndex = new Map<number, StreamingItem>();
|
||||
const openItemsByItemId = new Map<string, StreamingItem>();
|
||||
let lastOpenItem: StreamingItem | null = null;
|
||||
@@ -415,9 +475,11 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
outputIndex: number | undefined,
|
||||
itemId: string | undefined,
|
||||
entry: StreamingItem,
|
||||
alternateItemKey?: string,
|
||||
): void => {
|
||||
if (typeof outputIndex === "number") openItemsByOutputIndex.set(outputIndex, entry);
|
||||
if (itemId) openItemsByItemId.set(itemId, entry);
|
||||
if (alternateItemKey && alternateItemKey !== itemId) openItemsByItemId.set(alternateItemKey, entry);
|
||||
openItemsInOrder.push(entry);
|
||||
lastOpenItem = entry;
|
||||
};
|
||||
@@ -455,9 +517,11 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
outputIndex: number | undefined,
|
||||
itemId: string | undefined,
|
||||
entry: StreamingItem | undefined,
|
||||
alternateItemKey?: string,
|
||||
): void => {
|
||||
if (typeof outputIndex === "number") openItemsByOutputIndex.delete(outputIndex);
|
||||
if (itemId) openItemsByItemId.delete(itemId);
|
||||
if (alternateItemKey && alternateItemKey !== itemId) openItemsByItemId.delete(alternateItemKey);
|
||||
if (entry) {
|
||||
const index = openItemsInOrder.indexOf(entry);
|
||||
if (index >= 0) openItemsInOrder.splice(index, 1);
|
||||
@@ -497,7 +561,7 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
partialJson: item.arguments || "",
|
||||
};
|
||||
output.content.push(block);
|
||||
registerOpenItem(event.output_index, item.id, { item, block });
|
||||
registerOpenItem(event.output_index, item.id, { item, block }, item.call_id);
|
||||
stream.push({ type: "toolcall_start", contentIndex: contentIndexOf(block), partial: output });
|
||||
} else if (item.type === "custom_tool_call") {
|
||||
const block: StreamingToolCallBlock = {
|
||||
@@ -515,7 +579,7 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
partialJson: item.input ?? "",
|
||||
};
|
||||
output.content.push(block);
|
||||
registerOpenItem(event.output_index, item.id, { item, block });
|
||||
registerOpenItem(event.output_index, item.id, { item, block }, item.call_id);
|
||||
stream.push({ type: "toolcall_start", contentIndex: contentIndexOf(block), partial: output });
|
||||
}
|
||||
} else if (event.type === "response.reasoning_summary_part.added") {
|
||||
@@ -656,7 +720,10 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
} else if (event.type === "response.output_item.done") {
|
||||
const item = structuredCloneJSON(event.item);
|
||||
options?.onOutputItemDone?.(item);
|
||||
const entry = lookupOpenItem({ output_index: event.output_index, item_id: item.id });
|
||||
const entry =
|
||||
item.type === "function_call" || item.type === "custom_tool_call"
|
||||
? lookupOpenItem({ output_index: event.output_index, item_id: item.id ?? item.call_id })
|
||||
: lookupOpenItem({ output_index: event.output_index, item_id: item.id });
|
||||
if (item.type === "reasoning") {
|
||||
const thinking =
|
||||
item.summary?.length > 0
|
||||
@@ -715,7 +782,7 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
delete (block as { argumentsDone?: boolean }).argumentsDone;
|
||||
}
|
||||
const contentIndex = block ? contentIndexOf(block) : output.content.length - 1;
|
||||
closeOpenItem(event.output_index, item.id, entry);
|
||||
closeOpenItem(event.output_index, item.id, entry, item.call_id);
|
||||
stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
|
||||
} else if (item.type === "custom_tool_call") {
|
||||
const block = entry?.block.type === "toolCall" ? entry.block : undefined;
|
||||
@@ -728,7 +795,7 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
customWireName: item.name,
|
||||
};
|
||||
const contentIndex = block ? contentIndexOf(block) : output.content.length - 1;
|
||||
closeOpenItem(event.output_index, item.id, entry);
|
||||
closeOpenItem(event.output_index, item.id, entry, item.call_id);
|
||||
stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
|
||||
}
|
||||
} else if (event.type === "response.completed") {
|
||||
|
||||
@@ -62,6 +62,7 @@ import {
|
||||
isOpenAIResponsesProgressEvent,
|
||||
normalizeResponsesToolCallIdForTransform,
|
||||
processResponsesStream,
|
||||
repairOrphanResponsesToolCalls,
|
||||
repairOrphanResponsesToolOutputs,
|
||||
} from "./openai-responses-shared";
|
||||
import { transformMessages } from "./transform-messages";
|
||||
@@ -614,7 +615,7 @@ function convertConversationMessages(
|
||||
msgIndex++;
|
||||
}
|
||||
|
||||
return repairOrphanResponsesToolOutputs(messages);
|
||||
return repairOrphanResponsesToolCalls(repairOrphanResponsesToolOutputs(messages));
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -149,7 +149,10 @@ export function streamPiNative<TApi extends Api>(
|
||||
try {
|
||||
const url = resolveStreamUrl(model as Model<Api>);
|
||||
const fetchImpl = options?.fetch ?? globalThis.fetch;
|
||||
const headers = buildHeaders(model as Model<Api>, options?.apiKey);
|
||||
const headers = buildHeaders(
|
||||
model as Model<Api>,
|
||||
typeof options?.apiKey === "string" ? options.apiKey : undefined,
|
||||
);
|
||||
const body = JSON.stringify({
|
||||
modelId: model.id,
|
||||
context,
|
||||
|
||||
@@ -17,6 +17,96 @@ const enum ToolCallStatus {
|
||||
Aborted = 2,
|
||||
}
|
||||
|
||||
/**
|
||||
* Maximum tool-call id length the strictest replay provider accepts.
|
||||
*
|
||||
* Anthropic requires `^[a-zA-Z0-9_-]+$` with a 64-char cap; Google and Codex
|
||||
* `normalizeToolCallId` implementations cap individual id segments to the same
|
||||
* 64-char ceiling. Replacement ids minted here flow back through
|
||||
* `convertAnthropicMessages` (and friends) unchanged, so the `_dupN` suffix
|
||||
* MUST not push a normalized id past this bound.
|
||||
*/
|
||||
const MAX_TOOL_CALL_ID_LENGTH = 64;
|
||||
|
||||
function appendDuplicateSuffix(originalId: string, suffix: string): string {
|
||||
if (originalId.length + suffix.length <= MAX_TOOL_CALL_ID_LENGTH) return `${originalId}${suffix}`;
|
||||
const prefixBudget = Math.max(0, MAX_TOOL_CALL_ID_LENGTH - suffix.length);
|
||||
return `${originalId.slice(0, prefixBudget)}${suffix}`;
|
||||
}
|
||||
|
||||
type PendingToolResultRewrite = { replacementId: string } | undefined;
|
||||
|
||||
function deduplicateToolCallIds(messages: Message[]): Message[] {
|
||||
const seenToolCallIds = new Map<string, number>();
|
||||
const pendingToolResultRewrites = new Map<string, PendingToolResultRewrite[]>();
|
||||
|
||||
return messages.map(msg => {
|
||||
if (msg.role === "toolResult") {
|
||||
const rewrites = pendingToolResultRewrites.get(msg.toolCallId);
|
||||
if (!rewrites || rewrites.length === 0) return msg;
|
||||
|
||||
const rewrite = rewrites.shift();
|
||||
if (rewrites.length === 0) pendingToolResultRewrites.delete(msg.toolCallId);
|
||||
if (rewrite) return { ...msg, toolCallId: rewrite.replacementId };
|
||||
return msg;
|
||||
}
|
||||
|
||||
if (msg.role !== "assistant") return msg;
|
||||
|
||||
const enqueueToolResultRewrite = (id: string, rewrite: PendingToolResultRewrite): void => {
|
||||
const rewrites = pendingToolResultRewrites.get(id);
|
||||
if (rewrites) {
|
||||
rewrites.push(rewrite);
|
||||
return;
|
||||
}
|
||||
pendingToolResultRewrites.set(id, [rewrite]);
|
||||
};
|
||||
|
||||
// Ids this turn has already touched; used to scope the "drop carried-over
|
||||
// pending rewrites" semantics to the FIRST occurrence per turn so multiple
|
||||
// blocks of the same id within one turn still accumulate as duplicates.
|
||||
const idsTouchedInTurn = new Set<string>();
|
||||
let contentChanged = false;
|
||||
const content = msg.content.map(block => {
|
||||
if (block.type !== "toolCall") return block;
|
||||
|
||||
// Drop any pending rewrites carried over from a prior assistant turn
|
||||
// for this id on its first appearance this turn. When a later turn
|
||||
// re-emits the same id, the older duplicate call's expected result
|
||||
// never landed in time — the second pass synthesizes
|
||||
// "No result provided" for it, and the upcoming real result(id) must
|
||||
// route to one of THIS turn's calls. Without this guard the older
|
||||
// `_dup` id would steal the next result.
|
||||
if (!idsTouchedInTurn.has(block.id)) {
|
||||
pendingToolResultRewrites.delete(block.id);
|
||||
idsTouchedInTurn.add(block.id);
|
||||
}
|
||||
|
||||
const previousCount = seenToolCallIds.get(block.id) ?? 0;
|
||||
if (previousCount === 0) {
|
||||
seenToolCallIds.set(block.id, 1);
|
||||
enqueueToolResultRewrite(block.id, undefined);
|
||||
return block;
|
||||
}
|
||||
|
||||
let duplicateIndex = previousCount;
|
||||
let replacementId = appendDuplicateSuffix(block.id, `_dup${duplicateIndex}`);
|
||||
while (seenToolCallIds.has(replacementId)) {
|
||||
duplicateIndex += 1;
|
||||
replacementId = appendDuplicateSuffix(block.id, `_dup${duplicateIndex}`);
|
||||
}
|
||||
seenToolCallIds.set(block.id, duplicateIndex + 1);
|
||||
seenToolCallIds.set(replacementId, 1);
|
||||
enqueueToolResultRewrite(block.id, { replacementId });
|
||||
contentChanged = true;
|
||||
return { ...block, id: replacementId };
|
||||
});
|
||||
|
||||
if (!contentChanged) return msg;
|
||||
return { ...msg, content };
|
||||
});
|
||||
}
|
||||
|
||||
function shouldDropTruncatedThinkingOnlyAssistant(msg: AssistantMessage): boolean {
|
||||
const isTruncatedStop = msg.stopReason === "length" || msg.stopReason === "error" || msg.stopReason === "aborted";
|
||||
return isTruncatedStop && !msg.content.some(block => block.type === "toolCall" || block.type === "text");
|
||||
@@ -52,116 +142,120 @@ export function transformMessages<TApi extends Api>(
|
||||
|
||||
const latestSurvivingAssistantIndex = getLatestSurvivingAssistantIndex(messages);
|
||||
// First pass: transform messages (thinking blocks, tool call ID normalization)
|
||||
const transformed = messages.map((msg, index) => {
|
||||
// User and developer messages pass through unchanged
|
||||
if (msg.role === "user" || msg.role === "developer") {
|
||||
return msg;
|
||||
}
|
||||
const transformed = deduplicateToolCallIds(
|
||||
messages.map((msg, index) => {
|
||||
// User and developer messages pass through unchanged
|
||||
if (msg.role === "user" || msg.role === "developer") {
|
||||
return msg;
|
||||
}
|
||||
|
||||
// Handle toolResult messages - normalize toolCallId if we have a mapping
|
||||
if (msg.role === "toolResult") {
|
||||
const normalizedId = toolCallIdMap.get(msg.toolCallId);
|
||||
if (normalizedId && normalizedId !== msg.toolCallId) {
|
||||
return { ...msg, toolCallId: normalizedId };
|
||||
// Handle toolResult messages - normalize toolCallId if we have a mapping
|
||||
if (msg.role === "toolResult") {
|
||||
const normalizedId = toolCallIdMap.get(msg.toolCallId);
|
||||
if (normalizedId && normalizedId !== msg.toolCallId) {
|
||||
return { ...msg, toolCallId: normalizedId };
|
||||
}
|
||||
return msg;
|
||||
}
|
||||
|
||||
// Assistant messages need transformation check
|
||||
if (msg.role === "assistant") {
|
||||
const assistantMsg = msg as AssistantMessage;
|
||||
const isSameModel =
|
||||
assistantMsg.provider === model.provider &&
|
||||
assistantMsg.api === model.api &&
|
||||
assistantMsg.model === model.id;
|
||||
|
||||
const mustPreserveLatestAnthropicThinking =
|
||||
index === latestSurvivingAssistantIndex &&
|
||||
model.api === "anthropic-messages" &&
|
||||
assistantMsg.api === "anthropic-messages";
|
||||
// Aborted/errored messages may have partially-streamed thinking signatures.
|
||||
// A partial signature is invalid and will be rejected by the API, so we must
|
||||
// strip signatures from thinking blocks in these messages.
|
||||
//
|
||||
// Abandoned tool-use turns get the same treatment once they are no longer
|
||||
// the latest assistant message. When a turn carries toolCall blocks but did
|
||||
// NOT request tool execution (stopReason !== "toolUse" — e.g.
|
||||
// adaptive-thinking Opus emitting tool calls and then ending the turn on
|
||||
// `end_turn`/`stop`), the agent loop pairs those calls with placeholder
|
||||
// tool_results to keep the tool_use/tool_result contract valid. Historical
|
||||
// abandoned turns cannot safely replay their end_turn-bound signatures in
|
||||
// that continuation, so stripping downgrades them to plain text downstream.
|
||||
// Latest abandoned turns are exempt because Anthropic requires thinking
|
||||
// blocks from its most recent response to remain byte-for-byte unmodified.
|
||||
const invalidStopReason = assistantMsg.stopReason === "aborted" || assistantMsg.stopReason === "error";
|
||||
const abandonedToolUse =
|
||||
!invalidStopReason &&
|
||||
assistantMsg.stopReason !== "toolUse" &&
|
||||
assistantMsg.content.some(b => b.type === "toolCall");
|
||||
const hasInvalidSignatures = invalidStopReason || abandonedToolUse;
|
||||
|
||||
const transformedContent = assistantMsg.content.flatMap(block => {
|
||||
if (block.type === "thinking") {
|
||||
// Strip untrustworthy signatures so the encoder can downgrade to text.
|
||||
const sanitized =
|
||||
hasInvalidSignatures && block.thinkingSignature
|
||||
? { ...block, thinkingSignature: undefined }
|
||||
: block;
|
||||
if (mustPreserveLatestAnthropicThinking) return abandonedToolUse ? block : sanitized;
|
||||
// For same model: keep thinking blocks with signatures (needed for replay)
|
||||
// even if the thinking text is empty (OpenAI encrypted reasoning)
|
||||
if (isSameModel && sanitized.thinkingSignature) return sanitized;
|
||||
// Skip empty thinking blocks, convert others to plain text
|
||||
if (!sanitized.thinking || sanitized.thinking.trim() === "") return [];
|
||||
if (isSameModel) return sanitized;
|
||||
return {
|
||||
type: "text" as const,
|
||||
text: sanitized.thinking,
|
||||
};
|
||||
}
|
||||
|
||||
if (block.type === "redactedThinking") {
|
||||
if (mustPreserveLatestAnthropicThinking) return block;
|
||||
if (isSameModel) return block;
|
||||
return [];
|
||||
}
|
||||
|
||||
if (block.type === "text") {
|
||||
if (isSameModel) return block;
|
||||
return {
|
||||
type: "text" as const,
|
||||
text: block.text,
|
||||
};
|
||||
}
|
||||
|
||||
if (block.type === "toolCall") {
|
||||
const toolCall = block as ToolCall;
|
||||
let normalizedToolCall: ToolCall = toolCall;
|
||||
|
||||
if (!isSameModel && toolCall.thoughtSignature) {
|
||||
normalizedToolCall = { ...toolCall };
|
||||
delete (normalizedToolCall as { thoughtSignature?: string }).thoughtSignature;
|
||||
}
|
||||
|
||||
if (!isSameModel && normalizeToolCallId) {
|
||||
const normalizedId = normalizeToolCallId(toolCall.id, model, assistantMsg);
|
||||
if (normalizedId !== toolCall.id) {
|
||||
toolCallIdMap.set(toolCall.id, normalizedId);
|
||||
normalizedToolCall = { ...normalizedToolCall, id: normalizedId };
|
||||
}
|
||||
}
|
||||
|
||||
return normalizedToolCall;
|
||||
}
|
||||
|
||||
return block;
|
||||
});
|
||||
|
||||
return {
|
||||
...assistantMsg,
|
||||
content: transformedContent,
|
||||
};
|
||||
}
|
||||
return msg;
|
||||
}
|
||||
|
||||
// Assistant messages need transformation check
|
||||
if (msg.role === "assistant") {
|
||||
const assistantMsg = msg as AssistantMessage;
|
||||
const isSameModel =
|
||||
assistantMsg.provider === model.provider &&
|
||||
assistantMsg.api === model.api &&
|
||||
assistantMsg.model === model.id;
|
||||
|
||||
const mustPreserveLatestAnthropicThinking =
|
||||
index === latestSurvivingAssistantIndex &&
|
||||
model.api === "anthropic-messages" &&
|
||||
assistantMsg.api === "anthropic-messages";
|
||||
// Aborted/errored messages may have partially-streamed thinking signatures.
|
||||
// A partial signature is invalid and will be rejected by the API, so we must
|
||||
// strip signatures from thinking blocks in these messages.
|
||||
//
|
||||
// Abandoned tool-use turns get the same treatment once they are no longer
|
||||
// the latest assistant message. When a turn carries toolCall blocks but did
|
||||
// NOT request tool execution (stopReason !== "toolUse" — e.g.
|
||||
// adaptive-thinking Opus emitting tool calls and then ending the turn on
|
||||
// `end_turn`/`stop`), the agent loop pairs those calls with placeholder
|
||||
// tool_results to keep the tool_use/tool_result contract valid. Historical
|
||||
// abandoned turns cannot safely replay their end_turn-bound signatures in
|
||||
// that continuation, so stripping downgrades them to plain text downstream.
|
||||
// Latest abandoned turns are exempt because Anthropic requires thinking
|
||||
// blocks from its most recent response to remain byte-for-byte unmodified.
|
||||
const invalidStopReason = assistantMsg.stopReason === "aborted" || assistantMsg.stopReason === "error";
|
||||
const abandonedToolUse =
|
||||
!invalidStopReason &&
|
||||
assistantMsg.stopReason !== "toolUse" &&
|
||||
assistantMsg.content.some(b => b.type === "toolCall");
|
||||
const hasInvalidSignatures = invalidStopReason || abandonedToolUse;
|
||||
|
||||
const transformedContent = assistantMsg.content.flatMap(block => {
|
||||
if (block.type === "thinking") {
|
||||
// Strip untrustworthy signatures so the encoder can downgrade to text.
|
||||
const sanitized =
|
||||
hasInvalidSignatures && block.thinkingSignature ? { ...block, thinkingSignature: undefined } : block;
|
||||
if (mustPreserveLatestAnthropicThinking) return abandonedToolUse ? block : sanitized;
|
||||
// For same model: keep thinking blocks with signatures (needed for replay)
|
||||
// even if the thinking text is empty (OpenAI encrypted reasoning)
|
||||
if (isSameModel && sanitized.thinkingSignature) return sanitized;
|
||||
// Skip empty thinking blocks, convert others to plain text
|
||||
if (!sanitized.thinking || sanitized.thinking.trim() === "") return [];
|
||||
if (isSameModel) return sanitized;
|
||||
return {
|
||||
type: "text" as const,
|
||||
text: sanitized.thinking,
|
||||
};
|
||||
}
|
||||
|
||||
if (block.type === "redactedThinking") {
|
||||
if (mustPreserveLatestAnthropicThinking) return block;
|
||||
if (isSameModel) return block;
|
||||
return [];
|
||||
}
|
||||
|
||||
if (block.type === "text") {
|
||||
if (isSameModel) return block;
|
||||
return {
|
||||
type: "text" as const,
|
||||
text: block.text,
|
||||
};
|
||||
}
|
||||
|
||||
if (block.type === "toolCall") {
|
||||
const toolCall = block as ToolCall;
|
||||
let normalizedToolCall: ToolCall = toolCall;
|
||||
|
||||
if (!isSameModel && toolCall.thoughtSignature) {
|
||||
normalizedToolCall = { ...toolCall };
|
||||
delete (normalizedToolCall as { thoughtSignature?: string }).thoughtSignature;
|
||||
}
|
||||
|
||||
if (!isSameModel && normalizeToolCallId) {
|
||||
const normalizedId = normalizeToolCallId(toolCall.id, model, assistantMsg);
|
||||
if (normalizedId !== toolCall.id) {
|
||||
toolCallIdMap.set(toolCall.id, normalizedId);
|
||||
normalizedToolCall = { ...normalizedToolCall, id: normalizedId };
|
||||
}
|
||||
}
|
||||
|
||||
return normalizedToolCall;
|
||||
}
|
||||
|
||||
return block;
|
||||
});
|
||||
|
||||
return {
|
||||
...assistantMsg,
|
||||
content: transformedContent,
|
||||
};
|
||||
}
|
||||
return msg;
|
||||
});
|
||||
}),
|
||||
);
|
||||
const realToolResultsById = new Map<string, ToolResultMessage>();
|
||||
for (const msg of transformed) {
|
||||
if (msg.role === "toolResult" && !realToolResultsById.has(msg.toolCallId)) {
|
||||
|
||||
+38
-19
@@ -3,7 +3,8 @@ import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { $env, $pickenv, extractHttpStatusFromError } from "@oh-my-pi/pi-utils";
|
||||
import { getCustomApi } from "./api-registry";
|
||||
import type { Effort } from "./model-thinking";
|
||||
import { AUTH_RETRY_STEPS, isApiKeyResolver, resolveRetryKey } from "./auth-retry";
|
||||
import type { Effort } from "./effort";
|
||||
import {
|
||||
mapEffortToAnthropicAdaptiveEffort,
|
||||
mapEffortToGoogleThinkingLevel,
|
||||
@@ -208,8 +209,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
|
||||
tavily: "TAVILY_API_KEY",
|
||||
parallel: "PARALLEL_API_KEY",
|
||||
kagi: "KAGI_API_KEY",
|
||||
// GitHub Copilot uses GitHub personal access token
|
||||
"github-copilot": () => $pickenv("COPILOT_GITHUB_TOKEN", "GH_TOKEN", "GITHUB_TOKEN"),
|
||||
"github-copilot": "COPILOT_GITHUB_TOKEN",
|
||||
// Foundry mode optionally switches Anthropic auth to enterprise gateway credentials.
|
||||
anthropic: () =>
|
||||
isFoundryEnabled()
|
||||
@@ -443,12 +443,15 @@ export function streamSimple<TApi extends Api>(
|
||||
options?: SimpleStreamOptions,
|
||||
): AssistantMessageEventStream {
|
||||
const requestOptions = withRequestDebugFetch(options);
|
||||
const retryApiKey = requestOptions?.onAuthError
|
||||
? (requestOptions.apiKey ?? getEnvApiKey(model.provider))
|
||||
: undefined;
|
||||
if (retryApiKey) {
|
||||
const apiKeyResolver = isApiKeyResolver(requestOptions?.apiKey) ? requestOptions.apiKey : undefined;
|
||||
if (apiKeyResolver) {
|
||||
const outer = new AssistantMessageEventStream();
|
||||
const onAuthError = requestOptions!.onAuthError!;
|
||||
const signal = requestOptions?.signal;
|
||||
// One inner attempt against a resolved string key. When
|
||||
// `captureAuthFailure` is set, a retryable auth error that arrives before
|
||||
// any replay-unsafe event is buffered and returned (so the caller can
|
||||
// retry with a fresh key) instead of surfaced. The terminal attempt
|
||||
// clears the flag and emits whatever it gets.
|
||||
const runAttempt = async (apiKey: string, captureAuthFailure: boolean): Promise<AuthRetryFailure | undefined> => {
|
||||
const bufferedEvents: AssistantMessageEvent[] = [];
|
||||
let emittedReplayUnsafeEvent = false;
|
||||
@@ -458,7 +461,7 @@ export function streamSimple<TApi extends Api>(
|
||||
};
|
||||
|
||||
try {
|
||||
const inner = streamSimple(model, context, { ...requestOptions, apiKey, onAuthError: undefined });
|
||||
const inner = streamSimple(model, context, { ...requestOptions, apiKey });
|
||||
for await (const event of inner) {
|
||||
if (!emittedReplayUnsafeEvent && event.type === "start") {
|
||||
bufferedEvents.push(event);
|
||||
@@ -510,19 +513,32 @@ export function streamSimple<TApi extends Api>(
|
||||
};
|
||||
|
||||
void (async () => {
|
||||
const failure = await runAttempt(retryApiKey, true);
|
||||
if (!failure) return;
|
||||
let nextKey: string | undefined;
|
||||
let lastKey: string | undefined;
|
||||
try {
|
||||
nextKey = await onAuthError(model.provider, retryApiKey, failure.error);
|
||||
lastKey = (await apiKeyResolver({ lastChance: false, error: undefined, signal })) || undefined;
|
||||
} catch {
|
||||
nextKey = undefined;
|
||||
lastKey = undefined;
|
||||
}
|
||||
if (!nextKey || nextKey === retryApiKey) {
|
||||
emitFailure(failure);
|
||||
if (lastKey === undefined) {
|
||||
outer.fail(new Error(`No API key for provider: ${model.provider}`));
|
||||
return;
|
||||
}
|
||||
await runAttempt(nextKey, false);
|
||||
let failure = await runAttempt(lastKey, true);
|
||||
if (!failure) return;
|
||||
// a/b/c policy: refresh the same account (lastChance=false), then
|
||||
// switch to a sibling (lastChance=true). A step is skipped when the
|
||||
// resolver yields the same key it just tried or `undefined`; the
|
||||
// final step's attempt clears the capture flag so it emits directly.
|
||||
for (let step = 0; step < AUTH_RETRY_STEPS.length; step++) {
|
||||
const nextKey = await resolveRetryKey(apiKeyResolver, AUTH_RETRY_STEPS[step]!, failure.error, signal);
|
||||
if (nextKey === undefined || nextKey === lastKey) continue;
|
||||
lastKey = nextKey;
|
||||
const isLastStep = step === AUTH_RETRY_STEPS.length - 1;
|
||||
const next = await runAttempt(nextKey, !isLastStep);
|
||||
if (!next) return;
|
||||
failure = next;
|
||||
}
|
||||
emitFailure(failure);
|
||||
})();
|
||||
return outer;
|
||||
}
|
||||
@@ -553,7 +569,10 @@ export function streamSimple<TApi extends Api>(
|
||||
return stream(model, context, providerOptions);
|
||||
}
|
||||
|
||||
const apiKey = requestOptions?.apiKey || getEnvApiKey(model.provider);
|
||||
// The resolver form is handled by the wrapper above; only a static string
|
||||
// key reaches this point.
|
||||
const apiKey =
|
||||
(typeof requestOptions?.apiKey === "string" ? requestOptions.apiKey : undefined) || getEnvApiKey(model.provider);
|
||||
if (!apiKey) {
|
||||
throw new Error(`No API key for provider: ${model.provider}`);
|
||||
}
|
||||
@@ -724,7 +743,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
repetitionPenalty: options?.repetitionPenalty,
|
||||
maxTokens: options?.maxTokens ?? model.maxTokens,
|
||||
signal: options?.signal,
|
||||
apiKey: apiKey || options?.apiKey,
|
||||
apiKey: apiKey ?? (typeof options?.apiKey === "string" ? options.apiKey : undefined),
|
||||
cacheRetention: options?.cacheRetention,
|
||||
headers: options?.headers,
|
||||
initiatorOverride: options?.initiatorOverride,
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import type { ZodType, z } from "zod/v4";
|
||||
import type { ApiKey } from "./auth-retry";
|
||||
import type { BedrockOptions } from "./providers/amazon-bedrock";
|
||||
import type { AnthropicOptions } from "./providers/anthropic";
|
||||
import type { AzureOpenAIResponsesOptions } from "./providers/azure-openai-responses";
|
||||
@@ -151,7 +152,7 @@ export type KnownProvider =
|
||||
| "lm-studio";
|
||||
export type Provider = KnownProvider | string;
|
||||
|
||||
import type { Effort } from "./model-thinking";
|
||||
import type { Effort } from "./effort";
|
||||
|
||||
/** Token budgets for each thinking level (token-based providers only) */
|
||||
export type ThinkingBudgets = { [key in Effort]?: number };
|
||||
@@ -298,12 +299,6 @@ export interface StreamOptions {
|
||||
maxTokens?: number;
|
||||
signal?: AbortSignal;
|
||||
apiKey?: string;
|
||||
/**
|
||||
* Called when a provider returns 401 before any replay-unsafe assistant
|
||||
* event has been emitted. Returning a different key retries the provider
|
||||
* request once.
|
||||
*/
|
||||
onAuthError?: (provider: string, apiKey: string, error: unknown) => Promise<string | undefined>;
|
||||
cacheRetention?: CacheRetention;
|
||||
/**
|
||||
* Additional headers to include in provider requests.
|
||||
@@ -415,7 +410,15 @@ export interface StreamOptions {
|
||||
}
|
||||
|
||||
// Unified options with reasoning passed to streamSimple() and completeSimple()
|
||||
export interface SimpleStreamOptions extends StreamOptions {
|
||||
export interface SimpleStreamOptions extends Omit<StreamOptions, "apiKey"> {
|
||||
/**
|
||||
* API key for the request: either a static bearer string, or an
|
||||
* {@link ApiKeyResolver} that mints/rotates the key across the central
|
||||
* a/b/c auth-retry policy. `streamSimple`/`completeSimple` resolve a
|
||||
* resolver to a string before per-provider dispatch, so providers only
|
||||
* ever see the resolved {@link StreamOptions.apiKey} string.
|
||||
*/
|
||||
apiKey?: ApiKey;
|
||||
reasoning?: Effort;
|
||||
/**
|
||||
* Force-disable reasoning for the request even when the model supports it.
|
||||
|
||||
@@ -148,10 +148,17 @@ async function fetchAntigravityUsage(params: UsageFetchParams, ctx: UsageFetchCo
|
||||
}
|
||||
|
||||
const data = (await response.json()) as AntigravityUsageResponse;
|
||||
const limits: UsageLimit[] = [];
|
||||
|
||||
// The API returns per-model quota entries, but quota is shared across
|
||||
// models within the same tier. Deduplicate by (tier, windowId) so one
|
||||
// account doesn't produce 15 redundant bars.
|
||||
const deduped = new Map<
|
||||
string,
|
||||
{ amount: UsageAmount; window: UsageWindow | undefined; tier: string | undefined }
|
||||
>();
|
||||
let earliestReset: number | undefined;
|
||||
|
||||
for (const [modelId, modelInfo] of Object.entries(data.models ?? {})) {
|
||||
for (const [_modelId, modelInfo] of Object.entries(data.models ?? {})) {
|
||||
const quotaInfos = normalizeQuotaInfos(modelInfo);
|
||||
for (const quotaInfo of quotaInfos) {
|
||||
const amount = buildAmount(quotaInfo);
|
||||
@@ -159,35 +166,81 @@ async function fetchAntigravityUsage(params: UsageFetchParams, ctx: UsageFetchCo
|
||||
if (window?.resetsAt) {
|
||||
earliestReset = earliestReset ? Math.min(earliestReset, window.resetsAt) : window.resetsAt;
|
||||
}
|
||||
const labelBase = modelInfo.displayName || modelId;
|
||||
const label = quotaInfo.tier ? `${labelBase} (${quotaInfo.tier})` : labelBase;
|
||||
const windowId = window?.id ?? "default";
|
||||
limits.push({
|
||||
id: `${modelId}:${quotaInfo.tier ?? "default"}:${windowId}`,
|
||||
label,
|
||||
scope: {
|
||||
provider: params.provider,
|
||||
accountId: credential.accountId,
|
||||
projectId: credential.projectId,
|
||||
modelId,
|
||||
tier: quotaInfo.tier,
|
||||
windowId,
|
||||
},
|
||||
window,
|
||||
amount,
|
||||
status: getUsageStatus(amount.remainingFraction),
|
||||
});
|
||||
const tier = (quotaInfo.tier ?? "default").toLowerCase();
|
||||
// Use quotaInfo.windowId even when parseWindow returns undefined
|
||||
// (no resetTime) — separate windows must not collapse to "default".
|
||||
const windowId = quotaInfo.windowId ?? window?.id ?? "default";
|
||||
const key = `${tier}|${windowId}`;
|
||||
const existing = deduped.get(key);
|
||||
if (!existing) {
|
||||
deduped.set(key, { amount, window, tier: quotaInfo.tier });
|
||||
continue;
|
||||
}
|
||||
// Merge: keep the entry with fraction data for the bar, but
|
||||
// also keep any window with a reset time so "resets in…" survives.
|
||||
const eFrac = existing.amount.remainingFraction;
|
||||
const cFrac = amount.remainingFraction;
|
||||
const eHasFrac = eFrac !== undefined;
|
||||
const cHasFrac = cFrac !== undefined;
|
||||
|
||||
let bestAmount = existing.amount;
|
||||
let bestWindow = existing.window?.resetsAt ? existing.window : (window ?? existing.window);
|
||||
let bestTier = existing.tier ?? quotaInfo.tier;
|
||||
|
||||
if (!eHasFrac && cHasFrac) {
|
||||
bestAmount = amount;
|
||||
bestTier = quotaInfo.tier ?? existing.tier;
|
||||
} else if (eHasFrac && cHasFrac && cFrac! < eFrac!) {
|
||||
bestAmount = amount;
|
||||
bestTier = quotaInfo.tier ?? existing.tier;
|
||||
}
|
||||
// Always merge in window with reset time if the current
|
||||
// best doesn't have one.
|
||||
if (!bestWindow?.resetsAt && window?.resetsAt) {
|
||||
bestWindow = window;
|
||||
}
|
||||
deduped.set(key, { amount: bestAmount, window: bestWindow, tier: bestTier });
|
||||
}
|
||||
}
|
||||
|
||||
const limits: UsageLimit[] = [];
|
||||
for (const [key, entry] of deduped) {
|
||||
const [tier, windowId] = key.split("|") as [string, string];
|
||||
const label = "Usage";
|
||||
limits.push({
|
||||
id: `${params.provider}:${tier}:${windowId}`,
|
||||
label,
|
||||
scope: {
|
||||
provider: params.provider,
|
||||
accountId: credential.accountId,
|
||||
projectId: credential.projectId,
|
||||
tier: entry.tier,
|
||||
windowId,
|
||||
},
|
||||
window: entry.window,
|
||||
amount: entry.amount,
|
||||
status: getUsageStatus(entry.amount.remainingFraction),
|
||||
});
|
||||
}
|
||||
|
||||
limits.sort((a, b) => {
|
||||
const aFraction = a.amount.remainingFraction ?? 1;
|
||||
const bFraction = b.amount.remainingFraction ?? 1;
|
||||
return aFraction - bFraction;
|
||||
});
|
||||
|
||||
const metadata: UsageReport["metadata"] = {
|
||||
endpoint: url,
|
||||
projectId: credential.projectId,
|
||||
};
|
||||
if (credential.email) metadata.email = credential.email;
|
||||
if (credential.accountId) metadata.accountId = credential.accountId;
|
||||
|
||||
const report: UsageReport = {
|
||||
provider: params.provider,
|
||||
fetchedAt: nowMs,
|
||||
limits,
|
||||
metadata: {
|
||||
endpoint: url,
|
||||
projectId: credential.projectId,
|
||||
},
|
||||
metadata,
|
||||
raw: data,
|
||||
};
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import * as path from "node:path";
|
||||
import { extractHttpStatusFromError, getLogsDir } from "@oh-my-pi/pi-utils";
|
||||
import { extractHttpStatusFromError, getLogsDir, isBunTestRuntime } from "@oh-my-pi/pi-utils";
|
||||
import { isCopilotTransientModelError } from "./retry.js";
|
||||
import { formatErrorMessageWithRetryAfter } from "./retry-after.js";
|
||||
|
||||
@@ -31,7 +31,9 @@ export async function appendRawHttpRequestDumpFor400(
|
||||
error: unknown,
|
||||
dump: RawHttpRequestDump | undefined,
|
||||
): Promise<string> {
|
||||
if (!dump || extractHttpStatusFromError(error) !== 400) {
|
||||
// Never persist dumps under the test runner: providers exercise the 400 path
|
||||
// with mocked fetch responses, which would otherwise litter the real ~/.omp logs.
|
||||
if (!dump || isBunTestRuntime() || extractHttpStatusFromError(error) !== 400) {
|
||||
return message;
|
||||
}
|
||||
|
||||
|
||||
@@ -170,6 +170,7 @@ class FileRequestDebugSession implements RequestDebugSession {
|
||||
class FileRequestDebugResponseLog implements RequestDebugResponseLog {
|
||||
#handle: fs.FileHandle | undefined;
|
||||
#pending: Promise<void> = Promise.resolve();
|
||||
#closed: Promise<void> | undefined;
|
||||
|
||||
constructor(handle: fs.FileHandle) {
|
||||
this.#handle = handle;
|
||||
@@ -184,15 +185,19 @@ class FileRequestDebugResponseLog implements RequestDebugResponseLog {
|
||||
});
|
||||
}
|
||||
|
||||
async close(): Promise<void> {
|
||||
close(): Promise<void> {
|
||||
if (this.#closed) return this.#closed;
|
||||
const handle = this.#handle;
|
||||
if (!handle) return;
|
||||
if (!handle) return Promise.resolve();
|
||||
this.#handle = undefined;
|
||||
try {
|
||||
await this.#pending;
|
||||
} finally {
|
||||
await handle.close();
|
||||
}
|
||||
this.#closed = (async () => {
|
||||
try {
|
||||
await this.#pending;
|
||||
} finally {
|
||||
await handle.close();
|
||||
}
|
||||
})();
|
||||
return this.#closed;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -154,6 +154,22 @@ export const CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS: Record<string, Record<string,
|
||||
null: {},
|
||||
};
|
||||
|
||||
/**
|
||||
* Flat set of every type-specific key across all CCA types.
|
||||
* Used to identify sibling keys that need filtering during mixed-type collapse.
|
||||
*/
|
||||
export const ALL_CCA_TYPE_SPECIFIC_KEYS: Record<string, true> = buildAllCcaTypeSpecificKeys();
|
||||
|
||||
function buildAllCcaTypeSpecificKeys(): Record<string, true> {
|
||||
const all: Record<string, true> = {};
|
||||
for (const typeKeys of Object.values(CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS)) {
|
||||
for (const key in typeKeys) {
|
||||
all[key] = true;
|
||||
}
|
||||
}
|
||||
return all;
|
||||
}
|
||||
|
||||
/**
|
||||
* Cloud Code Assist shared schema keys allowed on any type.
|
||||
* Used alongside CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS for CCA combiner collapsing.
|
||||
|
||||
@@ -20,6 +20,14 @@ export interface JsonSchemaValidationIssue {
|
||||
message: string;
|
||||
expectedTypes?: string[];
|
||||
keyword?: string;
|
||||
/**
|
||||
* Marks issues that originate inside a failed `anyOf` / `oneOf` branch.
|
||||
* Consumers such as the tool-argument coercion layer use this to avoid
|
||||
* applying type repairs (e.g. singleton-array wrapping) that would be
|
||||
* authoritative outside of a combinator but are only one candidate
|
||||
* branch's expectation here.
|
||||
*/
|
||||
fromUnionBranch?: boolean;
|
||||
}
|
||||
|
||||
export interface JsonSchemaValidationResult {
|
||||
@@ -242,7 +250,17 @@ function validateSchemaNode(
|
||||
const branchValid = keyword === "anyOf" ? matches > 0 : matches === 1;
|
||||
if (!branchValid) {
|
||||
if (matches === 0 && firstIssues && firstIssues.length > 0) {
|
||||
issues.push(...firstIssues);
|
||||
// Only tag issues that sit at the combinator's own path as
|
||||
// union-branch; deeper issues describe a specific field within
|
||||
// the failed branch and should remain individually repairable.
|
||||
const unionDepth = path.length;
|
||||
for (const branchIssue of firstIssues) {
|
||||
if (branchIssue.path.length === unionDepth) {
|
||||
issues.push({ ...branchIssue, fromUnionBranch: true });
|
||||
} else {
|
||||
issues.push(branchIssue);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
pushIssue(
|
||||
issues,
|
||||
|
||||
@@ -11,6 +11,7 @@ import { dereferenceJsonSchema } from "./dereference";
|
||||
import { upgradeJsonSchemaTo202012 } from "./draft";
|
||||
import { areJsonValuesEqual, mergePropertySchemas } from "./equality";
|
||||
import {
|
||||
ALL_CCA_TYPE_SPECIFIC_KEYS,
|
||||
CLOUD_CODE_ASSIST_SHARED_SCHEMA_KEYS,
|
||||
CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS,
|
||||
COMBINATOR_KEYS,
|
||||
@@ -501,12 +502,32 @@ function collapseMixedTypeCombinerVariants(schema: JsonObject, combiner: "anyOf"
|
||||
if (variantTypes.length < 2 || variantTypes.every(type => type === "object")) {
|
||||
return schema;
|
||||
}
|
||||
|
||||
const nextSchema = copySchemaWithout(schema, combiner);
|
||||
const nonNullTypes = variantTypes.filter(t => t !== "null");
|
||||
nextSchema.type = nonNullTypes[0] ?? variantTypes[0];
|
||||
const chosenType: string = nonNullTypes[0] ?? variantTypes[0];
|
||||
nextSchema.type = chosenType;
|
||||
const chosenTypeAllowedKeys = CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS[chosenType] ?? {};
|
||||
|
||||
// Strip sibling keys that were copied from the parent and belong to a
|
||||
// different type (e.g. `items` sibling on a now-string-typed schema).
|
||||
for (const key in nextSchema) {
|
||||
if (!Object.hasOwn(nextSchema, key)) continue;
|
||||
if (key === "type") continue;
|
||||
if (
|
||||
Object.hasOwn(ALL_CCA_TYPE_SPECIFIC_KEYS, key) &&
|
||||
!Object.hasOwn(chosenTypeAllowedKeys, key) &&
|
||||
!Object.hasOwn(CLOUD_CODE_ASSIST_SHARED_SCHEMA_KEYS, key)
|
||||
) {
|
||||
delete nextSchema[key];
|
||||
}
|
||||
}
|
||||
|
||||
for (const key in mergedVariantFields) {
|
||||
if (!Object.hasOwn(mergedVariantFields, key)) continue;
|
||||
// Drop type-specific keys that don't belong to the chosen type
|
||||
if (!Object.hasOwn(chosenTypeAllowedKeys, key) && !Object.hasOwn(CLOUD_CODE_ASSIST_SHARED_SCHEMA_KEYS, key)) {
|
||||
continue;
|
||||
}
|
||||
const value = mergedVariantFields[key];
|
||||
const existingValue = nextSchema[key];
|
||||
if (existingValue !== undefined && !areJsonValuesEqual(existingValue, value)) {
|
||||
|
||||
@@ -600,12 +600,17 @@ export function modelMayLeakDsmlToolCalls(provider: string, modelId: string): bo
|
||||
);
|
||||
}
|
||||
|
||||
/** Cheap model/provider gate for MiniMax plain thinking tag leaks. */
|
||||
export function modelMayLeakThinkingTags(provider: string, modelId: string): boolean {
|
||||
return /minimax/i.test(provider) || /minimax/i.test(modelId);
|
||||
}
|
||||
|
||||
export function getStreamMarkupHealingPattern(
|
||||
provider: string,
|
||||
modelId: string,
|
||||
options?: { readonly parseThinkingTags?: boolean },
|
||||
): StreamMarkupHealingPattern | undefined {
|
||||
if (options?.parseThinkingTags) return "thinking";
|
||||
if (options?.parseThinkingTags || modelMayLeakThinkingTags(provider, modelId)) return "thinking";
|
||||
if (modelMayLeakKimiToolCalls(provider, modelId)) return "kimi";
|
||||
if (modelMayLeakDsmlToolCalls(provider, modelId)) return "dsml";
|
||||
return undefined;
|
||||
|
||||
@@ -724,6 +724,7 @@ interface FlatIssue {
|
||||
keyword: "type" | "unrecognized" | "other";
|
||||
instancePath: string;
|
||||
expectedTypes: string[];
|
||||
unionBranch: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -759,12 +760,12 @@ function mapZodExpectedToJsonSchemaType(expected: unknown): string | null {
|
||||
*/
|
||||
function flattenIssues(issues: ReadonlyArray<ZodIssue>): FlatIssue[] {
|
||||
const out: FlatIssue[] = [];
|
||||
const walk = (issue: ZodIssue, prefix: ReadonlyArray<PropertyKey>): void => {
|
||||
const walk = (issue: ZodIssue, prefix: ReadonlyArray<PropertyKey>, unionBranch: boolean): void => {
|
||||
const fullPath = prefix.length === 0 ? issue.path : [...prefix, ...issue.path];
|
||||
if (issue.code === "invalid_type") {
|
||||
const mapped = mapZodExpectedToJsonSchemaType((issue as { expected?: unknown }).expected);
|
||||
if (mapped) {
|
||||
out.push({ keyword: "type", instancePath: pathToPointer(fullPath), expectedTypes: [mapped] });
|
||||
out.push({ keyword: "type", instancePath: pathToPointer(fullPath), expectedTypes: [mapped], unionBranch });
|
||||
return;
|
||||
}
|
||||
}
|
||||
@@ -775,6 +776,7 @@ function flattenIssues(issues: ReadonlyArray<ZodIssue>): FlatIssue[] {
|
||||
keyword: "unrecognized",
|
||||
instancePath: pathToPointer([...fullPath, key]),
|
||||
expectedTypes: [],
|
||||
unionBranch,
|
||||
});
|
||||
}
|
||||
return;
|
||||
@@ -782,17 +784,21 @@ function flattenIssues(issues: ReadonlyArray<ZodIssue>): FlatIssue[] {
|
||||
if (issue.code === "invalid_union") {
|
||||
const inner = (issue as unknown as { errors?: ReadonlyArray<ReadonlyArray<ZodIssue>> }).errors;
|
||||
if (inner) {
|
||||
// A union-branch issue only competes with a sibling branch when it
|
||||
// sits at the union node's own path. Issues whose own path is
|
||||
// non-empty live on a deeper field that an already-identified
|
||||
// branch owns, so the singleton-array repair should still apply.
|
||||
for (const branch of inner) {
|
||||
for (const child of branch) {
|
||||
walk(child, fullPath);
|
||||
walk(child, fullPath, child.path.length === 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
out.push({ keyword: "other", instancePath: pathToPointer(fullPath), expectedTypes: [] });
|
||||
out.push({ keyword: "other", instancePath: pathToPointer(fullPath), expectedTypes: [], unionBranch });
|
||||
};
|
||||
for (const issue of issues) walk(issue, []);
|
||||
for (const issue of issues) walk(issue, [], false);
|
||||
return out;
|
||||
}
|
||||
|
||||
@@ -801,7 +807,9 @@ function flattenIssues(issues: ReadonlyArray<ZodIssue>): FlatIssue[] {
|
||||
*
|
||||
* Two kinds of repair are applied:
|
||||
* - **type**: when a value is a JSON-encoded string and the schema wants
|
||||
* something else, parse it and substitute the parsed value.
|
||||
* something else, parse it and substitute the parsed value. When a
|
||||
* non-union schema wants an array but receives a singleton value, wrap that
|
||||
* value in a one-element array.
|
||||
* - **unrecognized**: when a strict object received an extra key (Zod's
|
||||
* `unrecognized_keys` or JSON Schema's `additionalProperties: false`),
|
||||
* drop that key so re-validation succeeds. This effectively coerces every
|
||||
@@ -811,6 +819,7 @@ function flattenIssues(issues: ReadonlyArray<ZodIssue>): FlatIssue[] {
|
||||
* The function is safe and conservative:
|
||||
* - Only processes "type" and "unrecognized" issues
|
||||
* - Only attempts JSON coercion on string values
|
||||
* - Only wraps singleton array values for non-union type expectations
|
||||
* - Only accepts parsed results that match the expected type
|
||||
* - Clones the args object before mutation (copy-on-write)
|
||||
*/
|
||||
@@ -836,12 +845,16 @@ function coerceArgsFromIssues(args: unknown, issues: FlatIssue[]): { value: unkn
|
||||
if (issue.expectedTypes.length === 0) continue;
|
||||
|
||||
const currentValue = getValueAtPointer(nextArgs, issue.instancePath);
|
||||
if (typeof currentValue !== "string") continue;
|
||||
|
||||
const result = tryParseJsonForTypes(currentValue, issue.expectedTypes);
|
||||
const result =
|
||||
typeof currentValue === "string"
|
||||
? tryParseJsonForTypes(currentValue, issue.expectedTypes)
|
||||
: { value: currentValue, changed: false };
|
||||
const coercedValue = result.changed
|
||||
? result.value
|
||||
: issue.expectedTypes.includes("array")
|
||||
: issue.expectedTypes.includes("array") &&
|
||||
!issue.unionBranch &&
|
||||
currentValue !== undefined &&
|
||||
!Array.isArray(currentValue)
|
||||
? [currentValue]
|
||||
: undefined;
|
||||
if (coercedValue === undefined) continue;
|
||||
@@ -905,17 +918,20 @@ function preserveUnknownRootFields(input: unknown, parsed: unknown): unknown {
|
||||
|
||||
function flattenJsonSchemaIssues(issues: ReadonlyArray<JsonSchemaValidationIssue>): FlatIssue[] {
|
||||
return issues.map(issue => {
|
||||
const unionBranch = issue.fromUnionBranch === true;
|
||||
if (issue.keyword === "additionalProperties") {
|
||||
return {
|
||||
keyword: "unrecognized",
|
||||
instancePath: pathToPointer(issue.path),
|
||||
expectedTypes: [],
|
||||
unionBranch,
|
||||
};
|
||||
}
|
||||
return {
|
||||
keyword: issue.keyword === "type" ? "type" : "other",
|
||||
instancePath: pathToPointer(issue.path),
|
||||
expectedTypes: issue.expectedTypes ?? [],
|
||||
unionBranch,
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
@@ -0,0 +1,164 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { convertAnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic";
|
||||
import type { AssistantMessage, Message, Model, ToolResultMessage, UserMessage } from "@oh-my-pi/pi-ai/types";
|
||||
|
||||
/**
|
||||
* Regression: Anthropic-compatible reasoning endpoints often emit `thinking`
|
||||
* blocks without a first-party Anthropic signature, but still expect those
|
||||
* blocks back as native `type: "thinking"` on continuation. Demoting unsigned
|
||||
* thinking to text strips the reasoning chain and can destabilize follow-up
|
||||
* tool-call argument serialization (the upstream cause behind #2005's `todo`
|
||||
* renderer crash).
|
||||
*
|
||||
* Official Anthropic remains conservative: unsigned thinking is demoted to text
|
||||
* there because the first-party API enforces signature-based integrity.
|
||||
*/
|
||||
function makeModel(overrides: Partial<Model<"anthropic-messages">> = {}): Model<"anthropic-messages"> {
|
||||
return {
|
||||
api: "anthropic-messages",
|
||||
provider: "custom-anthropic",
|
||||
id: "reasoning-model",
|
||||
name: "Reasoning Anthropic-Compatible Model",
|
||||
baseUrl: "https://llm.example.com/anthropic",
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
maxTokens: 8_192,
|
||||
contextWindow: 200_000,
|
||||
reasoning: true,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
function makeUser(text = "continue"): UserMessage {
|
||||
return { role: "user", content: text, timestamp: 0 };
|
||||
}
|
||||
|
||||
function makeAssistantThinking(thinking: string, tail: AssistantMessage["content"][number][] = []): AssistantMessage {
|
||||
return {
|
||||
role: "assistant",
|
||||
content: [{ type: "thinking", thinking, thinkingSignature: "" }, ...tail],
|
||||
api: "anthropic-messages",
|
||||
provider: "custom-anthropic",
|
||||
model: "reasoning-model",
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "toolUse",
|
||||
timestamp: 0,
|
||||
};
|
||||
}
|
||||
|
||||
interface WireThinkingBlock {
|
||||
type: "thinking";
|
||||
thinking: string;
|
||||
signature: string;
|
||||
}
|
||||
interface WireTextBlock {
|
||||
type: "text";
|
||||
text: string;
|
||||
}
|
||||
interface WireToolUseBlock {
|
||||
type: "tool_use";
|
||||
id: string;
|
||||
name: string;
|
||||
input: Record<string, unknown>;
|
||||
}
|
||||
type WireBlock = WireThinkingBlock | WireTextBlock | WireToolUseBlock | { type: string; [key: string]: unknown };
|
||||
|
||||
function assistantWireBlocks(messages: Message[], model: Model<"anthropic-messages">): WireBlock[] {
|
||||
const params = convertAnthropicMessages(messages, model, false);
|
||||
const assistant = params.find(p => p.role === "assistant");
|
||||
return (assistant?.content as WireBlock[] | undefined) ?? [];
|
||||
}
|
||||
|
||||
describe("Anthropic-compatible unsigned thinking replay (#2005)", () => {
|
||||
it("preserves unsigned thinking for non-official reasoning endpoints", () => {
|
||||
const blocks = assistantWireBlocks(
|
||||
[
|
||||
makeUser("solve x"),
|
||||
makeAssistantThinking("plan: read the file, then edit", [{ type: "text", text: "Sure." }]),
|
||||
],
|
||||
makeModel(),
|
||||
);
|
||||
expect(blocks[0]).toEqual({
|
||||
type: "thinking",
|
||||
thinking: "plan: read the file, then edit",
|
||||
signature: "",
|
||||
});
|
||||
expect(blocks[1]).toEqual({ type: "text", text: "Sure." });
|
||||
});
|
||||
|
||||
it("covers the Xiaomi MiMo Anthropic-compatible reporter configuration without provider allowlists", () => {
|
||||
const model = makeModel({
|
||||
provider: "user-custom",
|
||||
id: "mimo-v2.5-pro",
|
||||
name: "MiMo V2.5 Pro (Singapore)",
|
||||
baseUrl: "https://token-plan-sgp.xiaomimimo.com/anthropic",
|
||||
maxTokens: 131_072,
|
||||
contextWindow: 1_048_576,
|
||||
});
|
||||
const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("hidden reasoning")], model);
|
||||
expect(blocks[0]).toEqual({ type: "thinking", thinking: "hidden reasoning", signature: "" });
|
||||
});
|
||||
|
||||
it("preserves legacy known non-signing endpoints even if model.reasoning is false", () => {
|
||||
const model = makeModel({ provider: "custom", baseUrl: "https://api.deepseek.com/v1", reasoning: false });
|
||||
const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("deepseek reasoning")], model);
|
||||
expect(blocks[0]?.type).toBe("thinking");
|
||||
});
|
||||
|
||||
it("still degrades unsigned thinking to text for official Anthropic", () => {
|
||||
const model = makeModel({ provider: "anthropic", baseUrl: "https://api.anthropic.com" });
|
||||
const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model);
|
||||
expect(blocks[0]?.type).toBe("text");
|
||||
expect((blocks[0] as WireTextBlock).text).toBe("internal scratch");
|
||||
});
|
||||
|
||||
it("treats a missing baseUrl as official Anthropic (resolveAnthropicBaseUrl default)", () => {
|
||||
// `isAnthropicApiBaseUrl(undefined) === true` because the actual HTTP
|
||||
// dispatch falls back to https://api.anthropic.com. Same-id custom
|
||||
// overrides that only tweak model metadata (no baseUrl override) must
|
||||
// not regress to native-thinking replay against the first-party API.
|
||||
const model = { ...makeModel(), provider: "anthropic", baseUrl: "" };
|
||||
const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model);
|
||||
expect(blocks[0]?.type).toBe("text");
|
||||
expect((blocks[0] as WireTextBlock).text).toBe("internal scratch");
|
||||
});
|
||||
|
||||
it("still degrades unsigned thinking to text for non-reasoning unknown endpoints", () => {
|
||||
const model = makeModel({ reasoning: false, baseUrl: "https://plain.example.com/anthropic" });
|
||||
const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("scratch")], model);
|
||||
expect(blocks[0]?.type).toBe("text");
|
||||
expect((blocks[0] as WireTextBlock).text).toBe("scratch");
|
||||
});
|
||||
|
||||
it("keeps thinking → tool_use pairing intact across continuation conversion", () => {
|
||||
const toolResult: ToolResultMessage = {
|
||||
role: "toolResult",
|
||||
toolCallId: "toolu_reasoning_1",
|
||||
toolName: "read",
|
||||
content: [{ type: "text", text: "file body" }],
|
||||
isError: false,
|
||||
timestamp: 0,
|
||||
};
|
||||
const model = makeModel();
|
||||
const messages: Message[] = [
|
||||
makeUser("read README"),
|
||||
makeAssistantThinking("I need to call the read tool", [
|
||||
{ type: "toolCall", id: "toolu_reasoning_1", name: "read", arguments: { path: "README.md" } },
|
||||
]),
|
||||
toolResult,
|
||||
];
|
||||
const params = convertAnthropicMessages(messages, model, false);
|
||||
expect(params.map(p => p.role)).toEqual(["user", "assistant", "user"]);
|
||||
const assistantBlocks = params[1].content as WireBlock[];
|
||||
expect(assistantBlocks[0]?.type).toBe("thinking");
|
||||
expect(assistantBlocks[1]?.type).toBe("tool_use");
|
||||
expect((assistantBlocks[1] as WireToolUseBlock).id).toBe("toolu_reasoning_1");
|
||||
});
|
||||
});
|
||||
@@ -1,5 +1,5 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { Effort } from "../src/model-thinking";
|
||||
import { Effort } from "../src/effort";
|
||||
import { encodeResponse, encodeStream, parseRequest } from "../src/providers/openai-responses-server";
|
||||
import type { AssistantMessage } from "../src/types";
|
||||
import { AssistantMessageEventStream } from "../src/utils/event-stream";
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { Effort } from "../src/model-thinking";
|
||||
import { Effort } from "../src/effort";
|
||||
import { encodeStream, formatError, parseRequest } from "../src/providers/pi-native-server";
|
||||
import type {
|
||||
AssistantMessage,
|
||||
|
||||
@@ -0,0 +1,162 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import type { ApiKeyResolveContext } from "@oh-my-pi/pi-ai";
|
||||
import { isApiKeyResolver, isAuthRetryableError, resolveApiKeyOnce, withAuth } from "@oh-my-pi/pi-ai";
|
||||
|
||||
function authError(status = 401): Error & { status: number } {
|
||||
return Object.assign(new Error(`${status} authentication_error`), { status });
|
||||
}
|
||||
|
||||
function usageLimitError(): Error & { status: number } {
|
||||
return Object.assign(new Error("You have hit your ChatGPT usage limit (pro plan). Try again in ~158 min."), {
|
||||
status: 429,
|
||||
});
|
||||
}
|
||||
|
||||
describe("isApiKeyResolver / resolveApiKeyOnce", () => {
|
||||
it("narrows resolver vs static key and resolves the initial value", async () => {
|
||||
expect(isApiKeyResolver("static")).toBe(false);
|
||||
expect(isApiKeyResolver(undefined)).toBe(false);
|
||||
expect(isApiKeyResolver(() => "k")).toBe(true);
|
||||
|
||||
expect(await resolveApiKeyOnce("static")).toBe("static");
|
||||
expect(await resolveApiKeyOnce(undefined)).toBeUndefined();
|
||||
|
||||
let seen: ApiKeyResolveContext | undefined;
|
||||
const resolved = await resolveApiKeyOnce(ctx => {
|
||||
seen = ctx;
|
||||
return "minted";
|
||||
});
|
||||
expect(resolved).toBe("minted");
|
||||
// Initial resolve must look like an initial resolve, not a retry.
|
||||
expect(seen).toEqual({ lastChance: false, error: undefined, signal: undefined });
|
||||
});
|
||||
});
|
||||
|
||||
describe("isAuthRetryableError", () => {
|
||||
it("treats 401 and usage-limit phrasing as retryable, everything else as not", () => {
|
||||
expect(isAuthRetryableError(authError(401))).toBe(true);
|
||||
expect(isAuthRetryableError(usageLimitError())).toBe(true);
|
||||
// A 429 whose body names the *account's* rate limit is rotatable (switch
|
||||
// account), even though it isn't a 401 and isn't phrased "usage limit".
|
||||
expect(
|
||||
isAuthRetryableError(
|
||||
Object.assign(
|
||||
new Error(
|
||||
'429 {"type":"error","error":{"type":"rate_limit_error","message":"This request would exceed your account\'s rate limit. Please try again later."}} retry-after-ms=9779000',
|
||||
),
|
||||
{ status: 429 },
|
||||
),
|
||||
),
|
||||
).toBe(true);
|
||||
// A generic (non-account) 429 rate limit is NOT rotatable — switching
|
||||
// credentials won't help an org/global limit.
|
||||
expect(isAuthRetryableError(Object.assign(new Error("429 too many requests"), { status: 429 }))).toBe(false);
|
||||
expect(isAuthRetryableError("Error: 401 unauthorized")).toBe(true);
|
||||
expect(isAuthRetryableError(authError(403))).toBe(false);
|
||||
expect(isAuthRetryableError(authError(500))).toBe(false);
|
||||
expect(isAuthRetryableError(new Error("network blip"))).toBe(false);
|
||||
expect(isAuthRetryableError(undefined)).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("withAuth", () => {
|
||||
it("runs a single attempt for a static string key (no retry)", async () => {
|
||||
const keys: Array<string | undefined> = [];
|
||||
const result = await withAuth("static-key", async key => {
|
||||
keys.push(key);
|
||||
return `ok:${key}`;
|
||||
});
|
||||
expect(result).toBe("ok:static-key");
|
||||
expect(keys).toEqual(["static-key"]);
|
||||
});
|
||||
|
||||
it("throws when a static key is missing", async () => {
|
||||
await expect(withAuth(undefined, async () => "never", { missingKeyMessage: "no key for foo" })).rejects.toThrow(
|
||||
"no key for foo",
|
||||
);
|
||||
});
|
||||
|
||||
it("refreshes the same account, then switches, in order", async () => {
|
||||
const keys: string[] = [];
|
||||
const contexts: ApiKeyResolveContext[] = [];
|
||||
const result = await withAuth(
|
||||
ctx => {
|
||||
contexts.push(ctx);
|
||||
return ctx.error === undefined ? "k0" : ctx.lastChance ? "k2" : "k1";
|
||||
},
|
||||
async key => {
|
||||
keys.push(key);
|
||||
if (key === "k2") return "success";
|
||||
throw authError();
|
||||
},
|
||||
);
|
||||
expect(result).toBe("success");
|
||||
expect(keys).toEqual(["k0", "k1", "k2"]);
|
||||
expect(contexts.map(ctx => ({ lastChance: ctx.lastChance, hasError: ctx.error !== undefined }))).toEqual([
|
||||
{ lastChance: false, hasError: false },
|
||||
{ lastChance: false, hasError: true },
|
||||
{ lastChance: true, hasError: true },
|
||||
]);
|
||||
});
|
||||
|
||||
it("stops retrying when the resolver returns undefined", async () => {
|
||||
const keys: string[] = [];
|
||||
const original = authError();
|
||||
await expect(
|
||||
withAuth(
|
||||
ctx => (ctx.error === undefined ? "k0" : undefined),
|
||||
async key => {
|
||||
keys.push(key);
|
||||
throw original;
|
||||
},
|
||||
),
|
||||
).rejects.toBe(original);
|
||||
expect(keys).toEqual(["k0"]);
|
||||
});
|
||||
|
||||
it("does not re-attempt when the re-resolved key is unchanged", async () => {
|
||||
const keys: string[] = [];
|
||||
const original = authError();
|
||||
// refresh-same returns the same key (skip), switch returns the same key (skip).
|
||||
await expect(
|
||||
withAuth(
|
||||
() => "same",
|
||||
async key => {
|
||||
keys.push(key);
|
||||
throw original;
|
||||
},
|
||||
),
|
||||
).rejects.toBe(original);
|
||||
expect(keys).toEqual(["same"]);
|
||||
});
|
||||
|
||||
it("propagates non-auth errors without retrying", async () => {
|
||||
const keys: string[] = [];
|
||||
const boom = new Error("network blip");
|
||||
await expect(
|
||||
withAuth(
|
||||
ctx => (ctx.error === undefined ? "k0" : "k1"),
|
||||
async key => {
|
||||
keys.push(key);
|
||||
throw boom;
|
||||
},
|
||||
),
|
||||
).rejects.toBe(boom);
|
||||
expect(keys).toEqual(["k0"]);
|
||||
});
|
||||
|
||||
it("honors a custom isAuthError classifier", async () => {
|
||||
const keys: string[] = [];
|
||||
const result = await withAuth(
|
||||
ctx => (ctx.error === undefined ? "k0" : "k1"),
|
||||
async key => {
|
||||
keys.push(key);
|
||||
if (key === "k0") throw new Error("CUSTOM_RETRY");
|
||||
return "ok";
|
||||
},
|
||||
{ isAuthError: error => error instanceof Error && error.message === "CUSTOM_RETRY" },
|
||||
);
|
||||
expect(result).toBe("ok");
|
||||
expect(keys).toEqual(["k0", "k1"]);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,163 @@
|
||||
import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test";
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { type AuthCredentialStore, AuthStorage, SqliteAuthCredentialStore } from "../src/auth-storage";
|
||||
import { registerOAuthProvider, unregisterOAuthProviders } from "../src/utils/oauth";
|
||||
|
||||
const PROVIDER = "unit-rotate-oauth";
|
||||
const SOURCE = "auth-storage-force-refresh-rotate-test";
|
||||
|
||||
function farExpiry(): number {
|
||||
return Date.now() + 60 * 60_000;
|
||||
}
|
||||
|
||||
function authError(): Error & { status: number } {
|
||||
return Object.assign(new Error("401 authentication_error"), { status: 401 });
|
||||
}
|
||||
|
||||
function usageLimitError(): Error & { status: number } {
|
||||
return Object.assign(new Error("You have hit your ChatGPT usage limit (pro plan). Try again in ~158 min."), {
|
||||
status: 429,
|
||||
});
|
||||
}
|
||||
|
||||
describe("AuthStorage forceRefresh + rotateSessionCredential", () => {
|
||||
let tempDir = "";
|
||||
let store: AuthCredentialStore | undefined;
|
||||
let authStorage: AuthStorage | undefined;
|
||||
|
||||
beforeEach(async () => {
|
||||
tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-rotate-"));
|
||||
store = await SqliteAuthCredentialStore.open(path.join(tempDir, "agent.db"));
|
||||
authStorage = new AuthStorage(store);
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
vi.restoreAllMocks();
|
||||
unregisterOAuthProviders(SOURCE);
|
||||
store?.close();
|
||||
store = undefined;
|
||||
authStorage = undefined;
|
||||
if (tempDir) {
|
||||
await fs.rm(tempDir, { recursive: true, force: true });
|
||||
tempDir = "";
|
||||
}
|
||||
});
|
||||
|
||||
function registerProvider(onRefresh?: () => void): void {
|
||||
registerOAuthProvider({
|
||||
id: PROVIDER,
|
||||
name: "Rotate Unit",
|
||||
sourceId: SOURCE,
|
||||
async login() {
|
||||
return { access: "login", refresh: "login", expires: farExpiry() };
|
||||
},
|
||||
async refreshToken(credentials) {
|
||||
onRefresh?.();
|
||||
return {
|
||||
...credentials,
|
||||
access: "minted-access",
|
||||
refresh: "minted-refresh",
|
||||
expires: farExpiry(),
|
||||
};
|
||||
},
|
||||
getApiKey(credentials) {
|
||||
return credentials.access;
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
test("forceRefresh re-mints a not-yet-expired token; a normal resolve uses the cached token", async () => {
|
||||
if (!authStorage) throw new Error("test setup failed");
|
||||
let refreshCalls = 0;
|
||||
registerProvider(() => {
|
||||
refreshCalls += 1;
|
||||
});
|
||||
await authStorage.set(PROVIDER, [
|
||||
{ type: "oauth", access: "cached-access", refresh: "cached-refresh", expires: farExpiry() },
|
||||
]);
|
||||
|
||||
const cached = await authStorage.getApiKey(PROVIDER, "s-control");
|
||||
expect(cached).toBe("cached-access");
|
||||
expect(refreshCalls).toBe(0);
|
||||
|
||||
const forced = await authStorage.getApiKey(PROVIDER, "s-force", { forceRefresh: true });
|
||||
expect(forced).toBe("minted-access");
|
||||
expect(refreshCalls).toBe(1);
|
||||
|
||||
// The re-minted credential is persisted, so the next plain resolve sees it.
|
||||
const after = await authStorage.getApiKey(PROVIDER, "s-after");
|
||||
expect(after).toBe("minted-access");
|
||||
});
|
||||
|
||||
test("rotateSessionCredential(401) blocks + clears the sticky and rotates to a sibling", async () => {
|
||||
if (!authStorage) throw new Error("test setup failed");
|
||||
registerProvider();
|
||||
await authStorage.set(PROVIDER, [
|
||||
{ type: "oauth", access: "acc-A", refresh: "ref-A", expires: farExpiry() },
|
||||
{ type: "oauth", access: "acc-B", refresh: "ref-B", expires: farExpiry() },
|
||||
]);
|
||||
|
||||
const first = await authStorage.getApiKey(PROVIDER, "sess");
|
||||
expect(["acc-A", "acc-B"]).toContain(first ?? "");
|
||||
|
||||
const usageLimitSpy = vi.spyOn(authStorage, "markUsageLimitReached");
|
||||
const rotated = await authStorage.rotateSessionCredential(PROVIDER, "sess", { error: authError() });
|
||||
|
||||
expect(rotated).toBe(true);
|
||||
// A hard 401 must NOT take the usage-limit code path.
|
||||
expect(usageLimitSpy).not.toHaveBeenCalled();
|
||||
|
||||
const second = await authStorage.getApiKey(PROVIDER, "sess");
|
||||
expect(["acc-A", "acc-B"]).toContain(second ?? "");
|
||||
expect(second).not.toBe(first);
|
||||
});
|
||||
|
||||
test("rotateSessionCredential(usage-limit) delegates to markUsageLimitReached", async () => {
|
||||
if (!authStorage) throw new Error("test setup failed");
|
||||
registerProvider();
|
||||
await authStorage.set(PROVIDER, [
|
||||
{ type: "oauth", access: "acc-A", refresh: "ref-A", expires: farExpiry() },
|
||||
{ type: "oauth", access: "acc-B", refresh: "ref-B", expires: farExpiry() },
|
||||
]);
|
||||
|
||||
const first = await authStorage.getApiKey(PROVIDER, "sess");
|
||||
const usageLimitSpy = vi.spyOn(authStorage, "markUsageLimitReached");
|
||||
|
||||
const rotated = await authStorage.rotateSessionCredential(PROVIDER, "sess", {
|
||||
error: usageLimitError(),
|
||||
});
|
||||
|
||||
expect(rotated).toBe(true);
|
||||
// Usage / account-rate-limit errors route to markUsageLimitReached, which
|
||||
// owns the block duration (default + server usage-report reset) — the
|
||||
// resolver never parses retry-after itself.
|
||||
expect(usageLimitSpy).toHaveBeenCalledTimes(1);
|
||||
expect(usageLimitSpy.mock.calls[0]?.[0]).toBe(PROVIDER);
|
||||
expect(usageLimitSpy.mock.calls[0]?.[1]).toBe("sess");
|
||||
|
||||
const second = await authStorage.getApiKey(PROVIDER, "sess");
|
||||
expect(second).not.toBe(first);
|
||||
});
|
||||
|
||||
test("rotateSessionCredential reports no sibling for a single-credential setup", async () => {
|
||||
if (!authStorage) throw new Error("test setup failed");
|
||||
registerProvider();
|
||||
await authStorage.set(PROVIDER, [
|
||||
{ type: "oauth", access: "only-access", refresh: "only-refresh", expires: farExpiry() },
|
||||
]);
|
||||
|
||||
await authStorage.getApiKey(PROVIDER, "sess");
|
||||
expect(await authStorage.rotateSessionCredential(PROVIDER, "sess", { error: authError() })).toBe(false);
|
||||
});
|
||||
|
||||
test("rotateSessionCredential returns false when the session has no sticky credential", async () => {
|
||||
if (!authStorage) throw new Error("test setup failed");
|
||||
registerProvider();
|
||||
await authStorage.set(PROVIDER, [{ type: "oauth", access: "acc-A", refresh: "ref-A", expires: farExpiry() }]);
|
||||
|
||||
// Never resolved a key for this session → nothing to rotate away from.
|
||||
expect(await authStorage.rotateSessionCredential(PROVIDER, "untouched", { error: authError() })).toBe(false);
|
||||
});
|
||||
});
|
||||
@@ -32,6 +32,43 @@ describe("Duplicate Tool Results Regression", () => {
|
||||
reasoning: true,
|
||||
};
|
||||
|
||||
const makeEvalAssistantMessage = (id: string, timestamp: number): AssistantMessage => ({
|
||||
role: "assistant",
|
||||
content: [{ type: "toolCall", id, name: "eval", arguments: {} }],
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
model: "claude-3-5-sonnet-20241022",
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "toolUse",
|
||||
timestamp,
|
||||
});
|
||||
|
||||
const makeEvalToolResult = (id: string, text: string, timestamp: number): ToolResultMessage => ({
|
||||
role: "toolResult",
|
||||
toolCallId: id,
|
||||
toolName: "eval",
|
||||
content: [{ type: "text", text }],
|
||||
isError: false,
|
||||
timestamp,
|
||||
});
|
||||
|
||||
const getAssistantToolIds = (messages: Message[]): string[] =>
|
||||
messages.flatMap(message =>
|
||||
message.role === "assistant"
|
||||
? message.content.filter((block): block is ToolCall => block.type === "toolCall").map(block => block.id)
|
||||
: [],
|
||||
);
|
||||
|
||||
const getToolResults = (messages: Message[]): ToolResultMessage[] =>
|
||||
messages.filter((message): message is ToolResultMessage => message.role === "toolResult");
|
||||
|
||||
it("should not duplicate tool results for errored messages when results already exist", () => {
|
||||
const toolCallId = "toolu_019xqMTvqWZiTDy8XxmjxrTo";
|
||||
|
||||
@@ -316,6 +353,144 @@ describe("Duplicate Tool Results Regression", () => {
|
||||
expect(result2.length).toBe(1);
|
||||
expect(result3.length).toBe(1);
|
||||
});
|
||||
|
||||
it("deduplicates repeated tool call ids and preserves call/result pairing", () => {
|
||||
const duplicateId = "functions.eval:301";
|
||||
const distinctId = "functions.eval:302";
|
||||
|
||||
const messages: Message[] = [
|
||||
makeEvalAssistantMessage(duplicateId, 1),
|
||||
makeEvalToolResult(duplicateId, "first", 2),
|
||||
makeEvalAssistantMessage(duplicateId, 3),
|
||||
makeEvalToolResult(duplicateId, "second", 4),
|
||||
makeEvalAssistantMessage(duplicateId, 5),
|
||||
makeEvalAssistantMessage(distinctId, 6),
|
||||
makeEvalToolResult(distinctId, "third", 7),
|
||||
];
|
||||
|
||||
const transformed = transformMessages(messages, model);
|
||||
const assistantToolIds = getAssistantToolIds(transformed);
|
||||
const toolResults = getToolResults(transformed);
|
||||
|
||||
expect(assistantToolIds).toEqual([duplicateId, `${duplicateId}_dup1`, `${duplicateId}_dup2`, distinctId]);
|
||||
expect(toolResults.map(result => result.toolCallId)).toEqual([
|
||||
duplicateId,
|
||||
`${duplicateId}_dup1`,
|
||||
`${duplicateId}_dup2`,
|
||||
distinctId,
|
||||
]);
|
||||
expect(toolResults.find(result => result.toolCallId === `${duplicateId}_dup1`)?.content).toEqual([
|
||||
{ type: "text", text: "second" },
|
||||
]);
|
||||
expect(toolResults.find(result => result.toolCallId === `${duplicateId}_dup2`)?.content).toEqual([
|
||||
{ type: "text", text: "No result provided" },
|
||||
]);
|
||||
});
|
||||
|
||||
it("deduplicates repeated ids without colliding with existing generated-looking ids", () => {
|
||||
const duplicateId = "functions.eval:301";
|
||||
const generatedLookingId = `${duplicateId}_dup1`;
|
||||
const messages: Message[] = [
|
||||
makeEvalAssistantMessage(duplicateId, 1),
|
||||
makeEvalToolResult(duplicateId, "first", 2),
|
||||
makeEvalAssistantMessage(generatedLookingId, 3),
|
||||
makeEvalToolResult(generatedLookingId, "already-used", 4),
|
||||
makeEvalAssistantMessage(duplicateId, 5),
|
||||
makeEvalToolResult(duplicateId, "second", 6),
|
||||
];
|
||||
|
||||
const transformed = transformMessages(messages, model);
|
||||
const assistantToolIds = getAssistantToolIds(transformed);
|
||||
const toolResults = getToolResults(transformed);
|
||||
|
||||
expect(assistantToolIds).toEqual([duplicateId, generatedLookingId, `${duplicateId}_dup2`]);
|
||||
expect(toolResults.map(result => result.toolCallId)).toEqual([
|
||||
duplicateId,
|
||||
generatedLookingId,
|
||||
`${duplicateId}_dup2`,
|
||||
]);
|
||||
expect(toolResults.find(result => result.toolCallId === `${duplicateId}_dup2`)?.content).toEqual([
|
||||
{ type: "text", text: "second" },
|
||||
]);
|
||||
});
|
||||
|
||||
it("preserves delayed duplicate tool results across message gaps", () => {
|
||||
const duplicateId = "functions.eval:301";
|
||||
const developerMessage: DeveloperMessage = { role: "developer", content: "handoff summary", timestamp: 4 };
|
||||
const messages: Message[] = [
|
||||
makeEvalAssistantMessage(duplicateId, 1),
|
||||
makeEvalToolResult(duplicateId, "first", 2),
|
||||
makeEvalAssistantMessage(duplicateId, 3),
|
||||
developerMessage,
|
||||
makeEvalToolResult(duplicateId, "second", 5),
|
||||
];
|
||||
|
||||
const transformed = transformMessages(messages, model);
|
||||
const toolResults = getToolResults(transformed);
|
||||
|
||||
expect(getAssistantToolIds(transformed)).toEqual([duplicateId, `${duplicateId}_dup1`]);
|
||||
expect(toolResults.map(result => result.toolCallId)).toEqual([duplicateId, `${duplicateId}_dup1`]);
|
||||
expect(toolResults.find(result => result.toolCallId === `${duplicateId}_dup1`)?.content).toEqual([
|
||||
{ type: "text", text: "second" },
|
||||
]);
|
||||
});
|
||||
|
||||
it("routes the late result to the most recent duplicate call when a new turn re-emits the id across a gap", () => {
|
||||
const duplicateId = "functions.eval:301";
|
||||
const developerMessage: DeveloperMessage = { role: "developer", content: "handoff summary", timestamp: 4 };
|
||||
const messages: Message[] = [
|
||||
makeEvalAssistantMessage(duplicateId, 1),
|
||||
makeEvalToolResult(duplicateId, "first", 2),
|
||||
makeEvalAssistantMessage(duplicateId, 3),
|
||||
developerMessage,
|
||||
makeEvalAssistantMessage(duplicateId, 5),
|
||||
makeEvalToolResult(duplicateId, "second", 6),
|
||||
];
|
||||
|
||||
const transformed = transformMessages(messages, model);
|
||||
const toolResults = getToolResults(transformed);
|
||||
|
||||
expect(getAssistantToolIds(transformed)).toEqual([duplicateId, `${duplicateId}_dup1`, `${duplicateId}_dup2`]);
|
||||
expect(toolResults.map(result => result.toolCallId)).toEqual([
|
||||
duplicateId,
|
||||
`${duplicateId}_dup1`,
|
||||
`${duplicateId}_dup2`,
|
||||
]);
|
||||
expect(toolResults.find(result => result.toolCallId === `${duplicateId}_dup1`)?.content).toEqual([
|
||||
{ type: "text", text: "No result provided" },
|
||||
]);
|
||||
expect(toolResults.find(result => result.toolCallId === `${duplicateId}_dup2`)?.content).toEqual([
|
||||
{ type: "text", text: "second" },
|
||||
]);
|
||||
});
|
||||
|
||||
it("keeps duplicate-id rewrites within the 64-char tool-call id limit", () => {
|
||||
const baseId = `toolu_${"a".repeat(58)}`;
|
||||
expect(baseId.length).toBe(64);
|
||||
const messages: Message[] = [
|
||||
makeEvalAssistantMessage(baseId, 1),
|
||||
makeEvalToolResult(baseId, "first", 2),
|
||||
makeEvalAssistantMessage(baseId, 3),
|
||||
makeEvalToolResult(baseId, "second", 4),
|
||||
];
|
||||
|
||||
const transformed = transformMessages(messages, model);
|
||||
const assistantToolIds = getAssistantToolIds(transformed);
|
||||
const toolResults = getToolResults(transformed);
|
||||
|
||||
expect(assistantToolIds).toHaveLength(2);
|
||||
for (const id of assistantToolIds) {
|
||||
expect(id.length).toBeLessThanOrEqual(64);
|
||||
expect(id).toMatch(/^[A-Za-z0-9_-]+$/);
|
||||
}
|
||||
const rewrittenId = assistantToolIds[1];
|
||||
expect(rewrittenId).not.toBe(baseId);
|
||||
expect(rewrittenId.endsWith("_dup1")).toBe(true);
|
||||
expect(toolResults.map(result => result.toolCallId)).toEqual([baseId, rewrittenId]);
|
||||
expect(toolResults.find(result => result.toolCallId === rewrittenId)?.content).toEqual([
|
||||
{ type: "text", text: "second" },
|
||||
]);
|
||||
});
|
||||
});
|
||||
|
||||
/**
|
||||
|
||||
@@ -2,8 +2,8 @@ import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { Effort } from "../src/effort";
|
||||
import { createModelManager } from "../src/model-manager";
|
||||
import { Effort } from "../src/model-thinking";
|
||||
import { getBundledModel } from "../src/models";
|
||||
import { githubCopilotModelManagerOptions } from "../src/provider-models/openai-compat";
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { Effort } from "../src/model-thinking";
|
||||
import { Effort } from "../src/effort";
|
||||
import { getBundledModel } from "../src/models";
|
||||
import { streamAnthropic } from "../src/providers/anthropic";
|
||||
import { streamOpenAIResponses } from "../src/providers/openai-responses";
|
||||
|
||||
@@ -0,0 +1,196 @@
|
||||
/**
|
||||
* Antigravity usage provider contract tests. The merge logic
|
||||
* deduplicates per-model quota entries by (tier, windowId),
|
||||
* preserves reset times when bar data and window data come from
|
||||
* different model entries, and handles mixed-case tier names.
|
||||
*/
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import type { UsageFetchContext, UsageFetchParams } from "../src/usage";
|
||||
import { antigravityUsageProvider } from "../src/usage/google-antigravity";
|
||||
|
||||
const accessTokenFixture = (() => {
|
||||
const header = Buffer.from(JSON.stringify({ alg: "none", typ: "JWT" })).toString("base64url");
|
||||
const body = Buffer.from(JSON.stringify({ sub: "user-fixture" })).toString("base64url");
|
||||
return `${header}.${body}.sig`;
|
||||
})();
|
||||
|
||||
function makeCredential(overrides?: Partial<UsageFetchParams["credential"]>) {
|
||||
return {
|
||||
type: "oauth" as const,
|
||||
accessToken: accessTokenFixture,
|
||||
refreshToken: "refresh-fixture",
|
||||
expiresAt: Date.now() + 3600_000,
|
||||
projectId: "test-project",
|
||||
email: "test@example.com",
|
||||
accountId: "acct-1",
|
||||
...overrides,
|
||||
} satisfies UsageFetchParams["credential"];
|
||||
}
|
||||
|
||||
function fakeFetch(json: unknown): typeof fetch {
|
||||
const fn = async () =>
|
||||
new Response(JSON.stringify(json), {
|
||||
status: 200,
|
||||
headers: { "content-type": "application/json" },
|
||||
});
|
||||
return fn as unknown as typeof fetch;
|
||||
}
|
||||
|
||||
function makeCtx(fetchImpl?: typeof fetch): UsageFetchContext {
|
||||
return { fetch: fetchImpl ?? fakeFetch({}) };
|
||||
}
|
||||
|
||||
// ── helpers ──────────────────────────────────────────────────────────
|
||||
|
||||
function makeApiModel(
|
||||
displayName: string,
|
||||
quota: { remainingFraction?: number; resetTime?: string; tier?: string; windowId?: string },
|
||||
) {
|
||||
return {
|
||||
displayName,
|
||||
quotaInfo: {
|
||||
remainingFraction: quota.remainingFraction,
|
||||
resetTime: quota.resetTime,
|
||||
tier: quota.tier,
|
||||
windowId: quota.windowId,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// ── tests ────────────────────────────────────────────────────────────
|
||||
|
||||
describe("antigravity usage provider", () => {
|
||||
it("merges two models with same tier into one limit", async () => {
|
||||
const payload = {
|
||||
models: {
|
||||
modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "premium" }),
|
||||
modelB: makeApiModel("Model B", { remainingFraction: 0.5, tier: "premium" }),
|
||||
},
|
||||
};
|
||||
const report = await antigravityUsageProvider.fetchUsage!(
|
||||
{ provider: "google-antigravity", credential: makeCredential(), signal: undefined },
|
||||
makeCtx(fakeFetch(payload)),
|
||||
);
|
||||
expect(report).not.toBeNull();
|
||||
expect(report!.limits.length).toBe(1);
|
||||
});
|
||||
|
||||
it("keeps the worst remainingFraction when merging same tier", async () => {
|
||||
const payload = {
|
||||
models: {
|
||||
modelA: makeApiModel("Model A", { remainingFraction: 0.1, tier: "premium" }),
|
||||
modelB: makeApiModel("Model B", { remainingFraction: 0.8, tier: "premium" }),
|
||||
},
|
||||
};
|
||||
const report = await antigravityUsageProvider.fetchUsage!(
|
||||
{ provider: "google-antigravity", credential: makeCredential(), signal: undefined },
|
||||
makeCtx(fakeFetch(payload)),
|
||||
);
|
||||
expect(report!.limits.length).toBe(1);
|
||||
expect(report!.limits[0]!.amount.remainingFraction).toBe(0.1);
|
||||
});
|
||||
|
||||
it("merges mixed-case tier names under lowercased key", async () => {
|
||||
const payload = {
|
||||
models: {
|
||||
modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "Default" }),
|
||||
modelB: makeApiModel("Model B", { remainingFraction: 0.6, tier: "default" }),
|
||||
},
|
||||
};
|
||||
const report = await antigravityUsageProvider.fetchUsage!(
|
||||
{ provider: "google-antigravity", credential: makeCredential(), signal: undefined },
|
||||
makeCtx(fakeFetch(payload)),
|
||||
);
|
||||
expect(report!.limits.length).toBe(1);
|
||||
});
|
||||
|
||||
it("preserves reset time from an entry even when bar data comes from another", async () => {
|
||||
const now = Date.now();
|
||||
const resetTime = new Date(now + 4 * 3600_000).toISOString();
|
||||
const payload = {
|
||||
models: {
|
||||
modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "default" }),
|
||||
modelB: makeApiModel("Model B", { remainingFraction: undefined, tier: "default", resetTime }),
|
||||
},
|
||||
};
|
||||
const report = await antigravityUsageProvider.fetchUsage!(
|
||||
{ provider: "google-antigravity", credential: makeCredential(), signal: undefined },
|
||||
makeCtx(fakeFetch(payload)),
|
||||
);
|
||||
expect(report!.limits.length).toBe(1);
|
||||
expect(report!.limits[0]!.amount.remainingFraction).toBe(0.3);
|
||||
expect(report!.limits[0]!.window).toBeDefined();
|
||||
expect(report!.limits[0]!.window!.resetsAt).toBeGreaterThan(now);
|
||||
});
|
||||
|
||||
it("separates models with different windowIds in the same tier", async () => {
|
||||
const now = Date.now();
|
||||
const t1 = new Date(now + 5 * 3600_000).toISOString();
|
||||
const t2 = new Date(now + 24 * 3600_000).toISOString();
|
||||
const payload = {
|
||||
models: {
|
||||
modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "premium", windowId: "5h", resetTime: t1 }),
|
||||
modelB: makeApiModel("Model B", {
|
||||
remainingFraction: 0.7,
|
||||
tier: "premium",
|
||||
windowId: "daily",
|
||||
resetTime: t2,
|
||||
}),
|
||||
},
|
||||
};
|
||||
const report = await antigravityUsageProvider.fetchUsage!(
|
||||
{ provider: "google-antigravity", credential: makeCredential(), signal: undefined },
|
||||
makeCtx(fakeFetch(payload)),
|
||||
);
|
||||
expect(report!.limits.length).toBe(2);
|
||||
});
|
||||
|
||||
it("includes email and projectId in report metadata", async () => {
|
||||
const payload = { models: { m: makeApiModel("M", { remainingFraction: 1 }) } };
|
||||
const report = await antigravityUsageProvider.fetchUsage!(
|
||||
{
|
||||
provider: "google-antigravity",
|
||||
credential: makeCredential({ email: "user@example.com", projectId: "proj-1" }),
|
||||
signal: undefined,
|
||||
},
|
||||
makeCtx(fakeFetch(payload)),
|
||||
);
|
||||
expect(report!.metadata?.email).toBe("user@example.com");
|
||||
expect(report!.metadata?.projectId).toBe("proj-1");
|
||||
});
|
||||
|
||||
it("does not include email when credential has none", async () => {
|
||||
const payload = { models: { m: makeApiModel("M", { remainingFraction: 1 }) } };
|
||||
const report = await antigravityUsageProvider.fetchUsage!(
|
||||
{ provider: "google-antigravity", credential: makeCredential({ email: undefined }), signal: undefined },
|
||||
makeCtx(fakeFetch(payload)),
|
||||
);
|
||||
expect(report!.metadata?.email).toBeUndefined();
|
||||
});
|
||||
|
||||
it("sorts limits by remainingFraction ascending (worst first)", async () => {
|
||||
const payload = {
|
||||
models: {
|
||||
modelA: makeApiModel("Model A", { remainingFraction: 0.9, tier: "high" }),
|
||||
modelB: makeApiModel("Model B", { remainingFraction: 0.2, tier: "low" }),
|
||||
modelC: makeApiModel("Model C", { remainingFraction: 0.5, tier: "mid" }),
|
||||
},
|
||||
};
|
||||
const report = await antigravityUsageProvider.fetchUsage!(
|
||||
{ provider: "google-antigravity", credential: makeCredential(), signal: undefined },
|
||||
makeCtx(fakeFetch(payload)),
|
||||
);
|
||||
expect(report!.limits.length).toBe(3);
|
||||
expect(report!.limits[0]!.amount.remainingFraction).toBe(0.2);
|
||||
expect(report!.limits[1]!.amount.remainingFraction).toBe(0.5);
|
||||
expect(report!.limits[2]!.amount.remainingFraction).toBe(0.9);
|
||||
});
|
||||
|
||||
it("returns null when credential has no projectId", async () => {
|
||||
const report = await antigravityUsageProvider.fetchUsage!(
|
||||
{ provider: "google-antigravity", credential: makeCredential({ projectId: undefined }), signal: undefined },
|
||||
makeCtx(),
|
||||
);
|
||||
expect(report).toBeNull();
|
||||
});
|
||||
});
|
||||
@@ -91,6 +91,19 @@ describe("issue #1207 — DeepSeek V4 keeps reasoning with tools", () => {
|
||||
expect(body.max_completion_tokens).toBeUndefined();
|
||||
});
|
||||
|
||||
it("does not mix Fireworks DeepSeek effort with the native thinking toggle", async () => {
|
||||
const model = getBundledModel("fireworks", "deepseek-v4-pro") as Model<"openai-completions">;
|
||||
const compat = resolveOpenAICompat(model);
|
||||
const body = await capturePayload(model);
|
||||
|
||||
expect(compat.extraBody).toBeUndefined();
|
||||
expect(body.tools).toBeDefined();
|
||||
expect(body.tool_choice).toBeUndefined();
|
||||
expect(body.reasoning_effort).toBe("high");
|
||||
expect(body.thinking).toBeUndefined();
|
||||
expect(body.max_tokens).toBe(123);
|
||||
});
|
||||
|
||||
it("preserves OpenRouter reasoning when tool_choice auto is present", async () => {
|
||||
const model = getBundledModel("openrouter", "deepseek/deepseek-v4-flash") as Model<"openai-completions">;
|
||||
const compat = detectOpenAICompat(model);
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { afterAll, beforeAll, describe, expect, it } from "bun:test";
|
||||
import { Effort } from "../src/model-thinking";
|
||||
import { Effort } from "../src/effort";
|
||||
import { streamBedrock } from "../src/providers/amazon-bedrock";
|
||||
import type { Context, Model } from "../src/types";
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { Effort } from "../src/model-thinking";
|
||||
import { Effort } from "../src/effort";
|
||||
import { streamAnthropic } from "../src/providers/anthropic";
|
||||
import type { Context, Model, Tool } from "../src/types";
|
||||
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { afterEach, describe, expect, it } from "bun:test";
|
||||
import { Effort, getSupportedEfforts } from "../src/model-thinking";
|
||||
import { Effort } from "../src/effort";
|
||||
import { getSupportedEfforts } from "../src/model-thinking";
|
||||
import { streamOpenAICompletions } from "../src/providers/openai-completions";
|
||||
import type { Context, Model } from "../src/types";
|
||||
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { Effort } from "@oh-my-pi/pi-ai/effort";
|
||||
import {
|
||||
applyGeneratedModelPolicies,
|
||||
clampThinkingLevelForModel,
|
||||
Effort,
|
||||
enrichModelThinking,
|
||||
linkOpenAIPromotionTargets,
|
||||
mapEffortToAnthropicAdaptiveEffort,
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import { Effort } from "../src/model-thinking";
|
||||
import { Effort } from "../src/effort";
|
||||
import { nanoGptModelManagerOptions } from "../src/provider-models/openai-compat";
|
||||
|
||||
const originalFetch = global.fetch;
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { afterEach, describe, expect, test, vi } from "bun:test";
|
||||
import { Effort } from "../src/model-thinking";
|
||||
import { Effort } from "../src/effort";
|
||||
import { ollamaModelManagerOptions } from "../src/provider-models/openai-compat";
|
||||
import { streamOllama } from "../src/providers/ollama";
|
||||
import type { Context, Model, Tool } from "../src/types";
|
||||
|
||||
@@ -71,6 +71,62 @@ describe("openai-codex request transformer", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("openai-codex orphan tool-call repair", () => {
|
||||
it("synthesizes a function_call_output for a function_call with no result", async () => {
|
||||
const body: RequestBody = {
|
||||
model: "gpt-5.1-codex",
|
||||
input: [
|
||||
{ type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] },
|
||||
{ type: "function_call", call_id: "call_orphan", name: "read", arguments: "{}" },
|
||||
{ type: "message", role: "user", content: [{ type: "input_text", text: "next" }] },
|
||||
],
|
||||
};
|
||||
|
||||
const transformed = await transformRequestBody(body, createCodexModel(body.model), {});
|
||||
const input = transformed.input || [];
|
||||
|
||||
const callIndex = input.findIndex(item => item.type === "function_call" && item.call_id === "call_orphan");
|
||||
expect(callIndex).toBeGreaterThanOrEqual(0);
|
||||
// The synthesized output sits immediately after the orphan call.
|
||||
const output = input[callIndex + 1];
|
||||
expect(output?.type).toBe("function_call_output");
|
||||
expect(output?.call_id).toBe("call_orphan");
|
||||
expect(typeof output?.output).toBe("string");
|
||||
expect(output?.output as string).toMatch(/interrupted/i);
|
||||
});
|
||||
|
||||
it("leaves a paired function_call untouched", async () => {
|
||||
const body: RequestBody = {
|
||||
model: "gpt-5.1-codex",
|
||||
input: [
|
||||
{ type: "function_call", call_id: "call_paired", name: "read", arguments: "{}" },
|
||||
{ type: "function_call_output", call_id: "call_paired", output: "real result" },
|
||||
],
|
||||
};
|
||||
|
||||
const transformed = await transformRequestBody(body, createCodexModel(body.model), {});
|
||||
const input = transformed.input || [];
|
||||
|
||||
const outputs = input.filter(item => item.type === "function_call_output" && item.call_id === "call_paired");
|
||||
expect(outputs).toHaveLength(1);
|
||||
expect(outputs[0]?.output).toBe("real result");
|
||||
});
|
||||
|
||||
it("synthesizes a custom_tool_call_output for an orphan custom_tool_call", async () => {
|
||||
const body: RequestBody = {
|
||||
model: "gpt-5.1-codex",
|
||||
input: [{ type: "custom_tool_call", call_id: "call_custom", name: "apply_patch" }],
|
||||
};
|
||||
|
||||
const transformed = await transformRequestBody(body, createCodexModel(body.model), {});
|
||||
const input = transformed.input || [];
|
||||
|
||||
const output = input.find(item => item.type === "custom_tool_call_output" && item.call_id === "call_custom");
|
||||
expect(output).toBeDefined();
|
||||
expect(output?.output as string).toMatch(/interrupted/i);
|
||||
});
|
||||
});
|
||||
|
||||
describe("openai-codex reasoning effort validation", () => {
|
||||
it("rejects gpt-5.1 xhigh when metadata does not list it", async () => {
|
||||
const body: RequestBody = { model: "gpt-5.1", input: [] };
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { afterEach, describe, expect, it } from "bun:test";
|
||||
import { Effort } from "../src/model-thinking";
|
||||
import { Effort } from "../src/effort";
|
||||
import { streamOpenAICompletions } from "../src/providers/openai-completions";
|
||||
import type { Context, Model } from "../src/types";
|
||||
|
||||
|
||||
@@ -0,0 +1,69 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import {
|
||||
repairOrphanResponsesToolCalls,
|
||||
repairOrphanResponsesToolOutputs,
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
|
||||
import type { ResponseInput } from "openai/resources/responses/responses";
|
||||
|
||||
describe("repairOrphanResponsesToolCalls", () => {
|
||||
it("appends a synthetic function_call_output after a call with no result", () => {
|
||||
const input: ResponseInput = [
|
||||
{ type: "function_call", call_id: "call_a", name: "read", arguments: "{}" },
|
||||
{ role: "user", content: [{ type: "input_text", text: "continue" }] },
|
||||
];
|
||||
|
||||
const repaired = repairOrphanResponsesToolCalls(input);
|
||||
const callIndex = repaired.findIndex(
|
||||
item =>
|
||||
(item as { type?: string }).type === "function_call" && (item as { call_id?: string }).call_id === "call_a",
|
||||
);
|
||||
const output = repaired[callIndex + 1] as { type?: string; call_id?: string; output?: unknown };
|
||||
expect(output.type).toBe("function_call_output");
|
||||
expect(output.call_id).toBe("call_a");
|
||||
expect(output.output).toMatch(/interrupted/i);
|
||||
});
|
||||
|
||||
it("uses custom_tool_call_output for an orphan custom_tool_call", () => {
|
||||
const input: ResponseInput = [
|
||||
{ type: "custom_tool_call", call_id: "call_c", name: "apply_patch", input: "patch" } as ResponseInput[number],
|
||||
];
|
||||
|
||||
const repaired = repairOrphanResponsesToolCalls(input);
|
||||
const output = repaired.find(item => (item as { type?: string }).type === "custom_tool_call_output") as
|
||||
| { call_id?: string }
|
||||
| undefined;
|
||||
expect(output?.call_id).toBe("call_c");
|
||||
});
|
||||
|
||||
it("returns the input unchanged when every call is paired", () => {
|
||||
const input: ResponseInput = [
|
||||
{ type: "function_call", call_id: "call_a", name: "read", arguments: "{}" },
|
||||
{ type: "function_call_output", call_id: "call_a", output: "ok" } as ResponseInput[number],
|
||||
];
|
||||
|
||||
const repaired = repairOrphanResponsesToolCalls(input);
|
||||
expect(repaired).toBe(input);
|
||||
});
|
||||
|
||||
it("composes with output repair so a tree-branch snapshot stays API-valid", () => {
|
||||
// Branching to a node that ends on a tool call drops the result child:
|
||||
// the assistant turn keeps the call, but no matching output remains.
|
||||
const input: ResponseInput = [
|
||||
{ role: "user", content: [{ type: "input_text", text: "do it" }] },
|
||||
{ type: "function_call", call_id: "call_x", name: "bash", arguments: "{}" },
|
||||
];
|
||||
|
||||
const repaired = repairOrphanResponsesToolCalls(repairOrphanResponsesToolOutputs(input));
|
||||
const callIds = new Set(
|
||||
repaired
|
||||
.filter(i => (i as { type?: string }).type === "function_call")
|
||||
.map(i => (i as { call_id: string }).call_id),
|
||||
);
|
||||
const outputIds = new Set(
|
||||
repaired
|
||||
.filter(i => (i as { type?: string }).type === "function_call_output")
|
||||
.map(i => (i as { call_id: string }).call_id),
|
||||
);
|
||||
for (const id of callIds) expect(outputIds.has(id)).toBe(true);
|
||||
});
|
||||
});
|
||||
@@ -260,4 +260,79 @@ describe("processResponsesStream: parallel function_call items", () => {
|
||||
expect(byCallId.get("call_a")?.toolCall.arguments).toEqual({ command: "printf a" });
|
||||
expect(byCallId.get("call_b")?.toolCall.arguments).toEqual({ command: "printf b" });
|
||||
});
|
||||
|
||||
test("routes deltas by item.call_id when llama.cpp omits item.id and output_index (issue #2015)", async () => {
|
||||
// llama.cpp's `to_json_oaicompat_resp` (tools/server/server-task.cpp) emits a
|
||||
// function_call's `output_item.added` with only `item.call_id` — no `item.id`,
|
||||
// no `output_index`. The matching `function_call_arguments.delta` then carries
|
||||
// `item_id: "fc_<call_id>"` and again no `output_index`. Without secondary
|
||||
// indexing on `call_id`, `processResponsesStream`'s lookup map stays empty and
|
||||
// every delta lands on the trailing block, leaving earlier calls with empty
|
||||
// arguments (= `{}`) — the read tool then rejects them with
|
||||
// `path: Invalid input: expected string, received undefined`.
|
||||
const output = makeOutput();
|
||||
const emitted: EmittedEvent[] = [];
|
||||
const stream = { push: (e: unknown) => emitted.push(e as EmittedEvent), end: () => {} } as never;
|
||||
|
||||
const argsA = JSON.stringify({ path: "a.txt" });
|
||||
const argsB = JSON.stringify({ path: "b.txt" });
|
||||
const argsC = JSON.stringify({ path: "c.txt" });
|
||||
|
||||
await processResponsesStream(
|
||||
makeStream([
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
item: { type: "function_call", call_id: "fc_a", name: "read", arguments: "" },
|
||||
},
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
item: { type: "function_call", call_id: "fc_b", name: "read", arguments: "" },
|
||||
},
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
item: { type: "function_call", call_id: "fc_c", name: "read", arguments: "" },
|
||||
},
|
||||
{ type: "response.function_call_arguments.delta", item_id: "fc_a", delta: argsA },
|
||||
{ type: "response.function_call_arguments.delta", item_id: "fc_b", delta: argsB },
|
||||
{ type: "response.function_call_arguments.delta", item_id: "fc_c", delta: argsC },
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
item: { type: "function_call", call_id: "fc_a", name: "read", arguments: argsA },
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
item: { type: "function_call", call_id: "fc_b", name: "read", arguments: argsB },
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
item: { type: "function_call", call_id: "fc_c", name: "read", arguments: argsC },
|
||||
},
|
||||
]),
|
||||
output,
|
||||
stream,
|
||||
makeModel(),
|
||||
);
|
||||
|
||||
expect(output.content).toHaveLength(3);
|
||||
const [a, b, c] = output.content;
|
||||
if (a?.type !== "toolCall" || b?.type !== "toolCall" || c?.type !== "toolCall") {
|
||||
throw new Error("expected toolCalls");
|
||||
}
|
||||
expect(a.arguments).toEqual({ path: "a.txt" });
|
||||
expect(b.arguments).toEqual({ path: "b.txt" });
|
||||
expect(c.arguments).toEqual({ path: "c.txt" });
|
||||
|
||||
const ends = emitted.filter(e => e.type === "toolcall_end") as Array<{
|
||||
toolCall: { id: string; arguments: Record<string, unknown> };
|
||||
contentIndex: number;
|
||||
}>;
|
||||
expect(ends).toHaveLength(3);
|
||||
const byCallId = new Map(ends.map(e => [e.toolCall.id.split("|")[0], e]));
|
||||
expect(byCallId.get("fc_a")?.toolCall.arguments).toEqual({ path: "a.txt" });
|
||||
expect(byCallId.get("fc_b")?.toolCall.arguments).toEqual({ path: "b.txt" });
|
||||
expect(byCallId.get("fc_c")?.toolCall.arguments).toEqual({ path: "c.txt" });
|
||||
expect(byCallId.get("fc_a")?.contentIndex).toBe(0);
|
||||
expect(byCallId.get("fc_b")?.contentIndex).toBe(1);
|
||||
expect(byCallId.get("fc_c")?.contentIndex).toBe(2);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -728,6 +728,18 @@ describe("stripResidualCombiners", () => {
|
||||
expect(normalized.anyOf).toBeUndefined();
|
||||
expect(normalized.oneOf).toBeUndefined();
|
||||
});
|
||||
|
||||
it("drops array-only keys when mixed-type collapse picks string from anyOf fixpoint", () => {
|
||||
const stripped = stripResidualCombiners({
|
||||
anyOf: [{ type: "string" }, { type: "array", items: { type: "string" } }],
|
||||
description: "pr number, url, or branch",
|
||||
}) as Record<string, unknown>;
|
||||
|
||||
expect(stripped.type).toBe("string");
|
||||
expect(stripped.items).toBeUndefined();
|
||||
expect(stripped.anyOf).toBeUndefined();
|
||||
expect(stripped.description).toBe("pr number, url, or branch");
|
||||
});
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -952,6 +964,36 @@ describe("normalizeSchemaForCCA", () => {
|
||||
properties: {},
|
||||
});
|
||||
});
|
||||
|
||||
it("strips array-only keys when mixed-type collapse picks a non-array type", () => {
|
||||
// Regression: anyOf [{type:"string"}, {type:"array", items:{type:"string"}}]
|
||||
// collapsed to {type:"string", items:{type:"string"}} which is invalid.
|
||||
// The fix filters mergedVariantFields against the chosen type's allowed keys.
|
||||
const normalized = normalizeSchemaForCCA({
|
||||
anyOf: [{ type: "string" }, { type: "array", items: { type: "string" } }],
|
||||
description: "pr number, url, or branch",
|
||||
});
|
||||
|
||||
expect(normalized).toEqual({
|
||||
type: "string",
|
||||
description: "pr number, url, or branch",
|
||||
});
|
||||
});
|
||||
|
||||
it("strips sibling type-specific keys copied from parent when mixed-type collapse picks opposing type", () => {
|
||||
// Edge case: parent has a sibling `items` outside the anyOf,
|
||||
// and the chosen type is string. The sibling must be stripped.
|
||||
const normalized = normalizeSchemaForCCA({
|
||||
anyOf: [{ type: "string" }, { type: "array", items: { type: "number" } }],
|
||||
items: { type: "string" },
|
||||
description: "pr number, url, or branch",
|
||||
});
|
||||
|
||||
expect(normalized).toEqual({
|
||||
type: "string",
|
||||
description: "pr number, url, or branch",
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { afterEach, describe, expect, it } from "bun:test";
|
||||
import type { ApiKeyResolveContext } from "@oh-my-pi/pi-ai";
|
||||
import { registerCustomApi, unregisterCustomApis } from "@oh-my-pi/pi-ai";
|
||||
import { streamSimple } from "@oh-my-pi/pi-ai/stream";
|
||||
import type { Api, AssistantMessage, Context, Model, SimpleStreamOptions, Usage } from "@oh-my-pi/pi-ai/types";
|
||||
@@ -25,9 +26,9 @@ function assistant(content: string[] = []): AssistantMessage {
|
||||
api: API,
|
||||
provider: "test-provider",
|
||||
model: "test-model",
|
||||
usage: usage(),
|
||||
timestamp: 1,
|
||||
stopReason: "stop",
|
||||
timestamp: Date.now(),
|
||||
usage: usage(),
|
||||
};
|
||||
}
|
||||
|
||||
@@ -39,19 +40,21 @@ function authError(): Error & { status: number } {
|
||||
return Object.assign(new Error("401 authentication_error"), { status: 401 });
|
||||
}
|
||||
|
||||
function usageLimitError(): Error & { status: number } {
|
||||
return Object.assign(new Error("You have hit your ChatGPT usage limit (pro plan). Try again in ~158 min."), {
|
||||
status: 429,
|
||||
});
|
||||
}
|
||||
|
||||
function model(): Model<Api> {
|
||||
return {
|
||||
id: "test-model",
|
||||
name: "test-model",
|
||||
name: "Test Model",
|
||||
api: API,
|
||||
provider: "test-provider",
|
||||
baseUrl: "mock://",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 1024,
|
||||
maxTokens: 1024,
|
||||
};
|
||||
contextWindow: 1000,
|
||||
maxTokens: 100,
|
||||
} as Model<Api>;
|
||||
}
|
||||
|
||||
const context: Context = {
|
||||
@@ -59,61 +62,64 @@ const context: Context = {
|
||||
messages: [{ role: "user", content: "hello", timestamp: 1 }],
|
||||
};
|
||||
|
||||
describe("streamSimple auth retry", () => {
|
||||
/** Records the static key each inner attempt actually received. */
|
||||
function pushKey(keys: unknown[], options?: SimpleStreamOptions): void {
|
||||
keys.push(options?.apiKey);
|
||||
}
|
||||
|
||||
function ok(stream: AssistantMessageEventStream): void {
|
||||
const message = assistant(["ok"]);
|
||||
stream.push({ type: "start", partial: message });
|
||||
stream.push({ type: "done", reason: "stop", message });
|
||||
}
|
||||
|
||||
describe("streamSimple resolver auth retry", () => {
|
||||
afterEach(() => {
|
||||
unregisterCustomApis(SOURCE_ID);
|
||||
});
|
||||
|
||||
it("retries once with a fresh key when 401 happens before the first event", async () => {
|
||||
const keys: Array<string | undefined> = [];
|
||||
let authCalls = 0;
|
||||
it("retries with a refreshed key when a 401 is thrown before the first event", async () => {
|
||||
const keys: unknown[] = [];
|
||||
const contexts: ApiKeyResolveContext[] = [];
|
||||
registerCustomApi(
|
||||
API,
|
||||
(_model: Model<Api>, _context: Context, options?: SimpleStreamOptions) => {
|
||||
keys.push(options?.apiKey);
|
||||
pushKey(keys, options);
|
||||
const stream = new AssistantMessageEventStream();
|
||||
queueMicrotask(() => {
|
||||
if (keys.length === 1) {
|
||||
stream.fail(authError());
|
||||
return;
|
||||
}
|
||||
const message = assistant(["ok"]);
|
||||
stream.push({ type: "start", partial: message });
|
||||
stream.push({ type: "done", reason: "stop", message });
|
||||
});
|
||||
queueMicrotask(() => (keys.length === 1 ? stream.fail(authError()) : ok(stream)));
|
||||
return stream;
|
||||
},
|
||||
SOURCE_ID,
|
||||
);
|
||||
|
||||
const stream = streamSimple(model(), context, {
|
||||
apiKey: "old-key",
|
||||
onAuthError: async (provider, oldKey, error) => {
|
||||
authCalls += 1;
|
||||
expect(provider).toBe("test-provider");
|
||||
expect(oldKey).toBe("old-key");
|
||||
expect((error as { status?: number }).status).toBe(401);
|
||||
return "new-key";
|
||||
apiKey: async ctx => {
|
||||
contexts.push(ctx);
|
||||
return ctx.error === undefined ? "old-key" : ctx.lastChance ? "switch-key" : "refresh-key";
|
||||
},
|
||||
});
|
||||
|
||||
for await (const _event of stream) {
|
||||
// drain
|
||||
}
|
||||
|
||||
expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]);
|
||||
expect(keys).toEqual(["old-key", "new-key"]);
|
||||
expect(authCalls).toBe(1);
|
||||
// Initial resolve, then step (b) refresh-same — the switch step is never reached.
|
||||
expect(keys).toEqual(["old-key", "refresh-key"]);
|
||||
expect(keys.every(key => typeof key === "string")).toBe(true);
|
||||
expect(contexts.map(ctx => ({ lastChance: ctx.lastChance, hasError: ctx.error !== undefined }))).toEqual([
|
||||
{ lastChance: false, hasError: false },
|
||||
{ lastChance: false, hasError: true },
|
||||
]);
|
||||
expect((contexts[1]?.error as { status?: number }).status).toBe(401);
|
||||
});
|
||||
|
||||
it("retries when a provider emits start then a 401 error event before content", async () => {
|
||||
const keys: Array<string | undefined> = [];
|
||||
it("buffers the start event and retries on a 401 error event before content", async () => {
|
||||
const keys: unknown[] = [];
|
||||
const eventTypes: string[] = [];
|
||||
let authCalls = 0;
|
||||
registerCustomApi(
|
||||
API,
|
||||
(_model: Model<Api>, _context: Context, options?: SimpleStreamOptions) => {
|
||||
keys.push(options?.apiKey);
|
||||
pushKey(keys, options);
|
||||
const stream = new AssistantMessageEventStream();
|
||||
queueMicrotask(() => {
|
||||
if (keys.length === 1) {
|
||||
@@ -127,9 +133,7 @@ describe("streamSimple auth retry", () => {
|
||||
});
|
||||
return;
|
||||
}
|
||||
const message = assistant(["ok"]);
|
||||
stream.push({ type: "start", partial: message });
|
||||
stream.push({ type: "done", reason: "stop", message });
|
||||
ok(stream);
|
||||
});
|
||||
return stream;
|
||||
},
|
||||
@@ -137,28 +141,59 @@ describe("streamSimple auth retry", () => {
|
||||
);
|
||||
|
||||
const stream = streamSimple(model(), context, {
|
||||
apiKey: "old-key",
|
||||
onAuthError: async (provider, oldKey, error) => {
|
||||
authCalls += 1;
|
||||
expect(provider).toBe("test-provider");
|
||||
expect(oldKey).toBe("old-key");
|
||||
expect((error as { status?: number }).status).toBe(401);
|
||||
return "new-key";
|
||||
},
|
||||
apiKey: async ctx => (ctx.error === undefined ? "old-key" : "new-key"),
|
||||
});
|
||||
|
||||
for await (const event of stream) {
|
||||
eventTypes.push(event.type);
|
||||
}
|
||||
|
||||
expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]);
|
||||
expect(keys).toEqual(["old-key", "new-key"]);
|
||||
// The buffered `start` of the failed attempt must not leak — the user
|
||||
// sees exactly one clean start/done pair.
|
||||
expect(eventTypes).toEqual(["start", "done"]);
|
||||
expect(authCalls).toBe(1);
|
||||
});
|
||||
|
||||
it("retries on a 401 carried only via errorStatus", async () => {
|
||||
const keys: unknown[] = [];
|
||||
registerCustomApi(
|
||||
API,
|
||||
(_model: Model<Api>, _context: Context, options?: SimpleStreamOptions) => {
|
||||
pushKey(keys, options);
|
||||
const stream = new AssistantMessageEventStream();
|
||||
queueMicrotask(() => {
|
||||
if (keys.length === 1) {
|
||||
stream.push({ type: "start", partial: assistant() });
|
||||
stream.push({
|
||||
type: "error",
|
||||
reason: "error",
|
||||
error: assistantError(
|
||||
'{"type":"error","error":{"type":"authentication_error","message":"Invalid authentication credentials"}}',
|
||||
401,
|
||||
),
|
||||
});
|
||||
return;
|
||||
}
|
||||
ok(stream);
|
||||
});
|
||||
return stream;
|
||||
},
|
||||
SOURCE_ID,
|
||||
);
|
||||
|
||||
const stream = streamSimple(model(), context, {
|
||||
apiKey: async ctx => (ctx.error === undefined ? "old-key" : "new-key"),
|
||||
});
|
||||
for await (const _event of stream) {
|
||||
// drain
|
||||
}
|
||||
|
||||
expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]);
|
||||
expect(keys).toEqual(["old-key", "new-key"]);
|
||||
});
|
||||
|
||||
it("does not retry after replay-unsafe content has been emitted", async () => {
|
||||
let authCalls = 0;
|
||||
let retryResolves = 0;
|
||||
const failure = authError();
|
||||
registerCustomApi(
|
||||
API,
|
||||
@@ -175,10 +210,9 @@ describe("streamSimple auth retry", () => {
|
||||
);
|
||||
|
||||
const stream = streamSimple(model(), context, {
|
||||
apiKey: "old-key",
|
||||
onAuthError: async () => {
|
||||
authCalls += 1;
|
||||
return "new-key";
|
||||
apiKey: async ctx => {
|
||||
if (ctx.error !== undefined) retryResolves += 1;
|
||||
return ctx.error === undefined ? "old-key" : "new-key";
|
||||
},
|
||||
});
|
||||
|
||||
@@ -192,94 +226,79 @@ describe("streamSimple auth retry", () => {
|
||||
}
|
||||
|
||||
expect(caught).toBe(failure);
|
||||
expect(authCalls).toBe(0);
|
||||
// The resolver is never asked for a retry key once a replay-unsafe event shipped.
|
||||
expect(retryResolves).toBe(0);
|
||||
});
|
||||
|
||||
it("retries on 401 carried via errorStatus when the message has no parseable status", async () => {
|
||||
const keys: Array<string | undefined> = [];
|
||||
let authCalls = 0;
|
||||
it("escalates refresh-same then switch in order (2-retry ordering)", async () => {
|
||||
const keys: unknown[] = [];
|
||||
registerCustomApi(
|
||||
API,
|
||||
(_model: Model<Api>, _context: Context, options?: SimpleStreamOptions) => {
|
||||
keys.push(options?.apiKey);
|
||||
pushKey(keys, options);
|
||||
const stream = new AssistantMessageEventStream();
|
||||
queueMicrotask(() => {
|
||||
if (keys.length === 1) {
|
||||
stream.push({ type: "start", partial: assistant() });
|
||||
stream.push({
|
||||
type: "error",
|
||||
reason: "error",
|
||||
// Realistic Anthropic SDK shape: message begins with `<status> <body>`
|
||||
// and the regex fallback in extractHttpStatusFromError cannot find 401
|
||||
// inside the JSON body. Only `errorStatus` carries the signal.
|
||||
error: assistantError(
|
||||
'{"type":"error","error":{"type":"authentication_error","message":"Invalid authentication credentials"}}',
|
||||
401,
|
||||
),
|
||||
});
|
||||
return;
|
||||
}
|
||||
const message = assistant(["ok"]);
|
||||
stream.push({ type: "start", partial: message });
|
||||
stream.push({ type: "done", reason: "stop", message });
|
||||
});
|
||||
queueMicrotask(() => (options?.apiKey === "switch-key" ? ok(stream) : stream.fail(authError())));
|
||||
return stream;
|
||||
},
|
||||
SOURCE_ID,
|
||||
);
|
||||
|
||||
const stream = streamSimple(model(), context, {
|
||||
apiKey: "old-key",
|
||||
onAuthError: async () => {
|
||||
authCalls += 1;
|
||||
return "new-key";
|
||||
},
|
||||
apiKey: async ctx => (ctx.error === undefined ? "old-key" : ctx.lastChance ? "switch-key" : "refresh-key"),
|
||||
});
|
||||
|
||||
for await (const _event of stream) {
|
||||
// drain
|
||||
}
|
||||
|
||||
expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]);
|
||||
expect(keys).toEqual(["old-key", "new-key"]);
|
||||
expect(authCalls).toBe(1);
|
||||
expect(keys).toEqual(["old-key", "refresh-key", "switch-key"]);
|
||||
});
|
||||
|
||||
it("retries on a thrown usage_limit_reached error before any event has been emitted", async () => {
|
||||
const keys: Array<string | undefined> = [];
|
||||
const errors: unknown[] = [];
|
||||
it("skips the refresh-same step when the resolver returns an unchanged key", async () => {
|
||||
const keys: unknown[] = [];
|
||||
registerCustomApi(
|
||||
API,
|
||||
(_model: Model<Api>, _context: Context, options?: SimpleStreamOptions) => {
|
||||
keys.push(options?.apiKey);
|
||||
pushKey(keys, options);
|
||||
const stream = new AssistantMessageEventStream();
|
||||
queueMicrotask(() => {
|
||||
if (keys.length === 1) {
|
||||
stream.fail(
|
||||
Object.assign(
|
||||
new Error("You have hit your ChatGPT usage limit (pro plan). Try again in ~158 min."),
|
||||
{ status: 429 },
|
||||
),
|
||||
);
|
||||
return;
|
||||
}
|
||||
const message = assistant(["ok"]);
|
||||
stream.push({ type: "start", partial: message });
|
||||
stream.push({ type: "done", reason: "stop", message });
|
||||
});
|
||||
queueMicrotask(() => (options?.apiKey === "switch-key" ? ok(stream) : stream.fail(authError())));
|
||||
return stream;
|
||||
},
|
||||
SOURCE_ID,
|
||||
);
|
||||
|
||||
const stream = streamSimple(model(), context, {
|
||||
apiKey: "old-key",
|
||||
onAuthError: async (_provider, _oldKey, error) => {
|
||||
errors.push(error);
|
||||
return "new-key";
|
||||
// refresh-same yields the same failing key → that attempt is skipped.
|
||||
apiKey: async ctx => (ctx.error === undefined ? "old-key" : ctx.lastChance ? "switch-key" : "old-key"),
|
||||
});
|
||||
for await (const _event of stream) {
|
||||
// drain
|
||||
}
|
||||
|
||||
expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]);
|
||||
expect(keys).toEqual(["old-key", "switch-key"]);
|
||||
});
|
||||
|
||||
it("retries a thrown usage-limit error and passes the cause to the resolver", async () => {
|
||||
const keys: unknown[] = [];
|
||||
const errors: unknown[] = [];
|
||||
registerCustomApi(
|
||||
API,
|
||||
(_model: Model<Api>, _context: Context, options?: SimpleStreamOptions) => {
|
||||
pushKey(keys, options);
|
||||
const stream = new AssistantMessageEventStream();
|
||||
queueMicrotask(() => (keys.length === 1 ? stream.fail(usageLimitError()) : ok(stream)));
|
||||
return stream;
|
||||
},
|
||||
SOURCE_ID,
|
||||
);
|
||||
|
||||
const stream = streamSimple(model(), context, {
|
||||
apiKey: async ctx => {
|
||||
if (ctx.error !== undefined) errors.push(ctx.error);
|
||||
return ctx.error === undefined ? "old-key" : "new-key";
|
||||
},
|
||||
});
|
||||
|
||||
for await (const _event of stream) {
|
||||
// drain
|
||||
}
|
||||
@@ -287,18 +306,17 @@ describe("streamSimple auth retry", () => {
|
||||
expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]);
|
||||
expect(keys).toEqual(["old-key", "new-key"]);
|
||||
expect(errors).toHaveLength(1);
|
||||
// The error surfaced to onAuthError carries the original 429 status so
|
||||
// the gateway's refresh hook can branch on it.
|
||||
// The cause carries the original 429 so the resolver can branch usage-limit vs 401.
|
||||
expect((errors[0] as { status?: number }).status).toBe(429);
|
||||
expect((errors[0] as Error).message).toMatch(/usage limit/i);
|
||||
});
|
||||
|
||||
it("retries when a provider emits a usage_limit_reached error event before content", async () => {
|
||||
const keys: Array<string | undefined> = [];
|
||||
it("retries a usage-limit error event before content", async () => {
|
||||
const keys: unknown[] = [];
|
||||
registerCustomApi(
|
||||
API,
|
||||
(_model: Model<Api>, _context: Context, options?: SimpleStreamOptions) => {
|
||||
keys.push(options?.apiKey);
|
||||
pushKey(keys, options);
|
||||
const stream = new AssistantMessageEventStream();
|
||||
queueMicrotask(() => {
|
||||
if (keys.length === 1) {
|
||||
@@ -306,16 +324,11 @@ describe("streamSimple auth retry", () => {
|
||||
stream.push({
|
||||
type: "error",
|
||||
reason: "error",
|
||||
// errorStatus deliberately omitted: matches how the codex
|
||||
// provider serializes the message-only path that used to
|
||||
// 502 in the gateway.
|
||||
error: assistantError("You have hit your ChatGPT usage limit (pro plan). Try again in ~158 min."),
|
||||
});
|
||||
return;
|
||||
}
|
||||
const message = assistant(["ok"]);
|
||||
stream.push({ type: "start", partial: message });
|
||||
stream.push({ type: "done", reason: "stop", message });
|
||||
ok(stream);
|
||||
});
|
||||
return stream;
|
||||
},
|
||||
@@ -323,10 +336,8 @@ describe("streamSimple auth retry", () => {
|
||||
);
|
||||
|
||||
const stream = streamSimple(model(), context, {
|
||||
apiKey: "old-key",
|
||||
onAuthError: async () => "new-key",
|
||||
apiKey: async ctx => (ctx.error === undefined ? "old-key" : "new-key"),
|
||||
});
|
||||
|
||||
for await (const _event of stream) {
|
||||
// drain
|
||||
}
|
||||
@@ -335,13 +346,13 @@ describe("streamSimple auth retry", () => {
|
||||
expect(keys).toEqual(["old-key", "new-key"]);
|
||||
});
|
||||
|
||||
it("surfaces the original usage_limit error when the retry callback declines", async () => {
|
||||
const keys: Array<string | undefined> = [];
|
||||
const original = Object.assign(new Error("You have hit your ChatGPT usage limit (pro plan)."), { status: 429 });
|
||||
it("surfaces the original error when the resolver declines every retry", async () => {
|
||||
const keys: unknown[] = [];
|
||||
const original = usageLimitError();
|
||||
registerCustomApi(
|
||||
API,
|
||||
(_model: Model<Api>, _context: Context, options?: SimpleStreamOptions) => {
|
||||
keys.push(options?.apiKey);
|
||||
pushKey(keys, options);
|
||||
const stream = new AssistantMessageEventStream();
|
||||
queueMicrotask(() => stream.fail(original));
|
||||
return stream;
|
||||
@@ -349,12 +360,9 @@ describe("streamSimple auth retry", () => {
|
||||
SOURCE_ID,
|
||||
);
|
||||
|
||||
// Callback returns undefined → no sibling credential to rotate to.
|
||||
// The original failure must reach the caller untouched so the client
|
||||
// can decide what to do (back off, surface to user, …).
|
||||
const stream = streamSimple(model(), context, {
|
||||
apiKey: "old-key",
|
||||
onAuthError: async () => undefined,
|
||||
// Decline all retries: no sibling credential to rotate to.
|
||||
apiKey: async ctx => (ctx.error === undefined ? "old-key" : undefined),
|
||||
});
|
||||
|
||||
let caught: unknown;
|
||||
@@ -367,8 +375,32 @@ describe("streamSimple auth retry", () => {
|
||||
}
|
||||
|
||||
expect(caught).toBe(original);
|
||||
// Single attempt — the inner stream is only re-invoked when a new key
|
||||
// is provided.
|
||||
expect(keys).toEqual(["old-key"]);
|
||||
});
|
||||
|
||||
it("fails the stream when the initial resolve yields no key", async () => {
|
||||
let attempts = 0;
|
||||
registerCustomApi(
|
||||
API,
|
||||
() => {
|
||||
attempts += 1;
|
||||
return new AssistantMessageEventStream();
|
||||
},
|
||||
SOURCE_ID,
|
||||
);
|
||||
|
||||
const stream = streamSimple(model(), context, { apiKey: async () => undefined });
|
||||
|
||||
let caught: unknown;
|
||||
try {
|
||||
for await (const _event of stream) {
|
||||
// drain
|
||||
}
|
||||
} catch (error) {
|
||||
caught = error;
|
||||
}
|
||||
|
||||
expect((caught as Error).message).toMatch(/No API key for provider/);
|
||||
expect(attempts).toBe(0);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -148,6 +148,7 @@ describe("StreamMarkupHealing pattern selection", () => {
|
||||
expect(getStreamMarkupHealingPattern("minimax-code", "MiniMax-M2.5", { parseThinkingTags: true })).toBe(
|
||||
"thinking",
|
||||
);
|
||||
expect(getStreamMarkupHealingPattern("opencode-zen", "minimax-m3")).toBe("thinking");
|
||||
expect(getStreamMarkupHealingPattern("nanogpt", "deepseek/deepseek-v4-pro")).toBe("dsml");
|
||||
expect(getStreamMarkupHealingPattern("ollama-cloud", "gpt-oss:120b")).toBeUndefined();
|
||||
expect(getStreamMarkupHealingPattern("openai", "deepseek-v4-pro")).toBeUndefined();
|
||||
@@ -582,6 +583,39 @@ describe("Ollama provider DSML envelope healing", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("OpenAI completions MiniMax thinking healing", () => {
|
||||
it("parses OpenCode Zen MiniMax think tags into a thinking block", async () => {
|
||||
const model: Model<"openai-completions"> = {
|
||||
id: "minimax-m3",
|
||||
name: "MiniMax M3",
|
||||
api: "openai-completions",
|
||||
provider: "opencode-zen",
|
||||
baseUrl: "https://opencode.ai/zen/v1",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 8_192,
|
||||
};
|
||||
global.fetch = mockFetch([
|
||||
chunk(model.id, { content: "visible <thin" }),
|
||||
chunk(model.id, { content: "k>hidden reasoning</think" }),
|
||||
chunk(model.id, { content: ">" }),
|
||||
chunk(model.id, { content: " answer" }),
|
||||
chunk(model.id, {}, "stop"),
|
||||
"[DONE]",
|
||||
]);
|
||||
|
||||
const result = await streamOpenAICompletions(model, baseContext(), { apiKey: "test-key" }).result();
|
||||
|
||||
expect(result.content).toEqual([
|
||||
{ type: "text", text: "visible " },
|
||||
{ type: "thinking", thinking: "hidden reasoning", thinkingSignature: undefined },
|
||||
{ type: "text", text: " answer" },
|
||||
]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("OpenAI completions provider DSML envelope healing", () => {
|
||||
it("heals the envelope into a structured tool call and suppresses leaked text", async () => {
|
||||
const model: Model<"openai-completions"> = {
|
||||
|
||||
@@ -652,6 +652,138 @@ describe("Generate E2E Tests", () => {
|
||||
else Bun.env.GOOGLE_APPLICATION_CREDENTIALS = originalGac;
|
||||
}
|
||||
});
|
||||
|
||||
it("routes impersonated_service_account ADC through IAM to the Vertex request", async () => {
|
||||
const originalProject = Bun.env.GOOGLE_CLOUD_PROJECT;
|
||||
const originalGcpProject = Bun.env.GCP_PROJECT;
|
||||
const originalGcloudProject = Bun.env.GCLOUD_PROJECT;
|
||||
const originalVertexLocation = Bun.env.GOOGLE_VERTEX_LOCATION;
|
||||
const originalCloudLocation = Bun.env.GOOGLE_CLOUD_LOCATION;
|
||||
const originalLocation = Bun.env.VERTEX_LOCATION;
|
||||
const originalApiKey = Bun.env.GOOGLE_CLOUD_API_KEY;
|
||||
const originalGac = Bun.env.GOOGLE_APPLICATION_CREDENTIALS;
|
||||
const tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-vertex-impersonation-"));
|
||||
const adcPath = path.join(tmpDir, "impersonated-adc.json");
|
||||
await Bun.write(
|
||||
adcPath,
|
||||
JSON.stringify({
|
||||
type: "impersonated_service_account",
|
||||
service_account_impersonation_url:
|
||||
"https://iamcredentials.googleapis.com/v1/projects/-/serviceAccounts/target@project.iam.gserviceaccount.com:generateAccessToken",
|
||||
source_credentials: {
|
||||
type: "authorized_user",
|
||||
client_id: "client-id",
|
||||
client_secret: "client-secret",
|
||||
refresh_token: "refresh-token",
|
||||
},
|
||||
delegates: ["projects/-/serviceAccounts/delegate@project.iam.gserviceaccount.com"],
|
||||
}),
|
||||
);
|
||||
const model: Model<"anthropic-messages"> = {
|
||||
id: "claude-sonnet-4@20250514",
|
||||
name: "Claude Sonnet 4",
|
||||
api: "anthropic-messages",
|
||||
provider: "google-vertex",
|
||||
baseUrl:
|
||||
"https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/publishers/anthropic/models/claude-sonnet-4@20250514:streamRawPredict",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 64_000,
|
||||
};
|
||||
const callOrder: string[] = [];
|
||||
let iamRequest: { url: string; authorization: string | null; body: unknown } | undefined;
|
||||
const captured = Promise.withResolvers<{ url: string; authorization: string | null }>();
|
||||
|
||||
try {
|
||||
__resetVertexTokenCache();
|
||||
Bun.env.GOOGLE_CLOUD_PROJECT = "vertex-project";
|
||||
Bun.env.GOOGLE_VERTEX_LOCATION = "global";
|
||||
delete Bun.env.GCP_PROJECT;
|
||||
delete Bun.env.GCLOUD_PROJECT;
|
||||
delete Bun.env.GOOGLE_CLOUD_LOCATION;
|
||||
delete Bun.env.VERTEX_LOCATION;
|
||||
delete Bun.env.GOOGLE_CLOUD_API_KEY;
|
||||
Bun.env.GOOGLE_APPLICATION_CREDENTIALS = adcPath;
|
||||
|
||||
const events = stream(
|
||||
model,
|
||||
{ messages: [{ role: "user", content: "Hello", timestamp: Date.now() }] },
|
||||
{
|
||||
apiKey: "<authenticated>",
|
||||
fetch: async (input, init) => {
|
||||
const url = input instanceof Request ? input.url : input.toString();
|
||||
const headers = input instanceof Request ? input.headers : new Headers(init?.headers);
|
||||
if (url === "https://oauth2.googleapis.com/token") {
|
||||
callOrder.push("source");
|
||||
return new Response(JSON.stringify({ access_token: "source-token", expires_in: 3600 }));
|
||||
}
|
||||
if (url.startsWith("https://iamcredentials.googleapis.com/")) {
|
||||
callOrder.push("iam");
|
||||
const bodyText =
|
||||
input instanceof Request ? await input.clone().text() : String(init?.body ?? "");
|
||||
iamRequest = { url, authorization: headers.get("authorization"), body: JSON.parse(bodyText) };
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
accessToken: "impersonated-token",
|
||||
expireTime: new Date(Date.now() + 3_600_000).toISOString(),
|
||||
}),
|
||||
);
|
||||
}
|
||||
callOrder.push("vertex");
|
||||
captured.resolve({ url, authorization: headers.get("authorization") });
|
||||
return new Response(JSON.stringify({ error: { message: "stop after capture" } }), { status: 400 });
|
||||
},
|
||||
},
|
||||
);
|
||||
|
||||
for await (const _event of events) {
|
||||
}
|
||||
|
||||
const request = await captured.promise;
|
||||
|
||||
// Source refresh, then IAM generateAccessToken, then the actual Vertex call.
|
||||
expect(callOrder).toEqual(["source", "iam", "vertex"]);
|
||||
|
||||
// IAM exchange is authorized by the freshly minted source token, posts the
|
||||
// reconstructed canonical URL, and forwards the configured delegates verbatim.
|
||||
expect(iamRequest?.url).toBe(
|
||||
"https://iamcredentials.googleapis.com/v1/projects/-/serviceAccounts/target@project.iam.gserviceaccount.com:generateAccessToken",
|
||||
);
|
||||
expect(iamRequest?.authorization).toBe("Bearer source-token");
|
||||
expect(iamRequest?.body).toEqual({
|
||||
delegates: ["projects/-/serviceAccounts/delegate@project.iam.gserviceaccount.com"],
|
||||
scope: ["https://www.googleapis.com/auth/cloud-platform"],
|
||||
lifetime: "3600s",
|
||||
});
|
||||
|
||||
// The impersonated token (not the source token) authorizes the Vertex request.
|
||||
expect(request.url).toBe(
|
||||
"https://aiplatform.googleapis.com/v1/projects/vertex-project/locations/global/publishers/anthropic/models/claude-sonnet-4@20250514:streamRawPredict",
|
||||
);
|
||||
expect(request.authorization).toBe("Bearer impersonated-token");
|
||||
} finally {
|
||||
__resetVertexTokenCache();
|
||||
await fs.rm(tmpDir, { recursive: true, force: true });
|
||||
if (originalProject === undefined) delete Bun.env.GOOGLE_CLOUD_PROJECT;
|
||||
else Bun.env.GOOGLE_CLOUD_PROJECT = originalProject;
|
||||
if (originalGcpProject === undefined) delete Bun.env.GCP_PROJECT;
|
||||
else Bun.env.GCP_PROJECT = originalGcpProject;
|
||||
if (originalGcloudProject === undefined) delete Bun.env.GCLOUD_PROJECT;
|
||||
else Bun.env.GCLOUD_PROJECT = originalGcloudProject;
|
||||
if (originalVertexLocation === undefined) delete Bun.env.GOOGLE_VERTEX_LOCATION;
|
||||
else Bun.env.GOOGLE_VERTEX_LOCATION = originalVertexLocation;
|
||||
if (originalCloudLocation === undefined) delete Bun.env.GOOGLE_CLOUD_LOCATION;
|
||||
else Bun.env.GOOGLE_CLOUD_LOCATION = originalCloudLocation;
|
||||
if (originalLocation === undefined) delete Bun.env.VERTEX_LOCATION;
|
||||
else Bun.env.VERTEX_LOCATION = originalLocation;
|
||||
if (originalApiKey === undefined) delete Bun.env.GOOGLE_CLOUD_API_KEY;
|
||||
else Bun.env.GOOGLE_CLOUD_API_KEY = originalApiKey;
|
||||
if (originalGac === undefined) delete Bun.env.GOOGLE_APPLICATION_CREDENTIALS;
|
||||
else Bun.env.GOOGLE_APPLICATION_CREDENTIALS = originalGac;
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe("Google Vertex Provider (gemini-3-flash-preview)", () => {
|
||||
|
||||
@@ -78,6 +78,178 @@ describe("Tool argument coercion", () => {
|
||||
expect(result.paths).toEqual(["src/**/*.ts"]);
|
||||
});
|
||||
|
||||
it("wraps a singleton object in an array when schema expects object array", () => {
|
||||
const tool: Tool = {
|
||||
name: "todo_like",
|
||||
description: "",
|
||||
parameters: z.object({
|
||||
ops: z.array(
|
||||
z.object({
|
||||
op: z.literal("init"),
|
||||
list: z.array(
|
||||
z.object({
|
||||
phase: z.string(),
|
||||
items: z.array(z.string()),
|
||||
}),
|
||||
),
|
||||
}),
|
||||
),
|
||||
}),
|
||||
};
|
||||
|
||||
const result = validateToolArguments(tool, {
|
||||
type: "toolCall",
|
||||
id: "call-singleton-object-array",
|
||||
name: "todo_like",
|
||||
arguments: {
|
||||
ops: {
|
||||
op: "init",
|
||||
list: [{ phase: "Repro", items: ["capture"] }],
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(result).toEqual({
|
||||
ops: [{ op: "init", list: [{ phase: "Repro", items: ["capture"] }] }],
|
||||
});
|
||||
});
|
||||
|
||||
it("wraps a singleton number in an array when schema expects number array", () => {
|
||||
const tool: Tool = {
|
||||
name: "numeric_list",
|
||||
description: "",
|
||||
parameters: z.object({ values: z.array(z.number()) }),
|
||||
};
|
||||
|
||||
const result = validateToolArguments(tool, {
|
||||
type: "toolCall",
|
||||
id: "call-singleton-number-array",
|
||||
name: "numeric_list",
|
||||
arguments: { values: 7 },
|
||||
});
|
||||
|
||||
expect(result).toEqual({ values: [7] });
|
||||
});
|
||||
|
||||
it("does not wrap singleton values for array expectations from failed union branches", () => {
|
||||
const entry = z.object({ id: z.number() });
|
||||
const tool: Tool = {
|
||||
name: "union_shape",
|
||||
description: "",
|
||||
parameters: z.union([z.array(entry), entry]),
|
||||
};
|
||||
|
||||
const result = validateToolArguments(tool, {
|
||||
type: "toolCall",
|
||||
id: "call-union-shape",
|
||||
name: "union_shape",
|
||||
arguments: { id: "1" },
|
||||
});
|
||||
|
||||
expect(result).toEqual({ id: 1 });
|
||||
});
|
||||
|
||||
it("does not wrap singleton values for JSON Schema anyOf array branches", () => {
|
||||
const tool: Tool = {
|
||||
name: "json_schema_union",
|
||||
description: "",
|
||||
parameters: {
|
||||
type: "object",
|
||||
properties: {
|
||||
target: {
|
||||
anyOf: [
|
||||
{
|
||||
type: "array",
|
||||
items: {
|
||||
type: "object",
|
||||
properties: { a: { type: "boolean" } },
|
||||
required: ["a"],
|
||||
additionalProperties: false,
|
||||
},
|
||||
},
|
||||
{
|
||||
type: "object",
|
||||
properties: { a: { type: "boolean" } },
|
||||
required: ["a"],
|
||||
additionalProperties: false,
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
required: ["target"],
|
||||
additionalProperties: false,
|
||||
},
|
||||
};
|
||||
|
||||
// The bug would silently coerce `{ a: "true" }` into `[{ a: true }]` by
|
||||
// wrapping the object to satisfy the failed `anyOf` array branch and
|
||||
// then coercing the inner string into a boolean. Branch tracking keeps
|
||||
// the wrap from firing so the wrong shape never makes it through.
|
||||
expect(() =>
|
||||
validateToolArguments(tool, {
|
||||
type: "toolCall",
|
||||
id: "call-jsonschema-union",
|
||||
name: "json_schema_union",
|
||||
arguments: { target: { a: "true" } },
|
||||
}),
|
||||
).toThrow("Validation failed");
|
||||
});
|
||||
|
||||
it("still wraps nested array fields inside a tag-selected Zod union branch", () => {
|
||||
const tool: Tool = {
|
||||
name: "tagged_union",
|
||||
description: "",
|
||||
parameters: z.union([
|
||||
z.object({ type: z.literal("indices"), indices: z.array(z.number()) }),
|
||||
z.object({ type: z.literal("all") }),
|
||||
]),
|
||||
};
|
||||
|
||||
const result = validateToolArguments(tool, {
|
||||
type: "toolCall",
|
||||
id: "call-tagged-union",
|
||||
name: "tagged_union",
|
||||
arguments: { type: "indices", indices: 1 },
|
||||
});
|
||||
|
||||
expect(result).toEqual({ type: "indices", indices: [1] });
|
||||
});
|
||||
|
||||
it("still wraps nested array fields inside a tag-selected JSON Schema anyOf branch", () => {
|
||||
const tool: Tool = {
|
||||
name: "tagged_json_union",
|
||||
description: "",
|
||||
parameters: {
|
||||
anyOf: [
|
||||
{
|
||||
type: "object",
|
||||
properties: {
|
||||
type: { const: "indices" },
|
||||
indices: { type: "array", items: { type: "number" } },
|
||||
},
|
||||
required: ["type", "indices"],
|
||||
additionalProperties: false,
|
||||
},
|
||||
{
|
||||
type: "object",
|
||||
properties: { type: { const: "all" } },
|
||||
required: ["type"],
|
||||
additionalProperties: false,
|
||||
},
|
||||
],
|
||||
},
|
||||
};
|
||||
|
||||
const result = validateToolArguments(tool, {
|
||||
type: "toolCall",
|
||||
id: "call-tagged-json-union",
|
||||
name: "tagged_json_union",
|
||||
arguments: { type: "indices", indices: 1 },
|
||||
});
|
||||
|
||||
expect(result).toEqual({ type: "indices", indices: [1] });
|
||||
});
|
||||
|
||||
it("parses JSON objects in string values when schema expects object", () => {
|
||||
const tool: Tool = {
|
||||
name: "t4",
|
||||
|
||||
@@ -1,23 +1,162 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Fixed
|
||||
|
||||
- Fixed the working spinner appearing to ignore Esc for 2-3 seconds when an interrupt lands mid-tool. Esc fires the abort synchronously, but the agent loop only stops the loader at `agent_end`, which it cannot reach until every in-flight tool settles in `executeToolCalls`' `await Promise.allSettled(...)` — and process/subagent/kernel-owning tools tear down gracefully (SIGTERM, 2-3s grace, SIGKILL), so the loader kept showing the unchanged "Working…/<intent>" line and read as a dropped keypress. The loader now switches to "Interrupting…" the instant Esc requests the abort and freezes intent-driven label updates until the turn unwinds (`EventController.notifyInterrupting`), so the interrupt is acknowledged immediately even while teardown completes.
|
||||
- Fixed a flaky JS eval worker startup that intermittently failed unrelated CI runs. The worker-ready wait reused Bun's 5s default per-test timeout as its floor, so a slow cold-start under `--isolate` + high concurrency was aborted mid-init; terminating a still-initializing Bun worker is the documented SIGILL/SIGTRAP crash trigger, which took down the whole test file. Worker init now floors at a fixed 15s infrastructure budget (independent of, and still dominated by, a larger per-cell `timeout`), and the JS eval test suites set a 20s file-local timeout so cold starts complete instead of being torn down.
|
||||
- Fixed reviewer-style subagent yields crashing the calling eval cell when a caller-supplied output schema declares `additionalProperties: false` without a `findings` property. `normalizeCompleteData` now consults the active validator before splicing collected `report_finding` entries onto the yielded payload, so injection is suppressed when the schema would reject it — keeping the executor's post-mortem validation in lockstep with the in-tool `yield` validation that already accepted the same raw payload ([#2070](https://github.com/can1357/oh-my-pi/issues/2070))
|
||||
- Fixed Anthropic empty `toolUse` stops without tool calls corrupting session history by retrying them and removing orphaned turns even at the retry cap.
|
||||
- Fixed MCP tools hanging in non-yolo modes by declaring `approval = "write"` on `MCPTool` and `DeferredMCPTool`, and propagating the `approval` property through `customToolToDefinition()` in `sdk.ts`
|
||||
- Fixed session resumption after a working directory is moved/renamed (e.g. `git worktree move`): `--continue` now re-roots the terminal's last session into the new directory when its original directory no longer exists, instead of silently starting a fresh empty session; cross-project `--resume <id>` offers to move (re-root) the session rather than only forking a duplicate copy when the source directory is gone
|
||||
- Fixed Kitty OSC 5522 paste rejecting plain text as "no supported text or image data": the listing parser now decodes the `mime="."` DATA payload (whitespace-separated MIME list) Kitty actually sends, in addition to the per-type DATA packets described by the ancillary 5522-mode spec ([#2051](https://github.com/can1357/oh-my-pi/issues/2051))
|
||||
- Fixed follow-up shortcut submission of builtin slash commands so `/goal set ...` applies goal mode instead of queueing as plain text.
|
||||
- Fixed Ctrl+Z crashing the agent on Windows with `TypeError: Unknown signal: SIGTSTP`. `InputController.handleCtrlZ` called `process.kill(0, "SIGTSTP")` unconditionally, but `SIGTSTP` is POSIX job-control and Bun/Node on Windows rejects the signal name from the JS side; the throw propagated out of the TUI input dispatcher as an uncaught exception. The handler now no-ops with a "Suspend (Ctrl+Z) is not supported on this platform" status on Windows, and on POSIX wraps `process.kill` in a try/catch that detaches the registered SIGCONT resume hook and re-`start()`s the TUI on failure so a rejected signal can never leave the UI stranded with a leaked listener ([#2036](https://github.com/can1357/oh-my-pi/issues/2036)).
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed the `--cwd` launch flag so it is parsed and can override the startup directory instead of always falling back to the current process directory or home auto-switch target.
|
||||
|
||||
## [15.10.1] - 2026-06-07
|
||||
|
||||
### Added
|
||||
|
||||
- Added `display.smoothStreaming` setting (default `true`) to let users enable or disable smooth assistant-stream text reveal
|
||||
- Added `/tan <work>` slash command to fork the current conversation into a background agent so tangential work can continue asynchronously while your main session stays active
|
||||
- Added a background `/tan` dispatch message that records the handoff in the transcript and marks the delegated work as non-blocking
|
||||
- Added `providerPromptCacheKey` support to `CreateAgentSessionOptions` so `/tan` background sessions can reuse the parent session’s prompt-cache lineage
|
||||
- Added session cloning for `/tan` runs with copied artifacts and shared MCP proxy tools
|
||||
- Added `SessionManager.forkFrom`’s optional `suppressBreadcrumb` mode to avoid breadcrumb updates when forking background `/tan` sessions
|
||||
- Added OSC 5522 enhanced paste handling in `InputController`, so terminal clipboard events are decoded as image or text payloads and inserted without passing raw paste sequences to the editor
|
||||
- Added bracketed image-path paste support in `CustomEditor` so a single pasted image file path (PNG/JPEG/GIF/WEBP) is loaded from disk and inserted as an image candidate
|
||||
- Added direct support for `Image #N` insertion from pasted local image paths by routing successful image-path pastes through the same image normalization and resize flow as clipboard image pastes
|
||||
- Added `/fresh` to rotate the provider-facing session id and clear in-memory provider stream/cache state without changing the local session file.
|
||||
- Added a `ChatBlock` transcript primitive (`modes/components/chat-block.ts`) and a single `ctx.present(...)` sink (with `ctx.resetTranscript()`) so chat output is mounted in one place instead of the repeated `chatContainer.addChild(...)` + `ui.requestRender()` pattern scattered across controllers. `ChatBlock` carries a React/Svelte-style lifecycle — `onMount` starts effects, `onCleanup` registers teardown, `finish()` self-completes (stops timers and freezes the block at its final content), and `dispose()`/`resetTranscript()` tears everything down — so animated blocks own their own resources instead of leaking `setInterval`/`requestRender` bookkeeping into callers. The MCP "Connecting…" spinner is now such a block.
|
||||
- Added a `framedBlock` output-block helper (`tui/output-block.ts`) plus a `borderColor` override and `applyBg: false` (no background fill) on output blocks, a `renderStatusLine` `iconOverride`, and an `icon.search` (magnifier) theme symbol — so tool renderers can draw self-contained muted-outline frames and search-family tools can show a magnifier instead of a checkmark.
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed the bash tool frame to use a plain top rule instead of repeating "Bash" in the title bar, and folded minimizer raw-output artifact links into the status footer as `Artifact: <id>`.
|
||||
|
||||
- Changed grouped `read` output to use a white filled-circle mark for the group/single-read success state and omit duplicate per-file success marks inside multi-read groups.
|
||||
|
||||
- Changed assistant streaming output to reveal text incrementally at 30 FPS with grapheme-safe adaptive catch-up, instead of replacing the whole message chunk-by-chunk
|
||||
- Changed shimmer-driven TUI animations (working text, pending bash/eval borders, and theme activity-spinner documentation) to render at 30fps instead of 60fps.
|
||||
- Changed running `task` tool agent rows to use a static `•` marker and shimmer only the subagent name, leaving descriptions, stats, and nested tool detail text solid while removing the rotating status glyph from those rows.
|
||||
- Changed settings singleton method access to reuse bound methods for the active instance instead of allocating a new bound function on every `settings.get` lookup.
|
||||
- Changed plan-mode approval to keep the drafted `local://<slug>-plan.md` file at its original name as the canonical plan path, so approved plans are no longer renamed when leaving plan mode
|
||||
- Changed plan-mode write enforcement so only `local://` artifact files are writable during planning, blocking working-tree edits and allowing scratch or draft plan files in the local artifact area
|
||||
- Changed the `todo` tool result renderer to stop redrawing every phase's full task list on each update: when a multi-phase list is rendered collapsed (the default, not manually expanded), only phases the latest update touched — the phase holding the in_progress task, any phase with a just-completed task, and phases named by the ops that ran (`init` counts as touching all) — render their tasks; untouched phases collapse to a one-line `N. Name done/total` summary. When call args are unavailable (e.g. transcript rebuilds) it falls back to the in_progress/completed-transition signals, and the manual expand toggle still shows every task. Also dropped the blank separator line previously inserted between phases.
|
||||
- Changed non-agent API operations (title and commit-message generation, image generation, web search, eval `llm()`, auto-thinking classifier, memory consolidation) to use session-aware API key resolution with auth retries via `registry.resolver()` / `authStorage.resolver()`, refreshing the active credential before rotating to another account
|
||||
- Changed image generation to wrap every provider fetch branch in `withAuth`, so 401 / usage-limit errors trigger credential force-refresh and rotation for authStorage-backed providers (OpenAI-hosted, antigravity, xai-oauth) while env-only providers (openrouter, gemini) stay single-attempt
|
||||
- Changed web-search providers using `authStorage.getApiKey` (anthropic, exa, tavily, parallel, synthetic, zai, kimi) to wrap HTTP calls in `withAuth` for automatic credential rotation on 401 / usage-limit errors
|
||||
- Changed the directory grouping for `find`, `search`, `ast_grep`, `ast_edit`, and `lsp` diagnostics from a single flat `# dir/` heading per immediate directory to a multi-level tree that folds the common path prefix into one heading. Previously every group repeated the full directory path — so results rooted outside cwd printed the absolute prefix (e.g. `/Users/me/proj/`) on every heading and nested directories were never collapsed. Now a single-child directory chain folds into one heading (`# packages/pkg/src/`, including an absolute root for out-of-cwd results), subdirectories nest one `#` deeper (`## nested/` → `### child.ts`), and each directory's own files are listed before its subdirectories. TUI hyperlink reconstruction tracks the nested directory stack across the whole output so file and code-frame links keep resolving to the correct absolute paths.
|
||||
- Changed the plan-mode approval surface from an inline transcript block plus a separate bottom selector into a single fullscreen overlay (like `/copy`) and overhauled its navigation. The overlay now renders the plan per-section through `ScrollView` (line-level ↑/↓ scroll, Shift+↑/↓ to scroll faster, PgUp/PgDn, g/G) with no stray per-line `…`, and — when the terminal is wide enough and the plan has ≥2 headings — shows a compact VS Code-style section sidebar (the redundant plan-title heading and any "Contents" label are omitted). Focus moves between regions with Tab/Shift+Tab (and flows at the edges: Down past the last section or the bottom of the body drops into the approval options; Up steps back), while the sidebar glows to track the scrolled section. The sidebar can fast-jump between sections, delete a section (with `u` undo), and annotate sections with feedback (`a`); deletions and annotations are collected into refinement feedback that is submitted back to the model when the operator picks "Refine plan". Mouse works too: clicking an approval option activates it, clicking a sidebar section jumps to it, and the wheel scrolls the plan. ←/→ always drive the model-tier slider, Enter confirms, the external-editor key opens the plan, and Esc cancels. The overlay borrows the terminal's alternate screen buffer for its lifetime (`fullscreen` overlay), so the transcript stays put on the normal screen instead of bleeding through scrollback behind the modal.
|
||||
- Changed the interactive controllers (command, MCP, selector, extension-UI, event), debug panels, and the status/error/warning helpers to render chat output through `ctx.present(...)` instead of appending to `chatContainer` and calling `ui.requestRender()` directly; transcript rebuilds dispose live blocks via `ctx.resetTranscript()` so animated blocks' timers stop on reset.
|
||||
- Changed tool-execution block rendering so the container (`ToolExecutionComponent`) is a transparent passthrough — it no longer inserts a top/bottom blank line, adds left/right padding, or paints a state-colored background behind tool output. Tools with substantial body now self-frame with a muted outline and the tool title in the frame's top bar (`edit`/`apply_patch`, `write`, `ask`, `todo`, `github`, `goal`, `inspect_image`, `search_tool_bm25`, `task`), matching the already-framed `bash`/`read`/`eval`/`debug`/`web_search`/`lsp` blocks, while streaming/in-progress and trivial results collapse to a clean status line. The search-family list tools (`find`, `search`, `ast_grep`) and `job` render frameless/minimal; `find`/`search`/`ast_grep` show a magnifier on success instead of a checkmark, and `job` drops its `Job:` label prefix (the per-job rows are self-describing). The `search_tool_bm25`, `github`, and `inspect_image` frames draw with no background fill, and `inspect_image`'s label was shortened to `Inspect`.
|
||||
- Changed the plan-mode active prompt (`prompts/system/plan-mode-active.md`) to make plans decision-complete and cut filler. Added an Objective framing ("another engineer can execute end-to-end without making a single design decision"), a shared "Resolving Unknowns" section (explore discoverable facts before asking; reserve `ask` for non-derivable preferences/tradeoffs with 2–4 options + a recommended default), and a single shared "The Plan" structure (Context / Approach grouped by behavior not file-by-file / ≤5 Critical files / Verification / Assumptions) that replaces the per-branch structure guidance previously duplicated across the iterative and parallel workflows. Added explicit prohibitions on sections that decide nothing (Non-Goals, Out of Scope, Alternatives Considered, Risks/Mitigations boilerplate, Future Work), on enumerating every file/line, and on inventing schema/validation/precedence policy the request never established.
|
||||
- Changed completion notifications (`completion.notify`) to fire whenever the agent yields its turn, including in the foreground. The `agent_end` notification was previously gated behind background mode (`isBackgrounded`), so an ordinary foreground turn never emitted one; the gate is gone and the desktop toast now fires on every normal turn completion (still skipped for aborted/error turns and when `completion.notify` is `off`).
|
||||
- Changed the in-progress `task` tool block to keep the shared `context` brief (`# Goal` / `# Constraints` background) visible after the first progress snapshot arrives, instead of dropping it the moment the streaming call view was replaced by the result frame, and to stop animating a spinner/clock next to the `Task` frame header while running — the per-agent body lines already carry their own running spinner, so the header now shows a static state icon (matching the completed/failed header icons). The context is rendered through a shared `buildContextSection` helper that also undoes per-field double-encoding, so the brief reads cleanly in the result frame even though `renderResult` receives the raw (un-repaired) tool args.
|
||||
- Changed the messaging shown when you press Esc to interrupt a streaming turn from the ambiguous `Operation aborted` / `Tool execution was aborted: Request was aborted` to `Interrupted by user`, so a deliberate user interrupt no longer reads like an internal failure. Every Esc/flush interrupt path (`onEscape` while streaming, the queued-message restore-and-abort path, and the empty-submit queue flush) threads the reason through `AgentSession.abort({ reason })` → `Agent.abort(reason)` so it rides the `AbortController` onto the aborted assistant message's `errorMessage`; the turn label renders it verbatim on both the live and replay paths, and the synthetic placeholder results paired with in-flight tool calls now read `Tool execution was aborted: Interrupted by user`. Aborts that carry no reason still fall back to the retry-aware `Operation aborted` generic. Transcript label resolution is centralized in `resolveAbortLabel` (`session/messages.ts`).
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed the `/background` (and `/bg`) slash command and the background-mode subsystem it was the sole entry point for — `InteractiveMode.isBackgrounded`, `createBackgroundUiContext`, `handleBackgroundEvent`, and every `isBackgrounded` guard across the input/event/extension-UI controllers and UI helpers. The command suspended the whole process group via `SIGTSTP` (a leftover testing shortcut) instead of detaching the running agent, which is not the expected workflow — use terminal panes or a multiplexer instead.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed session auto-retry for generic `upstream_error: Upstream request failed` gateway failures.
|
||||
|
||||
- Fixed inline `find` and `search` result blocks to align with grouped `read` output and render their success headers with the normal tool-title color instead of accent blue.
|
||||
|
||||
- Fixed the working-status shimmer to opt into the loader's 30fps animated-message repaint path while keeping both the status spinner and pending bash/eval tool spinners on their normal 80 ms glyph cadence.
|
||||
- Fixed consecutive `read` tool calls failing to collapse into a single grouped block when a reasoning model emits one read per completion (`[thinking, read]`). The read group was reset on every assistant `message_start`, so each read rendered as its own one-entry `Read …` line; now a read run accretes across completions and is broken only by a rendered non-empty text/thinking block, a non-read tool, or a user/IRC message — matching the transcript-rebuild path. `ReadToolGroupComponent` now reports its live/finalized state so the growing `Read (N)` header repaints correctly on native-scrollback (risk) terminals.
|
||||
- Fixed the `task` tool shared-context brief rendering raw Markdown headings (`# Goal`, `# Constraints`) inside framed call/result blocks instead of using the normal Markdown renderer.
|
||||
- Fixed the animated pending border on `bash`/`eval` blocks leaving a frozen dark "bar" segment behind after a backgrounded command finalized through the async update path. Once a command is auto-backgrounded (`details.async.state === "running"`) the block stays "partial" in the TUI until the async job-manager delivers the final result, but it also gets committed to native scrollback — so a mid-sweep shimmer frame baked a stray darkened border segment into the committed copy. The border now stops animating (and the 60fps redraw loop stops) the moment a block enters the backgrounded state, so the committed frame is a clean static border.
|
||||
- Fixed cold `omp` launch to clear native terminal history on the first paint, avoiding a once-per-launch duplicate welcome/transcript copy before the normal session replay.
|
||||
- Fixed plan approval resolution so `resolve` with `action: "apply"` can still find the plan file when `extra.title` is missing or stale by falling back to the current plan path and most-recent local plan artifacts
|
||||
- Fixed the search-family tool magnifier glyph (`find`, `search`, `ast_grep`, `search_tool_bm25`) to use the `accent` title color instead of `success` green, so the icon matches the tool title in the status header instead of standing out
|
||||
- Fixed TTSR stream interrupts to pass the matched rule name through the abort reason, so aborted in-flight tool placeholders say why they were stopped instead of `Request was aborted`.
|
||||
- Fixed URL reads for binary/special payloads to reuse local readers: remote archives list their root entries, SQLite databases show their table overview, notebooks render as editable cells, and unrenderable binary returns a metadata notice instead of decoded byte garbage.
|
||||
- Fixed pasted image-file paths that cannot be loaded to fall back to normal text paste with status feedback instead of disappearing.
|
||||
- Fixed tool-output file paths not being clickable OSC 8 `file://` hyperlinks in several renderers. `read` titles for plain text and image files (the common case) emitted no link at all because the renderer only linked when a `resolvedPath` was recorded — which the ordinary file/image read paths never set, keeping the absolute path only in `meta.source`; the renderer now falls back to that source path. `write` headers were never wrapped in a hyperlink and now link to the absolute path written (file, archive entry, SQLite, and conflict resolutions). `edit`/`apply_patch` headers wrapped the model-supplied (often cwd-relative) argument path, producing a root-anchored `file:///rel/path` URI; they now link the absolute `details.path` instead. Finally, `search`, `ast_grep`, and `ast_edit` produced doubled link targets (`/proj/src/src/file.ts`) for searches scoped to a subdirectory, because the renderer resolved the cwd-relative display paths against the scope directory rather than cwd — the scoped-search base is now the session cwd (with the scoped file's absolute path still seeding single-file body lines).
|
||||
- Fixed `omp dry-balance --bench` to recover from 401 token failures by re-minting the failing OAuth credential in place before switching accounts
|
||||
- Fixed the bash tool corrupting commands that embed multi-byte UTF-8 (e.g. `✓`/`×` inside a `grep -E` pattern) ahead of a trailing `| head`/`| tail`. The `bash.stripTrailingHeadTail` rewrite cut at char-offset positions reported by `brush-parser` while slicing the command by byte offset, so the trailing-pipe strip landed mid-pattern and dropped the closing quote — turning `… |✓|×|XCTAssert" | tail -80` into `… |✓|×-80` and making execution fail with `pi-natives:command: unterminated double quote`. Fixed in `pi_shell::fixup` (`@oh-my-pi/pi-natives`).
|
||||
- Fixed `omp dry-balance --bench` to recover from 401 token failures by re-minting the failing OAuth credential in place before switching accounts
|
||||
- Fixed duplicate file entries in grouped outputs for `find`, `search`, `ast_grep`, `ast_edit`, and `lsp` diagnostics when the same path appeared multiple times
|
||||
- Fixed search, grep, and edit output rendering so repeated directory group blank-line boundaries no longer break nested path/link reconstruction
|
||||
- Fixed `omp dry-balance --bench` flooding the terminal with staircased, duplicated spinner/status lines (and an indented summary) when the tty has ONLCR/OPOST disabled (raw mode). The interactive progress region separated rows with a bare LF and repositioned with a column-preserving `\x1b[<n>A` cursor-up, both of which only land at column 0 when the terminal translates LF→CRLF; with that translation off, every 80 ms redraw cascaded down and to the right into scrollback. The live region now carriage-returns before every cleared row, terminates each row with CRLF, and caps each row to the terminal width so a wrapped line cannot desync the cursor-up from the logical line count.
|
||||
- Fixed inconsistent vertical spacing between transcript blocks: some blocks (tool results from `search`/`find` and other renderer-backed tools) rendered with a doubled gap (a leading `Spacer` plus the content box's own `paddingY`), while others (the grouped `read` card, file-mention lists, IRC cards) rendered with no gap at all. Vertical spacing is now owned entirely by the chat renderer: `TranscriptContainer` strips each block's plain-blank top/bottom edges and inserts exactly one blank line between consecutive blocks, so every block is separated by a single consistent gap regardless of which component produced it. Individual components (assistant/user/tool/read-group/bash/eval/skill/custom/hook/compaction/branch/todo-reminder/plan-review messages) no longer emit their own leading `Spacer`/`paddingY` for separation, and multi-row groups (IRC cards, file-mention lists, completed-job batches, and the bordered command/`/changelog`/`/context`/version/OAuth/debug panels) are wrapped as single `TranscriptBlock` children so the renderer spaces them as one unit. Background-colored box padding is preserved as block-internal design.
|
||||
- Fixed `resolve` with `action: "discard"` surfacing a hard `isError` "No pending action to resolve" failure to the model when the agent asked to cancel a staged action (e.g. an `ast_edit` preview) but nothing was pending. A discard is a request to reach the "no staged change" end-state, which already holds in that case, so it is now honored as a successful cancellation (`"Nothing to discard; no pending action remains."` with `details.action: "discard"`) instead of an error. `action: "apply"` with no pending action still errors.
|
||||
- Fixed the collapsed tool-output expand hint rendering double brackets (e.g. `((Ctrl+O for more))`) — the `EXPAND_HINT` text already carried its own parentheses and then `formatExpandHint` wrapped it again with the theme's bracket glyphs. The hint now resolves the key actually bound to `app.tools.expand` at render time and reads `⟨<key>: Expand⟩` (e.g. `⟨Ctrl+O: Expand⟩`), so a single bracket pair surrounds it and a user remap of the expand keybinding is reflected instead of a hard-coded `Ctrl+O`.
|
||||
- Fixed the `edit`/`apply_patch` tool dropping its outlined frame while streaming/in-progress (only the final result was framed); the in-progress diff preview now renders inside the same muted frame as the completed result.
|
||||
- Fixed the `todo` and `job` tools rendering a success icon and success styling on a failed/error result; error results now show the error icon and a red frame border.
|
||||
- Fixed `debug` tool refusing every `dlv` launch on Go modules. The launch handler ran `validateLaunchProgram` before adapter selection and rejected any directory program with `launch program resolves to a directory`, while dlv's default `mode=debug` requires a Go package path (a directory or `.go` source file). Adapter resolution now precedes validation, directory programs prefer adapters that advertise `acceptsDirectoryProgram` before falling back to native extensionless debuggers, the rejection only fires when the resolved adapter does not advertise that flag (set on `dlv` in `dap/defaults.json`), and dlv's `mode` is derived from the program shape — directories and `.go` files launch as `mode=debug`, other files as `mode=exec` — so `omp` can debug both Go packages and pre-built binaries ([#2020](https://github.com/can1357/oh-my-pi/issues/2020)).
|
||||
|
||||
## [15.10.0] - 2026-06-06
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
- Replaced the `providers.parallelFetch` boolean setting with the `providers.fetch` enum (`auto` / `native` / `trafilatura` / `lynx` / `parallel` / `jina`) that selects the URL reader-backend priority for the `read`/`fetch` tool, mirroring `providers.image`/`providers.webSearch`. Existing configs are migrated automatically: the legacy key is dropped and the new `auto` default applies.
|
||||
|
||||
### Added
|
||||
|
||||
- Added a GitHub Actions read handler to the `read`/web-fetch GitHub scraper. Fetching `github.com/{owner}/{repo}/actions/runs/{id}` renders the run metadata plus a per-job breakdown (steps listed for any job that did not succeed), and `…/actions/runs/{id}/job/{id}` (also the API-style `…/jobs/{id}`) renders a single job's metadata, step table, and full plain-text logs. Logs are fetched via the `actions/jobs/{id}/logs` redirect using `GITHUB_TOKEN`/`GH_TOKEN` when present, with the per-line ISO timestamp prefix and leading BOM stripped; the section degrades to an explicit notice when logs are unavailable (no token, private repo, or expired/unfinalized run).
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed eval `agent()` subagents so they are never subject to the `task.maxRuntimeMs` wall-clock cap. The parent cell's idle watchdog is already suspended for the entire bridge call (`withBridgeTimeoutPause`), so a long-running fan-out/recovery workflow must not be killed by a per-subagent runtime limit. `runEvalAgent` now passes `maxRuntimeMs: 0` to `runSubprocess`, which honors an explicit `ExecutorOptions.maxRuntimeMs` override over the inherited setting.
|
||||
- Changed interactive timing behavior so `PI_TIMING=x pi` preloads the module timer before the CLI graph loads and includes the `(modules)` report. `PI_TIMING=full` now also exits after printing, matching `PI_TIMING=x`, so full module reports are usable for cold-start measurement without launching the TUI. Added the root `dev:timing` script for the same profiled startup path.
|
||||
- Changed coding-agent startup imports so normal TUI launch imports `InteractiveMode` directly, keeps print/RPC/ACP runners on their branch-only paths, and moves marketplace auto-update work behind a lightweight deferred starter.
|
||||
- Changed cold-launch setup gating so the full setup wizard (every scene plus the overlay and their TUI/OAuth/web-search/theme dependencies) is no longer statically imported by `main.ts`. The current setup version now lives in a tiny dependency-free `modes/setup-version` module, and the wizard barrel is lazy-loaded only when the stored setup version is stale or the wizard is forced — the common up-to-date launch skips loading it entirely.
|
||||
- Changed cold-launch startup imports so the hot-path CLI files no longer pull the full `@oh-my-pi/pi-ai` barrel: `commands/launch.ts` and `cli/args.ts` import `THINKING_EFFORTS`/`Effort` from the tiny `@oh-my-pi/pi-ai/effort` module, and `config/model-registry.ts` now imports its ~20 symbols from narrow subpaths (`api-registry`, `model-cache`, `model-manager`, `model-thinking`, `models`, `provider-models`, `types`, `utils/event-stream`) instead of the barrel — so launching no longer eagerly loads every provider, auth, OAuth, and usage module re-exported by the barrel.
|
||||
- Changed the `read`/`fetch` HTML reader-backend priority to `native > trafilatura > lynx > parallel > jina` (was `parallel > jina > trafilatura > lynx > native`). The in-process native `htmlToMarkdown` runs first — instant, no network, full-fidelity — so the common case no longer depends on a remote service, and a stalled remote backend can no longer mask it. Selecting a specific backend via `providers.fetch` tries it first, then the rest fall back. The low-quality gate (`>100` chars and not `isLowQualityOutput`) now applies uniformly to every backend; when none clears it, the highest-priority substantial-but-low-quality output is still surfaced so the `llms.txt` / document-extraction fallbacks keep running.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed eval `agent()` failures surfacing as an opaque `RuntimeError: bridge call '__agent__' failed` with no reason. When a subagent aborted, `runEvalAgent` built its failure message with `result.error ?? result.stderr ?? result.abortReason ?? …`, but `result.stderr` is the empty string on a clean abort (and `result.error` is gated on a non-empty `stderr`), so the nullish chain stopped at `""` and never reached `abortReason`. The empty string propagated through the loopback bridge and the Python prelude's `RuntimeError(msg or "bridge call … failed")`, discarding the real reason. The chain now uses `||` so an empty `stderr` falls through to `abortReason`.
|
||||
- Fixed subagent aborts being mislabeled as the generic "Cancelled by caller" when the abort originated inside the subagent's own turn (`stopReason: "aborted"` with no caller signal and no runtime-limit timer). `runSubprocess` now prefers the aborted assistant message's `errorMessage` (e.g. "Request was aborted" or a specific stream error) for that case, while a real caller signal or wall-clock abort still reports its precise reason.
|
||||
- Fixed a long streaming tool preview that alone overflows the viewport dropping its scrolled-off head on ED3-risk terminals (ghostty/kitty/iTerm2/…). When expanded with `Ctrl+O`, a streaming `write` (content streaming in) and a streaming `eval` (stdout streaming below its fixed code cell) render top-anchored and grow append-only, but the tool block never reported itself append-only to the transcript, so the renderer's commit-as-you-go boundary stopped at the block start and the earlier rows that scrolled above the viewport were committed nowhere — they vanished, leaving the preview looking like a viewport-tall circular buffer. `ToolExecutionComponent` now implements `isTranscriptBlockAppendOnly()` (gated on `isTranscriptBlockFinalized()`, so it also covers partial-result streams like `eval`), delegating to a renderer-declared `isStreamingPreviewAppendOnly` predicate so the expanded stream commits its head exactly like a streamed assistant reply. Collapsed previews (bounded sliding tail windows) and finalized/result previews (which can collapse to a capped view) stay deferred.
|
||||
- Fixed `read`/`fetch` silently dropping whole list sections on pages with malformed list markup — stray `<img>`, text, or `<video>` nodes as direct children of `<ul>` (e.g. the Alacritty changelog). The Jina reader (previously tried before the local renderers) mis-extracts such lists, returning empty `Added`/`Changed` sections; the native in-process renderer now runs first and preserves the full content.
|
||||
- Fixed `/usage` aggregate amount fallback using raw `limits.length` as account count — now counts unique `accountId` values from limit scopes, so N limits from a single account no longer display as "N accts".
|
||||
- Fixed `/usage` account labeling falling back to "account N" for providers that use `projectId` as their primary identity (e.g. Google Antigravity, Gemini CLI) — `projectId` from report metadata is now considered before the generic fallback.
|
||||
- Fixed the `todo` tool's TUI renderer crashing with `TypeError: args?.ops?.map is not a function` when a streaming tool-call delta surfaced a non-array `ops` field (mid-stream `parseStreamingJson` shapes like `{ ops: "[{" }`, or `[null]` entries before fields arrive). The renderer now treats non-array `ops`, non-object entries, and non-array `items` as missing structure instead of crashing, which also stops the spam-warn cascade that followed each malformed delta. Paired with the Anthropic-side reasoning-replay fix in `packages/ai` ([#2005](https://github.com/can1357/oh-my-pi/issues/2005)).
|
||||
- Fixed Python eval `agent()` collapsing subagent runtime-limit aborts (and other empty-stderr aborts) into a generic `RuntimeError: bridge call '__agent__' failed`. `runEvalAgent` coalesced the failure message with `??`, which stopped at the empty `stderr` and never reached `abortReason`, shipping an empty error through the loopback bridge. The bridge now prefers `abortReason` for aborts and trims empty `stderr`/`error` out of the fallback chain, so Python surfaces the actionable reason (e.g. `Subagent runtime limit exceeded (task.maxRuntimeMs=900000)`) ([#2006](https://github.com/can1357/oh-my-pi/issues/2006)).
|
||||
|
||||
## [15.9.69] - 2026-06-06
|
||||
|
||||
### Added
|
||||
|
||||
- Added anonymous fallback for Perplexity web search, allowing `web_search` and explicit Perplexity provider usage when no Perplexity credentials are configured
|
||||
- Added `gallery` CLI command to render built-in tool renderer output across streaming, in-progress, success, and failure states
|
||||
- Added `omp gallery` filtering and rendering options (`--tool`, `--state`, `--width`, `--expanded`, and `--plain`) for focused renderer previews and plain-text output
|
||||
- Added `omp gallery` fidelity for tools whose renderers are attached on the tool instance (`lsp`, `task`): the gallery now drives them through the same custom-tool render branch production uses, so regressions in that path surface in the gallery rather than only in a live session.
|
||||
- Added `omp gallery --screenshot`, which renders the gallery through a real virtual terminal (VHS) and writes PNG screenshot(s) instead of ANSI, so agents (and anything that can only read raw bytes) can actually see the rendered output. The capture forces truecolor and matches the active theme/symbol preset; tall galleries split across multiple images (whole renderers are never cut). Tune with `--out`, `--font`, and `--font-size`; requires `vhs` on `PATH` and fails with install guidance when absent.
|
||||
- Added `app.display.reset`, bound to `Ctrl+L` by default, to force an immediate terminal display reset/redraw without resizing the window.
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed authenticated Perplexity ask requests to default to the `experimental` model preference, matching the anonymous fallback.
|
||||
- Changed Perplexity explicit provider availability checks so the setup wizard can mark `perplexity` as available for manual selection without credentials, while auto provider discovery still requires auth
|
||||
- Changed the default persistent model selector shortcut from `Ctrl+L` to `Alt+M`, leaving `Ctrl+L` for display reset. Existing user remaps in `keybindings.yml` are preserved.
|
||||
- Changed the edit tool result header to carry the diff change stats (`+N / -M / K hunks`) inline next to the file path, and removed the redundant lone language-icon metadata row and the blank line between the header and the diff body, so a single-hunk edit renders as `✔ Edit: <icon> path:LINE ⟨+3 / 1 hunk⟩` immediately followed by the diff.
|
||||
- Changed the `web_search` tool result rendering: the answer text now shows in full instead of being truncated to a "… N more lines" preview (the `omp q` CLI still caps its compact output), each source renders as a single `title (domain) · age` line with the URL linked on the title (dropping the snippet and bare-URL rows), and the metadata block collapses to one `Provider: <model> @ <provider> (<auth>)` line plus `Usage:` (removing the redundant Sources/Citations/Request/Queries rows).
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Perplexity `perplexity_ask` response parsing so OAuth, cookie, and anonymous searches correctly extract answer text and sources from JSON payloads
|
||||
- Fixed framed read results rendering with an extra blank row above and below the output block.
|
||||
- Fixed collapsed search result previews that could show only "… N more matches" when the first grouped section exceeded the preview budget. Collapsed search output now compacts to match rows, fills the budget with visible hits before the summary, and keeps truncation details out of the bottom user-visible notice.
|
||||
- Fixed the expanded search result view dumping every match when all hits live in one file. A single file's matches collapse into one blank-line group, and the expanded tree list ignored the line budget, so a hot file whose matches span its whole length rendered every row (e.g. lines 374–2858). Expanded search output is now bounded by a larger-than-collapsed budget (`EXPANDED_LINES × 2`), keeps surrounding context rows, and appends a `… N more matches` summary when truncated.
|
||||
- Fixed boolean environment flag overrides that were ORed with settings, so `PI_INTENT_TRACING=0`, `PI_AUTO_QA=0`, and per-backend eval flags now take precedence when present while falling back to config when unset.
|
||||
- Stripped read-output line-number prefixes (`N:`) from auto-piped bare body rows in the hashline edit parser, so pasting `3:text` without a `+` prefix no longer injects `3:` as literal content. Uses single-pass stripping to avoid corrupting content whose own text starts with `digits:` ([#1492](https://github.com/can1357/oh-my-pi/issues/1492)).
|
||||
- Fixed custom-rendered tools that set `mergeCallAndResult` (e.g. `lsp`) rendering a redundant tool-name line above the framed result once a result arrived. `ToolExecutionComponent`'s custom-tool branch now emits the fallback label only when the tool has no `renderCall` and the call is not suppressed by an existing result, matching the built-in renderer branch.
|
||||
- Fixed the `write` tool result rendering with a green success checkmark even when the write failed. `writeToolRenderer.renderResult` now branches on `result.isError`, rendering the error status icon plus the failure message instead of the success header and content preview.
|
||||
- Fixed Perplexity OAuth/cookie web search returning a refusal answer ("I don't currently have access to the web-search tools in this turn") despite returning real sources. `callPerplexityOAuth` was prepending the API-style `web-search` system prompt to the query (`query_str = systemPrompt + "\n\n" + query`), but the consumer `www.perplexity.ai/rest/sse/perplexity_ask` endpoint has no system-message slot and reads the prepended instruction as a meta-prompt, making the model decline. The OAuth/cookie path now sends the bare query; the API-key path still passes the system prompt as a proper `system` message.
|
||||
- Fixed Perplexity OAuth web search always returning the free `turbo` model instead of the account's Pro model (e.g. `pplx_pro_upgraded`/Sonar). The `www.perplexity.ai/rest/sse/perplexity_ask` endpoint authenticates via the `__Secure-next-auth.session-token` cookie and ignores the `Authorization: Bearer` header entirely — so sending the OAuth session token as a bearer was treated as an anonymous request, which silently downgrades to `turbo` regardless of `model_preference`. The stored Perplexity OAuth token is itself the next-auth session JWT (the macOS app injects the same value as that cookie), so `callPerplexityAsk` now sends it as the `__Secure-next-auth.session-token` cookie, unlocking Pro model selection.
|
||||
|
||||
## [15.9.67] - 2026-06-06
|
||||
|
||||
### Added
|
||||
|
||||
- Added `timeout-pause` and `timeout-resume` eval bridge status events emitted around `agent()`/`llm()` operations
|
||||
|
||||
@@ -1017,7 +1017,7 @@ Core pipeline in `renderUrl(url, timeout, raw, signal)`:
|
||||
- HTML-to-text conversion (`renderHtmlToText`)
|
||||
7. If HTML conversion is low-quality (`isLowQualityOutput`), try document-link extraction (`extractDocumentLinks`) + markit.
|
||||
|
||||
`renderHtmlToText(...)` fallback order is explicit in code: Jina reader endpoint (`https://r.jina.ai/<url>`), then `trafilatura` (via `ensureTool`), then `lynx`, then native `htmlToMarkdown`.
|
||||
`renderHtmlToText(...)` tries reader backends in priority order: native in-process `htmlToMarkdown`, then `trafilatura` (via `ensureTool`), then `lynx`, then Parallel extract, then the Jina reader endpoint (`https://r.jina.ai/<url>`). The `providers.fetch` setting picks the order — `auto` uses the default above; selecting a specific backend tries it first, then the rest as fallbacks. Every backend's output must clear the same quality gate (`>100` chars and not `isLowQualityOutput`); if none clears it, the highest-priority substantial-but-low-quality output is surfaced so step 7's targeted fallbacks still run. native operates on already-fetched HTML (instant, no network), so it both wins the common case and survives a stalled remote backend.
|
||||
|
||||
### Scraper registry role (`web/scrapers/index.ts`)
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"type": "module",
|
||||
"name": "@oh-my-pi/pi-coding-agent",
|
||||
"version": "15.9.67",
|
||||
"version": "15.10.1",
|
||||
"description": "Coding agent CLI with read, bash, edit, write tools and session management",
|
||||
"homepage": "https://omp.sh",
|
||||
"author": "Can Boluk",
|
||||
|
||||
Executable
+42
@@ -0,0 +1,42 @@
|
||||
#!/bin/sh
|
||||
# Dev launcher for the omp CLI, installed by `bun run install:dev`.
|
||||
#
|
||||
# Problem it solves: Bun reads `bunfig.toml` from the *current working
|
||||
# directory* at startup and evaluates its `preload` entries before running the
|
||||
# script. A bun-shebang bin (what `bun link` creates for `src/cli.ts`)
|
||||
# therefore inherits whatever `preload` the directory you happen to be in
|
||||
# declares. Running `omp`/`pi` inside an unrelated Bun project can execute — and
|
||||
# crash on — that project's preload, e.g.
|
||||
# error: Cannot find module '@v12sh/utils/frontmatter' from '.../loader.ts'
|
||||
#
|
||||
# Bun only reads the *exact* cwd (it does not walk parents) and ignores
|
||||
# `--config`/`BUN_BE_BUN` for this, so the fix is to launch Bun from an empty,
|
||||
# bunfig-free directory and restore the real cwd inside the process via the
|
||||
# preload shim alongside this file.
|
||||
set -e
|
||||
|
||||
# Resolve this script's real location even when invoked through a symlink
|
||||
# (`$HOME/.bun/bin/omp` -> this file).
|
||||
self=$0
|
||||
while [ -L "$self" ]; do
|
||||
link=$(readlink "$self")
|
||||
case $link in
|
||||
/*) self=$link ;;
|
||||
*) self=$(dirname "$self")/$link ;;
|
||||
esac
|
||||
done
|
||||
scripts_dir=$(CDPATH= cd -- "$(dirname -- "$self")" && pwd -P)
|
||||
cli=$scripts_dir/../src/cli.ts
|
||||
preload=$scripts_dir/dev-launch-preload.ts
|
||||
timing_preload=$scripts_dir/../../utils/src/module-timer.ts
|
||||
|
||||
launch_dir=${OMP_DEV_LAUNCH_DIR:-${HOME}/.omp/.dev-cwd}
|
||||
mkdir -p "$launch_dir"
|
||||
|
||||
OMP_LAUNCH_CWD=$PWD
|
||||
export OMP_LAUNCH_CWD
|
||||
cd "$launch_dir"
|
||||
if [ -n "${PI_TIMING:-}" ]; then
|
||||
exec bun --preload "$preload" --preload "$timing_preload" "$cli" "$@"
|
||||
fi
|
||||
exec bun --preload "$preload" "$cli" "$@"
|
||||
@@ -0,0 +1,19 @@
|
||||
/**
|
||||
* Bun `--preload` shim for the omp dev launcher (`scripts/dev-launch`).
|
||||
*
|
||||
* The launcher starts Bun from an empty, bunfig-free directory so a foreign
|
||||
* project's `bunfig.toml` `preload` cannot run inside the omp CLI: Bun reads
|
||||
* `bunfig.toml` from the *current working directory* on startup and evaluates
|
||||
* its `preload` entries before the entrypoint, so a bun-shebang bin inherits
|
||||
* whatever `preload` the directory you launched from declares (and crashes if
|
||||
* that preload can't resolve). This shim is loaded before the entrypoint's
|
||||
* imports run, so it restores the user's real working directory in time for
|
||||
* import-time snapshots (e.g. `getProjectDir()` in `@oh-my-pi/pi-utils/dirs`).
|
||||
*/
|
||||
const launchCwd = process.env.OMP_LAUNCH_CWD;
|
||||
if (launchCwd) {
|
||||
delete process.env.OMP_LAUNCH_CWD;
|
||||
try {
|
||||
process.chdir(launchCwd);
|
||||
} catch {}
|
||||
}
|
||||
@@ -15,6 +15,7 @@
|
||||
*/
|
||||
import { type AssistantMessage, completeSimple, Effort, type Model } from "@oh-my-pi/pi-ai";
|
||||
import { prompt } from "@oh-my-pi/pi-utils";
|
||||
|
||||
import type { ModelRegistry } from "../config/model-registry";
|
||||
import { resolveRoleSelection } from "../config/model-resolver";
|
||||
import type { Settings } from "../config/settings";
|
||||
@@ -82,7 +83,10 @@ async function classifyOnline(input: string, deps: ClassifyDifficultyDeps): Prom
|
||||
messages: [{ role: "user", content: input, timestamp: Date.now() }],
|
||||
},
|
||||
{
|
||||
apiKey,
|
||||
apiKey: deps.registry.resolver(model.provider, {
|
||||
sessionId: deps.sessionId,
|
||||
baseUrl: model.baseUrl,
|
||||
}),
|
||||
maxTokens,
|
||||
disableReasoning: true,
|
||||
metadata,
|
||||
|
||||
@@ -22,6 +22,7 @@ export const commands: CommandEntry[] = [
|
||||
{ name: "config", load: () => import("./commands/config").then(m => m.default) },
|
||||
{ name: "dry-balance", load: () => import("./commands/dry-balance").then(m => m.default) },
|
||||
{ name: "grep", load: () => import("./commands/grep").then(m => m.default) },
|
||||
{ name: "gallery", load: () => import("./commands/gallery").then(m => m.default) },
|
||||
{ name: "grievances", load: () => import("./commands/grievances").then(m => m.default) },
|
||||
{ name: "install", load: () => import("./commands/install").then(m => m.default) },
|
||||
{ name: "plugin", load: () => import("./commands/plugin").then(m => m.default) },
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
/**
|
||||
* CLI argument parsing and help display
|
||||
*/
|
||||
import { type Effort, THINKING_EFFORTS } from "@oh-my-pi/pi-ai";
|
||||
import { type Effort, THINKING_EFFORTS } from "@oh-my-pi/pi-ai/effort";
|
||||
import { APP_NAME, CONFIG_DIR_NAME, logger } from "@oh-my-pi/pi-utils";
|
||||
import chalk from "chalk";
|
||||
import { parseEffort } from "../thinking";
|
||||
@@ -109,6 +109,8 @@ export function parseArgs(inputArgs: string[], extensionFlags?: Map<string, { ty
|
||||
result.version = true;
|
||||
} else if (arg === "--allow-home") {
|
||||
result.allowHome = true;
|
||||
} else if (arg === "--cwd" && i + 1 < args.length) {
|
||||
result.cwd = args[++i];
|
||||
} else if (arg === "--mode" && i + 1 < args.length) {
|
||||
const mode = args[++i];
|
||||
if (mode === "text" || mode === "json" || mode === "rpc" || mode === "acp" || mode === "rpc-ui") {
|
||||
@@ -258,7 +260,7 @@ export function getExtraHelpText(): string {
|
||||
NODE_EXTRA_CA_CERTS - CA bundle path (or inline PEM) for server certificate validation
|
||||
OPENAI_API_KEY - OpenAI GPT models
|
||||
GEMINI_API_KEY - Google Gemini models
|
||||
GITHUB_TOKEN - GitHub Copilot (or GH_TOKEN, COPILOT_GITHUB_TOKEN)
|
||||
COPILOT_GITHUB_TOKEN - GitHub Copilot
|
||||
|
||||
${chalk.dim("# Additional LLM Providers")}
|
||||
AZURE_OPENAI_API_KEY - Azure OpenAI models
|
||||
@@ -284,7 +286,7 @@ export function getExtraHelpText(): string {
|
||||
${chalk.dim("# Search & Tools")}
|
||||
EXA_API_KEY - Exa web search
|
||||
BRAVE_API_KEY - Brave web search
|
||||
PERPLEXITY_API_KEY - Perplexity web search (API)
|
||||
PERPLEXITY_API_KEY - Perplexity web search API key (optional; anonymous fallback)
|
||||
PERPLEXITY_COOKIES - Perplexity web search (session cookie)
|
||||
TAVILY_API_KEY - Tavily web search
|
||||
ANTHROPIC_SEARCH_API_KEY - Anthropic web search (override; isolates search from main ANTHROPIC_API_KEY)
|
||||
|
||||
@@ -1,8 +1,10 @@
|
||||
import type {
|
||||
Api,
|
||||
ApiKeyResolver,
|
||||
AssistantMessage,
|
||||
AssistantMessageEvent,
|
||||
AssistantMessageEventStream,
|
||||
AuthCredentialSnapshotEntry,
|
||||
Context,
|
||||
Model,
|
||||
OAuthAccess,
|
||||
@@ -59,6 +61,12 @@ export interface DryBalanceAuthStorage {
|
||||
options?: DryBalanceAuthOptions,
|
||||
): Promise<OAuthAccess | undefined>;
|
||||
getOAuthAccesses?(provider: string, options?: DryBalanceAuthOptions): Promise<OAuthAccessResolution[]>;
|
||||
/**
|
||||
* Force-refresh a single credential by id (step (b) of the auth-retry
|
||||
* policy). The bench re-mints the failing account's token in place on a
|
||||
* 401 rather than rotating accounts — it is measuring each account.
|
||||
*/
|
||||
forceRefreshCredentialById?(id: number, signal?: AbortSignal): Promise<AuthCredentialSnapshotEntry>;
|
||||
}
|
||||
|
||||
export interface DryBalanceModelRegistry {
|
||||
@@ -152,6 +160,8 @@ export interface DryBalanceDependencies {
|
||||
now?: () => number;
|
||||
stdoutIsTTY?: boolean;
|
||||
stderrIsTTY?: boolean;
|
||||
stdoutColumns?: number;
|
||||
stderrColumns?: number;
|
||||
}
|
||||
|
||||
type DryBalanceAttemptResult =
|
||||
@@ -181,6 +191,7 @@ type DryBalanceBenchTarget =
|
||||
ok: true;
|
||||
account: string;
|
||||
accessToken: string;
|
||||
credentialId?: number;
|
||||
}
|
||||
| {
|
||||
ok: false;
|
||||
@@ -310,10 +321,11 @@ function renderBenchStatusLine(
|
||||
}
|
||||
}
|
||||
|
||||
function createBenchProgressSink(
|
||||
export function createBenchProgressSink(
|
||||
total: number,
|
||||
write: (text: string) => void,
|
||||
interactive: boolean,
|
||||
columns: number,
|
||||
): DryBalanceBenchProgressSink {
|
||||
const statuses: DryBalanceBenchProgressStatus[] = Array.from({ length: total }, () => ({ state: "waiting" }));
|
||||
if (!interactive) {
|
||||
@@ -333,13 +345,21 @@ function createBenchProgressSink(
|
||||
let frame = 0;
|
||||
let lineCount = 0;
|
||||
let timer: NodeJS.Timeout | undefined;
|
||||
const width = Number.isFinite(columns) && columns > 0 ? Math.trunc(columns) : 80;
|
||||
const render = (): void => {
|
||||
const lines = [
|
||||
chalk.bold("bench requests"),
|
||||
...statuses.map((status, index) => renderBenchStatusLine(status, index, total, frame)),
|
||||
];
|
||||
if (lineCount > 0) write(`\x1b[${lineCount}A`);
|
||||
write(`${lines.map(line => `\x1b[2K${line}`).join("\n")}\n`);
|
||||
// Anchor every redraw at column 0 and terminate each row with CRLF: a
|
||||
// bare `\n` only returns to column 0 when the tty performs ONLCR
|
||||
// translation, which is off whenever the terminal is in raw mode — there
|
||||
// the old column-preserving cursor-up staircased each frame into
|
||||
// scrollback. Cap each line to the terminal width so a wrapped row never
|
||||
// desyncs the `\x1b[<n>A` cursor-up from the logical line count.
|
||||
const move = lineCount > 0 ? `\x1b[${lineCount}A` : "";
|
||||
const body = lines.map(line => `\x1b[2K${truncateToWidth(line, width)}`).join("\r\n");
|
||||
write(`${move}\r${body}\r\n`);
|
||||
lineCount = lines.length;
|
||||
};
|
||||
render();
|
||||
@@ -370,13 +390,23 @@ function createBenchProgressSink(
|
||||
async function runBenchRequest(
|
||||
model: Model<Api>,
|
||||
sessionId: string,
|
||||
account: string,
|
||||
accessToken: string,
|
||||
target: Extract<DryBalanceBenchTarget, { ok: true }>,
|
||||
authStorage: DryBalanceAuthStorage,
|
||||
streamFn: DryBalanceStreamSimple,
|
||||
now: () => number,
|
||||
): Promise<DryBalanceBenchResult> {
|
||||
const { account, accessToken, credentialId } = target;
|
||||
const startedAt = now();
|
||||
let firstTokenAt: number | undefined;
|
||||
// Re-mint the cached token on a 401: a peer/broker may have rotated it out
|
||||
// from under our snapshot (Anthropic rotates refresh tokens on every use).
|
||||
// The bench measures one account, so the switch step intentionally declines.
|
||||
const apiKey: ApiKeyResolver = async ({ lastChance, error }) => {
|
||||
if (error === undefined) return accessToken;
|
||||
if (lastChance || credentialId === undefined || !authStorage.forceRefreshCredentialById) return undefined;
|
||||
const refreshed = await authStorage.forceRefreshCredentialById(credentialId);
|
||||
return refreshed.credential.type === "oauth" ? refreshed.credential.access : undefined;
|
||||
};
|
||||
try {
|
||||
const context: Context = {
|
||||
messages: [
|
||||
@@ -389,7 +419,7 @@ async function runBenchRequest(
|
||||
],
|
||||
};
|
||||
const stream = streamFn(model, context, {
|
||||
apiKey: accessToken,
|
||||
apiKey,
|
||||
sessionId,
|
||||
maxTokens: resolveBenchMaxTokens(model),
|
||||
temperature: 0.2,
|
||||
@@ -454,7 +484,7 @@ async function resolveBenchTargets(
|
||||
seen.add(key);
|
||||
const account = extractAccount(entry);
|
||||
if (entry.ok) {
|
||||
targets.push({ ok: true, account, accessToken: entry.accessToken });
|
||||
targets.push({ ok: true, account, accessToken: entry.accessToken, credentialId: entry.credentialId });
|
||||
} else {
|
||||
targets.push({ ok: false, account, error: entry.error });
|
||||
}
|
||||
@@ -465,6 +495,7 @@ async function resolveBenchTargets(
|
||||
async function runBenchTargets(
|
||||
model: Model<Api>,
|
||||
targets: DryBalanceBenchTarget[],
|
||||
authStorage: DryBalanceAuthStorage,
|
||||
randomSessionId: () => string,
|
||||
progress: DryBalanceBenchProgressSink | undefined,
|
||||
streamFn: DryBalanceStreamSimple,
|
||||
@@ -482,14 +513,7 @@ async function runBenchTargets(
|
||||
return result;
|
||||
}
|
||||
progress?.markRunning(index, target.account);
|
||||
const result = await runBenchRequest(
|
||||
model,
|
||||
randomSessionId(),
|
||||
target.account,
|
||||
target.accessToken,
|
||||
streamFn,
|
||||
now,
|
||||
);
|
||||
const result = await runBenchRequest(model, randomSessionId(), target, authStorage, streamFn, now);
|
||||
progress?.complete(index, result);
|
||||
return result;
|
||||
}),
|
||||
@@ -792,8 +816,19 @@ export async function runDryBalanceCommand(
|
||||
const progressInteractive = command.flags.json
|
||||
? (deps.stderrIsTTY ?? process.stderr.isTTY === true)
|
||||
: (deps.stdoutIsTTY ?? process.stdout.isTTY === true);
|
||||
progress = createBenchProgressSink(targets.length, progressWrite, progressInteractive);
|
||||
benchResults = await runBenchTargets(model, targets, randomSessionId, progress, streamFn, now);
|
||||
const progressColumns = command.flags.json
|
||||
? (deps.stderrColumns ?? process.stderr.columns ?? 80)
|
||||
: (deps.stdoutColumns ?? process.stdout.columns ?? 80);
|
||||
progress = createBenchProgressSink(targets.length, progressWrite, progressInteractive, progressColumns);
|
||||
benchResults = await runBenchTargets(
|
||||
model,
|
||||
targets,
|
||||
runtime.modelRegistry.authStorage,
|
||||
randomSessionId,
|
||||
progress,
|
||||
streamFn,
|
||||
now,
|
||||
);
|
||||
results = targets.map(target =>
|
||||
target.ok ? { ok: true, account: target.account } : { ok: false, reason: target.error },
|
||||
);
|
||||
|
||||
@@ -0,0 +1,226 @@
|
||||
/**
|
||||
* `omp gallery` — render every built-in tool's renderer across its lifecycle.
|
||||
*
|
||||
* For each tool with a registered renderer, the gallery drives a real
|
||||
* {@link ToolExecutionComponent} through four states — streaming arguments,
|
||||
* arguments complete (in progress), success, and failure — and prints the
|
||||
* rendered output to stdout. It exists for visual QA of tool renderers without
|
||||
* having to provoke each state through a live agent session.
|
||||
*/
|
||||
import type { AgentTool } from "@oh-my-pi/pi-agent-core";
|
||||
import type { TUI } from "@oh-my-pi/pi-tui";
|
||||
import { getProjectDir } from "@oh-my-pi/pi-utils";
|
||||
import { Settings } from "../config/settings";
|
||||
import { ToolExecutionComponent } from "../modes/components/tool-execution";
|
||||
import { initTheme, theme } from "../modes/theme/theme";
|
||||
import { toolRenderers } from "../tools/renderers";
|
||||
import { type GalleryFixture, type GalleryResult, galleryFixtures } from "./gallery-fixtures";
|
||||
import { captureGalleryScreenshots } from "./gallery-screenshot";
|
||||
|
||||
/** Lifecycle states the gallery renders, in display order. */
|
||||
export const GALLERY_STATES = ["streaming", "progress", "success", "error"] as const;
|
||||
export type GalleryState = (typeof GALLERY_STATES)[number];
|
||||
|
||||
const STATE_LABELS: Record<GalleryState, string> = {
|
||||
streaming: "streaming args",
|
||||
progress: "in progress",
|
||||
success: "done",
|
||||
error: "failed",
|
||||
};
|
||||
|
||||
export interface GalleryCommandArgs {
|
||||
/** Render width in columns (defaults to terminal width, clamped). */
|
||||
width?: number;
|
||||
/** Restrict to a single tool name. */
|
||||
tool?: string;
|
||||
/** Restrict to specific lifecycle states. */
|
||||
states?: GalleryState[];
|
||||
/** Render the expanded variant of each renderer. */
|
||||
expanded?: boolean;
|
||||
/** Strip ANSI styling from the output (useful when redirecting to a file). */
|
||||
plain?: boolean;
|
||||
/** Capture the rendered gallery as PNG screenshot(s) via VHS instead of printing ANSI. */
|
||||
screenshot?: boolean;
|
||||
/** Screenshot output path (single image) or base path (suffixed when split across images). */
|
||||
out?: string;
|
||||
/** Font family for screenshots (must be installed; Nerd Font recommended for icon glyphs). */
|
||||
font?: string;
|
||||
/** Font size in points for screenshots. */
|
||||
fontSize?: number;
|
||||
}
|
||||
|
||||
/** One tool's rendered lifecycle, as ANSI lines: a leading blank, the section rule, then each state. */
|
||||
export interface GallerySection {
|
||||
heading: string;
|
||||
lines: string[];
|
||||
}
|
||||
|
||||
const GENERIC_ERROR: GalleryResult = {
|
||||
content: [{ type: "text", text: "Error: operation failed" }],
|
||||
isError: true,
|
||||
};
|
||||
|
||||
/**
|
||||
* Build the fake `AgentTool` the component needs for its label, edit mode, and —
|
||||
* for `customRendered` fixtures — the renderer functions that route it through
|
||||
* the same custom-tool branch production uses (see {@link GalleryFixture}).
|
||||
*/
|
||||
function fakeToolFor(name: string, fixture: GalleryFixture | undefined): AgentTool | undefined {
|
||||
if (!fixture?.label && !fixture?.editMode && !fixture?.customRendered) return undefined;
|
||||
const tool: Record<string, unknown> = { name, label: fixture.label ?? name, mode: fixture.editMode };
|
||||
if (fixture.customRendered) {
|
||||
const renderer = toolRenderers[name] as
|
||||
| { renderCall?: unknown; renderResult?: unknown; mergeCallAndResult?: unknown; inline?: unknown }
|
||||
| undefined;
|
||||
if (renderer) {
|
||||
tool.renderCall = renderer.renderCall;
|
||||
tool.renderResult = renderer.renderResult;
|
||||
tool.mergeCallAndResult = renderer.mergeCallAndResult;
|
||||
tool.inline = renderer.inline;
|
||||
}
|
||||
}
|
||||
return tool as unknown as AgentTool;
|
||||
}
|
||||
|
||||
/** The curated fixture for a tool, or a generic one for registry tools lacking sample data. */
|
||||
export function resolveFixture(name: string): GalleryFixture {
|
||||
return (
|
||||
galleryFixtures[name] ??
|
||||
({
|
||||
args: { note: `sample ${name} call` },
|
||||
result: { content: [{ type: "text", text: `${name} completed` }] },
|
||||
} satisfies GalleryFixture)
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Render a single tool/state pair to lines. Builds a fresh component, drives it
|
||||
* to the requested state, settles any async edit preview, then snapshots the
|
||||
* render and stops all animation timers.
|
||||
*/
|
||||
export async function renderGalleryState(
|
||||
name: string,
|
||||
fixture: GalleryFixture,
|
||||
state: GalleryState,
|
||||
width: number,
|
||||
expanded = false,
|
||||
): Promise<string[]> {
|
||||
const tool = fakeToolFor(name, fixture);
|
||||
const streamingArgs = state === "streaming" ? (fixture.streamingArgs ?? fixture.args) : fixture.args;
|
||||
// The component only calls `requestRender` during a static render;
|
||||
// `imageBudget` is consulted solely when images render, which the gallery
|
||||
// disables. A cast avoids constructing a real terminal.
|
||||
const ui = { requestRender() {} } as unknown as TUI;
|
||||
const component = new ToolExecutionComponent(name, streamingArgs, { showImages: false }, tool, ui, getProjectDir());
|
||||
component.setExpanded(expanded);
|
||||
|
||||
if (state !== "streaming") {
|
||||
component.setArgsComplete();
|
||||
}
|
||||
if (state === "success") {
|
||||
component.updateResult(fixture.result, false);
|
||||
} else if (state === "error") {
|
||||
component.updateResult(fixture.errorResult ?? GENERIC_ERROR, false);
|
||||
}
|
||||
|
||||
// Edit-like renderers compute their diff preview off the render path; wait
|
||||
// for it to settle so the snapshot is deterministic instead of racing a tick.
|
||||
await component.whenPreviewSettled();
|
||||
|
||||
const lines = component.render(width);
|
||||
component.stopAnimation();
|
||||
return lines;
|
||||
}
|
||||
|
||||
function resolveWidth(requested: number | undefined): number {
|
||||
const fallback = process.stdout.columns ?? 100;
|
||||
const width = requested ?? fallback;
|
||||
return Math.max(40, Math.min(200, width));
|
||||
}
|
||||
|
||||
function sectionRule(label: string, width: number): string {
|
||||
const prefix = `── ${label} `;
|
||||
const fill = Math.max(0, width - prefix.length);
|
||||
return theme.fg("accent", theme.bold(`${prefix}${"─".repeat(fill)}`));
|
||||
}
|
||||
|
||||
/**
|
||||
* Render each requested tool's lifecycle into ANSI section blocks. The block
|
||||
* layout (leading blank, section rule, then a blank + dim label + body per
|
||||
* state) is shared by the stdout and screenshot paths so both stay identical.
|
||||
*/
|
||||
async function renderGallerySections(
|
||||
names: string[],
|
||||
states: GalleryState[],
|
||||
width: number,
|
||||
expanded: boolean,
|
||||
): Promise<GallerySection[]> {
|
||||
const sections: GallerySection[] = [];
|
||||
for (const name of names) {
|
||||
const fixture = resolveFixture(name);
|
||||
const heading = fixture.label && fixture.label !== name ? `${name} — ${fixture.label}` : name;
|
||||
const lines: string[] = ["", sectionRule(heading, width)];
|
||||
for (const state of states) {
|
||||
lines.push("", theme.fg("dim", ` · ${STATE_LABELS[state]}`));
|
||||
try {
|
||||
for (const line of await renderGalleryState(name, fixture, state, width, expanded)) lines.push(line);
|
||||
} catch (err) {
|
||||
lines.push(theme.fg("error", ` render failed: ${String(err)}`));
|
||||
}
|
||||
}
|
||||
sections.push({ heading, lines });
|
||||
}
|
||||
return sections;
|
||||
}
|
||||
|
||||
/**
|
||||
* Render the gallery. Iterates the renderer registry (or a single tool),
|
||||
* printing each requested lifecycle state under a labeled section — or, with
|
||||
* `screenshot`, capturing the rendered output as PNG(s) via VHS.
|
||||
*/
|
||||
export async function runGalleryCommand(args: GalleryCommandArgs): Promise<void> {
|
||||
const settingsInstance = await Settings.init();
|
||||
// Screenshots must carry exact theme RGB regardless of how the invoking
|
||||
// terminal advertises its color support, so force truecolor before the theme
|
||||
// (and therefore every SGR escape it emits) is built.
|
||||
if (args.screenshot) process.env.COLORTERM = "truecolor";
|
||||
await initTheme(
|
||||
false,
|
||||
settingsInstance.get("symbolPreset"),
|
||||
settingsInstance.get("colorBlindMode"),
|
||||
settingsInstance.get("theme.dark"),
|
||||
settingsInstance.get("theme.light"),
|
||||
);
|
||||
|
||||
const width = resolveWidth(args.width);
|
||||
const expanded = args.expanded ?? false;
|
||||
const states = args.states && args.states.length > 0 ? args.states : [...GALLERY_STATES];
|
||||
|
||||
// Renderer-registry tools plus fixture-only tools (no dedicated renderer,
|
||||
// e.g. `report_tool_issue` / custom extension tools) so the gallery covers
|
||||
// the generic fallback + custom-tool branches too.
|
||||
const allNames = Array.from(new Set([...Object.keys(toolRenderers), ...Object.keys(galleryFixtures)])).sort();
|
||||
const names = args.tool ? allNames.filter(name => name === args.tool) : allNames;
|
||||
if (args.tool && names.length === 0) {
|
||||
process.stdout.write(`Unknown tool '${args.tool}'. Known tools: ${allNames.join(", ")}\n`);
|
||||
return;
|
||||
}
|
||||
|
||||
const sections = await renderGallerySections(names, states, width, expanded);
|
||||
|
||||
if (args.screenshot) {
|
||||
const paths = await captureGalleryScreenshots(sections, {
|
||||
width,
|
||||
font: args.font,
|
||||
fontSize: args.fontSize,
|
||||
out: args.out,
|
||||
});
|
||||
process.stdout.write(`${paths.join("\n")}\n`);
|
||||
return;
|
||||
}
|
||||
|
||||
const lines = sections.flatMap(section => section.lines);
|
||||
lines.push("");
|
||||
const text = lines.map(line => (args.plain ? Bun.stripANSI(line) : line)).join("\n");
|
||||
process.stdout.write(`${text}\n`);
|
||||
}
|
||||
@@ -0,0 +1,292 @@
|
||||
// Gallery fixtures for the agentic orchestration tools (task, goal, job).
|
||||
import type { GalleryFixture } from "./types";
|
||||
|
||||
export const agenticFixtures: Record<string, GalleryFixture> = {
|
||||
task: {
|
||||
label: "Task",
|
||||
customRendered: true,
|
||||
// Streaming: agent chosen, first task fully arrived, second still landing.
|
||||
streamingArgs: {
|
||||
agent: "task",
|
||||
tasks: [
|
||||
{
|
||||
id: "AuthLoader",
|
||||
description: "Load auth middleware",
|
||||
assignment: "Read packages/server/src/auth/*.ts and summarize the session-cookie flow.",
|
||||
},
|
||||
{ id: "RateLimiter", description: "Audit rate limiter" },
|
||||
],
|
||||
},
|
||||
args: {
|
||||
agent: "task",
|
||||
context: [
|
||||
"# Goal",
|
||||
"Harden the HTTP auth stack before the release cut.",
|
||||
"# Constraints",
|
||||
"Touch only files under packages/server/src/auth/. Do not run gates.",
|
||||
].join("\n"),
|
||||
tasks: [
|
||||
{
|
||||
id: "AuthLoader",
|
||||
description: "Load auth middleware",
|
||||
assignment:
|
||||
"Read packages/server/src/auth/session.ts and middleware.ts, then document the session-cookie validation flow and any TODOs.",
|
||||
},
|
||||
{
|
||||
id: "RateLimiter",
|
||||
description: "Audit rate limiter",
|
||||
assignment:
|
||||
"Inspect packages/server/src/auth/rate-limit.ts. Confirm the 429 path sets Retry-After and report gaps.",
|
||||
},
|
||||
{
|
||||
id: "TokenRotation",
|
||||
description: "Check token rotation",
|
||||
assignment:
|
||||
"Trace refresh-token rotation in packages/server/src/auth/tokens.ts and flag any reuse window.",
|
||||
},
|
||||
],
|
||||
},
|
||||
result: {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: "3 agents completed: AuthLoader, RateLimiter, TokenRotation.",
|
||||
},
|
||||
],
|
||||
details: {
|
||||
projectAgentsDir: null,
|
||||
totalDurationMs: 48_200,
|
||||
usage: { cost: { total: 0.34 } },
|
||||
results: [
|
||||
{
|
||||
index: 0,
|
||||
id: "AuthLoader",
|
||||
agent: "task",
|
||||
agentSource: "bundled",
|
||||
description: "Load auth middleware",
|
||||
task: "Read packages/server/src/auth/session.ts and middleware.ts",
|
||||
assignment:
|
||||
"Read packages/server/src/auth/session.ts and middleware.ts, then document the session-cookie validation flow and any TODOs.",
|
||||
exitCode: 0,
|
||||
output: [
|
||||
"Session validation runs in middleware.ts:42 via verifySessionCookie().",
|
||||
"Cookies are HMAC-signed (SHA-256) and checked against the session store.",
|
||||
"TODO at session.ts:88 — sliding-expiration refresh is stubbed.",
|
||||
].join("\n"),
|
||||
stderr: "",
|
||||
truncated: false,
|
||||
durationMs: 41_900,
|
||||
tokens: 61_400,
|
||||
contextTokens: 23_100,
|
||||
contextWindow: 200_000,
|
||||
resolvedModel: "anthropic/claude-sonnet",
|
||||
usage: { cost: { total: 0.12 } },
|
||||
outputMeta: { lineCount: 3, charCount: 214 },
|
||||
},
|
||||
{
|
||||
index: 1,
|
||||
id: "RateLimiter",
|
||||
agent: "task",
|
||||
agentSource: "bundled",
|
||||
description: "Audit rate limiter",
|
||||
task: "Inspect packages/server/src/auth/rate-limit.ts",
|
||||
assignment:
|
||||
"Inspect packages/server/src/auth/rate-limit.ts. Confirm the 429 path sets Retry-After and report gaps.",
|
||||
exitCode: 0,
|
||||
output: [
|
||||
"rate-limit.ts uses a fixed-window counter keyed by client IP.",
|
||||
"429 responses set Retry-After (rate-limit.ts:57).",
|
||||
"Gap: no per-account limit, so a botnet across IPs bypasses the cap.",
|
||||
].join("\n"),
|
||||
stderr: "",
|
||||
truncated: false,
|
||||
durationMs: 38_500,
|
||||
tokens: 54_800,
|
||||
contextTokens: 19_700,
|
||||
contextWindow: 200_000,
|
||||
resolvedModel: "anthropic/claude-sonnet",
|
||||
usage: { cost: { total: 0.1 } },
|
||||
outputMeta: { lineCount: 3, charCount: 198 },
|
||||
},
|
||||
{
|
||||
index: 2,
|
||||
id: "TokenRotation",
|
||||
agent: "task",
|
||||
agentSource: "bundled",
|
||||
description: "Check token rotation",
|
||||
task: "Trace refresh-token rotation in packages/server/src/auth/tokens.ts",
|
||||
assignment:
|
||||
"Trace refresh-token rotation in packages/server/src/auth/tokens.ts and flag any reuse window.",
|
||||
exitCode: 0,
|
||||
output: [
|
||||
"Refresh tokens rotate on every use (tokens.ts:120) and the old jti is revoked.",
|
||||
"Reuse of a rotated token triggers full-family revocation — no reuse window found.",
|
||||
].join("\n"),
|
||||
stderr: "",
|
||||
truncated: false,
|
||||
durationMs: 48_200,
|
||||
tokens: 49_200,
|
||||
contextTokens: 17_500,
|
||||
contextWindow: 200_000,
|
||||
resolvedModel: "anthropic/claude-sonnet",
|
||||
usage: { cost: { total: 0.12 } },
|
||||
outputMeta: { lineCount: 2, charCount: 160 },
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
errorResult: {
|
||||
isError: true,
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: "1 of 3 agents failed: RateLimiter.",
|
||||
},
|
||||
],
|
||||
details: {
|
||||
projectAgentsDir: null,
|
||||
totalDurationMs: 39_400,
|
||||
usage: { cost: { total: 0.21 } },
|
||||
results: [
|
||||
{
|
||||
index: 0,
|
||||
id: "AuthLoader",
|
||||
agent: "task",
|
||||
agentSource: "bundled",
|
||||
description: "Load auth middleware",
|
||||
task: "Read packages/server/src/auth/session.ts and middleware.ts",
|
||||
assignment:
|
||||
"Read packages/server/src/auth/session.ts and middleware.ts, then document the session-cookie validation flow and any TODOs.",
|
||||
exitCode: 0,
|
||||
output: "Session validation runs in middleware.ts:42 via verifySessionCookie().",
|
||||
stderr: "",
|
||||
truncated: false,
|
||||
durationMs: 31_200,
|
||||
tokens: 58_100,
|
||||
contextTokens: 21_900,
|
||||
contextWindow: 200_000,
|
||||
resolvedModel: "anthropic/claude-sonnet",
|
||||
usage: { cost: { total: 0.11 } },
|
||||
outputMeta: { lineCount: 1, charCount: 70 },
|
||||
},
|
||||
{
|
||||
index: 1,
|
||||
id: "RateLimiter",
|
||||
agent: "task",
|
||||
agentSource: "bundled",
|
||||
description: "Audit rate limiter",
|
||||
task: "Inspect packages/server/src/auth/rate-limit.ts",
|
||||
assignment:
|
||||
"Inspect packages/server/src/auth/rate-limit.ts. Confirm the 429 path sets Retry-After and report gaps.",
|
||||
exitCode: 1,
|
||||
output: "",
|
||||
stderr: "ENOENT: packages/server/src/auth/rate-limit.ts",
|
||||
truncated: false,
|
||||
durationMs: 9_800,
|
||||
tokens: 12_300,
|
||||
contextTokens: 6_400,
|
||||
contextWindow: 200_000,
|
||||
resolvedModel: "anthropic/claude-sonnet",
|
||||
usage: { cost: { total: 0.1 } },
|
||||
error: "Subagent exited 1: target file packages/server/src/auth/rate-limit.ts does not exist.",
|
||||
outputMeta: { lineCount: 0, charCount: 0 },
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
},
|
||||
|
||||
goal: {
|
||||
label: "Goal",
|
||||
// Streaming: op is "create"; objective text still being typed.
|
||||
streamingArgs: { op: "create", objective: "Ship the auth hardening" },
|
||||
args: {
|
||||
op: "create",
|
||||
objective: "Ship the auth hardening pass: per-account rate limits and sliding session expiry.",
|
||||
token_budget: 500_000,
|
||||
},
|
||||
result: {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: "Goal set. Working toward: Ship the auth hardening pass.",
|
||||
},
|
||||
],
|
||||
details: {
|
||||
op: "create",
|
||||
remainingTokens: 451_800,
|
||||
completionBudgetReport: null,
|
||||
goal: {
|
||||
id: "goal_8f2a",
|
||||
objective: "Ship the auth hardening pass: per-account rate limits and sliding session expiry.",
|
||||
status: "active",
|
||||
tokenBudget: 500_000,
|
||||
tokensUsed: 48_200,
|
||||
timeUsedSeconds: 312,
|
||||
createdAt: 1_749_200_000_000,
|
||||
updatedAt: 1_749_200_312_000,
|
||||
},
|
||||
},
|
||||
},
|
||||
errorResult: {
|
||||
isError: true,
|
||||
content: [{ type: "text", text: "Goal tool failed: objective is required when op=create." }],
|
||||
details: { op: "create" },
|
||||
},
|
||||
},
|
||||
|
||||
job: {
|
||||
label: "Job",
|
||||
// Streaming: polling a single job id; the second id is still arriving.
|
||||
streamingArgs: { poll: ["job_a1"] },
|
||||
args: { poll: ["job_a1", "job_b2", "job_c3"] },
|
||||
result: {
|
||||
content: [{ type: "text", text: "3 jobs settled." }],
|
||||
details: {
|
||||
jobs: [
|
||||
{
|
||||
id: "job_a1",
|
||||
type: "bash",
|
||||
status: "completed",
|
||||
label: "bun test packages/server/test/auth.test.ts",
|
||||
durationMs: 18_400,
|
||||
resultText: "42 pass, 0 fail (18.4s)",
|
||||
},
|
||||
{
|
||||
id: "job_b2",
|
||||
type: "task",
|
||||
status: "completed",
|
||||
label: "Migrate rate limiter to a sliding window",
|
||||
durationMs: 96_700,
|
||||
resultText: "Rewrote rate-limit.ts to a token-bucket; added per-account keys.",
|
||||
},
|
||||
{
|
||||
id: "job_c3",
|
||||
type: "bash",
|
||||
status: "failed",
|
||||
label: "bunx biome check packages/server/src/auth",
|
||||
durationMs: 4_100,
|
||||
errorText: "biome: 2 errors in tokens.ts — noUnusedVariables, useConst",
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
errorResult: {
|
||||
isError: true,
|
||||
content: [{ type: "text", text: "Job cancelled by user." }],
|
||||
details: {
|
||||
jobs: [
|
||||
{
|
||||
id: "job_d4",
|
||||
type: "task",
|
||||
status: "cancelled",
|
||||
label: "Refactor the session store to Redis",
|
||||
durationMs: 52_300,
|
||||
errorText: "Aborted: superseded by goal re-scope.",
|
||||
},
|
||||
],
|
||||
cancelled: [{ id: "job_d4", status: "cancelled" }],
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
@@ -0,0 +1,188 @@
|
||||
/** Gallery fixtures for the code-intelligence tools (lsp, debug). */
|
||||
import type { GalleryFixture } from "./types";
|
||||
|
||||
export const codeintelFixtures: Record<string, GalleryFixture> = {
|
||||
lsp: {
|
||||
label: "LSP",
|
||||
customRendered: true,
|
||||
streamingArgs: {
|
||||
action: "references",
|
||||
file: "src/server/auth.ts",
|
||||
},
|
||||
args: {
|
||||
action: "references",
|
||||
file: "src/server/auth.ts",
|
||||
line: 42,
|
||||
symbol: "validateToken",
|
||||
},
|
||||
result: {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: [
|
||||
"Found 6 reference(s):",
|
||||
" src/server/auth.ts:42:14",
|
||||
" 41: ",
|
||||
" 42: export function validateToken(token: string): Claims {",
|
||||
" 43: const claims = verifyJwt(token);",
|
||||
" src/server/auth.ts:118:21",
|
||||
' 117: if (!header) throw new HttpError(401, "missing token");',
|
||||
" 118: const claims = validateToken(stripBearer(header));",
|
||||
" 119: return claims.sub;",
|
||||
" src/server/middleware/session.ts:57:18",
|
||||
" 56: const token = req.cookies.session;",
|
||||
" 57: const claims = validateToken(token);",
|
||||
" 58: req.userId = claims.sub;",
|
||||
" src/server/router.ts:153:20",
|
||||
" 152: router.use(async (req, res, next) => {",
|
||||
" 153: req.claims = await validateToken(req.token);",
|
||||
" 154: next();",
|
||||
" test/auth.test.ts:24:9",
|
||||
' 23: it("rejects expired tokens", () => {',
|
||||
" 24: expect(() => validateToken(expired)).toThrow(/expired/);",
|
||||
" 25: });",
|
||||
" test/auth.test.ts:41:9",
|
||||
' 40: it("accepts valid tokens", () => {',
|
||||
" 41: const claims = validateToken(signed);",
|
||||
' 42: expect(claims.sub).toBe("u_123");',
|
||||
].join("\n"),
|
||||
},
|
||||
],
|
||||
details: {
|
||||
serverName: "typescript-language-server",
|
||||
action: "references",
|
||||
success: true,
|
||||
request: {
|
||||
action: "references",
|
||||
file: "src/server/auth.ts",
|
||||
line: 42,
|
||||
symbol: "validateToken",
|
||||
},
|
||||
},
|
||||
},
|
||||
errorResult: {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: "No language server found for this file",
|
||||
},
|
||||
],
|
||||
isError: true,
|
||||
details: {
|
||||
serverName: "typescript-language-server",
|
||||
action: "references",
|
||||
success: false,
|
||||
request: {
|
||||
action: "references",
|
||||
file: "src/server/auth.ts",
|
||||
line: 42,
|
||||
symbol: "validateToken",
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
|
||||
debug: {
|
||||
label: "Debug",
|
||||
streamingArgs: {
|
||||
action: "stack_trace",
|
||||
},
|
||||
args: {
|
||||
action: "stack_trace",
|
||||
levels: 20,
|
||||
},
|
||||
result: {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: [
|
||||
"Stack trace:",
|
||||
"- #1000 validate_token @ app/server.py:42:14",
|
||||
"- #1001 authenticate @ app/server.py:88:9",
|
||||
"- #1002 handle_request @ app/router.py:153:20",
|
||||
"- #1003 dispatch @ app/router.py:97:5",
|
||||
"- #1004 <module> @ app/server.py:212:1",
|
||||
].join("\n"),
|
||||
},
|
||||
],
|
||||
details: {
|
||||
action: "stack_trace",
|
||||
success: true,
|
||||
snapshot: {
|
||||
id: "dbg-1",
|
||||
adapter: "debugpy",
|
||||
cwd: "/Users/dev/project",
|
||||
program: "./app/server.py",
|
||||
status: "stopped",
|
||||
launchedAt: "2026-06-06T14:21:08.412Z",
|
||||
lastUsedAt: "2026-06-06T14:22:55.901Z",
|
||||
threadId: 1,
|
||||
frameId: 1000,
|
||||
stopReason: "breakpoint",
|
||||
stopDescription: "breakpoint 2",
|
||||
frameName: "validate_token",
|
||||
instructionPointerReference: "0x00000001000034a8",
|
||||
source: { name: "server.py", path: "app/server.py" },
|
||||
line: 42,
|
||||
column: 14,
|
||||
breakpointFiles: 1,
|
||||
breakpointCount: 2,
|
||||
functionBreakpointCount: 0,
|
||||
outputBytes: 248,
|
||||
outputTruncated: false,
|
||||
needsConfigurationDone: false,
|
||||
},
|
||||
stackFrames: [
|
||||
{
|
||||
id: 1000,
|
||||
name: "validate_token",
|
||||
source: { name: "server.py", path: "app/server.py" },
|
||||
line: 42,
|
||||
column: 14,
|
||||
},
|
||||
{
|
||||
id: 1001,
|
||||
name: "authenticate",
|
||||
source: { name: "server.py", path: "app/server.py" },
|
||||
line: 88,
|
||||
column: 9,
|
||||
},
|
||||
{
|
||||
id: 1002,
|
||||
name: "handle_request",
|
||||
source: { name: "router.py", path: "app/router.py" },
|
||||
line: 153,
|
||||
column: 20,
|
||||
},
|
||||
{
|
||||
id: 1003,
|
||||
name: "dispatch",
|
||||
source: { name: "router.py", path: "app/router.py" },
|
||||
line: 97,
|
||||
column: 5,
|
||||
},
|
||||
{
|
||||
id: 1004,
|
||||
name: "<module>",
|
||||
source: { name: "server.py", path: "app/server.py" },
|
||||
line: 212,
|
||||
column: 1,
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
errorResult: {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: "No active debug session. Launch or attach first.",
|
||||
},
|
||||
],
|
||||
isError: true,
|
||||
details: {
|
||||
action: "stack_trace",
|
||||
success: false,
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
@@ -0,0 +1,194 @@
|
||||
/** Gallery fixtures for the edit tools (edit, apply_patch, ast_edit). */
|
||||
import type { GalleryFixture } from "./types";
|
||||
|
||||
export const editFixtures: Record<string, GalleryFixture> = {
|
||||
edit: {
|
||||
label: "Edit",
|
||||
editMode: "replace",
|
||||
// `previewDiff` is surfaced verbatim by the renderer's call preview, and the
|
||||
// harness diff strategy skips `{ file_path, previewDiff }` (no `path`/`edits`),
|
||||
// so the canned diff survives the streaming and progress states.
|
||||
streamingArgs: {
|
||||
file_path: "packages/coding-agent/src/tools/read.ts",
|
||||
previewDiff: [
|
||||
"@@ -88,3 +88,4 @@",
|
||||
" const offset = args.offset ?? 1;",
|
||||
"- const limit = args.limit ?? 2000;",
|
||||
"+ const limit = args.limit ?? 4000;",
|
||||
].join("\n"),
|
||||
},
|
||||
args: {
|
||||
file_path: "packages/coding-agent/src/tools/read.ts",
|
||||
previewDiff: [
|
||||
"@@ -88,5 +88,6 @@",
|
||||
" const offset = args.offset ?? 1;",
|
||||
"- const limit = args.limit ?? 2000;",
|
||||
"+ const limit = args.limit ?? 4000;",
|
||||
" const raw = await Bun.file(path).text();",
|
||||
"- return raw.slice(offset, offset + limit);",
|
||||
'+ return raw.split("\\n").slice(offset - 1, offset - 1 + limit).join("\\n");',
|
||||
].join("\n"),
|
||||
},
|
||||
result: {
|
||||
content: [{ type: "text", text: "Edited packages/coding-agent/src/tools/read.ts (1 hunk, +3 -2)" }],
|
||||
details: {
|
||||
path: "packages/coding-agent/src/tools/read.ts",
|
||||
firstChangedLine: 89,
|
||||
diff: [
|
||||
"@@ -88,5 +88,6 @@",
|
||||
" const offset = args.offset ?? 1;",
|
||||
"- const limit = args.limit ?? 2000;",
|
||||
"+ const limit = args.limit ?? 4000;",
|
||||
" const raw = await Bun.file(path).text();",
|
||||
"- return raw.slice(offset, offset + limit);",
|
||||
'+ return raw.split("\\n").slice(offset - 1, offset - 1 + limit).join("\\n");',
|
||||
].join("\n"),
|
||||
},
|
||||
},
|
||||
errorResult: {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: "Edit failed: the search text was not found in packages/coding-agent/src/tools/read.ts",
|
||||
},
|
||||
],
|
||||
isError: true,
|
||||
details: {
|
||||
path: "packages/coding-agent/src/tools/read.ts",
|
||||
diff: "",
|
||||
errorText:
|
||||
"No match for the search text. Expected `const limit = args.limit ?? 2000;` near line 89, but the file has `const limit = args.limit ?? 1000;`. Re-read the file and retry with the current contents.",
|
||||
},
|
||||
},
|
||||
},
|
||||
|
||||
apply_patch: {
|
||||
label: "Apply Patch",
|
||||
editMode: "apply_patch",
|
||||
streamingArgs: {
|
||||
file_path: "packages/coding-agent/src/edit/renderer.ts",
|
||||
previewDiff: [
|
||||
"@@ -464,2 +464,2 @@",
|
||||
"- fileCount = countEditFiles(editArgs.edits);",
|
||||
"+ fileCount = countDistinctFiles(editArgs.edits);",
|
||||
].join("\n"),
|
||||
},
|
||||
args: {
|
||||
file_path: "packages/coding-agent/src/edit/renderer.ts",
|
||||
previewDiff: [
|
||||
"@@ -177,4 +177,4 @@",
|
||||
" /** Count distinct file paths in an edits array. */",
|
||||
"-function countEditFiles(edits: EditRenderEntry[]): number {",
|
||||
"+function countDistinctFiles(edits: EditRenderEntry[]): number {",
|
||||
" return new Set(edits.map(edit => filePathFromEditEntry(edit.path)).filter(Boolean)).size;",
|
||||
" }",
|
||||
"@@ -467,2 +467,2 @@",
|
||||
"- fileCount = countEditFiles(editArgs.edits);",
|
||||
"+ fileCount = countDistinctFiles(editArgs.edits);",
|
||||
].join("\n"),
|
||||
},
|
||||
result: {
|
||||
content: [
|
||||
{ type: "text", text: "Applied patch to packages/coding-agent/src/edit/renderer.ts (2 hunks, +2 -2)" },
|
||||
],
|
||||
details: {
|
||||
op: "update",
|
||||
path: "packages/coding-agent/src/edit/renderer.ts",
|
||||
firstChangedLine: 178,
|
||||
diff: [
|
||||
"@@ -177,4 +177,4 @@",
|
||||
" /** Count distinct file paths in an edits array. */",
|
||||
"-function countEditFiles(edits: EditRenderEntry[]): number {",
|
||||
"+function countDistinctFiles(edits: EditRenderEntry[]): number {",
|
||||
" return new Set(edits.map(edit => filePathFromEditEntry(edit.path)).filter(Boolean)).size;",
|
||||
" }",
|
||||
"@@ -467,2 +467,2 @@",
|
||||
"- fileCount = countEditFiles(editArgs.edits);",
|
||||
"+ fileCount = countDistinctFiles(editArgs.edits);",
|
||||
].join("\n"),
|
||||
},
|
||||
},
|
||||
errorResult: {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: "Apply patch failed: context does not match at line 177 of packages/coding-agent/src/edit/renderer.ts",
|
||||
},
|
||||
],
|
||||
isError: true,
|
||||
details: {
|
||||
op: "update",
|
||||
path: "packages/coding-agent/src/edit/renderer.ts",
|
||||
diff: "",
|
||||
errorText:
|
||||
"Hunk @@ -177,4 +177,4 @@ failed to apply: the context line `function countEditFiles(edits: EditRenderEntry[]): number {` does not match the file. The file may have changed since it was read.",
|
||||
},
|
||||
},
|
||||
},
|
||||
|
||||
ast_edit: {
|
||||
label: "AST Edit",
|
||||
streamingArgs: {
|
||||
ops: [{ pat: "countEditFiles($$$ARGS)" }],
|
||||
paths: ["packages/coding-agent/src/**/*.ts"],
|
||||
},
|
||||
args: {
|
||||
ops: [{ pat: "countEditFiles($$$ARGS)", out: "countDistinctFiles($$$ARGS)" }],
|
||||
paths: ["packages/coding-agent/src/**/*.ts"],
|
||||
},
|
||||
result: {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: [
|
||||
"# edit/renderer.ts (2 replacements)",
|
||||
"-468: fileCount = countEditFiles(editArgs.edits);",
|
||||
"+468: fileCount = countDistinctFiles(editArgs.edits);",
|
||||
"-488: const totalFiles = args?.edits ? countEditFiles(args.edits) : 0;",
|
||||
"+488: const totalFiles = args?.edits ? countDistinctFiles(args.edits) : 0;",
|
||||
"",
|
||||
"# tools/tool-result.ts (1 replacement)",
|
||||
"-42: return countEditFiles(files);",
|
||||
"+42: return countDistinctFiles(files);",
|
||||
].join("\n"),
|
||||
},
|
||||
],
|
||||
details: {
|
||||
totalReplacements: 3,
|
||||
filesTouched: 2,
|
||||
filesSearched: 214,
|
||||
applied: false,
|
||||
limitReached: false,
|
||||
scopePath: "packages/coding-agent/src",
|
||||
searchPath: "/Users/dev/Projects/pi/packages/coding-agent/src",
|
||||
files: ["edit/renderer.ts", "tools/tool-result.ts"],
|
||||
fileReplacements: [
|
||||
{ path: "edit/renderer.ts", count: 2 },
|
||||
{ path: "tools/tool-result.ts", count: 1 },
|
||||
],
|
||||
displayContent: [
|
||||
"# edit/",
|
||||
"## renderer.ts (2 replacements)",
|
||||
"-468│ fileCount = countEditFiles(editArgs.edits);",
|
||||
"+468│ fileCount = countDistinctFiles(editArgs.edits);",
|
||||
"-488│ const totalFiles = args?.edits ? countEditFiles(args.edits) : 0;",
|
||||
"+488│ const totalFiles = args?.edits ? countDistinctFiles(args.edits) : 0;",
|
||||
"",
|
||||
"# tools/",
|
||||
"## tool-result.ts (1 replacement)",
|
||||
"-42│ return countEditFiles(files);",
|
||||
"+42│ return countDistinctFiles(files);",
|
||||
].join("\n"),
|
||||
},
|
||||
},
|
||||
errorResult: {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: "Pattern parse error in ops[0].pat: unbalanced parenthesis in `countEditFiles($$$ARGS`",
|
||||
},
|
||||
],
|
||||
isError: true,
|
||||
},
|
||||
},
|
||||
};
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user