diff --git a/Cargo.lock b/Cargo.lock index bdf91c7ba..8ae5ba80e 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2064,9 +2064,9 @@ checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" [[package]] name = "jiff" -version = "0.2.29" +version = "0.2.31" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "34f877a98676d2fb664698d74cc6a51ce6c484ce8c770f05d0108ec9090aeb46" +checksum = "ccfe6121cbe750cf81efa362d85c0bde7ea298ec43092d3a193baca59cdbd634" dependencies = [ "defmt", "jiff-static", @@ -2091,9 +2091,9 @@ dependencies = [ [[package]] name = "jiff-static" -version = "0.2.29" +version = "0.2.31" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0666b5ab5ecaca213fc2a85b8c0083d9004e84ee2d5f9a7e0017aaf50986f25f" +checksum = "e165e897f662d428f3cd3828a919dbe067c2d42bb1031eede74ef9d27ecdedd2" dependencies = [ "proc-macro2", "quote", @@ -2911,7 +2911,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "16.2.6" +version = "16.2.9" dependencies = [ "anyhow", "ast-grep-core", @@ -2981,7 +2981,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "16.2.6" +version = "16.2.9" dependencies = [ "async-trait", "libc", @@ -2993,7 +2993,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "16.2.6" +version = "16.2.9" dependencies = [ "anyhow", "arboard", @@ -3043,7 +3043,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "16.2.6" +version = "16.2.9" dependencies = [ "anyhow", "brush-builtins", diff --git a/Cargo.toml b/Cargo.toml index e4033c8f8..5097e0d56 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/vendor/brush-core", "crates/vendor/brush-builtins"] resolver = "3" [workspace.package] -version = "16.2.6" +version = "16.2.9" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index d9c1512f9..dde66abfd 100644 --- a/bun.lock +++ b/bun.lock @@ -21,7 +21,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "16.2.6", + "version": "16.2.9", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -39,7 +39,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "16.2.6", + "version": "16.2.9", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -55,7 +55,7 @@ }, "packages/catalog": { "name": "@oh-my-pi/pi-catalog", - "version": "16.2.6", + "version": "16.2.9", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -69,7 +69,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "16.2.6", + "version": "16.2.9", "bin": { "omp": "src/cli.ts", }, @@ -137,7 +137,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "16.2.6", + "version": "16.2.9", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -148,7 +148,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "16.2.6", + "version": "16.2.9", "bin": { "mnemopi": "src/cli.ts", }, @@ -174,7 +174,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "16.2.6", + "version": "16.2.9", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -182,7 +182,7 @@ }, "packages/snapcompact": { "name": "@oh-my-pi/snapcompact", - "version": "16.2.6", + "version": "16.2.9", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -195,7 +195,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "16.2.6", + "version": "16.2.9", "bin": { "omp-stats": "./src/index.ts", }, @@ -221,7 +221,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "16.2.6", + "version": "16.2.9", "bin": { "omp-swarm": "src/cli.ts", }, @@ -247,7 +247,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "16.2.6", + "version": "16.2.9", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -288,7 +288,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "16.2.6", + "version": "16.2.9", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "handlebars": "catalog:", @@ -301,7 +301,7 @@ }, "packages/wire": { "name": "@oh-my-pi/pi-wire", - "version": "16.2.6", + "version": "16.2.9", "devDependencies": { "@types/bun": "catalog:", }, @@ -337,18 +337,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.2.6", - "@oh-my-pi/omp-stats": "16.2.6", - "@oh-my-pi/pi-agent-core": "16.2.6", - "@oh-my-pi/pi-ai": "16.2.6", - "@oh-my-pi/pi-catalog": "16.2.6", - "@oh-my-pi/pi-coding-agent": "16.2.6", - "@oh-my-pi/pi-mnemopi": "16.2.6", - "@oh-my-pi/pi-natives": "16.2.6", - "@oh-my-pi/pi-tui": "16.2.6", - "@oh-my-pi/pi-utils": "16.2.6", - "@oh-my-pi/pi-wire": "16.2.6", - "@oh-my-pi/snapcompact": "16.2.6", + "@oh-my-pi/hashline": "16.2.9", + "@oh-my-pi/omp-stats": "16.2.9", + "@oh-my-pi/pi-agent-core": "16.2.9", + "@oh-my-pi/pi-ai": "16.2.9", + "@oh-my-pi/pi-catalog": "16.2.9", + "@oh-my-pi/pi-coding-agent": "16.2.9", + "@oh-my-pi/pi-mnemopi": "16.2.9", + "@oh-my-pi/pi-natives": "16.2.9", + "@oh-my-pi/pi-tui": "16.2.9", + "@oh-my-pi/pi-utils": "16.2.9", + "@oh-my-pi/pi-wire": "16.2.9", + "@oh-my-pi/snapcompact": "16.2.9", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -1042,7 +1042,7 @@ "duck": ["duck@0.1.12", "", { "dependencies": { "underscore": "^1.13.1" } }, "sha512-wkctla1O6VfP89gQ+J/yDesM0S7B7XLXjKGzXxMDVFg7uEn706niAtyYovKbyq1oT9YwDcly721/iUWoc8MVRg=="], - "electron-to-chromium": ["electron-to-chromium@1.5.379", "", {}, "sha512-v/qV5aV5EUA2pGilzUCq5/eyOloZAqDZBu9UMBIzgPpLlprjSR6zswsWBTv0KpqxLGUAZEwhO95ZCt7srymNVA=="], + "electron-to-chromium": ["electron-to-chromium@1.5.380", "", {}, "sha512-W6d5AbuEoRayO447cqrg6lKJIlscgRnnxOZl/08kfV71BQDoEBC7Wwis68z87LjyK6f4kWyTaubuDbhHKrZkbA=="], "emnapi": ["emnapi@1.11.1", "", { "peerDependencies": { "node-addon-api": ">= 6.1.0" }, "optionalPeers": ["node-addon-api"] }, "sha512-kSRjhIcxjMFsBqk7ORvoc9aA5SBKDmecrtF5RMcmOTao0kD/zamaxsuTxMI8C1//wGUuvE7a+19pCE7AEhGVnA=="], @@ -1146,7 +1146,7 @@ "js-tokens": ["js-tokens@4.0.0", "", {}, "sha512-RdJUflcE3cUzKiMqQgsCu06FPu9UdIJO0beYbPhHN4k6apgJtifcoCtT9bcxOpYBtpD2kCM6Sbzg4CausW/PKQ=="], - "js-yaml": ["js-yaml@4.2.0", "", { "dependencies": { "argparse": "^2.0.1" }, "bin": { "js-yaml": "bin/js-yaml.js" } }, "sha512-ePWsvanv0DWuDRsW8dnt+R4jQ31SCRCQ7hhNcPXZPsoBZiemuZNYGf7adZdqX2D86j6rvKp3RpCxVTSb8WQlOw=="], + "js-yaml": ["js-yaml@4.3.0", "", { "dependencies": { "argparse": "^2.0.1" }, "bin": { "js-yaml": "bin/js-yaml.js" } }, "sha512-1td788aAnnZ5qs7V2QIRl1owjtYpbKt749Y3xauqQgwIIGF/xXWz1wMTEBx5O3LK3lXLVuqXPdPxj2BoFHaW9Q=="], "jsesc": ["jsesc@3.1.0", "", { "bin": { "jsesc": "bin/jsesc" } }, "sha512-/sM3dO2FOzXjKQhJuo0Q173wf2KOo8t4I8vHy6lF9poUp7bKT0/NHE8fPX23PwfhnykfqnC2xRxOnVw5XuGIaA=="], @@ -1282,7 +1282,7 @@ "postcss": ["postcss@8.5.15", "", { "dependencies": { "nanoid": "^3.3.12", "picocolors": "^1.1.1", "source-map-js": "^1.2.1" } }, "sha512-FfR8sjd4em2T6fb3I2MwAJU7HWVMr9zba+enmQeeWFfCbm+UOC/0X4DS8XtpUTMwWMGbjKYP7xjfNekzyGmB3A=="], - "prettier": ["prettier@3.8.5", "", { "bin": { "prettier": "bin/prettier.cjs" } }, "sha512-zxcTTCedNGJM4R8sj/Cq/F0W/c4iE0afWBcBwMTRtw4WHYP9TWkYjdiH3npPRUYsXQCPR0hTU9yjovOu+E6EQA=="], + "prettier": ["prettier@3.9.0", "", { "bin": { "prettier": "bin/prettier.cjs" } }, "sha512-LjIqSIC5VYLzs9WedVmJ2ljNAGnU+DteIClbahu4L/DBeWjZ6iT/k1lAYyu9JUh+1xINxWadaPw/Pl63y/agAw=="], "process-nextick-args": ["process-nextick-args@2.0.1", "", {}, "sha512-3ouUOpQhtgrbOa17J7+uxOTpITYWaGP7/AhoR3+A+/1e9skrzelGi/dXzEYyvbxubEF6Wn2ypscTKiKJFFn1ag=="], diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index c1b4f9742..58d216b0b 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -173,7 +173,7 @@ fn create_windows_napi_tokio_runtime() -> Option { /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV16_2_6")] +#[napi(js_name = "__piNativesV16_2_9")] pub const fn pi_natives_version_sentinel() {} /// Native module entry point: install crash diagnostics before any tool can diff --git a/docs/environment-variables.md b/docs/environment-variables.md index 078ea5fe4..81efdc319 100644 --- a/docs/environment-variables.md +++ b/docs/environment-variables.md @@ -207,8 +207,8 @@ OAuth host chain: `KIMI_CODE_OAUTH_HOST` → `KIMI_OAUTH_HOST` → `https://auth | `PI_CODEX_WEBSOCKET_IDLE_TIMEOUT_MS` | Positive integer override (default 300000) | | `PI_CODEX_WEBSOCKET_RETRY_BUDGET` | Non-negative integer override (default 5) | | `PI_CODEX_WEBSOCKET_RETRY_DELAY_MS` | Positive integer base backoff override (default 500) | -| `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` | Positive integer OpenAI first-event timeout override | -| `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` | Positive integer OpenAI stream idle timeout override | +| `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` | Positive integer OpenAI first-event timeout override; `0` disables. `omp config set providers.streamFirstEventTimeoutSeconds ` provides the persisted config equivalent | +| `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` | Positive integer OpenAI stream idle timeout override; `0` disables. `omp config set providers.streamIdleTimeoutSeconds ` provides the persisted config equivalent | ### Cursor provider debug diff --git a/docs/local-models.md b/docs/local-models.md index 68502d27b..2a5a494d9 100644 --- a/docs/local-models.md +++ b/docs/local-models.md @@ -127,7 +127,7 @@ Extraction favors **precision** (do not pollute long-term memory) → **Qwen3-1. pick** (its consolidation is good enough). If running a second model for consolidation, **gemma-3-1b** wins that task. -**Shipped local options**: `qwen3-1.7b` (recommended), `gemma-3-1b`, `qwen2.5-1.5b`, `lfm2-1.2b`. +**Shipped local options**: `llama3.2:3b`, `qwen3-1.7b` (recommended), `gemma-3-1b`, `qwen2.5-1.5b`, `lfm2-1.2b`. **Default**: `online` (the configured smol model). ### Known Mnemopi parser bugs (surfaced by these experiments) diff --git a/docs/session-switching-and-recent-listing.md b/docs/session-switching-and-recent-listing.md index 24987c6c7..51dc768fd 100644 --- a/docs/session-switching-and-recent-listing.md +++ b/docs/session-switching-and-recent-listing.md @@ -179,7 +179,7 @@ Lifecycle/state transition: 15. restore model via `getRestorableSessionModels(sessionContext.models, lastModelChangeRole)` — tries the recorded models in fallback order and uses the first one present in the model registry 16. restore thinking level and service tier: - thinking uses persisted `thinking_level_change`, otherwise the configured default clamped to model capability - - service tier uses persisted `service_tier_change`, otherwise the configured `serviceTier` setting (`"none"` becomes unset) + - service tier uses persisted `service_tier_change`, otherwise the configured per-family `tier.openai`/`tier.anthropic`/`tier.google` settings (`"none"` becomes unset) 17. reconnect agent listeners, run the registered session-switch reconciler if any (interactive mode re-enters persisted modes; errors logged, not fatal), and return `true` ## UI state rebuild after interactive switch diff --git a/docs/session.md b/docs/session.md index 1103919fa..84182abe3 100644 --- a/docs/session.md +++ b/docs/session.md @@ -177,11 +177,11 @@ Stores an `AgentMessage` directly. "id": "c1d2e3f4", "parentId": "b1c2d3e4", "timestamp": "2026-02-16T10:21:45.000Z", - "serviceTier": "flex" + "serviceTier": { "openai": "priority", "google": "flex" } } ``` -`serviceTier` can also be `null`. +`serviceTier` is a per-family map keyed by `openai`/`anthropic`/`google` (each value `auto`/`default`/`flex`/`scale`/`priority`), or `null` when no tier is active. Legacy entries that stored a single string (`"flex"`, `"openai-only"`, `"claude-only"`, …) are normalized to this map on read. ### `thinking_level_change` diff --git a/docs/settings.md b/docs/settings.md index 71c36134b..fdf61ce83 100644 --- a/docs/settings.md +++ b/docs/settings.md @@ -366,7 +366,11 @@ A value of `-1` means "use the provider/model default" — `omp` does not send t | `minP` | number | `-1` | Minimum-probability cutoff. | | `presencePenalty` | number | `-1` | Presence penalty. | | `repetitionPenalty` | number | `-1` | Repetition penalty. | -| `serviceTier` | enum | `none` | `none`, `auto`, `default`, `flex`, `scale`, `priority`, `openai-only`, `claude-only`. | +| `tier.openai` | enum | `none` | `none`, `auto`, `default`, `flex`, `scale`, `priority`. Sent as `service_tier` for OpenAI / OpenAI-Codex and OpenAI-family OpenRouter models. | +| `tier.anthropic` | enum | `none` | `none`, `priority`. `priority` realizes fast mode on supported direct Claude models (ignored on Bedrock/Vertex and via OpenRouter). | +| `tier.google` | enum | `none` | `none`, `flex`, `priority`. Gemini API sends it in the body; Vertex sends `priority` via header (`flex` is a no-op on Vertex). | +| `tier.subagent` | enum | `inherit` | `inherit`, `none`, `auto`, `default`, `flex`, `scale`, `priority`. Applied to the spawned model's family; `inherit` tracks the main agent. | +| `tier.advisor` | enum | `none` | `inherit`, `none`, `auto`, `default`, `flex`, `scale`, `priority`. Applied to the advisor model's family. | | `personality` | enum | `default` | `default`, `friendly`, `pragmatic`, `none`. | ### Retry and fallback diff --git a/docs/task-agent-discovery.md b/docs/task-agent-discovery.md index ea2acebcd..7e35066e6 100644 --- a/docs/task-agent-discovery.md +++ b/docs/task-agent-discovery.md @@ -44,7 +44,7 @@ Bundled agents are embedded at build time (`src/task/agents.ts`) using text impo `EMBEDDED_AGENT_DEFS` defines: - `explore`, `plan`, `designer`, `reviewer`, `librarian`, `oracle` from prompt files -- `task` and `quick_task` from shared `task.md` body plus injected frontmatter +- `task` and `sonic` from shared `task.md` body plus injected frontmatter Loading path: @@ -130,7 +130,7 @@ Runtime output schema precedence in `TaskTool.#runSpawn`: (`effectiveOutputSchema = effectiveAgent.output ?? this.session.outputSchema` — the task call itself never carries a schema; ad-hoc structured workflows go through the eval bridge's `agent(prompt, schema)`.) -The model-facing prompt (`src/prompts/tools/task.md`) no longer carries the old structured-output mismatch warning; it tags read-only agents and warns against offloading reasoning to `explore`/`quick_task` instead. +The model-facing prompt (`src/prompts/tools/task.md`) no longer carries the old structured-output mismatch warning; it tags read-only agents and warns against offloading reasoning to `explore`/`sonic` instead. ## Command discovery interaction diff --git a/docs/tools/task.md b/docs/tools/task.md index fb27feb76..abf8bdcbe 100644 --- a/docs/tools/task.md +++ b/docs/tools/task.md @@ -106,7 +106,7 @@ Artifacts and side channels: - off — single spawn per call; `tasks`/`context` are rejected and removed from the schema. - Isolation mode (`task.isolation.mode`): `none`, `auto`, `apfs`, `btrfs`, `zfs`, `reflink`, `overlayfs`, `projfs`, `block-clone`, `rcopy` (legacy `worktree`, `fuse-overlay`, `fuse-projfs` accepted for back-compat); the PAL resolves the actual backend with fallback. - Isolation merge strategy: patch mode (capture/apply root patches) or branch mode (commit to `omp/task/`, cherry-pick into parent). -- Agent source precedence: project custom agents, then user custom agents, then bundled agents (`explore`, `plan`, `designer`, `reviewer`, `task`, `quick_task`, `librarian`, `oracle`). +- Agent source precedence: project custom agents, then user custom agents, then bundled agents (`explore`, `plan`, `designer`, `reviewer`, `task`, `sonic`, `librarian`, `oracle`). ## Side Effects - Filesystem diff --git a/package.json b/package.json index dd1b7c615..7851c6acc 100644 --- a/package.json +++ b/package.json @@ -25,18 +25,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.2.6", - "@oh-my-pi/omp-stats": "16.2.6", - "@oh-my-pi/pi-agent-core": "16.2.6", - "@oh-my-pi/pi-ai": "16.2.6", - "@oh-my-pi/pi-catalog": "16.2.6", - "@oh-my-pi/pi-coding-agent": "16.2.6", - "@oh-my-pi/pi-mnemopi": "16.2.6", - "@oh-my-pi/pi-natives": "16.2.6", - "@oh-my-pi/pi-tui": "16.2.6", - "@oh-my-pi/pi-utils": "16.2.6", - "@oh-my-pi/pi-wire": "16.2.6", - "@oh-my-pi/snapcompact": "16.2.6", + "@oh-my-pi/hashline": "16.2.9", + "@oh-my-pi/omp-stats": "16.2.9", + "@oh-my-pi/pi-agent-core": "16.2.9", + "@oh-my-pi/pi-ai": "16.2.9", + "@oh-my-pi/pi-catalog": "16.2.9", + "@oh-my-pi/pi-coding-agent": "16.2.9", + "@oh-my-pi/pi-mnemopi": "16.2.9", + "@oh-my-pi/pi-natives": "16.2.9", + "@oh-my-pi/pi-tui": "16.2.9", + "@oh-my-pi/pi-utils": "16.2.9", + "@oh-my-pi/pi-wire": "16.2.9", + "@oh-my-pi/snapcompact": "16.2.9", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", diff --git a/packages/agent/package.json b/packages/agent/package.json index bb031de05..046f34860 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "16.2.6", + "version": "16.2.9", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/agent/src/compaction/entries.ts b/packages/agent/src/compaction/entries.ts index 70831433f..f06299990 100644 --- a/packages/agent/src/compaction/entries.ts +++ b/packages/agent/src/compaction/entries.ts @@ -1,4 +1,4 @@ -import type { ImageContent, MessageAttribution, ServiceTier, TextContent } from "@oh-my-pi/pi-ai"; +import type { ImageContent, MessageAttribution, ServiceTierByFamily, TextContent } from "@oh-my-pi/pi-ai"; import type { AgentMessage } from "../types"; export interface SessionEntryBase { @@ -28,7 +28,7 @@ export interface ModelChangeEntry extends SessionEntryBase { export interface ServiceTierChangeEntry extends SessionEntryBase { type: "service_tier_change"; - serviceTier: ServiceTier | null; + serviceTier: ServiceTierByFamily | null; } export interface CompactionEntry extends SessionEntryBase { diff --git a/packages/agent/src/telemetry.ts b/packages/agent/src/telemetry.ts index d2a855ba7..e6314a325 100644 --- a/packages/agent/src/telemetry.ts +++ b/packages/agent/src/telemetry.ts @@ -30,7 +30,6 @@ import { completeSimple, type Message, type Model, - resolveServiceTier, type ServiceTier, type SimpleStreamOptions, type StopReason, @@ -752,8 +751,7 @@ function buildChatRequestAttributes(stepNumber: number, request: ChatRequestSnap attrs[GenAIAttr.RequestStopSequences] = [...request.stopSequences]; } if (request.serviceTier && shouldSendServiceTier(request.serviceTier, provider)) { - const resolved = resolveServiceTier(request.serviceTier, provider); - if (resolved) attrs[OpenAIAttr.RequestServiceTier] = resolved; + attrs[OpenAIAttr.RequestServiceTier] = request.serviceTier; } if (request.reasoningEffort) attrs[PiGenAIAttr.RequestReasoningEffort] = request.reasoningEffort; const toolChoice = serializeToolChoice(request.toolChoice); diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 3a4f8c78d..0f4c612a6 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,43 @@ ## [Unreleased] +## [16.2.9] - 2026-06-30 + +### Added + +- Added `OAuthCallbackFlowOptions.allowPortFallback` to allow disabling random-port fallback, enabling strict port enforcement and early configuration errors for OAuth flows with static redirect URIs. + +### Changed + +- Improved `OAuthCallbackFlow` port conflict error messages to include the busy port, configured redirect URI, and actionable remediation steps. + +### Fixed + +- Fixed an issue where malformed tool-call JSON from local Ollama or llama.cpp models was incorrectly retried as generic 500 errors, now surfacing a clear recovery message. +- Fixed a race condition in OAuth callback flows where abort signals triggered before the callback listener was registered were ignored. + +## [16.2.7] - 2026-06-30 + +### Added + +- Added service tier support for Google Gemini and Vertex AI, including model-specific service tier configurations via ServiceTierByFamily. +- Added Google Vertex AI Interactions API support for Gemini 3+ models by default, with automatic fallback to :streamGenerateContent and a useInteractionsApi: false option to force standard generation. +- Added support for explicit Vertex bearer access tokens via GOOGLE_CLOUD_ACCESS_TOKEN or CLOUDSDK_AUTH_ACCESS_TOKEN environment variables. + +### Changed + +- Updated service tier logic to use per-provider configurations instead of global scopes. +- Refactored priority request billing and accounting to better align with specific provider capabilities. +- Updated API key resolution precedence so explicit environment variables (e.g., GEMINI_API_KEY) override stored or broker-migrated static API keys, while deliberate OAuth logins still take highest precedence. + +### Fixed + +- Improved Vertex AI reliability by automatically falling back to global endpoints on 404 errors. +- Fixed safety setting application for Google Vertex AI models. +- Fixed Kimi Code's Anthropic-compatible request path to keep thinking enabled and downgrade forced tool choice for Kimi K2.7 Code title generation. +- Fixed leaked reasoning fences (such as ```thinking or ) across all providers by splitting them into structured thinking blocks during streaming. +- Fixed Codex requests failing with unsupported all_turns errors on older models (gpt-5.1 and gpt-5.3) by gating the reasoning.context: "all_turns" default to gpt-5.4+ models. + ## [16.2.6] - 2026-06-29 ### Fixed diff --git a/packages/ai/package.json b/packages/ai/package.json index 3ccd37c1f..b5fbf1b58 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "16.2.6", + "version": "16.2.9", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 6060fcb76..d6d91c9f0 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -1736,8 +1736,8 @@ export class AuthStorage { /** * Classify where a provider's auth comes from, following the same precedence * as {@link AuthStorage.getApiKey}: runtime override → config override → - * stored credential (api_key before oauth, matching getApiKey) → env var → - * fallback resolver. Returns undefined when no auth is configured. + * stored OAuth → env var → stored api_key → fallback resolver. Returns + * undefined when no auth is configured. * * Compact, structured counterpart to {@link describeCredentialSource}. */ @@ -1745,10 +1745,9 @@ export class AuthStorage { if (this.#runtimeOverrides.has(provider)) return { kind: "runtime" }; if (this.#configOverrides.has(provider)) return { kind: "config" }; const stored = this.#getCredentialsForProvider(provider); - if (stored.length > 0) { - return { kind: stored.some(credential => credential.type === "api_key") ? "api_key" : "oauth" }; - } + if (stored.some(credential => credential.type === "oauth")) return { kind: "oauth" }; if (getEnvApiKey(provider)) return { kind: "env", envVar: getEnvApiKeyName(provider) }; + if (stored.some(credential => credential.type === "api_key")) return { kind: "api_key" }; if (this.#fallbackResolver?.(provider)) return { kind: "fallback" }; return undefined; } @@ -3760,12 +3759,8 @@ export class AuthStorage { return configKey; } - const apiKeySelection = this.#selectCredentialByType(provider, "api_key"); - if (apiKeySelection) { - return this.#configValueResolver(apiKeySelection.credential.key); - } - - // Return current OAuth access token only if it is not already expired. + // Precedence: a deliberate OAuth login wins, then an explicit env var, then a stored + // static api_key (which may be a stale broker-migrated copy) as a last resort. const oauthSelection = this.#selectCredentialByType(provider, "oauth"); if (oauthSelection) { const expiresAt = oauthSelection.credential.expires; @@ -3784,17 +3779,22 @@ export class AuthStorage { const envKey = getEnvApiKey(provider); if (envKey) return envKey; + const apiKeySelection = this.#selectCredentialByType(provider, "api_key"); + if (apiKeySelection) { + return this.#configValueResolver(apiKeySelection.credential.key); + } + return this.#fallbackResolver?.(provider) ?? undefined; } /** * Get API key for a provider. - * Priority: + * Priority (first match wins): * 1. Runtime override (CLI --api-key) * 2. Config override (models.yml `providers..apiKey`) - * 3. API key from storage - * 4. OAuth token from storage (auto-refreshed) - * 5. Environment variable + * 3. OAuth token from storage (auto-refreshed) + * 4. Environment variable + * 5. Stored API key (e.g. a broker-migrated copy) — last resort, so an explicit env var wins * 6. Fallback resolver (models.yml custom providers, last-resort) */ async getApiKey(provider: string, sessionId?: string, options?: AuthApiKeyOptions): Promise { @@ -3814,25 +3814,27 @@ export class AuthStorage { return configKey; } - const apiKeySelection = this.#selectCredentialByType(provider, "api_key", sessionId); - if (apiKeySelection) { - this.#recordSessionCredential(provider, sessionId, "api_key", apiKeySelection.index); - return this.#configValueResolver(apiKeySelection.credential.key); - } - + // Precedence: a deliberate OAuth login wins, then an explicit env var, then a stored + // static api_key (which may be a stale broker-migrated copy) as a last resort. const oauthResolved = await this.#resolveOAuthSelection(provider, sessionId, options); if (oauthResolved) { return oauthResolved.apiKey; } - // Fall back to environment variable or custom resolver. If we reach here after - // an OAuth miss, the session sticky (if any) is stale — the request will - // authenticate via env/fallback, not OAuth, so clear the sticky now so that - // getOAuthAccountId() correctly suppresses account_uuid for this session. + // Past OAuth: the session sticky (if any) is stale — the request authenticates via + // env/api_key/fallback, not OAuth, so clear it now so getOAuthAccountId() correctly + // suppresses account_uuid for this session. if (sessionId) this.#sessionLastCredential.get(provider)?.delete(sessionId); + const envKey = getEnvApiKey(provider); if (envKey) return envKey; + const apiKeySelection = this.#selectCredentialByType(provider, "api_key", sessionId); + if (apiKeySelection) { + this.#recordSessionCredential(provider, sessionId, "api_key", apiKeySelection.index); + return this.#configValueResolver(apiKeySelection.credential.key); + } + // Fall back to custom resolver (e.g., models.json custom providers) return this.#fallbackResolver?.(provider) ?? undefined; } @@ -4486,12 +4488,13 @@ export class AuthStorage { /** * Describe where the active credential for a provider came from. * - * Surfaces four layers, highest precedence first: + * Mirrors {@link AuthStorage.getApiKey} precedence, highest first: * 1. Runtime override (`--api-key`). * 2. Config override (`models.yml` `providers..apiKey`). - * 3. Stored credential (the one this session is currently sticky to, or the - * one round-robin would pick next when no session id is supplied). - * 4. Env var / fallback resolver — when no stored credential exists. + * 3. Stored OAuth credential. + * 4. Env var — overrides a stored static api_key (e.g. a stale broker copy). + * 5. Stored api_key credential. + * 6. Fallback resolver. * * The string is purely informational; consumers must not parse it. */ @@ -4505,30 +4508,31 @@ export class AuthStorage { const baseLabel = this.#sourceLabel ?? "local store"; const stored = this.#getStoredCredentials(provider); - if (stored.length === 0) { - if (getEnvApiKey(provider)) return `env ${baseLabel ? `(fallback over ${baseLabel})` : ""}`.trim(); - if (this.#fallbackResolver?.(provider) !== undefined) return `fallback resolver`; - return undefined; - } - const session = sessionId ? this.#sessionLastCredential.get(provider)?.get(sessionId) : undefined; - // Same selection logic as #selectCredentialByType for "no session" lookups: prefer - // the type with stored credentials, lean OAuth before api_key. We don't run the - // full round-robin here because describing the source shouldn't advance the index. - const preferredType: AuthCredential["type"] = - session?.type ?? (stored.some(entry => entry.credential.type === "oauth") ? "oauth" : "api_key"); - const typed = stored - .map((entry, index) => ({ entry, index })) - .filter(({ entry }) => entry.credential.type === preferredType); - if (typed.length === 0) return baseLabel; - const index = session?.index ?? typed[0].index; - const chosen = stored[index] ?? typed[0].entry; - const credential = chosen.credential; - const identity = - credential.type === "oauth" - ? (credential.email ?? credential.accountId ?? credential.projectId ?? `cred ${chosen.id}`) - : `cred ${chosen.id}`; - return `${baseLabel} · ${preferredType} #${chosen.id} (${identity})`; + // Describe the stored credential of a given type, honoring the session sticky index. + const describeStored = (type: AuthCredential["type"]): string | undefined => { + const typed = stored + .map((entry, index) => ({ entry, index })) + .filter(({ entry }) => entry.credential.type === type); + if (typed.length === 0) return undefined; + const index = session?.type === type ? session.index : typed[0].index; + const chosen = stored[index] ?? typed[0].entry; + const credential = chosen.credential; + const identity = + credential.type === "oauth" + ? (credential.email ?? credential.accountId ?? credential.projectId ?? `cred ${chosen.id}`) + : `cred ${chosen.id}`; + return `${baseLabel} · ${type} #${chosen.id} (${identity})`; + }; + + // A deliberate OAuth login wins; then an explicit env var; then a stored static api_key. + const oauthSource = describeStored("oauth"); + if (oauthSource) return oauthSource; + if (getEnvApiKey(provider)) return `env (over ${baseLabel})`; + const apiKeySource = describeStored("api_key"); + if (apiKeySource) return apiKeySource; + if (this.#fallbackResolver?.(provider) !== undefined) return "fallback resolver"; + return undefined; } } diff --git a/packages/ai/src/error/flags.ts b/packages/ai/src/error/flags.ts index 16294594b..9c9f4a897 100644 --- a/packages/ai/src/error/flags.ts +++ b/packages/ai/src/error/flags.ts @@ -94,6 +94,14 @@ const MALFORMED_FUNCTION_CALL_PATTERN = /\bmalformed.?function.?call\b/i; const PROVIDER_FINISH_ERROR_PATTERN = /\bProvider (?:returned error finish_reason|finish_reason:\s*error)\b/i; const STALE_RESPONSE_ITEM_PATTERNS = [/\bItem with id ['"][^'"]+['"] not found\.?/i, /previous[ _]?response/i] as const; const STALE_RESPONSE_ITEM_DETAIL_PATTERN = /not[ _]?found|invalid|expired|stale|zero[ _-]?data[ _-]?retention/i; +/** + * Local llama.cpp / Ollama deterministic tool-call argument JSON parse failure. + * The model emitted invalid JSON in a tool call and the server returned HTTP 500 + * with this exact text — replaying the same prompt yields the same malformed + * output, so callers strip {@link Flag.Transient} when this matches. + */ +export const LLAMA_CPP_TOOL_CALL_PARSE_PATTERN = + /failed to parse tool call arguments as json|\[json\.exception\.parse_error\.101\]/i; // Copilot routing flap: HTTP 400 `model_not_supported` (structural code on the // error, also surfaced in text). Treated as transient — a retry usually lands @@ -441,7 +449,13 @@ export function classifyMessage(message: { const currentStatus = message.errorStatus ?? statusFromId(existingId); const textId = classifyText(message.errorMessage, currentStatus, message.api); - const kinds = ((existingId ?? 0) | textId) & KIND_MASK; + let kinds = ((existingId ?? 0) | textId) & KIND_MASK; + if (message.errorMessage && LLAMA_CPP_TOOL_CALL_PARSE_PATTERN.test(message.errorMessage)) { + // Deterministic local-model tool-call JSON parse failure: HTTP 500 is misleading + // because the same prompt reproduces the same malformed output, so the agent-level + // auto-retry would loop. Strip Transient so the recovery message surfaces immediately. + kinds &= ~Flag.Transient; + } const id = kinds !== 0 ? create(kinds) : (statusFromId(textId) ?? statusFromId(existingId) ?? currentStatus ?? 0); message.errorId = id; diff --git a/packages/ai/src/error/format.ts b/packages/ai/src/error/format.ts index 7e9765c8c..58360712f 100644 --- a/packages/ai/src/error/format.ts +++ b/packages/ai/src/error/format.ts @@ -5,6 +5,12 @@ import { rewriteCopilotError, } from "../utils/http-inspector"; import { formatErrorMessageWithRetryAfter } from "../utils/retry-after"; +import { LLAMA_CPP_TOOL_CALL_PARSE_PATTERN } from "./flags"; + +function rewriteOllamaToolCallJsonError(message: string): string { + if (!LLAMA_CPP_TOOL_CALL_PARSE_PATTERN.test(message)) return message; + return `Local Ollama model emitted malformed tool-call JSON and llama.cpp rejected it (HTTP 500). This is usually a deterministic model-output failure after context degradation, not a transient server outage; reload the model or reduce context, then retry.\n${message}`; +} /** Inputs that steer {@link formatMessage}'s formatter selection. */ export interface FormatMessageOptions { @@ -12,7 +18,7 @@ export interface FormatMessageOptions { rawRequestDump?: RawHttpRequestDump; /** Captured non-2xx response body, appended to the message when available. */ capturedErrorResponse?: CapturedHttpErrorResponse; - /** Provider id; `"github-copilot"` triggers the copilot message rewrite. */ + /** Provider id; gates provider-specific user-facing rewrites. */ provider?: string; } @@ -32,5 +38,8 @@ export async function formatMessage(error: unknown, opts: FormatMessageOptions = if (opts.provider === "github-copilot") { message = rewriteCopilotError(message, error, opts.provider); } + if (opts.provider === "ollama") { + message = rewriteOllamaToolCallJsonError(message); + } return message; } diff --git a/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts b/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts new file mode 100644 index 000000000..b76cbd6cb --- /dev/null +++ b/packages/ai/src/providers/__tests__/kimi-code-thinking.test.ts @@ -0,0 +1,112 @@ +import { describe, expect, it } from "bun:test"; +import { getBundledModel } from "@oh-my-pi/pi-catalog"; +import type { Context } from "../../types"; +import type { MessageCreateParamsStreaming } from "../anthropic-wire"; +import { streamOpenAIAnthropicShim } from "../openai-anthropic-shim"; +import { + applyChatCompletionsCompatPolicy, + type OpenAICompletionsParams, + resolveOpenAICompatPolicy, +} from "../openai-shared"; + +const BASE_CHAT_COMPLETIONS_PARAMS: OpenAICompletionsParams = { messages: [], model: "unused", stream: true }; +const TITLE_CONTEXT: Context = { + systemPrompt: ["Generate a title."], + messages: [{ role: "user", content: "Explain the login failure", timestamp: 0 }], + tools: [ + { + name: "set_title", + description: "Set title", + parameters: { + type: "object", + properties: { title: { type: "string" } }, + required: ["title"], + additionalProperties: false, + }, + }, + ], +}; + +describe("Kimi K2.7 Code thinking policy", () => { + it("omits disabled thinking for title-generator-style Kimi Code requests", () => { + const model = getBundledModel<"openai-completions">("kimi-code", "kimi-for-coding"); + const policy = resolveOpenAICompatPolicy(model, { + endpoint: "chat-completions", + disableReasoning: true, + toolChoice: { type: "tool", name: "set_title" }, + }); + const params = { ...BASE_CHAT_COMPLETIONS_PARAMS }; + + applyChatCompletionsCompatPolicy(params, policy); + + expect("thinking" in params).toBe(false); + expect(model.compat.supportsForcedToolChoice).toBe(false); + }); + + it("enables thinking and downgrades forced tool choice on Kimi Code's Anthropic endpoint", async () => { + const model = getBundledModel<"openai-completions">("kimi-code", "kimi-for-coding"); + let payload: MessageCreateParamsStreaming | undefined; + const stream = streamOpenAIAnthropicShim( + model, + TITLE_CONTEXT, + { + apiKey: "test-key", + maxTokens: 1024, + disableReasoning: true, + toolChoice: { type: "tool", name: "set_title" }, + onPayload: body => { + payload = body as MessageCreateParamsStreaming; + throw new Error("stop after payload capture"); + }, + }, + { + anthropicBaseUrl: "https://api.kimi.com/coding", + defaultFormat: "anthropic", + }, + ); + + await stream.result(); + + expect(payload?.thinking?.type).toBe("enabled"); + expect(payload?.tool_choice).toEqual({ type: "auto" }); + }); + + it("omits disabled thinking for native Moonshot Kimi K2.7 Code variants", () => { + for (const modelId of ["kimi-k2.7-code", "kimi-k2.7-code-highspeed"]) { + const model = getBundledModel<"openai-completions">("moonshot", modelId); + const policy = resolveOpenAICompatPolicy(model, { + endpoint: "chat-completions", + disableReasoning: true, + }); + const params = { ...BASE_CHAT_COMPLETIONS_PARAMS }; + applyChatCompletionsCompatPolicy(params, policy); + + expect("thinking" in params).toBe(false); + expect(model.compat.supportsForcedToolChoice).toBe(false); + } + }); + + it("keeps the openai disable shape for non-native Kimi K2.7 Code aliases", () => { + for (const { provider, id } of [ + { provider: "fireworks", id: "kimi-k2.7-code" }, + { provider: "openrouter", id: "moonshotai/kimi-k2.7-code" }, + ] as const) { + const model = getBundledModel<"openai-completions">(provider, id); + expect(model.compat.supportsForcedToolChoice).toBe(true); + expect(model.compat.reasoningDisableMode).not.toBe("omit"); + } + }); + + it("keeps explicit disabled thinking for Kimi K2.6", () => { + const model = getBundledModel<"openai-completions">("moonshot", "kimi-k2.6"); + const policy = resolveOpenAICompatPolicy(model, { + endpoint: "chat-completions", + disableReasoning: true, + }); + const params = { ...BASE_CHAT_COMPLETIONS_PARAMS }; + + applyChatCompletionsCompatPolicy(params, policy); + + expect(params.thinking).toEqual({ type: "disabled" }); + }); +}); diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 442cfd59e..14b027dab 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -43,7 +43,6 @@ import type { ToolResultMessage, Usage, } from "../types"; -import { resolveServiceTier } from "../types"; import { isRecord, normalizeSystemPrompts, normalizeToolCallId, resolveCacheRetention } from "../utils"; import { createAbortSourceTracker } from "../utils/abort"; import { @@ -1595,7 +1594,7 @@ const streamAnthropicOnce = ( isOAuthToken = false; } else { const extraBetas = normalizeExtraBetas(options?.betas); - const wantsAnthropicPriority = resolveServiceTier(options?.serviceTier, model.provider) === "priority"; + const wantsAnthropicPriority = model.provider === "anthropic" && options?.serviceTier === "priority"; // Skip the fast-mode beta when this session already learned the // endpoint+model rejects fast mode; `speed` is dropped from the params // too (dropFastMode), so the request stays a faithful non-fast request. @@ -2191,7 +2190,8 @@ const streamAnthropicOnce = ( } if ( !dropFastMode && - resolveServiceTier(options?.serviceTier, model.provider) === "priority" && + model.provider === "anthropic" && + options?.serviceTier === "priority" && firstTokenTime === undefined && AIError.isFastModeUnsupported(streamFailure) ) { @@ -2259,7 +2259,7 @@ const streamAnthropicOnce = ( } output.duration = performance.now() - startTime; if (firstTokenTime) output.ttft = firstTokenTime - startTime; - if (dropFastMode && resolveServiceTier(options?.serviceTier, model.provider) === "priority") { + if (dropFastMode && model.provider === "anthropic" && options?.serviceTier === "priority") { output.disabledFeatures = [...(output.disabledFeatures ?? []), "priority"]; } stream.push({ type: "done", reason: output.stopReason, message: output }); @@ -2876,9 +2876,10 @@ function buildParams( let thinking: MessageCreateParamsStreaming["thinking"] | undefined; let outputConfigEffort: AnthropicOutputEffort | undefined; if (model.reasoning) { - if (options?.thinkingEnabled) { + if (options?.thinkingEnabled || model.compat.requiresThinkingEnabled) { + const thinkingOptions = options ?? {}; const mode = model.thinking?.mode; - const effort = resolveAnthropicAdaptiveEffort(model, options); + const effort = resolveAnthropicAdaptiveEffort(model, thinkingOptions); const compat = model.compat; if (mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) { const adaptive: { type: "adaptive"; display?: AnthropicThinkingDisplay } = { type: "adaptive" }; @@ -2889,15 +2890,15 @@ function buildParams( // support: Opus 4.6 / Sonnet 4.6+ reject it with a 400, so an explicit // `thinkingDisplay` MUST NOT force it onto a model that can't accept it. if (model.thinking?.supportsDisplay) { - adaptive.display = options.thinkingDisplay ?? "summarized"; + adaptive.display = thinkingOptions.thinkingDisplay ?? "summarized"; } thinking = adaptive; if (effort && effort !== "adaptive") outputConfigEffort = effort; } else { thinking = { type: "enabled", - budget_tokens: options.thinkingBudgetTokens || 1024, - display: options.thinkingDisplay ?? "summarized", + budget_tokens: thinkingOptions.thinkingBudgetTokens || 1024, + display: thinkingOptions.thinkingDisplay ?? "summarized", }; if (mode === "anthropic-budget-effort" && effort && effort !== "adaptive") outputConfigEffort = effort; } @@ -2996,7 +2997,7 @@ function buildParams( seqs.length > ANTHROPIC_STOP_SEQUENCES_MAX ? seqs.slice(0, ANTHROPIC_STOP_SEQUENCES_MAX) : seqs; } - if (resolveServiceTier(options?.serviceTier, model.provider) === "priority") { + if (model.provider === "anthropic" && options?.serviceTier === "priority") { params.speed = "fast"; } diff --git a/packages/ai/src/providers/google-auth.ts b/packages/ai/src/providers/google-auth.ts index 11a2cd6c4..7aabaf5cc 100644 --- a/packages/ai/src/providers/google-auth.ts +++ b/packages/ai/src/providers/google-auth.ts @@ -13,6 +13,7 @@ */ import { Buffer } from "node:buffer"; +import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { $envpos, isEnoent, logger } from "@oh-my-pi/pi-utils"; @@ -281,6 +282,11 @@ const SHARED_TOKEN_RESOLVE_TIMEOUT_MS = 30_000; * The token is cached in module scope and refreshed `GOOGLE_VERTEX_REFRESH_SKEW_MS` ms before it expires. */ export async function getVertexAccessToken(options?: { signal?: AbortSignal; fetch?: FetchImpl }): Promise { + // An explicit access token (e.g. `gcloud auth print-access-token`) bypasses the cache so a + // refreshed env token takes effect immediately. `CLOUDSDK_AUTH_ACCESS_TOKEN` is gcloud's own + // override var; `GOOGLE_CLOUD_ACCESS_TOKEN` is the omp-facing alias. + const explicitToken = Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN || Bun.env.CLOUDSDK_AUTH_ACCESS_TOKEN; + if (explicitToken) return explicitToken; const fetchImpl = options?.fetch ?? globalThis.fetch.bind(globalThis); const skew = getRefreshSkewMs(); const now = Date.now(); @@ -323,3 +329,22 @@ export function __resetVertexTokenCache(): void { tokenCache.clear(); inflight.clear(); } + +/** + * Sync best-effort probe for a usable Vertex bearer credential source — an explicit access-token + * env var, `GOOGLE_APPLICATION_CREDENTIALS`, a user ADC file, or a GCP runtime whose metadata + * server can mint ADC (GCE/Cloud Run/App Engine/Functions). Lets callers prefer the bearer + * Interactions transport only when ADC is actually reachable, without paying the async + * metadata-probe cost for API-key-only setups. + */ +export function hasVertexBearerCredentialsHint(): boolean { + if (Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN || Bun.env.CLOUDSDK_AUTH_ACCESS_TOKEN) return true; + if (Bun.env.GOOGLE_APPLICATION_CREDENTIALS) return true; + // GCP-hosted runtimes expose ADC via the metadata server; these env vars mark those runtimes. + if (Bun.env.K_SERVICE || Bun.env.FUNCTION_TARGET || Bun.env.GAE_ENV || Bun.env.GCE_METADATA_HOST) return true; + try { + return fs.existsSync(userAdcPath()); + } catch { + return false; + } +} diff --git a/packages/ai/src/providers/google-gemini-cli.ts b/packages/ai/src/providers/google-gemini-cli.ts index 9338eb972..adbe02a87 100644 --- a/packages/ai/src/providers/google-gemini-cli.ts +++ b/packages/ai/src/providers/google-gemini-cli.ts @@ -35,6 +35,7 @@ import { armPreResponseTimeout, getStreamFirstEventTimeoutMs } from "../utils/id // Refresh is the sole responsibility of AuthStorage (broker-aware, single-flighted); // the stream provider trusts the access token threaded through `options.apiKey`. import { normalizeSchemaForCCA } from "../utils/schema"; +import { StreamMarkupHealing, type StreamMarkupHealingEvent } from "../utils/stream-markup-healing"; import type { Content, FunctionCallingConfigMode, ThinkingConfig } from "./google-shared"; import { convertMessages, @@ -665,15 +666,43 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( let currentBlock: TextContent | ThinkingContent | null = null; const blocks = output.content; const blockIndex = () => blocks.length - 1; + const visibleTextHealing = new StreamMarkupHealing({ pattern: "thinking" }); let isBuffering = false; let textBuffer = ""; let bufferedTextSignature: string | undefined; - const emitVisibleText = (delta: string, thoughtSignature?: string) => { - if (!delta || !currentBlock || currentBlock.type !== "text") return; - currentBlock.text += delta; - currentBlock.textSignature = retainThoughtSignature(currentBlock.textSignature, thoughtSignature); + const endCurrentBlock = (): void => { + if (!currentBlock) return; + pushBlockEndEvent(currentBlock, blockIndex(), output, stream); + currentBlock = null; + }; + + const startTextBlock = (): TextContent => { + let block = currentBlock; + if (block?.type !== "text") { + endCurrentBlock(); + block = startTextOrThinkingBlock(false, output, stream, ensureStarted); + currentBlock = block; + } + return block; + }; + + const startThinkingBlock = (): ThinkingContent => { + let block = currentBlock; + if (block?.type !== "thinking") { + endCurrentBlock(); + block = startTextOrThinkingBlock(true, output, stream, ensureStarted); + currentBlock = block; + } + return block; + }; + + const emitVisibleText = (delta: string, thoughtSignature?: string): void => { + if (!delta) return; + const block = startTextBlock(); + block.text += delta; + block.textSignature = retainThoughtSignature(block.textSignature, thoughtSignature); stream.push({ type: "text_delta", contentIndex: blockIndex(), @@ -682,6 +711,48 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( }); }; + const emitVisibleThinking = (delta: string): void => { + if (!delta) return; + const block = startThinkingBlock(); + block.thinking += delta; + stream.push({ + type: "thinking_delta", + contentIndex: blockIndex(), + delta, + partial: output, + }); + }; + + const emitHealingEvent = (event: StreamMarkupHealingEvent, thoughtSignature?: string): void => { + if (event.type === "text") { + emitVisibleText(event.text, thoughtSignature); + } else if (event.type === "thinking") { + emitVisibleThinking(event.thinking); + } + }; + + const feedVisibleText = (delta: string, thoughtSignature?: string): void => { + for (const event of visibleTextHealing.feedEvents(delta)) { + emitHealingEvent(event, thoughtSignature); + } + }; + + const flushVisibleText = (thoughtSignature?: string): void => { + for (const event of visibleTextHealing.flushEvents()) { + emitHealingEvent(event, thoughtSignature); + } + }; + + const retainCurrentBlockThoughtSignature = (thoughtSignature: string): void => { + const block = currentBlock; + if (!block) return; + if (block.type === "thinking") { + block.thinkingSignature = retainThoughtSignature(block.thinkingSignature, thoughtSignature); + } else { + block.textSignature = retainThoughtSignature(block.textSignature, thoughtSignature); + } + }; + for await (const chunk of readSseJson( activeResponse.body!, options?.signal, @@ -710,20 +781,12 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( for (const part of candidate.content.parts) { if (part.text !== undefined && part.text !== "") { const isThinking = isThinkingPart(part); - if ( - !currentBlock || - (isThinking && currentBlock.type !== "thinking") || - (!isThinking && currentBlock.type !== "text") - ) { - if (currentBlock) { - pushBlockEndEvent(currentBlock, blockIndex(), output, stream); - } - currentBlock = startTextOrThinkingBlock(isThinking, output, stream, ensureStarted); - } - if (currentBlock.type === "thinking") { - currentBlock.thinking += part.text; - currentBlock.thinkingSignature = retainThoughtSignature( - currentBlock.thinkingSignature, + if (isThinking) { + flushVisibleText(); + const block = startThinkingBlock(); + block.thinking += part.text; + block.thinkingSignature = retainThoughtSignature( + block.thinkingSignature, part.thoughtSignature, ); stream.push({ @@ -744,7 +807,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( textBuffer = part.text; bufferedTextSignature = part.thoughtSignature; } else { - emitVisibleText(part.text, part.thoughtSignature); + feedVisibleText(part.text, part.thoughtSignature); } if (isBuffering) { @@ -757,32 +820,19 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( isBuffering = false; textBuffer = ""; bufferedTextSignature = undefined; - emitVisibleText(buffered.visibleText, visibleSignature); + feedVisibleText(buffered.visibleText, visibleSignature); } } } - } else if (part.text === "" && part.thoughtSignature && currentBlock && !part.functionCall) { - if (currentBlock.type === "thinking") { - currentBlock.thinkingSignature = retainThoughtSignature( - currentBlock.thinkingSignature, - part.thoughtSignature, - ); - } else { - currentBlock.textSignature = retainThoughtSignature( - currentBlock.textSignature, - part.thoughtSignature, - ); - } + } else if (part.text === "" && part.thoughtSignature && !part.functionCall) { + retainCurrentBlockThoughtSignature(part.thoughtSignature); } if (part.functionCall) { - if (currentBlock) { - pushBlockEndEvent(currentBlock, blockIndex(), output, stream); - currentBlock = null; - } + flushVisibleText(); + endCurrentBlock(); isBuffering = false; textBuffer = ""; - const providedId = part.functionCall.id; const needsNewId = !providedId || output.content.some(b => b.type === "toolCall" && b.id === providedId); @@ -848,16 +898,15 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = ( sawLeak = true; } if (buffered.kind !== "incomplete") { - emitVisibleText(buffered.visibleText, bufferedTextSignature); + feedVisibleText(buffered.visibleText, bufferedTextSignature); } bufferedTextSignature = undefined; isBuffering = false; textBuffer = ""; } - if (currentBlock) { - pushBlockEndEvent(currentBlock, blockIndex(), output, stream); - } + flushVisibleText(bufferedTextSignature); + endCurrentBlock(); return hasMeaningfulGoogleContent(output) || sawLeak; }; diff --git a/packages/ai/src/providers/google-interactions.ts b/packages/ai/src/providers/google-interactions.ts new file mode 100644 index 000000000..3f5612809 --- /dev/null +++ b/packages/ai/src/providers/google-interactions.ts @@ -0,0 +1,753 @@ +import { parseGeminiModel } from "@oh-my-pi/pi-catalog/identity"; +import { calculateCost } from "@oh-my-pi/pi-catalog/models"; +import { fetchWithRetry, readSseJson } from "@oh-my-pi/pi-utils"; +import * as AIError from "../error"; +import type { + AssistantMessage, + Context, + FetchImpl, + ImageContent, + Message, + Model, + ProviderSessionState, + TextContent, + ToolCall, + Usage, +} from "../types"; +import { shouldSendServiceTier } from "../types"; +import { normalizeSystemPrompts } from "../utils"; +import { AssistantMessageEventStream } from "../utils/event-stream"; +import { convertTools, type GoogleSharedStreamOptions, type GoogleThinkingLevel } from "./google-shared"; + +type GoogleInteractionsApi = "google-generative-ai" | "google-vertex"; +type GoogleInteractionsModel = Model; +type GoogleOptions = GoogleSharedStreamOptions; + +const GOOGLE_INTERACTIONS_STATE_KEY = "google-interactions-state"; + +/** Provider session state storing the last Gemini Interactions response id. */ +export interface GoogleInteractionsProviderSessionState extends ProviderSessionState { + lastInteractionId?: string; +} + +/** Conversation anchor for continuing an Interactions turn from a prior assistant response. */ +export interface InteractionAnchor { + id?: string; + messageIndex?: number; +} + +type InteractionContent = { type: "text"; text: string } | { type: "image"; data: string; mime_type: string }; + +interface InteractionUserInputStep { + type: "user_input"; + content: InteractionContent[]; +} + +interface InteractionModelOutputStep { + type: "model_output"; + content: InteractionContent[]; +} + +interface InteractionFunctionCallStep { + type: "function_call"; + id: string; + name: string; + arguments: Record; +} + +interface InteractionFunctionResultStep { + type: "function_result"; + name: string; + call_id: string; + result: InteractionContent[]; + is_error?: boolean; +} + +interface InteractionThoughtStep { + type: "thought"; + summary?: InteractionContent[]; + signature?: string; +} + +interface InteractionThoughtSummaryDelta { + type: "thought_summary"; + content?: InteractionContent; +} + +interface InteractionThoughtSignatureDelta { + type: "thought_signature"; + signature?: string; +} + +interface InteractionArgumentsDelta { + type: "arguments_delta"; + arguments?: string; +} + +interface PendingInteractionToolCall { + id: string; + name: string; + argumentsText: string; + argumentsObject: Record; +} + +type InteractionInputStep = + | InteractionUserInputStep + | InteractionModelOutputStep + | InteractionFunctionCallStep + | InteractionFunctionResultStep; + +type InteractionStep = + | InteractionModelOutputStep + | InteractionFunctionCallStep + | InteractionFunctionResultStep + | InteractionThoughtStep + | InteractionUserInputStep; + +type InteractionThinkingLevel = "minimal" | "low" | "medium" | "high"; + +interface InteractionGenerationConfig { + temperature?: number; + top_p?: number; + top_k?: number; + min_p?: number; + presence_penalty?: number; + frequency_penalty?: number; + repetition_penalty?: number; + max_output_tokens?: number; + thinking_level?: InteractionThinkingLevel; + thinking_budget?: number; +} + +interface GoogleInteractionRequest { + model: string; + input: InteractionInputStep[]; + stream: true; + previous_interaction_id?: string; + system_instruction?: string; + tools?: { functionDeclarations: Record[] }[]; + store?: boolean; + generation_config?: InteractionGenerationConfig; + service_tier?: string; +} + +interface InteractionUsage { + total_input_tokens?: number; + total_cached_tokens?: number; + total_output_tokens?: number; + total_thought_tokens?: number; + total_tokens?: number; +} + +interface InteractionResource { + id?: string; + status?: string; + usage?: InteractionUsage; +} + +interface InteractionStreamMetadata { + total_usage?: InteractionUsage; +} + +interface InteractionSseEvent { + event_type?: string; + index?: number; + step?: InteractionStep; + delta?: + | InteractionContent + | InteractionFunctionCallStep + | InteractionThoughtStep + | InteractionThoughtSummaryDelta + | InteractionThoughtSignatureDelta + | InteractionArgumentsDelta; + interaction?: InteractionResource; + interaction_id?: string; + status?: string; + metadata?: InteractionStreamMetadata; + error?: { message?: string; code?: string | number }; +} + +function emptyUsage(): Usage { + return { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }; +} + +function getGoogleInteractionsState( + providerSessionState: Map | undefined, + create: boolean, +): GoogleInteractionsProviderSessionState | undefined { + if (!providerSessionState) return undefined; + const existing = providerSessionState.get(GOOGLE_INTERACTIONS_STATE_KEY) as + | GoogleInteractionsProviderSessionState + | undefined; + if (existing || !create) return existing; + const state: GoogleInteractionsProviderSessionState = { close: () => {} }; + providerSessionState.set(GOOGLE_INTERACTIONS_STATE_KEY, state); + return state; +} + +function interactionContentFromText(text: string): InteractionContent[] { + return text.length === 0 ? [] : [{ type: "text", text }]; +} + +function interactionContentFromParts(parts: readonly (TextContent | ImageContent)[]): InteractionContent[] { + const content: InteractionContent[] = []; + for (const part of parts) { + if (part.type === "text") { + if (part.text.length > 0) content.push({ type: "text", text: part.text }); + } else { + content.push({ type: "image", data: part.data, mime_type: part.mimeType }); + } + } + return content; +} + +function userInputStepFromMessage(message: Extract): InteractionUserInputStep { + const content = + typeof message.content === "string" + ? interactionContentFromText(message.content) + : interactionContentFromParts(message.content); + return { type: "user_input", content }; +} + +function functionResultStepFromMessage( + message: Extract, +): InteractionFunctionResultStep { + const result = interactionContentFromParts(message.content); + return { + type: "function_result", + name: message.toolName, + call_id: message.toolCallId, + result: result.length > 0 ? result : [{ type: "text", text: "" }], + ...(message.isError ? { is_error: true } : {}), + }; +} + +function appendAssistantInteractionSteps(message: AssistantMessage, steps: InteractionInputStep[]): void { + let modelContent: InteractionContent[] = []; + const flushModelContent = (): void => { + if (modelContent.length === 0) return; + steps.push({ type: "model_output", content: modelContent }); + modelContent = []; + }; + + for (const block of message.content) { + if (block.type === "text") { + if (block.text.length > 0) modelContent.push({ type: "text", text: block.text }); + } else if (block.type === "toolCall") { + flushModelContent(); + steps.push({ type: "function_call", id: block.id, name: block.name, arguments: block.arguments }); + } + } + flushModelContent(); +} + +function interactionMessagesAfterAnchor( + messages: readonly Message[], + anchorIndex: number | undefined, +): readonly Message[] { + return anchorIndex === undefined ? messages : messages.slice(anchorIndex + 1); +} + +function buildInteractionInput(context: Context, anchorIndex: number | undefined): InteractionInputStep[] { + const input: InteractionInputStep[] = []; + for (const message of interactionMessagesAfterAnchor(context.messages, anchorIndex)) { + if (message.role === "user" || message.role === "developer") { + const step = userInputStepFromMessage(message); + if (step.content.length > 0) input.push(step); + } else if (message.role === "toolResult") { + input.push(functionResultStepFromMessage(message)); + } else if (anchorIndex === undefined) { + appendAssistantInteractionSteps(message, input); + } + } + return input.length > 0 ? input : [{ type: "user_input", content: [{ type: "text", text: "" }] }]; +} + +function toInteractionThinkingLevel(level: GoogleThinkingLevel): InteractionThinkingLevel | undefined { + switch (level) { + case "MINIMAL": + return "minimal"; + case "LOW": + return "low"; + case "MEDIUM": + return "medium"; + case "HIGH": + return "high"; + case "THINKING_LEVEL_UNSPECIFIED": + return undefined; + } +} + +function buildInteractionGenerationConfig(options: GoogleOptions | undefined): InteractionGenerationConfig | undefined { + const config: InteractionGenerationConfig = {}; + if (options?.temperature !== undefined) config.temperature = options.temperature; + if (options?.topP !== undefined) config.top_p = options.topP; + if (options?.topK !== undefined) config.top_k = options.topK; + if (options?.minP !== undefined) config.min_p = options.minP; + if (options?.presencePenalty !== undefined) config.presence_penalty = options.presencePenalty; + if (options?.frequencyPenalty !== undefined) config.frequency_penalty = options.frequencyPenalty; + if (options?.repetitionPenalty !== undefined) config.repetition_penalty = options.repetitionPenalty; + if (options?.maxTokens !== undefined) config.max_output_tokens = options.maxTokens; + if (options?.thinking?.level !== undefined) { + const thinkingLevel = toInteractionThinkingLevel(options.thinking.level); + if (thinkingLevel !== undefined) config.thinking_level = thinkingLevel; + } else if (options?.thinking?.budgetTokens !== undefined) { + config.thinking_budget = options.thinking.budgetTokens; + } + return Object.keys(config).length > 0 ? config : undefined; +} + +function buildInteractionRequest( + model: GoogleInteractionsModel, + context: Context, + options: GoogleOptions | undefined, + anchor: InteractionAnchor, +): GoogleInteractionRequest { + const systemInstruction = normalizeSystemPrompts(context.systemPrompt).join("\n\n"); + const generationConfig = buildInteractionGenerationConfig(options); + return { + model: model.id, + input: buildInteractionInput(context, anchor.messageIndex), + stream: true, + ...(anchor.id !== undefined ? { previous_interaction_id: anchor.id } : {}), + ...(systemInstruction.length > 0 ? { system_instruction: systemInstruction } : {}), + ...(context.tools && context.tools.length > 0 ? { tools: convertTools(context.tools, model) } : {}), + ...(options?.storeInteraction !== undefined ? { store: options.storeInteraction } : {}), + ...(generationConfig !== undefined ? { generation_config: generationConfig } : {}), + ...(shouldSendServiceTier(options?.serviceTier, model.provider) ? { service_tier: options?.serviceTier } : {}), + }; +} + +function applyInteractionUsage( + model: GoogleInteractionsModel, + output: AssistantMessage, + usage: InteractionUsage, +): void { + const thinkingTokens = usage.total_thought_tokens ?? 0; + output.usage = { + input: (usage.total_input_tokens ?? 0) - (usage.total_cached_tokens ?? 0), + output: (usage.total_output_tokens ?? 0) + thinkingTokens, + cacheRead: usage.total_cached_tokens ?? 0, + cacheWrite: 0, + totalTokens: usage.total_tokens ?? 0, + ...(thinkingTokens > 0 ? { reasoningTokens: thinkingTokens } : {}), + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }; + calculateCost(model, output.usage); +} + +function parseInteractionFunctionCall(value: unknown): InteractionFunctionCallStep | undefined { + if (!value || typeof value !== "object" || Array.isArray(value)) return undefined; + const record = value as Record; + if (record.type !== "function_call") return undefined; + if (typeof record.id !== "string" || typeof record.name !== "string") return undefined; + const args = record.arguments; + return { + type: "function_call", + id: record.id, + name: record.name, + arguments: args && typeof args === "object" && !Array.isArray(args) ? { ...args } : {}, + }; +} + +function pendingToolCallFromStep(call: InteractionFunctionCallStep): PendingInteractionToolCall { + return { + id: call.id, + name: call.name, + argumentsText: "", + argumentsObject: call.arguments, + }; +} + +function parseInteractionArguments(text: string): Record { + if (text.trim().length === 0) return {}; + try { + const parsed: unknown = JSON.parse(text); + if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) return { ...parsed }; + } catch { + return {}; + } + return {}; +} + +/** Provider-specific URL, headers, and fetch implementation for an Interactions request. */ +export interface GoogleInteractionsPlan { + url: string; + headers: Record; + fetch?: FetchImpl; +} + +/** + * Streams Gemini Interactions API model-mode responses for direct Google and Vertex providers. + * + * `fallback`, when supplied, is the legacy `:streamGenerateContent` stream factory. It runs + * transparently — forwarding its events into this stream — when the Interactions attempt fails + * before any content is emitted with a signal that the model/endpoint does not support + * Interactions (HTTP 404/400). Provide it only for auto-selected Interactions requests so an + * explicit `useInteractionsApi: true` still surfaces failures. + */ +export function streamGoogleInteractions(args: { + model: Model; + context: Context; + options: GoogleSharedStreamOptions | undefined; + api: T; + anchor: InteractionAnchor; + state: GoogleInteractionsProviderSessionState | undefined; + prepare: () => GoogleInteractionsPlan | Promise; + fallback?: () => AssistantMessageEventStream; +}): AssistantMessageEventStream { + const { model, context, options, anchor, state } = args; + const stream = new AssistantMessageEventStream(); + const output: AssistantMessage = { + role: "assistant", + content: [], + api: args.api, + provider: model.provider, + model: model.id, + usage: emptyUsage(), + stopReason: "stop", + timestamp: Date.now(), + }; + const storeInteraction = options?.storeInteraction !== false; + + void (async () => { + let started = false; + let sawTerminal = false; + let currentTextBlock: TextContent | undefined; + let currentThinkingBlock: Extract | undefined; + let pendingThinkingSignature: string | undefined; + const stepKinds = new Map(); + const pendingToolCalls = new Map(); + const ensureStarted = (): void => { + if (started) return; + stream.push({ type: "start", partial: output }); + started = true; + }; + const endOpenBlocks = (): void => { + if (currentTextBlock) { + stream.push({ + type: "text_end", + contentIndex: output.content.indexOf(currentTextBlock), + content: currentTextBlock.text, + partial: output, + }); + currentTextBlock = undefined; + } + if (currentThinkingBlock) { + stream.push({ + type: "thinking_end", + contentIndex: output.content.indexOf(currentThinkingBlock), + content: currentThinkingBlock.thinking, + partial: output, + }); + currentThinkingBlock = undefined; + } + }; + const emitText = (text: string): void => { + if (text.length === 0) return; + ensureStarted(); + if (!currentTextBlock) { + if (currentThinkingBlock) endOpenBlocks(); + currentTextBlock = { type: "text", text: "" }; + output.content.push(currentTextBlock); + stream.push({ type: "text_start", contentIndex: output.content.length - 1, partial: output }); + } + currentTextBlock.text += text; + stream.push({ + type: "text_delta", + contentIndex: output.content.indexOf(currentTextBlock), + delta: text, + partial: output, + }); + }; + const applyThinkingSignature = (signature: string | undefined): void => { + if (!signature) return; + if (currentThinkingBlock) { + currentThinkingBlock.thinkingSignature = signature; + } else { + pendingThinkingSignature = signature; + } + }; + const emitThinking = (text: string): void => { + if (text.length === 0) return; + ensureStarted(); + if (!currentThinkingBlock) { + if (currentTextBlock) endOpenBlocks(); + currentThinkingBlock = { type: "thinking", thinking: "", thinkingSignature: pendingThinkingSignature }; + pendingThinkingSignature = undefined; + output.content.push(currentThinkingBlock); + stream.push({ type: "thinking_start", contentIndex: output.content.length - 1, partial: output }); + } + currentThinkingBlock.thinking += text; + stream.push({ + type: "thinking_delta", + contentIndex: output.content.indexOf(currentThinkingBlock), + delta: text, + partial: output, + }); + }; + const emitToolCall = (call: InteractionFunctionCallStep): void => { + ensureStarted(); + endOpenBlocks(); + const toolCall: ToolCall = { + type: "toolCall", + id: call.id, + name: call.name, + arguments: call.arguments, + }; + output.content.push(toolCall); + const contentIndex = output.content.length - 1; + stream.push({ type: "toolcall_start", contentIndex, partial: output }); + stream.push({ + type: "toolcall_delta", + contentIndex, + delta: JSON.stringify(toolCall.arguments), + partial: output, + }); + stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); + }; + const emitPendingToolCall = (pending: PendingInteractionToolCall): void => { + emitToolCall({ + type: "function_call", + id: pending.id, + name: pending.name, + arguments: + pending.argumentsText.length > 0 + ? parseInteractionArguments(pending.argumentsText) + : pending.argumentsObject, + }); + }; + + let prepared = false; + try { + const plan = await args.prepare(); + prepared = true; + let requestBody: unknown = buildInteractionRequest(model, context, options, anchor); + const replacement = await options?.onPayload?.(requestBody, model); + if (replacement !== undefined) requestBody = replacement; + const response = await fetchWithRetry(() => plan.url, { + method: "POST", + headers: { + ...plan.headers, + "Content-Type": "application/json", + Accept: "text/event-stream", + }, + body: JSON.stringify(requestBody), + signal: options?.signal, + fetch: plan.fetch, + }); + if (!response.ok) { + const errorText = await response.text().catch(() => ""); + throw new AIError.GoogleApiError( + `Google Interactions API error (${response.status}): ${errorText}`, + response.status, + { + headers: response.headers, + }, + ); + } + if (!response.body) { + throw new AIError.ProviderResponseError("Google Interactions API returned an empty response body", { + provider: model.provider, + kind: "empty-body", + }); + } + for await (const event of readSseJson(response.body, options?.signal, sse => + options?.onSseEvent?.({ event: sse.event, data: sse.data, raw: [...sse.raw] }, model), + )) { + if (event.error) { + throw new AIError.ProviderResponseError(event.error.message ?? "Google Interactions API stream error", { + provider: model.provider, + kind: "runtime", + }); + } + if (event.metadata?.total_usage) applyInteractionUsage(model, output, event.metadata.total_usage); + if (event.event_type === "interaction.created") { + if (storeInteraction && event.interaction?.id) output.responseId = event.interaction.id; + } else if (event.event_type === "step.start" && event.index !== undefined && event.step) { + stepKinds.set(event.index, event.step.type); + const call = parseInteractionFunctionCall(event.step); + if (call) { + pendingToolCalls.set(event.index, pendingToolCallFromStep(call)); + } else if (event.step.type === "thought") { + applyThinkingSignature(event.step.signature); + for (const item of event.step.summary ?? []) { + if (item.type === "text") emitThinking(item.text); + } + } + } else if (event.event_type === "step.delta" && event.index !== undefined && event.delta) { + const call = parseInteractionFunctionCall(event.delta); + if (call) { + pendingToolCalls.set(event.index, pendingToolCallFromStep(call)); + } else if (event.delta.type === "text") { + if (stepKinds.get(event.index) === "thought") emitThinking(event.delta.text); + else emitText(event.delta.text); + } else if (event.delta.type === "thought_summary") { + if (event.delta.content?.type === "text") emitThinking(event.delta.content.text); + } else if (event.delta.type === "thought_signature") { + applyThinkingSignature(event.delta.signature); + } else if (event.delta.type === "arguments_delta") { + const pending = pendingToolCalls.get(event.index); + if (pending && event.delta.arguments) pending.argumentsText += event.delta.arguments; + } + } else if (event.event_type === "step.stop" && event.index !== undefined) { + const stepKind = stepKinds.get(event.index); + const pending = pendingToolCalls.get(event.index); + if (pending) { + emitPendingToolCall(pending); + pendingToolCalls.delete(event.index); + } else { + endOpenBlocks(); + if (stepKind === "thought") pendingThinkingSignature = undefined; + } + stepKinds.delete(event.index); + } else if (event.event_type === "interaction.completed" || event.event_type === "interaction.complete") { + if (storeInteraction && event.interaction?.id) output.responseId = event.interaction.id; + if (event.interaction?.usage) applyInteractionUsage(model, output, event.interaction.usage); + for (const pending of pendingToolCalls.values()) emitPendingToolCall(pending); + pendingToolCalls.clear(); + endOpenBlocks(); + output.stopReason = + event.interaction?.status === "requires_action" || + output.content.some(block => block.type === "toolCall") + ? "toolUse" + : "stop"; + if (storeInteraction && state) state.lastInteractionId = output.responseId; + sawTerminal = true; + ensureStarted(); + stream.push({ type: "done", reason: output.stopReason, message: output }); + } + } + if (!sawTerminal) { + throw new AIError.ProviderResponseError("Google Interactions API stream ended without a terminal event", { + provider: model.provider, + kind: "incomplete-stream", + }); + } + } catch (error) { + // Auto-selected Interactions degrades to `:streamGenerateContent` when no content has + // streamed yet and the failure means Interactions can't serve this request — the bearer + // credential couldn't be resolved (prepare threw) or the model/endpoint rejected it + // (HTTP 404/400). Provider 401/403/429/5xx still surface. Mirrors the OpenAI Responses + // `previous_response_id` fallback. + const unsupported = error instanceof AIError.GoogleApiError && (error.status === 404 || error.status === 400); + if (!started && args.fallback && !options?.signal?.aborted && (!prepared || unsupported)) { + for await (const event of args.fallback()) stream.push(event); + return; + } + output.stopReason = options?.signal?.aborted ? "aborted" : "error"; + output.errorMessage = error instanceof Error ? error.message : String(error); + stream.push({ type: "error", reason: output.stopReason, error: output }); + } + })(); + + return stream; +} + +function findAssistantInteractionAnchor( + context: Context, + interactionId: string, + provider: string, +): InteractionAnchor | undefined { + for (let index = context.messages.length - 1; index >= 0; index -= 1) { + const message = context.messages[index]; + if (message?.role === "assistant" && message.provider === provider && message.responseId === interactionId) { + return { id: interactionId, messageIndex: index }; + } + } + return undefined; +} + +function latestAssistantInteractionAnchor(context: Context, provider: string): InteractionAnchor | undefined { + for (let index = context.messages.length - 1; index >= 0; index -= 1) { + const message = context.messages[index]; + if (message?.role === "assistant" && message.provider === provider && message.responseId) { + return { id: message.responseId, messageIndex: index }; + } + } + return undefined; +} + +function resolveInteractionAnchor( + context: Context, + explicitPreviousInteractionId: string | undefined, + state: GoogleInteractionsProviderSessionState | undefined, + provider: string, +): InteractionAnchor { + if (explicitPreviousInteractionId !== undefined) { + return ( + findAssistantInteractionAnchor(context, explicitPreviousInteractionId, provider) ?? { + id: explicitPreviousInteractionId, + } + ); + } + const lineageAnchor = latestAssistantInteractionAnchor(context, provider); + if (lineageAnchor) return lineageAnchor; + if (state?.lastInteractionId) + return findAssistantInteractionAnchor(context, state.lastInteractionId, provider) ?? {}; + return {}; +} + +/** + * Whether a model is served by the Gemini Interactions API. Interactions is a Gemini 3-era + * transport, so the catalog subset that supports it is Gemini 3.0+. Older Gemini and non-Gemini + * ids keep `:streamGenerateContent`, which covers the full catalog. + */ +export function modelSupportsInteractions(model: Pick): boolean { + const parsed = parseGeminiModel(model.id); + return parsed !== null && parsed.version.major >= 3; +} + +/** + * Resolves whether a Google provider call should use Interactions and which lineage anchor to send. + * + * Precedence: explicit `useInteractionsApi: false` always wins (force generateContent); otherwise + * Interactions engages when explicitly requested, when continuing a stored interaction + * (`previousInteractionId`/assistant lineage/session state), or when `autoEligible` (the + * zero-config default for the capable model subset on the official endpoint). `auto` flags the + * last case for the caller — it is the only mode that wires up the generateContent fallback. + */ +export function resolveInteractionDispatch(args: { + context: Context; + options: GoogleSharedStreamOptions | undefined; + provider: string; + autoEligible: boolean; +}): { + useInteractions: boolean; + auto: boolean; + anchor: InteractionAnchor; + state: GoogleInteractionsProviderSessionState | undefined; +} { + const explicitPreviousInteractionId = args.options?.previousInteractionId; + if (args.options?.storeInteraction === false && explicitPreviousInteractionId !== undefined) { + throw new AIError.ConfigurationError( + "Google Interactions API cannot combine storeInteraction:false with previousInteractionId.", + ); + } + const explicitOptOut = args.options?.useInteractionsApi === false; + const explicitOptIn = args.options?.useInteractionsApi === true || explicitPreviousInteractionId !== undefined; + const storageEnabled = args.options?.storeInteraction !== false; + const existingState = storageEnabled + ? getGoogleInteractionsState(args.options?.providerSessionState, false) + : undefined; + const anchor = explicitOptOut + ? {} + : resolveInteractionAnchor(args.context, explicitPreviousInteractionId, existingState, args.provider); + const useInteractions = !explicitOptOut && (explicitOptIn || anchor.id !== undefined || args.autoEligible); + const auto = useInteractions && !explicitOptIn; + const interactionState = + useInteractions && storageEnabled + ? getGoogleInteractionsState( + args.options?.providerSessionState, + args.options?.providerSessionState !== undefined, + ) + : undefined; + return { useInteractions, auto, anchor, state: interactionState }; +} diff --git a/packages/ai/src/providers/google-shared.ts b/packages/ai/src/providers/google-shared.ts index de75ec704..8f9b70cad 100644 --- a/packages/ai/src/providers/google-shared.ts +++ b/packages/ai/src/providers/google-shared.ts @@ -14,6 +14,7 @@ import type { FetchImpl, ImageContent, Model, + ServiceTier, StopReason, StreamOptions, TextContent, @@ -21,6 +22,7 @@ import type { Tool, ToolCall, } from "../types"; +import { shouldSendServiceTier } from "../types"; import { normalizeSystemPrompts } from "../utils"; import { AssistantMessageEventStream } from "../utils/event-stream"; import type { RawHttpRequestDump } from "../utils/http-inspector"; @@ -73,6 +75,21 @@ export interface GoogleSharedStreamOptions extends StreamOptions { budgetTokens?: number; level?: GoogleThinkingLevel; }; + /** Gemini/Vertex serving tier (`flex`/`priority`); other values are omitted. */ + serviceTier?: ServiceTier; + /** + * Continues a Gemini Interactions API conversation from a stored interaction. + * When set on the direct Google provider, the request uses `/interactions` + * with `previous_interaction_id` instead of the legacy generateContent stream. + */ + previousInteractionId?: string; + /** + * Uses the Gemini Interactions API for direct Google requests, storing the + * returned interaction id on the assistant response for follow-up turns. + */ + useInteractionsApi?: boolean; + /** Overrides Interactions API request storage; default is the API default (`true`). */ + storeInteraction?: boolean; } /** @@ -126,11 +143,9 @@ function resolveThoughtSignature(isSameProviderAndModel: boolean, signature: str return isSameProviderAndModel && isValidThoughtSignature(signature) ? signature : undefined; } -/** - * Claude models via Google APIs require explicit tool call IDs in function calls/responses. - */ -export function requiresToolCallId(modelId: string): boolean { - return modelId.startsWith("claude-"); +function supportsFunctionPartId(model: Model): boolean { + if (model.api === "google-vertex") return false; + return model.id.startsWith("claude-") || (model.api === "google-generative-ai" && isGemini3Model(model.id)); } function getGeminiMajorVersion(modelId: string): number | undefined { @@ -156,8 +171,9 @@ function isGemini3Model(modelId: string): boolean { */ export function convertMessages(model: Model, context: Context): Content[] { const contents: Content[] = []; + const emittedToolCallNames = new Map(); + const normalizeToolCallId = (id: string): string => { - if (!requiresToolCallId(model.id)) return id; return id.replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 64); }; @@ -243,6 +259,7 @@ export function convertMessages(model: Model, contex }); } } else if (block.type === "toolCall") { + emittedToolCallNames.set(block.id, block.name); const thoughtSignature = resolveThoughtSignature(isSameProviderAndModel, block.thoughtSignature); const effectiveSignature = thoughtSignature || (isGemini3Model(model.id) ? SKIP_THOUGHT_SIGNATURE : undefined); @@ -251,11 +268,11 @@ export function convertMessages(model: Model, contex functionCall: { name: block.name, args: block.arguments ?? {}, - ...(requiresToolCallId(model.id) ? { id: block.id } : {}), + ...(supportsFunctionPartId(model) ? { id: block.id } : {}), }, }; if (model.provider === "google-vertex" && part?.functionCall?.id) { - delete part.functionCall.id; // Vertex AI does not support 'id' in functionCall + delete part.functionCall.id; // Vertex AI GenerateContent rejects 'id' in functionCall parts. } if (effectiveSignature) { part.thoughtSignature = effectiveSignature; @@ -301,10 +318,11 @@ export function convertMessages(model: Model, contex }, })); - const includeId = requiresToolCallId(model.id); + const includeId = supportsFunctionPartId(model); + const emittedName = emittedToolCallNames.get(msg.toolCallId); const functionResponsePart: Part = { functionResponse: { - name: msg.toolName, + name: emittedName ?? msg.toolName, response: msg.isError ? { error: responseValue } : { output: responseValue }, ...(hasImages && modelSupportsMultimodalFunctionResponse && { parts: imageParts }), ...(includeId ? { id: msg.toolCallId } : {}), @@ -312,7 +330,7 @@ export function convertMessages(model: Model, contex }; if (model.provider === "google-vertex" && functionResponsePart.functionResponse?.id) { - delete functionResponsePart.functionResponse.id; // Vertex AI does not support 'id' in functionResponse + delete functionResponsePart.functionResponse.id; // Vertex AI GenerateContent rejects 'id' in functionResponse parts. } // Cloud Code Assist API requires all function responses to be in a single user turn. @@ -529,6 +547,24 @@ export function pushToolCallEvents( * `text_start` / `thinking_start` event. `onBeforeStartEvent` lets the SSE consumer * inject its `ensureStarted()` first-token side effect into the canonical event order. */ +export function startTextOrThinkingBlock( + isThinking: true, + output: AssistantMessage, + stream: AssistantMessageEventStream, + onBeforeStartEvent?: () => void, +): ThinkingContent; +export function startTextOrThinkingBlock( + isThinking: false, + output: AssistantMessage, + stream: AssistantMessageEventStream, + onBeforeStartEvent?: () => void, +): TextContent; +export function startTextOrThinkingBlock( + isThinking: boolean, + output: AssistantMessage, + stream: AssistantMessageEventStream, + onBeforeStartEvent?: () => void, +): TextContent | ThinkingContent; export function startTextOrThinkingBlock( isThinking: boolean, output: AssistantMessage, @@ -791,6 +827,14 @@ export function buildGoogleGenerateContentParams 0 && { tools: convertTools(context.tools, model) }), }; + // Gemini API (google-generative-ai) reads the tier from the request body; + // Vertex AI ignores a body field and requires the + // `X-Vertex-AI-LLM-Shared-Request-Type` header instead (added in + // streamGoogleVertex), so only emit the body field for the direct API. + if (model.provider === "google" && shouldSendServiceTier(options.serviceTier, model.provider)) { + config.serviceTier = options.serviceTier; + } + if (context.tools && context.tools.length > 0 && options.toolChoice) { const choice = options.toolChoice; if (typeof choice === "string") { @@ -852,6 +896,8 @@ export interface GoogleGenAIRequestPlan { url: string; headers: Record; fetch?: FetchImpl; + /** Optional URL retried once when {@link url} returns 404 (regional Vertex endpoint missing a global-only model). */ + fallbackUrl?: string; } export function streamGoogleGenAI(args: { @@ -906,8 +952,8 @@ export function streamGoogleGenAI> => { - const response = await fetchImpl(plan.url, { + const openStreamAt = async (requestUrl: string): Promise> => { + const response = await fetchImpl(requestUrl, { method: "POST", headers: { ...plan.headers, "Content-Type": "application/json", Accept: "text/event-stream" }, body: bodyJson, @@ -929,6 +975,20 @@ export function streamGoogleGenAI; }; + // A regional Vertex endpoint 404s for models published only on the + // global endpoint; retry global once so a stale/ambient region never + // breaks a request that worked before regional routing existed. + const openStream = async (): Promise> => { + if (!plan.fallbackUrl) return openStreamAt(plan.url); + try { + return await openStreamAt(plan.url); + } catch (error) { + if (error instanceof AIError.GoogleApiError && error.status === 404) { + return openStreamAt(plan.fallbackUrl); + } + throw error; + } + }; let body = await openStream(); stream.push({ type: "start", partial: output }); @@ -1007,6 +1067,7 @@ function paramsToWireBody(params: GenerateContentParameters): Record = {}; if (config.temperature !== undefined) gen.temperature = config.temperature; diff --git a/packages/ai/src/providers/google-types.ts b/packages/ai/src/providers/google-types.ts index 086448dea..0c5c2fe9f 100644 --- a/packages/ai/src/providers/google-types.ts +++ b/packages/ai/src/providers/google-types.ts @@ -9,7 +9,6 @@ * - `POST {generativelanguage,aiplatform}.googleapis.com/.../models/{model}:streamGenerateContent?alt=sse` * - The Cloud Code Assist endpoint used by `google-gemini-cli.ts` */ - /** Mirror of `@google/genai`'s `FinishReason` string enum. */ export type FinishReason = | "FINISH_REASON_UNSPECIFIED" @@ -131,6 +130,11 @@ export interface GenerateContentConfig { safetySettings?: Array>; cachedContent?: string; thinkingConfig?: ThinkingConfig; + /** + * Gemini/Vertex serving tier. Serialized to the request body root as + * `serviceTier` (camelCase) by the transformer in `google-shared.ts`. + */ + serviceTier?: "auto" | "default" | "flex" | "scale" | "priority"; abortSignal?: AbortSignal; } diff --git a/packages/ai/src/providers/google-vertex.ts b/packages/ai/src/providers/google-vertex.ts index 5400da7bd..66b1868b9 100644 --- a/packages/ai/src/providers/google-vertex.ts +++ b/packages/ai/src/providers/google-vertex.ts @@ -2,7 +2,13 @@ import { $env } from "@oh-my-pi/pi-utils"; import * as AIError from "../error"; import type { Context, Model, StreamFunction } from "../types"; import type { AssistantMessageEventStream } from "../utils/event-stream"; -import { getVertexAccessToken } from "./google-auth"; +import { getVertexAccessToken, hasVertexBearerCredentialsHint } from "./google-auth"; +import { + type GoogleInteractionsPlan, + modelSupportsInteractions, + resolveInteractionDispatch, + streamGoogleInteractions, +} from "./google-interactions"; import { buildGoogleGenerateContentParams, type GoogleGenAIRequestPlan, @@ -16,48 +22,128 @@ export interface GoogleVertexOptions extends GoogleSharedStreamOptions { } const API_VERSION = "v1"; +const INTERACTIONS_API_VERSION = "v1beta1"; +const INTERACTIONS_API_REVISION = "2026-05-20"; export const streamGoogleVertex: StreamFunction<"google-vertex"> = ( model: Model<"google-vertex">, context: Context, options?: GoogleVertexOptions, -): AssistantMessageEventStream => - streamGoogleGenAI({ - model, - options, - api: "google-vertex", - retainTextSignature: true, - prepare: async (): Promise => { - const apiKey = resolveApiKey(options); - const params = buildGoogleGenerateContentParams(model, context, options ?? {}); - const baseHeaders: Record = { - ...(model.headers ?? {}), - ...(options?.headers ?? {}), - }; +): AssistantMessageEventStream => { + const runGenerateContent = (): AssistantMessageEventStream => + streamGoogleGenAI({ + model, + options, + api: "google-vertex", + retainTextSignature: true, + prepare: async (): Promise => { + const apiKey = resolveApiKey(options); + const params = buildGoogleGenerateContentParams(model, context, options ?? {}); + params.config ||= {}; + if (!params.config.safetySettings) { + params.config.safetySettings = [ + { + category: "HARM_CATEGORY_HATE_SPEECH", + threshold: "OFF", + }, + { + category: "HARM_CATEGORY_DANGEROUS_CONTENT", + threshold: "OFF", + }, + { + category: "HARM_CATEGORY_SEXUALLY_EXPLICIT", + threshold: "OFF", + }, + { + category: "HARM_CATEGORY_HARASSMENT", + threshold: "OFF", + }, + ]; + } + const baseHeaders: Record = { + ...(model.headers ?? {}), + ...(options?.headers ?? {}), + }; + // Vertex AI ignores a `serviceTier` request-body field (unlike the direct + // Gemini API); priority must travel as a request header. Only `priority` + // has a documented Vertex request control — `flex` has none, so it's a no-op. + if (options?.serviceTier === "priority") { + baseHeaders["X-Vertex-AI-LLM-Shared-Request-Type"] = "priority"; + } - if (apiKey) { - const url = `https://aiplatform.googleapis.com/${API_VERSION}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; + if (apiKey) { + // Explicit `location` is a deliberate residency choice: honor it and let + // a 404 surface. An ambient env-derived region falls back to the global + // endpoint so a stray GOOGLE_*_LOCATION never breaks a previously-working + // global-only request. + const explicitLocation = options?.location; + const location = explicitLocation ?? resolveAmbientLocation() ?? "global"; + const host = resolveEndpointHost(location); + const path = `${API_VERSION}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; + const useGlobalFallback = !explicitLocation && host !== "aiplatform.googleapis.com"; + return { + params, + url: `https://${host}/${path}`, + fallbackUrl: useGlobalFallback ? `https://aiplatform.googleapis.com/${path}` : undefined, + headers: { + ...baseHeaders, + "x-goog-api-key": apiKey, + }, + fetch: options?.fetch, + }; + } + + const project = resolveProject(options); + const location = resolveLocation(options); + const accessToken = await getVertexAccessToken({ signal: options?.signal, fetch: options?.fetch }); + const host = resolveEndpointHost(location); + const url = `https://${host}/${API_VERSION}/projects/${project}/locations/${location}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; return { params, url, - headers: { ...baseHeaders, "x-goog-api-key": apiKey }, + headers: { ...baseHeaders, Authorization: `Bearer ${accessToken}` }, fetch: options?.fetch, }; - } + }, + }); + // Default Gemini 3+ onto Interactions whenever a bearer credential source exists (ADC file, + // `GOOGLE_APPLICATION_CREDENTIALS`, or an explicit access-token env). Interactions needs bearer + // auth, so express API-key-only setups stay on generateContent — and an express key, when + // present, still serves the generateContent fallback. Interactions always targets the official + // global `aiplatform` host; the fallback also recovers ids the endpoint rejects. + const { useInteractions, auto, anchor, state } = resolveInteractionDispatch({ + context, + options, + provider: model.provider, + autoEligible: modelSupportsInteractions(model) && hasVertexBearerCredentialsHint(), + }); + if (!useInteractions) return runGenerateContent(); + + return streamGoogleInteractions({ + model, + context, + options, + api: "google-vertex", + anchor, + state, + prepare: async (): Promise => { const project = resolveProject(options); - const location = resolveLocation(options); const accessToken = await getVertexAccessToken({ signal: options?.signal, fetch: options?.fetch }); - const host = resolveEndpointHost(location); - const url = `https://${host}/${API_VERSION}/projects/${project}/locations/${location}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; return { - params, - url, - headers: { ...baseHeaders, Authorization: `Bearer ${accessToken}` }, + url: `https://aiplatform.googleapis.com/${INTERACTIONS_API_VERSION}/projects/${project}/locations/global/interactions`, + headers: { + ...(model.headers ?? {}), + ...(options?.headers ?? {}), + Authorization: `Bearer ${accessToken}`, + "Api-Revision": INTERACTIONS_API_REVISION, + }, fetch: options?.fetch, }; }, + fallback: auto ? runGenerateContent : undefined, }); +}; function resolveApiKey(options?: GoogleVertexOptions): string | undefined { // options.apiKey may contain sentinel values like "" or "N/A" @@ -80,9 +166,14 @@ function resolveProject(options?: GoogleVertexOptions): string { function resolveEndpointHost(location: string): string { return location === "global" ? "aiplatform.googleapis.com" : `${location}-aiplatform.googleapis.com`; } +function resolveAmbientLocation(): string | undefined { + return $env.GOOGLE_VERTEX_LOCATION || $env.GOOGLE_CLOUD_LOCATION || $env.VERTEX_LOCATION || undefined; +} +function resolveOptionalLocation(options?: GoogleVertexOptions): string | undefined { + return options?.location || resolveAmbientLocation(); +} function resolveLocation(options?: GoogleVertexOptions): string { - const location = - options?.location || $env.GOOGLE_VERTEX_LOCATION || $env.GOOGLE_CLOUD_LOCATION || $env.VERTEX_LOCATION; + const location = resolveOptionalLocation(options); if (!location) { throw new AIError.ConfigurationError( "Vertex AI requires a location. Set GOOGLE_VERTEX_LOCATION/GOOGLE_CLOUD_LOCATION/VERTEX_LOCATION or pass location in options.", diff --git a/packages/ai/src/providers/google.ts b/packages/ai/src/providers/google.ts index 24b5a1280..3bd8ed7ff 100644 --- a/packages/ai/src/providers/google.ts +++ b/packages/ai/src/providers/google.ts @@ -2,6 +2,7 @@ import * as AIError from "../error"; import { getEnvApiKey } from "../stream"; import type { Context, Model, StreamFunction } from "../types"; import type { AssistantMessageEventStream } from "../utils/event-stream"; +import { modelSupportsInteractions, resolveInteractionDispatch, streamGoogleInteractions } from "./google-interactions"; import { buildGoogleGenerateContentParams, type GoogleGenAIRequestPlan, @@ -17,29 +18,70 @@ export const streamGoogle: StreamFunction<"google-generative-ai"> = ( model: Model<"google-generative-ai">, context: Context, options?: GoogleOptions, -): AssistantMessageEventStream => - streamGoogleGenAI({ +): AssistantMessageEventStream => { + const apiKey = options?.apiKey || getEnvApiKey(model.provider); + if (!apiKey) { + throw new AIError.MissingApiKeyError( + undefined, + "Google Generative AI requires an API key (GEMINI_API_KEY or options.apiKey).", + ); + } + + const runGenerateContent = (): AssistantMessageEventStream => + streamGoogleGenAI({ + model, + options, + api: "google-generative-ai", + prepare: (): GoogleGenAIRequestPlan => { + const params = buildGoogleGenerateContentParams(model, context, options ?? {}); + // `model.baseUrl` already includes the API version segment when set (mirrors the + // `apiVersion: ""` reset that the SDK relied on for custom base URLs). + const base = model.baseUrl?.trim() || DEFAULT_GENERATIVE_LANGUAGE_BASE; + const url = `${base}/models/${model.id}:streamGenerateContent?alt=sse`; + const headers: Record = { + "x-goog-api-key": apiKey, + ...(model.headers ?? {}), + ...(options?.headers ?? {}), + }; + return { params, url, headers, fetch: options?.fetch }; + }, + }); + + // Default Gemini 3+ on the official endpoint onto Interactions (custom proxy base URLs keep + // generateContent, which serves the full catalog). The fallback recovers ids the endpoint rejects. + const trimmedBase = model.baseUrl?.trim(); + let officialEndpoint = !trimmedBase; + if (trimmedBase) { + try { + officialEndpoint = new URL(trimmedBase).hostname === "generativelanguage.googleapis.com"; + } catch { + officialEndpoint = false; + } + } + const { useInteractions, auto, anchor, state } = resolveInteractionDispatch({ + context, + options, + provider: model.provider, + autoEligible: officialEndpoint && modelSupportsInteractions(model), + }); + if (!useInteractions) return runGenerateContent(); + + return streamGoogleInteractions({ model, + context, options, api: "google-generative-ai", - prepare: (): GoogleGenAIRequestPlan => { - const apiKey = options?.apiKey || getEnvApiKey(model.provider); - if (!apiKey) { - throw new AIError.MissingApiKeyError( - undefined, - "Google Generative AI requires an API key (GEMINI_API_KEY or options.apiKey).", - ); - } - const params = buildGoogleGenerateContentParams(model, context, options ?? {}); - // `model.baseUrl` already includes the API version segment when set (mirrors the - // `apiVersion: ""` reset that the SDK relied on for custom base URLs). - const base = model.baseUrl?.trim() || DEFAULT_GENERATIVE_LANGUAGE_BASE; - const url = `${base}/models/${model.id}:streamGenerateContent?alt=sse`; - const headers: Record = { + anchor, + state, + prepare: () => ({ + url: `${trimmedBase || DEFAULT_GENERATIVE_LANGUAGE_BASE}/interactions`, + headers: { "x-goog-api-key": apiKey, ...(model.headers ?? {}), ...(options?.headers ?? {}), - }; - return { params, url, headers, fetch: options?.fetch }; - }, + }, + fetch: options?.fetch, + }), + fallback: auto ? runGenerateContent : undefined, }); +}; diff --git a/packages/ai/src/providers/ollama.ts b/packages/ai/src/providers/ollama.ts index d46a50453..d2b0b67df 100644 --- a/packages/ai/src/providers/ollama.ts +++ b/packages/ai/src/providers/ollama.ts @@ -323,6 +323,10 @@ function createChatBody(model: Model<"ollama-chat">, context: Context, options: }; } +function shouldRetryOllamaResponse(response: Response, bodyText: string): boolean { + return response.status < 500 || !AIError.LLAMA_CPP_TOOL_CALL_PARSE_PATTERN.test(bodyText); +} + async function captureHttpErrorResponse(response: Response): Promise { let bodyText: string | undefined; let bodyJson: unknown; @@ -598,6 +602,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = ( body: JSON.stringify(body), signal: watchdog.signal, defaultDelayMs: OLLAMA_RETRY_DELAYS_MS, + shouldRetryResponse: shouldRetryOllamaResponse, fetch: options.fetch, timeout: false, }); @@ -747,6 +752,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = ( } const result = await AIError.finalize(error, { api: model.api, + provider: model.provider, signal: options.signal, rawRequestDump, capturedErrorResponse, diff --git a/packages/ai/src/providers/openai-chat-server.ts b/packages/ai/src/providers/openai-chat-server.ts index da2d02627..441138ea9 100644 --- a/packages/ai/src/providers/openai-chat-server.ts +++ b/packages/ai/src/providers/openai-chat-server.ts @@ -13,7 +13,7 @@ import type { Context, ImageContent, Message, - ResolvedServiceTier, + ServiceTier, StopReason, TextContent, Tool, @@ -38,7 +38,7 @@ function isReasoningEffort(value: unknown): value is ReasoningEffort { return value === "minimal" || value === "low" || value === "medium" || value === "high" || value === "xhigh"; } -function isServiceTier(value: unknown): value is ResolvedServiceTier { +function isServiceTier(value: unknown): value is ServiceTier { return value === "auto" || value === "default" || value === "flex" || value === "scale" || value === "priority"; } diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 2a15b7bc9..446727a9f 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -104,7 +104,7 @@ import { transformMessages } from "./transform-messages"; export interface OpenAICodexResponsesOptions extends StreamOptions { reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh"; reasoningSummary?: "auto" | "concise" | "detailed" | null; - /** `reasoning.context` replay scope; defaults to `all_turns` for every Codex request when unset. */ + /** `reasoning.context` replay scope; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */ reasoningContext?: CodexReasoningContext; textVerbosity?: "low" | "medium" | "high"; include?: string[]; diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index b627fcbfa..77882a60e 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -1,4 +1,5 @@ import type { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { supportsAllTurnsReasoningContext } from "@oh-my-pi/pi-catalog/identity"; import { requireSupportedEffort } from "@oh-my-pi/pi-catalog/model-thinking"; import type { Api, Model } from "../../types"; @@ -14,7 +15,7 @@ export interface ReasoningConfig { export interface CodexRequestOptions { reasoningEffort?: ReasoningConfig["effort"]; reasoningSummary?: ReasoningConfig["summary"] | null; - /** Explicit `reasoning.context` override; defaults to `all_turns` for every Codex request when unset. */ + /** Explicit `reasoning.context` override; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */ reasoningContext?: CodexReasoningContext; textVerbosity?: "low" | "medium" | "high"; include?: string[]; @@ -254,9 +255,21 @@ export async function transformRequestBody( ...body.reasoning, ...reasoningConfig, }; - // Default reasoning replay to `all_turns` for every Codex request, - // mirroring codex-rs; an explicit `reasoningContext` overrides it. - body.reasoning.context = options.reasoningContext ?? "all_turns"; + // Default reasoning replay to `all_turns`, mirroring codex-rs; an + // explicit `reasoningContext` overrides the default. The `all_turns` + // value is only accepted from gpt-5.4 onward — earlier Codex ids + // (gpt-5.1-codex, gpt-5.3-codex, gpt-5.3-codex-spark) reject it with + // "Unsupported value: 'all_turns' is not supported with this model". + // For those, drop `context` so the server applies its `current_turn` + // default. The version gate is authoritative: even an explicit + // `all_turns` override is suppressed on unsupported models, while + // `current_turn`/`auto` (universally supported) always pass through. + const context = options.reasoningContext ?? "all_turns"; + if (context === "all_turns" && !supportsAllTurnsReasoningContext(model.id)) { + delete body.reasoning.context; + } else { + body.reasoning.context = context; + } } else { delete body.reasoning; } diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index a024d79b5..b6cd43daa 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -39,8 +39,6 @@ import { type Model, OPENAI_MAX_OUTPUT_TOKENS, type Provider, - type ResolvedServiceTier, - resolveServiceTier, type ServiceTier, type StopReason, type StreamOptions, @@ -269,14 +267,13 @@ export function resolveOpenAIRequestSetup( } export function applyOpenAIServiceTier( - params: { service_tier?: ResolvedServiceTier | "auto" | "default" | null | undefined }, + params: { service_tier?: ServiceTier | null | undefined }, serviceTier: ServiceTier | null | undefined, provider: Provider | undefined, ): void { if (!shouldSendServiceTier(serviceTier, provider)) return; - const resolved = resolveServiceTier(serviceTier, provider); - if (resolved === "flex" || resolved === "scale" || resolved === "priority") { - params.service_tier = resolved; + if (serviceTier === "flex" || serviceTier === "scale" || serviceTier === "priority") { + params.service_tier = serviceTier; } } @@ -315,10 +312,7 @@ export function applyOpenAIResponsesServiceTierCost( // The response echo is authoritative when present (OpenAI may downgrade a // requested priority/flex turn to default under load); only fall back to the // requested tier when the response omits the echo entirely. - const served = - typeof responseServiceTier === "string" - ? responseServiceTier - : resolveServiceTier(requestServiceTier, model.provider); + const served = typeof responseServiceTier === "string" ? responseServiceTier : (requestServiceTier ?? undefined); const multiplier = getOpenAIResponsesServiceTierCostMultiplier(served); if (multiplier === 1) return; usage.cost.input *= multiplier; @@ -623,7 +617,7 @@ export type OpenAICompletionsParams = Omit void; #callbackReject?: (error: string) => void; @@ -50,6 +65,7 @@ export abstract class OAuthCallbackFlow { this.preferredPort = preferredPortOrOptions; this.callbackPath = callbackPath; this.callbackHostname = DEFAULT_HOSTNAME; + this.allowPortFallback = true; return; } @@ -57,6 +73,7 @@ export abstract class OAuthCallbackFlow { this.callbackPath = preferredPortOrOptions.callbackPath ?? CALLBACK_PATH; this.callbackHostname = preferredPortOrOptions.callbackHostname ?? DEFAULT_HOSTNAME; this.redirectUri = preferredPortOrOptions.redirectUri; + this.allowPortFallback = preferredPortOrOptions.allowPortFallback ?? true; } /** @@ -87,18 +104,29 @@ export abstract class OAuthCallbackFlow { .join(""); } + #loginCancelledError(): AIError.LoginCancelledError { + return new AIError.LoginCancelledError(`OAuth callback cancelled: ${this.ctrl.signal?.reason}`); + } + + #throwIfCancelled(): void { + if (this.ctrl.signal?.aborted) throw this.#loginCancelledError(); + } + /** * Execute the OAuth login flow. */ async login(): Promise { const state = this.generateState(); + this.#throwIfCancelled(); // Start callback server first to get actual redirect URI const { server, redirectUri } = await this.#startCallbackServer(state); try { + this.#throwIfCancelled(); // Generate auth URL with the ACTUAL redirect URI (may differ from expected if port was busy) const { url: authUrl, instructions } = await this.generateAuthUrl(state, redirectUri); + this.#throwIfCancelled(); // Notify controller that auth is ready this.ctrl.onAuth?.({ url: authUrl, instructions }); @@ -106,6 +134,7 @@ export abstract class OAuthCallbackFlow { // Wait for callback or manual input const { code } = await this.#waitForCallback(state); + this.#throwIfCancelled(); this.ctrl.onProgress?.("Exchanging authorization code for tokens..."); @@ -126,10 +155,17 @@ export abstract class OAuthCallbackFlow { } const redirectUri = `http://${this.callbackHostname}:${this.preferredPort}${this.callbackPath}`; return { server, redirectUri }; - } catch { + } catch (cause) { if (this.redirectUri) { throw new AIError.ConfigurationError( - `OAuth callback port ${this.preferredPort} unavailable; cannot fall back to a random port when oauth.redirectUri is set`, + `OAuth callback port ${this.preferredPort} is in use, but oauth.redirectUri (${this.redirectUri}) requires this exact port. Free port ${this.preferredPort} (e.g. stop the process bound to it) and retry, or change oauth.redirectUri to point at an available port.`, + { cause }, + ); + } + if (!this.allowPortFallback) { + throw new AIError.ConfigurationError( + `OAuth callback port ${this.preferredPort} is in use. The OAuth provider validates redirect URIs against its registered callback, so falling back to a random port would be rejected. Free port ${this.preferredPort} (e.g. stop the process bound to it) and retry, or set oauth.callbackPort/oauth.redirectUri to a port the provider has registered.`, + { cause }, ); } const server = this.#createServer(0, expectedState); @@ -208,17 +244,18 @@ export abstract class OAuthCallbackFlow { #waitForCallback(expectedState: string): Promise { const timeoutSignal = AbortSignal.timeout(DEFAULT_TIMEOUT); const signal = this.ctrl.signal ? AbortSignal.any([this.ctrl.signal, timeoutSignal]) : timeoutSignal; + if (signal.aborted) return Promise.reject(this.#loginCancelledError()); - const callbackPromise = new Promise((resolve, reject) => { - this.#callbackResolve = resolve; - this.#callbackReject = reject; + const callback = Promise.withResolvers(); + this.#callbackResolve = callback.resolve; + this.#callbackReject = callback.reject; - signal.addEventListener("abort", () => { - this.#callbackResolve = undefined; - this.#callbackReject = undefined; - reject(new AIError.LoginCancelledError(`OAuth callback cancelled: ${signal.reason}`)); - }); + signal.addEventListener("abort", () => { + this.#callbackResolve = undefined; + this.#callbackReject = undefined; + callback.reject(new AIError.LoginCancelledError(`OAuth callback cancelled: ${signal.reason}`)); }); + const callbackPromise = callback.promise; // Manual input race (if supported) if (this.ctrl.onManualCodeInput) { diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 649fa0391..62d572ba0 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -71,6 +71,7 @@ import type { ToolChoice, } from "./types"; import { AssistantMessageEventStream } from "./utils/event-stream"; +import { wrapLeakedThinkingStream } from "./utils/leaked-thinking-stream"; import { wrapFetchForProxy } from "./utils/proxy"; import { withRequestDebugFetch } from "./utils/request-debug"; import { withGeminiThinkingLoopGuard } from "./utils/thinking-loop"; @@ -499,8 +500,11 @@ function withProviderInFlightLimit AssistantMessageEventStream, ): AssistantMessageEventStream { + // Leaked-thinking healing folds in here — the one shared provider-dispatch + // chokepoint — so the loop guard (which wraps this) sees healed events and all + // six provider exits are covered by one wrap. Healing is idempotent. const limit = resolveProviderInFlightLimit(model.provider, options); - if (limit === undefined) return dispatch(); + if (limit === undefined) return wrapLeakedThinkingStream(dispatch()); const outer = new AssistantMessageEventStream(); void (async () => { @@ -520,7 +524,7 @@ function withProviderInFlightLimit( // GitLab Duo Workflow - IDE workflow protocol + WebSocket action bridge if (model.api === "gitlab-duo-agent") { - return streamGitLabDuoWorkflow(model as Model<"gitlab-duo-agent">, context, { - ...requestOptions, - apiKey, - }); + // Does not route through withProviderInFlightLimit, so heal explicitly. + return wrapLeakedThinkingStream( + streamGitLabDuoWorkflow(model as Model<"gitlab-duo-agent">, context, { + ...requestOptions, + apiKey, + }), + ); } // Kimi Code - route to dedicated handler that wraps OpenAI or Anthropic API @@ -1355,6 +1362,9 @@ function mapOptionsForApi( streamFirstEventTimeoutMs: options?.streamFirstEventTimeoutMs, streamIdleTimeoutMs: options?.streamIdleTimeoutMs, providerSessionState: options?.providerSessionState, + useInteractionsApi: options?.useInteractionsApi, + storeInteraction: options?.storeInteraction, + previousInteractionId: options?.previousInteractionId, maxInFlightRequests: options?.maxInFlightRequests, onPayload: options?.onPayload, onResponse: options?.onResponse, @@ -1565,6 +1575,7 @@ function mapOptionsForApi( if (!reasoning || !model.reasoning) { return castApi<"google-generative-ai">({ ...base, + serviceTier: options?.serviceTier, thinking: { enabled: false }, toolChoice: mapGoogleToolChoice(options?.toolChoice), }); @@ -1578,6 +1589,7 @@ function mapOptionsForApi( if (googleModel.thinking?.mode === "google-level") { return castApi<"google-generative-ai">({ ...base, + serviceTier: options?.serviceTier, thinking: { enabled: true, level: mapEffortToGoogleThinkingLevel(effort), @@ -1661,6 +1673,7 @@ function mapOptionsForApi( if (!reasoning || !model.reasoning) { return castApi<"google-vertex">({ ...base, + serviceTier: options?.serviceTier, thinking: { enabled: false }, toolChoice: mapGoogleToolChoice(options?.toolChoice), }); @@ -1673,6 +1686,7 @@ function mapOptionsForApi( if (geminiModel.thinking?.mode === "google-level") { return castApi<"google-vertex">({ ...base, + serviceTier: options?.serviceTier, thinking: { enabled: true, level: mapEffortToGoogleThinkingLevel(effort), @@ -1683,6 +1697,7 @@ function mapOptionsForApi( return castApi<"google-vertex">({ ...base, + serviceTier: options?.serviceTier, thinking: { enabled: true, budgetTokens: getGoogleBudget(geminiModel, effort, options?.thinkingBudgets), diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index f19e736c5..b6c0467df 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -105,84 +105,181 @@ export type ToolChoice = export type CacheRetention = "none" | "short" | "long"; /** - * Service tier hint for processing priority / cost control. + * Service tier hint for processing priority / cost control. These are the + * values providers consume on the wire: * - * The unscoped values (`"auto"`, `"default"`, `"flex"`, `"scale"`, - * `"priority"`) are passed through to providers that understand them - * (OpenAI's `service_tier` field directly; Anthropic translates - * `"priority"` into `speed: "fast"` on supported Opus models). + * - OpenAI / OpenAI-Codex: sent verbatim as the `service_tier` field + * (`flex`/`scale`/`priority`). + * - Google (Gemini API + Vertex AI): sent as the top-level `serviceTier` + * field (`flex`/`priority`). + * - OpenRouter: passed through as `service_tier`; OpenRouter realizes it for + * the OpenAI- and Google-family upstreams it supports and ignores it + * otherwise. + * - Direct Anthropic: `"priority"` is translated into `speed: "fast"` plus the + * fast-mode beta on supported Opus models. Other tiers are ignored. * - * The scoped values target a specific provider family and behave as the - * unscoped value on the matching provider, or `undefined` everywhere else. - * They let users opt into priority on one family without paying premium - * costs on the other when switching models mid-session. - * - * - `"openai-only"` → `"priority"` on `openai` and `openai-codex`; ignored elsewhere. - * - `"claude-only"` → `"priority"` on direct `anthropic` (not Bedrock/Vertex Claude). + * Per-family scoping is expressed by {@link ServiceTierByFamily}, not by + * scoped sentinel values — see {@link serviceTierFamily}. */ -export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority" | "openai-only" | "claude-only"; +export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority"; -/** Resolved tier — one of the values that providers actually consume on the wire. */ -export type ResolvedServiceTier = Exclude; +/** Provider families that expose an independent service-tier knob. */ +export type ServiceTierFamily = "openai" | "anthropic" | "google"; /** - * Resolves a possibly scoped `ServiceTier` to the effective tier for the - * given provider. Scoped values match their target family and otherwise - * collapse to `undefined`; unscoped values pass through unchanged. + * Per-family service-tier selection. A request consults only the entry for the + * family its model belongs to (see {@link resolveModelServiceTier}), so a user + * can opt one family into priority without affecting the others when switching + * models mid-session. */ -export function resolveServiceTier( - serviceTier: ServiceTier | null | undefined, - provider: Provider | undefined, -): ResolvedServiceTier | undefined { - if (!serviceTier) return undefined; - switch (serviceTier) { - case "openai-only": - return provider === "openai" || provider === "openai-codex" ? "priority" : undefined; - case "claude-only": - return provider === "anthropic" ? "priority" : undefined; - default: - return serviceTier; +export type ServiceTierByFamily = Partial>; + +/** + * Classify a model into the service-tier family whose knob governs it, or + * `undefined` when the model exposes no serving-priority control. + * + * OpenRouter models are classified by id namespace (`anthropic/`, `google/`, + * `openai/`); Claude on Bedrock/Vertex (api `anthropic-messages`) is the + * anthropic family even though its provider is `amazon-bedrock`/`google-vertex`. + */ +export function serviceTierFamily(model: Pick): ServiceTierFamily | undefined { + const provider = model.provider; + if (provider === "openrouter") { + const id = model.id.toLowerCase(); + if (id.startsWith("anthropic/")) return "anthropic"; + if (id.startsWith("google/")) return "google"; + if (id.startsWith("openai/")) return "openai"; + return undefined; } + if (provider === "openai" || provider === "openai-codex") return "openai"; + if (model.api === "anthropic-messages") return "anthropic"; + if (provider === "google" || provider === "google-vertex") return "google"; + return undefined; } /** - * True when the (possibly scoped) tier should be sent on the wire as the - * `service_tier` request field for the given provider. OpenAI / OpenAI-Codex - * accept `flex`/`scale`/`priority`; Fireworks Serverless realizes only its - * Priority serving path (`service_tier: "priority"`) on the OpenAI-compatible - * chat-completions endpoint. Unsupported tiers (`"auto"`, `"default"`), other - * providers, and scope mismatches all return false. + * Reduce a per-family tier map to the single wire tier for `model` — the entry + * for the model's family, or `undefined` when the model has no family. + */ +export function resolveModelServiceTier( + tiers: ServiceTierByFamily | null | undefined, + model: Pick, +): ServiceTier | undefined { + if (!tiers) return undefined; + const family = serviceTierFamily(model); + return family ? tiers[family] : undefined; +} + +/** + * True when the tier should be sent on the wire as the provider's service-tier + * request field. OpenAI / OpenAI-Codex accept `flex`/`scale`/`priority`; Google + * (Gemini API + Vertex) and OpenRouter accept `flex`/`priority`; Fireworks + * Serverless realizes only its Priority serving path. Anthropic is absent — it + * realizes `priority` via `speed: "fast"`, not a service-tier field. */ export function shouldSendServiceTier( serviceTier: ServiceTier | null | undefined, provider: Provider | undefined, ): boolean { - const resolved = resolveServiceTier(serviceTier, provider); - if (provider === "openai" || provider === "openai-codex") { - return resolved === "flex" || resolved === "scale" || resolved === "priority"; + if (!serviceTier) return false; + if (provider === "openai" || provider === "openai-codex" || provider === "openrouter") { + return serviceTier === "flex" || serviceTier === "scale" || serviceTier === "priority"; } - if (provider === "fireworks") { - return resolved === "priority"; + if (provider === "google") { + return serviceTier === "flex" || serviceTier === "priority"; + } + // Vertex realizes only priority (via header); flex has no documented control. + if (provider === "google-vertex" || provider === "fireworks") { + return serviceTier === "priority"; } return false; } /** - * Premium-request weight contributed by sending priority to a provider - * that supports it. Mirrors GitHub Copilot's `premiumRequests` accounting - * so the "premium requests" stat aggregates priority traffic across the - * OpenAI family and Anthropic fast-mode realizations. + * True when `priority` will actually be realized on the wire for `model`. + * Direct Anthropic realizes fast mode; OpenAI/Google/Fireworks emit the + * service-tier field; OpenRouter realizes it only for its OpenAI- and + * Google-family upstreams. Bedrock/Vertex Claude and OpenRouter Anthropic + * models do not realize priority and return `false`. + */ +export function realizesPriorityServiceTier( + serviceTier: ServiceTier | null | undefined, + model: Pick, +): boolean { + if (serviceTier !== "priority") return false; + if (model.provider === "anthropic") return true; + if (model.provider === "openrouter") { + const family = serviceTierFamily(model); + return family === "openai" || family === "google"; + } + if (model.api === "anthropic-messages") return false; + return shouldSendServiceTier(serviceTier, model.provider); +} + +/** + * Premium-request weight contributed by a priority request to a provider that + * realizes it and bills extra. Mirrors GitHub Copilot's `premiumRequests` + * accounting so the "premium requests" stat aggregates priority traffic across + * the OpenAI family, direct Anthropic fast mode, and Google priority. * - * Returns 1 per resolved priority request, 0 otherwise. + * Returns 1 only when priority is actually realized on the wire for `model` + * (see {@link realizesPriorityServiceTier}) and the provider bills it as a + * premium request. OpenRouter is excluded — it bills per its own pricing, not + * Copilot-premium semantics — as are Bedrock/Vertex Claude, where priority is + * silently dropped. */ export function getPriorityPremiumRequests( serviceTier: ServiceTier | null | undefined, - provider: Provider | undefined, + model: Pick, ): number { - if (resolveServiceTier(serviceTier, provider) !== "priority") return 0; - // Only providers that realize `priority` on the wire bill the user. - // Everywhere else, the field is silently dropped and nothing is charged. - return provider === "openai" || provider === "openai-codex" || provider === "anthropic" ? 1 : 0; + if (!realizesPriorityServiceTier(serviceTier, model)) return 0; + const provider = model.provider; + return provider === "openai" || + provider === "openai-codex" || + provider === "anthropic" || + provider === "google" || + provider === "google-vertex" + ? 1 + : 0; +} + +/** + * Coerce a persisted service-tier value to a {@link ServiceTierByFamily}. Newer + * sessions store the family map directly; legacy sessions stored a single + * scalar — `"priority"` applied everywhere, `"openai-only"`/`"claude-only"` + * scoped to one family, and the remaining values were OpenAI-only semantics. + */ +export function coerceServiceTierByFamily(value: unknown): ServiceTierByFamily | undefined { + if (value === null || value === undefined) return undefined; + if (typeof value === "object") { + const src = value as Record; + const out: ServiceTierByFamily = {}; + for (const family of ["openai", "anthropic", "google"] as const) { + const tier = src[family]; + if (tier === "auto" || tier === "default" || tier === "flex" || tier === "scale" || tier === "priority") { + out[family] = tier; + } + } + return Object.keys(out).length > 0 ? out : undefined; + } + switch (value) { + case "priority": + return { openai: "priority", anthropic: "priority", google: "priority" }; + case "openai-only": + return { openai: "priority" }; + case "claude-only": + return { anthropic: "priority" }; + case "auto": + return { openai: "auto" }; + case "default": + return { openai: "default" }; + case "flex": + return { openai: "flex" }; + case "scale": + return { openai: "scale" }; + default: + return undefined; + } } export interface ProviderSessionState { @@ -277,6 +374,22 @@ export interface StreamOptions { * Providers can use this to persist transport/session state between turns. */ providerSessionState?: Map; + /** + * Force Gemini model-mode Interactions API transport for providers that support it. + * When unset, those providers may still use Interactions to continue known + * server-side conversation lineage via `previousInteractionId` or stored state. + */ + useInteractionsApi?: boolean; + /** + * Whether supported Interactions transports should store server-side conversation + * state and return response ids for follow-up turns. Defaults to true. + */ + storeInteraction?: boolean; + /** + * Explicit Interactions response id to continue. Mutually exclusive with + * `storeInteraction: false` because the follow-up itself must be storable. + */ + previousInteractionId?: string; /** * Optional per-provider concurrent request cap for LLM stream calls. Keys are * provider ids (`model.provider`); positive numeric values cap in-flight diff --git a/packages/ai/src/utils/leaked-thinking-stream.ts b/packages/ai/src/utils/leaked-thinking-stream.ts new file mode 100644 index 000000000..9596a9b73 --- /dev/null +++ b/packages/ai/src/utils/leaked-thinking-stream.ts @@ -0,0 +1,260 @@ +/** + * Central live healing for leaked reasoning markup in the visible text channel. + * + * Some providers emit their canonical reasoning idioms (` ```thinking `, + * ``, Gemma/Harmony channels, …) into the *visible* text stream instead + * of a structured thinking part. {@link wrapLeakedThinkingStream} re-projects any + * provider stream into a fresh {@link AssistantMessageEventStream}, splitting the + * leaked fences out into proper `thinking` blocks *live* as deltas arrive — so + * every provider gets the same healing, not just the three with provider-local + * {@link StreamMarkupHealing} loops. + * + * The healing is idempotent: a second pass over already-clean text finds no + * fences, so wrapping a provider that already heals (or wrapping twice) is a + * harmless pass-through. Signatures are load-bearing for Google/Gemini/Vertex + * thought round-tripping, so text sub-blocks carry the source `textSignature`, + * forwarded thinking blocks their `thinkingSignature`, and forwarded tool calls + * their `thoughtSignature`. + * + * Modeled on {@link wrapInbandToolStream} / `InbandStreamProjector` in + * `../dialect/owned-stream.ts`, minus all in-band tool-call grammar: tool-call + * events are forwarded verbatim. + */ + +import type { AssistantMessage, TextContent, ThinkingContent, ToolCall } from "../types"; +import { AssistantMessageEventStream } from "./event-stream"; +import { StreamMarkupHealing, type StreamMarkupHealingEvent } from "./stream-markup-healing"; + +/** + * Wrap a provider stream so leaked reasoning fences are healed into thinking + * blocks live, for every provider. Returns a new stream that re-projects the + * inner one; the inner stream is fully consumed. + */ +export function wrapLeakedThinkingStream(inner: AssistantMessageEventStream): AssistantMessageEventStream { + const out = new AssistantMessageEventStream(); + void (async () => { + try { + let projector: LeakedThinkingProjector | undefined; + for await (const event of inner) { + switch (event.type) { + case "start": + projector = new LeakedThinkingProjector(out, event.partial); + break; + case "text_delta": { + projector ??= new LeakedThinkingProjector(out, event.partial); + const block = event.partial.content[event.contentIndex]; + projector.text(event.delta, block?.type === "text" ? block.textSignature : undefined); + break; + } + case "thinking_delta": { + projector ??= new LeakedThinkingProjector(out, event.partial); + const block = event.partial.content[event.contentIndex]; + projector.thinking(event.delta, block?.type === "thinking" ? block.thinkingSignature : undefined); + break; + } + case "toolcall_start": { + projector ??= new LeakedThinkingProjector(out, event.partial); + const block = event.partial.content[event.contentIndex]; + projector.toolStart(event.contentIndex, block?.type === "toolCall" ? block.name : ""); + break; + } + case "toolcall_delta": + projector?.toolDelta(event.contentIndex, event.delta); + break; + case "toolcall_end": + projector?.toolEnd(event.contentIndex, event.toolCall); + break; + case "done": { + projector ??= new LeakedThinkingProjector(out, event.message); + const content = projector.finish(event.message); + out.push({ type: "done", reason: event.reason, message: { ...event.message, content } }); + return; + } + case "error": { + projector ??= new LeakedThinkingProjector(out, event.error); + const content = projector.finish(event.error); + out.push({ type: "error", reason: event.reason, error: { ...event.error, content } }); + return; + } + // text_start/text_end/thinking_start/thinking_end are ignored: the + // projector owns block boundaries (matches wrapInbandToolStream). + } + } + // Inner ended via end(result) without a terminal event. + if (!out.done) { + const result = await inner.result(); + projector ??= new LeakedThinkingProjector(out, result); + const content = projector.finish(result); + out.end({ ...result, content }); + } + } catch (err) { + if (!out.done) out.fail(err); + } + })(); + return out; +} + +type OpenBlock = { index: number } | undefined; + +/** + * Re-projects an inner stream's events into `out`, healing leaked reasoning out + * of the visible text channel while forwarding native thinking and tool calls. + */ +class LeakedThinkingProjector { + readonly #out: AssistantMessageEventStream; + readonly #healer = new StreamMarkupHealing({ pattern: "thinking" }); + #partial: AssistantMessage; + #text: OpenBlock; + #thinking: OpenBlock; + /** Total visible text length fed to the healer, to replay any un-streamed tail in {@link finish}. */ + #fedLen = 0; + /** Latest non-undefined text signature seen, stamped onto held-back text flushed later. */ + #lastTextSignature: string | undefined; + /** Forwarded native tool calls, keyed by the inner stream's `contentIndex`. */ + #toolBlocks = new Map(); + + constructor(out: AssistantMessageEventStream, seed: AssistantMessage) { + this.#out = out; + this.#partial = { ...seed, content: [] }; + this.#out.push({ type: "start", partial: this.#partial }); + } + + /** Feed a visible-text delta through the healer, splitting leaked fences live. */ + text(delta: string, signature: string | undefined): void { + this.#fedLen += delta.length; + if (signature !== undefined) this.#lastTextSignature = signature; + this.#apply(this.#healer.feedEvents(delta), this.#lastTextSignature); + } + + /** Forward a native thinking delta, preserving its signature. */ + thinking(delta: string, signature: string | undefined): void { + const index = this.#openThinking(); + const block = this.#partial.content[index] as ThinkingContent; + block.thinking += delta; + if (signature !== undefined) block.thinkingSignature = signature; + this.#out.push({ type: "thinking_delta", contentIndex: index, delta, partial: this.#partial }); + } + + /** Forward a native tool call's start, releasing any held-back text first. */ + toolStart(srcIndex: number, name: string): void { + this.#apply(this.#healer.flushEvents(), this.#lastTextSignature); + this.#closeText(); + this.#closeThinking(); + const block: ToolCall = { type: "toolCall", id: "", name, arguments: {} }; + this.#partial.content.push(block); + const index = this.#partial.content.length - 1; + this.#toolBlocks.set(srcIndex, { index }); + this.#out.push({ type: "toolcall_start", contentIndex: index, partial: this.#partial }); + } + + toolDelta(srcIndex: number, delta: string): void { + const entry = this.#toolBlocks.get(srcIndex); + if (!entry) return; + this.#out.push({ type: "toolcall_delta", contentIndex: entry.index, delta, partial: this.#partial }); + } + + toolEnd(srcIndex: number, toolCall: ToolCall): void { + const entry = this.#toolBlocks.get(srcIndex); + if (entry) { + const block = this.#partial.content[entry.index] as ToolCall; + Object.assign(block, toolCall); + this.#out.push({ type: "toolcall_end", contentIndex: entry.index, toolCall: block, partial: this.#partial }); + this.#toolBlocks.delete(srcIndex); + return; + } + // `end` without a matching `start` — release held text, then forward whole. + this.#apply(this.#healer.flushEvents(), this.#lastTextSignature); + this.#closeText(); + this.#closeThinking(); + const block: ToolCall = { ...toolCall }; + this.#partial.content.push(block); + const index = this.#partial.content.length - 1; + this.#out.push({ type: "toolcall_start", contentIndex: index, partial: this.#partial }); + this.#out.push({ type: "toolcall_end", contentIndex: index, toolCall: block, partial: this.#partial }); + } + + /** + * Finalize: replay any un-streamed visible-text tail from `message.content`, + * flush held-back fragments, close open blocks, and return the healed content. + */ + finish(message: AssistantMessage): AssistantMessage["content"] { + let fullText = ""; + let tailSignature: string | undefined; + for (const block of message.content) { + if (block.type === "text") { + fullText += block.text; + tailSignature = block.textSignature; + } + } + if (tailSignature !== undefined) this.#lastTextSignature = tailSignature; + if (fullText.length > this.#fedLen) { + this.#apply(this.#healer.feedEvents(fullText.slice(this.#fedLen)), this.#lastTextSignature); + } + this.#apply(this.#healer.flushEvents(), this.#lastTextSignature); + this.#closeText(); + this.#closeThinking(); + return this.#partial.content; + } + + #apply(events: readonly StreamMarkupHealingEvent[], signature?: string): void { + for (const event of events) { + if (event.type === "text") this.#emitText(event.text, signature); + else if (event.type === "thinking") this.#emitHealedThinking(event.thinking); + } + } + + #emitText(text: string, signature: string | undefined): void { + if (text.length === 0) return; + this.#closeThinking(); + if (!this.#text) { + const block: TextContent = + signature === undefined ? { type: "text", text: "" } : { type: "text", text: "", textSignature: signature }; + this.#partial.content.push(block); + this.#text = { index: this.#partial.content.length - 1 }; + this.#out.push({ type: "text_start", contentIndex: this.#text.index, partial: this.#partial }); + } else if (signature !== undefined) { + (this.#partial.content[this.#text.index] as TextContent).textSignature = signature; + } + const block = this.#partial.content[this.#text.index] as TextContent; + block.text += text; + this.#out.push({ type: "text_delta", contentIndex: this.#text.index, delta: text, partial: this.#partial }); + } + + /** Healed (leaked) thinking carries no signature, matching the source fence. */ + #emitHealedThinking(text: string): void { + if (text.length === 0) return; + const index = this.#openThinking(); + const block = this.#partial.content[index] as ThinkingContent; + block.thinking += text; + this.#out.push({ type: "thinking_delta", contentIndex: index, delta: text, partial: this.#partial }); + } + + #openThinking(): number { + this.#closeText(); + if (!this.#thinking) { + this.#partial.content.push({ type: "thinking", thinking: "" }); + this.#thinking = { index: this.#partial.content.length - 1 }; + this.#out.push({ type: "thinking_start", contentIndex: this.#thinking.index, partial: this.#partial }); + } + return this.#thinking.index; + } + + #closeText(): void { + if (!this.#text) return; + const block = this.#partial.content[this.#text.index] as TextContent; + this.#out.push({ type: "text_end", contentIndex: this.#text.index, content: block.text, partial: this.#partial }); + this.#text = undefined; + } + + #closeThinking(): void { + if (!this.#thinking) return; + const block = this.#partial.content[this.#thinking.index] as ThinkingContent; + this.#out.push({ + type: "thinking_end", + contentIndex: this.#thinking.index, + content: block.thinking, + partial: this.#partial, + }); + this.#thinking = undefined; + } +} diff --git a/packages/ai/test/anthropic-fast-mode.test.ts b/packages/ai/test/anthropic-fast-mode.test.ts index b70601c8a..a1d8fa904 100644 --- a/packages/ai/test/anthropic-fast-mode.test.ts +++ b/packages/ai/test/anthropic-fast-mode.test.ts @@ -88,22 +88,6 @@ describe("Anthropic priority service tier → speed='fast'", () => { expect(payload.speed).toBeUndefined(); } }); - - it("sets speed='fast' on direct anthropic when serviceTier='claude-only'", async () => { - const payload = (await capturePayload(makeAnthropicModel("claude-opus-4-7"), { - serviceTier: "claude-only", - })) as { speed?: string }; - expect(payload.speed).toBe("fast"); - }); - - it("omits speed when serviceTier='openai-only' on an anthropic model", async () => { - // Scoped to OpenAI — on this anthropic request, the scope doesn't match, - // so `speed` must not be set on the wire. - const payload = (await capturePayload(makeAnthropicModel("claude-opus-4-7"), { - serviceTier: "openai-only", - })) as Record; - expect(payload.speed).toBeUndefined(); - }); }); describe("clearAnthropicFastModeFallback", () => { diff --git a/packages/ai/test/auth-storage-api-key-login.test.ts b/packages/ai/test/auth-storage-api-key-login.test.ts index e50e0fd4f..879a8a9ec 100644 --- a/packages/ai/test/auth-storage-api-key-login.test.ts +++ b/packages/ai/test/auth-storage-api-key-login.test.ts @@ -8,6 +8,7 @@ import { AuthStorage, SqliteAuthCredentialStore } from "@oh-my-pi/pi-ai/auth-sto import * as deepseekModule from "@oh-my-pi/pi-ai/registry/deepseek"; import * as kagiModule from "@oh-my-pi/pi-ai/registry/kagi"; import * as ollamaCloudModule from "@oh-my-pi/pi-ai/registry/ollama-cloud"; +import * as aiStream from "@oh-my-pi/pi-ai/stream"; import { removeWithRetries } from "../../utils/src/temp"; function countCredentialRows(dbPath: string, provider: string): number { @@ -38,6 +39,9 @@ function countCredentialRowsByDisabledState(dbPath: string, provider: string, di } describe("AuthStorage api-key login upsert", () => { + // A live env var now (correctly) overrides a stored static api_key. These tests verify that a + // freshly stored api_key resolves through AuthStorage.getApiKey, so neutralize the env leg + // entirely — this ignores every provider's ambient env key, not just the few set locally. let tempDir = ""; let dbPath = ""; let store: SqliteAuthCredentialStore | null = null; @@ -47,6 +51,7 @@ describe("AuthStorage api-key login upsert", () => { let loginOllamaCloudSpy: Mock; beforeEach(async () => { + vi.spyOn(aiStream, "getEnvApiKey").mockReturnValue(undefined); tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-auth-api-key-login-")); dbPath = path.join(tempDir, "agent.db"); store = await SqliteAuthCredentialStore.open(dbPath); diff --git a/packages/ai/test/auth-storage-credential-origin.test.ts b/packages/ai/test/auth-storage-credential-origin.test.ts index 822ffeff4..e3d7b2d10 100644 --- a/packages/ai/test/auth-storage-credential-origin.test.ts +++ b/packages/ai/test/auth-storage-credential-origin.test.ts @@ -68,14 +68,24 @@ describe("AuthStorage.getCredentialOrigin", () => { }); }); - test("a stored api key reports api_key and outranks a co-stored OAuth credential", async () => { + test("a stored OAuth credential outranks a co-stored api key", async () => { await withEnv(SUPPRESS_ENV, async () => { - // getApiKey() prefers api_key before oauth, so the origin must match. + // getApiKey() resolves stored OAuth before a stored api_key, so the origin must match. await auth?.set("openai", [ { type: "oauth", access: "a", refresh: "r", expires: Date.now() + 60_000 }, { type: "api_key", key: "sk-stored" }, ]); - expect(auth?.getCredentialOrigin("openai")).toEqual({ kind: "api_key" }); + expect(auth?.getCredentialOrigin("openai")).toEqual({ kind: "oauth" }); + }); + }); + + test("an explicit env var outranks a stored api key", async () => { + // Regression: a live env var is the user's current choice and must win over a stored + // static api_key (e.g. a stale broker-migrated copy) so `GEMINI_API_KEY` etc. take effect. + await withEnv({ ...SUPPRESS_ENV, OPENAI_API_KEY: "sk-env" }, async () => { + await auth?.set("openai", [{ type: "api_key", key: "sk-stored" }]); + expect(auth?.getCredentialOrigin("openai")).toEqual({ kind: "env", envVar: "OPENAI_API_KEY" }); + expect(await auth?.getApiKey("openai")).toBe("sk-env"); }); }); diff --git a/packages/ai/test/callback-server-port-fallback.test.ts b/packages/ai/test/callback-server-port-fallback.test.ts new file mode 100644 index 000000000..65bdd5282 --- /dev/null +++ b/packages/ai/test/callback-server-port-fallback.test.ts @@ -0,0 +1,123 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import { ConfigurationError } from "@oh-my-pi/pi-ai/error"; +import { OAuthCallbackFlow } from "@oh-my-pi/pi-ai/registry/oauth/callback-server"; +import type { OAuthCredentials } from "@oh-my-pi/pi-ai/registry/oauth/types"; + +/** + * Minimal callback flow we can drive without a real authorization server. + * `generateAuthUrl` is never expected to run in the strict-port tests — + * `#startCallbackServer` must throw before `login()` can reach it — so a stray + * invocation surfaces as a counter bump the test asserts on. + */ +class TestCallbackFlow extends OAuthCallbackFlow { + authUrlCalls = 0; + lastRedirectUri?: string; + + async generateAuthUrl(_state: string, redirectUri: string): Promise<{ url: string }> { + this.authUrlCalls += 1; + this.lastRedirectUri = redirectUri; + return { url: `${redirectUri}?started=1` }; + } + + async exchangeToken(code: string, _state: string, _redirectUri: string): Promise { + return { access: `access-${code}`, refresh: "refresh", expires: Date.now() + 60_000 }; + } +} + +/** + * Bind a real loopback port so the next `Bun.serve({ port })` against the + * same port fails with EADDRINUSE. Returns the bound port plus a `release` + * callback for teardown. + */ +function occupyLoopbackPort(): { port: number; release: () => void } { + const server = Bun.serve({ port: 0, fetch: () => new Response("blocker") }); + const port = server.port; + if (typeof port !== "number") { + server.stop(true); + throw new Error("Bun.serve({ port: 0 }) did not assign a numeric port"); + } + return { port, release: () => server.stop(true) }; +} + +afterEach(() => { + vi.restoreAllMocks(); +}); + +describe("OAuthCallbackFlow port fallback policy", () => { + it("falls back to a random port by default so historical AI-provider flows keep working", async () => { + const blocker = occupyLoopbackPort(); + const progress: string[] = []; + const flow = new TestCallbackFlow( + { + onAuth: () => {}, + onProgress: msg => progress.push(msg), + // Short abort — we only care that the flow advertised the fallback URI. + signal: AbortSignal.timeout(100), + }, + { preferredPort: blocker.port }, + ); + + try { + await expect(flow.login()).rejects.toThrow(); // aborted while waiting for the browser callback + const fallbackNotice = progress.find(msg => msg.startsWith(`Preferred port ${blocker.port} unavailable`)); + expect(fallbackNotice).toBeDefined(); + // Notice carries a different (random) port, never the blocked one. + expect(fallbackNotice).not.toContain(`using port ${blocker.port}`); + // generateAuthUrl ran with the random-port redirect URI — that's the + // silent fallback behavior that MCP flows now opt out of. + expect(flow.authUrlCalls).toBe(1); + expect(flow.lastRedirectUri).toMatch(/^http:\/\/localhost:\d+\/callback$/); + expect(flow.lastRedirectUri).not.toContain(`:${blocker.port}/`); + } finally { + blocker.release(); + } + }); + + it("throws a ConfigurationError when allowPortFallback is false", async () => { + const serveSpy = vi.spyOn(Bun, "serve").mockImplementation(() => { + throw new Error("EADDRINUSE"); + }); + + const flow = new TestCallbackFlow( + { + onAuth: () => {}, + signal: AbortSignal.timeout(1_000), + }, + { preferredPort: 14581, allowPortFallback: false }, + ); + + await expect(flow.login()).rejects.toThrow(ConfigurationError); + await expect(flow.login()).rejects.toThrow( + /OAuth callback port 14581 is in use\. The OAuth provider validates redirect URIs/, + ); + // Fallback to port 0 must never be attempted: every serve call uses the preferred port. + const portArgs = serveSpy.mock.calls.map(([opts]) => opts.port); + expect(portArgs.every(port => port === 14581)).toBe(true); + // generateAuthUrl never runs: the error fires before login() opens the browser. + expect(flow.authUrlCalls).toBe(0); + }); + + it("preserves redirectUri-strict behavior with the updated error message", async () => { + vi.spyOn(Bun, "serve").mockImplementation(() => { + throw new Error("EADDRINUSE"); + }); + + const flow = new TestCallbackFlow( + { + onAuth: () => {}, + signal: AbortSignal.timeout(1_000), + }, + { + preferredPort: 14582, + redirectUri: "http://localhost:14582/callback", + }, + ); + + // redirectUri takes precedence over allowPortFallback in the error + // message so users learn exactly which configuration knob is forcing + // the strict port match. + await expect(flow.login()).rejects.toThrow( + /oauth\.redirectUri \(http:\/\/localhost:14582\/callback\) requires this exact port/, + ); + }); +}); diff --git a/packages/ai/test/google-empty-response-retry.test.ts b/packages/ai/test/google-empty-response-retry.test.ts index 005c6f8ad..b1e4db870 100644 --- a/packages/ai/test/google-empty-response-retry.test.ts +++ b/packages/ai/test/google-empty-response-retry.test.ts @@ -89,7 +89,8 @@ describe("Google empty-response retry (public + Vertex path)", () => { return calls === 1 ? sse(genaiChunk("")) : sse(genaiChunk("Hello!")); }; - const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock }); + // Pin the generateContent transport: gemini-3 ids now auto-route to Interactions by default. + const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock, useInteractionsApi: false }); const { events, starts } = await drain(stream); const result = await stream.result(); @@ -107,7 +108,7 @@ describe("Google empty-response retry (public + Vertex path)", () => { return sse(genaiChunk("")); }; - const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock }); + const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock, useInteractionsApi: false }); const result = await stream.result(); expect(calls).toBe(3); // MAX_EMPTY_STREAM_RETRIES (2) + 1 initial attempt @@ -137,6 +138,7 @@ describe("Google empty-response retry (public + Vertex path)", () => { project: "project", location: "location", fetch: fetchMock, + useInteractionsApi: false, }); const { events } = await drain(stream); const result = await stream.result(); @@ -194,6 +196,7 @@ describe("Google empty-response retry (public + Vertex path)", () => { project: "project", location: "location", fetch: fetchMock, + useInteractionsApi: false, }); const result = await stream.result(); diff --git a/packages/ai/test/google-function-calling-matching.test.ts b/packages/ai/test/google-function-calling-matching.test.ts new file mode 100644 index 000000000..9f610e19e --- /dev/null +++ b/packages/ai/test/google-function-calling-matching.test.ts @@ -0,0 +1,145 @@ +import { describe, expect, it } from "bun:test"; +import { convertMessages } from "@oh-my-pi/pi-ai/providers/google-shared"; +import type { Context, Model, Usage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +const ZERO_USAGE: Usage = { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, +}; + +function createGoogleModel( + id: string, + api: "google-generative-ai" | "google-vertex" = "google-generative-ai", +): Model { + return buildModel({ + id, + name: id, + api, + provider: api === "google-vertex" ? "google-vertex" : "google", + baseUrl: "https://example.com", + reasoning: false, + input: ["text", "image"], + cost: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + }, + contextWindow: 200000, + maxTokens: 8192, + }); +} + +function contextWithToolResult(toolName = "stale_tool_name"): Context { + return { + messages: [ + { + role: "user", + content: "Call the tool", + timestamp: 1000, + }, + { + role: "assistant", + provider: "google-generative-ai", + api: "google-generative-ai", + model: "gemini-2.5-flash", + content: [ + { + type: "toolCall", + id: "call_12345_abc", + name: "actual_tool_name", + arguments: { query: "pi" }, + }, + ], + usage: ZERO_USAGE, + stopReason: "toolUse", + timestamp: 2000, + }, + { + role: "toolResult", + toolCallId: "call_12345_abc", + toolName, + isError: false, + content: [ + { + type: "text", + text: "Tool result text", + }, + ], + timestamp: 3000, + }, + ], + }; +} + +function functionCallAndResponse(model: Model<"google-generative-ai" | "google-vertex">, context: Context) { + const contents = convertMessages(model, context); + const functionCall = contents.find(c => c.role === "model")?.parts?.find(part => part.functionCall)?.functionCall; + const functionResponse = contents + .find(c => c.role === "user" && c.parts?.some(part => part.functionResponse)) + ?.parts?.find(part => part.functionResponse)?.functionResponse; + + return { functionCall, functionResponse }; +} + +describe("Google GenerateContent function response matching", () => { + it("uses emitted functionCall IDs and names for direct Gemini 3 functionResponse parts", () => { + const model = createGoogleModel("gemini-3.5-flash"); + const { functionCall, functionResponse } = functionCallAndResponse(model, contextWithToolResult()); + + expect(functionCall?.id).toBe("call_12345_abc"); + expect(functionResponse?.id).toBe("call_12345_abc"); + expect(functionCall?.name).toBe("actual_tool_name"); + expect(functionResponse?.name).toBe("actual_tool_name"); + expect(functionResponse?.name).toBe(functionCall?.name); + }); + + it("omits unsupported Part IDs for Vertex Gemini 3.5 GenerateContent", () => { + const model = createGoogleModel("gemini-3.5-flash", "google-vertex"); + const { functionCall, functionResponse } = functionCallAndResponse(model, contextWithToolResult()); + + expect(functionCall?.id).toBeUndefined(); + expect(functionResponse?.id).toBeUndefined(); + expect(functionResponse?.name).toBe(functionCall?.name); + }); + + it("keeps multimodal tool output inside Gemini 3 functionResponse parts", () => { + const context = contextWithToolResult("actual_tool_name"); + const toolResult = context.messages[2]; + if (toolResult.role !== "toolResult") throw new Error("expected tool result fixture"); + toolResult.content.push({ + type: "image", + mimeType: "image/png", + data: "base64-image-data", + }); + + const model = createGoogleModel("gemini-3.5-flash"); + const { functionResponse } = functionCallAndResponse(model, context); + + expect(functionResponse?.parts).toEqual([ + { + inlineData: { + mimeType: "image/png", + data: "base64-image-data", + }, + }, + ]); + }); + + it("keeps Claude call IDs on non-Vertex Google-compatible endpoints", () => { + const model = createGoogleModel("claude-sonnet-4-5"); + const { functionCall, functionResponse } = functionCallAndResponse( + model, + contextWithToolResult("actual_tool_name"), + ); + + expect(functionCall?.id).toBe("call_12345_abc"); + expect(functionResponse?.id).toBe("call_12345_abc"); + expect(functionResponse?.name).toBe(functionCall?.name); + }); +}); diff --git a/packages/ai/test/google-gemini-cli-alignment.test.ts b/packages/ai/test/google-gemini-cli-alignment.test.ts index b7de79719..3e8c2f463 100644 --- a/packages/ai/test/google-gemini-cli-alignment.test.ts +++ b/packages/ai/test/google-gemini-cli-alignment.test.ts @@ -596,11 +596,9 @@ describe("Google Gemini CLI alignment", () => { } const result = await stream.result(); - expect(result.content).toHaveLength(1); - expect(result.content[0]).toEqual({ - type: "text", - text: "", - }); + // A fully-discarded planning leak leaves no residual content — no empty + // text block survives (the central healing wrapper strips empties too). + expect(result.content).toHaveLength(0); expect(result.stopReason).toBe("stop"); const textDeltaEvents = events.filter(e => e.type === "text_delta"); @@ -838,15 +836,11 @@ describe("Google Gemini CLI alignment", () => { } const result = await stream.result(); - expect(result.content).toHaveLength(2); - expect(result.content[0]).toEqual({ - type: "text", - text: "", - }); - expect(result.content[1].type).toBe("toolCall"); - if (result.content[1].type === "toolCall") { - expect(result.content[1].name).toBe("read"); - expect(result.content[1].arguments).toEqual({ path: "src/main.ts" }); + expect(result.content).toHaveLength(1); + expect(result.content[0].type).toBe("toolCall"); + if (result.content[0].type === "toolCall") { + expect(result.content[0].name).toBe("read"); + expect(result.content[0].arguments).toEqual({ path: "src/main.ts" }); } expect(events.filter(e => e.type === "toolcall_start")).toHaveLength(1); @@ -917,11 +911,7 @@ describe("Google Gemini CLI alignment", () => { events.push(event); } const result = await stream.result(); - expect(result.content).toHaveLength(1); - expect(result.content[0]).toEqual({ - type: "text", - text: "", - }); + expect(result.content).toHaveLength(0); }); }); }); diff --git a/packages/ai/test/google-interactions.test.ts b/packages/ai/test/google-interactions.test.ts new file mode 100644 index 000000000..136fe2d95 --- /dev/null +++ b/packages/ai/test/google-interactions.test.ts @@ -0,0 +1,511 @@ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { streamGoogle } from "@oh-my-pi/pi-ai/providers/google"; +import { __resetVertexTokenCache } from "@oh-my-pi/pi-ai/providers/google-auth"; +import { streamGoogleVertex } from "@oh-my-pi/pi-ai/providers/google-vertex"; +import { streamSimple } from "@oh-my-pi/pi-ai/stream"; +import type { AssistantMessage, Context, FetchImpl, Model, Tool, Usage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +function googleModel(baseUrl = "https://generativelanguage.googleapis.com/v1beta"): Model<"google-generative-ai"> { + return buildModel({ + id: "gemini-3.5-flash", + name: "Gemini 3.5 Flash", + api: "google-generative-ai", + provider: "google", + baseUrl, + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1_000_000, + maxTokens: 8_192, + }); +} + +function vertexModel(id = "gemini-3.5-flash"): Model<"google-vertex"> { + return buildModel({ + id, + name: id, + api: "google-vertex", + provider: "google-vertex", + baseUrl: "", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1_000_000, + maxTokens: 8_192, + }); +} + +function sseResponse(events: readonly unknown[]): Response { + const payload = `${events.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`; + return new Response(payload, { status: 200, headers: { "content-type": "text/event-stream" } }); +} + +const weatherTool: Tool = { + name: "get_weather", + description: "Get weather", + parameters: { + type: "object", + properties: { city: { type: "string" } }, + required: ["city"], + additionalProperties: false, + }, +}; + +describe("Google Interactions API", () => { + it("chains tool results with previous_interaction_id from the prior assistant response", async () => { + const model = googleModel(); + const requestBodies: unknown[] = []; + let calls = 0; + const fetchMock: FetchImpl = async (_input, init) => { + requestBodies.push(JSON.parse(String(init?.body ?? "{}"))); + calls += 1; + if (calls === 1) { + return sseResponse([ + { + event_type: "interaction.created", + interaction: { id: "int_1", status: "in_progress" }, + }, + { event_type: "step.start", index: 0, step: { type: "thought" } }, + { + event_type: "step.delta", + index: 0, + delta: { type: "thought_signature", signature: "thought_sig_1" }, + }, + { + event_type: "step.delta", + index: 0, + delta: { type: "thought_summary", content: { type: "text", text: "Checking weather.\n" } }, + }, + { event_type: "step.stop", index: 0 }, + { + event_type: "step.start", + index: 1, + step: { + type: "function_call", + id: "call_weather", + name: "get_weather", + arguments: {}, + }, + }, + { event_type: "step.delta", index: 1, delta: { type: "arguments_delta", arguments: '{"city":"Bos' } }, + { event_type: "step.delta", index: 1, delta: { type: "arguments_delta", arguments: 'ton"}' } }, + { event_type: "step.stop", index: 1 }, + { + event_type: "interaction.completed", + interaction: { + id: "int_1", + status: "requires_action", + usage: { total_input_tokens: 10, total_output_tokens: 2, total_tokens: 12 }, + }, + }, + ]); + } + return sseResponse([ + { event_type: "interaction.created", interaction: { id: "int_2", status: "in_progress" } }, + { event_type: "step.start", index: 0, step: { type: "model_output" } }, + { event_type: "step.delta", index: 0, delta: { type: "text", text: "Sunny." } }, + { event_type: "step.stop", index: 0 }, + { + event_type: "interaction.completed", + interaction: { + id: "int_2", + status: "completed", + usage: { total_input_tokens: 3, total_output_tokens: 1, total_tokens: 4 }, + }, + }, + ]); + }; + Object.assign(fetchMock, { preconnect: fetch.preconnect }); + + const firstContext: Context = { + systemPrompt: ["Use concise weather reports."], + messages: [{ role: "user", content: "Need weather", timestamp: 1 }], + tools: [weatherTool], + }; + const first = await streamGoogle(model, firstContext, { + apiKey: "test-key", + fetch: fetchMock, + useInteractionsApi: true, + thinking: { enabled: true, level: "HIGH", budgetTokens: 123 }, + }).result(); + + expect(first.responseId).toBe("int_1"); + expect(first.stopReason).toBe("toolUse"); + expect(first.content).toEqual([ + { type: "thinking", thinking: "Checking weather.\n", thinkingSignature: "thought_sig_1" }, + { type: "toolCall", id: "call_weather", name: "get_weather", arguments: { city: "Boston" } }, + ]); + expect(requestBodies[0]).toMatchObject({ + model: "gemini-3.5-flash", + stream: true, + input: [{ type: "user_input", content: [{ type: "text", text: "Need weather" }] }], + system_instruction: "Use concise weather reports.", + tools: [{ functionDeclarations: [{ name: "get_weather" }] }], + generation_config: { thinking_level: "high" }, + }); + expect(requestBodies[0]).not.toHaveProperty("previous_interaction_id"); + expect(JSON.stringify(requestBodies[0])).not.toContain("thinking_budget"); + + const secondContext: Context = { + messages: [ + { role: "user", content: "Need weather", timestamp: 1 }, + first, + { + role: "toolResult", + toolCallId: "call_weather", + toolName: "get_weather", + content: [{ type: "text", text: "72F and sunny" }], + isError: false, + timestamp: 2, + }, + ], + tools: [weatherTool], + systemPrompt: ["Use concise weather reports."], + }; + const second = await streamGoogle(model, secondContext, { + apiKey: "test-key", + fetch: fetchMock, + thinking: { enabled: true, level: "HIGH", budgetTokens: 123 }, + }).result(); + + expect(second.responseId).toBe("int_2"); + expect(second.content).toEqual([{ type: "text", text: "Sunny." }]); + expect(requestBodies[1]).toMatchObject({ + previous_interaction_id: "int_1", + input: [ + { + type: "function_result", + name: "get_weather", + call_id: "call_weather", + result: [{ type: "text", text: "72F and sunny" }], + }, + ], + tools: [{ functionDeclarations: [{ name: "get_weather" }] }], + system_instruction: "Use concise weather reports.", + generation_config: { thinking_level: "high" }, + }); + expect(JSON.stringify(requestBodies[1])).not.toContain("Need weather"); + expect(JSON.stringify(requestBodies[1])).not.toContain("thinking_budget"); + }); + + it("does not expose or reuse interaction ids when storage is disabled", async () => { + const model = googleModel(); + const requestBodies: unknown[] = []; + const fetchMock: FetchImpl = async (_input, init) => { + requestBodies.push(JSON.parse(String(init?.body ?? "{}"))); + return sseResponse([ + { event_type: "interaction.created", interaction: { id: "unstored_int", status: "in_progress" } }, + { event_type: "step.start", index: 0, step: { type: "model_output" } }, + { event_type: "step.delta", index: 0, delta: { type: "text", text: "Done." } }, + { event_type: "step.stop", index: 0 }, + { event_type: "interaction.completed", interaction: { id: "unstored_int", status: "completed" } }, + ]); + }; + Object.assign(fetchMock, { preconnect: fetch.preconnect }); + + const result = await streamGoogle( + model, + { messages: [{ role: "user", content: "Hello", timestamp: 1 }] }, + { + apiKey: "test-key", + fetch: fetchMock, + useInteractionsApi: true, + storeInteraction: false, + }, + ).result(); + + expect(result.responseId).toBeUndefined(); + expect(requestBodies[0]).toMatchObject({ store: false }); + expect(() => + streamGoogle( + model, + { messages: [{ role: "user", content: "Hello", timestamp: 1 }] }, + { + apiKey: "test-key", + fetch: fetchMock, + storeInteraction: false, + previousInteractionId: "unstored_int", + }, + ), + ).toThrow(/storeInteraction:false/); + }); + + it("reads thought payload from step.start without leaking a prior signature", async () => { + const model = googleModel(); + const fetchMock: FetchImpl = async () => + sseResponse([ + { event_type: "interaction.created", interaction: { id: "int_3", status: "in_progress" } }, + { event_type: "step.start", index: 0, step: { type: "thought", signature: "stale_sig" } }, + { event_type: "step.stop", index: 0 }, + { + event_type: "step.start", + index: 1, + step: { type: "thought", summary: [{ type: "text", text: "Fresh plan.\n" }] }, + }, + { event_type: "step.stop", index: 1 }, + { event_type: "step.start", index: 2, step: { type: "model_output" } }, + { event_type: "step.delta", index: 2, delta: { type: "text", text: "Answer." } }, + { event_type: "step.stop", index: 2 }, + { event_type: "interaction.completed", interaction: { id: "int_3", status: "completed" } }, + ]); + Object.assign(fetchMock, { preconnect: fetch.preconnect }); + + const result = await streamGoogle( + model, + { messages: [{ role: "user", content: "Hello", timestamp: 1 }] }, + { apiKey: "test-key", fetch: fetchMock, useInteractionsApi: true }, + ).result(); + + expect(result.content).toEqual([ + { type: "thinking", thinking: "Fresh plan.\n" }, + { type: "text", text: "Answer." }, + ]); + }); +}); + +function genaiSse(text: string): Response { + return new Response( + `data: ${JSON.stringify({ + candidates: [{ content: { parts: [{ text }] }, finishReason: "STOP" }], + usageMetadata: { promptTokenCount: 1, candidatesTokenCount: 1, totalTokenCount: 2 }, + })}\n\n`, + { status: 200, headers: { "content-type": "text/event-stream" } }, + ); +} + +function interactionsTextSse( + id: string, + text: string, + terminal: "interaction.completed" | "interaction.complete" = "interaction.completed", +): Response { + return sseResponse([ + { event_type: "interaction.created", interaction: { id, status: "in_progress" } }, + { event_type: "step.start", index: 0, step: { type: "model_output" } }, + { event_type: "step.delta", index: 0, delta: { type: "text", text } }, + { event_type: "step.stop", index: 0 }, + { + event_type: terminal, + interaction: { + id, + status: "completed", + usage: { total_input_tokens: 10, total_output_tokens: 5, total_tokens: 15 }, + }, + }, + ]); +} + +interface CapturedCall { + url: string; + method: string; + headers: Headers; + body: unknown; +} + +function captureFetch(handler: (url: string) => Response): { fetch: FetchImpl; calls: CapturedCall[] } { + const calls: CapturedCall[] = []; + const fetchMock: FetchImpl = async (input, init) => { + const url = input instanceof Request ? input.url : String(input); + calls.push({ + url, + method: String(init?.method ?? "GET"), + headers: new Headers(init?.headers), + body: init?.body ? JSON.parse(String(init.body)) : undefined, + }); + return handler(url); + }; + Object.assign(fetchMock, { preconnect: fetch.preconnect }); + return { fetch: fetchMock, calls }; +} + +const ZERO_USAGE: Usage = { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, +}; + +function assistantWithResponse( + api: "google-vertex" | "google-generative-ai", + provider: string, + responseId: string, +): AssistantMessage { + return { + role: "assistant", + api, + provider, + model: "gemini-3.5-flash", + content: [{ type: "text", text: "prev" }], + usage: ZERO_USAGE, + stopReason: "stop", + timestamp: 2, + responseId, + }; +} + +const userTurn: Context = { messages: [{ role: "user", content: "Hi there", timestamp: 1 }] }; + +describe("Google Interactions API — zero-config default + fallback", () => { + let savedToken: string | undefined; + beforeEach(() => { + // A bearer source makes the Vertex auto-gate fire deterministically on any machine, and + // `getVertexAccessToken` returns it directly (no OAuth/metadata round-trip in tests). + savedToken = Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN; + Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN = "test-bearer"; + }); + afterEach(() => { + if (savedToken === undefined) delete Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN; + else Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN = savedToken; + __resetVertexTokenCache(); + }); + + it("auto-routes a capable Vertex model to Interactions under bearer auth", async () => { + const { fetch, calls } = captureFetch(() => interactionsTextSse("vint_1", "Hi")); + const result = await streamGoogleVertex(vertexModel(), userTurn, { + project: "p", + location: "us", + fetch, + }).result(); + + expect(calls).toHaveLength(1); + expect(calls[0].url).toBe("https://aiplatform.googleapis.com/v1beta1/projects/p/locations/global/interactions"); + expect(calls[0].method).toBe("POST"); + expect(calls[0].headers.get("Api-Revision")).toBe("2026-05-20"); + expect(calls[0].headers.get("Authorization")).toBe("Bearer test-bearer"); + expect(calls[0].body).toMatchObject({ + model: "gemini-3.5-flash", + stream: true, + input: [{ type: "user_input", content: [{ type: "text", text: "Hi there" }] }], + }); + expect(calls[0].body).not.toHaveProperty("contents"); + expect(calls[0].body).not.toHaveProperty("agent"); + expect(calls[0].body).not.toHaveProperty("environment"); + expect(result.stopReason).toBe("stop"); + expect(result.content).toEqual([{ type: "text", text: "Hi" }]); + expect(result.responseId).toBe("vint_1"); + expect(result.usage.totalTokens).toBe(15); + }); + + it("keeps an older (sub-3) Vertex model on generateContent", async () => { + const { fetch, calls } = captureFetch(() => genaiSse("ok")); + await streamGoogleVertex(vertexModel("gemini-2.5-flash"), userTurn, { + project: "p", + location: "us", + fetch, + }).result(); + + expect(calls[0].url).toContain(":streamGenerateContent"); + expect(calls.some(c => c.url.includes("/interactions"))).toBe(false); + }); + + it("honors useInteractionsApi:false on a capable Vertex model", async () => { + const { fetch, calls } = captureFetch(() => genaiSse("ok")); + await streamGoogleVertex(vertexModel(), userTurn, { + project: "p", + location: "us", + useInteractionsApi: false, + fetch, + }).result(); + + expect(calls[0].url).toContain(":streamGenerateContent"); + expect(calls.some(c => c.url.includes("/interactions"))).toBe(false); + }); + + it("falls back to generateContent when auto Interactions is unsupported (404)", async () => { + const { fetch, calls } = captureFetch(url => + url.includes("/interactions") ? new Response("nope", { status: 404 }) : genaiSse("recovered"), + ); + const result = await streamGoogleVertex(vertexModel(), userTurn, { + project: "p", + location: "us", + fetch, + }).result(); + + expect(calls[0].url).toContain("/interactions"); + expect(calls[1].url).toContain(":streamGenerateContent"); + expect(result.stopReason).toBe("stop"); + expect(result.content).toEqual([{ type: "text", text: "recovered" }]); + }); + + it("surfaces the error (no fallback) when explicit Interactions is unsupported", async () => { + const { fetch, calls } = captureFetch(url => + url.includes("/interactions") ? new Response("nope", { status: 404 }) : genaiSse("unexpected"), + ); + const result = await streamGoogleVertex(vertexModel(), userTurn, { + project: "p", + useInteractionsApi: true, + fetch, + }).result(); + + expect(result.stopReason).toBe("error"); + expect(calls.some(c => c.url.includes(":streamGenerateContent"))).toBe(false); + }); + + it("accepts interaction.complete as a terminal-event alias", async () => { + const { fetch } = captureFetch(() => interactionsTextSse("vint_2", "Done", "interaction.complete")); + const result = await streamGoogleVertex(vertexModel(), userTurn, { project: "p", fetch }).result(); + + expect(result.stopReason).toBe("stop"); + expect(result.content).toEqual([{ type: "text", text: "Done" }]); + }); + + it("sends previous_interaction_id only for same-provider assistant lineage", async () => { + const sameProvider = captureFetch(() => interactionsTextSse("vint_3", "ok")); + await streamGoogleVertex( + vertexModel(), + { + messages: [ + { role: "user", content: "a", timestamp: 1 }, + assistantWithResponse("google-vertex", "google-vertex", "vint_prev"), + { role: "user", content: "b", timestamp: 3 }, + ], + }, + { project: "p", fetch: sameProvider.fetch }, + ).result(); + expect(sameProvider.calls[0].body).toMatchObject({ previous_interaction_id: "vint_prev" }); + + const wrongProvider = captureFetch(() => interactionsTextSse("vint_4", "ok")); + await streamGoogleVertex( + vertexModel(), + { + messages: [ + { role: "user", content: "a", timestamp: 1 }, + assistantWithResponse("google-generative-ai", "google", "gint_prev"), + { role: "user", content: "b", timestamp: 3 }, + ], + }, + { project: "p", fetch: wrongProvider.fetch }, + ).result(); + expect(wrongProvider.calls[0].body).not.toHaveProperty("previous_interaction_id"); + }); + + it("auto-routes a capable direct Google model on the official endpoint to Interactions", async () => { + const { fetch, calls } = captureFetch(() => interactionsTextSse("gint_1", "Hi")); + const result = await streamGoogle(googleModel(), userTurn, { apiKey: "k", fetch }).result(); + + expect(calls[0].url).toBe("https://generativelanguage.googleapis.com/v1beta/interactions"); + expect(calls[0].headers.get("x-goog-api-key")).toBe("k"); + expect(result.responseId).toBe("gint_1"); + }); + + it("keeps a custom-baseUrl direct Google model on generateContent", async () => { + const { fetch, calls } = captureFetch(() => genaiSse("ok")); + await streamGoogle(googleModel("https://proxy.example.com/v1beta"), userTurn, { + apiKey: "k", + fetch, + }).result(); + + expect(calls[0].url).toContain(":streamGenerateContent"); + expect(calls[0].url.startsWith("https://proxy.example.com/")).toBe(true); + expect(calls.some(c => c.url.includes("/interactions"))).toBe(false); + }); + + it("threads the auto-default through streamSimple for a capable Google model", async () => { + const { fetch, calls } = captureFetch(() => interactionsTextSse("sint_1", "Hi")); + await streamSimple(googleModel(), userTurn, { apiKey: "k", fetch }).result(); + + expect(calls.some(c => c.url.includes("/interactions"))).toBe(true); + }); +}); diff --git a/packages/ai/test/google-service-tier.test.ts b/packages/ai/test/google-service-tier.test.ts new file mode 100644 index 000000000..ec541d847 --- /dev/null +++ b/packages/ai/test/google-service-tier.test.ts @@ -0,0 +1,107 @@ +import { describe, expect, it } from "bun:test"; +import { streamGoogle } from "@oh-my-pi/pi-ai/providers/google"; +import { streamGoogleVertex } from "@oh-my-pi/pi-ai/providers/google-vertex"; +import type { AssistantMessageEvent, Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +const context: Context = { messages: [{ role: "user", content: "hi", timestamp: 1 }] }; + +function sseStop(): Response { + const chunk = { + candidates: [{ content: { parts: [{ text: "ok" }] }, finishReason: "STOP" }], + usageMetadata: { promptTokenCount: 1, candidatesTokenCount: 1, totalTokenCount: 2 }, + }; + return new Response(`data: ${JSON.stringify(chunk)}\n\n`, { + status: 200, + headers: { "content-type": "text/event-stream" }, + }); +} + +async function drain(stream: AsyncIterable): Promise { + for await (const _ of stream) { + // consume + } +} + +interface Captured { + headers: Headers; + body: Record; +} + +function capturingFetch(): { fetch: FetchImpl; captured: () => Captured } { + let cap: Captured | undefined; + const fetch: FetchImpl = async (_url, init) => { + cap = { + headers: new Headers(init?.headers), + body: JSON.parse(String(init?.body ?? "{}")), + }; + return sseStop(); + }; + return { + fetch, + captured: () => { + if (!cap) throw new Error("fetch was not called"); + return cap; + }, + }; +} + +const geminiModel: Model<"google-generative-ai"> = buildModel({ + id: "gemini-3-flash", + name: "Gemini 3 Flash", + api: "google-generative-ai", + provider: "google", + baseUrl: "https://generativelanguage.googleapis.com/v1beta", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 32_000, +}); + +const vertexModel: Model<"google-vertex"> = buildModel({ + id: "gemini-3-flash", + name: "Gemini 3 Flash (Vertex)", + api: "google-vertex", + provider: "google-vertex", + baseUrl: "", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 32_000, +}); + +describe("Google service tier wire encoding", () => { + it("Gemini API sends the tier in the request body, not a header", async () => { + const { fetch, captured } = capturingFetch(); + await drain( + streamGoogle(geminiModel, context, { apiKey: "k", serviceTier: "priority", fetch, useInteractionsApi: false }), + ); + const { headers, body } = captured(); + expect(body.serviceTier).toBe("priority"); + expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBeNull(); + }); + + it("Vertex sends priority via header and omits the body tier field", async () => { + const { fetch, captured } = capturingFetch(); + await drain(streamGoogleVertex(vertexModel, context, { apiKey: "k", serviceTier: "priority", fetch })); + const { headers, body } = captured(); + expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBe("priority"); + expect(body.serviceTier).toBeUndefined(); + }); + + it("Vertex omits both header and body for flex (no documented control)", async () => { + const { fetch, captured } = capturingFetch(); + await drain(streamGoogleVertex(vertexModel, context, { apiKey: "k", serviceTier: "flex", fetch })); + const { headers, body } = captured(); + expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBeNull(); + expect(body.serviceTier).toBeUndefined(); + }); + + it("omits the tier entirely when unset", async () => { + const { fetch, captured } = capturingFetch(); + await drain(streamGoogle(geminiModel, context, { apiKey: "k", fetch, useInteractionsApi: false })); + expect(captured().body.serviceTier).toBeUndefined(); + }); +}); diff --git a/packages/ai/test/google-system-prompt.test.ts b/packages/ai/test/google-system-prompt.test.ts index 1ca718188..f5ef242be 100644 --- a/packages/ai/test/google-system-prompt.test.ts +++ b/packages/ai/test/google-system-prompt.test.ts @@ -26,6 +26,8 @@ async function captureGooglePayload( await streamGoogle(model, context, { apiKey: "test-key", + // Capture the generateContent request shape; gemini-3 ids auto-route to Interactions by default. + useInteractionsApi: false, onPayload: payload => { captured = payload as { config: { systemInstruction?: unknown }; contents: unknown[] }; }, diff --git a/packages/ai/test/issue-1270-repro.test.ts b/packages/ai/test/issue-1270-repro.test.ts index 5d3582254..2154ff79a 100644 --- a/packages/ai/test/issue-1270-repro.test.ts +++ b/packages/ai/test/issue-1270-repro.test.ts @@ -44,6 +44,8 @@ describe("issue #1270: Vertex AI global endpoint", () => { const stream = streamGoogleVertex(model, context, { project: "vertex-project", location: "global", + // This asserts the generateContent URL; gemini-3 ids auto-route to Interactions by default. + useInteractionsApi: false, fetch: async input => { const url = input instanceof Request ? input.url : input.toString(); urls.push(url); diff --git a/packages/ai/test/leaked-thinking-stream.test.ts b/packages/ai/test/leaked-thinking-stream.test.ts new file mode 100644 index 000000000..d2b57166c --- /dev/null +++ b/packages/ai/test/leaked-thinking-stream.test.ts @@ -0,0 +1,285 @@ +import { describe, expect, it } from "bun:test"; +import { stream } from "@oh-my-pi/pi-ai/stream"; +import type { + AssistantMessage, + AssistantMessageEvent, + Context, + FetchImpl, + Model, + TextContent, + ThinkingContent, + ToolCall, +} from "@oh-my-pi/pi-ai/types"; +import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; +import { wrapLeakedThinkingStream } from "@oh-my-pi/pi-ai/utils/leaked-thinking-stream"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +/** Minimal assistant message; `content`/`stopReason` overridden per event. */ +function msg(overrides: Partial = {}): AssistantMessage { + return { + role: "assistant", + content: [], + api: "mock", + provider: "mock", + model: "mock", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp: 0, + ...overrides, + }; +} + +/** + * Drive the wrapper: push inner events synchronously, then drain the healed + * output. Returns every emitted event plus the resolved final message. + */ +async function runWrapper( + feed: (inner: AssistantMessageEventStream) => void, +): Promise<{ events: AssistantMessageEvent[]; result: AssistantMessage }> { + const inner = new AssistantMessageEventStream(); + const out = wrapLeakedThinkingStream(inner); + feed(inner); + const events: AssistantMessageEvent[] = []; + for await (const event of out) events.push(event); + const result = await out.result(); + return { events, result }; +} + +function texts(message: AssistantMessage): string[] { + return message.content.filter((b): b is TextContent => b.type === "text").map(b => b.text); +} + +function thinks(message: AssistantMessage): ThinkingContent[] { + return message.content.filter((b): b is ThinkingContent => b.type === "thinking"); +} + +describe("wrapLeakedThinkingStream", () => { + it("splits a leaked fence into structured blocks live during streaming", async () => { + const leaked = "Visible before.```thinking\nplan\n```Visible after."; + const { events, result } = await runWrapper(inner => { + inner.push({ type: "start", partial: msg() }); + inner.push({ type: "text_start", contentIndex: 0, partial: msg({ content: [{ type: "text", text: "" }] }) }); + inner.push({ + type: "text_delta", + contentIndex: 0, + delta: leaked, + partial: msg({ content: [{ type: "text", text: leaked }] }), + }); + inner.push({ + type: "text_end", + contentIndex: 0, + content: leaked, + partial: msg({ content: [{ type: "text", text: leaked }] }), + }); + inner.push({ type: "done", reason: "stop", message: msg({ content: [{ type: "text", text: leaked }] }) }); + }); + + expect(result.content.map(b => b.type)).toEqual(["text", "thinking", "text"]); + expect(texts(result)).toEqual(["Visible before.", "Visible after."]); + expect(thinks(result).map(b => b.thinking)).toEqual(["plan\n"]); + // The split happened live, not only in the terminal message. + expect(events.some(e => e.type === "thinking_delta")).toBe(true); + }); + + it("preserves text, thinking, and tool-call signatures across the split", async () => { + const leaked = "before ```thinking\nhmm\n``` after"; + const call: ToolCall = { + type: "toolCall", + id: "call_1", + name: "read", + arguments: { path: "x" }, + thoughtSignature: "tsig", + }; + const { result } = await runWrapper(inner => { + inner.push({ type: "start", partial: msg() }); + inner.push({ + type: "text_start", + contentIndex: 0, + partial: msg({ content: [{ type: "text", text: "", textSignature: "sig" }] }), + }); + inner.push({ + type: "text_delta", + contentIndex: 0, + delta: leaked, + partial: msg({ content: [{ type: "text", text: leaked, textSignature: "sig" }] }), + }); + const withCall = msg({ + content: [{ type: "text", text: leaked, textSignature: "sig" }, call], + stopReason: "toolUse", + }); + inner.push({ type: "toolcall_start", contentIndex: 1, partial: withCall }); + inner.push({ type: "toolcall_end", contentIndex: 1, toolCall: call, partial: withCall }); + inner.push({ type: "done", reason: "toolUse", message: withCall }); + }); + + const textBlocks = result.content.filter((b): b is TextContent => b.type === "text"); + expect(textBlocks.map(b => b.text)).toEqual(["before ", " after"]); + expect(textBlocks.map(b => b.textSignature)).toEqual(["sig", "sig"]); + // Healed (leaked) thinking carries no signature. + expect(thinks(result).every(b => b.thinkingSignature === undefined)).toBe(true); + const calls = result.content.filter((b): b is ToolCall => b.type === "toolCall"); + expect(calls[0]?.thoughtSignature).toBe("tsig"); + }); + + it("heals a fence that only appears in the terminal message (no prior text deltas)", async () => { + const leaked = "Intro.```thinking\nquiet\n```Outro."; + const { result } = await runWrapper(inner => { + inner.push({ type: "start", partial: msg() }); + inner.push({ + type: "done", + reason: "stop", + message: msg({ content: [{ type: "text", text: leaked, textSignature: "sig" }] }), + }); + }); + + expect(result.content.map(b => b.type)).toEqual(["text", "thinking", "text"]); + expect(texts(result)).toEqual(["Intro.", "Outro."]); + // Tail-replayed text still carries the source signature. + expect(result.content.filter((b): b is TextContent => b.type === "text").map(b => b.textSignature)).toEqual([ + "sig", + "sig", + ]); + }); + + it("passes clean text through unchanged and forwards native thinking", async () => { + const clean = "Just a normal answer."; + const cleanRun = await runWrapper(inner => { + inner.push({ type: "start", partial: msg() }); + inner.push({ type: "text_start", contentIndex: 0, partial: msg({ content: [{ type: "text", text: "" }] }) }); + inner.push({ + type: "text_delta", + contentIndex: 0, + delta: clean, + partial: msg({ content: [{ type: "text", text: clean }] }), + }); + inner.push({ type: "done", reason: "stop", message: msg({ content: [{ type: "text", text: clean }] }) }); + }); + expect(cleanRun.result.content.map(b => b.type)).toEqual(["text"]); + expect(texts(cleanRun.result)).toEqual([clean]); + + const nativeThinking = msg({ + content: [ + { type: "thinking", thinking: "native reasoning", thinkingSignature: "tk" }, + { type: "text", text: "answer" }, + ], + }); + const nativeRun = await runWrapper(inner => { + inner.push({ type: "start", partial: msg() }); + inner.push({ + type: "thinking_start", + contentIndex: 0, + partial: msg({ content: [{ type: "thinking", thinking: "" }] }), + }); + inner.push({ + type: "thinking_delta", + contentIndex: 0, + delta: "native reasoning", + partial: msg({ content: [{ type: "thinking", thinking: "native reasoning", thinkingSignature: "tk" }] }), + }); + inner.push({ + type: "thinking_end", + contentIndex: 0, + content: "native reasoning", + partial: msg({ content: [{ type: "thinking", thinking: "native reasoning", thinkingSignature: "tk" }] }), + }); + inner.push({ type: "text_start", contentIndex: 1, partial: nativeThinking }); + inner.push({ type: "text_delta", contentIndex: 1, delta: "answer", partial: nativeThinking }); + inner.push({ type: "done", reason: "stop", message: nativeThinking }); + }); + expect(nativeRun.result.content.map(b => b.type)).toEqual(["thinking", "text"]); + expect(thinks(nativeRun.result)[0]?.thinking).toBe("native reasoning"); + expect(thinks(nativeRun.result)[0]?.thinkingSignature).toBe("tk"); + expect(texts(nativeRun.result)).toEqual(["answer"]); + }); + + it("heals a terminal error message and keeps its error stop reason", async () => { + const leaked = "Partial.```thinking\noops\n```Recovered."; + const { result } = await runWrapper(inner => { + inner.push({ type: "start", partial: msg() }); + inner.push({ + type: "error", + reason: "error", + error: msg({ content: [{ type: "text", text: leaked }], stopReason: "error" }), + }); + }); + + expect(result.content.map(b => b.type)).toEqual(["text", "thinking", "text"]); + expect(texts(result)).toEqual(["Partial.", "Recovered."]); + expect(result.stopReason).toBe("error"); + }); +}); + +describe("leaked thinking healing through stream()", () => { + function sseFrame(event: string, data: unknown): string { + return `event: ${event}\ndata: ${JSON.stringify(data)}\n\n`; + } + + function anthropicLeakFetch(text: string): FetchImpl { + const body = [ + sseFrame("message_start", { + type: "message_start", + message: { id: "msg_leak", usage: { input_tokens: 5, output_tokens: 0 } }, + }), + sseFrame("content_block_start", { + type: "content_block_start", + index: 0, + content_block: { type: "text", text: "" }, + }), + sseFrame("content_block_delta", { + type: "content_block_delta", + index: 0, + delta: { type: "text_delta", text }, + }), + sseFrame("content_block_stop", { type: "content_block_stop", index: 0 }), + sseFrame("message_delta", { + type: "message_delta", + delta: { stop_reason: "end_turn" }, + usage: { input_tokens: 5, output_tokens: 4 }, + }), + sseFrame("message_stop", { type: "message_stop" }), + ].join(""); + const fn = async (_input: string | URL | Request, _init?: RequestInit): Promise => + new Response(body, { + status: 200, + headers: { "content-type": "text/event-stream", "request-id": "req_mock" }, + }); + return Object.assign(fn, { preconnect: fetch.preconnect }); + } + + it("splits a leaked fence from a provider with no own healer", async () => { + // Anthropic has no provider-local visible-text healer, so a split here + // proves the central wrapper is composed into stream(). + const model: Model<"anthropic-messages"> = buildModel({ + id: "claude-sonnet-4-5", + name: "Claude Sonnet 4.5", + api: "anthropic-messages", + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 8_192, + }); + const leaked = "```thinking\nDeliberate.\n```\nFinal answer."; + const context: Context = { messages: [{ role: "user", content: "hi", timestamp: Date.now() }] }; + const result = await stream(model, context, { + apiKey: "test", + fetch: anthropicLeakFetch(leaked), + }).result(); + + expect(result.content.map(b => b.type)).toEqual(["thinking", "text"]); + const thinking = thinks(result) + .map(b => b.thinking) + .join(""); + expect(thinking).toContain("Deliberate."); + expect(texts(result).join("").trim()).toBe("Final answer."); + }); +}); diff --git a/packages/ai/test/ollama-tool-call-json-parse-error.test.ts b/packages/ai/test/ollama-tool-call-json-parse-error.test.ts new file mode 100644 index 000000000..7b2addee1 --- /dev/null +++ b/packages/ai/test/ollama-tool-call-json-parse-error.test.ts @@ -0,0 +1,68 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import { scheduler } from "node:timers/promises"; +import * as AIError from "@oh-my-pi/pi-ai/error"; +import { streamOllama } from "@oh-my-pi/pi-ai/providers/ollama"; +import type { Context, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +const model: Model<"ollama-chat"> = buildModel({ + id: "qwen3.6-coder:27b", + name: "Qwen 3.6 Coder 27B", + api: "ollama-chat", + provider: "ollama", + baseUrl: "http://127.0.0.1:11434", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 131_072, + maxTokens: 8192, +}); + +const context: Context = { + messages: [{ role: "user", content: "Use bash to run a Python command.", timestamp: 0 }], +}; + +const llamaToolParseFailure = JSON.stringify({ + error: { + code: 500, + message: + "Failed to parse tool call arguments as JSON: [json.exception.parse_error.101] parse error at line 1, column 557: syntax error while parsing value - invalid string: missing closing quote; last read: '\"uv run python -c \\\"\\nimport jax.numpy as jnp\\nimport'", + }, +}); + +describe("Ollama malformed tool-call JSON errors", () => { + afterEach(() => vi.restoreAllMocks()); + + it("does not retry deterministic llama.cpp tool argument parse failures", async () => { + vi.spyOn(scheduler, "wait").mockResolvedValue(undefined); + let calls = 0; + const fetchMock = async () => { + calls += 1; + return new Response(llamaToolParseFailure, { status: 500 }); + }; + + const result = await streamOllama(model, context, { + apiKey: "ollama", + fetch: fetchMock, + }).result(); + + expect(calls).toBe(1); + expect(result.stopReason).toBe("error"); + expect(result.errorStatus).toBe(500); + expect(result.errorMessage).toContain("Local Ollama model emitted malformed tool-call JSON"); + expect(result.errorMessage).toContain("reload the model"); + }); + + it("strips Transient so agent-level auto-retry will not replay the deterministic failure", async () => { + vi.spyOn(scheduler, "wait").mockResolvedValue(undefined); + const fetchMock = async () => new Response(llamaToolParseFailure, { status: 500 }); + + const result = await streamOllama(model, context, { + apiKey: "ollama", + fetch: fetchMock, + }).result(); + + expect(AIError.is(result.errorId, AIError.Flag.Transient)).toBe(false); + expect(AIError.retriable(result.errorId)).toBe(false); + }); +}); diff --git a/packages/ai/test/openai-codex-responses-lite.test.ts b/packages/ai/test/openai-codex-responses-lite.test.ts index 575c3cdd3..c6aee594f 100644 --- a/packages/ai/test/openai-codex-responses-lite.test.ts +++ b/packages/ai/test/openai-codex-responses-lite.test.ts @@ -83,21 +83,21 @@ function createCodexFetchMock(sse: string, onRequest: (captured: CapturedCodexRe } describe("openai-codex reasoning.context", () => { - it("forwards an explicit reasoning.context and defaults to all_turns", async () => { - const model = createCodexModel("gpt-5.1-codex"); + it("defaults to all_turns on gpt-5.4+ models and forwards explicit overrides", async () => { + const model = createCodexModel("gpt-5.4"); + + const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" }); + expect(defaulted.reasoning?.context).toBe("all_turns"); const explicit = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium", reasoningContext: "current_turn", }); expect(explicit.reasoning?.context).toBe("current_turn"); - - const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" }); - expect(defaulted.reasoning?.context).toBe("all_turns"); }); - it("defaults reasoning.context to all_turns under Responses Lite unless overridden", async () => { - const model = createCodexModel("gpt-5.1-codex"); + it("keeps the all_turns default for the lite transport on supported models", async () => { + const model = createCodexModel("gpt-5.5"); const lite = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium", @@ -112,6 +112,39 @@ describe("openai-codex reasoning.context", () => { }); expect(overridden.reasoning?.context).toBe("auto"); }); + + // gpt-5.1-codex / gpt-5.3-codex / gpt-5.3-codex-spark reject `all_turns` + // ("Unsupported value: 'all_turns' is not supported with this model"). + it.each([ + "gpt-5.1-codex", + "gpt-5.3-codex", + "gpt-5.3-codex-spark", + ])("omits the all_turns default for pre-5.4 model %s", async modelId => { + const model = createCodexModel(modelId); + + const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" }); + expect(defaulted.reasoning).toBeDefined(); + expect(defaulted.reasoning?.context).toBeUndefined(); + expect("context" in (defaulted.reasoning ?? {})).toBe(false); + + // A supported override (current_turn/auto) is still honored. + const overridden = await transformRequestBody({ model: model.id }, model, { + reasoningEffort: "medium", + reasoningContext: "current_turn", + }); + expect(overridden.reasoning?.context).toBe("current_turn"); + }); + + it("suppresses an explicit all_turns override on a pre-5.4 model", async () => { + const model = createCodexModel("gpt-5.3-codex-spark"); + + const forced = await transformRequestBody({ model: model.id }, model, { + reasoningEffort: "medium", + reasoningContext: "all_turns", + }); + expect(forced.reasoning).toBeDefined(); + expect(forced.reasoning?.context).toBeUndefined(); + }); }); describe("openai-codex Responses Lite input shaping", () => { diff --git a/packages/ai/test/openai-completions-tool-result-images.test.ts b/packages/ai/test/openai-completions-tool-result-images.test.ts index 6b13398be..bac431780 100644 --- a/packages/ai/test/openai-completions-tool-result-images.test.ts +++ b/packages/ai/test/openai-completions-tool-result-images.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from "bun:test"; import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { NON_VISION_IMAGE_PLACEHOLDER } from "@oh-my-pi/pi-ai/providers/vision-guard"; import type { AssistantMessage, Context, Model, ToolResultMessage, Usage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types"; @@ -75,6 +76,48 @@ function buildToolResult(toolCallId: string, timestamp: number): ToolResultMessa } describe("openai-completions convertMessages", () => { + it("serializes Cerebras gemma image inputs as Chat Completions data URIs", () => { + const model = buildModel({ + id: "gemma-4-31b", + name: "Gemma 4 31B", + api: "openai-completions", + provider: "cerebras", + baseUrl: "https://api.cerebras.ai/v1", + reasoning: false, + input: ["text", "image"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128_000, + maxTokens: 8_192, + }); + const context: Context = { + messages: [ + { + role: "user", + content: [ + { type: "text", text: "Identify the shapes and colors. Return JSON only." }, + { type: "image", mimeType: "image/png", data: "ZmFrZQ==" }, + ], + timestamp: 1, + }, + ], + }; + + const messages = convertMessages(model, context, compat); + + expect(messages).toEqual([ + { + role: "user", + content: [ + { type: "text", text: "Identify the shapes and colors. Return JSON only." }, + { + type: "image_url", + image_url: { url: "data:image/png;base64,ZmFrZQ==" }, + }, + ], + }, + ]); + }); + it("batches tool-result images after consecutive tool results", () => { const baseModel = getBundledModel("openai", "gpt-4o-mini") as Model<"openai-completions">; const model: Model<"openai-completions"> = { diff --git a/packages/ai/test/service-tier-premium-requests.test.ts b/packages/ai/test/service-tier-premium-requests.test.ts index 17f12c1cc..7a217d3be 100644 --- a/packages/ai/test/service-tier-premium-requests.test.ts +++ b/packages/ai/test/service-tier-premium-requests.test.ts @@ -1,138 +1,149 @@ import { describe, expect, it } from "bun:test"; -import { getPriorityPremiumRequests, resolveServiceTier, shouldSendServiceTier } from "@oh-my-pi/pi-ai/types"; +import type { Api } from "@oh-my-pi/pi-ai/types"; +import { + coerceServiceTierByFamily, + getPriorityPremiumRequests, + realizesPriorityServiceTier, + resolveModelServiceTier, + serviceTierFamily, + shouldSendServiceTier, +} from "@oh-my-pi/pi-ai/types"; -describe("getPriorityPremiumRequests", () => { - it("counts priority tier as one premium request on OpenAI", () => { - expect(getPriorityPremiumRequests("priority", "openai")).toBe(1); +const m = (provider: string, api: Api, id: string): { provider: string; api: Api; id: string } => ({ + provider, + api, + id, +}); + +const openai = m("openai", "openai-responses", "gpt-5"); +const codex = m("openai-codex", "openai-codex-responses", "gpt-5.5"); +const anthropic = m("anthropic", "anthropic-messages", "claude-opus-4-6"); +const vertexClaude = m("google-vertex", "anthropic-messages", "claude-opus-4-6"); +const gemini = m("google", "google-generative-ai", "gemini-3-flash"); +const vertexGemini = m("google-vertex", "google-vertex", "gemini-3-flash"); +const fireworks = m("fireworks", "openai-completions", "qwen3"); +const orOpenAI = m("openrouter", "openai-responses", "openai/gpt-5.5"); +const orGoogle = m("openrouter", "openai-completions", "google/gemini-3-flash"); +const orAnthropic = m("openrouter", "openai-completions", "anthropic/claude-opus-4-6"); + +describe("serviceTierFamily", () => { + it("classifies first-party providers by provider/api", () => { + expect(serviceTierFamily(openai)).toBe("openai"); + expect(serviceTierFamily(codex)).toBe("openai"); + expect(serviceTierFamily(anthropic)).toBe("anthropic"); + expect(serviceTierFamily(vertexClaude)).toBe("anthropic"); // Claude on Vertex is the anthropic family + expect(serviceTierFamily(gemini)).toBe("google"); + expect(serviceTierFamily(vertexGemini)).toBe("google"); + expect(serviceTierFamily(fireworks)).toBeUndefined(); }); - it("counts priority tier as one premium request on OpenAI Codex", () => { - expect(getPriorityPremiumRequests("priority", "openai-codex")).toBe(1); - }); - - it("ignores non-priority paid tiers", () => { - expect(getPriorityPremiumRequests("flex", "openai")).toBe(0); - expect(getPriorityPremiumRequests("scale", "openai")).toBe(0); - }); - - it("ignores default and auto tiers", () => { - expect(getPriorityPremiumRequests("default", "openai")).toBe(0); - expect(getPriorityPremiumRequests("auto", "openai")).toBe(0); - }); - - it("ignores priority tier on providers that drop service_tier", () => { - // `priority` is realized on `openai`, `openai-codex`, and direct `anthropic` - // (as fast mode). Everywhere else it's silently dropped, so it must not - // be billed as premium. - expect(getPriorityPremiumRequests("priority", "github-copilot")).toBe(0); - expect(getPriorityPremiumRequests("priority", "azure")).toBe(0); - expect(getPriorityPremiumRequests("priority", "bedrock")).toBe(0); - }); - - it("counts priority on direct Anthropic as one premium request (fast mode)", () => { - expect(getPriorityPremiumRequests("priority", "anthropic")).toBe(1); - }); - - it("returns zero when service tier is unset", () => { - expect(getPriorityPremiumRequests(undefined, "openai")).toBe(0); - expect(getPriorityPremiumRequests(null, "openai")).toBe(0); - }); - - describe("scoped tiers", () => { - it("treats `openai-only` as priority on OpenAI and OpenAI-Codex", () => { - expect(getPriorityPremiumRequests("openai-only", "openai")).toBe(1); - expect(getPriorityPremiumRequests("openai-only", "openai-codex")).toBe(1); - }); - - it("treats `openai-only` as inactive on Anthropic and everywhere else", () => { - expect(getPriorityPremiumRequests("openai-only", "anthropic")).toBe(0); - expect(getPriorityPremiumRequests("openai-only", "github-copilot")).toBe(0); - expect(getPriorityPremiumRequests("openai-only", "bedrock")).toBe(0); - }); - - it("treats `claude-only` as priority on direct Anthropic", () => { - expect(getPriorityPremiumRequests("claude-only", "anthropic")).toBe(1); - }); - - it("treats `claude-only` as inactive on OpenAI, Bedrock/Vertex, and elsewhere", () => { - expect(getPriorityPremiumRequests("claude-only", "openai")).toBe(0); - expect(getPriorityPremiumRequests("claude-only", "openai-codex")).toBe(0); - expect(getPriorityPremiumRequests("claude-only", "bedrock")).toBe(0); - expect(getPriorityPremiumRequests("claude-only", "vertex")).toBe(0); - }); + it("classifies OpenRouter models by id namespace", () => { + expect(serviceTierFamily(orOpenAI)).toBe("openai"); + expect(serviceTierFamily(orGoogle)).toBe("google"); + expect(serviceTierFamily(orAnthropic)).toBe("anthropic"); + expect(serviceTierFamily(m("openrouter", "openai-completions", "z-ai/glm-4.7"))).toBeUndefined(); }); }); -describe("resolveServiceTier", () => { - it("passes unscoped tiers through unchanged for any provider", () => { - expect(resolveServiceTier("flex", "openai")).toBe("flex"); - expect(resolveServiceTier("priority", "anthropic")).toBe("priority"); - expect(resolveServiceTier("auto", "openai-codex")).toBe("auto"); - expect(resolveServiceTier("default", "github-copilot")).toBe("default"); - }); - - it("scopes `openai-only` to OpenAI providers", () => { - expect(resolveServiceTier("openai-only", "openai")).toBe("priority"); - expect(resolveServiceTier("openai-only", "openai-codex")).toBe("priority"); - expect(resolveServiceTier("openai-only", "anthropic")).toBeUndefined(); - expect(resolveServiceTier("openai-only", "bedrock")).toBeUndefined(); - expect(resolveServiceTier("openai-only", undefined)).toBeUndefined(); - }); - - it("scopes `claude-only` to direct Anthropic", () => { - expect(resolveServiceTier("claude-only", "anthropic")).toBe("priority"); - expect(resolveServiceTier("claude-only", "openai")).toBeUndefined(); - expect(resolveServiceTier("claude-only", "bedrock")).toBeUndefined(); - expect(resolveServiceTier("claude-only", "vertex")).toBeUndefined(); - }); - - it("returns undefined for null/undefined input", () => { - expect(resolveServiceTier(undefined, "openai")).toBeUndefined(); - expect(resolveServiceTier(null, "openai")).toBeUndefined(); +describe("resolveModelServiceTier", () => { + it("reduces a per-family map to the model's family entry", () => { + const tiers = { openai: "priority", anthropic: "priority", google: "flex" } as const; + expect(resolveModelServiceTier(tiers, openai)).toBe("priority"); + expect(resolveModelServiceTier(tiers, gemini)).toBe("flex"); + expect(resolveModelServiceTier(tiers, orAnthropic)).toBe("priority"); + expect(resolveModelServiceTier(tiers, fireworks)).toBeUndefined(); // no family + expect(resolveModelServiceTier(undefined, openai)).toBeUndefined(); + expect(resolveModelServiceTier({ google: "priority" }, openai)).toBeUndefined(); }); }); describe("shouldSendServiceTier", () => { - it("returns false for non-OpenAI/non-Fireworks providers", () => { - expect(shouldSendServiceTier("flex", "azure-openai-responses")).toBe(false); - expect(shouldSendServiceTier("scale", "firepass")).toBe(false); + it("sends flex/scale/priority on the OpenAI family and OpenRouter", () => { + for (const p of ["openai", "openai-codex", "openrouter"]) { + expect(shouldSendServiceTier("flex", p)).toBe(true); + expect(shouldSendServiceTier("scale", p)).toBe(true); + expect(shouldSendServiceTier("priority", p)).toBe(true); + expect(shouldSendServiceTier("default", p)).toBe(false); + expect(shouldSendServiceTier("auto", p)).toBe(false); + } + }); + + it("sends flex/priority on direct Google, priority-only on Vertex (no scale)", () => { + expect(shouldSendServiceTier("flex", "google")).toBe(true); + expect(shouldSendServiceTier("priority", "google")).toBe(true); + expect(shouldSendServiceTier("scale", "google")).toBe(false); + expect(shouldSendServiceTier("priority", "google-vertex")).toBe(true); + expect(shouldSendServiceTier("flex", "google-vertex")).toBe(false); // Vertex flex has no wire control + }); + + it("sends only priority on Fireworks, nothing on Anthropic", () => { + expect(shouldSendServiceTier("priority", "fireworks")).toBe(true); + expect(shouldSendServiceTier("flex", "fireworks")).toBe(false); expect(shouldSendServiceTier("priority", "anthropic")).toBe(false); }); - it("returns true for fireworks only with the priority tier", () => { - expect(shouldSendServiceTier("priority", "fireworks")).toBe(true); - // Fireworks realizes only the Priority serving path — flex/scale are OpenAI-only. - expect(shouldSendServiceTier("flex", "fireworks")).toBe(false); - expect(shouldSendServiceTier("scale", "fireworks")).toBe(false); - expect(shouldSendServiceTier("auto", "fireworks")).toBe(false); - expect(shouldSendServiceTier("default", "fireworks")).toBe(false); - expect(shouldSendServiceTier(undefined, "fireworks")).toBe(false); - }); - - it("returns true for openai with priority/flex/scale tiers", () => { - expect(shouldSendServiceTier("priority", "openai")).toBe(true); - expect(shouldSendServiceTier("flex", "openai")).toBe(true); - expect(shouldSendServiceTier("scale", "openai")).toBe(true); - }); - - it("returns true for openai-codex with priority/flex/scale tiers", () => { - expect(shouldSendServiceTier("priority", "openai-codex")).toBe(true); - expect(shouldSendServiceTier("flex", "openai-codex")).toBe(true); - expect(shouldSendServiceTier("scale", "openai-codex")).toBe(true); - }); - - it("returns false for default tier on OpenAI providers", () => { - expect(shouldSendServiceTier("default", "openai")).toBe(false); - expect(shouldSendServiceTier("default", "openai-codex")).toBe(false); - }); - - it("returns false for auto tier on OpenAI providers", () => { - expect(shouldSendServiceTier("auto", "openai")).toBe(false); - expect(shouldSendServiceTier("auto", "openai-codex")).toBe(false); - }); - - it("returns false for undefined/null tier", () => { + it("returns false for unset tiers", () => { expect(shouldSendServiceTier(undefined, "openai")).toBe(false); expect(shouldSendServiceTier(null, "openai")).toBe(false); }); }); + +describe("realizesPriorityServiceTier", () => { + it("realizes priority where the wire actually applies it", () => { + expect(realizesPriorityServiceTier("priority", openai)).toBe(true); + expect(realizesPriorityServiceTier("priority", anthropic)).toBe(true); // direct fast mode + expect(realizesPriorityServiceTier("priority", gemini)).toBe(true); + expect(realizesPriorityServiceTier("priority", vertexGemini)).toBe(true); + expect(realizesPriorityServiceTier("priority", fireworks)).toBe(true); + expect(realizesPriorityServiceTier("priority", orOpenAI)).toBe(true); + expect(realizesPriorityServiceTier("priority", orGoogle)).toBe(true); + }); + + it("does not realize priority where the wire drops it", () => { + expect(realizesPriorityServiceTier("priority", vertexClaude)).toBe(false); // no fast mode on Vertex + expect(realizesPriorityServiceTier("priority", orAnthropic)).toBe(false); // OpenRouter Anthropic + expect(realizesPriorityServiceTier("flex", openai)).toBe(false); + expect(realizesPriorityServiceTier(undefined, openai)).toBe(false); + }); +}); + +describe("getPriorityPremiumRequests", () => { + it("counts one premium request per realized priority on billing providers", () => { + expect(getPriorityPremiumRequests("priority", openai)).toBe(1); + expect(getPriorityPremiumRequests("priority", codex)).toBe(1); + expect(getPriorityPremiumRequests("priority", anthropic)).toBe(1); + expect(getPriorityPremiumRequests("priority", gemini)).toBe(1); + expect(getPriorityPremiumRequests("priority", vertexGemini)).toBe(1); + }); + + it("does not bill OpenRouter, unrealized, or non-priority traffic", () => { + expect(getPriorityPremiumRequests("priority", orOpenAI)).toBe(0); // OpenRouter bills its own way + expect(getPriorityPremiumRequests("priority", vertexClaude)).toBe(0); // not realized + expect(getPriorityPremiumRequests("priority", fireworks)).toBe(0); // realized but not Copilot-premium + expect(getPriorityPremiumRequests("flex", openai)).toBe(0); + expect(getPriorityPremiumRequests(undefined, openai)).toBe(0); + }); +}); + +describe("coerceServiceTierByFamily", () => { + it("migrates legacy scalar values to a per-family map", () => { + expect(coerceServiceTierByFamily("priority")).toEqual({ + openai: "priority", + anthropic: "priority", + google: "priority", + }); + expect(coerceServiceTierByFamily("openai-only")).toEqual({ openai: "priority" }); + expect(coerceServiceTierByFamily("claude-only")).toEqual({ anthropic: "priority" }); + expect(coerceServiceTierByFamily("flex")).toEqual({ openai: "flex" }); + expect(coerceServiceTierByFamily("none")).toBeUndefined(); + expect(coerceServiceTierByFamily(null)).toBeUndefined(); + }); + + it("passes a per-family map through, dropping invalid entries", () => { + expect(coerceServiceTierByFamily({ openai: "priority", google: "flex" })).toEqual({ + openai: "priority", + google: "flex", + }); + expect(coerceServiceTierByFamily({ openai: "bogus" })).toBeUndefined(); + }); +}); diff --git a/packages/ai/test/stream-auth-retry.test.ts b/packages/ai/test/stream-auth-retry.test.ts index 307b37412..484d80f01 100644 --- a/packages/ai/test/stream-auth-retry.test.ts +++ b/packages/ai/test/stream-auth-retry.test.ts @@ -153,9 +153,9 @@ describe("streamSimple resolver auth retry", () => { expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); expect(keys).toEqual(["old-key", "new-key"]); - // The buffered `start` of the failed attempt must not leak — the user - // sees exactly one clean start/done pair. - expect(eventTypes).toEqual(["start", "done"]); + // The failed attempt's buffered start must not leak — the user sees a + // single start from the successful attempt, then its healed content. + expect(eventTypes).toEqual(["start", "text_start", "text_delta", "text_end", "done"]); }); it("retries on a 401 carried only via errorStatus", async () => { @@ -206,6 +206,12 @@ describe("streamSimple resolver auth retry", () => { queueMicrotask(() => { stream.push({ type: "start", partial: assistant() }); stream.push({ type: "text_start", contentIndex: 0, partial: assistant([""]) }); + stream.push({ + type: "text_delta", + contentIndex: 0, + delta: "partial", + partial: assistant(["partial"]), + }); stream.fail(failure); }); return stream; @@ -398,7 +404,7 @@ describe("streamSimple resolver auth retry", () => { expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); expect(keys).toEqual(["credential-A", "credential-B"]); - expect(eventTypes).toEqual(["start", "done"]); + expect(eventTypes).toEqual(["start", "text_start", "text_delta", "text_end", "done"]); expect(retryContexts.map(ctx => ctx.lastChance)).toEqual([false, true]); } }); diff --git a/packages/ai/test/stream-markup-healing.test.ts b/packages/ai/test/stream-markup-healing.test.ts index 27e72d26a..23f525eeb 100644 --- a/packages/ai/test/stream-markup-healing.test.ts +++ b/packages/ai/test/stream-markup-healing.test.ts @@ -5,9 +5,10 @@ import { type InbandScanEvent, ThinkingInbandScanner, } from "@oh-my-pi/pi-ai/dialect"; +import { streamGoogleGeminiCli } from "@oh-my-pi/pi-ai/providers/google-gemini-cli"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { stream } from "@oh-my-pi/pi-ai/stream"; -import type { Context, FetchImpl, Model, ThinkingContent, Tool, ToolCall } from "@oh-my-pi/pi-ai/types"; +import type { Context, FetchImpl, Model, TextContent, ThinkingContent, Tool, ToolCall } from "@oh-my-pi/pi-ai/types"; import { getStreamMarkupHealingPattern, StreamMarkupHealing } from "@oh-my-pi/pi-ai/utils/stream-markup-healing"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; @@ -38,7 +39,7 @@ interface SseChunk { }>; } -function sseResponse(events: ReadonlyArray): Response { +function sseResponse(events: ReadonlyArray): Response { const payload = `${events .map(event => `data: ${typeof event === "string" ? event : JSON.stringify(event)}`) .join("\n\n")}\n\n`; @@ -48,7 +49,7 @@ function sseResponse(events: ReadonlyArray): Response { }); } -function mockFetch(events: ReadonlyArray): FetchImpl { +function mockFetch(events: ReadonlyArray): FetchImpl { const fn = async (_input: string | URL | Request, _init?: RequestInit): Promise => sseResponse(events); return Object.assign(fn, { preconnect: fetch.preconnect }); } @@ -124,6 +125,21 @@ const deepseekCloudModel: Model<"ollama-chat"> = buildModel({ maxTokens: 8_192, }); +function geminiCliModel(): Model<"google-gemini-cli"> { + return buildModel({ + id: "gemini-3.5-flash", + name: "Gemini 3.5 Flash", + api: "google-gemini-cli", + provider: "google-antigravity", + baseUrl: "https://antigravity.test", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1_000_000, + maxTokens: 8_192, + }); +} + function ndjsonResponse(lines: ReadonlyArray): Response { const body = `${lines.map(line => JSON.stringify(line)).join("\n")}\n`; const encoder = new TextEncoder(); @@ -230,6 +246,73 @@ describe("openai-completions leaked thinking healing", () => { }); }); +describe("google-gemini-cli leaked thinking healing", () => { + it("lifts a leaked Gemini thinking fence before a native tool call", async () => { + const model = geminiCliModel(); + const fetchMock = mockFetch([ + { + response: { + candidates: [ + { + content: { + role: "model", + parts: [ + { + text: "```thinking\nCheck the provider path.\n```\nI will inspect the file.", + thoughtSignature: "visible-text-signature", + }, + { + functionCall: { + name: "read", + args: { path: "packages/ai/src/providers/google-gemini-cli.ts" }, + id: "call_read_1", + }, + thoughtSignature: "function-call-signature", + }, + ], + }, + finishReason: "STOP", + }, + ], + usageMetadata: { + promptTokenCount: 10, + candidatesTokenCount: 5, + thoughtsTokenCount: 3, + totalTokenCount: 18, + }, + }, + }, + ]); + + const result = await streamGoogleGeminiCli( + model, + { ...baseContext(), tools: [readTool] }, + { + apiKey: JSON.stringify({ token: "test-token", projectId: "test-project" }), + fetch: fetchMock, + }, + ).result(); + + expect(result.content.map(block => block.type)).toEqual(["thinking", "text", "toolCall"]); + const thinking = result.content + .filter((block): block is ThinkingContent => block.type === "thinking") + .map(block => block.thinking) + .join(""); + const textBlocks = result.content.filter((block): block is TextContent => block.type === "text"); + const text = textBlocks.map(block => block.text).join(""); + const calls = result.content.filter((block): block is ToolCall => block.type === "toolCall"); + + expect(thinking).toBe("Check the provider path.\n"); + expect(text).toBe("\nI will inspect the file."); + expect(text).not.toContain("```thinking"); + expect(calls).toHaveLength(1); + expect(textBlocks[0]?.textSignature).toBe("visible-text-signature"); + expect(calls[0]?.id).toBe("call_read_1"); + expect(calls[0]?.thoughtSignature).toBe("function-call-signature"); + expect(result.stopReason).toBe("toolUse"); + }); +}); + describe("StreamMarkupHealing DSML envelope pattern", () => { it("parses the reporter's verbatim leak into a structured tool call", () => { const healing = new StreamMarkupHealing({ pattern: "dsml" }); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 09b0580c6..bf9d34037 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,23 @@ ## [Unreleased] +## [16.2.9] - 2026-06-30 + +### Added + +- Added full capability support for Claude Sonnet 5, aligning it with Claude Opus 4.8 and Fable 5. This includes adaptive thinking display, mid-conversation system messages, sampling parameter and thinking omission API restrictions, and 5-tier adaptive reasoning effort mapping (including xhigh and max levels) across direct APIs, OpenRouter, and Bedrock Converse. + +### Changed + +- Updated input and output costs for models in the catalog. + +## [16.2.7] - 2026-06-30 + +### Fixed + +- Fixed compatibility with Kimi K2.7 Code on native endpoints to ensure thinking mode is preserved and tool choice is not forced. +- Fixed Cerebras gemma-4-31b dynamic discovery to correctly identify the model as image-capable, enabling proper serialization of attached images. + ## [16.2.6] - 2026-06-29 ### Fixed diff --git a/packages/catalog/package.json b/packages/catalog/package.json index 9f09ce14e..966cf2524 100644 --- a/packages/catalog/package.json +++ b/packages/catalog/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-catalog", - "version": "16.2.6", + "version": "16.2.9", "description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/catalog/src/compat/anthropic.ts b/packages/catalog/src/compat/anthropic.ts index 93390b0e3..cdff05ec2 100644 --- a/packages/catalog/src/compat/anthropic.ts +++ b/packages/catalog/src/compat/anthropic.ts @@ -28,6 +28,14 @@ export function isOfficialAnthropicApiUrl(baseUrl?: string): boolean { return lower === OFFICIAL_ANTHROPIC_URL || lower.startsWith(`${OFFICIAL_ANTHROPIC_URL}/`); } +/** Mirrors `compat/openai.ts`; native-only host gating is the caller's responsibility. */ +const KIMI_K27_CODE_MODEL_PATTERN = /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code(?:[-._]?highspeed)?$/i; + +function matchesKimiK27CodeFamily(spec: ModelSpec<"anthropic-messages">): boolean { + if (KIMI_K27_CODE_MODEL_PATTERN.test(spec.id)) return true; + return spec.id === "kimi-for-coding" && /k2\.?7 code/i.test(spec.name ?? ""); +} + /** Build the resolved anthropic-messages compat record for a model spec. */ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): ResolvedAnthropicCompat { const baseUrl = spec.baseUrl; @@ -40,6 +48,7 @@ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): Res // doesn't whitelist the `fine-grained-tool-streaming-2025-05-14` beta either // (issue #2558), so eager tool-input streaming is unavailable on this host. const isCopilot = modelMatchesHost(spec, "githubCopilot"); + const requiresThinkingEnabled = modelMatchesHost(spec, "moonshotNative") && matchesKimiK27CodeFamily(spec); const compat: ResolvedAnthropicCompat = { officialEndpoint: official, disableStrictTools: false, @@ -53,12 +62,13 @@ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): Res // detection requires the canonical api.anthropic.com host plus a // supported model id. supportsMidConversationSystem: official && supportsMidConversationSystemMessages(spec.id), - supportsForcedToolChoice: !isAnthropicFableOrMythosModel(spec.id), + supportsForcedToolChoice: !requiresThinkingEnabled && !isAnthropicFableOrMythosModel(spec.id), // Opus 4.7+ and Fable/Mythos reject temperature/top_p/top_k with a 400. supportsSamplingParams: !hasOpus47ApiRestrictions(spec.id), // Z.AI workaround (issue #814): its proxy deserializes tool_result blocks // into a class that reads `.id`. requiresToolResultId: isZai, + requiresThinkingEnabled, // Official Anthropic enforces signature-based thinking-chain integrity, so // unsigned thinking blocks must stay text there. Anthropic-compatible // reasoning endpoints commonly emit unsigned thinking blocks while still diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index 26ae8b357..d0ca9447e 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -39,6 +39,20 @@ const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000; const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; /** Kimi K2.6 can spend several minutes reasoning before the first visible token. */ const KIMI_K26_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; +/** + * Native Kimi K2.7 Code requires `thinking.type: "enabled"` and rejects + * disabled thinking. Match the public id, its Fast variant, and the + * `kimi-code/kimi-for-coding` alias (which keeps the family name). + * Caller-disabled requests on non-native dialects (Fireworks `openai`, + * OpenRouter `openrouter`, …) MUST keep their per-dialect disable shape — + * gating on `isMoonshotKimi` is the caller's responsibility. + */ +const KIMI_K27_CODE_MODEL_PATTERN = /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code(?:[-._]?highspeed)?$/i; + +function matchesKimiK27CodeFamily(spec: ModelSpec<"openai-completions">): boolean { + if (KIMI_K27_CODE_MODEL_PATTERN.test(spec.id)) return true; + return spec.id === "kimi-for-coding" && /k2\.?7 code/i.test(spec.name ?? ""); +} /** Xiaomi MiMo Pro on api.xiaomimimo.com can stall ~2min before the first event (issue #1770). */ const XIAOMI_MIMO_STREAM_IDLE_TIMEOUT_MS = 300_000; /** Alibaba Coding Plan (coding-intl.dashscope) qwen models idle before the first event (issue #1770). */ @@ -231,6 +245,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv const isKimiModel = isKimiModelId(spec.id); const isMoonshotNative = modelMatchesHost(hostModel, "moonshotNative"); const isMoonshotKimi = isKimiModel && isMoonshotNative; + const requiresEnabledThinking = isMoonshotKimi && matchesKimiK27CodeFamily(spec); const usesMoonshotKimiPreservedThinking = isMoonshotKimi && isKimiK26ModelId(spec.id); const isAnthropicModel = modelMatchesHost(hostModel, "anthropic") || isClaudeModelId(spec.id) || isAnthropicNamespacedModelId(spec.id); @@ -403,7 +418,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel, disableReasoningOnToolChoice: isDeepseekFamily && Boolean(spec.reasoning) && !isOpenRouter, supportsToolChoice: !isDirectDeepseekReasoning, - supportsForcedToolChoice: true, + supportsForcedToolChoice: !requiresEnabledThinking, supportsNamedToolChoice: provider !== "llama.cpp", maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens", requiresToolResultName: isMistral, @@ -509,7 +524,9 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv applyCompatOverrides(compat, spec.compat); if (spec.compat?.reasoningDisableMode === undefined) { - compat.reasoningDisableMode = resolveReasoningDisableMode(compat.thinkingFormat); + compat.reasoningDisableMode = requiresEnabledThinking + ? "omit" + : resolveReasoningDisableMode(compat.thinkingFormat); } if (spec.compat?.omitReasoningEffort === undefined && !compat.supportsReasoningEffort) { compat.omitReasoningEffort = true; diff --git a/packages/catalog/src/identity/classify.ts b/packages/catalog/src/identity/classify.ts index 624b071c7..7028b56e5 100644 --- a/packages/catalog/src/identity/classify.ts +++ b/packages/catalog/src/identity/classify.ts @@ -158,6 +158,20 @@ export function isFableOrMythos(kind: AnthropicKind): boolean { return kind === "fable" || kind === "mythos"; } +/** + * Returns true if the parsed Anthropic model is part of the adaptive-thinking + * Claude generation at or above a specific capability threshold. + * - Opus has a configurable minimum version floor (e.g. "4.6", "4.7", "4.8"). + * - Sonnet, Fable, and Mythos all require version 5 or higher. + */ +export function isAnthropicAdaptiveGenAtLeast(parsed: AnthropicModel, opusMin: "4.6" | "4.7" | "4.8"): boolean { + if (parsed.kind === "opus") { + return semverGte(parsed.version, opusMin); + } + // Sonnet 5+, Fable 5+, Mythos 5+, and any future gen-5+ models + return semverGte(parsed.version, "5"); +} + function createSemVer(major: number, minor: number, patch = 0): SemVer { return { major, minor, patch }; } diff --git a/packages/catalog/src/identity/family.ts b/packages/catalog/src/identity/family.ts index c48d91f35..1f80ead61 100644 --- a/packages/catalog/src/identity/family.ts +++ b/packages/catalog/src/identity/family.ts @@ -9,10 +9,12 @@ import { bareModelId, + isAnthropicAdaptiveGenAtLeast, isFableOrMythos, parseAnthropicModel, parseGlmModel, parseKnownModel, + parseOpenAIModel, semverGte, } from "./classify"; @@ -121,6 +123,22 @@ export const isOpenAIModelId = memo((modelId: string): boolean => { return /(^|\/)(gpt|o1|o3|o4)[-.]/i.test(modelId) || modelId.toLowerCase().includes("openai/"); }); +/** + * OpenAI Codex models that honor `reasoning.context: "all_turns"` (full + * cross-turn reasoning replay). The `reasoning.context` field itself exists for + * the whole gpt-5/o-series family, but the `all_turns` value is only accepted + * from gpt-5.4 onward; earlier ids (`gpt-5.1-codex`, `gpt-5.3-codex`, and + * `gpt-5.3-codex-spark`) reject it with + * `Unsupported value: 'all_turns' is not supported with this model`. Version + * floor (not an allowlist) so 5.6/6.x inherit support automatically. Callers + * fall back to omitting `context`, letting the server default to `current_turn`. + */ +export const supportsAllTurnsReasoningContext = memo((modelId: string): boolean => { + const parsed = parseOpenAIModel(bareModelId(modelId)); + if (!parsed) return false; + return semverGte(parsed.version, "5.4"); +}); + /** * Reasoning-capable GLM coding SKUs: glm-4.5 and up on the base / `-air` / * `-turbo` lines. Excludes the vision (`…v`) shape, the non-reasoning @@ -184,41 +202,38 @@ export const modelFamilyToken = memo((modelId: string): string => { }); /** - * Adaptive thinking `display` is supported starting with Claude Opus 4.7 and - * the Claude Fable/Mythos 5 generation. Older adaptive-thinking models - * (Opus 4.6, Sonnet 4.6+) reject the field. Classifier-based, so dotted and - * dashed version forms both match while bare dated ids + * Adaptive thinking `display` is supported starting with Claude Opus 4.7+, + * Sonnet 5+, and the Claude Fable/Mythos 5 generation. Older adaptive-thinking + * models (Opus 4.6, Sonnet 4.6) reject the field. Classifier-based, so dotted + * and dashed version forms both match while bare dated ids * (`claude-opus-4-20250514` = Opus 4.0) stay excluded. */ export const supportsAdaptiveThinkingDisplay = memo((modelId: string): boolean => { const parsed = parseAnthropicModel(bareModelId(modelId)); - if (!parsed) return false; - if (isFableOrMythos(parsed.kind)) return semverGte(parsed.version, "5"); - return parsed.kind === "opus" && semverGte(parsed.version, "4.7"); + return parsed !== null && isAnthropicAdaptiveGenAtLeast(parsed, "4.7"); }); /** - * Returns true for Anthropic models with Opus 4.7+/Fable/Mythos API restrictions: + * Returns true for Anthropic models with Opus 4.7+, Sonnet 5+, and Fable/Mythos 5+ + * API restrictions: * - Sampling parameters (temperature/top_p/top_k) return 400 error * - Thinking content is omitted by default (needs display: "summarized") */ export const hasOpus47ApiRestrictions = memo((modelId: string): boolean => { const parsed = parseAnthropicModel(bareModelId(modelId)); - if (!parsed) return false; - return (parsed.kind === "opus" && semverGte(parsed.version, "4.7")) || isFableOrMythos(parsed.kind); + return parsed !== null && isAnthropicAdaptiveGenAtLeast(parsed, "4.7"); }); /** * Mid-conversation `role: "system"` messages (system instructions appended at * non-first positions in the `messages` array) are supported starting with - * Claude Opus 4.8 and the Claude Fable/Mythos 5 generation. Earlier Claude - * models reject the role. + * Claude Opus 4.8+, Sonnet 5+, and the Claude Fable/Mythos 5 generation. + * Earlier Claude models reject the role. * @see https://platform.claude.com/docs/en/build-with-claude/mid-conversation-system-messages */ export const supportsMidConversationSystemMessages = memo((modelId: string): boolean => { const parsed = parseAnthropicModel(bareModelId(modelId)); - if (!parsed) return false; - return (parsed.kind === "opus" && semverGte(parsed.version, "4.8")) || isFableOrMythos(parsed.kind); + return parsed !== null && isAnthropicAdaptiveGenAtLeast(parsed, "4.8"); }); export const isAnthropicFableOrMythosModel = memo((modelId: string): boolean => { diff --git a/packages/catalog/src/model-thinking.ts b/packages/catalog/src/model-thinking.ts index 03daa8b45..b0e01af59 100644 --- a/packages/catalog/src/model-thinking.ts +++ b/packages/catalog/src/model-thinking.ts @@ -12,7 +12,7 @@ import { type AnthropicModel, bareModelId, type GeminiModel, - isFableOrMythos, + isAnthropicAdaptiveGenAtLeast, type OpenAIModel, type ParsedModel, parseAnthropicModel, @@ -441,12 +441,11 @@ function isDeepseekReasoningModel(spec: ModelSpec): bool function getOpenRouterAnthropicReasoningEffortMap(modelId: string): EffortMap | undefined { const parsed = parseAnthropicModel(bareModelId(modelId)); if (!parsed) return undefined; - // Adaptive efforts on OpenRouter's completions front: Fable/Mythos and - // Opus 4.6+ only — Sonnet stays on the plain effort vocabulary there. - const isOpusAdaptive = parsed.kind === "opus" && semverGte(parsed.version, "4.6"); - if (!isFableOrMythos(parsed.kind) && !isOpusAdaptive) return undefined; + // Adaptive efforts on OpenRouter's completions front: Fable/Mythos, Sonnet 5+, + // and Opus 4.6+ only — older Sonnet versions stay on the plain effort vocabulary there. + if (!isAnthropicAdaptiveGenAtLeast(parsed, "4.6")) return undefined; - const hasRealXHigh = isFableOrMythos(parsed.kind) || semverGte(parsed.version, "4.7"); + const hasRealXHigh = isAnthropicAdaptiveGenAtLeast(parsed, "4.7"); return hasRealXHigh ? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER; } @@ -521,7 +520,7 @@ function inferAnthropicSupportedEfforts( (spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") && semverGte(parsedModel.version, "4.6") ) { - return parsedModel.kind === "opus" || isFableOrMythos(parsedModel.kind) + return isAnthropicAdaptiveGenAtLeast(parsedModel, "4.6") ? DEFAULT_REASONING_EFFORTS_WITH_XHIGH : DEFAULT_REASONING_EFFORTS; } @@ -601,10 +600,7 @@ function inferThinkingControlMode( case "bedrock-converse-stream": if (parsedModel.family === "anthropic") { - if ( - semverGte(parsedModel.version, "4.6") && - (parsedModel.kind === "opus" || isFableOrMythos(parsedModel.kind)) - ) { + if (isAnthropicAdaptiveGenAtLeast(parsedModel, "4.6")) { return "anthropic-adaptive"; } // Opus 4.5 on Bedrock metadata mirrors the direct-Anthropic @@ -627,19 +623,18 @@ function isOpenRouterAnthropicAdaptiveReasoningModel( ): boolean { if (!isOpenAICompatReasoningApi(spec.api)) return false; if (!modelMatchesHost(spec, "openrouter")) return false; - return isFableOrMythos(parsedModel.kind) || (parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.6")); + return isAnthropicAdaptiveGenAtLeast(parsedModel, "4.6"); } /** - * Opus 4.7+ and Fable/Mythos on the Messages API expose the full five-tier + * Opus 4.7+, Sonnet 5+, and Fable/Mythos 5+ on the Messages API expose the full five-tier * adaptive scale (low/medium/high/xhigh/max). Bedrock Converse stays on the * four-tier scale regardless of model version. */ function anthropicModelHasRealXHighEffort(spec: ModelSpec, parsedModel: ParsedModel): boolean { if (spec.api !== "anthropic-messages") return false; if (parsedModel.family !== "anthropic") return false; - if (isFableOrMythos(parsedModel.kind)) return true; - return parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.7"); + return isAnthropicAdaptiveGenAtLeast(parsedModel, "4.7"); } // --------------------------------------------------------------------------- diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 492404b0c..959efab5b 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -2476,7 +2476,7 @@ "api": "openai-completions", "provider": "aimlapi", "baseUrl": "https://api.aimlapi.com/v1", - "reasoning": true, + "reasoning": false, "input": [ "text" ], @@ -2487,24 +2487,7 @@ "cacheWrite": 0 }, "contextWindow": 163840, - "maxTokens": 16384, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } + "maxTokens": 16384 }, "deepseek/deepseek-chat-v3.1": { "id": "deepseek/deepseek-chat-v3.1", @@ -10620,6 +10603,35 @@ ] } }, + "xai.grok-4.3": { + "id": "xai.grok-4.3", + "name": "Grok 4.3", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.25, + "output": 2.5, + "cacheRead": 0.2, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 131072, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "zai.glm-4.7": { "id": "zai.glm-4.7", "name": "GLM-4.7", @@ -12287,6 +12299,26 @@ } }, "cerebras": { + "gemma-4-31b": { + "id": "gemma-4-31b", + "name": "gemma-4-31b", + "api": "openai-completions", + "provider": "cerebras", + "baseUrl": "https://api.cerebras.ai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 8192 + }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "GPT OSS 120B", @@ -13672,7 +13704,7 @@ "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", - "reasoning": true, + "reasoning": false, "input": [ "text" ], @@ -13683,17 +13715,7 @@ "cacheWrite": 0 }, "contextWindow": 128000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } + "maxTokens": 128000 }, "meta-llama/Llama-3.3-70B-Instruct": { "id": "meta-llama/Llama-3.3-70B-Instruct", @@ -13701,7 +13723,7 @@ "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", - "reasoning": true, + "reasoning": false, "input": [ "text" ], @@ -13712,17 +13734,7 @@ "cacheWrite": 0 }, "contextWindow": 128000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } + "maxTokens": 128000 }, "meta-llama/Llama-4-Scout-17B-16E-Instruct": { "id": "meta-llama/Llama-4-Scout-17B-16E-Instruct", @@ -13730,7 +13742,7 @@ "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", - "reasoning": true, + "reasoning": false, "input": [ "text", "image" @@ -13742,17 +13754,7 @@ "cacheWrite": 0 }, "contextWindow": 64000, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } + "maxTokens": 64000 }, "microsoft/Phi-4-mini-instruct": { "id": "microsoft/Phi-4-mini-instruct", @@ -13760,7 +13762,7 @@ "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", - "reasoning": true, + "reasoning": false, "input": [ "text" ], @@ -13771,17 +13773,7 @@ "cacheWrite": 0 }, "contextWindow": 128000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } + "maxTokens": 128000 }, "MiniMaxAI/MiniMax-M2.5": { "id": "MiniMaxAI/MiniMax-M2.5", @@ -13789,7 +13781,7 @@ "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -13800,7 +13792,16 @@ "cacheWrite": 0 }, "contextWindow": 196608, - "maxTokens": 196608 + "maxTokens": 196608, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } }, "moonshotai/Kimi-K2.5": { "id": "moonshotai/Kimi-K2.5", @@ -13898,7 +13899,7 @@ "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -13909,7 +13910,17 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144 + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": { "id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B", @@ -15449,46 +15460,6 @@ "trustExplicitThinkingOnly": true } }, - "claude-opus-4-6-fast": { - "id": "claude-opus-4-6-fast", - "name": "Claude Opus 4.6 Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000, - "thinking": { - "mode": "budget", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ], - "effortRouting": { - "off": "claude-opus-4-6-fast", - "minimal": "claude-opus-4-6-thinking-fast", - "low": "claude-opus-4-6-thinking-fast", - "medium": "claude-opus-4-6-thinking-fast", - "high": "claude-opus-4-6-thinking-fast" - } - }, - "compat": { - "trustExplicitThinkingOnly": true - } - }, "claude-opus-4-7": { "id": "claude-opus-4-7", "name": "Claude Opus 4.7", @@ -21734,7 +21705,7 @@ "cost": { "input": 0.075, "output": 0.3, - "cacheRead": 0.037, + "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 131072, @@ -22420,13 +22391,13 @@ "text" ], "cost": { - "input": 0, - "output": 0, + "input": 0.25, + "output": 0.69, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 65536, + "maxTokens": 32768, "thinking": { "mode": "effort", "efforts": [ @@ -25048,7 +25019,7 @@ "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": true, + "reasoning": false, "input": [ "text" ], @@ -25059,24 +25030,7 @@ "cacheWrite": 0 }, "contextWindow": 163840, - "maxTokens": 65536, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high", - "xhigh": "max" - } - } + "maxTokens": 65536 }, "deepseek/deepseek-chat-v3.1": { "id": "deepseek/deepseek-chat-v3.1", @@ -30816,6 +30770,7 @@ "supportsStrictMode": false, "toolStrictMode": "mixed", "stripDeepseekSpecialTokens": false, + "streamMarkupHealingPattern": "thinking", "reasoningDeltasMayBeCumulative": false, "emptyLengthFinishIsContextError": false, "usesOpenAIToolCallIdLimit": false, @@ -50500,7 +50455,7 @@ "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -50511,7 +50466,20 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 131072 + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "xhigh": "max" + } + } }, "TEE/gpt-oss-120b": { "id": "TEE/gpt-oss-120b", @@ -54444,6 +54412,36 @@ "requiresEffort": true } }, + "minimaxai/minimax-m3": { + "id": "minimaxai/minimax-m3", + "name": "MiniMax-M3", + "api": "openai-completions", + "provider": "nvidia", + "baseUrl": "https://integrate.api.nvidia.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "mistralai/codestral-22b-instruct-v0.1": { "id": "mistralai/codestral-22b-instruct-v0.1", "name": "Codestral 22b Instruct V0.1", @@ -62195,9 +62193,9 @@ "image" ], "cost": { - "input": 0.66, - "output": 3.41, - "cacheRead": 0.144, + "input": 0.55, + "output": 3.1999999999999997, + "cacheRead": 0.11, "cacheWrite": 0 }, "contextWindow": 262144, @@ -63522,7 +63520,7 @@ "api": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", "provider": "openrouter", - "reasoning": true, + "reasoning": false, "input": [ "text" ], @@ -63533,22 +63531,7 @@ "cacheWrite": 0 }, "contextWindow": 163840, - "maxTokens": 16384, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ], - "effortMap": { - "minimal": "high", - "low": "high", - "medium": "high", - "high": "high" - } - } + "maxTokens": 16384 }, "deepseek/deepseek-chat-v3.1": { "id": "deepseek/deepseek-chat-v3.1", @@ -65751,7 +65734,7 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 32768 + "maxTokens": 100352 }, "moonshotai/kimi-k2-0905": { "id": "moonshotai/kimi-k2-0905", @@ -65770,7 +65753,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144 + "maxTokens": 100352 }, "moonshotai/kimi-k2-0905:exacto": { "id": "moonshotai/kimi-k2-0905:exacto", @@ -65861,9 +65844,9 @@ "image" ], "cost": { - "input": 0.66, - "output": 3.41, - "cacheRead": 0.144, + "input": 0.55, + "output": 3.1999999999999997, + "cacheRead": 0.11, "cacheWrite": 0 }, "contextWindow": 262144, @@ -69347,8 +69330,8 @@ "image" ], "cost": { - "input": 0.28850000000000003, - "output": 2.65, + "input": 0.28500000000000003, + "output": 2.4, "cacheRead": 0.15, "cacheWrite": 0 }, @@ -69758,13 +69741,13 @@ "text" ], "cost": { - "input": 0.09, + "input": 0.09999999999999999, "output": 0.3, "cacheRead": 0.02, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 16384, + "maxTokens": 65536, "thinking": { "mode": "effort", "efforts": [ @@ -70846,8 +70829,8 @@ "text" ], "cost": { - "input": 0.98, - "output": 3.08, + "input": 0.975, + "output": 4.300000000000001, "cacheRead": 0.182, "cacheWrite": 0 }, @@ -70874,7 +70857,7 @@ "text" ], "cost": { - "input": 0.95, + "input": 0.94, "output": 3, "cacheRead": 0.18, "cacheWrite": 0 @@ -71154,6 +71137,25 @@ ] } }, + "hf:moonshotai/Kimi-K2.7-Code": { + "id": "hf:moonshotai/Kimi-K2.7-Code", + "name": "moonshotai/Kimi-K2.7-Code", + "api": "openai-completions", + "provider": "synthetic", + "baseUrl": "https://api.synthetic.new/openai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 8192 + }, "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4": { "id": "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4", "name": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4", @@ -71196,7 +71198,7 @@ "cost": { "input": 0.1, "output": 0.1, - "cacheRead": 0, + "cacheRead": 0.1, "cacheWrite": 0 }, "contextWindow": 131072, @@ -71210,9 +71212,9 @@ ] } }, - "hf:Qwen/Qwen3.5-397B-A17B": { - "id": "hf:Qwen/Qwen3.5-397B-A17B", - "name": "Qwen/Qwen3.5-397B-A17B", + "hf:Qwen/Qwen3.6-27B": { + "id": "hf:Qwen/Qwen3.6-27B", + "name": "Qwen/Qwen3.6-27B", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71222,9 +71224,9 @@ "image" ], "cost": { - "input": 0.6, - "output": 3, - "cacheRead": 0.6, + "input": 0.45, + "output": 3.6, + "cacheRead": 0.45, "cacheWrite": 0 }, "contextWindow": 262144, @@ -71239,54 +71241,6 @@ ] } }, - "hf:Qwen/Qwen3.6-27B": { - "id": "hf:Qwen/Qwen3.6-27B", - "name": "Qwen/Qwen3.6-27B", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 8192 - }, - "hf:zai-org/GLM-4.7": { - "id": "hf:zai-org/GLM-4.7", - "name": "zai-org/GLM-4.7", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0.55, - "output": 2.19, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 202752, - "maxTokens": 64000, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, "hf:zai-org/GLM-4.7-Flash": { "id": "hf:zai-org/GLM-4.7-Flash", "name": "zai-org/GLM-4.7-Flash", @@ -71298,9 +71252,9 @@ "text" ], "cost": { - "input": 0.06, - "output": 0.4, - "cacheRead": 0.06, + "input": 0.1, + "output": 0.5, + "cacheRead": 0.1, "cacheWrite": 0 }, "contextWindow": 196608, @@ -71351,18 +71305,28 @@ "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.4, + "output": 4.4, + "cacheRead": 1.4, "cacheWrite": 0 }, "contextWindow": 524288, - "maxTokens": 8192 + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "syn:large:text": { "id": "syn:large:text", @@ -71682,7 +71646,7 @@ "api": "openai-completions", "provider": "together", "baseUrl": "https://api.together.xyz/v1", - "reasoning": true, + "reasoning": false, "input": [ "text", "image" @@ -71694,17 +71658,7 @@ "cacheWrite": 0.18 }, "contextWindow": 10000000, - "maxTokens": 32768, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } + "maxTokens": 32768 }, "MiniMaxAI/MiniMax-M2.5": { "id": "MiniMaxAI/MiniMax-M2.5", @@ -72381,6 +72335,38 @@ "escapeBuiltinToolNames": true } }, + "umans-glm-5.2-nvfp4": { + "id": "umans-glm-5.2-nvfp4", + "name": "Umans GLM 5.2 NVFP4 (experimental, short test from Jun 29)", + "api": "anthropic-messages", + "provider": "umans", + "baseUrl": "https://api.code.umans.ai", + "reasoning": true, + "thinking": { + "mode": "anthropic-budget-effort", + "efforts": [ + "high", + "xhigh" + ], + "effortMap": { + "xhigh": "max" + } + }, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 405504, + "maxTokens": 131071, + "compat": { + "escapeBuiltinToolNames": true + } + }, "umans-kimi-k2.7": { "id": "umans-kimi-k2.7", "name": "Umans Kimi K2.7 Code", @@ -75747,6 +75733,7 @@ "supportsForcedToolChoice": true, "supportsSamplingParams": true, "requiresToolResultId": false, + "requiresThinkingEnabled": false, "replayUnsignedThinking": false, "escapeBuiltinToolNames": false } @@ -80633,7 +80620,7 @@ }, "zai/glm-4.5": { "id": "zai/glm-4.5", - "name": "GLM-4.5", + "name": "GLM 4.5", "api": "anthropic-messages", "baseUrl": "https://ai-gateway.vercel.sh", "provider": "vercel-ai-gateway", @@ -82174,12 +82161,12 @@ "contextWindow": 256000, "maxTokens": 256000, "compat": { - "includeEncryptedReasoning": false, - "filterReasoningHistory": true, - "omitReasoningEffort": true, "reasoningEffortMap": { "minimal": "low" }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "omitReasoningEffort": true, "supportsReasoningEffort": false } }, @@ -82223,9 +82210,9 @@ "text" ], "cost": { - "input": 0.1, - "output": 0.3, - "cacheRead": 0.01, + "input": 0.14, + "output": 0.28, + "cacheRead": 0.0028, "cacheWrite": 0 }, "contextWindow": 262144, @@ -82259,9 +82246,9 @@ "image" ], "cost": { - "input": 0.4, - "output": 2, - "cacheRead": 0.08, + "input": 0.14, + "output": 0.28, + "cacheRead": 0.0028, "cacheWrite": 0 }, "contextWindow": 262144, @@ -82294,9 +82281,9 @@ "text" ], "cost": { - "input": 1, - "output": 3, - "cacheRead": 0.2, + "input": 0.435, + "output": 0.87, + "cacheRead": 0.0036, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -82330,9 +82317,9 @@ "image" ], "cost": { - "input": 0.4, - "output": 2, - "cacheRead": 0.08, + "input": 0.14, + "output": 0.28, + "cacheRead": 0.0028, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -82365,9 +82352,9 @@ "text" ], "cost": { - "input": 1, - "output": 3, - "cacheRead": 0.2, + "input": 0.435, + "output": 0.87, + "cacheRead": 0.0036, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -85046,6 +85033,35 @@ "contextWindow": 256000, "maxTokens": 80000 }, + "meituan/longcat-2.0": { + "id": "meituan/longcat-2.0", + "name": "LongCat-2.0", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 1.18, + "cacheRead": 0.006, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": null, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "meta/llama-3.3-70b-instruct": { "id": "meta/llama-3.3-70b-instruct", "name": "Llama 3.3 70b Instruct", diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 12776ea48..53aaed7c2 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -803,6 +803,18 @@ export function groqModelManagerOptions(config?: GroqModelManagerConfig): ModelM // 3. Cerebras // --------------------------------------------------------------------------- +const CEREBRAS_IMAGE_INPUT_MODEL_IDS = new Set(["gemma-4-31b"]); + +function applyCerebrasDiscoveryOverrides(model: ModelSpec<"openai-completions">): ModelSpec<"openai-completions"> { + if (!CEREBRAS_IMAGE_INPUT_MODEL_IDS.has(model.id)) { + return model; + } + return { + ...model, + input: ["text", "image"], + }; +} + export interface CerebrasModelManagerConfig { apiKey?: string; baseUrl?: string; @@ -812,7 +824,27 @@ export interface CerebrasModelManagerConfig { export function cerebrasModelManagerOptions( config?: CerebrasModelManagerConfig, ): ModelManagerOptions<"openai-completions"> { - return createSimpleOpenAICompletionsOptions("cerebras", "https://api.cerebras.ai/v1", config); + const apiKey = config?.apiKey; + const baseUrl = config?.baseUrl ?? "https://api.cerebras.ai/v1"; + const references = createBundledReferenceMap<"openai-completions">("cerebras"); + return { + providerId: "cerebras", + ...(apiKey && { + fetchDynamicModels: () => + fetchOpenAICompatibleModels({ + api: "openai-completions", + provider: "cerebras", + baseUrl, + apiKey, + mapModel: (entry, defaults) => { + const reference = references.get(defaults.id); + const model = mapWithBundledReference(entry, defaults, reference); + return applyCerebrasDiscoveryOverrides(model); + }, + fetch: config?.fetch, + }), + }), + }; } // --------------------------------------------------------------------------- @@ -1294,7 +1326,7 @@ export function clampFireworksKimiMaxTokens(modelId: string, candidate: number | export const KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS = 32_768; export function isKimiK27CodeModelId(modelId: string): boolean { - return /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code$/i.test(modelId); + return /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code(?:[-._]?highspeed)?$/i.test(modelId); } export function clampKimiK27CodeMaxTokens(modelId: string, candidate: number): number; diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts index 29c0a9571..c41d76d3c 100644 --- a/packages/catalog/src/types.ts +++ b/packages/catalog/src/types.ts @@ -420,6 +420,11 @@ export interface AnthropicCompat { * Default: auto-detected from provider/baseUrl and `model.reasoning`. */ replayUnsignedThinking?: boolean; + /** + * Whether the endpoint requires `thinking.type: "enabled"` whenever the + * model reasons. Use for models that reject omitted or disabled thinking. + */ + requiresThinkingEnabled?: boolean; /** * Prefix Anthropic built-in tool names (`web_search`, `code_execution`, ...) * when they are ordinary client tools. Some Anthropic-compatible gateways diff --git a/packages/catalog/test/cerebras-provider.test.ts b/packages/catalog/test/cerebras-provider.test.ts new file mode 100644 index 000000000..7a829d1bc --- /dev/null +++ b/packages/catalog/test/cerebras-provider.test.ts @@ -0,0 +1,41 @@ +import { describe, expect, test } from "bun:test"; +import { cerebrasModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; +import type { FetchImpl } from "@oh-my-pi/pi-catalog/types"; + +describe("Cerebras provider discovery", () => { + test("discovers gemma-4-31b as image-capable", async () => { + const calls: Array<{ url: string; authorization: string | null }> = []; + const fetchMock: FetchImpl = async (input: string | URL | Request, init?: RequestInit) => { + const headers = new Headers(init?.headers); + calls.push({ + url: String(input), + authorization: headers.get("authorization"), + }); + return new Response( + JSON.stringify({ + data: [ + { id: "gemma-4-31b", object: "model" }, + { id: "llama3.1-8b", object: "model" }, + ], + }), + { status: 200, headers: { "content-type": "application/json" } }, + ); + }; + + const options = cerebrasModelManagerOptions({ apiKey: "cerebras-test-key", fetch: fetchMock }); + const models = await options.fetchDynamicModels?.(); + + expect(calls).toEqual([ + { + url: "https://api.cerebras.ai/v1/models", + authorization: "Bearer cerebras-test-key", + }, + ]); + expect(models?.find(model => model.id === "gemma-4-31b")).toMatchObject({ + provider: "cerebras", + api: "openai-completions", + input: ["text", "image"], + }); + expect(models?.find(model => model.id === "llama3.1-8b")?.input).toEqual(["text"]); + }); +}); diff --git a/packages/catalog/test/identity-family.test.ts b/packages/catalog/test/identity-family.test.ts index b6b204b41..7d8fd4095 100644 --- a/packages/catalog/test/identity-family.test.ts +++ b/packages/catalog/test/identity-family.test.ts @@ -1,5 +1,6 @@ import { describe, expect, test } from "bun:test"; import { + hasOpus47ApiRestrictions, isClaudeModelId, isGlmVisionModelId, isGrokReasoningEffortCapable, @@ -11,6 +12,7 @@ import { isReasoningGlmModelId, modelFamilyToken, supportsAdaptiveThinkingDisplay, + supportsMidConversationSystemMessages, } from "@oh-my-pi/pi-catalog/identity"; describe("isKimiModelId", () => { @@ -44,10 +46,12 @@ describe("isClaudeModelId", () => { }); describe("supportsAdaptiveThinkingDisplay", () => { - test("allows Claude Fable 5 and Opus 4.7 or newer only", () => { + test("allows Claude Fable 5, Opus 4.7 or newer, and Sonnet 5 or newer only", () => { expect(supportsAdaptiveThinkingDisplay("claude-fable-5")).toBe(true); expect(supportsAdaptiveThinkingDisplay("claude-opus-4-7")).toBe(true); expect(supportsAdaptiveThinkingDisplay("claude-opus-5-0")).toBe(true); + expect(supportsAdaptiveThinkingDisplay("claude-sonnet-5")).toBe(true); + expect(supportsAdaptiveThinkingDisplay("us.anthropic.claude-sonnet-5")).toBe(true); // Dotted and dashed version separators are equivalent. expect(supportsAdaptiveThinkingDisplay("claude-opus-4.7")).toBe(true); expect(supportsAdaptiveThinkingDisplay("anthropic/claude-opus-4.8")).toBe(true); @@ -58,6 +62,30 @@ describe("supportsAdaptiveThinkingDisplay", () => { }); }); +describe("hasOpus47ApiRestrictions", () => { + test("allows Claude Fable 5, Opus 4.7 or newer, and Sonnet 5 or newer only", () => { + expect(hasOpus47ApiRestrictions("claude-fable-5")).toBe(true); + expect(hasOpus47ApiRestrictions("claude-opus-4-7")).toBe(true); + expect(hasOpus47ApiRestrictions("claude-opus-4.8")).toBe(true); + expect(hasOpus47ApiRestrictions("claude-sonnet-5")).toBe(true); + expect(hasOpus47ApiRestrictions("us.anthropic.claude-sonnet-5")).toBe(true); + expect(hasOpus47ApiRestrictions("claude-opus-4-6")).toBe(false); + expect(hasOpus47ApiRestrictions("claude-sonnet-4-6")).toBe(false); + expect(hasOpus47ApiRestrictions("claude-sonnet-4-5")).toBe(false); + }); +}); + +describe("supportsMidConversationSystemMessages", () => { + test("allows Claude Fable 5, Opus 4.8 or newer, and Sonnet 5 or newer only", () => { + expect(supportsMidConversationSystemMessages("claude-fable-5")).toBe(true); + expect(supportsMidConversationSystemMessages("claude-opus-4-8")).toBe(true); + expect(supportsMidConversationSystemMessages("claude-sonnet-5")).toBe(true); + expect(supportsMidConversationSystemMessages("us.anthropic.claude-sonnet-5")).toBe(true); + expect(supportsMidConversationSystemMessages("claude-opus-4-7")).toBe(false); + expect(supportsMidConversationSystemMessages("claude-sonnet-4-6")).toBe(false); + }); +}); + describe("isMinimaxM2FamilyModelId", () => { test("matches every M2-generation id shape served by aggregator/native hosts", () => { // Fireworks/OpenCode/openrouter direct ids and `-highspeed`/`-lightning` variants. diff --git a/packages/catalog/test/model-thinking.test.ts b/packages/catalog/test/model-thinking.test.ts index ee20b0369..6d6019fe1 100644 --- a/packages/catalog/test/model-thinking.test.ts +++ b/packages/catalog/test/model-thinking.test.ts @@ -367,6 +367,12 @@ describe("model thinking derivation", () => { provider: "amazon-bedrock", }); const sonnet46 = createModel({ id: "claude-sonnet-4.6", api: "anthropic-messages", provider: "anthropic" }); + const sonnet5 = createModel({ id: "claude-sonnet-5", api: "anthropic-messages", provider: "anthropic" }); + const sonnet5Bedrock = createModel({ + id: "global.anthropic.claude-sonnet-5", + api: "bedrock-converse-stream", + provider: "amazon-bedrock", + }); const mythos = createModel({ id: "claude-mythos-5", api: "anthropic-messages", provider: "anthropic" }); const mythosBedrock = createModel({ id: "global.anthropic.claude-mythos-5", @@ -399,6 +405,8 @@ describe("model thinking derivation", () => { expect(sonnet45Bedrock.thinking?.mode).toBe("budget"); expect(opus46.thinking?.mode).toBe("anthropic-adaptive"); expect(sonnet46.thinking?.mode).toBe("anthropic-adaptive"); + expect(sonnet5.thinking?.mode).toBe("anthropic-adaptive"); + expect(sonnet5Bedrock.thinking?.mode).toBe("anthropic-adaptive"); expect(mythosBedrock.thinking?.mode).toBe("anthropic-adaptive"); expect(minimaxM2.thinking).toEqual({ mode: "anthropic-adaptive", @@ -437,13 +445,16 @@ describe("model thinking derivation", () => { expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.XHigh)).toBe("max"); expect(mapEffortToAnthropicAdaptiveEffort(mythos, Effort.High)).toBe("xhigh"); expect(mapEffortToAnthropicAdaptiveEffort(mythosBedrock, Effort.XHigh)).toBe("max"); + expect(mapEffortToAnthropicAdaptiveEffort(sonnet5, Effort.High)).toBe("xhigh"); + expect(mapEffortToAnthropicAdaptiveEffort(sonnet5Bedrock, Effort.XHigh)).toBe("max"); // Bedrock Converse keeps the four-tier legacy mapping; xhigh aliases to "max". expect(opus47Bedrock.thinking?.effortMap).toEqual({ minimal: "low", xhigh: "max" }); expect(mapEffortToAnthropicAdaptiveEffort(opus47Bedrock, Effort.High)).toBe("high"); + expect(mapEffortToAnthropicAdaptiveEffort(sonnet5Bedrock, Effort.High)).toBe("high"); expect(() => mapEffortToAnthropicAdaptiveEffort(sonnet46, Effort.XHigh)).toThrow(/not supported/); }); - it("bakes adaptive display support for Opus 4.7+ and Fable/Mythos 5", () => { + it("bakes adaptive display support for Opus 4.7+, Sonnet 5+, and Fable/Mythos 5", () => { const opus46 = createModel({ id: "claude-opus-4.6", api: "anthropic-messages", provider: "anthropic" }); const opus47 = createModel({ id: "claude-opus-4-7", api: "anthropic-messages", provider: "anthropic" }); // Dotted and dashed version forms are equivalent; bare dated ids stay Opus 4.0. @@ -459,6 +470,12 @@ describe("model thinking derivation", () => { api: "bedrock-converse-stream", provider: "amazon-bedrock", }); + const sonnet5 = createModel({ id: "claude-sonnet-5", api: "anthropic-messages", provider: "anthropic" }); + const sonnet5Bedrock = createModel({ + id: "global.anthropic.claude-sonnet-5", + api: "bedrock-converse-stream", + provider: "amazon-bedrock", + }); expect(opus46.thinking?.supportsDisplay).toBeUndefined(); expect(opus47.thinking?.supportsDisplay).toBe(true); @@ -466,6 +483,8 @@ describe("model thinking derivation", () => { expect(opus4Dated.thinking?.supportsDisplay).toBeUndefined(); expect(fable.thinking?.supportsDisplay).toBe(true); expect(fableBedrock.thinking?.supportsDisplay).toBe(true); + expect(sonnet5.thinking?.supportsDisplay).toBe(true); + expect(sonnet5Bedrock.thinking?.supportsDisplay).toBe(true); }); it("backfills wire facts onto explicit thinking, explicit values winning", () => { @@ -526,10 +545,12 @@ describe("model thinking derivation", () => { it("bakes sampling-param rejection into anthropic compat", () => { const sonnet45 = createModel({ id: "claude-sonnet-4-5", api: "anthropic-messages", provider: "anthropic" }); const opus47 = createModel({ id: "claude-opus-4.7", api: "anthropic-messages", provider: "anthropic" }); + const sonnet5 = createModel({ id: "claude-sonnet-5", api: "anthropic-messages", provider: "anthropic" }); const fable = createModel({ id: "claude-fable-5", api: "anthropic-messages", provider: "anthropic" }); expect(sonnet45.compat.supportsSamplingParams).toBe(true); expect(opus47.compat.supportsSamplingParams).toBe(false); + expect(sonnet5.compat.supportsSamplingParams).toBe(false); expect(fable.compat.supportsSamplingParams).toBe(false); }); @@ -685,11 +706,17 @@ describe("model thinking runtime helpers", () => { api: "openai-completions", provider: "openrouter", }); - + const sonnet5 = createModel({ + id: "anthropic/claude-sonnet-5", + api: "openai-completions", + provider: "openrouter", + }); expect(fable.thinking?.efforts.at(-1)).toBe(Effort.XHigh); expect(opus46.thinking?.efforts.at(-1)).toBe(Effort.XHigh); expect(sonnet46.thinking?.efforts.at(-1)).toBe(Effort.High); + expect(sonnet5.thinking?.efforts.at(-1)).toBe(Effort.XHigh); expect(requireSupportedEffort(fable, Effort.XHigh)).toBe(Effort.XHigh); + expect(requireSupportedEffort(sonnet5, Effort.XHigh)).toBe(Effort.XHigh); }); it("enables xhigh for openai-responses and openai-codex-responses APIs", () => { diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index af340e97e..cbe0c7938 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -18,13 +18,81 @@ - Fixed a multi-character `mode: "replace"` regex remainder (the bytes of a match outside a preserved `#…#` placeholder) drifting across an obfuscator restart, which invalidated provider prompt-cache prefixes even with a stable key. The remainder was redacted to a content-derived `ZZ`+hash marker that was only recognized as already-redacted within the generating session (via an in-memory set), so a fresh obfuscator reprocessing persisted text re-redacted it to a different value (`ZZPL#…#` → `ZZ7f#…#`). The remainder marker now derives from a keyed run of the per-install key and the remainder length, so any instance sharing the key reproduces it byte-identically (idempotent across restart) while staying unpredictable enough that raw sentinel-shaped bytes (`ZZZZ`) still differ from it and are redacted rather than passed through ([#2465](https://github.com/can1357/oh-my-pi/issues/2465)). - Fixed a secret regex match that starts in outside text and ends inside a previously generated `#…#` placeholder's expanded value leaving an independently-matching outside prefix provider-visible. Resuming the scan past the cut placeholder skipped the whole straddling span, so a pattern like `[A-Z0-9]{8,12}` greedily spanning `SECRETUV` into an `ABCDEFGH` placeholder returned `SECRETUV#…#` even though `SECRETUV` satisfies the regex on its own. The cut handling now re-runs the regex bounded to just before the placeholder (full left context kept, so lookbehind still evaluates) and redacts the standalone prefix match — to its own reversible placeholder in obfuscate mode, or a one-way redaction in replace mode — while the cut secret stays as its existing placeholder. The replace-mode redaction's fixed point is verified against the placeholder-expanded view re-obfuscation actually scans, so it does not drift when the adjacent placeholder expands ([#2465](https://github.com/can1357/oh-my-pi/issues/2465)). +## [16.2.9] - 2026-06-30 + +### Breaking Changes + +- Renamed the built-in quick_task subagent to sonic; update any task spawns or configurations referencing quick_task by name. + +### Added + +- Added the llama3.2:3b local model option for memory and auto-thinking tasks, utilizing a quantized ONNX model. +- Added a built-in Tester subagent designed to write high-signal tests for behavior, invariants, and edge cases while avoiding redundant or low-value tests. +- Added a Speech-to-Text submit trigger setting to auto-submit dictation on release, on complete sentences, or via a spoken submit command. +- Added a loop-guard mechanism that detects thinking/response loops and injects a system notice during auto-retries to guide the model to break the pattern and take a concrete next step. + ### Changed -- Enabled contextual snapcompact shape resolution based on rendered text content +- Renamed the Speech-to-Text (STT) setting label from "TTS Submit Trigger" to "Speech-to-Text Submit Trigger". ### Fixed -- Fixed snapcompact preflight to use the same font-aware renderability probe as compaction, including prior preserved archive text, so CJK history remains renderable through per-glyph Silver fallback across repeated compactions. +- Fixed an issue where mid-run compaction was incorrectly skipped when a persisted assistant display variant shared a persistence key but differed in content from the live message. +- Fixed duplicate placeholder cards being created when streamed tool blocks started with an empty ID. +- Fixed omp debug --profile failing on Bun by treating the optional --allow-natives-syntax flag as best-effort when v8.setFlagsFromString is unavailable. +- Fixed release binaries missing the compiled tiny-model Transformers.js version pin, preventing runtime resolution issues on certain platforms like Homebrew Darwin arm64. +- Fixed MCP OAuth flows silently falling back to a random redirect port when the preferred port (default 3000) was busy, which caused authentication failures with strict providers. The flow now fails fast with a configuration error when a static client ID is pinned, while dynamic registration flows continue to use fallback ports. +- Fixed /mcp reauth and /mcp add commands ignoring the Escape key during OAuth authentication, allowing users to cancel the flow immediately instead of waiting for the timeout. + +### Removed + +- Removed the built-in oracle subagent. + +## [16.2.8] - 2026-06-30 + +### Added + +- Added built-in Go coding rules including `go-add-cleanup`, `go-bench-loop`, `go-exp-promoted`, `go-ioutil`, `go-join-hostport`, `go-new-expr`, `go-rand-v2`, and `go-range-int` + +### Changed + +- Relaxed strict bash tool constraints regarding the use of search, grep, ls, and find commands + +### Fixed + +- Fixed auto-compaction dead-ends by automatically triggering a shake rescue to elide oversized tails +- Improved compaction warning message to suggest running `/shake images` for irreducible image tails +- Fixed `grep`/`search` direct execution to accept JSON-array string `paths` for string-or-array inputs. ([#3873](https://github.com/can1357/oh-my-pi/issues/3873)) +- Fixed auto-compaction dead-ending with "Compaction freed too little context to make progress" when a single recent turn (large tool output, heavy fenced/XML block) is itself bigger than the recovery band — `findCutPoint` can't cut inside one message, so the summarizer had no lever left. The guard now runs an artifact-backed `shake` elide pass over the oversized tail and re-tests headroom before pausing, and the remaining warning points at `/shake images` for image-only tails it can't elide. ([#3786](https://github.com/can1357/oh-my-pi/issues/3786)) +- Fixed reviewer/`task` subagents whose incremental `yield` (`type: ["overall_correctness"]`, `type: ["findings"]`, …) carried a value that mismatched the matching property's sub-schema being silently accepted and then post-mortem rejected with `schema_violation` — opaquely swapping the agent's accepted output for an error blob. The yield tool now validates each incremental section's `data` against its top-level property's sub-schema (items schema for array-typed labels) and surfaces the same retry feedback as terminal yields, so models like `deepseek-v4-pro` that emit `"Correct"`/`"correct."`/`"approved"` for an enum field get up to three corrective retries; the existing `MAX_SCHEMA_RETRIES` override then accepts the value with `SUBAGENT_WARNING_SCHEMA_OVERRIDDEN` instead of losing the entire result. Unknown labels stay unconstrained ([#3870](https://github.com/can1357/oh-my-pi/issues/3870)). +- Fixed streaming tool-call previews (notably `write`) showing an empty body for the entire streaming phase by surfacing the partial JSON already in hand on the first reveal, then pacing only subsequent growth ([#3881](https://github.com/can1357/oh-my-pi/issues/3881)). +- Fixed hashline edit mode preserving UTF-8 BOM bytes on edited files. ([#3867](https://github.com/can1357/oh-my-pi/issues/3867)) +- Fixed slow local LLM streams by forwarding persisted stream timeout settings (`providers.streamFirstEventTimeoutSeconds`, `providers.streamIdleTimeoutSeconds`) into model requests, so users can widen or disable watchdogs without environment variables. ([#3878](https://github.com/can1357/oh-my-pi/issues/3878)) +- Fixed the bash interceptor blocking `echo` / `printf` redirects to `/dev/null`, `/dev/tty`, `/dev/stdout`, and `/dev/stderr` device sinks while still directing real file writes to the write tool. ([#3763](https://github.com/can1357/oh-my-pi/issues/3763)) +- Fixed long snapcompact sessions re-sending multi-megabyte standing image archives on every provider request by enforcing a per-request frame byte budget, letting auto-compaction fall back to context-full summaries when snapcompact output is too large, and omitting legacy over-budget frames from rebuilt LLM contexts. ([#3792](https://github.com/can1357/oh-my-pi/issues/3792)) + +## [16.2.7] - 2026-06-30 + +### Breaking Changes + +- Replaced the global `serviceTier` and `fastModeScope` settings with granular, per-family settings (`tier.openai`, `tier.anthropic`, and `tier.google`) to control service tiers, subagents, advisors, and `/fast` mode targets. + +### Changed + +- Improved binary file detection and terminal handling to prevent corruption from non-UTF-8 content, and updated file summaries to explicitly note skipped binary files. +- Enhanced context compaction (snapcompact) to resolve shapes contextually based on rendered text content. + +### Fixed + +- Improved reliability of DuckDuckGo web searches by updating browser request headers and parameters +- Fixed an issue where CJK (Chinese, Japanese, Korean) history could become unrenderable during repeated context compactions. +- Fixed a memory exhaustion bug in the TUI when using `/resume` on large previous sessions. +- Fixed an issue where the `irc` inbox missed messages that arrived while the recipient agent was already running. +- Fixed a startup hang caused by system-prompt GPU detection blocking and repeatedly running failed probes. +- Improved error reporting for `omp tiny-models download` by displaying the actual worker-side download error. +- Resolved status inconsistencies between `/extensions`, `/mcp list`, and the dashboard, ensuring MCP server states, allowlists/denylists, and configuration files (like `mcp.json`) stay fully synchronized. +- Improved branch-mode task merges to preserve the agent's original commit history (messages and authors) and fixed a bug where merges were rejected due to unrelated dirty changes in the parent checkout. +- Fixed an issue where the `Working...` loader spinner would prematurely disappear or fail to re-arm after a subagent (`task`) tool completed or during transient overlays (such as auto-compaction or auto-retry). ## [16.2.6] - 2026-06-29 diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index af8c5dd51..7fbb49f8d 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-coding-agent", - "version": "16.2.6", + "version": "16.2.9", "description": "Coding agent CLI with read, bash, edit, write tools and session management", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/coding-agent/src/cli/bench-cli.ts b/packages/coding-agent/src/cli/bench-cli.ts index 52ed29b28..9e8a0a537 100644 --- a/packages/coding-agent/src/cli/bench-cli.ts +++ b/packages/coding-agent/src/cli/bench-cli.ts @@ -10,9 +10,10 @@ import type { Model, ProviderSessionState, ServiceTier, + ServiceTierByFamily, SimpleStreamOptions, } from "@oh-my-pi/pi-ai"; -import { streamSimple } from "@oh-my-pi/pi-ai"; +import { resolveModelServiceTier, streamSimple } from "@oh-my-pi/pi-ai"; import { buildModelProviderPriorityRank, type CanonicalModelVariant } from "@oh-my-pi/pi-catalog/identity"; import { replaceTabs, truncateToWidth } from "@oh-my-pi/pi-tui"; import { formatDuration, getProjectDir } from "@oh-my-pi/pi-utils"; @@ -25,7 +26,7 @@ import { getModelMatchPreferences, resolveCliModel, } from "../config/model-resolver"; -import { resolveServiceTierSetting } from "../config/service-tier"; +import { buildServiceTierByFamily, serviceTierForAllFamilies, serviceTierSettingToTier } from "../config/service-tier"; import { Settings } from "../config/settings"; import benchPrompt from "../prompts/bench.md" with { type: "text" }; import { discoverAuthStorage, loadCliExtensionProviders } from "../sdk"; @@ -106,8 +107,8 @@ export interface BenchSummary { maxTokens: number; models: BenchModelReport[]; failures: number; - /** Requested service tier passed to every request; absent when none was requested. Scoped tiers (`openai-only`/`claude-only`) may be dropped per-provider downstream. */ - serviceTier?: ServiceTier; + /** Requested per-family service tiers, resolved per model before reaching the wire. */ + serviceTierByFamily?: ServiceTierByFamily; } type BenchStreamSimple = ( @@ -518,12 +519,18 @@ export async function runBenchCommand(command: BenchCommandArgs, deps: BenchDepe const runtime = await (deps.createRuntime ?? createDefaultRuntime)(); try { const targets = resolveBenchModels(command.models, runtime.modelRegistry, runtime.settings, writeStderr); - // Explicit `--service-tier` wins; otherwise fall back to the configured - // `serviceTier` setting (`none`/unset omits the wire field). Scope-aware - // gating to the model's provider happens downstream in the provider layer. - const serviceTierValue = command.flags.serviceTier ?? runtime.settings?.get("serviceTier"); - const serviceTier = serviceTierValue ? resolveServiceTierSetting(serviceTierValue, undefined) : undefined; - if (!json && serviceTier) writeStdout(`${chalk.dim(`service tier: ${serviceTier}`)}\n`); + // Explicit `--service-tier` (a single value broadcast across families) wins; + // otherwise fall back to the configured per-family `tier.*` settings. Each + // model resolves its own family's tier below before reaching the wire. + const flagTier = command.flags.serviceTier ? serviceTierSettingToTier(command.flags.serviceTier) : undefined; + const serviceTierByFamily = command.flags.serviceTier + ? serviceTierForAllFamilies(flagTier) + : buildServiceTierByFamily( + runtime.settings?.get("tier.openai") ?? "none", + runtime.settings?.get("tier.anthropic") ?? "none", + runtime.settings?.get("tier.google") ?? "none", + ); + if (!json && flagTier) writeStdout(`${chalk.dim(`service tier: ${flagTier}`)}\n`); const reports: BenchModelReport[] = []; for (const { selector, model, thinking } of targets) { if (!json) { @@ -564,7 +571,7 @@ export async function runBenchCommand(command: BenchCommandArgs, deps: BenchDepe maxTokens, reasoning: toReasoningEffort(thinking), disableReasoning: shouldDisableReasoning(thinking) ? true : undefined, - serviceTier, + serviceTier: resolveModelServiceTier(serviceTierByFamily, model), }, streamFn, now, @@ -606,7 +613,7 @@ export async function runBenchCommand(command: BenchCommandArgs, deps: BenchDepe reports.push(buildModelReport(selector, model, thinking, results)); } const failures = reports.reduce((sum, report) => sum + report.results.filter(result => !result.ok).length, 0); - const summary: BenchSummary = { runs, maxTokens, models: reports, failures, serviceTier }; + const summary: BenchSummary = { runs, maxTokens, models: reports, failures, serviceTierByFamily }; if (json) { writeStdout(`${JSON.stringify(summary, null, 2)}\n`); } else if (reports.length > 1 || runs > 1) { diff --git a/packages/coding-agent/src/cli/tiny-models-cli.ts b/packages/coding-agent/src/cli/tiny-models-cli.ts index 02a3be475..b2efc23b0 100644 --- a/packages/coding-agent/src/cli/tiny-models-cli.ts +++ b/packages/coding-agent/src/cli/tiny-models-cli.ts @@ -28,12 +28,21 @@ interface ProgressReporter { interface DownloadResult { model: TinyLocalModelKey; ok: boolean; + error?: string; } function writeLine(text = ""): void { process.stdout.write(`${text}\n`); } +function downloadErrorSummary(error: string | undefined): string | undefined { + return error + ?.split(/\r?\n/) + .map(line => line.trim()) + .find(line => line.length > 0) + ?.replace(/^Error:\s*/, ""); +} + export function resolveModels(model: string | undefined): TinyLocalModelKey[] { if (!model) return [DEFAULT_TINY_TITLE_LOCAL_MODEL_KEY]; // `all` is a prefetch convenience: skip models that fail before load (unsupported @@ -101,10 +110,15 @@ async function downloadOne(modelKey: TinyLocalModelKey, json: boolean | undefine const label = getTinyLocalModelSpec(modelKey)?.label ?? modelKey; if (!json && !process.stdout.isTTY) writeLine(`Downloading ${label} (${modelKey})...`); const progress = makeProgressReporter(modelKey, json); - const ok = await tinyTitleClient.downloadModel(modelKey, { onProgress: progress.onProgress }); - progress.finish(ok); - if (!json && !process.stdout.isTTY) writeLine(ok ? `Downloaded ${label}.` : `Failed to download ${label}.`); - return { model: modelKey, ok }; + const result = await tinyTitleClient.downloadModel(modelKey, { onProgress: progress.onProgress }); + progress.finish(result.ok); + const error = downloadErrorSummary(result.error); + if (!json && !process.stdout.isTTY) { + writeLine(result.ok ? `Downloaded ${label}.` : `Failed to download ${label}${error ? `: ${error}` : ""}.`); + } else if (!json && !result.ok && error) { + writeLine(`${label} failed: ${error}`); + } + return result.error ? { model: modelKey, ok: result.ok, error: result.error } : { model: modelKey, ok: result.ok }; } export async function runTinyModelsCommand(command: TinyModelsCommandArgs): Promise { diff --git a/packages/coding-agent/src/commands/bench.ts b/packages/coding-agent/src/commands/bench.ts index 20d0a3a04..606530377 100644 --- a/packages/coding-agent/src/commands/bench.ts +++ b/packages/coding-agent/src/commands/bench.ts @@ -1,6 +1,6 @@ import { Args, Command, Flags } from "@oh-my-pi/pi-utils/cli"; import { runBenchCommand } from "../cli/bench-cli"; -import { SERVICE_TIER_SETTING_VALUES } from "../config/service-tier"; +import { SERVICE_TIER_OPENAI_VALUES } from "../config/service-tier"; export default class Bench extends Command { static description = @@ -19,8 +19,8 @@ export default class Bench extends Command { "max-tokens": Flags.integer({ description: "Max output tokens per request", default: 512 }), prompt: Flags.string({ description: "Custom prompt text (default: bundled bench prompt)" }), "service-tier": Flags.string({ - description: "Service tier hint (default: configured `serviceTier` setting; `none` omits it)", - options: SERVICE_TIER_SETTING_VALUES, + description: "Service tier applied per model family (default: configured `tier.*` settings; `none` omits it)", + options: SERVICE_TIER_OPENAI_VALUES, }), json: Flags.boolean({ description: "Output JSON" }), par: Flags.integer({ description: "Execute runs with N parallel queries/requests", default: 4 }), diff --git a/packages/coding-agent/src/commit/agentic/agent.ts b/packages/coding-agent/src/commit/agentic/agent.ts index 98b4c8867..68b529e97 100644 --- a/packages/coding-agent/src/commit/agentic/agent.ts +++ b/packages/coding-agent/src/commit/agentic/agent.ts @@ -42,7 +42,7 @@ export async function runCommitAgentSession(input: CommitAgentInput): Promise