Merge remote-tracking branch 'upstream/main' into feat/secret-friendly-names

This commit is contained in:
Mathews-Tom
2026-07-01 00:34:03 +05:30
228 changed files with 9670 additions and 1571 deletions
Generated
+8 -8
View File
@@ -2064,9 +2064,9 @@ checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
[[package]]
name = "jiff"
version = "0.2.29"
version = "0.2.31"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "34f877a98676d2fb664698d74cc6a51ce6c484ce8c770f05d0108ec9090aeb46"
checksum = "ccfe6121cbe750cf81efa362d85c0bde7ea298ec43092d3a193baca59cdbd634"
dependencies = [
"defmt",
"jiff-static",
@@ -2091,9 +2091,9 @@ dependencies = [
[[package]]
name = "jiff-static"
version = "0.2.29"
version = "0.2.31"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0666b5ab5ecaca213fc2a85b8c0083d9004e84ee2d5f9a7e0017aaf50986f25f"
checksum = "e165e897f662d428f3cd3828a919dbe067c2d42bb1031eede74ef9d27ecdedd2"
dependencies = [
"proc-macro2",
"quote",
@@ -2911,7 +2911,7 @@ dependencies = [
[[package]]
name = "pi-ast"
version = "16.2.6"
version = "16.2.9"
dependencies = [
"anyhow",
"ast-grep-core",
@@ -2981,7 +2981,7 @@ dependencies = [
[[package]]
name = "pi-iso"
version = "16.2.6"
version = "16.2.9"
dependencies = [
"async-trait",
"libc",
@@ -2993,7 +2993,7 @@ dependencies = [
[[package]]
name = "pi-natives"
version = "16.2.6"
version = "16.2.9"
dependencies = [
"anyhow",
"arboard",
@@ -3043,7 +3043,7 @@ dependencies = [
[[package]]
name = "pi-shell"
version = "16.2.6"
version = "16.2.9"
dependencies = [
"anyhow",
"brush-builtins",
+1 -1
View File
@@ -4,7 +4,7 @@ exclude = ["crates/vendor/brush-core", "crates/vendor/brush-builtins"]
resolver = "3"
[workspace.package]
version = "16.2.6"
version = "16.2.9"
edition = "2024"
license = "MIT"
authors = ["Can Boluk"]
+28 -28
View File
@@ -21,7 +21,7 @@
},
"packages/agent": {
"name": "@oh-my-pi/pi-agent-core",
"version": "16.2.6",
"version": "16.2.9",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
@@ -39,7 +39,7 @@
},
"packages/ai": {
"name": "@oh-my-pi/pi-ai",
"version": "16.2.6",
"version": "16.2.9",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
@@ -55,7 +55,7 @@
},
"packages/catalog": {
"name": "@oh-my-pi/pi-catalog",
"version": "16.2.6",
"version": "16.2.9",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
@@ -69,7 +69,7 @@
},
"packages/coding-agent": {
"name": "@oh-my-pi/pi-coding-agent",
"version": "16.2.6",
"version": "16.2.9",
"bin": {
"omp": "src/cli.ts",
},
@@ -137,7 +137,7 @@
},
"packages/hashline": {
"name": "@oh-my-pi/hashline",
"version": "16.2.6",
"version": "16.2.9",
"dependencies": {
"diff": "catalog:",
"lru-cache": "catalog:",
@@ -148,7 +148,7 @@
},
"packages/mnemopi": {
"name": "@oh-my-pi/pi-mnemopi",
"version": "16.2.6",
"version": "16.2.9",
"bin": {
"mnemopi": "src/cli.ts",
},
@@ -174,7 +174,7 @@
},
"packages/natives": {
"name": "@oh-my-pi/pi-natives",
"version": "16.2.6",
"version": "16.2.9",
"devDependencies": {
"@napi-rs/cli": "catalog:",
"@types/bun": "catalog:",
@@ -182,7 +182,7 @@
},
"packages/snapcompact": {
"name": "@oh-my-pi/snapcompact",
"version": "16.2.6",
"version": "16.2.9",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-natives": "catalog:",
@@ -195,7 +195,7 @@
},
"packages/stats": {
"name": "@oh-my-pi/omp-stats",
"version": "16.2.6",
"version": "16.2.9",
"bin": {
"omp-stats": "./src/index.ts",
},
@@ -221,7 +221,7 @@
},
"packages/swarm-extension": {
"name": "@oh-my-pi/swarm-extension",
"version": "16.2.6",
"version": "16.2.9",
"bin": {
"omp-swarm": "src/cli.ts",
},
@@ -247,7 +247,7 @@
},
"packages/tui": {
"name": "@oh-my-pi/pi-tui",
"version": "16.2.6",
"version": "16.2.9",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
@@ -288,7 +288,7 @@
},
"packages/utils": {
"name": "@oh-my-pi/pi-utils",
"version": "16.2.6",
"version": "16.2.9",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"handlebars": "catalog:",
@@ -301,7 +301,7 @@
},
"packages/wire": {
"name": "@oh-my-pi/pi-wire",
"version": "16.2.6",
"version": "16.2.9",
"devDependencies": {
"@types/bun": "catalog:",
},
@@ -337,18 +337,18 @@
"@huggingface/transformers": "^4.2.0",
"@mozilla/readability": "^0.6.0",
"@napi-rs/cli": "3.7.0",
"@oh-my-pi/hashline": "16.2.6",
"@oh-my-pi/omp-stats": "16.2.6",
"@oh-my-pi/pi-agent-core": "16.2.6",
"@oh-my-pi/pi-ai": "16.2.6",
"@oh-my-pi/pi-catalog": "16.2.6",
"@oh-my-pi/pi-coding-agent": "16.2.6",
"@oh-my-pi/pi-mnemopi": "16.2.6",
"@oh-my-pi/pi-natives": "16.2.6",
"@oh-my-pi/pi-tui": "16.2.6",
"@oh-my-pi/pi-utils": "16.2.6",
"@oh-my-pi/pi-wire": "16.2.6",
"@oh-my-pi/snapcompact": "16.2.6",
"@oh-my-pi/hashline": "16.2.9",
"@oh-my-pi/omp-stats": "16.2.9",
"@oh-my-pi/pi-agent-core": "16.2.9",
"@oh-my-pi/pi-ai": "16.2.9",
"@oh-my-pi/pi-catalog": "16.2.9",
"@oh-my-pi/pi-coding-agent": "16.2.9",
"@oh-my-pi/pi-mnemopi": "16.2.9",
"@oh-my-pi/pi-natives": "16.2.9",
"@oh-my-pi/pi-tui": "16.2.9",
"@oh-my-pi/pi-utils": "16.2.9",
"@oh-my-pi/pi-wire": "16.2.9",
"@oh-my-pi/snapcompact": "16.2.9",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/context-async-hooks": "^2.7.1",
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
@@ -1042,7 +1042,7 @@
"duck": ["duck@0.1.12", "", { "dependencies": { "underscore": "^1.13.1" } }, "sha512-wkctla1O6VfP89gQ+J/yDesM0S7B7XLXjKGzXxMDVFg7uEn706niAtyYovKbyq1oT9YwDcly721/iUWoc8MVRg=="],
"electron-to-chromium": ["electron-to-chromium@1.5.379", "", {}, "sha512-v/qV5aV5EUA2pGilzUCq5/eyOloZAqDZBu9UMBIzgPpLlprjSR6zswsWBTv0KpqxLGUAZEwhO95ZCt7srymNVA=="],
"electron-to-chromium": ["electron-to-chromium@1.5.380", "", {}, "sha512-W6d5AbuEoRayO447cqrg6lKJIlscgRnnxOZl/08kfV71BQDoEBC7Wwis68z87LjyK6f4kWyTaubuDbhHKrZkbA=="],
"emnapi": ["emnapi@1.11.1", "", { "peerDependencies": { "node-addon-api": ">= 6.1.0" }, "optionalPeers": ["node-addon-api"] }, "sha512-kSRjhIcxjMFsBqk7ORvoc9aA5SBKDmecrtF5RMcmOTao0kD/zamaxsuTxMI8C1//wGUuvE7a+19pCE7AEhGVnA=="],
@@ -1146,7 +1146,7 @@
"js-tokens": ["js-tokens@4.0.0", "", {}, "sha512-RdJUflcE3cUzKiMqQgsCu06FPu9UdIJO0beYbPhHN4k6apgJtifcoCtT9bcxOpYBtpD2kCM6Sbzg4CausW/PKQ=="],
"js-yaml": ["js-yaml@4.2.0", "", { "dependencies": { "argparse": "^2.0.1" }, "bin": { "js-yaml": "bin/js-yaml.js" } }, "sha512-ePWsvanv0DWuDRsW8dnt+R4jQ31SCRCQ7hhNcPXZPsoBZiemuZNYGf7adZdqX2D86j6rvKp3RpCxVTSb8WQlOw=="],
"js-yaml": ["js-yaml@4.3.0", "", { "dependencies": { "argparse": "^2.0.1" }, "bin": { "js-yaml": "bin/js-yaml.js" } }, "sha512-1td788aAnnZ5qs7V2QIRl1owjtYpbKt749Y3xauqQgwIIGF/xXWz1wMTEBx5O3LK3lXLVuqXPdPxj2BoFHaW9Q=="],
"jsesc": ["jsesc@3.1.0", "", { "bin": { "jsesc": "bin/jsesc" } }, "sha512-/sM3dO2FOzXjKQhJuo0Q173wf2KOo8t4I8vHy6lF9poUp7bKT0/NHE8fPX23PwfhnykfqnC2xRxOnVw5XuGIaA=="],
@@ -1282,7 +1282,7 @@
"postcss": ["postcss@8.5.15", "", { "dependencies": { "nanoid": "^3.3.12", "picocolors": "^1.1.1", "source-map-js": "^1.2.1" } }, "sha512-FfR8sjd4em2T6fb3I2MwAJU7HWVMr9zba+enmQeeWFfCbm+UOC/0X4DS8XtpUTMwWMGbjKYP7xjfNekzyGmB3A=="],
"prettier": ["prettier@3.8.5", "", { "bin": { "prettier": "bin/prettier.cjs" } }, "sha512-zxcTTCedNGJM4R8sj/Cq/F0W/c4iE0afWBcBwMTRtw4WHYP9TWkYjdiH3npPRUYsXQCPR0hTU9yjovOu+E6EQA=="],
"prettier": ["prettier@3.9.0", "", { "bin": { "prettier": "bin/prettier.cjs" } }, "sha512-LjIqSIC5VYLzs9WedVmJ2ljNAGnU+DteIClbahu4L/DBeWjZ6iT/k1lAYyu9JUh+1xINxWadaPw/Pl63y/agAw=="],
"process-nextick-args": ["process-nextick-args@2.0.1", "", {}, "sha512-3ouUOpQhtgrbOa17J7+uxOTpITYWaGP7/AhoR3+A+/1e9skrzelGi/dXzEYyvbxubEF6Wn2ypscTKiKJFFn1ag=="],
+1 -1
View File
@@ -173,7 +173,7 @@ fn create_windows_napi_tokio_runtime() -> Option<tokio::runtime::Runtime> {
/// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in
/// `packages/natives/native/index.js` (which derives the name from
/// `package.json#version`).
#[napi(js_name = "__piNativesV16_2_6")]
#[napi(js_name = "__piNativesV16_2_9")]
pub const fn pi_natives_version_sentinel() {}
/// Native module entry point: install crash diagnostics before any tool can
+2 -2
View File
@@ -207,8 +207,8 @@ OAuth host chain: `KIMI_CODE_OAUTH_HOST` → `KIMI_OAUTH_HOST` → `https://auth
| `PI_CODEX_WEBSOCKET_IDLE_TIMEOUT_MS` | Positive integer override (default 300000) |
| `PI_CODEX_WEBSOCKET_RETRY_BUDGET` | Non-negative integer override (default 5) |
| `PI_CODEX_WEBSOCKET_RETRY_DELAY_MS` | Positive integer base backoff override (default 500) |
| `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` | Positive integer OpenAI first-event timeout override |
| `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` | Positive integer OpenAI stream idle timeout override |
| `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` | Positive integer OpenAI first-event timeout override; `0` disables. `omp config set providers.streamFirstEventTimeoutSeconds <seconds>` provides the persisted config equivalent |
| `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` | Positive integer OpenAI stream idle timeout override; `0` disables. `omp config set providers.streamIdleTimeoutSeconds <seconds>` provides the persisted config equivalent |
### Cursor provider debug
+1 -1
View File
@@ -127,7 +127,7 @@ Extraction favors **precision** (do not pollute long-term memory) → **Qwen3-1.
pick** (its consolidation is good enough). If running a second model for consolidation, **gemma-3-1b**
wins that task.
**Shipped local options**: `qwen3-1.7b` (recommended), `gemma-3-1b`, `qwen2.5-1.5b`, `lfm2-1.2b`.
**Shipped local options**: `llama3.2:3b`, `qwen3-1.7b` (recommended), `gemma-3-1b`, `qwen2.5-1.5b`, `lfm2-1.2b`.
**Default**: `online` (the configured smol model).
### Known Mnemopi parser bugs (surfaced by these experiments)
+1 -1
View File
@@ -179,7 +179,7 @@ Lifecycle/state transition:
15. restore model via `getRestorableSessionModels(sessionContext.models, lastModelChangeRole)` — tries the recorded models in fallback order and uses the first one present in the model registry
16. restore thinking level and service tier:
- thinking uses persisted `thinking_level_change`, otherwise the configured default clamped to model capability
- service tier uses persisted `service_tier_change`, otherwise the configured `serviceTier` setting (`"none"` becomes unset)
- service tier uses persisted `service_tier_change`, otherwise the configured per-family `tier.openai`/`tier.anthropic`/`tier.google` settings (`"none"` becomes unset)
17. reconnect agent listeners, run the registered session-switch reconciler if any (interactive mode re-enters persisted modes; errors logged, not fatal), and return `true`
## UI state rebuild after interactive switch
+2 -2
View File
@@ -177,11 +177,11 @@ Stores an `AgentMessage` directly.
"id": "c1d2e3f4",
"parentId": "b1c2d3e4",
"timestamp": "2026-02-16T10:21:45.000Z",
"serviceTier": "flex"
"serviceTier": { "openai": "priority", "google": "flex" }
}
```
`serviceTier` can also be `null`.
`serviceTier` is a per-family map keyed by `openai`/`anthropic`/`google` (each value `auto`/`default`/`flex`/`scale`/`priority`), or `null` when no tier is active. Legacy entries that stored a single string (`"flex"`, `"openai-only"`, `"claude-only"`, …) are normalized to this map on read.
### `thinking_level_change`
+5 -1
View File
@@ -366,7 +366,11 @@ A value of `-1` means "use the provider/model default" — `omp` does not send t
| `minP` | number | `-1` | Minimum-probability cutoff. |
| `presencePenalty` | number | `-1` | Presence penalty. |
| `repetitionPenalty` | number | `-1` | Repetition penalty. |
| `serviceTier` | enum | `none` | `none`, `auto`, `default`, `flex`, `scale`, `priority`, `openai-only`, `claude-only`. |
| `tier.openai` | enum | `none` | `none`, `auto`, `default`, `flex`, `scale`, `priority`. Sent as `service_tier` for OpenAI / OpenAI-Codex and OpenAI-family OpenRouter models. |
| `tier.anthropic` | enum | `none` | `none`, `priority`. `priority` realizes fast mode on supported direct Claude models (ignored on Bedrock/Vertex and via OpenRouter). |
| `tier.google` | enum | `none` | `none`, `flex`, `priority`. Gemini API sends it in the body; Vertex sends `priority` via header (`flex` is a no-op on Vertex). |
| `tier.subagent` | enum | `inherit` | `inherit`, `none`, `auto`, `default`, `flex`, `scale`, `priority`. Applied to the spawned model's family; `inherit` tracks the main agent. |
| `tier.advisor` | enum | `none` | `inherit`, `none`, `auto`, `default`, `flex`, `scale`, `priority`. Applied to the advisor model's family. |
| `personality` | enum | `default` | `default`, `friendly`, `pragmatic`, `none`. |
### Retry and fallback
+2 -2
View File
@@ -44,7 +44,7 @@ Bundled agents are embedded at build time (`src/task/agents.ts`) using text impo
`EMBEDDED_AGENT_DEFS` defines:
- `explore`, `plan`, `designer`, `reviewer`, `librarian`, `oracle` from prompt files
- `task` and `quick_task` from shared `task.md` body plus injected frontmatter
- `task` and `sonic` from shared `task.md` body plus injected frontmatter
Loading path:
@@ -130,7 +130,7 @@ Runtime output schema precedence in `TaskTool.#runSpawn`:
(`effectiveOutputSchema = effectiveAgent.output ?? this.session.outputSchema` — the task call itself never carries a schema; ad-hoc structured workflows go through the eval bridge's `agent(prompt, schema)`.)
The model-facing prompt (`src/prompts/tools/task.md`) no longer carries the old structured-output mismatch warning; it tags read-only agents and warns against offloading reasoning to `explore`/`quick_task` instead.
The model-facing prompt (`src/prompts/tools/task.md`) no longer carries the old structured-output mismatch warning; it tags read-only agents and warns against offloading reasoning to `explore`/`sonic` instead.
## Command discovery interaction
+1 -1
View File
@@ -106,7 +106,7 @@ Artifacts and side channels:
- off — single spawn per call; `tasks`/`context` are rejected and removed from the schema.
- Isolation mode (`task.isolation.mode`): `none`, `auto`, `apfs`, `btrfs`, `zfs`, `reflink`, `overlayfs`, `projfs`, `block-clone`, `rcopy` (legacy `worktree`, `fuse-overlay`, `fuse-projfs` accepted for back-compat); the PAL resolves the actual backend with fallback.
- Isolation merge strategy: patch mode (capture/apply root patches) or branch mode (commit to `omp/task/<id>`, cherry-pick into parent).
- Agent source precedence: project custom agents, then user custom agents, then bundled agents (`explore`, `plan`, `designer`, `reviewer`, `task`, `quick_task`, `librarian`, `oracle`).
- Agent source precedence: project custom agents, then user custom agents, then bundled agents (`explore`, `plan`, `designer`, `reviewer`, `task`, `sonic`, `librarian`, `oracle`).
## Side Effects
- Filesystem
+12 -12
View File
@@ -25,18 +25,18 @@
"@huggingface/transformers": "^4.2.0",
"@mozilla/readability": "^0.6.0",
"@napi-rs/cli": "3.7.0",
"@oh-my-pi/hashline": "16.2.6",
"@oh-my-pi/omp-stats": "16.2.6",
"@oh-my-pi/pi-agent-core": "16.2.6",
"@oh-my-pi/pi-ai": "16.2.6",
"@oh-my-pi/pi-catalog": "16.2.6",
"@oh-my-pi/pi-coding-agent": "16.2.6",
"@oh-my-pi/pi-mnemopi": "16.2.6",
"@oh-my-pi/pi-natives": "16.2.6",
"@oh-my-pi/pi-tui": "16.2.6",
"@oh-my-pi/pi-utils": "16.2.6",
"@oh-my-pi/pi-wire": "16.2.6",
"@oh-my-pi/snapcompact": "16.2.6",
"@oh-my-pi/hashline": "16.2.9",
"@oh-my-pi/omp-stats": "16.2.9",
"@oh-my-pi/pi-agent-core": "16.2.9",
"@oh-my-pi/pi-ai": "16.2.9",
"@oh-my-pi/pi-catalog": "16.2.9",
"@oh-my-pi/pi-coding-agent": "16.2.9",
"@oh-my-pi/pi-mnemopi": "16.2.9",
"@oh-my-pi/pi-natives": "16.2.9",
"@oh-my-pi/pi-tui": "16.2.9",
"@oh-my-pi/pi-utils": "16.2.9",
"@oh-my-pi/pi-wire": "16.2.9",
"@oh-my-pi/snapcompact": "16.2.9",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/context-async-hooks": "^2.7.1",
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-agent-core",
"version": "16.2.6",
"version": "16.2.9",
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+2 -2
View File
@@ -1,4 +1,4 @@
import type { ImageContent, MessageAttribution, ServiceTier, TextContent } from "@oh-my-pi/pi-ai";
import type { ImageContent, MessageAttribution, ServiceTierByFamily, TextContent } from "@oh-my-pi/pi-ai";
import type { AgentMessage } from "../types";
export interface SessionEntryBase {
@@ -28,7 +28,7 @@ export interface ModelChangeEntry extends SessionEntryBase {
export interface ServiceTierChangeEntry extends SessionEntryBase {
type: "service_tier_change";
serviceTier: ServiceTier | null;
serviceTier: ServiceTierByFamily | null;
}
export interface CompactionEntry<T = unknown> extends SessionEntryBase {
+1 -3
View File
@@ -30,7 +30,6 @@ import {
completeSimple,
type Message,
type Model,
resolveServiceTier,
type ServiceTier,
type SimpleStreamOptions,
type StopReason,
@@ -752,8 +751,7 @@ function buildChatRequestAttributes(stepNumber: number, request: ChatRequestSnap
attrs[GenAIAttr.RequestStopSequences] = [...request.stopSequences];
}
if (request.serviceTier && shouldSendServiceTier(request.serviceTier, provider)) {
const resolved = resolveServiceTier(request.serviceTier, provider);
if (resolved) attrs[OpenAIAttr.RequestServiceTier] = resolved;
attrs[OpenAIAttr.RequestServiceTier] = request.serviceTier;
}
if (request.reasoningEffort) attrs[PiGenAIAttr.RequestReasoningEffort] = request.reasoningEffort;
const toolChoice = serializeToolChoice(request.toolChoice);
+37
View File
@@ -2,6 +2,43 @@
## [Unreleased]
## [16.2.9] - 2026-06-30
### Added
- Added `OAuthCallbackFlowOptions.allowPortFallback` to allow disabling random-port fallback, enabling strict port enforcement and early configuration errors for OAuth flows with static redirect URIs.
### Changed
- Improved `OAuthCallbackFlow` port conflict error messages to include the busy port, configured redirect URI, and actionable remediation steps.
### Fixed
- Fixed an issue where malformed tool-call JSON from local Ollama or llama.cpp models was incorrectly retried as generic 500 errors, now surfacing a clear recovery message.
- Fixed a race condition in OAuth callback flows where abort signals triggered before the callback listener was registered were ignored.
## [16.2.7] - 2026-06-30
### Added
- Added service tier support for Google Gemini and Vertex AI, including model-specific service tier configurations via ServiceTierByFamily.
- Added Google Vertex AI Interactions API support for Gemini 3+ models by default, with automatic fallback to :streamGenerateContent and a useInteractionsApi: false option to force standard generation.
- Added support for explicit Vertex bearer access tokens via GOOGLE_CLOUD_ACCESS_TOKEN or CLOUDSDK_AUTH_ACCESS_TOKEN environment variables.
### Changed
- Updated service tier logic to use per-provider configurations instead of global scopes.
- Refactored priority request billing and accounting to better align with specific provider capabilities.
- Updated API key resolution precedence so explicit environment variables (e.g., GEMINI_API_KEY) override stored or broker-migrated static API keys, while deliberate OAuth logins still take highest precedence.
### Fixed
- Improved Vertex AI reliability by automatically falling back to global endpoints on 404 errors.
- Fixed safety setting application for Google Vertex AI models.
- Fixed Kimi Code's Anthropic-compatible request path to keep thinking enabled and downgrade forced tool choice for Kimi K2.7 Code title generation.
- Fixed leaked reasoning fences (such as ```thinking or <think>) across all providers by splitting them into structured thinking blocks during streaming.
- Fixed Codex requests failing with unsupported all_turns errors on older models (gpt-5.1 and gpt-5.3) by gating the reasoning.context: "all_turns" default to gpt-5.4+ models.
## [16.2.6] - 2026-06-29
### Fixed
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-ai",
"version": "16.2.6",
"version": "16.2.9",
"description": "Unified LLM API with automatic model discovery and provider configuration",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+56 -52
View File
@@ -1736,8 +1736,8 @@ export class AuthStorage {
/**
* Classify where a provider's auth comes from, following the same precedence
* as {@link AuthStorage.getApiKey}: runtime override → config override →
* stored credential (api_key before oauth, matching getApiKey) → env var →
* fallback resolver. Returns undefined when no auth is configured.
* stored OAuth → env var → stored api_key → fallback resolver. Returns
* undefined when no auth is configured.
*
* Compact, structured counterpart to {@link describeCredentialSource}.
*/
@@ -1745,10 +1745,9 @@ export class AuthStorage {
if (this.#runtimeOverrides.has(provider)) return { kind: "runtime" };
if (this.#configOverrides.has(provider)) return { kind: "config" };
const stored = this.#getCredentialsForProvider(provider);
if (stored.length > 0) {
return { kind: stored.some(credential => credential.type === "api_key") ? "api_key" : "oauth" };
}
if (stored.some(credential => credential.type === "oauth")) return { kind: "oauth" };
if (getEnvApiKey(provider)) return { kind: "env", envVar: getEnvApiKeyName(provider) };
if (stored.some(credential => credential.type === "api_key")) return { kind: "api_key" };
if (this.#fallbackResolver?.(provider)) return { kind: "fallback" };
return undefined;
}
@@ -3760,12 +3759,8 @@ export class AuthStorage {
return configKey;
}
const apiKeySelection = this.#selectCredentialByType(provider, "api_key");
if (apiKeySelection) {
return this.#configValueResolver(apiKeySelection.credential.key);
}
// Return current OAuth access token only if it is not already expired.
// Precedence: a deliberate OAuth login wins, then an explicit env var, then a stored
// static api_key (which may be a stale broker-migrated copy) as a last resort.
const oauthSelection = this.#selectCredentialByType(provider, "oauth");
if (oauthSelection) {
const expiresAt = oauthSelection.credential.expires;
@@ -3784,17 +3779,22 @@ export class AuthStorage {
const envKey = getEnvApiKey(provider);
if (envKey) return envKey;
const apiKeySelection = this.#selectCredentialByType(provider, "api_key");
if (apiKeySelection) {
return this.#configValueResolver(apiKeySelection.credential.key);
}
return this.#fallbackResolver?.(provider) ?? undefined;
}
/**
* Get API key for a provider.
* Priority:
* Priority (first match wins):
* 1. Runtime override (CLI --api-key)
* 2. Config override (models.yml `providers.<name>.apiKey`)
* 3. API key from storage
* 4. OAuth token from storage (auto-refreshed)
* 5. Environment variable
* 3. OAuth token from storage (auto-refreshed)
* 4. Environment variable
* 5. Stored API key (e.g. a broker-migrated copy) — last resort, so an explicit env var wins
* 6. Fallback resolver (models.yml custom providers, last-resort)
*/
async getApiKey(provider: string, sessionId?: string, options?: AuthApiKeyOptions): Promise<string | undefined> {
@@ -3814,25 +3814,27 @@ export class AuthStorage {
return configKey;
}
const apiKeySelection = this.#selectCredentialByType(provider, "api_key", sessionId);
if (apiKeySelection) {
this.#recordSessionCredential(provider, sessionId, "api_key", apiKeySelection.index);
return this.#configValueResolver(apiKeySelection.credential.key);
}
// Precedence: a deliberate OAuth login wins, then an explicit env var, then a stored
// static api_key (which may be a stale broker-migrated copy) as a last resort.
const oauthResolved = await this.#resolveOAuthSelection(provider, sessionId, options);
if (oauthResolved) {
return oauthResolved.apiKey;
}
// Fall back to environment variable or custom resolver. If we reach here after
// an OAuth miss, the session sticky (if any) is stale — the request will
// authenticate via env/fallback, not OAuth, so clear the sticky now so that
// getOAuthAccountId() correctly suppresses account_uuid for this session.
// Past OAuth: the session sticky (if any) is stale — the request authenticates via
// env/api_key/fallback, not OAuth, so clear it now so getOAuthAccountId() correctly
// suppresses account_uuid for this session.
if (sessionId) this.#sessionLastCredential.get(provider)?.delete(sessionId);
const envKey = getEnvApiKey(provider);
if (envKey) return envKey;
const apiKeySelection = this.#selectCredentialByType(provider, "api_key", sessionId);
if (apiKeySelection) {
this.#recordSessionCredential(provider, sessionId, "api_key", apiKeySelection.index);
return this.#configValueResolver(apiKeySelection.credential.key);
}
// Fall back to custom resolver (e.g., models.json custom providers)
return this.#fallbackResolver?.(provider) ?? undefined;
}
@@ -4486,12 +4488,13 @@ export class AuthStorage {
/**
* Describe where the active credential for a provider came from.
*
* Surfaces four layers, highest precedence first:
* Mirrors {@link AuthStorage.getApiKey} precedence, highest first:
* 1. Runtime override (`--api-key`).
* 2. Config override (`models.yml` `providers.<name>.apiKey`).
* 3. Stored credential (the one this session is currently sticky to, or the
* one round-robin would pick next when no session id is supplied).
* 4. Env var / fallback resolver — when no stored credential exists.
* 3. Stored OAuth credential.
* 4. Env var — overrides a stored static api_key (e.g. a stale broker copy).
* 5. Stored api_key credential.
* 6. Fallback resolver.
*
* The string is purely informational; consumers must not parse it.
*/
@@ -4505,30 +4508,31 @@ export class AuthStorage {
const baseLabel = this.#sourceLabel ?? "local store";
const stored = this.#getStoredCredentials(provider);
if (stored.length === 0) {
if (getEnvApiKey(provider)) return `env ${baseLabel ? `(fallback over ${baseLabel})` : ""}`.trim();
if (this.#fallbackResolver?.(provider) !== undefined) return `fallback resolver`;
return undefined;
}
const session = sessionId ? this.#sessionLastCredential.get(provider)?.get(sessionId) : undefined;
// Same selection logic as #selectCredentialByType for "no session" lookups: prefer
// the type with stored credentials, lean OAuth before api_key. We don't run the
// full round-robin here because describing the source shouldn't advance the index.
const preferredType: AuthCredential["type"] =
session?.type ?? (stored.some(entry => entry.credential.type === "oauth") ? "oauth" : "api_key");
const typed = stored
.map((entry, index) => ({ entry, index }))
.filter(({ entry }) => entry.credential.type === preferredType);
if (typed.length === 0) return baseLabel;
const index = session?.index ?? typed[0].index;
const chosen = stored[index] ?? typed[0].entry;
const credential = chosen.credential;
const identity =
credential.type === "oauth"
? (credential.email ?? credential.accountId ?? credential.projectId ?? `cred ${chosen.id}`)
: `cred ${chosen.id}`;
return `${baseLabel} · ${preferredType} #${chosen.id} (${identity})`;
// Describe the stored credential of a given type, honoring the session sticky index.
const describeStored = (type: AuthCredential["type"]): string | undefined => {
const typed = stored
.map((entry, index) => ({ entry, index }))
.filter(({ entry }) => entry.credential.type === type);
if (typed.length === 0) return undefined;
const index = session?.type === type ? session.index : typed[0].index;
const chosen = stored[index] ?? typed[0].entry;
const credential = chosen.credential;
const identity =
credential.type === "oauth"
? (credential.email ?? credential.accountId ?? credential.projectId ?? `cred ${chosen.id}`)
: `cred ${chosen.id}`;
return `${baseLabel} · ${type} #${chosen.id} (${identity})`;
};
// A deliberate OAuth login wins; then an explicit env var; then a stored static api_key.
const oauthSource = describeStored("oauth");
if (oauthSource) return oauthSource;
if (getEnvApiKey(provider)) return `env (over ${baseLabel})`;
const apiKeySource = describeStored("api_key");
if (apiKeySource) return apiKeySource;
if (this.#fallbackResolver?.(provider) !== undefined) return "fallback resolver";
return undefined;
}
}
+15 -1
View File
@@ -94,6 +94,14 @@ const MALFORMED_FUNCTION_CALL_PATTERN = /\bmalformed.?function.?call\b/i;
const PROVIDER_FINISH_ERROR_PATTERN = /\bProvider (?:returned error finish_reason|finish_reason:\s*error)\b/i;
const STALE_RESPONSE_ITEM_PATTERNS = [/\bItem with id ['"][^'"]+['"] not found\.?/i, /previous[ _]?response/i] as const;
const STALE_RESPONSE_ITEM_DETAIL_PATTERN = /not[ _]?found|invalid|expired|stale|zero[ _-]?data[ _-]?retention/i;
/**
* Local llama.cpp / Ollama deterministic tool-call argument JSON parse failure.
* The model emitted invalid JSON in a tool call and the server returned HTTP 500
* with this exact text — replaying the same prompt yields the same malformed
* output, so callers strip {@link Flag.Transient} when this matches.
*/
export const LLAMA_CPP_TOOL_CALL_PARSE_PATTERN =
/failed to parse tool call arguments as json|\[json\.exception\.parse_error\.101\]/i;
// Copilot routing flap: HTTP 400 `model_not_supported` (structural code on the
// error, also surfaced in text). Treated as transient — a retry usually lands
@@ -441,7 +449,13 @@ export function classifyMessage(message: {
const currentStatus = message.errorStatus ?? statusFromId(existingId);
const textId = classifyText(message.errorMessage, currentStatus, message.api);
const kinds = ((existingId ?? 0) | textId) & KIND_MASK;
let kinds = ((existingId ?? 0) | textId) & KIND_MASK;
if (message.errorMessage && LLAMA_CPP_TOOL_CALL_PARSE_PATTERN.test(message.errorMessage)) {
// Deterministic local-model tool-call JSON parse failure: HTTP 500 is misleading
// because the same prompt reproduces the same malformed output, so the agent-level
// auto-retry would loop. Strip Transient so the recovery message surfaces immediately.
kinds &= ~Flag.Transient;
}
const id = kinds !== 0 ? create(kinds) : (statusFromId(textId) ?? statusFromId(existingId) ?? currentStatus ?? 0);
message.errorId = id;
+10 -1
View File
@@ -5,6 +5,12 @@ import {
rewriteCopilotError,
} from "../utils/http-inspector";
import { formatErrorMessageWithRetryAfter } from "../utils/retry-after";
import { LLAMA_CPP_TOOL_CALL_PARSE_PATTERN } from "./flags";
function rewriteOllamaToolCallJsonError(message: string): string {
if (!LLAMA_CPP_TOOL_CALL_PARSE_PATTERN.test(message)) return message;
return `Local Ollama model emitted malformed tool-call JSON and llama.cpp rejected it (HTTP 500). This is usually a deterministic model-output failure after context degradation, not a transient server outage; reload the model or reduce context, then retry.\n${message}`;
}
/** Inputs that steer {@link formatMessage}'s formatter selection. */
export interface FormatMessageOptions {
@@ -12,7 +18,7 @@ export interface FormatMessageOptions {
rawRequestDump?: RawHttpRequestDump;
/** Captured non-2xx response body, appended to the message when available. */
capturedErrorResponse?: CapturedHttpErrorResponse;
/** Provider id; `"github-copilot"` triggers the copilot message rewrite. */
/** Provider id; gates provider-specific user-facing rewrites. */
provider?: string;
}
@@ -32,5 +38,8 @@ export async function formatMessage(error: unknown, opts: FormatMessageOptions =
if (opts.provider === "github-copilot") {
message = rewriteCopilotError(message, error, opts.provider);
}
if (opts.provider === "ollama") {
message = rewriteOllamaToolCallJsonError(message);
}
return message;
}
@@ -0,0 +1,112 @@
import { describe, expect, it } from "bun:test";
import { getBundledModel } from "@oh-my-pi/pi-catalog";
import type { Context } from "../../types";
import type { MessageCreateParamsStreaming } from "../anthropic-wire";
import { streamOpenAIAnthropicShim } from "../openai-anthropic-shim";
import {
applyChatCompletionsCompatPolicy,
type OpenAICompletionsParams,
resolveOpenAICompatPolicy,
} from "../openai-shared";
const BASE_CHAT_COMPLETIONS_PARAMS: OpenAICompletionsParams = { messages: [], model: "unused", stream: true };
const TITLE_CONTEXT: Context = {
systemPrompt: ["Generate a title."],
messages: [{ role: "user", content: "Explain the login failure", timestamp: 0 }],
tools: [
{
name: "set_title",
description: "Set title",
parameters: {
type: "object",
properties: { title: { type: "string" } },
required: ["title"],
additionalProperties: false,
},
},
],
};
describe("Kimi K2.7 Code thinking policy", () => {
it("omits disabled thinking for title-generator-style Kimi Code requests", () => {
const model = getBundledModel<"openai-completions">("kimi-code", "kimi-for-coding");
const policy = resolveOpenAICompatPolicy(model, {
endpoint: "chat-completions",
disableReasoning: true,
toolChoice: { type: "tool", name: "set_title" },
});
const params = { ...BASE_CHAT_COMPLETIONS_PARAMS };
applyChatCompletionsCompatPolicy(params, policy);
expect("thinking" in params).toBe(false);
expect(model.compat.supportsForcedToolChoice).toBe(false);
});
it("enables thinking and downgrades forced tool choice on Kimi Code's Anthropic endpoint", async () => {
const model = getBundledModel<"openai-completions">("kimi-code", "kimi-for-coding");
let payload: MessageCreateParamsStreaming | undefined;
const stream = streamOpenAIAnthropicShim(
model,
TITLE_CONTEXT,
{
apiKey: "test-key",
maxTokens: 1024,
disableReasoning: true,
toolChoice: { type: "tool", name: "set_title" },
onPayload: body => {
payload = body as MessageCreateParamsStreaming;
throw new Error("stop after payload capture");
},
},
{
anthropicBaseUrl: "https://api.kimi.com/coding",
defaultFormat: "anthropic",
},
);
await stream.result();
expect(payload?.thinking?.type).toBe("enabled");
expect(payload?.tool_choice).toEqual({ type: "auto" });
});
it("omits disabled thinking for native Moonshot Kimi K2.7 Code variants", () => {
for (const modelId of ["kimi-k2.7-code", "kimi-k2.7-code-highspeed"]) {
const model = getBundledModel<"openai-completions">("moonshot", modelId);
const policy = resolveOpenAICompatPolicy(model, {
endpoint: "chat-completions",
disableReasoning: true,
});
const params = { ...BASE_CHAT_COMPLETIONS_PARAMS };
applyChatCompletionsCompatPolicy(params, policy);
expect("thinking" in params).toBe(false);
expect(model.compat.supportsForcedToolChoice).toBe(false);
}
});
it("keeps the openai disable shape for non-native Kimi K2.7 Code aliases", () => {
for (const { provider, id } of [
{ provider: "fireworks", id: "kimi-k2.7-code" },
{ provider: "openrouter", id: "moonshotai/kimi-k2.7-code" },
] as const) {
const model = getBundledModel<"openai-completions">(provider, id);
expect(model.compat.supportsForcedToolChoice).toBe(true);
expect(model.compat.reasoningDisableMode).not.toBe("omit");
}
});
it("keeps explicit disabled thinking for Kimi K2.6", () => {
const model = getBundledModel<"openai-completions">("moonshot", "kimi-k2.6");
const policy = resolveOpenAICompatPolicy(model, {
endpoint: "chat-completions",
disableReasoning: true,
});
const params = { ...BASE_CHAT_COMPLETIONS_PARAMS };
applyChatCompletionsCompatPolicy(params, policy);
expect(params.thinking).toEqual({ type: "disabled" });
});
});
+11 -10
View File
@@ -43,7 +43,6 @@ import type {
ToolResultMessage,
Usage,
} from "../types";
import { resolveServiceTier } from "../types";
import { isRecord, normalizeSystemPrompts, normalizeToolCallId, resolveCacheRetention } from "../utils";
import { createAbortSourceTracker } from "../utils/abort";
import {
@@ -1595,7 +1594,7 @@ const streamAnthropicOnce = (
isOAuthToken = false;
} else {
const extraBetas = normalizeExtraBetas(options?.betas);
const wantsAnthropicPriority = resolveServiceTier(options?.serviceTier, model.provider) === "priority";
const wantsAnthropicPriority = model.provider === "anthropic" && options?.serviceTier === "priority";
// Skip the fast-mode beta when this session already learned the
// endpoint+model rejects fast mode; `speed` is dropped from the params
// too (dropFastMode), so the request stays a faithful non-fast request.
@@ -2191,7 +2190,8 @@ const streamAnthropicOnce = (
}
if (
!dropFastMode &&
resolveServiceTier(options?.serviceTier, model.provider) === "priority" &&
model.provider === "anthropic" &&
options?.serviceTier === "priority" &&
firstTokenTime === undefined &&
AIError.isFastModeUnsupported(streamFailure)
) {
@@ -2259,7 +2259,7 @@ const streamAnthropicOnce = (
}
output.duration = performance.now() - startTime;
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
if (dropFastMode && resolveServiceTier(options?.serviceTier, model.provider) === "priority") {
if (dropFastMode && model.provider === "anthropic" && options?.serviceTier === "priority") {
output.disabledFeatures = [...(output.disabledFeatures ?? []), "priority"];
}
stream.push({ type: "done", reason: output.stopReason, message: output });
@@ -2876,9 +2876,10 @@ function buildParams(
let thinking: MessageCreateParamsStreaming["thinking"] | undefined;
let outputConfigEffort: AnthropicOutputEffort | undefined;
if (model.reasoning) {
if (options?.thinkingEnabled) {
if (options?.thinkingEnabled || model.compat.requiresThinkingEnabled) {
const thinkingOptions = options ?? {};
const mode = model.thinking?.mode;
const effort = resolveAnthropicAdaptiveEffort(model, options);
const effort = resolveAnthropicAdaptiveEffort(model, thinkingOptions);
const compat = model.compat;
if (mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) {
const adaptive: { type: "adaptive"; display?: AnthropicThinkingDisplay } = { type: "adaptive" };
@@ -2889,15 +2890,15 @@ function buildParams(
// support: Opus 4.6 / Sonnet 4.6+ reject it with a 400, so an explicit
// `thinkingDisplay` MUST NOT force it onto a model that can't accept it.
if (model.thinking?.supportsDisplay) {
adaptive.display = options.thinkingDisplay ?? "summarized";
adaptive.display = thinkingOptions.thinkingDisplay ?? "summarized";
}
thinking = adaptive;
if (effort && effort !== "adaptive") outputConfigEffort = effort;
} else {
thinking = {
type: "enabled",
budget_tokens: options.thinkingBudgetTokens || 1024,
display: options.thinkingDisplay ?? "summarized",
budget_tokens: thinkingOptions.thinkingBudgetTokens || 1024,
display: thinkingOptions.thinkingDisplay ?? "summarized",
};
if (mode === "anthropic-budget-effort" && effort && effort !== "adaptive") outputConfigEffort = effort;
}
@@ -2996,7 +2997,7 @@ function buildParams(
seqs.length > ANTHROPIC_STOP_SEQUENCES_MAX ? seqs.slice(0, ANTHROPIC_STOP_SEQUENCES_MAX) : seqs;
}
if (resolveServiceTier(options?.serviceTier, model.provider) === "priority") {
if (model.provider === "anthropic" && options?.serviceTier === "priority") {
params.speed = "fast";
}
+25
View File
@@ -13,6 +13,7 @@
*/
import { Buffer } from "node:buffer";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { $envpos, isEnoent, logger } from "@oh-my-pi/pi-utils";
@@ -281,6 +282,11 @@ const SHARED_TOKEN_RESOLVE_TIMEOUT_MS = 30_000;
* The token is cached in module scope and refreshed `GOOGLE_VERTEX_REFRESH_SKEW_MS` ms before it expires.
*/
export async function getVertexAccessToken(options?: { signal?: AbortSignal; fetch?: FetchImpl }): Promise<string> {
// An explicit access token (e.g. `gcloud auth print-access-token`) bypasses the cache so a
// refreshed env token takes effect immediately. `CLOUDSDK_AUTH_ACCESS_TOKEN` is gcloud's own
// override var; `GOOGLE_CLOUD_ACCESS_TOKEN` is the omp-facing alias.
const explicitToken = Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN || Bun.env.CLOUDSDK_AUTH_ACCESS_TOKEN;
if (explicitToken) return explicitToken;
const fetchImpl = options?.fetch ?? globalThis.fetch.bind(globalThis);
const skew = getRefreshSkewMs();
const now = Date.now();
@@ -323,3 +329,22 @@ export function __resetVertexTokenCache(): void {
tokenCache.clear();
inflight.clear();
}
/**
* Sync best-effort probe for a usable Vertex bearer credential source — an explicit access-token
* env var, `GOOGLE_APPLICATION_CREDENTIALS`, a user ADC file, or a GCP runtime whose metadata
* server can mint ADC (GCE/Cloud Run/App Engine/Functions). Lets callers prefer the bearer
* Interactions transport only when ADC is actually reachable, without paying the async
* metadata-probe cost for API-key-only setups.
*/
export function hasVertexBearerCredentialsHint(): boolean {
if (Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN || Bun.env.CLOUDSDK_AUTH_ACCESS_TOKEN) return true;
if (Bun.env.GOOGLE_APPLICATION_CREDENTIALS) return true;
// GCP-hosted runtimes expose ADC via the metadata server; these env vars mark those runtimes.
if (Bun.env.K_SERVICE || Bun.env.FUNCTION_TARGET || Bun.env.GAE_ENV || Bun.env.GCE_METADATA_HOST) return true;
try {
return fs.existsSync(userAdcPath());
} catch {
return false;
}
}
+90 -41
View File
@@ -35,6 +35,7 @@ import { armPreResponseTimeout, getStreamFirstEventTimeoutMs } from "../utils/id
// Refresh is the sole responsibility of AuthStorage (broker-aware, single-flighted);
// the stream provider trusts the access token threaded through `options.apiKey`.
import { normalizeSchemaForCCA } from "../utils/schema";
import { StreamMarkupHealing, type StreamMarkupHealingEvent } from "../utils/stream-markup-healing";
import type { Content, FunctionCallingConfigMode, ThinkingConfig } from "./google-shared";
import {
convertMessages,
@@ -665,15 +666,43 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
let currentBlock: TextContent | ThinkingContent | null = null;
const blocks = output.content;
const blockIndex = () => blocks.length - 1;
const visibleTextHealing = new StreamMarkupHealing({ pattern: "thinking" });
let isBuffering = false;
let textBuffer = "";
let bufferedTextSignature: string | undefined;
const emitVisibleText = (delta: string, thoughtSignature?: string) => {
if (!delta || !currentBlock || currentBlock.type !== "text") return;
currentBlock.text += delta;
currentBlock.textSignature = retainThoughtSignature(currentBlock.textSignature, thoughtSignature);
const endCurrentBlock = (): void => {
if (!currentBlock) return;
pushBlockEndEvent(currentBlock, blockIndex(), output, stream);
currentBlock = null;
};
const startTextBlock = (): TextContent => {
let block = currentBlock;
if (block?.type !== "text") {
endCurrentBlock();
block = startTextOrThinkingBlock(false, output, stream, ensureStarted);
currentBlock = block;
}
return block;
};
const startThinkingBlock = (): ThinkingContent => {
let block = currentBlock;
if (block?.type !== "thinking") {
endCurrentBlock();
block = startTextOrThinkingBlock(true, output, stream, ensureStarted);
currentBlock = block;
}
return block;
};
const emitVisibleText = (delta: string, thoughtSignature?: string): void => {
if (!delta) return;
const block = startTextBlock();
block.text += delta;
block.textSignature = retainThoughtSignature(block.textSignature, thoughtSignature);
stream.push({
type: "text_delta",
contentIndex: blockIndex(),
@@ -682,6 +711,48 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
});
};
const emitVisibleThinking = (delta: string): void => {
if (!delta) return;
const block = startThinkingBlock();
block.thinking += delta;
stream.push({
type: "thinking_delta",
contentIndex: blockIndex(),
delta,
partial: output,
});
};
const emitHealingEvent = (event: StreamMarkupHealingEvent, thoughtSignature?: string): void => {
if (event.type === "text") {
emitVisibleText(event.text, thoughtSignature);
} else if (event.type === "thinking") {
emitVisibleThinking(event.thinking);
}
};
const feedVisibleText = (delta: string, thoughtSignature?: string): void => {
for (const event of visibleTextHealing.feedEvents(delta)) {
emitHealingEvent(event, thoughtSignature);
}
};
const flushVisibleText = (thoughtSignature?: string): void => {
for (const event of visibleTextHealing.flushEvents()) {
emitHealingEvent(event, thoughtSignature);
}
};
const retainCurrentBlockThoughtSignature = (thoughtSignature: string): void => {
const block = currentBlock;
if (!block) return;
if (block.type === "thinking") {
block.thinkingSignature = retainThoughtSignature(block.thinkingSignature, thoughtSignature);
} else {
block.textSignature = retainThoughtSignature(block.textSignature, thoughtSignature);
}
};
for await (const chunk of readSseJson<CloudCodeAssistResponseChunk>(
activeResponse.body!,
options?.signal,
@@ -710,20 +781,12 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
for (const part of candidate.content.parts) {
if (part.text !== undefined && part.text !== "") {
const isThinking = isThinkingPart(part);
if (
!currentBlock ||
(isThinking && currentBlock.type !== "thinking") ||
(!isThinking && currentBlock.type !== "text")
) {
if (currentBlock) {
pushBlockEndEvent(currentBlock, blockIndex(), output, stream);
}
currentBlock = startTextOrThinkingBlock(isThinking, output, stream, ensureStarted);
}
if (currentBlock.type === "thinking") {
currentBlock.thinking += part.text;
currentBlock.thinkingSignature = retainThoughtSignature(
currentBlock.thinkingSignature,
if (isThinking) {
flushVisibleText();
const block = startThinkingBlock();
block.thinking += part.text;
block.thinkingSignature = retainThoughtSignature(
block.thinkingSignature,
part.thoughtSignature,
);
stream.push({
@@ -744,7 +807,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
textBuffer = part.text;
bufferedTextSignature = part.thoughtSignature;
} else {
emitVisibleText(part.text, part.thoughtSignature);
feedVisibleText(part.text, part.thoughtSignature);
}
if (isBuffering) {
@@ -757,32 +820,19 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
isBuffering = false;
textBuffer = "";
bufferedTextSignature = undefined;
emitVisibleText(buffered.visibleText, visibleSignature);
feedVisibleText(buffered.visibleText, visibleSignature);
}
}
}
} else if (part.text === "" && part.thoughtSignature && currentBlock && !part.functionCall) {
if (currentBlock.type === "thinking") {
currentBlock.thinkingSignature = retainThoughtSignature(
currentBlock.thinkingSignature,
part.thoughtSignature,
);
} else {
currentBlock.textSignature = retainThoughtSignature(
currentBlock.textSignature,
part.thoughtSignature,
);
}
} else if (part.text === "" && part.thoughtSignature && !part.functionCall) {
retainCurrentBlockThoughtSignature(part.thoughtSignature);
}
if (part.functionCall) {
if (currentBlock) {
pushBlockEndEvent(currentBlock, blockIndex(), output, stream);
currentBlock = null;
}
flushVisibleText();
endCurrentBlock();
isBuffering = false;
textBuffer = "";
const providedId = part.functionCall.id;
const needsNewId =
!providedId || output.content.some(b => b.type === "toolCall" && b.id === providedId);
@@ -848,16 +898,15 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
sawLeak = true;
}
if (buffered.kind !== "incomplete") {
emitVisibleText(buffered.visibleText, bufferedTextSignature);
feedVisibleText(buffered.visibleText, bufferedTextSignature);
}
bufferedTextSignature = undefined;
isBuffering = false;
textBuffer = "";
}
if (currentBlock) {
pushBlockEndEvent(currentBlock, blockIndex(), output, stream);
}
flushVisibleText(bufferedTextSignature);
endCurrentBlock();
return hasMeaningfulGoogleContent(output) || sawLeak;
};
@@ -0,0 +1,753 @@
import { parseGeminiModel } from "@oh-my-pi/pi-catalog/identity";
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
import { fetchWithRetry, readSseJson } from "@oh-my-pi/pi-utils";
import * as AIError from "../error";
import type {
AssistantMessage,
Context,
FetchImpl,
ImageContent,
Message,
Model,
ProviderSessionState,
TextContent,
ToolCall,
Usage,
} from "../types";
import { shouldSendServiceTier } from "../types";
import { normalizeSystemPrompts } from "../utils";
import { AssistantMessageEventStream } from "../utils/event-stream";
import { convertTools, type GoogleSharedStreamOptions, type GoogleThinkingLevel } from "./google-shared";
type GoogleInteractionsApi = "google-generative-ai" | "google-vertex";
type GoogleInteractionsModel = Model<GoogleInteractionsApi>;
type GoogleOptions = GoogleSharedStreamOptions;
const GOOGLE_INTERACTIONS_STATE_KEY = "google-interactions-state";
/** Provider session state storing the last Gemini Interactions response id. */
export interface GoogleInteractionsProviderSessionState extends ProviderSessionState {
lastInteractionId?: string;
}
/** Conversation anchor for continuing an Interactions turn from a prior assistant response. */
export interface InteractionAnchor {
id?: string;
messageIndex?: number;
}
type InteractionContent = { type: "text"; text: string } | { type: "image"; data: string; mime_type: string };
interface InteractionUserInputStep {
type: "user_input";
content: InteractionContent[];
}
interface InteractionModelOutputStep {
type: "model_output";
content: InteractionContent[];
}
interface InteractionFunctionCallStep {
type: "function_call";
id: string;
name: string;
arguments: Record<string, unknown>;
}
interface InteractionFunctionResultStep {
type: "function_result";
name: string;
call_id: string;
result: InteractionContent[];
is_error?: boolean;
}
interface InteractionThoughtStep {
type: "thought";
summary?: InteractionContent[];
signature?: string;
}
interface InteractionThoughtSummaryDelta {
type: "thought_summary";
content?: InteractionContent;
}
interface InteractionThoughtSignatureDelta {
type: "thought_signature";
signature?: string;
}
interface InteractionArgumentsDelta {
type: "arguments_delta";
arguments?: string;
}
interface PendingInteractionToolCall {
id: string;
name: string;
argumentsText: string;
argumentsObject: Record<string, unknown>;
}
type InteractionInputStep =
| InteractionUserInputStep
| InteractionModelOutputStep
| InteractionFunctionCallStep
| InteractionFunctionResultStep;
type InteractionStep =
| InteractionModelOutputStep
| InteractionFunctionCallStep
| InteractionFunctionResultStep
| InteractionThoughtStep
| InteractionUserInputStep;
type InteractionThinkingLevel = "minimal" | "low" | "medium" | "high";
interface InteractionGenerationConfig {
temperature?: number;
top_p?: number;
top_k?: number;
min_p?: number;
presence_penalty?: number;
frequency_penalty?: number;
repetition_penalty?: number;
max_output_tokens?: number;
thinking_level?: InteractionThinkingLevel;
thinking_budget?: number;
}
interface GoogleInteractionRequest {
model: string;
input: InteractionInputStep[];
stream: true;
previous_interaction_id?: string;
system_instruction?: string;
tools?: { functionDeclarations: Record<string, unknown>[] }[];
store?: boolean;
generation_config?: InteractionGenerationConfig;
service_tier?: string;
}
interface InteractionUsage {
total_input_tokens?: number;
total_cached_tokens?: number;
total_output_tokens?: number;
total_thought_tokens?: number;
total_tokens?: number;
}
interface InteractionResource {
id?: string;
status?: string;
usage?: InteractionUsage;
}
interface InteractionStreamMetadata {
total_usage?: InteractionUsage;
}
interface InteractionSseEvent {
event_type?: string;
index?: number;
step?: InteractionStep;
delta?:
| InteractionContent
| InteractionFunctionCallStep
| InteractionThoughtStep
| InteractionThoughtSummaryDelta
| InteractionThoughtSignatureDelta
| InteractionArgumentsDelta;
interaction?: InteractionResource;
interaction_id?: string;
status?: string;
metadata?: InteractionStreamMetadata;
error?: { message?: string; code?: string | number };
}
function emptyUsage(): Usage {
return {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
}
function getGoogleInteractionsState(
providerSessionState: Map<string, ProviderSessionState> | undefined,
create: boolean,
): GoogleInteractionsProviderSessionState | undefined {
if (!providerSessionState) return undefined;
const existing = providerSessionState.get(GOOGLE_INTERACTIONS_STATE_KEY) as
| GoogleInteractionsProviderSessionState
| undefined;
if (existing || !create) return existing;
const state: GoogleInteractionsProviderSessionState = { close: () => {} };
providerSessionState.set(GOOGLE_INTERACTIONS_STATE_KEY, state);
return state;
}
function interactionContentFromText(text: string): InteractionContent[] {
return text.length === 0 ? [] : [{ type: "text", text }];
}
function interactionContentFromParts(parts: readonly (TextContent | ImageContent)[]): InteractionContent[] {
const content: InteractionContent[] = [];
for (const part of parts) {
if (part.type === "text") {
if (part.text.length > 0) content.push({ type: "text", text: part.text });
} else {
content.push({ type: "image", data: part.data, mime_type: part.mimeType });
}
}
return content;
}
function userInputStepFromMessage(message: Extract<Message, { role: "user" | "developer" }>): InteractionUserInputStep {
const content =
typeof message.content === "string"
? interactionContentFromText(message.content)
: interactionContentFromParts(message.content);
return { type: "user_input", content };
}
function functionResultStepFromMessage(
message: Extract<Message, { role: "toolResult" }>,
): InteractionFunctionResultStep {
const result = interactionContentFromParts(message.content);
return {
type: "function_result",
name: message.toolName,
call_id: message.toolCallId,
result: result.length > 0 ? result : [{ type: "text", text: "" }],
...(message.isError ? { is_error: true } : {}),
};
}
function appendAssistantInteractionSteps(message: AssistantMessage, steps: InteractionInputStep[]): void {
let modelContent: InteractionContent[] = [];
const flushModelContent = (): void => {
if (modelContent.length === 0) return;
steps.push({ type: "model_output", content: modelContent });
modelContent = [];
};
for (const block of message.content) {
if (block.type === "text") {
if (block.text.length > 0) modelContent.push({ type: "text", text: block.text });
} else if (block.type === "toolCall") {
flushModelContent();
steps.push({ type: "function_call", id: block.id, name: block.name, arguments: block.arguments });
}
}
flushModelContent();
}
function interactionMessagesAfterAnchor(
messages: readonly Message[],
anchorIndex: number | undefined,
): readonly Message[] {
return anchorIndex === undefined ? messages : messages.slice(anchorIndex + 1);
}
function buildInteractionInput(context: Context, anchorIndex: number | undefined): InteractionInputStep[] {
const input: InteractionInputStep[] = [];
for (const message of interactionMessagesAfterAnchor(context.messages, anchorIndex)) {
if (message.role === "user" || message.role === "developer") {
const step = userInputStepFromMessage(message);
if (step.content.length > 0) input.push(step);
} else if (message.role === "toolResult") {
input.push(functionResultStepFromMessage(message));
} else if (anchorIndex === undefined) {
appendAssistantInteractionSteps(message, input);
}
}
return input.length > 0 ? input : [{ type: "user_input", content: [{ type: "text", text: "" }] }];
}
function toInteractionThinkingLevel(level: GoogleThinkingLevel): InteractionThinkingLevel | undefined {
switch (level) {
case "MINIMAL":
return "minimal";
case "LOW":
return "low";
case "MEDIUM":
return "medium";
case "HIGH":
return "high";
case "THINKING_LEVEL_UNSPECIFIED":
return undefined;
}
}
function buildInteractionGenerationConfig(options: GoogleOptions | undefined): InteractionGenerationConfig | undefined {
const config: InteractionGenerationConfig = {};
if (options?.temperature !== undefined) config.temperature = options.temperature;
if (options?.topP !== undefined) config.top_p = options.topP;
if (options?.topK !== undefined) config.top_k = options.topK;
if (options?.minP !== undefined) config.min_p = options.minP;
if (options?.presencePenalty !== undefined) config.presence_penalty = options.presencePenalty;
if (options?.frequencyPenalty !== undefined) config.frequency_penalty = options.frequencyPenalty;
if (options?.repetitionPenalty !== undefined) config.repetition_penalty = options.repetitionPenalty;
if (options?.maxTokens !== undefined) config.max_output_tokens = options.maxTokens;
if (options?.thinking?.level !== undefined) {
const thinkingLevel = toInteractionThinkingLevel(options.thinking.level);
if (thinkingLevel !== undefined) config.thinking_level = thinkingLevel;
} else if (options?.thinking?.budgetTokens !== undefined) {
config.thinking_budget = options.thinking.budgetTokens;
}
return Object.keys(config).length > 0 ? config : undefined;
}
function buildInteractionRequest(
model: GoogleInteractionsModel,
context: Context,
options: GoogleOptions | undefined,
anchor: InteractionAnchor,
): GoogleInteractionRequest {
const systemInstruction = normalizeSystemPrompts(context.systemPrompt).join("\n\n");
const generationConfig = buildInteractionGenerationConfig(options);
return {
model: model.id,
input: buildInteractionInput(context, anchor.messageIndex),
stream: true,
...(anchor.id !== undefined ? { previous_interaction_id: anchor.id } : {}),
...(systemInstruction.length > 0 ? { system_instruction: systemInstruction } : {}),
...(context.tools && context.tools.length > 0 ? { tools: convertTools(context.tools, model) } : {}),
...(options?.storeInteraction !== undefined ? { store: options.storeInteraction } : {}),
...(generationConfig !== undefined ? { generation_config: generationConfig } : {}),
...(shouldSendServiceTier(options?.serviceTier, model.provider) ? { service_tier: options?.serviceTier } : {}),
};
}
function applyInteractionUsage(
model: GoogleInteractionsModel,
output: AssistantMessage,
usage: InteractionUsage,
): void {
const thinkingTokens = usage.total_thought_tokens ?? 0;
output.usage = {
input: (usage.total_input_tokens ?? 0) - (usage.total_cached_tokens ?? 0),
output: (usage.total_output_tokens ?? 0) + thinkingTokens,
cacheRead: usage.total_cached_tokens ?? 0,
cacheWrite: 0,
totalTokens: usage.total_tokens ?? 0,
...(thinkingTokens > 0 ? { reasoningTokens: thinkingTokens } : {}),
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
calculateCost(model, output.usage);
}
function parseInteractionFunctionCall(value: unknown): InteractionFunctionCallStep | undefined {
if (!value || typeof value !== "object" || Array.isArray(value)) return undefined;
const record = value as Record<string, unknown>;
if (record.type !== "function_call") return undefined;
if (typeof record.id !== "string" || typeof record.name !== "string") return undefined;
const args = record.arguments;
return {
type: "function_call",
id: record.id,
name: record.name,
arguments: args && typeof args === "object" && !Array.isArray(args) ? { ...args } : {},
};
}
function pendingToolCallFromStep(call: InteractionFunctionCallStep): PendingInteractionToolCall {
return {
id: call.id,
name: call.name,
argumentsText: "",
argumentsObject: call.arguments,
};
}
function parseInteractionArguments(text: string): Record<string, unknown> {
if (text.trim().length === 0) return {};
try {
const parsed: unknown = JSON.parse(text);
if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) return { ...parsed };
} catch {
return {};
}
return {};
}
/** Provider-specific URL, headers, and fetch implementation for an Interactions request. */
export interface GoogleInteractionsPlan {
url: string;
headers: Record<string, string>;
fetch?: FetchImpl;
}
/**
* Streams Gemini Interactions API model-mode responses for direct Google and Vertex providers.
*
* `fallback`, when supplied, is the legacy `:streamGenerateContent` stream factory. It runs
* transparently — forwarding its events into this stream — when the Interactions attempt fails
* before any content is emitted with a signal that the model/endpoint does not support
* Interactions (HTTP 404/400). Provide it only for auto-selected Interactions requests so an
* explicit `useInteractionsApi: true` still surfaces failures.
*/
export function streamGoogleInteractions<T extends GoogleInteractionsApi>(args: {
model: Model<T>;
context: Context;
options: GoogleSharedStreamOptions | undefined;
api: T;
anchor: InteractionAnchor;
state: GoogleInteractionsProviderSessionState | undefined;
prepare: () => GoogleInteractionsPlan | Promise<GoogleInteractionsPlan>;
fallback?: () => AssistantMessageEventStream;
}): AssistantMessageEventStream {
const { model, context, options, anchor, state } = args;
const stream = new AssistantMessageEventStream();
const output: AssistantMessage = {
role: "assistant",
content: [],
api: args.api,
provider: model.provider,
model: model.id,
usage: emptyUsage(),
stopReason: "stop",
timestamp: Date.now(),
};
const storeInteraction = options?.storeInteraction !== false;
void (async () => {
let started = false;
let sawTerminal = false;
let currentTextBlock: TextContent | undefined;
let currentThinkingBlock: Extract<AssistantMessage["content"][number], { type: "thinking" }> | undefined;
let pendingThinkingSignature: string | undefined;
const stepKinds = new Map<number, string>();
const pendingToolCalls = new Map<number, PendingInteractionToolCall>();
const ensureStarted = (): void => {
if (started) return;
stream.push({ type: "start", partial: output });
started = true;
};
const endOpenBlocks = (): void => {
if (currentTextBlock) {
stream.push({
type: "text_end",
contentIndex: output.content.indexOf(currentTextBlock),
content: currentTextBlock.text,
partial: output,
});
currentTextBlock = undefined;
}
if (currentThinkingBlock) {
stream.push({
type: "thinking_end",
contentIndex: output.content.indexOf(currentThinkingBlock),
content: currentThinkingBlock.thinking,
partial: output,
});
currentThinkingBlock = undefined;
}
};
const emitText = (text: string): void => {
if (text.length === 0) return;
ensureStarted();
if (!currentTextBlock) {
if (currentThinkingBlock) endOpenBlocks();
currentTextBlock = { type: "text", text: "" };
output.content.push(currentTextBlock);
stream.push({ type: "text_start", contentIndex: output.content.length - 1, partial: output });
}
currentTextBlock.text += text;
stream.push({
type: "text_delta",
contentIndex: output.content.indexOf(currentTextBlock),
delta: text,
partial: output,
});
};
const applyThinkingSignature = (signature: string | undefined): void => {
if (!signature) return;
if (currentThinkingBlock) {
currentThinkingBlock.thinkingSignature = signature;
} else {
pendingThinkingSignature = signature;
}
};
const emitThinking = (text: string): void => {
if (text.length === 0) return;
ensureStarted();
if (!currentThinkingBlock) {
if (currentTextBlock) endOpenBlocks();
currentThinkingBlock = { type: "thinking", thinking: "", thinkingSignature: pendingThinkingSignature };
pendingThinkingSignature = undefined;
output.content.push(currentThinkingBlock);
stream.push({ type: "thinking_start", contentIndex: output.content.length - 1, partial: output });
}
currentThinkingBlock.thinking += text;
stream.push({
type: "thinking_delta",
contentIndex: output.content.indexOf(currentThinkingBlock),
delta: text,
partial: output,
});
};
const emitToolCall = (call: InteractionFunctionCallStep): void => {
ensureStarted();
endOpenBlocks();
const toolCall: ToolCall = {
type: "toolCall",
id: call.id,
name: call.name,
arguments: call.arguments,
};
output.content.push(toolCall);
const contentIndex = output.content.length - 1;
stream.push({ type: "toolcall_start", contentIndex, partial: output });
stream.push({
type: "toolcall_delta",
contentIndex,
delta: JSON.stringify(toolCall.arguments),
partial: output,
});
stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
};
const emitPendingToolCall = (pending: PendingInteractionToolCall): void => {
emitToolCall({
type: "function_call",
id: pending.id,
name: pending.name,
arguments:
pending.argumentsText.length > 0
? parseInteractionArguments(pending.argumentsText)
: pending.argumentsObject,
});
};
let prepared = false;
try {
const plan = await args.prepare();
prepared = true;
let requestBody: unknown = buildInteractionRequest(model, context, options, anchor);
const replacement = await options?.onPayload?.(requestBody, model);
if (replacement !== undefined) requestBody = replacement;
const response = await fetchWithRetry(() => plan.url, {
method: "POST",
headers: {
...plan.headers,
"Content-Type": "application/json",
Accept: "text/event-stream",
},
body: JSON.stringify(requestBody),
signal: options?.signal,
fetch: plan.fetch,
});
if (!response.ok) {
const errorText = await response.text().catch(() => "");
throw new AIError.GoogleApiError(
`Google Interactions API error (${response.status}): ${errorText}`,
response.status,
{
headers: response.headers,
},
);
}
if (!response.body) {
throw new AIError.ProviderResponseError("Google Interactions API returned an empty response body", {
provider: model.provider,
kind: "empty-body",
});
}
for await (const event of readSseJson<InteractionSseEvent>(response.body, options?.signal, sse =>
options?.onSseEvent?.({ event: sse.event, data: sse.data, raw: [...sse.raw] }, model),
)) {
if (event.error) {
throw new AIError.ProviderResponseError(event.error.message ?? "Google Interactions API stream error", {
provider: model.provider,
kind: "runtime",
});
}
if (event.metadata?.total_usage) applyInteractionUsage(model, output, event.metadata.total_usage);
if (event.event_type === "interaction.created") {
if (storeInteraction && event.interaction?.id) output.responseId = event.interaction.id;
} else if (event.event_type === "step.start" && event.index !== undefined && event.step) {
stepKinds.set(event.index, event.step.type);
const call = parseInteractionFunctionCall(event.step);
if (call) {
pendingToolCalls.set(event.index, pendingToolCallFromStep(call));
} else if (event.step.type === "thought") {
applyThinkingSignature(event.step.signature);
for (const item of event.step.summary ?? []) {
if (item.type === "text") emitThinking(item.text);
}
}
} else if (event.event_type === "step.delta" && event.index !== undefined && event.delta) {
const call = parseInteractionFunctionCall(event.delta);
if (call) {
pendingToolCalls.set(event.index, pendingToolCallFromStep(call));
} else if (event.delta.type === "text") {
if (stepKinds.get(event.index) === "thought") emitThinking(event.delta.text);
else emitText(event.delta.text);
} else if (event.delta.type === "thought_summary") {
if (event.delta.content?.type === "text") emitThinking(event.delta.content.text);
} else if (event.delta.type === "thought_signature") {
applyThinkingSignature(event.delta.signature);
} else if (event.delta.type === "arguments_delta") {
const pending = pendingToolCalls.get(event.index);
if (pending && event.delta.arguments) pending.argumentsText += event.delta.arguments;
}
} else if (event.event_type === "step.stop" && event.index !== undefined) {
const stepKind = stepKinds.get(event.index);
const pending = pendingToolCalls.get(event.index);
if (pending) {
emitPendingToolCall(pending);
pendingToolCalls.delete(event.index);
} else {
endOpenBlocks();
if (stepKind === "thought") pendingThinkingSignature = undefined;
}
stepKinds.delete(event.index);
} else if (event.event_type === "interaction.completed" || event.event_type === "interaction.complete") {
if (storeInteraction && event.interaction?.id) output.responseId = event.interaction.id;
if (event.interaction?.usage) applyInteractionUsage(model, output, event.interaction.usage);
for (const pending of pendingToolCalls.values()) emitPendingToolCall(pending);
pendingToolCalls.clear();
endOpenBlocks();
output.stopReason =
event.interaction?.status === "requires_action" ||
output.content.some(block => block.type === "toolCall")
? "toolUse"
: "stop";
if (storeInteraction && state) state.lastInteractionId = output.responseId;
sawTerminal = true;
ensureStarted();
stream.push({ type: "done", reason: output.stopReason, message: output });
}
}
if (!sawTerminal) {
throw new AIError.ProviderResponseError("Google Interactions API stream ended without a terminal event", {
provider: model.provider,
kind: "incomplete-stream",
});
}
} catch (error) {
// Auto-selected Interactions degrades to `:streamGenerateContent` when no content has
// streamed yet and the failure means Interactions can't serve this request — the bearer
// credential couldn't be resolved (prepare threw) or the model/endpoint rejected it
// (HTTP 404/400). Provider 401/403/429/5xx still surface. Mirrors the OpenAI Responses
// `previous_response_id` fallback.
const unsupported = error instanceof AIError.GoogleApiError && (error.status === 404 || error.status === 400);
if (!started && args.fallback && !options?.signal?.aborted && (!prepared || unsupported)) {
for await (const event of args.fallback()) stream.push(event);
return;
}
output.stopReason = options?.signal?.aborted ? "aborted" : "error";
output.errorMessage = error instanceof Error ? error.message : String(error);
stream.push({ type: "error", reason: output.stopReason, error: output });
}
})();
return stream;
}
function findAssistantInteractionAnchor(
context: Context,
interactionId: string,
provider: string,
): InteractionAnchor | undefined {
for (let index = context.messages.length - 1; index >= 0; index -= 1) {
const message = context.messages[index];
if (message?.role === "assistant" && message.provider === provider && message.responseId === interactionId) {
return { id: interactionId, messageIndex: index };
}
}
return undefined;
}
function latestAssistantInteractionAnchor(context: Context, provider: string): InteractionAnchor | undefined {
for (let index = context.messages.length - 1; index >= 0; index -= 1) {
const message = context.messages[index];
if (message?.role === "assistant" && message.provider === provider && message.responseId) {
return { id: message.responseId, messageIndex: index };
}
}
return undefined;
}
function resolveInteractionAnchor(
context: Context,
explicitPreviousInteractionId: string | undefined,
state: GoogleInteractionsProviderSessionState | undefined,
provider: string,
): InteractionAnchor {
if (explicitPreviousInteractionId !== undefined) {
return (
findAssistantInteractionAnchor(context, explicitPreviousInteractionId, provider) ?? {
id: explicitPreviousInteractionId,
}
);
}
const lineageAnchor = latestAssistantInteractionAnchor(context, provider);
if (lineageAnchor) return lineageAnchor;
if (state?.lastInteractionId)
return findAssistantInteractionAnchor(context, state.lastInteractionId, provider) ?? {};
return {};
}
/**
* Whether a model is served by the Gemini Interactions API. Interactions is a Gemini 3-era
* transport, so the catalog subset that supports it is Gemini 3.0+. Older Gemini and non-Gemini
* ids keep `:streamGenerateContent`, which covers the full catalog.
*/
export function modelSupportsInteractions(model: Pick<Model, "id">): boolean {
const parsed = parseGeminiModel(model.id);
return parsed !== null && parsed.version.major >= 3;
}
/**
* Resolves whether a Google provider call should use Interactions and which lineage anchor to send.
*
* Precedence: explicit `useInteractionsApi: false` always wins (force generateContent); otherwise
* Interactions engages when explicitly requested, when continuing a stored interaction
* (`previousInteractionId`/assistant lineage/session state), or when `autoEligible` (the
* zero-config default for the capable model subset on the official endpoint). `auto` flags the
* last case for the caller — it is the only mode that wires up the generateContent fallback.
*/
export function resolveInteractionDispatch(args: {
context: Context;
options: GoogleSharedStreamOptions | undefined;
provider: string;
autoEligible: boolean;
}): {
useInteractions: boolean;
auto: boolean;
anchor: InteractionAnchor;
state: GoogleInteractionsProviderSessionState | undefined;
} {
const explicitPreviousInteractionId = args.options?.previousInteractionId;
if (args.options?.storeInteraction === false && explicitPreviousInteractionId !== undefined) {
throw new AIError.ConfigurationError(
"Google Interactions API cannot combine storeInteraction:false with previousInteractionId.",
);
}
const explicitOptOut = args.options?.useInteractionsApi === false;
const explicitOptIn = args.options?.useInteractionsApi === true || explicitPreviousInteractionId !== undefined;
const storageEnabled = args.options?.storeInteraction !== false;
const existingState = storageEnabled
? getGoogleInteractionsState(args.options?.providerSessionState, false)
: undefined;
const anchor = explicitOptOut
? {}
: resolveInteractionAnchor(args.context, explicitPreviousInteractionId, existingState, args.provider);
const useInteractions = !explicitOptOut && (explicitOptIn || anchor.id !== undefined || args.autoEligible);
const auto = useInteractions && !explicitOptIn;
const interactionState =
useInteractions && storageEnabled
? getGoogleInteractionsState(
args.options?.providerSessionState,
args.options?.providerSessionState !== undefined,
)
: undefined;
return { useInteractions, auto, anchor, state: interactionState };
}
+74 -13
View File
@@ -14,6 +14,7 @@ import type {
FetchImpl,
ImageContent,
Model,
ServiceTier,
StopReason,
StreamOptions,
TextContent,
@@ -21,6 +22,7 @@ import type {
Tool,
ToolCall,
} from "../types";
import { shouldSendServiceTier } from "../types";
import { normalizeSystemPrompts } from "../utils";
import { AssistantMessageEventStream } from "../utils/event-stream";
import type { RawHttpRequestDump } from "../utils/http-inspector";
@@ -73,6 +75,21 @@ export interface GoogleSharedStreamOptions extends StreamOptions {
budgetTokens?: number;
level?: GoogleThinkingLevel;
};
/** Gemini/Vertex serving tier (`flex`/`priority`); other values are omitted. */
serviceTier?: ServiceTier;
/**
* Continues a Gemini Interactions API conversation from a stored interaction.
* When set on the direct Google provider, the request uses `/interactions`
* with `previous_interaction_id` instead of the legacy generateContent stream.
*/
previousInteractionId?: string;
/**
* Uses the Gemini Interactions API for direct Google requests, storing the
* returned interaction id on the assistant response for follow-up turns.
*/
useInteractionsApi?: boolean;
/** Overrides Interactions API request storage; default is the API default (`true`). */
storeInteraction?: boolean;
}
/**
@@ -126,11 +143,9 @@ function resolveThoughtSignature(isSameProviderAndModel: boolean, signature: str
return isSameProviderAndModel && isValidThoughtSignature(signature) ? signature : undefined;
}
/**
* Claude models via Google APIs require explicit tool call IDs in function calls/responses.
*/
export function requiresToolCallId(modelId: string): boolean {
return modelId.startsWith("claude-");
function supportsFunctionPartId<T extends GoogleApiType>(model: Model<T>): boolean {
if (model.api === "google-vertex") return false;
return model.id.startsWith("claude-") || (model.api === "google-generative-ai" && isGemini3Model(model.id));
}
function getGeminiMajorVersion(modelId: string): number | undefined {
@@ -156,8 +171,9 @@ function isGemini3Model(modelId: string): boolean {
*/
export function convertMessages<T extends GoogleApiType>(model: Model<T>, context: Context): Content[] {
const contents: Content[] = [];
const emittedToolCallNames = new Map<string, string>();
const normalizeToolCallId = (id: string): string => {
if (!requiresToolCallId(model.id)) return id;
return id.replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 64);
};
@@ -243,6 +259,7 @@ export function convertMessages<T extends GoogleApiType>(model: Model<T>, contex
});
}
} else if (block.type === "toolCall") {
emittedToolCallNames.set(block.id, block.name);
const thoughtSignature = resolveThoughtSignature(isSameProviderAndModel, block.thoughtSignature);
const effectiveSignature =
thoughtSignature || (isGemini3Model(model.id) ? SKIP_THOUGHT_SIGNATURE : undefined);
@@ -251,11 +268,11 @@ export function convertMessages<T extends GoogleApiType>(model: Model<T>, contex
functionCall: {
name: block.name,
args: block.arguments ?? {},
...(requiresToolCallId(model.id) ? { id: block.id } : {}),
...(supportsFunctionPartId(model) ? { id: block.id } : {}),
},
};
if (model.provider === "google-vertex" && part?.functionCall?.id) {
delete part.functionCall.id; // Vertex AI does not support 'id' in functionCall
delete part.functionCall.id; // Vertex AI GenerateContent rejects 'id' in functionCall parts.
}
if (effectiveSignature) {
part.thoughtSignature = effectiveSignature;
@@ -301,10 +318,11 @@ export function convertMessages<T extends GoogleApiType>(model: Model<T>, contex
},
}));
const includeId = requiresToolCallId(model.id);
const includeId = supportsFunctionPartId(model);
const emittedName = emittedToolCallNames.get(msg.toolCallId);
const functionResponsePart: Part = {
functionResponse: {
name: msg.toolName,
name: emittedName ?? msg.toolName,
response: msg.isError ? { error: responseValue } : { output: responseValue },
...(hasImages && modelSupportsMultimodalFunctionResponse && { parts: imageParts }),
...(includeId ? { id: msg.toolCallId } : {}),
@@ -312,7 +330,7 @@ export function convertMessages<T extends GoogleApiType>(model: Model<T>, contex
};
if (model.provider === "google-vertex" && functionResponsePart.functionResponse?.id) {
delete functionResponsePart.functionResponse.id; // Vertex AI does not support 'id' in functionResponse
delete functionResponsePart.functionResponse.id; // Vertex AI GenerateContent rejects 'id' in functionResponse parts.
}
// Cloud Code Assist API requires all function responses to be in a single user turn.
@@ -529,6 +547,24 @@ export function pushToolCallEvents(
* `text_start` / `thinking_start` event. `onBeforeStartEvent` lets the SSE consumer
* inject its `ensureStarted()` first-token side effect into the canonical event order.
*/
export function startTextOrThinkingBlock(
isThinking: true,
output: AssistantMessage,
stream: AssistantMessageEventStream,
onBeforeStartEvent?: () => void,
): ThinkingContent;
export function startTextOrThinkingBlock(
isThinking: false,
output: AssistantMessage,
stream: AssistantMessageEventStream,
onBeforeStartEvent?: () => void,
): TextContent;
export function startTextOrThinkingBlock(
isThinking: boolean,
output: AssistantMessage,
stream: AssistantMessageEventStream,
onBeforeStartEvent?: () => void,
): TextContent | ThinkingContent;
export function startTextOrThinkingBlock(
isThinking: boolean,
output: AssistantMessage,
@@ -791,6 +827,14 @@ export function buildGoogleGenerateContentParams<T extends "google-generative-ai
...(context.tools && context.tools.length > 0 && { tools: convertTools(context.tools, model) }),
};
// Gemini API (google-generative-ai) reads the tier from the request body;
// Vertex AI ignores a body field and requires the
// `X-Vertex-AI-LLM-Shared-Request-Type` header instead (added in
// streamGoogleVertex), so only emit the body field for the direct API.
if (model.provider === "google" && shouldSendServiceTier(options.serviceTier, model.provider)) {
config.serviceTier = options.serviceTier;
}
if (context.tools && context.tools.length > 0 && options.toolChoice) {
const choice = options.toolChoice;
if (typeof choice === "string") {
@@ -852,6 +896,8 @@ export interface GoogleGenAIRequestPlan {
url: string;
headers: Record<string, string>;
fetch?: FetchImpl;
/** Optional URL retried once when {@link url} returns 404 (regional Vertex endpoint missing a global-only model). */
fallbackUrl?: string;
}
export function streamGoogleGenAI<T extends "google-generative-ai" | "google-vertex">(args: {
@@ -906,8 +952,8 @@ export function streamGoogleGenAI<T extends "google-generative-ai" | "google-ver
const bodyJson = JSON.stringify(paramsToWireBody(params));
const fetchImpl = plan.fetch ?? options?.fetch ?? (globalThis.fetch.bind(globalThis) as FetchImpl);
const openStream = async (): Promise<ReadableStream<Uint8Array>> => {
const response = await fetchImpl(plan.url, {
const openStreamAt = async (requestUrl: string): Promise<ReadableStream<Uint8Array>> => {
const response = await fetchImpl(requestUrl, {
method: "POST",
headers: { ...plan.headers, "Content-Type": "application/json", Accept: "text/event-stream" },
body: bodyJson,
@@ -929,6 +975,20 @@ export function streamGoogleGenAI<T extends "google-generative-ai" | "google-ver
}
return response.body as ReadableStream<Uint8Array>;
};
// A regional Vertex endpoint 404s for models published only on the
// global endpoint; retry global once so a stale/ambient region never
// breaks a request that worked before regional routing existed.
const openStream = async (): Promise<ReadableStream<Uint8Array>> => {
if (!plan.fallbackUrl) return openStreamAt(plan.url);
try {
return await openStreamAt(plan.url);
} catch (error) {
if (error instanceof AIError.GoogleApiError && error.status === 404) {
return openStreamAt(plan.fallbackUrl);
}
throw error;
}
};
let body = await openStream();
stream.push({ type: "start", partial: output });
@@ -1007,6 +1067,7 @@ function paramsToWireBody(params: GenerateContentParameters): Record<string, unk
if (config.toolConfig !== undefined) body.toolConfig = config.toolConfig;
if (config.safetySettings !== undefined) body.safetySettings = config.safetySettings;
if (config.cachedContent !== undefined) body.cachedContent = config.cachedContent;
if (config.serviceTier !== undefined) body.serviceTier = config.serviceTier;
const gen: Record<string, unknown> = {};
if (config.temperature !== undefined) gen.temperature = config.temperature;
+5 -1
View File
@@ -9,7 +9,6 @@
* - `POST {generativelanguage,aiplatform}.googleapis.com/.../models/{model}:streamGenerateContent?alt=sse`
* - The Cloud Code Assist endpoint used by `google-gemini-cli.ts`
*/
/** Mirror of `@google/genai`'s `FinishReason` string enum. */
export type FinishReason =
| "FINISH_REASON_UNSPECIFIED"
@@ -131,6 +130,11 @@ export interface GenerateContentConfig {
safetySettings?: Array<Record<string, unknown>>;
cachedContent?: string;
thinkingConfig?: ThinkingConfig;
/**
* Gemini/Vertex serving tier. Serialized to the request body root as
* `serviceTier` (camelCase) by the transformer in `google-shared.ts`.
*/
serviceTier?: "auto" | "default" | "flex" | "scale" | "priority";
abortSignal?: AbortSignal;
}
+117 -26
View File
@@ -2,7 +2,13 @@ import { $env } from "@oh-my-pi/pi-utils";
import * as AIError from "../error";
import type { Context, Model, StreamFunction } from "../types";
import type { AssistantMessageEventStream } from "../utils/event-stream";
import { getVertexAccessToken } from "./google-auth";
import { getVertexAccessToken, hasVertexBearerCredentialsHint } from "./google-auth";
import {
type GoogleInteractionsPlan,
modelSupportsInteractions,
resolveInteractionDispatch,
streamGoogleInteractions,
} from "./google-interactions";
import {
buildGoogleGenerateContentParams,
type GoogleGenAIRequestPlan,
@@ -16,48 +22,128 @@ export interface GoogleVertexOptions extends GoogleSharedStreamOptions {
}
const API_VERSION = "v1";
const INTERACTIONS_API_VERSION = "v1beta1";
const INTERACTIONS_API_REVISION = "2026-05-20";
export const streamGoogleVertex: StreamFunction<"google-vertex"> = (
model: Model<"google-vertex">,
context: Context,
options?: GoogleVertexOptions,
): AssistantMessageEventStream =>
streamGoogleGenAI({
model,
options,
api: "google-vertex",
retainTextSignature: true,
prepare: async (): Promise<GoogleGenAIRequestPlan> => {
const apiKey = resolveApiKey(options);
const params = buildGoogleGenerateContentParams(model, context, options ?? {});
const baseHeaders: Record<string, string> = {
...(model.headers ?? {}),
...(options?.headers ?? {}),
};
): AssistantMessageEventStream => {
const runGenerateContent = (): AssistantMessageEventStream =>
streamGoogleGenAI({
model,
options,
api: "google-vertex",
retainTextSignature: true,
prepare: async (): Promise<GoogleGenAIRequestPlan> => {
const apiKey = resolveApiKey(options);
const params = buildGoogleGenerateContentParams(model, context, options ?? {});
params.config ||= {};
if (!params.config.safetySettings) {
params.config.safetySettings = [
{
category: "HARM_CATEGORY_HATE_SPEECH",
threshold: "OFF",
},
{
category: "HARM_CATEGORY_DANGEROUS_CONTENT",
threshold: "OFF",
},
{
category: "HARM_CATEGORY_SEXUALLY_EXPLICIT",
threshold: "OFF",
},
{
category: "HARM_CATEGORY_HARASSMENT",
threshold: "OFF",
},
];
}
const baseHeaders: Record<string, string> = {
...(model.headers ?? {}),
...(options?.headers ?? {}),
};
// Vertex AI ignores a `serviceTier` request-body field (unlike the direct
// Gemini API); priority must travel as a request header. Only `priority`
// has a documented Vertex request control — `flex` has none, so it's a no-op.
if (options?.serviceTier === "priority") {
baseHeaders["X-Vertex-AI-LLM-Shared-Request-Type"] = "priority";
}
if (apiKey) {
const url = `https://aiplatform.googleapis.com/${API_VERSION}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`;
if (apiKey) {
// Explicit `location` is a deliberate residency choice: honor it and let
// a 404 surface. An ambient env-derived region falls back to the global
// endpoint so a stray GOOGLE_*_LOCATION never breaks a previously-working
// global-only request.
const explicitLocation = options?.location;
const location = explicitLocation ?? resolveAmbientLocation() ?? "global";
const host = resolveEndpointHost(location);
const path = `${API_VERSION}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`;
const useGlobalFallback = !explicitLocation && host !== "aiplatform.googleapis.com";
return {
params,
url: `https://${host}/${path}`,
fallbackUrl: useGlobalFallback ? `https://aiplatform.googleapis.com/${path}` : undefined,
headers: {
...baseHeaders,
"x-goog-api-key": apiKey,
},
fetch: options?.fetch,
};
}
const project = resolveProject(options);
const location = resolveLocation(options);
const accessToken = await getVertexAccessToken({ signal: options?.signal, fetch: options?.fetch });
const host = resolveEndpointHost(location);
const url = `https://${host}/${API_VERSION}/projects/${project}/locations/${location}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`;
return {
params,
url,
headers: { ...baseHeaders, "x-goog-api-key": apiKey },
headers: { ...baseHeaders, Authorization: `Bearer ${accessToken}` },
fetch: options?.fetch,
};
}
},
});
// Default Gemini 3+ onto Interactions whenever a bearer credential source exists (ADC file,
// `GOOGLE_APPLICATION_CREDENTIALS`, or an explicit access-token env). Interactions needs bearer
// auth, so express API-key-only setups stay on generateContent — and an express key, when
// present, still serves the generateContent fallback. Interactions always targets the official
// global `aiplatform` host; the fallback also recovers ids the endpoint rejects.
const { useInteractions, auto, anchor, state } = resolveInteractionDispatch({
context,
options,
provider: model.provider,
autoEligible: modelSupportsInteractions(model) && hasVertexBearerCredentialsHint(),
});
if (!useInteractions) return runGenerateContent();
return streamGoogleInteractions({
model,
context,
options,
api: "google-vertex",
anchor,
state,
prepare: async (): Promise<GoogleInteractionsPlan> => {
const project = resolveProject(options);
const location = resolveLocation(options);
const accessToken = await getVertexAccessToken({ signal: options?.signal, fetch: options?.fetch });
const host = resolveEndpointHost(location);
const url = `https://${host}/${API_VERSION}/projects/${project}/locations/${location}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`;
return {
params,
url,
headers: { ...baseHeaders, Authorization: `Bearer ${accessToken}` },
url: `https://aiplatform.googleapis.com/${INTERACTIONS_API_VERSION}/projects/${project}/locations/global/interactions`,
headers: {
...(model.headers ?? {}),
...(options?.headers ?? {}),
Authorization: `Bearer ${accessToken}`,
"Api-Revision": INTERACTIONS_API_REVISION,
},
fetch: options?.fetch,
};
},
fallback: auto ? runGenerateContent : undefined,
});
};
function resolveApiKey(options?: GoogleVertexOptions): string | undefined {
// options.apiKey may contain sentinel values like "<authenticated>" or "N/A"
@@ -80,9 +166,14 @@ function resolveProject(options?: GoogleVertexOptions): string {
function resolveEndpointHost(location: string): string {
return location === "global" ? "aiplatform.googleapis.com" : `${location}-aiplatform.googleapis.com`;
}
function resolveAmbientLocation(): string | undefined {
return $env.GOOGLE_VERTEX_LOCATION || $env.GOOGLE_CLOUD_LOCATION || $env.VERTEX_LOCATION || undefined;
}
function resolveOptionalLocation(options?: GoogleVertexOptions): string | undefined {
return options?.location || resolveAmbientLocation();
}
function resolveLocation(options?: GoogleVertexOptions): string {
const location =
options?.location || $env.GOOGLE_VERTEX_LOCATION || $env.GOOGLE_CLOUD_LOCATION || $env.VERTEX_LOCATION;
const location = resolveOptionalLocation(options);
if (!location) {
throw new AIError.ConfigurationError(
"Vertex AI requires a location. Set GOOGLE_VERTEX_LOCATION/GOOGLE_CLOUD_LOCATION/VERTEX_LOCATION or pass location in options.",
+61 -19
View File
@@ -2,6 +2,7 @@ import * as AIError from "../error";
import { getEnvApiKey } from "../stream";
import type { Context, Model, StreamFunction } from "../types";
import type { AssistantMessageEventStream } from "../utils/event-stream";
import { modelSupportsInteractions, resolveInteractionDispatch, streamGoogleInteractions } from "./google-interactions";
import {
buildGoogleGenerateContentParams,
type GoogleGenAIRequestPlan,
@@ -17,29 +18,70 @@ export const streamGoogle: StreamFunction<"google-generative-ai"> = (
model: Model<"google-generative-ai">,
context: Context,
options?: GoogleOptions,
): AssistantMessageEventStream =>
streamGoogleGenAI({
): AssistantMessageEventStream => {
const apiKey = options?.apiKey || getEnvApiKey(model.provider);
if (!apiKey) {
throw new AIError.MissingApiKeyError(
undefined,
"Google Generative AI requires an API key (GEMINI_API_KEY or options.apiKey).",
);
}
const runGenerateContent = (): AssistantMessageEventStream =>
streamGoogleGenAI({
model,
options,
api: "google-generative-ai",
prepare: (): GoogleGenAIRequestPlan => {
const params = buildGoogleGenerateContentParams(model, context, options ?? {});
// `model.baseUrl` already includes the API version segment when set (mirrors the
// `apiVersion: ""` reset that the SDK relied on for custom base URLs).
const base = model.baseUrl?.trim() || DEFAULT_GENERATIVE_LANGUAGE_BASE;
const url = `${base}/models/${model.id}:streamGenerateContent?alt=sse`;
const headers: Record<string, string> = {
"x-goog-api-key": apiKey,
...(model.headers ?? {}),
...(options?.headers ?? {}),
};
return { params, url, headers, fetch: options?.fetch };
},
});
// Default Gemini 3+ on the official endpoint onto Interactions (custom proxy base URLs keep
// generateContent, which serves the full catalog). The fallback recovers ids the endpoint rejects.
const trimmedBase = model.baseUrl?.trim();
let officialEndpoint = !trimmedBase;
if (trimmedBase) {
try {
officialEndpoint = new URL(trimmedBase).hostname === "generativelanguage.googleapis.com";
} catch {
officialEndpoint = false;
}
}
const { useInteractions, auto, anchor, state } = resolveInteractionDispatch({
context,
options,
provider: model.provider,
autoEligible: officialEndpoint && modelSupportsInteractions(model),
});
if (!useInteractions) return runGenerateContent();
return streamGoogleInteractions({
model,
context,
options,
api: "google-generative-ai",
prepare: (): GoogleGenAIRequestPlan => {
const apiKey = options?.apiKey || getEnvApiKey(model.provider);
if (!apiKey) {
throw new AIError.MissingApiKeyError(
undefined,
"Google Generative AI requires an API key (GEMINI_API_KEY or options.apiKey).",
);
}
const params = buildGoogleGenerateContentParams(model, context, options ?? {});
// `model.baseUrl` already includes the API version segment when set (mirrors the
// `apiVersion: ""` reset that the SDK relied on for custom base URLs).
const base = model.baseUrl?.trim() || DEFAULT_GENERATIVE_LANGUAGE_BASE;
const url = `${base}/models/${model.id}:streamGenerateContent?alt=sse`;
const headers: Record<string, string> = {
anchor,
state,
prepare: () => ({
url: `${trimmedBase || DEFAULT_GENERATIVE_LANGUAGE_BASE}/interactions`,
headers: {
"x-goog-api-key": apiKey,
...(model.headers ?? {}),
...(options?.headers ?? {}),
};
return { params, url, headers, fetch: options?.fetch };
},
},
fetch: options?.fetch,
}),
fallback: auto ? runGenerateContent : undefined,
});
};
+6
View File
@@ -323,6 +323,10 @@ function createChatBody(model: Model<"ollama-chat">, context: Context, options:
};
}
function shouldRetryOllamaResponse(response: Response, bodyText: string): boolean {
return response.status < 500 || !AIError.LLAMA_CPP_TOOL_CALL_PARSE_PATTERN.test(bodyText);
}
async function captureHttpErrorResponse(response: Response): Promise<CapturedHttpErrorResponse> {
let bodyText: string | undefined;
let bodyJson: unknown;
@@ -598,6 +602,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
body: JSON.stringify(body),
signal: watchdog.signal,
defaultDelayMs: OLLAMA_RETRY_DELAYS_MS,
shouldRetryResponse: shouldRetryOllamaResponse,
fetch: options.fetch,
timeout: false,
});
@@ -747,6 +752,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
}
const result = await AIError.finalize(error, {
api: model.api,
provider: model.provider,
signal: options.signal,
rawRequestDump,
capturedErrorResponse,
@@ -13,7 +13,7 @@ import type {
Context,
ImageContent,
Message,
ResolvedServiceTier,
ServiceTier,
StopReason,
TextContent,
Tool,
@@ -38,7 +38,7 @@ function isReasoningEffort(value: unknown): value is ReasoningEffort {
return value === "minimal" || value === "low" || value === "medium" || value === "high" || value === "xhigh";
}
function isServiceTier(value: unknown): value is ResolvedServiceTier {
function isServiceTier(value: unknown): value is ServiceTier {
return value === "auto" || value === "default" || value === "flex" || value === "scale" || value === "priority";
}
@@ -104,7 +104,7 @@ import { transformMessages } from "./transform-messages";
export interface OpenAICodexResponsesOptions extends StreamOptions {
reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh";
reasoningSummary?: "auto" | "concise" | "detailed" | null;
/** `reasoning.context` replay scope; defaults to `all_turns` for every Codex request when unset. */
/** `reasoning.context` replay scope; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */
reasoningContext?: CodexReasoningContext;
textVerbosity?: "low" | "medium" | "high";
include?: string[];
@@ -1,4 +1,5 @@
import type { Effort } from "@oh-my-pi/pi-catalog/effort";
import { supportsAllTurnsReasoningContext } from "@oh-my-pi/pi-catalog/identity";
import { requireSupportedEffort } from "@oh-my-pi/pi-catalog/model-thinking";
import type { Api, Model } from "../../types";
@@ -14,7 +15,7 @@ export interface ReasoningConfig {
export interface CodexRequestOptions {
reasoningEffort?: ReasoningConfig["effort"];
reasoningSummary?: ReasoningConfig["summary"] | null;
/** Explicit `reasoning.context` override; defaults to `all_turns` for every Codex request when unset. */
/** Explicit `reasoning.context` override; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */
reasoningContext?: CodexReasoningContext;
textVerbosity?: "low" | "medium" | "high";
include?: string[];
@@ -254,9 +255,21 @@ export async function transformRequestBody(
...body.reasoning,
...reasoningConfig,
};
// Default reasoning replay to `all_turns` for every Codex request,
// mirroring codex-rs; an explicit `reasoningContext` overrides it.
body.reasoning.context = options.reasoningContext ?? "all_turns";
// Default reasoning replay to `all_turns`, mirroring codex-rs; an
// explicit `reasoningContext` overrides the default. The `all_turns`
// value is only accepted from gpt-5.4 onward — earlier Codex ids
// (gpt-5.1-codex, gpt-5.3-codex, gpt-5.3-codex-spark) reject it with
// "Unsupported value: 'all_turns' is not supported with this model".
// For those, drop `context` so the server applies its `current_turn`
// default. The version gate is authoritative: even an explicit
// `all_turns` override is suppressed on unsupported models, while
// `current_turn`/`auto` (universally supported) always pass through.
const context = options.reasoningContext ?? "all_turns";
if (context === "all_turns" && !supportsAllTurnsReasoningContext(model.id)) {
delete body.reasoning.context;
} else {
body.reasoning.context = context;
}
} else {
delete body.reasoning;
}
+5 -11
View File
@@ -39,8 +39,6 @@ import {
type Model,
OPENAI_MAX_OUTPUT_TOKENS,
type Provider,
type ResolvedServiceTier,
resolveServiceTier,
type ServiceTier,
type StopReason,
type StreamOptions,
@@ -269,14 +267,13 @@ export function resolveOpenAIRequestSetup(
}
export function applyOpenAIServiceTier(
params: { service_tier?: ResolvedServiceTier | "auto" | "default" | null | undefined },
params: { service_tier?: ServiceTier | null | undefined },
serviceTier: ServiceTier | null | undefined,
provider: Provider | undefined,
): void {
if (!shouldSendServiceTier(serviceTier, provider)) return;
const resolved = resolveServiceTier(serviceTier, provider);
if (resolved === "flex" || resolved === "scale" || resolved === "priority") {
params.service_tier = resolved;
if (serviceTier === "flex" || serviceTier === "scale" || serviceTier === "priority") {
params.service_tier = serviceTier;
}
}
@@ -315,10 +312,7 @@ export function applyOpenAIResponsesServiceTierCost(
// The response echo is authoritative when present (OpenAI may downgrade a
// requested priority/flex turn to default under load); only fall back to the
// requested tier when the response omits the echo entirely.
const served =
typeof responseServiceTier === "string"
? responseServiceTier
: resolveServiceTier(requestServiceTier, model.provider);
const served = typeof responseServiceTier === "string" ? responseServiceTier : (requestServiceTier ?? undefined);
const multiplier = getOpenAIResponsesServiceTierCostMultiplier(served);
if (multiplier === 1) return;
usage.cost.input *= multiplier;
@@ -623,7 +617,7 @@ export type OpenAICompletionsParams = Omit<ChatCompletionCreateParamsStreaming,
chat_template_kwargs?: { enable_thinking?: boolean; preserve_thinking?: boolean };
reasoning?: { effort?: string } | { enabled: false };
reasoning_effort?: string | null;
service_tier?: ResolvedServiceTier;
service_tier?: ServiceTier;
tool_stream?: boolean;
provider?: OpenAICompat["openRouterRouting"];
providerOptions?: { gateway?: { only?: string[]; order?: string[] } };
@@ -26,6 +26,20 @@ export interface OAuthCallbackFlowOptions {
callbackHostname?: string;
/** Exact redirect URI advertised to the provider; disables port fallback. */
redirectUri?: string;
/**
* Whether the flow may bind to a random port when {@link preferredPort} is
* unavailable. Defaults to `true` so historical AI-provider flows (which
* pick uncommon ports and tolerate any loopback callback) keep working.
*
* Set to `false` for providers that validate the redirect URI against a
* registered callback — silently advertising a random-port URI would be
* rejected by the authorization server, leaving the browser on an opaque
* 500 page and the local callback waiting until the 5-minute timeout fires.
* With fallback disabled, {@link OAuthCallbackFlow.login} throws a
* {@link AIError.ConfigurationError} immediately so the caller can surface
* an actionable message before opening the browser.
*/
allowPortFallback?: boolean;
}
/**
@@ -37,6 +51,7 @@ export abstract class OAuthCallbackFlow {
callbackPath: string;
callbackHostname: string;
redirectUri?: string;
allowPortFallback: boolean;
#callbackResolve?: (result: CallbackResult) => void;
#callbackReject?: (error: string) => void;
@@ -50,6 +65,7 @@ export abstract class OAuthCallbackFlow {
this.preferredPort = preferredPortOrOptions;
this.callbackPath = callbackPath;
this.callbackHostname = DEFAULT_HOSTNAME;
this.allowPortFallback = true;
return;
}
@@ -57,6 +73,7 @@ export abstract class OAuthCallbackFlow {
this.callbackPath = preferredPortOrOptions.callbackPath ?? CALLBACK_PATH;
this.callbackHostname = preferredPortOrOptions.callbackHostname ?? DEFAULT_HOSTNAME;
this.redirectUri = preferredPortOrOptions.redirectUri;
this.allowPortFallback = preferredPortOrOptions.allowPortFallback ?? true;
}
/**
@@ -87,18 +104,29 @@ export abstract class OAuthCallbackFlow {
.join("");
}
#loginCancelledError(): AIError.LoginCancelledError {
return new AIError.LoginCancelledError(`OAuth callback cancelled: ${this.ctrl.signal?.reason}`);
}
#throwIfCancelled(): void {
if (this.ctrl.signal?.aborted) throw this.#loginCancelledError();
}
/**
* Execute the OAuth login flow.
*/
async login(): Promise<OAuthCredentials> {
const state = this.generateState();
this.#throwIfCancelled();
// Start callback server first to get actual redirect URI
const { server, redirectUri } = await this.#startCallbackServer(state);
try {
this.#throwIfCancelled();
// Generate auth URL with the ACTUAL redirect URI (may differ from expected if port was busy)
const { url: authUrl, instructions } = await this.generateAuthUrl(state, redirectUri);
this.#throwIfCancelled();
// Notify controller that auth is ready
this.ctrl.onAuth?.({ url: authUrl, instructions });
@@ -106,6 +134,7 @@ export abstract class OAuthCallbackFlow {
// Wait for callback or manual input
const { code } = await this.#waitForCallback(state);
this.#throwIfCancelled();
this.ctrl.onProgress?.("Exchanging authorization code for tokens...");
@@ -126,10 +155,17 @@ export abstract class OAuthCallbackFlow {
}
const redirectUri = `http://${this.callbackHostname}:${this.preferredPort}${this.callbackPath}`;
return { server, redirectUri };
} catch {
} catch (cause) {
if (this.redirectUri) {
throw new AIError.ConfigurationError(
`OAuth callback port ${this.preferredPort} unavailable; cannot fall back to a random port when oauth.redirectUri is set`,
`OAuth callback port ${this.preferredPort} is in use, but oauth.redirectUri (${this.redirectUri}) requires this exact port. Free port ${this.preferredPort} (e.g. stop the process bound to it) and retry, or change oauth.redirectUri to point at an available port.`,
{ cause },
);
}
if (!this.allowPortFallback) {
throw new AIError.ConfigurationError(
`OAuth callback port ${this.preferredPort} is in use. The OAuth provider validates redirect URIs against its registered callback, so falling back to a random port would be rejected. Free port ${this.preferredPort} (e.g. stop the process bound to it) and retry, or set oauth.callbackPort/oauth.redirectUri to a port the provider has registered.`,
{ cause },
);
}
const server = this.#createServer(0, expectedState);
@@ -208,17 +244,18 @@ export abstract class OAuthCallbackFlow {
#waitForCallback(expectedState: string): Promise<CallbackResult> {
const timeoutSignal = AbortSignal.timeout(DEFAULT_TIMEOUT);
const signal = this.ctrl.signal ? AbortSignal.any([this.ctrl.signal, timeoutSignal]) : timeoutSignal;
if (signal.aborted) return Promise.reject(this.#loginCancelledError());
const callbackPromise = new Promise<CallbackResult>((resolve, reject) => {
this.#callbackResolve = resolve;
this.#callbackReject = reject;
const callback = Promise.withResolvers<CallbackResult>();
this.#callbackResolve = callback.resolve;
this.#callbackReject = callback.reject;
signal.addEventListener("abort", () => {
this.#callbackResolve = undefined;
this.#callbackReject = undefined;
reject(new AIError.LoginCancelledError(`OAuth callback cancelled: ${signal.reason}`));
});
signal.addEventListener("abort", () => {
this.#callbackResolve = undefined;
this.#callbackReject = undefined;
callback.reject(new AIError.LoginCancelledError(`OAuth callback cancelled: ${signal.reason}`));
});
const callbackPromise = callback.promise;
// Manual input race (if supported)
if (this.ctrl.onManualCodeInput) {
+21 -6
View File
@@ -71,6 +71,7 @@ import type {
ToolChoice,
} from "./types";
import { AssistantMessageEventStream } from "./utils/event-stream";
import { wrapLeakedThinkingStream } from "./utils/leaked-thinking-stream";
import { wrapFetchForProxy } from "./utils/proxy";
import { withRequestDebugFetch } from "./utils/request-debug";
import { withGeminiThinkingLoopGuard } from "./utils/thinking-loop";
@@ -499,8 +500,11 @@ function withProviderInFlightLimit<TOptions extends Pick<StreamOptions, "signal"
options: TOptions | undefined,
dispatch: () => AssistantMessageEventStream,
): AssistantMessageEventStream {
// Leaked-thinking healing folds in here — the one shared provider-dispatch
// chokepoint — so the loop guard (which wraps this) sees healed events and all
// six provider exits are covered by one wrap. Healing is idempotent.
const limit = resolveProviderInFlightLimit(model.provider, options);
if (limit === undefined) return dispatch();
if (limit === undefined) return wrapLeakedThinkingStream(dispatch());
const outer = new AssistantMessageEventStream();
void (async () => {
@@ -520,7 +524,7 @@ function withProviderInFlightLimit<TOptions extends Pick<StreamOptions, "signal"
if (options?.signal?.aborted) {
throw options.signal.reason ?? new AIError.AbortError("Provider request aborted before dispatch");
}
const inner = dispatch();
const inner = wrapLeakedThinkingStream(dispatch());
try {
for await (const event of inner) {
outer.push(event);
@@ -1105,10 +1109,13 @@ export function streamSimple<TApi extends Api>(
// GitLab Duo Workflow - IDE workflow protocol + WebSocket action bridge
if (model.api === "gitlab-duo-agent") {
return streamGitLabDuoWorkflow(model as Model<"gitlab-duo-agent">, context, {
...requestOptions,
apiKey,
});
// Does not route through withProviderInFlightLimit, so heal explicitly.
return wrapLeakedThinkingStream(
streamGitLabDuoWorkflow(model as Model<"gitlab-duo-agent">, context, {
...requestOptions,
apiKey,
}),
);
}
// Kimi Code - route to dedicated handler that wraps OpenAI or Anthropic API
@@ -1355,6 +1362,9 @@ function mapOptionsForApi<TApi extends Api>(
streamFirstEventTimeoutMs: options?.streamFirstEventTimeoutMs,
streamIdleTimeoutMs: options?.streamIdleTimeoutMs,
providerSessionState: options?.providerSessionState,
useInteractionsApi: options?.useInteractionsApi,
storeInteraction: options?.storeInteraction,
previousInteractionId: options?.previousInteractionId,
maxInFlightRequests: options?.maxInFlightRequests,
onPayload: options?.onPayload,
onResponse: options?.onResponse,
@@ -1565,6 +1575,7 @@ function mapOptionsForApi<TApi extends Api>(
if (!reasoning || !model.reasoning) {
return castApi<"google-generative-ai">({
...base,
serviceTier: options?.serviceTier,
thinking: { enabled: false },
toolChoice: mapGoogleToolChoice(options?.toolChoice),
});
@@ -1578,6 +1589,7 @@ function mapOptionsForApi<TApi extends Api>(
if (googleModel.thinking?.mode === "google-level") {
return castApi<"google-generative-ai">({
...base,
serviceTier: options?.serviceTier,
thinking: {
enabled: true,
level: mapEffortToGoogleThinkingLevel(effort),
@@ -1661,6 +1673,7 @@ function mapOptionsForApi<TApi extends Api>(
if (!reasoning || !model.reasoning) {
return castApi<"google-vertex">({
...base,
serviceTier: options?.serviceTier,
thinking: { enabled: false },
toolChoice: mapGoogleToolChoice(options?.toolChoice),
});
@@ -1673,6 +1686,7 @@ function mapOptionsForApi<TApi extends Api>(
if (geminiModel.thinking?.mode === "google-level") {
return castApi<"google-vertex">({
...base,
serviceTier: options?.serviceTier,
thinking: {
enabled: true,
level: mapEffortToGoogleThinkingLevel(effort),
@@ -1683,6 +1697,7 @@ function mapOptionsForApi<TApi extends Api>(
return castApi<"google-vertex">({
...base,
serviceTier: options?.serviceTier,
thinking: {
enabled: true,
budgetTokens: getGoogleBudget(geminiModel, effort, options?.thinkingBudgets),
+164 -51
View File
@@ -105,84 +105,181 @@ export type ToolChoice =
export type CacheRetention = "none" | "short" | "long";
/**
* Service tier hint for processing priority / cost control.
* Service tier hint for processing priority / cost control. These are the
* values providers consume on the wire:
*
* The unscoped values (`"auto"`, `"default"`, `"flex"`, `"scale"`,
* `"priority"`) are passed through to providers that understand them
* (OpenAI's `service_tier` field directly; Anthropic translates
* `"priority"` into `speed: "fast"` on supported Opus models).
* - OpenAI / OpenAI-Codex: sent verbatim as the `service_tier` field
* (`flex`/`scale`/`priority`).
* - Google (Gemini API + Vertex AI): sent as the top-level `serviceTier`
* field (`flex`/`priority`).
* - OpenRouter: passed through as `service_tier`; OpenRouter realizes it for
* the OpenAI- and Google-family upstreams it supports and ignores it
* otherwise.
* - Direct Anthropic: `"priority"` is translated into `speed: "fast"` plus the
* fast-mode beta on supported Opus models. Other tiers are ignored.
*
* The scoped values target a specific provider family and behave as the
* unscoped value on the matching provider, or `undefined` everywhere else.
* They let users opt into priority on one family without paying premium
* costs on the other when switching models mid-session.
*
* - `"openai-only"` → `"priority"` on `openai` and `openai-codex`; ignored elsewhere.
* - `"claude-only"` → `"priority"` on direct `anthropic` (not Bedrock/Vertex Claude).
* Per-family scoping is expressed by {@link ServiceTierByFamily}, not by
* scoped sentinel values — see {@link serviceTierFamily}.
*/
export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority" | "openai-only" | "claude-only";
export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority";
/** Resolved tier — one of the values that providers actually consume on the wire. */
export type ResolvedServiceTier = Exclude<ServiceTier, "openai-only" | "claude-only">;
/** Provider families that expose an independent service-tier knob. */
export type ServiceTierFamily = "openai" | "anthropic" | "google";
/**
* Resolves a possibly scoped `ServiceTier` to the effective tier for the
* given provider. Scoped values match their target family and otherwise
* collapse to `undefined`; unscoped values pass through unchanged.
* Per-family service-tier selection. A request consults only the entry for the
* family its model belongs to (see {@link resolveModelServiceTier}), so a user
* can opt one family into priority without affecting the others when switching
* models mid-session.
*/
export function resolveServiceTier(
serviceTier: ServiceTier | null | undefined,
provider: Provider | undefined,
): ResolvedServiceTier | undefined {
if (!serviceTier) return undefined;
switch (serviceTier) {
case "openai-only":
return provider === "openai" || provider === "openai-codex" ? "priority" : undefined;
case "claude-only":
return provider === "anthropic" ? "priority" : undefined;
default:
return serviceTier;
export type ServiceTierByFamily = Partial<Record<ServiceTierFamily, ServiceTier>>;
/**
* Classify a model into the service-tier family whose knob governs it, or
* `undefined` when the model exposes no serving-priority control.
*
* OpenRouter models are classified by id namespace (`anthropic/`, `google/`,
* `openai/`); Claude on Bedrock/Vertex (api `anthropic-messages`) is the
* anthropic family even though its provider is `amazon-bedrock`/`google-vertex`.
*/
export function serviceTierFamily(model: Pick<Model, "provider" | "api" | "id">): ServiceTierFamily | undefined {
const provider = model.provider;
if (provider === "openrouter") {
const id = model.id.toLowerCase();
if (id.startsWith("anthropic/")) return "anthropic";
if (id.startsWith("google/")) return "google";
if (id.startsWith("openai/")) return "openai";
return undefined;
}
if (provider === "openai" || provider === "openai-codex") return "openai";
if (model.api === "anthropic-messages") return "anthropic";
if (provider === "google" || provider === "google-vertex") return "google";
return undefined;
}
/**
* True when the (possibly scoped) tier should be sent on the wire as the
* `service_tier` request field for the given provider. OpenAI / OpenAI-Codex
* accept `flex`/`scale`/`priority`; Fireworks Serverless realizes only its
* Priority serving path (`service_tier: "priority"`) on the OpenAI-compatible
* chat-completions endpoint. Unsupported tiers (`"auto"`, `"default"`), other
* providers, and scope mismatches all return false.
* Reduce a per-family tier map to the single wire tier for `model` — the entry
* for the model's family, or `undefined` when the model has no family.
*/
export function resolveModelServiceTier(
tiers: ServiceTierByFamily | null | undefined,
model: Pick<Model, "provider" | "api" | "id">,
): ServiceTier | undefined {
if (!tiers) return undefined;
const family = serviceTierFamily(model);
return family ? tiers[family] : undefined;
}
/**
* True when the tier should be sent on the wire as the provider's service-tier
* request field. OpenAI / OpenAI-Codex accept `flex`/`scale`/`priority`; Google
* (Gemini API + Vertex) and OpenRouter accept `flex`/`priority`; Fireworks
* Serverless realizes only its Priority serving path. Anthropic is absent — it
* realizes `priority` via `speed: "fast"`, not a service-tier field.
*/
export function shouldSendServiceTier(
serviceTier: ServiceTier | null | undefined,
provider: Provider | undefined,
): boolean {
const resolved = resolveServiceTier(serviceTier, provider);
if (provider === "openai" || provider === "openai-codex") {
return resolved === "flex" || resolved === "scale" || resolved === "priority";
if (!serviceTier) return false;
if (provider === "openai" || provider === "openai-codex" || provider === "openrouter") {
return serviceTier === "flex" || serviceTier === "scale" || serviceTier === "priority";
}
if (provider === "fireworks") {
return resolved === "priority";
if (provider === "google") {
return serviceTier === "flex" || serviceTier === "priority";
}
// Vertex realizes only priority (via header); flex has no documented control.
if (provider === "google-vertex" || provider === "fireworks") {
return serviceTier === "priority";
}
return false;
}
/**
* Premium-request weight contributed by sending priority to a provider
* that supports it. Mirrors GitHub Copilot's `premiumRequests` accounting
* so the "premium requests" stat aggregates priority traffic across the
* OpenAI family and Anthropic fast-mode realizations.
* True when `priority` will actually be realized on the wire for `model`.
* Direct Anthropic realizes fast mode; OpenAI/Google/Fireworks emit the
* service-tier field; OpenRouter realizes it only for its OpenAI- and
* Google-family upstreams. Bedrock/Vertex Claude and OpenRouter Anthropic
* models do not realize priority and return `false`.
*/
export function realizesPriorityServiceTier(
serviceTier: ServiceTier | null | undefined,
model: Pick<Model, "provider" | "api" | "id">,
): boolean {
if (serviceTier !== "priority") return false;
if (model.provider === "anthropic") return true;
if (model.provider === "openrouter") {
const family = serviceTierFamily(model);
return family === "openai" || family === "google";
}
if (model.api === "anthropic-messages") return false;
return shouldSendServiceTier(serviceTier, model.provider);
}
/**
* Premium-request weight contributed by a priority request to a provider that
* realizes it and bills extra. Mirrors GitHub Copilot's `premiumRequests`
* accounting so the "premium requests" stat aggregates priority traffic across
* the OpenAI family, direct Anthropic fast mode, and Google priority.
*
* Returns 1 per resolved priority request, 0 otherwise.
* Returns 1 only when priority is actually realized on the wire for `model`
* (see {@link realizesPriorityServiceTier}) and the provider bills it as a
* premium request. OpenRouter is excluded — it bills per its own pricing, not
* Copilot-premium semantics — as are Bedrock/Vertex Claude, where priority is
* silently dropped.
*/
export function getPriorityPremiumRequests(
serviceTier: ServiceTier | null | undefined,
provider: Provider | undefined,
model: Pick<Model, "provider" | "api" | "id">,
): number {
if (resolveServiceTier(serviceTier, provider) !== "priority") return 0;
// Only providers that realize `priority` on the wire bill the user.
// Everywhere else, the field is silently dropped and nothing is charged.
return provider === "openai" || provider === "openai-codex" || provider === "anthropic" ? 1 : 0;
if (!realizesPriorityServiceTier(serviceTier, model)) return 0;
const provider = model.provider;
return provider === "openai" ||
provider === "openai-codex" ||
provider === "anthropic" ||
provider === "google" ||
provider === "google-vertex"
? 1
: 0;
}
/**
* Coerce a persisted service-tier value to a {@link ServiceTierByFamily}. Newer
* sessions store the family map directly; legacy sessions stored a single
* scalar — `"priority"` applied everywhere, `"openai-only"`/`"claude-only"`
* scoped to one family, and the remaining values were OpenAI-only semantics.
*/
export function coerceServiceTierByFamily(value: unknown): ServiceTierByFamily | undefined {
if (value === null || value === undefined) return undefined;
if (typeof value === "object") {
const src = value as Record<string, unknown>;
const out: ServiceTierByFamily = {};
for (const family of ["openai", "anthropic", "google"] as const) {
const tier = src[family];
if (tier === "auto" || tier === "default" || tier === "flex" || tier === "scale" || tier === "priority") {
out[family] = tier;
}
}
return Object.keys(out).length > 0 ? out : undefined;
}
switch (value) {
case "priority":
return { openai: "priority", anthropic: "priority", google: "priority" };
case "openai-only":
return { openai: "priority" };
case "claude-only":
return { anthropic: "priority" };
case "auto":
return { openai: "auto" };
case "default":
return { openai: "default" };
case "flex":
return { openai: "flex" };
case "scale":
return { openai: "scale" };
default:
return undefined;
}
}
export interface ProviderSessionState {
@@ -277,6 +374,22 @@ export interface StreamOptions {
* Providers can use this to persist transport/session state between turns.
*/
providerSessionState?: Map<string, ProviderSessionState>;
/**
* Force Gemini model-mode Interactions API transport for providers that support it.
* When unset, those providers may still use Interactions to continue known
* server-side conversation lineage via `previousInteractionId` or stored state.
*/
useInteractionsApi?: boolean;
/**
* Whether supported Interactions transports should store server-side conversation
* state and return response ids for follow-up turns. Defaults to true.
*/
storeInteraction?: boolean;
/**
* Explicit Interactions response id to continue. Mutually exclusive with
* `storeInteraction: false` because the follow-up itself must be storable.
*/
previousInteractionId?: string;
/**
* Optional per-provider concurrent request cap for LLM stream calls. Keys are
* provider ids (`model.provider`); positive numeric values cap in-flight
@@ -0,0 +1,260 @@
/**
* Central live healing for leaked reasoning markup in the visible text channel.
*
* Some providers emit their canonical reasoning idioms (` ```thinking `,
* `<think>`, Gemma/Harmony channels, …) into the *visible* text stream instead
* of a structured thinking part. {@link wrapLeakedThinkingStream} re-projects any
* provider stream into a fresh {@link AssistantMessageEventStream}, splitting the
* leaked fences out into proper `thinking` blocks *live* as deltas arrive — so
* every provider gets the same healing, not just the three with provider-local
* {@link StreamMarkupHealing} loops.
*
* The healing is idempotent: a second pass over already-clean text finds no
* fences, so wrapping a provider that already heals (or wrapping twice) is a
* harmless pass-through. Signatures are load-bearing for Google/Gemini/Vertex
* thought round-tripping, so text sub-blocks carry the source `textSignature`,
* forwarded thinking blocks their `thinkingSignature`, and forwarded tool calls
* their `thoughtSignature`.
*
* Modeled on {@link wrapInbandToolStream} / `InbandStreamProjector` in
* `../dialect/owned-stream.ts`, minus all in-band tool-call grammar: tool-call
* events are forwarded verbatim.
*/
import type { AssistantMessage, TextContent, ThinkingContent, ToolCall } from "../types";
import { AssistantMessageEventStream } from "./event-stream";
import { StreamMarkupHealing, type StreamMarkupHealingEvent } from "./stream-markup-healing";
/**
* Wrap a provider stream so leaked reasoning fences are healed into thinking
* blocks live, for every provider. Returns a new stream that re-projects the
* inner one; the inner stream is fully consumed.
*/
export function wrapLeakedThinkingStream(inner: AssistantMessageEventStream): AssistantMessageEventStream {
const out = new AssistantMessageEventStream();
void (async () => {
try {
let projector: LeakedThinkingProjector | undefined;
for await (const event of inner) {
switch (event.type) {
case "start":
projector = new LeakedThinkingProjector(out, event.partial);
break;
case "text_delta": {
projector ??= new LeakedThinkingProjector(out, event.partial);
const block = event.partial.content[event.contentIndex];
projector.text(event.delta, block?.type === "text" ? block.textSignature : undefined);
break;
}
case "thinking_delta": {
projector ??= new LeakedThinkingProjector(out, event.partial);
const block = event.partial.content[event.contentIndex];
projector.thinking(event.delta, block?.type === "thinking" ? block.thinkingSignature : undefined);
break;
}
case "toolcall_start": {
projector ??= new LeakedThinkingProjector(out, event.partial);
const block = event.partial.content[event.contentIndex];
projector.toolStart(event.contentIndex, block?.type === "toolCall" ? block.name : "");
break;
}
case "toolcall_delta":
projector?.toolDelta(event.contentIndex, event.delta);
break;
case "toolcall_end":
projector?.toolEnd(event.contentIndex, event.toolCall);
break;
case "done": {
projector ??= new LeakedThinkingProjector(out, event.message);
const content = projector.finish(event.message);
out.push({ type: "done", reason: event.reason, message: { ...event.message, content } });
return;
}
case "error": {
projector ??= new LeakedThinkingProjector(out, event.error);
const content = projector.finish(event.error);
out.push({ type: "error", reason: event.reason, error: { ...event.error, content } });
return;
}
// text_start/text_end/thinking_start/thinking_end are ignored: the
// projector owns block boundaries (matches wrapInbandToolStream).
}
}
// Inner ended via end(result) without a terminal event.
if (!out.done) {
const result = await inner.result();
projector ??= new LeakedThinkingProjector(out, result);
const content = projector.finish(result);
out.end({ ...result, content });
}
} catch (err) {
if (!out.done) out.fail(err);
}
})();
return out;
}
type OpenBlock = { index: number } | undefined;
/**
* Re-projects an inner stream's events into `out`, healing leaked reasoning out
* of the visible text channel while forwarding native thinking and tool calls.
*/
class LeakedThinkingProjector {
readonly #out: AssistantMessageEventStream;
readonly #healer = new StreamMarkupHealing({ pattern: "thinking" });
#partial: AssistantMessage;
#text: OpenBlock;
#thinking: OpenBlock;
/** Total visible text length fed to the healer, to replay any un-streamed tail in {@link finish}. */
#fedLen = 0;
/** Latest non-undefined text signature seen, stamped onto held-back text flushed later. */
#lastTextSignature: string | undefined;
/** Forwarded native tool calls, keyed by the inner stream's `contentIndex`. */
#toolBlocks = new Map<number, { index: number }>();
constructor(out: AssistantMessageEventStream, seed: AssistantMessage) {
this.#out = out;
this.#partial = { ...seed, content: [] };
this.#out.push({ type: "start", partial: this.#partial });
}
/** Feed a visible-text delta through the healer, splitting leaked fences live. */
text(delta: string, signature: string | undefined): void {
this.#fedLen += delta.length;
if (signature !== undefined) this.#lastTextSignature = signature;
this.#apply(this.#healer.feedEvents(delta), this.#lastTextSignature);
}
/** Forward a native thinking delta, preserving its signature. */
thinking(delta: string, signature: string | undefined): void {
const index = this.#openThinking();
const block = this.#partial.content[index] as ThinkingContent;
block.thinking += delta;
if (signature !== undefined) block.thinkingSignature = signature;
this.#out.push({ type: "thinking_delta", contentIndex: index, delta, partial: this.#partial });
}
/** Forward a native tool call's start, releasing any held-back text first. */
toolStart(srcIndex: number, name: string): void {
this.#apply(this.#healer.flushEvents(), this.#lastTextSignature);
this.#closeText();
this.#closeThinking();
const block: ToolCall = { type: "toolCall", id: "", name, arguments: {} };
this.#partial.content.push(block);
const index = this.#partial.content.length - 1;
this.#toolBlocks.set(srcIndex, { index });
this.#out.push({ type: "toolcall_start", contentIndex: index, partial: this.#partial });
}
toolDelta(srcIndex: number, delta: string): void {
const entry = this.#toolBlocks.get(srcIndex);
if (!entry) return;
this.#out.push({ type: "toolcall_delta", contentIndex: entry.index, delta, partial: this.#partial });
}
toolEnd(srcIndex: number, toolCall: ToolCall): void {
const entry = this.#toolBlocks.get(srcIndex);
if (entry) {
const block = this.#partial.content[entry.index] as ToolCall;
Object.assign(block, toolCall);
this.#out.push({ type: "toolcall_end", contentIndex: entry.index, toolCall: block, partial: this.#partial });
this.#toolBlocks.delete(srcIndex);
return;
}
// `end` without a matching `start` — release held text, then forward whole.
this.#apply(this.#healer.flushEvents(), this.#lastTextSignature);
this.#closeText();
this.#closeThinking();
const block: ToolCall = { ...toolCall };
this.#partial.content.push(block);
const index = this.#partial.content.length - 1;
this.#out.push({ type: "toolcall_start", contentIndex: index, partial: this.#partial });
this.#out.push({ type: "toolcall_end", contentIndex: index, toolCall: block, partial: this.#partial });
}
/**
* Finalize: replay any un-streamed visible-text tail from `message.content`,
* flush held-back fragments, close open blocks, and return the healed content.
*/
finish(message: AssistantMessage): AssistantMessage["content"] {
let fullText = "";
let tailSignature: string | undefined;
for (const block of message.content) {
if (block.type === "text") {
fullText += block.text;
tailSignature = block.textSignature;
}
}
if (tailSignature !== undefined) this.#lastTextSignature = tailSignature;
if (fullText.length > this.#fedLen) {
this.#apply(this.#healer.feedEvents(fullText.slice(this.#fedLen)), this.#lastTextSignature);
}
this.#apply(this.#healer.flushEvents(), this.#lastTextSignature);
this.#closeText();
this.#closeThinking();
return this.#partial.content;
}
#apply(events: readonly StreamMarkupHealingEvent[], signature?: string): void {
for (const event of events) {
if (event.type === "text") this.#emitText(event.text, signature);
else if (event.type === "thinking") this.#emitHealedThinking(event.thinking);
}
}
#emitText(text: string, signature: string | undefined): void {
if (text.length === 0) return;
this.#closeThinking();
if (!this.#text) {
const block: TextContent =
signature === undefined ? { type: "text", text: "" } : { type: "text", text: "", textSignature: signature };
this.#partial.content.push(block);
this.#text = { index: this.#partial.content.length - 1 };
this.#out.push({ type: "text_start", contentIndex: this.#text.index, partial: this.#partial });
} else if (signature !== undefined) {
(this.#partial.content[this.#text.index] as TextContent).textSignature = signature;
}
const block = this.#partial.content[this.#text.index] as TextContent;
block.text += text;
this.#out.push({ type: "text_delta", contentIndex: this.#text.index, delta: text, partial: this.#partial });
}
/** Healed (leaked) thinking carries no signature, matching the source fence. */
#emitHealedThinking(text: string): void {
if (text.length === 0) return;
const index = this.#openThinking();
const block = this.#partial.content[index] as ThinkingContent;
block.thinking += text;
this.#out.push({ type: "thinking_delta", contentIndex: index, delta: text, partial: this.#partial });
}
#openThinking(): number {
this.#closeText();
if (!this.#thinking) {
this.#partial.content.push({ type: "thinking", thinking: "" });
this.#thinking = { index: this.#partial.content.length - 1 };
this.#out.push({ type: "thinking_start", contentIndex: this.#thinking.index, partial: this.#partial });
}
return this.#thinking.index;
}
#closeText(): void {
if (!this.#text) return;
const block = this.#partial.content[this.#text.index] as TextContent;
this.#out.push({ type: "text_end", contentIndex: this.#text.index, content: block.text, partial: this.#partial });
this.#text = undefined;
}
#closeThinking(): void {
if (!this.#thinking) return;
const block = this.#partial.content[this.#thinking.index] as ThinkingContent;
this.#out.push({
type: "thinking_end",
contentIndex: this.#thinking.index,
content: block.thinking,
partial: this.#partial,
});
this.#thinking = undefined;
}
}
@@ -88,22 +88,6 @@ describe("Anthropic priority service tier → speed='fast'", () => {
expect(payload.speed).toBeUndefined();
}
});
it("sets speed='fast' on direct anthropic when serviceTier='claude-only'", async () => {
const payload = (await capturePayload(makeAnthropicModel("claude-opus-4-7"), {
serviceTier: "claude-only",
})) as { speed?: string };
expect(payload.speed).toBe("fast");
});
it("omits speed when serviceTier='openai-only' on an anthropic model", async () => {
// Scoped to OpenAI — on this anthropic request, the scope doesn't match,
// so `speed` must not be set on the wire.
const payload = (await capturePayload(makeAnthropicModel("claude-opus-4-7"), {
serviceTier: "openai-only",
})) as Record<string, unknown>;
expect(payload.speed).toBeUndefined();
});
});
describe("clearAnthropicFastModeFallback", () => {
@@ -8,6 +8,7 @@ import { AuthStorage, SqliteAuthCredentialStore } from "@oh-my-pi/pi-ai/auth-sto
import * as deepseekModule from "@oh-my-pi/pi-ai/registry/deepseek";
import * as kagiModule from "@oh-my-pi/pi-ai/registry/kagi";
import * as ollamaCloudModule from "@oh-my-pi/pi-ai/registry/ollama-cloud";
import * as aiStream from "@oh-my-pi/pi-ai/stream";
import { removeWithRetries } from "../../utils/src/temp";
function countCredentialRows(dbPath: string, provider: string): number {
@@ -38,6 +39,9 @@ function countCredentialRowsByDisabledState(dbPath: string, provider: string, di
}
describe("AuthStorage api-key login upsert", () => {
// A live env var now (correctly) overrides a stored static api_key. These tests verify that a
// freshly stored api_key resolves through AuthStorage.getApiKey, so neutralize the env leg
// entirely — this ignores every provider's ambient env key, not just the few set locally.
let tempDir = "";
let dbPath = "";
let store: SqliteAuthCredentialStore | null = null;
@@ -47,6 +51,7 @@ describe("AuthStorage api-key login upsert", () => {
let loginOllamaCloudSpy: Mock<typeof ollamaCloudModule.loginOllamaCloud>;
beforeEach(async () => {
vi.spyOn(aiStream, "getEnvApiKey").mockReturnValue(undefined);
tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-auth-api-key-login-"));
dbPath = path.join(tempDir, "agent.db");
store = await SqliteAuthCredentialStore.open(dbPath);
@@ -68,14 +68,24 @@ describe("AuthStorage.getCredentialOrigin", () => {
});
});
test("a stored api key reports api_key and outranks a co-stored OAuth credential", async () => {
test("a stored OAuth credential outranks a co-stored api key", async () => {
await withEnv(SUPPRESS_ENV, async () => {
// getApiKey() prefers api_key before oauth, so the origin must match.
// getApiKey() resolves stored OAuth before a stored api_key, so the origin must match.
await auth?.set("openai", [
{ type: "oauth", access: "a", refresh: "r", expires: Date.now() + 60_000 },
{ type: "api_key", key: "sk-stored" },
]);
expect(auth?.getCredentialOrigin("openai")).toEqual({ kind: "api_key" });
expect(auth?.getCredentialOrigin("openai")).toEqual({ kind: "oauth" });
});
});
test("an explicit env var outranks a stored api key", async () => {
// Regression: a live env var is the user's current choice and must win over a stored
// static api_key (e.g. a stale broker-migrated copy) so `GEMINI_API_KEY` etc. take effect.
await withEnv({ ...SUPPRESS_ENV, OPENAI_API_KEY: "sk-env" }, async () => {
await auth?.set("openai", [{ type: "api_key", key: "sk-stored" }]);
expect(auth?.getCredentialOrigin("openai")).toEqual({ kind: "env", envVar: "OPENAI_API_KEY" });
expect(await auth?.getApiKey("openai")).toBe("sk-env");
});
});
@@ -0,0 +1,123 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { ConfigurationError } from "@oh-my-pi/pi-ai/error";
import { OAuthCallbackFlow } from "@oh-my-pi/pi-ai/registry/oauth/callback-server";
import type { OAuthCredentials } from "@oh-my-pi/pi-ai/registry/oauth/types";
/**
* Minimal callback flow we can drive without a real authorization server.
* `generateAuthUrl` is never expected to run in the strict-port tests —
* `#startCallbackServer` must throw before `login()` can reach it — so a stray
* invocation surfaces as a counter bump the test asserts on.
*/
class TestCallbackFlow extends OAuthCallbackFlow {
authUrlCalls = 0;
lastRedirectUri?: string;
async generateAuthUrl(_state: string, redirectUri: string): Promise<{ url: string }> {
this.authUrlCalls += 1;
this.lastRedirectUri = redirectUri;
return { url: `${redirectUri}?started=1` };
}
async exchangeToken(code: string, _state: string, _redirectUri: string): Promise<OAuthCredentials> {
return { access: `access-${code}`, refresh: "refresh", expires: Date.now() + 60_000 };
}
}
/**
* Bind a real loopback port so the next `Bun.serve({ port })` against the
* same port fails with EADDRINUSE. Returns the bound port plus a `release`
* callback for teardown.
*/
function occupyLoopbackPort(): { port: number; release: () => void } {
const server = Bun.serve({ port: 0, fetch: () => new Response("blocker") });
const port = server.port;
if (typeof port !== "number") {
server.stop(true);
throw new Error("Bun.serve({ port: 0 }) did not assign a numeric port");
}
return { port, release: () => server.stop(true) };
}
afterEach(() => {
vi.restoreAllMocks();
});
describe("OAuthCallbackFlow port fallback policy", () => {
it("falls back to a random port by default so historical AI-provider flows keep working", async () => {
const blocker = occupyLoopbackPort();
const progress: string[] = [];
const flow = new TestCallbackFlow(
{
onAuth: () => {},
onProgress: msg => progress.push(msg),
// Short abort — we only care that the flow advertised the fallback URI.
signal: AbortSignal.timeout(100),
},
{ preferredPort: blocker.port },
);
try {
await expect(flow.login()).rejects.toThrow(); // aborted while waiting for the browser callback
const fallbackNotice = progress.find(msg => msg.startsWith(`Preferred port ${blocker.port} unavailable`));
expect(fallbackNotice).toBeDefined();
// Notice carries a different (random) port, never the blocked one.
expect(fallbackNotice).not.toContain(`using port ${blocker.port}`);
// generateAuthUrl ran with the random-port redirect URI — that's the
// silent fallback behavior that MCP flows now opt out of.
expect(flow.authUrlCalls).toBe(1);
expect(flow.lastRedirectUri).toMatch(/^http:\/\/localhost:\d+\/callback$/);
expect(flow.lastRedirectUri).not.toContain(`:${blocker.port}/`);
} finally {
blocker.release();
}
});
it("throws a ConfigurationError when allowPortFallback is false", async () => {
const serveSpy = vi.spyOn(Bun, "serve").mockImplementation(() => {
throw new Error("EADDRINUSE");
});
const flow = new TestCallbackFlow(
{
onAuth: () => {},
signal: AbortSignal.timeout(1_000),
},
{ preferredPort: 14581, allowPortFallback: false },
);
await expect(flow.login()).rejects.toThrow(ConfigurationError);
await expect(flow.login()).rejects.toThrow(
/OAuth callback port 14581 is in use\. The OAuth provider validates redirect URIs/,
);
// Fallback to port 0 must never be attempted: every serve call uses the preferred port.
const portArgs = serveSpy.mock.calls.map(([opts]) => opts.port);
expect(portArgs.every(port => port === 14581)).toBe(true);
// generateAuthUrl never runs: the error fires before login() opens the browser.
expect(flow.authUrlCalls).toBe(0);
});
it("preserves redirectUri-strict behavior with the updated error message", async () => {
vi.spyOn(Bun, "serve").mockImplementation(() => {
throw new Error("EADDRINUSE");
});
const flow = new TestCallbackFlow(
{
onAuth: () => {},
signal: AbortSignal.timeout(1_000),
},
{
preferredPort: 14582,
redirectUri: "http://localhost:14582/callback",
},
);
// redirectUri takes precedence over allowPortFallback in the error
// message so users learn exactly which configuration knob is forcing
// the strict port match.
await expect(flow.login()).rejects.toThrow(
/oauth\.redirectUri \(http:\/\/localhost:14582\/callback\) requires this exact port/,
);
});
});
@@ -89,7 +89,8 @@ describe("Google empty-response retry (public + Vertex path)", () => {
return calls === 1 ? sse(genaiChunk("")) : sse(genaiChunk("Hello!"));
};
const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock });
// Pin the generateContent transport: gemini-3 ids now auto-route to Interactions by default.
const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock, useInteractionsApi: false });
const { events, starts } = await drain(stream);
const result = await stream.result();
@@ -107,7 +108,7 @@ describe("Google empty-response retry (public + Vertex path)", () => {
return sse(genaiChunk(""));
};
const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock });
const stream = streamGoogle(genaiModel, context, { apiKey: "k", fetch: fetchMock, useInteractionsApi: false });
const result = await stream.result();
expect(calls).toBe(3); // MAX_EMPTY_STREAM_RETRIES (2) + 1 initial attempt
@@ -137,6 +138,7 @@ describe("Google empty-response retry (public + Vertex path)", () => {
project: "project",
location: "location",
fetch: fetchMock,
useInteractionsApi: false,
});
const { events } = await drain(stream);
const result = await stream.result();
@@ -194,6 +196,7 @@ describe("Google empty-response retry (public + Vertex path)", () => {
project: "project",
location: "location",
fetch: fetchMock,
useInteractionsApi: false,
});
const result = await stream.result();
@@ -0,0 +1,145 @@
import { describe, expect, it } from "bun:test";
import { convertMessages } from "@oh-my-pi/pi-ai/providers/google-shared";
import type { Context, Model, Usage } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
const ZERO_USAGE: Usage = {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
function createGoogleModel(
id: string,
api: "google-generative-ai" | "google-vertex" = "google-generative-ai",
): Model<typeof api> {
return buildModel({
id,
name: id,
api,
provider: api === "google-vertex" ? "google-vertex" : "google",
baseUrl: "https://example.com",
reasoning: false,
input: ["text", "image"],
cost: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 200000,
maxTokens: 8192,
});
}
function contextWithToolResult(toolName = "stale_tool_name"): Context {
return {
messages: [
{
role: "user",
content: "Call the tool",
timestamp: 1000,
},
{
role: "assistant",
provider: "google-generative-ai",
api: "google-generative-ai",
model: "gemini-2.5-flash",
content: [
{
type: "toolCall",
id: "call_12345_abc",
name: "actual_tool_name",
arguments: { query: "pi" },
},
],
usage: ZERO_USAGE,
stopReason: "toolUse",
timestamp: 2000,
},
{
role: "toolResult",
toolCallId: "call_12345_abc",
toolName,
isError: false,
content: [
{
type: "text",
text: "Tool result text",
},
],
timestamp: 3000,
},
],
};
}
function functionCallAndResponse(model: Model<"google-generative-ai" | "google-vertex">, context: Context) {
const contents = convertMessages(model, context);
const functionCall = contents.find(c => c.role === "model")?.parts?.find(part => part.functionCall)?.functionCall;
const functionResponse = contents
.find(c => c.role === "user" && c.parts?.some(part => part.functionResponse))
?.parts?.find(part => part.functionResponse)?.functionResponse;
return { functionCall, functionResponse };
}
describe("Google GenerateContent function response matching", () => {
it("uses emitted functionCall IDs and names for direct Gemini 3 functionResponse parts", () => {
const model = createGoogleModel("gemini-3.5-flash");
const { functionCall, functionResponse } = functionCallAndResponse(model, contextWithToolResult());
expect(functionCall?.id).toBe("call_12345_abc");
expect(functionResponse?.id).toBe("call_12345_abc");
expect(functionCall?.name).toBe("actual_tool_name");
expect(functionResponse?.name).toBe("actual_tool_name");
expect(functionResponse?.name).toBe(functionCall?.name);
});
it("omits unsupported Part IDs for Vertex Gemini 3.5 GenerateContent", () => {
const model = createGoogleModel("gemini-3.5-flash", "google-vertex");
const { functionCall, functionResponse } = functionCallAndResponse(model, contextWithToolResult());
expect(functionCall?.id).toBeUndefined();
expect(functionResponse?.id).toBeUndefined();
expect(functionResponse?.name).toBe(functionCall?.name);
});
it("keeps multimodal tool output inside Gemini 3 functionResponse parts", () => {
const context = contextWithToolResult("actual_tool_name");
const toolResult = context.messages[2];
if (toolResult.role !== "toolResult") throw new Error("expected tool result fixture");
toolResult.content.push({
type: "image",
mimeType: "image/png",
data: "base64-image-data",
});
const model = createGoogleModel("gemini-3.5-flash");
const { functionResponse } = functionCallAndResponse(model, context);
expect(functionResponse?.parts).toEqual([
{
inlineData: {
mimeType: "image/png",
data: "base64-image-data",
},
},
]);
});
it("keeps Claude call IDs on non-Vertex Google-compatible endpoints", () => {
const model = createGoogleModel("claude-sonnet-4-5");
const { functionCall, functionResponse } = functionCallAndResponse(
model,
contextWithToolResult("actual_tool_name"),
);
expect(functionCall?.id).toBe("call_12345_abc");
expect(functionResponse?.id).toBe("call_12345_abc");
expect(functionResponse?.name).toBe(functionCall?.name);
});
});
@@ -596,11 +596,9 @@ describe("Google Gemini CLI alignment", () => {
}
const result = await stream.result();
expect(result.content).toHaveLength(1);
expect(result.content[0]).toEqual({
type: "text",
text: "",
});
// A fully-discarded planning leak leaves no residual content — no empty
// text block survives (the central healing wrapper strips empties too).
expect(result.content).toHaveLength(0);
expect(result.stopReason).toBe("stop");
const textDeltaEvents = events.filter(e => e.type === "text_delta");
@@ -838,15 +836,11 @@ describe("Google Gemini CLI alignment", () => {
}
const result = await stream.result();
expect(result.content).toHaveLength(2);
expect(result.content[0]).toEqual({
type: "text",
text: "",
});
expect(result.content[1].type).toBe("toolCall");
if (result.content[1].type === "toolCall") {
expect(result.content[1].name).toBe("read");
expect(result.content[1].arguments).toEqual({ path: "src/main.ts" });
expect(result.content).toHaveLength(1);
expect(result.content[0].type).toBe("toolCall");
if (result.content[0].type === "toolCall") {
expect(result.content[0].name).toBe("read");
expect(result.content[0].arguments).toEqual({ path: "src/main.ts" });
}
expect(events.filter(e => e.type === "toolcall_start")).toHaveLength(1);
@@ -917,11 +911,7 @@ describe("Google Gemini CLI alignment", () => {
events.push(event);
}
const result = await stream.result();
expect(result.content).toHaveLength(1);
expect(result.content[0]).toEqual({
type: "text",
text: "",
});
expect(result.content).toHaveLength(0);
});
});
});
@@ -0,0 +1,511 @@
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import { streamGoogle } from "@oh-my-pi/pi-ai/providers/google";
import { __resetVertexTokenCache } from "@oh-my-pi/pi-ai/providers/google-auth";
import { streamGoogleVertex } from "@oh-my-pi/pi-ai/providers/google-vertex";
import { streamSimple } from "@oh-my-pi/pi-ai/stream";
import type { AssistantMessage, Context, FetchImpl, Model, Tool, Usage } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
function googleModel(baseUrl = "https://generativelanguage.googleapis.com/v1beta"): Model<"google-generative-ai"> {
return buildModel({
id: "gemini-3.5-flash",
name: "Gemini 3.5 Flash",
api: "google-generative-ai",
provider: "google",
baseUrl,
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 1_000_000,
maxTokens: 8_192,
});
}
function vertexModel(id = "gemini-3.5-flash"): Model<"google-vertex"> {
return buildModel({
id,
name: id,
api: "google-vertex",
provider: "google-vertex",
baseUrl: "",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 1_000_000,
maxTokens: 8_192,
});
}
function sseResponse(events: readonly unknown[]): Response {
const payload = `${events.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`;
return new Response(payload, { status: 200, headers: { "content-type": "text/event-stream" } });
}
const weatherTool: Tool = {
name: "get_weather",
description: "Get weather",
parameters: {
type: "object",
properties: { city: { type: "string" } },
required: ["city"],
additionalProperties: false,
},
};
describe("Google Interactions API", () => {
it("chains tool results with previous_interaction_id from the prior assistant response", async () => {
const model = googleModel();
const requestBodies: unknown[] = [];
let calls = 0;
const fetchMock: FetchImpl = async (_input, init) => {
requestBodies.push(JSON.parse(String(init?.body ?? "{}")));
calls += 1;
if (calls === 1) {
return sseResponse([
{
event_type: "interaction.created",
interaction: { id: "int_1", status: "in_progress" },
},
{ event_type: "step.start", index: 0, step: { type: "thought" } },
{
event_type: "step.delta",
index: 0,
delta: { type: "thought_signature", signature: "thought_sig_1" },
},
{
event_type: "step.delta",
index: 0,
delta: { type: "thought_summary", content: { type: "text", text: "Checking weather.\n" } },
},
{ event_type: "step.stop", index: 0 },
{
event_type: "step.start",
index: 1,
step: {
type: "function_call",
id: "call_weather",
name: "get_weather",
arguments: {},
},
},
{ event_type: "step.delta", index: 1, delta: { type: "arguments_delta", arguments: '{"city":"Bos' } },
{ event_type: "step.delta", index: 1, delta: { type: "arguments_delta", arguments: 'ton"}' } },
{ event_type: "step.stop", index: 1 },
{
event_type: "interaction.completed",
interaction: {
id: "int_1",
status: "requires_action",
usage: { total_input_tokens: 10, total_output_tokens: 2, total_tokens: 12 },
},
},
]);
}
return sseResponse([
{ event_type: "interaction.created", interaction: { id: "int_2", status: "in_progress" } },
{ event_type: "step.start", index: 0, step: { type: "model_output" } },
{ event_type: "step.delta", index: 0, delta: { type: "text", text: "Sunny." } },
{ event_type: "step.stop", index: 0 },
{
event_type: "interaction.completed",
interaction: {
id: "int_2",
status: "completed",
usage: { total_input_tokens: 3, total_output_tokens: 1, total_tokens: 4 },
},
},
]);
};
Object.assign(fetchMock, { preconnect: fetch.preconnect });
const firstContext: Context = {
systemPrompt: ["Use concise weather reports."],
messages: [{ role: "user", content: "Need weather", timestamp: 1 }],
tools: [weatherTool],
};
const first = await streamGoogle(model, firstContext, {
apiKey: "test-key",
fetch: fetchMock,
useInteractionsApi: true,
thinking: { enabled: true, level: "HIGH", budgetTokens: 123 },
}).result();
expect(first.responseId).toBe("int_1");
expect(first.stopReason).toBe("toolUse");
expect(first.content).toEqual([
{ type: "thinking", thinking: "Checking weather.\n", thinkingSignature: "thought_sig_1" },
{ type: "toolCall", id: "call_weather", name: "get_weather", arguments: { city: "Boston" } },
]);
expect(requestBodies[0]).toMatchObject({
model: "gemini-3.5-flash",
stream: true,
input: [{ type: "user_input", content: [{ type: "text", text: "Need weather" }] }],
system_instruction: "Use concise weather reports.",
tools: [{ functionDeclarations: [{ name: "get_weather" }] }],
generation_config: { thinking_level: "high" },
});
expect(requestBodies[0]).not.toHaveProperty("previous_interaction_id");
expect(JSON.stringify(requestBodies[0])).not.toContain("thinking_budget");
const secondContext: Context = {
messages: [
{ role: "user", content: "Need weather", timestamp: 1 },
first,
{
role: "toolResult",
toolCallId: "call_weather",
toolName: "get_weather",
content: [{ type: "text", text: "72F and sunny" }],
isError: false,
timestamp: 2,
},
],
tools: [weatherTool],
systemPrompt: ["Use concise weather reports."],
};
const second = await streamGoogle(model, secondContext, {
apiKey: "test-key",
fetch: fetchMock,
thinking: { enabled: true, level: "HIGH", budgetTokens: 123 },
}).result();
expect(second.responseId).toBe("int_2");
expect(second.content).toEqual([{ type: "text", text: "Sunny." }]);
expect(requestBodies[1]).toMatchObject({
previous_interaction_id: "int_1",
input: [
{
type: "function_result",
name: "get_weather",
call_id: "call_weather",
result: [{ type: "text", text: "72F and sunny" }],
},
],
tools: [{ functionDeclarations: [{ name: "get_weather" }] }],
system_instruction: "Use concise weather reports.",
generation_config: { thinking_level: "high" },
});
expect(JSON.stringify(requestBodies[1])).not.toContain("Need weather");
expect(JSON.stringify(requestBodies[1])).not.toContain("thinking_budget");
});
it("does not expose or reuse interaction ids when storage is disabled", async () => {
const model = googleModel();
const requestBodies: unknown[] = [];
const fetchMock: FetchImpl = async (_input, init) => {
requestBodies.push(JSON.parse(String(init?.body ?? "{}")));
return sseResponse([
{ event_type: "interaction.created", interaction: { id: "unstored_int", status: "in_progress" } },
{ event_type: "step.start", index: 0, step: { type: "model_output" } },
{ event_type: "step.delta", index: 0, delta: { type: "text", text: "Done." } },
{ event_type: "step.stop", index: 0 },
{ event_type: "interaction.completed", interaction: { id: "unstored_int", status: "completed" } },
]);
};
Object.assign(fetchMock, { preconnect: fetch.preconnect });
const result = await streamGoogle(
model,
{ messages: [{ role: "user", content: "Hello", timestamp: 1 }] },
{
apiKey: "test-key",
fetch: fetchMock,
useInteractionsApi: true,
storeInteraction: false,
},
).result();
expect(result.responseId).toBeUndefined();
expect(requestBodies[0]).toMatchObject({ store: false });
expect(() =>
streamGoogle(
model,
{ messages: [{ role: "user", content: "Hello", timestamp: 1 }] },
{
apiKey: "test-key",
fetch: fetchMock,
storeInteraction: false,
previousInteractionId: "unstored_int",
},
),
).toThrow(/storeInteraction:false/);
});
it("reads thought payload from step.start without leaking a prior signature", async () => {
const model = googleModel();
const fetchMock: FetchImpl = async () =>
sseResponse([
{ event_type: "interaction.created", interaction: { id: "int_3", status: "in_progress" } },
{ event_type: "step.start", index: 0, step: { type: "thought", signature: "stale_sig" } },
{ event_type: "step.stop", index: 0 },
{
event_type: "step.start",
index: 1,
step: { type: "thought", summary: [{ type: "text", text: "Fresh plan.\n" }] },
},
{ event_type: "step.stop", index: 1 },
{ event_type: "step.start", index: 2, step: { type: "model_output" } },
{ event_type: "step.delta", index: 2, delta: { type: "text", text: "Answer." } },
{ event_type: "step.stop", index: 2 },
{ event_type: "interaction.completed", interaction: { id: "int_3", status: "completed" } },
]);
Object.assign(fetchMock, { preconnect: fetch.preconnect });
const result = await streamGoogle(
model,
{ messages: [{ role: "user", content: "Hello", timestamp: 1 }] },
{ apiKey: "test-key", fetch: fetchMock, useInteractionsApi: true },
).result();
expect(result.content).toEqual([
{ type: "thinking", thinking: "Fresh plan.\n" },
{ type: "text", text: "Answer." },
]);
});
});
function genaiSse(text: string): Response {
return new Response(
`data: ${JSON.stringify({
candidates: [{ content: { parts: [{ text }] }, finishReason: "STOP" }],
usageMetadata: { promptTokenCount: 1, candidatesTokenCount: 1, totalTokenCount: 2 },
})}\n\n`,
{ status: 200, headers: { "content-type": "text/event-stream" } },
);
}
function interactionsTextSse(
id: string,
text: string,
terminal: "interaction.completed" | "interaction.complete" = "interaction.completed",
): Response {
return sseResponse([
{ event_type: "interaction.created", interaction: { id, status: "in_progress" } },
{ event_type: "step.start", index: 0, step: { type: "model_output" } },
{ event_type: "step.delta", index: 0, delta: { type: "text", text } },
{ event_type: "step.stop", index: 0 },
{
event_type: terminal,
interaction: {
id,
status: "completed",
usage: { total_input_tokens: 10, total_output_tokens: 5, total_tokens: 15 },
},
},
]);
}
interface CapturedCall {
url: string;
method: string;
headers: Headers;
body: unknown;
}
function captureFetch(handler: (url: string) => Response): { fetch: FetchImpl; calls: CapturedCall[] } {
const calls: CapturedCall[] = [];
const fetchMock: FetchImpl = async (input, init) => {
const url = input instanceof Request ? input.url : String(input);
calls.push({
url,
method: String(init?.method ?? "GET"),
headers: new Headers(init?.headers),
body: init?.body ? JSON.parse(String(init.body)) : undefined,
});
return handler(url);
};
Object.assign(fetchMock, { preconnect: fetch.preconnect });
return { fetch: fetchMock, calls };
}
const ZERO_USAGE: Usage = {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
function assistantWithResponse(
api: "google-vertex" | "google-generative-ai",
provider: string,
responseId: string,
): AssistantMessage {
return {
role: "assistant",
api,
provider,
model: "gemini-3.5-flash",
content: [{ type: "text", text: "prev" }],
usage: ZERO_USAGE,
stopReason: "stop",
timestamp: 2,
responseId,
};
}
const userTurn: Context = { messages: [{ role: "user", content: "Hi there", timestamp: 1 }] };
describe("Google Interactions API — zero-config default + fallback", () => {
let savedToken: string | undefined;
beforeEach(() => {
// A bearer source makes the Vertex auto-gate fire deterministically on any machine, and
// `getVertexAccessToken` returns it directly (no OAuth/metadata round-trip in tests).
savedToken = Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN;
Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN = "test-bearer";
});
afterEach(() => {
if (savedToken === undefined) delete Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN;
else Bun.env.GOOGLE_CLOUD_ACCESS_TOKEN = savedToken;
__resetVertexTokenCache();
});
it("auto-routes a capable Vertex model to Interactions under bearer auth", async () => {
const { fetch, calls } = captureFetch(() => interactionsTextSse("vint_1", "Hi"));
const result = await streamGoogleVertex(vertexModel(), userTurn, {
project: "p",
location: "us",
fetch,
}).result();
expect(calls).toHaveLength(1);
expect(calls[0].url).toBe("https://aiplatform.googleapis.com/v1beta1/projects/p/locations/global/interactions");
expect(calls[0].method).toBe("POST");
expect(calls[0].headers.get("Api-Revision")).toBe("2026-05-20");
expect(calls[0].headers.get("Authorization")).toBe("Bearer test-bearer");
expect(calls[0].body).toMatchObject({
model: "gemini-3.5-flash",
stream: true,
input: [{ type: "user_input", content: [{ type: "text", text: "Hi there" }] }],
});
expect(calls[0].body).not.toHaveProperty("contents");
expect(calls[0].body).not.toHaveProperty("agent");
expect(calls[0].body).not.toHaveProperty("environment");
expect(result.stopReason).toBe("stop");
expect(result.content).toEqual([{ type: "text", text: "Hi" }]);
expect(result.responseId).toBe("vint_1");
expect(result.usage.totalTokens).toBe(15);
});
it("keeps an older (sub-3) Vertex model on generateContent", async () => {
const { fetch, calls } = captureFetch(() => genaiSse("ok"));
await streamGoogleVertex(vertexModel("gemini-2.5-flash"), userTurn, {
project: "p",
location: "us",
fetch,
}).result();
expect(calls[0].url).toContain(":streamGenerateContent");
expect(calls.some(c => c.url.includes("/interactions"))).toBe(false);
});
it("honors useInteractionsApi:false on a capable Vertex model", async () => {
const { fetch, calls } = captureFetch(() => genaiSse("ok"));
await streamGoogleVertex(vertexModel(), userTurn, {
project: "p",
location: "us",
useInteractionsApi: false,
fetch,
}).result();
expect(calls[0].url).toContain(":streamGenerateContent");
expect(calls.some(c => c.url.includes("/interactions"))).toBe(false);
});
it("falls back to generateContent when auto Interactions is unsupported (404)", async () => {
const { fetch, calls } = captureFetch(url =>
url.includes("/interactions") ? new Response("nope", { status: 404 }) : genaiSse("recovered"),
);
const result = await streamGoogleVertex(vertexModel(), userTurn, {
project: "p",
location: "us",
fetch,
}).result();
expect(calls[0].url).toContain("/interactions");
expect(calls[1].url).toContain(":streamGenerateContent");
expect(result.stopReason).toBe("stop");
expect(result.content).toEqual([{ type: "text", text: "recovered" }]);
});
it("surfaces the error (no fallback) when explicit Interactions is unsupported", async () => {
const { fetch, calls } = captureFetch(url =>
url.includes("/interactions") ? new Response("nope", { status: 404 }) : genaiSse("unexpected"),
);
const result = await streamGoogleVertex(vertexModel(), userTurn, {
project: "p",
useInteractionsApi: true,
fetch,
}).result();
expect(result.stopReason).toBe("error");
expect(calls.some(c => c.url.includes(":streamGenerateContent"))).toBe(false);
});
it("accepts interaction.complete as a terminal-event alias", async () => {
const { fetch } = captureFetch(() => interactionsTextSse("vint_2", "Done", "interaction.complete"));
const result = await streamGoogleVertex(vertexModel(), userTurn, { project: "p", fetch }).result();
expect(result.stopReason).toBe("stop");
expect(result.content).toEqual([{ type: "text", text: "Done" }]);
});
it("sends previous_interaction_id only for same-provider assistant lineage", async () => {
const sameProvider = captureFetch(() => interactionsTextSse("vint_3", "ok"));
await streamGoogleVertex(
vertexModel(),
{
messages: [
{ role: "user", content: "a", timestamp: 1 },
assistantWithResponse("google-vertex", "google-vertex", "vint_prev"),
{ role: "user", content: "b", timestamp: 3 },
],
},
{ project: "p", fetch: sameProvider.fetch },
).result();
expect(sameProvider.calls[0].body).toMatchObject({ previous_interaction_id: "vint_prev" });
const wrongProvider = captureFetch(() => interactionsTextSse("vint_4", "ok"));
await streamGoogleVertex(
vertexModel(),
{
messages: [
{ role: "user", content: "a", timestamp: 1 },
assistantWithResponse("google-generative-ai", "google", "gint_prev"),
{ role: "user", content: "b", timestamp: 3 },
],
},
{ project: "p", fetch: wrongProvider.fetch },
).result();
expect(wrongProvider.calls[0].body).not.toHaveProperty("previous_interaction_id");
});
it("auto-routes a capable direct Google model on the official endpoint to Interactions", async () => {
const { fetch, calls } = captureFetch(() => interactionsTextSse("gint_1", "Hi"));
const result = await streamGoogle(googleModel(), userTurn, { apiKey: "k", fetch }).result();
expect(calls[0].url).toBe("https://generativelanguage.googleapis.com/v1beta/interactions");
expect(calls[0].headers.get("x-goog-api-key")).toBe("k");
expect(result.responseId).toBe("gint_1");
});
it("keeps a custom-baseUrl direct Google model on generateContent", async () => {
const { fetch, calls } = captureFetch(() => genaiSse("ok"));
await streamGoogle(googleModel("https://proxy.example.com/v1beta"), userTurn, {
apiKey: "k",
fetch,
}).result();
expect(calls[0].url).toContain(":streamGenerateContent");
expect(calls[0].url.startsWith("https://proxy.example.com/")).toBe(true);
expect(calls.some(c => c.url.includes("/interactions"))).toBe(false);
});
it("threads the auto-default through streamSimple for a capable Google model", async () => {
const { fetch, calls } = captureFetch(() => interactionsTextSse("sint_1", "Hi"));
await streamSimple(googleModel(), userTurn, { apiKey: "k", fetch }).result();
expect(calls.some(c => c.url.includes("/interactions"))).toBe(true);
});
});
@@ -0,0 +1,107 @@
import { describe, expect, it } from "bun:test";
import { streamGoogle } from "@oh-my-pi/pi-ai/providers/google";
import { streamGoogleVertex } from "@oh-my-pi/pi-ai/providers/google-vertex";
import type { AssistantMessageEvent, Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
const context: Context = { messages: [{ role: "user", content: "hi", timestamp: 1 }] };
function sseStop(): Response {
const chunk = {
candidates: [{ content: { parts: [{ text: "ok" }] }, finishReason: "STOP" }],
usageMetadata: { promptTokenCount: 1, candidatesTokenCount: 1, totalTokenCount: 2 },
};
return new Response(`data: ${JSON.stringify(chunk)}\n\n`, {
status: 200,
headers: { "content-type": "text/event-stream" },
});
}
async function drain(stream: AsyncIterable<AssistantMessageEvent>): Promise<void> {
for await (const _ of stream) {
// consume
}
}
interface Captured {
headers: Headers;
body: Record<string, unknown>;
}
function capturingFetch(): { fetch: FetchImpl; captured: () => Captured } {
let cap: Captured | undefined;
const fetch: FetchImpl = async (_url, init) => {
cap = {
headers: new Headers(init?.headers),
body: JSON.parse(String(init?.body ?? "{}")),
};
return sseStop();
};
return {
fetch,
captured: () => {
if (!cap) throw new Error("fetch was not called");
return cap;
},
};
}
const geminiModel: Model<"google-generative-ai"> = buildModel({
id: "gemini-3-flash",
name: "Gemini 3 Flash",
api: "google-generative-ai",
provider: "google",
baseUrl: "https://generativelanguage.googleapis.com/v1beta",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 32_000,
});
const vertexModel: Model<"google-vertex"> = buildModel({
id: "gemini-3-flash",
name: "Gemini 3 Flash (Vertex)",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 32_000,
});
describe("Google service tier wire encoding", () => {
it("Gemini API sends the tier in the request body, not a header", async () => {
const { fetch, captured } = capturingFetch();
await drain(
streamGoogle(geminiModel, context, { apiKey: "k", serviceTier: "priority", fetch, useInteractionsApi: false }),
);
const { headers, body } = captured();
expect(body.serviceTier).toBe("priority");
expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBeNull();
});
it("Vertex sends priority via header and omits the body tier field", async () => {
const { fetch, captured } = capturingFetch();
await drain(streamGoogleVertex(vertexModel, context, { apiKey: "k", serviceTier: "priority", fetch }));
const { headers, body } = captured();
expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBe("priority");
expect(body.serviceTier).toBeUndefined();
});
it("Vertex omits both header and body for flex (no documented control)", async () => {
const { fetch, captured } = capturingFetch();
await drain(streamGoogleVertex(vertexModel, context, { apiKey: "k", serviceTier: "flex", fetch }));
const { headers, body } = captured();
expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBeNull();
expect(body.serviceTier).toBeUndefined();
});
it("omits the tier entirely when unset", async () => {
const { fetch, captured } = capturingFetch();
await drain(streamGoogle(geminiModel, context, { apiKey: "k", fetch, useInteractionsApi: false }));
expect(captured().body.serviceTier).toBeUndefined();
});
});
@@ -26,6 +26,8 @@ async function captureGooglePayload(
await streamGoogle(model, context, {
apiKey: "test-key",
// Capture the generateContent request shape; gemini-3 ids auto-route to Interactions by default.
useInteractionsApi: false,
onPayload: payload => {
captured = payload as { config: { systemInstruction?: unknown }; contents: unknown[] };
},
@@ -44,6 +44,8 @@ describe("issue #1270: Vertex AI global endpoint", () => {
const stream = streamGoogleVertex(model, context, {
project: "vertex-project",
location: "global",
// This asserts the generateContent URL; gemini-3 ids auto-route to Interactions by default.
useInteractionsApi: false,
fetch: async input => {
const url = input instanceof Request ? input.url : input.toString();
urls.push(url);
@@ -0,0 +1,285 @@
import { describe, expect, it } from "bun:test";
import { stream } from "@oh-my-pi/pi-ai/stream";
import type {
AssistantMessage,
AssistantMessageEvent,
Context,
FetchImpl,
Model,
TextContent,
ThinkingContent,
ToolCall,
} from "@oh-my-pi/pi-ai/types";
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
import { wrapLeakedThinkingStream } from "@oh-my-pi/pi-ai/utils/leaked-thinking-stream";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
/** Minimal assistant message; `content`/`stopReason` overridden per event. */
function msg(overrides: Partial<AssistantMessage> = {}): AssistantMessage {
return {
role: "assistant",
content: [],
api: "mock",
provider: "mock",
model: "mock",
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "stop",
timestamp: 0,
...overrides,
};
}
/**
* Drive the wrapper: push inner events synchronously, then drain the healed
* output. Returns every emitted event plus the resolved final message.
*/
async function runWrapper(
feed: (inner: AssistantMessageEventStream) => void,
): Promise<{ events: AssistantMessageEvent[]; result: AssistantMessage }> {
const inner = new AssistantMessageEventStream();
const out = wrapLeakedThinkingStream(inner);
feed(inner);
const events: AssistantMessageEvent[] = [];
for await (const event of out) events.push(event);
const result = await out.result();
return { events, result };
}
function texts(message: AssistantMessage): string[] {
return message.content.filter((b): b is TextContent => b.type === "text").map(b => b.text);
}
function thinks(message: AssistantMessage): ThinkingContent[] {
return message.content.filter((b): b is ThinkingContent => b.type === "thinking");
}
describe("wrapLeakedThinkingStream", () => {
it("splits a leaked fence into structured blocks live during streaming", async () => {
const leaked = "Visible before.```thinking\nplan\n```Visible after.";
const { events, result } = await runWrapper(inner => {
inner.push({ type: "start", partial: msg() });
inner.push({ type: "text_start", contentIndex: 0, partial: msg({ content: [{ type: "text", text: "" }] }) });
inner.push({
type: "text_delta",
contentIndex: 0,
delta: leaked,
partial: msg({ content: [{ type: "text", text: leaked }] }),
});
inner.push({
type: "text_end",
contentIndex: 0,
content: leaked,
partial: msg({ content: [{ type: "text", text: leaked }] }),
});
inner.push({ type: "done", reason: "stop", message: msg({ content: [{ type: "text", text: leaked }] }) });
});
expect(result.content.map(b => b.type)).toEqual(["text", "thinking", "text"]);
expect(texts(result)).toEqual(["Visible before.", "Visible after."]);
expect(thinks(result).map(b => b.thinking)).toEqual(["plan\n"]);
// The split happened live, not only in the terminal message.
expect(events.some(e => e.type === "thinking_delta")).toBe(true);
});
it("preserves text, thinking, and tool-call signatures across the split", async () => {
const leaked = "before ```thinking\nhmm\n``` after";
const call: ToolCall = {
type: "toolCall",
id: "call_1",
name: "read",
arguments: { path: "x" },
thoughtSignature: "tsig",
};
const { result } = await runWrapper(inner => {
inner.push({ type: "start", partial: msg() });
inner.push({
type: "text_start",
contentIndex: 0,
partial: msg({ content: [{ type: "text", text: "", textSignature: "sig" }] }),
});
inner.push({
type: "text_delta",
contentIndex: 0,
delta: leaked,
partial: msg({ content: [{ type: "text", text: leaked, textSignature: "sig" }] }),
});
const withCall = msg({
content: [{ type: "text", text: leaked, textSignature: "sig" }, call],
stopReason: "toolUse",
});
inner.push({ type: "toolcall_start", contentIndex: 1, partial: withCall });
inner.push({ type: "toolcall_end", contentIndex: 1, toolCall: call, partial: withCall });
inner.push({ type: "done", reason: "toolUse", message: withCall });
});
const textBlocks = result.content.filter((b): b is TextContent => b.type === "text");
expect(textBlocks.map(b => b.text)).toEqual(["before ", " after"]);
expect(textBlocks.map(b => b.textSignature)).toEqual(["sig", "sig"]);
// Healed (leaked) thinking carries no signature.
expect(thinks(result).every(b => b.thinkingSignature === undefined)).toBe(true);
const calls = result.content.filter((b): b is ToolCall => b.type === "toolCall");
expect(calls[0]?.thoughtSignature).toBe("tsig");
});
it("heals a fence that only appears in the terminal message (no prior text deltas)", async () => {
const leaked = "Intro.```thinking\nquiet\n```Outro.";
const { result } = await runWrapper(inner => {
inner.push({ type: "start", partial: msg() });
inner.push({
type: "done",
reason: "stop",
message: msg({ content: [{ type: "text", text: leaked, textSignature: "sig" }] }),
});
});
expect(result.content.map(b => b.type)).toEqual(["text", "thinking", "text"]);
expect(texts(result)).toEqual(["Intro.", "Outro."]);
// Tail-replayed text still carries the source signature.
expect(result.content.filter((b): b is TextContent => b.type === "text").map(b => b.textSignature)).toEqual([
"sig",
"sig",
]);
});
it("passes clean text through unchanged and forwards native thinking", async () => {
const clean = "Just a normal answer.";
const cleanRun = await runWrapper(inner => {
inner.push({ type: "start", partial: msg() });
inner.push({ type: "text_start", contentIndex: 0, partial: msg({ content: [{ type: "text", text: "" }] }) });
inner.push({
type: "text_delta",
contentIndex: 0,
delta: clean,
partial: msg({ content: [{ type: "text", text: clean }] }),
});
inner.push({ type: "done", reason: "stop", message: msg({ content: [{ type: "text", text: clean }] }) });
});
expect(cleanRun.result.content.map(b => b.type)).toEqual(["text"]);
expect(texts(cleanRun.result)).toEqual([clean]);
const nativeThinking = msg({
content: [
{ type: "thinking", thinking: "native reasoning", thinkingSignature: "tk" },
{ type: "text", text: "answer" },
],
});
const nativeRun = await runWrapper(inner => {
inner.push({ type: "start", partial: msg() });
inner.push({
type: "thinking_start",
contentIndex: 0,
partial: msg({ content: [{ type: "thinking", thinking: "" }] }),
});
inner.push({
type: "thinking_delta",
contentIndex: 0,
delta: "native reasoning",
partial: msg({ content: [{ type: "thinking", thinking: "native reasoning", thinkingSignature: "tk" }] }),
});
inner.push({
type: "thinking_end",
contentIndex: 0,
content: "native reasoning",
partial: msg({ content: [{ type: "thinking", thinking: "native reasoning", thinkingSignature: "tk" }] }),
});
inner.push({ type: "text_start", contentIndex: 1, partial: nativeThinking });
inner.push({ type: "text_delta", contentIndex: 1, delta: "answer", partial: nativeThinking });
inner.push({ type: "done", reason: "stop", message: nativeThinking });
});
expect(nativeRun.result.content.map(b => b.type)).toEqual(["thinking", "text"]);
expect(thinks(nativeRun.result)[0]?.thinking).toBe("native reasoning");
expect(thinks(nativeRun.result)[0]?.thinkingSignature).toBe("tk");
expect(texts(nativeRun.result)).toEqual(["answer"]);
});
it("heals a terminal error message and keeps its error stop reason", async () => {
const leaked = "Partial.```thinking\noops\n```Recovered.";
const { result } = await runWrapper(inner => {
inner.push({ type: "start", partial: msg() });
inner.push({
type: "error",
reason: "error",
error: msg({ content: [{ type: "text", text: leaked }], stopReason: "error" }),
});
});
expect(result.content.map(b => b.type)).toEqual(["text", "thinking", "text"]);
expect(texts(result)).toEqual(["Partial.", "Recovered."]);
expect(result.stopReason).toBe("error");
});
});
describe("leaked thinking healing through stream()", () => {
function sseFrame(event: string, data: unknown): string {
return `event: ${event}\ndata: ${JSON.stringify(data)}\n\n`;
}
function anthropicLeakFetch(text: string): FetchImpl {
const body = [
sseFrame("message_start", {
type: "message_start",
message: { id: "msg_leak", usage: { input_tokens: 5, output_tokens: 0 } },
}),
sseFrame("content_block_start", {
type: "content_block_start",
index: 0,
content_block: { type: "text", text: "" },
}),
sseFrame("content_block_delta", {
type: "content_block_delta",
index: 0,
delta: { type: "text_delta", text },
}),
sseFrame("content_block_stop", { type: "content_block_stop", index: 0 }),
sseFrame("message_delta", {
type: "message_delta",
delta: { stop_reason: "end_turn" },
usage: { input_tokens: 5, output_tokens: 4 },
}),
sseFrame("message_stop", { type: "message_stop" }),
].join("");
const fn = async (_input: string | URL | Request, _init?: RequestInit): Promise<Response> =>
new Response(body, {
status: 200,
headers: { "content-type": "text/event-stream", "request-id": "req_mock" },
});
return Object.assign(fn, { preconnect: fetch.preconnect });
}
it("splits a leaked fence from a provider with no own healer", async () => {
// Anthropic has no provider-local visible-text healer, so a split here
// proves the central wrapper is composed into stream().
const model: Model<"anthropic-messages"> = buildModel({
id: "claude-sonnet-4-5",
name: "Claude Sonnet 4.5",
api: "anthropic-messages",
provider: "anthropic",
baseUrl: "https://api.anthropic.com",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 8_192,
});
const leaked = "```thinking\nDeliberate.\n```\nFinal answer.";
const context: Context = { messages: [{ role: "user", content: "hi", timestamp: Date.now() }] };
const result = await stream(model, context, {
apiKey: "test",
fetch: anthropicLeakFetch(leaked),
}).result();
expect(result.content.map(b => b.type)).toEqual(["thinking", "text"]);
const thinking = thinks(result)
.map(b => b.thinking)
.join("");
expect(thinking).toContain("Deliberate.");
expect(texts(result).join("").trim()).toBe("Final answer.");
});
});
@@ -0,0 +1,68 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { scheduler } from "node:timers/promises";
import * as AIError from "@oh-my-pi/pi-ai/error";
import { streamOllama } from "@oh-my-pi/pi-ai/providers/ollama";
import type { Context, Model } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
const model: Model<"ollama-chat"> = buildModel({
id: "qwen3.6-coder:27b",
name: "Qwen 3.6 Coder 27B",
api: "ollama-chat",
provider: "ollama",
baseUrl: "http://127.0.0.1:11434",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 131_072,
maxTokens: 8192,
});
const context: Context = {
messages: [{ role: "user", content: "Use bash to run a Python command.", timestamp: 0 }],
};
const llamaToolParseFailure = JSON.stringify({
error: {
code: 500,
message:
"Failed to parse tool call arguments as JSON: [json.exception.parse_error.101] parse error at line 1, column 557: syntax error while parsing value - invalid string: missing closing quote; last read: '\"uv run python -c \\\"\\nimport jax.numpy as jnp\\nimport'",
},
});
describe("Ollama malformed tool-call JSON errors", () => {
afterEach(() => vi.restoreAllMocks());
it("does not retry deterministic llama.cpp tool argument parse failures", async () => {
vi.spyOn(scheduler, "wait").mockResolvedValue(undefined);
let calls = 0;
const fetchMock = async () => {
calls += 1;
return new Response(llamaToolParseFailure, { status: 500 });
};
const result = await streamOllama(model, context, {
apiKey: "ollama",
fetch: fetchMock,
}).result();
expect(calls).toBe(1);
expect(result.stopReason).toBe("error");
expect(result.errorStatus).toBe(500);
expect(result.errorMessage).toContain("Local Ollama model emitted malformed tool-call JSON");
expect(result.errorMessage).toContain("reload the model");
});
it("strips Transient so agent-level auto-retry will not replay the deterministic failure", async () => {
vi.spyOn(scheduler, "wait").mockResolvedValue(undefined);
const fetchMock = async () => new Response(llamaToolParseFailure, { status: 500 });
const result = await streamOllama(model, context, {
apiKey: "ollama",
fetch: fetchMock,
}).result();
expect(AIError.is(result.errorId, AIError.Flag.Transient)).toBe(false);
expect(AIError.retriable(result.errorId)).toBe(false);
});
});
@@ -83,21 +83,21 @@ function createCodexFetchMock(sse: string, onRequest: (captured: CapturedCodexRe
}
describe("openai-codex reasoning.context", () => {
it("forwards an explicit reasoning.context and defaults to all_turns", async () => {
const model = createCodexModel("gpt-5.1-codex");
it("defaults to all_turns on gpt-5.4+ models and forwards explicit overrides", async () => {
const model = createCodexModel("gpt-5.4");
const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" });
expect(defaulted.reasoning?.context).toBe("all_turns");
const explicit = await transformRequestBody({ model: model.id }, model, {
reasoningEffort: "medium",
reasoningContext: "current_turn",
});
expect(explicit.reasoning?.context).toBe("current_turn");
const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" });
expect(defaulted.reasoning?.context).toBe("all_turns");
});
it("defaults reasoning.context to all_turns under Responses Lite unless overridden", async () => {
const model = createCodexModel("gpt-5.1-codex");
it("keeps the all_turns default for the lite transport on supported models", async () => {
const model = createCodexModel("gpt-5.5");
const lite = await transformRequestBody({ model: model.id }, model, {
reasoningEffort: "medium",
@@ -112,6 +112,39 @@ describe("openai-codex reasoning.context", () => {
});
expect(overridden.reasoning?.context).toBe("auto");
});
// gpt-5.1-codex / gpt-5.3-codex / gpt-5.3-codex-spark reject `all_turns`
// ("Unsupported value: 'all_turns' is not supported with this model").
it.each([
"gpt-5.1-codex",
"gpt-5.3-codex",
"gpt-5.3-codex-spark",
])("omits the all_turns default for pre-5.4 model %s", async modelId => {
const model = createCodexModel(modelId);
const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" });
expect(defaulted.reasoning).toBeDefined();
expect(defaulted.reasoning?.context).toBeUndefined();
expect("context" in (defaulted.reasoning ?? {})).toBe(false);
// A supported override (current_turn/auto) is still honored.
const overridden = await transformRequestBody({ model: model.id }, model, {
reasoningEffort: "medium",
reasoningContext: "current_turn",
});
expect(overridden.reasoning?.context).toBe("current_turn");
});
it("suppresses an explicit all_turns override on a pre-5.4 model", async () => {
const model = createCodexModel("gpt-5.3-codex-spark");
const forced = await transformRequestBody({ model: model.id }, model, {
reasoningEffort: "medium",
reasoningContext: "all_turns",
});
expect(forced.reasoning).toBeDefined();
expect(forced.reasoning?.context).toBeUndefined();
});
});
describe("openai-codex Responses Lite input shaping", () => {
@@ -2,6 +2,7 @@ import { describe, expect, it } from "bun:test";
import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions";
import { NON_VISION_IMAGE_PLACEHOLDER } from "@oh-my-pi/pi-ai/providers/vision-guard";
import type { AssistantMessage, Context, Model, ToolResultMessage, Usage } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types";
@@ -75,6 +76,48 @@ function buildToolResult(toolCallId: string, timestamp: number): ToolResultMessa
}
describe("openai-completions convertMessages", () => {
it("serializes Cerebras gemma image inputs as Chat Completions data URIs", () => {
const model = buildModel({
id: "gemma-4-31b",
name: "Gemma 4 31B",
api: "openai-completions",
provider: "cerebras",
baseUrl: "https://api.cerebras.ai/v1",
reasoning: false,
input: ["text", "image"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128_000,
maxTokens: 8_192,
});
const context: Context = {
messages: [
{
role: "user",
content: [
{ type: "text", text: "Identify the shapes and colors. Return JSON only." },
{ type: "image", mimeType: "image/png", data: "ZmFrZQ==" },
],
timestamp: 1,
},
],
};
const messages = convertMessages(model, context, compat);
expect(messages).toEqual([
{
role: "user",
content: [
{ type: "text", text: "Identify the shapes and colors. Return JSON only." },
{
type: "image_url",
image_url: { url: "data:image/png;base64,ZmFrZQ==" },
},
],
},
]);
});
it("batches tool-result images after consecutive tool results", () => {
const baseModel = getBundledModel("openai", "gpt-4o-mini") as Model<"openai-completions">;
const model: Model<"openai-completions"> = {
@@ -1,138 +1,149 @@
import { describe, expect, it } from "bun:test";
import { getPriorityPremiumRequests, resolveServiceTier, shouldSendServiceTier } from "@oh-my-pi/pi-ai/types";
import type { Api } from "@oh-my-pi/pi-ai/types";
import {
coerceServiceTierByFamily,
getPriorityPremiumRequests,
realizesPriorityServiceTier,
resolveModelServiceTier,
serviceTierFamily,
shouldSendServiceTier,
} from "@oh-my-pi/pi-ai/types";
describe("getPriorityPremiumRequests", () => {
it("counts priority tier as one premium request on OpenAI", () => {
expect(getPriorityPremiumRequests("priority", "openai")).toBe(1);
const m = (provider: string, api: Api, id: string): { provider: string; api: Api; id: string } => ({
provider,
api,
id,
});
const openai = m("openai", "openai-responses", "gpt-5");
const codex = m("openai-codex", "openai-codex-responses", "gpt-5.5");
const anthropic = m("anthropic", "anthropic-messages", "claude-opus-4-6");
const vertexClaude = m("google-vertex", "anthropic-messages", "claude-opus-4-6");
const gemini = m("google", "google-generative-ai", "gemini-3-flash");
const vertexGemini = m("google-vertex", "google-vertex", "gemini-3-flash");
const fireworks = m("fireworks", "openai-completions", "qwen3");
const orOpenAI = m("openrouter", "openai-responses", "openai/gpt-5.5");
const orGoogle = m("openrouter", "openai-completions", "google/gemini-3-flash");
const orAnthropic = m("openrouter", "openai-completions", "anthropic/claude-opus-4-6");
describe("serviceTierFamily", () => {
it("classifies first-party providers by provider/api", () => {
expect(serviceTierFamily(openai)).toBe("openai");
expect(serviceTierFamily(codex)).toBe("openai");
expect(serviceTierFamily(anthropic)).toBe("anthropic");
expect(serviceTierFamily(vertexClaude)).toBe("anthropic"); // Claude on Vertex is the anthropic family
expect(serviceTierFamily(gemini)).toBe("google");
expect(serviceTierFamily(vertexGemini)).toBe("google");
expect(serviceTierFamily(fireworks)).toBeUndefined();
});
it("counts priority tier as one premium request on OpenAI Codex", () => {
expect(getPriorityPremiumRequests("priority", "openai-codex")).toBe(1);
});
it("ignores non-priority paid tiers", () => {
expect(getPriorityPremiumRequests("flex", "openai")).toBe(0);
expect(getPriorityPremiumRequests("scale", "openai")).toBe(0);
});
it("ignores default and auto tiers", () => {
expect(getPriorityPremiumRequests("default", "openai")).toBe(0);
expect(getPriorityPremiumRequests("auto", "openai")).toBe(0);
});
it("ignores priority tier on providers that drop service_tier", () => {
// `priority` is realized on `openai`, `openai-codex`, and direct `anthropic`
// (as fast mode). Everywhere else it's silently dropped, so it must not
// be billed as premium.
expect(getPriorityPremiumRequests("priority", "github-copilot")).toBe(0);
expect(getPriorityPremiumRequests("priority", "azure")).toBe(0);
expect(getPriorityPremiumRequests("priority", "bedrock")).toBe(0);
});
it("counts priority on direct Anthropic as one premium request (fast mode)", () => {
expect(getPriorityPremiumRequests("priority", "anthropic")).toBe(1);
});
it("returns zero when service tier is unset", () => {
expect(getPriorityPremiumRequests(undefined, "openai")).toBe(0);
expect(getPriorityPremiumRequests(null, "openai")).toBe(0);
});
describe("scoped tiers", () => {
it("treats `openai-only` as priority on OpenAI and OpenAI-Codex", () => {
expect(getPriorityPremiumRequests("openai-only", "openai")).toBe(1);
expect(getPriorityPremiumRequests("openai-only", "openai-codex")).toBe(1);
});
it("treats `openai-only` as inactive on Anthropic and everywhere else", () => {
expect(getPriorityPremiumRequests("openai-only", "anthropic")).toBe(0);
expect(getPriorityPremiumRequests("openai-only", "github-copilot")).toBe(0);
expect(getPriorityPremiumRequests("openai-only", "bedrock")).toBe(0);
});
it("treats `claude-only` as priority on direct Anthropic", () => {
expect(getPriorityPremiumRequests("claude-only", "anthropic")).toBe(1);
});
it("treats `claude-only` as inactive on OpenAI, Bedrock/Vertex, and elsewhere", () => {
expect(getPriorityPremiumRequests("claude-only", "openai")).toBe(0);
expect(getPriorityPremiumRequests("claude-only", "openai-codex")).toBe(0);
expect(getPriorityPremiumRequests("claude-only", "bedrock")).toBe(0);
expect(getPriorityPremiumRequests("claude-only", "vertex")).toBe(0);
});
it("classifies OpenRouter models by id namespace", () => {
expect(serviceTierFamily(orOpenAI)).toBe("openai");
expect(serviceTierFamily(orGoogle)).toBe("google");
expect(serviceTierFamily(orAnthropic)).toBe("anthropic");
expect(serviceTierFamily(m("openrouter", "openai-completions", "z-ai/glm-4.7"))).toBeUndefined();
});
});
describe("resolveServiceTier", () => {
it("passes unscoped tiers through unchanged for any provider", () => {
expect(resolveServiceTier("flex", "openai")).toBe("flex");
expect(resolveServiceTier("priority", "anthropic")).toBe("priority");
expect(resolveServiceTier("auto", "openai-codex")).toBe("auto");
expect(resolveServiceTier("default", "github-copilot")).toBe("default");
});
it("scopes `openai-only` to OpenAI providers", () => {
expect(resolveServiceTier("openai-only", "openai")).toBe("priority");
expect(resolveServiceTier("openai-only", "openai-codex")).toBe("priority");
expect(resolveServiceTier("openai-only", "anthropic")).toBeUndefined();
expect(resolveServiceTier("openai-only", "bedrock")).toBeUndefined();
expect(resolveServiceTier("openai-only", undefined)).toBeUndefined();
});
it("scopes `claude-only` to direct Anthropic", () => {
expect(resolveServiceTier("claude-only", "anthropic")).toBe("priority");
expect(resolveServiceTier("claude-only", "openai")).toBeUndefined();
expect(resolveServiceTier("claude-only", "bedrock")).toBeUndefined();
expect(resolveServiceTier("claude-only", "vertex")).toBeUndefined();
});
it("returns undefined for null/undefined input", () => {
expect(resolveServiceTier(undefined, "openai")).toBeUndefined();
expect(resolveServiceTier(null, "openai")).toBeUndefined();
describe("resolveModelServiceTier", () => {
it("reduces a per-family map to the model's family entry", () => {
const tiers = { openai: "priority", anthropic: "priority", google: "flex" } as const;
expect(resolveModelServiceTier(tiers, openai)).toBe("priority");
expect(resolveModelServiceTier(tiers, gemini)).toBe("flex");
expect(resolveModelServiceTier(tiers, orAnthropic)).toBe("priority");
expect(resolveModelServiceTier(tiers, fireworks)).toBeUndefined(); // no family
expect(resolveModelServiceTier(undefined, openai)).toBeUndefined();
expect(resolveModelServiceTier({ google: "priority" }, openai)).toBeUndefined();
});
});
describe("shouldSendServiceTier", () => {
it("returns false for non-OpenAI/non-Fireworks providers", () => {
expect(shouldSendServiceTier("flex", "azure-openai-responses")).toBe(false);
expect(shouldSendServiceTier("scale", "firepass")).toBe(false);
it("sends flex/scale/priority on the OpenAI family and OpenRouter", () => {
for (const p of ["openai", "openai-codex", "openrouter"]) {
expect(shouldSendServiceTier("flex", p)).toBe(true);
expect(shouldSendServiceTier("scale", p)).toBe(true);
expect(shouldSendServiceTier("priority", p)).toBe(true);
expect(shouldSendServiceTier("default", p)).toBe(false);
expect(shouldSendServiceTier("auto", p)).toBe(false);
}
});
it("sends flex/priority on direct Google, priority-only on Vertex (no scale)", () => {
expect(shouldSendServiceTier("flex", "google")).toBe(true);
expect(shouldSendServiceTier("priority", "google")).toBe(true);
expect(shouldSendServiceTier("scale", "google")).toBe(false);
expect(shouldSendServiceTier("priority", "google-vertex")).toBe(true);
expect(shouldSendServiceTier("flex", "google-vertex")).toBe(false); // Vertex flex has no wire control
});
it("sends only priority on Fireworks, nothing on Anthropic", () => {
expect(shouldSendServiceTier("priority", "fireworks")).toBe(true);
expect(shouldSendServiceTier("flex", "fireworks")).toBe(false);
expect(shouldSendServiceTier("priority", "anthropic")).toBe(false);
});
it("returns true for fireworks only with the priority tier", () => {
expect(shouldSendServiceTier("priority", "fireworks")).toBe(true);
// Fireworks realizes only the Priority serving path — flex/scale are OpenAI-only.
expect(shouldSendServiceTier("flex", "fireworks")).toBe(false);
expect(shouldSendServiceTier("scale", "fireworks")).toBe(false);
expect(shouldSendServiceTier("auto", "fireworks")).toBe(false);
expect(shouldSendServiceTier("default", "fireworks")).toBe(false);
expect(shouldSendServiceTier(undefined, "fireworks")).toBe(false);
});
it("returns true for openai with priority/flex/scale tiers", () => {
expect(shouldSendServiceTier("priority", "openai")).toBe(true);
expect(shouldSendServiceTier("flex", "openai")).toBe(true);
expect(shouldSendServiceTier("scale", "openai")).toBe(true);
});
it("returns true for openai-codex with priority/flex/scale tiers", () => {
expect(shouldSendServiceTier("priority", "openai-codex")).toBe(true);
expect(shouldSendServiceTier("flex", "openai-codex")).toBe(true);
expect(shouldSendServiceTier("scale", "openai-codex")).toBe(true);
});
it("returns false for default tier on OpenAI providers", () => {
expect(shouldSendServiceTier("default", "openai")).toBe(false);
expect(shouldSendServiceTier("default", "openai-codex")).toBe(false);
});
it("returns false for auto tier on OpenAI providers", () => {
expect(shouldSendServiceTier("auto", "openai")).toBe(false);
expect(shouldSendServiceTier("auto", "openai-codex")).toBe(false);
});
it("returns false for undefined/null tier", () => {
it("returns false for unset tiers", () => {
expect(shouldSendServiceTier(undefined, "openai")).toBe(false);
expect(shouldSendServiceTier(null, "openai")).toBe(false);
});
});
describe("realizesPriorityServiceTier", () => {
it("realizes priority where the wire actually applies it", () => {
expect(realizesPriorityServiceTier("priority", openai)).toBe(true);
expect(realizesPriorityServiceTier("priority", anthropic)).toBe(true); // direct fast mode
expect(realizesPriorityServiceTier("priority", gemini)).toBe(true);
expect(realizesPriorityServiceTier("priority", vertexGemini)).toBe(true);
expect(realizesPriorityServiceTier("priority", fireworks)).toBe(true);
expect(realizesPriorityServiceTier("priority", orOpenAI)).toBe(true);
expect(realizesPriorityServiceTier("priority", orGoogle)).toBe(true);
});
it("does not realize priority where the wire drops it", () => {
expect(realizesPriorityServiceTier("priority", vertexClaude)).toBe(false); // no fast mode on Vertex
expect(realizesPriorityServiceTier("priority", orAnthropic)).toBe(false); // OpenRouter Anthropic
expect(realizesPriorityServiceTier("flex", openai)).toBe(false);
expect(realizesPriorityServiceTier(undefined, openai)).toBe(false);
});
});
describe("getPriorityPremiumRequests", () => {
it("counts one premium request per realized priority on billing providers", () => {
expect(getPriorityPremiumRequests("priority", openai)).toBe(1);
expect(getPriorityPremiumRequests("priority", codex)).toBe(1);
expect(getPriorityPremiumRequests("priority", anthropic)).toBe(1);
expect(getPriorityPremiumRequests("priority", gemini)).toBe(1);
expect(getPriorityPremiumRequests("priority", vertexGemini)).toBe(1);
});
it("does not bill OpenRouter, unrealized, or non-priority traffic", () => {
expect(getPriorityPremiumRequests("priority", orOpenAI)).toBe(0); // OpenRouter bills its own way
expect(getPriorityPremiumRequests("priority", vertexClaude)).toBe(0); // not realized
expect(getPriorityPremiumRequests("priority", fireworks)).toBe(0); // realized but not Copilot-premium
expect(getPriorityPremiumRequests("flex", openai)).toBe(0);
expect(getPriorityPremiumRequests(undefined, openai)).toBe(0);
});
});
describe("coerceServiceTierByFamily", () => {
it("migrates legacy scalar values to a per-family map", () => {
expect(coerceServiceTierByFamily("priority")).toEqual({
openai: "priority",
anthropic: "priority",
google: "priority",
});
expect(coerceServiceTierByFamily("openai-only")).toEqual({ openai: "priority" });
expect(coerceServiceTierByFamily("claude-only")).toEqual({ anthropic: "priority" });
expect(coerceServiceTierByFamily("flex")).toEqual({ openai: "flex" });
expect(coerceServiceTierByFamily("none")).toBeUndefined();
expect(coerceServiceTierByFamily(null)).toBeUndefined();
});
it("passes a per-family map through, dropping invalid entries", () => {
expect(coerceServiceTierByFamily({ openai: "priority", google: "flex" })).toEqual({
openai: "priority",
google: "flex",
});
expect(coerceServiceTierByFamily({ openai: "bogus" })).toBeUndefined();
});
});
+10 -4
View File
@@ -153,9 +153,9 @@ describe("streamSimple resolver auth retry", () => {
expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]);
expect(keys).toEqual(["old-key", "new-key"]);
// The buffered `start` of the failed attempt must not leak — the user
// sees exactly one clean start/done pair.
expect(eventTypes).toEqual(["start", "done"]);
// The failed attempt's buffered start must not leak — the user sees a
// single start from the successful attempt, then its healed content.
expect(eventTypes).toEqual(["start", "text_start", "text_delta", "text_end", "done"]);
});
it("retries on a 401 carried only via errorStatus", async () => {
@@ -206,6 +206,12 @@ describe("streamSimple resolver auth retry", () => {
queueMicrotask(() => {
stream.push({ type: "start", partial: assistant() });
stream.push({ type: "text_start", contentIndex: 0, partial: assistant([""]) });
stream.push({
type: "text_delta",
contentIndex: 0,
delta: "partial",
partial: assistant(["partial"]),
});
stream.fail(failure);
});
return stream;
@@ -398,7 +404,7 @@ describe("streamSimple resolver auth retry", () => {
expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]);
expect(keys).toEqual(["credential-A", "credential-B"]);
expect(eventTypes).toEqual(["start", "done"]);
expect(eventTypes).toEqual(["start", "text_start", "text_delta", "text_end", "done"]);
expect(retryContexts.map(ctx => ctx.lastChance)).toEqual([false, true]);
}
});
+86 -3
View File
@@ -5,9 +5,10 @@ import {
type InbandScanEvent,
ThinkingInbandScanner,
} from "@oh-my-pi/pi-ai/dialect";
import { streamGoogleGeminiCli } from "@oh-my-pi/pi-ai/providers/google-gemini-cli";
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import { stream } from "@oh-my-pi/pi-ai/stream";
import type { Context, FetchImpl, Model, ThinkingContent, Tool, ToolCall } from "@oh-my-pi/pi-ai/types";
import type { Context, FetchImpl, Model, TextContent, ThinkingContent, Tool, ToolCall } from "@oh-my-pi/pi-ai/types";
import { getStreamMarkupHealingPattern, StreamMarkupHealing } from "@oh-my-pi/pi-ai/utils/stream-markup-healing";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
@@ -38,7 +39,7 @@ interface SseChunk {
}>;
}
function sseResponse(events: ReadonlyArray<SseChunk | "[DONE]">): Response {
function sseResponse(events: ReadonlyArray<unknown | "[DONE]">): Response {
const payload = `${events
.map(event => `data: ${typeof event === "string" ? event : JSON.stringify(event)}`)
.join("\n\n")}\n\n`;
@@ -48,7 +49,7 @@ function sseResponse(events: ReadonlyArray<SseChunk | "[DONE]">): Response {
});
}
function mockFetch(events: ReadonlyArray<SseChunk | "[DONE]">): FetchImpl {
function mockFetch(events: ReadonlyArray<unknown | "[DONE]">): FetchImpl {
const fn = async (_input: string | URL | Request, _init?: RequestInit): Promise<Response> => sseResponse(events);
return Object.assign(fn, { preconnect: fetch.preconnect });
}
@@ -124,6 +125,21 @@ const deepseekCloudModel: Model<"ollama-chat"> = buildModel({
maxTokens: 8_192,
});
function geminiCliModel(): Model<"google-gemini-cli"> {
return buildModel({
id: "gemini-3.5-flash",
name: "Gemini 3.5 Flash",
api: "google-gemini-cli",
provider: "google-antigravity",
baseUrl: "https://antigravity.test",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 1_000_000,
maxTokens: 8_192,
});
}
function ndjsonResponse(lines: ReadonlyArray<unknown>): Response {
const body = `${lines.map(line => JSON.stringify(line)).join("\n")}\n`;
const encoder = new TextEncoder();
@@ -230,6 +246,73 @@ describe("openai-completions leaked thinking healing", () => {
});
});
describe("google-gemini-cli leaked thinking healing", () => {
it("lifts a leaked Gemini thinking fence before a native tool call", async () => {
const model = geminiCliModel();
const fetchMock = mockFetch([
{
response: {
candidates: [
{
content: {
role: "model",
parts: [
{
text: "```thinking\nCheck the provider path.\n```\nI will inspect the file.",
thoughtSignature: "visible-text-signature",
},
{
functionCall: {
name: "read",
args: { path: "packages/ai/src/providers/google-gemini-cli.ts" },
id: "call_read_1",
},
thoughtSignature: "function-call-signature",
},
],
},
finishReason: "STOP",
},
],
usageMetadata: {
promptTokenCount: 10,
candidatesTokenCount: 5,
thoughtsTokenCount: 3,
totalTokenCount: 18,
},
},
},
]);
const result = await streamGoogleGeminiCli(
model,
{ ...baseContext(), tools: [readTool] },
{
apiKey: JSON.stringify({ token: "test-token", projectId: "test-project" }),
fetch: fetchMock,
},
).result();
expect(result.content.map(block => block.type)).toEqual(["thinking", "text", "toolCall"]);
const thinking = result.content
.filter((block): block is ThinkingContent => block.type === "thinking")
.map(block => block.thinking)
.join("");
const textBlocks = result.content.filter((block): block is TextContent => block.type === "text");
const text = textBlocks.map(block => block.text).join("");
const calls = result.content.filter((block): block is ToolCall => block.type === "toolCall");
expect(thinking).toBe("Check the provider path.\n");
expect(text).toBe("\nI will inspect the file.");
expect(text).not.toContain("```thinking");
expect(calls).toHaveLength(1);
expect(textBlocks[0]?.textSignature).toBe("visible-text-signature");
expect(calls[0]?.id).toBe("call_read_1");
expect(calls[0]?.thoughtSignature).toBe("function-call-signature");
expect(result.stopReason).toBe("toolUse");
});
});
describe("StreamMarkupHealing DSML envelope pattern", () => {
it("parses the reporter's verbatim leak into a structured tool call", () => {
const healing = new StreamMarkupHealing({ pattern: "dsml" });
+17
View File
@@ -2,6 +2,23 @@
## [Unreleased]
## [16.2.9] - 2026-06-30
### Added
- Added full capability support for Claude Sonnet 5, aligning it with Claude Opus 4.8 and Fable 5. This includes adaptive thinking display, mid-conversation system messages, sampling parameter and thinking omission API restrictions, and 5-tier adaptive reasoning effort mapping (including xhigh and max levels) across direct APIs, OpenRouter, and Bedrock Converse.
### Changed
- Updated input and output costs for models in the catalog.
## [16.2.7] - 2026-06-30
### Fixed
- Fixed compatibility with Kimi K2.7 Code on native endpoints to ensure thinking mode is preserved and tool choice is not forced.
- Fixed Cerebras gemma-4-31b dynamic discovery to correctly identify the model as image-capable, enabling proper serialization of attached images.
## [16.2.6] - 2026-06-29
### Fixed
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-catalog",
"version": "16.2.6",
"version": "16.2.9",
"description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+11 -1
View File
@@ -28,6 +28,14 @@ export function isOfficialAnthropicApiUrl(baseUrl?: string): boolean {
return lower === OFFICIAL_ANTHROPIC_URL || lower.startsWith(`${OFFICIAL_ANTHROPIC_URL}/`);
}
/** Mirrors `compat/openai.ts`; native-only host gating is the caller's responsibility. */
const KIMI_K27_CODE_MODEL_PATTERN = /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code(?:[-._]?highspeed)?$/i;
function matchesKimiK27CodeFamily(spec: ModelSpec<"anthropic-messages">): boolean {
if (KIMI_K27_CODE_MODEL_PATTERN.test(spec.id)) return true;
return spec.id === "kimi-for-coding" && /k2\.?7 code/i.test(spec.name ?? "");
}
/** Build the resolved anthropic-messages compat record for a model spec. */
export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): ResolvedAnthropicCompat {
const baseUrl = spec.baseUrl;
@@ -40,6 +48,7 @@ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): Res
// doesn't whitelist the `fine-grained-tool-streaming-2025-05-14` beta either
// (issue #2558), so eager tool-input streaming is unavailable on this host.
const isCopilot = modelMatchesHost(spec, "githubCopilot");
const requiresThinkingEnabled = modelMatchesHost(spec, "moonshotNative") && matchesKimiK27CodeFamily(spec);
const compat: ResolvedAnthropicCompat = {
officialEndpoint: official,
disableStrictTools: false,
@@ -53,12 +62,13 @@ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): Res
// detection requires the canonical api.anthropic.com host plus a
// supported model id.
supportsMidConversationSystem: official && supportsMidConversationSystemMessages(spec.id),
supportsForcedToolChoice: !isAnthropicFableOrMythosModel(spec.id),
supportsForcedToolChoice: !requiresThinkingEnabled && !isAnthropicFableOrMythosModel(spec.id),
// Opus 4.7+ and Fable/Mythos reject temperature/top_p/top_k with a 400.
supportsSamplingParams: !hasOpus47ApiRestrictions(spec.id),
// Z.AI workaround (issue #814): its proxy deserializes tool_result blocks
// into a class that reads `.id`.
requiresToolResultId: isZai,
requiresThinkingEnabled,
// Official Anthropic enforces signature-based thinking-chain integrity, so
// unsigned thinking blocks must stay text there. Anthropic-compatible
// reasoning endpoints commonly emit unsigned thinking blocks while still
+19 -2
View File
@@ -39,6 +39,20 @@ const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000;
const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
/** Kimi K2.6 can spend several minutes reasoning before the first visible token. */
const KIMI_K26_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
/**
* Native Kimi K2.7 Code requires `thinking.type: "enabled"` and rejects
* disabled thinking. Match the public id, its Fast variant, and the
* `kimi-code/kimi-for-coding` alias (which keeps the family name).
* Caller-disabled requests on non-native dialects (Fireworks `openai`,
* OpenRouter `openrouter`, …) MUST keep their per-dialect disable shape —
* gating on `isMoonshotKimi` is the caller's responsibility.
*/
const KIMI_K27_CODE_MODEL_PATTERN = /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code(?:[-._]?highspeed)?$/i;
function matchesKimiK27CodeFamily(spec: ModelSpec<"openai-completions">): boolean {
if (KIMI_K27_CODE_MODEL_PATTERN.test(spec.id)) return true;
return spec.id === "kimi-for-coding" && /k2\.?7 code/i.test(spec.name ?? "");
}
/** Xiaomi MiMo Pro on api.xiaomimimo.com can stall ~2min before the first event (issue #1770). */
const XIAOMI_MIMO_STREAM_IDLE_TIMEOUT_MS = 300_000;
/** Alibaba Coding Plan (coding-intl.dashscope) qwen models idle before the first event (issue #1770). */
@@ -231,6 +245,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
const isKimiModel = isKimiModelId(spec.id);
const isMoonshotNative = modelMatchesHost(hostModel, "moonshotNative");
const isMoonshotKimi = isKimiModel && isMoonshotNative;
const requiresEnabledThinking = isMoonshotKimi && matchesKimiK27CodeFamily(spec);
const usesMoonshotKimiPreservedThinking = isMoonshotKimi && isKimiK26ModelId(spec.id);
const isAnthropicModel =
modelMatchesHost(hostModel, "anthropic") || isClaudeModelId(spec.id) || isAnthropicNamespacedModelId(spec.id);
@@ -403,7 +418,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel,
disableReasoningOnToolChoice: isDeepseekFamily && Boolean(spec.reasoning) && !isOpenRouter,
supportsToolChoice: !isDirectDeepseekReasoning,
supportsForcedToolChoice: true,
supportsForcedToolChoice: !requiresEnabledThinking,
supportsNamedToolChoice: provider !== "llama.cpp",
maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens",
requiresToolResultName: isMistral,
@@ -509,7 +524,9 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
applyCompatOverrides(compat, spec.compat);
if (spec.compat?.reasoningDisableMode === undefined) {
compat.reasoningDisableMode = resolveReasoningDisableMode(compat.thinkingFormat);
compat.reasoningDisableMode = requiresEnabledThinking
? "omit"
: resolveReasoningDisableMode(compat.thinkingFormat);
}
if (spec.compat?.omitReasoningEffort === undefined && !compat.supportsReasoningEffort) {
compat.omitReasoningEffort = true;
+14
View File
@@ -158,6 +158,20 @@ export function isFableOrMythos(kind: AnthropicKind): boolean {
return kind === "fable" || kind === "mythos";
}
/**
* Returns true if the parsed Anthropic model is part of the adaptive-thinking
* Claude generation at or above a specific capability threshold.
* - Opus has a configurable minimum version floor (e.g. "4.6", "4.7", "4.8").
* - Sonnet, Fable, and Mythos all require version 5 or higher.
*/
export function isAnthropicAdaptiveGenAtLeast(parsed: AnthropicModel, opusMin: "4.6" | "4.7" | "4.8"): boolean {
if (parsed.kind === "opus") {
return semverGte(parsed.version, opusMin);
}
// Sonnet 5+, Fable 5+, Mythos 5+, and any future gen-5+ models
return semverGte(parsed.version, "5");
}
function createSemVer(major: number, minor: number, patch = 0): SemVer {
return { major, minor, patch };
}
+29 -14
View File
@@ -9,10 +9,12 @@
import {
bareModelId,
isAnthropicAdaptiveGenAtLeast,
isFableOrMythos,
parseAnthropicModel,
parseGlmModel,
parseKnownModel,
parseOpenAIModel,
semverGte,
} from "./classify";
@@ -121,6 +123,22 @@ export const isOpenAIModelId = memo((modelId: string): boolean => {
return /(^|\/)(gpt|o1|o3|o4)[-.]/i.test(modelId) || modelId.toLowerCase().includes("openai/");
});
/**
* OpenAI Codex models that honor `reasoning.context: "all_turns"` (full
* cross-turn reasoning replay). The `reasoning.context` field itself exists for
* the whole gpt-5/o-series family, but the `all_turns` value is only accepted
* from gpt-5.4 onward; earlier ids (`gpt-5.1-codex`, `gpt-5.3-codex`, and
* `gpt-5.3-codex-spark`) reject it with
* `Unsupported value: 'all_turns' is not supported with this model`. Version
* floor (not an allowlist) so 5.6/6.x inherit support automatically. Callers
* fall back to omitting `context`, letting the server default to `current_turn`.
*/
export const supportsAllTurnsReasoningContext = memo((modelId: string): boolean => {
const parsed = parseOpenAIModel(bareModelId(modelId));
if (!parsed) return false;
return semverGte(parsed.version, "5.4");
});
/**
* Reasoning-capable GLM coding SKUs: glm-4.5 and up on the base / `-air` /
* `-turbo` lines. Excludes the vision (`…v`) shape, the non-reasoning
@@ -184,41 +202,38 @@ export const modelFamilyToken = memo((modelId: string): string => {
});
/**
* Adaptive thinking `display` is supported starting with Claude Opus 4.7 and
* the Claude Fable/Mythos 5 generation. Older adaptive-thinking models
* (Opus 4.6, Sonnet 4.6+) reject the field. Classifier-based, so dotted and
* dashed version forms both match while bare dated ids
* Adaptive thinking `display` is supported starting with Claude Opus 4.7+,
* Sonnet 5+, and the Claude Fable/Mythos 5 generation. Older adaptive-thinking
* models (Opus 4.6, Sonnet 4.6) reject the field. Classifier-based, so dotted
* and dashed version forms both match while bare dated ids
* (`claude-opus-4-20250514` = Opus 4.0) stay excluded.
*/
export const supportsAdaptiveThinkingDisplay = memo((modelId: string): boolean => {
const parsed = parseAnthropicModel(bareModelId(modelId));
if (!parsed) return false;
if (isFableOrMythos(parsed.kind)) return semverGte(parsed.version, "5");
return parsed.kind === "opus" && semverGte(parsed.version, "4.7");
return parsed !== null && isAnthropicAdaptiveGenAtLeast(parsed, "4.7");
});
/**
* Returns true for Anthropic models with Opus 4.7+/Fable/Mythos API restrictions:
* Returns true for Anthropic models with Opus 4.7+, Sonnet 5+, and Fable/Mythos 5+
* API restrictions:
* - Sampling parameters (temperature/top_p/top_k) return 400 error
* - Thinking content is omitted by default (needs display: "summarized")
*/
export const hasOpus47ApiRestrictions = memo((modelId: string): boolean => {
const parsed = parseAnthropicModel(bareModelId(modelId));
if (!parsed) return false;
return (parsed.kind === "opus" && semverGte(parsed.version, "4.7")) || isFableOrMythos(parsed.kind);
return parsed !== null && isAnthropicAdaptiveGenAtLeast(parsed, "4.7");
});
/**
* Mid-conversation `role: "system"` messages (system instructions appended at
* non-first positions in the `messages` array) are supported starting with
* Claude Opus 4.8 and the Claude Fable/Mythos 5 generation. Earlier Claude
* models reject the role.
* Claude Opus 4.8+, Sonnet 5+, and the Claude Fable/Mythos 5 generation.
* Earlier Claude models reject the role.
* @see https://platform.claude.com/docs/en/build-with-claude/mid-conversation-system-messages
*/
export const supportsMidConversationSystemMessages = memo((modelId: string): boolean => {
const parsed = parseAnthropicModel(bareModelId(modelId));
if (!parsed) return false;
return (parsed.kind === "opus" && semverGte(parsed.version, "4.8")) || isFableOrMythos(parsed.kind);
return parsed !== null && isAnthropicAdaptiveGenAtLeast(parsed, "4.8");
});
export const isAnthropicFableOrMythosModel = memo((modelId: string): boolean => {
+10 -15
View File
@@ -12,7 +12,7 @@ import {
type AnthropicModel,
bareModelId,
type GeminiModel,
isFableOrMythos,
isAnthropicAdaptiveGenAtLeast,
type OpenAIModel,
type ParsedModel,
parseAnthropicModel,
@@ -441,12 +441,11 @@ function isDeepseekReasoningModel<TApi extends Api>(spec: ModelSpec<TApi>): bool
function getOpenRouterAnthropicReasoningEffortMap(modelId: string): EffortMap | undefined {
const parsed = parseAnthropicModel(bareModelId(modelId));
if (!parsed) return undefined;
// Adaptive efforts on OpenRouter's completions front: Fable/Mythos and
// Opus 4.6+ only — Sonnet stays on the plain effort vocabulary there.
const isOpusAdaptive = parsed.kind === "opus" && semverGte(parsed.version, "4.6");
if (!isFableOrMythos(parsed.kind) && !isOpusAdaptive) return undefined;
// Adaptive efforts on OpenRouter's completions front: Fable/Mythos, Sonnet 5+,
// and Opus 4.6+ only — older Sonnet versions stay on the plain effort vocabulary there.
if (!isAnthropicAdaptiveGenAtLeast(parsed, "4.6")) return undefined;
const hasRealXHigh = isFableOrMythos(parsed.kind) || semverGte(parsed.version, "4.7");
const hasRealXHigh = isAnthropicAdaptiveGenAtLeast(parsed, "4.7");
return hasRealXHigh ? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER;
}
@@ -521,7 +520,7 @@ function inferAnthropicSupportedEfforts<TApi extends Api>(
(spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") &&
semverGte(parsedModel.version, "4.6")
) {
return parsedModel.kind === "opus" || isFableOrMythos(parsedModel.kind)
return isAnthropicAdaptiveGenAtLeast(parsedModel, "4.6")
? DEFAULT_REASONING_EFFORTS_WITH_XHIGH
: DEFAULT_REASONING_EFFORTS;
}
@@ -601,10 +600,7 @@ function inferThinkingControlMode<TApi extends Api>(
case "bedrock-converse-stream":
if (parsedModel.family === "anthropic") {
if (
semverGte(parsedModel.version, "4.6") &&
(parsedModel.kind === "opus" || isFableOrMythos(parsedModel.kind))
) {
if (isAnthropicAdaptiveGenAtLeast(parsedModel, "4.6")) {
return "anthropic-adaptive";
}
// Opus 4.5 on Bedrock metadata mirrors the direct-Anthropic
@@ -627,19 +623,18 @@ function isOpenRouterAnthropicAdaptiveReasoningModel<TApi extends Api>(
): boolean {
if (!isOpenAICompatReasoningApi(spec.api)) return false;
if (!modelMatchesHost(spec, "openrouter")) return false;
return isFableOrMythos(parsedModel.kind) || (parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.6"));
return isAnthropicAdaptiveGenAtLeast(parsedModel, "4.6");
}
/**
* Opus 4.7+ and Fable/Mythos on the Messages API expose the full five-tier
* Opus 4.7+, Sonnet 5+, and Fable/Mythos 5+ on the Messages API expose the full five-tier
* adaptive scale (low/medium/high/xhigh/max). Bedrock Converse stays on the
* four-tier scale regardless of model version.
*/
function anthropicModelHasRealXHighEffort<TApi extends Api>(spec: ModelSpec<TApi>, parsedModel: ParsedModel): boolean {
if (spec.api !== "anthropic-messages") return false;
if (parsedModel.family !== "anthropic") return false;
if (isFableOrMythos(parsedModel.kind)) return true;
return parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.7");
return isAnthropicAdaptiveGenAtLeast(parsedModel, "4.7");
}
// ---------------------------------------------------------------------------
+278 -262
View File
@@ -2476,7 +2476,7 @@
"api": "openai-completions",
"provider": "aimlapi",
"baseUrl": "https://api.aimlapi.com/v1",
"reasoning": true,
"reasoning": false,
"input": [
"text"
],
@@ -2487,24 +2487,7 @@
"cacheWrite": 0
},
"contextWindow": 163840,
"maxTokens": 16384,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"effortMap": {
"minimal": "high",
"low": "high",
"medium": "high",
"high": "high",
"xhigh": "max"
}
}
"maxTokens": 16384
},
"deepseek/deepseek-chat-v3.1": {
"id": "deepseek/deepseek-chat-v3.1",
@@ -10620,6 +10603,35 @@
]
}
},
"xai.grok-4.3": {
"id": "xai.grok-4.3",
"name": "Grok 4.3",
"api": "bedrock-converse-stream",
"provider": "amazon-bedrock",
"baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 1.25,
"output": 2.5,
"cacheRead": 0.2,
"cacheWrite": 0
},
"contextWindow": 1000000,
"maxTokens": 131072,
"thinking": {
"mode": "budget",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"zai.glm-4.7": {
"id": "zai.glm-4.7",
"name": "GLM-4.7",
@@ -12287,6 +12299,26 @@
}
},
"cerebras": {
"gemma-4-31b": {
"id": "gemma-4-31b",
"name": "gemma-4-31b",
"api": "openai-completions",
"provider": "cerebras",
"baseUrl": "https://api.cerebras.ai/v1",
"reasoning": false,
"input": [
"text",
"image"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 256000,
"maxTokens": 8192
},
"gpt-oss-120b": {
"id": "gpt-oss-120b",
"name": "GPT OSS 120B",
@@ -13672,7 +13704,7 @@
"api": "openai-completions",
"provider": "coreweave",
"baseUrl": "https://api.inference.wandb.ai/v1",
"reasoning": true,
"reasoning": false,
"input": [
"text"
],
@@ -13683,17 +13715,7 @@
"cacheWrite": 0
},
"contextWindow": 128000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
"maxTokens": 128000
},
"meta-llama/Llama-3.3-70B-Instruct": {
"id": "meta-llama/Llama-3.3-70B-Instruct",
@@ -13701,7 +13723,7 @@
"api": "openai-completions",
"provider": "coreweave",
"baseUrl": "https://api.inference.wandb.ai/v1",
"reasoning": true,
"reasoning": false,
"input": [
"text"
],
@@ -13712,17 +13734,7 @@
"cacheWrite": 0
},
"contextWindow": 128000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
"maxTokens": 128000
},
"meta-llama/Llama-4-Scout-17B-16E-Instruct": {
"id": "meta-llama/Llama-4-Scout-17B-16E-Instruct",
@@ -13730,7 +13742,7 @@
"api": "openai-completions",
"provider": "coreweave",
"baseUrl": "https://api.inference.wandb.ai/v1",
"reasoning": true,
"reasoning": false,
"input": [
"text",
"image"
@@ -13742,17 +13754,7 @@
"cacheWrite": 0
},
"contextWindow": 64000,
"maxTokens": 64000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
"maxTokens": 64000
},
"microsoft/Phi-4-mini-instruct": {
"id": "microsoft/Phi-4-mini-instruct",
@@ -13760,7 +13762,7 @@
"api": "openai-completions",
"provider": "coreweave",
"baseUrl": "https://api.inference.wandb.ai/v1",
"reasoning": true,
"reasoning": false,
"input": [
"text"
],
@@ -13771,17 +13773,7 @@
"cacheWrite": 0
},
"contextWindow": 128000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
"maxTokens": 128000
},
"MiniMaxAI/MiniMax-M2.5": {
"id": "MiniMaxAI/MiniMax-M2.5",
@@ -13789,7 +13781,7 @@
"api": "openai-completions",
"provider": "coreweave",
"baseUrl": "https://api.inference.wandb.ai/v1",
"reasoning": false,
"reasoning": true,
"input": [
"text"
],
@@ -13800,7 +13792,16 @@
"cacheWrite": 0
},
"contextWindow": 196608,
"maxTokens": 196608
"maxTokens": 196608,
"thinking": {
"mode": "effort",
"efforts": [
"low",
"medium",
"high"
],
"requiresEffort": true
}
},
"moonshotai/Kimi-K2.5": {
"id": "moonshotai/Kimi-K2.5",
@@ -13898,7 +13899,7 @@
"api": "openai-completions",
"provider": "coreweave",
"baseUrl": "https://api.inference.wandb.ai/v1",
"reasoning": false,
"reasoning": true,
"input": [
"text"
],
@@ -13909,7 +13910,17 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 262144
"maxTokens": 262144,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": {
"id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B",
@@ -15449,46 +15460,6 @@
"trustExplicitThinkingOnly": true
}
},
"claude-opus-4-6-fast": {
"id": "claude-opus-4-6-fast",
"name": "Claude Opus 4.6 Fast",
"api": "devin-agent",
"provider": "devin",
"baseUrl": "https://server.codeium.com",
"reasoning": true,
"input": [
"text",
"image"
],
"supportsTools": true,
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 200000,
"maxTokens": 64000,
"thinking": {
"mode": "budget",
"efforts": [
"minimal",
"low",
"medium",
"high"
],
"effortRouting": {
"off": "claude-opus-4-6-fast",
"minimal": "claude-opus-4-6-thinking-fast",
"low": "claude-opus-4-6-thinking-fast",
"medium": "claude-opus-4-6-thinking-fast",
"high": "claude-opus-4-6-thinking-fast"
}
},
"compat": {
"trustExplicitThinkingOnly": true
}
},
"claude-opus-4-7": {
"id": "claude-opus-4-7",
"name": "Claude Opus 4.7",
@@ -21734,7 +21705,7 @@
"cost": {
"input": 0.075,
"output": 0.3,
"cacheRead": 0.037,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 131072,
@@ -22420,13 +22391,13 @@
"text"
],
"cost": {
"input": 0,
"output": 0,
"input": 0.25,
"output": 0.69,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 131072,
"maxTokens": 65536,
"maxTokens": 32768,
"thinking": {
"mode": "effort",
"efforts": [
@@ -25048,7 +25019,7 @@
"api": "openai-completions",
"provider": "kilo",
"baseUrl": "https://api.kilo.ai/api/gateway",
"reasoning": true,
"reasoning": false,
"input": [
"text"
],
@@ -25059,24 +25030,7 @@
"cacheWrite": 0
},
"contextWindow": 163840,
"maxTokens": 65536,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"effortMap": {
"minimal": "high",
"low": "high",
"medium": "high",
"high": "high",
"xhigh": "max"
}
}
"maxTokens": 65536
},
"deepseek/deepseek-chat-v3.1": {
"id": "deepseek/deepseek-chat-v3.1",
@@ -30816,6 +30770,7 @@
"supportsStrictMode": false,
"toolStrictMode": "mixed",
"stripDeepseekSpecialTokens": false,
"streamMarkupHealingPattern": "thinking",
"reasoningDeltasMayBeCumulative": false,
"emptyLengthFinishIsContextError": false,
"usesOpenAIToolCallIdLimit": false,
@@ -50500,7 +50455,7 @@
"api": "openai-completions",
"provider": "nanogpt",
"baseUrl": "https://nano-gpt.com/api/v1",
"reasoning": false,
"reasoning": true,
"input": [
"text"
],
@@ -50511,7 +50466,20 @@
"cacheWrite": 0
},
"contextWindow": 1048576,
"maxTokens": 131072
"maxTokens": 131072,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"effortMap": {
"xhigh": "max"
}
}
},
"TEE/gpt-oss-120b": {
"id": "TEE/gpt-oss-120b",
@@ -54444,6 +54412,36 @@
"requiresEffort": true
}
},
"minimaxai/minimax-m3": {
"id": "minimaxai/minimax-m3",
"name": "MiniMax-M3",
"api": "openai-completions",
"provider": "nvidia",
"baseUrl": "https://integrate.api.nvidia.com/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 1000000,
"maxTokens": 16384,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"mistralai/codestral-22b-instruct-v0.1": {
"id": "mistralai/codestral-22b-instruct-v0.1",
"name": "Codestral 22b Instruct V0.1",
@@ -62195,9 +62193,9 @@
"image"
],
"cost": {
"input": 0.66,
"output": 3.41,
"cacheRead": 0.144,
"input": 0.55,
"output": 3.1999999999999997,
"cacheRead": 0.11,
"cacheWrite": 0
},
"contextWindow": 262144,
@@ -63522,7 +63520,7 @@
"api": "openrouter",
"baseUrl": "https://openrouter.ai/api/v1",
"provider": "openrouter",
"reasoning": true,
"reasoning": false,
"input": [
"text"
],
@@ -63533,22 +63531,7 @@
"cacheWrite": 0
},
"contextWindow": 163840,
"maxTokens": 16384,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
],
"effortMap": {
"minimal": "high",
"low": "high",
"medium": "high",
"high": "high"
}
}
"maxTokens": 16384
},
"deepseek/deepseek-chat-v3.1": {
"id": "deepseek/deepseek-chat-v3.1",
@@ -65751,7 +65734,7 @@
"cacheWrite": 0
},
"contextWindow": 131072,
"maxTokens": 32768
"maxTokens": 100352
},
"moonshotai/kimi-k2-0905": {
"id": "moonshotai/kimi-k2-0905",
@@ -65770,7 +65753,7 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 262144
"maxTokens": 100352
},
"moonshotai/kimi-k2-0905:exacto": {
"id": "moonshotai/kimi-k2-0905:exacto",
@@ -65861,9 +65844,9 @@
"image"
],
"cost": {
"input": 0.66,
"output": 3.41,
"cacheRead": 0.144,
"input": 0.55,
"output": 3.1999999999999997,
"cacheRead": 0.11,
"cacheWrite": 0
},
"contextWindow": 262144,
@@ -69347,8 +69330,8 @@
"image"
],
"cost": {
"input": 0.28850000000000003,
"output": 2.65,
"input": 0.28500000000000003,
"output": 2.4,
"cacheRead": 0.15,
"cacheWrite": 0
},
@@ -69758,13 +69741,13 @@
"text"
],
"cost": {
"input": 0.09,
"input": 0.09999999999999999,
"output": 0.3,
"cacheRead": 0.02,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 16384,
"maxTokens": 65536,
"thinking": {
"mode": "effort",
"efforts": [
@@ -70846,8 +70829,8 @@
"text"
],
"cost": {
"input": 0.98,
"output": 3.08,
"input": 0.975,
"output": 4.300000000000001,
"cacheRead": 0.182,
"cacheWrite": 0
},
@@ -70874,7 +70857,7 @@
"text"
],
"cost": {
"input": 0.95,
"input": 0.94,
"output": 3,
"cacheRead": 0.18,
"cacheWrite": 0
@@ -71154,6 +71137,25 @@
]
}
},
"hf:moonshotai/Kimi-K2.7-Code": {
"id": "hf:moonshotai/Kimi-K2.7-Code",
"name": "moonshotai/Kimi-K2.7-Code",
"api": "openai-completions",
"provider": "synthetic",
"baseUrl": "https://api.synthetic.new/openai/v1",
"reasoning": false,
"input": [
"text"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 8192
},
"hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4": {
"id": "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4",
"name": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4",
@@ -71196,7 +71198,7 @@
"cost": {
"input": 0.1,
"output": 0.1,
"cacheRead": 0,
"cacheRead": 0.1,
"cacheWrite": 0
},
"contextWindow": 131072,
@@ -71210,9 +71212,9 @@
]
}
},
"hf:Qwen/Qwen3.5-397B-A17B": {
"id": "hf:Qwen/Qwen3.5-397B-A17B",
"name": "Qwen/Qwen3.5-397B-A17B",
"hf:Qwen/Qwen3.6-27B": {
"id": "hf:Qwen/Qwen3.6-27B",
"name": "Qwen/Qwen3.6-27B",
"api": "openai-completions",
"provider": "synthetic",
"baseUrl": "https://api.synthetic.new/openai/v1",
@@ -71222,9 +71224,9 @@
"image"
],
"cost": {
"input": 0.6,
"output": 3,
"cacheRead": 0.6,
"input": 0.45,
"output": 3.6,
"cacheRead": 0.45,
"cacheWrite": 0
},
"contextWindow": 262144,
@@ -71239,54 +71241,6 @@
]
}
},
"hf:Qwen/Qwen3.6-27B": {
"id": "hf:Qwen/Qwen3.6-27B",
"name": "Qwen/Qwen3.6-27B",
"api": "openai-completions",
"provider": "synthetic",
"baseUrl": "https://api.synthetic.new/openai/v1",
"reasoning": false,
"input": [
"text"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 8192
},
"hf:zai-org/GLM-4.7": {
"id": "hf:zai-org/GLM-4.7",
"name": "zai-org/GLM-4.7",
"api": "openai-completions",
"provider": "synthetic",
"baseUrl": "https://api.synthetic.new/openai/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0.55,
"output": 2.19,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 202752,
"maxTokens": 64000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"hf:zai-org/GLM-4.7-Flash": {
"id": "hf:zai-org/GLM-4.7-Flash",
"name": "zai-org/GLM-4.7-Flash",
@@ -71298,9 +71252,9 @@
"text"
],
"cost": {
"input": 0.06,
"output": 0.4,
"cacheRead": 0.06,
"input": 0.1,
"output": 0.5,
"cacheRead": 0.1,
"cacheWrite": 0
},
"contextWindow": 196608,
@@ -71351,18 +71305,28 @@
"api": "openai-completions",
"provider": "synthetic",
"baseUrl": "https://api.synthetic.new/openai/v1",
"reasoning": false,
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"input": 1.4,
"output": 4.4,
"cacheRead": 1.4,
"cacheWrite": 0
},
"contextWindow": 524288,
"maxTokens": 8192
"maxTokens": 65536,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"syn:large:text": {
"id": "syn:large:text",
@@ -71682,7 +71646,7 @@
"api": "openai-completions",
"provider": "together",
"baseUrl": "https://api.together.xyz/v1",
"reasoning": true,
"reasoning": false,
"input": [
"text",
"image"
@@ -71694,17 +71658,7 @@
"cacheWrite": 0.18
},
"contextWindow": 10000000,
"maxTokens": 32768,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
"maxTokens": 32768
},
"MiniMaxAI/MiniMax-M2.5": {
"id": "MiniMaxAI/MiniMax-M2.5",
@@ -72381,6 +72335,38 @@
"escapeBuiltinToolNames": true
}
},
"umans-glm-5.2-nvfp4": {
"id": "umans-glm-5.2-nvfp4",
"name": "Umans GLM 5.2 NVFP4 (experimental, short test from Jun 29)",
"api": "anthropic-messages",
"provider": "umans",
"baseUrl": "https://api.code.umans.ai",
"reasoning": true,
"thinking": {
"mode": "anthropic-budget-effort",
"efforts": [
"high",
"xhigh"
],
"effortMap": {
"xhigh": "max"
}
},
"input": [
"text"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 405504,
"maxTokens": 131071,
"compat": {
"escapeBuiltinToolNames": true
}
},
"umans-kimi-k2.7": {
"id": "umans-kimi-k2.7",
"name": "Umans Kimi K2.7 Code",
@@ -75747,6 +75733,7 @@
"supportsForcedToolChoice": true,
"supportsSamplingParams": true,
"requiresToolResultId": false,
"requiresThinkingEnabled": false,
"replayUnsignedThinking": false,
"escapeBuiltinToolNames": false
}
@@ -80633,7 +80620,7 @@
},
"zai/glm-4.5": {
"id": "zai/glm-4.5",
"name": "GLM-4.5",
"name": "GLM 4.5",
"api": "anthropic-messages",
"baseUrl": "https://ai-gateway.vercel.sh",
"provider": "vercel-ai-gateway",
@@ -82174,12 +82161,12 @@
"contextWindow": 256000,
"maxTokens": 256000,
"compat": {
"includeEncryptedReasoning": false,
"filterReasoningHistory": true,
"omitReasoningEffort": true,
"reasoningEffortMap": {
"minimal": "low"
},
"includeEncryptedReasoning": false,
"filterReasoningHistory": true,
"omitReasoningEffort": true,
"supportsReasoningEffort": false
}
},
@@ -82223,9 +82210,9 @@
"text"
],
"cost": {
"input": 0.1,
"output": 0.3,
"cacheRead": 0.01,
"input": 0.14,
"output": 0.28,
"cacheRead": 0.0028,
"cacheWrite": 0
},
"contextWindow": 262144,
@@ -82259,9 +82246,9 @@
"image"
],
"cost": {
"input": 0.4,
"output": 2,
"cacheRead": 0.08,
"input": 0.14,
"output": 0.28,
"cacheRead": 0.0028,
"cacheWrite": 0
},
"contextWindow": 262144,
@@ -82294,9 +82281,9 @@
"text"
],
"cost": {
"input": 1,
"output": 3,
"cacheRead": 0.2,
"input": 0.435,
"output": 0.87,
"cacheRead": 0.0036,
"cacheWrite": 0
},
"contextWindow": 1048576,
@@ -82330,9 +82317,9 @@
"image"
],
"cost": {
"input": 0.4,
"output": 2,
"cacheRead": 0.08,
"input": 0.14,
"output": 0.28,
"cacheRead": 0.0028,
"cacheWrite": 0
},
"contextWindow": 1048576,
@@ -82365,9 +82352,9 @@
"text"
],
"cost": {
"input": 1,
"output": 3,
"cacheRead": 0.2,
"input": 0.435,
"output": 0.87,
"cacheRead": 0.0036,
"cacheWrite": 0
},
"contextWindow": 1048576,
@@ -85046,6 +85033,35 @@
"contextWindow": 256000,
"maxTokens": 80000
},
"meituan/longcat-2.0": {
"id": "meituan/longcat-2.0",
"name": "LongCat-2.0",
"api": "openai-completions",
"provider": "zenmux",
"baseUrl": "https://zenmux.ai/api/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0.3,
"output": 1.18,
"cacheRead": 0.006,
"cacheWrite": 0
},
"contextWindow": 1000000,
"maxTokens": null,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"meta/llama-3.3-70b-instruct": {
"id": "meta/llama-3.3-70b-instruct",
"name": "Llama 3.3 70b Instruct",
@@ -803,6 +803,18 @@ export function groqModelManagerOptions(config?: GroqModelManagerConfig): ModelM
// 3. Cerebras
// ---------------------------------------------------------------------------
const CEREBRAS_IMAGE_INPUT_MODEL_IDS = new Set(["gemma-4-31b"]);
function applyCerebrasDiscoveryOverrides(model: ModelSpec<"openai-completions">): ModelSpec<"openai-completions"> {
if (!CEREBRAS_IMAGE_INPUT_MODEL_IDS.has(model.id)) {
return model;
}
return {
...model,
input: ["text", "image"],
};
}
export interface CerebrasModelManagerConfig {
apiKey?: string;
baseUrl?: string;
@@ -812,7 +824,27 @@ export interface CerebrasModelManagerConfig {
export function cerebrasModelManagerOptions(
config?: CerebrasModelManagerConfig,
): ModelManagerOptions<"openai-completions"> {
return createSimpleOpenAICompletionsOptions("cerebras", "https://api.cerebras.ai/v1", config);
const apiKey = config?.apiKey;
const baseUrl = config?.baseUrl ?? "https://api.cerebras.ai/v1";
const references = createBundledReferenceMap<"openai-completions">("cerebras");
return {
providerId: "cerebras",
...(apiKey && {
fetchDynamicModels: () =>
fetchOpenAICompatibleModels({
api: "openai-completions",
provider: "cerebras",
baseUrl,
apiKey,
mapModel: (entry, defaults) => {
const reference = references.get(defaults.id);
const model = mapWithBundledReference(entry, defaults, reference);
return applyCerebrasDiscoveryOverrides(model);
},
fetch: config?.fetch,
}),
}),
};
}
// ---------------------------------------------------------------------------
@@ -1294,7 +1326,7 @@ export function clampFireworksKimiMaxTokens(modelId: string, candidate: number |
export const KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS = 32_768;
export function isKimiK27CodeModelId(modelId: string): boolean {
return /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code$/i.test(modelId);
return /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code(?:[-._]?highspeed)?$/i.test(modelId);
}
export function clampKimiK27CodeMaxTokens(modelId: string, candidate: number): number;
+5
View File
@@ -420,6 +420,11 @@ export interface AnthropicCompat {
* Default: auto-detected from provider/baseUrl and `model.reasoning`.
*/
replayUnsignedThinking?: boolean;
/**
* Whether the endpoint requires `thinking.type: "enabled"` whenever the
* model reasons. Use for models that reject omitted or disabled thinking.
*/
requiresThinkingEnabled?: boolean;
/**
* Prefix Anthropic built-in tool names (`web_search`, `code_execution`, ...)
* when they are ordinary client tools. Some Anthropic-compatible gateways
@@ -0,0 +1,41 @@
import { describe, expect, test } from "bun:test";
import { cerebrasModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
import type { FetchImpl } from "@oh-my-pi/pi-catalog/types";
describe("Cerebras provider discovery", () => {
test("discovers gemma-4-31b as image-capable", async () => {
const calls: Array<{ url: string; authorization: string | null }> = [];
const fetchMock: FetchImpl = async (input: string | URL | Request, init?: RequestInit) => {
const headers = new Headers(init?.headers);
calls.push({
url: String(input),
authorization: headers.get("authorization"),
});
return new Response(
JSON.stringify({
data: [
{ id: "gemma-4-31b", object: "model" },
{ id: "llama3.1-8b", object: "model" },
],
}),
{ status: 200, headers: { "content-type": "application/json" } },
);
};
const options = cerebrasModelManagerOptions({ apiKey: "cerebras-test-key", fetch: fetchMock });
const models = await options.fetchDynamicModels?.();
expect(calls).toEqual([
{
url: "https://api.cerebras.ai/v1/models",
authorization: "Bearer cerebras-test-key",
},
]);
expect(models?.find(model => model.id === "gemma-4-31b")).toMatchObject({
provider: "cerebras",
api: "openai-completions",
input: ["text", "image"],
});
expect(models?.find(model => model.id === "llama3.1-8b")?.input).toEqual(["text"]);
});
});
+29 -1
View File
@@ -1,5 +1,6 @@
import { describe, expect, test } from "bun:test";
import {
hasOpus47ApiRestrictions,
isClaudeModelId,
isGlmVisionModelId,
isGrokReasoningEffortCapable,
@@ -11,6 +12,7 @@ import {
isReasoningGlmModelId,
modelFamilyToken,
supportsAdaptiveThinkingDisplay,
supportsMidConversationSystemMessages,
} from "@oh-my-pi/pi-catalog/identity";
describe("isKimiModelId", () => {
@@ -44,10 +46,12 @@ describe("isClaudeModelId", () => {
});
describe("supportsAdaptiveThinkingDisplay", () => {
test("allows Claude Fable 5 and Opus 4.7 or newer only", () => {
test("allows Claude Fable 5, Opus 4.7 or newer, and Sonnet 5 or newer only", () => {
expect(supportsAdaptiveThinkingDisplay("claude-fable-5")).toBe(true);
expect(supportsAdaptiveThinkingDisplay("claude-opus-4-7")).toBe(true);
expect(supportsAdaptiveThinkingDisplay("claude-opus-5-0")).toBe(true);
expect(supportsAdaptiveThinkingDisplay("claude-sonnet-5")).toBe(true);
expect(supportsAdaptiveThinkingDisplay("us.anthropic.claude-sonnet-5")).toBe(true);
// Dotted and dashed version separators are equivalent.
expect(supportsAdaptiveThinkingDisplay("claude-opus-4.7")).toBe(true);
expect(supportsAdaptiveThinkingDisplay("anthropic/claude-opus-4.8")).toBe(true);
@@ -58,6 +62,30 @@ describe("supportsAdaptiveThinkingDisplay", () => {
});
});
describe("hasOpus47ApiRestrictions", () => {
test("allows Claude Fable 5, Opus 4.7 or newer, and Sonnet 5 or newer only", () => {
expect(hasOpus47ApiRestrictions("claude-fable-5")).toBe(true);
expect(hasOpus47ApiRestrictions("claude-opus-4-7")).toBe(true);
expect(hasOpus47ApiRestrictions("claude-opus-4.8")).toBe(true);
expect(hasOpus47ApiRestrictions("claude-sonnet-5")).toBe(true);
expect(hasOpus47ApiRestrictions("us.anthropic.claude-sonnet-5")).toBe(true);
expect(hasOpus47ApiRestrictions("claude-opus-4-6")).toBe(false);
expect(hasOpus47ApiRestrictions("claude-sonnet-4-6")).toBe(false);
expect(hasOpus47ApiRestrictions("claude-sonnet-4-5")).toBe(false);
});
});
describe("supportsMidConversationSystemMessages", () => {
test("allows Claude Fable 5, Opus 4.8 or newer, and Sonnet 5 or newer only", () => {
expect(supportsMidConversationSystemMessages("claude-fable-5")).toBe(true);
expect(supportsMidConversationSystemMessages("claude-opus-4-8")).toBe(true);
expect(supportsMidConversationSystemMessages("claude-sonnet-5")).toBe(true);
expect(supportsMidConversationSystemMessages("us.anthropic.claude-sonnet-5")).toBe(true);
expect(supportsMidConversationSystemMessages("claude-opus-4-7")).toBe(false);
expect(supportsMidConversationSystemMessages("claude-sonnet-4-6")).toBe(false);
});
});
describe("isMinimaxM2FamilyModelId", () => {
test("matches every M2-generation id shape served by aggregator/native hosts", () => {
// Fireworks/OpenCode/openrouter direct ids and `-highspeed`/`-lightning` variants.
+29 -2
View File
@@ -367,6 +367,12 @@ describe("model thinking derivation", () => {
provider: "amazon-bedrock",
});
const sonnet46 = createModel({ id: "claude-sonnet-4.6", api: "anthropic-messages", provider: "anthropic" });
const sonnet5 = createModel({ id: "claude-sonnet-5", api: "anthropic-messages", provider: "anthropic" });
const sonnet5Bedrock = createModel({
id: "global.anthropic.claude-sonnet-5",
api: "bedrock-converse-stream",
provider: "amazon-bedrock",
});
const mythos = createModel({ id: "claude-mythos-5", api: "anthropic-messages", provider: "anthropic" });
const mythosBedrock = createModel({
id: "global.anthropic.claude-mythos-5",
@@ -399,6 +405,8 @@ describe("model thinking derivation", () => {
expect(sonnet45Bedrock.thinking?.mode).toBe("budget");
expect(opus46.thinking?.mode).toBe("anthropic-adaptive");
expect(sonnet46.thinking?.mode).toBe("anthropic-adaptive");
expect(sonnet5.thinking?.mode).toBe("anthropic-adaptive");
expect(sonnet5Bedrock.thinking?.mode).toBe("anthropic-adaptive");
expect(mythosBedrock.thinking?.mode).toBe("anthropic-adaptive");
expect(minimaxM2.thinking).toEqual({
mode: "anthropic-adaptive",
@@ -437,13 +445,16 @@ describe("model thinking derivation", () => {
expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.XHigh)).toBe("max");
expect(mapEffortToAnthropicAdaptiveEffort(mythos, Effort.High)).toBe("xhigh");
expect(mapEffortToAnthropicAdaptiveEffort(mythosBedrock, Effort.XHigh)).toBe("max");
expect(mapEffortToAnthropicAdaptiveEffort(sonnet5, Effort.High)).toBe("xhigh");
expect(mapEffortToAnthropicAdaptiveEffort(sonnet5Bedrock, Effort.XHigh)).toBe("max");
// Bedrock Converse keeps the four-tier legacy mapping; xhigh aliases to "max".
expect(opus47Bedrock.thinking?.effortMap).toEqual({ minimal: "low", xhigh: "max" });
expect(mapEffortToAnthropicAdaptiveEffort(opus47Bedrock, Effort.High)).toBe("high");
expect(mapEffortToAnthropicAdaptiveEffort(sonnet5Bedrock, Effort.High)).toBe("high");
expect(() => mapEffortToAnthropicAdaptiveEffort(sonnet46, Effort.XHigh)).toThrow(/not supported/);
});
it("bakes adaptive display support for Opus 4.7+ and Fable/Mythos 5", () => {
it("bakes adaptive display support for Opus 4.7+, Sonnet 5+, and Fable/Mythos 5", () => {
const opus46 = createModel({ id: "claude-opus-4.6", api: "anthropic-messages", provider: "anthropic" });
const opus47 = createModel({ id: "claude-opus-4-7", api: "anthropic-messages", provider: "anthropic" });
// Dotted and dashed version forms are equivalent; bare dated ids stay Opus 4.0.
@@ -459,6 +470,12 @@ describe("model thinking derivation", () => {
api: "bedrock-converse-stream",
provider: "amazon-bedrock",
});
const sonnet5 = createModel({ id: "claude-sonnet-5", api: "anthropic-messages", provider: "anthropic" });
const sonnet5Bedrock = createModel({
id: "global.anthropic.claude-sonnet-5",
api: "bedrock-converse-stream",
provider: "amazon-bedrock",
});
expect(opus46.thinking?.supportsDisplay).toBeUndefined();
expect(opus47.thinking?.supportsDisplay).toBe(true);
@@ -466,6 +483,8 @@ describe("model thinking derivation", () => {
expect(opus4Dated.thinking?.supportsDisplay).toBeUndefined();
expect(fable.thinking?.supportsDisplay).toBe(true);
expect(fableBedrock.thinking?.supportsDisplay).toBe(true);
expect(sonnet5.thinking?.supportsDisplay).toBe(true);
expect(sonnet5Bedrock.thinking?.supportsDisplay).toBe(true);
});
it("backfills wire facts onto explicit thinking, explicit values winning", () => {
@@ -526,10 +545,12 @@ describe("model thinking derivation", () => {
it("bakes sampling-param rejection into anthropic compat", () => {
const sonnet45 = createModel({ id: "claude-sonnet-4-5", api: "anthropic-messages", provider: "anthropic" });
const opus47 = createModel({ id: "claude-opus-4.7", api: "anthropic-messages", provider: "anthropic" });
const sonnet5 = createModel({ id: "claude-sonnet-5", api: "anthropic-messages", provider: "anthropic" });
const fable = createModel({ id: "claude-fable-5", api: "anthropic-messages", provider: "anthropic" });
expect(sonnet45.compat.supportsSamplingParams).toBe(true);
expect(opus47.compat.supportsSamplingParams).toBe(false);
expect(sonnet5.compat.supportsSamplingParams).toBe(false);
expect(fable.compat.supportsSamplingParams).toBe(false);
});
@@ -685,11 +706,17 @@ describe("model thinking runtime helpers", () => {
api: "openai-completions",
provider: "openrouter",
});
const sonnet5 = createModel({
id: "anthropic/claude-sonnet-5",
api: "openai-completions",
provider: "openrouter",
});
expect(fable.thinking?.efforts.at(-1)).toBe(Effort.XHigh);
expect(opus46.thinking?.efforts.at(-1)).toBe(Effort.XHigh);
expect(sonnet46.thinking?.efforts.at(-1)).toBe(Effort.High);
expect(sonnet5.thinking?.efforts.at(-1)).toBe(Effort.XHigh);
expect(requireSupportedEffort(fable, Effort.XHigh)).toBe(Effort.XHigh);
expect(requireSupportedEffort(sonnet5, Effort.XHigh)).toBe(Effort.XHigh);
});
it("enables xhigh for openai-responses and openai-codex-responses APIs", () => {
+70 -2
View File
@@ -18,13 +18,81 @@
- Fixed a multi-character `mode: "replace"` regex remainder (the bytes of a match outside a preserved `#…#` placeholder) drifting across an obfuscator restart, which invalidated provider prompt-cache prefixes even with a stable key. The remainder was redacted to a content-derived `ZZ`+hash marker that was only recognized as already-redacted within the generating session (via an in-memory set), so a fresh obfuscator reprocessing persisted text re-redacted it to a different value (`ZZPL#…#` → `ZZ7f#…#`). The remainder marker now derives from a keyed run of the per-install key and the remainder length, so any instance sharing the key reproduces it byte-identically (idempotent across restart) while staying unpredictable enough that raw sentinel-shaped bytes (`ZZZZ`) still differ from it and are redacted rather than passed through ([#2465](https://github.com/can1357/oh-my-pi/issues/2465)).
- Fixed a secret regex match that starts in outside text and ends inside a previously generated `#…#` placeholder's expanded value leaving an independently-matching outside prefix provider-visible. Resuming the scan past the cut placeholder skipped the whole straddling span, so a pattern like `[A-Z0-9]{8,12}` greedily spanning `SECRETUV` into an `ABCDEFGH` placeholder returned `SECRETUV#…#` even though `SECRETUV` satisfies the regex on its own. The cut handling now re-runs the regex bounded to just before the placeholder (full left context kept, so lookbehind still evaluates) and redacts the standalone prefix match — to its own reversible placeholder in obfuscate mode, or a one-way redaction in replace mode — while the cut secret stays as its existing placeholder. The replace-mode redaction's fixed point is verified against the placeholder-expanded view re-obfuscation actually scans, so it does not drift when the adjacent placeholder expands ([#2465](https://github.com/can1357/oh-my-pi/issues/2465)).
## [16.2.9] - 2026-06-30
### Breaking Changes
- Renamed the built-in quick_task subagent to sonic; update any task spawns or configurations referencing quick_task by name.
### Added
- Added the llama3.2:3b local model option for memory and auto-thinking tasks, utilizing a quantized ONNX model.
- Added a built-in Tester subagent designed to write high-signal tests for behavior, invariants, and edge cases while avoiding redundant or low-value tests.
- Added a Speech-to-Text submit trigger setting to auto-submit dictation on release, on complete sentences, or via a spoken submit command.
- Added a loop-guard mechanism that detects thinking/response loops and injects a system notice during auto-retries to guide the model to break the pattern and take a concrete next step.
### Changed
- Enabled contextual snapcompact shape resolution based on rendered text content
- Renamed the Speech-to-Text (STT) setting label from "TTS Submit Trigger" to "Speech-to-Text Submit Trigger".
### Fixed
- Fixed snapcompact preflight to use the same font-aware renderability probe as compaction, including prior preserved archive text, so CJK history remains renderable through per-glyph Silver fallback across repeated compactions.
- Fixed an issue where mid-run compaction was incorrectly skipped when a persisted assistant display variant shared a persistence key but differed in content from the live message.
- Fixed duplicate placeholder cards being created when streamed tool blocks started with an empty ID.
- Fixed omp debug --profile failing on Bun by treating the optional --allow-natives-syntax flag as best-effort when v8.setFlagsFromString is unavailable.
- Fixed release binaries missing the compiled tiny-model Transformers.js version pin, preventing runtime resolution issues on certain platforms like Homebrew Darwin arm64.
- Fixed MCP OAuth flows silently falling back to a random redirect port when the preferred port (default 3000) was busy, which caused authentication failures with strict providers. The flow now fails fast with a configuration error when a static client ID is pinned, while dynamic registration flows continue to use fallback ports.
- Fixed /mcp reauth and /mcp add commands ignoring the Escape key during OAuth authentication, allowing users to cancel the flow immediately instead of waiting for the timeout.
### Removed
- Removed the built-in oracle subagent.
## [16.2.8] - 2026-06-30
### Added
- Added built-in Go coding rules including `go-add-cleanup`, `go-bench-loop`, `go-exp-promoted`, `go-ioutil`, `go-join-hostport`, `go-new-expr`, `go-rand-v2`, and `go-range-int`
### Changed
- Relaxed strict bash tool constraints regarding the use of search, grep, ls, and find commands
### Fixed
- Fixed auto-compaction dead-ends by automatically triggering a shake rescue to elide oversized tails
- Improved compaction warning message to suggest running `/shake images` for irreducible image tails
- Fixed `grep`/`search` direct execution to accept JSON-array string `paths` for string-or-array inputs. ([#3873](https://github.com/can1357/oh-my-pi/issues/3873))
- Fixed auto-compaction dead-ending with "Compaction freed too little context to make progress" when a single recent turn (large tool output, heavy fenced/XML block) is itself bigger than the recovery band — `findCutPoint` can't cut inside one message, so the summarizer had no lever left. The guard now runs an artifact-backed `shake` elide pass over the oversized tail and re-tests headroom before pausing, and the remaining warning points at `/shake images` for image-only tails it can't elide. ([#3786](https://github.com/can1357/oh-my-pi/issues/3786))
- Fixed reviewer/`task` subagents whose incremental `yield` (`type: ["overall_correctness"]`, `type: ["findings"]`, …) carried a value that mismatched the matching property's sub-schema being silently accepted and then post-mortem rejected with `schema_violation` — opaquely swapping the agent's accepted output for an error blob. The yield tool now validates each incremental section's `data` against its top-level property's sub-schema (items schema for array-typed labels) and surfaces the same retry feedback as terminal yields, so models like `deepseek-v4-pro` that emit `"Correct"`/`"correct."`/`"approved"` for an enum field get up to three corrective retries; the existing `MAX_SCHEMA_RETRIES` override then accepts the value with `SUBAGENT_WARNING_SCHEMA_OVERRIDDEN` instead of losing the entire result. Unknown labels stay unconstrained ([#3870](https://github.com/can1357/oh-my-pi/issues/3870)).
- Fixed streaming tool-call previews (notably `write`) showing an empty body for the entire streaming phase by surfacing the partial JSON already in hand on the first reveal, then pacing only subsequent growth ([#3881](https://github.com/can1357/oh-my-pi/issues/3881)).
- Fixed hashline edit mode preserving UTF-8 BOM bytes on edited files. ([#3867](https://github.com/can1357/oh-my-pi/issues/3867))
- Fixed slow local LLM streams by forwarding persisted stream timeout settings (`providers.streamFirstEventTimeoutSeconds`, `providers.streamIdleTimeoutSeconds`) into model requests, so users can widen or disable watchdogs without environment variables. ([#3878](https://github.com/can1357/oh-my-pi/issues/3878))
- Fixed the bash interceptor blocking `echo` / `printf` redirects to `/dev/null`, `/dev/tty`, `/dev/stdout`, and `/dev/stderr` device sinks while still directing real file writes to the write tool. ([#3763](https://github.com/can1357/oh-my-pi/issues/3763))
- Fixed long snapcompact sessions re-sending multi-megabyte standing image archives on every provider request by enforcing a per-request frame byte budget, letting auto-compaction fall back to context-full summaries when snapcompact output is too large, and omitting legacy over-budget frames from rebuilt LLM contexts. ([#3792](https://github.com/can1357/oh-my-pi/issues/3792))
## [16.2.7] - 2026-06-30
### Breaking Changes
- Replaced the global `serviceTier` and `fastModeScope` settings with granular, per-family settings (`tier.openai`, `tier.anthropic`, and `tier.google`) to control service tiers, subagents, advisors, and `/fast` mode targets.
### Changed
- Improved binary file detection and terminal handling to prevent corruption from non-UTF-8 content, and updated file summaries to explicitly note skipped binary files.
- Enhanced context compaction (snapcompact) to resolve shapes contextually based on rendered text content.
### Fixed
- Improved reliability of DuckDuckGo web searches by updating browser request headers and parameters
- Fixed an issue where CJK (Chinese, Japanese, Korean) history could become unrenderable during repeated context compactions.
- Fixed a memory exhaustion bug in the TUI when using `/resume` on large previous sessions.
- Fixed an issue where the `irc` inbox missed messages that arrived while the recipient agent was already running.
- Fixed a startup hang caused by system-prompt GPU detection blocking and repeatedly running failed probes.
- Improved error reporting for `omp tiny-models download` by displaying the actual worker-side download error.
- Resolved status inconsistencies between `/extensions`, `/mcp list`, and the dashboard, ensuring MCP server states, allowlists/denylists, and configuration files (like `mcp.json`) stay fully synchronized.
- Improved branch-mode task merges to preserve the agent's original commit history (messages and authors) and fixed a bug where merges were rejected due to unrelated dirty changes in the parent checkout.
- Fixed an issue where the `Working...` loader spinner would prematurely disappear or fail to re-arm after a subagent (`task`) tool completed or during transient overlays (such as auto-compaction or auto-retry).
## [16.2.6] - 2026-06-29
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-coding-agent",
"version": "16.2.6",
"version": "16.2.9",
"description": "Coding agent CLI with read, bash, edit, write tools and session management",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+19 -12
View File
@@ -10,9 +10,10 @@ import type {
Model,
ProviderSessionState,
ServiceTier,
ServiceTierByFamily,
SimpleStreamOptions,
} from "@oh-my-pi/pi-ai";
import { streamSimple } from "@oh-my-pi/pi-ai";
import { resolveModelServiceTier, streamSimple } from "@oh-my-pi/pi-ai";
import { buildModelProviderPriorityRank, type CanonicalModelVariant } from "@oh-my-pi/pi-catalog/identity";
import { replaceTabs, truncateToWidth } from "@oh-my-pi/pi-tui";
import { formatDuration, getProjectDir } from "@oh-my-pi/pi-utils";
@@ -25,7 +26,7 @@ import {
getModelMatchPreferences,
resolveCliModel,
} from "../config/model-resolver";
import { resolveServiceTierSetting } from "../config/service-tier";
import { buildServiceTierByFamily, serviceTierForAllFamilies, serviceTierSettingToTier } from "../config/service-tier";
import { Settings } from "../config/settings";
import benchPrompt from "../prompts/bench.md" with { type: "text" };
import { discoverAuthStorage, loadCliExtensionProviders } from "../sdk";
@@ -106,8 +107,8 @@ export interface BenchSummary {
maxTokens: number;
models: BenchModelReport[];
failures: number;
/** Requested service tier passed to every request; absent when none was requested. Scoped tiers (`openai-only`/`claude-only`) may be dropped per-provider downstream. */
serviceTier?: ServiceTier;
/** Requested per-family service tiers, resolved per model before reaching the wire. */
serviceTierByFamily?: ServiceTierByFamily;
}
type BenchStreamSimple = (
@@ -518,12 +519,18 @@ export async function runBenchCommand(command: BenchCommandArgs, deps: BenchDepe
const runtime = await (deps.createRuntime ?? createDefaultRuntime)();
try {
const targets = resolveBenchModels(command.models, runtime.modelRegistry, runtime.settings, writeStderr);
// Explicit `--service-tier` wins; otherwise fall back to the configured
// `serviceTier` setting (`none`/unset omits the wire field). Scope-aware
// gating to the model's provider happens downstream in the provider layer.
const serviceTierValue = command.flags.serviceTier ?? runtime.settings?.get("serviceTier");
const serviceTier = serviceTierValue ? resolveServiceTierSetting(serviceTierValue, undefined) : undefined;
if (!json && serviceTier) writeStdout(`${chalk.dim(`service tier: ${serviceTier}`)}\n`);
// Explicit `--service-tier` (a single value broadcast across families) wins;
// otherwise fall back to the configured per-family `tier.*` settings. Each
// model resolves its own family's tier below before reaching the wire.
const flagTier = command.flags.serviceTier ? serviceTierSettingToTier(command.flags.serviceTier) : undefined;
const serviceTierByFamily = command.flags.serviceTier
? serviceTierForAllFamilies(flagTier)
: buildServiceTierByFamily(
runtime.settings?.get("tier.openai") ?? "none",
runtime.settings?.get("tier.anthropic") ?? "none",
runtime.settings?.get("tier.google") ?? "none",
);
if (!json && flagTier) writeStdout(`${chalk.dim(`service tier: ${flagTier}`)}\n`);
const reports: BenchModelReport[] = [];
for (const { selector, model, thinking } of targets) {
if (!json) {
@@ -564,7 +571,7 @@ export async function runBenchCommand(command: BenchCommandArgs, deps: BenchDepe
maxTokens,
reasoning: toReasoningEffort(thinking),
disableReasoning: shouldDisableReasoning(thinking) ? true : undefined,
serviceTier,
serviceTier: resolveModelServiceTier(serviceTierByFamily, model),
},
streamFn,
now,
@@ -606,7 +613,7 @@ export async function runBenchCommand(command: BenchCommandArgs, deps: BenchDepe
reports.push(buildModelReport(selector, model, thinking, results));
}
const failures = reports.reduce((sum, report) => sum + report.results.filter(result => !result.ok).length, 0);
const summary: BenchSummary = { runs, maxTokens, models: reports, failures, serviceTier };
const summary: BenchSummary = { runs, maxTokens, models: reports, failures, serviceTierByFamily };
if (json) {
writeStdout(`${JSON.stringify(summary, null, 2)}\n`);
} else if (reports.length > 1 || runs > 1) {
@@ -28,12 +28,21 @@ interface ProgressReporter {
interface DownloadResult {
model: TinyLocalModelKey;
ok: boolean;
error?: string;
}
function writeLine(text = ""): void {
process.stdout.write(`${text}\n`);
}
function downloadErrorSummary(error: string | undefined): string | undefined {
return error
?.split(/\r?\n/)
.map(line => line.trim())
.find(line => line.length > 0)
?.replace(/^Error:\s*/, "");
}
export function resolveModels(model: string | undefined): TinyLocalModelKey[] {
if (!model) return [DEFAULT_TINY_TITLE_LOCAL_MODEL_KEY];
// `all` is a prefetch convenience: skip models that fail before load (unsupported
@@ -101,10 +110,15 @@ async function downloadOne(modelKey: TinyLocalModelKey, json: boolean | undefine
const label = getTinyLocalModelSpec(modelKey)?.label ?? modelKey;
if (!json && !process.stdout.isTTY) writeLine(`Downloading ${label} (${modelKey})...`);
const progress = makeProgressReporter(modelKey, json);
const ok = await tinyTitleClient.downloadModel(modelKey, { onProgress: progress.onProgress });
progress.finish(ok);
if (!json && !process.stdout.isTTY) writeLine(ok ? `Downloaded ${label}.` : `Failed to download ${label}.`);
return { model: modelKey, ok };
const result = await tinyTitleClient.downloadModel(modelKey, { onProgress: progress.onProgress });
progress.finish(result.ok);
const error = downloadErrorSummary(result.error);
if (!json && !process.stdout.isTTY) {
writeLine(result.ok ? `Downloaded ${label}.` : `Failed to download ${label}${error ? `: ${error}` : ""}.`);
} else if (!json && !result.ok && error) {
writeLine(`${label} failed: ${error}`);
}
return result.error ? { model: modelKey, ok: result.ok, error: result.error } : { model: modelKey, ok: result.ok };
}
export async function runTinyModelsCommand(command: TinyModelsCommandArgs): Promise<void> {
+3 -3
View File
@@ -1,6 +1,6 @@
import { Args, Command, Flags } from "@oh-my-pi/pi-utils/cli";
import { runBenchCommand } from "../cli/bench-cli";
import { SERVICE_TIER_SETTING_VALUES } from "../config/service-tier";
import { SERVICE_TIER_OPENAI_VALUES } from "../config/service-tier";
export default class Bench extends Command {
static description =
@@ -19,8 +19,8 @@ export default class Bench extends Command {
"max-tokens": Flags.integer({ description: "Max output tokens per request", default: 512 }),
prompt: Flags.string({ description: "Custom prompt text (default: bundled bench prompt)" }),
"service-tier": Flags.string({
description: "Service tier hint (default: configured `serviceTier` setting; `none` omits it)",
options: SERVICE_TIER_SETTING_VALUES,
description: "Service tier applied per model family (default: configured `tier.*` settings; `none` omits it)",
options: SERVICE_TIER_OPENAI_VALUES,
}),
json: Flags.boolean({ description: "Output JSON" }),
par: Flags.integer({ description: "Execute runs with N parallel queries/requests", default: 4 }),
@@ -42,7 +42,7 @@ export async function runCommitAgentSession(input: CommitAgentInput): Promise<Co
types_description: typesDescription,
});
const state: CommitAgentState = { diffText: input.diffText };
const spawns = "quick_task";
const spawns = "sonic";
const tools = createCommitTools({
cwd: input.cwd,
authStorage: input.authStorage,
@@ -27,7 +27,7 @@ Tool guidance:
- git_file_diff: diff for specific files
- git_hunk: specific hunks for large diffs
- recent_commits: recent commit subjects + style stats
- analyze_files: spawn quick_task subagents in parallel for analysis
- analyze_files: spawn sonic subagents in parallel for analysis
- propose_changelog: provide changelog entries for each changelog target
- propose_commit: submit final commit proposal and run validation
- split_commit: propose multiple commit groups (no overlapping files; all staged files covered)
@@ -63,7 +63,7 @@ export function createAnalyzeFileTool(options: {
return {
name: "analyze_files",
label: "Analyze Files",
description: "Spawn quick_task agents to analyze files.",
description: "Spawn sonic agents to analyze files.",
parameters: analyzeFileSchema,
async execute(toolCallId, params, _onUpdate, ctx, signal) {
const toolSession = buildToolSession(ctx, options);
@@ -83,7 +83,7 @@ export function createAnalyzeFileTool(options: {
related_files: relatedFiles,
});
const taskParams: TaskParams = {
agent: "quick_task",
agent: "sonic",
id: `AnalyzeFile${index + 1}`,
description: `Analyze ${file}`,
assignment,
@@ -22,7 +22,16 @@
},
"disabledServers": {
"type": "array",
"description": "User-level denylist for disabling discovered servers by name.",
"description": "User-level denylist for disabling discovered servers by name. Highest precedence: a server here is hidden regardless of any other source.",
"items": {
"type": "string",
"minLength": 1
},
"uniqueItems": true
},
"enabledServers": {
"type": "array",
"description": "User-level allowlist that overrides a discovered server's `enabled: false` flag (e.g. when the source config is owned by another tool such as opencode.json). The denylist still wins.",
"items": {
"type": "string",
"minLength": 1
@@ -1,87 +1,116 @@
import type { ServiceTier } from "@oh-my-pi/pi-ai";
import type { ServiceTier, ServiceTierByFamily } from "@oh-my-pi/pi-ai";
import type { SubmenuOption } from "./settings-schema";
/**
* Service-tier setting values shared by every "Service Tier" setting. `"none"`
* is the omit-the-parameter sentinel; the remaining values mirror
* {@link ServiceTier}.
* Per-family service-tier setting values. `"none"` is the omit-the-parameter
* sentinel; the rest mirror the wire {@link ServiceTier} values each provider
* family actually realizes. OpenAI accepts the full set; Anthropic realizes
* only `priority` (fast mode); Google (Gemini API + Vertex) realizes
* `flex`/`priority`.
*/
export const SERVICE_TIER_SETTING_VALUES = [
export const SERVICE_TIER_OPENAI_VALUES = ["none", "auto", "default", "flex", "scale", "priority"] as const;
export const SERVICE_TIER_ANTHROPIC_VALUES = ["none", "priority"] as const;
export const SERVICE_TIER_GOOGLE_VALUES = ["none", "flex", "priority"] as const;
export type ServiceTierOpenAISettingValue = (typeof SERVICE_TIER_OPENAI_VALUES)[number];
export type ServiceTierAnthropicSettingValue = (typeof SERVICE_TIER_ANTHROPIC_VALUES)[number];
export type ServiceTierGoogleSettingValue = (typeof SERVICE_TIER_GOOGLE_VALUES)[number];
/**
* Inherit-capable single value for the subagent/advisor tiers. The chosen tier
* is broadcast across families and applied to whichever family the spawned
* model belongs to (clamped to what that family realizes); `"inherit"` defers
* to the main agent's live per-family selection.
*/
export const SERVICE_TIER_INHERIT_SETTING_VALUES = [
"inherit",
"none",
"auto",
"default",
"flex",
"scale",
"priority",
"openai-only",
"claude-only",
] as const;
export type ServiceTierSettingValue = (typeof SERVICE_TIER_SETTING_VALUES)[number];
/** Variant value set for scoped service-tier settings (subagent/advisor) that can defer to the main agent. */
export const SERVICE_TIER_INHERIT_SETTING_VALUES = ["inherit", ...SERVICE_TIER_SETTING_VALUES] as const;
export type ServiceTierInheritSettingValue = (typeof SERVICE_TIER_INHERIT_SETTING_VALUES)[number];
/** Submenu descriptions shared by the base `serviceTier` setting. */
export const SERVICE_TIER_OPTIONS: ReadonlyArray<SubmenuOption<ServiceTierSettingValue>> = [
{ value: "none", label: "None", description: "Omit service_tier parameter" },
{ value: "auto", label: "Auto", description: "Use provider default tier selection (OpenAI)" },
{ value: "default", label: "Default", description: "Standard priority processing (OpenAI)" },
{ value: "flex", label: "Flex", description: "Flexible capacity tier when available (OpenAI)" },
{ value: "scale", label: "Scale", description: "Scale Tier credits when available (OpenAI)" },
export const SERVICE_TIER_OPENAI_OPTIONS: ReadonlyArray<SubmenuOption<ServiceTierOpenAISettingValue>> = [
{ value: "none", label: "None", description: "Omit service_tier (standard processing)" },
{ value: "auto", label: "Auto", description: "Provider default tier selection" },
{ value: "default", label: "Default", description: "Standard priority processing" },
{ value: "flex", label: "Flex", description: "Lower cost, higher latency when available" },
{ value: "scale", label: "Scale", description: "Scale Tier credits when available" },
{ value: "priority", label: "Priority", description: "Faster, higher cost (premium request)" },
];
export const SERVICE_TIER_ANTHROPIC_OPTIONS: ReadonlyArray<SubmenuOption<ServiceTierAnthropicSettingValue>> = [
{ value: "none", label: "None", description: "Standard processing" },
{
value: "priority",
label: "Priority",
description: "Priority on every supported provider (OpenAI `service_tier`, Anthropic fast mode)",
},
{
value: "openai-only",
label: "Priority (OpenAI only)",
description: "Priority on OpenAI/OpenAI-Codex requests; ignored elsewhere",
},
{
value: "claude-only",
label: "Priority (Claude only)",
description: "Anthropic fast mode on direct Claude requests; ignored elsewhere (incl. Bedrock/Vertex)",
description: 'Fast mode (`speed: "fast"`) on supported direct Claude models; ignored on Bedrock/Vertex',
},
];
/** Submenu descriptions for inherit-capable service-tier settings. */
export const SERVICE_TIER_GOOGLE_OPTIONS: ReadonlyArray<SubmenuOption<ServiceTierGoogleSettingValue>> = [
{ value: "none", label: "None", description: "Standard processing" },
{ value: "flex", label: "Flex", description: "Lower cost, higher latency (Gemini API + Vertex)" },
{ value: "priority", label: "Priority", description: "Faster, higher reliability (Gemini API + Vertex)" },
];
export const SERVICE_TIER_INHERIT_OPTIONS: ReadonlyArray<SubmenuOption<ServiceTierInheritSettingValue>> = [
{ value: "inherit", label: "Inherit", description: "Use the main agent's Service Tier" },
...SERVICE_TIER_OPTIONS,
{ value: "inherit", label: "Inherit", description: "Match the main agent's live per-family tiers" },
{ value: "none", label: "None", description: "Standard processing" },
{ value: "auto", label: "Auto", description: "Provider default tier selection (OpenAI family)" },
{ value: "default", label: "Default", description: "Standard priority processing (OpenAI family)" },
{ value: "flex", label: "Flex", description: "Flexible capacity tier (OpenAI/Google families)" },
{ value: "scale", label: "Scale", description: "Scale Tier credits (OpenAI family)" },
{ value: "priority", label: "Priority", description: "Priority on every supported family of the spawned model" },
];
/**
* Resolve a service-tier setting value to the wire {@link ServiceTier} (or
* `undefined` to omit). `"inherit"` defers to `inherited`; `"none"` omits.
*/
export function resolveServiceTierSetting(value: string, inherited: ServiceTier | undefined): ServiceTier | undefined {
if (value === "inherit") return inherited;
if (value === "none" || value === "") return undefined;
/** Map a per-family setting value to a wire {@link ServiceTier}, or `undefined` to omit. */
export function serviceTierSettingToTier(value: string): ServiceTier | undefined {
if (value === "none" || value === "" || value === "inherit") return undefined;
return value as ServiceTier;
}
/**
* Resolve the `serviceTier` *setting value* to stamp onto a subagent's settings
* snapshot.
*
* - A concrete `subagentSetting` (`"none"` or a tier) wins outright.
* - `"inherit"` defers to the parent's live effective tier when the caller has a
* live session (`inherited` passed as `ServiceTier | null`, where `null` means
* the parent explicitly has no tier — e.g. `/fast off`). When no live session
* is available (`inherited === undefined`, e.g. cold subagent revive) it falls
* back to the parent's configured `serviceTier` setting so behavior matches a
* plain settings snapshot.
*/
export function resolveSubagentServiceTier(
subagentSetting: string,
configuredTier: ServiceTierSettingValue,
inherited: ServiceTier | null | undefined,
): ServiceTierSettingValue {
if (subagentSetting !== "inherit") return subagentSetting as ServiceTierSettingValue;
if (inherited === undefined) return configuredTier;
return inherited ?? "none";
/** Assemble the live per-family tier map from the three `tier.*` setting values. */
export function buildServiceTierByFamily(openai: string, anthropic: string, google: string): ServiceTierByFamily {
const out: ServiceTierByFamily = {};
const o = serviceTierSettingToTier(openai);
if (o) out.openai = o;
const a = serviceTierSettingToTier(anthropic);
if (a) out.anthropic = a;
const g = serviceTierSettingToTier(google);
if (g) out.google = g;
return out;
}
/**
* Broadcast a single chosen tier across families, clamped to what each family
* realizes: OpenAI takes any tier, Anthropic only `priority`, Google only
* `flex`/`priority`. Used by the subagent/advisor single-value settings and the
* `omp bench --service-tier` flag, which apply one tier to whatever family the
* target model belongs to.
*/
export function serviceTierForAllFamilies(tier: ServiceTier | undefined): ServiceTierByFamily {
if (!tier) return {};
const out: ServiceTierByFamily = { openai: tier };
if (tier === "priority") out.anthropic = "priority";
if (tier === "flex" || tier === "priority") out.google = tier;
return out;
}
/**
* Resolve a subagent/advisor service-tier setting to a per-family map.
*
* - A concrete tier is broadcast across families (see
* {@link serviceTierForAllFamilies}).
* - `"none"` yields an empty map.
* - `"inherit"` defers to `inherited` — the parent's live per-family tiers when
* a live session supplied them, else the empty map.
*/
export function resolveSubagentServiceTier(setting: string, inherited: ServiceTierByFamily): ServiceTierByFamily {
if (setting === "inherit") return inherited;
return serviceTierForAllFamilies(serviceTierSettingToTier(setting));
}
@@ -3,6 +3,7 @@ import { DEFAULT_SHARE_URL } from "@oh-my-pi/pi-wire";
import { SHAPE_VARIANT_NAMES } from "@oh-my-pi/snapcompact";
import { DEFAULT_RELAY_URL } from "../collab/protocol";
import { DEFAULT_STT_MODEL_KEY, STT_MODEL_OPTIONS, STT_MODEL_VALUES } from "../stt/models";
import { STT_SUBMIT_TRIGGER_OPTIONS, STT_SUBMIT_TRIGGER_VALUES } from "../stt/submit-trigger";
import { AUTO_THINKING, getConfiguredThinkingLevelMetadata, getThinkingLevelMetadata } from "../thinking";
import {
TINY_MODEL_DEVICE_DEFAULT,
@@ -36,10 +37,14 @@ import {
import { EDIT_MODES } from "../utils/edit-mode";
import { SEARCH_PROVIDER_OPTIONS, SEARCH_PROVIDER_PREFERENCES, type SearchProviderId } from "../web/search/types";
import {
SERVICE_TIER_ANTHROPIC_OPTIONS,
SERVICE_TIER_ANTHROPIC_VALUES,
SERVICE_TIER_GOOGLE_OPTIONS,
SERVICE_TIER_GOOGLE_VALUES,
SERVICE_TIER_INHERIT_OPTIONS,
SERVICE_TIER_INHERIT_SETTING_VALUES,
SERVICE_TIER_OPTIONS,
SERVICE_TIER_SETTING_VALUES,
SERVICE_TIER_OPENAI_OPTIONS,
SERVICE_TIER_OPENAI_VALUES,
} from "./service-tier";
/** Unified settings schema - single source of truth for all settings.
@@ -140,7 +145,7 @@ export const TAB_GROUPS: Record<SettingTab, readonly string[]> = {
"Developer",
],
tasks: ["Modes", "Subagents", "Isolation", "Commands & Skills"],
providers: ["Services", "Fireworks", "Tiny Model", "Protocol", "Privacy"],
providers: ["Services", "Fireworks", "Tiny Model", "Protocol", "Timeouts", "Privacy"],
};
/** Status line segment identifiers */
@@ -1214,75 +1219,77 @@ export const SETTINGS_SCHEMA = {
},
},
serviceTier: {
"tier.openai": {
type: "enum",
values: SERVICE_TIER_SETTING_VALUES,
values: SERVICE_TIER_OPENAI_VALUES,
default: "none",
ui: {
tab: "model",
group: "Sampling",
label: "Service Tier",
label: "Service Tier — OpenAI",
description:
'Processing priority hint (none = omit). OpenAI accepts the tier values directly; Anthropic realizes `priority` as `speed: "fast"` on supported Opus models. Scoped values target one family.',
options: SERVICE_TIER_OPTIONS,
"Processing tier for OpenAI / OpenAI-Codex requests, and OpenAI-family models routed via OpenRouter (none = omit). Sent as `service_tier`.",
options: SERVICE_TIER_OPENAI_OPTIONS,
},
},
serviceTierSubagent: {
"tier.anthropic": {
type: "enum",
values: SERVICE_TIER_ANTHROPIC_VALUES,
default: "none",
ui: {
tab: "model",
group: "Sampling",
label: "Service Tier — Anthropic",
description:
'Processing tier for Claude requests. `priority` realizes fast mode (`speed: "fast"`) on supported direct Anthropic models; ignored on Bedrock/Vertex Claude and via OpenRouter.',
options: SERVICE_TIER_ANTHROPIC_OPTIONS,
},
},
"tier.google": {
type: "enum",
values: SERVICE_TIER_GOOGLE_VALUES,
default: "none",
ui: {
tab: "model",
group: "Sampling",
label: "Service Tier — Google",
description:
"Processing tier for Gemini (Google AI Studio + Vertex) requests, and Google-family models routed via OpenRouter (none = omit). Sent as the top-level `serviceTier` field.",
options: SERVICE_TIER_GOOGLE_OPTIONS,
},
},
"tier.subagent": {
type: "enum",
values: SERVICE_TIER_INHERIT_SETTING_VALUES,
default: "inherit",
ui: {
tab: "model",
group: "Sampling",
label: "Service Tier - Subagent",
label: "Service Tier — Subagent",
description:
"Service Tier for spawned task/eval subagents. Inherit = match the main agent's live tier (tracks /fast); pick a value to scope subagents independently.",
"Service Tier for spawned task/eval subagents. Inherit = match the main agent's live per-family tiers (tracks /fast); pick a value to apply it to whichever family the subagent's model belongs to.",
options: SERVICE_TIER_INHERIT_OPTIONS,
},
},
serviceTierAdvisor: {
"tier.advisor": {
type: "enum",
values: SERVICE_TIER_INHERIT_SETTING_VALUES,
default: "none",
ui: {
tab: "model",
group: "Sampling",
label: "Service Tier - Advisor",
label: "Service Tier — Advisor",
description:
"Service Tier for the advisor model. None = standard processing; Inherit = match the main agent's live tier; pick a value (e.g. Priority) to run the advisor on a faster serving path.",
"Service Tier for the advisor model. None = standard processing; Inherit = match the main agent's live per-family tiers; pick a value to apply it to the advisor model's family.",
options: SERVICE_TIER_INHERIT_OPTIONS,
condition: "advisorEnabled",
},
},
fastModeScope: {
type: "enum",
values: ["both", "openai", "claude"] as const,
default: "both",
ui: {
tab: "model",
group: "Sampling",
label: "Fast Mode Scope",
description:
'Which providers `/fast on` (and the fast-mode toggle) target. "both" = priority on every supported provider; "openai"/"claude" scope it to one family (mirrors serviceTier openai-only/claude-only).',
options: [
{ value: "both", label: "Both", description: "Priority on every supported provider" },
{
value: "openai",
label: "OpenAI only",
description: "Priority on OpenAI/OpenAI-Codex requests; ignored elsewhere",
},
{
value: "claude",
label: "Claude only",
description: "Anthropic fast mode on direct Claude requests; ignored elsewhere",
},
],
},
},
// Retries
"retry.enabled": { type: "boolean", default: true },
@@ -1788,6 +1795,19 @@ export const SETTINGS_SCHEMA = {
options: STT_MODEL_OPTIONS,
},
},
"stt.submitTrigger": {
type: "enum",
values: STT_SUBMIT_TRIGGER_VALUES,
default: "never",
ui: {
tab: "interaction",
group: "Speech",
label: "Speech-to-Text Submit Trigger",
description:
"Choose when speech dictation automatically submits: Never, Release (2+ words), Release with complete sentence, or When I Say Submit.",
options: STT_SUBMIT_TRIGGER_OPTIONS,
},
},
// ────────────────────────────────────────────────────────────────────────
// Context
@@ -4075,7 +4095,7 @@ export const SETTINGS_SCHEMA = {
group: "Subagents",
label: "Soft Subagent Request Budget",
description:
"Soft per-subagent request budget (assistant requests per run). Crossing it injects one steering notice asking the subagent to wrap up; at 1.5x the budget the run is aborted gracefully, salvaging partial output. 0 disables the guard. Bundled explore/quick_task agents use a lower built-in budget.",
"Soft per-subagent request budget (assistant requests per run). Crossing it injects one steering notice asking the subagent to wrap up; at 1.5x the budget the run is aborted gracefully, salvaging partial output. 0 disables the guard. Bundled explore/sonic agents use a lower built-in budget.",
options: [
{ value: "0", label: "Disabled" },
{ value: "40", label: "40 requests" },
@@ -4548,6 +4568,44 @@ export const SETTINGS_SCHEMA = {
},
},
"providers.streamFirstEventTimeoutSeconds": {
type: "number",
default: -1,
ui: {
tab: "providers",
group: "Timeouts",
label: "Stream First Event Timeout",
description:
"Seconds to wait for the first model stream event; -1 uses provider/env defaults, 0 disables the watchdog",
options: [
{ value: "-1", label: "Auto", description: "Use provider defaults and PI_* timeout env vars" },
{ value: "0", label: "Off", description: "Disable first-event timeout" },
{ value: "300", label: "5 minutes" },
{ value: "600", label: "10 minutes" },
{ value: "1800", label: "30 minutes" },
],
},
},
"providers.streamIdleTimeoutSeconds": {
type: "number",
default: -1,
ui: {
tab: "providers",
group: "Timeouts",
label: "Stream Idle Timeout",
description:
"Seconds a model stream may stay silent between events; -1 uses provider/env defaults, 0 disables the watchdog",
options: [
{ value: "-1", label: "Auto", description: "Use provider defaults and PI_* timeout env vars" },
{ value: "0", label: "Off", description: "Disable idle timeout" },
{ value: "300", label: "5 minutes" },
{ value: "600", label: "10 minutes" },
{ value: "1800", label: "30 minutes" },
],
},
},
"providers.openrouterVariant": {
type: "enum",
values: ["default", "nitro", "floor", "online", "exacto"] as const,
@@ -1148,6 +1148,53 @@ export class Settings {
// the incoherent "hashline edits without addressable anchors" state.
delete raw.readHashLines;
// serviceTier (single enum with scoped openai-only/claude-only sentinels)
// → per-family tier.openai/tier.anthropic/tier.google; serviceTierSubagent
// → tier.subagent; serviceTierAdvisor → tier.advisor. `fastModeScope` is
// dropped — per-family scoping is now expressed by the three tier settings.
const tierObj = isRecord(raw.tier) ? raw.tier : {};
let tierTouched = false;
const setTier = (family: string, value: unknown): void => {
if (value !== undefined && !(family in tierObj)) {
tierObj[family] = value;
tierTouched = true;
}
};
if (typeof raw.serviceTier === "string") {
switch (raw.serviceTier) {
case "priority":
setTier("openai", "priority");
setTier("anthropic", "priority");
setTier("google", "priority");
break;
case "openai-only":
setTier("openai", "priority");
break;
case "claude-only":
setTier("anthropic", "priority");
break;
case "auto":
case "default":
case "flex":
case "scale":
setTier("openai", raw.serviceTier);
break;
}
delete raw.serviceTier;
}
const mapInheritTier = (value: unknown): unknown =>
value === "openai-only" || value === "claude-only" ? "priority" : value;
if ("serviceTierSubagent" in raw) {
setTier("subagent", mapInheritTier(raw.serviceTierSubagent));
delete raw.serviceTierSubagent;
}
if ("serviceTierAdvisor" in raw) {
setTier("advisor", mapInheritTier(raw.serviceTierAdvisor));
delete raw.serviceTierAdvisor;
}
if (tierTouched) raw.tier = tierObj;
delete raw.fastModeScope;
return raw;
}
+7 -1
View File
@@ -114,7 +114,13 @@ function formatProfileAsMarkdown(profileJson: string): string {
*/
export async function startCpuProfile(): Promise<ProfilerSession> {
const v8 = await import("node:v8");
v8.setFlagsFromString("--allow-natives-syntax");
try {
// Enables `%GetOptimizationStatus` and friends when V8 natives are needed
// for ad-hoc profiling. Best-effort: Bun does not implement
// `setFlagsFromString` (oven-sh/bun#1702) but the CPU profiler itself
// works without it, so swallow the error and continue.
v8.setFlagsFromString("--allow-natives-syntax");
} catch {}
const { Session } = await import("node:inspector/promises");
const session = new Session();
@@ -0,0 +1,32 @@
---
description: "Prefer runtime.AddCleanup over runtime.SetFinalizer for new code (Go 1.24)"
condition: 'runtime\.SetFinalizer'
scope: "tool:edit(*.go), tool:write(*.go)"
---
Go 1.24 added `runtime.AddCleanup`, a finalization mechanism that is more flexible and less error-prone than `runtime.SetFinalizer`. The release notes state plainly: **new code should prefer `AddCleanup` over `SetFinalizer`.**
## Why AddCleanup wins
- Multiple cleanups may attach to one object; `SetFinalizer` allows only one.
- Cleanups may attach to interior pointers.
- Objects that form a reference cycle still get cleaned up — finalizers leak them.
- A cleanup does not resurrect its object or delay freeing it (and what it points to) by an extra GC cycle.
## Migration
```go
// Before
runtime.SetFinalizer(obj, func(o *T) { o.release() })
// After — the cleanup func receives a value you supply, NOT the object,
// so it cannot accidentally keep the object alive.
runtime.AddCleanup(obj, func(h handle) { h.release() }, obj.handle)
```
The cleanup argument must not reference `obj` itself (that would keep it reachable forever). Capture only the data the cleanup needs — a file descriptor, handle, or pointer that is independent of `obj`.
## Keep SetFinalizer only when
- The module targets a Go release older than 1.24.
- You depend on finalizer-specific behavior (e.g. object resurrection) that `AddCleanup` deliberately does not provide.
@@ -0,0 +1,36 @@
---
description: "Use for b.Loop() in benchmarks instead of the for i := 0; i < b.N; i++ loop (Go 1.24)"
interruptMode: never
scope: "tool:edit(*_test.go), tool:write(*_test.go)"
astCondition:
- "func $F($B *testing.B) { $$$PRE for $I := 0; $I < $B.N; $I++ { $$$BODY } $$$POST }"
---
Go 1.24 added `testing.B.Loop`. Write `for b.Loop() { ... }` instead of looping over `b.N`.
## Why
- Setup and teardown outside the loop run exactly once per `-count`, not once per `b.N` re-estimation, so expensive fixtures are no longer timed or repeated.
- The compiler keeps the loop's parameters and results alive, so it can't optimize away the body you are trying to measure — a classic `b.N` benchmarking footgun.
## Avoid
```go
func BenchmarkEncode(b *testing.B) {
for i := 0; i < b.N; i++ {
Encode(input)
}
}
```
## Use
```go
func BenchmarkEncode(b *testing.B) {
for b.Loop() {
Encode(input)
}
}
```
Requires Go 1.24+. If the module targets an older Go, keep the `b.N` loop.
@@ -0,0 +1,39 @@
---
description: "Use the standard library slices and maps packages instead of golang.org/x/exp/{slices,maps}"
condition:
- '"golang.org/x/exp/slices"'
- '"golang.org/x/exp/maps"'
scope: "tool:edit(*.go), tool:write(*.go)"
---
`golang.org/x/exp/slices` and `golang.org/x/exp/maps` were promoted into the standard library as `slices` and `maps` in Go 1.21. Import the stdlib packages in new code instead of the experimental ones.
## Migration
```go
// Before
import (
"golang.org/x/exp/slices"
"golang.org/x/exp/maps"
)
// After
import (
"slices"
"maps"
)
```
Most call sites are unchanged: `slices.Sort`, `slices.Contains`, `slices.Index`, `slices.Equal`, `maps.Clone`, etc.
## Watch the signature differences
The promoted APIs were tweaked, so a blind path swap can break the build:
- `x/exp/maps.Keys(m)` / `Values(m)` returned a slice; the stdlib `maps.Keys(m)` / `maps.Values(m)` return an **iterator** (`iter.Seq`). Use `slices.Collect(maps.Keys(m))` to recover a slice, or range over the iterator.
- `slices.SortFunc` takes a comparison returning `int` (cmp-style), matching the stdlib signature.
## Keep x/exp when
- The module's `go` directive is below 1.21 (stdlib `slices`/`maps` don't exist yet).
- You need an `x/exp` helper that was not promoted (e.g. parts of `x/exp/constraints` still live outside the stdlib).
@@ -0,0 +1,36 @@
---
description: "Use io and os instead of the deprecated io/ioutil package"
condition: '"io/ioutil"'
scope: "tool:edit(*.go), tool:write(*.go)"
---
`io/ioutil` has been deprecated since Go 1.16. Every function moved to `io` or `os` with the same behavior. Do not import it in new code.
## Mapping
| io/ioutil | Replacement |
| --- | --- |
| `ioutil.ReadAll` | `io.ReadAll` |
| `ioutil.ReadFile` | `os.ReadFile` |
| `ioutil.WriteFile` | `os.WriteFile` |
| `ioutil.ReadDir` | `os.ReadDir` (returns `[]os.DirEntry`, not `[]os.FileInfo`) |
| `ioutil.TempFile` | `os.CreateTemp` |
| `ioutil.TempDir` | `os.MkdirTemp` |
| `ioutil.NopCloser` | `io.NopCloser` |
| `ioutil.Discard` | `io.Discard` |
## Migration
```go
// Before
import "io/ioutil"
data, err := ioutil.ReadFile(path)
_ = ioutil.WriteFile(out, data, 0o644)
// After
import "os"
data, err := os.ReadFile(path)
_ = os.WriteFile(out, data, 0o644)
```
`os.ReadDir` returns `[]os.DirEntry` rather than `[]os.FileInfo` — call `entry.Info()` if you need the old `FileInfo`. Everything else is a drop-in rename.
@@ -0,0 +1,29 @@
---
description: "Build network addresses with net.JoinHostPort, not fmt.Sprintf(\"%s:%d\", host, port) — the Sprintf form breaks on IPv6"
condition: 'fmt\.Sprintf\("%s:%d"'
scope: "tool:edit(*.go), tool:write(*.go)"
---
Use `net.JoinHostPort(host, port)` to assemble a `host:port` address. `fmt.Sprintf("%s:%d", host, port)` produces invalid addresses for IPv6 hosts, which must be bracketed (`[::1]:80`). Go 1.25's `go vet` `hostport` analyzer flags exactly this pattern.
## Why
- An IPv6 literal like `::1` has its own colons, so `fmt.Sprintf("%s:%d", "::1", 80)` yields `::1:80` — unparseable by `net.Dial`.
- `net.JoinHostPort` adds the brackets when the host contains a colon and leaves IPv4/hostnames untouched.
## Avoid
```go
addr := fmt.Sprintf("%s:%d", host, port)
conn, err := net.Dial("tcp", addr)
```
## Use
```go
// port is a string here; convert an int with strconv.Itoa.
addr := net.JoinHostPort(host, strconv.Itoa(port))
conn, err := net.Dial("tcp", addr)
```
`net.JoinHostPort` takes the port as a string. For an `int` port, wrap it in `strconv.Itoa`. The function is available in every supported Go version.
@@ -0,0 +1,44 @@
---
description: "Use new(expr) for pointer-to-value helpers instead of `func ptr[T any](v T) *T { return &v }` (Go 1.26)"
interruptMode: never
scope: "tool:edit(*.go), tool:write(*.go)"
astCondition:
- "func $F($V $T) *$T { return &$V }"
- "func $F[$$$TP]($V $T) *$T { return &$V }"
---
Go 1.26 lets `new` take an expression: `new(expr)` allocates, stores `expr`, and returns its `*T`. That removes the need for hand-written `Ptr`/`boolPtr`/`Int64`-style helpers and the `x := v; p := &x` two-step.
## Why
- One builtin replaces a helper per type (`boolPtr`, `strPtr`, `int64Ptr`, …) and the generic `func Ptr[T any](v T) *T`.
- No extra function-call frame and no separate heap escape — the value is constructed directly in the allocation.
- The intent (`new(false)`) reads at the call site instead of hiding behind a helper name.
## Avoid
```go
// A helper that just takes a value and returns its address.
func boolPtr(v bool) *bool { return &v }
func strPtr(v string) *string { return &v }
func Ptr[T any](v T) *T { return &v }
cfg := Config{Enabled: boolPtr(true), Name: strPtr("svc")}
```
## Use
```go
cfg := Config{Enabled: new(true), Name: new("svc")}
// Was: x := int64(300); p := &x
p := new(int64(300))
```
`new(true)` / `new(false)` give you `*bool`; `new(expr)` works for any expression, including function results (`new(time.Now())`).
## Notes
- Requires Go 1.26+. If the module's `go` directive is older, keep the helper or the temp-variable form until the toolchain is bumped.
- This is for helpers that *only* take a value and return its address. A function that does real work before taking an address is not in scope.
- `new(T)` (a bare type) is unchanged and still zero-initializes.
@@ -0,0 +1,40 @@
---
description: Prefer math/rand/v2 over the legacy math/rand package
condition: '"math/rand"'
scope: "tool:edit(*.go), tool:write(*.go)"
---
Use `math/rand/v2` instead of the legacy `math/rand` package (stable since Go 1.22).
## Why
- No global `Seed`: `math/rand`'s top-level functions read a process-global generator (auto-seeded since Go 1.20), so a fixed seed is global mutable state that's easy to misuse; `v2` drops the global `Seed` entirely.
- Cleaner, better-bounded API: `rand.IntN(n)` / generic `rand.N(n)` replace `rand.Intn(n)`, and `Shuffle`, `Perm`, `Float64` carry over with clearer names.
- Modern generators: `v2` exposes `PCG` and `ChaCha8` sources instead of the old default LCG.
## Migration
```go
// Before
import "math/rand"
n := rand.Intn(100)
f := rand.Float64()
// After
import "math/rand/v2"
n := rand.IntN(100)
f := rand.Float64()
```
| math/rand | math/rand/v2 |
| --- | --- |
| `rand.Intn(n)` | `rand.IntN(n)` |
| `rand.Int63n(n)` | `rand.Int64N(n)` |
| `rand.Intn`/`Int31n` on a `*Rand` | `(*Rand).IntN` / `Int32N` |
| `rand.Seed(x)` | drop it — `v2` has no global seed |
| explicit `rand.New(rand.NewSource(seed))` | `rand.New(rand.NewPCG(s1, s2))` or `rand.NewChaCha8(seed)` |
## Keep math/rand only when
- You need a reproducible stream from a fixed seed via the classic `NewSource`/`Seed` API that a caller already depends on.
- Reach for `crypto/rand` instead when the values are security-sensitive — neither `math/rand` variant is cryptographically secure.
@@ -0,0 +1,45 @@
---
description: "Use for i := range n instead of the C-style for i := 0; i < n; i++ loop (Go 1.22)"
interruptMode: never
scope: "tool:edit(*.go), tool:write(*.go)"
astCondition:
- "for $I := 0; $I < $N; $I++ { $$$BODY }"
---
Go 1.22 lets `for` range over an integer. A plain counting loop from `0` to `n` with step `1` reads better as `for i := range n` (or `for range n` when the index is unused).
## Avoid
```go
for i := 0; i < n; i++ {
use(i)
}
for i := 0; i < len(s); i++ {
use(s[i])
}
```
## Use
```go
for i := range n {
use(i)
}
// Ranging the slice directly is usually clearer than indexing.
for i := range s {
use(s[i])
}
// Index unused → drop it entirely.
for range n {
tick()
}
```
## When it does not apply
- Non-zero start, step other than `++`, or a descending loop (`for i := n - 1; i >= 0; i--`) — keep the explicit form.
- The body reassigns the loop variable or depends on `i` surviving past the loop.
- Requires Go 1.22+. If the module's `go` directive is older, keep the classic loop.
@@ -8,6 +8,14 @@
* Registered by the lowest-priority `builtin-defaults` rule provider so any
* user/project/tool rule with the same name overrides the bundled copy.
*/
import goAddCleanup from "./go-add-cleanup.md" with { type: "text" };
import goBenchLoop from "./go-bench-loop.md" with { type: "text" };
import goExpPromoted from "./go-exp-promoted.md" with { type: "text" };
import goIoutil from "./go-ioutil.md" with { type: "text" };
import goJoinHostport from "./go-join-hostport.md" with { type: "text" };
import goNewExpr from "./go-new-expr.md" with { type: "text" };
import goRandV2 from "./go-rand-v2.md" with { type: "text" };
import goRangeInt from "./go-range-int.md" with { type: "text" };
import rsBoxLeak from "./rs-box-leak.md" with { type: "text" };
import rsFuturePrelude from "./rs-future-prelude.md" with { type: "text" };
import rsLazylock from "./rs-lazylock.md" with { type: "text" };
@@ -35,6 +43,14 @@ export interface BuiltinRuleSource {
/** All bundled default rules, ordered by name. */
export const BUILTIN_RULE_SOURCES: readonly BuiltinRuleSource[] = [
{ name: "go-add-cleanup", content: goAddCleanup },
{ name: "go-bench-loop", content: goBenchLoop },
{ name: "go-exp-promoted", content: goExpPromoted },
{ name: "go-ioutil", content: goIoutil },
{ name: "go-join-hostport", content: goJoinHostport },
{ name: "go-new-expr", content: goNewExpr },
{ name: "go-rand-v2", content: goRandV2 },
{ name: "go-range-int", content: goRangeInt },
{ name: "rs-box-leak", content: rsBoxLeak },
{ name: "rs-future-prelude", content: rsFuturePrelude },
{ name: "rs-lazylock", content: rsLazylock },
@@ -28,6 +28,7 @@ import { invalidateFsScanAfterWrite } from "../../tools/fs-cache-invalidation";
import { isInternalUrlPath } from "../../tools/path-utils";
import { enforcePlanModeWrite, resolvePlanPath, targetsLocalSandbox } from "../../tools/plan-mode-guard";
import { canonicalSnapshotKey } from "../file-snapshot-store";
import { isNotebookPath } from "../notebook";
import { readEditFileText, serializeEditFileText } from "../read-file";
import type { LspBatchRequest } from "../renderer";
@@ -123,6 +124,17 @@ export class HashlineFilesystem extends Filesystem {
return content;
}
async readBinary(relativePath: string): Promise<Uint8Array | undefined> {
const absolutePath = this.resolveAbsolute(relativePath);
if (isNotebookPath(absolutePath)) return undefined;
try {
return await fs.readFile(absolutePath);
} catch (error) {
if (isEnoent(error)) throw new NotFoundError(relativePath, error);
throw error;
}
}
async preflightWrite(relativePath: string, options?: PreflightWriteOptions): Promise<void> {
const fileOp = options?.fileOp;
if (fileOp?.kind === "rem") {
@@ -405,8 +405,10 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption
parentMnemopiSessionState: options.session.getMnemopiSessionState?.(),
parentTelemetry: options.session.getTelemetry?.(),
parentAgentId: options.session.getAgentId?.() ?? MAIN_AGENT_ID,
// Live source of truth for `serviceTierSubagent: inherit` (null = explicit none).
parentServiceTier: options.session.getServiceTier ? (options.session.getServiceTier() ?? null) : undefined,
// Live source of truth for `tier.subagent: inherit` (null = explicit none).
parentServiceTier: options.session.getServiceTierByFamily
? (options.session.getServiceTierByFamily() ?? null)
: undefined,
// Deliberately omit parentEvalSessionId: the parent's Python kernel is
// blocked on this bridge call, so sharing the eval session would deadlock
// (subagent queues behind the parent's in-flight execution, parent waits
+1 -1
View File
@@ -140,7 +140,7 @@ const HOST_DEFAULTED_SETTING_PATHS: SettingPath[] = [
"advisor.subagents",
"advisor.syncBacklog",
"advisor.immuneTurns",
"serviceTierAdvisor",
"tier.advisor",
];
const RPC_BACKGROUND_DEFAULTED_SETTING_PATHS: SettingPath[] = [
@@ -227,3 +227,124 @@ export async function setServerDisabled(filePath: string, name: string, disabled
await writeMCPConfigFile(filePath, updated);
}
/**
* Read the user-level force-enable list (allowlist that overrides a
* non-writable source config's `enabled: false`).
*/
export async function readEnabledServers(filePath: string): Promise<string[]> {
const config = await readMCPConfigFile(filePath);
return Array.isArray(config.enabledServers) ? config.enabledServers : [];
}
/**
* Add or remove a server name from the user-level force-enable list.
* The list overrides a discovered server's `enabled: false` flag but does
* NOT override the `disabledServers` denylist.
*/
export async function setServerForceEnabled(filePath: string, name: string, force: boolean): Promise<void> {
const config = await readMCPConfigFile(filePath);
const current = new Set(config.enabledServers ?? []);
if (force) {
current.add(name);
} else {
current.delete(name);
}
const updated: MCPConfigFile = {
...config,
enabledServers: current.size > 0 ? Array.from(current).sort() : undefined,
};
if (!updated.enabledServers) {
delete updated.enabledServers;
}
await writeMCPConfigFile(filePath, updated);
}
/** Paths and target state for toggling one MCP server across known config files. */
export interface SetMcpServerEnabledOptions {
userPath: string;
projectPath: string;
/**
* Absolute path to the loaded row's source mcp.json. Provide ONLY for
* formats this codebase owns (native `.omp/mcp.json` and `mcp-json`
* `mcp.json`/`.mcp.json`). Tool-owned configs (opencode.json, claude.json,
* settings.json …) MUST be omitted; we never mutate another tool's file.
*/
sourcePath?: string;
name: string;
enabled: boolean;
}
/**
* Flip a server's enabled/disabled state regardless of where it lives.
*
* Resolution order, mirroring `/mcp enable` / `/mcp disable` plus the dashboard
* fix for non-writable source configs:
*
* - Server found in `sourcePath` (writable) → write `enabled` on that entry.
* - Else server in project mcp.json → write `enabled` there.
* - Else server in user mcp.json → write `enabled` there.
* - Else (server defined in a tool-owned source like opencode.json, OR a
* purely discovered server):
* - Disable → add to the user-level `disabledServers` denylist.
* - Enable → add to the user-level `enabledServers` allowlist so the
* dashboard / runtime override the non-writable source's
* `enabled: false` flag.
*
* Cleanup invariants — on every call:
* - Re-enable clears any stale denylist entry so a server disabled via
* `/mcp disable` and re-enabled here doesn't stay suppressed.
* - Disable clears any stale allowlist entry so re-disabling a
* force-enabled server actually takes effect.
*/
export async function setMcpServerEnabled(options: SetMcpServerEnabledOptions): Promise<void> {
const { userPath, projectPath, sourcePath, name, enabled } = options;
const candidatePaths = [...new Set([sourcePath, projectPath, userPath].filter(path => path !== undefined))];
let updatedInConfig = false;
for (const filePath of candidatePaths) {
const config = await readMCPConfigFile(filePath);
const server = config.mcpServers?.[name];
if (server === undefined) continue;
await updateMCPServer(filePath, name, { ...server, enabled });
updatedInConfig = true;
break;
}
if (enabled) {
// Either we just wrote `enabled: true` on a writable source, or the
// server lives in a non-writable source whose `enabled: false` flag we
// need to override via the user allowlist. Either way the denylist
// entry (if any) must clear so the row becomes active.
const denied = await readDisabledServers(userPath);
if (denied.includes(name)) {
await setServerDisabled(userPath, name, false);
}
const forced = await readEnabledServers(userPath);
const isForced = forced.includes(name);
if (!updatedInConfig && !isForced) {
await setServerForceEnabled(userPath, name, true);
} else if (updatedInConfig && isForced) {
// Writable source now carries `enabled: true`; the override is
// redundant. Drop it so the user's allowlist stays tidy.
await setServerForceEnabled(userPath, name, false);
}
return;
}
// Disable path. Clear any force-enable override regardless of source so the
// disable actually sticks.
const forced = await readEnabledServers(userPath);
if (forced.includes(name)) {
await setServerForceEnabled(userPath, name, false);
}
if (!updatedInConfig) {
await setServerDisabled(userPath, name, true);
}
}
+10 -6
View File
@@ -9,7 +9,7 @@ import { mcpCapability } from "../capability/mcp";
import type { SourceMeta } from "../capability/types";
import type { MCPServer } from "../discovery";
import { loadCapability } from "../discovery";
import { readDisabledServers } from "./config-writer";
import { readDisabledServers, readEnabledServers } from "./config-writer";
import type { MCPServerConfig } from "./types";
/** Options for loading MCP configs */
@@ -105,16 +105,20 @@ export async function loadAllMCPConfigs(cwd: string, options?: LoadMCPConfigsOpt
? result.items
: result.items.filter(server => server._source.level !== "project");
// Load user-level disabled servers list
const disabledServers = new Set(await readDisabledServers(getMCPConfigPath("user", cwd)));
// Load user-level disable/force-enable lists. The denylist always wins; the
// allowlist overrides a non-writable source config's `enabled: false`.
const userPath = getMCPConfigPath("user", cwd);
const [disabledServers, forcedEnabled] = await Promise.all([
readDisabledServers(userPath).then(list => new Set(list)),
readEnabledServers(userPath).then(list => new Set(list)),
]);
// Convert to legacy format and preserve source metadata
let configs: Record<string, MCPServerConfig> = {};
let sources: Record<string, SourceMeta> = {};
for (const server of servers) {
const config = convertToLegacyConfig(server);
if (config.enabled === false || disabledServers.has(server.name)) {
continue;
}
if (disabledServers.has(server.name)) continue;
if (config.enabled === false && !forcedEnabled.has(server.name)) continue;
configs[server.name] = config;
sources[server.name] = server._source;
}
+35 -8
View File
@@ -153,14 +153,44 @@ function resolveCallbackHostname(redirectUri: string | undefined): string | unde
return parsed.hostname;
}
/**
* Resolve the client_id MCPOAuthFlow would use without doing any I/O —
* either the explicitly configured value or one embedded as a query parameter
* in the authorization URL. Returns `undefined` when no client_id is known
* statically, which is the trigger for dynamic client registration in
* {@link MCPOAuthFlow.#tryRegisterClient}.
*/
function staticClientIdFromConfig(config: MCPOAuthConfig): string | undefined {
const fromConfig = config.clientId?.trim();
if (fromConfig) return fromConfig;
try {
return new URL(config.authorizationUrl).searchParams.get("client_id") ?? undefined;
} catch {
return undefined;
}
}
function resolveCallbackOptions(config: MCPOAuthConfig): OAuthCallbackFlowOptions {
const redirectUri = resolveRedirectUri(config.redirectUri);
validateRedirectConfig(config, redirectUri);
// When a client_id is already pinned (config-supplied or embedded in the
// authorization URL), it was registered against a specific redirect URI.
// Silently advertising a different port at the authorize endpoint would
// be rejected by providers like Atlassian (HTTP 500 in the browser, local
// flow hangs until the 5-minute timeout), so fail fast instead.
//
// When no client_id is pinned, MCPOAuthFlow will attempt dynamic client
// registration on demand with whichever loopback URI we actually bound —
// the provider issues a client_id tied to *that* URI, so the random-port
// fallback remains safe for first-install DCR flows whose preferred port
// happens to be occupied.
const allowPortFallback = staticClientIdFromConfig(config) === undefined;
return {
preferredPort: resolveCallbackPort(config.callbackPort, redirectUri),
callbackPath: resolveCallbackPath(config.callbackPath, redirectUri),
callbackHostname: resolveCallbackHostname(redirectUri),
redirectUri,
allowPortFallback,
};
}
@@ -400,6 +430,7 @@ export class MCPOAuthFlow extends OAuthCallbackFlow {
"Content-Type": "application/x-www-form-urlencoded",
},
body: params.toString(),
signal: this.ctrl.signal,
});
if (!response.ok) {
@@ -453,14 +484,7 @@ export class MCPOAuthFlow extends OAuthCallbackFlow {
}
#resolveClientId(config: MCPOAuthConfig): string | undefined {
const fromConfig = config.clientId?.trim();
if (fromConfig) return fromConfig;
try {
return new URL(config.authorizationUrl).searchParams.get("client_id") ?? undefined;
} catch {
return undefined;
}
return staticClientIdFromConfig(config);
}
#resourceFromAuthorizationUrl(authorizationUrl: string): string | undefined {
try {
@@ -495,6 +519,7 @@ export class MCPOAuthFlow extends OAuthCallbackFlow {
"Content-Type": "application/json",
Accept: "application/json",
},
signal: this.ctrl.signal,
body: JSON.stringify({
client_name: "oh-my-pi",
redirect_uris: [redirectUri],
@@ -560,6 +585,7 @@ export class MCPOAuthFlow extends OAuthCallbackFlow {
const response = await this.#fetch(wellKnownUrl, {
method: "GET",
headers: { Accept: "application/json" },
signal: this.ctrl.signal,
});
if (!response.ok) return null;
const metadata = (await response.json()) as { registration_endpoint?: string };
@@ -578,6 +604,7 @@ export class MCPOAuthFlow extends OAuthCallbackFlow {
method: "GET",
redirect: "manual",
headers: { Accept: "text/plain,text/html,application/json" },
signal: this.ctrl.signal,
});
if (response.status < 400) return;
const body = await response.text();
+3
View File
@@ -111,7 +111,10 @@ export const MCP_CONFIG_SCHEMA_URL =
export interface MCPConfigFile {
$schema?: string;
mcpServers?: Record<string, MCPServerConfig>;
/** Names to hide regardless of any source `enabled` flag. Highest precedence. */
disabledServers?: string[];
/** Names to force-enable when a non-writable source reports `enabled: false`. */
enabledServers?: string[];
}
// =============================================================================
@@ -24,7 +24,9 @@ import {
truncateToWidth,
visibleWidth,
} from "@oh-my-pi/pi-tui";
import { getMCPConfigPath, logger } from "@oh-my-pi/pi-utils";
import { Settings } from "../../../config/settings";
import { setMcpServerEnabled } from "../../../mcp/config-writer";
import { getTabBarTheme } from "../../../modes/shared";
import { theme } from "../../../modes/theme/theme";
import { matchesAppInterrupt } from "../../../modes/utils/keybinding-matchers";
@@ -263,6 +265,14 @@ export class ExtensionDashboard implements Component {
const sm = this.settings ?? Settings.instance;
if (!sm) return;
// MCP toggles route through the canonical denylist in
// `~/.omp/agent/mcp.json` so `/mcp list`, the MCP runtime, and this
// dashboard agree on every server's enabled state (issue #3827).
if (extensionId.startsWith("mcp:")) {
void this.#toggleMcpExtension(extensionId, enabled, sm);
return;
}
const disabled = ((sm.get("disabledExtensions") as string[]) ?? []).slice();
if (enabled) {
const index = disabled.indexOf(extensionId);
@@ -281,6 +291,42 @@ export class ExtensionDashboard implements Component {
void this.#refreshFromState();
}
async #toggleMcpExtension(extensionId: string, enabled: boolean, sm: Settings): Promise<void> {
const name = extensionId.slice("mcp:".length);
try {
await setMcpServerEnabled({
userPath: getMCPConfigPath("user", this.cwd),
projectPath: getMCPConfigPath("project", this.cwd),
sourcePath: this.#writableMcpSourcePath(extensionId),
name,
enabled,
});
} catch (error) {
logger.warn("Failed to persist MCP toggle", { name, enabled, error: String(error) });
}
// Reconcile `settings.disabledExtensions` with the canonical mcp.json
// state so a legacy `mcp:<name>` flag from before this routing change
// doesn't keep the server marked disabled after the user re-enables it
// via the UI.
const stored = ((sm.get("disabledExtensions") as string[]) ?? []).slice();
const had = stored.indexOf(extensionId);
if (enabled && had !== -1) {
stored.splice(had, 1);
sm.set("disabledExtensions", stored);
this.#applyDisabledExtensions(stored);
}
await this.#refreshFromState();
}
#writableMcpSourcePath(extensionId: string): string | undefined {
const extension = this.#state.extensions.find(ext => ext.id === extensionId);
if (!extension) return undefined;
if (extension.source.provider !== "native" && extension.source.provider !== "mcp-json") return undefined;
return extension.path;
}
async #refreshFromState(): Promise<void> {
const refreshToken = ++this.#refreshToken;
// Remember the current tab so it survives the re-sort.
@@ -4,7 +4,7 @@
*/
import * as path from "node:path";
import { fuzzyMatch } from "@oh-my-pi/pi-tui";
import { logger } from "@oh-my-pi/pi-utils";
import { getMCPConfigPath, logger } from "@oh-my-pi/pi-utils";
import type { ContextFile } from "../../../capability/context-file";
import type { ExtensionModule } from "../../../capability/extension-module";
import type { Hook } from "../../../capability/hook";
@@ -22,6 +22,7 @@ import {
isProviderEnabled,
loadCapability,
} from "../../../discovery";
import { readDisabledServers, readEnabledServers } from "../../../mcp/config-writer";
import type {
DashboardState,
Extension,
@@ -141,12 +142,32 @@ export async function loadAllExtensions(cwd?: string, disabledIds?: string[]): P
logger.warn("Failed to load extension-modules capability", { error: String(error) });
}
// Load MCP servers
// Load MCP servers. The dashboard mirrors `/mcp list` (issue #3827) by
// honoring the same disable signals: the dashboard-private settings list,
// the per-server `enabled: false` flag, and the user-level `disabledServers`
// denylist that `/mcp disable` writes through `setServerDisabled`. The
// user-level `enabledServers` allowlist overrides a non-writable source's
// `enabled: false` (e.g. opencode.json) but never the denylist.
try {
const userMcpPath = cwd ? getMCPConfigPath("user", cwd) : undefined;
const [mcpDisabledNames, mcpForcedEnabled] = await Promise.all([
userMcpPath
? readDisabledServers(userMcpPath)
.then(list => new Set(list))
.catch(() => new Set<string>())
: Promise.resolve(new Set<string>()),
userMcpPath
? readEnabledServers(userMcpPath)
.then(list => new Set(list))
.catch(() => new Set<string>())
: Promise.resolve(new Set<string>()),
]);
const mcps = await loadCapability<MCPServer>("mcps", loadOpts);
for (const server of mcps.all) {
const id = makeExtensionId("mcp", server.name);
const isDisabled = disabledExtensions.has(id);
const forced = mcpForcedEnabled.has(server.name);
const sourceSaysDisabled = server.enabled === false && !forced;
const isDisabled = mcpDisabledNames.has(server.name) || disabledExtensions.has(id) || sourceSaysDisabled;
const isShadowed = (server as { _shadowed?: boolean })._shadowed;
const providerEnabled = isProviderEnabled(server._source.provider);

Some files were not shown because too many files have changed in this diff Show More