feat(ai): consolidated OpenAI-family streaming and add OpenRouter API support

This change introduces a new `openrouter` API type and extensively refactors OpenAI-family streaming providers, centralizing shared logic and improving robustness.

Key changes include:
- **Unified OpenAI-family Logic:** Consolidated core utilities, compat resolution, request shaping, and stream processing into `openai-shared.ts`, reducing duplication across `openai-completions`, `openai-responses`, and `openai-codex-responses`.
- **OpenRouter API Type:** Introduced a dedicated `openrouter` API type with dual-surface compatibility, allowing it to dispatch requests as either OpenAI Chat Completions or Responses.
- **Enhanced Provider Integration:**
    - Improved Perplexity search to leverage shared OpenAI streaming transports, including API-key fallback to OpenRouter and support for Perplexity's Responses API.
    - Integrated xAI-specific logic directly into the shared `stream.ts` dispatch, removing the dedicated `xai-responses` provider.
    - Refined credential parsing for Google Gemini CLI and handling of Azure deployment names.
- **Robustness & Consistency:** Improved error handling for Codex, standardized output token parameter resolution, and ensured consistent application of reasoning suppression across all Chat Completions dialects.
- **New Documentation:** Added `provider-endpoint-constraints.md` to detail endpoint-specific behaviors and quirks for various providers.
- **Telemetry & Debugging:** Extended telemetry propagation to advisor calls and overflow compaction tasks. Improved debugging for Codex WebSocket failures and stream error messages.
- **Tooling & Security:** Updated browser stealth scripts to prevent detection and added a new `ts-no-inline-cast-access` TTSR rule.
This commit is contained in:
can1357
2026-06-17 21:16:15 +02:00
parent eb67863e75
commit d4317d3d20
74 changed files with 8799 additions and 3608 deletions
+28
View File
@@ -15,6 +15,7 @@
"@typescript/native-preview": "catalog:",
"lint-staged": "catalog:",
"prettier": "catalog:",
"ts-morph": "catalog:",
"typescript": "catalog:",
},
},
@@ -42,6 +43,7 @@
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
"arktype": "catalog:",
"partial-json": "catalog:",
"zod": "catalog:",
},
@@ -55,6 +57,7 @@
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
"arktype": "catalog:",
"zod": "catalog:",
},
"devDependencies": {
@@ -92,6 +95,7 @@
"@puppeteer/browsers": "catalog:",
"@types/turndown": "catalog:",
"@xterm/headless": "catalog:",
"arktype": "catalog:",
"chalk": "catalog:",
"diff": "catalog:",
"fflate": "catalog:",
@@ -349,6 +353,7 @@
"@types/turndown": "5.0.6",
"@typescript/native-preview": "7.0.0-dev.20260609.1",
"@xterm/headless": "^6.0.0",
"arktype": "^2.2.0",
"beautiful-mermaid": "^1.1.3",
"chalk": "^5.6.2",
"chart.js": "^4.5.1",
@@ -375,6 +380,7 @@
"regexp-tree": "^0.1.27",
"solid-js": "^1.9.13",
"tailwindcss": "^4.3.0",
"ts-morph": "^28.0.0",
"turndown": "7.2.4",
"turndown-plugin-gfm": "1.0.2",
"typescript": "^6.0.3",
@@ -395,6 +401,10 @@
"@anush008/tokenizers-win32-x64-msvc": ["@anush008/tokenizers-win32-x64-msvc@0.0.0", "", { "os": "win32", "cpu": "x64" }, "sha512-/5kP0G96+Cr6947F0ZetXnmL31YCaN15dbNbh2NHg7TXXRwfqk95+JtPP5Q7v4jbR2xxAmuseBqB4H/V7zKWuw=="],
"@ark/schema": ["@ark/schema@0.56.0", "", { "dependencies": { "@ark/util": "0.56.0" } }, "sha512-ECg3hox/6Z/nLajxXqNhgPtNdHWC9zNsDyskwO28WinoFEnWow4IsERNz9AnXRhTZJnYIlAJ4uGn3nlLk65vZA=="],
"@ark/util": ["@ark/util@0.56.0", "", {}, "sha512-BghfRC8b9pNs3vBoDJhcta0/c1J1rsoS1+HgVUreMFPdhz/CRAKReAu57YEllNaSy98rWAdY1gE+gFup7OXpgA=="],
"@babel/code-frame": ["@babel/code-frame@7.29.7", "", { "dependencies": { "@babel/helper-validator-identifier": "^7.29.7", "js-tokens": "^4.0.0", "picocolors": "^1.1.1" } }, "sha512-Aup7aUOfpbAUg2ROOJN6Iw5f9DMBlzu0mIkm/malLQFN/YQgO48wCj0Kxa3sEHJvPVFg7siR+qRInwXd2qhQKw=="],
"@babel/compat-data": ["@babel/compat-data@7.29.7", "", {}, "sha512-locTkQyKvwIEgBzVrn8693ebc97F2U8ZHjbXwDXJ5Fn2TCpNwTlKcaKLkdHop5c/icOFE7qt7Q9JC5hnKNa6Gg=="],
@@ -853,6 +863,8 @@
"@tokenizer/token": ["@tokenizer/token@0.3.0", "", {}, "sha512-OvjF+z51L3ov0OyAU0duzsYuvO01PH7x4t6DJx+guahgTnBHkhJdG7soQeTSFLWN3efnHyibZ4Z8l2EuWwJN3A=="],
"@ts-morph/common": ["@ts-morph/common@0.29.0", "", { "dependencies": { "minimatch": "^10.0.1", "path-browserify": "^1.0.1", "tinyglobby": "^0.2.14" } }, "sha512-35oUmphHbJvQ/+UTwFNme/t2p3FoKiGJ5auTjjpNTop2dyREspirjMy82PLSC1pnDJ8ah1GU98hwpVt64YXQsg=="],
"@tybys/wasm-util": ["@tybys/wasm-util@0.10.2", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-RoBvJ2X0wuKlWFIjrwffGw1IqZHKQqzIchKaadZZfnNpsAYp2mM0h36JtPCjNDAHGgYez/15uMBpfGwchhiMgg=="],
"@types/babel__core": ["@types/babel__core@7.20.5", "", { "dependencies": { "@babel/parser": "^7.20.7", "@babel/types": "^7.20.7", "@types/babel__generator": "*", "@types/babel__template": "*", "@types/babel__traverse": "*" } }, "sha512-qoQprZvz5wQFJwMDqeseRXWv3rqMvhgpbXFfVyWhbx9X47POIA6i/+dXefEmZKoAgOaTdaIgNSMqMIU61yRyzA=="],
@@ -907,12 +919,18 @@
"argparse": ["argparse@1.0.10", "", { "dependencies": { "sprintf-js": "~1.0.2" } }, "sha512-o5Roy6tNG4SL/FOkCAN6RzjiakZS25RLYFrcMttJqbdd8BWrnA+fGz57iN5Pb06pvBGvl5gQ0B48dJlslXvoTg=="],
"arkregex": ["arkregex@0.0.5", "", { "dependencies": { "@ark/util": "0.56.0" } }, "sha512-ncYjBdLlh5/QnVsAA8De16Tc9EqmYM7y/WU9j+236KcyYNUXogpz3sC4ATIZYzzLxwI+0sEOaQLEmLmRleaEXw=="],
"arktype": ["arktype@2.2.0", "", { "dependencies": { "@ark/schema": "0.56.0", "@ark/util": "0.56.0", "arkregex": "0.0.5" } }, "sha512-t54MZ7ti5BhOEvzEkgKnWvqj+UbDfWig+DHr5I34xatymPusKLS0lQpNJd8M6DzmIto2QGszHfNKoFIT8tMCZQ=="],
"async": ["async@3.2.6", "", {}, "sha512-htCUDlxyyCLMgaM3xXg0C0LW2xqfuQ6p05pCEIsXuyQ+a1koYKTuBMzRNwmybfLgvJDMd0r1LTn4+E0Ti6C2AA=="],
"babel-plugin-jsx-dom-expressions": ["babel-plugin-jsx-dom-expressions@0.40.7", "", { "dependencies": { "@babel/helper-module-imports": "7.18.6", "@babel/plugin-syntax-jsx": "^7.18.6", "@babel/types": "^7.20.7", "html-entities": "2.3.3", "parse5": "^7.1.2" }, "peerDependencies": { "@babel/core": "^7.20.12" } }, "sha512-/O6JWUmjv03OI9lL2ry9bUjpD5S3PclM55RRJEyCdcFZ5W2SEA/59d+l2hNsk3gI6kiWRdRPdOtqZmsQzFN1pQ=="],
"babel-preset-solid": ["babel-preset-solid@1.9.12", "", { "dependencies": { "babel-plugin-jsx-dom-expressions": "^0.40.6" }, "peerDependencies": { "@babel/core": "^7.0.0", "solid-js": "^1.9.12" }, "optionalPeers": ["solid-js"] }, "sha512-LLqnuKVDlKpyBlMPcH6qEvs/wmS9a+NczppxJ3ryS/c0O5IiSFOIBQi9GzyiGDSbcJpx4Gr87jyFTos1MyEuWg=="],
"balanced-match": ["balanced-match@4.0.4", "", {}, "sha512-BLrgEcRTwX2o6gGxGOCNyMvGSp35YofuYzw9h1IMTRmKqttAZZVU67bdb9Pr2vUHA8+j3i2tJfjO6C6+4myGTA=="],
"base64-js": ["base64-js@1.5.1", "", {}, "sha512-AKpaYlHn8t4SVbOHCy+b5+KKgvR4vrsD8vbvrbiQJps7fKDTkjkDry6ji0rUJjC0kzbNePLwzxq8iypo41qeWA=="],
"baseline-browser-mapping": ["baseline-browser-mapping@2.10.37", "", { "bin": { "baseline-browser-mapping": "dist/cli.cjs" } }, "sha512-girxaJ7WZssDOFhzCGZTDKoTa1gk6A1TbflaYTpykLJ4UU9Fz9kx1aREM8JCuoVHbL8X8T/mJg7w2oYSq72Oig=="],
@@ -927,6 +945,8 @@
"boolean": ["boolean@3.2.0", "", {}, "sha512-d0II/GO9uf9lfUHH2BQsjxzRJZBdsjgsBiW4BvhWk/3qoKwQFjIDVN19PfX8F2D/r9PCMTtLWjYVCFrpeYUzsw=="],
"brace-expansion": ["brace-expansion@5.0.6", "", { "dependencies": { "balanced-match": "^4.0.2" } }, "sha512-kLpxurY4Z4r9sgMsyG0Z9uzsBlgiU/EFKhj/h91/8yHu0edo7XuixOIH3VcJ8kkxs6/jPzoI6U9Vj3WqbMQ94g=="],
"browserslist": ["browserslist@4.28.2", "", { "dependencies": { "baseline-browser-mapping": "^2.10.12", "caniuse-lite": "^1.0.30001782", "electron-to-chromium": "^1.5.328", "node-releases": "^2.0.36", "update-browserslist-db": "^1.2.3" }, "bin": { "browserslist": "cli.js" } }, "sha512-48xSriZYYg+8qXna9kwqjIVzuQxi+KYWp2+5nCYnYKPTr0LvD89Jqk2Or5ogxz0NUMfIjhh2lIUX/LyX9B4oIg=="],
"bun-types": ["bun-types@1.3.14", "", { "dependencies": { "@types/node": "*" } }, "sha512-4N0ig0fEomHt5R0KCFWjovxow98rIoRwKolrYdCcknNwMekCXRnWEUvgu5soYV8QXtVsrUD8B95MBOZGPvr6KQ=="],
@@ -955,6 +975,8 @@
"cliui": ["cliui@8.0.1", "", { "dependencies": { "string-width": "^4.2.0", "strip-ansi": "^6.0.1", "wrap-ansi": "^7.0.0" } }, "sha512-BSeNnyus75C4//NQ9gQt1/csTXyo/8Sb+afLAkzAptFuMsod9HFokGNudZpi/oQV73hnVK+sR+5PVRMd+Dr7YQ=="],
"code-block-writer": ["code-block-writer@13.0.3", "", {}, "sha512-Oofo0pq3IKnsFtuHqSF7TqBfr71aeyZDVJ0HpmqB7FBM2qEigL0iPONSCZSO9pE9dZTAxANe5XHG9Uy0YMv8cg=="],
"color": ["color@5.0.3", "", { "dependencies": { "color-convert": "^3.1.3", "color-string": "^2.1.3" } }, "sha512-ezmVcLR3xAVp8kYOm4GS45ZLLgIE6SPAFoduLr6hTDajwb3KZ2F46gulK3XpcwRFb5KKGCSezCBAY4Dw4HsyXA=="],
"color-convert": ["color-convert@3.1.3", "", { "dependencies": { "color-name": "^2.0.0" } }, "sha512-fasDH2ont2GqF5HpyO4w0+BcewlhHEZOFn9c1ckZdHpJ56Qb7MHhH/IcJZbBGgvdtwdwNbLvxiBEdg336iA9Sg=="],
@@ -1195,6 +1217,8 @@
"mimic-function": ["mimic-function@5.0.1", "", {}, "sha512-VP79XUPxV2CigYP3jWwAUFSku2aKqBH7uTAapFWCBqutsbmDo96KY5o8uh6U+/YSIn5OxJnXp73beVkpqMIGhA=="],
"minimatch": ["minimatch@10.2.5", "", { "dependencies": { "brace-expansion": "^5.0.5" } }, "sha512-MULkVLfKGYDFYejP07QOurDLLQpcjk7Fw+7jXS2R2czRQzR56yHRveU5NDJEOviH+hETZKSkIk5c+T23GjFUMg=="],
"minimist": ["minimist@1.2.8", "", {}, "sha512-2yyAR8qBkN3YuheJanUpWC5U3bb5osDywNB8RzDVlDwDHbocAJveqqj1u8+SVD7jkWT4yvsHCpWqqWqAxb0zCA=="],
"minipass": ["minipass@5.0.0", "", {}, "sha512-3FnjYuehv9k6ovOEbyOswadCDPX1piCfhV8ncmYtHOjuPwylVWsghTLo7rabjC3Rx5xD4HDx8Wm1xnMF7S5qFQ=="],
@@ -1249,6 +1273,8 @@
"partial-json": ["partial-json@0.1.7", "", {}, "sha512-Njv/59hHaokb/hRUjce3Hdv12wd60MtM9Z5Olmn+nehe0QDAsRtRbJPvJ0Z91TusF0SuZRIvnM+S4l6EIP8leA=="],
"path-browserify": ["path-browserify@1.0.1", "", {}, "sha512-b7uo2UCUOYZcnF/3ID0lulOJi/bafxa1xPe7ZPsammBSpjSWQkjNxlt635YGS2MiR9GjvuXCtz2emr3jbsz98g=="],
"path-expression-matcher": ["path-expression-matcher@1.5.0", "", {}, "sha512-cbrerZV+6rvdQrrD+iGMcZFEiiSrbv9Tfdkvnusy6y0x0GKBXREFg/Y65GhIfm0tnLntThhzCnfKwp1WRjeCyQ=="],
"path-is-absolute": ["path-is-absolute@1.0.1", "", {}, "sha512-AVbw3UJ2e9bq64vSaS9Am0fje1Pa8pbGqTTsmXfaIiMpnr5DlDhfJOuLj9Sf95ZPVDAUerDfEk88MPmPe7UCQg=="],
@@ -1379,6 +1405,8 @@
"triple-beam": ["triple-beam@1.4.1", "", {}, "sha512-aZbgViZrg1QNcG+LULa7nhZpJTZSLm/mXnHXnbAbjmN5aSa0y7V+wvv6+4WaBtpISJzThKy+PIPxc1Nq1EJ9mg=="],
"ts-morph": ["ts-morph@28.0.0", "", { "dependencies": { "@ts-morph/common": "~0.29.0", "code-block-writer": "^13.0.3" } }, "sha512-Wp3tnZ2bzwxyTZMtgWVzXDfm7lB1Drz+y9DmmYH/L702PQhPyVrp3pkou3yIz4qjS14GY9kcpmLiOOMvl8oG1g=="],
"tslib": ["tslib@2.8.1", "", {}, "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w=="],
"turndown": ["turndown@7.2.4", "", { "dependencies": { "@mixmark-io/domino": "^2.2.0" } }, "sha512-I8yFsfRzmzK0WV1pNNOA4A7y4RDfFxPRxb3t+e3ui14qSGOxGtiSP6GjeX+Y6CHb7HYaFj7ECUD7VE5kQMZWGQ=="],
+2
View File
@@ -540,6 +540,8 @@ The built-in model policy currently links OpenAI `codex-spark` variants to `gpt-
The `compat` block on a provider or model overrides the URL-based auto-detection in `packages/catalog/src/compat/openai.ts` (`buildOpenAICompat`). It is validated by `OpenAICompatSchema` in `packages/coding-agent/src/config/models-config-schema.ts` and consumed by every `openai-completions` transport (`packages/ai/src/providers/openai-completions.ts`). The canonical type is `OpenAICompat` in `packages/catalog/src/types.ts`.
Endpoint-specific exceptions that interact with these fields are cataloged in [Provider endpoint constraints](./provider-endpoint-constraints.md).
`models.yml` accepts the following keys (all optional; unset falls back to URL detection):
Request shaping:
+394
View File
@@ -0,0 +1,394 @@
# Provider endpoint constraints
Provider integrations are not interchangeable just because they speak an
OpenAI-shaped HTTP protocol. A request is shaped by four layers at once:
1. endpoint family: `openai-completions`, `openai-responses`,
`openai-codex-responses`, `anthropic-messages`, etc.
2. gateway/auth surface: OpenRouter, Vercel AI Gateway, Azure OpenAI, Copilot,
Alibaba Coding Plan, Kimi Code, Fireworks/Firepass, and similar hosts
3. model metadata and `compat` overrides
4. request context: tools, images, reasoning mode, stateful session, service tier
Use this page when adding a provider, adding a compat flag, or moving logic out
of a provider-specific branch. The goal is to encode endpoint constraints once,
at the narrowest layer that actually owns the behavior.
Related references:
- [Providers](./providers.md) — provider availability, credentials, custom providers
- [Model and Provider Configuration](./models.md) — `models.yml`, routing, and compat fields
- [Provider streaming internals](./provider-streaming-internals.md) — stream event normalization
- [Adding a provider](./adding-a-provider.md) — catalog/auth wiring for a new provider
## Baseline rules
- Prefer compat metadata over provider-name branches when behavior is model or
endpoint configurable.
- Keep transport mechanics transport-local. Codex websocket replay, Responses
item routing, and Chat Completions SSE decoding are protocol behavior, not
generic compat flags.
- Scope fallbacks to the failing capability. A strict-tool failure should not
disable unrelated features. A stale Responses chain should reset chain state,
not disable Responses entirely.
- Do not emit defaults that alter gateway routing. OpenRouter is the known case
for default `max_tokens`, but any gateway can treat optional fields as routing
hints.
- Stop retrying after visible side effects. Once text or a tool call is visible
to the user/session, retry policy must avoid duplicate output and duplicate
tool execution.
## 1. Choose the endpoint family first
### OpenAI Chat Completions compatible
Preserve these differences instead of treating every host as stock OpenAI:
- `stream_options.include_usage` is only safe when compat says streaming usage
is supported.
- `store: false` is accepted only by some hosts.
- max-output caps use either `max_tokens` or `max_completion_tokens`.
- stop sequences and frequency penalty live on this path among the current
OpenAI-like endpoint set.
- OpenRouter-style reasoning and routing fields are not portable to other
OpenAI-compatible hosts unless compat says so.
### OpenAI Responses compatible
Responses request shape is its own dialect:
- uses `input`, `instructions`, `store`, `prompt_cache_key`, optional
`previous_response_id`, and `max_output_tokens`
- can default official OpenAI requests to stateful chaining with
`previous_response_id` plus `store: true`
- third-party Responses proxies may reject native reasoning history, encrypted
reasoning replay, or `previous_response_id`
- stream completion is authoritative only after `response.completed` or
`response.incomplete`; a stream close before either terminal event should fail
for OpenAI Responses rather than surface partial output as success
### OpenAI Codex Responses
Codex is not plain Responses with a different URL. Keep these as Codex transport
policy:
- Codex account headers and beta headers
- `x-codex-turn-state` and `x-models-etag`
- optional websocket transport plus SSE fallback
- `responsesLite`
- prompt-cache/session ids used as transport state
- websocket-only `previous_response_id` chaining; SSE never chains
- Codex retry/replay rules, including reconnect and SSE replay boundaries
- provider retry only before user-visible content has been emitted
- whitespace-only tool-call argument loop breaker
Codex intentionally does not forward caller max-token caps because the backend
rejects them.
### Anthropic/OpenAI dual-surface providers
Kimi Code and Synthetic can be called as OpenAI-compatible or
Anthropic-compatible. The shim may need to:
- switch `format`
- rebuild an Anthropic model when needed
- map internal reasoning to Anthropic thinking budgets
- delegate back to OpenAI Completions
Do not encode these as one-way provider migrations; they are runtime surface
selection decisions.
## 2. Apply gateway and auth overlays
These constraints sit above the endpoint family. They affect auth, headers,
routing, model ids, or usage accounting.
### Azure OpenAI
- Chat Completions base URL reshapes to
`/deployments/{deployment}/chat/completions?api-version=...`.
- Deployment names may differ from model ids through
`AZURE_OPENAI_DEPLOYMENT_NAME_MAP`.
### GitHub Copilot
- The API key is parsed into an access token.
- Dynamic Copilot headers depend on messages/images.
- `premiumRequests` must survive usage population and replacement.
- Base URL may be resolved from the raw key.
### OpenRouter
- Adds attribution/cache headers.
- Supports routing suffixes such as `:nitro` and `:floor`.
- Appends a routing suffix only when the model id has no explicit suffix after
the last provider path segment.
- Uses nested `reasoning` request fields.
- Routes providers through the OpenRouter `provider` object.
- Has special cache-write usage accounting.
- Has strict-tool fallback for Anthropic grammar-size failures.
- Should omit catalog-default `max_tokens` unless the caller explicitly set a
cap, so upstream routing is not biased.
### Vercel AI Gateway
- Routing preferences go under `providerOptions.gateway.only` and
`providerOptions.gateway.order`.
- Do not reuse OpenRouter's `provider` object.
### Alibaba Coding Plan
- API key bytes may be JSON carrying `{ token, enterpriseUrl }`.
- Auth and base URL resolution are provider-specific.
### Kimi Code
- The OpenAI-compatible path needs common Kimi headers.
- It also participates in the OpenAI/Anthropic dual-surface shim.
### Fireworks and Firepass
- Wire model ids need provider-specific mapping.
- Fireworks can conflict when DeepSeek-style `thinking` and OpenAI-style
`reasoning_effort` are both present after extra body fields are merged.
## 3. Serialize request parameters by dialect
Check these before adding or forwarding a field:
- **Model id.** Some models resolve a wire id from reasoning effort.
Firepass/Fireworks transform ids. OpenRouter suffix handling is path-segment
aware.
- **Max output tokens.** Kimi-family models may require a max-token field even
when the caller did not set one. OpenRouter should omit catalog defaults unless
explicit. Codex drops caller caps. Responses uses `max_output_tokens`; Chat
Completions uses `max_tokens` or `max_completion_tokens`.
- **Service tier.** Completions, Responses, and Codex all handle service tiers,
but allowed values and pricing multipliers differ. Codex has a special
priority multiplier for `gpt-5.5`.
- **Prompt cache/session.** OpenAI Responses uses `prompt_cache_key`.
OpenRouter Responses uses `session_id`. Codex uses prompt cache/session ids for
transport state. Anthropic-style cache control requires `cache_control` on a
text part.
- **Stateful chaining.** Official OpenAI Responses may chain by default.
Third-party endpoints generally should not. Codex chains only on websocket
`response.create`.
## 4. Map reasoning and thinking explicitly
Reasoning fields are not interchangeable.
### OpenAI-style `reasoning_effort`
- Effort values come from compat/model metadata.
- If reasoning is disabled but the host has no real off switch, map to the
lowest supported effort rather than inventing an unsupported value.
### Responses `reasoning`
- Uses `reasoning: { effort, summary }`.
- Can include `reasoning.encrypted_content` for replay.
- xAI Grok models may require omitting `reasoning.effort`.
- Some compat paths inject the GPT-5 `# Juice: 0 !important` developer scaffold.
### OpenRouter `reasoning`
- Uses nested `reasoning: { effort }`.
- Disabling reasoning must send `reasoning: { enabled: false }`; OpenRouter can
otherwise default reasoning models into thinking.
### Z.AI / GLM
- Uses `thinking: { type: "enabled" }` or
`thinking: { type: "disabled" }`.
- GLM 5.2 reasoning-effort models may also receive `reasoning_effort`.
- Tool requests need `tool_stream: true`.
### Qwen
- One dialect uses top-level `enable_thinking`.
- Another uses `chat_template_kwargs.enable_thinking`.
### Anthropic-compatible format
- Reasoning maps to Anthropic thinking enablement and thinking-budget tokens,
not OpenAI-style fields.
### DeepSeek reasoning history
- DeepSeek-compatible reasoning models may require exact `reasoning_content`
replay.
- Some variants require replay on every assistant turn, not only tool-call turns.
- Synthetic `"."` placeholders are acceptable for Kimi/OpenRouter-style compat,
but not DeepSeek V4 exact replay.
### Reasoning plus tool choice
- DeepSeek reasoning models can reject `tool_choice` while thinking is enabled.
- Kimi can reject forced tool choice while thinking is enabled.
- Compat needs both policies: disable reasoning for any tool choice, and disable
reasoning only for forced tool choice.
### xAI Grok through Responses/SuperGrok
Keep these independent:
- omit `reasoning.effort`
- include or drop encrypted reasoning replay
- filter reasoning-history wrappers
Some models reject only one of those fields; do not collapse them into one
"Grok mode" branch.
## 5. Normalize tools and schemas per endpoint
### Strict tools
Strict schemas are not a universal capability:
- some providers support strict tools
- some reject mixed strict/non-strict tools
- some reject strictified schemas
- OpenRouter Anthropic models can fail with “compiled grammar too large”
Retry-without-strict should be a compat recovery policy scoped to the current
session/provider path.
### Responses and Codex custom tools
Responses and Codex both support freeform custom grammar tools for `apply_patch`.
Both disable request-level parallel tool calls when any custom grammar tool is
present. Responses additionally:
- sanitizes schemas differently
- quarantines invalid enum/const schema contradictions
- repairs orphan tool outputs into assistant notes
- synthesizes placeholder outputs for orphan tool calls
Codex applies its own request transformation before sending.
### Tool choice
Before emitting `tool_choice`:
- confirm the endpoint supports it
- downgrade forced choice to `auto` if forced choice is unsupported
- drop `tool_choice: "none"` when no tools are emitted
- drop forced named tool choice if that named tool was filtered out
### Anthropic through LiteLLM/Bedrock
- If history contains tool calls/results and `context.tools` is undefined, send
`tools: []` as a sentinel.
- If `context.tools = []`, treat it as explicit opt-out and do not emit the
sentinel.
### Mistral / Devstral
- Tool-call ids must be exactly 9 alphanumeric characters.
- Some flows need a synthetic assistant bridge after tool results before the next
user message.
### Custom tool outputs
Responses/Codex must remember whether a call was `custom_tool_call`; the paired
output must then be `custom_tool_call_output`, not `function_call_output`.
### MiniMax-compatible streaming arguments
Tool arguments can stream as objects instead of JSON strings. Deep-merge object
deltas, then emit one final concat-safe JSON delta.
## 6. Convert messages and replay history safely
- **System/developer roles.** Reasoning models may require `developer`. Some
providers do not support `developer` and must downgrade to `user`. Some reject
multiple system messages and need coalescing.
- **Responses system prompts.** Responses usually uses top-level `instructions`.
Reasoning models that support `developer` put system prompts inline as
developer messages.
- **Assistant content.** Some OpenAI-compatible backends mirror array content
literally, so assistant content is normalized to a string. Tool-call replay may
require `content: ""` or `content: "."` instead of `null`.
- **Thinking replay.** Some models want thinking as visible text. Others need a
provider-specific reasoning field. Some permit synthetic placeholders; others
need exact replay.
- **Vision.** If the model/provider cannot accept images, convert image input and
tool-result images to placeholders. Some Qwen/Dashscope-compatible modes are
text-only even when the high-level model is multimodal.
- **Native Responses history.** Native provider payload replay is model-bound.
Strip or normalize foreign reasoning signatures. Shared code normalizes
Responses pipe-separated tool ids, hashes foreign item ids, and can filter
reasoning history.
## 7. Decode streams by provider behavior, not just schema
- **Generic OpenAI-compatible streams.** Keepalive chunks, role-only deltas, and
empty `choices: []` are not progress. Idle watchdogs must not sleep forever
because of them.
- **Mistral Medium 3.5-style content.** `delta.content` can be an array/object of
text parts, not a string; normalize it to text.
- **DeepSeek via NVIDIA/native/proxies.** Some endpoints leak chat-template
markers like `<|...|>` into visible content. Buffering is required because
markers can be split across chunks.
- **DeepSeek/template-leak tool calls.** Some providers leak tool-call markup in
text while also producing structured tool calls. Markup healing belongs in the
stream decoder policy, not endpoint business logic.
- **MiniMax-M3 cumulative reasoning.** Reasoning deltas may be cumulative
snapshots. Deduplicate by reasoning field signature.
- **Responses streams.** Route parallel items by `output_index`, `item_id`,
call-id aliases, and prefixed `fc_` aliases. Tolerate missing
`content_part.added` or `output_item.added`. Finalize pending tool calls at the
terminal event.
- **Terminal behavior.** Chat Completions can break after `finish_reason` plus
usage. Responses breaks on `response.completed` or `response.incomplete`. Tool
calls with `stop` promote to `toolUse`. Codex/Responses `end_turn:false` maps
to `pause_turn`.
- **Ollama length failures.** `finish_reason: length` with no visible content is
treated as context-window failure and mapped to an error.
## 8. Preserve usage and cost semantics
- OpenRouter `prompt_tokens_details.cache_write_tokens` is billed differently:
subtract it from input tokens and emit it as cache-write usage.
- DeepSeek native `prompt_cache_miss_tokens` is the billed input portion, not a
separate cache-write charge. Do not double-count it.
- GitHub Copilot `premiumRequests` must survive when usage is populated or
replaced.
- Responses and Codex both adjust cost by resolved service tier, but Codex uses
different multipliers.
## 9. Implement recovery at the right boundary
- **Strict tool fallback.** `400`/`422` schema or strict-tool failures should
disable strict tools for the appropriate session scope and retry non-strict.
- **OpenAI Responses stateful fallback.** Stale, invalid, or unsupported
`previous_response_id` resets chain state and retries with full context. Zero
Data Retention disables chaining immediately.
- **Codex websocket fallback.** Websocket connection errors, stale sockets,
connection limits, retry-budget exhaustion, or unsafe partial output can
trigger reconnect or SSE replay.
- **Codex whitespace tool-loop breaker.** Codex can stream whitespace-only
tool-call argument deltas indefinitely. Cap events/chars, drop the degenerate
partial tool call, and retry only when safe.
- **Codex `previous_response_id` fallback.** Stale or unsupported ids are chain
breaks and retry with full context, but only for websocket because SSE never
chains.
- **Provider retry before content.** Codex retries retryable provider stream
errors only before user-visible content has been emitted.
## 10. Checklist for a new constraint
Before adding a branch or compat field, answer these in order:
1. Is this endpoint-family behavior, gateway behavior, model behavior, or request
context behavior?
2. Can it be represented by existing `compat` metadata?
3. If not, is a new compat field better than a provider-name branch?
4. Does the field need provider-level defaults, model-level overrides, or both?
5. Does it interact with tools, images, reasoning, stateful Responses chains, or
service tier?
6. Can retry happen before visible text/tool calls only?
7. Does usage accounting still preserve cache reads/writes, billed input, service
tier multipliers, and provider-specific counters such as Copilot
`premiumRequests`?
+1 -1
View File
@@ -208,7 +208,7 @@ Provider-specific (not fully abstracted):
- [`../../ai/src/utils/event-stream.ts`](../packages/ai/src/utils/event-stream.ts) — generic stream queue + final-result resolution.
- [`../../ai/src/utils/json-parse.ts`](../packages/ai/src/utils/json-parse.ts) — partial JSON parsing for streamed tool arguments.
- [`../../ai/src/providers/anthropic.ts`](../packages/ai/src/providers/anthropic.ts) — Anthropic event translation and tool JSON delta accumulation.
- [`../../ai/src/providers/openai-responses.ts`](../packages/ai/src/providers/openai-responses.ts), [`openai-responses-shared.ts`](../packages/ai/src/providers/openai-responses-shared.ts), [`openai-codex-responses.ts`](../packages/ai/src/providers/openai-codex-responses.ts), [`azure-openai-responses.ts`](../packages/ai/src/providers/azure-openai-responses.ts) — Responses-family event translation and status mapping.
- [`../../ai/src/providers/openai-responses.ts`](../packages/ai/src/providers/openai-responses.ts), [`openai-shared.ts`](../packages/ai/src/providers/openai-shared.ts), [`openai-codex-responses.ts`](../packages/ai/src/providers/openai-codex-responses.ts), [`azure-openai-responses.ts`](../packages/ai/src/providers/azure-openai-responses.ts) — Responses-family event translation and status mapping.
- [`../../ai/src/providers/google.ts`](../packages/ai/src/providers/google.ts), [`google-gemini-cli.ts`](../packages/ai/src/providers/google-gemini-cli.ts), [`google-vertex.ts`](../packages/ai/src/providers/google-vertex.ts) — Gemini stream chunk-to-block translation variants.
- [`../../ai/src/providers/google-shared.ts`](../packages/ai/src/providers/google-shared.ts) — Gemini finish-reason mapping and shared conversion rules.
- [`../../ai/src/providers/amazon-bedrock.ts`](../packages/ai/src/providers/amazon-bedrock.ts), [`openai-completions.ts`](../packages/ai/src/providers/openai-completions.ts), [`ollama.ts`](../packages/ai/src/providers/ollama.ts), [`cursor.ts`](../packages/ai/src/providers/cursor.ts), [`pi-native-client.ts`](../packages/ai/src/providers/pi-native-client.ts) — additional built-in stream adapters using the same event contract.
+1 -1
View File
@@ -4,7 +4,7 @@ Providers are the model backends `omp` can route requests to: Anthropic, OpenAI,
A **provider** is the account or backend namespace, such as `anthropic`, `openai`, `google`, or `ollama`. A **model** is a concrete model under that provider, selected as `provider/model-id`, such as `anthropic/claude-opus-4-6`. Disabling a provider removes every model under it from selection; if you only want to narrow individual models, use model settings instead.
This page covers how providers become available, how credentials are resolved, the provider/environment-variable map, local engines, disabling providers, and custom providers. For model selection and the full `models.yml` schema, see [Model and Provider Configuration](./models.md). For config-file locations and merge precedence, see [Settings](./settings.md). For credential storage and login flows in depth, see [Secrets and credentials](./secrets.md). For the complete environment-variable reference, see [Environment variables](./environment-variables.md). For local engine setup, see [Local models](./local-models.md). For context-file discovery providers, see [Context files](./context-files.md).
This page covers how providers become available, how credentials are resolved, the provider/environment-variable map, local engines, disabling providers, and custom providers. For endpoint-specific request, reasoning, tool, stream, usage, and retry constraints, see [Provider endpoint constraints](./provider-endpoint-constraints.md). For model selection and the full `models.yml` schema, see [Model and Provider Configuration](./models.md). For config-file locations and merge precedence, see [Settings](./settings.md). For credential storage and login flows in depth, see [Secrets and credentials](./secrets.md). For the complete environment-variable reference, see [Environment variables](./environment-variables.md). For local engine setup, see [Local models](./local-models.md). For context-file discovery providers, see [Context files](./context-files.md).
## How `omp` decides a provider is available
+6 -3
View File
@@ -53,6 +53,7 @@
"@types/turndown": "5.0.6",
"@typescript/native-preview": "7.0.0-dev.20260609.1",
"@xterm/headless": "^6.0.0",
"arktype": "^2.2.0",
"beautiful-mermaid": "^1.1.3",
"chalk": "^5.6.2",
"chart.js": "^4.5.1",
@@ -79,6 +80,7 @@
"regexp-tree": "^0.1.27",
"solid-js": "^1.9.13",
"tailwindcss": "^4.3.0",
"ts-morph": "^28.0.0",
"turndown": "7.2.4",
"turndown-plugin-gfm": "1.0.2",
"typescript": "^6.0.3",
@@ -179,11 +181,12 @@
},
"devDependencies": {
"@biomejs/biome": "catalog:",
"prettier": "catalog:",
"@types/bun": "catalog:",
"@typescript/native-preview": "catalog:",
"typescript": "catalog:",
"lint-staged": "catalog:"
"lint-staged": "catalog:",
"prettier": "catalog:",
"ts-morph": "catalog:",
"typescript": "catalog:"
},
"lint-staged": {
"*.{js,ts,jsx,tsx,json,jsonc,css}": "biome check --write --no-errors-on-unmatched"
+1 -1
View File
@@ -13,7 +13,7 @@
*/
import { ProviderHttpError } from "@oh-my-pi/pi-ai/errors";
import { parseTextSignature } from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
import { parseTextSignature } from "@oh-my-pi/pi-ai/providers/openai-shared";
import { transformMessages } from "@oh-my-pi/pi-ai/providers/transform-messages";
import type { AssistantMessage, FetchImpl, Message, Model } from "@oh-my-pi/pi-ai/types";
import {
+22 -1
View File
@@ -1,6 +1,27 @@
# Changelog
## [Unreleased]
### Added
- Added `getOpenRouterHeaders` utility to export standard OpenRouter integration headers
### Changed
- Replaced the dedicated `xai-responses` provider with a unified `openai-responses` path that handles xAI-specific reasoning effort stripping dynamically
- Updated OpenAI Responses stream handling to throw a clearer error message when a stream closes without a terminal response event
- Consolidated shared OpenAI-compatible routing and strict-tool fallback helpers across Chat Completions and Responses providers.
- Consolidated the OpenAI-family provider stack: merged `openai-responses-shared` into `openai-shared` and removed the now-dead `openai-responses-shared` re-export shim; folded the three duplicated `service_tier` request blocks and the per-provider wire model-id transform into shared `applyOpenAIServiceTier`/`applyWireModelIdTransform` helpers; moved residual provider-name wire-quirk checks (DeepSeek special-token strip, cumulative reasoning deltas, Ollama empty-length context error, OpenAI tool-call-id cap, Fireworks thinking drop, OpenRouter/OpenAI Responses request fields) into resolved compat fields; shared the Responses stream per-block accumulation helpers plus the terminal pending-tool-call finalization (`finalizePendingResponsesToolCalls`) and toolUse/pause stop-reason promotion (`promoteResponsesToolUseStopReason`) between `processResponsesStream` and the Codex stream handler; and removed the redundant `getOpenAIResponsesCacheSessionId` alias in favor of `getOpenAIResponsesPromptCacheKey`.
- Centralized OpenAI-family request-param policy into shared `resolveOpenAIOutputTokenParam` (output-token field selection, OpenRouter default-cap omission, `alwaysSendMaxTokens` defaulting, model/provider clamp), `applyOpenAIGatewayRouting` (OpenRouter `provider` + Vercel AI Gateway `providerOptions`), and `applyOpenAIExtraBody` (extra-body merge + Fireworks thinking drop) helpers used by both Chat Completions and Responses `buildParams`, and moved the Chat Completions reasoning/thinking dialect dispatch (`applyChatCompletionsReasoningParams` + `disableChatCompletionsReasoningForDialect`) plus the `OpenAICompletionsParams` request type into `openai-shared` alongside `applyResponsesReasoningParams`. As a consistency consequence, direct `streamOpenAIResponses` calls (bypassing `streamSimple`) now emit `max_output_tokens` for `alwaysSendMaxTokens` (Kimi-family) models even without a caller cap — matching Chat Completions and the value `streamSimple` already supplied.
- Centralized OpenAI-family reasoning compat resolution behind a shared `resolveOpenAICompatPolicy` consumed by both Chat Completions and Responses request builders. Shared policy now drives tool-choice reasoning suppression, dialect-specific disable encoding, reasoning-history replay filters, encrypted-reasoning inclusion, Mistral/OpenAI tool-call-id modes, stream healing/DeepSeek token stripping, and xAI/OpenRouter cache-affinity wiring instead of endpoint-local provider/model checks.
### Fixed
- Fixed Google Gemini CLI credential parsing to correctly prioritize `projectId` over `project_id` even when empty, and drop non-string values gracefully
- Fixed OpenRouter Responses requests to omit default max token fields unless an explicit caller cap is provided, preventing upstream filtering issues
- Fixed Chat Completions reasoning suppression (`disableReasoningOnToolChoice` / `disableReasoningOnForcedToolChoice`) to turn thinking off symmetrically across every dialect via a shared `disableChatCompletionsReasoningForDialect` helper. Previously the conflict path only deleted `reasoning_effort`/`reasoning` (and set Z.AI `thinking: { type: "disabled" }` on the forced branch alone), leaving Qwen `enable_thinking`, Qwen chat-template `chat_template_kwargs.enable_thinking`, and OpenRouter nested `reasoning` enabled — so those hosts could keep thinking on under forced/required tool choice and re-trip the incompatibility the policy guards against. OpenRouter is now set to `{ reasoning: { enabled: false } }` (not deleted, which OpenRouter treats as default-on).
- Fixed OpenRouter Responses requests to send `session_id` from `sessionId` in the request body for sticky provider routing and observability grouping.
- Fixed OpenRouter Responses request shaping to preserve provider routing, variant suffixes, caller header overrides, and strict-tool fallback behavior while omitting only unsafe default max-token caps.
- Fixed OpenAI Responses stateful chaining so a non-ZDR stale `previous_response_id` retry keeps `store: true`: the full-context retry stays chainable on the next turn and the consecutive stale-failure circuit breaker trips after the configured limit instead of alternating cold turns. Zero Data Retention rejections still disable chaining on the first strike.
## [16.0.5] - 2026-06-17
@@ -3828,4 +3849,4 @@ _Dedicated to Peter's shoulder ([@steipete](https://twitter.com/steipete))_
## [0.9.4] - 2025-11-26
Initial release with multi-provider LLM support.
Initial release with multi-provider LLM support.
+1
View File
@@ -40,6 +40,7 @@
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
"arktype": "catalog:",
"partial-json": "catalog:",
"zod": "catalog:"
},
+2
View File
@@ -1,4 +1,5 @@
export { type ZodType, z } from "zod/v4";
export { type, type Type } from "arktype";
export * from "./api-registry";
export * from "./auth-broker";
export { type AuthGatewayBootOptions, type ModelResolver, startAuthGateway } from "./auth-gateway/server";
@@ -38,6 +39,7 @@ export * from "./usage/openai-codex-reset";
export * from "./usage/zai";
export * from "./utils/anthropic-auth";
export * from "./utils/event-stream";
export * from "./utils/openrouter-headers";
export * from "./utils/overflow";
export * from "./utils/retry";
export * from "./utils/schema";
@@ -0,0 +1,84 @@
import { describe, expect, it } from "bun:test";
import { createCodexProviderStreamError, isRetryableCodexFailureEvent } from "../openai-codex-responses";
describe("isRetryableCodexFailureEvent", () => {
it("classifies retryable codes from nested error.code, error.type, then rawEvent.code", () => {
expect(isRetryableCodexFailureEvent({ error: { code: "server_error" } })).toBe(true);
expect(isRetryableCodexFailureEvent({ error: { type: "internal_error" } })).toBe(true);
expect(isRetryableCodexFailureEvent({ code: "model_error" })).toBe(true);
});
it("prefers nested error.code over the top-level code (distinct precedence from the factory)", () => {
// The error.code chain wins, so a non-retryable nested code with no retryable message
// is NOT retryable even though the top-level code is retryable.
expect(isRetryableCodexFailureEvent({ code: "server_error", error: { code: "bad_request" } })).toBe(false);
});
it("detects retryable messages when the code is absent or unknown", () => {
expect(isRetryableCodexFailureEvent({ message: "Please retry your request shortly" })).toBe(true);
expect(isRetryableCodexFailureEvent({ code: "bad_request", message: "we are overloaded" })).toBe(true);
expect(isRetryableCodexFailureEvent({ response: { message: "service unavailable" } })).toBe(true);
});
it("returns false for non-retryable code and message", () => {
expect(isRetryableCodexFailureEvent({ code: "bad_request", message: "invalid input" })).toBe(false);
expect(isRetryableCodexFailureEvent({})).toBe(false);
});
it("falls back to response.error when rawEvent.error is not an object", () => {
expect(isRetryableCodexFailureEvent({ error: "boom", response: { error: { code: "server_error" } } })).toBe(true);
});
it("ignores mistyped fields instead of failing the whole parse", () => {
// A numeric top-level `code` must not poison parsing; the retryable message is still honored.
expect(isRetryableCodexFailureEvent({ code: 500, message: "internal error while processing" })).toBe(true);
// Same tolerance nested: a non-string error.code is dropped while a valid error.message survives.
expect(isRetryableCodexFailureEvent({ error: { code: 123, message: "server error happened" } })).toBe(true);
});
});
describe("createCodexProviderStreamError", () => {
it("prefers top-level rawEvent.code over nested error code", () => {
expect(createCodexProviderStreamError({ code: "outer_code", error: { code: "inner_code" } }).code).toBe(
"outer_code",
);
});
it("falls back to nested error.code then error.type", () => {
expect(createCodexProviderStreamError({ error: { code: "inner_code" } }).code).toBe("inner_code");
expect(createCodexProviderStreamError({ error: { type: "inner_type" } }).code).toBe("inner_type");
});
it("leaves code undefined when nothing supplies one", () => {
expect(createCodexProviderStreamError({ message: "boom" }).code).toBeUndefined();
});
it("marks retryable error events and formats them as error events", () => {
const err = createCodexProviderStreamError({ type: "error", code: "server_error", message: "kaboom" });
expect(err.retryable).toBe(true);
expect(err.code).toBe("server_error");
expect(err.message).toContain("error event");
expect(err.message).toContain("kaboom");
});
it("formats non-error failures via the response-failure path", () => {
const err = createCodexProviderStreamError({
type: "response.failed",
response: { error: { message: "downstream blew up" } },
});
expect(err.retryable).toBe(false);
expect(err.code).toBeUndefined();
expect(err.message).toContain("response failed");
expect(err.message).toContain("downstream blew up");
});
it("falls back to response.error when rawEvent.error is not an object", () => {
const err = createCodexProviderStreamError({
error: "boom",
response: { error: { code: "server_error", message: "nested boom" } },
});
expect(err.code).toBe("server_error");
expect(err.retryable).toBe(true);
expect(err.message).toContain("nested boom");
});
});
@@ -8,10 +8,8 @@ import type {
ServiceTier,
StreamFunction,
StreamOptions,
Tool,
ToolChoice,
} from "../types";
import { normalizeSystemPrompts } from "../utils";
import { createAbortSourceTracker } from "../utils/abort";
import { AssistantMessageEventStream } from "../utils/event-stream";
import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
@@ -23,44 +21,24 @@ import {
import { postOpenAIStream } from "../utils/openai-http";
import { sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
import { mapToOpenAIResponsesToolChoice } from "../utils/tool-choice";
import { getOpenAIResponsesCacheSessionId } from "./openai-responses";
import type { ResponseCreateParamsStreaming, ResponseStreamEvent } from "./openai-responses-wire";
import {
appendResponsesToolResultMessages,
applyCommonResponsesSamplingParams,
applyResponsesReasoningParams,
convertResponsesAssistantMessage,
convertResponsesInputContent,
buildResponsesInput,
createInitialResponsesAssistantMessage,
getOpenAIResponsesPromptCacheKey,
isOpenAIResponsesProgressEvent,
normalizeResponsesToolCallIdForTransform,
parseAzureDeploymentNameMap,
processResponsesStream,
repairOrphanResponsesToolCalls,
} from "./openai-responses-shared";
import type {
Tool as OpenAITool,
ResponseCreateParamsStreaming,
ResponseInput,
ResponseStreamEvent,
} from "./openai-responses-wire";
import { transformMessages } from "./transform-messages";
} from "./openai-shared";
export { parseAzureDeploymentNameMap } from "./openai-shared";
const DEFAULT_AZURE_API_VERSION = "v1";
const AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE =
"Azure OpenAI responses stream timed out while waiting for the first event";
export function parseAzureDeploymentNameMap(value: string | undefined): Map<string, string> {
const map = new Map<string, string>();
if (!value) return map;
for (const entry of value.split(",")) {
const trimmed = entry.trim();
if (!trimmed) continue;
const [modelId, deploymentName] = trimmed.split("=", 2);
if (!modelId || !deploymentName) continue;
map.set(modelId.trim(), deploymentName.trim());
}
return map;
}
function resolveDeploymentName(model: Model<"azure-openai-responses">, options?: AzureOpenAIResponsesOptions): string {
if (options?.azureDeploymentName) {
return options.azureDeploymentName;
@@ -193,13 +171,13 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
abortSignal: options?.signal,
isProgressItem: isOpenAIResponsesProgressEvent,
});
let sawCompleted = false;
let sawTerminalResponseEvent = false;
await processResponsesStream(timedOpenaiStream, output, stream, model, {
onFirstToken: () => {
if (!firstTokenTime) firstTokenTime = Date.now();
},
onCompleted: () => {
sawCompleted = true;
sawTerminalResponseEvent = true;
},
});
@@ -212,8 +190,8 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
throw new Error("Request was aborted");
}
if (!sawCompleted) {
throw new Error("Azure OpenAI responses stream closed before response.completed was received");
if (!sawTerminalResponseEvent) {
throw new Error("Azure OpenAI responses stream closed before a terminal response event was received");
}
if (output.stopReason === "aborted" || output.stopReason === "error") {
@@ -240,14 +218,6 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
return stream;
};
function normalizeAzureBaseUrl(baseUrl: string): string {
return baseUrl.replace(/\/+$/, "");
}
function buildDefaultBaseUrl(resourceName: string): string {
return `https://${resourceName}.openai.azure.com/openai/v1`;
}
function resolveAzureConfig(
model: Model<"azure-openai-responses">,
options?: AzureOpenAIResponsesOptions,
@@ -260,7 +230,7 @@ function resolveAzureConfig(
let resolvedBaseUrl = baseUrl;
if (!resolvedBaseUrl && resourceName) {
resolvedBaseUrl = buildDefaultBaseUrl(resourceName);
resolvedBaseUrl = `https://${resourceName}.openai.azure.com/openai/v1`;
}
if (!resolvedBaseUrl && model.baseUrl) {
@@ -274,7 +244,7 @@ function resolveAzureConfig(
}
return {
baseUrl: normalizeAzureBaseUrl(resolvedBaseUrl),
baseUrl: resolvedBaseUrl.replace(/\/+$/, ""),
apiVersion,
};
}
@@ -322,13 +292,21 @@ function buildParams(
options: AzureOpenAIResponsesOptions | undefined,
deploymentName: string,
) {
const messages = convertMessages(model, context, true);
const systemRole = model.reasoning && model.compat.supportsDeveloperRole ? "developer" : "system";
const messages = buildResponsesInput({
model,
context,
strictResponsesPairing: true,
systemRole,
includeThinkingSignatures: true,
developerStringContent: true,
});
const params: AzureOpenAIResponsesSamplingParams = {
model: deploymentName,
input: messages,
stream: true,
prompt_cache_key: getOpenAIResponsesCacheSessionId(options),
prompt_cache_key: getOpenAIResponsesPromptCacheKey(options),
// Encrypted reasoning replay (applyResponsesReasoningParams) requires
// stateless responses, matching the openai provider.
store: false,
@@ -337,7 +315,13 @@ function buildParams(
applyCommonResponsesSamplingParams(params, options, model);
if (context.tools) {
params.tools = convertTools(context.tools);
params.tools = context.tools.map(tool => ({
type: "function" as const,
name: tool.name,
description: tool.description || "",
parameters: sanitizeSchemaForOpenAIResponses(toolWireSchema(tool)),
strict: false,
}));
if (options?.toolChoice) {
const toolChoice = mapToOpenAIResponsesToolChoice(options.toolChoice);
if (
@@ -355,60 +339,3 @@ function buildParams(
return params;
}
function convertMessages(
model: Model<"azure-openai-responses">,
context: Context,
strictResponsesPairing: boolean,
): ResponseInput {
const messages: ResponseInput = [];
const transformedMessages = transformMessages(context.messages, model, normalizeResponsesToolCallIdForTransform);
const knownCallIds = new Set<string>();
const customCallIds = new Set<string>();
const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
if (systemPrompts.length > 0) {
const role = model.reasoning && model.compat.supportsDeveloperRole ? "developer" : "system";
for (const systemPrompt of systemPrompts) {
messages.push({ role, content: systemPrompt });
}
}
let msgIndex = 0;
for (const msg of transformedMessages) {
if (msg.role === "user" || msg.role === "developer") {
const content = convertResponsesInputContent(msg.content, model.input.includes("image"));
if (!content) continue;
messages.push({
role: "user",
content: msg.role === "developer" && typeof msg.content === "string" ? msg.content.toWellFormed() : content,
});
} else if (msg.role === "assistant") {
const outputItems = convertResponsesAssistantMessage(
msg as AssistantMessage,
model,
msgIndex,
knownCallIds,
true,
customCallIds,
);
if (outputItems.length === 0) continue;
messages.push(...outputItems);
} else if (msg.role === "toolResult") {
appendResponsesToolResultMessages(messages, msg, model, strictResponsesPairing, knownCallIds, customCallIds);
}
msgIndex++;
}
return repairOrphanResponsesToolCalls(messages);
}
function convertTools(tools: Tool[]): OpenAITool[] {
return tools.map(tool => ({
type: "function",
name: tool.name,
description: tool.description || "",
parameters: sanitizeSchemaForOpenAIResponses(toolWireSchema(tool)),
strict: false,
}));
}
+24 -27
View File
@@ -13,6 +13,7 @@ import {
getGeminiCliHeaders,
} from "@oh-my-pi/pi-catalog/wire/gemini-headers";
import { extractHttpStatusFromError, fetchWithRetry, readSseJson } from "@oh-my-pi/pi-utils";
import { z } from "zod/v4";
import { ProviderHttpError } from "../errors";
import type {
Api,
@@ -185,16 +186,22 @@ function extractErrorMessage(errorText: string): string {
return errorText;
}
interface GeminiCliApiKeyPayload {
token?: unknown;
projectId?: unknown;
project_id?: unknown;
refreshToken?: unknown;
expiresAt?: unknown;
email?: unknown;
refresh?: unknown;
expires?: unknown;
}
const optionalCredentialString = z.string().optional().catch(undefined);
const geminiCliCredentialsSchema = z
.object({
token: optionalCredentialString,
projectId: optionalCredentialString,
project_id: optionalCredentialString,
refreshToken: optionalCredentialString,
refresh: optionalCredentialString,
email: optionalCredentialString,
expiresAt: z.unknown().optional(),
expires: z.unknown().optional(),
})
.loose()
.catch({});
interface ParsedGeminiCliCredentials {
accessToken: string;
projectId: string;
@@ -215,32 +222,22 @@ export function parseGeminiCliCredentials(apiKeyRaw: string): ParsedGeminiCliCre
const missingCredentialsMessage =
"Missing token or projectId in Google Cloud credentials. Use /login to re-authenticate.";
let parsed: GeminiCliApiKeyPayload;
let rawCredentials: unknown;
try {
parsed = JSON.parse(apiKeyRaw) as GeminiCliApiKeyPayload;
rawCredentials = JSON.parse(apiKeyRaw);
} catch {
throw new Error(invalidCredentialsMessage);
}
const parsed = geminiCliCredentialsSchema.parse(rawCredentials);
const projectId =
typeof parsed.projectId === "string"
? parsed.projectId
: typeof parsed.project_id === "string"
? parsed.project_id
: undefined;
if (typeof parsed.token !== "string" || typeof projectId !== "string") {
const projectId = parsed.projectId ?? parsed.project_id;
if (parsed.token === undefined || projectId === undefined) {
throw new Error(missingCredentialsMessage);
}
const refreshToken =
typeof parsed.refreshToken === "string"
? parsed.refreshToken
: typeof parsed.refresh === "string"
? parsed.refresh
: undefined;
const refreshToken = parsed.refreshToken ?? parsed.refresh;
const expiresAt = normalizeExpiryMs(parsed.expiresAt ?? parsed.expires);
const email = typeof parsed.email === "string" && parsed.email.length > 0 ? parsed.email : undefined;
const email = parsed.email && parsed.email.length > 0 ? parsed.email : undefined;
return {
accessToken: parsed.token,
File diff suppressed because it is too large Load Diff
@@ -57,6 +57,7 @@ export interface RequestBody {
client_metadata?: Record<string, string>;
max_output_tokens?: number;
max_completion_tokens?: number;
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
[key: string]: unknown;
}
File diff suppressed because it is too large Load Diff
@@ -360,7 +360,7 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest {
if (effectiveType === "custom_tool_call") {
const call = item as { id?: string; call_id: string; name: string; input: string };
// Custom tools carry a raw input string. We stash it in `arguments.input`
// matching pi-ai's openai-responses-shared convention, and tag the call
// matching pi-ai's openai-shared convention, and tag the call
// with `customWireName` so encoders re-emit it as `custom_tool_call`.
const toolCall: ToolCall = {
type: "toolCall",
File diff suppressed because it is too large Load Diff
+270 -302
View File
@@ -1,12 +1,11 @@
import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts";
import { parseGitHubCopilotApiKey } from "@oh-my-pi/pi-catalog/wire/github-copilot";
import { $env, $flag, extractHttpStatusFromError, logger, structuredCloneJSON } from "@oh-my-pi/pi-utils";
import { $flag, extractHttpStatusFromError, logger, structuredCloneJSON } from "@oh-my-pi/pi-utils";
import { getEnvApiKey } from "../stream";
import type {
AssistantMessage,
Context,
MessageAttribution,
Model,
OpenAICompat,
ProviderSessionState,
RawSseEvent,
ServiceTier,
@@ -17,8 +16,6 @@ import type {
} from "../types";
import {
createOpenAIResponsesHistoryPayload,
getOpenAIResponsesHistoryItems,
getOpenAIResponsesHistoryPayload,
normalizeSystemPrompts,
resolveCacheRetention,
sanitizeOpenAIResponsesHistoryItemsForReplay,
@@ -31,7 +28,7 @@ import {
getOpenAIStreamIdleTimeoutMs,
iterateWithIdleTimeout,
} from "../utils/idle-iterator";
import { postOpenAIStream } from "../utils/openai-http";
import { OpenAIHttpError, postOpenAIStream } from "../utils/openai-http";
import { notifyProviderResponse } from "../utils/provider-response";
import { callWithCopilotModelRetry } from "../utils/retry";
import {
@@ -42,42 +39,41 @@ import {
toolWireSchema,
} from "../utils/schema";
import { mapToOpenAIResponsesToolChoice, type OpenAIResponsesToolChoice } from "../utils/tool-choice";
import {
buildCopilotDynamicHeaders,
hasCopilotVisionInput,
resolveGitHubCopilotBaseUrl,
} from "./github-copilot-headers";
import { compactGrammarDefinition } from "./grammar";
import {
appendResponsesToolResultMessages,
applyCommonResponsesSamplingParams,
applyResponsesReasoningParams,
buildResponsesDeltaInput,
collectCustomCallIds,
collectKnownCallIds,
convertResponsesAssistantMessage,
convertResponsesInputContent,
createInitialResponsesAssistantMessage,
isOpenAIResponsesProgressEvent,
normalizeResponsesToolCallIdForTransform,
processResponsesStream,
repairOrphanResponsesToolCalls,
repairOrphanResponsesToolOutputs,
} from "./openai-responses-shared";
import type {
Tool as OpenAITool,
ResponseCreateParamsStreaming,
ResponseInput,
ResponseStreamEvent,
} from "./openai-responses-wire";
import { transformMessages } from "./transform-messages";
export function normalizeOpenAIResponsesPromptCacheKey(sessionId: string | undefined): string | undefined {
if (!sessionId || sessionId.length === 0) return undefined;
const wellFormed = sessionId.toWellFormed();
if (wellFormed.length <= 64) return wellFormed;
return `pc_${Bun.hash(wellFormed).toString(36)}`;
}
import {
applyCommonResponsesSamplingParams,
applyOpenAIExtraBody,
applyOpenAIGatewayRouting,
applyResponsesCompatPolicy,
applyWireModelIdTransform,
buildResponsesDeltaInput,
buildResponsesInput,
clearOpenAIStrictToolsState,
createInitialResponsesAssistantMessage,
createOpenAIStrictToolsState,
disableStrictToolsForScope,
getOpenAIResponsesPromptCacheKey,
getOpenAIResponsesRoutingSessionId,
getOpenAIStrictToolsScope,
getOpenRouterResponsesSessionId,
isCompiledGrammarTooLargeStrictError,
isOpenAIResponsesProgressEvent,
isOpenRouterAnthropicModel,
isStrictToolsDisabledForScope,
type OpenAIStrictToolsScope,
type OpenAIStrictToolsState,
processResponsesStream,
resolveOpenAIOutputTokenParam,
resolveOpenAICompatPolicy,
resolveOpenAIRequestSetup,
shouldRetryWithoutStrictTools,
} from "./openai-shared";
// OpenAI Responses-specific options
export interface OpenAIResponsesOptions extends StreamOptions {
@@ -85,6 +81,9 @@ export interface OpenAIResponsesOptions extends StreamOptions {
reasoningSummary?: "auto" | "detailed" | "concise" | null;
serviceTier?: ServiceTier;
toolChoice?: ToolChoice;
openrouterVariant?: string;
maxTokensExplicit?: boolean;
disableReasoning?: boolean;
/**
* Stateful turns: chain via `previous_response_id` + delta input instead of
* replaying the full transcript. Forces `store: true` (the platform only
@@ -96,29 +95,25 @@ export interface OpenAIResponsesOptions extends StreamOptions {
*/
statefulResponses?: boolean;
/**
* Enforce strict tool call/result pairing when building Responses API inputs.
* Azure OpenAI and GitHub Copilot Responses paths require tool results to match prior tool calls.
* Override catalog compat for strict tool call/result pairing when building
* Responses API inputs. Default behavior is catalog compat; this is only for
* debugging/adapter wrappers.
*/
strictResponsesPairing?: boolean;
/**
* Pass `include: ["reasoning.encrypted_content"]` on requests when the
* model supports reasoning. Default: true (preserves current behavior).
* Set to false when the upstream Responses endpoint rejects replayed
* encrypted reasoning (e.g., xAI Grok under SuperGrok OAuth).
* Override catalog compat for `include: ["reasoning.encrypted_content"]`.
* Default behavior is catalog compat; this is only for debugging/adapter wrappers.
*/
includeEncryptedReasoning?: boolean;
/**
* Strip `type: "reasoning"` items from replayed conversation history
* before they hit the wire. Default: false (preserves current behavior).
* Set to true when the upstream rejects replayed reasoning wrappers.
* Override catalog compat for stripping `type: "reasoning"` items from
* replayed conversation history before request encoding. Default behavior is
* catalog compat; this is only for debugging/adapter wrappers.
*/
filterReasoningHistory?: boolean;
/**
* Suppress the `reasoning.effort` wire param when set, even if
* `options.reasoning` is requested. Default: false. xAI Grok models
* outside the effort-capable allowlist 400 with "Model X does not
* support parameter reasoningEffort" — the xAI Responses adapter sets
* this when the target model is not in GROK_EFFORT_CAPABLE_PREFIXES.
* Override catalog compat for suppressing the `reasoning.effort` wire param.
* Default behavior is catalog compat; this is only for debugging/adapter wrappers.
*/
omitReasoningEffort?: boolean;
/**
@@ -141,7 +136,7 @@ const OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE =
/** Consecutive stale-previous-response failures before chaining is disabled for the session. */
const OPENAI_RESPONSES_CHAIN_STALE_FAILURE_LIMIT = 3;
interface OpenAIResponsesProviderSessionState extends ProviderSessionState {
interface OpenAIResponsesProviderSessionState extends ProviderSessionState, OpenAIStrictToolsState {
nativeHistoryReplayWarmed: boolean;
/** Stateful `previous_response_id` chain baselines, keyed by baseUrl/model/session. */
chains: Map<string, OpenAIResponsesChainState>;
@@ -164,27 +159,26 @@ interface OpenAIResponsesChainState {
}
function createOpenAIResponsesProviderSessionState(): OpenAIResponsesProviderSessionState {
const strictToolsState = createOpenAIStrictToolsState();
const state: OpenAIResponsesProviderSessionState = {
...strictToolsState,
nativeHistoryReplayWarmed: false,
chains: new Map(),
close: () => {
state.nativeHistoryReplayWarmed = false;
state.chains.clear();
clearOpenAIStrictToolsState(state);
},
};
return state;
}
function getOpenAIResponsesProviderSessionStateKey(model: Model<"openai-responses">): string {
return `${OPENAI_RESPONSES_PROVIDER_SESSION_STATE_PREFIX}${model.provider}`;
}
function getOpenAIResponsesProviderSessionState(
model: Model<"openai-responses">,
providerSessionState: Map<string, ProviderSessionState> | undefined,
): OpenAIResponsesProviderSessionState | undefined {
if (!providerSessionState) return undefined;
const key = getOpenAIResponsesProviderSessionStateKey(model);
const key = `${OPENAI_RESPONSES_PROVIDER_SESSION_STATE_PREFIX}${model.provider}`;
const existing = providerSessionState.get(key) as OpenAIResponsesProviderSessionState | undefined;
if (existing) return existing;
const created = createOpenAIResponsesProviderSessionState();
@@ -192,12 +186,6 @@ function getOpenAIResponsesProviderSessionState(
return created;
}
function canReplayOpenAIResponsesNativeHistory(
providerSessionState: OpenAIResponsesProviderSessionState | undefined,
): boolean {
return providerSessionState?.nativeHistoryReplayWarmed ?? true;
}
function isOpenAIResponsesStatefulEnabled(
options: OpenAIResponsesOptions | undefined,
baseUrl: string | undefined,
@@ -214,9 +202,10 @@ function isOpenAIResponsesStatefulEnabled(
function getOpenAIResponsesChainState(
providerSessionState: OpenAIResponsesProviderSessionState,
model: Model<"openai-responses">,
resolvedBaseUrl: string | undefined,
sessionId: string,
): OpenAIResponsesChainState {
const key = `${model.baseUrl ?? ""}\u0000${model.id}\u0000${sessionId}`;
const key = `${resolvedBaseUrl ?? model.baseUrl ?? ""}\u0000${model.id}\u0000${sessionId}`;
const existing = providerSessionState.chains.get(key);
if (existing) return existing;
const created: OpenAIResponsesChainState = { canAppend: false, staleFailures: 0, disabled: false };
@@ -237,18 +226,6 @@ interface OpenAIResponsesChainedParams {
previousResponseId?: string;
}
/**
* Drop the per-turn trailing scaffolding (the GPT-5 "Juice: 0" developer item)
* from `input`, yielding the wire form of the conversation arguments alone.
*/
function stripTrailingScaffolding(
params: OpenAIResponsesSamplingParams,
trailingScaffoldingItems: number,
): OpenAIResponsesSamplingParams {
if (trailingScaffoldingItems <= 0 || !Array.isArray(params.input)) return params;
return { ...params, input: params.input.slice(0, params.input.length - trailingScaffoldingItems) };
}
/**
* Shape the next turn's request: when the session's append baseline is intact
* (same options, strict history prefix), chain via `previous_response_id` +
@@ -264,7 +241,10 @@ function buildOpenAIResponsesChainedParams(
trailingScaffoldingItems: number,
chain: OpenAIResponsesChainState,
): OpenAIResponsesChainedParams {
const historyParams = stripTrailingScaffolding(params, trailingScaffoldingItems);
const historyParams =
trailingScaffoldingItems > 0 && Array.isArray(params.input)
? { ...params, input: params.input.slice(0, params.input.length - trailingScaffoldingItems) }
: params;
const deltaInput = chain.canAppend
? buildResponsesDeltaInput<ResponseInput[number]>(chain.lastParams, chain.lastResponseItems, historyParams)
: null;
@@ -296,18 +276,6 @@ function isOpenAIResponsesStalePreviousResponseError(error: unknown): boolean {
);
}
/**
* Zero Data Retention orgs accept `store: true` but refuse to resolve any
* `previous_response_id` — the prior response was never persisted server-side.
* The 400 carries a fixed phrasing ("Zero Data Retention") that the generic
* stale-id regex above does not match, so it is classified separately and
* disables chaining categorically (one strike, not three).
*/
function isOpenAIResponsesZeroDataRetentionError(error: unknown): boolean {
if (!(error instanceof Error)) return false;
return /previous[ _]?response/i.test(error.message) && /zero[ _-]?data[ _-]?retention/i.test(error.message);
}
function registerOpenAIResponsesChainStaleFailure(chain: OpenAIResponsesChainState, error: unknown): void {
resetOpenAIResponsesChainState(chain);
chain.staleFailures += 1;
@@ -340,7 +308,10 @@ type OpenAIResponsesSamplingParams = ResponseCreateParamsStreaming & {
min_p?: number;
presence_penalty?: number;
repetition_penalty?: number;
session_id?: string;
stream_options?: { include_obfuscation?: boolean };
provider?: OpenAICompat["openRouterRouting"];
reasoning?: { effort?: string } | { enabled: false };
};
/**
@@ -396,19 +367,26 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
// stable prompt-cache key independently. Side-channel calls use this to
// avoid perturbing provider conversation state without cold-starting the cache.
const routingSessionId = getOpenAIResponsesRoutingSessionId(options);
const promptCacheSessionId = getOpenAIResponsesPromptCacheKey(options);
const apiKey = options?.apiKey || getEnvApiKey(model.provider) || "";
const {
headers: requestHeaders,
copilotPremiumRequests,
baseUrl,
} = createRequestSetup(model, context, apiKey, options?.headers, options?.initiatorOverride, routingSessionId);
const { headers, copilotPremiumRequests, baseUrl } = resolveOpenAIRequestSetup(model, {
apiKey,
extraHeaders: options?.headers,
initiatorOverride: options?.initiatorOverride,
messages: context.messages,
openAISessionId: routingSessionId,
promptCacheSessionId,
});
const premiumRequestsTotal = copilotPremiumRequests;
const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState);
const builtParams = buildParams(model, context, options, providerSessionState);
const strictToolsScope = getOpenAIStrictToolsScope(model, baseUrl);
const builtParams = buildParams(model, context, options, providerSessionState, strictToolsScope);
const params = builtParams.params;
const { trailingScaffoldingItems } = builtParams;
let activeParams = params;
let activeTrailingScaffoldingItems = trailingScaffoldingItems;
if (isOpenAIResponsesStatefulEnabled(options, baseUrl) && routingSessionId && providerSessionState) {
chainState = getOpenAIResponsesChainState(providerSessionState, model, routingSessionId);
chainState = getOpenAIResponsesChainState(providerSessionState, model, baseUrl, routingSessionId);
if (!chainState.disabled) {
// Platform `previous_response_id` chaining only resolves stored responses.
params.store = true;
@@ -451,13 +429,13 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
);
}
try {
const headers = { ...requestHeaders };
const headersWithTimeout = { ...headers };
if (requestTimeoutMs !== undefined) {
headers["X-Stainless-Timeout"] = Math.floor(requestTimeoutMs / 1000).toString();
headersWithTimeout["X-Stainless-Timeout"] = Math.floor(requestTimeoutMs / 1000).toString();
}
const { events, response, requestId } = await postOpenAIStream<ResponseStreamEvent>({
url: requestUrl,
headers,
headers: headersWithTimeout,
body: requestParams,
signal: requestSignal,
fetch: options?.fetch,
@@ -481,40 +459,114 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
{ provider: model.provider, signal: requestSignal },
);
let openaiStream: AsyncIterable<ResponseStreamEvent>;
try {
openaiStream = await openResponsesStream(chained.params);
} catch (error) {
if (!chainState || !sentPreviousResponseId || requestSignal.aborted) {
throw error;
let strictRetryAvailable = true;
let activeStrictToolsApplied = builtParams.strictToolsApplied;
let forceDisableStrictTools = false;
while (true) {
try {
openaiStream = await openResponsesStream(chained.params);
break;
} catch (error) {
const capturedErrorResponse = error instanceof OpenAIHttpError ? error.captured : undefined;
const compiledGrammarTooLarge =
isOpenRouterAnthropicModel(model) &&
isCompiledGrammarTooLargeStrictError(error, capturedErrorResponse);
const canRetryWithoutStrictTools =
strictRetryAvailable &&
!requestSignal.aborted &&
(compiledGrammarTooLarge ||
shouldRetryWithoutStrictTools(
error,
capturedErrorResponse,
activeStrictToolsApplied,
context.tools,
));
if (canRetryWithoutStrictTools) {
strictRetryAvailable = false;
forceDisableStrictTools = true;
const strictFallbackError = compiledGrammarTooLarge
? await finalizeErrorMessage(error, rawRequestDump, capturedErrorResponse)
: undefined;
disableStrictToolsForScope(providerSessionState, strictToolsScope);
const fallbackBuilt = buildParams(
model,
context,
options,
providerSessionState,
strictToolsScope,
true,
);
const fallbackParams = fallbackBuilt.params;
if (chainState && !chainState.disabled) fallbackParams.store = true;
let fallbackChained: OpenAIResponsesChainedParams =
chainState && !chainState.disabled
? buildOpenAIResponsesChainedParams(
fallbackParams,
fallbackBuilt.trailingScaffoldingItems,
chainState,
)
: { params: fallbackParams };
sentPreviousResponseId = fallbackChained.previousResponseId;
fallbackChained = {
...fallbackChained,
params: await applyPayloadReplacement(fallbackChained.params),
};
chained = fallbackChained;
rawRequestDump.body = chained.params;
activeParams = fallbackParams;
activeTrailingScaffoldingItems = fallbackBuilt.trailingScaffoldingItems;
activeStrictToolsApplied = fallbackBuilt.strictToolsApplied;
if (strictFallbackError) output.errorMessage = strictFallbackError;
continue;
}
if (!chainState || !sentPreviousResponseId || requestSignal.aborted) {
throw error;
}
const zdrRejection =
error instanceof Error &&
/previous[ _]?response/i.test(error.message) &&
/zero[ _-]?data[ _-]?retention/i.test(error.message);
if (!zdrRejection && !isOpenAIResponsesStalePreviousResponseError(error)) {
throw error;
}
// Server rejected the chain baseline: reset, count the failure (or
// disable categorically on ZDR), and retry once with the full
// transcript. Structurally cannot loop — the retry carries no
// previous_response_id.
if (zdrRejection) {
markOpenAIResponsesChainZeroDataRetention(chainState, error);
// ZDR orgs cannot store responses; the retry uses `store: false`.
} else {
registerOpenAIResponsesChainStaleFailure(chainState, error);
}
sentPreviousResponseId = undefined;
const currentBuilt = buildParams(
model,
context,
options,
providerSessionState,
strictToolsScope,
forceDisableStrictTools,
);
const currentParams = currentBuilt.params;
// Only ZDR forces `store: false` (the org never persists responses). A
// non-ZDR stale baseline is transient, so keep storing: the full-context
// retry must be chainable next turn, and the consecutive stale-failure
// breaker only trips when each retry stores and the next turn re-chains.
currentParams.store = !zdrRejection;
const retryParams = await applyPayloadReplacement(currentParams);
chained = { params: retryParams };
rawRequestDump.body = retryParams;
activeParams = currentParams;
activeTrailingScaffoldingItems = currentBuilt.trailingScaffoldingItems;
activeStrictToolsApplied = currentBuilt.strictToolsApplied;
}
const zdrRejection = isOpenAIResponsesZeroDataRetentionError(error);
if (!zdrRejection && !isOpenAIResponsesStalePreviousResponseError(error)) {
throw error;
}
// Server rejected the chain baseline: reset, count the failure (or
// disable categorically on ZDR), and retry once with the full
// transcript. Structurally cannot loop — the retry carries no
// previous_response_id.
if (zdrRejection) {
markOpenAIResponsesChainZeroDataRetention(chainState, error);
// ZDR orgs cannot store responses; the original request forced
// `store: true` for chaining, which is meaningless here and would
// otherwise leave subsequent turns asking the server to retain
// data it must discard.
params.store = false;
} else {
registerOpenAIResponsesChainStaleFailure(chainState, error);
}
sentPreviousResponseId = undefined;
const retryParams = await applyPayloadReplacement(params);
rawRequestDump.body = retryParams;
openaiStream = await openResponsesStream(retryParams);
}
if (premiumRequestsTotal !== undefined) output.usage.premiumRequests = premiumRequestsTotal;
stream.push({ type: "start", partial: output });
const nativeOutputItems: Array<Record<string, unknown>> = [];
let sawCompleted = false;
let sawTerminalResponseEvent = false;
const timedOpenaiStream = iterateWithIdleTimeout(openaiStream, {
idleTimeoutMs,
firstItemTimeoutMs: firstEventTimeoutMs,
@@ -535,7 +587,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
nativeOutputItems.push(item as unknown as Record<string, unknown>);
},
onCompleted: () => {
sawCompleted = true;
sawTerminalResponseEvent = true;
},
});
@@ -548,11 +600,12 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
}
// Detect premature stream closure: the HTTP stream ended without the
// provider sending `response.completed`. Custom/proxy providers may
// drop the connection mid-stream; without this guard the incomplete
// output is silently surfaced as a successful "stop".
if (!sawCompleted) {
throw new Error("OpenAI responses stream closed before response.completed was received");
// provider sending `response.completed` or `response.incomplete`.
// Custom/proxy providers may drop the connection mid-stream; without
// this guard the incomplete output is silently surfaced as a successful
// "stop".
if (!sawTerminalResponseEvent) {
throw new Error("OpenAI responses stream closed before a terminal response event was received");
}
if (output.stopReason === "aborted" || output.stopReason === "error") {
@@ -562,7 +615,14 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
output.providerPayload = createOpenAIResponsesHistoryPayload(model.provider, nativeOutputItems);
if (providerSessionState) providerSessionState.nativeHistoryReplayWarmed = true;
if (chainState) {
chainState.lastParams = structuredCloneJSON(stripTrailingScaffolding(params, trailingScaffoldingItems));
chainState.lastParams = structuredCloneJSON(
activeTrailingScaffoldingItems > 0 && Array.isArray(activeParams.input)
? {
...activeParams,
input: activeParams.input.slice(0, activeParams.input.length - activeTrailingScaffoldingItems),
}
: activeParams,
);
if (output.responseId) {
chainState.lastResponseId = output.responseId;
chainState.lastResponseItems = sanitizeOpenAIResponsesHistoryItemsForReplay(
@@ -587,8 +647,14 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
if (chainState) resetOpenAIResponsesChainState(chainState);
const firstEventTimeoutError = abortTracker.getLocalAbortReason();
output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
output.errorStatus = extractHttpStatusFromError(error);
output.errorMessage = firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
const capturedErrorResponse = error instanceof OpenAIHttpError ? error.captured : undefined;
output.errorStatus = extractHttpStatusFromError(error) ?? capturedErrorResponse?.status;
output.errorMessage =
firstEventTimeoutError?.message ??
(await finalizeErrorMessage(error, rawRequestDump, capturedErrorResponse));
// Some providers via OpenRouter include extra details here.
const rawMetadata = (error as { error?: { metadata?: { raw?: string } } })?.error?.metadata?.raw;
if (rawMetadata) output.errorMessage += `\n${rawMetadata}`;
output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
output.duration = Date.now() - startTime;
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
@@ -600,87 +666,42 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
return stream;
};
function createRequestSetup(
model: Model<"openai-responses">,
context: Context,
apiKey?: string,
extraHeaders?: Record<string, string>,
initiatorOverride?: MessageAttribution,
sessionId?: string,
): {
headers: Record<string, string>;
copilotPremiumRequests: number | undefined;
baseUrl: string | undefined;
} {
if (!apiKey) {
if (!$env.OPENAI_API_KEY) {
throw new Error(
"OpenAI API key is required. Set OPENAI_API_KEY environment variable or pass it as an argument.",
);
}
apiKey = $env.OPENAI_API_KEY;
}
const rawApiKey = apiKey;
const headers = { ...(model.headers ?? {}), ...(extraHeaders ?? {}) };
let copilotPremiumRequests: number | undefined;
let baseUrl = model.baseUrl;
if (model.provider === "github-copilot") {
apiKey = parseGitHubCopilotApiKey(rawApiKey).accessToken;
const hasImages = hasCopilotVisionInput(context.messages);
const copilot = buildCopilotDynamicHeaders({
messages: context.messages,
hasImages,
premiumMultiplier: model.premiumMultiplier,
headers,
initiatorOverride,
});
Object.assign(headers, copilot.headers);
copilotPremiumRequests = copilot.premiumRequests;
baseUrl = resolveGitHubCopilotBaseUrl(model.baseUrl, rawApiKey) ?? model.baseUrl;
}
if (sessionId && model.provider === "openai") {
headers.session_id ??= sessionId;
headers["x-client-request-id"] ??= sessionId;
}
headers.Authorization ??= `Bearer ${apiKey}`;
return { headers, copilotPremiumRequests, baseUrl };
}
function getOpenAIResponsesPromptCacheKey(
options: Pick<OpenAIResponsesOptions, "cacheRetention" | "promptCacheKey" | "sessionId"> | undefined,
): string | undefined {
if (resolveCacheRetention(options?.cacheRetention) === "none") return undefined;
return normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId);
}
export function getOpenAIResponsesCacheSessionId(
options: Pick<OpenAIResponsesOptions, "cacheRetention" | "sessionId" | "promptCacheKey"> | undefined,
): string | undefined {
return getOpenAIResponsesPromptCacheKey(options);
}
function getOpenAIResponsesRoutingSessionId(
options: Pick<OpenAIResponsesOptions, "cacheRetention" | "sessionId"> | undefined,
): string | undefined {
if (resolveCacheRetention(options?.cacheRetention) === "none") return undefined;
return normalizeOpenAIResponsesPromptCacheKey(options?.sessionId);
}
/** @internal Exported for tests. */
export function buildParams(
model: Model<"openai-responses">,
context: Context,
options: OpenAIResponsesOptions | undefined,
providerSessionState: OpenAIResponsesProviderSessionState | undefined,
): { params: OpenAIResponsesSamplingParams; trailingScaffoldingItems: number } {
const strictResponsesPairing = options?.strictResponsesPairing ?? model.compat.strictResponsesPairing;
const messages = convertConversationMessages(model, context, strictResponsesPairing, providerSessionState, options);
strictToolsScope?: OpenAIStrictToolsScope,
disableStrictToolsOverride = false,
): { params: OpenAIResponsesSamplingParams; trailingScaffoldingItems: number; strictToolsApplied: boolean } {
const policy = resolveOpenAICompatPolicy(model, {
endpoint: "responses",
reasoning: options?.reasoning,
disableReasoning: options?.disableReasoning,
toolChoice: options?.toolChoice,
strictResponsesPairing: options?.strictResponsesPairing,
includeEncryptedReasoning: options?.includeEncryptedReasoning,
filterReasoningHistory: options?.filterReasoningHistory,
omitReasoningEffort: options?.omitReasoningEffort,
});
const strictResponsesPairing = policy.tools.strictResponsesPairing;
const shouldReplayNativeHistory = providerSessionState?.nativeHistoryReplayWarmed ?? true;
const messages = buildResponsesInput({
model,
context,
strictResponsesPairing,
nativeHistory: {
replay: shouldReplayNativeHistory,
filterReasoning: policy.reasoning.filterReasoningHistory,
},
includeThinkingSignatures: shouldReplayNativeHistory,
repairOrphanOutputs: true,
});
const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
let systemInstructions: string | undefined;
if (systemPrompts.length > 0) {
const needsDeveloperRole = model.reasoning && model.compat.supportsDeveloperRole;
const needsDeveloperRole = policy.messages.systemRole === "developer";
if (needsDeveloperRole) {
// Reasoning models on known OpenAI-compatible endpoints require the
// `developer` role. Send all system prompts inline in `input`.
@@ -697,8 +718,13 @@ export function buildParams(
const cacheRetention = resolveCacheRetention(options?.cacheRetention);
const promptCacheKey = getOpenAIResponsesPromptCacheKey(options);
const modelId = applyWireModelIdTransform(
model.requestModelId ?? model.id,
model.compat.wireModelIdMode,
options?.openrouterVariant,
);
const params: OpenAIResponsesSamplingParams = {
model: model.requestModelId ?? model.id,
model: modelId,
input: messages,
instructions: systemInstructions,
stream: true,
@@ -708,18 +734,35 @@ export function buildParams(
? "24h"
: undefined
: undefined,
// Gateway routing: OpenRouter-only Responses wire field for sticky upstream
// routing + observability grouping; no equivalent on direct OpenAI.
session_id: model.compat.isOpenRouterHost ? getOpenRouterResponsesSessionId(options) : undefined,
store: false,
stream_options: model.provider === "openai" ? { include_obfuscation: false } : undefined,
stream_options: model.compat.supportsObfuscationOptOut ? { include_obfuscation: false } : undefined,
};
const outputToken = resolveOpenAIOutputTokenParam({
field: "max_output_tokens",
maxTokens: options?.maxTokens,
maxTokensExplicit: options?.maxTokensExplicit ?? options?.maxTokens !== undefined,
modelMaxTokens: model.maxTokens,
omitMaxOutputTokens: model.omitMaxOutputTokens ?? false,
isOpenRouterHost: model.compat.isOpenRouterHost,
alwaysSendMaxTokens: model.compat.alwaysSendMaxTokens,
});
applyCommonResponsesSamplingParams(params, options, model);
applyCommonResponsesSamplingParams(params, { ...options, maxTokens: outputToken?.value }, model);
// TODO: openai responses has no top-level `stop`/`stop_sequences`; surface via reasoning.stop?
// `StreamOptions.stopSequences` is intentionally dropped for this provider.
// TODO: openai responses has no top-level `frequency_penalty` field as of the current SDK;
// `StreamOptions.frequencyPenalty` is intentionally dropped for this provider.
let strictToolsApplied = false;
if (context.tools) {
params.tools = convertTools(context.tools, model.compat.supportsStrictMode, model);
const disableStrictTools =
disableStrictToolsOverride || isStrictToolsDisabledForScope(providerSessionState, strictToolsScope);
const strictMode = !disableStrictTools && model.compat.supportsStrictMode;
params.tools = convertTools(context.tools, strictMode, model);
strictToolsApplied = params.tools.some(t => (t as { strict?: boolean }).strict === true);
if (options?.toolChoice) {
// Map tool_choice against the tools that survived quarantine, not the
// original list: a forced choice for a dropped tool — or "required" when
@@ -748,104 +791,29 @@ export function buildParams(
}
}
const trailingScaffoldingItems = applyResponsesReasoningParams(
params,
model,
options,
messages,
effort =>
const reasoningPolicy = resolveOpenAICompatPolicy(model, {
endpoint: "responses",
reasoning: options?.reasoning,
disableReasoning: options?.disableReasoning,
toolChoice: params.tool_choice,
strictResponsesPairing: options?.strictResponsesPairing,
includeEncryptedReasoning: options?.includeEncryptedReasoning,
filterReasoningHistory: options?.filterReasoningHistory,
omitReasoningEffort: options?.omitReasoningEffort,
});
const trailingScaffoldingItems = applyResponsesCompatPolicy(params, messages, reasoningPolicy, {
reasoningSummary: options?.reasoningSummary,
mapEffort: effort =>
model.compat.reasoningEffortMap?.[effort as NonNullable<OpenAIResponsesOptions["reasoning"]>] ??
model.thinking?.effortMap?.[effort as NonNullable<OpenAIResponsesOptions["reasoning"]>] ??
effort,
options?.includeEncryptedReasoning ?? true,
options?.omitReasoningEffort ?? false,
);
});
if (options?.extraBody) {
Object.assign(params, options.extraBody);
}
applyOpenAIGatewayRouting(params, model.compat);
return { params, trailingScaffoldingItems };
}
applyOpenAIExtraBody(params, options?.extraBody);
function convertConversationMessages(
model: Model<"openai-responses">,
context: Context,
strictResponsesPairing: boolean,
providerSessionState: OpenAIResponsesProviderSessionState | undefined,
options?: OpenAIResponsesOptions,
): ResponseInput {
const filterReasoning = <T extends { type?: string }>(items: T[]): T[] =>
options?.filterReasoningHistory ? items.filter(i => i?.type !== "reasoning") : items;
const messages: ResponseInput = [];
let knownCallIds = new Set<string>();
const customCallIds = new Set<string>();
const shouldReplayNativeHistory = canReplayOpenAIResponsesNativeHistory(providerSessionState);
const transformedMessages = transformMessages(context.messages, model, normalizeResponsesToolCallIdForTransform);
let msgIndex = 0;
for (const msg of transformedMessages) {
if (msg.role === "user" || msg.role === "developer") {
const providerPayload = (msg as { providerPayload?: AssistantMessage["providerPayload"] }).providerPayload;
const historyItems = getOpenAIResponsesHistoryItems(providerPayload, model.provider);
const shouldReplayPayloadItems =
shouldReplayNativeHistory ||
(historyItems?.some(item => {
if (!item || typeof item !== "object") return false;
const candidate = item as { type?: unknown };
return candidate.type === "compaction" || candidate.type === "compaction_summary";
}) ??
false);
if (historyItems && shouldReplayPayloadItems) {
messages.push(...sanitizeOpenAIResponsesHistoryItemsForReplay(filterReasoning(historyItems)));
knownCallIds = collectKnownCallIds(messages);
for (const id of collectCustomCallIds(messages)) customCallIds.add(id);
msgIndex++;
continue;
}
const content = convertResponsesInputContent(msg.content, model.input.includes("image"));
if (!content) continue;
messages.push({ role: "user", content });
} else if (msg.role === "assistant") {
const assistantMsg = msg as AssistantMessage;
// Native items are model-bound (reasoning carries encrypted content minted
// by the producing model); after a mid-session model switch fall back to
// block re-encode, which strips foreign signatures.
const providerPayload =
shouldReplayNativeHistory && assistantMsg.api === model.api && assistantMsg.model === model.id
? getOpenAIResponsesHistoryPayload(assistantMsg.providerPayload, model.provider, assistantMsg.provider)
: undefined;
const historyItems = providerPayload?.items;
if (historyItems) {
const sanitizedHistoryItems = sanitizeOpenAIResponsesHistoryItemsForReplay(filterReasoning(historyItems));
if (providerPayload?.dt) {
messages.push(...sanitizedHistoryItems);
} else {
messages.splice(0, messages.length, ...sanitizedHistoryItems);
}
knownCallIds = collectKnownCallIds(messages);
for (const id of collectCustomCallIds(messages)) customCallIds.add(id);
msgIndex++;
continue;
}
const outputItems = convertResponsesAssistantMessage(
assistantMsg,
model,
msgIndex,
knownCallIds,
shouldReplayNativeHistory,
customCallIds,
);
if (outputItems.length === 0) continue;
messages.push(...outputItems);
} else if (msg.role === "toolResult") {
appendResponsesToolResultMessages(messages, msg, model, strictResponsesPairing, knownCallIds, customCallIds);
}
msgIndex++;
}
return repairOrphanResponsesToolCalls(repairOrphanResponsesToolOutputs(messages));
return { params, trailingScaffoldingItems, strictToolsApplied };
}
/**
File diff suppressed because it is too large Load Diff
@@ -1,82 +0,0 @@
// Ported from NousResearch/hermes-agent (MIT) — agent/transports/codex.py:182-193,
// agent/codex_responses_adapter.py:247-311, agent/model_metadata.py:263-285.
// Logic EXTRACTED into a dedicated xAI adapter so the generic OpenAI Responses
// path stays provider-agnostic and the OpenAI Codex Responses path is unaffected.
import type { Context, Model, StreamFunction } from "../types";
import {
getOpenAIResponsesCacheSessionId,
type OpenAIResponsesOptions,
streamOpenAIResponses,
} from "./openai-responses";
// xAI rejects `reasoning.effort` on grok-4 / grok-4-fast / grok-3 /
// grok-code-fast / grok-4.20-0309-* / grok-build with HTTP 400 ("Model X does
// not support parameter reasoningEffort") even though those models reason
// natively (hermes-agent/agent/transports/codex.py:127-133). Only send the
// effort dial when the target model is on this allowlist; otherwise suppress
// it via OpenAIResponsesOptions.omitReasoningEffort and let the model reason
// on its own. grok-build was previously on this list per user spec; the live
// xAI server contradicts that assumption (HTTP 400 confirmed against
// api.x.ai/v1/responses on 2026-05-17).
const GROK_EFFORT_CAPABLE_PREFIXES = ["grok-3-mini", "grok-4.20-multi-agent", "grok-4.3"] as const;
function grokSupportsReasoningEffort(modelId: string): boolean {
const name = (modelId || "").trim().toLowerCase();
if (!name) return false;
// Strip common aggregator prefixes (x-ai/, openrouter/x-ai/, xai/, ...) before matching.
const bare = name.includes("/") ? name.slice(name.lastIndexOf("/") + 1) : name;
return GROK_EFFORT_CAPABLE_PREFIXES.some(prefix => bare.startsWith(prefix));
}
/**
* xAI Grok Responses adapter (SuperGrok OAuth path).
*
* Three xAI-specific behaviors vs the generic OpenAI Responses adapter:
*
* 1. `x-grok-conv-id` header + body `prompt_cache_key` route prompt-cache
* hits on xAI's edge. Hermes uses both (agent/transports/codex.py:182-193).
* The header is undocumented by xAI; `previous_response_id` is the
* documented alternative — switch if xAI deprecates the header.
* 2. includeEncryptedReasoning=false — xAI's /v1/responses rejects replayed
* `encrypted_content` blobs minted under SuperGrok OAuth.
* 3. filterReasoningHistory=true — strip `type: "reasoning"` items from
* replayed conversation history; the blob inside is non-replayable under
* OAuth and the wrapper item 404s without it (store=false; server cannot
* resolve by id).
*
* Everything else is the generic OpenAI Responses transport. The xAI bearer
* token arrives in `options.apiKey` via AuthStorage.getApiKey() upstream, and
* the xAI base URL (`https://api.x.ai/v1`) arrives via `model.baseUrl` from
* the provider registry — not routed through this wrapper.
*/
export const streamXAIResponses: StreamFunction<"openai-responses"> = (
model: Model<"openai-responses">,
context: Context,
options: OpenAIResponsesOptions = {},
) => {
const cacheSessionId = getOpenAIResponsesCacheSessionId(options);
const xaiHeaders: Record<string, string> = { ...options?.headers };
if (cacheSessionId) {
xaiHeaders["x-grok-conv-id"] = cacheSessionId;
}
const xaiBody: Record<string, unknown> = { ...(options?.extraBody ?? {}) };
if (cacheSessionId) {
xaiBody.prompt_cache_key = cacheSessionId;
}
const xaiOptions: OpenAIResponsesOptions = {
...options,
headers: xaiHeaders,
extraBody: xaiBody,
includeEncryptedReasoning: false,
filterReasoningHistory: true,
// Caller-passed value always wins (escape hatch for future xAI behavior
// changes); otherwise gate the effort dial on the allowlist.
omitReasoningEffort: options?.omitReasoningEffort ?? !grokSupportsReasoningEffort(model.id),
};
return streamOpenAIResponses(model, context, xaiOptions);
};
+42 -11
View File
@@ -46,7 +46,6 @@ import {
streamOpenAIResponses,
} from "./providers/register-builtins";
import { isSyntheticModel, streamSynthetic } from "./providers/synthetic";
import { streamXAIResponses } from "./providers/xai-responses";
import { isUsageLimitError } from "./rate-limit-utils";
import { PROVIDER_REGISTRY } from "./registry";
import type {
@@ -74,6 +73,7 @@ function isGoogleVertexAuthenticatedModel(model: Model<Api>): boolean {
);
}
function createVertexAuthenticatedFetch(options: StreamOptions | undefined): FetchImpl {
const baseFetch = options?.fetch ?? fetch;
const vertexFetch = async (input: string | URL | Request, init?: RequestInit): Promise<Response> => {
@@ -292,15 +292,19 @@ function streamDispatch<TApi extends Api>(
});
}
case "openrouter": {
const useResponses = $env.PI_OPENROUTER_RESPONSES !== "0";
if (useResponses) {
return streamOpenAIResponses(model as Model<"openai-responses">, context, providerOptions as any);
}
return streamOpenAICompletions(model as Model<"openai-completions">, context, providerOptions as any);
}
case "openai-completions":
return streamOpenAICompletions(model as Model<"openai-completions">, context, providerOptions as any);
case "openai-responses": {
if (model.provider === "xai-oauth") {
return streamXAIResponses(model as Model<"openai-responses">, context, providerOptions as any);
}
case "openai-responses":
return streamOpenAIResponses(model as Model<"openai-responses">, context, providerOptions as any);
}
case "azure-openai-responses":
return streamAzureOpenAIResponses(model as Model<"azure-openai-responses">, context, providerOptions as any);
@@ -679,11 +683,10 @@ function resolveOpenAiReasoningEffort<TApi extends Api>(
// Models that reason natively but expose no effort dial carry
// `thinking: undefined` (baked at build time from
// `compat.supportsReasoningEffort: false` on openai-responses*). The
// wire-side omitReasoningEffort gate (providers/xai-responses.ts:78) is the
// actual strip; returning undefined here avoids a redundant
// requireSupportedEffort throw that would defeat the gate and surface a
// confusing "Compaction failed: Thinking effort high is not supported
// by..." to the user.
// wire-side omitReasoningEffort gate (stream.ts) is the actual strip; returning
// undefined here avoids a redundant requireSupportedEffort throw that would
// defeat the gate and surface a confusing "Compaction failed: Thinking effort
// high is not supported by..." to the user.
if (!model.thinking) return undefined;
return requireSupportedEffort(model, reasoning);
}
@@ -869,6 +872,31 @@ function mapOptionsForApi<TApi extends Api>(
return castApi<"bedrock-converse-stream">({ ...bedrockBase, maxTokens, thinkingBudgets });
}
case "openrouter": {
const useResponses = $env.PI_OPENROUTER_RESPONSES !== "0";
if (useResponses) {
return castApi<"openai-responses">({
...base,
reasoning: resolveOpenAiReasoningEffort(model, options),
toolChoice: mapOpenAiToolChoice(options?.toolChoice),
serviceTier: options?.serviceTier,
reasoningSummary: options?.hideThinkingSummary ? null : undefined,
openrouterVariant: options?.openrouterVariant,
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
disableReasoning: options?.disableReasoning,
});
}
return castApi<"openai-completions">({
...base,
reasoning: resolveOpenAiReasoningEffort(model, options),
disableReasoning: options?.disableReasoning,
toolChoice: mapOpenAiToolChoice(options?.toolChoice),
serviceTier: options?.serviceTier,
openrouterVariant: options?.openrouterVariant,
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
});
}
case "openai-completions":
return castApi<"openai-completions">({
...base,
@@ -887,6 +915,9 @@ function mapOptionsForApi<TApi extends Api>(
toolChoice: mapOpenAiToolChoice(options?.toolChoice),
serviceTier: options?.serviceTier,
reasoningSummary: options?.hideThinkingSummary ? null : undefined,
openrouterVariant: options?.openrouterVariant,
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
disableReasoning: options?.disableReasoning,
});
case "azure-openai-responses":
+1
View File
@@ -56,6 +56,7 @@ export interface ApiOptionsMap {
"anthropic-messages": AnthropicOptions;
"bedrock-converse-stream": BedrockOptions;
"openai-completions": OpenAICompletionsOptions;
openrouter: OpenAIResponsesOptions | OpenAICompletionsOptions;
"openai-responses": OpenAIResponsesOptions;
"openai-codex-responses": OpenAICodexResponsesOptions;
"azure-openai-responses": AzureOpenAIResponsesOptions;
@@ -0,0 +1,12 @@
import packageJson from "../../package.json" with { type: "json" };
export function getOpenRouterHeaders(): Record<string, string> {
return {
"User-Agent": `Oh-My-Pi/${packageJson.version}`,
"HTTP-Referer": "https://omp.sh/",
"X-OpenRouter-Title": "Oh-My-Pi",
"X-OpenRouter-Categories": "cli-agent",
"X-OpenRouter-Cache": "true",
"X-OpenRouter-Cache-TTL": "3600",
};
}
@@ -8,12 +8,12 @@ import {
mapOpenAIResponsesToolChoiceForTools,
supportsFreeformApplyPatch,
} from "@oh-my-pi/pi-ai/providers/openai-responses";
import type { ResponseStreamEvent } from "@oh-my-pi/pi-ai/providers/openai-responses-wire";
import {
appendResponsesToolResultMessages,
convertResponsesAssistantMessage,
processResponsesStream,
} from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
import type { ResponseStreamEvent } from "@oh-my-pi/pi-ai/providers/openai-responses-wire";
} from "@oh-my-pi/pi-ai/providers/openai-shared";
import type { AssistantMessage, Model, ModelSpec, Tool, ToolResultMessage } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { z } from "zod/v4";
@@ -4,6 +4,24 @@ import type { AssistantMessage, Model, ModelSpec, ThinkingContent, ToolCall } fr
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
interface OpenAICompletionAssistantWireMessage {
role: "assistant";
content?: unknown;
reasoning_content?: unknown;
rs_6f3a1b2c4d5e6f7a8b9c0d1e2f3a4b5c?: unknown;
}
function isOpenAICompletionAssistantWireMessage(message: unknown): message is OpenAICompletionAssistantWireMessage {
if (typeof message !== "object" || message === null) return false;
return (message as { role?: unknown }).role === "assistant";
}
function findOpenAICompletionAssistantWireMessage(
messages: readonly unknown[] | undefined,
): OpenAICompletionAssistantWireMessage | undefined {
return messages?.find(isOpenAICompletionAssistantWireMessage);
}
function deepseekModel(overrides: Partial<ModelSpec<"openai-completions">>): Model<"openai-completions"> {
const base = getBundledModel("openai", "gpt-4o-mini");
return buildModel({
@@ -188,10 +206,10 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
timestamp: Date.now(),
};
const messages = convertMessages(model, { messages: [msg] }, compat);
const assistant = messages.find(m => m.role === "assistant");
const assistant = findOpenAICompletionAssistantWireMessage(messages);
expect(assistant).toBeDefined();
// The reasoning_content field should be set from the signature, even if empty.
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("");
expect(assistant?.reasoning_content).toBe("");
});
it("recovers reasoning_content from non-empty thinking block with signature", () => {
@@ -231,9 +249,9 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
timestamp: Date.now(),
};
const messages = convertMessages(model, { messages: [msg] }, compat);
const assistant = messages.find(m => m.role === "assistant");
const assistant = findOpenAICompletionAssistantWireMessage(messages);
expect(assistant).toBeDefined();
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("I need to read the file first.");
expect(assistant?.reasoning_content).toBe("I need to read the file first.");
});
it("normalizes OpenRouter reasoning deltas to DeepSeek reasoning_content on replay", () => {
@@ -256,9 +274,9 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
} as ToolCall,
]);
const messages = convertMessages(model, { messages: [msg] }, compat);
const assistant = messages.find(m => m.role === "assistant");
const assistant = findOpenAICompletionAssistantWireMessage(messages);
expect(assistant).toBeDefined();
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("I should inspect the requested file.");
expect(assistant?.reasoning_content).toBe("I should inspect the requested file.");
});
it("does not use opaque signature as property name but still sets reasoning_content from thinking text", () => {
const model = deepseekModel({
@@ -302,12 +320,12 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
timestamp: Date.now(),
};
const messages = convertMessages(model, { messages: [msg] }, compat);
const assistant = messages.find(m => m.role === "assistant");
const assistant = findOpenAICompletionAssistantWireMessage(messages);
expect(assistant).toBeDefined();
// Should NOT have used the opaque signature as a property name.
expect(Reflect.get(assistant as object, "rs_6f3a1b2c4d5e6f7a8b9c0d1e2f3a4b5c")).toBeUndefined();
expect(assistant?.rs_6f3a1b2c4d5e6f7a8b9c0d1e2f3a4b5c).toBeUndefined();
// Should have set reasoning_content from the thinking text via the openai path.
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("some reasoning");
expect(assistant?.reasoning_content).toBe("some reasoning");
});
it("falls through to empty-string when thinking block has opaque signature and empty text", () => {
const model = deepseekModel({
@@ -349,10 +367,10 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
timestamp: Date.now(),
};
const messages = convertMessages(model, { messages: [msg] }, compat);
const assistant = messages.find(m => m.role === "assistant");
const assistant = findOpenAICompletionAssistantWireMessage(messages);
expect(assistant).toBeDefined();
expect(Reflect.get(assistant as object, "rs_6f3a1b2c4d5e6f7a8b9c0d1e2f3a4b5c")).toBeUndefined();
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("");
expect(assistant?.rs_6f3a1b2c4d5e6f7a8b9c0d1e2f3a4b5c).toBeUndefined();
expect(assistant?.reasoning_content).toBe("");
});
});
@@ -379,10 +397,10 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
} as ToolCall,
]);
const messages = convertMessages(model, { messages: [msg] }, compat);
const assistant = messages.find(m => m.role === "assistant");
const assistant = findOpenAICompletionAssistantWireMessage(messages);
expect(assistant).toBeDefined();
// reasoning_content must be present (empty string) — not absent and not "."
const rc = Reflect.get(assistant as object, "reasoning_content");
const rc = assistant?.reasoning_content;
expect(rc).toBeDefined();
expect(rc).toBe("");
});
@@ -402,10 +420,10 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
} as ToolCall,
]);
const messages = convertMessages(model, { messages: [msg] }, compat);
const assistant = messages.find(m => m.role === "assistant");
const assistant = findOpenAICompletionAssistantWireMessage(messages);
expect(assistant).toBeDefined();
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("");
expect((assistant as { content: unknown }).content).toBe("");
expect(assistant?.reasoning_content).toBe("");
expect(assistant?.content).toBe("");
});
it("sets content to empty string (not null) when reasoning_content is present", () => {
@@ -424,9 +442,9 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
} as ToolCall,
]);
const messages = convertMessages(model, { messages: [msg] }, compat);
const assistant = messages.find(m => m.role === "assistant");
const assistant = findOpenAICompletionAssistantWireMessage(messages);
expect(assistant).toBeDefined();
expect((assistant as { content: unknown }).content).toBe("");
expect(assistant?.content).toBe("");
});
});
@@ -463,10 +481,10 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
timestamp: Date.now(),
};
const messages = convertMessages(model, { messages: [msg] }, compat);
const assistant = messages.find(m => m.role === "assistant");
const assistant = findOpenAICompletionAssistantWireMessage(messages);
expect(assistant).toBeDefined();
// reasoning_content must be present — even on non-tool-call turns
const rc = Reflect.get(assistant as object, "reasoning_content");
const rc = assistant?.reasoning_content;
expect(rc).toBeDefined();
expect(rc).toBe("");
});
@@ -503,10 +521,10 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
timestamp: Date.now(),
};
const messages = convertMessages(model, { messages: [msg] }, compat);
const assistant = messages.find(m => m.role === "assistant");
const assistant = findOpenAICompletionAssistantWireMessage(messages);
expect(assistant).toBeDefined();
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("Let me think about this.");
expect((assistant as { content: unknown }).content).toBe("The answer is 42.");
expect(assistant?.reasoning_content).toBe("Let me think about this.");
expect(assistant?.content).toBe("The answer is 42.");
});
it("does NOT inject reasoning_content on non-tool-call turn for non-DeepSeek providers", () => {
@@ -539,10 +557,10 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
timestamp: Date.now(),
};
const messages = convertMessages(model, { messages: [msg] }, compat);
const assistant = messages.find(m => m.role === "assistant");
const assistant = findOpenAICompletionAssistantWireMessage(messages);
expect(assistant).toBeDefined();
// OpenRouter reasoning models only need reasoning_content on tool-call turns
expect(Reflect.get(assistant as object, "reasoning_content")).toBeUndefined();
expect(assistant?.reasoning_content).toBeUndefined();
});
});
@@ -573,9 +591,9 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
} as ToolCall,
]);
const messages = convertMessages(model, { messages: [msg] }, compat);
const assistant = messages.find(m => m.role === "assistant");
const assistant = findOpenAICompletionAssistantWireMessage(messages);
expect(assistant).toBeDefined();
expect(Reflect.get(assistant as object, "reasoning_content")).toBe(".");
expect(assistant?.reasoning_content).toBe(".");
});
});
});
@@ -129,6 +129,18 @@ describe("Google Gemini CLI alignment", () => {
});
});
it("keeps projectId alias precedence and tolerates a mistyped projectId", () => {
// projectId wins over project_id even when it is an empty string.
const emptyPrimary = parseGeminiCliCredentials(
JSON.stringify({ token: "t", projectId: "", project_id: "fallback" }),
);
expect(emptyPrimary.projectId).toBe("");
// A non-string projectId is dropped, so project_id is used instead.
const mistyped = parseGeminiCliCredentials(JSON.stringify({ token: "t", projectId: 42, project_id: "fallback" }));
expect(mistyped.projectId).toBe("fallback");
});
it("avoids excessive antigravity refresh churn with pre-buffered OAuth expiry", () => {
const issuedAt = 1_700_000_000_000;
const preBufferedExpiry = issuedAt + 55 * 60 * 1000;
+21 -3
View File
@@ -104,8 +104,26 @@ async function capturePayload(
return promise;
}
interface CompletionAssistantWireMessage {
role: "assistant";
reasoning_content?: unknown;
}
interface CompletionBody {
thinking?: { type?: string; keep?: string };
messages?: unknown[];
stream?: boolean;
}
function isCompletionAssistantWireMessage(message: unknown): message is CompletionAssistantWireMessage {
if (typeof message !== "object" || message === null) return false;
return (message as { role?: unknown }).role === "assistant";
}
function findCompletionAssistantWireMessage(
messages: readonly unknown[] | undefined,
): CompletionAssistantWireMessage | undefined {
return messages?.find(isCompletionAssistantWireMessage);
}
describe("issue #1838 — kimi-k2.6 preserves historical reasoning across tool calls", () => {
@@ -248,11 +266,11 @@ describe("issue #1838 — kimi-k2.6 preserves historical reasoning across tool c
},
],
},
)) as CompletionBody & { messages?: Array<Record<string, unknown>>; stream?: boolean };
)) as CompletionBody;
expect(payload.thinking).toEqual({ type: "enabled", keep: "all" });
expect(payload.stream).toBe(true);
const assistant = payload.messages?.find(m => m.role === "assistant");
const assistant = findCompletionAssistantWireMessage(payload.messages);
expect(assistant).toBeDefined();
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("Need to read the file first.");
expect(assistant?.reasoning_content).toBe("Need to read the file first.");
});
});
+2 -1
View File
@@ -212,7 +212,8 @@ describe("issue #2080 - MiniMax multi-chunk object tool arguments", () => {
},
},
]);
expect(Reflect.get(Object.prototype, "polluted")).toBeUndefined();
const objectPrototype = Object.prototype as { polluted?: unknown };
expect(objectPrototype.polluted).toBeUndefined();
});
it("emits a concat-safe `toolcall_delta` sequence — accumulated deltas parse to the merged args", async () => {
+1 -1
View File
@@ -111,7 +111,7 @@ describe("issue #827 — kimi reasoning models drop reasoning under forced tool_
})) as CompletionsBody;
expect(body.tool_choice).toMatchObject({ type: "function", function: { name: "echo" } });
expect(body.reasoning).toBeUndefined();
expect(body.reasoning).toEqual({ enabled: false });
expect(body.reasoning_effort).toBeUndefined();
});
it("sends explicit thinking disabled for Moonshot Kimi K2.6 when a named tool is forced", async () => {
+21 -4
View File
@@ -4,6 +4,23 @@ import type { AssistantMessage, Model, ModelSpec } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
interface OpenAICompletionAssistantWireMessage {
role: "assistant";
content?: unknown;
reasoning_content?: unknown;
}
function isOpenAICompletionAssistantWireMessage(message: unknown): message is OpenAICompletionAssistantWireMessage {
if (typeof message !== "object" || message === null) return false;
return (message as { role?: unknown }).role === "assistant";
}
function findOpenAICompletionAssistantWireMessage(
messages: readonly unknown[] | undefined,
): OpenAICompletionAssistantWireMessage | undefined {
return messages?.find(isOpenAICompletionAssistantWireMessage);
}
function deepseekModel(overrides: Partial<ModelSpec<"openai-completions">>): Model<"openai-completions"> {
const base = getBundledModel("openai", "gpt-4o-mini");
return buildModel({
@@ -70,9 +87,9 @@ describe("issue #883 / #810 — DeepSeek V4 reasoning_content tool-call replay",
});
const compat = model.compat;
const messages = convertMessages(model, { messages: [assistantWithToolCall(model)] }, compat);
const assistant = messages.find(m => m.role === "assistant");
const assistant = findOpenAICompletionAssistantWireMessage(messages);
expect(assistant).toBeDefined();
const reasoningContent = Reflect.get(assistant as object, "reasoning_content");
const reasoningContent = assistant?.reasoning_content;
expect(reasoningContent).toBeDefined();
// DeepSeek rejects synthetic "." — when no thinking blocks exist, we emit empty string
expect(reasoningContent).toBe("");
@@ -113,8 +130,8 @@ describe("issue #883 / #810 — DeepSeek V4 reasoning_content tool-call replay",
timestamp: Date.now(),
};
const messages = convertMessages(model, { messages: [toolOnly] }, compat);
const assistant = messages.find(m => m.role === "assistant");
const assistant = findOpenAICompletionAssistantWireMessage(messages);
expect(assistant).toBeDefined();
expect((assistant as { content: unknown }).content).toBe("");
expect(assistant?.content).toBe("");
});
});
+1 -1
View File
@@ -46,7 +46,7 @@ function customResponsesModel(compat: OpenAICompat): Model<"openai-responses"> {
effortMap: compat.reasoningEffortMap,
},
compat,
} as Model<"openai-responses">;
} as unknown as Model<"openai-responses">;
}
describe("issue #931 — openai-responses reasoning effort mapping", () => {
@@ -6,7 +6,7 @@ import { convertMessages as convertOpenAICompletionsMessages } from "@oh-my-pi/p
import {
appendResponsesToolResultMessages,
convertResponsesInputContent,
} from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
} from "@oh-my-pi/pi-ai/providers/openai-shared";
import { NON_VISION_IMAGE_PLACEHOLDER } from "@oh-my-pi/pi-ai/providers/vision-guard";
import type { Api, AssistantMessage, Context, Model, ModelSpec, ToolResultMessage, Usage } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
@@ -42,8 +42,13 @@ const compat: ResolvedOpenAICompat = {
requiresThinkingAsText: false,
requiresMistralToolIds: false,
thinkingFormat: "openai",
reasoningDisableMode: "lowest-effort",
omitReasoningEffort: false,
includeEncryptedReasoning: true,
filterReasoningHistory: false,
reasoningContentField: "reasoning_content",
requiresReasoningContentForToolCalls: false,
requiresReasoningContentForAllAssistantTurns: false,
allowsSyntheticReasoningContentForToolCalls: true,
requiresAssistantContentForToolCalls: false,
openRouterRouting: {},
@@ -51,6 +56,12 @@ const compat: ResolvedOpenAICompat = {
extraBody: {},
supportsStrictMode: true,
toolStrictMode: "none",
wireModelIdMode: "raw",
stripDeepseekSpecialTokens: false,
reasoningDeltasMayBeCumulative: false,
emptyLengthFinishIsContextError: false,
usesOpenAIToolCallIdLimit: false,
dropThinkingWhenReasoningEffort: false,
};
function makeModel<TApi extends Api>(api: TApi, provider: Model["provider"]): Model<TApi> {
@@ -3,6 +3,14 @@ import type { Context } from "@oh-my-pi/pi-ai";
import { streamOllama } from "@oh-my-pi/pi-ai/providers/ollama";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
interface OllamaChatRequestPayload {
think?: unknown;
}
function isOllamaChatRequestPayload(value: unknown): value is OllamaChatRequestPayload {
return value !== null && typeof value === "object";
}
function createReasoningOllamaModel() {
return buildModel({
id: "deepseek-v4-flash",
@@ -20,10 +28,10 @@ function createReasoningOllamaModel() {
describe("Ollama chat thinking controls", () => {
it("sends think false when reasoning is explicitly disabled", async () => {
let payload: object | undefined;
let payload: OllamaChatRequestPayload | undefined;
const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
const parsed: unknown = JSON.parse(String(init?.body));
if (parsed === null || typeof parsed !== "object") {
if (!isOllamaChatRequestPayload(parsed)) {
throw new Error("Expected Ollama payload object");
}
payload = parsed;
@@ -41,6 +49,6 @@ describe("Ollama chat thinking controls", () => {
fetch: fetchMock,
}).result();
expect(payload ? Reflect.get(payload, "think") : undefined).toBe(false);
expect(payload?.think).toBe(false);
});
});
@@ -0,0 +1,154 @@
import { describe, expect, it } from "bun:test";
import {
applyChatCompletionsCompatPolicy,
applyResponsesCompatPolicy,
resolveOpenAICompatPolicy,
type OpenAICompletionsParams,
} from "@oh-my-pi/pi-ai/providers/openai-shared";
import type { ResponseCreateParamsStreaming, ResponseInput } from "@oh-my-pi/pi-ai/providers/openai-responses-wire";
import type { Model, ModelSpec, OpenAICompat } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { Effort } from "@oh-my-pi/pi-catalog/effort";
function chatModel(compat: OpenAICompat): Model<"openai-completions"> {
return buildModel({
id: "compat-reasoner",
name: "Compat Reasoner",
api: "openai-completions",
provider: "test-provider",
baseUrl: "https://example.com/v1",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128_000,
maxTokens: 4096,
compat,
} satisfies ModelSpec<"openai-completions">);
}
function responsesModel(compat: OpenAICompat): Model<"openai-responses"> {
return buildModel({
id: "compat-reasoner",
name: "Compat Reasoner",
api: "openai-responses",
provider: "test-provider",
baseUrl: "https://example.com/v1",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128_000,
maxTokens: 4096,
compat,
} satisfies ModelSpec<"openai-responses">);
}
function chatParams(): OpenAICompletionsParams {
return { model: "compat-reasoner", messages: [], stream: true };
}
function responsesParams(): ResponseCreateParamsStreaming {
return { model: "compat-reasoner", input: [], stream: true };
}
describe("OpenAI compat policy", () => {
it("suppresses reasoning on forced tool choice for both endpoints", () => {
const compat: OpenAICompat = {
disableReasoningOnForcedToolChoice: true,
thinkingFormat: "openrouter",
reasoningDisableMode: "openrouter-enabled-false",
};
const toolChoice = { type: "function", name: "search" };
const chatPolicy = resolveOpenAICompatPolicy(chatModel(compat), {
endpoint: "chat-completions",
reasoning: Effort.High,
toolChoice,
});
const responsesPolicy = resolveOpenAICompatPolicy(responsesModel(compat), {
endpoint: "responses",
reasoning: Effort.High,
toolChoice,
});
expect(chatPolicy.reasoning.enabled).toBe(false);
expect(responsesPolicy.reasoning.enabled).toBe(false);
expect(chatPolicy.reasoning.disableReason).toBe("forced-tool-choice");
expect(responsesPolicy.reasoning.disableReason).toBe("forced-tool-choice");
});
it("encodes OpenRouter disabled reasoning through both wire adapters", () => {
const compat: OpenAICompat = {
thinkingFormat: "openrouter",
reasoningDisableMode: "openrouter-enabled-false",
};
const chatBody = chatParams();
const responseBody = responsesParams();
const responseInput: ResponseInput = [];
applyChatCompletionsCompatPolicy(
chatBody,
resolveOpenAICompatPolicy(chatModel(compat), {
endpoint: "chat-completions",
disableReasoning: true,
}),
);
applyResponsesCompatPolicy(
responseBody,
responseInput,
resolveOpenAICompatPolicy(responsesModel(compat), { endpoint: "responses", disableReasoning: true }),
undefined,
);
expect(chatBody.reasoning).toEqual({ enabled: false });
expect(responseBody.reasoning as unknown).toEqual({ enabled: false });
});
it("omits effort for both wire adapters from one catalog flag", () => {
const compat: OpenAICompat = { omitReasoningEffort: true };
const chatBody = chatParams();
const responseBody = responsesParams();
const responseInput: ResponseInput = [];
applyChatCompletionsCompatPolicy(
chatBody,
resolveOpenAICompatPolicy(chatModel(compat), { endpoint: "chat-completions", reasoning: Effort.High }),
);
applyResponsesCompatPolicy(
responseBody,
responseInput,
resolveOpenAICompatPolicy(responsesModel(compat), { endpoint: "responses", reasoning: Effort.High }),
undefined,
);
expect(chatBody.reasoning_effort).toBeUndefined();
expect(responseBody.reasoning).toBeUndefined();
});
it("exposes reasoning replay constraints independent of endpoint", () => {
const compat: OpenAICompat = {
requiresReasoningContentForToolCalls: true,
requiresReasoningContentForAllAssistantTurns: true,
allowsSyntheticReasoningContentForToolCalls: false,
reasoningContentField: "reasoning_content",
};
const chatPolicy = resolveOpenAICompatPolicy(chatModel(compat), { endpoint: "chat-completions" });
const responsesPolicy = resolveOpenAICompatPolicy(responsesModel(compat), { endpoint: "responses" });
expect(chatPolicy.reasoning.requiresReasoningContentForToolCalls).toBe(true);
expect(responsesPolicy.reasoning.requiresReasoningContentForToolCalls).toBe(true);
expect(chatPolicy.reasoning.requiresReasoningContentForAllAssistantTurns).toBe(true);
expect(responsesPolicy.reasoning.requiresReasoningContentForAllAssistantTurns).toBe(true);
expect(chatPolicy.reasoning.allowsSyntheticReasoningContentForToolCalls).toBe(false);
expect(responsesPolicy.reasoning.allowsSyntheticReasoningContentForToolCalls).toBe(false);
});
it("exposes tool id and cumulative reasoning stream constraints for both endpoints", () => {
const compat: OpenAICompat = { requiresMistralToolIds: true, reasoningDeltasMayBeCumulative: true };
const chatPolicy = resolveOpenAICompatPolicy(chatModel(compat), { endpoint: "chat-completions" });
const responsesPolicy = resolveOpenAICompatPolicy(responsesModel(compat), { endpoint: "responses" });
expect(chatPolicy.tools.toolCallIdKind).toBe("mistral-9-alnum");
expect(responsesPolicy.tools.toolCallIdKind).toBe("mistral-9-alnum");
expect(chatPolicy.stream.reasoningDeltasMayBeCumulative).toBe(true);
expect(responsesPolicy.stream.reasoningDeltasMayBeCumulative).toBe(true);
});
});
@@ -33,20 +33,20 @@ function createAbortedSignal(): AbortSignal {
return controller.signal;
}
function toObject(value: unknown): object | null {
return typeof value === "object" && value !== null ? value : null;
function toObject(value: unknown): Record<string, unknown> | null {
return typeof value === "object" && value !== null ? (value as Record<string, unknown>) : null;
}
function getNestedObject(value: unknown, key: string): object | null {
function getNestedObject(value: unknown, key: string): Record<string, unknown> | null {
const obj = toObject(value);
if (!obj) return null;
return toObject(Reflect.get(obj, key));
return toObject(obj[key]);
}
function getNestedBoolean(value: unknown, key: string): boolean | undefined {
const obj = toObject(value);
if (!obj) return undefined;
const property = Reflect.get(obj, key);
const property = obj[key];
return typeof property === "boolean" ? property : undefined;
}
@@ -98,6 +98,17 @@ function zaiGlm52Model(): Model<"openai-completions"> {
} satisfies ModelSpec<"openai-completions">);
}
function kimiZaiModel(): Model<"openai-completions"> {
return buildModel({
...gpt4oMiniSpec,
api: "openai-completions",
provider: "moonshot",
baseUrl: "https://api.moonshot.ai/v1",
id: "kimi-k2.6",
reasoning: true,
} as ModelSpec<"openai-completions">);
}
async function captureOpenAICompletionsPayload(
model: Model<"openai-completions">,
context: Context = baseContext(),
@@ -115,9 +126,9 @@ async function captureOpenAICompletionsPayload(
return promise;
}
function getPayloadMessages(payload: unknown): object[] {
function getPayloadMessages(payload: unknown): Record<string, unknown>[] {
const payloadObject = toObject(payload);
const messages = payloadObject ? Reflect.get(payloadObject, "messages") : undefined;
const messages = payloadObject?.messages;
if (!Array.isArray(messages)) throw new Error("payload messages missing");
return messages.map(message => {
const messageObject = toObject(message);
@@ -129,14 +140,14 @@ function getPayloadMessages(payload: unknown): object[] {
function getLastPayloadContent(payload: unknown): unknown {
const lastMessage = getPayloadMessages(payload).at(-1);
if (!lastMessage) throw new Error("payload has no messages");
return Reflect.get(lastMessage, "content");
return lastMessage.content;
}
function getLastTextPart(content: unknown): object | undefined {
function getLastTextPart(content: unknown): Record<string, unknown> | undefined {
if (!Array.isArray(content)) return undefined;
for (let index = content.length - 1; index >= 0; index--) {
const part = toObject(content[index]);
if (part && Reflect.get(part, "type") === "text") return part;
if (part?.type === "text") return part;
}
return undefined;
}
@@ -164,8 +175,13 @@ describe("openai-completions compatibility", () => {
requiresThinkingAsText: false,
requiresMistralToolIds: false,
thinkingFormat: "openai",
reasoningDisableMode: "lowest-effort",
omitReasoningEffort: false,
includeEncryptedReasoning: true,
filterReasoningHistory: false,
reasoningContentField: "reasoning_content",
requiresReasoningContentForToolCalls: false,
requiresReasoningContentForAllAssistantTurns: false,
allowsSyntheticReasoningContentForToolCalls: true,
requiresAssistantContentForToolCalls: false,
openRouterRouting: {},
@@ -177,6 +193,12 @@ describe("openai-completions compatibility", () => {
alwaysSendMaxTokens: false,
isOpenRouterHost: false,
isVercelGatewayHost: false,
wireModelIdMode: "raw",
stripDeepseekSpecialTokens: false,
reasoningDeltasMayBeCumulative: false,
emptyLengthFinishIsContextError: false,
usesOpenAIToolCallIdLimit: false,
dropThinkingWhenReasoningEffort: false,
} satisfies ResolvedOpenAICompat;
const assistantMessage: AssistantMessage = {
role: "assistant",
@@ -350,7 +372,7 @@ describe("openai-completions compatibility", () => {
// Regression: Moonshot's Kimi chat template rejects the `developer` role
// with `400 Invalid request: tokenization failed` because `developer` is
// an OpenAI extension and most other hosts don't carry it through their
// tokenizer. The default for any non-OpenAI/Azure host MUST be `system`,
// tokenizer. The default for non-OpenAI/Azure hosts MUST be `system`,
// so reasoning models on those hosts cannot accidentally emit `developer`.
const cases: Array<{ provider: string; baseUrl: string; expected: boolean }> = [
{ provider: "openai", baseUrl: "https://api.openai.com/v1", expected: true },
@@ -612,10 +634,46 @@ describe("openai-completions compatibility", () => {
const payload = await promise;
const thinking = getNestedObject(payload, "thinking");
expect(Reflect.get(thinking ?? {}, "type")).toBe("enabled");
expect(Reflect.get(toObject(payload) ?? {}, "reasoning_effort")).toBe("max");
expect(Reflect.get(toObject(payload) ?? {}, "tool_stream")).toBe(true);
expect(Reflect.get(toObject(payload) ?? {}, "max_tokens")).toBe(65_536);
const payloadObject = toObject(payload);
expect(thinking?.type).toBe("enabled");
expect(payloadObject?.reasoning_effort).toBe("max");
expect(payloadObject?.tool_stream).toBe(true);
expect(payloadObject?.max_tokens).toBe(65_536);
});
it("keeps Z.AI tool streaming disabled for native Kimi reasoning models", async () => {
const model = kimiZaiModel();
expect(model.compat.thinkingFormat).toBe("zai");
expect(model.compat.supportsReasoningEffort).toBe(true);
const readTool: Tool = {
name: "read",
description: "Read a file",
parameters: {
type: "object",
properties: { path: { type: "string" } },
required: ["path"],
},
};
const { promise, resolve } = Promise.withResolvers<unknown>();
const fetchMock = createMockFetch(["[DONE]"]);
streamOpenAICompletions(
model,
{ ...baseContext(), tools: [readTool] },
{
apiKey: "test-key",
fetch: fetchMock,
reasoning: "high",
toolChoice: { type: "tool", name: "read" },
signal: createAbortedSignal(),
onPayload: payload => resolve(payload),
},
);
const payload = await promise;
const payloadObject = toObject(payload);
expect(payloadObject?.tool_stream).toBeUndefined();
});
it("maps GLM-5.2 minimal reasoning to disabled Z.AI thinking", async () => {
@@ -631,8 +689,8 @@ describe("openai-completions compatibility", () => {
const payload = await promise;
const thinking = getNestedObject(payload, "thinking");
expect(Reflect.get(thinking ?? {}, "type")).toBe("disabled");
expect(Reflect.get(toObject(payload) ?? {}, "reasoning_effort")).toBeUndefined();
expect(thinking?.type).toBe("disabled");
expect(toObject(payload)?.reasoning_effort).toBeUndefined();
});
it("treats finish_reason end as stop", async () => {
@@ -818,8 +876,8 @@ describe("openai-completions compatibility", () => {
expect(assistant).toBeDefined();
const assistantObject = toObject(assistant);
expect(assistantObject).toBeDefined();
expect(assistantObject ? Reflect.get(assistantObject, "reasoning_text") : undefined).toBe("inspect tool output");
expect(assistantObject ? Reflect.get(assistantObject, "reasoning_content") : undefined).toBeUndefined();
expect(assistantObject?.reasoning_text).toBe("inspect tool output");
expect(assistantObject?.reasoning_content).toBeUndefined();
});
});
@@ -950,7 +1008,7 @@ describe("kimi model detection via detectCompat", () => {
const messages = convertMessages(model, { messages: [toolCallMessage] }, compat);
const assistant = messages.find(m => m.role === "assistant");
expect(assistant).toBeDefined();
expect(Reflect.get(assistant as object, "reasoning_content")).toBeUndefined();
expect(toObject(assistant)?.reasoning_content).toBeUndefined();
});
it("does not replay streamed reasoning fields for kimi on opencode-go", () => {
@@ -995,9 +1053,9 @@ describe("kimi model detection via detectCompat", () => {
if (!assistantObject) {
throw new Error("assistant message missing");
}
expect(Reflect.get(assistantObject, "reasoning")).toBeUndefined();
expect(Reflect.get(assistantObject, "reasoning_content")).toBeUndefined();
expect(Reflect.get(assistantObject, "reasoning_text")).toBeUndefined();
expect(assistantObject.reasoning).toBeUndefined();
expect(assistantObject.reasoning_content).toBeUndefined();
expect(assistantObject.reasoning_text).toBeUndefined();
});
// #1484: OpenCode Zen's Kimi gateway now 400s with `thinking is enabled but
@@ -1071,10 +1129,10 @@ describe("kimi model detection via detectCompat", () => {
const payload = (await promise) as { messages: Array<Record<string, unknown>> };
const assistant = payload.messages.find(m => m.role === "assistant");
expect(assistant).toBeDefined();
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("Need to read the file before answering.");
expect(assistant?.reasoning_content).toBe("Need to read the file before answering.");
// The streamed `reasoning` key must NOT land in the wire body alongside
// `reasoning_content`; opencode's strict schema rejects unknown fields.
expect(Reflect.get(assistant as object, "reasoning")).toBeUndefined();
expect(assistant?.reasoning).toBeUndefined();
});
// #1071 regression guard alongside the #1484 fix: with thinking disabled the
@@ -1137,9 +1195,9 @@ describe("kimi model detection via detectCompat", () => {
const payload = (await promise) as { messages: Array<Record<string, unknown>> };
const assistant = payload.messages.find(m => m.role === "assistant");
expect(assistant).toBeDefined();
expect(Reflect.get(assistant as object, "reasoning_content")).toBeUndefined();
expect(Reflect.get(assistant as object, "reasoning")).toBeUndefined();
expect(Reflect.get(assistant as object, "reasoning_text")).toBeUndefined();
expect(assistant?.reasoning_content).toBeUndefined();
expect(assistant?.reasoning).toBeUndefined();
expect(assistant?.reasoning_text).toBeUndefined();
});
// #1485 review: `disableReasoningOnForcedToolChoice` strips thinking from
@@ -1226,9 +1284,9 @@ describe("kimi model detection via detectCompat", () => {
};
const assistant = payload.messages.find(m => m.role === "assistant");
expect(assistant).toBeDefined();
expect(Reflect.get(assistant as object, "reasoning_content")).toBeUndefined();
expect(Reflect.get(assistant as object, "reasoning")).toBeUndefined();
expect(Reflect.get(assistant as object, "reasoning_text")).toBeUndefined();
expect(assistant?.reasoning_content).toBeUndefined();
expect(assistant?.reasoning).toBeUndefined();
expect(assistant?.reasoning_text).toBeUndefined();
// The forced-tool guard must still strip the request-level thinking
// signal so neither end of the wire mentions reasoning.
expect(payload.reasoning_effort).toBeUndefined();
@@ -1319,7 +1377,7 @@ describe("kimi model detection via detectCompat", () => {
};
const assistant = payload.messages.find(m => m.role === "assistant");
expect(assistant).toBeDefined();
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("Plan first, then call the tool.");
expect(assistant?.reasoning_content).toBe("Plan first, then call the tool.");
expect(payload.reasoning_effort).toBe("high");
expect(payload.tool_choice).toBe("auto");
});
@@ -1400,10 +1458,10 @@ describe("kimi model detection via detectCompat", () => {
const payload = (await promise) as { messages: Array<Record<string, unknown>> };
const assistant = payload.messages.find(m => m.role === "assistant");
expect(assistant).toBeDefined();
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("Need to read the file before answering.");
expect(assistant?.reasoning_content).toBe("Need to read the file before answering.");
// DeepSeek's allowsSynthetic=false must keep the stale `reasoning` key
// off the wire body so opencode's schema validation does not flag it.
expect(Reflect.get(assistant as object, "reasoning")).toBeUndefined();
expect(assistant?.reasoning).toBeUndefined();
});
// #1484 follow-up: the Zen gateway invariant applies to every opencode-go
@@ -1485,13 +1543,13 @@ describe("kimi model detection via detectCompat", () => {
const assistant = payload.messages.find(m => m.role === "assistant");
expect(assistant).toBeDefined();
if (expectReplay) {
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("Plan before acting.");
expect(assistant?.reasoning_content).toBe("Plan before acting.");
// The stale streamed `reasoning` key must never land in the wire body.
expect(Reflect.get(assistant as object, "reasoning")).toBeUndefined();
expect(assistant?.reasoning).toBeUndefined();
} else {
expect(Reflect.get(assistant as object, "reasoning_content")).toBeUndefined();
expect(Reflect.get(assistant as object, "reasoning")).toBeUndefined();
expect(Reflect.get(assistant as object, "reasoning_text")).toBeUndefined();
expect(assistant?.reasoning_content).toBeUndefined();
expect(assistant?.reasoning).toBeUndefined();
expect(assistant?.reasoning_text).toBeUndefined();
}
});
@@ -1526,7 +1584,7 @@ describe("kimi model detection via detectCompat", () => {
const messages = convertMessages(model, { messages: [toolCallMessage] }, compat);
const assistant = messages.find(m => m.role === "assistant");
expect(assistant).toBeDefined();
const reasoningContent = Reflect.get(assistant as object, "reasoning_content");
const reasoningContent = toObject(assistant)?.reasoning_content;
expect(reasoningContent).toBeDefined();
expect(typeof reasoningContent).toBe("string");
expect((reasoningContent as string).length).toBeGreaterThan(0);
@@ -1572,7 +1630,7 @@ describe("kimi model detection via detectCompat", () => {
const messages = convertMessages(model, { messages: [toolCallMessage] }, compat);
const assistant = messages.find(m => m.role === "assistant");
expect(assistant).toBeDefined();
expect(Reflect.get(assistant as object, "reasoning_content")).toBe(".");
expect(toObject(assistant)?.reasoning_content).toBe(".");
});
it("does not inject reasoning_content when model is not kimi", () => {
@@ -1825,8 +1883,8 @@ describe("anthropic cache control for OpenAI-compatible chat completions", () =>
const content = getLastPayloadContent(payload);
const textPart = getLastTextPart(content);
expect(Reflect.get(textPart ?? {}, "text")).toBe("cache me");
expect(Reflect.get(textPart ?? {}, "cache_control")).toEqual({ type: "ephemeral" });
expect(textPart?.text).toBe("cache me");
expect(textPart?.cache_control).toEqual({ type: "ephemeral" });
});
it("preserves OpenRouter Anthropic cache_control detection", async () => {
@@ -1835,7 +1893,7 @@ describe("anthropic cache control for OpenAI-compatible chat completions", () =>
const content = getLastPayloadContent(payload);
const textPart = getLastTextPart(content);
expect(Reflect.get(textPart ?? {}, "cache_control")).toEqual({ type: "ephemeral" });
expect(textPart?.cache_control).toEqual({ type: "ephemeral" });
});
it("does not attach Anthropic cache_control to empty assistant tool-call content", async () => {
@@ -1877,16 +1935,16 @@ describe("anthropic cache control for OpenAI-compatible chat completions", () =>
});
const messages = getPayloadMessages(payload);
const assistant = messages.find(message => {
const toolCalls = Reflect.get(message, "tool_calls");
return Reflect.get(message, "role") === "assistant" && Array.isArray(toolCalls);
const toolCalls = message.tool_calls;
return message.role === "assistant" && Array.isArray(toolCalls);
});
const firstUser = messages.find(message => Reflect.get(message, "role") === "user");
const userContent = firstUser ? Reflect.get(firstUser, "content") : undefined;
const firstUser = messages.find(message => message.role === "user");
const userContent = firstUser?.content;
const textPart = getLastTextPart(userContent);
expect(assistant ? Reflect.get(assistant, "content") : undefined).toBe("");
expect(Reflect.get(textPart ?? {}, "text")).toBe("cache me");
expect(Reflect.get(textPart ?? {}, "cache_control")).toEqual({ type: "ephemeral" });
expect(assistant?.content).toBe("");
expect(textPart?.text).toBe("cache me");
expect(textPart?.cache_control).toEqual({ type: "ephemeral" });
});
it("does not infer Anthropic cache_control for custom Claude ids without compat", async () => {
@@ -84,7 +84,7 @@ async function captureDisableReasoningPayload(model: Model<"openai-completions">
return payload;
}
describe("OpenAI completions disableReasoning", () => {
describe("OpenAI completions disableReasoning and thinking dialects", () => {
it("sends the lowest supported reasoning effort for generic effort-mode models", async () => {
const payload = await captureDisableReasoningPayload(createReasoningEffortModel());
@@ -98,4 +98,186 @@ describe("OpenAI completions disableReasoning", () => {
expect(payload.reasoning_effort).toBe("none");
expect(payload.reasoning).toBeUndefined();
});
// Additional requested tests for applyChatCompletionsReasoningParams dialect / behavior verification
it("sets OpenRouter thinking disabled when disableReasoning: true", async () => {
const model = buildModel({
id: "openrouter-reasoner",
name: "OpenRouter Reasoner",
api: "openai-completions",
provider: "openrouter",
baseUrl: "https://openrouter.ai/api/v1",
reasoning: true,
compat: {
thinkingFormat: "openrouter",
supportsReasoningParams: true,
},
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128_000,
maxTokens: 16_384,
});
const payload = await captureDisableReasoningPayload(model);
expect(payload.reasoning).toEqual({ enabled: false });
});
it("sets Qwen enable_thinking: true with reasoning enabled, and false with forced tool choice", async () => {
const model = buildModel({
id: "qwen-reasoner",
name: "Qwen Reasoner",
api: "openai-completions",
provider: "custom",
baseUrl: "https://proxy.example.com/v1",
reasoning: true,
compat: {
thinkingFormat: "qwen",
supportsReasoningParams: true,
supportsToolChoice: true,
supportsForcedToolChoice: true,
disableReasoningOnForcedToolChoice: true,
},
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128_000,
maxTokens: 16_384,
});
// 1. Enabled check using stream call
const { promise, resolve } = Promise.withResolvers<unknown>();
streamOpenAICompletions(model, testContext, {
apiKey: "test-key",
fetch: createMockFetchForQwen(resolve),
reasoning: "medium",
});
const payloadEnabled = (await promise) as Record<string, unknown>;
expect(payloadEnabled.enable_thinking).toBe(true);
// 2. Disabled on forced tool choice check
const { promise: p2, resolve: r2 } = Promise.withResolvers<unknown>();
streamOpenAICompletions(
model,
{
messages: testContext.messages,
tools: [
{
name: "read",
description: "Read a file",
parameters: { type: "object", properties: {} },
},
],
},
{
apiKey: "test-key",
fetch: createMockFetchForQwen(r2),
reasoning: "medium",
toolChoice: { type: "tool", name: "read" },
},
);
const payloadDisabled = (await p2) as Record<string, unknown>;
expect(payloadDisabled.enable_thinking).toBe(false);
});
it("sets Qwen chat-template thinking format properly", async () => {
const model = buildModel({
id: "qwen-ct-reasoner",
name: "Qwen Chat Template Reasoner",
api: "openai-completions",
provider: "custom",
baseUrl: "https://proxy.example.com/v1",
reasoning: true,
compat: {
thinkingFormat: "qwen-chat-template",
supportsReasoningParams: true,
},
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128_000,
maxTokens: 16_384,
});
const { promise, resolve } = Promise.withResolvers<unknown>();
streamOpenAICompletions(model, testContext, {
apiKey: "test-key",
fetch: createMockFetchForQwen(resolve),
reasoning: "medium",
});
const payload = (await promise) as Record<string, unknown>;
expect(payload.chat_template_kwargs).toEqual({ enable_thinking: true });
});
it("sets Z.AI thinking format and toggles type logically based on forced tool choice", async () => {
const model = buildModel({
id: "zai-reasoner",
name: "Z.AI Reasoner",
api: "openai-completions",
provider: "custom",
baseUrl: "https://proxy.example.com/v1",
reasoning: true,
compat: {
thinkingFormat: "zai",
supportsReasoningParams: true,
supportsToolChoice: true,
supportsForcedToolChoice: true,
disableReasoningOnForcedToolChoice: true,
},
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128_000,
maxTokens: 16_384,
});
const { promise, resolve } = Promise.withResolvers<unknown>();
streamOpenAICompletions(model, testContext, {
apiKey: "test-key",
fetch: createMockFetchForQwen(resolve),
reasoning: "medium",
});
const payloadEnabled = (await promise) as Record<string, unknown>;
expect(payloadEnabled.thinking).toEqual({ type: "enabled" });
const { promise: p2, resolve: r2 } = Promise.withResolvers<unknown>();
streamOpenAICompletions(
model,
{
messages: testContext.messages,
tools: [
{
name: "read",
description: "Read a file",
parameters: { type: "object", properties: {} },
},
],
},
{
apiKey: "test-key",
fetch: createMockFetchForQwen(r2),
reasoning: "medium",
toolChoice: { type: "tool", name: "read" },
},
);
const payloadDisabled = (await p2) as Record<string, unknown>;
expect(payloadDisabled.thinking).toEqual({ type: "disabled" });
});
});
function createMockFetchForQwen(resolve: (value: unknown) => void): FetchImpl {
const fetchMock: FetchImpl = Object.assign(
async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
const payload = JSON.parse(typeof init?.body === "string" ? init.body : "{}") as Record<string, unknown>;
resolve(payload);
return createSseResponse([
{
id: "chatcmpl",
object: "chat.completion.chunk",
created: 0,
model: "model",
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
},
"[DONE]",
]);
},
{ preconnect: fetch.preconnect },
);
return fetchMock;
}
@@ -0,0 +1,144 @@
import { describe, expect, it } from "bun:test";
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import type { Context, FetchImpl, Model, ModelSpec, OpenAICompat, Tool } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
// Each Chat Completions reasoning dialect carries "thinking is on" on a
// different wire field. The `disableReasoningOnForcedToolChoice` /
// `disableReasoningOnToolChoice` conflict policies must turn thinking OFF on the
// *matching* field, not just delete `reasoning_effort` — otherwise a Qwen /
// Qwen-template / OpenRouter request keeps thinking enabled and re-trips the
// very 400 the policy exists to dodge. Both conflict branches funnel through the
// same `disableChatCompletionsReasoningForDialect` helper, so the forced-tool
// path below exercises that helper for every dialect.
function createAbortedSignal(): AbortSignal {
const controller = new AbortController();
controller.abort();
return controller.signal;
}
function createMockFetch(): FetchImpl {
async function mockFetch(_input: string | URL | Request, _init?: RequestInit): Promise<Response> {
return new Response("data: [DONE]\n\n", {
status: 200,
headers: { "content-type": "text/event-stream" },
});
}
return Object.assign(mockFetch, { preconnect: fetch.preconnect });
}
const readTool: Tool = {
name: "read",
description: "Read a file",
parameters: {
type: "object",
properties: { path: { type: "string" } },
required: ["path"],
},
};
function forcedToolContext(): Context {
return {
messages: [{ role: "user", content: "Summarize the README", timestamp: Date.now() }],
tools: [readTool],
};
}
function reasoningDialectModel(
thinkingFormat: NonNullable<OpenAICompat["thinkingFormat"]>,
compatOverrides: Partial<OpenAICompat> = {},
): Model<"openai-completions"> {
return buildModel({
id: "test-reasoning-model",
name: "Test Reasoning Model",
api: "openai-completions",
provider: "custom-openai-compatible",
baseUrl: "https://example.test/v1",
reasoning: true,
compat: {
thinkingFormat,
supportsReasoningParams: true,
supportsReasoningEffort: true,
supportsToolChoice: true,
supportsForcedToolChoice: true,
disableReasoningOnForcedToolChoice: true,
...compatOverrides,
},
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 1_000_000,
maxTokens: 131_072,
} satisfies ModelSpec<"openai-completions">);
}
async function captureForcedToolPayload(model: Model<"openai-completions">): Promise<Record<string, unknown>> {
const { promise, resolve } = Promise.withResolvers<unknown>();
streamOpenAICompletions(model, forcedToolContext(), {
apiKey: "test-key",
fetch: createMockFetch(),
signal: createAbortedSignal(),
reasoning: "high",
toolChoice: { type: "tool", name: "read" },
onPayload: payload => resolve(payload),
});
const payload = await promise;
if (typeof payload !== "object" || payload === null) throw new Error("Expected captured request payload");
return payload as Record<string, unknown>;
}
describe("Chat Completions reasoning-disable conflict policy (per dialect)", () => {
it("disables Z.AI thinking and drops reasoning_effort on forced tool choice", async () => {
const payload = await captureForcedToolPayload(reasoningDialectModel("zai"));
expect(payload.thinking).toEqual({ type: "disabled" });
expect(payload.reasoning_effort).toBeUndefined();
});
it("sets Qwen enable_thinking=false on forced tool choice", async () => {
const payload = await captureForcedToolPayload(reasoningDialectModel("qwen"));
expect(payload.enable_thinking).toBe(false);
expect(payload.reasoning_effort).toBeUndefined();
});
it("sets Qwen chat-template enable_thinking=false on forced tool choice", async () => {
const payload = await captureForcedToolPayload(reasoningDialectModel("qwen-chat-template"));
expect(payload.chat_template_kwargs).toEqual({ enable_thinking: false });
expect(payload.reasoning_effort).toBeUndefined();
});
it("sets OpenRouter reasoning={enabled:false} (never just deleted) on forced tool choice", async () => {
const payload = await captureForcedToolPayload(reasoningDialectModel("openrouter"));
// OpenRouter defaults reasoning models back to thinking when `reasoning` is
// absent, so suppression must explicitly disable rather than delete it.
expect(payload.reasoning).toEqual({ enabled: false });
expect(payload.reasoning_effort).toBeUndefined();
});
it("drops stale reasoning_effort for OpenAI-style effort endpoints on forced tool choice", async () => {
const payload = await captureForcedToolPayload(reasoningDialectModel("openai"));
expect(payload.reasoning_effort).toBeUndefined();
expect(payload.reasoning).toBeUndefined();
});
it("applies the same per-dialect disable on the non-forced disableReasoningOnToolChoice path", async () => {
// Non-forced branch (any tool_choice) shares the dialect-aware helper:
// an OpenRouter reasoning model with disableReasoningOnToolChoice must emit
// reasoning={enabled:false} rather than leaving the effort object in place.
const model = reasoningDialectModel("openrouter", {
disableReasoningOnForcedToolChoice: false,
disableReasoningOnToolChoice: true,
});
const { promise, resolve } = Promise.withResolvers<unknown>();
streamOpenAICompletions(model, forcedToolContext(), {
apiKey: "test-key",
fetch: createMockFetch(),
signal: createAbortedSignal(),
reasoning: "high",
toolChoice: "auto",
onPayload: payload => resolve(payload),
});
const payload = (await promise) as Record<string, unknown>;
expect(payload.tool_choice).toBe("auto");
expect(payload.reasoning).toEqual({ enabled: false });
});
});
@@ -31,8 +31,13 @@ const compat: ResolvedOpenAICompat = {
requiresThinkingAsText: false,
requiresMistralToolIds: false,
thinkingFormat: "openai",
reasoningDisableMode: "lowest-effort",
omitReasoningEffort: false,
includeEncryptedReasoning: true,
filterReasoningHistory: false,
reasoningContentField: "reasoning_content",
requiresReasoningContentForToolCalls: false,
requiresReasoningContentForAllAssistantTurns: false,
allowsSyntheticReasoningContentForToolCalls: true,
requiresAssistantContentForToolCalls: false,
openRouterRouting: {},
@@ -44,6 +49,12 @@ const compat: ResolvedOpenAICompat = {
alwaysSendMaxTokens: false,
isOpenRouterHost: false,
isVercelGatewayHost: false,
wireModelIdMode: "raw",
stripDeepseekSpecialTokens: false,
reasoningDeltasMayBeCumulative: false,
emptyLengthFinishIsContextError: false,
usesOpenAIToolCallIdLimit: false,
dropThinkingWhenReasoningEffort: false,
};
function buildToolResult(toolCallId: string, timestamp: number): ToolResultMessage {
@@ -672,7 +672,7 @@ describe("OpenAI-family first-event timeouts", () => {
);
});
it("errors when OpenAI responses stream closes without response.completed", async () => {
it("errors when OpenAI responses stream closes without a terminal response event", async () => {
const incompleteResponse = createSseResponse([
{ type: "response.created", response: { id: "resp_incomplete" } },
{
@@ -691,7 +691,7 @@ describe("OpenAI-family first-event timeouts", () => {
content: [{ type: "output_text", text: "Hello" }],
},
},
// Intentionally no response.completed — simulates premature provider disconnect.
// Intentionally no response.completed/incomplete — simulates premature provider disconnect.
]);
const fetchMock: FetchImpl = () => Promise.resolve(incompleteResponse);
const result = await streamOpenAIResponses(openAIResponsesModel, baseContext(), {
@@ -700,13 +700,13 @@ describe("OpenAI-family first-event timeouts", () => {
}).result();
expect(result.stopReason).toBe("error");
expect(result.errorMessage).toBe("OpenAI responses stream closed before response.completed was received");
expect(result.errorMessage).toBe("OpenAI responses stream closed before a terminal response event was received");
expect(result.content as unknown[]).toEqual([
{ type: "text", text: "Hello", textSignature: '{"v":1,"id":"msg_incomplete"}' },
]);
});
it("errors when Azure OpenAI responses stream closes without response.completed", async () => {
it("errors when Azure OpenAI responses stream closes without a terminal response event", async () => {
const incompleteResponse = createSseResponse([
{ type: "response.created", response: { id: "resp_incomplete_azure" } },
{
@@ -731,7 +731,7 @@ describe("OpenAI-family first-event timeouts", () => {
content: [{ type: "output_text", text: "Hello azure" }],
},
},
// Intentionally no response.completed — simulates premature provider disconnect.
// Intentionally no response.completed/incomplete — simulates premature provider disconnect.
]);
const fetchMock: FetchImpl = () => Promise.resolve(incompleteResponse);
const result = await streamAzureOpenAIResponses(azureOpenAIResponsesModel, baseContext(), {
@@ -742,7 +742,9 @@ describe("OpenAI-family first-event timeouts", () => {
}).result();
expect(result.stopReason).toBe("error");
expect(result.errorMessage).toBe("Azure OpenAI responses stream closed before response.completed was received");
expect(result.errorMessage).toBe(
"Azure OpenAI responses stream closed before a terminal response event was received",
);
expect(result.content as unknown[]).toEqual([
{ type: "text", text: "Hello azure", textSignature: '{"v":1,"id":"msg_incomplete_azure"}' },
]);
@@ -6,13 +6,13 @@ import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
// Output-token wire policy for OpenAI-family providers:
// - Non-aggregator completions + all responses: clamp to OPENAI_MAX_OUTPUT_TOKENS
// (mirrors Anthropic's cap) so a catalog maxTokens that tracks the context
// window never overflows the upstream.
// - OpenRouter completions: omit default max_tokens entirely. OpenRouter filters
// out any upstream whose output cap is below the requested value (e.g. Cerebras
// GLM-4.7 ~40k), silently defeating provider routing. Explicit caller caps and
// Kimi via OpenRouter are still sent.
// - Non-aggregator completions + non-OpenRouter responses: clamp to
// OPENAI_MAX_OUTPUT_TOKENS (mirrors Anthropic's cap) so a catalog maxTokens
// that tracks the context window never overflows the upstream.
// - OpenRouter completions/responses: omit default max token fields entirely.
// OpenRouter filters out any upstream whose output cap is below the requested
// value (e.g. Cerebras GLM-4.7 ~40k), silently defeating provider routing.
// Explicit caller caps and Kimi chat-completions via OpenRouter are still sent.
const ctx: Context = {
systemPrompt: ["hi"],
@@ -43,13 +43,30 @@ function captureResponsesBody(): { fetchMock: FetchImpl; captured: Record<string
return { fetchMock, captured };
}
async function drainResponses(model: Model<"openai-responses">): Promise<Record<string, unknown>> {
const { fetchMock, captured } = captureResponsesBody();
const stream = streamSimple(model, ctx, { apiKey: "k", fetch: fetchMock });
for await (const event of stream) {
if (event.type === "done" || event.type === "error") break;
async function drainResponses(
model: Model<"openai-responses" | "openrouter">,
maxTokens?: number,
): Promise<Record<string, unknown>> {
const previousOpenRouterResponses = Bun.env.PI_OPENROUTER_RESPONSES;
if (model.api === "openrouter") Bun.env.PI_OPENROUTER_RESPONSES = "1";
try {
const { fetchMock, captured } = captureResponsesBody();
const stream = streamSimple(model, ctx, {
apiKey: "k",
...(maxTokens === undefined ? {} : { maxTokens }),
fetch: fetchMock,
});
for await (const event of stream) {
if (event.type === "done" || event.type === "error") break;
}
return captured;
} finally {
if (previousOpenRouterResponses === undefined) {
delete Bun.env.PI_OPENROUTER_RESPONSES;
} else {
Bun.env.PI_OPENROUTER_RESPONSES = previousOpenRouterResponses;
}
}
return captured;
}
function completionsSse(): Response {
@@ -110,6 +127,21 @@ function glmCompletionsModel(maxTokens: number): Model<"openai-completions"> {
});
}
function openRouterResponsesModel(maxTokens: number): Model<"openrouter"> {
return buildModel({
id: "z-ai/glm-4.7",
name: "GLM 4.7",
api: "openrouter",
provider: "openrouter",
baseUrl: "https://openrouter.ai/api/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 202_752,
maxTokens,
});
}
// Non-aggregator completions model: the 64k clamp applies (max_tokens is sent).
function directCompletionsModel(maxTokens: number): Model<"openai-completions"> {
return buildModel({
@@ -155,6 +187,16 @@ describe("OpenAI-family output-token cap", () => {
expect(body.max_output_tokens).toBe(OPENAI_MAX_OUTPUT_TOKENS);
});
it("omits default max_output_tokens for OpenRouter Responses so provider routing is not filtered", async () => {
const body = await drainResponses(openRouterResponsesModel(131_072));
expect(body.max_output_tokens).toBeUndefined();
});
it("sends explicit maxTokens for OpenRouter Responses caller caps", async () => {
const body = await drainResponses(openRouterResponsesModel(131_072), 2_048);
expect(body.max_output_tokens).toBe(2_048);
});
it("clamps non-aggregator completions output to the 64k ceiling", async () => {
const body = await captureCompletionsBody(directCompletionsModel(131_072), 131_072);
expect(body.max_completion_tokens ?? body.max_tokens).toBe(OPENAI_MAX_OUTPUT_TOKENS);
@@ -0,0 +1,169 @@
import { describe, expect, it } from "bun:test";
import {
applyOpenAIGatewayRouting,
type OpenAIGatewayRoutingCompat,
type OpenAIGatewayRoutingParams,
type ResolveOpenAIOutputTokenInput,
resolveOpenAIOutputTokenParam,
} from "@oh-my-pi/pi-ai/providers/openai-shared";
const OPENAI_MAX_OUTPUT_TOKENS = 64_000;
function tokenInput(overrides: Partial<ResolveOpenAIOutputTokenInput> = {}): ResolveOpenAIOutputTokenInput {
return {
field: "max_completion_tokens",
maxTokens: undefined,
maxTokensExplicit: false,
modelMaxTokens: 131_072,
omitMaxOutputTokens: false,
isOpenRouterHost: false,
alwaysSendMaxTokens: false,
...overrides,
};
}
describe("resolveOpenAIOutputTokenParam", () => {
it("returns the requested cap clamped to the provider ceiling", () => {
expect(resolveOpenAIOutputTokenParam(tokenInput({ maxTokens: 200_000 }))).toEqual({
field: "max_completion_tokens",
value: OPENAI_MAX_OUTPUT_TOKENS,
});
});
it("clamps to the model cap when it is below the provider ceiling", () => {
expect(resolveOpenAIOutputTokenParam(tokenInput({ maxTokens: 50_000, modelMaxTokens: 32_000 }))).toEqual({
field: "max_completion_tokens",
value: 32_000,
});
});
it("honors a raised provider clamp (GLM-5.2 reasoning) above the default ceiling", () => {
expect(
resolveOpenAIOutputTokenParam(
tokenInput({ maxTokens: 120_000, modelMaxTokens: 131_072, providerOutputClamp: 131_072 }),
),
).toEqual({ field: "max_completion_tokens", value: 120_000 });
});
it("selects the wire field name the endpoint requested", () => {
expect(resolveOpenAIOutputTokenParam(tokenInput({ field: "max_tokens", maxTokens: 1024 }))?.field).toBe(
"max_tokens",
);
expect(resolveOpenAIOutputTokenParam(tokenInput({ field: "max_output_tokens", maxTokens: 1024 }))?.field).toBe(
"max_output_tokens",
);
});
it("returns undefined when no cap was requested and the endpoint does not require one", () => {
expect(resolveOpenAIOutputTokenParam(tokenInput({ maxTokens: undefined }))).toBeUndefined();
});
it("defaults from the model cap when alwaysSendMaxTokens and the caller omitted one", () => {
// Kimi-family TPM math: clamp(model cap, provider ceiling).
expect(resolveOpenAIOutputTokenParam(tokenInput({ alwaysSendMaxTokens: true, maxTokens: undefined }))).toEqual({
field: "max_completion_tokens",
value: OPENAI_MAX_OUTPUT_TOKENS,
});
});
it("falls back to the provider ceiling when alwaysSendMaxTokens and no model cap exists", () => {
expect(
resolveOpenAIOutputTokenParam(
tokenInput({ alwaysSendMaxTokens: true, maxTokens: undefined, modelMaxTokens: undefined }),
),
).toEqual({ field: "max_completion_tokens", value: OPENAI_MAX_OUTPUT_TOKENS });
});
it("omits the catalog default for OpenRouter when the caller did not explicitly set a cap", () => {
expect(
resolveOpenAIOutputTokenParam(
tokenInput({ isOpenRouterHost: true, maxTokens: 131_072, maxTokensExplicit: false }),
),
).toBeUndefined();
});
it("preserves an explicit caller cap for OpenRouter", () => {
expect(
resolveOpenAIOutputTokenParam(
tokenInput({ isOpenRouterHost: true, maxTokens: 2048, maxTokensExplicit: true }),
),
).toEqual({ field: "max_completion_tokens", value: 2048 });
});
it("keeps the cap for OpenRouter models that require it (alwaysSendMaxTokens overrides routing omission)", () => {
expect(
resolveOpenAIOutputTokenParam(
tokenInput({
field: "max_output_tokens",
isOpenRouterHost: true,
alwaysSendMaxTokens: true,
maxTokens: 131_072,
maxTokensExplicit: false,
}),
),
).toEqual({ field: "max_output_tokens", value: OPENAI_MAX_OUTPUT_TOKENS });
});
it("drops the field entirely when omitMaxOutputTokens is set (Ollama-style proxies)", () => {
expect(resolveOpenAIOutputTokenParam(tokenInput({ maxTokens: 4096, omitMaxOutputTokens: true }))).toBeUndefined();
});
it("treats a null caller cap like an omitted one", () => {
expect(resolveOpenAIOutputTokenParam(tokenInput({ maxTokens: null }))).toBeUndefined();
});
});
function routingParams(): OpenAIGatewayRoutingParams {
return {};
}
describe("applyOpenAIGatewayRouting", () => {
it("sets the OpenRouter provider routing block", () => {
const routing = { only: ["anthropic"], order: ["anthropic", "openai"] };
const params = routingParams();
applyOpenAIGatewayRouting(params, { isOpenRouterHost: true, openRouterRouting: routing });
expect(params.provider).toEqual(routing);
expect(params.providerOptions).toBeUndefined();
});
it("does not set provider when the host is not OpenRouter", () => {
const params = routingParams();
applyOpenAIGatewayRouting(params, {
isOpenRouterHost: false,
openRouterRouting: { only: ["anthropic"] },
});
expect(params.provider).toBeUndefined();
});
it("maps Vercel gateway routing to providerOptions.gateway", () => {
const params = routingParams();
const compat: OpenAIGatewayRoutingCompat = {
isOpenRouterHost: false,
isVercelGatewayHost: true,
vercelGatewayRouting: { only: ["bedrock"], order: ["bedrock", "anthropic"] },
};
applyOpenAIGatewayRouting(params, compat);
expect(params.providerOptions).toEqual({ gateway: { only: ["bedrock"], order: ["bedrock", "anthropic"] } });
expect(params.provider).toBeUndefined();
});
it("ignores Vercel gateway routing with neither only nor order", () => {
const params = routingParams();
applyOpenAIGatewayRouting(params, {
isOpenRouterHost: false,
isVercelGatewayHost: true,
vercelGatewayRouting: {},
});
expect(params.providerOptions).toBeUndefined();
});
it("does not apply Vercel routing when the host is not the Vercel gateway", () => {
const params = routingParams();
applyOpenAIGatewayRouting(params, {
isOpenRouterHost: false,
isVercelGatewayHost: false,
vercelGatewayRouting: { only: ["bedrock"] },
});
expect(params.providerOptions).toBeUndefined();
});
});
@@ -1,9 +1,38 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { type OpenAIResponsesOptions, streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
import { stream as streamModel } from "@oh-my-pi/pi-ai/stream";
import type { Context, FetchImpl, Model, ProviderSessionState } from "@oh-my-pi/pi-ai/types";
import { buildOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
const model = getBundledModel("openai", "gpt-5-mini") as Model<"openai-responses">;
const openRouterResponsesModel: Model<"openai-responses"> = {
...model,
id: "openai/gpt-5.5",
name: "OpenRouter GPT 5.5",
provider: "openrouter",
baseUrl: "https://openrouter.ai/api/v1",
compat: buildOpenAIResponsesCompat({
id: "openai/gpt-5.5",
name: "OpenRouter GPT 5.5",
provider: "openrouter",
baseUrl: "https://openrouter.ai/api/v1",
}),
};
const xaiOAuthResponsesModel: Model<"openai-responses"> = {
...model,
id: "grok-build",
name: "Grok Build",
provider: "xai-oauth",
baseUrl: "https://api.x.ai/v1",
compat: buildOpenAIResponsesCompat({
id: "grok-build",
name: "Grok Build",
provider: "xai-oauth",
baseUrl: "https://api.x.ai/v1",
reasoning: true,
}),
};
function createSseResponse(events: unknown[]): Response {
const payload = `${events.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`;
@@ -19,15 +48,23 @@ function getHeader(headers: RequestInit["headers"], name: string): string | null
async function captureOpenAIResponseHeaders(
options: OpenAIResponsesOptions,
): Promise<{ sessionId: string | null; clientRequestId: string | null; body: Record<string, unknown> | null }> {
requestModel: Model<"openai-responses"> = model,
): Promise<{
sessionId: string | null;
clientRequestId: string | null;
headers: Headers;
body: Record<string, unknown> | null;
}> {
const captured = {
sessionId: null as string | null,
clientRequestId: null as string | null,
headers: new Headers(),
body: null as Record<string, unknown> | null,
};
const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => {
captured.sessionId = getHeader(init?.headers, "session_id");
captured.clientRequestId = getHeader(init?.headers, "x-client-request-id");
captured.headers = new Headers(init?.headers);
captured.body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : null;
return createSseResponse([
{
@@ -65,7 +102,72 @@ async function captureOpenAIResponseHeaders(
systemPrompt: ["stable system", "stable durable context"],
messages: [{ role: "user", content: "hi", timestamp: Date.now() }],
};
const stream = streamOpenAIResponses(model, context, { apiKey: "test-key", ...options, fetch: fetchMock });
const stream = streamOpenAIResponses(requestModel, context, { apiKey: "test-key", ...options, fetch: fetchMock });
for await (const event of stream) {
if (event.type === "done" || event.type === "error") break;
}
return captured;
}
async function captureDispatchedOpenAIResponseHeaders(
options: OpenAIResponsesOptions,
requestModel: Model<"openai-responses">,
): Promise<{
sessionId: string | null;
clientRequestId: string | null;
headers: Headers;
body: Record<string, unknown> | null;
}> {
const captured = {
sessionId: null as string | null,
clientRequestId: null as string | null,
headers: new Headers(),
body: null as Record<string, unknown> | null,
};
const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => {
captured.sessionId = getHeader(init?.headers, "session_id");
captured.clientRequestId = getHeader(init?.headers, "x-client-request-id");
captured.headers = new Headers(init?.headers);
captured.body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : null;
return createSseResponse([
{
type: "response.output_item.added",
item: { type: "message", id: "msg_1", role: "assistant", status: "in_progress", content: [] },
},
{ type: "response.content_part.added", part: { type: "output_text", text: "" } },
{ type: "response.output_text.delta", delta: "Hello" },
{
type: "response.output_item.done",
item: {
type: "message",
id: "msg_1",
role: "assistant",
status: "completed",
content: [{ type: "output_text", text: "Hello" }],
},
},
{
type: "response.completed",
response: {
status: "completed",
usage: {
input_tokens: 5,
output_tokens: 3,
total_tokens: 8,
input_tokens_details: { cached_tokens: 0 },
},
},
},
]);
});
const context: Context = {
systemPrompt: ["stable system", "stable durable context"],
messages: [{ role: "user", content: "hi", timestamp: Date.now() }],
};
const stream = streamModel(requestModel, context, { apiKey: "test-key", ...options, fetch: fetchMock });
for await (const event of stream) {
if (event.type === "done" || event.type === "error") break;
@@ -111,6 +213,102 @@ describe("openai-responses cache affinity", () => {
expect(captured.body?.prompt_cache_key).toBe("session-123");
});
it("xAI OAuth adapter request shaping does not mutate reused options", async () => {
const options: OpenAIResponsesOptions = {
sessionId: "session-123",
headers: { existing: "header" },
extraBody: { existing: true },
};
const first = await captureDispatchedOpenAIResponseHeaders(options, xaiOAuthResponsesModel);
const second = await captureDispatchedOpenAIResponseHeaders(options, xaiOAuthResponsesModel);
expect(options).toEqual({
sessionId: "session-123",
headers: { existing: "header" },
extraBody: { existing: true },
});
for (const captured of [first, second]) {
expect(getHeader(captured.headers, "x-grok-conv-id")).toBe("session-123");
expect(captured.body?.prompt_cache_key).toBe("session-123");
expect(captured.body?.existing).toBe(true);
expect(captured.body?.reasoning).toBeUndefined();
}
});
it("sets OpenRouter Responses session_id from sessionId in the body", async () => {
const captured = await captureOpenAIResponseHeaders(
{ sessionId: "workflow-123", promptCacheKey: "cache-key-123" },
openRouterResponsesModel,
);
expect(captured.sessionId).toBeNull();
expect(captured.clientRequestId).toBeNull();
expect(captured.body?.session_id).toBe("workflow-123");
expect(captured.body?.prompt_cache_key).toBe("cache-key-123");
});
it("lets explicit headers override OpenRouter Responses defaults", async () => {
const captured = await captureOpenAIResponseHeaders(
{
headers: {
"HTTP-Referer": "https://example.test/",
"X-OpenRouter-Title": "Custom App",
"X-OpenRouter-Cache": "false",
},
},
openRouterResponsesModel,
);
expect(getHeader(captured.headers, "HTTP-Referer")).toBe("https://example.test/");
expect(getHeader(captured.headers, "X-OpenRouter-Title")).toBe("Custom App");
expect(getHeader(captured.headers, "X-OpenRouter-Cache")).toBe("false");
});
it("applies OpenRouter Responses model variants and provider routing to the body", async () => {
const routedModel: Model<"openai-responses"> = {
...openRouterResponsesModel,
compat: {
...openRouterResponsesModel.compat,
openRouterRouting: { only: ["anthropic"], order: ["anthropic"] },
},
};
const captured = await captureOpenAIResponseHeaders({ openrouterVariant: "nitro" }, routedModel);
expect(captured.body?.model).toBe("openai/gpt-5.5:nitro");
expect(captured.body?.provider).toEqual({ only: ["anthropic"], order: ["anthropic"] });
});
it("keeps OpenRouter session_id on values longer than OpenAI prompt cache keys", async () => {
const longSessionId = "s".repeat(100);
const captured = await captureOpenAIResponseHeaders({ sessionId: longSessionId }, openRouterResponsesModel);
expect(captured.body?.session_id).toBe(longSessionId);
expect(captured.body?.prompt_cache_key).not.toBe(longSessionId);
});
it("hashes OpenRouter session_id only past the 256 character limit", async () => {
const tooLongSessionId = "s".repeat(300);
const captured = await captureOpenAIResponseHeaders({ sessionId: tooLongSessionId }, openRouterResponsesModel);
const sessionId = captured.body?.session_id;
expect(typeof sessionId).toBe("string");
expect((sessionId as string).length).toBeLessThanOrEqual(256);
expect(sessionId).not.toBe(tooLongSessionId);
});
it("lets explicit extraBody override OpenRouter Responses session_id", async () => {
const captured = await captureOpenAIResponseHeaders(
{
sessionId: "workflow-123",
extraBody: { session_id: "body-wins" },
},
openRouterResponsesModel,
);
expect(captured.body?.session_id).toBe("body-wins");
});
it("merges adapter extra body fields into the Responses request payload", async () => {
const captured = await captureOpenAIResponseHeaders({
sessionId: "session-123",
@@ -235,4 +433,14 @@ describe("openai-responses cache affinity", () => {
expect(captured.clientRequestId).toBeNull();
expect(captured.body?.prompt_cache_key).toBeUndefined();
});
it("omits OpenRouter Responses session_id when cache retention is disabled", async () => {
const captured = await captureOpenAIResponseHeaders(
{ cacheRetention: "none", sessionId: "workflow-123" },
openRouterResponsesModel,
);
expect(captured.body?.session_id).toBeUndefined();
expect(captured.body?.prompt_cache_key).toBeUndefined();
});
});
@@ -0,0 +1,368 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
import { streamSimple } from "@oh-my-pi/pi-ai/stream";
import type { Context, FetchImpl, Model, ModelSpec, OpenAICompat, StreamOptions } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { Effort } from "@oh-my-pi/pi-catalog/effort";
const context: Context = {
systemPrompt: ["Stay concise."],
messages: [{ role: "user", content: "ping", timestamp: 0 }],
};
function createSseResponse(): Response {
return new Response(
`data: ${JSON.stringify({
type: "response.completed",
response: {
status: "completed",
usage: { input_tokens: 1, output_tokens: 1, total_tokens: 2, input_tokens_details: { cached_tokens: 0 } },
},
})}\n\n`,
{ status: 200, headers: { "content-type": "text/event-stream" } },
);
}
function createChatDoneResponse(): Response {
return new Response("data: [DONE]\n\n", { status: 200, headers: { "content-type": "text/event-stream" } });
}
function buildOpenRouterModel(
overrides: Partial<ModelSpec<"openrouter">> = {},
compat?: OpenAICompat,
): Model<"openrouter"> {
return buildModel({
id: "anthropic/claude-haiku-latest",
name: "Claude Haiku via OpenRouter",
api: "openrouter",
provider: "openrouter",
baseUrl: "https://openrouter.ai/api/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 131_072,
compat,
...overrides,
} as ModelSpec<"openrouter">);
}
function buildOpenRouterResponsesModel(
overrides: Partial<ModelSpec<"openai-responses">> = {},
): Model<"openai-responses"> {
return buildModel({
id: "anthropic/claude-haiku-latest",
name: "Claude Haiku via OpenRouter Responses",
api: "openai-responses",
provider: "openrouter",
baseUrl: "https://openrouter.ai/api/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 131_072,
...overrides,
} as ModelSpec<"openai-responses">);
}
async function capturePseudoChatRequest(
model: Model<"openrouter">,
options: Omit<StreamOptions, "apiKey"> & {
reasoning?: Effort;
disableReasoning?: boolean;
openrouterVariant?: string;
} = {},
): Promise<Record<string, unknown>> {
let body: Record<string, unknown> | undefined;
const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => {
body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : {};
return createChatDoneResponse();
});
const stream = streamOpenAICompletions(model as unknown as Model<"openai-completions">, context, {
apiKey: "test-key",
...options,
fetch: fetchMock,
});
for await (const event of stream) {
if (event.type === "done" || event.type === "error") break;
}
expect(fetchMock).toHaveBeenCalledTimes(1);
if (!body) throw new Error("Expected captured OpenRouter chat request");
return body;
}
async function capturePseudoResponsesRequest(
model: Model<"openrouter">,
options: Omit<StreamOptions, "apiKey"> & {
reasoning?: Effort;
disableReasoning?: boolean;
openrouterVariant?: string;
} = {},
): Promise<Record<string, unknown>> {
let body: Record<string, unknown> | undefined;
const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => {
body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : {};
return createSseResponse();
});
const stream = streamOpenAIResponses(model as unknown as Model<"openai-responses">, context, {
apiKey: "test-key",
...options,
fetch: fetchMock,
});
for await (const event of stream) {
if (event.type === "done" || event.type === "error") break;
}
expect(fetchMock).toHaveBeenCalledTimes(1);
if (!body) throw new Error("Expected captured OpenRouter Responses request");
return body;
}
async function captureRequest<TApi extends "openrouter" | "openai-responses">(
model: Model<TApi>,
options: Omit<StreamOptions, "apiKey"> & {
reasoning?: Effort;
disableReasoning?: boolean;
openrouterVariant?: string;
} = {},
): Promise<{ body: Record<string, unknown>; headers: Headers }> {
let body: Record<string, unknown> | undefined;
let headers: Headers | undefined;
const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => {
body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : {};
headers = new Headers(init?.headers);
return createSseResponse();
});
const stream = streamSimple(model, context, { apiKey: "test-key", ...options, fetch: fetchMock });
for await (const event of stream) {
if (event.type === "done" || event.type === "error") break;
}
expect(fetchMock).toHaveBeenCalledTimes(1);
if (!body || !headers) throw new Error("Expected captured OpenRouter Responses request");
return { body, headers };
}
afterEach(() => {
vi.restoreAllMocks();
});
describe("OpenRouter pseudo API dual-surface request parity", () => {
it("builds equivalent Anthropic reasoning payloads for chat and Responses", async () => {
const routing = { only: ["anthropic"], order: ["anthropic", "openai"] };
const model = buildOpenRouterModel(
{ id: "anthropic/claude-fable-5", name: "Claude Fable 5", reasoning: true },
{ openRouterRouting: routing },
);
const chatBody = await capturePseudoChatRequest(model, {
openrouterVariant: "nitro",
reasoning: Effort.High,
sessionId: "workflow-123",
});
const responsesBody = await capturePseudoResponsesRequest(model, {
openrouterVariant: "nitro",
reasoning: Effort.High,
sessionId: "workflow-123",
});
expect(chatBody).toEqual({
model: "anthropic/claude-fable-5:nitro",
messages: [
{ role: "system", content: "Stay concise." },
{
role: "user",
content: [{ type: "text", text: "ping", cache_control: { type: "ephemeral" } }],
},
],
stream: true,
stream_options: { include_usage: true },
store: false,
reasoning: { effort: "xhigh" },
provider: routing,
});
expect(responsesBody).toEqual({
model: "anthropic/claude-fable-5:nitro",
instructions: "Stay concise.",
stream: true,
input: [{ role: "user", content: [{ type: "input_text", text: "ping" }] }],
store: false,
reasoning: { effort: "xhigh", summary: "auto" },
prompt_cache_key: "workflow-123",
session_id: "workflow-123",
provider: routing,
include: ["reasoning.encrypted_content"],
});
expect(chatBody).not.toHaveProperty("max_tokens");
expect(chatBody).not.toHaveProperty("max_completion_tokens");
expect(responsesBody).not.toHaveProperty("max_output_tokens");
});
it("builds equivalent non-Anthropic reasoning payloads for chat and Responses", async () => {
const routing = { order: ["deepseek", "openai"] };
const model = buildOpenRouterModel(
{ id: "deepseek/deepseek-r1", name: "DeepSeek R1", reasoning: true },
{ openRouterRouting: routing },
);
const chatBody = await capturePseudoChatRequest(model, {
openrouterVariant: "nitro",
reasoning: Effort.High,
sessionId: "workflow-456",
});
const responsesBody = await capturePseudoResponsesRequest(model, {
openrouterVariant: "nitro",
reasoning: Effort.High,
sessionId: "workflow-456",
});
expect(chatBody).toEqual({
model: "deepseek/deepseek-r1:nitro",
messages: [
{ role: "system", content: "Stay concise." },
{ role: "user", content: "ping" },
],
stream: true,
stream_options: { include_usage: true },
store: false,
reasoning: { effort: "high" },
provider: routing,
});
expect(responsesBody).toEqual({
model: "deepseek/deepseek-r1:nitro",
instructions: "Stay concise.",
stream: true,
input: [{ role: "user", content: [{ type: "input_text", text: "ping" }] }],
store: false,
reasoning: { effort: "high", summary: "auto" },
prompt_cache_key: "workflow-456",
session_id: "workflow-456",
provider: routing,
include: ["reasoning.encrypted_content"],
});
expect(chatBody).not.toHaveProperty("max_tokens");
expect(chatBody).not.toHaveProperty("max_completion_tokens");
expect(responsesBody).not.toHaveProperty("max_output_tokens");
});
it("keeps stop and frequency penalty on Chat Completions but drops them for Responses", async () => {
const model = buildOpenRouterModel();
const options = {
stopSequences: ["</stop>"],
frequencyPenalty: 0.75,
};
const chatBody = await capturePseudoChatRequest(model, options);
const responsesBody = await capturePseudoResponsesRequest(model, options);
expect(chatBody.stop).toBe("</stop>");
expect(chatBody.frequency_penalty).toBe(0.75);
expect(responsesBody).not.toHaveProperty("stop");
expect(responsesBody).not.toHaveProperty("frequency_penalty");
});
});
describe("OpenRouter Responses request shape", () => {
it("appends openrouterVariant only when the resolved model id has no variant after the final slash", async () => {
const suffixed = await captureRequest(buildOpenRouterResponsesModel(), { openrouterVariant: "nitro" });
expect(suffixed.body.model).toBe("anthropic/claude-haiku-latest:nitro");
const explicit = await captureRequest(
buildOpenRouterResponsesModel({ id: "anthropic/claude-haiku-latest:online" }),
{ openrouterVariant: "nitro" },
);
expect(explicit.body.model).toBe("anthropic/claude-haiku-latest:online");
});
it("resolves pseudo OpenRouter API compat and preserves provider routing in the Responses body", async () => {
const routing = { only: ["anthropic"], order: ["anthropic", "openai"] };
const pseudoModel = buildOpenRouterModel({}, { openRouterRouting: routing });
expect(pseudoModel.compat.openRouterRouting).toEqual(routing);
const { body } = await captureRequest(buildOpenRouterResponsesModel({ compat: { openRouterRouting: routing } }));
expect(body.provider).toEqual(routing);
});
it("explicitly disables OpenRouter Responses reasoning with enabled=false", async () => {
const { body } = await captureRequest(
buildOpenRouterResponsesModel({ id: "deepseek/deepseek-r1", name: "DeepSeek R1", reasoning: true }),
{ disableReasoning: true },
);
expect(body.reasoning).toEqual({ enabled: false });
});
it("omits default max_output_tokens for OpenRouter but sends explicit caller caps", async () => {
const defaultRequest = await captureRequest(buildOpenRouterResponsesModel());
expect(defaultRequest.body).not.toHaveProperty("max_output_tokens");
const explicitRequest = await captureRequest(buildOpenRouterResponsesModel(), { maxTokens: 2048 });
expect(explicitRequest.body.max_output_tokens).toBe(2048);
});
it("keeps default max_output_tokens for OpenRouter models that require it", async () => {
const request = await captureRequest(
buildOpenRouterResponsesModel({ id: "moonshotai/kimi-k2.6", name: "Kimi K2.6" }),
);
expect(request.body.max_output_tokens).toBe(64_000);
});
it("lets caller headers override OpenRouter attribution and cache defaults", async () => {
const { headers } = await captureRequest(buildOpenRouterResponsesModel(), {
headers: {
"HTTP-Referer": "https://caller.example/",
"X-OpenRouter-Title": "Caller App",
"X-OpenRouter-Cache": "false",
"X-OpenRouter-Cache-TTL": "7",
},
});
expect(headers.get("HTTP-Referer")).toBe("https://caller.example/");
expect(headers.get("X-OpenRouter-Title")).toBe("Caller App");
expect(headers.get("X-OpenRouter-Cache")).toBe("false");
expect(headers.get("X-OpenRouter-Cache-TTL")).toBe("7");
});
});
async function captureDirectResponsesRequest(
model: Model<"openai-responses">,
options: Omit<StreamOptions, "apiKey"> & { reasoning?: Effort } = {},
): Promise<Record<string, unknown>> {
let body: Record<string, unknown> | undefined;
const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => {
body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : {};
return createSseResponse();
});
const stream = streamOpenAIResponses(model, context, { apiKey: "test-key", ...options, fetch: fetchMock });
for await (const event of stream) {
if (event.type === "done" || event.type === "error") break;
}
expect(fetchMock).toHaveBeenCalledTimes(1);
if (!body) throw new Error("Expected captured direct Responses request");
return body;
}
describe("Responses direct-provider max-token defaults", () => {
// `streamOpenAIResponses` is public; callers may bypass `streamSimple` (which
// pre-fills `options.maxTokens` from the model cap). After centralizing the
// output-token policy in `resolveOpenAIOutputTokenParam`, the provider itself
// supplies the `alwaysSendMaxTokens` default — so a direct call now emits
// `max_output_tokens` for Kimi-style models even with no caller cap. This is
// an intentional consistency change vs. the prior inline logic, which only
// emitted the field because `streamSimple` had injected it.
it("emits max_output_tokens for an alwaysSendMaxTokens model on a direct call with no caller cap", async () => {
const body = await captureDirectResponsesRequest(
buildOpenRouterResponsesModel({ id: "moonshotai/kimi-k2.6", name: "Kimi K2.6" }),
);
expect(body.max_output_tokens).toBe(64_000);
});
it("still omits max_output_tokens for a routing-only OpenRouter model on a direct call", async () => {
const body = await captureDirectResponsesRequest(buildOpenRouterResponsesModel());
expect(body).not.toHaveProperty("max_output_tokens");
});
});
@@ -1,9 +1,9 @@
import { describe, expect, it } from "bun:test";
import type { ResponseInput } from "@oh-my-pi/pi-ai/providers/openai-responses-wire";
import {
repairOrphanResponsesToolCalls,
repairOrphanResponsesToolOutputs,
} from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
import type { ResponseInput } from "@oh-my-pi/pi-ai/providers/openai-responses-wire";
} from "@oh-my-pi/pi-ai/providers/openai-shared";
describe("repairOrphanResponsesToolCalls", () => {
it("appends a synthetic function_call_output after a call with no result", () => {
@@ -13,8 +13,8 @@
// `output_item.done` event must be routed by `output_index`/`item_id`, not by
// arrival order.
import { describe, expect, test } from "bun:test";
import { processResponsesStream } from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
import type { ResponseStreamEvent } from "@oh-my-pi/pi-ai/providers/openai-responses-wire";
import { processResponsesStream } from "@oh-my-pi/pi-ai/providers/openai-shared";
import type { AssistantMessage, Model } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
@@ -8,8 +8,8 @@
// input on the stored content block and drop the transient `partialJson`
// accumulation buffer, mirroring the function_call branch.
import { describe, expect, test } from "bun:test";
import { processResponsesStream } from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
import type { ResponseStreamEvent } from "@oh-my-pi/pi-ai/providers/openai-responses-wire";
import { processResponsesStream } from "@oh-my-pi/pi-ai/providers/openai-shared";
import type { AssistantMessage, Model } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
@@ -1,6 +1,7 @@
import { describe, expect, it } from "bun:test";
import { applyAnthropicUsageExtras } from "@oh-my-pi/pi-ai/providers/anthropic";
import { parseChunkUsage } from "@oh-my-pi/pi-ai/providers/openai-completions";
import { calculateOpenAIUsageAccounting } from "@oh-my-pi/pi-ai/providers/openai-shared";
import type { Model, Usage } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
@@ -203,6 +204,59 @@ describe("openai-completions parseChunkUsage", () => {
});
});
describe("shared OpenAI usage accounting", () => {
it("uses provider cache-write details ahead of native DeepSeek passthrough fields", () => {
const usage = calculateOpenAIUsageAccounting({
promptTokens: 6_000,
outputTokens: 250,
cachedTokens: 200,
reasoningTokens: 0,
cacheWriteOpenRouter: 5_000,
cacheWriteDeepSeek: 50,
hasDeepSeekCacheHitAndMiss: true,
});
expect(usage.input).toBe(800);
expect(usage.cacheRead).toBe(200);
expect(usage.cacheWrite).toBe(5_000);
expect(usage.totalTokens).toBe(6_250);
});
it("does not emit DeepSeek cache misses as cache writes", () => {
const usage = calculateOpenAIUsageAccounting({
promptTokens: 150,
outputTokens: 200,
cachedTokens: 100,
reasoningTokens: 0,
cacheWriteOpenRouter: undefined,
cacheWriteDeepSeek: 50,
hasDeepSeekCacheHitAndMiss: true,
});
expect(usage.input).toBe(50);
expect(usage.cacheRead).toBe(100);
expect(usage.cacheWrite).toBe(0);
expect(usage.totalTokens).toBe(350);
});
it("treats zero provider cache-write as present when native fields pass through", () => {
const usage = calculateOpenAIUsageAccounting({
promptTokens: 150,
outputTokens: 25,
cachedTokens: 100,
reasoningTokens: 0,
cacheWriteOpenRouter: 0,
cacheWriteDeepSeek: 50,
hasDeepSeekCacheHitAndMiss: true,
});
expect(usage.input).toBe(50);
expect(usage.cacheRead).toBe(100);
expect(usage.cacheWrite).toBe(0);
expect(usage.totalTokens).toBe(175);
});
});
describe("anthropic applyAnthropicUsageExtras", () => {
it("captures cache TTL breakdown when both buckets are non-zero", () => {
const usage = blankUsage();
@@ -10,8 +10,8 @@ import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
// returns undefined for them instead of tripping `requireSupportedEffort`
// (the old user-visible "Compaction failed: Thinking effort high is not
// supported by xai-oauth/grok-build. Supported efforts:" with an empty list),
// and the wire-side `omitReasoningEffort` gate (providers/xai-responses.ts)
// remains the single source of truth for the actual strip.
// and the wire-side `omitReasoningEffort` gate (stream.ts) remains the single
// source of truth for the actual strip.
describe("effort-dial-less reasoner encoding (regression)", () => {
test("xai-oauth/grok-build reasons but carries no thinking config", () => {
const grokBuild = getBundledModel("xai-oauth", "grok-build");
+11 -1
View File
@@ -1,6 +1,15 @@
# Changelog
## [Unreleased]
### Added
- Added a dedicated `openrouter` API type and `ResolvedOpenRouterCompat` configuration to support unified chat-completions and Responses-API compatibility for OpenRouter models
### Changed
- Migrated bundled OpenRouter models in the catalog from `openai-completions` to the new `openrouter` API type
- Consolidated the resolved OpenAI compat shape: extracted a shared `ResolvedOpenAISharedCompat` core that both `ResolvedOpenAICompat` and `ResolvedOpenAIResponsesCompat` extend (each builder still computes its own per-surface value, preserving chat↔Responses divergence), added internal resolved wire-quirk fields (`wireModelIdMode`, `stripDeepseekSpecialTokens`, `reasoningDeltasMayBeCumulative`, `emptyLengthFinishIsContextError`, `usesOpenAIToolCallIdLimit`, `dropThinkingWhenReasoningEffort`, `supportsObfuscationOptOut`), and replaced `buildOpenRouterCompat`'s cast-and-copy with an exhaustive `pickResponsesOnly` composition that fails to compile if a new Responses-only field is added without handling. The public `OpenAICompat` config vocabulary is unchanged.
- Expanded `OpenAICompat`/`ResolvedOpenAISharedCompat` with shared reasoning/history/stream/request flags (`reasoningDisableMode`, `omitReasoningEffort`, `includeEncryptedReasoning`, `filterReasoningHistory`, `requiresReasoningContentForAllAssistantTurns`, `streamMarkupHealingPattern`, `promptCacheSessionHeader`, etc.) so model/provider/gateway constraints are declared once in catalog compat and then consumed uniformly by Chat Completions and Responses endpoints.
## [16.0.5] - 2026-06-17
@@ -19,6 +28,7 @@
### Fixed
- Fixed OpenRouter pseudo-API model construction so bundled OpenRouter models resolve shared OpenAI compatibility metadata instead of an undefined compat record.
- Fixed `off` effort routing for `claude-opus-4-5` and `claude-opus-4-6` to use their base model IDs when thinking is disabled
- Fixed `gemini-2.5-flash` effort routing so all non-off effort levels resolve to `gemini-2.5-flash-thinking`
- Fixed shared variant alias provider resolution so `resolveBareVariantAlias` reports all matching providers when model aliases are present in both CCA collapse tables
@@ -257,4 +267,4 @@
### Removed
- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads.
- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads.
+1
View File
@@ -35,6 +35,7 @@
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
"arktype": "catalog:",
"zod": "catalog:"
},
"devDependencies": {
+3 -1
View File
@@ -10,7 +10,7 @@
* compat per request.
*/
import { buildAnthropicCompat } from "./compat/anthropic";
import { buildOpenAICompat, buildOpenAIResponsesCompat } from "./compat/openai";
import { buildOpenAICompat, buildOpenAIResponsesCompat, buildOpenRouterCompat } from "./compat/openai";
import { resolveModelThinking } from "./model-thinking";
import type { Api, CompatOf, Model, ModelSpec } from "./types";
import { cleanModelName } from "./utils";
@@ -28,6 +28,8 @@ export function buildModel<TApi extends Api>(spec: ModelSpec<TApi>): Model<TApi>
export function buildCompat(spec: ModelSpec<Api>): CompatOf<Api> {
switch (spec.api) {
case "openrouter":
return buildOpenRouterCompat(spec as ModelSpec<"openrouter">);
case "openai-completions":
return buildOpenAICompat(spec as ModelSpec<"openai-completions">);
case "openai-responses":
+175 -17
View File
@@ -19,7 +19,15 @@ import {
isQwenModelId,
modelFamilyToken,
} from "../identity/family";
import type { ModelSpec, OpenAICompat, ResolvedOpenAICompat, ResolvedOpenAIResponsesCompat } from "../types";
import type {
ModelSpec,
OpenAICompat,
OpenAIStreamMarkupHealingPattern,
ResolvedOpenAICompat,
ResolvedOpenAIResponsesCompat,
ResolvedOpenAISharedCompat,
ResolvedOpenRouterCompat,
} from "../types";
import { applyCompatOverrides } from "./apply";
/** GLM coding-plan SKUs idle for minutes mid-reasoning; see `streamIdleTimeoutMs`. */
@@ -29,6 +37,50 @@ const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000;
const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
/** Kimi K2.6 can spend several minutes reasoning before the first visible token. */
const KIMI_K26_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
const MINIMAX_PROVIDER_OR_ID_PATTERN = /minimax/i;
const DSML_HEALING_PROVIDERS = new Set([
"ollama",
"ollama-cloud",
"nvidia",
"deepseek",
"fireworks",
"nanogpt",
"opencode-go",
"openrouter",
]);
function resolveReasoningDisableMode(
thinkingFormat: ResolvedOpenAISharedCompat["thinkingFormat"],
): ResolvedOpenAISharedCompat["reasoningDisableMode"] {
switch (thinkingFormat) {
case "openrouter":
return "openrouter-enabled-false";
case "zai":
return "zai-thinking-disabled";
case "qwen":
return "qwen-enable-thinking-false";
case "qwen-chat-template":
return "qwen-template-false";
default:
return "lowest-effort";
}
}
function detectStreamMarkupHealingPattern(
provider: string,
modelId: string,
): OpenAIStreamMarkupHealingPattern | undefined {
if (MINIMAX_PROVIDER_OR_ID_PATTERN.test(provider) || MINIMAX_PROVIDER_OR_ID_PATTERN.test(modelId)) {
return "thinking";
}
if (provider === "kimi-code" || provider === "moonshot" || /kimi[-/_.]?k2/i.test(modelId)) {
return "kimi";
}
if (isDeepseekModelIdOrName(modelId) && DSML_HEALING_PROVIDERS.has(provider)) {
return "dsml";
}
return undefined;
}
/**
* OpenCode's gateways (https://opencode.ai/zen|go) gate `reasoning_content`
@@ -197,6 +249,26 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
? DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS
: undefined;
const wireModelIdMode: ResolvedOpenAISharedCompat["wireModelIdMode"] =
provider === "firepass"
? "firepass"
: provider === "fireworks"
? "fireworks"
: isOpenRouter
? "openrouter"
: "raw";
const thinkingFormat: ResolvedOpenAISharedCompat["thinkingFormat"] =
isZai || isZhipu || isMoonshotKimi || isXiaomiMimo
? "zai"
: isOpenRouter
? "openrouter"
: isQwen && isNvidiaNim
? "qwen-chat-template"
: isAlibaba || isQwen
? "qwen"
: "openai";
const compat: ResolvedOpenAICompat = {
supportsStore: !isNonStandard,
// `developer` is an OpenAI-Responses-era extension to the chat-completions schema. Almost
@@ -241,16 +313,11 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
// (`chat_template_kwargs.enable_thinking`); top-level `enable_thinking`
// is rejected by NIM's `additionalProperties: false` request schema
// (issue #2299).
thinkingFormat:
isZai || isZhipu || isMoonshotKimi || isXiaomiMimo
? "zai"
: isOpenRouter
? "openrouter"
: isQwen && isNvidiaNim
? "qwen-chat-template"
: isAlibaba || isQwen
? "qwen"
: "openai",
thinkingFormat,
reasoningDisableMode: resolveReasoningDisableMode(thinkingFormat),
omitReasoningEffort: false,
includeEncryptedReasoning: true,
filterReasoningHistory: false,
thinkingKeep: usesMoonshotKimiPreservedThinking ? "all" : undefined,
reasoningContentField: "reasoning_content",
// Backends that 400 follow-up requests when prior assistant tool-call turns lack `reasoning_content`:
@@ -271,6 +338,8 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
(isDeepseekFamily && Boolean(spec.reasoning)) ||
isXiaomiMimo ||
(isOpenRouter && Boolean(spec.reasoning)),
requiresReasoningContentForAllAssistantTurns:
((isDeepseekFamily && Boolean(spec.reasoning)) || isXiaomiMimo) && !isOpenRouter,
// DeepSeek V4 and Xiaomi MiMo reject synthetic reasoning_content placeholders (".") on tool-call turns.
// Kimi and OpenRouter accept them when actual reasoning is unavailable.
allowsSyntheticReasoningContentForToolCalls: (!isDeepseekFamily || !spec.reasoning) && !isXiaomiMimo,
@@ -279,20 +348,43 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
openRouterRouting: undefined,
vercelGatewayRouting: undefined,
isOpenRouterHost: isOpenRouter,
wireModelIdMode,
isVercelGatewayHost: isVercelGateway,
supportsStrictMode: detectStrictModeSupport(provider, baseUrl),
extraBody: isDirectDeepseekReasoning ? { thinking: { type: "enabled" } } : undefined,
toolStrictMode: isCerebras ? "all_strict" : "mixed",
streamIdleTimeoutMs,
stripDeepseekSpecialTokens:
isDeepseekModelIdOrName(spec.id) && (provider === "nvidia" || provider === "deepseek"),
streamMarkupHealingPattern: detectStreamMarkupHealingPattern(provider, spec.id),
reasoningDeltasMayBeCumulative:
MINIMAX_PROVIDER_OR_ID_PATTERN.test(provider) || MINIMAX_PROVIDER_OR_ID_PATTERN.test(spec.id),
emptyLengthFinishIsContextError: provider === "ollama",
usesOpenAIToolCallIdLimit: provider === "openai",
promptCacheSessionHeader: undefined,
dropThinkingWhenReasoningEffort: provider === "fireworks",
};
applyCompatOverrides(compat, spec.compat);
if (spec.compat?.reasoningDisableMode === undefined) {
compat.reasoningDisableMode = resolveReasoningDisableMode(compat.thinkingFormat);
}
if (spec.compat?.omitReasoningEffort === undefined && !compat.supportsReasoningEffort) {
compat.omitReasoningEffort = true;
}
const whenThinkingPolicy =
spec.compat?.whenThinking ?? (isOpenCodeProvider && spec.reasoning ? OPENCODE_WHEN_THINKING : undefined);
if (whenThinkingPolicy) {
const variant: ResolvedOpenAICompat = { ...compat };
applyCompatOverrides(variant, whenThinkingPolicy);
if (whenThinkingPolicy.reasoningDisableMode === undefined) {
variant.reasoningDisableMode = resolveReasoningDisableMode(variant.thinkingFormat);
}
if (whenThinkingPolicy.omitReasoningEffort === undefined && !variant.supportsReasoningEffort) {
variant.omitReasoningEffort = true;
}
compat.whenThinking = variant;
}
@@ -304,6 +396,7 @@ interface OpenAIResponsesSpecLike {
provider: string;
name: string;
baseUrl: string;
reasoning?: boolean;
compat?: OpenAICompat;
}
@@ -321,22 +414,87 @@ interface OpenAIResponsesSpecLike {
export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): ResolvedOpenAIResponsesCompat {
const baseUrl = spec.baseUrl ?? "";
const isAzure = modelMatchesHost({ provider: spec.provider, baseUrl }, "azureOpenAI");
const isOpenRouter = modelMatchesHost({ provider: spec.provider, baseUrl }, "openrouter");
const isOpenAIUrl = hostMatchesUrl(baseUrl, "openai");
const id = spec.id ?? "";
const thinkingFormat: ResolvedOpenAISharedCompat["thinkingFormat"] = isOpenRouter ? "openrouter" : "openai";
const isKimiModel = id ? isKimiModelId(id) : false;
const isDeepseekFamily = id ? isDeepseekModelIdOrName(id) || isDeepseekModelIdOrName(spec.name) : false;
const reasoningCapable = Boolean(spec.reasoning);
const compat: ResolvedOpenAIResponsesCompat = {
supportsDeveloperRole: isAzure || hostMatchesUrl(baseUrl, "openai") || hostMatchesUrl(baseUrl, "githubCopilot"),
supportsDeveloperRole: isAzure || isOpenAIUrl || hostMatchesUrl(baseUrl, "githubCopilot"),
supportsStrictMode:
spec.provider === "openai" ||
isAzure ||
spec.provider === "github-copilot" ||
hostMatchesUrl(baseUrl, "openai"),
spec.provider === "openai" || isAzure || spec.provider === "github-copilot" || isOpenRouter || isOpenAIUrl,
supportsReasoningEffort: true,
supportsLongPromptCacheRetention: hostMatchesUrl(baseUrl, "openai"),
supportsLongPromptCacheRetention: isOpenAIUrl,
// Azure OpenAI and GitHub Copilot Responses paths require tool results
// to strictly match prior tool calls when building Responses inputs.
strictResponsesPairing: isAzure || spec.provider === "github-copilot",
requiresJuiceZeroHack: spec.name.toLowerCase().startsWith("gpt-5"),
reasoningEffortMap: {},
supportsReasoningParams: true,
thinkingFormat,
reasoningDisableMode: resolveReasoningDisableMode(thinkingFormat),
omitReasoningEffort: false,
includeEncryptedReasoning: spec.provider !== "xai-oauth",
filterReasoningHistory: spec.provider === "xai-oauth",
disableReasoningOnForcedToolChoice: isKimiModel,
disableReasoningOnToolChoice: isDeepseekFamily && reasoningCapable && !isOpenRouter,
supportsToolChoice: true,
supportsForcedToolChoice: true,
reasoningContentField: "reasoning_content",
requiresReasoningContentForToolCalls:
(isKimiModel || (isDeepseekFamily && reasoningCapable) || (isOpenRouter && reasoningCapable)) &&
reasoningCapable,
requiresReasoningContentForAllAssistantTurns: isDeepseekFamily && reasoningCapable && !isOpenRouter,
allowsSyntheticReasoningContentForToolCalls: !isDeepseekFamily || !reasoningCapable,
requiresThinkingAsText: false,
requiresMistralToolIds: false,
requiresToolResultName: false,
requiresAssistantAfterToolResult: false,
requiresAssistantContentForToolCalls: isKimiModel,
openRouterRouting: undefined,
isOpenRouterHost: isOpenRouter,
wireModelIdMode: isOpenRouter ? "openrouter" : "raw",
alwaysSendMaxTokens: spec.id ? isKimiModelId(spec.id) : false,
enableGeminiThinkingLoopGuard: modelFamilyToken(spec.id ?? "") === "gemini",
supportsObfuscationOptOut: isOpenAIUrl || spec.provider === "openai",
stripDeepseekSpecialTokens:
Boolean(id) && isDeepseekModelIdOrName(id) && (spec.provider === "nvidia" || spec.provider === "deepseek"),
streamMarkupHealingPattern: id ? detectStreamMarkupHealingPattern(spec.provider, id) : undefined,
reasoningDeltasMayBeCumulative:
MINIMAX_PROVIDER_OR_ID_PATTERN.test(spec.provider) || (id ? MINIMAX_PROVIDER_OR_ID_PATTERN.test(id) : false),
emptyLengthFinishIsContextError: spec.provider === "ollama",
usesOpenAIToolCallIdLimit: spec.provider === "openai",
promptCacheSessionHeader: spec.provider === "xai-oauth" ? "x-grok-conv-id" : undefined,
};
applyCompatOverrides(compat, spec.compat);
if (spec.compat?.reasoningDisableMode === undefined) {
compat.reasoningDisableMode = resolveReasoningDisableMode(compat.thinkingFormat);
}
if (spec.compat?.omitReasoningEffort === undefined && !compat.supportsReasoningEffort) {
compat.omitReasoningEffort = true;
}
return compat;
}
type ResponsesOnlyCompat = Omit<ResolvedOpenAIResponsesCompat, keyof ResolvedOpenAISharedCompat>;
function pickResponsesOnly(compat: ResolvedOpenAIResponsesCompat): ResponsesOnlyCompat {
return {
supportsLongPromptCacheRetention: compat.supportsLongPromptCacheRetention,
strictResponsesPairing: compat.strictResponsesPairing,
requiresJuiceZeroHack: compat.requiresJuiceZeroHack,
supportsObfuscationOptOut: compat.supportsObfuscationOptOut,
} satisfies ResponsesOnlyCompat;
}
export function buildOpenRouterCompat(spec: ModelSpec<"openrouter">): ResolvedOpenRouterCompat {
const chat = buildOpenAICompat({
...spec,
api: "openai-completions",
} as ModelSpec<"openai-completions">);
const responses = buildOpenAIResponsesCompat(spec);
return { ...chat, ...pickResponsesOnly(responses) } as ResolvedOpenRouterCompat;
}
+9 -5
View File
@@ -266,11 +266,15 @@ function sameEffortList(left: readonly Effort[], right: readonly Effort[]): bool
return true;
}
function isOpenAICompatReasoningApi(api: Api): boolean {
return api === "openai-completions" || api === "openrouter";
}
function getModelDefinedEfforts<TApi extends Api>(spec: ModelSpec<TApi>): readonly Effort[] | undefined {
if (spec.api === "openai-completions" && isZaiGlm52ReasoningEffortModel(spec)) {
if (isOpenAICompatReasoningApi(spec.api) && isZaiGlm52ReasoningEffortModel(spec)) {
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
}
return spec.api === "openai-completions" && (isMinimaxM2FamilyModelId(spec.id) || isOpenAIGptOssModelId(spec.id))
return isOpenAICompatReasoningApi(spec.api) && (isMinimaxM2FamilyModelId(spec.id) || isOpenAIGptOssModelId(spec.id))
? LOW_MEDIUM_HIGH_REASONING_EFFORTS
: undefined;
}
@@ -298,7 +302,7 @@ function inferDetectedEffortMap<TApi extends Api>(
? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER
: ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER;
}
if (spec.api !== "openai-completions") {
if (!isOpenAICompatReasoningApi(spec.api)) {
return undefined;
}
if (spec.provider === "groq" && spec.id === "qwen/qwen3-32b") {
@@ -437,7 +441,7 @@ function inferFallbackEfforts<TApi extends Api>(spec: ModelSpec<TApi>, compat: C
if (spec.api === "bedrock-converse-stream") {
return DEFAULT_REASONING_EFFORTS;
}
if (spec.api === "openai-completions") {
if (isOpenAICompatReasoningApi(spec.api)) {
const resolved = compat as ResolvedOpenAICompat;
if (resolved.thinkingFormat === "openai" && resolved.supportsReasoningEffort) {
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
@@ -503,7 +507,7 @@ function isOpenRouterAnthropicAdaptiveReasoningModel<TApi extends Api>(
parsedModel: AnthropicModel,
spec: ModelSpec<TApi>,
): boolean {
if (spec.api !== "openai-completions") return false;
if (!isOpenAICompatReasoningApi(spec.api)) return false;
if (!modelMatchesHost(spec, "openrouter")) return false;
return isFableOrMythos(parsedModel.kind) || (parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.6"));
}
File diff suppressed because it is too large Load Diff
@@ -847,7 +847,7 @@ interface XAICuratedModel {
* Whether xAI accepts the `reasoning.effort` wire param for this model.
* Default true. When false: picker hides the effort dial (via
* getSupportedEfforts in model-thinking.ts) AND wire-side already omits
* the param via GROK_EFFORT_CAPABLE_PREFIXES in providers/xai-responses.ts.
* the param via GROK_EFFORT_CAPABLE_PREFIXES in pi-ai's stream.ts.
* Must agree with that allowlist; two truths kept in sync by curated-catalog
* author convention until a follow-up Op: compress unifies them.
*/
@@ -867,11 +867,9 @@ interface XAICuratedModel {
// coding-fine-tuned chat model; 512K context per user spec (2026-05-17).
//
// supportsReasoningEffort=false entries reason natively but reject the wire
// `reasoning.effort` param (api.x.ai returns HTTP 400). Mirrors the HTTP-side
// GROK_EFFORT_CAPABLE_PREFIXES allowlist in providers/xai-responses.ts. The
// curated flag is the picker-visible truth; the HTTP allowlist is the wire
// truth. omitReasoningEffort in xai-responses.ts already prevents 400s; this
// fixes the picker UX wart of advertising an inert dial.
// `reasoning.effort` param (api.x.ai returns HTTP 400). The corresponding
// omit/include/history replay defaults live in catalog compat so every
// OpenAI-family endpoint consumes the same constraint.
export const XAI_OAUTH_CURATED_MODELS: readonly XAICuratedModel[] = [
// grok-build is text-only per the bundled catalog; omit `input` for the default.
{ id: "grok-build", contextWindow: 512_000, name: "Grok Build", supportsReasoningEffort: false },
@@ -894,8 +892,7 @@ export const XAI_OAUTH_CURATED_MODELS: readonly XAICuratedModel[] = [
},
// Cursor's "Composer 2.5 Fast" exposed via SuperGrok: non-reasoning,
// text-only, 200K context (mirrors Cursor's composer-* catalog entries).
// Off the GROK_EFFORT_CAPABLE_PREFIXES allowlist, so the wire side already
// sets omitReasoningEffort=true; reasoning:false also hides the effort dial.
// Off the Grok effort-capable allowlist; reasoning:false also hides the effort dial.
{
id: "grok-composer-2.5-fast",
contextWindow: 200_000,
@@ -910,12 +907,31 @@ export const XAI_OAUTH_CURATED_MODELS: readonly XAICuratedModel[] = [
// strings; the chat picker MUST exclude these prefixes or selecting them 400s.
const XAI_NON_CHAT_PREFIXES = ["grok-imagine-", "grok-stt-", "grok-voice-"] as const;
const GROK_EFFORT_CAPABLE_PREFIXES = ["grok-3-mini", "grok-4.20-multi-agent", "grok-4.3"] as const;
function grokSupportsReasoningEffort(modelId: string): boolean {
const name = modelId.trim().toLowerCase();
if (!name) return false;
const bare = name.includes("/") ? name.slice(name.lastIndexOf("/") + 1) : name;
return GROK_EFFORT_CAPABLE_PREFIXES.some(prefix => bare.startsWith(prefix));
}
function withXaiOAuthCompatDefaults(model: ModelSpec<"openai-responses">): ModelSpec<"openai-responses"> {
const compat = {
...(model.compat ?? {}),
includeEncryptedReasoning: model.compat?.includeEncryptedReasoning ?? false,
filterReasoningHistory: model.compat?.filterReasoningHistory ?? true,
omitReasoningEffort: model.compat?.omitReasoningEffort ?? !grokSupportsReasoningEffort(model.id),
};
return { ...model, compat };
}
// Hermes-agent parity: only the `minimal -> low` clamp is applied (see
// hermes-agent/agent/transports/codex.py:92 `_effort_clamp = {"minimal":
// "low"}`). Hermes sends `xhigh` to xAI verbatim and we match that contract
// — let xAI decide if the level is valid for the specific Grok model.
// `resolveModelThinking` folds this into `model.thinking.effortMap`, downstream
// of the omitReasoningEffort gate in xai-responses.ts.
// of the omitReasoningEffort gate in pi-ai's stream.ts.
const XAI_REASONING_EFFORT_MAP = { minimal: "low" } as const;
// xai-oauth's /v1/models exposes no per-request output limit on the OAuth
@@ -942,6 +958,9 @@ function mergeCuratedIntoModel(
const compat = {
...(base.compat ?? {}),
reasoningEffortMap: { ...XAI_REASONING_EFFORT_MAP, ...(base.compat?.reasoningEffortMap ?? {}) },
includeEncryptedReasoning: base.compat?.includeEncryptedReasoning ?? false,
filterReasoningHistory: base.compat?.filterReasoningHistory ?? true,
omitReasoningEffort: base.compat?.omitReasoningEffort ?? !grokSupportsReasoningEffort(base.id),
...(effort === undefined ? {} : { supportsReasoningEffort: effort }),
};
return {
@@ -1005,7 +1024,7 @@ function applyXAIOAuthCuration(dynamic: readonly ModelSpec<"openai-responses">[]
const curatedFirst = XAI_OAUTH_CURATED_MODELS.map(c => byId.get(c.id)).filter(
(e): e is ModelSpec<"openai-responses"> => e !== undefined,
);
const rest = filtered.filter(e => !curatedIds.has(e.id));
const rest = filtered.filter(e => !curatedIds.has(e.id)).map(withXaiOAuthCompatDefaults);
return [...curatedFirst, ...rest];
}
+163 -48
View File
@@ -5,6 +5,7 @@ export type { KnownProvider } from "./provider-models/descriptors";
export type KnownApi =
| "openai-completions"
| "openai-responses"
| "openrouter"
| "openai-codex-responses"
| "azure-openai-responses"
| "anthropic-messages"
@@ -142,6 +143,19 @@ export interface Usage {
};
}
export type OpenAIReasoningFormat = "openai" | "openrouter" | "zai" | "qwen" | "qwen-chat-template";
export type OpenAIReasoningDisableMode =
| "omit"
| "lowest-effort"
| "openrouter-enabled-false"
| "zai-thinking-disabled"
| "qwen-enable-thinking-false"
| "qwen-template-false"
| "juice-zero-developer-message";
export type OpenAIStreamMarkupHealingPattern = "kimi" | "dsml" | "thinking";
/**
* Compatibility settings for openai-completions API.
* Use this to override URL-based auto-detection for custom providers.
@@ -188,13 +202,23 @@ export interface OpenAICompat {
/** Whether tool call IDs must be normalized to Mistral format (exactly 9 alphanumeric chars). Default: auto-detected from URL. */
requiresMistralToolIds?: boolean;
/** Format for reasoning/thinking parameter. "openai" uses reasoning_effort, "openrouter" uses reasoning: { effort }, "zai" uses thinking: { type: "enabled" | "disabled" } (also used by Moonshot Kimi), "qwen" uses top-level enable_thinking, and "qwen-chat-template" uses chat_template_kwargs.enable_thinking. Default: "openai". */
thinkingFormat?: "openai" | "openrouter" | "zai" | "qwen" | "qwen-chat-template";
thinkingFormat?: OpenAIReasoningFormat;
/** Request-time disable encoding for the selected reasoning/thinking format. Default: derived from `thinkingFormat`. */
reasoningDisableMode?: OpenAIReasoningDisableMode;
/** Whether the provider rejects `reasoning.effort`/`reasoning_effort` even when the model reasons natively. Default: false unless reasoning effort is unsupported. */
omitReasoningEffort?: boolean;
/** Whether Responses requests should ask for encrypted reasoning replay items. Default: true. */
includeEncryptedReasoning?: boolean;
/** Whether replayed Responses history should strip native `type: "reasoning"` items before request encoding. Default: false. */
filterReasoningHistory?: boolean;
/** Optional `thinking.keep` value for Z.ai/Moonshot-style thinking params. Set false to suppress auto-detected keep. Default: auto-detected. */
thinkingKeep?: "all" | false;
/** Which reasoning content field to emit on assistant messages. Default: auto-detected. */
reasoningContentField?: "reasoning_content" | "reasoning" | "reasoning_text";
/** Whether assistant tool-call messages must include reasoning content. Default: false. */
requiresReasoningContentForToolCalls?: boolean;
/** Whether all assistant messages must include reasoning content. Default: false. */
requiresReasoningContentForAllAssistantTurns?: boolean;
/** Whether the provider accepts a synthetic placeholder (e.g. ".") for missing reasoning_content on tool-call turns. Default: true. Set to false for providers like DeepSeek that validate the exact reasoning_content value. */
allowsSyntheticReasoningContentForToolCalls?: boolean;
/** Whether assistant tool-call messages must include non-empty content. Default: false. */
@@ -228,6 +252,8 @@ export interface OpenAICompat {
vercelGatewayRouting?: VercelGatewayRouting;
/** Extra fields to include in request body (e.g. gateway routing hints for OpenClaw-style proxies). */
extraBody?: Record<string, unknown>;
/** Request-session header that should mirror the normalized prompt-cache key. Default: unset. */
promptCacheSessionHeader?: "x-grok-conv-id";
/** Whether chat-completions payloads should include provider-specific prompt-cache markers. */
cacheControlFormat?: "anthropic" | undefined;
/** Whether the provider supports the `strict` field in tool definitions. Default: auto-detected per provider/baseUrl (conservative for unknown providers). */
@@ -255,6 +281,16 @@ export interface OpenAICompat {
* Default: auto-detected (GPT-5-family model names).
*/
requiresJuiceZeroHack?: boolean;
/** Whether streamed reasoning deltas for the same field may repeat the full cumulative text snapshot. Default: false. */
reasoningDeltasMayBeCumulative?: boolean;
/** Strip leaked DeepSeek chat-template special tokens from visible content deltas. Default: auto-detected. */
stripDeepseekSpecialTokens?: boolean;
/** Heal leaked chat-template/tool-call/thinking markup from visible content deltas. Default: auto-detected. */
streamMarkupHealingPattern?: OpenAIStreamMarkupHealingPattern;
/** Treat an empty length-finished stream as a context-window error. Default: auto-detected. */
emptyLengthFinishIsContextError?: boolean;
/** Normalize tool call ids to OpenAI's 40-character limit. Default: auto-detected. */
usesOpenAIToolCallIdLimit?: boolean;
/**
* Compat deltas applied when a request actually engages thinking mode
* (reasoning requested and not disabled, model reasoning-capable, and not
@@ -361,59 +397,135 @@ export interface VercelGatewayRouting {
type ResolvedToolStrictMode = NonNullable<OpenAICompat["toolStrictMode"]> | "mixed";
/**
* Fields whose meaning is identical across chat-completions and Responses surfaces.
* Each builder still computes its own per-surface value when defaults diverge.
*/
export interface ResolvedOpenAISharedCompat {
supportsDeveloperRole: boolean;
supportsStrictMode: boolean;
supportsReasoningEffort: boolean;
reasoningEffortMap: Partial<Record<Effort, string>>;
supportsReasoningParams: boolean;
thinkingFormat: OpenAIReasoningFormat;
reasoningDisableMode: OpenAIReasoningDisableMode;
omitReasoningEffort: boolean;
includeEncryptedReasoning: boolean;
filterReasoningHistory: boolean;
disableReasoningOnForcedToolChoice: boolean;
disableReasoningOnToolChoice: boolean;
supportsToolChoice: boolean;
supportsForcedToolChoice: boolean;
reasoningContentField?: OpenAICompat["reasoningContentField"];
requiresReasoningContentForToolCalls: boolean;
requiresReasoningContentForAllAssistantTurns: boolean;
allowsSyntheticReasoningContentForToolCalls: boolean;
requiresThinkingAsText: boolean;
requiresMistralToolIds: boolean;
requiresToolResultName: boolean;
requiresAssistantAfterToolResult: boolean;
requiresAssistantContentForToolCalls: boolean;
stripDeepseekSpecialTokens: boolean;
streamMarkupHealingPattern?: OpenAIStreamMarkupHealingPattern;
reasoningDeltasMayBeCumulative: boolean;
emptyLengthFinishIsContextError: boolean;
usesOpenAIToolCallIdLimit: boolean;
promptCacheSessionHeader?: OpenAICompat["promptCacheSessionHeader"];
/** The model sits behind OpenRouter (routing prefs and max-token omission apply). */
isOpenRouterHost: boolean;
/** Whether this endpoint needs a max-token field even when caller did not set one. */
alwaysSendMaxTokens: boolean;
/** See {@link OpenAICompat.enableGeminiThinkingLoopGuard}. Set by the builder from the family classifier. */
enableGeminiThinkingLoopGuard?: boolean;
openRouterRouting?: OpenAICompat["openRouterRouting"];
/** Provider-specific wire model-id transform applied to the base id. */
wireModelIdMode: "raw" | "firepass" | "fireworks" | "openrouter";
}
/**
* Fully-resolved chat-completions compat view: every detected default
* materialized and user overrides applied. Built once per model by
* `buildModel`; request handlers read fields and never detect, resolve, or
* allocate.
*/
export type ResolvedOpenAICompat = Required<
Omit<
OpenAICompat,
| "openRouterRouting"
| "vercelGatewayRouting"
| "extraBody"
| "toolStrictMode"
| "streamIdleTimeoutMs"
| "supportsLongPromptCacheRetention"
| "cacheControlFormat"
| "thinkingKeep"
| "strictResponsesPairing"
| "requiresJuiceZeroHack"
| "enableGeminiThinkingLoopGuard"
| "whenThinking"
>
> & {
openRouterRouting?: OpenAICompat["openRouterRouting"];
vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"];
extraBody?: OpenAICompat["extraBody"];
cacheControlFormat?: OpenAICompat["cacheControlFormat"];
thinkingKeep?: OpenAICompat["thinkingKeep"];
streamIdleTimeoutMs?: number;
toolStrictMode: ResolvedToolStrictMode;
/** The model sits behind OpenRouter (routing prefs and max-token omission apply). */
isOpenRouterHost: boolean;
/** The model sits behind Vercel AI Gateway. */
isVercelGatewayHost: boolean;
/** See {@link OpenAICompat.enableGeminiThinkingLoopGuard}. Set by the builder from the family classifier. */
enableGeminiThinkingLoopGuard?: boolean;
/** Complete alternate view for thinking-engaged requests; swap pointers, never spread. */
whenThinking?: ResolvedOpenAICompat;
};
export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
Required<
Omit<
OpenAICompat,
| "supportsDeveloperRole"
| "supportsReasoningEffort"
| "reasoningEffortMap"
| "supportsReasoningParams"
| "thinkingFormat"
| "reasoningDisableMode"
| "omitReasoningEffort"
| "includeEncryptedReasoning"
| "filterReasoningHistory"
| "disableReasoningOnForcedToolChoice"
| "disableReasoningOnToolChoice"
| "supportsToolChoice"
| "supportsForcedToolChoice"
| "reasoningContentField"
| "requiresReasoningContentForToolCalls"
| "requiresReasoningContentForAllAssistantTurns"
| "allowsSyntheticReasoningContentForToolCalls"
| "requiresThinkingAsText"
| "requiresMistralToolIds"
| "requiresToolResultName"
| "requiresAssistantAfterToolResult"
| "requiresAssistantContentForToolCalls"
| "stripDeepseekSpecialTokens"
| "streamMarkupHealingPattern"
| "reasoningDeltasMayBeCumulative"
| "emptyLengthFinishIsContextError"
| "usesOpenAIToolCallIdLimit"
| "promptCacheSessionHeader"
| "openRouterRouting"
| "isOpenRouterHost"
| "supportsStrictMode"
| "supportsLongPromptCacheRetention"
| "alwaysSendMaxTokens"
| "wireModelIdMode"
| "vercelGatewayRouting"
| "extraBody"
| "toolStrictMode"
| "streamIdleTimeoutMs"
| "cacheControlFormat"
| "thinkingKeep"
| "strictResponsesPairing"
| "requiresJuiceZeroHack"
| "enableGeminiThinkingLoopGuard"
| "whenThinking"
>
> & {
vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"];
extraBody?: OpenAICompat["extraBody"];
cacheControlFormat?: OpenAICompat["cacheControlFormat"];
thinkingKeep?: OpenAICompat["thinkingKeep"];
streamIdleTimeoutMs?: number;
toolStrictMode: ResolvedToolStrictMode;
/** The model sits behind Vercel AI Gateway. */
isVercelGatewayHost: boolean;
dropThinkingWhenReasoningEffort: boolean;
/** Complete alternate view for thinking-engaged requests; swap pointers, never spread. */
whenThinking?: ResolvedOpenAICompat;
};
/** Fully-resolved Responses-API compat view (same contract as `ResolvedOpenAICompat`). */
export interface ResolvedOpenAIResponsesCompat {
supportsDeveloperRole: boolean;
supportsStrictMode: boolean;
supportsReasoningEffort: boolean;
export interface ResolvedOpenAIResponsesCompat extends ResolvedOpenAISharedCompat {
supportsLongPromptCacheRetention: boolean;
strictResponsesPairing: boolean;
requiresJuiceZeroHack: boolean;
reasoningEffortMap: Partial<Record<Effort, string>>;
/** See {@link OpenAICompat.enableGeminiThinkingLoopGuard}. */
enableGeminiThinkingLoopGuard?: boolean;
supportsObfuscationOptOut: boolean;
}
/**
* OpenRouter is a pseudo API: runtime dispatch can use either Responses
* (default) or Chat Completions (`PI_OPENROUTER_RESPONSES=0`) with the same
* model object, so its resolved compat must satisfy both handlers.
*/
export type ResolvedOpenRouterCompat = ResolvedOpenAICompat & ResolvedOpenAIResponsesCompat;
/** Fully-resolved anthropic-messages compat view (same contract as `ResolvedOpenAICompat`). */
export type ResolvedAnthropicCompat = Required<AnthropicCompat> & {
/**
@@ -428,6 +540,7 @@ export type ResolvedAnthropicCompat = Required<AnthropicCompat> & {
/** Sparse, user-authored compat overrides for a given API (models.json / config vocabulary). */
export type CompatConfigOf<TApi extends Api> = TApi extends
| "openai-completions"
| "openrouter"
| "openai-responses"
| "azure-openai-responses"
| "openai-codex-responses"
@@ -437,13 +550,15 @@ export type CompatConfigOf<TApi extends Api> = TApi extends
: undefined;
/** Resolved compat for a given API: complete record, materialized once by `buildModel`. */
export type CompatOf<TApi extends Api> = TApi extends "openai-completions"
? ResolvedOpenAICompat
: TApi extends "openai-responses" | "azure-openai-responses" | "openai-codex-responses"
? ResolvedOpenAIResponsesCompat
: TApi extends "anthropic-messages"
? ResolvedAnthropicCompat
: undefined;
export type CompatOf<TApi extends Api> = TApi extends "openrouter"
? ResolvedOpenRouterCompat
: TApi extends "openai-completions"
? ResolvedOpenAICompat
: TApi extends "openai-responses" | "azure-openai-responses" | "openai-codex-responses"
? ResolvedOpenAIResponsesCompat
: TApi extends "anthropic-messages"
? ResolvedAnthropicCompat
: undefined;
// Model interface for the unified model system
export interface Model<TApi extends Api = Api> {
+138
View File
@@ -5,7 +5,9 @@ import * as os from "node:os";
import * as path from "node:path";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { isOfficialAnthropicApiUrl } from "@oh-my-pi/pi-catalog/compat/anthropic";
import { buildOpenAICompat, buildOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
import { resolveProviderModels } from "@oh-my-pi/pi-catalog/model-manager";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import type { ModelSpec } from "@oh-my-pi/pi-catalog/types";
function completionsSpec(overrides: Partial<ModelSpec<"openai-completions">> = {}): ModelSpec<"openai-completions"> {
@@ -24,6 +26,22 @@ function completionsSpec(overrides: Partial<ModelSpec<"openai-completions">> = {
};
}
function openrouterSpec(overrides: Partial<ModelSpec<"openrouter">> = {}): ModelSpec<"openrouter"> {
return {
id: "anthropic/claude-sonnet-4",
name: "Claude Sonnet 4",
api: "openrouter",
provider: "openrouter",
baseUrl: "https://openrouter.ai/api/v1",
reasoning: true,
input: ["text", "image"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 64_000,
...overrides,
};
}
describe("buildModel", () => {
it("resolves a complete compat record for an openai-completions spec with no compat", () => {
const model = buildModel(completionsSpec());
@@ -75,6 +93,29 @@ describe("buildModel", () => {
expect(model.compat.whenThinking).toBeUndefined();
});
it("builds OpenRouter pseudo-API models with shared chat and Responses compat", () => {
const model = buildModel(
openrouterSpec({
compat: { openRouterRouting: { only: ["anthropic"], order: ["anthropic"] } },
}),
);
expect(model.compat).toBeDefined();
expect(model.compat.isOpenRouterHost).toBe(true);
expect(model.compat.thinkingFormat).toBe("openrouter");
expect(model.compat.supportsStrictMode).toBe(true);
expect(model.compat.strictResponsesPairing).toBe(false);
expect(model.compat.openRouterRouting).toEqual({ only: ["anthropic"], order: ["anthropic"] });
});
it("loads bundled OpenRouter models with resolved compat", () => {
const model = getBundledModel<"openrouter">("openrouter", "anthropic/claude-sonnet-4");
expect(model.compat).toBeDefined();
expect(model.compat?.isOpenRouterHost).toBe(true);
expect(model.compat?.supportsStrictMode).toBe(true);
});
it("strips gateway author prefixes and extrinsic tags from display names", () => {
const cases: [string, string][] = [
["Anthropic: Claude Opus 4.6 (Fast) ($$$$)", "Claude Opus 4.6 (Fast)"],
@@ -103,6 +144,103 @@ describe("buildModel", () => {
});
});
describe("openai-completions wire-quirk compat detection", () => {
it("derives wireModelIdMode from provider/host", () => {
expect(buildOpenAICompat(completionsSpec({ provider: "firepass" })).wireModelIdMode).toBe("firepass");
expect(
buildOpenAICompat(completionsSpec({ provider: "fireworks", baseUrl: "https://api.fireworks.ai/inference/v1" }))
.wireModelIdMode,
).toBe("fireworks");
expect(
buildOpenAICompat(completionsSpec({ provider: "openrouter", baseUrl: "https://openrouter.ai/api/v1" }))
.wireModelIdMode,
).toBe("openrouter");
expect(buildOpenAICompat(completionsSpec()).wireModelIdMode).toBe("raw");
});
it("strips DeepSeek special tokens only for deepseek ids on nvidia/deepseek providers", () => {
expect(
buildOpenAICompat(
completionsSpec({
provider: "nvidia",
id: "deepseek-ai/deepseek-v3.1",
baseUrl: "https://integrate.api.nvidia.com/v1",
}),
).stripDeepseekSpecialTokens,
).toBe(true);
expect(
buildOpenAICompat(
completionsSpec({ provider: "deepseek", id: "deepseek-chat", baseUrl: "https://api.deepseek.com/v1" }),
).stripDeepseekSpecialTokens,
).toBe(true);
// DeepSeek id behind another host must NOT strip (only nvidia/deepseek hosts emit the raw tokens).
expect(
buildOpenAICompat(
completionsSpec({
provider: "openrouter",
id: "deepseek/deepseek-v3.1",
baseUrl: "https://openrouter.ai/api/v1",
}),
).stripDeepseekSpecialTokens,
).toBe(false);
// Non-deepseek id on nvidia must NOT strip.
expect(
buildOpenAICompat(
completionsSpec({
provider: "nvidia",
id: "meta/llama-3.1",
baseUrl: "https://integrate.api.nvidia.com/v1",
}),
).stripDeepseekSpecialTokens,
).toBe(false);
});
it("flags cumulative reasoning deltas for MiniMax provider or id", () => {
expect(buildOpenAICompat(completionsSpec({ provider: "minimax" })).reasoningDeltasMayBeCumulative).toBe(true);
expect(buildOpenAICompat(completionsSpec({ id: "MiniMax-M2" })).reasoningDeltasMayBeCumulative).toBe(true);
expect(buildOpenAICompat(completionsSpec()).reasoningDeltasMayBeCumulative).toBe(false);
});
it("maps the remaining provider-keyed wire quirks", () => {
expect(buildOpenAICompat(completionsSpec({ provider: "ollama" })).emptyLengthFinishIsContextError).toBe(true);
expect(buildOpenAICompat(completionsSpec()).emptyLengthFinishIsContextError).toBe(false);
expect(
buildOpenAICompat(completionsSpec({ provider: "openai", baseUrl: "https://api.openai.com/v1" }))
.usesOpenAIToolCallIdLimit,
).toBe(true);
expect(buildOpenAICompat(completionsSpec()).usesOpenAIToolCallIdLimit).toBe(false);
expect(
buildOpenAICompat(completionsSpec({ provider: "fireworks", baseUrl: "https://api.fireworks.ai/inference/v1" }))
.dropThinkingWhenReasoningEffort,
).toBe(true);
expect(buildOpenAICompat(completionsSpec()).dropThinkingWhenReasoningEffort).toBe(false);
});
it("derives Responses obfuscation opt-out and wire mode per surface", () => {
expect(
buildOpenAIResponsesCompat({
id: "gpt-5",
provider: "openai",
name: "GPT 5",
baseUrl: "https://api.openai.com/v1",
}).supportsObfuscationOptOut,
).toBe(true);
// Azure mirrors the schema but is NOT the OpenAI host: no obfuscation opt-out.
expect(
buildOpenAIResponsesCompat({ id: "gpt-5", provider: "azure", name: "gpt-5", baseUrl: "" })
.supportsObfuscationOptOut,
).toBe(false);
const openrouterResponses = buildOpenAIResponsesCompat({
id: "anthropic/claude-sonnet-4",
provider: "openrouter",
name: "Claude Sonnet 4",
baseUrl: "https://openrouter.ai/api/v1",
});
expect(openrouterResponses.supportsObfuscationOptOut).toBe(false);
expect(openrouterResponses.wireModelIdMode).toBe("openrouter");
});
});
describe("model cache spec round trip", () => {
it("persists sparse specs and rebuilds resolved models on cache reads", async () => {
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-model-cache-"));
+11 -1
View File
@@ -12,6 +12,14 @@ import type { FetchImpl, Model } from "@oh-my-pi/pi-catalog/types";
const TP_KEY = "tp-ci1p8t1w4e1sbxgyc8v65tnrjbzro287igmvyf25van9mt76";
const SGP_BASE_URL = "https://token-plan-sgp.xiaomimimo.com/v1";
interface MessageWithReasoningContent {
reasoning_content?: unknown;
}
function isMessageWithReasoningContent(value: unknown): value is MessageWithReasoningContent {
return value !== null && typeof value === "object";
}
afterEach(() => {
vi.restoreAllMocks();
});
@@ -129,6 +137,8 @@ describe("issue #1846: Xiaomi Token Plan provider support", () => {
expect(compat.thinkingFormat).toBe("zai");
expect(compat.requiresReasoningContentForToolCalls).toBe(true);
expect(compat.allowsSyntheticReasoningContentForToolCalls).toBe(false);
expect(Reflect.get(assistant ?? {}, "reasoning_content")).toBe("I need to inspect the file before answering.");
expect(isMessageWithReasoningContent(assistant) ? assistant.reasoning_content : undefined).toBe(
"I need to inspect the file before answering.",
);
});
});
+9 -1
View File
@@ -1,23 +1,31 @@
# Changelog
## [Unreleased]
### Added
- Added support for OpenRouter fallback in Perplexity web search when direct Perplexity API keys fail or are unavailable
- Added support for streaming the Perplexity Responses API (`/v1/responses`) via the `PI_PERPLEXITY_RESPONSES=1` environment variable
- Added `omp ttsr` top-level CLI command to inspect and test Time-Traveling Stream Rules
- Added `omp ttsr list` to enumerate all project/user-loaded TTSR rules with their conditions, scope, and source metadata
- Added `omp ttsr test` to run snippets through the real TTSR matching pipeline with inline text, `--file <path>`, or stdin via `--file -`
- Added `--json` output to `omp ttsr test` and `omp ttsr list` for machine-readable reporting
- Added `--rule`, `--source`, `--tool`, `--path`, and `--verbose` options to `omp ttsr test` to control matching context and inspection details
- Added `omp ttsr` subcommand for inspecting and testing Time-Traveling Stream Rules: `omp ttsr list` shows every TTSR-registered rule the current project/user config would load, and `omp ttsr test` feeds a snippet (inline, `--file`, or stdin) through the real TTSR matching pipeline (`TtsrManager.checkSnapshot`/`checkAstSnapshot`) and reports which rules would trigger. A positional that resolves to a file defaults to tool/edit context; `--source`, `--tool`, and `--path` override the inferred match context so glob/AST/scope-scoped rules evaluate the same way they do in a live session. `--rule` tests a single rule markdown file in isolation.
- Added support for reading embedded PDF images via `read <pdf>:<image>.png` and listing available image members with `read <pdf>:`
- Added a built-in `ts-no-inline-cast-access` TTSR rule that interrupts inline object-type assertions read immediately as a property (`(x as { y: T }).y`, including `?.` and bracket access), steering toward schema validation, `in`/`typeof` narrowing, or a validated named type
### Changed
- Changed advisor model calls and overflow-compaction tasks to inherit and propagate primary telemetry spans, usage, and cost tracking
- Changed PDF read output to replace `<!-- image: ... -->` placeholders with clickable `read <pdf>:<image>.png` handles, including line-range and multi-range reads
- Changed the built-in `ts-no-any` rule to recommend a schema parse at trust boundaries and `in`-narrowing (instead of an inline `as`-cast) when reading fields off `unknown`
### Fixed
- Fixed Perplexity web search to use shared OpenAI streaming transports while preserving streamed sources, citations, and related questions
- Fixed `read <pdf>:<member>` errors for unknown PDF images to surface available extracted image names
- Fixed puppeteer stealth scripts to use cached Reflect methods (`Reflect_get`, `Reflect_apply`) and `Reflect.apply` instead of live `Reflect`/`Function.prototype.apply` calls, preventing page tampering from leaking through proxy traps.
- Fixed Perplexity API-key web search to use shared OpenAI streaming transports while preserving streamed sources and OpenRouter fallback.
### Security
+1
View File
@@ -67,6 +67,7 @@
"@puppeteer/browsers": "catalog:",
"@types/turndown": "catalog:",
"@xterm/headless": "catalog:",
"arktype": "catalog:",
"chalk": "catalog:",
"diff": "catalog:",
"fflate": "catalog:",
@@ -1,5 +1,5 @@
import { describe, expect, it, vi } from "bun:test";
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
import type { AgentMessage, AgentTelemetryConfig } from "@oh-my-pi/pi-agent-core";
import { createAdvisorMessageCard } from "../../modes/components/advisor-message";
import { getThemeByName } from "../../modes/theme/theme";
import { formatSessionHistoryMarkdown } from "../../session/session-history-format";
@@ -11,6 +11,7 @@ import {
type AdvisorNote,
AdvisorRuntime,
type AdvisorRuntimeHost,
deriveAdvisorTelemetry,
formatAdvisorBatchContent,
isAdvisorInterruptImmuneTurnActive,
isInterruptingSeverity,
@@ -181,6 +182,37 @@ describe("advisor", () => {
});
});
describe("deriveAdvisorTelemetry", () => {
it("returns undefined when the primary has no telemetry so the advisor stays a no-op", () => {
expect(deriveAdvisorTelemetry(undefined, { id: "s-advisor", name: "Advisor" })).toBeUndefined();
});
it("inherits the primary's usage/cost hooks but restamps identity and clears the conversation", () => {
const onChatUsage = vi.fn();
const costEstimator = vi.fn();
const primary: AgentTelemetryConfig = {
agent: { id: "main", name: "Main" },
conversationId: "session-1",
attributes: { "deployment.id": "prod" },
onChatUsage,
costEstimator,
};
const identity = { id: "session-1-advisor", name: "Advisor", description: "anthropic/claude-sonnet-4-5" };
const derived = deriveAdvisorTelemetry(primary, identity);
// Usage/cost hooks are inherited so the advisor model's calls report through
// the same pipeline as the primary — the whole point of the fix.
expect(derived?.onChatUsage).toBe(onChatUsage);
expect(derived?.costEstimator).toBe(costEstimator);
expect(derived?.attributes).toEqual({ "deployment.id": "prod" });
// Advisor identity replaces the primary's so spans are attributable to the advisor.
expect(derived?.agent).toEqual(identity);
// Conversation cleared so the advisor loop falls back to its own `-advisor` session id.
expect(derived?.conversationId).toBeUndefined();
});
});
describe("AdvisorRuntime", () => {
function makeAgent(promptInputs: string[]): AdvisorAgent {
return {
@@ -1,4 +1,11 @@
import type { AgentTool, AgentToolContext, AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core";
import type {
AgentIdentity,
AgentTelemetryConfig,
AgentTool,
AgentToolContext,
AgentToolResult,
AgentToolUpdateCallback,
} from "@oh-my-pi/pi-agent-core";
import { escapeXmlText } from "@oh-my-pi/pi-utils";
import { z } from "zod/v4";
import adviseDescription from "../prompts/advisor/advise-tool.md" with { type: "text" };
@@ -109,6 +116,25 @@ export function resolveAdvisorDeliveryChannel(opts: {
return "steer";
}
/**
* Derive the advisor loop's telemetry from the primary session's config so the
* advisor model's GenAI spans and usage/cost hooks (onChatUsage, onCostDelta,
* costEstimator) fire under the same pipeline as every other model call —
* stamped with the advisor's own agent identity. `conversationId` is cleared so
* the advisor loop falls back to its own `-advisor` session id for
* `gen_ai.conversation.id` instead of inheriting the primary's conversation.
*
* Returns undefined when the primary has no telemetry (instrumentation off), so
* the advisor `Agent` stays a zero-overhead no-op as well.
*/
export function deriveAdvisorTelemetry(
primaryTelemetry: AgentTelemetryConfig | undefined,
identity: AgentIdentity,
): AgentTelemetryConfig | undefined {
if (!primaryTelemetry) return undefined;
return { ...primaryTelemetry, agent: identity, conversationId: undefined };
}
/**
* Side-effect-free investigation tools handed to the advisor agent so it can
* inspect the workspace before weighing in. Names match the primary session's
@@ -126,6 +126,7 @@ import {
type AdvisorNote,
AdvisorRuntime,
type AdvisorSeverity,
deriveAdvisorTelemetry,
formatAdvisorBatchContent,
isAdvisorInterruptImmuneTurnActive,
isInterruptingSeverity,
@@ -150,7 +151,7 @@ import {
resolveModelRoleValue,
resolveRoleSelection,
} from "../config/model-resolver";
import { MODEL_ROLE_IDS } from "../config/model-roles";
import { MODEL_ROLE_IDS, MODEL_ROLES } from "../config/model-roles";
import { expandPromptTemplate, type PromptTemplate } from "../config/prompt-templates";
import type { Settings, SkillsSettings } from "../config/settings";
import { onAppendOnlyModeChanged } from "../config/settings";
@@ -1755,6 +1756,15 @@ export class AgentSession {
systemPrompt.push(this.#advisorWatchdogPrompt);
}
const advisorSessionId = this.sessionId ? `${this.sessionId}-advisor` : undefined;
// Thread the primary's telemetry into the advisor loop so the advisor
// model's GenAI spans + usage/cost hooks fire like every other model call,
// stamped with the advisor's own identity (see deriveAdvisorTelemetry).
const advisorTelemetry = deriveAdvisorTelemetry(this.agent.telemetry, {
id: advisorSessionId,
name: MODEL_ROLES.advisor.name,
description: formatModelString(advisorSel.model),
});
const advisorAgent = new Agent({
initialState: {
systemPrompt,
@@ -1766,6 +1776,7 @@ export class AgentSession {
sessionId: advisorSessionId,
getApiKey: requestModel => this.#modelRegistry.resolver(requestModel, advisorSessionId),
intentTracing: false,
telemetry: advisorTelemetry,
});
advisorAgent.setDisableReasoning(shouldDisableReasoning(advisorThinkingLevel));
@@ -1941,24 +1952,26 @@ export class AgentSession {
let compactResult: CompactionResult | undefined;
let lastError: unknown;
const advisorSessionId = this.sessionId ? `${this.sessionId}-advisor` : undefined;
// Instrument the advisor's overflow-compaction one-shot like the primary
// compaction path so the advisor model's maintenance call also emits spans.
const telemetry = resolveTelemetry(advisor.telemetry, advisorSessionId);
for (const candidate of candidates) {
const apiKey = await this.#modelRegistry.getApiKey(
candidate,
this.sessionId ? `${this.sessionId}-advisor` : undefined,
);
const apiKey = await this.#modelRegistry.getApiKey(candidate, advisorSessionId);
if (!apiKey) continue;
try {
compactResult = await compact(
preparation,
candidate,
this.#modelRegistry.resolver(candidate, this.sessionId ? `${this.sessionId}-advisor` : undefined),
this.#modelRegistry.resolver(candidate, advisorSessionId),
undefined,
undefined,
{
thinkingLevel: advisorCompactionThinkingLevel,
convertToLlm: messages => this.#convertToLlmForSideRequest(messages),
telemetry,
},
);
break;
+10 -5
View File
@@ -1,6 +1,14 @@
import * as os from "node:os";
import * as path from "node:path";
import { type ApiKey, type FetchImpl, getEnvApiKey, type Model, ProviderHttpError, withAuth } from "@oh-my-pi/pi-ai";
import {
type ApiKey,
type FetchImpl,
getEnvApiKey,
getOpenRouterHeaders,
type Model,
ProviderHttpError,
withAuth,
} from "@oh-my-pi/pi-ai";
import {
CODEX_BASE_URL,
getCodexAccountId,
@@ -20,7 +28,6 @@ import {
untilAborted,
} from "@oh-my-pi/pi-utils";
import { z } from "zod/v4";
import packageJson from "../../package.json" with { type: "json" };
import { isAuthenticated, type ModelRegistry } from "../config/model-registry";
import { settings } from "../config/settings";
import type { CustomTool } from "../extensibility/custom-tools/types";
@@ -1407,9 +1414,7 @@ export const imageGenTool: CustomTool<typeof imageGenSchema, ImageGenToolDetails
headers: {
"Content-Type": "application/json",
Authorization: `Bearer ${key}`,
"HTTP-Referer": "https://omp.sh/",
"X-OpenRouter-Title": "Oh-My-Pi",
"X-OpenRouter-Categories": "cli-agent",
...getOpenRouterHeaders(),
},
body: JSON.stringify(requestBody),
signal: requestSignal,
@@ -8,12 +8,24 @@
* - Anonymous via `www.perplexity.ai/rest/sse/perplexity_ask`
*/
import { type AuthStorage, type FetchImpl, getEnvApiKey, type OAuthAccess, withOAuthAccess } from "@oh-my-pi/pi-ai";
import {
type AssistantMessage,
type AssistantMessageEventStream,
type AuthStorage,
type Context,
type FetchImpl,
type OAuthAccess,
type Usage,
withOAuthAccess,
} from "@oh-my-pi/pi-ai";
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import type { Model, ModelSpec } from "@oh-my-pi/pi-catalog/types";
import { $env, readSseJson } from "@oh-my-pi/pi-utils";
import type {
PerplexityMessageOutput,
PerplexityRequest,
PerplexityResponse,
PerplexitySearchResult,
SearchCitation,
SearchResponse,
SearchSource,
@@ -24,7 +36,9 @@ import type { SearchParams } from "./base";
import { SearchProvider } from "./base";
import { classifyProviderHttpError, withHardTimeout } from "./utils";
const PERPLEXITY_API_URL = "https://api.perplexity.ai/chat/completions";
const PERPLEXITY_CHAT_BASE_URL = "https://api.perplexity.ai";
const PERPLEXITY_RESPONSES_BASE_URL = "https://api.perplexity.ai/v1";
const OPENROUTER_BASE_URL = "https://openrouter.io/api/v1";
const PERPLEXITY_OAUTH_ASK_URL = "https://www.perplexity.ai/rest/sse/perplexity_ask";
const DEFAULT_MAX_TOKENS = 8192;
@@ -37,10 +51,7 @@ const ANONYMOUS_USER_AGENT =
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36";
type PerplexityAuth =
| {
type: "api_key";
token: string;
}
| ApiConfig
| {
type: "oauth";
access: OAuthAccess;
@@ -278,9 +289,52 @@ export interface PerplexitySearchParams {
fetch?: FetchImpl;
}
/** Find PERPLEXITY_API_KEY from environment or .env files (also checks PPLX_API_KEY) */
export function findApiKey(): string | null {
return getEnvApiKey("perplexity") ?? null;
interface ApiConfig {
type: "api_key";
apiKey: string;
provider: "perplexity" | "openrouter";
chatBaseUrl: string;
responsesBaseUrl: string;
modelPrefix: string;
useResponses: boolean;
}
/** Detect API-key endpoints to try in priority order (Perplexity direct, then OpenRouter). */
async function getApiConfigs(
authStorage: AuthStorage,
sessionId: string | undefined,
signal: AbortSignal | undefined,
): Promise<ApiConfig[]> {
const useResponses = $env.PI_PERPLEXITY_RESPONSES === "1";
const configs: ApiConfig[] = [];
const perplexityKey = await authStorage.getApiKey("perplexity", sessionId, { signal });
if (perplexityKey) {
configs.push({
type: "api_key",
apiKey: perplexityKey,
provider: "perplexity",
chatBaseUrl: PERPLEXITY_CHAT_BASE_URL,
responsesBaseUrl: PERPLEXITY_RESPONSES_BASE_URL,
modelPrefix: "",
useResponses,
});
}
const openrouterKey = await authStorage.getApiKey("openrouter", sessionId, { signal });
if (openrouterKey) {
configs.push({
type: "api_key",
apiKey: openrouterKey,
provider: "openrouter",
chatBaseUrl: OPENROUTER_BASE_URL,
responsesBaseUrl: OPENROUTER_BASE_URL,
modelPrefix: "perplexity/",
useResponses,
});
}
return configs;
}
/**
@@ -302,86 +356,236 @@ function jwtExpiryMs(token: string): number | undefined {
}
}
async function findOAuthAccess(
/** Collect all available auth methods to try in priority order */
async function getAvailableAuthMethods(
authStorage: AuthStorage,
sessionId: string | undefined,
signal: AbortSignal | undefined,
): Promise<OAuthAccess | null> {
): Promise<PerplexityAuth[]> {
const methods: PerplexityAuth[] = [];
// 1. Perplexity OAuth & Cookies (same priority - highest)
try {
// `getOAuthAccess` returns the raw OAuth bearer only — runtime/config
// api_key overrides and stored api_key credentials are intentionally
// suppressed so we don't POST an `api.perplexity.ai` key to the
// `www.perplexity.ai` session/SSE endpoint.
const access = await authStorage.getOAuthAccess("perplexity", sessionId, { signal });
const token = access?.accessToken;
if (!access || !token) return null;
// Trust the JWT's own `exp` claim if it has one; otherwise treat as
// non-expiring. Perplexity session JWTs commonly omit `exp`.
const jwtExpiry = jwtExpiryMs(token);
if (jwtExpiry !== undefined && jwtExpiry <= Date.now() + OAUTH_EXPIRY_BUFFER_MS) return null;
return access;
if (access && token) {
const jwtExpiry = jwtExpiryMs(token);
if (jwtExpiry === undefined || jwtExpiry > Date.now() + OAUTH_EXPIRY_BUFFER_MS) {
methods.push({ type: "oauth", access });
}
}
} catch {
return null;
// ignored
}
}
async function findPerplexityAuth(
authStorage: AuthStorage,
sessionId: string | undefined,
signal: AbortSignal | undefined,
): Promise<PerplexityAuth> {
// 1. PERPLEXITY_COOKIES env var
const cookies = $env.PERPLEXITY_COOKIES?.trim();
if (cookies) {
return { type: "cookies", cookies };
methods.push({ type: "cookies", cookies });
}
const apiKey = findApiKey();
const apiConfigs = await getApiConfigs(authStorage, sessionId, signal);
methods.push(...apiConfigs);
// 2. OAuth/session bearer from AuthStorage.
const oauthAccess = await findOAuthAccess(authStorage, sessionId, signal);
if (oauthAccess) {
return { type: "oauth", access: oauthAccess };
// 5. Fallback to Perplexity free (anonymous)
if (methods.length === 0) {
methods.push({ type: "anonymous" });
}
// 3. PERPLEXITY_API_KEY env var
if (apiKey) {
return { type: "api_key", token: apiKey };
}
// 4. The consumer ask endpoint currently accepts unauthenticated browser-style requests.
return { type: "anonymous" };
return methods;
}
/** Call Perplexity API-key endpoint. */
interface PerplexityApiStreamMetadata {
id?: string;
model?: string;
citations?: unknown;
search_results?: unknown;
related_questions?: unknown;
}
function buildPerplexityCompletionsModel(config: ApiConfig, request: PerplexityRequest): Model<"openai-completions"> {
const model = config.modelPrefix ? `${config.modelPrefix}${request.model}` : request.model;
const spec: ModelSpec<"openai-completions"> = {
id: model,
name: model,
api: "openai-completions",
provider: config.provider,
baseUrl: config.chatBaseUrl,
reasoning: false,
input: ["text"],
supportsTools: false,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: null,
maxTokens: null,
compat: {
supportsStore: false,
supportsMultipleSystemMessages: true,
supportsReasoningParams: false,
supportsUsageInStreaming: true,
maxTokensField: "max_tokens",
},
};
return buildModel(spec);
}
function buildPerplexityResponsesModel(config: ApiConfig, request: PerplexityRequest): Model<"openai-responses"> {
const model = config.modelPrefix ? `${config.modelPrefix}${request.model}` : request.model;
const spec: ModelSpec<"openai-responses"> = {
id: model,
name: model,
api: "openai-responses",
provider: config.provider,
baseUrl: config.responsesBaseUrl,
reasoning: false,
input: ["text"],
supportsTools: false,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: null,
maxTokens: null,
compat: {
alwaysSendMaxTokens: true,
supportsReasoningParams: false,
},
};
return buildModel(spec);
}
function buildPerplexityContext(request: PerplexityRequest): Context {
const systemPrompt: string[] = [];
const messages: Context["messages"] = [];
for (const message of request.messages) {
if (typeof message.content !== "string" || message.content.length === 0) continue;
if (message.role === "system") {
systemPrompt.push(message.content);
continue;
}
if (message.role === "user") {
messages.push({ role: "user", content: message.content, timestamp: 0 });
}
}
return { systemPrompt: systemPrompt.length > 0 ? systemPrompt : undefined, messages };
}
function buildPerplexityExtraBody(request: PerplexityRequest): Record<string, unknown> {
return {
search_mode: request.search_mode,
num_search_results: request.num_search_results,
web_search_options: request.web_search_options,
enable_search_classifier: request.enable_search_classifier,
reasoning_effort: request.reasoning_effort,
language_preference: request.language_preference,
return_related_questions: request.return_related_questions,
search_recency_filter: request.search_recency_filter,
};
}
function applyPerplexityExtraBody(payload: unknown, request: PerplexityRequest): void {
const record = asRecord(payload);
if (!record) return;
Object.assign(record, buildPerplexityExtraBody(request));
}
function collectPerplexityOutputMetadata(metadata: PerplexityApiStreamMetadata, output: unknown): void {
if (!Array.isArray(output)) return;
for (const item of output) {
const record = asRecord(item);
if (!record) continue;
if (Array.isArray(record.search_results)) metadata.search_results = record.search_results;
if (Array.isArray(record.results)) metadata.search_results = record.results;
if (Array.isArray(record.citations)) metadata.citations = record.citations;
if (Array.isArray(record.related_questions)) metadata.related_questions = record.related_questions;
collectPerplexityOutputMetadata(metadata, record.content);
}
}
function collectPerplexityMetadataFromRecord(
metadata: PerplexityApiStreamMetadata,
record: Record<string, unknown>,
): void {
const id = record.id;
if (typeof id === "string" && id.length > 0) metadata.id = id;
const model = record.model;
if (typeof model === "string" && model.length > 0) metadata.model = model;
if (Array.isArray(record.citations)) metadata.citations = record.citations;
if (Array.isArray(record.search_results)) metadata.search_results = record.search_results;
if (Array.isArray(record.related_questions)) metadata.related_questions = record.related_questions;
if (Array.isArray(record.results)) metadata.search_results = record.results;
collectPerplexityOutputMetadata(metadata, record.output);
const response = asRecord(record.response);
if (response) {
collectPerplexityOutputMetadata(metadata, response.output);
collectPerplexityMetadataFromRecord(metadata, response);
}
}
function collectPerplexityMetadata(metadata: PerplexityApiStreamMetadata, data: string): void {
if (data === "[DONE]") return;
const record = asRecord(parseJson(data));
if (record) collectPerplexityMetadataFromRecord(metadata, record);
}
async function drainAssistantStream(stream: AssistantMessageEventStream): Promise<AssistantMessage> {
let finalMessage: AssistantMessage | undefined;
for await (const event of stream) {
if (event.type === "done") {
finalMessage = event.message;
} else if (event.type === "error") {
finalMessage = event.error;
}
}
return finalMessage ?? stream.result();
}
function throwPerplexityStreamError(message: AssistantMessage): never {
const status = message.errorStatus ?? 500;
const details = message.errorMessage ?? "Perplexity API stream failed";
const classified = classifyProviderHttpError("perplexity", status, details);
if (classified) throw classified;
throw new SearchProviderError("perplexity", `Perplexity API error (${status}): ${details}`, status);
}
/** Call Perplexity API-key endpoint (or OpenRouter) through the shared OpenAI streaming providers. */
async function callPerplexityApi(
apiKey: string,
config: ApiConfig,
request: PerplexityRequest,
fetchImpl: FetchImpl | undefined,
signal?: AbortSignal,
): Promise<PerplexityResponse> {
const response = await (fetchImpl ?? fetch)(PERPLEXITY_API_URL, {
method: "POST",
headers: {
Authorization: `Bearer ${apiKey}`,
"Content-Type": "application/json",
},
body: JSON.stringify(request),
signal: withHardTimeout(signal),
});
): Promise<SearchResponse> {
const metadata: PerplexityApiStreamMetadata = {};
const context = buildPerplexityContext(request);
const requestSignal = withHardTimeout(signal);
const onSseEvent = (event: { data: string }): void => {
collectPerplexityMetadata(metadata, event.data);
};
if (!response.ok) {
const errorText = await response.text();
const classified = classifyProviderHttpError("perplexity", response.status, errorText);
if (classified) throw classified;
throw new SearchProviderError(
"perplexity",
`Perplexity API error (${response.status}): ${errorText}`,
response.status,
);
const message = config.useResponses
? await drainAssistantStream(
streamOpenAIResponses(buildPerplexityResponsesModel(config, request), context, {
apiKey: config.apiKey,
maxTokens: request.max_tokens ?? undefined,
temperature: request.temperature ?? undefined,
signal: requestSignal,
fetch: fetchImpl,
extraBody: buildPerplexityExtraBody(request),
onSseEvent,
}),
)
: await drainAssistantStream(
streamOpenAICompletions(buildPerplexityCompletionsModel(config, request), context, {
apiKey: config.apiKey,
maxTokens: request.max_tokens ?? undefined,
temperature: request.temperature ?? undefined,
signal: requestSignal,
fetch: fetchImpl,
onPayload: payload => applyPerplexityExtraBody(payload, request),
onSseEvent,
}),
);
if (message.stopReason === "error" || message.stopReason === "aborted") {
throwPerplexityStreamError(message);
}
return response.json() as Promise<PerplexityResponse>;
return parseStreamedApiResponse(message, metadata);
}
function buildOAuthSources(event: PerplexityOAuthStreamEvent): SearchSource[] {
@@ -570,26 +774,49 @@ async function callPerplexityAsk(
};
}
function messageContentToText(content: PerplexityMessageOutput["content"]): string {
if (!content) return "";
if (typeof content === "string") return content;
return content.map(chunk => (chunk.type === "text" ? chunk.text : "")).join("");
function assistantText(message: AssistantMessage): string {
let text = "";
for (const block of message.content) {
if (block.type === "text") text += block.text;
}
return text;
}
/** Parse API response into unified SearchResponse */
function parseResponse(response: PerplexityResponse): SearchResponse {
const messageContent = response.choices[0]?.message?.content ?? null;
const answer = messageContentToText(messageContent);
function isPerplexitySearchResult(value: unknown): value is PerplexitySearchResult {
const record = asRecord(value);
return typeof record?.url === "string" && record.url.length > 0;
}
function searchResultsFromMetadata(metadata: PerplexityApiStreamMetadata): PerplexitySearchResult[] {
return Array.isArray(metadata.search_results) ? metadata.search_results.filter(isPerplexitySearchResult) : [];
}
function citationUrlsFromMetadata(metadata: PerplexityApiStreamMetadata): string[] {
return Array.isArray(metadata.citations)
? metadata.citations.filter((url): url is string => typeof url === "string" && url.length > 0)
: [];
}
function relatedQuestionsFromMetadata(metadata: PerplexityApiStreamMetadata): string[] {
return Array.isArray(metadata.related_questions)
? metadata.related_questions.filter(
(question): question is string => typeof question === "string" && question.trim().length > 0,
)
: [];
}
function buildApiSources(metadata: PerplexityApiStreamMetadata): {
sources: SearchSource[];
citations: SearchCitation[];
} {
const sources: SearchSource[] = [];
const citations: SearchCitation[] = [];
const citationUrls = response.citations ?? [];
const searchResults = response.search_results ?? [];
const searchResults = searchResultsFromMetadata(metadata);
const citationUrls = citationUrlsFromMetadata(metadata);
if (citationUrls.length > 0) {
for (const url of citationUrls) {
const searchResult = searchResults.find(r => r.url === url);
const searchResult = searchResults.find(result => result.url === url);
sources.push({
title: searchResult?.title ?? url,
url,
@@ -597,10 +824,7 @@ function parseResponse(response: PerplexityResponse): SearchResponse {
publishedDate: searchResult?.date ?? undefined,
ageSeconds: dateToAgeSeconds(searchResult?.date),
});
citations.push({
url,
title: searchResult?.title ?? url,
});
citations.push({ url, title: searchResult?.title ?? url });
}
} else {
for (const searchResult of searchResults) {
@@ -614,7 +838,22 @@ function parseResponse(response: PerplexityResponse): SearchResponse {
}
}
const relatedQuestions = (response.related_questions ?? []).filter(q => q.trim().length > 0);
return { sources, citations };
}
function usageFromAssistant(usage: Usage): SearchResponse["usage"] | undefined {
if (usage.input === 0 && usage.output === 0 && usage.totalTokens === 0) return undefined;
return {
inputTokens: usage.input,
outputTokens: usage.output,
totalTokens: usage.totalTokens,
};
}
function parseStreamedApiResponse(message: AssistantMessage, metadata: PerplexityApiStreamMetadata): SearchResponse {
const { sources, citations } = buildApiSources(metadata);
const relatedQuestions = relatedQuestionsFromMetadata(metadata);
const answer = assistantText(message);
return {
provider: "perplexity",
@@ -622,15 +861,9 @@ function parseResponse(response: PerplexityResponse): SearchResponse {
sources,
citations: citations.length > 0 ? citations : undefined,
relatedQuestions: relatedQuestions.length > 0 ? relatedQuestions : undefined,
usage: response.usage
? {
inputTokens: response.usage.prompt_tokens,
outputTokens: response.usage.completion_tokens,
totalTokens: response.usage.total_tokens,
}
: undefined,
model: response.model,
requestId: response.id,
usage: usageFromAssistant(message.usage),
model: metadata.model ?? message.model,
requestId: metadata.id ?? message.responseId,
};
}
@@ -643,35 +876,6 @@ function applySourceLimit(result: SearchResponse, limit?: number): SearchRespons
/** Execute Perplexity web search */
export async function searchPerplexity(params: PerplexitySearchParams): Promise<SearchResponse> {
const auth = await findPerplexityAuth(params.authStorage, params.sessionId, params.signal);
if (auth.type !== "api_key") {
// OAuth bearer mode routes the whole authenticated unit (the ask
// session/SSE request) through the central auth-retry policy so a 401 or
// usage-limit force-refreshes, then rotates to a sibling credential.
// Cookie/env/anonymous modes have no rotatable credential — untouched.
const askResult =
auth.type === "oauth"
? await withOAuthAccess(
params.authStorage,
"perplexity",
access => callPerplexityAsk({ type: "oauth", token: access.accessToken }, params),
{ sessionId: params.sessionId, signal: params.signal, seed: auth.access },
)
: await callPerplexityAsk(auth, params);
return applySourceLimit(
{
provider: "perplexity",
answer: askResult.answer || undefined,
sources: askResult.sources,
model: askResult.model,
requestId: askResult.requestId,
authMode: auth.type === "anonymous" ? "anonymous" : "oauth",
},
params.num_results,
);
}
const systemPrompt = params.system_prompt;
const messages: PerplexityRequest["messages"] = [];
if (systemPrompt) {
@@ -700,10 +904,48 @@ export async function searchPerplexity(params: PerplexitySearchParams): Promise<
request.search_recency_filter = params.search_recency_filter;
}
const response = await callPerplexityApi(auth.token, request, params.fetch, params.signal);
const result = parseResponse(response);
result.authMode = "api_key";
return applySourceLimit(result, params.num_results);
const authMethods = await getAvailableAuthMethods(params.authStorage, params.sessionId, params.signal);
for (const auth of authMethods) {
if (auth.type === "api_key") {
try {
const result = await callPerplexityApi(auth, request, params.fetch, params.signal);
result.authMode = "api_key";
return applySourceLimit(result, params.num_results);
} catch (error) {
if (params.signal?.aborted) throw error;
// Try next method
}
} else {
// Use OAuth/cookies/anonymous path
try {
const askResult =
auth.type === "oauth"
? await withOAuthAccess(
params.authStorage,
"perplexity",
access => callPerplexityAsk({ type: "oauth", token: access.accessToken }, params),
{ sessionId: params.sessionId, signal: params.signal, seed: auth.access },
)
: await callPerplexityAsk(auth, params);
return applySourceLimit(
{
provider: "perplexity",
answer: askResult.answer || undefined,
sources: askResult.sources,
model: askResult.model,
requestId: askResult.requestId,
authMode: auth.type === "anonymous" ? "anonymous" : "oauth",
},
params.num_results,
);
} catch {
// Try next method
}
}
}
throw new SearchProviderError("perplexity", "No authentication method available.", 401);
}
/** Search provider for Perplexity. */
@@ -712,7 +954,9 @@ export class PerplexityProvider extends SearchProvider {
readonly label = "Perplexity";
isAvailable(authStorage: AuthStorage): boolean {
return !!$env.PERPLEXITY_COOKIES?.trim() || authStorage.hasAuth("perplexity") || !!findApiKey();
return (
!!$env.PERPLEXITY_COOKIES?.trim() || authStorage.hasAuth("perplexity") || authStorage.hasAuth("openrouter")
);
}
/**
@@ -12,7 +12,7 @@ import {
shouldCompact,
} from "@oh-my-pi/pi-agent-core/compaction/compaction";
import * as ai from "@oh-my-pi/pi-ai";
import { encodeTextSignatureV1 } from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
import { encodeTextSignatureV1 } from "@oh-my-pi/pi-ai/providers/openai-shared";
import type { AssistantMessage, Model, ProviderPayload, Usage } from "@oh-my-pi/pi-ai/types";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import { buildSessionContext } from "@oh-my-pi/pi-coding-agent/session/session-context";
@@ -10,7 +10,7 @@ import { BUILTIN_DEFAULTS_PROVIDER_ID, type Rule, ruleCapability } from "@oh-my-
import type { LoadContext } from "@oh-my-pi/pi-coding-agent/capability/types";
// Register all discovery providers as a side effect.
import "@oh-my-pi/pi-coding-agent/discovery";
import { TtsrManager } from "@oh-my-pi/pi-coding-agent/export/ttsr";
import { TtsrManager, type TtsrMatchContext } from "@oh-my-pi/pi-coding-agent/export/ttsr";
function ruleProvider() {
const cap = getCapability(ruleCapability.id);
@@ -97,6 +97,56 @@ describe("builtin-defaults rule provider", () => {
).toEqual([]);
});
it("fires ts-no-inline-cast-access on inline cast-and-access but not named-type casts", async () => {
const rules = await loadBuiltinRules();
const rule = rules.find(r => r.name === "ts-no-inline-cast-access");
if (!rule) throw new Error("ts-no-inline-cast-access rule missing");
const manager = new TtsrManager();
expect(manager.addRule(rule)).toBe(true);
// AST conditions only run on edit/write streams, with the language inferred from the path.
const ctx: TtsrMatchContext = { source: "tool", toolName: "edit", filePaths: ["src/foo.ts"] };
// Inline object-type assertion immediately read — every access form is flagged.
const violations = [
"const a = (value as { content: unknown }).content;",
"const b = (value as { content: unknown })?.content;",
'const c = (opts as { enabled: boolean })["enabled"];',
"const d = (value as unknown as { content: unknown }).content;",
];
for (const snippet of violations) {
manager.resetBuffer();
const matches = await manager.checkAstSnapshot(snippet, ctx);
expect(
matches.map(r => r.name),
snippet,
).toEqual(["ts-no-inline-cast-access"]);
}
// A cast to a named type, plain member access, and a bare cast (no read) are all left alone.
const allowed = [
"const e = (value as Foo).bar;",
"const f = obj.content;",
"const g = value as { content: unknown };",
];
for (const snippet of allowed) {
manager.resetBuffer();
const matches = await manager.checkAstSnapshot(snippet, ctx);
expect(matches, snippet).toEqual([]);
}
// Out of scope: the same violation in a non-TS file never reaches the matcher.
manager.resetBuffer();
expect(
await manager.checkAstSnapshot("const h = (value as { content: unknown }).content;", {
source: "tool",
toolName: "edit",
filePaths: ["src/foo.js"],
}),
).toEqual([]);
});
it("is the lowest-priority rule provider so user/project rules override defaults", () => {
const { cap, provider } = ruleProvider();
const others = cap.providers.filter(p => p.id !== BUILTIN_DEFAULTS_PROVIDER_ID);
@@ -93,7 +93,17 @@ describe("browser stealth bootstrap", () => {
it("uses documentElement as the iframe container when document.head is unavailable", () => {
type NativeWindow = Pick<
typeof globalThis,
"Function" | "Object" | "setTimeout" | "Math" | "Event" | "Promise" | "Blob" | "Proxy" | "Intl" | "Date"
| "Function"
| "Object"
| "setTimeout"
| "Math"
| "Event"
| "Promise"
| "Blob"
| "Proxy"
| "Intl"
| "Date"
| "Reflect"
>;
type FakeContainer = {
appendChild(node: FakeIframe): void;
@@ -128,6 +138,7 @@ describe("browser stealth bootstrap", () => {
Proxy,
Intl,
Date,
Reflect,
};
const iframe: FakeIframe = { contentWindow: nativeWindow, parentNode: null, style: {} };
const createdTags: string[] = [];
@@ -149,6 +160,14 @@ describe("browser stealth bootstrap", () => {
expect(removed).toEqual([iframe]);
expect(iframe.parentNode).toBeNull();
});
it("caches Reflect methods before any stealth scripts run", () => {
const script = buildStealthInjectionScriptForTest();
expect(script).toContain("const Reflect_get = nativeWindow.Reflect.get");
expect(script).toContain("const Reflect_apply = nativeWindow.Reflect.apply");
expect(script).not.toContain("Reflect" + ".get(");
expect(script).not.toContain("Reflect.apply(");
});
});
describe("browser stealth target setup", () => {
it("attempts user-agent override for page-like existing targets and skips non-page worker/browser targets", async () => {
@@ -3,6 +3,8 @@ import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai";
import { PerplexityProvider, searchPerplexity } from "@oh-my-pi/pi-coding-agent/web/search/providers/perplexity";
const API_URL = "https://api.perplexity.ai/chat/completions";
const OPENROUTER_API_URL = "https://openrouter.io/api/v1/chat/completions";
const RESPONSES_URL = "https://api.perplexity.ai/v1/responses";
// API-key path only: getOAuthAccess returns undefined so findPerplexityAuth
// falls through to PERPLEXITY_API_KEY (set per-test, restored in afterEach).
@@ -10,51 +12,79 @@ const apiKeyAuthStorage = {
async getOAuthAccess() {
return undefined;
},
async getApiKey(provider: string) {
if (provider === "perplexity") return process.env.PERPLEXITY_API_KEY;
if (provider === "openrouter") return process.env.OPENROUTER_API_KEY;
return undefined;
},
hasAuth() {
return false;
},
} as unknown as AuthStorage;
function mockApi(capture: (body: Record<string, unknown>) => void, response: Record<string, unknown>): FetchImpl {
function sseResponse(events: Record<string, unknown>[]): Response {
return new Response(`${events.map(event => `data: ${JSON.stringify(event)}\n\n`).join("")}data: [DONE]\n\n`, {
status: 200,
headers: { "Content-Type": "text/event-stream" },
});
}
function mockApi(capture: (body: Record<string, unknown>) => void, events: Record<string, unknown>[]): FetchImpl {
return async (input, init) => {
const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
if (url === API_URL) {
capture(JSON.parse(init?.body as string));
return new Response(JSON.stringify(response), {
status: 200,
headers: { "Content-Type": "application/json" },
});
return sseResponse(events);
}
return new Response("not mocked", { status: 500 });
};
}
function baseResponse(extra: Record<string, unknown> = {}) {
return {
id: "req-1",
model: "sonar-pro",
created: 0,
choices: [{ index: 0, message: { role: "assistant", content: "answer" }, delta: {} }],
search_results: [{ title: "T", url: "https://example.com", snippet: "s" }],
...extra,
};
function baseResponse(extra: Record<string, unknown> = {}): Record<string, unknown>[] {
return [
{
id: "req-1",
model: "sonar-pro",
choices: [{ index: 0, delta: { role: "assistant", content: "answer" }, finish_reason: null }],
},
{
id: "req-1",
model: "sonar-pro",
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
search_results: [{ title: "T", url: "https://example.com", snippet: "s" }],
...extra,
},
{
id: "req-1",
model: "sonar-pro",
choices: [],
usage: { prompt_tokens: 1, completion_tokens: 2, total_tokens: 3 },
},
];
}
describe("Perplexity API-key request shape", () => {
const savedKey = process.env.PERPLEXITY_API_KEY;
const savedOpenRouterKey = process.env.OPENROUTER_API_KEY;
const savedCookies = process.env.PERPLEXITY_COOKIES;
const savedResponsesMode = process.env.PI_PERPLEXITY_RESPONSES;
beforeEach(() => {
process.env.PERPLEXITY_API_KEY = "test-key";
delete process.env.PERPLEXITY_COOKIES;
delete process.env.PI_PERPLEXITY_RESPONSES;
});
afterEach(() => {
vi.restoreAllMocks();
if (savedKey === undefined) delete process.env.PERPLEXITY_API_KEY;
else process.env.PERPLEXITY_API_KEY = savedKey;
if (savedOpenRouterKey === undefined) delete process.env.OPENROUTER_API_KEY;
else process.env.OPENROUTER_API_KEY = savedOpenRouterKey;
if (savedCookies === undefined) delete process.env.PERPLEXITY_COOKIES;
else process.env.PERPLEXITY_COOKIES = savedCookies;
if (savedResponsesMode === undefined) delete process.env.PI_PERPLEXITY_RESPONSES;
else process.env.PI_PERPLEXITY_RESPONSES = savedResponsesMode;
});
it("requests comprehensive defaults: 20 results, high context, related questions", async () => {
@@ -107,6 +137,99 @@ describe("Perplexity API-key request shape", () => {
expect(response.relatedQuestions).toBeUndefined();
});
it("falls back to OpenRouter with the selected API-key config after direct Perplexity fails", async () => {
process.env.OPENROUTER_API_KEY = "openrouter-test-key";
const urls: string[] = [];
const bodies: Record<string, unknown>[] = [];
const fetchMock: FetchImpl = async (input, init) => {
const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
urls.push(url);
bodies.push(JSON.parse(init?.body as string));
if (url === API_URL) return new Response("direct failed", { status: 500 });
if (url === OPENROUTER_API_URL) return sseResponse(baseResponse());
return new Response("not mocked", { status: 500 });
};
const response = await searchPerplexity({
query: "quic vs tcp",
authStorage: apiKeyAuthStorage,
fetch: fetchMock,
});
expect(urls).toEqual([API_URL, OPENROUTER_API_URL]);
expect(bodies[0]?.model).toBe("sonar-pro");
expect(bodies[1]?.model).toBe("perplexity/sonar-pro");
expect(response.authMode).toBe("api_key");
expect(response.answer).toBe("answer");
});
it("streams the Responses API and captures Perplexity search result events", async () => {
process.env.PI_PERPLEXITY_RESPONSES = "1";
let body: Record<string, unknown> | undefined;
const fetchMock: FetchImpl = async (input, init) => {
const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
if (url !== RESPONSES_URL) return new Response("not mocked", { status: 500 });
body = JSON.parse(init?.body as string);
return sseResponse([
{
type: "response.created",
response: { id: "resp-1", status: "in_progress" },
},
{
type: "response.reasoning.search_results",
results: [{ title: "Responses source", url: "https://example.org", snippet: "rs" }],
},
{
type: "response.output_item.added",
output_index: 0,
item: { id: "msg-1", type: "message", role: "assistant", content: [], status: "in_progress" },
},
{
type: "response.output_text.delta",
output_index: 0,
item_id: "msg-1",
content_index: 0,
delta: "answer",
},
{
type: "response.output_item.done",
output_index: 0,
item: {
id: "msg-1",
type: "message",
role: "assistant",
status: "completed",
content: [{ type: "output_text", text: "answer", annotations: [] }],
},
},
{
type: "response.completed",
response: {
id: "resp-1",
model: "sonar-pro",
status: "completed",
output: [],
usage: { input_tokens: 1, output_tokens: 2, total_tokens: 3 },
},
},
]);
};
const response = await searchPerplexity({
query: "quic vs tcp",
authStorage: apiKeyAuthStorage,
fetch: fetchMock,
});
expect(body?.num_search_results).toBe(20);
expect(body?.max_output_tokens).toBe(8192);
expect(response.answer).toBe("answer");
expect(response.sources[0]).toMatchObject({
title: "Responses source",
url: "https://example.org",
snippet: "rs",
});
expect(response.requestId).toBe("resp-1");
});
});
const OAUTH_ASK_URL = "https://www.perplexity.ai/rest/sse/perplexity_ask";
@@ -117,6 +240,9 @@ const oauthAuthStorage = {
async getOAuthAccess() {
return { accessToken: "test-oauth-token" };
},
async getApiKey() {
return undefined;
},
hasAuth() {
return true;
},
@@ -126,6 +252,9 @@ const anonymousAuthStorage = {
async getOAuthAccess() {
return undefined;
},
async getApiKey() {
return undefined;
},
hasAuth() {
return false;
},
+4 -1
View File
@@ -1,6 +1,9 @@
# Changelog
## [Unreleased]
### Changed
- Updated OpenRouter request headers to use standard shared headers from the pi-ai package
## [16.0.5] - 2026-06-17
@@ -129,4 +132,4 @@
- Fixed `rememberBatch(..., { extract: true })` to run background fact extraction for batch uploads (including per-item `extract` flags) so extracted facts are generated and recallable after extraction
- Fixed `extract: true` fact extraction to continue safely when no LLM is configured by turning extraction failures into no-op background tasks
- Fixed configured LLM fact extraction by using temperature 0 so re-ingesting the same text is deterministic and avoids near-duplicate extractions
- Fixed `remember(..., { extract: true })` silently dropping the flag: it now schedules the LLM fact extractor (`extractFactsSafe`) over the stored content and persists the extracted facts so they become recallable. Previously the LLM extractor had no production callers and `extract` was dead.
- Fixed `remember(..., { extract: true })` silently dropping the flag: it now schedules the LLM fact extractor (`extractFactsSafe`) over the stored content and persists the extracted facts so they become recallable. Previously the LLM extractor had no production callers and `extract` was dead.
+2 -6
View File
@@ -1,5 +1,5 @@
import { mkdirSync } from "node:fs";
import { type ApiKey, ProviderHttpError, withAuth } from "@oh-my-pi/pi-ai";
import { type ApiKey, getOpenRouterHeaders, ProviderHttpError, withAuth } from "@oh-my-pi/pi-ai";
import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts";
import {
$env,
@@ -11,7 +11,6 @@ import {
} from "@oh-my-pi/pi-utils";
import type { EmbeddingModel } from "fastembed";
import { LRUCache } from "lru-cache/raw";
import packageJson from "../../package.json" with { type: "json" };
import { loadFastembed } from "./fastembed-runtime";
import {
type EmbeddingOutput,
@@ -268,10 +267,7 @@ async function embedApi(texts: readonly string[]): Promise<EmbeddingMatrix | nul
const response = await withAuth(apiKey, async key => {
const headers: Record<string, string> = {
"Content-Type": "application/json",
"User-Agent": `Oh-My-Pi/${packageJson.version}`,
"HTTP-Referer": "https://omp.sh/",
"X-OpenRouter-Title": "Oh-My-Pi",
"X-OpenRouter-Categories": "cli-agent",
...getOpenRouterHeaders(),
};
if (key !== "") {
headers.Authorization = `Bearer ${key}`;
+546
View File
@@ -0,0 +1,546 @@
import { describe, expect, test } from "bun:test";
import { IndentationText, Project } from "ts-morph";
import { inlineFile, type Options } from "./inline-functions";
function opts(overrides: Partial<Options> = {}): Options {
return {
maxStatements: overrides.maxStatements ?? 3,
nameFilter: overrides.nameFilter,
verbose: false,
strictEffects: overrides.strictEffects ?? false,
};
}
/** Run the inliner on an in-memory source and return { text, inlined names }. */
function run(src: string, overrides: Partial<Options> = {}): { text: string; inlined: string[] } {
const project = new Project({
useInMemoryFileSystem: true,
manipulationSettings: { indentationText: IndentationText.Tab },
});
const sf = project.createSourceFile("input.ts", src);
const inlined = inlineFile(sf, opts(overrides));
return { text: sf.getFullText(), inlined };
}
/** Collapse whitespace so assertions ignore the inliner's pre-format indentation. */
function norm(s: string): string {
return s.replace(/\s+/g, " ").trim();
}
describe("inline-functions: guard inversion", () => {
test("single `!==` guard becomes a positive `===` wrapper with params substituted", () => {
const { text, inlined } = run(`
function handlePart(currentItem: Item | null, rawEvent: Record<string, unknown>): void {
if (currentItem?.type !== "reasoning") return;
appendPart(currentItem, (rawEvent as { part: P }).part);
}
function dispatch(runtime: Runtime, rawEvent: Record<string, unknown>): void {
handlePart(runtime.currentItem, rawEvent);
}
`);
expect(inlined).toEqual(["handlePart"]);
expect(text).not.toContain("function handlePart");
const n = norm(text);
expect(n).toContain('if (runtime.currentItem?.type === "reasoning") {');
expect(n).toContain("appendPart(runtime.currentItem, (rawEvent as { part: P }).part);");
});
test("`||` guard with two comparisons becomes `&&` via De Morgan", () => {
const { text } = run(`
function handleDelta(currentItem: Item | null, currentBlock: Block | null, rawEvent: Record<string, unknown>): void {
if (currentItem?.type !== "reasoning" || currentBlock?.type !== "thinking") return;
const delta = (rawEvent as { delta?: string }).delta || "";
appendDelta(currentItem, currentBlock, delta);
}
function dispatch(runtime: Runtime, rawEvent: Record<string, unknown>): void {
handleDelta(runtime.currentItem, runtime.currentBlock, rawEvent);
}
`);
const n = norm(text);
expect(n).toContain(
'if (runtime.currentItem?.type === "reasoning" && runtime.currentBlock?.type === "thinking") {',
);
expect(n).toContain('const delta = (rawEvent as { delta?: string }).delta || "";');
expect(n).toContain("appendDelta(runtime.currentItem, runtime.currentBlock, delta);");
});
test("comparator flips: `=== null` guard becomes `!== null`", () => {
const { text } = run(`
function h(x: string | null): void {
if (x === null) return;
use(x);
}
function call(value: string | null): void {
h(value);
}
`);
expect(norm(text)).toContain("if (value !== null) {");
});
test("multiple guards combine with `&&`, wrapping disjunctions", () => {
const { text } = run(`
function h(a: A, b: B): void {
if (a.x !== 1) return;
if (b.y === 2 || b.z === 3) return;
done(a, b);
}
function call(p: A, q: B): void {
h(p, q);
}
`);
// !(a.x !== 1) -> p.x === 1 ; !(b.y === 2 || b.z === 3) -> q.y !== 2 && q.z !== 3.
// The disjunction-negation is an && itself, so it nests flat with no parens.
expect(norm(text)).toContain("if (p.x === 1 && q.y !== 2 && q.z !== 3) {");
});
test("non-comparison guard uses a `!(...)` fallback", () => {
const { text } = run(`
function h(x: string): void {
if (isBad(x)) return;
use(x);
}
function call(v: string): void {
h(v);
}
`);
expect(norm(text)).toContain("if (!(isBad(v))) {");
});
});
describe("inline-functions: argument handling", () => {
test("guardless helper splices its statements straight into the caller", () => {
const { text, inlined } = run(`
function h(a: number, b: string): void {
first(a);
second(b);
}
function call(x: string): void {
h(1, x);
}
`);
expect(inlined).toEqual(["h"]);
const n = norm(text);
expect(n).not.toContain("function h");
expect(n).toContain("function call(x: string): void { first(1); second(x); }");
});
test("impure argument is hoisted to a const evaluated unconditionally before the guard", () => {
const { text } = run(`
function h(x: number): void {
if (bad(x)) return;
use(x);
}
function call(): void {
h(make());
}
`);
const n = norm(text);
// make() has a side effect and is used twice (guard + tail) -> hoist once, unconditionally.
expect(n).toContain("const __inl_x = make();");
expect(n.indexOf("const __inl_x = make();")).toBeLessThan(n.indexOf("if (!(bad(__inl_x)))"));
expect(n).toContain("use(__inl_x);");
// the call expression must appear exactly once (single evaluation)
expect(n.match(/make\(\)/g)?.length).toBe(1);
});
test("pure but non-trivial argument used twice is hoisted, not duplicated", () => {
const { text } = run(`
function h(x: number): void {
if (x > 0) return;
use(x);
}
function call(a: number, b: number): void {
h(a + b);
}
`);
const n = norm(text);
expect(n).toContain("const __inl_x = a + b;");
expect(n.match(/a \+ b/g)?.length).toBe(1);
});
test("default mode treats a member chain as pure and inlines it (the pretty result)", () => {
const { text } = run(`
function h(x: number): void {
if (x > 0) return;
use(x);
}
function call(runtime: Runtime): void {
h(runtime.count);
}
`);
const n = norm(text);
expect(n).not.toContain("const __inl_x");
expect(n).toContain("if (runtime.count <= 0) {");
expect(n).toContain("use(runtime.count);");
});
test("low-precedence inline argument gets wrapped in parens at member-access use sites", () => {
const { text } = run(`
function h(x: number): void {
take(x.toFixed());
}
function call(a: number, b: number): void {
h(a ? b : a);
}
`);
// conditional arg used once -> inlined, parenthesized because it feeds `.toFixed()`
expect(norm(text)).toContain("take((a ? b : a).toFixed());");
});
test("unused impure argument is still evaluated for its side effects", () => {
const { text } = run(`
function h(unusedArg: number): void {
sideEffectFree();
}
function call(): void {
h(makeNoise());
}
`);
const n = norm(text);
expect(n).toContain("makeNoise();");
expect(n).toContain("sideEffectFree();");
});
});
describe("inline-functions: --strict-effects", () => {
test("hoists a member-chain argument so it evaluates eagerly, exactly once", () => {
const { text } = run(
`
function h(x: number): void {
if (x > 0) return;
use(x);
}
function call(runtime: Runtime): void {
h(runtime.count);
}
`,
{ strictEffects: true },
);
const n = norm(text);
expect(n).toContain("const __inl_x = runtime.count;");
expect(n).toContain("if (__inl_x <= 0) {");
expect(n).toContain("use(__inl_x);");
expect(n.match(/runtime\.count/g)?.length).toBe(1);
});
test("snapshots every used argument left-to-right before the body", () => {
const { text } = run(
`
function h(a: number, b: number): void {
use(a);
use2(b);
}
function call(x: number): void {
h(x, (x = 2));
}
`,
{ strictEffects: true },
);
const n = norm(text);
// param a is read BEFORE the second argument mutates x.
expect(n).toContain("const __inl_a = x;");
expect(n).toContain("const __inl_b = (x = 2);");
expect(n.indexOf("const __inl_a = x;")).toBeLessThan(n.indexOf("const __inl_b = (x = 2);"));
expect(n).toContain("use(__inl_a);");
expect(n).toContain("use2(__inl_b);");
});
});
describe("inline-functions: multiple call sites", () => {
test("every call site is inlined and the declaration removed", () => {
const { text, inlined } = run(`
function h(item: Item | null): void {
if (item?.type !== "x") return;
touch(item);
}
function a(r: Runtime): void { h(r.currentItem); }
function b(r: Runtime): void { h(r.other); }
`);
expect(inlined).toEqual(["h"]);
const n = norm(text);
expect(n).not.toContain("function h(");
expect(n).toContain('if (r.currentItem?.type === "x") {');
expect(n).toContain('if (r.other?.type === "x") {');
});
});
describe("inline-functions: safety skips", () => {
const cases: Array<{ name: string; src: string }> = [
{
name: "exported helper",
src: `export function h(x: number): void { if (x > 0) return; use(x); }
function call(): void { h(1); }`,
},
{
name: "recursive helper",
src: `function h(x: number): void { if (x > 0) return; h(x - 1); }
function call(): void { h(3); }`,
},
{
name: "call result is used",
src: `function h(x: number): boolean { if (x > 0) return false; return true; }
function call(): void { const r = h(1); use(r); }`,
},
{
name: "parameter is written",
src: `function h(x: number): void { x = x + 1; use(x); }
function call(): void { h(1); }`,
},
{
name: "tail contains a return",
src: `function h(x: number): void { use(x); if (x > 0) return; use2(x); }
function call(): void { h(1); }`,
},
{
name: "async helper",
src: `async function h(x: number): Promise<void> { use(x); }
function call(): void { void h(1); }`,
},
{
name: "default parameter value",
src: `function h(x: number = 5): void { use(x); }
function call(): void { h(); }`,
},
{
name: "tail too large for --max-statements",
src: `function h(x: number): void { if (x > 0) return; one(x); two(x); three(x); four(x); }
function call(): void { h(1); }`,
},
{
name: "function-scoped var in tail",
src: `function h(x: number): void { if (x > 0) return; var y = x; use(y); }
function call(): void { h(1); }`,
},
{
name: "for (var ...) in tail",
src: `function h(x: number): void { for (var i = 0; i < x; i++) use(i); }
function call(): void { h(2); }`,
},
{
name: "tail declares a local function",
src: `function h(x: number): void { if (x > 0) return; function inner() { return x; } use(inner()); }
function call(): void { h(1); }`,
},
];
for (const c of cases) {
test(`skips: ${c.name}`, () => {
const { inlined } = run(c.src);
expect(inlined).toEqual([]);
});
}
test("skips when a free body identifier is shadowed at a call site", () => {
const { inlined, text } = run(`
function h(x: number): void { if (x > 0) return; helper(x); }
function call(): void {
const helper = (n: number) => n;
h(1);
}
`);
expect(inlined).toEqual([]);
expect(text).toContain("function h(");
});
test("inlines when the same name lives only at module scope (no shadow)", () => {
const { inlined } = run(`
function helper(n: number): number { return n; }
function h(x: number): void { if (x > 0) return; helper(x); }
function call(): void { h(1); }
`);
expect(inlined).toEqual(["h"]);
});
});
describe("inline-functions: guardless local-name collision", () => {
test("renames a tail local that would redeclare a name already live in the target block", () => {
const { text } = run(`
function h(x: number): void {
const delta = x + 1;
use(delta);
}
function call(value: number): void {
const delta = compute();
h(value);
log(delta);
}
`);
const n = norm(text);
// caller's `delta` is untouched; the inlined local is renamed.
expect(n).toContain("const delta = compute();");
expect(n).toContain("const delta_2 = value + 1;");
expect(n).toContain("use(delta_2);");
expect(n).toContain("log(delta);");
});
});
describe("inline-functions: type soundness", () => {
test("a fully-typed inline introduces no new diagnostics", () => {
const project = new Project({ useInMemoryFileSystem: true, compilerOptions: { strict: true } });
const sf = project.createSourceFile(
"typed.ts",
`
interface Block { type: "thinking" | "text"; }
interface Item { type: "reasoning" | "message"; }
interface Runtime { currentItem: Item | null; currentBlock: Block | null; }
declare function appendDelta(item: Item, block: Block, delta: string, n: number): void;
function handleDelta(
currentItem: Item | null,
currentBlock: Block | null,
rawEvent: Record<string, unknown>,
n: number,
): void {
if (currentItem?.type !== "reasoning" || currentBlock?.type !== "thinking") return;
const delta = (rawEvent as { delta?: string }).delta || "";
appendDelta(currentItem, currentBlock, delta, n);
}
export function dispatch(runtime: Runtime, rawEvent: Record<string, unknown>): void {
handleDelta(runtime.currentItem, runtime.currentBlock, rawEvent, 0);
}
`,
);
expect(sf.getPreEmitDiagnostics()).toHaveLength(0);
const inlined = inlineFile(sf, opts());
expect(inlined).toEqual(["handleDelta"]);
expect(sf.getPreEmitDiagnostics()).toHaveLength(0);
expect(norm(sf.getFullText())).toContain(
'if (runtime.currentItem?.type === "reasoning" && runtime.currentBlock?.type === "thinking") {',
);
});
});
describe("inline-functions: formatting safety", () => {
test("wraps the tail exactly one indent level deeper than the inverted guard", () => {
const { text } = run(
'function dispatch(r: Runtime): void {\n\thandle(r.item);\n}\n' +
'function handle(item: Item | null): void {\n\tif (item?.type !== "x") return;\n\ttouch(item);\n}\n',
);
// `if` sits at the function-body indent (one tab); its body is one deeper.
expect(text).toContain('\tif (r.item?.type === "x") {\n\t\ttouch(r.item);\n\t}');
});
test("never indents the interior of a multi-line template literal in the tail", () => {
const { text } = run(
"function h(x: number): void {\n\tif (x > 0) return;\n\tconst s = `line1\nline2`;\n\tuse(s);\n}\n" +
"function call(v: number): void {\n\th(v);\n}\n",
);
// `line2` must stay at column 0 — no tab injected into the string contents.
expect(text).toContain("`line1\nline2`");
});
});
describe("inline-functions: hoisted temp names", () => {
test("distinct temp names for multiple impure calls in the same block", () => {
const { text } = run(`
function h(x: number): void { if (bad(x)) return; use(x); }
function call(): void {
h(make1());
h(make2());
}
`);
const n = norm(text);
expect(n).toContain("const __inl_x = make1();");
expect(n).toContain("const __inl_x_2 = make2();");
});
test("temp names may be reused across separate, non-overlapping blocks", () => {
const { text } = run(`
function h(x: number): void { if (bad(x)) return; use(x); }
function call(flag: boolean): void {
if (flag) {
h(make1());
} else {
h(make2());
}
}
`);
const n = norm(text);
expect((n.match(/const __inl_x =/g) ?? []).length).toBe(2);
expect(n).not.toContain("__inl_x_2");
});
test("switch cases share one scope, so temp names stay distinct across cases", () => {
const { text } = run(`
function h(x: number): void { if (bad(x)) return; use(x); }
function call(k: number): void {
switch (k) {
case 1:
h(make1());
break;
case 2:
h(make2());
break;
}
}
`);
const n = norm(text);
expect(n).toContain("const __inl_x = make1();");
expect(n).toContain("const __inl_x_2 = make2();");
});
});
describe("inline-functions: object shorthand safety", () => {
test("expands shorthand when a substituted parameter is not a bare identifier", () => {
const { text } = run(`
function warn(sourceType: string): void {
logger.warn("msg", { sourceType });
}
function call(source: { type: string }): void {
warn(source.type);
}
`);
expect(norm(text)).toContain('logger.warn("msg", { sourceType: source.type });');
});
test("expands shorthand when a colliding tail local is renamed", () => {
const { text } = run(`
function h(): void {
const id = compute();
use({ id });
}
function call(): void {
const id = 1;
h();
log(id);
}
`);
const n = norm(text);
expect(n).toContain("const id_2 = compute();");
expect(n).toContain("use({ id: id_2 });");
});
test("does not rename a tail local that only matches a property name in the block", () => {
const { text } = run(`
function h(): void {
const index = compute();
use(index);
}
function call(obj: { index: number }): void {
read(obj.index);
h();
}
`);
const n = norm(text);
expect(n).toContain("const index = compute();");
expect(n).not.toContain("index_2");
});
test("does not rename a tail local that only matches a type name in the block", () => {
const { text } = run(`
interface Thing { id: number }
function h(): void {
const Thing = makeThing();
use(Thing);
}
function call(): void {
const x: Thing = { id: 1 };
h();
}
`);
const n = norm(text);
expect(n).toContain("const Thing = makeThing();");
expect(n).not.toContain("Thing_2");
});
});
+1051
View File
File diff suppressed because it is too large Load Diff