feat(ai): consolidated OpenAI-family streaming and add OpenRouter API support
This change introduces a new `openrouter` API type and extensively refactors OpenAI-family streaming providers, centralizing shared logic and improving robustness.
Key changes include:
- **Unified OpenAI-family Logic:** Consolidated core utilities, compat resolution, request shaping, and stream processing into `openai-shared.ts`, reducing duplication across `openai-completions`, `openai-responses`, and `openai-codex-responses`.
- **OpenRouter API Type:** Introduced a dedicated `openrouter` API type with dual-surface compatibility, allowing it to dispatch requests as either OpenAI Chat Completions or Responses.
- **Enhanced Provider Integration:**
- Improved Perplexity search to leverage shared OpenAI streaming transports, including API-key fallback to OpenRouter and support for Perplexity's Responses API.
- Integrated xAI-specific logic directly into the shared `stream.ts` dispatch, removing the dedicated `xai-responses` provider.
- Refined credential parsing for Google Gemini CLI and handling of Azure deployment names.
- **Robustness & Consistency:** Improved error handling for Codex, standardized output token parameter resolution, and ensured consistent application of reasoning suppression across all Chat Completions dialects.
- **New Documentation:** Added `provider-endpoint-constraints.md` to detail endpoint-specific behaviors and quirks for various providers.
- **Telemetry & Debugging:** Extended telemetry propagation to advisor calls and overflow compaction tasks. Improved debugging for Codex WebSocket failures and stream error messages.
- **Tooling & Security:** Updated browser stealth scripts to prevent detection and added a new `ts-no-inline-cast-access` TTSR rule.
This commit is contained in:
@@ -15,6 +15,7 @@
|
||||
"@typescript/native-preview": "catalog:",
|
||||
"lint-staged": "catalog:",
|
||||
"prettier": "catalog:",
|
||||
"ts-morph": "catalog:",
|
||||
"typescript": "catalog:",
|
||||
},
|
||||
},
|
||||
@@ -42,6 +43,7 @@
|
||||
"@bufbuild/protobuf": "catalog:",
|
||||
"@oh-my-pi/pi-catalog": "catalog:",
|
||||
"@oh-my-pi/pi-utils": "catalog:",
|
||||
"arktype": "catalog:",
|
||||
"partial-json": "catalog:",
|
||||
"zod": "catalog:",
|
||||
},
|
||||
@@ -55,6 +57,7 @@
|
||||
"dependencies": {
|
||||
"@bufbuild/protobuf": "catalog:",
|
||||
"@oh-my-pi/pi-utils": "catalog:",
|
||||
"arktype": "catalog:",
|
||||
"zod": "catalog:",
|
||||
},
|
||||
"devDependencies": {
|
||||
@@ -92,6 +95,7 @@
|
||||
"@puppeteer/browsers": "catalog:",
|
||||
"@types/turndown": "catalog:",
|
||||
"@xterm/headless": "catalog:",
|
||||
"arktype": "catalog:",
|
||||
"chalk": "catalog:",
|
||||
"diff": "catalog:",
|
||||
"fflate": "catalog:",
|
||||
@@ -349,6 +353,7 @@
|
||||
"@types/turndown": "5.0.6",
|
||||
"@typescript/native-preview": "7.0.0-dev.20260609.1",
|
||||
"@xterm/headless": "^6.0.0",
|
||||
"arktype": "^2.2.0",
|
||||
"beautiful-mermaid": "^1.1.3",
|
||||
"chalk": "^5.6.2",
|
||||
"chart.js": "^4.5.1",
|
||||
@@ -375,6 +380,7 @@
|
||||
"regexp-tree": "^0.1.27",
|
||||
"solid-js": "^1.9.13",
|
||||
"tailwindcss": "^4.3.0",
|
||||
"ts-morph": "^28.0.0",
|
||||
"turndown": "7.2.4",
|
||||
"turndown-plugin-gfm": "1.0.2",
|
||||
"typescript": "^6.0.3",
|
||||
@@ -395,6 +401,10 @@
|
||||
|
||||
"@anush008/tokenizers-win32-x64-msvc": ["@anush008/tokenizers-win32-x64-msvc@0.0.0", "", { "os": "win32", "cpu": "x64" }, "sha512-/5kP0G96+Cr6947F0ZetXnmL31YCaN15dbNbh2NHg7TXXRwfqk95+JtPP5Q7v4jbR2xxAmuseBqB4H/V7zKWuw=="],
|
||||
|
||||
"@ark/schema": ["@ark/schema@0.56.0", "", { "dependencies": { "@ark/util": "0.56.0" } }, "sha512-ECg3hox/6Z/nLajxXqNhgPtNdHWC9zNsDyskwO28WinoFEnWow4IsERNz9AnXRhTZJnYIlAJ4uGn3nlLk65vZA=="],
|
||||
|
||||
"@ark/util": ["@ark/util@0.56.0", "", {}, "sha512-BghfRC8b9pNs3vBoDJhcta0/c1J1rsoS1+HgVUreMFPdhz/CRAKReAu57YEllNaSy98rWAdY1gE+gFup7OXpgA=="],
|
||||
|
||||
"@babel/code-frame": ["@babel/code-frame@7.29.7", "", { "dependencies": { "@babel/helper-validator-identifier": "^7.29.7", "js-tokens": "^4.0.0", "picocolors": "^1.1.1" } }, "sha512-Aup7aUOfpbAUg2ROOJN6Iw5f9DMBlzu0mIkm/malLQFN/YQgO48wCj0Kxa3sEHJvPVFg7siR+qRInwXd2qhQKw=="],
|
||||
|
||||
"@babel/compat-data": ["@babel/compat-data@7.29.7", "", {}, "sha512-locTkQyKvwIEgBzVrn8693ebc97F2U8ZHjbXwDXJ5Fn2TCpNwTlKcaKLkdHop5c/icOFE7qt7Q9JC5hnKNa6Gg=="],
|
||||
@@ -853,6 +863,8 @@
|
||||
|
||||
"@tokenizer/token": ["@tokenizer/token@0.3.0", "", {}, "sha512-OvjF+z51L3ov0OyAU0duzsYuvO01PH7x4t6DJx+guahgTnBHkhJdG7soQeTSFLWN3efnHyibZ4Z8l2EuWwJN3A=="],
|
||||
|
||||
"@ts-morph/common": ["@ts-morph/common@0.29.0", "", { "dependencies": { "minimatch": "^10.0.1", "path-browserify": "^1.0.1", "tinyglobby": "^0.2.14" } }, "sha512-35oUmphHbJvQ/+UTwFNme/t2p3FoKiGJ5auTjjpNTop2dyREspirjMy82PLSC1pnDJ8ah1GU98hwpVt64YXQsg=="],
|
||||
|
||||
"@tybys/wasm-util": ["@tybys/wasm-util@0.10.2", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-RoBvJ2X0wuKlWFIjrwffGw1IqZHKQqzIchKaadZZfnNpsAYp2mM0h36JtPCjNDAHGgYez/15uMBpfGwchhiMgg=="],
|
||||
|
||||
"@types/babel__core": ["@types/babel__core@7.20.5", "", { "dependencies": { "@babel/parser": "^7.20.7", "@babel/types": "^7.20.7", "@types/babel__generator": "*", "@types/babel__template": "*", "@types/babel__traverse": "*" } }, "sha512-qoQprZvz5wQFJwMDqeseRXWv3rqMvhgpbXFfVyWhbx9X47POIA6i/+dXefEmZKoAgOaTdaIgNSMqMIU61yRyzA=="],
|
||||
@@ -907,12 +919,18 @@
|
||||
|
||||
"argparse": ["argparse@1.0.10", "", { "dependencies": { "sprintf-js": "~1.0.2" } }, "sha512-o5Roy6tNG4SL/FOkCAN6RzjiakZS25RLYFrcMttJqbdd8BWrnA+fGz57iN5Pb06pvBGvl5gQ0B48dJlslXvoTg=="],
|
||||
|
||||
"arkregex": ["arkregex@0.0.5", "", { "dependencies": { "@ark/util": "0.56.0" } }, "sha512-ncYjBdLlh5/QnVsAA8De16Tc9EqmYM7y/WU9j+236KcyYNUXogpz3sC4ATIZYzzLxwI+0sEOaQLEmLmRleaEXw=="],
|
||||
|
||||
"arktype": ["arktype@2.2.0", "", { "dependencies": { "@ark/schema": "0.56.0", "@ark/util": "0.56.0", "arkregex": "0.0.5" } }, "sha512-t54MZ7ti5BhOEvzEkgKnWvqj+UbDfWig+DHr5I34xatymPusKLS0lQpNJd8M6DzmIto2QGszHfNKoFIT8tMCZQ=="],
|
||||
|
||||
"async": ["async@3.2.6", "", {}, "sha512-htCUDlxyyCLMgaM3xXg0C0LW2xqfuQ6p05pCEIsXuyQ+a1koYKTuBMzRNwmybfLgvJDMd0r1LTn4+E0Ti6C2AA=="],
|
||||
|
||||
"babel-plugin-jsx-dom-expressions": ["babel-plugin-jsx-dom-expressions@0.40.7", "", { "dependencies": { "@babel/helper-module-imports": "7.18.6", "@babel/plugin-syntax-jsx": "^7.18.6", "@babel/types": "^7.20.7", "html-entities": "2.3.3", "parse5": "^7.1.2" }, "peerDependencies": { "@babel/core": "^7.20.12" } }, "sha512-/O6JWUmjv03OI9lL2ry9bUjpD5S3PclM55RRJEyCdcFZ5W2SEA/59d+l2hNsk3gI6kiWRdRPdOtqZmsQzFN1pQ=="],
|
||||
|
||||
"babel-preset-solid": ["babel-preset-solid@1.9.12", "", { "dependencies": { "babel-plugin-jsx-dom-expressions": "^0.40.6" }, "peerDependencies": { "@babel/core": "^7.0.0", "solid-js": "^1.9.12" }, "optionalPeers": ["solid-js"] }, "sha512-LLqnuKVDlKpyBlMPcH6qEvs/wmS9a+NczppxJ3ryS/c0O5IiSFOIBQi9GzyiGDSbcJpx4Gr87jyFTos1MyEuWg=="],
|
||||
|
||||
"balanced-match": ["balanced-match@4.0.4", "", {}, "sha512-BLrgEcRTwX2o6gGxGOCNyMvGSp35YofuYzw9h1IMTRmKqttAZZVU67bdb9Pr2vUHA8+j3i2tJfjO6C6+4myGTA=="],
|
||||
|
||||
"base64-js": ["base64-js@1.5.1", "", {}, "sha512-AKpaYlHn8t4SVbOHCy+b5+KKgvR4vrsD8vbvrbiQJps7fKDTkjkDry6ji0rUJjC0kzbNePLwzxq8iypo41qeWA=="],
|
||||
|
||||
"baseline-browser-mapping": ["baseline-browser-mapping@2.10.37", "", { "bin": { "baseline-browser-mapping": "dist/cli.cjs" } }, "sha512-girxaJ7WZssDOFhzCGZTDKoTa1gk6A1TbflaYTpykLJ4UU9Fz9kx1aREM8JCuoVHbL8X8T/mJg7w2oYSq72Oig=="],
|
||||
@@ -927,6 +945,8 @@
|
||||
|
||||
"boolean": ["boolean@3.2.0", "", {}, "sha512-d0II/GO9uf9lfUHH2BQsjxzRJZBdsjgsBiW4BvhWk/3qoKwQFjIDVN19PfX8F2D/r9PCMTtLWjYVCFrpeYUzsw=="],
|
||||
|
||||
"brace-expansion": ["brace-expansion@5.0.6", "", { "dependencies": { "balanced-match": "^4.0.2" } }, "sha512-kLpxurY4Z4r9sgMsyG0Z9uzsBlgiU/EFKhj/h91/8yHu0edo7XuixOIH3VcJ8kkxs6/jPzoI6U9Vj3WqbMQ94g=="],
|
||||
|
||||
"browserslist": ["browserslist@4.28.2", "", { "dependencies": { "baseline-browser-mapping": "^2.10.12", "caniuse-lite": "^1.0.30001782", "electron-to-chromium": "^1.5.328", "node-releases": "^2.0.36", "update-browserslist-db": "^1.2.3" }, "bin": { "browserslist": "cli.js" } }, "sha512-48xSriZYYg+8qXna9kwqjIVzuQxi+KYWp2+5nCYnYKPTr0LvD89Jqk2Or5ogxz0NUMfIjhh2lIUX/LyX9B4oIg=="],
|
||||
|
||||
"bun-types": ["bun-types@1.3.14", "", { "dependencies": { "@types/node": "*" } }, "sha512-4N0ig0fEomHt5R0KCFWjovxow98rIoRwKolrYdCcknNwMekCXRnWEUvgu5soYV8QXtVsrUD8B95MBOZGPvr6KQ=="],
|
||||
@@ -955,6 +975,8 @@
|
||||
|
||||
"cliui": ["cliui@8.0.1", "", { "dependencies": { "string-width": "^4.2.0", "strip-ansi": "^6.0.1", "wrap-ansi": "^7.0.0" } }, "sha512-BSeNnyus75C4//NQ9gQt1/csTXyo/8Sb+afLAkzAptFuMsod9HFokGNudZpi/oQV73hnVK+sR+5PVRMd+Dr7YQ=="],
|
||||
|
||||
"code-block-writer": ["code-block-writer@13.0.3", "", {}, "sha512-Oofo0pq3IKnsFtuHqSF7TqBfr71aeyZDVJ0HpmqB7FBM2qEigL0iPONSCZSO9pE9dZTAxANe5XHG9Uy0YMv8cg=="],
|
||||
|
||||
"color": ["color@5.0.3", "", { "dependencies": { "color-convert": "^3.1.3", "color-string": "^2.1.3" } }, "sha512-ezmVcLR3xAVp8kYOm4GS45ZLLgIE6SPAFoduLr6hTDajwb3KZ2F46gulK3XpcwRFb5KKGCSezCBAY4Dw4HsyXA=="],
|
||||
|
||||
"color-convert": ["color-convert@3.1.3", "", { "dependencies": { "color-name": "^2.0.0" } }, "sha512-fasDH2ont2GqF5HpyO4w0+BcewlhHEZOFn9c1ckZdHpJ56Qb7MHhH/IcJZbBGgvdtwdwNbLvxiBEdg336iA9Sg=="],
|
||||
@@ -1195,6 +1217,8 @@
|
||||
|
||||
"mimic-function": ["mimic-function@5.0.1", "", {}, "sha512-VP79XUPxV2CigYP3jWwAUFSku2aKqBH7uTAapFWCBqutsbmDo96KY5o8uh6U+/YSIn5OxJnXp73beVkpqMIGhA=="],
|
||||
|
||||
"minimatch": ["minimatch@10.2.5", "", { "dependencies": { "brace-expansion": "^5.0.5" } }, "sha512-MULkVLfKGYDFYejP07QOurDLLQpcjk7Fw+7jXS2R2czRQzR56yHRveU5NDJEOviH+hETZKSkIk5c+T23GjFUMg=="],
|
||||
|
||||
"minimist": ["minimist@1.2.8", "", {}, "sha512-2yyAR8qBkN3YuheJanUpWC5U3bb5osDywNB8RzDVlDwDHbocAJveqqj1u8+SVD7jkWT4yvsHCpWqqWqAxb0zCA=="],
|
||||
|
||||
"minipass": ["minipass@5.0.0", "", {}, "sha512-3FnjYuehv9k6ovOEbyOswadCDPX1piCfhV8ncmYtHOjuPwylVWsghTLo7rabjC3Rx5xD4HDx8Wm1xnMF7S5qFQ=="],
|
||||
@@ -1249,6 +1273,8 @@
|
||||
|
||||
"partial-json": ["partial-json@0.1.7", "", {}, "sha512-Njv/59hHaokb/hRUjce3Hdv12wd60MtM9Z5Olmn+nehe0QDAsRtRbJPvJ0Z91TusF0SuZRIvnM+S4l6EIP8leA=="],
|
||||
|
||||
"path-browserify": ["path-browserify@1.0.1", "", {}, "sha512-b7uo2UCUOYZcnF/3ID0lulOJi/bafxa1xPe7ZPsammBSpjSWQkjNxlt635YGS2MiR9GjvuXCtz2emr3jbsz98g=="],
|
||||
|
||||
"path-expression-matcher": ["path-expression-matcher@1.5.0", "", {}, "sha512-cbrerZV+6rvdQrrD+iGMcZFEiiSrbv9Tfdkvnusy6y0x0GKBXREFg/Y65GhIfm0tnLntThhzCnfKwp1WRjeCyQ=="],
|
||||
|
||||
"path-is-absolute": ["path-is-absolute@1.0.1", "", {}, "sha512-AVbw3UJ2e9bq64vSaS9Am0fje1Pa8pbGqTTsmXfaIiMpnr5DlDhfJOuLj9Sf95ZPVDAUerDfEk88MPmPe7UCQg=="],
|
||||
@@ -1379,6 +1405,8 @@
|
||||
|
||||
"triple-beam": ["triple-beam@1.4.1", "", {}, "sha512-aZbgViZrg1QNcG+LULa7nhZpJTZSLm/mXnHXnbAbjmN5aSa0y7V+wvv6+4WaBtpISJzThKy+PIPxc1Nq1EJ9mg=="],
|
||||
|
||||
"ts-morph": ["ts-morph@28.0.0", "", { "dependencies": { "@ts-morph/common": "~0.29.0", "code-block-writer": "^13.0.3" } }, "sha512-Wp3tnZ2bzwxyTZMtgWVzXDfm7lB1Drz+y9DmmYH/L702PQhPyVrp3pkou3yIz4qjS14GY9kcpmLiOOMvl8oG1g=="],
|
||||
|
||||
"tslib": ["tslib@2.8.1", "", {}, "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w=="],
|
||||
|
||||
"turndown": ["turndown@7.2.4", "", { "dependencies": { "@mixmark-io/domino": "^2.2.0" } }, "sha512-I8yFsfRzmzK0WV1pNNOA4A7y4RDfFxPRxb3t+e3ui14qSGOxGtiSP6GjeX+Y6CHb7HYaFj7ECUD7VE5kQMZWGQ=="],
|
||||
|
||||
@@ -540,6 +540,8 @@ The built-in model policy currently links OpenAI `codex-spark` variants to `gpt-
|
||||
|
||||
The `compat` block on a provider or model overrides the URL-based auto-detection in `packages/catalog/src/compat/openai.ts` (`buildOpenAICompat`). It is validated by `OpenAICompatSchema` in `packages/coding-agent/src/config/models-config-schema.ts` and consumed by every `openai-completions` transport (`packages/ai/src/providers/openai-completions.ts`). The canonical type is `OpenAICompat` in `packages/catalog/src/types.ts`.
|
||||
|
||||
Endpoint-specific exceptions that interact with these fields are cataloged in [Provider endpoint constraints](./provider-endpoint-constraints.md).
|
||||
|
||||
`models.yml` accepts the following keys (all optional; unset falls back to URL detection):
|
||||
|
||||
Request shaping:
|
||||
|
||||
@@ -0,0 +1,394 @@
|
||||
# Provider endpoint constraints
|
||||
|
||||
Provider integrations are not interchangeable just because they speak an
|
||||
OpenAI-shaped HTTP protocol. A request is shaped by four layers at once:
|
||||
|
||||
1. endpoint family: `openai-completions`, `openai-responses`,
|
||||
`openai-codex-responses`, `anthropic-messages`, etc.
|
||||
2. gateway/auth surface: OpenRouter, Vercel AI Gateway, Azure OpenAI, Copilot,
|
||||
Alibaba Coding Plan, Kimi Code, Fireworks/Firepass, and similar hosts
|
||||
3. model metadata and `compat` overrides
|
||||
4. request context: tools, images, reasoning mode, stateful session, service tier
|
||||
|
||||
Use this page when adding a provider, adding a compat flag, or moving logic out
|
||||
of a provider-specific branch. The goal is to encode endpoint constraints once,
|
||||
at the narrowest layer that actually owns the behavior.
|
||||
|
||||
Related references:
|
||||
|
||||
- [Providers](./providers.md) — provider availability, credentials, custom providers
|
||||
- [Model and Provider Configuration](./models.md) — `models.yml`, routing, and compat fields
|
||||
- [Provider streaming internals](./provider-streaming-internals.md) — stream event normalization
|
||||
- [Adding a provider](./adding-a-provider.md) — catalog/auth wiring for a new provider
|
||||
|
||||
## Baseline rules
|
||||
|
||||
- Prefer compat metadata over provider-name branches when behavior is model or
|
||||
endpoint configurable.
|
||||
- Keep transport mechanics transport-local. Codex websocket replay, Responses
|
||||
item routing, and Chat Completions SSE decoding are protocol behavior, not
|
||||
generic compat flags.
|
||||
- Scope fallbacks to the failing capability. A strict-tool failure should not
|
||||
disable unrelated features. A stale Responses chain should reset chain state,
|
||||
not disable Responses entirely.
|
||||
- Do not emit defaults that alter gateway routing. OpenRouter is the known case
|
||||
for default `max_tokens`, but any gateway can treat optional fields as routing
|
||||
hints.
|
||||
- Stop retrying after visible side effects. Once text or a tool call is visible
|
||||
to the user/session, retry policy must avoid duplicate output and duplicate
|
||||
tool execution.
|
||||
|
||||
## 1. Choose the endpoint family first
|
||||
|
||||
### OpenAI Chat Completions compatible
|
||||
|
||||
Preserve these differences instead of treating every host as stock OpenAI:
|
||||
|
||||
- `stream_options.include_usage` is only safe when compat says streaming usage
|
||||
is supported.
|
||||
- `store: false` is accepted only by some hosts.
|
||||
- max-output caps use either `max_tokens` or `max_completion_tokens`.
|
||||
- stop sequences and frequency penalty live on this path among the current
|
||||
OpenAI-like endpoint set.
|
||||
- OpenRouter-style reasoning and routing fields are not portable to other
|
||||
OpenAI-compatible hosts unless compat says so.
|
||||
|
||||
### OpenAI Responses compatible
|
||||
|
||||
Responses request shape is its own dialect:
|
||||
|
||||
- uses `input`, `instructions`, `store`, `prompt_cache_key`, optional
|
||||
`previous_response_id`, and `max_output_tokens`
|
||||
- can default official OpenAI requests to stateful chaining with
|
||||
`previous_response_id` plus `store: true`
|
||||
- third-party Responses proxies may reject native reasoning history, encrypted
|
||||
reasoning replay, or `previous_response_id`
|
||||
- stream completion is authoritative only after `response.completed` or
|
||||
`response.incomplete`; a stream close before either terminal event should fail
|
||||
for OpenAI Responses rather than surface partial output as success
|
||||
|
||||
### OpenAI Codex Responses
|
||||
|
||||
Codex is not plain Responses with a different URL. Keep these as Codex transport
|
||||
policy:
|
||||
|
||||
- Codex account headers and beta headers
|
||||
- `x-codex-turn-state` and `x-models-etag`
|
||||
- optional websocket transport plus SSE fallback
|
||||
- `responsesLite`
|
||||
- prompt-cache/session ids used as transport state
|
||||
- websocket-only `previous_response_id` chaining; SSE never chains
|
||||
- Codex retry/replay rules, including reconnect and SSE replay boundaries
|
||||
- provider retry only before user-visible content has been emitted
|
||||
- whitespace-only tool-call argument loop breaker
|
||||
|
||||
Codex intentionally does not forward caller max-token caps because the backend
|
||||
rejects them.
|
||||
|
||||
### Anthropic/OpenAI dual-surface providers
|
||||
|
||||
Kimi Code and Synthetic can be called as OpenAI-compatible or
|
||||
Anthropic-compatible. The shim may need to:
|
||||
|
||||
- switch `format`
|
||||
- rebuild an Anthropic model when needed
|
||||
- map internal reasoning to Anthropic thinking budgets
|
||||
- delegate back to OpenAI Completions
|
||||
|
||||
Do not encode these as one-way provider migrations; they are runtime surface
|
||||
selection decisions.
|
||||
|
||||
## 2. Apply gateway and auth overlays
|
||||
|
||||
These constraints sit above the endpoint family. They affect auth, headers,
|
||||
routing, model ids, or usage accounting.
|
||||
|
||||
### Azure OpenAI
|
||||
|
||||
- Chat Completions base URL reshapes to
|
||||
`/deployments/{deployment}/chat/completions?api-version=...`.
|
||||
- Deployment names may differ from model ids through
|
||||
`AZURE_OPENAI_DEPLOYMENT_NAME_MAP`.
|
||||
|
||||
### GitHub Copilot
|
||||
|
||||
- The API key is parsed into an access token.
|
||||
- Dynamic Copilot headers depend on messages/images.
|
||||
- `premiumRequests` must survive usage population and replacement.
|
||||
- Base URL may be resolved from the raw key.
|
||||
|
||||
### OpenRouter
|
||||
|
||||
- Adds attribution/cache headers.
|
||||
- Supports routing suffixes such as `:nitro` and `:floor`.
|
||||
- Appends a routing suffix only when the model id has no explicit suffix after
|
||||
the last provider path segment.
|
||||
- Uses nested `reasoning` request fields.
|
||||
- Routes providers through the OpenRouter `provider` object.
|
||||
- Has special cache-write usage accounting.
|
||||
- Has strict-tool fallback for Anthropic grammar-size failures.
|
||||
- Should omit catalog-default `max_tokens` unless the caller explicitly set a
|
||||
cap, so upstream routing is not biased.
|
||||
|
||||
### Vercel AI Gateway
|
||||
|
||||
- Routing preferences go under `providerOptions.gateway.only` and
|
||||
`providerOptions.gateway.order`.
|
||||
- Do not reuse OpenRouter's `provider` object.
|
||||
|
||||
### Alibaba Coding Plan
|
||||
|
||||
- API key bytes may be JSON carrying `{ token, enterpriseUrl }`.
|
||||
- Auth and base URL resolution are provider-specific.
|
||||
|
||||
### Kimi Code
|
||||
|
||||
- The OpenAI-compatible path needs common Kimi headers.
|
||||
- It also participates in the OpenAI/Anthropic dual-surface shim.
|
||||
|
||||
### Fireworks and Firepass
|
||||
|
||||
- Wire model ids need provider-specific mapping.
|
||||
- Fireworks can conflict when DeepSeek-style `thinking` and OpenAI-style
|
||||
`reasoning_effort` are both present after extra body fields are merged.
|
||||
|
||||
## 3. Serialize request parameters by dialect
|
||||
|
||||
Check these before adding or forwarding a field:
|
||||
|
||||
- **Model id.** Some models resolve a wire id from reasoning effort.
|
||||
Firepass/Fireworks transform ids. OpenRouter suffix handling is path-segment
|
||||
aware.
|
||||
- **Max output tokens.** Kimi-family models may require a max-token field even
|
||||
when the caller did not set one. OpenRouter should omit catalog defaults unless
|
||||
explicit. Codex drops caller caps. Responses uses `max_output_tokens`; Chat
|
||||
Completions uses `max_tokens` or `max_completion_tokens`.
|
||||
- **Service tier.** Completions, Responses, and Codex all handle service tiers,
|
||||
but allowed values and pricing multipliers differ. Codex has a special
|
||||
priority multiplier for `gpt-5.5`.
|
||||
- **Prompt cache/session.** OpenAI Responses uses `prompt_cache_key`.
|
||||
OpenRouter Responses uses `session_id`. Codex uses prompt cache/session ids for
|
||||
transport state. Anthropic-style cache control requires `cache_control` on a
|
||||
text part.
|
||||
- **Stateful chaining.** Official OpenAI Responses may chain by default.
|
||||
Third-party endpoints generally should not. Codex chains only on websocket
|
||||
`response.create`.
|
||||
|
||||
## 4. Map reasoning and thinking explicitly
|
||||
|
||||
Reasoning fields are not interchangeable.
|
||||
|
||||
### OpenAI-style `reasoning_effort`
|
||||
|
||||
- Effort values come from compat/model metadata.
|
||||
- If reasoning is disabled but the host has no real off switch, map to the
|
||||
lowest supported effort rather than inventing an unsupported value.
|
||||
|
||||
### Responses `reasoning`
|
||||
|
||||
- Uses `reasoning: { effort, summary }`.
|
||||
- Can include `reasoning.encrypted_content` for replay.
|
||||
- xAI Grok models may require omitting `reasoning.effort`.
|
||||
- Some compat paths inject the GPT-5 `# Juice: 0 !important` developer scaffold.
|
||||
|
||||
### OpenRouter `reasoning`
|
||||
|
||||
- Uses nested `reasoning: { effort }`.
|
||||
- Disabling reasoning must send `reasoning: { enabled: false }`; OpenRouter can
|
||||
otherwise default reasoning models into thinking.
|
||||
|
||||
### Z.AI / GLM
|
||||
|
||||
- Uses `thinking: { type: "enabled" }` or
|
||||
`thinking: { type: "disabled" }`.
|
||||
- GLM 5.2 reasoning-effort models may also receive `reasoning_effort`.
|
||||
- Tool requests need `tool_stream: true`.
|
||||
|
||||
### Qwen
|
||||
|
||||
- One dialect uses top-level `enable_thinking`.
|
||||
- Another uses `chat_template_kwargs.enable_thinking`.
|
||||
|
||||
### Anthropic-compatible format
|
||||
|
||||
- Reasoning maps to Anthropic thinking enablement and thinking-budget tokens,
|
||||
not OpenAI-style fields.
|
||||
|
||||
### DeepSeek reasoning history
|
||||
|
||||
- DeepSeek-compatible reasoning models may require exact `reasoning_content`
|
||||
replay.
|
||||
- Some variants require replay on every assistant turn, not only tool-call turns.
|
||||
- Synthetic `"."` placeholders are acceptable for Kimi/OpenRouter-style compat,
|
||||
but not DeepSeek V4 exact replay.
|
||||
|
||||
### Reasoning plus tool choice
|
||||
|
||||
- DeepSeek reasoning models can reject `tool_choice` while thinking is enabled.
|
||||
- Kimi can reject forced tool choice while thinking is enabled.
|
||||
- Compat needs both policies: disable reasoning for any tool choice, and disable
|
||||
reasoning only for forced tool choice.
|
||||
|
||||
### xAI Grok through Responses/SuperGrok
|
||||
|
||||
Keep these independent:
|
||||
|
||||
- omit `reasoning.effort`
|
||||
- include or drop encrypted reasoning replay
|
||||
- filter reasoning-history wrappers
|
||||
|
||||
Some models reject only one of those fields; do not collapse them into one
|
||||
"Grok mode" branch.
|
||||
|
||||
## 5. Normalize tools and schemas per endpoint
|
||||
|
||||
### Strict tools
|
||||
|
||||
Strict schemas are not a universal capability:
|
||||
|
||||
- some providers support strict tools
|
||||
- some reject mixed strict/non-strict tools
|
||||
- some reject strictified schemas
|
||||
- OpenRouter Anthropic models can fail with “compiled grammar too large”
|
||||
|
||||
Retry-without-strict should be a compat recovery policy scoped to the current
|
||||
session/provider path.
|
||||
|
||||
### Responses and Codex custom tools
|
||||
|
||||
Responses and Codex both support freeform custom grammar tools for `apply_patch`.
|
||||
Both disable request-level parallel tool calls when any custom grammar tool is
|
||||
present. Responses additionally:
|
||||
|
||||
- sanitizes schemas differently
|
||||
- quarantines invalid enum/const schema contradictions
|
||||
- repairs orphan tool outputs into assistant notes
|
||||
- synthesizes placeholder outputs for orphan tool calls
|
||||
|
||||
Codex applies its own request transformation before sending.
|
||||
|
||||
### Tool choice
|
||||
|
||||
Before emitting `tool_choice`:
|
||||
|
||||
- confirm the endpoint supports it
|
||||
- downgrade forced choice to `auto` if forced choice is unsupported
|
||||
- drop `tool_choice: "none"` when no tools are emitted
|
||||
- drop forced named tool choice if that named tool was filtered out
|
||||
|
||||
### Anthropic through LiteLLM/Bedrock
|
||||
|
||||
- If history contains tool calls/results and `context.tools` is undefined, send
|
||||
`tools: []` as a sentinel.
|
||||
- If `context.tools = []`, treat it as explicit opt-out and do not emit the
|
||||
sentinel.
|
||||
|
||||
### Mistral / Devstral
|
||||
|
||||
- Tool-call ids must be exactly 9 alphanumeric characters.
|
||||
- Some flows need a synthetic assistant bridge after tool results before the next
|
||||
user message.
|
||||
|
||||
### Custom tool outputs
|
||||
|
||||
Responses/Codex must remember whether a call was `custom_tool_call`; the paired
|
||||
output must then be `custom_tool_call_output`, not `function_call_output`.
|
||||
|
||||
### MiniMax-compatible streaming arguments
|
||||
|
||||
Tool arguments can stream as objects instead of JSON strings. Deep-merge object
|
||||
deltas, then emit one final concat-safe JSON delta.
|
||||
|
||||
## 6. Convert messages and replay history safely
|
||||
|
||||
- **System/developer roles.** Reasoning models may require `developer`. Some
|
||||
providers do not support `developer` and must downgrade to `user`. Some reject
|
||||
multiple system messages and need coalescing.
|
||||
- **Responses system prompts.** Responses usually uses top-level `instructions`.
|
||||
Reasoning models that support `developer` put system prompts inline as
|
||||
developer messages.
|
||||
- **Assistant content.** Some OpenAI-compatible backends mirror array content
|
||||
literally, so assistant content is normalized to a string. Tool-call replay may
|
||||
require `content: ""` or `content: "."` instead of `null`.
|
||||
- **Thinking replay.** Some models want thinking as visible text. Others need a
|
||||
provider-specific reasoning field. Some permit synthetic placeholders; others
|
||||
need exact replay.
|
||||
- **Vision.** If the model/provider cannot accept images, convert image input and
|
||||
tool-result images to placeholders. Some Qwen/Dashscope-compatible modes are
|
||||
text-only even when the high-level model is multimodal.
|
||||
- **Native Responses history.** Native provider payload replay is model-bound.
|
||||
Strip or normalize foreign reasoning signatures. Shared code normalizes
|
||||
Responses pipe-separated tool ids, hashes foreign item ids, and can filter
|
||||
reasoning history.
|
||||
|
||||
## 7. Decode streams by provider behavior, not just schema
|
||||
|
||||
- **Generic OpenAI-compatible streams.** Keepalive chunks, role-only deltas, and
|
||||
empty `choices: []` are not progress. Idle watchdogs must not sleep forever
|
||||
because of them.
|
||||
- **Mistral Medium 3.5-style content.** `delta.content` can be an array/object of
|
||||
text parts, not a string; normalize it to text.
|
||||
- **DeepSeek via NVIDIA/native/proxies.** Some endpoints leak chat-template
|
||||
markers like `<|...|>` into visible content. Buffering is required because
|
||||
markers can be split across chunks.
|
||||
- **DeepSeek/template-leak tool calls.** Some providers leak tool-call markup in
|
||||
text while also producing structured tool calls. Markup healing belongs in the
|
||||
stream decoder policy, not endpoint business logic.
|
||||
- **MiniMax-M3 cumulative reasoning.** Reasoning deltas may be cumulative
|
||||
snapshots. Deduplicate by reasoning field signature.
|
||||
- **Responses streams.** Route parallel items by `output_index`, `item_id`,
|
||||
call-id aliases, and prefixed `fc_` aliases. Tolerate missing
|
||||
`content_part.added` or `output_item.added`. Finalize pending tool calls at the
|
||||
terminal event.
|
||||
- **Terminal behavior.** Chat Completions can break after `finish_reason` plus
|
||||
usage. Responses breaks on `response.completed` or `response.incomplete`. Tool
|
||||
calls with `stop` promote to `toolUse`. Codex/Responses `end_turn:false` maps
|
||||
to `pause_turn`.
|
||||
- **Ollama length failures.** `finish_reason: length` with no visible content is
|
||||
treated as context-window failure and mapped to an error.
|
||||
|
||||
## 8. Preserve usage and cost semantics
|
||||
|
||||
- OpenRouter `prompt_tokens_details.cache_write_tokens` is billed differently:
|
||||
subtract it from input tokens and emit it as cache-write usage.
|
||||
- DeepSeek native `prompt_cache_miss_tokens` is the billed input portion, not a
|
||||
separate cache-write charge. Do not double-count it.
|
||||
- GitHub Copilot `premiumRequests` must survive when usage is populated or
|
||||
replaced.
|
||||
- Responses and Codex both adjust cost by resolved service tier, but Codex uses
|
||||
different multipliers.
|
||||
|
||||
## 9. Implement recovery at the right boundary
|
||||
|
||||
- **Strict tool fallback.** `400`/`422` schema or strict-tool failures should
|
||||
disable strict tools for the appropriate session scope and retry non-strict.
|
||||
- **OpenAI Responses stateful fallback.** Stale, invalid, or unsupported
|
||||
`previous_response_id` resets chain state and retries with full context. Zero
|
||||
Data Retention disables chaining immediately.
|
||||
- **Codex websocket fallback.** Websocket connection errors, stale sockets,
|
||||
connection limits, retry-budget exhaustion, or unsafe partial output can
|
||||
trigger reconnect or SSE replay.
|
||||
- **Codex whitespace tool-loop breaker.** Codex can stream whitespace-only
|
||||
tool-call argument deltas indefinitely. Cap events/chars, drop the degenerate
|
||||
partial tool call, and retry only when safe.
|
||||
- **Codex `previous_response_id` fallback.** Stale or unsupported ids are chain
|
||||
breaks and retry with full context, but only for websocket because SSE never
|
||||
chains.
|
||||
- **Provider retry before content.** Codex retries retryable provider stream
|
||||
errors only before user-visible content has been emitted.
|
||||
|
||||
## 10. Checklist for a new constraint
|
||||
|
||||
Before adding a branch or compat field, answer these in order:
|
||||
|
||||
1. Is this endpoint-family behavior, gateway behavior, model behavior, or request
|
||||
context behavior?
|
||||
2. Can it be represented by existing `compat` metadata?
|
||||
3. If not, is a new compat field better than a provider-name branch?
|
||||
4. Does the field need provider-level defaults, model-level overrides, or both?
|
||||
5. Does it interact with tools, images, reasoning, stateful Responses chains, or
|
||||
service tier?
|
||||
6. Can retry happen before visible text/tool calls only?
|
||||
7. Does usage accounting still preserve cache reads/writes, billed input, service
|
||||
tier multipliers, and provider-specific counters such as Copilot
|
||||
`premiumRequests`?
|
||||
@@ -208,7 +208,7 @@ Provider-specific (not fully abstracted):
|
||||
- [`../../ai/src/utils/event-stream.ts`](../packages/ai/src/utils/event-stream.ts) — generic stream queue + final-result resolution.
|
||||
- [`../../ai/src/utils/json-parse.ts`](../packages/ai/src/utils/json-parse.ts) — partial JSON parsing for streamed tool arguments.
|
||||
- [`../../ai/src/providers/anthropic.ts`](../packages/ai/src/providers/anthropic.ts) — Anthropic event translation and tool JSON delta accumulation.
|
||||
- [`../../ai/src/providers/openai-responses.ts`](../packages/ai/src/providers/openai-responses.ts), [`openai-responses-shared.ts`](../packages/ai/src/providers/openai-responses-shared.ts), [`openai-codex-responses.ts`](../packages/ai/src/providers/openai-codex-responses.ts), [`azure-openai-responses.ts`](../packages/ai/src/providers/azure-openai-responses.ts) — Responses-family event translation and status mapping.
|
||||
- [`../../ai/src/providers/openai-responses.ts`](../packages/ai/src/providers/openai-responses.ts), [`openai-shared.ts`](../packages/ai/src/providers/openai-shared.ts), [`openai-codex-responses.ts`](../packages/ai/src/providers/openai-codex-responses.ts), [`azure-openai-responses.ts`](../packages/ai/src/providers/azure-openai-responses.ts) — Responses-family event translation and status mapping.
|
||||
- [`../../ai/src/providers/google.ts`](../packages/ai/src/providers/google.ts), [`google-gemini-cli.ts`](../packages/ai/src/providers/google-gemini-cli.ts), [`google-vertex.ts`](../packages/ai/src/providers/google-vertex.ts) — Gemini stream chunk-to-block translation variants.
|
||||
- [`../../ai/src/providers/google-shared.ts`](../packages/ai/src/providers/google-shared.ts) — Gemini finish-reason mapping and shared conversion rules.
|
||||
- [`../../ai/src/providers/amazon-bedrock.ts`](../packages/ai/src/providers/amazon-bedrock.ts), [`openai-completions.ts`](../packages/ai/src/providers/openai-completions.ts), [`ollama.ts`](../packages/ai/src/providers/ollama.ts), [`cursor.ts`](../packages/ai/src/providers/cursor.ts), [`pi-native-client.ts`](../packages/ai/src/providers/pi-native-client.ts) — additional built-in stream adapters using the same event contract.
|
||||
|
||||
+1
-1
@@ -4,7 +4,7 @@ Providers are the model backends `omp` can route requests to: Anthropic, OpenAI,
|
||||
|
||||
A **provider** is the account or backend namespace, such as `anthropic`, `openai`, `google`, or `ollama`. A **model** is a concrete model under that provider, selected as `provider/model-id`, such as `anthropic/claude-opus-4-6`. Disabling a provider removes every model under it from selection; if you only want to narrow individual models, use model settings instead.
|
||||
|
||||
This page covers how providers become available, how credentials are resolved, the provider/environment-variable map, local engines, disabling providers, and custom providers. For model selection and the full `models.yml` schema, see [Model and Provider Configuration](./models.md). For config-file locations and merge precedence, see [Settings](./settings.md). For credential storage and login flows in depth, see [Secrets and credentials](./secrets.md). For the complete environment-variable reference, see [Environment variables](./environment-variables.md). For local engine setup, see [Local models](./local-models.md). For context-file discovery providers, see [Context files](./context-files.md).
|
||||
This page covers how providers become available, how credentials are resolved, the provider/environment-variable map, local engines, disabling providers, and custom providers. For endpoint-specific request, reasoning, tool, stream, usage, and retry constraints, see [Provider endpoint constraints](./provider-endpoint-constraints.md). For model selection and the full `models.yml` schema, see [Model and Provider Configuration](./models.md). For config-file locations and merge precedence, see [Settings](./settings.md). For credential storage and login flows in depth, see [Secrets and credentials](./secrets.md). For the complete environment-variable reference, see [Environment variables](./environment-variables.md). For local engine setup, see [Local models](./local-models.md). For context-file discovery providers, see [Context files](./context-files.md).
|
||||
|
||||
## How `omp` decides a provider is available
|
||||
|
||||
|
||||
+6
-3
@@ -53,6 +53,7 @@
|
||||
"@types/turndown": "5.0.6",
|
||||
"@typescript/native-preview": "7.0.0-dev.20260609.1",
|
||||
"@xterm/headless": "^6.0.0",
|
||||
"arktype": "^2.2.0",
|
||||
"beautiful-mermaid": "^1.1.3",
|
||||
"chalk": "^5.6.2",
|
||||
"chart.js": "^4.5.1",
|
||||
@@ -79,6 +80,7 @@
|
||||
"regexp-tree": "^0.1.27",
|
||||
"solid-js": "^1.9.13",
|
||||
"tailwindcss": "^4.3.0",
|
||||
"ts-morph": "^28.0.0",
|
||||
"turndown": "7.2.4",
|
||||
"turndown-plugin-gfm": "1.0.2",
|
||||
"typescript": "^6.0.3",
|
||||
@@ -179,11 +181,12 @@
|
||||
},
|
||||
"devDependencies": {
|
||||
"@biomejs/biome": "catalog:",
|
||||
"prettier": "catalog:",
|
||||
"@types/bun": "catalog:",
|
||||
"@typescript/native-preview": "catalog:",
|
||||
"typescript": "catalog:",
|
||||
"lint-staged": "catalog:"
|
||||
"lint-staged": "catalog:",
|
||||
"prettier": "catalog:",
|
||||
"ts-morph": "catalog:",
|
||||
"typescript": "catalog:"
|
||||
},
|
||||
"lint-staged": {
|
||||
"*.{js,ts,jsx,tsx,json,jsonc,css}": "biome check --write --no-errors-on-unmatched"
|
||||
|
||||
@@ -13,7 +13,7 @@
|
||||
*/
|
||||
|
||||
import { ProviderHttpError } from "@oh-my-pi/pi-ai/errors";
|
||||
import { parseTextSignature } from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
|
||||
import { parseTextSignature } from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
import { transformMessages } from "@oh-my-pi/pi-ai/providers/transform-messages";
|
||||
import type { AssistantMessage, FetchImpl, Message, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import {
|
||||
|
||||
@@ -1,6 +1,27 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Added
|
||||
|
||||
- Added `getOpenRouterHeaders` utility to export standard OpenRouter integration headers
|
||||
|
||||
### Changed
|
||||
|
||||
- Replaced the dedicated `xai-responses` provider with a unified `openai-responses` path that handles xAI-specific reasoning effort stripping dynamically
|
||||
- Updated OpenAI Responses stream handling to throw a clearer error message when a stream closes without a terminal response event
|
||||
- Consolidated shared OpenAI-compatible routing and strict-tool fallback helpers across Chat Completions and Responses providers.
|
||||
- Consolidated the OpenAI-family provider stack: merged `openai-responses-shared` into `openai-shared` and removed the now-dead `openai-responses-shared` re-export shim; folded the three duplicated `service_tier` request blocks and the per-provider wire model-id transform into shared `applyOpenAIServiceTier`/`applyWireModelIdTransform` helpers; moved residual provider-name wire-quirk checks (DeepSeek special-token strip, cumulative reasoning deltas, Ollama empty-length context error, OpenAI tool-call-id cap, Fireworks thinking drop, OpenRouter/OpenAI Responses request fields) into resolved compat fields; shared the Responses stream per-block accumulation helpers plus the terminal pending-tool-call finalization (`finalizePendingResponsesToolCalls`) and toolUse/pause stop-reason promotion (`promoteResponsesToolUseStopReason`) between `processResponsesStream` and the Codex stream handler; and removed the redundant `getOpenAIResponsesCacheSessionId` alias in favor of `getOpenAIResponsesPromptCacheKey`.
|
||||
- Centralized OpenAI-family request-param policy into shared `resolveOpenAIOutputTokenParam` (output-token field selection, OpenRouter default-cap omission, `alwaysSendMaxTokens` defaulting, model/provider clamp), `applyOpenAIGatewayRouting` (OpenRouter `provider` + Vercel AI Gateway `providerOptions`), and `applyOpenAIExtraBody` (extra-body merge + Fireworks thinking drop) helpers used by both Chat Completions and Responses `buildParams`, and moved the Chat Completions reasoning/thinking dialect dispatch (`applyChatCompletionsReasoningParams` + `disableChatCompletionsReasoningForDialect`) plus the `OpenAICompletionsParams` request type into `openai-shared` alongside `applyResponsesReasoningParams`. As a consistency consequence, direct `streamOpenAIResponses` calls (bypassing `streamSimple`) now emit `max_output_tokens` for `alwaysSendMaxTokens` (Kimi-family) models even without a caller cap — matching Chat Completions and the value `streamSimple` already supplied.
|
||||
- Centralized OpenAI-family reasoning compat resolution behind a shared `resolveOpenAICompatPolicy` consumed by both Chat Completions and Responses request builders. Shared policy now drives tool-choice reasoning suppression, dialect-specific disable encoding, reasoning-history replay filters, encrypted-reasoning inclusion, Mistral/OpenAI tool-call-id modes, stream healing/DeepSeek token stripping, and xAI/OpenRouter cache-affinity wiring instead of endpoint-local provider/model checks.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Google Gemini CLI credential parsing to correctly prioritize `projectId` over `project_id` even when empty, and drop non-string values gracefully
|
||||
- Fixed OpenRouter Responses requests to omit default max token fields unless an explicit caller cap is provided, preventing upstream filtering issues
|
||||
- Fixed Chat Completions reasoning suppression (`disableReasoningOnToolChoice` / `disableReasoningOnForcedToolChoice`) to turn thinking off symmetrically across every dialect via a shared `disableChatCompletionsReasoningForDialect` helper. Previously the conflict path only deleted `reasoning_effort`/`reasoning` (and set Z.AI `thinking: { type: "disabled" }` on the forced branch alone), leaving Qwen `enable_thinking`, Qwen chat-template `chat_template_kwargs.enable_thinking`, and OpenRouter nested `reasoning` enabled — so those hosts could keep thinking on under forced/required tool choice and re-trip the incompatibility the policy guards against. OpenRouter is now set to `{ reasoning: { enabled: false } }` (not deleted, which OpenRouter treats as default-on).
|
||||
- Fixed OpenRouter Responses requests to send `session_id` from `sessionId` in the request body for sticky provider routing and observability grouping.
|
||||
- Fixed OpenRouter Responses request shaping to preserve provider routing, variant suffixes, caller header overrides, and strict-tool fallback behavior while omitting only unsafe default max-token caps.
|
||||
- Fixed OpenAI Responses stateful chaining so a non-ZDR stale `previous_response_id` retry keeps `store: true`: the full-context retry stays chainable on the next turn and the consecutive stale-failure circuit breaker trips after the configured limit instead of alternating cold turns. Zero Data Retention rejections still disable chaining on the first strike.
|
||||
|
||||
## [16.0.5] - 2026-06-17
|
||||
|
||||
@@ -3828,4 +3849,4 @@ _Dedicated to Peter's shoulder ([@steipete](https://twitter.com/steipete))_
|
||||
|
||||
## [0.9.4] - 2025-11-26
|
||||
|
||||
Initial release with multi-provider LLM support.
|
||||
Initial release with multi-provider LLM support.
|
||||
@@ -40,6 +40,7 @@
|
||||
"@bufbuild/protobuf": "catalog:",
|
||||
"@oh-my-pi/pi-catalog": "catalog:",
|
||||
"@oh-my-pi/pi-utils": "catalog:",
|
||||
"arktype": "catalog:",
|
||||
"partial-json": "catalog:",
|
||||
"zod": "catalog:"
|
||||
},
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
export { type ZodType, z } from "zod/v4";
|
||||
export { type, type Type } from "arktype";
|
||||
export * from "./api-registry";
|
||||
export * from "./auth-broker";
|
||||
export { type AuthGatewayBootOptions, type ModelResolver, startAuthGateway } from "./auth-gateway/server";
|
||||
@@ -38,6 +39,7 @@ export * from "./usage/openai-codex-reset";
|
||||
export * from "./usage/zai";
|
||||
export * from "./utils/anthropic-auth";
|
||||
export * from "./utils/event-stream";
|
||||
export * from "./utils/openrouter-headers";
|
||||
export * from "./utils/overflow";
|
||||
export * from "./utils/retry";
|
||||
export * from "./utils/schema";
|
||||
|
||||
@@ -0,0 +1,84 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { createCodexProviderStreamError, isRetryableCodexFailureEvent } from "../openai-codex-responses";
|
||||
|
||||
describe("isRetryableCodexFailureEvent", () => {
|
||||
it("classifies retryable codes from nested error.code, error.type, then rawEvent.code", () => {
|
||||
expect(isRetryableCodexFailureEvent({ error: { code: "server_error" } })).toBe(true);
|
||||
expect(isRetryableCodexFailureEvent({ error: { type: "internal_error" } })).toBe(true);
|
||||
expect(isRetryableCodexFailureEvent({ code: "model_error" })).toBe(true);
|
||||
});
|
||||
|
||||
it("prefers nested error.code over the top-level code (distinct precedence from the factory)", () => {
|
||||
// The error.code chain wins, so a non-retryable nested code with no retryable message
|
||||
// is NOT retryable even though the top-level code is retryable.
|
||||
expect(isRetryableCodexFailureEvent({ code: "server_error", error: { code: "bad_request" } })).toBe(false);
|
||||
});
|
||||
|
||||
it("detects retryable messages when the code is absent or unknown", () => {
|
||||
expect(isRetryableCodexFailureEvent({ message: "Please retry your request shortly" })).toBe(true);
|
||||
expect(isRetryableCodexFailureEvent({ code: "bad_request", message: "we are overloaded" })).toBe(true);
|
||||
expect(isRetryableCodexFailureEvent({ response: { message: "service unavailable" } })).toBe(true);
|
||||
});
|
||||
|
||||
it("returns false for non-retryable code and message", () => {
|
||||
expect(isRetryableCodexFailureEvent({ code: "bad_request", message: "invalid input" })).toBe(false);
|
||||
expect(isRetryableCodexFailureEvent({})).toBe(false);
|
||||
});
|
||||
|
||||
it("falls back to response.error when rawEvent.error is not an object", () => {
|
||||
expect(isRetryableCodexFailureEvent({ error: "boom", response: { error: { code: "server_error" } } })).toBe(true);
|
||||
});
|
||||
|
||||
it("ignores mistyped fields instead of failing the whole parse", () => {
|
||||
// A numeric top-level `code` must not poison parsing; the retryable message is still honored.
|
||||
expect(isRetryableCodexFailureEvent({ code: 500, message: "internal error while processing" })).toBe(true);
|
||||
// Same tolerance nested: a non-string error.code is dropped while a valid error.message survives.
|
||||
expect(isRetryableCodexFailureEvent({ error: { code: 123, message: "server error happened" } })).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe("createCodexProviderStreamError", () => {
|
||||
it("prefers top-level rawEvent.code over nested error code", () => {
|
||||
expect(createCodexProviderStreamError({ code: "outer_code", error: { code: "inner_code" } }).code).toBe(
|
||||
"outer_code",
|
||||
);
|
||||
});
|
||||
|
||||
it("falls back to nested error.code then error.type", () => {
|
||||
expect(createCodexProviderStreamError({ error: { code: "inner_code" } }).code).toBe("inner_code");
|
||||
expect(createCodexProviderStreamError({ error: { type: "inner_type" } }).code).toBe("inner_type");
|
||||
});
|
||||
|
||||
it("leaves code undefined when nothing supplies one", () => {
|
||||
expect(createCodexProviderStreamError({ message: "boom" }).code).toBeUndefined();
|
||||
});
|
||||
|
||||
it("marks retryable error events and formats them as error events", () => {
|
||||
const err = createCodexProviderStreamError({ type: "error", code: "server_error", message: "kaboom" });
|
||||
expect(err.retryable).toBe(true);
|
||||
expect(err.code).toBe("server_error");
|
||||
expect(err.message).toContain("error event");
|
||||
expect(err.message).toContain("kaboom");
|
||||
});
|
||||
|
||||
it("formats non-error failures via the response-failure path", () => {
|
||||
const err = createCodexProviderStreamError({
|
||||
type: "response.failed",
|
||||
response: { error: { message: "downstream blew up" } },
|
||||
});
|
||||
expect(err.retryable).toBe(false);
|
||||
expect(err.code).toBeUndefined();
|
||||
expect(err.message).toContain("response failed");
|
||||
expect(err.message).toContain("downstream blew up");
|
||||
});
|
||||
|
||||
it("falls back to response.error when rawEvent.error is not an object", () => {
|
||||
const err = createCodexProviderStreamError({
|
||||
error: "boom",
|
||||
response: { error: { code: "server_error", message: "nested boom" } },
|
||||
});
|
||||
expect(err.code).toBe("server_error");
|
||||
expect(err.retryable).toBe(true);
|
||||
expect(err.message).toContain("nested boom");
|
||||
});
|
||||
});
|
||||
@@ -8,10 +8,8 @@ import type {
|
||||
ServiceTier,
|
||||
StreamFunction,
|
||||
StreamOptions,
|
||||
Tool,
|
||||
ToolChoice,
|
||||
} from "../types";
|
||||
import { normalizeSystemPrompts } from "../utils";
|
||||
import { createAbortSourceTracker } from "../utils/abort";
|
||||
import { AssistantMessageEventStream } from "../utils/event-stream";
|
||||
import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
|
||||
@@ -23,44 +21,24 @@ import {
|
||||
import { postOpenAIStream } from "../utils/openai-http";
|
||||
import { sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
|
||||
import { mapToOpenAIResponsesToolChoice } from "../utils/tool-choice";
|
||||
import { getOpenAIResponsesCacheSessionId } from "./openai-responses";
|
||||
import type { ResponseCreateParamsStreaming, ResponseStreamEvent } from "./openai-responses-wire";
|
||||
import {
|
||||
appendResponsesToolResultMessages,
|
||||
applyCommonResponsesSamplingParams,
|
||||
applyResponsesReasoningParams,
|
||||
convertResponsesAssistantMessage,
|
||||
convertResponsesInputContent,
|
||||
buildResponsesInput,
|
||||
createInitialResponsesAssistantMessage,
|
||||
getOpenAIResponsesPromptCacheKey,
|
||||
isOpenAIResponsesProgressEvent,
|
||||
normalizeResponsesToolCallIdForTransform,
|
||||
parseAzureDeploymentNameMap,
|
||||
processResponsesStream,
|
||||
repairOrphanResponsesToolCalls,
|
||||
} from "./openai-responses-shared";
|
||||
import type {
|
||||
Tool as OpenAITool,
|
||||
ResponseCreateParamsStreaming,
|
||||
ResponseInput,
|
||||
ResponseStreamEvent,
|
||||
} from "./openai-responses-wire";
|
||||
import { transformMessages } from "./transform-messages";
|
||||
} from "./openai-shared";
|
||||
|
||||
export { parseAzureDeploymentNameMap } from "./openai-shared";
|
||||
|
||||
const DEFAULT_AZURE_API_VERSION = "v1";
|
||||
const AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE =
|
||||
"Azure OpenAI responses stream timed out while waiting for the first event";
|
||||
|
||||
export function parseAzureDeploymentNameMap(value: string | undefined): Map<string, string> {
|
||||
const map = new Map<string, string>();
|
||||
if (!value) return map;
|
||||
for (const entry of value.split(",")) {
|
||||
const trimmed = entry.trim();
|
||||
if (!trimmed) continue;
|
||||
const [modelId, deploymentName] = trimmed.split("=", 2);
|
||||
if (!modelId || !deploymentName) continue;
|
||||
map.set(modelId.trim(), deploymentName.trim());
|
||||
}
|
||||
return map;
|
||||
}
|
||||
|
||||
function resolveDeploymentName(model: Model<"azure-openai-responses">, options?: AzureOpenAIResponsesOptions): string {
|
||||
if (options?.azureDeploymentName) {
|
||||
return options.azureDeploymentName;
|
||||
@@ -193,13 +171,13 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
||||
abortSignal: options?.signal,
|
||||
isProgressItem: isOpenAIResponsesProgressEvent,
|
||||
});
|
||||
let sawCompleted = false;
|
||||
let sawTerminalResponseEvent = false;
|
||||
await processResponsesStream(timedOpenaiStream, output, stream, model, {
|
||||
onFirstToken: () => {
|
||||
if (!firstTokenTime) firstTokenTime = Date.now();
|
||||
},
|
||||
onCompleted: () => {
|
||||
sawCompleted = true;
|
||||
sawTerminalResponseEvent = true;
|
||||
},
|
||||
});
|
||||
|
||||
@@ -212,8 +190,8 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
||||
throw new Error("Request was aborted");
|
||||
}
|
||||
|
||||
if (!sawCompleted) {
|
||||
throw new Error("Azure OpenAI responses stream closed before response.completed was received");
|
||||
if (!sawTerminalResponseEvent) {
|
||||
throw new Error("Azure OpenAI responses stream closed before a terminal response event was received");
|
||||
}
|
||||
|
||||
if (output.stopReason === "aborted" || output.stopReason === "error") {
|
||||
@@ -240,14 +218,6 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
||||
return stream;
|
||||
};
|
||||
|
||||
function normalizeAzureBaseUrl(baseUrl: string): string {
|
||||
return baseUrl.replace(/\/+$/, "");
|
||||
}
|
||||
|
||||
function buildDefaultBaseUrl(resourceName: string): string {
|
||||
return `https://${resourceName}.openai.azure.com/openai/v1`;
|
||||
}
|
||||
|
||||
function resolveAzureConfig(
|
||||
model: Model<"azure-openai-responses">,
|
||||
options?: AzureOpenAIResponsesOptions,
|
||||
@@ -260,7 +230,7 @@ function resolveAzureConfig(
|
||||
let resolvedBaseUrl = baseUrl;
|
||||
|
||||
if (!resolvedBaseUrl && resourceName) {
|
||||
resolvedBaseUrl = buildDefaultBaseUrl(resourceName);
|
||||
resolvedBaseUrl = `https://${resourceName}.openai.azure.com/openai/v1`;
|
||||
}
|
||||
|
||||
if (!resolvedBaseUrl && model.baseUrl) {
|
||||
@@ -274,7 +244,7 @@ function resolveAzureConfig(
|
||||
}
|
||||
|
||||
return {
|
||||
baseUrl: normalizeAzureBaseUrl(resolvedBaseUrl),
|
||||
baseUrl: resolvedBaseUrl.replace(/\/+$/, ""),
|
||||
apiVersion,
|
||||
};
|
||||
}
|
||||
@@ -322,13 +292,21 @@ function buildParams(
|
||||
options: AzureOpenAIResponsesOptions | undefined,
|
||||
deploymentName: string,
|
||||
) {
|
||||
const messages = convertMessages(model, context, true);
|
||||
const systemRole = model.reasoning && model.compat.supportsDeveloperRole ? "developer" : "system";
|
||||
const messages = buildResponsesInput({
|
||||
model,
|
||||
context,
|
||||
strictResponsesPairing: true,
|
||||
systemRole,
|
||||
includeThinkingSignatures: true,
|
||||
developerStringContent: true,
|
||||
});
|
||||
|
||||
const params: AzureOpenAIResponsesSamplingParams = {
|
||||
model: deploymentName,
|
||||
input: messages,
|
||||
stream: true,
|
||||
prompt_cache_key: getOpenAIResponsesCacheSessionId(options),
|
||||
prompt_cache_key: getOpenAIResponsesPromptCacheKey(options),
|
||||
// Encrypted reasoning replay (applyResponsesReasoningParams) requires
|
||||
// stateless responses, matching the openai provider.
|
||||
store: false,
|
||||
@@ -337,7 +315,13 @@ function buildParams(
|
||||
applyCommonResponsesSamplingParams(params, options, model);
|
||||
|
||||
if (context.tools) {
|
||||
params.tools = convertTools(context.tools);
|
||||
params.tools = context.tools.map(tool => ({
|
||||
type: "function" as const,
|
||||
name: tool.name,
|
||||
description: tool.description || "",
|
||||
parameters: sanitizeSchemaForOpenAIResponses(toolWireSchema(tool)),
|
||||
strict: false,
|
||||
}));
|
||||
if (options?.toolChoice) {
|
||||
const toolChoice = mapToOpenAIResponsesToolChoice(options.toolChoice);
|
||||
if (
|
||||
@@ -355,60 +339,3 @@ function buildParams(
|
||||
|
||||
return params;
|
||||
}
|
||||
|
||||
function convertMessages(
|
||||
model: Model<"azure-openai-responses">,
|
||||
context: Context,
|
||||
strictResponsesPairing: boolean,
|
||||
): ResponseInput {
|
||||
const messages: ResponseInput = [];
|
||||
const transformedMessages = transformMessages(context.messages, model, normalizeResponsesToolCallIdForTransform);
|
||||
const knownCallIds = new Set<string>();
|
||||
const customCallIds = new Set<string>();
|
||||
|
||||
const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
|
||||
if (systemPrompts.length > 0) {
|
||||
const role = model.reasoning && model.compat.supportsDeveloperRole ? "developer" : "system";
|
||||
for (const systemPrompt of systemPrompts) {
|
||||
messages.push({ role, content: systemPrompt });
|
||||
}
|
||||
}
|
||||
|
||||
let msgIndex = 0;
|
||||
for (const msg of transformedMessages) {
|
||||
if (msg.role === "user" || msg.role === "developer") {
|
||||
const content = convertResponsesInputContent(msg.content, model.input.includes("image"));
|
||||
if (!content) continue;
|
||||
messages.push({
|
||||
role: "user",
|
||||
content: msg.role === "developer" && typeof msg.content === "string" ? msg.content.toWellFormed() : content,
|
||||
});
|
||||
} else if (msg.role === "assistant") {
|
||||
const outputItems = convertResponsesAssistantMessage(
|
||||
msg as AssistantMessage,
|
||||
model,
|
||||
msgIndex,
|
||||
knownCallIds,
|
||||
true,
|
||||
customCallIds,
|
||||
);
|
||||
if (outputItems.length === 0) continue;
|
||||
messages.push(...outputItems);
|
||||
} else if (msg.role === "toolResult") {
|
||||
appendResponsesToolResultMessages(messages, msg, model, strictResponsesPairing, knownCallIds, customCallIds);
|
||||
}
|
||||
msgIndex++;
|
||||
}
|
||||
|
||||
return repairOrphanResponsesToolCalls(messages);
|
||||
}
|
||||
|
||||
function convertTools(tools: Tool[]): OpenAITool[] {
|
||||
return tools.map(tool => ({
|
||||
type: "function",
|
||||
name: tool.name,
|
||||
description: tool.description || "",
|
||||
parameters: sanitizeSchemaForOpenAIResponses(toolWireSchema(tool)),
|
||||
strict: false,
|
||||
}));
|
||||
}
|
||||
|
||||
@@ -13,6 +13,7 @@ import {
|
||||
getGeminiCliHeaders,
|
||||
} from "@oh-my-pi/pi-catalog/wire/gemini-headers";
|
||||
import { extractHttpStatusFromError, fetchWithRetry, readSseJson } from "@oh-my-pi/pi-utils";
|
||||
import { z } from "zod/v4";
|
||||
import { ProviderHttpError } from "../errors";
|
||||
import type {
|
||||
Api,
|
||||
@@ -185,16 +186,22 @@ function extractErrorMessage(errorText: string): string {
|
||||
return errorText;
|
||||
}
|
||||
|
||||
interface GeminiCliApiKeyPayload {
|
||||
token?: unknown;
|
||||
projectId?: unknown;
|
||||
project_id?: unknown;
|
||||
refreshToken?: unknown;
|
||||
expiresAt?: unknown;
|
||||
email?: unknown;
|
||||
refresh?: unknown;
|
||||
expires?: unknown;
|
||||
}
|
||||
const optionalCredentialString = z.string().optional().catch(undefined);
|
||||
|
||||
const geminiCliCredentialsSchema = z
|
||||
.object({
|
||||
token: optionalCredentialString,
|
||||
projectId: optionalCredentialString,
|
||||
project_id: optionalCredentialString,
|
||||
refreshToken: optionalCredentialString,
|
||||
refresh: optionalCredentialString,
|
||||
email: optionalCredentialString,
|
||||
expiresAt: z.unknown().optional(),
|
||||
expires: z.unknown().optional(),
|
||||
})
|
||||
.loose()
|
||||
.catch({});
|
||||
|
||||
interface ParsedGeminiCliCredentials {
|
||||
accessToken: string;
|
||||
projectId: string;
|
||||
@@ -215,32 +222,22 @@ export function parseGeminiCliCredentials(apiKeyRaw: string): ParsedGeminiCliCre
|
||||
const missingCredentialsMessage =
|
||||
"Missing token or projectId in Google Cloud credentials. Use /login to re-authenticate.";
|
||||
|
||||
let parsed: GeminiCliApiKeyPayload;
|
||||
let rawCredentials: unknown;
|
||||
try {
|
||||
parsed = JSON.parse(apiKeyRaw) as GeminiCliApiKeyPayload;
|
||||
rawCredentials = JSON.parse(apiKeyRaw);
|
||||
} catch {
|
||||
throw new Error(invalidCredentialsMessage);
|
||||
}
|
||||
const parsed = geminiCliCredentialsSchema.parse(rawCredentials);
|
||||
|
||||
const projectId =
|
||||
typeof parsed.projectId === "string"
|
||||
? parsed.projectId
|
||||
: typeof parsed.project_id === "string"
|
||||
? parsed.project_id
|
||||
: undefined;
|
||||
|
||||
if (typeof parsed.token !== "string" || typeof projectId !== "string") {
|
||||
const projectId = parsed.projectId ?? parsed.project_id;
|
||||
if (parsed.token === undefined || projectId === undefined) {
|
||||
throw new Error(missingCredentialsMessage);
|
||||
}
|
||||
|
||||
const refreshToken =
|
||||
typeof parsed.refreshToken === "string"
|
||||
? parsed.refreshToken
|
||||
: typeof parsed.refresh === "string"
|
||||
? parsed.refresh
|
||||
: undefined;
|
||||
const refreshToken = parsed.refreshToken ?? parsed.refresh;
|
||||
const expiresAt = normalizeExpiryMs(parsed.expiresAt ?? parsed.expires);
|
||||
const email = typeof parsed.email === "string" && parsed.email.length > 0 ? parsed.email : undefined;
|
||||
const email = parsed.email && parsed.email.length > 0 ? parsed.email : undefined;
|
||||
|
||||
return {
|
||||
accessToken: parsed.token,
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -57,6 +57,7 @@ export interface RequestBody {
|
||||
client_metadata?: Record<string, string>;
|
||||
max_output_tokens?: number;
|
||||
max_completion_tokens?: number;
|
||||
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
|
||||
[key: string]: unknown;
|
||||
}
|
||||
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -360,7 +360,7 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest {
|
||||
if (effectiveType === "custom_tool_call") {
|
||||
const call = item as { id?: string; call_id: string; name: string; input: string };
|
||||
// Custom tools carry a raw input string. We stash it in `arguments.input`
|
||||
// matching pi-ai's openai-responses-shared convention, and tag the call
|
||||
// matching pi-ai's openai-shared convention, and tag the call
|
||||
// with `customWireName` so encoders re-emit it as `custom_tool_call`.
|
||||
const toolCall: ToolCall = {
|
||||
type: "toolCall",
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,12 +1,11 @@
|
||||
import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts";
|
||||
import { parseGitHubCopilotApiKey } from "@oh-my-pi/pi-catalog/wire/github-copilot";
|
||||
import { $env, $flag, extractHttpStatusFromError, logger, structuredCloneJSON } from "@oh-my-pi/pi-utils";
|
||||
import { $flag, extractHttpStatusFromError, logger, structuredCloneJSON } from "@oh-my-pi/pi-utils";
|
||||
import { getEnvApiKey } from "../stream";
|
||||
import type {
|
||||
AssistantMessage,
|
||||
Context,
|
||||
MessageAttribution,
|
||||
Model,
|
||||
OpenAICompat,
|
||||
ProviderSessionState,
|
||||
RawSseEvent,
|
||||
ServiceTier,
|
||||
@@ -17,8 +16,6 @@ import type {
|
||||
} from "../types";
|
||||
import {
|
||||
createOpenAIResponsesHistoryPayload,
|
||||
getOpenAIResponsesHistoryItems,
|
||||
getOpenAIResponsesHistoryPayload,
|
||||
normalizeSystemPrompts,
|
||||
resolveCacheRetention,
|
||||
sanitizeOpenAIResponsesHistoryItemsForReplay,
|
||||
@@ -31,7 +28,7 @@ import {
|
||||
getOpenAIStreamIdleTimeoutMs,
|
||||
iterateWithIdleTimeout,
|
||||
} from "../utils/idle-iterator";
|
||||
import { postOpenAIStream } from "../utils/openai-http";
|
||||
import { OpenAIHttpError, postOpenAIStream } from "../utils/openai-http";
|
||||
import { notifyProviderResponse } from "../utils/provider-response";
|
||||
import { callWithCopilotModelRetry } from "../utils/retry";
|
||||
import {
|
||||
@@ -42,42 +39,41 @@ import {
|
||||
toolWireSchema,
|
||||
} from "../utils/schema";
|
||||
import { mapToOpenAIResponsesToolChoice, type OpenAIResponsesToolChoice } from "../utils/tool-choice";
|
||||
import {
|
||||
buildCopilotDynamicHeaders,
|
||||
hasCopilotVisionInput,
|
||||
resolveGitHubCopilotBaseUrl,
|
||||
} from "./github-copilot-headers";
|
||||
import { compactGrammarDefinition } from "./grammar";
|
||||
import {
|
||||
appendResponsesToolResultMessages,
|
||||
applyCommonResponsesSamplingParams,
|
||||
applyResponsesReasoningParams,
|
||||
buildResponsesDeltaInput,
|
||||
collectCustomCallIds,
|
||||
collectKnownCallIds,
|
||||
convertResponsesAssistantMessage,
|
||||
convertResponsesInputContent,
|
||||
createInitialResponsesAssistantMessage,
|
||||
isOpenAIResponsesProgressEvent,
|
||||
normalizeResponsesToolCallIdForTransform,
|
||||
processResponsesStream,
|
||||
repairOrphanResponsesToolCalls,
|
||||
repairOrphanResponsesToolOutputs,
|
||||
} from "./openai-responses-shared";
|
||||
import type {
|
||||
Tool as OpenAITool,
|
||||
ResponseCreateParamsStreaming,
|
||||
ResponseInput,
|
||||
ResponseStreamEvent,
|
||||
} from "./openai-responses-wire";
|
||||
import { transformMessages } from "./transform-messages";
|
||||
|
||||
export function normalizeOpenAIResponsesPromptCacheKey(sessionId: string | undefined): string | undefined {
|
||||
if (!sessionId || sessionId.length === 0) return undefined;
|
||||
const wellFormed = sessionId.toWellFormed();
|
||||
if (wellFormed.length <= 64) return wellFormed;
|
||||
return `pc_${Bun.hash(wellFormed).toString(36)}`;
|
||||
}
|
||||
import {
|
||||
applyCommonResponsesSamplingParams,
|
||||
applyOpenAIExtraBody,
|
||||
applyOpenAIGatewayRouting,
|
||||
applyResponsesCompatPolicy,
|
||||
applyWireModelIdTransform,
|
||||
buildResponsesDeltaInput,
|
||||
buildResponsesInput,
|
||||
clearOpenAIStrictToolsState,
|
||||
createInitialResponsesAssistantMessage,
|
||||
createOpenAIStrictToolsState,
|
||||
disableStrictToolsForScope,
|
||||
getOpenAIResponsesPromptCacheKey,
|
||||
getOpenAIResponsesRoutingSessionId,
|
||||
getOpenAIStrictToolsScope,
|
||||
getOpenRouterResponsesSessionId,
|
||||
isCompiledGrammarTooLargeStrictError,
|
||||
isOpenAIResponsesProgressEvent,
|
||||
isOpenRouterAnthropicModel,
|
||||
isStrictToolsDisabledForScope,
|
||||
type OpenAIStrictToolsScope,
|
||||
type OpenAIStrictToolsState,
|
||||
processResponsesStream,
|
||||
resolveOpenAIOutputTokenParam,
|
||||
resolveOpenAICompatPolicy,
|
||||
resolveOpenAIRequestSetup,
|
||||
shouldRetryWithoutStrictTools,
|
||||
} from "./openai-shared";
|
||||
|
||||
// OpenAI Responses-specific options
|
||||
export interface OpenAIResponsesOptions extends StreamOptions {
|
||||
@@ -85,6 +81,9 @@ export interface OpenAIResponsesOptions extends StreamOptions {
|
||||
reasoningSummary?: "auto" | "detailed" | "concise" | null;
|
||||
serviceTier?: ServiceTier;
|
||||
toolChoice?: ToolChoice;
|
||||
openrouterVariant?: string;
|
||||
maxTokensExplicit?: boolean;
|
||||
disableReasoning?: boolean;
|
||||
/**
|
||||
* Stateful turns: chain via `previous_response_id` + delta input instead of
|
||||
* replaying the full transcript. Forces `store: true` (the platform only
|
||||
@@ -96,29 +95,25 @@ export interface OpenAIResponsesOptions extends StreamOptions {
|
||||
*/
|
||||
statefulResponses?: boolean;
|
||||
/**
|
||||
* Enforce strict tool call/result pairing when building Responses API inputs.
|
||||
* Azure OpenAI and GitHub Copilot Responses paths require tool results to match prior tool calls.
|
||||
* Override catalog compat for strict tool call/result pairing when building
|
||||
* Responses API inputs. Default behavior is catalog compat; this is only for
|
||||
* debugging/adapter wrappers.
|
||||
*/
|
||||
strictResponsesPairing?: boolean;
|
||||
/**
|
||||
* Pass `include: ["reasoning.encrypted_content"]` on requests when the
|
||||
* model supports reasoning. Default: true (preserves current behavior).
|
||||
* Set to false when the upstream Responses endpoint rejects replayed
|
||||
* encrypted reasoning (e.g., xAI Grok under SuperGrok OAuth).
|
||||
* Override catalog compat for `include: ["reasoning.encrypted_content"]`.
|
||||
* Default behavior is catalog compat; this is only for debugging/adapter wrappers.
|
||||
*/
|
||||
includeEncryptedReasoning?: boolean;
|
||||
/**
|
||||
* Strip `type: "reasoning"` items from replayed conversation history
|
||||
* before they hit the wire. Default: false (preserves current behavior).
|
||||
* Set to true when the upstream rejects replayed reasoning wrappers.
|
||||
* Override catalog compat for stripping `type: "reasoning"` items from
|
||||
* replayed conversation history before request encoding. Default behavior is
|
||||
* catalog compat; this is only for debugging/adapter wrappers.
|
||||
*/
|
||||
filterReasoningHistory?: boolean;
|
||||
/**
|
||||
* Suppress the `reasoning.effort` wire param when set, even if
|
||||
* `options.reasoning` is requested. Default: false. xAI Grok models
|
||||
* outside the effort-capable allowlist 400 with "Model X does not
|
||||
* support parameter reasoningEffort" — the xAI Responses adapter sets
|
||||
* this when the target model is not in GROK_EFFORT_CAPABLE_PREFIXES.
|
||||
* Override catalog compat for suppressing the `reasoning.effort` wire param.
|
||||
* Default behavior is catalog compat; this is only for debugging/adapter wrappers.
|
||||
*/
|
||||
omitReasoningEffort?: boolean;
|
||||
/**
|
||||
@@ -141,7 +136,7 @@ const OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE =
|
||||
/** Consecutive stale-previous-response failures before chaining is disabled for the session. */
|
||||
const OPENAI_RESPONSES_CHAIN_STALE_FAILURE_LIMIT = 3;
|
||||
|
||||
interface OpenAIResponsesProviderSessionState extends ProviderSessionState {
|
||||
interface OpenAIResponsesProviderSessionState extends ProviderSessionState, OpenAIStrictToolsState {
|
||||
nativeHistoryReplayWarmed: boolean;
|
||||
/** Stateful `previous_response_id` chain baselines, keyed by baseUrl/model/session. */
|
||||
chains: Map<string, OpenAIResponsesChainState>;
|
||||
@@ -164,27 +159,26 @@ interface OpenAIResponsesChainState {
|
||||
}
|
||||
|
||||
function createOpenAIResponsesProviderSessionState(): OpenAIResponsesProviderSessionState {
|
||||
const strictToolsState = createOpenAIStrictToolsState();
|
||||
const state: OpenAIResponsesProviderSessionState = {
|
||||
...strictToolsState,
|
||||
nativeHistoryReplayWarmed: false,
|
||||
chains: new Map(),
|
||||
close: () => {
|
||||
state.nativeHistoryReplayWarmed = false;
|
||||
state.chains.clear();
|
||||
clearOpenAIStrictToolsState(state);
|
||||
},
|
||||
};
|
||||
return state;
|
||||
}
|
||||
|
||||
function getOpenAIResponsesProviderSessionStateKey(model: Model<"openai-responses">): string {
|
||||
return `${OPENAI_RESPONSES_PROVIDER_SESSION_STATE_PREFIX}${model.provider}`;
|
||||
}
|
||||
|
||||
function getOpenAIResponsesProviderSessionState(
|
||||
model: Model<"openai-responses">,
|
||||
providerSessionState: Map<string, ProviderSessionState> | undefined,
|
||||
): OpenAIResponsesProviderSessionState | undefined {
|
||||
if (!providerSessionState) return undefined;
|
||||
const key = getOpenAIResponsesProviderSessionStateKey(model);
|
||||
const key = `${OPENAI_RESPONSES_PROVIDER_SESSION_STATE_PREFIX}${model.provider}`;
|
||||
const existing = providerSessionState.get(key) as OpenAIResponsesProviderSessionState | undefined;
|
||||
if (existing) return existing;
|
||||
const created = createOpenAIResponsesProviderSessionState();
|
||||
@@ -192,12 +186,6 @@ function getOpenAIResponsesProviderSessionState(
|
||||
return created;
|
||||
}
|
||||
|
||||
function canReplayOpenAIResponsesNativeHistory(
|
||||
providerSessionState: OpenAIResponsesProviderSessionState | undefined,
|
||||
): boolean {
|
||||
return providerSessionState?.nativeHistoryReplayWarmed ?? true;
|
||||
}
|
||||
|
||||
function isOpenAIResponsesStatefulEnabled(
|
||||
options: OpenAIResponsesOptions | undefined,
|
||||
baseUrl: string | undefined,
|
||||
@@ -214,9 +202,10 @@ function isOpenAIResponsesStatefulEnabled(
|
||||
function getOpenAIResponsesChainState(
|
||||
providerSessionState: OpenAIResponsesProviderSessionState,
|
||||
model: Model<"openai-responses">,
|
||||
resolvedBaseUrl: string | undefined,
|
||||
sessionId: string,
|
||||
): OpenAIResponsesChainState {
|
||||
const key = `${model.baseUrl ?? ""}\u0000${model.id}\u0000${sessionId}`;
|
||||
const key = `${resolvedBaseUrl ?? model.baseUrl ?? ""}\u0000${model.id}\u0000${sessionId}`;
|
||||
const existing = providerSessionState.chains.get(key);
|
||||
if (existing) return existing;
|
||||
const created: OpenAIResponsesChainState = { canAppend: false, staleFailures: 0, disabled: false };
|
||||
@@ -237,18 +226,6 @@ interface OpenAIResponsesChainedParams {
|
||||
previousResponseId?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Drop the per-turn trailing scaffolding (the GPT-5 "Juice: 0" developer item)
|
||||
* from `input`, yielding the wire form of the conversation arguments alone.
|
||||
*/
|
||||
function stripTrailingScaffolding(
|
||||
params: OpenAIResponsesSamplingParams,
|
||||
trailingScaffoldingItems: number,
|
||||
): OpenAIResponsesSamplingParams {
|
||||
if (trailingScaffoldingItems <= 0 || !Array.isArray(params.input)) return params;
|
||||
return { ...params, input: params.input.slice(0, params.input.length - trailingScaffoldingItems) };
|
||||
}
|
||||
|
||||
/**
|
||||
* Shape the next turn's request: when the session's append baseline is intact
|
||||
* (same options, strict history prefix), chain via `previous_response_id` +
|
||||
@@ -264,7 +241,10 @@ function buildOpenAIResponsesChainedParams(
|
||||
trailingScaffoldingItems: number,
|
||||
chain: OpenAIResponsesChainState,
|
||||
): OpenAIResponsesChainedParams {
|
||||
const historyParams = stripTrailingScaffolding(params, trailingScaffoldingItems);
|
||||
const historyParams =
|
||||
trailingScaffoldingItems > 0 && Array.isArray(params.input)
|
||||
? { ...params, input: params.input.slice(0, params.input.length - trailingScaffoldingItems) }
|
||||
: params;
|
||||
const deltaInput = chain.canAppend
|
||||
? buildResponsesDeltaInput<ResponseInput[number]>(chain.lastParams, chain.lastResponseItems, historyParams)
|
||||
: null;
|
||||
@@ -296,18 +276,6 @@ function isOpenAIResponsesStalePreviousResponseError(error: unknown): boolean {
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Zero Data Retention orgs accept `store: true` but refuse to resolve any
|
||||
* `previous_response_id` — the prior response was never persisted server-side.
|
||||
* The 400 carries a fixed phrasing ("Zero Data Retention") that the generic
|
||||
* stale-id regex above does not match, so it is classified separately and
|
||||
* disables chaining categorically (one strike, not three).
|
||||
*/
|
||||
function isOpenAIResponsesZeroDataRetentionError(error: unknown): boolean {
|
||||
if (!(error instanceof Error)) return false;
|
||||
return /previous[ _]?response/i.test(error.message) && /zero[ _-]?data[ _-]?retention/i.test(error.message);
|
||||
}
|
||||
|
||||
function registerOpenAIResponsesChainStaleFailure(chain: OpenAIResponsesChainState, error: unknown): void {
|
||||
resetOpenAIResponsesChainState(chain);
|
||||
chain.staleFailures += 1;
|
||||
@@ -340,7 +308,10 @@ type OpenAIResponsesSamplingParams = ResponseCreateParamsStreaming & {
|
||||
min_p?: number;
|
||||
presence_penalty?: number;
|
||||
repetition_penalty?: number;
|
||||
session_id?: string;
|
||||
stream_options?: { include_obfuscation?: boolean };
|
||||
provider?: OpenAICompat["openRouterRouting"];
|
||||
reasoning?: { effort?: string } | { enabled: false };
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -396,19 +367,26 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
||||
// stable prompt-cache key independently. Side-channel calls use this to
|
||||
// avoid perturbing provider conversation state without cold-starting the cache.
|
||||
const routingSessionId = getOpenAIResponsesRoutingSessionId(options);
|
||||
const promptCacheSessionId = getOpenAIResponsesPromptCacheKey(options);
|
||||
const apiKey = options?.apiKey || getEnvApiKey(model.provider) || "";
|
||||
const {
|
||||
headers: requestHeaders,
|
||||
copilotPremiumRequests,
|
||||
baseUrl,
|
||||
} = createRequestSetup(model, context, apiKey, options?.headers, options?.initiatorOverride, routingSessionId);
|
||||
const { headers, copilotPremiumRequests, baseUrl } = resolveOpenAIRequestSetup(model, {
|
||||
apiKey,
|
||||
extraHeaders: options?.headers,
|
||||
initiatorOverride: options?.initiatorOverride,
|
||||
messages: context.messages,
|
||||
openAISessionId: routingSessionId,
|
||||
promptCacheSessionId,
|
||||
});
|
||||
const premiumRequestsTotal = copilotPremiumRequests;
|
||||
const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState);
|
||||
const builtParams = buildParams(model, context, options, providerSessionState);
|
||||
const strictToolsScope = getOpenAIStrictToolsScope(model, baseUrl);
|
||||
const builtParams = buildParams(model, context, options, providerSessionState, strictToolsScope);
|
||||
const params = builtParams.params;
|
||||
const { trailingScaffoldingItems } = builtParams;
|
||||
let activeParams = params;
|
||||
let activeTrailingScaffoldingItems = trailingScaffoldingItems;
|
||||
if (isOpenAIResponsesStatefulEnabled(options, baseUrl) && routingSessionId && providerSessionState) {
|
||||
chainState = getOpenAIResponsesChainState(providerSessionState, model, routingSessionId);
|
||||
chainState = getOpenAIResponsesChainState(providerSessionState, model, baseUrl, routingSessionId);
|
||||
if (!chainState.disabled) {
|
||||
// Platform `previous_response_id` chaining only resolves stored responses.
|
||||
params.store = true;
|
||||
@@ -451,13 +429,13 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
||||
);
|
||||
}
|
||||
try {
|
||||
const headers = { ...requestHeaders };
|
||||
const headersWithTimeout = { ...headers };
|
||||
if (requestTimeoutMs !== undefined) {
|
||||
headers["X-Stainless-Timeout"] = Math.floor(requestTimeoutMs / 1000).toString();
|
||||
headersWithTimeout["X-Stainless-Timeout"] = Math.floor(requestTimeoutMs / 1000).toString();
|
||||
}
|
||||
const { events, response, requestId } = await postOpenAIStream<ResponseStreamEvent>({
|
||||
url: requestUrl,
|
||||
headers,
|
||||
headers: headersWithTimeout,
|
||||
body: requestParams,
|
||||
signal: requestSignal,
|
||||
fetch: options?.fetch,
|
||||
@@ -481,40 +459,114 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
||||
{ provider: model.provider, signal: requestSignal },
|
||||
);
|
||||
let openaiStream: AsyncIterable<ResponseStreamEvent>;
|
||||
try {
|
||||
openaiStream = await openResponsesStream(chained.params);
|
||||
} catch (error) {
|
||||
if (!chainState || !sentPreviousResponseId || requestSignal.aborted) {
|
||||
throw error;
|
||||
let strictRetryAvailable = true;
|
||||
let activeStrictToolsApplied = builtParams.strictToolsApplied;
|
||||
let forceDisableStrictTools = false;
|
||||
while (true) {
|
||||
try {
|
||||
openaiStream = await openResponsesStream(chained.params);
|
||||
break;
|
||||
} catch (error) {
|
||||
const capturedErrorResponse = error instanceof OpenAIHttpError ? error.captured : undefined;
|
||||
const compiledGrammarTooLarge =
|
||||
isOpenRouterAnthropicModel(model) &&
|
||||
isCompiledGrammarTooLargeStrictError(error, capturedErrorResponse);
|
||||
const canRetryWithoutStrictTools =
|
||||
strictRetryAvailable &&
|
||||
!requestSignal.aborted &&
|
||||
(compiledGrammarTooLarge ||
|
||||
shouldRetryWithoutStrictTools(
|
||||
error,
|
||||
capturedErrorResponse,
|
||||
activeStrictToolsApplied,
|
||||
context.tools,
|
||||
));
|
||||
if (canRetryWithoutStrictTools) {
|
||||
strictRetryAvailable = false;
|
||||
forceDisableStrictTools = true;
|
||||
const strictFallbackError = compiledGrammarTooLarge
|
||||
? await finalizeErrorMessage(error, rawRequestDump, capturedErrorResponse)
|
||||
: undefined;
|
||||
disableStrictToolsForScope(providerSessionState, strictToolsScope);
|
||||
const fallbackBuilt = buildParams(
|
||||
model,
|
||||
context,
|
||||
options,
|
||||
providerSessionState,
|
||||
strictToolsScope,
|
||||
true,
|
||||
);
|
||||
const fallbackParams = fallbackBuilt.params;
|
||||
if (chainState && !chainState.disabled) fallbackParams.store = true;
|
||||
let fallbackChained: OpenAIResponsesChainedParams =
|
||||
chainState && !chainState.disabled
|
||||
? buildOpenAIResponsesChainedParams(
|
||||
fallbackParams,
|
||||
fallbackBuilt.trailingScaffoldingItems,
|
||||
chainState,
|
||||
)
|
||||
: { params: fallbackParams };
|
||||
sentPreviousResponseId = fallbackChained.previousResponseId;
|
||||
fallbackChained = {
|
||||
...fallbackChained,
|
||||
params: await applyPayloadReplacement(fallbackChained.params),
|
||||
};
|
||||
chained = fallbackChained;
|
||||
rawRequestDump.body = chained.params;
|
||||
activeParams = fallbackParams;
|
||||
activeTrailingScaffoldingItems = fallbackBuilt.trailingScaffoldingItems;
|
||||
activeStrictToolsApplied = fallbackBuilt.strictToolsApplied;
|
||||
if (strictFallbackError) output.errorMessage = strictFallbackError;
|
||||
continue;
|
||||
}
|
||||
if (!chainState || !sentPreviousResponseId || requestSignal.aborted) {
|
||||
throw error;
|
||||
}
|
||||
const zdrRejection =
|
||||
error instanceof Error &&
|
||||
/previous[ _]?response/i.test(error.message) &&
|
||||
/zero[ _-]?data[ _-]?retention/i.test(error.message);
|
||||
if (!zdrRejection && !isOpenAIResponsesStalePreviousResponseError(error)) {
|
||||
throw error;
|
||||
}
|
||||
// Server rejected the chain baseline: reset, count the failure (or
|
||||
// disable categorically on ZDR), and retry once with the full
|
||||
// transcript. Structurally cannot loop — the retry carries no
|
||||
// previous_response_id.
|
||||
if (zdrRejection) {
|
||||
markOpenAIResponsesChainZeroDataRetention(chainState, error);
|
||||
// ZDR orgs cannot store responses; the retry uses `store: false`.
|
||||
} else {
|
||||
registerOpenAIResponsesChainStaleFailure(chainState, error);
|
||||
}
|
||||
sentPreviousResponseId = undefined;
|
||||
const currentBuilt = buildParams(
|
||||
model,
|
||||
context,
|
||||
options,
|
||||
providerSessionState,
|
||||
strictToolsScope,
|
||||
forceDisableStrictTools,
|
||||
);
|
||||
const currentParams = currentBuilt.params;
|
||||
// Only ZDR forces `store: false` (the org never persists responses). A
|
||||
// non-ZDR stale baseline is transient, so keep storing: the full-context
|
||||
// retry must be chainable next turn, and the consecutive stale-failure
|
||||
// breaker only trips when each retry stores and the next turn re-chains.
|
||||
currentParams.store = !zdrRejection;
|
||||
const retryParams = await applyPayloadReplacement(currentParams);
|
||||
chained = { params: retryParams };
|
||||
rawRequestDump.body = retryParams;
|
||||
activeParams = currentParams;
|
||||
activeTrailingScaffoldingItems = currentBuilt.trailingScaffoldingItems;
|
||||
activeStrictToolsApplied = currentBuilt.strictToolsApplied;
|
||||
}
|
||||
const zdrRejection = isOpenAIResponsesZeroDataRetentionError(error);
|
||||
if (!zdrRejection && !isOpenAIResponsesStalePreviousResponseError(error)) {
|
||||
throw error;
|
||||
}
|
||||
// Server rejected the chain baseline: reset, count the failure (or
|
||||
// disable categorically on ZDR), and retry once with the full
|
||||
// transcript. Structurally cannot loop — the retry carries no
|
||||
// previous_response_id.
|
||||
if (zdrRejection) {
|
||||
markOpenAIResponsesChainZeroDataRetention(chainState, error);
|
||||
// ZDR orgs cannot store responses; the original request forced
|
||||
// `store: true` for chaining, which is meaningless here and would
|
||||
// otherwise leave subsequent turns asking the server to retain
|
||||
// data it must discard.
|
||||
params.store = false;
|
||||
} else {
|
||||
registerOpenAIResponsesChainStaleFailure(chainState, error);
|
||||
}
|
||||
sentPreviousResponseId = undefined;
|
||||
const retryParams = await applyPayloadReplacement(params);
|
||||
rawRequestDump.body = retryParams;
|
||||
openaiStream = await openResponsesStream(retryParams);
|
||||
}
|
||||
if (premiumRequestsTotal !== undefined) output.usage.premiumRequests = premiumRequestsTotal;
|
||||
stream.push({ type: "start", partial: output });
|
||||
|
||||
const nativeOutputItems: Array<Record<string, unknown>> = [];
|
||||
let sawCompleted = false;
|
||||
let sawTerminalResponseEvent = false;
|
||||
const timedOpenaiStream = iterateWithIdleTimeout(openaiStream, {
|
||||
idleTimeoutMs,
|
||||
firstItemTimeoutMs: firstEventTimeoutMs,
|
||||
@@ -535,7 +587,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
||||
nativeOutputItems.push(item as unknown as Record<string, unknown>);
|
||||
},
|
||||
onCompleted: () => {
|
||||
sawCompleted = true;
|
||||
sawTerminalResponseEvent = true;
|
||||
},
|
||||
});
|
||||
|
||||
@@ -548,11 +600,12 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
||||
}
|
||||
|
||||
// Detect premature stream closure: the HTTP stream ended without the
|
||||
// provider sending `response.completed`. Custom/proxy providers may
|
||||
// drop the connection mid-stream; without this guard the incomplete
|
||||
// output is silently surfaced as a successful "stop".
|
||||
if (!sawCompleted) {
|
||||
throw new Error("OpenAI responses stream closed before response.completed was received");
|
||||
// provider sending `response.completed` or `response.incomplete`.
|
||||
// Custom/proxy providers may drop the connection mid-stream; without
|
||||
// this guard the incomplete output is silently surfaced as a successful
|
||||
// "stop".
|
||||
if (!sawTerminalResponseEvent) {
|
||||
throw new Error("OpenAI responses stream closed before a terminal response event was received");
|
||||
}
|
||||
|
||||
if (output.stopReason === "aborted" || output.stopReason === "error") {
|
||||
@@ -562,7 +615,14 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
||||
output.providerPayload = createOpenAIResponsesHistoryPayload(model.provider, nativeOutputItems);
|
||||
if (providerSessionState) providerSessionState.nativeHistoryReplayWarmed = true;
|
||||
if (chainState) {
|
||||
chainState.lastParams = structuredCloneJSON(stripTrailingScaffolding(params, trailingScaffoldingItems));
|
||||
chainState.lastParams = structuredCloneJSON(
|
||||
activeTrailingScaffoldingItems > 0 && Array.isArray(activeParams.input)
|
||||
? {
|
||||
...activeParams,
|
||||
input: activeParams.input.slice(0, activeParams.input.length - activeTrailingScaffoldingItems),
|
||||
}
|
||||
: activeParams,
|
||||
);
|
||||
if (output.responseId) {
|
||||
chainState.lastResponseId = output.responseId;
|
||||
chainState.lastResponseItems = sanitizeOpenAIResponsesHistoryItemsForReplay(
|
||||
@@ -587,8 +647,14 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
||||
if (chainState) resetOpenAIResponsesChainState(chainState);
|
||||
const firstEventTimeoutError = abortTracker.getLocalAbortReason();
|
||||
output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
|
||||
output.errorStatus = extractHttpStatusFromError(error);
|
||||
output.errorMessage = firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
|
||||
const capturedErrorResponse = error instanceof OpenAIHttpError ? error.captured : undefined;
|
||||
output.errorStatus = extractHttpStatusFromError(error) ?? capturedErrorResponse?.status;
|
||||
output.errorMessage =
|
||||
firstEventTimeoutError?.message ??
|
||||
(await finalizeErrorMessage(error, rawRequestDump, capturedErrorResponse));
|
||||
// Some providers via OpenRouter include extra details here.
|
||||
const rawMetadata = (error as { error?: { metadata?: { raw?: string } } })?.error?.metadata?.raw;
|
||||
if (rawMetadata) output.errorMessage += `\n${rawMetadata}`;
|
||||
output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
|
||||
output.duration = Date.now() - startTime;
|
||||
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
|
||||
@@ -600,87 +666,42 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
||||
return stream;
|
||||
};
|
||||
|
||||
function createRequestSetup(
|
||||
model: Model<"openai-responses">,
|
||||
context: Context,
|
||||
apiKey?: string,
|
||||
extraHeaders?: Record<string, string>,
|
||||
initiatorOverride?: MessageAttribution,
|
||||
sessionId?: string,
|
||||
): {
|
||||
headers: Record<string, string>;
|
||||
copilotPremiumRequests: number | undefined;
|
||||
baseUrl: string | undefined;
|
||||
} {
|
||||
if (!apiKey) {
|
||||
if (!$env.OPENAI_API_KEY) {
|
||||
throw new Error(
|
||||
"OpenAI API key is required. Set OPENAI_API_KEY environment variable or pass it as an argument.",
|
||||
);
|
||||
}
|
||||
apiKey = $env.OPENAI_API_KEY;
|
||||
}
|
||||
const rawApiKey = apiKey;
|
||||
|
||||
const headers = { ...(model.headers ?? {}), ...(extraHeaders ?? {}) };
|
||||
let copilotPremiumRequests: number | undefined;
|
||||
|
||||
let baseUrl = model.baseUrl;
|
||||
if (model.provider === "github-copilot") {
|
||||
apiKey = parseGitHubCopilotApiKey(rawApiKey).accessToken;
|
||||
const hasImages = hasCopilotVisionInput(context.messages);
|
||||
const copilot = buildCopilotDynamicHeaders({
|
||||
messages: context.messages,
|
||||
hasImages,
|
||||
premiumMultiplier: model.premiumMultiplier,
|
||||
headers,
|
||||
initiatorOverride,
|
||||
});
|
||||
Object.assign(headers, copilot.headers);
|
||||
copilotPremiumRequests = copilot.premiumRequests;
|
||||
baseUrl = resolveGitHubCopilotBaseUrl(model.baseUrl, rawApiKey) ?? model.baseUrl;
|
||||
}
|
||||
if (sessionId && model.provider === "openai") {
|
||||
headers.session_id ??= sessionId;
|
||||
headers["x-client-request-id"] ??= sessionId;
|
||||
}
|
||||
headers.Authorization ??= `Bearer ${apiKey}`;
|
||||
return { headers, copilotPremiumRequests, baseUrl };
|
||||
}
|
||||
|
||||
function getOpenAIResponsesPromptCacheKey(
|
||||
options: Pick<OpenAIResponsesOptions, "cacheRetention" | "promptCacheKey" | "sessionId"> | undefined,
|
||||
): string | undefined {
|
||||
if (resolveCacheRetention(options?.cacheRetention) === "none") return undefined;
|
||||
return normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId);
|
||||
}
|
||||
|
||||
export function getOpenAIResponsesCacheSessionId(
|
||||
options: Pick<OpenAIResponsesOptions, "cacheRetention" | "sessionId" | "promptCacheKey"> | undefined,
|
||||
): string | undefined {
|
||||
return getOpenAIResponsesPromptCacheKey(options);
|
||||
}
|
||||
function getOpenAIResponsesRoutingSessionId(
|
||||
options: Pick<OpenAIResponsesOptions, "cacheRetention" | "sessionId"> | undefined,
|
||||
): string | undefined {
|
||||
if (resolveCacheRetention(options?.cacheRetention) === "none") return undefined;
|
||||
return normalizeOpenAIResponsesPromptCacheKey(options?.sessionId);
|
||||
}
|
||||
|
||||
/** @internal Exported for tests. */
|
||||
export function buildParams(
|
||||
model: Model<"openai-responses">,
|
||||
context: Context,
|
||||
options: OpenAIResponsesOptions | undefined,
|
||||
providerSessionState: OpenAIResponsesProviderSessionState | undefined,
|
||||
): { params: OpenAIResponsesSamplingParams; trailingScaffoldingItems: number } {
|
||||
const strictResponsesPairing = options?.strictResponsesPairing ?? model.compat.strictResponsesPairing;
|
||||
const messages = convertConversationMessages(model, context, strictResponsesPairing, providerSessionState, options);
|
||||
strictToolsScope?: OpenAIStrictToolsScope,
|
||||
disableStrictToolsOverride = false,
|
||||
): { params: OpenAIResponsesSamplingParams; trailingScaffoldingItems: number; strictToolsApplied: boolean } {
|
||||
const policy = resolveOpenAICompatPolicy(model, {
|
||||
endpoint: "responses",
|
||||
reasoning: options?.reasoning,
|
||||
disableReasoning: options?.disableReasoning,
|
||||
toolChoice: options?.toolChoice,
|
||||
strictResponsesPairing: options?.strictResponsesPairing,
|
||||
includeEncryptedReasoning: options?.includeEncryptedReasoning,
|
||||
filterReasoningHistory: options?.filterReasoningHistory,
|
||||
omitReasoningEffort: options?.omitReasoningEffort,
|
||||
});
|
||||
const strictResponsesPairing = policy.tools.strictResponsesPairing;
|
||||
const shouldReplayNativeHistory = providerSessionState?.nativeHistoryReplayWarmed ?? true;
|
||||
const messages = buildResponsesInput({
|
||||
model,
|
||||
context,
|
||||
strictResponsesPairing,
|
||||
nativeHistory: {
|
||||
replay: shouldReplayNativeHistory,
|
||||
filterReasoning: policy.reasoning.filterReasoningHistory,
|
||||
},
|
||||
includeThinkingSignatures: shouldReplayNativeHistory,
|
||||
repairOrphanOutputs: true,
|
||||
});
|
||||
|
||||
const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
|
||||
let systemInstructions: string | undefined;
|
||||
if (systemPrompts.length > 0) {
|
||||
const needsDeveloperRole = model.reasoning && model.compat.supportsDeveloperRole;
|
||||
const needsDeveloperRole = policy.messages.systemRole === "developer";
|
||||
if (needsDeveloperRole) {
|
||||
// Reasoning models on known OpenAI-compatible endpoints require the
|
||||
// `developer` role. Send all system prompts inline in `input`.
|
||||
@@ -697,8 +718,13 @@ export function buildParams(
|
||||
|
||||
const cacheRetention = resolveCacheRetention(options?.cacheRetention);
|
||||
const promptCacheKey = getOpenAIResponsesPromptCacheKey(options);
|
||||
const modelId = applyWireModelIdTransform(
|
||||
model.requestModelId ?? model.id,
|
||||
model.compat.wireModelIdMode,
|
||||
options?.openrouterVariant,
|
||||
);
|
||||
const params: OpenAIResponsesSamplingParams = {
|
||||
model: model.requestModelId ?? model.id,
|
||||
model: modelId,
|
||||
input: messages,
|
||||
instructions: systemInstructions,
|
||||
stream: true,
|
||||
@@ -708,18 +734,35 @@ export function buildParams(
|
||||
? "24h"
|
||||
: undefined
|
||||
: undefined,
|
||||
// Gateway routing: OpenRouter-only Responses wire field for sticky upstream
|
||||
// routing + observability grouping; no equivalent on direct OpenAI.
|
||||
session_id: model.compat.isOpenRouterHost ? getOpenRouterResponsesSessionId(options) : undefined,
|
||||
store: false,
|
||||
stream_options: model.provider === "openai" ? { include_obfuscation: false } : undefined,
|
||||
stream_options: model.compat.supportsObfuscationOptOut ? { include_obfuscation: false } : undefined,
|
||||
};
|
||||
const outputToken = resolveOpenAIOutputTokenParam({
|
||||
field: "max_output_tokens",
|
||||
maxTokens: options?.maxTokens,
|
||||
maxTokensExplicit: options?.maxTokensExplicit ?? options?.maxTokens !== undefined,
|
||||
modelMaxTokens: model.maxTokens,
|
||||
omitMaxOutputTokens: model.omitMaxOutputTokens ?? false,
|
||||
isOpenRouterHost: model.compat.isOpenRouterHost,
|
||||
alwaysSendMaxTokens: model.compat.alwaysSendMaxTokens,
|
||||
});
|
||||
|
||||
applyCommonResponsesSamplingParams(params, options, model);
|
||||
applyCommonResponsesSamplingParams(params, { ...options, maxTokens: outputToken?.value }, model);
|
||||
// TODO: openai responses has no top-level `stop`/`stop_sequences`; surface via reasoning.stop?
|
||||
// `StreamOptions.stopSequences` is intentionally dropped for this provider.
|
||||
// TODO: openai responses has no top-level `frequency_penalty` field as of the current SDK;
|
||||
// `StreamOptions.frequencyPenalty` is intentionally dropped for this provider.
|
||||
|
||||
let strictToolsApplied = false;
|
||||
if (context.tools) {
|
||||
params.tools = convertTools(context.tools, model.compat.supportsStrictMode, model);
|
||||
const disableStrictTools =
|
||||
disableStrictToolsOverride || isStrictToolsDisabledForScope(providerSessionState, strictToolsScope);
|
||||
const strictMode = !disableStrictTools && model.compat.supportsStrictMode;
|
||||
params.tools = convertTools(context.tools, strictMode, model);
|
||||
strictToolsApplied = params.tools.some(t => (t as { strict?: boolean }).strict === true);
|
||||
if (options?.toolChoice) {
|
||||
// Map tool_choice against the tools that survived quarantine, not the
|
||||
// original list: a forced choice for a dropped tool — or "required" when
|
||||
@@ -748,104 +791,29 @@ export function buildParams(
|
||||
}
|
||||
}
|
||||
|
||||
const trailingScaffoldingItems = applyResponsesReasoningParams(
|
||||
params,
|
||||
model,
|
||||
options,
|
||||
messages,
|
||||
effort =>
|
||||
const reasoningPolicy = resolveOpenAICompatPolicy(model, {
|
||||
endpoint: "responses",
|
||||
reasoning: options?.reasoning,
|
||||
disableReasoning: options?.disableReasoning,
|
||||
toolChoice: params.tool_choice,
|
||||
strictResponsesPairing: options?.strictResponsesPairing,
|
||||
includeEncryptedReasoning: options?.includeEncryptedReasoning,
|
||||
filterReasoningHistory: options?.filterReasoningHistory,
|
||||
omitReasoningEffort: options?.omitReasoningEffort,
|
||||
});
|
||||
const trailingScaffoldingItems = applyResponsesCompatPolicy(params, messages, reasoningPolicy, {
|
||||
reasoningSummary: options?.reasoningSummary,
|
||||
mapEffort: effort =>
|
||||
model.compat.reasoningEffortMap?.[effort as NonNullable<OpenAIResponsesOptions["reasoning"]>] ??
|
||||
model.thinking?.effortMap?.[effort as NonNullable<OpenAIResponsesOptions["reasoning"]>] ??
|
||||
effort,
|
||||
options?.includeEncryptedReasoning ?? true,
|
||||
options?.omitReasoningEffort ?? false,
|
||||
);
|
||||
});
|
||||
|
||||
if (options?.extraBody) {
|
||||
Object.assign(params, options.extraBody);
|
||||
}
|
||||
applyOpenAIGatewayRouting(params, model.compat);
|
||||
|
||||
return { params, trailingScaffoldingItems };
|
||||
}
|
||||
applyOpenAIExtraBody(params, options?.extraBody);
|
||||
|
||||
function convertConversationMessages(
|
||||
model: Model<"openai-responses">,
|
||||
context: Context,
|
||||
strictResponsesPairing: boolean,
|
||||
providerSessionState: OpenAIResponsesProviderSessionState | undefined,
|
||||
options?: OpenAIResponsesOptions,
|
||||
): ResponseInput {
|
||||
const filterReasoning = <T extends { type?: string }>(items: T[]): T[] =>
|
||||
options?.filterReasoningHistory ? items.filter(i => i?.type !== "reasoning") : items;
|
||||
const messages: ResponseInput = [];
|
||||
let knownCallIds = new Set<string>();
|
||||
const customCallIds = new Set<string>();
|
||||
const shouldReplayNativeHistory = canReplayOpenAIResponsesNativeHistory(providerSessionState);
|
||||
const transformedMessages = transformMessages(context.messages, model, normalizeResponsesToolCallIdForTransform);
|
||||
|
||||
let msgIndex = 0;
|
||||
for (const msg of transformedMessages) {
|
||||
if (msg.role === "user" || msg.role === "developer") {
|
||||
const providerPayload = (msg as { providerPayload?: AssistantMessage["providerPayload"] }).providerPayload;
|
||||
const historyItems = getOpenAIResponsesHistoryItems(providerPayload, model.provider);
|
||||
const shouldReplayPayloadItems =
|
||||
shouldReplayNativeHistory ||
|
||||
(historyItems?.some(item => {
|
||||
if (!item || typeof item !== "object") return false;
|
||||
const candidate = item as { type?: unknown };
|
||||
return candidate.type === "compaction" || candidate.type === "compaction_summary";
|
||||
}) ??
|
||||
false);
|
||||
if (historyItems && shouldReplayPayloadItems) {
|
||||
messages.push(...sanitizeOpenAIResponsesHistoryItemsForReplay(filterReasoning(historyItems)));
|
||||
knownCallIds = collectKnownCallIds(messages);
|
||||
for (const id of collectCustomCallIds(messages)) customCallIds.add(id);
|
||||
msgIndex++;
|
||||
continue;
|
||||
}
|
||||
const content = convertResponsesInputContent(msg.content, model.input.includes("image"));
|
||||
if (!content) continue;
|
||||
messages.push({ role: "user", content });
|
||||
} else if (msg.role === "assistant") {
|
||||
const assistantMsg = msg as AssistantMessage;
|
||||
// Native items are model-bound (reasoning carries encrypted content minted
|
||||
// by the producing model); after a mid-session model switch fall back to
|
||||
// block re-encode, which strips foreign signatures.
|
||||
const providerPayload =
|
||||
shouldReplayNativeHistory && assistantMsg.api === model.api && assistantMsg.model === model.id
|
||||
? getOpenAIResponsesHistoryPayload(assistantMsg.providerPayload, model.provider, assistantMsg.provider)
|
||||
: undefined;
|
||||
const historyItems = providerPayload?.items;
|
||||
if (historyItems) {
|
||||
const sanitizedHistoryItems = sanitizeOpenAIResponsesHistoryItemsForReplay(filterReasoning(historyItems));
|
||||
if (providerPayload?.dt) {
|
||||
messages.push(...sanitizedHistoryItems);
|
||||
} else {
|
||||
messages.splice(0, messages.length, ...sanitizedHistoryItems);
|
||||
}
|
||||
knownCallIds = collectKnownCallIds(messages);
|
||||
for (const id of collectCustomCallIds(messages)) customCallIds.add(id);
|
||||
msgIndex++;
|
||||
continue;
|
||||
}
|
||||
|
||||
const outputItems = convertResponsesAssistantMessage(
|
||||
assistantMsg,
|
||||
model,
|
||||
msgIndex,
|
||||
knownCallIds,
|
||||
shouldReplayNativeHistory,
|
||||
customCallIds,
|
||||
);
|
||||
if (outputItems.length === 0) continue;
|
||||
messages.push(...outputItems);
|
||||
} else if (msg.role === "toolResult") {
|
||||
appendResponsesToolResultMessages(messages, msg, model, strictResponsesPairing, knownCallIds, customCallIds);
|
||||
}
|
||||
msgIndex++;
|
||||
}
|
||||
|
||||
return repairOrphanResponsesToolCalls(repairOrphanResponsesToolOutputs(messages));
|
||||
return { params, trailingScaffoldingItems, strictToolsApplied };
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,82 +0,0 @@
|
||||
// Ported from NousResearch/hermes-agent (MIT) — agent/transports/codex.py:182-193,
|
||||
// agent/codex_responses_adapter.py:247-311, agent/model_metadata.py:263-285.
|
||||
// Logic EXTRACTED into a dedicated xAI adapter so the generic OpenAI Responses
|
||||
// path stays provider-agnostic and the OpenAI Codex Responses path is unaffected.
|
||||
|
||||
import type { Context, Model, StreamFunction } from "../types";
|
||||
import {
|
||||
getOpenAIResponsesCacheSessionId,
|
||||
type OpenAIResponsesOptions,
|
||||
streamOpenAIResponses,
|
||||
} from "./openai-responses";
|
||||
|
||||
// xAI rejects `reasoning.effort` on grok-4 / grok-4-fast / grok-3 /
|
||||
// grok-code-fast / grok-4.20-0309-* / grok-build with HTTP 400 ("Model X does
|
||||
// not support parameter reasoningEffort") even though those models reason
|
||||
// natively (hermes-agent/agent/transports/codex.py:127-133). Only send the
|
||||
// effort dial when the target model is on this allowlist; otherwise suppress
|
||||
// it via OpenAIResponsesOptions.omitReasoningEffort and let the model reason
|
||||
// on its own. grok-build was previously on this list per user spec; the live
|
||||
// xAI server contradicts that assumption (HTTP 400 confirmed against
|
||||
// api.x.ai/v1/responses on 2026-05-17).
|
||||
const GROK_EFFORT_CAPABLE_PREFIXES = ["grok-3-mini", "grok-4.20-multi-agent", "grok-4.3"] as const;
|
||||
|
||||
function grokSupportsReasoningEffort(modelId: string): boolean {
|
||||
const name = (modelId || "").trim().toLowerCase();
|
||||
if (!name) return false;
|
||||
// Strip common aggregator prefixes (x-ai/, openrouter/x-ai/, xai/, ...) before matching.
|
||||
const bare = name.includes("/") ? name.slice(name.lastIndexOf("/") + 1) : name;
|
||||
return GROK_EFFORT_CAPABLE_PREFIXES.some(prefix => bare.startsWith(prefix));
|
||||
}
|
||||
|
||||
/**
|
||||
* xAI Grok Responses adapter (SuperGrok OAuth path).
|
||||
*
|
||||
* Three xAI-specific behaviors vs the generic OpenAI Responses adapter:
|
||||
*
|
||||
* 1. `x-grok-conv-id` header + body `prompt_cache_key` route prompt-cache
|
||||
* hits on xAI's edge. Hermes uses both (agent/transports/codex.py:182-193).
|
||||
* The header is undocumented by xAI; `previous_response_id` is the
|
||||
* documented alternative — switch if xAI deprecates the header.
|
||||
* 2. includeEncryptedReasoning=false — xAI's /v1/responses rejects replayed
|
||||
* `encrypted_content` blobs minted under SuperGrok OAuth.
|
||||
* 3. filterReasoningHistory=true — strip `type: "reasoning"` items from
|
||||
* replayed conversation history; the blob inside is non-replayable under
|
||||
* OAuth and the wrapper item 404s without it (store=false; server cannot
|
||||
* resolve by id).
|
||||
*
|
||||
* Everything else is the generic OpenAI Responses transport. The xAI bearer
|
||||
* token arrives in `options.apiKey` via AuthStorage.getApiKey() upstream, and
|
||||
* the xAI base URL (`https://api.x.ai/v1`) arrives via `model.baseUrl` from
|
||||
* the provider registry — not routed through this wrapper.
|
||||
*/
|
||||
export const streamXAIResponses: StreamFunction<"openai-responses"> = (
|
||||
model: Model<"openai-responses">,
|
||||
context: Context,
|
||||
options: OpenAIResponsesOptions = {},
|
||||
) => {
|
||||
const cacheSessionId = getOpenAIResponsesCacheSessionId(options);
|
||||
|
||||
const xaiHeaders: Record<string, string> = { ...options?.headers };
|
||||
if (cacheSessionId) {
|
||||
xaiHeaders["x-grok-conv-id"] = cacheSessionId;
|
||||
}
|
||||
|
||||
const xaiBody: Record<string, unknown> = { ...(options?.extraBody ?? {}) };
|
||||
if (cacheSessionId) {
|
||||
xaiBody.prompt_cache_key = cacheSessionId;
|
||||
}
|
||||
|
||||
const xaiOptions: OpenAIResponsesOptions = {
|
||||
...options,
|
||||
headers: xaiHeaders,
|
||||
extraBody: xaiBody,
|
||||
includeEncryptedReasoning: false,
|
||||
filterReasoningHistory: true,
|
||||
// Caller-passed value always wins (escape hatch for future xAI behavior
|
||||
// changes); otherwise gate the effort dial on the allowlist.
|
||||
omitReasoningEffort: options?.omitReasoningEffort ?? !grokSupportsReasoningEffort(model.id),
|
||||
};
|
||||
|
||||
return streamOpenAIResponses(model, context, xaiOptions);
|
||||
};
|
||||
+42
-11
@@ -46,7 +46,6 @@ import {
|
||||
streamOpenAIResponses,
|
||||
} from "./providers/register-builtins";
|
||||
import { isSyntheticModel, streamSynthetic } from "./providers/synthetic";
|
||||
import { streamXAIResponses } from "./providers/xai-responses";
|
||||
import { isUsageLimitError } from "./rate-limit-utils";
|
||||
import { PROVIDER_REGISTRY } from "./registry";
|
||||
import type {
|
||||
@@ -74,6 +73,7 @@ function isGoogleVertexAuthenticatedModel(model: Model<Api>): boolean {
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
function createVertexAuthenticatedFetch(options: StreamOptions | undefined): FetchImpl {
|
||||
const baseFetch = options?.fetch ?? fetch;
|
||||
const vertexFetch = async (input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
||||
@@ -292,15 +292,19 @@ function streamDispatch<TApi extends Api>(
|
||||
});
|
||||
}
|
||||
|
||||
case "openrouter": {
|
||||
const useResponses = $env.PI_OPENROUTER_RESPONSES !== "0";
|
||||
if (useResponses) {
|
||||
return streamOpenAIResponses(model as Model<"openai-responses">, context, providerOptions as any);
|
||||
}
|
||||
return streamOpenAICompletions(model as Model<"openai-completions">, context, providerOptions as any);
|
||||
}
|
||||
|
||||
case "openai-completions":
|
||||
return streamOpenAICompletions(model as Model<"openai-completions">, context, providerOptions as any);
|
||||
|
||||
case "openai-responses": {
|
||||
if (model.provider === "xai-oauth") {
|
||||
return streamXAIResponses(model as Model<"openai-responses">, context, providerOptions as any);
|
||||
}
|
||||
case "openai-responses":
|
||||
return streamOpenAIResponses(model as Model<"openai-responses">, context, providerOptions as any);
|
||||
}
|
||||
|
||||
case "azure-openai-responses":
|
||||
return streamAzureOpenAIResponses(model as Model<"azure-openai-responses">, context, providerOptions as any);
|
||||
@@ -679,11 +683,10 @@ function resolveOpenAiReasoningEffort<TApi extends Api>(
|
||||
// Models that reason natively but expose no effort dial carry
|
||||
// `thinking: undefined` (baked at build time from
|
||||
// `compat.supportsReasoningEffort: false` on openai-responses*). The
|
||||
// wire-side omitReasoningEffort gate (providers/xai-responses.ts:78) is the
|
||||
// actual strip; returning undefined here avoids a redundant
|
||||
// requireSupportedEffort throw that would defeat the gate and surface a
|
||||
// confusing "Compaction failed: Thinking effort high is not supported
|
||||
// by..." to the user.
|
||||
// wire-side omitReasoningEffort gate (stream.ts) is the actual strip; returning
|
||||
// undefined here avoids a redundant requireSupportedEffort throw that would
|
||||
// defeat the gate and surface a confusing "Compaction failed: Thinking effort
|
||||
// high is not supported by..." to the user.
|
||||
if (!model.thinking) return undefined;
|
||||
return requireSupportedEffort(model, reasoning);
|
||||
}
|
||||
@@ -869,6 +872,31 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
return castApi<"bedrock-converse-stream">({ ...bedrockBase, maxTokens, thinkingBudgets });
|
||||
}
|
||||
|
||||
case "openrouter": {
|
||||
const useResponses = $env.PI_OPENROUTER_RESPONSES !== "0";
|
||||
if (useResponses) {
|
||||
return castApi<"openai-responses">({
|
||||
...base,
|
||||
reasoning: resolveOpenAiReasoningEffort(model, options),
|
||||
toolChoice: mapOpenAiToolChoice(options?.toolChoice),
|
||||
serviceTier: options?.serviceTier,
|
||||
reasoningSummary: options?.hideThinkingSummary ? null : undefined,
|
||||
openrouterVariant: options?.openrouterVariant,
|
||||
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
|
||||
disableReasoning: options?.disableReasoning,
|
||||
});
|
||||
}
|
||||
return castApi<"openai-completions">({
|
||||
...base,
|
||||
reasoning: resolveOpenAiReasoningEffort(model, options),
|
||||
disableReasoning: options?.disableReasoning,
|
||||
toolChoice: mapOpenAiToolChoice(options?.toolChoice),
|
||||
serviceTier: options?.serviceTier,
|
||||
openrouterVariant: options?.openrouterVariant,
|
||||
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
|
||||
});
|
||||
}
|
||||
|
||||
case "openai-completions":
|
||||
return castApi<"openai-completions">({
|
||||
...base,
|
||||
@@ -887,6 +915,9 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
toolChoice: mapOpenAiToolChoice(options?.toolChoice),
|
||||
serviceTier: options?.serviceTier,
|
||||
reasoningSummary: options?.hideThinkingSummary ? null : undefined,
|
||||
openrouterVariant: options?.openrouterVariant,
|
||||
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
|
||||
disableReasoning: options?.disableReasoning,
|
||||
});
|
||||
|
||||
case "azure-openai-responses":
|
||||
|
||||
@@ -56,6 +56,7 @@ export interface ApiOptionsMap {
|
||||
"anthropic-messages": AnthropicOptions;
|
||||
"bedrock-converse-stream": BedrockOptions;
|
||||
"openai-completions": OpenAICompletionsOptions;
|
||||
openrouter: OpenAIResponsesOptions | OpenAICompletionsOptions;
|
||||
"openai-responses": OpenAIResponsesOptions;
|
||||
"openai-codex-responses": OpenAICodexResponsesOptions;
|
||||
"azure-openai-responses": AzureOpenAIResponsesOptions;
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
import packageJson from "../../package.json" with { type: "json" };
|
||||
|
||||
export function getOpenRouterHeaders(): Record<string, string> {
|
||||
return {
|
||||
"User-Agent": `Oh-My-Pi/${packageJson.version}`,
|
||||
"HTTP-Referer": "https://omp.sh/",
|
||||
"X-OpenRouter-Title": "Oh-My-Pi",
|
||||
"X-OpenRouter-Categories": "cli-agent",
|
||||
"X-OpenRouter-Cache": "true",
|
||||
"X-OpenRouter-Cache-TTL": "3600",
|
||||
};
|
||||
}
|
||||
@@ -8,12 +8,12 @@ import {
|
||||
mapOpenAIResponsesToolChoiceForTools,
|
||||
supportsFreeformApplyPatch,
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-responses";
|
||||
import type { ResponseStreamEvent } from "@oh-my-pi/pi-ai/providers/openai-responses-wire";
|
||||
import {
|
||||
appendResponsesToolResultMessages,
|
||||
convertResponsesAssistantMessage,
|
||||
processResponsesStream,
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
|
||||
import type { ResponseStreamEvent } from "@oh-my-pi/pi-ai/providers/openai-responses-wire";
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
import type { AssistantMessage, Model, ModelSpec, Tool, ToolResultMessage } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { z } from "zod/v4";
|
||||
|
||||
@@ -4,6 +4,24 @@ import type { AssistantMessage, Model, ModelSpec, ThinkingContent, ToolCall } fr
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
|
||||
interface OpenAICompletionAssistantWireMessage {
|
||||
role: "assistant";
|
||||
content?: unknown;
|
||||
reasoning_content?: unknown;
|
||||
rs_6f3a1b2c4d5e6f7a8b9c0d1e2f3a4b5c?: unknown;
|
||||
}
|
||||
|
||||
function isOpenAICompletionAssistantWireMessage(message: unknown): message is OpenAICompletionAssistantWireMessage {
|
||||
if (typeof message !== "object" || message === null) return false;
|
||||
return (message as { role?: unknown }).role === "assistant";
|
||||
}
|
||||
|
||||
function findOpenAICompletionAssistantWireMessage(
|
||||
messages: readonly unknown[] | undefined,
|
||||
): OpenAICompletionAssistantWireMessage | undefined {
|
||||
return messages?.find(isOpenAICompletionAssistantWireMessage);
|
||||
}
|
||||
|
||||
function deepseekModel(overrides: Partial<ModelSpec<"openai-completions">>): Model<"openai-completions"> {
|
||||
const base = getBundledModel("openai", "gpt-4o-mini");
|
||||
return buildModel({
|
||||
@@ -188,10 +206,10 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const messages = convertMessages(model, { messages: [msg] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
const assistant = findOpenAICompletionAssistantWireMessage(messages);
|
||||
expect(assistant).toBeDefined();
|
||||
// The reasoning_content field should be set from the signature, even if empty.
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("");
|
||||
expect(assistant?.reasoning_content).toBe("");
|
||||
});
|
||||
|
||||
it("recovers reasoning_content from non-empty thinking block with signature", () => {
|
||||
@@ -231,9 +249,9 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const messages = convertMessages(model, { messages: [msg] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
const assistant = findOpenAICompletionAssistantWireMessage(messages);
|
||||
expect(assistant).toBeDefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("I need to read the file first.");
|
||||
expect(assistant?.reasoning_content).toBe("I need to read the file first.");
|
||||
});
|
||||
|
||||
it("normalizes OpenRouter reasoning deltas to DeepSeek reasoning_content on replay", () => {
|
||||
@@ -256,9 +274,9 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
} as ToolCall,
|
||||
]);
|
||||
const messages = convertMessages(model, { messages: [msg] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
const assistant = findOpenAICompletionAssistantWireMessage(messages);
|
||||
expect(assistant).toBeDefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("I should inspect the requested file.");
|
||||
expect(assistant?.reasoning_content).toBe("I should inspect the requested file.");
|
||||
});
|
||||
it("does not use opaque signature as property name but still sets reasoning_content from thinking text", () => {
|
||||
const model = deepseekModel({
|
||||
@@ -302,12 +320,12 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const messages = convertMessages(model, { messages: [msg] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
const assistant = findOpenAICompletionAssistantWireMessage(messages);
|
||||
expect(assistant).toBeDefined();
|
||||
// Should NOT have used the opaque signature as a property name.
|
||||
expect(Reflect.get(assistant as object, "rs_6f3a1b2c4d5e6f7a8b9c0d1e2f3a4b5c")).toBeUndefined();
|
||||
expect(assistant?.rs_6f3a1b2c4d5e6f7a8b9c0d1e2f3a4b5c).toBeUndefined();
|
||||
// Should have set reasoning_content from the thinking text via the openai path.
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("some reasoning");
|
||||
expect(assistant?.reasoning_content).toBe("some reasoning");
|
||||
});
|
||||
it("falls through to empty-string when thinking block has opaque signature and empty text", () => {
|
||||
const model = deepseekModel({
|
||||
@@ -349,10 +367,10 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const messages = convertMessages(model, { messages: [msg] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
const assistant = findOpenAICompletionAssistantWireMessage(messages);
|
||||
expect(assistant).toBeDefined();
|
||||
expect(Reflect.get(assistant as object, "rs_6f3a1b2c4d5e6f7a8b9c0d1e2f3a4b5c")).toBeUndefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("");
|
||||
expect(assistant?.rs_6f3a1b2c4d5e6f7a8b9c0d1e2f3a4b5c).toBeUndefined();
|
||||
expect(assistant?.reasoning_content).toBe("");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -379,10 +397,10 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
} as ToolCall,
|
||||
]);
|
||||
const messages = convertMessages(model, { messages: [msg] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
const assistant = findOpenAICompletionAssistantWireMessage(messages);
|
||||
expect(assistant).toBeDefined();
|
||||
// reasoning_content must be present (empty string) — not absent and not "."
|
||||
const rc = Reflect.get(assistant as object, "reasoning_content");
|
||||
const rc = assistant?.reasoning_content;
|
||||
expect(rc).toBeDefined();
|
||||
expect(rc).toBe("");
|
||||
});
|
||||
@@ -402,10 +420,10 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
} as ToolCall,
|
||||
]);
|
||||
const messages = convertMessages(model, { messages: [msg] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
const assistant = findOpenAICompletionAssistantWireMessage(messages);
|
||||
expect(assistant).toBeDefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("");
|
||||
expect((assistant as { content: unknown }).content).toBe("");
|
||||
expect(assistant?.reasoning_content).toBe("");
|
||||
expect(assistant?.content).toBe("");
|
||||
});
|
||||
|
||||
it("sets content to empty string (not null) when reasoning_content is present", () => {
|
||||
@@ -424,9 +442,9 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
} as ToolCall,
|
||||
]);
|
||||
const messages = convertMessages(model, { messages: [msg] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
const assistant = findOpenAICompletionAssistantWireMessage(messages);
|
||||
expect(assistant).toBeDefined();
|
||||
expect((assistant as { content: unknown }).content).toBe("");
|
||||
expect(assistant?.content).toBe("");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -463,10 +481,10 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const messages = convertMessages(model, { messages: [msg] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
const assistant = findOpenAICompletionAssistantWireMessage(messages);
|
||||
expect(assistant).toBeDefined();
|
||||
// reasoning_content must be present — even on non-tool-call turns
|
||||
const rc = Reflect.get(assistant as object, "reasoning_content");
|
||||
const rc = assistant?.reasoning_content;
|
||||
expect(rc).toBeDefined();
|
||||
expect(rc).toBe("");
|
||||
});
|
||||
@@ -503,10 +521,10 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const messages = convertMessages(model, { messages: [msg] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
const assistant = findOpenAICompletionAssistantWireMessage(messages);
|
||||
expect(assistant).toBeDefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("Let me think about this.");
|
||||
expect((assistant as { content: unknown }).content).toBe("The answer is 42.");
|
||||
expect(assistant?.reasoning_content).toBe("Let me think about this.");
|
||||
expect(assistant?.content).toBe("The answer is 42.");
|
||||
});
|
||||
|
||||
it("does NOT inject reasoning_content on non-tool-call turn for non-DeepSeek providers", () => {
|
||||
@@ -539,10 +557,10 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const messages = convertMessages(model, { messages: [msg] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
const assistant = findOpenAICompletionAssistantWireMessage(messages);
|
||||
expect(assistant).toBeDefined();
|
||||
// OpenRouter reasoning models only need reasoning_content on tool-call turns
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBeUndefined();
|
||||
expect(assistant?.reasoning_content).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -573,9 +591,9 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
} as ToolCall,
|
||||
]);
|
||||
const messages = convertMessages(model, { messages: [msg] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
const assistant = findOpenAICompletionAssistantWireMessage(messages);
|
||||
expect(assistant).toBeDefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBe(".");
|
||||
expect(assistant?.reasoning_content).toBe(".");
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -129,6 +129,18 @@ describe("Google Gemini CLI alignment", () => {
|
||||
});
|
||||
});
|
||||
|
||||
it("keeps projectId alias precedence and tolerates a mistyped projectId", () => {
|
||||
// projectId wins over project_id even when it is an empty string.
|
||||
const emptyPrimary = parseGeminiCliCredentials(
|
||||
JSON.stringify({ token: "t", projectId: "", project_id: "fallback" }),
|
||||
);
|
||||
expect(emptyPrimary.projectId).toBe("");
|
||||
|
||||
// A non-string projectId is dropped, so project_id is used instead.
|
||||
const mistyped = parseGeminiCliCredentials(JSON.stringify({ token: "t", projectId: 42, project_id: "fallback" }));
|
||||
expect(mistyped.projectId).toBe("fallback");
|
||||
});
|
||||
|
||||
it("avoids excessive antigravity refresh churn with pre-buffered OAuth expiry", () => {
|
||||
const issuedAt = 1_700_000_000_000;
|
||||
const preBufferedExpiry = issuedAt + 55 * 60 * 1000;
|
||||
|
||||
@@ -104,8 +104,26 @@ async function capturePayload(
|
||||
return promise;
|
||||
}
|
||||
|
||||
interface CompletionAssistantWireMessage {
|
||||
role: "assistant";
|
||||
reasoning_content?: unknown;
|
||||
}
|
||||
|
||||
interface CompletionBody {
|
||||
thinking?: { type?: string; keep?: string };
|
||||
messages?: unknown[];
|
||||
stream?: boolean;
|
||||
}
|
||||
|
||||
function isCompletionAssistantWireMessage(message: unknown): message is CompletionAssistantWireMessage {
|
||||
if (typeof message !== "object" || message === null) return false;
|
||||
return (message as { role?: unknown }).role === "assistant";
|
||||
}
|
||||
|
||||
function findCompletionAssistantWireMessage(
|
||||
messages: readonly unknown[] | undefined,
|
||||
): CompletionAssistantWireMessage | undefined {
|
||||
return messages?.find(isCompletionAssistantWireMessage);
|
||||
}
|
||||
|
||||
describe("issue #1838 — kimi-k2.6 preserves historical reasoning across tool calls", () => {
|
||||
@@ -248,11 +266,11 @@ describe("issue #1838 — kimi-k2.6 preserves historical reasoning across tool c
|
||||
},
|
||||
],
|
||||
},
|
||||
)) as CompletionBody & { messages?: Array<Record<string, unknown>>; stream?: boolean };
|
||||
)) as CompletionBody;
|
||||
expect(payload.thinking).toEqual({ type: "enabled", keep: "all" });
|
||||
expect(payload.stream).toBe(true);
|
||||
const assistant = payload.messages?.find(m => m.role === "assistant");
|
||||
const assistant = findCompletionAssistantWireMessage(payload.messages);
|
||||
expect(assistant).toBeDefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("Need to read the file first.");
|
||||
expect(assistant?.reasoning_content).toBe("Need to read the file first.");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -212,7 +212,8 @@ describe("issue #2080 - MiniMax multi-chunk object tool arguments", () => {
|
||||
},
|
||||
},
|
||||
]);
|
||||
expect(Reflect.get(Object.prototype, "polluted")).toBeUndefined();
|
||||
const objectPrototype = Object.prototype as { polluted?: unknown };
|
||||
expect(objectPrototype.polluted).toBeUndefined();
|
||||
});
|
||||
|
||||
it("emits a concat-safe `toolcall_delta` sequence — accumulated deltas parse to the merged args", async () => {
|
||||
|
||||
@@ -111,7 +111,7 @@ describe("issue #827 — kimi reasoning models drop reasoning under forced tool_
|
||||
})) as CompletionsBody;
|
||||
|
||||
expect(body.tool_choice).toMatchObject({ type: "function", function: { name: "echo" } });
|
||||
expect(body.reasoning).toBeUndefined();
|
||||
expect(body.reasoning).toEqual({ enabled: false });
|
||||
expect(body.reasoning_effort).toBeUndefined();
|
||||
});
|
||||
it("sends explicit thinking disabled for Moonshot Kimi K2.6 when a named tool is forced", async () => {
|
||||
|
||||
@@ -4,6 +4,23 @@ import type { AssistantMessage, Model, ModelSpec } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
|
||||
interface OpenAICompletionAssistantWireMessage {
|
||||
role: "assistant";
|
||||
content?: unknown;
|
||||
reasoning_content?: unknown;
|
||||
}
|
||||
|
||||
function isOpenAICompletionAssistantWireMessage(message: unknown): message is OpenAICompletionAssistantWireMessage {
|
||||
if (typeof message !== "object" || message === null) return false;
|
||||
return (message as { role?: unknown }).role === "assistant";
|
||||
}
|
||||
|
||||
function findOpenAICompletionAssistantWireMessage(
|
||||
messages: readonly unknown[] | undefined,
|
||||
): OpenAICompletionAssistantWireMessage | undefined {
|
||||
return messages?.find(isOpenAICompletionAssistantWireMessage);
|
||||
}
|
||||
|
||||
function deepseekModel(overrides: Partial<ModelSpec<"openai-completions">>): Model<"openai-completions"> {
|
||||
const base = getBundledModel("openai", "gpt-4o-mini");
|
||||
return buildModel({
|
||||
@@ -70,9 +87,9 @@ describe("issue #883 / #810 — DeepSeek V4 reasoning_content tool-call replay",
|
||||
});
|
||||
const compat = model.compat;
|
||||
const messages = convertMessages(model, { messages: [assistantWithToolCall(model)] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
const assistant = findOpenAICompletionAssistantWireMessage(messages);
|
||||
expect(assistant).toBeDefined();
|
||||
const reasoningContent = Reflect.get(assistant as object, "reasoning_content");
|
||||
const reasoningContent = assistant?.reasoning_content;
|
||||
expect(reasoningContent).toBeDefined();
|
||||
// DeepSeek rejects synthetic "." — when no thinking blocks exist, we emit empty string
|
||||
expect(reasoningContent).toBe("");
|
||||
@@ -113,8 +130,8 @@ describe("issue #883 / #810 — DeepSeek V4 reasoning_content tool-call replay",
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const messages = convertMessages(model, { messages: [toolOnly] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
const assistant = findOpenAICompletionAssistantWireMessage(messages);
|
||||
expect(assistant).toBeDefined();
|
||||
expect((assistant as { content: unknown }).content).toBe("");
|
||||
expect(assistant?.content).toBe("");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -46,7 +46,7 @@ function customResponsesModel(compat: OpenAICompat): Model<"openai-responses"> {
|
||||
effortMap: compat.reasoningEffortMap,
|
||||
},
|
||||
compat,
|
||||
} as Model<"openai-responses">;
|
||||
} as unknown as Model<"openai-responses">;
|
||||
}
|
||||
|
||||
describe("issue #931 — openai-responses reasoning effort mapping", () => {
|
||||
|
||||
@@ -6,7 +6,7 @@ import { convertMessages as convertOpenAICompletionsMessages } from "@oh-my-pi/p
|
||||
import {
|
||||
appendResponsesToolResultMessages,
|
||||
convertResponsesInputContent,
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
import { NON_VISION_IMAGE_PLACEHOLDER } from "@oh-my-pi/pi-ai/providers/vision-guard";
|
||||
import type { Api, AssistantMessage, Context, Model, ModelSpec, ToolResultMessage, Usage } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
@@ -42,8 +42,13 @@ const compat: ResolvedOpenAICompat = {
|
||||
requiresThinkingAsText: false,
|
||||
requiresMistralToolIds: false,
|
||||
thinkingFormat: "openai",
|
||||
reasoningDisableMode: "lowest-effort",
|
||||
omitReasoningEffort: false,
|
||||
includeEncryptedReasoning: true,
|
||||
filterReasoningHistory: false,
|
||||
reasoningContentField: "reasoning_content",
|
||||
requiresReasoningContentForToolCalls: false,
|
||||
requiresReasoningContentForAllAssistantTurns: false,
|
||||
allowsSyntheticReasoningContentForToolCalls: true,
|
||||
requiresAssistantContentForToolCalls: false,
|
||||
openRouterRouting: {},
|
||||
@@ -51,6 +56,12 @@ const compat: ResolvedOpenAICompat = {
|
||||
extraBody: {},
|
||||
supportsStrictMode: true,
|
||||
toolStrictMode: "none",
|
||||
wireModelIdMode: "raw",
|
||||
stripDeepseekSpecialTokens: false,
|
||||
reasoningDeltasMayBeCumulative: false,
|
||||
emptyLengthFinishIsContextError: false,
|
||||
usesOpenAIToolCallIdLimit: false,
|
||||
dropThinkingWhenReasoningEffort: false,
|
||||
};
|
||||
|
||||
function makeModel<TApi extends Api>(api: TApi, provider: Model["provider"]): Model<TApi> {
|
||||
|
||||
@@ -3,6 +3,14 @@ import type { Context } from "@oh-my-pi/pi-ai";
|
||||
import { streamOllama } from "@oh-my-pi/pi-ai/providers/ollama";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
|
||||
interface OllamaChatRequestPayload {
|
||||
think?: unknown;
|
||||
}
|
||||
|
||||
function isOllamaChatRequestPayload(value: unknown): value is OllamaChatRequestPayload {
|
||||
return value !== null && typeof value === "object";
|
||||
}
|
||||
|
||||
function createReasoningOllamaModel() {
|
||||
return buildModel({
|
||||
id: "deepseek-v4-flash",
|
||||
@@ -20,10 +28,10 @@ function createReasoningOllamaModel() {
|
||||
|
||||
describe("Ollama chat thinking controls", () => {
|
||||
it("sends think false when reasoning is explicitly disabled", async () => {
|
||||
let payload: object | undefined;
|
||||
let payload: OllamaChatRequestPayload | undefined;
|
||||
const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
||||
const parsed: unknown = JSON.parse(String(init?.body));
|
||||
if (parsed === null || typeof parsed !== "object") {
|
||||
if (!isOllamaChatRequestPayload(parsed)) {
|
||||
throw new Error("Expected Ollama payload object");
|
||||
}
|
||||
payload = parsed;
|
||||
@@ -41,6 +49,6 @@ describe("Ollama chat thinking controls", () => {
|
||||
fetch: fetchMock,
|
||||
}).result();
|
||||
|
||||
expect(payload ? Reflect.get(payload, "think") : undefined).toBe(false);
|
||||
expect(payload?.think).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import {
|
||||
applyChatCompletionsCompatPolicy,
|
||||
applyResponsesCompatPolicy,
|
||||
resolveOpenAICompatPolicy,
|
||||
type OpenAICompletionsParams,
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
import type { ResponseCreateParamsStreaming, ResponseInput } from "@oh-my-pi/pi-ai/providers/openai-responses-wire";
|
||||
import type { Model, ModelSpec, OpenAICompat } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
|
||||
function chatModel(compat: OpenAICompat): Model<"openai-completions"> {
|
||||
return buildModel({
|
||||
id: "compat-reasoner",
|
||||
name: "Compat Reasoner",
|
||||
api: "openai-completions",
|
||||
provider: "test-provider",
|
||||
baseUrl: "https://example.com/v1",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128_000,
|
||||
maxTokens: 4096,
|
||||
compat,
|
||||
} satisfies ModelSpec<"openai-completions">);
|
||||
}
|
||||
|
||||
function responsesModel(compat: OpenAICompat): Model<"openai-responses"> {
|
||||
return buildModel({
|
||||
id: "compat-reasoner",
|
||||
name: "Compat Reasoner",
|
||||
api: "openai-responses",
|
||||
provider: "test-provider",
|
||||
baseUrl: "https://example.com/v1",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128_000,
|
||||
maxTokens: 4096,
|
||||
compat,
|
||||
} satisfies ModelSpec<"openai-responses">);
|
||||
}
|
||||
|
||||
function chatParams(): OpenAICompletionsParams {
|
||||
return { model: "compat-reasoner", messages: [], stream: true };
|
||||
}
|
||||
|
||||
function responsesParams(): ResponseCreateParamsStreaming {
|
||||
return { model: "compat-reasoner", input: [], stream: true };
|
||||
}
|
||||
|
||||
describe("OpenAI compat policy", () => {
|
||||
it("suppresses reasoning on forced tool choice for both endpoints", () => {
|
||||
const compat: OpenAICompat = {
|
||||
disableReasoningOnForcedToolChoice: true,
|
||||
thinkingFormat: "openrouter",
|
||||
reasoningDisableMode: "openrouter-enabled-false",
|
||||
};
|
||||
const toolChoice = { type: "function", name: "search" };
|
||||
const chatPolicy = resolveOpenAICompatPolicy(chatModel(compat), {
|
||||
endpoint: "chat-completions",
|
||||
reasoning: Effort.High,
|
||||
toolChoice,
|
||||
});
|
||||
const responsesPolicy = resolveOpenAICompatPolicy(responsesModel(compat), {
|
||||
endpoint: "responses",
|
||||
reasoning: Effort.High,
|
||||
toolChoice,
|
||||
});
|
||||
|
||||
expect(chatPolicy.reasoning.enabled).toBe(false);
|
||||
expect(responsesPolicy.reasoning.enabled).toBe(false);
|
||||
expect(chatPolicy.reasoning.disableReason).toBe("forced-tool-choice");
|
||||
expect(responsesPolicy.reasoning.disableReason).toBe("forced-tool-choice");
|
||||
});
|
||||
|
||||
it("encodes OpenRouter disabled reasoning through both wire adapters", () => {
|
||||
const compat: OpenAICompat = {
|
||||
thinkingFormat: "openrouter",
|
||||
reasoningDisableMode: "openrouter-enabled-false",
|
||||
};
|
||||
const chatBody = chatParams();
|
||||
const responseBody = responsesParams();
|
||||
const responseInput: ResponseInput = [];
|
||||
|
||||
applyChatCompletionsCompatPolicy(
|
||||
chatBody,
|
||||
resolveOpenAICompatPolicy(chatModel(compat), {
|
||||
endpoint: "chat-completions",
|
||||
disableReasoning: true,
|
||||
}),
|
||||
);
|
||||
applyResponsesCompatPolicy(
|
||||
responseBody,
|
||||
responseInput,
|
||||
resolveOpenAICompatPolicy(responsesModel(compat), { endpoint: "responses", disableReasoning: true }),
|
||||
undefined,
|
||||
);
|
||||
|
||||
expect(chatBody.reasoning).toEqual({ enabled: false });
|
||||
expect(responseBody.reasoning as unknown).toEqual({ enabled: false });
|
||||
});
|
||||
|
||||
it("omits effort for both wire adapters from one catalog flag", () => {
|
||||
const compat: OpenAICompat = { omitReasoningEffort: true };
|
||||
const chatBody = chatParams();
|
||||
const responseBody = responsesParams();
|
||||
const responseInput: ResponseInput = [];
|
||||
|
||||
applyChatCompletionsCompatPolicy(
|
||||
chatBody,
|
||||
resolveOpenAICompatPolicy(chatModel(compat), { endpoint: "chat-completions", reasoning: Effort.High }),
|
||||
);
|
||||
applyResponsesCompatPolicy(
|
||||
responseBody,
|
||||
responseInput,
|
||||
resolveOpenAICompatPolicy(responsesModel(compat), { endpoint: "responses", reasoning: Effort.High }),
|
||||
undefined,
|
||||
);
|
||||
|
||||
expect(chatBody.reasoning_effort).toBeUndefined();
|
||||
expect(responseBody.reasoning).toBeUndefined();
|
||||
});
|
||||
|
||||
it("exposes reasoning replay constraints independent of endpoint", () => {
|
||||
const compat: OpenAICompat = {
|
||||
requiresReasoningContentForToolCalls: true,
|
||||
requiresReasoningContentForAllAssistantTurns: true,
|
||||
allowsSyntheticReasoningContentForToolCalls: false,
|
||||
reasoningContentField: "reasoning_content",
|
||||
};
|
||||
const chatPolicy = resolveOpenAICompatPolicy(chatModel(compat), { endpoint: "chat-completions" });
|
||||
const responsesPolicy = resolveOpenAICompatPolicy(responsesModel(compat), { endpoint: "responses" });
|
||||
|
||||
expect(chatPolicy.reasoning.requiresReasoningContentForToolCalls).toBe(true);
|
||||
expect(responsesPolicy.reasoning.requiresReasoningContentForToolCalls).toBe(true);
|
||||
expect(chatPolicy.reasoning.requiresReasoningContentForAllAssistantTurns).toBe(true);
|
||||
expect(responsesPolicy.reasoning.requiresReasoningContentForAllAssistantTurns).toBe(true);
|
||||
expect(chatPolicy.reasoning.allowsSyntheticReasoningContentForToolCalls).toBe(false);
|
||||
expect(responsesPolicy.reasoning.allowsSyntheticReasoningContentForToolCalls).toBe(false);
|
||||
});
|
||||
|
||||
it("exposes tool id and cumulative reasoning stream constraints for both endpoints", () => {
|
||||
const compat: OpenAICompat = { requiresMistralToolIds: true, reasoningDeltasMayBeCumulative: true };
|
||||
const chatPolicy = resolveOpenAICompatPolicy(chatModel(compat), { endpoint: "chat-completions" });
|
||||
const responsesPolicy = resolveOpenAICompatPolicy(responsesModel(compat), { endpoint: "responses" });
|
||||
|
||||
expect(chatPolicy.tools.toolCallIdKind).toBe("mistral-9-alnum");
|
||||
expect(responsesPolicy.tools.toolCallIdKind).toBe("mistral-9-alnum");
|
||||
expect(chatPolicy.stream.reasoningDeltasMayBeCumulative).toBe(true);
|
||||
expect(responsesPolicy.stream.reasoningDeltasMayBeCumulative).toBe(true);
|
||||
});
|
||||
});
|
||||
@@ -33,20 +33,20 @@ function createAbortedSignal(): AbortSignal {
|
||||
return controller.signal;
|
||||
}
|
||||
|
||||
function toObject(value: unknown): object | null {
|
||||
return typeof value === "object" && value !== null ? value : null;
|
||||
function toObject(value: unknown): Record<string, unknown> | null {
|
||||
return typeof value === "object" && value !== null ? (value as Record<string, unknown>) : null;
|
||||
}
|
||||
|
||||
function getNestedObject(value: unknown, key: string): object | null {
|
||||
function getNestedObject(value: unknown, key: string): Record<string, unknown> | null {
|
||||
const obj = toObject(value);
|
||||
if (!obj) return null;
|
||||
return toObject(Reflect.get(obj, key));
|
||||
return toObject(obj[key]);
|
||||
}
|
||||
|
||||
function getNestedBoolean(value: unknown, key: string): boolean | undefined {
|
||||
const obj = toObject(value);
|
||||
if (!obj) return undefined;
|
||||
const property = Reflect.get(obj, key);
|
||||
const property = obj[key];
|
||||
return typeof property === "boolean" ? property : undefined;
|
||||
}
|
||||
|
||||
@@ -98,6 +98,17 @@ function zaiGlm52Model(): Model<"openai-completions"> {
|
||||
} satisfies ModelSpec<"openai-completions">);
|
||||
}
|
||||
|
||||
function kimiZaiModel(): Model<"openai-completions"> {
|
||||
return buildModel({
|
||||
...gpt4oMiniSpec,
|
||||
api: "openai-completions",
|
||||
provider: "moonshot",
|
||||
baseUrl: "https://api.moonshot.ai/v1",
|
||||
id: "kimi-k2.6",
|
||||
reasoning: true,
|
||||
} as ModelSpec<"openai-completions">);
|
||||
}
|
||||
|
||||
async function captureOpenAICompletionsPayload(
|
||||
model: Model<"openai-completions">,
|
||||
context: Context = baseContext(),
|
||||
@@ -115,9 +126,9 @@ async function captureOpenAICompletionsPayload(
|
||||
return promise;
|
||||
}
|
||||
|
||||
function getPayloadMessages(payload: unknown): object[] {
|
||||
function getPayloadMessages(payload: unknown): Record<string, unknown>[] {
|
||||
const payloadObject = toObject(payload);
|
||||
const messages = payloadObject ? Reflect.get(payloadObject, "messages") : undefined;
|
||||
const messages = payloadObject?.messages;
|
||||
if (!Array.isArray(messages)) throw new Error("payload messages missing");
|
||||
return messages.map(message => {
|
||||
const messageObject = toObject(message);
|
||||
@@ -129,14 +140,14 @@ function getPayloadMessages(payload: unknown): object[] {
|
||||
function getLastPayloadContent(payload: unknown): unknown {
|
||||
const lastMessage = getPayloadMessages(payload).at(-1);
|
||||
if (!lastMessage) throw new Error("payload has no messages");
|
||||
return Reflect.get(lastMessage, "content");
|
||||
return lastMessage.content;
|
||||
}
|
||||
|
||||
function getLastTextPart(content: unknown): object | undefined {
|
||||
function getLastTextPart(content: unknown): Record<string, unknown> | undefined {
|
||||
if (!Array.isArray(content)) return undefined;
|
||||
for (let index = content.length - 1; index >= 0; index--) {
|
||||
const part = toObject(content[index]);
|
||||
if (part && Reflect.get(part, "type") === "text") return part;
|
||||
if (part?.type === "text") return part;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
@@ -164,8 +175,13 @@ describe("openai-completions compatibility", () => {
|
||||
requiresThinkingAsText: false,
|
||||
requiresMistralToolIds: false,
|
||||
thinkingFormat: "openai",
|
||||
reasoningDisableMode: "lowest-effort",
|
||||
omitReasoningEffort: false,
|
||||
includeEncryptedReasoning: true,
|
||||
filterReasoningHistory: false,
|
||||
reasoningContentField: "reasoning_content",
|
||||
requiresReasoningContentForToolCalls: false,
|
||||
requiresReasoningContentForAllAssistantTurns: false,
|
||||
allowsSyntheticReasoningContentForToolCalls: true,
|
||||
requiresAssistantContentForToolCalls: false,
|
||||
openRouterRouting: {},
|
||||
@@ -177,6 +193,12 @@ describe("openai-completions compatibility", () => {
|
||||
alwaysSendMaxTokens: false,
|
||||
isOpenRouterHost: false,
|
||||
isVercelGatewayHost: false,
|
||||
wireModelIdMode: "raw",
|
||||
stripDeepseekSpecialTokens: false,
|
||||
reasoningDeltasMayBeCumulative: false,
|
||||
emptyLengthFinishIsContextError: false,
|
||||
usesOpenAIToolCallIdLimit: false,
|
||||
dropThinkingWhenReasoningEffort: false,
|
||||
} satisfies ResolvedOpenAICompat;
|
||||
const assistantMessage: AssistantMessage = {
|
||||
role: "assistant",
|
||||
@@ -350,7 +372,7 @@ describe("openai-completions compatibility", () => {
|
||||
// Regression: Moonshot's Kimi chat template rejects the `developer` role
|
||||
// with `400 Invalid request: tokenization failed` because `developer` is
|
||||
// an OpenAI extension and most other hosts don't carry it through their
|
||||
// tokenizer. The default for any non-OpenAI/Azure host MUST be `system`,
|
||||
// tokenizer. The default for non-OpenAI/Azure hosts MUST be `system`,
|
||||
// so reasoning models on those hosts cannot accidentally emit `developer`.
|
||||
const cases: Array<{ provider: string; baseUrl: string; expected: boolean }> = [
|
||||
{ provider: "openai", baseUrl: "https://api.openai.com/v1", expected: true },
|
||||
@@ -612,10 +634,46 @@ describe("openai-completions compatibility", () => {
|
||||
const payload = await promise;
|
||||
const thinking = getNestedObject(payload, "thinking");
|
||||
|
||||
expect(Reflect.get(thinking ?? {}, "type")).toBe("enabled");
|
||||
expect(Reflect.get(toObject(payload) ?? {}, "reasoning_effort")).toBe("max");
|
||||
expect(Reflect.get(toObject(payload) ?? {}, "tool_stream")).toBe(true);
|
||||
expect(Reflect.get(toObject(payload) ?? {}, "max_tokens")).toBe(65_536);
|
||||
const payloadObject = toObject(payload);
|
||||
|
||||
expect(thinking?.type).toBe("enabled");
|
||||
expect(payloadObject?.reasoning_effort).toBe("max");
|
||||
expect(payloadObject?.tool_stream).toBe(true);
|
||||
expect(payloadObject?.max_tokens).toBe(65_536);
|
||||
});
|
||||
|
||||
it("keeps Z.AI tool streaming disabled for native Kimi reasoning models", async () => {
|
||||
const model = kimiZaiModel();
|
||||
expect(model.compat.thinkingFormat).toBe("zai");
|
||||
expect(model.compat.supportsReasoningEffort).toBe(true);
|
||||
const readTool: Tool = {
|
||||
name: "read",
|
||||
description: "Read a file",
|
||||
parameters: {
|
||||
type: "object",
|
||||
properties: { path: { type: "string" } },
|
||||
required: ["path"],
|
||||
},
|
||||
};
|
||||
|
||||
const { promise, resolve } = Promise.withResolvers<unknown>();
|
||||
const fetchMock = createMockFetch(["[DONE]"]);
|
||||
streamOpenAICompletions(
|
||||
model,
|
||||
{ ...baseContext(), tools: [readTool] },
|
||||
{
|
||||
apiKey: "test-key",
|
||||
fetch: fetchMock,
|
||||
reasoning: "high",
|
||||
toolChoice: { type: "tool", name: "read" },
|
||||
signal: createAbortedSignal(),
|
||||
onPayload: payload => resolve(payload),
|
||||
},
|
||||
);
|
||||
const payload = await promise;
|
||||
const payloadObject = toObject(payload);
|
||||
|
||||
expect(payloadObject?.tool_stream).toBeUndefined();
|
||||
});
|
||||
|
||||
it("maps GLM-5.2 minimal reasoning to disabled Z.AI thinking", async () => {
|
||||
@@ -631,8 +689,8 @@ describe("openai-completions compatibility", () => {
|
||||
const payload = await promise;
|
||||
const thinking = getNestedObject(payload, "thinking");
|
||||
|
||||
expect(Reflect.get(thinking ?? {}, "type")).toBe("disabled");
|
||||
expect(Reflect.get(toObject(payload) ?? {}, "reasoning_effort")).toBeUndefined();
|
||||
expect(thinking?.type).toBe("disabled");
|
||||
expect(toObject(payload)?.reasoning_effort).toBeUndefined();
|
||||
});
|
||||
|
||||
it("treats finish_reason end as stop", async () => {
|
||||
@@ -818,8 +876,8 @@ describe("openai-completions compatibility", () => {
|
||||
expect(assistant).toBeDefined();
|
||||
const assistantObject = toObject(assistant);
|
||||
expect(assistantObject).toBeDefined();
|
||||
expect(assistantObject ? Reflect.get(assistantObject, "reasoning_text") : undefined).toBe("inspect tool output");
|
||||
expect(assistantObject ? Reflect.get(assistantObject, "reasoning_content") : undefined).toBeUndefined();
|
||||
expect(assistantObject?.reasoning_text).toBe("inspect tool output");
|
||||
expect(assistantObject?.reasoning_content).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -950,7 +1008,7 @@ describe("kimi model detection via detectCompat", () => {
|
||||
const messages = convertMessages(model, { messages: [toolCallMessage] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
expect(assistant).toBeDefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBeUndefined();
|
||||
expect(toObject(assistant)?.reasoning_content).toBeUndefined();
|
||||
});
|
||||
|
||||
it("does not replay streamed reasoning fields for kimi on opencode-go", () => {
|
||||
@@ -995,9 +1053,9 @@ describe("kimi model detection via detectCompat", () => {
|
||||
if (!assistantObject) {
|
||||
throw new Error("assistant message missing");
|
||||
}
|
||||
expect(Reflect.get(assistantObject, "reasoning")).toBeUndefined();
|
||||
expect(Reflect.get(assistantObject, "reasoning_content")).toBeUndefined();
|
||||
expect(Reflect.get(assistantObject, "reasoning_text")).toBeUndefined();
|
||||
expect(assistantObject.reasoning).toBeUndefined();
|
||||
expect(assistantObject.reasoning_content).toBeUndefined();
|
||||
expect(assistantObject.reasoning_text).toBeUndefined();
|
||||
});
|
||||
|
||||
// #1484: OpenCode Zen's Kimi gateway now 400s with `thinking is enabled but
|
||||
@@ -1071,10 +1129,10 @@ describe("kimi model detection via detectCompat", () => {
|
||||
const payload = (await promise) as { messages: Array<Record<string, unknown>> };
|
||||
const assistant = payload.messages.find(m => m.role === "assistant");
|
||||
expect(assistant).toBeDefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("Need to read the file before answering.");
|
||||
expect(assistant?.reasoning_content).toBe("Need to read the file before answering.");
|
||||
// The streamed `reasoning` key must NOT land in the wire body alongside
|
||||
// `reasoning_content`; opencode's strict schema rejects unknown fields.
|
||||
expect(Reflect.get(assistant as object, "reasoning")).toBeUndefined();
|
||||
expect(assistant?.reasoning).toBeUndefined();
|
||||
});
|
||||
|
||||
// #1071 regression guard alongside the #1484 fix: with thinking disabled the
|
||||
@@ -1137,9 +1195,9 @@ describe("kimi model detection via detectCompat", () => {
|
||||
const payload = (await promise) as { messages: Array<Record<string, unknown>> };
|
||||
const assistant = payload.messages.find(m => m.role === "assistant");
|
||||
expect(assistant).toBeDefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBeUndefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning")).toBeUndefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_text")).toBeUndefined();
|
||||
expect(assistant?.reasoning_content).toBeUndefined();
|
||||
expect(assistant?.reasoning).toBeUndefined();
|
||||
expect(assistant?.reasoning_text).toBeUndefined();
|
||||
});
|
||||
|
||||
// #1485 review: `disableReasoningOnForcedToolChoice` strips thinking from
|
||||
@@ -1226,9 +1284,9 @@ describe("kimi model detection via detectCompat", () => {
|
||||
};
|
||||
const assistant = payload.messages.find(m => m.role === "assistant");
|
||||
expect(assistant).toBeDefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBeUndefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning")).toBeUndefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_text")).toBeUndefined();
|
||||
expect(assistant?.reasoning_content).toBeUndefined();
|
||||
expect(assistant?.reasoning).toBeUndefined();
|
||||
expect(assistant?.reasoning_text).toBeUndefined();
|
||||
// The forced-tool guard must still strip the request-level thinking
|
||||
// signal so neither end of the wire mentions reasoning.
|
||||
expect(payload.reasoning_effort).toBeUndefined();
|
||||
@@ -1319,7 +1377,7 @@ describe("kimi model detection via detectCompat", () => {
|
||||
};
|
||||
const assistant = payload.messages.find(m => m.role === "assistant");
|
||||
expect(assistant).toBeDefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("Plan first, then call the tool.");
|
||||
expect(assistant?.reasoning_content).toBe("Plan first, then call the tool.");
|
||||
expect(payload.reasoning_effort).toBe("high");
|
||||
expect(payload.tool_choice).toBe("auto");
|
||||
});
|
||||
@@ -1400,10 +1458,10 @@ describe("kimi model detection via detectCompat", () => {
|
||||
const payload = (await promise) as { messages: Array<Record<string, unknown>> };
|
||||
const assistant = payload.messages.find(m => m.role === "assistant");
|
||||
expect(assistant).toBeDefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("Need to read the file before answering.");
|
||||
expect(assistant?.reasoning_content).toBe("Need to read the file before answering.");
|
||||
// DeepSeek's allowsSynthetic=false must keep the stale `reasoning` key
|
||||
// off the wire body so opencode's schema validation does not flag it.
|
||||
expect(Reflect.get(assistant as object, "reasoning")).toBeUndefined();
|
||||
expect(assistant?.reasoning).toBeUndefined();
|
||||
});
|
||||
|
||||
// #1484 follow-up: the Zen gateway invariant applies to every opencode-go
|
||||
@@ -1485,13 +1543,13 @@ describe("kimi model detection via detectCompat", () => {
|
||||
const assistant = payload.messages.find(m => m.role === "assistant");
|
||||
expect(assistant).toBeDefined();
|
||||
if (expectReplay) {
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("Plan before acting.");
|
||||
expect(assistant?.reasoning_content).toBe("Plan before acting.");
|
||||
// The stale streamed `reasoning` key must never land in the wire body.
|
||||
expect(Reflect.get(assistant as object, "reasoning")).toBeUndefined();
|
||||
expect(assistant?.reasoning).toBeUndefined();
|
||||
} else {
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBeUndefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning")).toBeUndefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_text")).toBeUndefined();
|
||||
expect(assistant?.reasoning_content).toBeUndefined();
|
||||
expect(assistant?.reasoning).toBeUndefined();
|
||||
expect(assistant?.reasoning_text).toBeUndefined();
|
||||
}
|
||||
});
|
||||
|
||||
@@ -1526,7 +1584,7 @@ describe("kimi model detection via detectCompat", () => {
|
||||
const messages = convertMessages(model, { messages: [toolCallMessage] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
expect(assistant).toBeDefined();
|
||||
const reasoningContent = Reflect.get(assistant as object, "reasoning_content");
|
||||
const reasoningContent = toObject(assistant)?.reasoning_content;
|
||||
expect(reasoningContent).toBeDefined();
|
||||
expect(typeof reasoningContent).toBe("string");
|
||||
expect((reasoningContent as string).length).toBeGreaterThan(0);
|
||||
@@ -1572,7 +1630,7 @@ describe("kimi model detection via detectCompat", () => {
|
||||
const messages = convertMessages(model, { messages: [toolCallMessage] }, compat);
|
||||
const assistant = messages.find(m => m.role === "assistant");
|
||||
expect(assistant).toBeDefined();
|
||||
expect(Reflect.get(assistant as object, "reasoning_content")).toBe(".");
|
||||
expect(toObject(assistant)?.reasoning_content).toBe(".");
|
||||
});
|
||||
|
||||
it("does not inject reasoning_content when model is not kimi", () => {
|
||||
@@ -1825,8 +1883,8 @@ describe("anthropic cache control for OpenAI-compatible chat completions", () =>
|
||||
const content = getLastPayloadContent(payload);
|
||||
const textPart = getLastTextPart(content);
|
||||
|
||||
expect(Reflect.get(textPart ?? {}, "text")).toBe("cache me");
|
||||
expect(Reflect.get(textPart ?? {}, "cache_control")).toEqual({ type: "ephemeral" });
|
||||
expect(textPart?.text).toBe("cache me");
|
||||
expect(textPart?.cache_control).toEqual({ type: "ephemeral" });
|
||||
});
|
||||
|
||||
it("preserves OpenRouter Anthropic cache_control detection", async () => {
|
||||
@@ -1835,7 +1893,7 @@ describe("anthropic cache control for OpenAI-compatible chat completions", () =>
|
||||
const content = getLastPayloadContent(payload);
|
||||
const textPart = getLastTextPart(content);
|
||||
|
||||
expect(Reflect.get(textPart ?? {}, "cache_control")).toEqual({ type: "ephemeral" });
|
||||
expect(textPart?.cache_control).toEqual({ type: "ephemeral" });
|
||||
});
|
||||
|
||||
it("does not attach Anthropic cache_control to empty assistant tool-call content", async () => {
|
||||
@@ -1877,16 +1935,16 @@ describe("anthropic cache control for OpenAI-compatible chat completions", () =>
|
||||
});
|
||||
const messages = getPayloadMessages(payload);
|
||||
const assistant = messages.find(message => {
|
||||
const toolCalls = Reflect.get(message, "tool_calls");
|
||||
return Reflect.get(message, "role") === "assistant" && Array.isArray(toolCalls);
|
||||
const toolCalls = message.tool_calls;
|
||||
return message.role === "assistant" && Array.isArray(toolCalls);
|
||||
});
|
||||
const firstUser = messages.find(message => Reflect.get(message, "role") === "user");
|
||||
const userContent = firstUser ? Reflect.get(firstUser, "content") : undefined;
|
||||
const firstUser = messages.find(message => message.role === "user");
|
||||
const userContent = firstUser?.content;
|
||||
const textPart = getLastTextPart(userContent);
|
||||
|
||||
expect(assistant ? Reflect.get(assistant, "content") : undefined).toBe("");
|
||||
expect(Reflect.get(textPart ?? {}, "text")).toBe("cache me");
|
||||
expect(Reflect.get(textPart ?? {}, "cache_control")).toEqual({ type: "ephemeral" });
|
||||
expect(assistant?.content).toBe("");
|
||||
expect(textPart?.text).toBe("cache me");
|
||||
expect(textPart?.cache_control).toEqual({ type: "ephemeral" });
|
||||
});
|
||||
|
||||
it("does not infer Anthropic cache_control for custom Claude ids without compat", async () => {
|
||||
|
||||
@@ -84,7 +84,7 @@ async function captureDisableReasoningPayload(model: Model<"openai-completions">
|
||||
return payload;
|
||||
}
|
||||
|
||||
describe("OpenAI completions disableReasoning", () => {
|
||||
describe("OpenAI completions disableReasoning and thinking dialects", () => {
|
||||
it("sends the lowest supported reasoning effort for generic effort-mode models", async () => {
|
||||
const payload = await captureDisableReasoningPayload(createReasoningEffortModel());
|
||||
|
||||
@@ -98,4 +98,186 @@ describe("OpenAI completions disableReasoning", () => {
|
||||
expect(payload.reasoning_effort).toBe("none");
|
||||
expect(payload.reasoning).toBeUndefined();
|
||||
});
|
||||
|
||||
// Additional requested tests for applyChatCompletionsReasoningParams dialect / behavior verification
|
||||
it("sets OpenRouter thinking disabled when disableReasoning: true", async () => {
|
||||
const model = buildModel({
|
||||
id: "openrouter-reasoner",
|
||||
name: "OpenRouter Reasoner",
|
||||
api: "openai-completions",
|
||||
provider: "openrouter",
|
||||
baseUrl: "https://openrouter.ai/api/v1",
|
||||
reasoning: true,
|
||||
compat: {
|
||||
thinkingFormat: "openrouter",
|
||||
supportsReasoningParams: true,
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128_000,
|
||||
maxTokens: 16_384,
|
||||
});
|
||||
|
||||
const payload = await captureDisableReasoningPayload(model);
|
||||
expect(payload.reasoning).toEqual({ enabled: false });
|
||||
});
|
||||
|
||||
it("sets Qwen enable_thinking: true with reasoning enabled, and false with forced tool choice", async () => {
|
||||
const model = buildModel({
|
||||
id: "qwen-reasoner",
|
||||
name: "Qwen Reasoner",
|
||||
api: "openai-completions",
|
||||
provider: "custom",
|
||||
baseUrl: "https://proxy.example.com/v1",
|
||||
reasoning: true,
|
||||
compat: {
|
||||
thinkingFormat: "qwen",
|
||||
supportsReasoningParams: true,
|
||||
supportsToolChoice: true,
|
||||
supportsForcedToolChoice: true,
|
||||
disableReasoningOnForcedToolChoice: true,
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128_000,
|
||||
maxTokens: 16_384,
|
||||
});
|
||||
|
||||
// 1. Enabled check using stream call
|
||||
const { promise, resolve } = Promise.withResolvers<unknown>();
|
||||
streamOpenAICompletions(model, testContext, {
|
||||
apiKey: "test-key",
|
||||
fetch: createMockFetchForQwen(resolve),
|
||||
reasoning: "medium",
|
||||
});
|
||||
const payloadEnabled = (await promise) as Record<string, unknown>;
|
||||
expect(payloadEnabled.enable_thinking).toBe(true);
|
||||
|
||||
// 2. Disabled on forced tool choice check
|
||||
const { promise: p2, resolve: r2 } = Promise.withResolvers<unknown>();
|
||||
streamOpenAICompletions(
|
||||
model,
|
||||
{
|
||||
messages: testContext.messages,
|
||||
tools: [
|
||||
{
|
||||
name: "read",
|
||||
description: "Read a file",
|
||||
parameters: { type: "object", properties: {} },
|
||||
},
|
||||
],
|
||||
},
|
||||
{
|
||||
apiKey: "test-key",
|
||||
fetch: createMockFetchForQwen(r2),
|
||||
reasoning: "medium",
|
||||
toolChoice: { type: "tool", name: "read" },
|
||||
},
|
||||
);
|
||||
const payloadDisabled = (await p2) as Record<string, unknown>;
|
||||
expect(payloadDisabled.enable_thinking).toBe(false);
|
||||
});
|
||||
|
||||
it("sets Qwen chat-template thinking format properly", async () => {
|
||||
const model = buildModel({
|
||||
id: "qwen-ct-reasoner",
|
||||
name: "Qwen Chat Template Reasoner",
|
||||
api: "openai-completions",
|
||||
provider: "custom",
|
||||
baseUrl: "https://proxy.example.com/v1",
|
||||
reasoning: true,
|
||||
compat: {
|
||||
thinkingFormat: "qwen-chat-template",
|
||||
supportsReasoningParams: true,
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128_000,
|
||||
maxTokens: 16_384,
|
||||
});
|
||||
|
||||
const { promise, resolve } = Promise.withResolvers<unknown>();
|
||||
streamOpenAICompletions(model, testContext, {
|
||||
apiKey: "test-key",
|
||||
fetch: createMockFetchForQwen(resolve),
|
||||
reasoning: "medium",
|
||||
});
|
||||
const payload = (await promise) as Record<string, unknown>;
|
||||
expect(payload.chat_template_kwargs).toEqual({ enable_thinking: true });
|
||||
});
|
||||
|
||||
it("sets Z.AI thinking format and toggles type logically based on forced tool choice", async () => {
|
||||
const model = buildModel({
|
||||
id: "zai-reasoner",
|
||||
name: "Z.AI Reasoner",
|
||||
api: "openai-completions",
|
||||
provider: "custom",
|
||||
baseUrl: "https://proxy.example.com/v1",
|
||||
reasoning: true,
|
||||
compat: {
|
||||
thinkingFormat: "zai",
|
||||
supportsReasoningParams: true,
|
||||
supportsToolChoice: true,
|
||||
supportsForcedToolChoice: true,
|
||||
disableReasoningOnForcedToolChoice: true,
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128_000,
|
||||
maxTokens: 16_384,
|
||||
});
|
||||
|
||||
const { promise, resolve } = Promise.withResolvers<unknown>();
|
||||
streamOpenAICompletions(model, testContext, {
|
||||
apiKey: "test-key",
|
||||
fetch: createMockFetchForQwen(resolve),
|
||||
reasoning: "medium",
|
||||
});
|
||||
const payloadEnabled = (await promise) as Record<string, unknown>;
|
||||
expect(payloadEnabled.thinking).toEqual({ type: "enabled" });
|
||||
|
||||
const { promise: p2, resolve: r2 } = Promise.withResolvers<unknown>();
|
||||
streamOpenAICompletions(
|
||||
model,
|
||||
{
|
||||
messages: testContext.messages,
|
||||
tools: [
|
||||
{
|
||||
name: "read",
|
||||
description: "Read a file",
|
||||
parameters: { type: "object", properties: {} },
|
||||
},
|
||||
],
|
||||
},
|
||||
{
|
||||
apiKey: "test-key",
|
||||
fetch: createMockFetchForQwen(r2),
|
||||
reasoning: "medium",
|
||||
toolChoice: { type: "tool", name: "read" },
|
||||
},
|
||||
);
|
||||
const payloadDisabled = (await p2) as Record<string, unknown>;
|
||||
expect(payloadDisabled.thinking).toEqual({ type: "disabled" });
|
||||
});
|
||||
});
|
||||
|
||||
function createMockFetchForQwen(resolve: (value: unknown) => void): FetchImpl {
|
||||
const fetchMock: FetchImpl = Object.assign(
|
||||
async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
||||
const payload = JSON.parse(typeof init?.body === "string" ? init.body : "{}") as Record<string, unknown>;
|
||||
resolve(payload);
|
||||
return createSseResponse([
|
||||
{
|
||||
id: "chatcmpl",
|
||||
object: "chat.completion.chunk",
|
||||
created: 0,
|
||||
model: "model",
|
||||
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
|
||||
},
|
||||
"[DONE]",
|
||||
]);
|
||||
},
|
||||
{ preconnect: fetch.preconnect },
|
||||
);
|
||||
return fetchMock;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,144 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import type { Context, FetchImpl, Model, ModelSpec, OpenAICompat, Tool } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
|
||||
// Each Chat Completions reasoning dialect carries "thinking is on" on a
|
||||
// different wire field. The `disableReasoningOnForcedToolChoice` /
|
||||
// `disableReasoningOnToolChoice` conflict policies must turn thinking OFF on the
|
||||
// *matching* field, not just delete `reasoning_effort` — otherwise a Qwen /
|
||||
// Qwen-template / OpenRouter request keeps thinking enabled and re-trips the
|
||||
// very 400 the policy exists to dodge. Both conflict branches funnel through the
|
||||
// same `disableChatCompletionsReasoningForDialect` helper, so the forced-tool
|
||||
// path below exercises that helper for every dialect.
|
||||
|
||||
function createAbortedSignal(): AbortSignal {
|
||||
const controller = new AbortController();
|
||||
controller.abort();
|
||||
return controller.signal;
|
||||
}
|
||||
|
||||
function createMockFetch(): FetchImpl {
|
||||
async function mockFetch(_input: string | URL | Request, _init?: RequestInit): Promise<Response> {
|
||||
return new Response("data: [DONE]\n\n", {
|
||||
status: 200,
|
||||
headers: { "content-type": "text/event-stream" },
|
||||
});
|
||||
}
|
||||
return Object.assign(mockFetch, { preconnect: fetch.preconnect });
|
||||
}
|
||||
|
||||
const readTool: Tool = {
|
||||
name: "read",
|
||||
description: "Read a file",
|
||||
parameters: {
|
||||
type: "object",
|
||||
properties: { path: { type: "string" } },
|
||||
required: ["path"],
|
||||
},
|
||||
};
|
||||
|
||||
function forcedToolContext(): Context {
|
||||
return {
|
||||
messages: [{ role: "user", content: "Summarize the README", timestamp: Date.now() }],
|
||||
tools: [readTool],
|
||||
};
|
||||
}
|
||||
|
||||
function reasoningDialectModel(
|
||||
thinkingFormat: NonNullable<OpenAICompat["thinkingFormat"]>,
|
||||
compatOverrides: Partial<OpenAICompat> = {},
|
||||
): Model<"openai-completions"> {
|
||||
return buildModel({
|
||||
id: "test-reasoning-model",
|
||||
name: "Test Reasoning Model",
|
||||
api: "openai-completions",
|
||||
provider: "custom-openai-compatible",
|
||||
baseUrl: "https://example.test/v1",
|
||||
reasoning: true,
|
||||
compat: {
|
||||
thinkingFormat,
|
||||
supportsReasoningParams: true,
|
||||
supportsReasoningEffort: true,
|
||||
supportsToolChoice: true,
|
||||
supportsForcedToolChoice: true,
|
||||
disableReasoningOnForcedToolChoice: true,
|
||||
...compatOverrides,
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 1_000_000,
|
||||
maxTokens: 131_072,
|
||||
} satisfies ModelSpec<"openai-completions">);
|
||||
}
|
||||
|
||||
async function captureForcedToolPayload(model: Model<"openai-completions">): Promise<Record<string, unknown>> {
|
||||
const { promise, resolve } = Promise.withResolvers<unknown>();
|
||||
streamOpenAICompletions(model, forcedToolContext(), {
|
||||
apiKey: "test-key",
|
||||
fetch: createMockFetch(),
|
||||
signal: createAbortedSignal(),
|
||||
reasoning: "high",
|
||||
toolChoice: { type: "tool", name: "read" },
|
||||
onPayload: payload => resolve(payload),
|
||||
});
|
||||
const payload = await promise;
|
||||
if (typeof payload !== "object" || payload === null) throw new Error("Expected captured request payload");
|
||||
return payload as Record<string, unknown>;
|
||||
}
|
||||
|
||||
describe("Chat Completions reasoning-disable conflict policy (per dialect)", () => {
|
||||
it("disables Z.AI thinking and drops reasoning_effort on forced tool choice", async () => {
|
||||
const payload = await captureForcedToolPayload(reasoningDialectModel("zai"));
|
||||
expect(payload.thinking).toEqual({ type: "disabled" });
|
||||
expect(payload.reasoning_effort).toBeUndefined();
|
||||
});
|
||||
|
||||
it("sets Qwen enable_thinking=false on forced tool choice", async () => {
|
||||
const payload = await captureForcedToolPayload(reasoningDialectModel("qwen"));
|
||||
expect(payload.enable_thinking).toBe(false);
|
||||
expect(payload.reasoning_effort).toBeUndefined();
|
||||
});
|
||||
|
||||
it("sets Qwen chat-template enable_thinking=false on forced tool choice", async () => {
|
||||
const payload = await captureForcedToolPayload(reasoningDialectModel("qwen-chat-template"));
|
||||
expect(payload.chat_template_kwargs).toEqual({ enable_thinking: false });
|
||||
expect(payload.reasoning_effort).toBeUndefined();
|
||||
});
|
||||
|
||||
it("sets OpenRouter reasoning={enabled:false} (never just deleted) on forced tool choice", async () => {
|
||||
const payload = await captureForcedToolPayload(reasoningDialectModel("openrouter"));
|
||||
// OpenRouter defaults reasoning models back to thinking when `reasoning` is
|
||||
// absent, so suppression must explicitly disable rather than delete it.
|
||||
expect(payload.reasoning).toEqual({ enabled: false });
|
||||
expect(payload.reasoning_effort).toBeUndefined();
|
||||
});
|
||||
|
||||
it("drops stale reasoning_effort for OpenAI-style effort endpoints on forced tool choice", async () => {
|
||||
const payload = await captureForcedToolPayload(reasoningDialectModel("openai"));
|
||||
expect(payload.reasoning_effort).toBeUndefined();
|
||||
expect(payload.reasoning).toBeUndefined();
|
||||
});
|
||||
|
||||
it("applies the same per-dialect disable on the non-forced disableReasoningOnToolChoice path", async () => {
|
||||
// Non-forced branch (any tool_choice) shares the dialect-aware helper:
|
||||
// an OpenRouter reasoning model with disableReasoningOnToolChoice must emit
|
||||
// reasoning={enabled:false} rather than leaving the effort object in place.
|
||||
const model = reasoningDialectModel("openrouter", {
|
||||
disableReasoningOnForcedToolChoice: false,
|
||||
disableReasoningOnToolChoice: true,
|
||||
});
|
||||
const { promise, resolve } = Promise.withResolvers<unknown>();
|
||||
streamOpenAICompletions(model, forcedToolContext(), {
|
||||
apiKey: "test-key",
|
||||
fetch: createMockFetch(),
|
||||
signal: createAbortedSignal(),
|
||||
reasoning: "high",
|
||||
toolChoice: "auto",
|
||||
onPayload: payload => resolve(payload),
|
||||
});
|
||||
const payload = (await promise) as Record<string, unknown>;
|
||||
expect(payload.tool_choice).toBe("auto");
|
||||
expect(payload.reasoning).toEqual({ enabled: false });
|
||||
});
|
||||
});
|
||||
@@ -31,8 +31,13 @@ const compat: ResolvedOpenAICompat = {
|
||||
requiresThinkingAsText: false,
|
||||
requiresMistralToolIds: false,
|
||||
thinkingFormat: "openai",
|
||||
reasoningDisableMode: "lowest-effort",
|
||||
omitReasoningEffort: false,
|
||||
includeEncryptedReasoning: true,
|
||||
filterReasoningHistory: false,
|
||||
reasoningContentField: "reasoning_content",
|
||||
requiresReasoningContentForToolCalls: false,
|
||||
requiresReasoningContentForAllAssistantTurns: false,
|
||||
allowsSyntheticReasoningContentForToolCalls: true,
|
||||
requiresAssistantContentForToolCalls: false,
|
||||
openRouterRouting: {},
|
||||
@@ -44,6 +49,12 @@ const compat: ResolvedOpenAICompat = {
|
||||
alwaysSendMaxTokens: false,
|
||||
isOpenRouterHost: false,
|
||||
isVercelGatewayHost: false,
|
||||
wireModelIdMode: "raw",
|
||||
stripDeepseekSpecialTokens: false,
|
||||
reasoningDeltasMayBeCumulative: false,
|
||||
emptyLengthFinishIsContextError: false,
|
||||
usesOpenAIToolCallIdLimit: false,
|
||||
dropThinkingWhenReasoningEffort: false,
|
||||
};
|
||||
|
||||
function buildToolResult(toolCallId: string, timestamp: number): ToolResultMessage {
|
||||
|
||||
@@ -672,7 +672,7 @@ describe("OpenAI-family first-event timeouts", () => {
|
||||
);
|
||||
});
|
||||
|
||||
it("errors when OpenAI responses stream closes without response.completed", async () => {
|
||||
it("errors when OpenAI responses stream closes without a terminal response event", async () => {
|
||||
const incompleteResponse = createSseResponse([
|
||||
{ type: "response.created", response: { id: "resp_incomplete" } },
|
||||
{
|
||||
@@ -691,7 +691,7 @@ describe("OpenAI-family first-event timeouts", () => {
|
||||
content: [{ type: "output_text", text: "Hello" }],
|
||||
},
|
||||
},
|
||||
// Intentionally no response.completed — simulates premature provider disconnect.
|
||||
// Intentionally no response.completed/incomplete — simulates premature provider disconnect.
|
||||
]);
|
||||
const fetchMock: FetchImpl = () => Promise.resolve(incompleteResponse);
|
||||
const result = await streamOpenAIResponses(openAIResponsesModel, baseContext(), {
|
||||
@@ -700,13 +700,13 @@ describe("OpenAI-family first-event timeouts", () => {
|
||||
}).result();
|
||||
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorMessage).toBe("OpenAI responses stream closed before response.completed was received");
|
||||
expect(result.errorMessage).toBe("OpenAI responses stream closed before a terminal response event was received");
|
||||
expect(result.content as unknown[]).toEqual([
|
||||
{ type: "text", text: "Hello", textSignature: '{"v":1,"id":"msg_incomplete"}' },
|
||||
]);
|
||||
});
|
||||
|
||||
it("errors when Azure OpenAI responses stream closes without response.completed", async () => {
|
||||
it("errors when Azure OpenAI responses stream closes without a terminal response event", async () => {
|
||||
const incompleteResponse = createSseResponse([
|
||||
{ type: "response.created", response: { id: "resp_incomplete_azure" } },
|
||||
{
|
||||
@@ -731,7 +731,7 @@ describe("OpenAI-family first-event timeouts", () => {
|
||||
content: [{ type: "output_text", text: "Hello azure" }],
|
||||
},
|
||||
},
|
||||
// Intentionally no response.completed — simulates premature provider disconnect.
|
||||
// Intentionally no response.completed/incomplete — simulates premature provider disconnect.
|
||||
]);
|
||||
const fetchMock: FetchImpl = () => Promise.resolve(incompleteResponse);
|
||||
const result = await streamAzureOpenAIResponses(azureOpenAIResponsesModel, baseContext(), {
|
||||
@@ -742,7 +742,9 @@ describe("OpenAI-family first-event timeouts", () => {
|
||||
}).result();
|
||||
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorMessage).toBe("Azure OpenAI responses stream closed before response.completed was received");
|
||||
expect(result.errorMessage).toBe(
|
||||
"Azure OpenAI responses stream closed before a terminal response event was received",
|
||||
);
|
||||
expect(result.content as unknown[]).toEqual([
|
||||
{ type: "text", text: "Hello azure", textSignature: '{"v":1,"id":"msg_incomplete_azure"}' },
|
||||
]);
|
||||
|
||||
@@ -6,13 +6,13 @@ import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
|
||||
// Output-token wire policy for OpenAI-family providers:
|
||||
// - Non-aggregator completions + all responses: clamp to OPENAI_MAX_OUTPUT_TOKENS
|
||||
// (mirrors Anthropic's cap) so a catalog maxTokens that tracks the context
|
||||
// window never overflows the upstream.
|
||||
// - OpenRouter completions: omit default max_tokens entirely. OpenRouter filters
|
||||
// out any upstream whose output cap is below the requested value (e.g. Cerebras
|
||||
// GLM-4.7 ~40k), silently defeating provider routing. Explicit caller caps and
|
||||
// Kimi via OpenRouter are still sent.
|
||||
// - Non-aggregator completions + non-OpenRouter responses: clamp to
|
||||
// OPENAI_MAX_OUTPUT_TOKENS (mirrors Anthropic's cap) so a catalog maxTokens
|
||||
// that tracks the context window never overflows the upstream.
|
||||
// - OpenRouter completions/responses: omit default max token fields entirely.
|
||||
// OpenRouter filters out any upstream whose output cap is below the requested
|
||||
// value (e.g. Cerebras GLM-4.7 ~40k), silently defeating provider routing.
|
||||
// Explicit caller caps and Kimi chat-completions via OpenRouter are still sent.
|
||||
|
||||
const ctx: Context = {
|
||||
systemPrompt: ["hi"],
|
||||
@@ -43,13 +43,30 @@ function captureResponsesBody(): { fetchMock: FetchImpl; captured: Record<string
|
||||
return { fetchMock, captured };
|
||||
}
|
||||
|
||||
async function drainResponses(model: Model<"openai-responses">): Promise<Record<string, unknown>> {
|
||||
const { fetchMock, captured } = captureResponsesBody();
|
||||
const stream = streamSimple(model, ctx, { apiKey: "k", fetch: fetchMock });
|
||||
for await (const event of stream) {
|
||||
if (event.type === "done" || event.type === "error") break;
|
||||
async function drainResponses(
|
||||
model: Model<"openai-responses" | "openrouter">,
|
||||
maxTokens?: number,
|
||||
): Promise<Record<string, unknown>> {
|
||||
const previousOpenRouterResponses = Bun.env.PI_OPENROUTER_RESPONSES;
|
||||
if (model.api === "openrouter") Bun.env.PI_OPENROUTER_RESPONSES = "1";
|
||||
try {
|
||||
const { fetchMock, captured } = captureResponsesBody();
|
||||
const stream = streamSimple(model, ctx, {
|
||||
apiKey: "k",
|
||||
...(maxTokens === undefined ? {} : { maxTokens }),
|
||||
fetch: fetchMock,
|
||||
});
|
||||
for await (const event of stream) {
|
||||
if (event.type === "done" || event.type === "error") break;
|
||||
}
|
||||
return captured;
|
||||
} finally {
|
||||
if (previousOpenRouterResponses === undefined) {
|
||||
delete Bun.env.PI_OPENROUTER_RESPONSES;
|
||||
} else {
|
||||
Bun.env.PI_OPENROUTER_RESPONSES = previousOpenRouterResponses;
|
||||
}
|
||||
}
|
||||
return captured;
|
||||
}
|
||||
|
||||
function completionsSse(): Response {
|
||||
@@ -110,6 +127,21 @@ function glmCompletionsModel(maxTokens: number): Model<"openai-completions"> {
|
||||
});
|
||||
}
|
||||
|
||||
function openRouterResponsesModel(maxTokens: number): Model<"openrouter"> {
|
||||
return buildModel({
|
||||
id: "z-ai/glm-4.7",
|
||||
name: "GLM 4.7",
|
||||
api: "openrouter",
|
||||
provider: "openrouter",
|
||||
baseUrl: "https://openrouter.ai/api/v1",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 202_752,
|
||||
maxTokens,
|
||||
});
|
||||
}
|
||||
|
||||
// Non-aggregator completions model: the 64k clamp applies (max_tokens is sent).
|
||||
function directCompletionsModel(maxTokens: number): Model<"openai-completions"> {
|
||||
return buildModel({
|
||||
@@ -155,6 +187,16 @@ describe("OpenAI-family output-token cap", () => {
|
||||
expect(body.max_output_tokens).toBe(OPENAI_MAX_OUTPUT_TOKENS);
|
||||
});
|
||||
|
||||
it("omits default max_output_tokens for OpenRouter Responses so provider routing is not filtered", async () => {
|
||||
const body = await drainResponses(openRouterResponsesModel(131_072));
|
||||
expect(body.max_output_tokens).toBeUndefined();
|
||||
});
|
||||
|
||||
it("sends explicit maxTokens for OpenRouter Responses caller caps", async () => {
|
||||
const body = await drainResponses(openRouterResponsesModel(131_072), 2_048);
|
||||
expect(body.max_output_tokens).toBe(2_048);
|
||||
});
|
||||
|
||||
it("clamps non-aggregator completions output to the 64k ceiling", async () => {
|
||||
const body = await captureCompletionsBody(directCompletionsModel(131_072), 131_072);
|
||||
expect(body.max_completion_tokens ?? body.max_tokens).toBe(OPENAI_MAX_OUTPUT_TOKENS);
|
||||
|
||||
@@ -0,0 +1,169 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import {
|
||||
applyOpenAIGatewayRouting,
|
||||
type OpenAIGatewayRoutingCompat,
|
||||
type OpenAIGatewayRoutingParams,
|
||||
type ResolveOpenAIOutputTokenInput,
|
||||
resolveOpenAIOutputTokenParam,
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
|
||||
const OPENAI_MAX_OUTPUT_TOKENS = 64_000;
|
||||
|
||||
function tokenInput(overrides: Partial<ResolveOpenAIOutputTokenInput> = {}): ResolveOpenAIOutputTokenInput {
|
||||
return {
|
||||
field: "max_completion_tokens",
|
||||
maxTokens: undefined,
|
||||
maxTokensExplicit: false,
|
||||
modelMaxTokens: 131_072,
|
||||
omitMaxOutputTokens: false,
|
||||
isOpenRouterHost: false,
|
||||
alwaysSendMaxTokens: false,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
describe("resolveOpenAIOutputTokenParam", () => {
|
||||
it("returns the requested cap clamped to the provider ceiling", () => {
|
||||
expect(resolveOpenAIOutputTokenParam(tokenInput({ maxTokens: 200_000 }))).toEqual({
|
||||
field: "max_completion_tokens",
|
||||
value: OPENAI_MAX_OUTPUT_TOKENS,
|
||||
});
|
||||
});
|
||||
|
||||
it("clamps to the model cap when it is below the provider ceiling", () => {
|
||||
expect(resolveOpenAIOutputTokenParam(tokenInput({ maxTokens: 50_000, modelMaxTokens: 32_000 }))).toEqual({
|
||||
field: "max_completion_tokens",
|
||||
value: 32_000,
|
||||
});
|
||||
});
|
||||
|
||||
it("honors a raised provider clamp (GLM-5.2 reasoning) above the default ceiling", () => {
|
||||
expect(
|
||||
resolveOpenAIOutputTokenParam(
|
||||
tokenInput({ maxTokens: 120_000, modelMaxTokens: 131_072, providerOutputClamp: 131_072 }),
|
||||
),
|
||||
).toEqual({ field: "max_completion_tokens", value: 120_000 });
|
||||
});
|
||||
|
||||
it("selects the wire field name the endpoint requested", () => {
|
||||
expect(resolveOpenAIOutputTokenParam(tokenInput({ field: "max_tokens", maxTokens: 1024 }))?.field).toBe(
|
||||
"max_tokens",
|
||||
);
|
||||
expect(resolveOpenAIOutputTokenParam(tokenInput({ field: "max_output_tokens", maxTokens: 1024 }))?.field).toBe(
|
||||
"max_output_tokens",
|
||||
);
|
||||
});
|
||||
|
||||
it("returns undefined when no cap was requested and the endpoint does not require one", () => {
|
||||
expect(resolveOpenAIOutputTokenParam(tokenInput({ maxTokens: undefined }))).toBeUndefined();
|
||||
});
|
||||
|
||||
it("defaults from the model cap when alwaysSendMaxTokens and the caller omitted one", () => {
|
||||
// Kimi-family TPM math: clamp(model cap, provider ceiling).
|
||||
expect(resolveOpenAIOutputTokenParam(tokenInput({ alwaysSendMaxTokens: true, maxTokens: undefined }))).toEqual({
|
||||
field: "max_completion_tokens",
|
||||
value: OPENAI_MAX_OUTPUT_TOKENS,
|
||||
});
|
||||
});
|
||||
|
||||
it("falls back to the provider ceiling when alwaysSendMaxTokens and no model cap exists", () => {
|
||||
expect(
|
||||
resolveOpenAIOutputTokenParam(
|
||||
tokenInput({ alwaysSendMaxTokens: true, maxTokens: undefined, modelMaxTokens: undefined }),
|
||||
),
|
||||
).toEqual({ field: "max_completion_tokens", value: OPENAI_MAX_OUTPUT_TOKENS });
|
||||
});
|
||||
|
||||
it("omits the catalog default for OpenRouter when the caller did not explicitly set a cap", () => {
|
||||
expect(
|
||||
resolveOpenAIOutputTokenParam(
|
||||
tokenInput({ isOpenRouterHost: true, maxTokens: 131_072, maxTokensExplicit: false }),
|
||||
),
|
||||
).toBeUndefined();
|
||||
});
|
||||
|
||||
it("preserves an explicit caller cap for OpenRouter", () => {
|
||||
expect(
|
||||
resolveOpenAIOutputTokenParam(
|
||||
tokenInput({ isOpenRouterHost: true, maxTokens: 2048, maxTokensExplicit: true }),
|
||||
),
|
||||
).toEqual({ field: "max_completion_tokens", value: 2048 });
|
||||
});
|
||||
|
||||
it("keeps the cap for OpenRouter models that require it (alwaysSendMaxTokens overrides routing omission)", () => {
|
||||
expect(
|
||||
resolveOpenAIOutputTokenParam(
|
||||
tokenInput({
|
||||
field: "max_output_tokens",
|
||||
isOpenRouterHost: true,
|
||||
alwaysSendMaxTokens: true,
|
||||
maxTokens: 131_072,
|
||||
maxTokensExplicit: false,
|
||||
}),
|
||||
),
|
||||
).toEqual({ field: "max_output_tokens", value: OPENAI_MAX_OUTPUT_TOKENS });
|
||||
});
|
||||
|
||||
it("drops the field entirely when omitMaxOutputTokens is set (Ollama-style proxies)", () => {
|
||||
expect(resolveOpenAIOutputTokenParam(tokenInput({ maxTokens: 4096, omitMaxOutputTokens: true }))).toBeUndefined();
|
||||
});
|
||||
|
||||
it("treats a null caller cap like an omitted one", () => {
|
||||
expect(resolveOpenAIOutputTokenParam(tokenInput({ maxTokens: null }))).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
function routingParams(): OpenAIGatewayRoutingParams {
|
||||
return {};
|
||||
}
|
||||
|
||||
describe("applyOpenAIGatewayRouting", () => {
|
||||
it("sets the OpenRouter provider routing block", () => {
|
||||
const routing = { only: ["anthropic"], order: ["anthropic", "openai"] };
|
||||
const params = routingParams();
|
||||
applyOpenAIGatewayRouting(params, { isOpenRouterHost: true, openRouterRouting: routing });
|
||||
expect(params.provider).toEqual(routing);
|
||||
expect(params.providerOptions).toBeUndefined();
|
||||
});
|
||||
|
||||
it("does not set provider when the host is not OpenRouter", () => {
|
||||
const params = routingParams();
|
||||
applyOpenAIGatewayRouting(params, {
|
||||
isOpenRouterHost: false,
|
||||
openRouterRouting: { only: ["anthropic"] },
|
||||
});
|
||||
expect(params.provider).toBeUndefined();
|
||||
});
|
||||
|
||||
it("maps Vercel gateway routing to providerOptions.gateway", () => {
|
||||
const params = routingParams();
|
||||
const compat: OpenAIGatewayRoutingCompat = {
|
||||
isOpenRouterHost: false,
|
||||
isVercelGatewayHost: true,
|
||||
vercelGatewayRouting: { only: ["bedrock"], order: ["bedrock", "anthropic"] },
|
||||
};
|
||||
applyOpenAIGatewayRouting(params, compat);
|
||||
expect(params.providerOptions).toEqual({ gateway: { only: ["bedrock"], order: ["bedrock", "anthropic"] } });
|
||||
expect(params.provider).toBeUndefined();
|
||||
});
|
||||
|
||||
it("ignores Vercel gateway routing with neither only nor order", () => {
|
||||
const params = routingParams();
|
||||
applyOpenAIGatewayRouting(params, {
|
||||
isOpenRouterHost: false,
|
||||
isVercelGatewayHost: true,
|
||||
vercelGatewayRouting: {},
|
||||
});
|
||||
expect(params.providerOptions).toBeUndefined();
|
||||
});
|
||||
|
||||
it("does not apply Vercel routing when the host is not the Vercel gateway", () => {
|
||||
const params = routingParams();
|
||||
applyOpenAIGatewayRouting(params, {
|
||||
isOpenRouterHost: false,
|
||||
isVercelGatewayHost: false,
|
||||
vercelGatewayRouting: { only: ["bedrock"] },
|
||||
});
|
||||
expect(params.providerOptions).toBeUndefined();
|
||||
});
|
||||
});
|
||||
@@ -1,9 +1,38 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import { type OpenAIResponsesOptions, streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
|
||||
import { stream as streamModel } from "@oh-my-pi/pi-ai/stream";
|
||||
import type { Context, FetchImpl, Model, ProviderSessionState } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
|
||||
const model = getBundledModel("openai", "gpt-5-mini") as Model<"openai-responses">;
|
||||
const openRouterResponsesModel: Model<"openai-responses"> = {
|
||||
...model,
|
||||
id: "openai/gpt-5.5",
|
||||
name: "OpenRouter GPT 5.5",
|
||||
provider: "openrouter",
|
||||
baseUrl: "https://openrouter.ai/api/v1",
|
||||
compat: buildOpenAIResponsesCompat({
|
||||
id: "openai/gpt-5.5",
|
||||
name: "OpenRouter GPT 5.5",
|
||||
provider: "openrouter",
|
||||
baseUrl: "https://openrouter.ai/api/v1",
|
||||
}),
|
||||
};
|
||||
const xaiOAuthResponsesModel: Model<"openai-responses"> = {
|
||||
...model,
|
||||
id: "grok-build",
|
||||
name: "Grok Build",
|
||||
provider: "xai-oauth",
|
||||
baseUrl: "https://api.x.ai/v1",
|
||||
compat: buildOpenAIResponsesCompat({
|
||||
id: "grok-build",
|
||||
name: "Grok Build",
|
||||
provider: "xai-oauth",
|
||||
baseUrl: "https://api.x.ai/v1",
|
||||
reasoning: true,
|
||||
}),
|
||||
};
|
||||
|
||||
function createSseResponse(events: unknown[]): Response {
|
||||
const payload = `${events.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`;
|
||||
@@ -19,15 +48,23 @@ function getHeader(headers: RequestInit["headers"], name: string): string | null
|
||||
|
||||
async function captureOpenAIResponseHeaders(
|
||||
options: OpenAIResponsesOptions,
|
||||
): Promise<{ sessionId: string | null; clientRequestId: string | null; body: Record<string, unknown> | null }> {
|
||||
requestModel: Model<"openai-responses"> = model,
|
||||
): Promise<{
|
||||
sessionId: string | null;
|
||||
clientRequestId: string | null;
|
||||
headers: Headers;
|
||||
body: Record<string, unknown> | null;
|
||||
}> {
|
||||
const captured = {
|
||||
sessionId: null as string | null,
|
||||
clientRequestId: null as string | null,
|
||||
headers: new Headers(),
|
||||
body: null as Record<string, unknown> | null,
|
||||
};
|
||||
const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => {
|
||||
captured.sessionId = getHeader(init?.headers, "session_id");
|
||||
captured.clientRequestId = getHeader(init?.headers, "x-client-request-id");
|
||||
captured.headers = new Headers(init?.headers);
|
||||
captured.body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : null;
|
||||
return createSseResponse([
|
||||
{
|
||||
@@ -65,7 +102,72 @@ async function captureOpenAIResponseHeaders(
|
||||
systemPrompt: ["stable system", "stable durable context"],
|
||||
messages: [{ role: "user", content: "hi", timestamp: Date.now() }],
|
||||
};
|
||||
const stream = streamOpenAIResponses(model, context, { apiKey: "test-key", ...options, fetch: fetchMock });
|
||||
const stream = streamOpenAIResponses(requestModel, context, { apiKey: "test-key", ...options, fetch: fetchMock });
|
||||
|
||||
for await (const event of stream) {
|
||||
if (event.type === "done" || event.type === "error") break;
|
||||
}
|
||||
|
||||
return captured;
|
||||
}
|
||||
|
||||
async function captureDispatchedOpenAIResponseHeaders(
|
||||
options: OpenAIResponsesOptions,
|
||||
requestModel: Model<"openai-responses">,
|
||||
): Promise<{
|
||||
sessionId: string | null;
|
||||
clientRequestId: string | null;
|
||||
headers: Headers;
|
||||
body: Record<string, unknown> | null;
|
||||
}> {
|
||||
const captured = {
|
||||
sessionId: null as string | null,
|
||||
clientRequestId: null as string | null,
|
||||
headers: new Headers(),
|
||||
body: null as Record<string, unknown> | null,
|
||||
};
|
||||
const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => {
|
||||
captured.sessionId = getHeader(init?.headers, "session_id");
|
||||
captured.clientRequestId = getHeader(init?.headers, "x-client-request-id");
|
||||
captured.headers = new Headers(init?.headers);
|
||||
captured.body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : null;
|
||||
return createSseResponse([
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
item: { type: "message", id: "msg_1", role: "assistant", status: "in_progress", content: [] },
|
||||
},
|
||||
{ type: "response.content_part.added", part: { type: "output_text", text: "" } },
|
||||
{ type: "response.output_text.delta", delta: "Hello" },
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
item: {
|
||||
type: "message",
|
||||
id: "msg_1",
|
||||
role: "assistant",
|
||||
status: "completed",
|
||||
content: [{ type: "output_text", text: "Hello" }],
|
||||
},
|
||||
},
|
||||
{
|
||||
type: "response.completed",
|
||||
response: {
|
||||
status: "completed",
|
||||
usage: {
|
||||
input_tokens: 5,
|
||||
output_tokens: 3,
|
||||
total_tokens: 8,
|
||||
input_tokens_details: { cached_tokens: 0 },
|
||||
},
|
||||
},
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
const context: Context = {
|
||||
systemPrompt: ["stable system", "stable durable context"],
|
||||
messages: [{ role: "user", content: "hi", timestamp: Date.now() }],
|
||||
};
|
||||
const stream = streamModel(requestModel, context, { apiKey: "test-key", ...options, fetch: fetchMock });
|
||||
|
||||
for await (const event of stream) {
|
||||
if (event.type === "done" || event.type === "error") break;
|
||||
@@ -111,6 +213,102 @@ describe("openai-responses cache affinity", () => {
|
||||
expect(captured.body?.prompt_cache_key).toBe("session-123");
|
||||
});
|
||||
|
||||
it("xAI OAuth adapter request shaping does not mutate reused options", async () => {
|
||||
const options: OpenAIResponsesOptions = {
|
||||
sessionId: "session-123",
|
||||
headers: { existing: "header" },
|
||||
extraBody: { existing: true },
|
||||
};
|
||||
|
||||
const first = await captureDispatchedOpenAIResponseHeaders(options, xaiOAuthResponsesModel);
|
||||
const second = await captureDispatchedOpenAIResponseHeaders(options, xaiOAuthResponsesModel);
|
||||
|
||||
expect(options).toEqual({
|
||||
sessionId: "session-123",
|
||||
headers: { existing: "header" },
|
||||
extraBody: { existing: true },
|
||||
});
|
||||
for (const captured of [first, second]) {
|
||||
expect(getHeader(captured.headers, "x-grok-conv-id")).toBe("session-123");
|
||||
expect(captured.body?.prompt_cache_key).toBe("session-123");
|
||||
expect(captured.body?.existing).toBe(true);
|
||||
expect(captured.body?.reasoning).toBeUndefined();
|
||||
}
|
||||
});
|
||||
|
||||
it("sets OpenRouter Responses session_id from sessionId in the body", async () => {
|
||||
const captured = await captureOpenAIResponseHeaders(
|
||||
{ sessionId: "workflow-123", promptCacheKey: "cache-key-123" },
|
||||
openRouterResponsesModel,
|
||||
);
|
||||
|
||||
expect(captured.sessionId).toBeNull();
|
||||
expect(captured.clientRequestId).toBeNull();
|
||||
expect(captured.body?.session_id).toBe("workflow-123");
|
||||
expect(captured.body?.prompt_cache_key).toBe("cache-key-123");
|
||||
});
|
||||
|
||||
it("lets explicit headers override OpenRouter Responses defaults", async () => {
|
||||
const captured = await captureOpenAIResponseHeaders(
|
||||
{
|
||||
headers: {
|
||||
"HTTP-Referer": "https://example.test/",
|
||||
"X-OpenRouter-Title": "Custom App",
|
||||
"X-OpenRouter-Cache": "false",
|
||||
},
|
||||
},
|
||||
openRouterResponsesModel,
|
||||
);
|
||||
|
||||
expect(getHeader(captured.headers, "HTTP-Referer")).toBe("https://example.test/");
|
||||
expect(getHeader(captured.headers, "X-OpenRouter-Title")).toBe("Custom App");
|
||||
expect(getHeader(captured.headers, "X-OpenRouter-Cache")).toBe("false");
|
||||
});
|
||||
|
||||
it("applies OpenRouter Responses model variants and provider routing to the body", async () => {
|
||||
const routedModel: Model<"openai-responses"> = {
|
||||
...openRouterResponsesModel,
|
||||
compat: {
|
||||
...openRouterResponsesModel.compat,
|
||||
openRouterRouting: { only: ["anthropic"], order: ["anthropic"] },
|
||||
},
|
||||
};
|
||||
const captured = await captureOpenAIResponseHeaders({ openrouterVariant: "nitro" }, routedModel);
|
||||
|
||||
expect(captured.body?.model).toBe("openai/gpt-5.5:nitro");
|
||||
expect(captured.body?.provider).toEqual({ only: ["anthropic"], order: ["anthropic"] });
|
||||
});
|
||||
|
||||
it("keeps OpenRouter session_id on values longer than OpenAI prompt cache keys", async () => {
|
||||
const longSessionId = "s".repeat(100);
|
||||
const captured = await captureOpenAIResponseHeaders({ sessionId: longSessionId }, openRouterResponsesModel);
|
||||
|
||||
expect(captured.body?.session_id).toBe(longSessionId);
|
||||
expect(captured.body?.prompt_cache_key).not.toBe(longSessionId);
|
||||
});
|
||||
|
||||
it("hashes OpenRouter session_id only past the 256 character limit", async () => {
|
||||
const tooLongSessionId = "s".repeat(300);
|
||||
const captured = await captureOpenAIResponseHeaders({ sessionId: tooLongSessionId }, openRouterResponsesModel);
|
||||
const sessionId = captured.body?.session_id;
|
||||
|
||||
expect(typeof sessionId).toBe("string");
|
||||
expect((sessionId as string).length).toBeLessThanOrEqual(256);
|
||||
expect(sessionId).not.toBe(tooLongSessionId);
|
||||
});
|
||||
|
||||
it("lets explicit extraBody override OpenRouter Responses session_id", async () => {
|
||||
const captured = await captureOpenAIResponseHeaders(
|
||||
{
|
||||
sessionId: "workflow-123",
|
||||
extraBody: { session_id: "body-wins" },
|
||||
},
|
||||
openRouterResponsesModel,
|
||||
);
|
||||
|
||||
expect(captured.body?.session_id).toBe("body-wins");
|
||||
});
|
||||
|
||||
it("merges adapter extra body fields into the Responses request payload", async () => {
|
||||
const captured = await captureOpenAIResponseHeaders({
|
||||
sessionId: "session-123",
|
||||
@@ -235,4 +433,14 @@ describe("openai-responses cache affinity", () => {
|
||||
expect(captured.clientRequestId).toBeNull();
|
||||
expect(captured.body?.prompt_cache_key).toBeUndefined();
|
||||
});
|
||||
|
||||
it("omits OpenRouter Responses session_id when cache retention is disabled", async () => {
|
||||
const captured = await captureOpenAIResponseHeaders(
|
||||
{ cacheRetention: "none", sessionId: "workflow-123" },
|
||||
openRouterResponsesModel,
|
||||
);
|
||||
|
||||
expect(captured.body?.session_id).toBeUndefined();
|
||||
expect(captured.body?.prompt_cache_key).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,368 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
|
||||
import { streamSimple } from "@oh-my-pi/pi-ai/stream";
|
||||
import type { Context, FetchImpl, Model, ModelSpec, OpenAICompat, StreamOptions } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
|
||||
const context: Context = {
|
||||
systemPrompt: ["Stay concise."],
|
||||
messages: [{ role: "user", content: "ping", timestamp: 0 }],
|
||||
};
|
||||
|
||||
function createSseResponse(): Response {
|
||||
return new Response(
|
||||
`data: ${JSON.stringify({
|
||||
type: "response.completed",
|
||||
response: {
|
||||
status: "completed",
|
||||
usage: { input_tokens: 1, output_tokens: 1, total_tokens: 2, input_tokens_details: { cached_tokens: 0 } },
|
||||
},
|
||||
})}\n\n`,
|
||||
{ status: 200, headers: { "content-type": "text/event-stream" } },
|
||||
);
|
||||
}
|
||||
function createChatDoneResponse(): Response {
|
||||
return new Response("data: [DONE]\n\n", { status: 200, headers: { "content-type": "text/event-stream" } });
|
||||
}
|
||||
|
||||
function buildOpenRouterModel(
|
||||
overrides: Partial<ModelSpec<"openrouter">> = {},
|
||||
compat?: OpenAICompat,
|
||||
): Model<"openrouter"> {
|
||||
return buildModel({
|
||||
id: "anthropic/claude-haiku-latest",
|
||||
name: "Claude Haiku via OpenRouter",
|
||||
api: "openrouter",
|
||||
provider: "openrouter",
|
||||
baseUrl: "https://openrouter.ai/api/v1",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 131_072,
|
||||
compat,
|
||||
...overrides,
|
||||
} as ModelSpec<"openrouter">);
|
||||
}
|
||||
|
||||
function buildOpenRouterResponsesModel(
|
||||
overrides: Partial<ModelSpec<"openai-responses">> = {},
|
||||
): Model<"openai-responses"> {
|
||||
return buildModel({
|
||||
id: "anthropic/claude-haiku-latest",
|
||||
name: "Claude Haiku via OpenRouter Responses",
|
||||
api: "openai-responses",
|
||||
provider: "openrouter",
|
||||
baseUrl: "https://openrouter.ai/api/v1",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 131_072,
|
||||
...overrides,
|
||||
} as ModelSpec<"openai-responses">);
|
||||
}
|
||||
async function capturePseudoChatRequest(
|
||||
model: Model<"openrouter">,
|
||||
options: Omit<StreamOptions, "apiKey"> & {
|
||||
reasoning?: Effort;
|
||||
disableReasoning?: boolean;
|
||||
openrouterVariant?: string;
|
||||
} = {},
|
||||
): Promise<Record<string, unknown>> {
|
||||
let body: Record<string, unknown> | undefined;
|
||||
const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => {
|
||||
body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : {};
|
||||
return createChatDoneResponse();
|
||||
});
|
||||
|
||||
const stream = streamOpenAICompletions(model as unknown as Model<"openai-completions">, context, {
|
||||
apiKey: "test-key",
|
||||
...options,
|
||||
fetch: fetchMock,
|
||||
});
|
||||
for await (const event of stream) {
|
||||
if (event.type === "done" || event.type === "error") break;
|
||||
}
|
||||
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
if (!body) throw new Error("Expected captured OpenRouter chat request");
|
||||
return body;
|
||||
}
|
||||
|
||||
async function capturePseudoResponsesRequest(
|
||||
model: Model<"openrouter">,
|
||||
options: Omit<StreamOptions, "apiKey"> & {
|
||||
reasoning?: Effort;
|
||||
disableReasoning?: boolean;
|
||||
openrouterVariant?: string;
|
||||
} = {},
|
||||
): Promise<Record<string, unknown>> {
|
||||
let body: Record<string, unknown> | undefined;
|
||||
const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => {
|
||||
body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : {};
|
||||
return createSseResponse();
|
||||
});
|
||||
|
||||
const stream = streamOpenAIResponses(model as unknown as Model<"openai-responses">, context, {
|
||||
apiKey: "test-key",
|
||||
...options,
|
||||
fetch: fetchMock,
|
||||
});
|
||||
for await (const event of stream) {
|
||||
if (event.type === "done" || event.type === "error") break;
|
||||
}
|
||||
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
if (!body) throw new Error("Expected captured OpenRouter Responses request");
|
||||
return body;
|
||||
}
|
||||
|
||||
async function captureRequest<TApi extends "openrouter" | "openai-responses">(
|
||||
model: Model<TApi>,
|
||||
options: Omit<StreamOptions, "apiKey"> & {
|
||||
reasoning?: Effort;
|
||||
disableReasoning?: boolean;
|
||||
openrouterVariant?: string;
|
||||
} = {},
|
||||
): Promise<{ body: Record<string, unknown>; headers: Headers }> {
|
||||
let body: Record<string, unknown> | undefined;
|
||||
let headers: Headers | undefined;
|
||||
const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => {
|
||||
body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : {};
|
||||
headers = new Headers(init?.headers);
|
||||
return createSseResponse();
|
||||
});
|
||||
|
||||
const stream = streamSimple(model, context, { apiKey: "test-key", ...options, fetch: fetchMock });
|
||||
for await (const event of stream) {
|
||||
if (event.type === "done" || event.type === "error") break;
|
||||
}
|
||||
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
if (!body || !headers) throw new Error("Expected captured OpenRouter Responses request");
|
||||
return { body, headers };
|
||||
}
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
describe("OpenRouter pseudo API dual-surface request parity", () => {
|
||||
it("builds equivalent Anthropic reasoning payloads for chat and Responses", async () => {
|
||||
const routing = { only: ["anthropic"], order: ["anthropic", "openai"] };
|
||||
const model = buildOpenRouterModel(
|
||||
{ id: "anthropic/claude-fable-5", name: "Claude Fable 5", reasoning: true },
|
||||
{ openRouterRouting: routing },
|
||||
);
|
||||
|
||||
const chatBody = await capturePseudoChatRequest(model, {
|
||||
openrouterVariant: "nitro",
|
||||
reasoning: Effort.High,
|
||||
sessionId: "workflow-123",
|
||||
});
|
||||
const responsesBody = await capturePseudoResponsesRequest(model, {
|
||||
openrouterVariant: "nitro",
|
||||
reasoning: Effort.High,
|
||||
sessionId: "workflow-123",
|
||||
});
|
||||
|
||||
expect(chatBody).toEqual({
|
||||
model: "anthropic/claude-fable-5:nitro",
|
||||
messages: [
|
||||
{ role: "system", content: "Stay concise." },
|
||||
{
|
||||
role: "user",
|
||||
content: [{ type: "text", text: "ping", cache_control: { type: "ephemeral" } }],
|
||||
},
|
||||
],
|
||||
stream: true,
|
||||
stream_options: { include_usage: true },
|
||||
store: false,
|
||||
reasoning: { effort: "xhigh" },
|
||||
provider: routing,
|
||||
});
|
||||
expect(responsesBody).toEqual({
|
||||
model: "anthropic/claude-fable-5:nitro",
|
||||
instructions: "Stay concise.",
|
||||
stream: true,
|
||||
input: [{ role: "user", content: [{ type: "input_text", text: "ping" }] }],
|
||||
store: false,
|
||||
reasoning: { effort: "xhigh", summary: "auto" },
|
||||
prompt_cache_key: "workflow-123",
|
||||
session_id: "workflow-123",
|
||||
provider: routing,
|
||||
include: ["reasoning.encrypted_content"],
|
||||
});
|
||||
expect(chatBody).not.toHaveProperty("max_tokens");
|
||||
expect(chatBody).not.toHaveProperty("max_completion_tokens");
|
||||
expect(responsesBody).not.toHaveProperty("max_output_tokens");
|
||||
});
|
||||
|
||||
it("builds equivalent non-Anthropic reasoning payloads for chat and Responses", async () => {
|
||||
const routing = { order: ["deepseek", "openai"] };
|
||||
const model = buildOpenRouterModel(
|
||||
{ id: "deepseek/deepseek-r1", name: "DeepSeek R1", reasoning: true },
|
||||
{ openRouterRouting: routing },
|
||||
);
|
||||
|
||||
const chatBody = await capturePseudoChatRequest(model, {
|
||||
openrouterVariant: "nitro",
|
||||
reasoning: Effort.High,
|
||||
sessionId: "workflow-456",
|
||||
});
|
||||
const responsesBody = await capturePseudoResponsesRequest(model, {
|
||||
openrouterVariant: "nitro",
|
||||
reasoning: Effort.High,
|
||||
sessionId: "workflow-456",
|
||||
});
|
||||
|
||||
expect(chatBody).toEqual({
|
||||
model: "deepseek/deepseek-r1:nitro",
|
||||
messages: [
|
||||
{ role: "system", content: "Stay concise." },
|
||||
{ role: "user", content: "ping" },
|
||||
],
|
||||
stream: true,
|
||||
stream_options: { include_usage: true },
|
||||
store: false,
|
||||
reasoning: { effort: "high" },
|
||||
provider: routing,
|
||||
});
|
||||
expect(responsesBody).toEqual({
|
||||
model: "deepseek/deepseek-r1:nitro",
|
||||
instructions: "Stay concise.",
|
||||
stream: true,
|
||||
input: [{ role: "user", content: [{ type: "input_text", text: "ping" }] }],
|
||||
store: false,
|
||||
reasoning: { effort: "high", summary: "auto" },
|
||||
prompt_cache_key: "workflow-456",
|
||||
session_id: "workflow-456",
|
||||
provider: routing,
|
||||
include: ["reasoning.encrypted_content"],
|
||||
});
|
||||
expect(chatBody).not.toHaveProperty("max_tokens");
|
||||
expect(chatBody).not.toHaveProperty("max_completion_tokens");
|
||||
expect(responsesBody).not.toHaveProperty("max_output_tokens");
|
||||
});
|
||||
|
||||
it("keeps stop and frequency penalty on Chat Completions but drops them for Responses", async () => {
|
||||
const model = buildOpenRouterModel();
|
||||
const options = {
|
||||
stopSequences: ["</stop>"],
|
||||
frequencyPenalty: 0.75,
|
||||
};
|
||||
|
||||
const chatBody = await capturePseudoChatRequest(model, options);
|
||||
const responsesBody = await capturePseudoResponsesRequest(model, options);
|
||||
|
||||
expect(chatBody.stop).toBe("</stop>");
|
||||
expect(chatBody.frequency_penalty).toBe(0.75);
|
||||
expect(responsesBody).not.toHaveProperty("stop");
|
||||
expect(responsesBody).not.toHaveProperty("frequency_penalty");
|
||||
});
|
||||
});
|
||||
|
||||
describe("OpenRouter Responses request shape", () => {
|
||||
it("appends openrouterVariant only when the resolved model id has no variant after the final slash", async () => {
|
||||
const suffixed = await captureRequest(buildOpenRouterResponsesModel(), { openrouterVariant: "nitro" });
|
||||
expect(suffixed.body.model).toBe("anthropic/claude-haiku-latest:nitro");
|
||||
|
||||
const explicit = await captureRequest(
|
||||
buildOpenRouterResponsesModel({ id: "anthropic/claude-haiku-latest:online" }),
|
||||
{ openrouterVariant: "nitro" },
|
||||
);
|
||||
expect(explicit.body.model).toBe("anthropic/claude-haiku-latest:online");
|
||||
});
|
||||
|
||||
it("resolves pseudo OpenRouter API compat and preserves provider routing in the Responses body", async () => {
|
||||
const routing = { only: ["anthropic"], order: ["anthropic", "openai"] };
|
||||
const pseudoModel = buildOpenRouterModel({}, { openRouterRouting: routing });
|
||||
expect(pseudoModel.compat.openRouterRouting).toEqual(routing);
|
||||
|
||||
const { body } = await captureRequest(buildOpenRouterResponsesModel({ compat: { openRouterRouting: routing } }));
|
||||
expect(body.provider).toEqual(routing);
|
||||
});
|
||||
|
||||
it("explicitly disables OpenRouter Responses reasoning with enabled=false", async () => {
|
||||
const { body } = await captureRequest(
|
||||
buildOpenRouterResponsesModel({ id: "deepseek/deepseek-r1", name: "DeepSeek R1", reasoning: true }),
|
||||
{ disableReasoning: true },
|
||||
);
|
||||
expect(body.reasoning).toEqual({ enabled: false });
|
||||
});
|
||||
|
||||
it("omits default max_output_tokens for OpenRouter but sends explicit caller caps", async () => {
|
||||
const defaultRequest = await captureRequest(buildOpenRouterResponsesModel());
|
||||
expect(defaultRequest.body).not.toHaveProperty("max_output_tokens");
|
||||
|
||||
const explicitRequest = await captureRequest(buildOpenRouterResponsesModel(), { maxTokens: 2048 });
|
||||
expect(explicitRequest.body.max_output_tokens).toBe(2048);
|
||||
});
|
||||
|
||||
it("keeps default max_output_tokens for OpenRouter models that require it", async () => {
|
||||
const request = await captureRequest(
|
||||
buildOpenRouterResponsesModel({ id: "moonshotai/kimi-k2.6", name: "Kimi K2.6" }),
|
||||
);
|
||||
expect(request.body.max_output_tokens).toBe(64_000);
|
||||
});
|
||||
|
||||
it("lets caller headers override OpenRouter attribution and cache defaults", async () => {
|
||||
const { headers } = await captureRequest(buildOpenRouterResponsesModel(), {
|
||||
headers: {
|
||||
"HTTP-Referer": "https://caller.example/",
|
||||
"X-OpenRouter-Title": "Caller App",
|
||||
"X-OpenRouter-Cache": "false",
|
||||
"X-OpenRouter-Cache-TTL": "7",
|
||||
},
|
||||
});
|
||||
|
||||
expect(headers.get("HTTP-Referer")).toBe("https://caller.example/");
|
||||
expect(headers.get("X-OpenRouter-Title")).toBe("Caller App");
|
||||
expect(headers.get("X-OpenRouter-Cache")).toBe("false");
|
||||
expect(headers.get("X-OpenRouter-Cache-TTL")).toBe("7");
|
||||
});
|
||||
});
|
||||
|
||||
async function captureDirectResponsesRequest(
|
||||
model: Model<"openai-responses">,
|
||||
options: Omit<StreamOptions, "apiKey"> & { reasoning?: Effort } = {},
|
||||
): Promise<Record<string, unknown>> {
|
||||
let body: Record<string, unknown> | undefined;
|
||||
const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => {
|
||||
body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : {};
|
||||
return createSseResponse();
|
||||
});
|
||||
|
||||
const stream = streamOpenAIResponses(model, context, { apiKey: "test-key", ...options, fetch: fetchMock });
|
||||
for await (const event of stream) {
|
||||
if (event.type === "done" || event.type === "error") break;
|
||||
}
|
||||
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
if (!body) throw new Error("Expected captured direct Responses request");
|
||||
return body;
|
||||
}
|
||||
|
||||
describe("Responses direct-provider max-token defaults", () => {
|
||||
// `streamOpenAIResponses` is public; callers may bypass `streamSimple` (which
|
||||
// pre-fills `options.maxTokens` from the model cap). After centralizing the
|
||||
// output-token policy in `resolveOpenAIOutputTokenParam`, the provider itself
|
||||
// supplies the `alwaysSendMaxTokens` default — so a direct call now emits
|
||||
// `max_output_tokens` for Kimi-style models even with no caller cap. This is
|
||||
// an intentional consistency change vs. the prior inline logic, which only
|
||||
// emitted the field because `streamSimple` had injected it.
|
||||
it("emits max_output_tokens for an alwaysSendMaxTokens model on a direct call with no caller cap", async () => {
|
||||
const body = await captureDirectResponsesRequest(
|
||||
buildOpenRouterResponsesModel({ id: "moonshotai/kimi-k2.6", name: "Kimi K2.6" }),
|
||||
);
|
||||
expect(body.max_output_tokens).toBe(64_000);
|
||||
});
|
||||
|
||||
it("still omits max_output_tokens for a routing-only OpenRouter model on a direct call", async () => {
|
||||
const body = await captureDirectResponsesRequest(buildOpenRouterResponsesModel());
|
||||
expect(body).not.toHaveProperty("max_output_tokens");
|
||||
});
|
||||
});
|
||||
@@ -1,9 +1,9 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import type { ResponseInput } from "@oh-my-pi/pi-ai/providers/openai-responses-wire";
|
||||
import {
|
||||
repairOrphanResponsesToolCalls,
|
||||
repairOrphanResponsesToolOutputs,
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
|
||||
import type { ResponseInput } from "@oh-my-pi/pi-ai/providers/openai-responses-wire";
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
|
||||
describe("repairOrphanResponsesToolCalls", () => {
|
||||
it("appends a synthetic function_call_output after a call with no result", () => {
|
||||
|
||||
@@ -13,8 +13,8 @@
|
||||
// `output_item.done` event must be routed by `output_index`/`item_id`, not by
|
||||
// arrival order.
|
||||
import { describe, expect, test } from "bun:test";
|
||||
import { processResponsesStream } from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
|
||||
import type { ResponseStreamEvent } from "@oh-my-pi/pi-ai/providers/openai-responses-wire";
|
||||
import { processResponsesStream } from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
import type { AssistantMessage, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
|
||||
|
||||
@@ -8,8 +8,8 @@
|
||||
// input on the stored content block and drop the transient `partialJson`
|
||||
// accumulation buffer, mirroring the function_call branch.
|
||||
import { describe, expect, test } from "bun:test";
|
||||
import { processResponsesStream } from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
|
||||
import type { ResponseStreamEvent } from "@oh-my-pi/pi-ai/providers/openai-responses-wire";
|
||||
import { processResponsesStream } from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
import type { AssistantMessage, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { applyAnthropicUsageExtras } from "@oh-my-pi/pi-ai/providers/anthropic";
|
||||
import { parseChunkUsage } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import { calculateOpenAIUsageAccounting } from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
import type { Model, Usage } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
|
||||
@@ -203,6 +204,59 @@ describe("openai-completions parseChunkUsage", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("shared OpenAI usage accounting", () => {
|
||||
it("uses provider cache-write details ahead of native DeepSeek passthrough fields", () => {
|
||||
const usage = calculateOpenAIUsageAccounting({
|
||||
promptTokens: 6_000,
|
||||
outputTokens: 250,
|
||||
cachedTokens: 200,
|
||||
reasoningTokens: 0,
|
||||
cacheWriteOpenRouter: 5_000,
|
||||
cacheWriteDeepSeek: 50,
|
||||
hasDeepSeekCacheHitAndMiss: true,
|
||||
});
|
||||
|
||||
expect(usage.input).toBe(800);
|
||||
expect(usage.cacheRead).toBe(200);
|
||||
expect(usage.cacheWrite).toBe(5_000);
|
||||
expect(usage.totalTokens).toBe(6_250);
|
||||
});
|
||||
|
||||
it("does not emit DeepSeek cache misses as cache writes", () => {
|
||||
const usage = calculateOpenAIUsageAccounting({
|
||||
promptTokens: 150,
|
||||
outputTokens: 200,
|
||||
cachedTokens: 100,
|
||||
reasoningTokens: 0,
|
||||
cacheWriteOpenRouter: undefined,
|
||||
cacheWriteDeepSeek: 50,
|
||||
hasDeepSeekCacheHitAndMiss: true,
|
||||
});
|
||||
|
||||
expect(usage.input).toBe(50);
|
||||
expect(usage.cacheRead).toBe(100);
|
||||
expect(usage.cacheWrite).toBe(0);
|
||||
expect(usage.totalTokens).toBe(350);
|
||||
});
|
||||
|
||||
it("treats zero provider cache-write as present when native fields pass through", () => {
|
||||
const usage = calculateOpenAIUsageAccounting({
|
||||
promptTokens: 150,
|
||||
outputTokens: 25,
|
||||
cachedTokens: 100,
|
||||
reasoningTokens: 0,
|
||||
cacheWriteOpenRouter: 0,
|
||||
cacheWriteDeepSeek: 50,
|
||||
hasDeepSeekCacheHitAndMiss: true,
|
||||
});
|
||||
|
||||
expect(usage.input).toBe(50);
|
||||
expect(usage.cacheRead).toBe(100);
|
||||
expect(usage.cacheWrite).toBe(0);
|
||||
expect(usage.totalTokens).toBe(175);
|
||||
});
|
||||
});
|
||||
|
||||
describe("anthropic applyAnthropicUsageExtras", () => {
|
||||
it("captures cache TTL breakdown when both buckets are non-zero", () => {
|
||||
const usage = blankUsage();
|
||||
|
||||
@@ -10,8 +10,8 @@ import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
// returns undefined for them instead of tripping `requireSupportedEffort`
|
||||
// (the old user-visible "Compaction failed: Thinking effort high is not
|
||||
// supported by xai-oauth/grok-build. Supported efforts:" with an empty list),
|
||||
// and the wire-side `omitReasoningEffort` gate (providers/xai-responses.ts)
|
||||
// remains the single source of truth for the actual strip.
|
||||
// and the wire-side `omitReasoningEffort` gate (stream.ts) remains the single
|
||||
// source of truth for the actual strip.
|
||||
describe("effort-dial-less reasoner encoding (regression)", () => {
|
||||
test("xai-oauth/grok-build reasons but carries no thinking config", () => {
|
||||
const grokBuild = getBundledModel("xai-oauth", "grok-build");
|
||||
|
||||
@@ -1,6 +1,15 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Added
|
||||
|
||||
- Added a dedicated `openrouter` API type and `ResolvedOpenRouterCompat` configuration to support unified chat-completions and Responses-API compatibility for OpenRouter models
|
||||
|
||||
### Changed
|
||||
|
||||
- Migrated bundled OpenRouter models in the catalog from `openai-completions` to the new `openrouter` API type
|
||||
- Consolidated the resolved OpenAI compat shape: extracted a shared `ResolvedOpenAISharedCompat` core that both `ResolvedOpenAICompat` and `ResolvedOpenAIResponsesCompat` extend (each builder still computes its own per-surface value, preserving chat↔Responses divergence), added internal resolved wire-quirk fields (`wireModelIdMode`, `stripDeepseekSpecialTokens`, `reasoningDeltasMayBeCumulative`, `emptyLengthFinishIsContextError`, `usesOpenAIToolCallIdLimit`, `dropThinkingWhenReasoningEffort`, `supportsObfuscationOptOut`), and replaced `buildOpenRouterCompat`'s cast-and-copy with an exhaustive `pickResponsesOnly` composition that fails to compile if a new Responses-only field is added without handling. The public `OpenAICompat` config vocabulary is unchanged.
|
||||
- Expanded `OpenAICompat`/`ResolvedOpenAISharedCompat` with shared reasoning/history/stream/request flags (`reasoningDisableMode`, `omitReasoningEffort`, `includeEncryptedReasoning`, `filterReasoningHistory`, `requiresReasoningContentForAllAssistantTurns`, `streamMarkupHealingPattern`, `promptCacheSessionHeader`, etc.) so model/provider/gateway constraints are declared once in catalog compat and then consumed uniformly by Chat Completions and Responses endpoints.
|
||||
|
||||
## [16.0.5] - 2026-06-17
|
||||
|
||||
@@ -19,6 +28,7 @@
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed OpenRouter pseudo-API model construction so bundled OpenRouter models resolve shared OpenAI compatibility metadata instead of an undefined compat record.
|
||||
- Fixed `off` effort routing for `claude-opus-4-5` and `claude-opus-4-6` to use their base model IDs when thinking is disabled
|
||||
- Fixed `gemini-2.5-flash` effort routing so all non-off effort levels resolve to `gemini-2.5-flash-thinking`
|
||||
- Fixed shared variant alias provider resolution so `resolveBareVariantAlias` reports all matching providers when model aliases are present in both CCA collapse tables
|
||||
@@ -257,4 +267,4 @@
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads.
|
||||
- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads.
|
||||
@@ -35,6 +35,7 @@
|
||||
"dependencies": {
|
||||
"@bufbuild/protobuf": "catalog:",
|
||||
"@oh-my-pi/pi-utils": "catalog:",
|
||||
"arktype": "catalog:",
|
||||
"zod": "catalog:"
|
||||
},
|
||||
"devDependencies": {
|
||||
|
||||
@@ -10,7 +10,7 @@
|
||||
* compat per request.
|
||||
*/
|
||||
import { buildAnthropicCompat } from "./compat/anthropic";
|
||||
import { buildOpenAICompat, buildOpenAIResponsesCompat } from "./compat/openai";
|
||||
import { buildOpenAICompat, buildOpenAIResponsesCompat, buildOpenRouterCompat } from "./compat/openai";
|
||||
import { resolveModelThinking } from "./model-thinking";
|
||||
import type { Api, CompatOf, Model, ModelSpec } from "./types";
|
||||
import { cleanModelName } from "./utils";
|
||||
@@ -28,6 +28,8 @@ export function buildModel<TApi extends Api>(spec: ModelSpec<TApi>): Model<TApi>
|
||||
|
||||
export function buildCompat(spec: ModelSpec<Api>): CompatOf<Api> {
|
||||
switch (spec.api) {
|
||||
case "openrouter":
|
||||
return buildOpenRouterCompat(spec as ModelSpec<"openrouter">);
|
||||
case "openai-completions":
|
||||
return buildOpenAICompat(spec as ModelSpec<"openai-completions">);
|
||||
case "openai-responses":
|
||||
|
||||
@@ -19,7 +19,15 @@ import {
|
||||
isQwenModelId,
|
||||
modelFamilyToken,
|
||||
} from "../identity/family";
|
||||
import type { ModelSpec, OpenAICompat, ResolvedOpenAICompat, ResolvedOpenAIResponsesCompat } from "../types";
|
||||
import type {
|
||||
ModelSpec,
|
||||
OpenAICompat,
|
||||
OpenAIStreamMarkupHealingPattern,
|
||||
ResolvedOpenAICompat,
|
||||
ResolvedOpenAIResponsesCompat,
|
||||
ResolvedOpenAISharedCompat,
|
||||
ResolvedOpenRouterCompat,
|
||||
} from "../types";
|
||||
import { applyCompatOverrides } from "./apply";
|
||||
|
||||
/** GLM coding-plan SKUs idle for minutes mid-reasoning; see `streamIdleTimeoutMs`. */
|
||||
@@ -29,6 +37,50 @@ const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000;
|
||||
const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
|
||||
/** Kimi K2.6 can spend several minutes reasoning before the first visible token. */
|
||||
const KIMI_K26_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
|
||||
const MINIMAX_PROVIDER_OR_ID_PATTERN = /minimax/i;
|
||||
const DSML_HEALING_PROVIDERS = new Set([
|
||||
"ollama",
|
||||
"ollama-cloud",
|
||||
"nvidia",
|
||||
"deepseek",
|
||||
"fireworks",
|
||||
"nanogpt",
|
||||
"opencode-go",
|
||||
"openrouter",
|
||||
]);
|
||||
|
||||
function resolveReasoningDisableMode(
|
||||
thinkingFormat: ResolvedOpenAISharedCompat["thinkingFormat"],
|
||||
): ResolvedOpenAISharedCompat["reasoningDisableMode"] {
|
||||
switch (thinkingFormat) {
|
||||
case "openrouter":
|
||||
return "openrouter-enabled-false";
|
||||
case "zai":
|
||||
return "zai-thinking-disabled";
|
||||
case "qwen":
|
||||
return "qwen-enable-thinking-false";
|
||||
case "qwen-chat-template":
|
||||
return "qwen-template-false";
|
||||
default:
|
||||
return "lowest-effort";
|
||||
}
|
||||
}
|
||||
|
||||
function detectStreamMarkupHealingPattern(
|
||||
provider: string,
|
||||
modelId: string,
|
||||
): OpenAIStreamMarkupHealingPattern | undefined {
|
||||
if (MINIMAX_PROVIDER_OR_ID_PATTERN.test(provider) || MINIMAX_PROVIDER_OR_ID_PATTERN.test(modelId)) {
|
||||
return "thinking";
|
||||
}
|
||||
if (provider === "kimi-code" || provider === "moonshot" || /kimi[-/_.]?k2/i.test(modelId)) {
|
||||
return "kimi";
|
||||
}
|
||||
if (isDeepseekModelIdOrName(modelId) && DSML_HEALING_PROVIDERS.has(provider)) {
|
||||
return "dsml";
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* OpenCode's gateways (https://opencode.ai/zen|go) gate `reasoning_content`
|
||||
@@ -197,6 +249,26 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
? DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS
|
||||
: undefined;
|
||||
|
||||
const wireModelIdMode: ResolvedOpenAISharedCompat["wireModelIdMode"] =
|
||||
provider === "firepass"
|
||||
? "firepass"
|
||||
: provider === "fireworks"
|
||||
? "fireworks"
|
||||
: isOpenRouter
|
||||
? "openrouter"
|
||||
: "raw";
|
||||
const thinkingFormat: ResolvedOpenAISharedCompat["thinkingFormat"] =
|
||||
isZai || isZhipu || isMoonshotKimi || isXiaomiMimo
|
||||
? "zai"
|
||||
: isOpenRouter
|
||||
? "openrouter"
|
||||
: isQwen && isNvidiaNim
|
||||
? "qwen-chat-template"
|
||||
: isAlibaba || isQwen
|
||||
? "qwen"
|
||||
: "openai";
|
||||
|
||||
|
||||
const compat: ResolvedOpenAICompat = {
|
||||
supportsStore: !isNonStandard,
|
||||
// `developer` is an OpenAI-Responses-era extension to the chat-completions schema. Almost
|
||||
@@ -241,16 +313,11 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
// (`chat_template_kwargs.enable_thinking`); top-level `enable_thinking`
|
||||
// is rejected by NIM's `additionalProperties: false` request schema
|
||||
// (issue #2299).
|
||||
thinkingFormat:
|
||||
isZai || isZhipu || isMoonshotKimi || isXiaomiMimo
|
||||
? "zai"
|
||||
: isOpenRouter
|
||||
? "openrouter"
|
||||
: isQwen && isNvidiaNim
|
||||
? "qwen-chat-template"
|
||||
: isAlibaba || isQwen
|
||||
? "qwen"
|
||||
: "openai",
|
||||
thinkingFormat,
|
||||
reasoningDisableMode: resolveReasoningDisableMode(thinkingFormat),
|
||||
omitReasoningEffort: false,
|
||||
includeEncryptedReasoning: true,
|
||||
filterReasoningHistory: false,
|
||||
thinkingKeep: usesMoonshotKimiPreservedThinking ? "all" : undefined,
|
||||
reasoningContentField: "reasoning_content",
|
||||
// Backends that 400 follow-up requests when prior assistant tool-call turns lack `reasoning_content`:
|
||||
@@ -271,6 +338,8 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
(isDeepseekFamily && Boolean(spec.reasoning)) ||
|
||||
isXiaomiMimo ||
|
||||
(isOpenRouter && Boolean(spec.reasoning)),
|
||||
requiresReasoningContentForAllAssistantTurns:
|
||||
((isDeepseekFamily && Boolean(spec.reasoning)) || isXiaomiMimo) && !isOpenRouter,
|
||||
// DeepSeek V4 and Xiaomi MiMo reject synthetic reasoning_content placeholders (".") on tool-call turns.
|
||||
// Kimi and OpenRouter accept them when actual reasoning is unavailable.
|
||||
allowsSyntheticReasoningContentForToolCalls: (!isDeepseekFamily || !spec.reasoning) && !isXiaomiMimo,
|
||||
@@ -279,20 +348,43 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
openRouterRouting: undefined,
|
||||
vercelGatewayRouting: undefined,
|
||||
isOpenRouterHost: isOpenRouter,
|
||||
wireModelIdMode,
|
||||
isVercelGatewayHost: isVercelGateway,
|
||||
supportsStrictMode: detectStrictModeSupport(provider, baseUrl),
|
||||
extraBody: isDirectDeepseekReasoning ? { thinking: { type: "enabled" } } : undefined,
|
||||
toolStrictMode: isCerebras ? "all_strict" : "mixed",
|
||||
streamIdleTimeoutMs,
|
||||
stripDeepseekSpecialTokens:
|
||||
isDeepseekModelIdOrName(spec.id) && (provider === "nvidia" || provider === "deepseek"),
|
||||
streamMarkupHealingPattern: detectStreamMarkupHealingPattern(provider, spec.id),
|
||||
reasoningDeltasMayBeCumulative:
|
||||
MINIMAX_PROVIDER_OR_ID_PATTERN.test(provider) || MINIMAX_PROVIDER_OR_ID_PATTERN.test(spec.id),
|
||||
emptyLengthFinishIsContextError: provider === "ollama",
|
||||
usesOpenAIToolCallIdLimit: provider === "openai",
|
||||
promptCacheSessionHeader: undefined,
|
||||
dropThinkingWhenReasoningEffort: provider === "fireworks",
|
||||
};
|
||||
|
||||
applyCompatOverrides(compat, spec.compat);
|
||||
if (spec.compat?.reasoningDisableMode === undefined) {
|
||||
compat.reasoningDisableMode = resolveReasoningDisableMode(compat.thinkingFormat);
|
||||
}
|
||||
if (spec.compat?.omitReasoningEffort === undefined && !compat.supportsReasoningEffort) {
|
||||
compat.omitReasoningEffort = true;
|
||||
}
|
||||
|
||||
|
||||
const whenThinkingPolicy =
|
||||
spec.compat?.whenThinking ?? (isOpenCodeProvider && spec.reasoning ? OPENCODE_WHEN_THINKING : undefined);
|
||||
if (whenThinkingPolicy) {
|
||||
const variant: ResolvedOpenAICompat = { ...compat };
|
||||
applyCompatOverrides(variant, whenThinkingPolicy);
|
||||
if (whenThinkingPolicy.reasoningDisableMode === undefined) {
|
||||
variant.reasoningDisableMode = resolveReasoningDisableMode(variant.thinkingFormat);
|
||||
}
|
||||
if (whenThinkingPolicy.omitReasoningEffort === undefined && !variant.supportsReasoningEffort) {
|
||||
variant.omitReasoningEffort = true;
|
||||
}
|
||||
compat.whenThinking = variant;
|
||||
}
|
||||
|
||||
@@ -304,6 +396,7 @@ interface OpenAIResponsesSpecLike {
|
||||
provider: string;
|
||||
name: string;
|
||||
baseUrl: string;
|
||||
reasoning?: boolean;
|
||||
compat?: OpenAICompat;
|
||||
}
|
||||
|
||||
@@ -321,22 +414,87 @@ interface OpenAIResponsesSpecLike {
|
||||
export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): ResolvedOpenAIResponsesCompat {
|
||||
const baseUrl = spec.baseUrl ?? "";
|
||||
const isAzure = modelMatchesHost({ provider: spec.provider, baseUrl }, "azureOpenAI");
|
||||
const isOpenRouter = modelMatchesHost({ provider: spec.provider, baseUrl }, "openrouter");
|
||||
const isOpenAIUrl = hostMatchesUrl(baseUrl, "openai");
|
||||
const id = spec.id ?? "";
|
||||
const thinkingFormat: ResolvedOpenAISharedCompat["thinkingFormat"] = isOpenRouter ? "openrouter" : "openai";
|
||||
const isKimiModel = id ? isKimiModelId(id) : false;
|
||||
const isDeepseekFamily = id ? isDeepseekModelIdOrName(id) || isDeepseekModelIdOrName(spec.name) : false;
|
||||
const reasoningCapable = Boolean(spec.reasoning);
|
||||
|
||||
const compat: ResolvedOpenAIResponsesCompat = {
|
||||
supportsDeveloperRole: isAzure || hostMatchesUrl(baseUrl, "openai") || hostMatchesUrl(baseUrl, "githubCopilot"),
|
||||
supportsDeveloperRole: isAzure || isOpenAIUrl || hostMatchesUrl(baseUrl, "githubCopilot"),
|
||||
supportsStrictMode:
|
||||
spec.provider === "openai" ||
|
||||
isAzure ||
|
||||
spec.provider === "github-copilot" ||
|
||||
hostMatchesUrl(baseUrl, "openai"),
|
||||
spec.provider === "openai" || isAzure || spec.provider === "github-copilot" || isOpenRouter || isOpenAIUrl,
|
||||
supportsReasoningEffort: true,
|
||||
supportsLongPromptCacheRetention: hostMatchesUrl(baseUrl, "openai"),
|
||||
supportsLongPromptCacheRetention: isOpenAIUrl,
|
||||
// Azure OpenAI and GitHub Copilot Responses paths require tool results
|
||||
// to strictly match prior tool calls when building Responses inputs.
|
||||
strictResponsesPairing: isAzure || spec.provider === "github-copilot",
|
||||
requiresJuiceZeroHack: spec.name.toLowerCase().startsWith("gpt-5"),
|
||||
reasoningEffortMap: {},
|
||||
supportsReasoningParams: true,
|
||||
thinkingFormat,
|
||||
reasoningDisableMode: resolveReasoningDisableMode(thinkingFormat),
|
||||
omitReasoningEffort: false,
|
||||
includeEncryptedReasoning: spec.provider !== "xai-oauth",
|
||||
filterReasoningHistory: spec.provider === "xai-oauth",
|
||||
disableReasoningOnForcedToolChoice: isKimiModel,
|
||||
disableReasoningOnToolChoice: isDeepseekFamily && reasoningCapable && !isOpenRouter,
|
||||
supportsToolChoice: true,
|
||||
supportsForcedToolChoice: true,
|
||||
reasoningContentField: "reasoning_content",
|
||||
requiresReasoningContentForToolCalls:
|
||||
(isKimiModel || (isDeepseekFamily && reasoningCapable) || (isOpenRouter && reasoningCapable)) &&
|
||||
reasoningCapable,
|
||||
requiresReasoningContentForAllAssistantTurns: isDeepseekFamily && reasoningCapable && !isOpenRouter,
|
||||
allowsSyntheticReasoningContentForToolCalls: !isDeepseekFamily || !reasoningCapable,
|
||||
requiresThinkingAsText: false,
|
||||
requiresMistralToolIds: false,
|
||||
requiresToolResultName: false,
|
||||
requiresAssistantAfterToolResult: false,
|
||||
requiresAssistantContentForToolCalls: isKimiModel,
|
||||
openRouterRouting: undefined,
|
||||
isOpenRouterHost: isOpenRouter,
|
||||
wireModelIdMode: isOpenRouter ? "openrouter" : "raw",
|
||||
alwaysSendMaxTokens: spec.id ? isKimiModelId(spec.id) : false,
|
||||
enableGeminiThinkingLoopGuard: modelFamilyToken(spec.id ?? "") === "gemini",
|
||||
supportsObfuscationOptOut: isOpenAIUrl || spec.provider === "openai",
|
||||
stripDeepseekSpecialTokens:
|
||||
Boolean(id) && isDeepseekModelIdOrName(id) && (spec.provider === "nvidia" || spec.provider === "deepseek"),
|
||||
streamMarkupHealingPattern: id ? detectStreamMarkupHealingPattern(spec.provider, id) : undefined,
|
||||
reasoningDeltasMayBeCumulative:
|
||||
MINIMAX_PROVIDER_OR_ID_PATTERN.test(spec.provider) || (id ? MINIMAX_PROVIDER_OR_ID_PATTERN.test(id) : false),
|
||||
emptyLengthFinishIsContextError: spec.provider === "ollama",
|
||||
usesOpenAIToolCallIdLimit: spec.provider === "openai",
|
||||
promptCacheSessionHeader: spec.provider === "xai-oauth" ? "x-grok-conv-id" : undefined,
|
||||
};
|
||||
applyCompatOverrides(compat, spec.compat);
|
||||
if (spec.compat?.reasoningDisableMode === undefined) {
|
||||
compat.reasoningDisableMode = resolveReasoningDisableMode(compat.thinkingFormat);
|
||||
}
|
||||
if (spec.compat?.omitReasoningEffort === undefined && !compat.supportsReasoningEffort) {
|
||||
compat.omitReasoningEffort = true;
|
||||
}
|
||||
return compat;
|
||||
}
|
||||
|
||||
type ResponsesOnlyCompat = Omit<ResolvedOpenAIResponsesCompat, keyof ResolvedOpenAISharedCompat>;
|
||||
|
||||
function pickResponsesOnly(compat: ResolvedOpenAIResponsesCompat): ResponsesOnlyCompat {
|
||||
return {
|
||||
supportsLongPromptCacheRetention: compat.supportsLongPromptCacheRetention,
|
||||
strictResponsesPairing: compat.strictResponsesPairing,
|
||||
requiresJuiceZeroHack: compat.requiresJuiceZeroHack,
|
||||
supportsObfuscationOptOut: compat.supportsObfuscationOptOut,
|
||||
} satisfies ResponsesOnlyCompat;
|
||||
}
|
||||
|
||||
export function buildOpenRouterCompat(spec: ModelSpec<"openrouter">): ResolvedOpenRouterCompat {
|
||||
const chat = buildOpenAICompat({
|
||||
...spec,
|
||||
api: "openai-completions",
|
||||
} as ModelSpec<"openai-completions">);
|
||||
const responses = buildOpenAIResponsesCompat(spec);
|
||||
return { ...chat, ...pickResponsesOnly(responses) } as ResolvedOpenRouterCompat;
|
||||
}
|
||||
|
||||
@@ -266,11 +266,15 @@ function sameEffortList(left: readonly Effort[], right: readonly Effort[]): bool
|
||||
return true;
|
||||
}
|
||||
|
||||
function isOpenAICompatReasoningApi(api: Api): boolean {
|
||||
return api === "openai-completions" || api === "openrouter";
|
||||
}
|
||||
|
||||
function getModelDefinedEfforts<TApi extends Api>(spec: ModelSpec<TApi>): readonly Effort[] | undefined {
|
||||
if (spec.api === "openai-completions" && isZaiGlm52ReasoningEffortModel(spec)) {
|
||||
if (isOpenAICompatReasoningApi(spec.api) && isZaiGlm52ReasoningEffortModel(spec)) {
|
||||
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
||||
}
|
||||
return spec.api === "openai-completions" && (isMinimaxM2FamilyModelId(spec.id) || isOpenAIGptOssModelId(spec.id))
|
||||
return isOpenAICompatReasoningApi(spec.api) && (isMinimaxM2FamilyModelId(spec.id) || isOpenAIGptOssModelId(spec.id))
|
||||
? LOW_MEDIUM_HIGH_REASONING_EFFORTS
|
||||
: undefined;
|
||||
}
|
||||
@@ -298,7 +302,7 @@ function inferDetectedEffortMap<TApi extends Api>(
|
||||
? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER
|
||||
: ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER;
|
||||
}
|
||||
if (spec.api !== "openai-completions") {
|
||||
if (!isOpenAICompatReasoningApi(spec.api)) {
|
||||
return undefined;
|
||||
}
|
||||
if (spec.provider === "groq" && spec.id === "qwen/qwen3-32b") {
|
||||
@@ -437,7 +441,7 @@ function inferFallbackEfforts<TApi extends Api>(spec: ModelSpec<TApi>, compat: C
|
||||
if (spec.api === "bedrock-converse-stream") {
|
||||
return DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
if (spec.api === "openai-completions") {
|
||||
if (isOpenAICompatReasoningApi(spec.api)) {
|
||||
const resolved = compat as ResolvedOpenAICompat;
|
||||
if (resolved.thinkingFormat === "openai" && resolved.supportsReasoningEffort) {
|
||||
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
||||
@@ -503,7 +507,7 @@ function isOpenRouterAnthropicAdaptiveReasoningModel<TApi extends Api>(
|
||||
parsedModel: AnthropicModel,
|
||||
spec: ModelSpec<TApi>,
|
||||
): boolean {
|
||||
if (spec.api !== "openai-completions") return false;
|
||||
if (!isOpenAICompatReasoningApi(spec.api)) return false;
|
||||
if (!modelMatchesHost(spec, "openrouter")) return false;
|
||||
return isFableOrMythos(parsedModel.kind) || (parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.6"));
|
||||
}
|
||||
|
||||
+343
-343
File diff suppressed because it is too large
Load Diff
@@ -847,7 +847,7 @@ interface XAICuratedModel {
|
||||
* Whether xAI accepts the `reasoning.effort` wire param for this model.
|
||||
* Default true. When false: picker hides the effort dial (via
|
||||
* getSupportedEfforts in model-thinking.ts) AND wire-side already omits
|
||||
* the param via GROK_EFFORT_CAPABLE_PREFIXES in providers/xai-responses.ts.
|
||||
* the param via GROK_EFFORT_CAPABLE_PREFIXES in pi-ai's stream.ts.
|
||||
* Must agree with that allowlist; two truths kept in sync by curated-catalog
|
||||
* author convention until a follow-up Op: compress unifies them.
|
||||
*/
|
||||
@@ -867,11 +867,9 @@ interface XAICuratedModel {
|
||||
// coding-fine-tuned chat model; 512K context per user spec (2026-05-17).
|
||||
//
|
||||
// supportsReasoningEffort=false entries reason natively but reject the wire
|
||||
// `reasoning.effort` param (api.x.ai returns HTTP 400). Mirrors the HTTP-side
|
||||
// GROK_EFFORT_CAPABLE_PREFIXES allowlist in providers/xai-responses.ts. The
|
||||
// curated flag is the picker-visible truth; the HTTP allowlist is the wire
|
||||
// truth. omitReasoningEffort in xai-responses.ts already prevents 400s; this
|
||||
// fixes the picker UX wart of advertising an inert dial.
|
||||
// `reasoning.effort` param (api.x.ai returns HTTP 400). The corresponding
|
||||
// omit/include/history replay defaults live in catalog compat so every
|
||||
// OpenAI-family endpoint consumes the same constraint.
|
||||
export const XAI_OAUTH_CURATED_MODELS: readonly XAICuratedModel[] = [
|
||||
// grok-build is text-only per the bundled catalog; omit `input` for the default.
|
||||
{ id: "grok-build", contextWindow: 512_000, name: "Grok Build", supportsReasoningEffort: false },
|
||||
@@ -894,8 +892,7 @@ export const XAI_OAUTH_CURATED_MODELS: readonly XAICuratedModel[] = [
|
||||
},
|
||||
// Cursor's "Composer 2.5 Fast" exposed via SuperGrok: non-reasoning,
|
||||
// text-only, 200K context (mirrors Cursor's composer-* catalog entries).
|
||||
// Off the GROK_EFFORT_CAPABLE_PREFIXES allowlist, so the wire side already
|
||||
// sets omitReasoningEffort=true; reasoning:false also hides the effort dial.
|
||||
// Off the Grok effort-capable allowlist; reasoning:false also hides the effort dial.
|
||||
{
|
||||
id: "grok-composer-2.5-fast",
|
||||
contextWindow: 200_000,
|
||||
@@ -910,12 +907,31 @@ export const XAI_OAUTH_CURATED_MODELS: readonly XAICuratedModel[] = [
|
||||
// strings; the chat picker MUST exclude these prefixes or selecting them 400s.
|
||||
const XAI_NON_CHAT_PREFIXES = ["grok-imagine-", "grok-stt-", "grok-voice-"] as const;
|
||||
|
||||
const GROK_EFFORT_CAPABLE_PREFIXES = ["grok-3-mini", "grok-4.20-multi-agent", "grok-4.3"] as const;
|
||||
|
||||
function grokSupportsReasoningEffort(modelId: string): boolean {
|
||||
const name = modelId.trim().toLowerCase();
|
||||
if (!name) return false;
|
||||
const bare = name.includes("/") ? name.slice(name.lastIndexOf("/") + 1) : name;
|
||||
return GROK_EFFORT_CAPABLE_PREFIXES.some(prefix => bare.startsWith(prefix));
|
||||
}
|
||||
|
||||
function withXaiOAuthCompatDefaults(model: ModelSpec<"openai-responses">): ModelSpec<"openai-responses"> {
|
||||
const compat = {
|
||||
...(model.compat ?? {}),
|
||||
includeEncryptedReasoning: model.compat?.includeEncryptedReasoning ?? false,
|
||||
filterReasoningHistory: model.compat?.filterReasoningHistory ?? true,
|
||||
omitReasoningEffort: model.compat?.omitReasoningEffort ?? !grokSupportsReasoningEffort(model.id),
|
||||
};
|
||||
return { ...model, compat };
|
||||
}
|
||||
|
||||
// Hermes-agent parity: only the `minimal -> low` clamp is applied (see
|
||||
// hermes-agent/agent/transports/codex.py:92 `_effort_clamp = {"minimal":
|
||||
// "low"}`). Hermes sends `xhigh` to xAI verbatim and we match that contract
|
||||
// — let xAI decide if the level is valid for the specific Grok model.
|
||||
// `resolveModelThinking` folds this into `model.thinking.effortMap`, downstream
|
||||
// of the omitReasoningEffort gate in xai-responses.ts.
|
||||
// of the omitReasoningEffort gate in pi-ai's stream.ts.
|
||||
const XAI_REASONING_EFFORT_MAP = { minimal: "low" } as const;
|
||||
|
||||
// xai-oauth's /v1/models exposes no per-request output limit on the OAuth
|
||||
@@ -942,6 +958,9 @@ function mergeCuratedIntoModel(
|
||||
const compat = {
|
||||
...(base.compat ?? {}),
|
||||
reasoningEffortMap: { ...XAI_REASONING_EFFORT_MAP, ...(base.compat?.reasoningEffortMap ?? {}) },
|
||||
includeEncryptedReasoning: base.compat?.includeEncryptedReasoning ?? false,
|
||||
filterReasoningHistory: base.compat?.filterReasoningHistory ?? true,
|
||||
omitReasoningEffort: base.compat?.omitReasoningEffort ?? !grokSupportsReasoningEffort(base.id),
|
||||
...(effort === undefined ? {} : { supportsReasoningEffort: effort }),
|
||||
};
|
||||
return {
|
||||
@@ -1005,7 +1024,7 @@ function applyXAIOAuthCuration(dynamic: readonly ModelSpec<"openai-responses">[]
|
||||
const curatedFirst = XAI_OAUTH_CURATED_MODELS.map(c => byId.get(c.id)).filter(
|
||||
(e): e is ModelSpec<"openai-responses"> => e !== undefined,
|
||||
);
|
||||
const rest = filtered.filter(e => !curatedIds.has(e.id));
|
||||
const rest = filtered.filter(e => !curatedIds.has(e.id)).map(withXaiOAuthCompatDefaults);
|
||||
return [...curatedFirst, ...rest];
|
||||
}
|
||||
|
||||
|
||||
+163
-48
@@ -5,6 +5,7 @@ export type { KnownProvider } from "./provider-models/descriptors";
|
||||
export type KnownApi =
|
||||
| "openai-completions"
|
||||
| "openai-responses"
|
||||
| "openrouter"
|
||||
| "openai-codex-responses"
|
||||
| "azure-openai-responses"
|
||||
| "anthropic-messages"
|
||||
@@ -142,6 +143,19 @@ export interface Usage {
|
||||
};
|
||||
}
|
||||
|
||||
export type OpenAIReasoningFormat = "openai" | "openrouter" | "zai" | "qwen" | "qwen-chat-template";
|
||||
|
||||
export type OpenAIReasoningDisableMode =
|
||||
| "omit"
|
||||
| "lowest-effort"
|
||||
| "openrouter-enabled-false"
|
||||
| "zai-thinking-disabled"
|
||||
| "qwen-enable-thinking-false"
|
||||
| "qwen-template-false"
|
||||
| "juice-zero-developer-message";
|
||||
|
||||
export type OpenAIStreamMarkupHealingPattern = "kimi" | "dsml" | "thinking";
|
||||
|
||||
/**
|
||||
* Compatibility settings for openai-completions API.
|
||||
* Use this to override URL-based auto-detection for custom providers.
|
||||
@@ -188,13 +202,23 @@ export interface OpenAICompat {
|
||||
/** Whether tool call IDs must be normalized to Mistral format (exactly 9 alphanumeric chars). Default: auto-detected from URL. */
|
||||
requiresMistralToolIds?: boolean;
|
||||
/** Format for reasoning/thinking parameter. "openai" uses reasoning_effort, "openrouter" uses reasoning: { effort }, "zai" uses thinking: { type: "enabled" | "disabled" } (also used by Moonshot Kimi), "qwen" uses top-level enable_thinking, and "qwen-chat-template" uses chat_template_kwargs.enable_thinking. Default: "openai". */
|
||||
thinkingFormat?: "openai" | "openrouter" | "zai" | "qwen" | "qwen-chat-template";
|
||||
thinkingFormat?: OpenAIReasoningFormat;
|
||||
/** Request-time disable encoding for the selected reasoning/thinking format. Default: derived from `thinkingFormat`. */
|
||||
reasoningDisableMode?: OpenAIReasoningDisableMode;
|
||||
/** Whether the provider rejects `reasoning.effort`/`reasoning_effort` even when the model reasons natively. Default: false unless reasoning effort is unsupported. */
|
||||
omitReasoningEffort?: boolean;
|
||||
/** Whether Responses requests should ask for encrypted reasoning replay items. Default: true. */
|
||||
includeEncryptedReasoning?: boolean;
|
||||
/** Whether replayed Responses history should strip native `type: "reasoning"` items before request encoding. Default: false. */
|
||||
filterReasoningHistory?: boolean;
|
||||
/** Optional `thinking.keep` value for Z.ai/Moonshot-style thinking params. Set false to suppress auto-detected keep. Default: auto-detected. */
|
||||
thinkingKeep?: "all" | false;
|
||||
/** Which reasoning content field to emit on assistant messages. Default: auto-detected. */
|
||||
reasoningContentField?: "reasoning_content" | "reasoning" | "reasoning_text";
|
||||
/** Whether assistant tool-call messages must include reasoning content. Default: false. */
|
||||
requiresReasoningContentForToolCalls?: boolean;
|
||||
/** Whether all assistant messages must include reasoning content. Default: false. */
|
||||
requiresReasoningContentForAllAssistantTurns?: boolean;
|
||||
/** Whether the provider accepts a synthetic placeholder (e.g. ".") for missing reasoning_content on tool-call turns. Default: true. Set to false for providers like DeepSeek that validate the exact reasoning_content value. */
|
||||
allowsSyntheticReasoningContentForToolCalls?: boolean;
|
||||
/** Whether assistant tool-call messages must include non-empty content. Default: false. */
|
||||
@@ -228,6 +252,8 @@ export interface OpenAICompat {
|
||||
vercelGatewayRouting?: VercelGatewayRouting;
|
||||
/** Extra fields to include in request body (e.g. gateway routing hints for OpenClaw-style proxies). */
|
||||
extraBody?: Record<string, unknown>;
|
||||
/** Request-session header that should mirror the normalized prompt-cache key. Default: unset. */
|
||||
promptCacheSessionHeader?: "x-grok-conv-id";
|
||||
/** Whether chat-completions payloads should include provider-specific prompt-cache markers. */
|
||||
cacheControlFormat?: "anthropic" | undefined;
|
||||
/** Whether the provider supports the `strict` field in tool definitions. Default: auto-detected per provider/baseUrl (conservative for unknown providers). */
|
||||
@@ -255,6 +281,16 @@ export interface OpenAICompat {
|
||||
* Default: auto-detected (GPT-5-family model names).
|
||||
*/
|
||||
requiresJuiceZeroHack?: boolean;
|
||||
/** Whether streamed reasoning deltas for the same field may repeat the full cumulative text snapshot. Default: false. */
|
||||
reasoningDeltasMayBeCumulative?: boolean;
|
||||
/** Strip leaked DeepSeek chat-template special tokens from visible content deltas. Default: auto-detected. */
|
||||
stripDeepseekSpecialTokens?: boolean;
|
||||
/** Heal leaked chat-template/tool-call/thinking markup from visible content deltas. Default: auto-detected. */
|
||||
streamMarkupHealingPattern?: OpenAIStreamMarkupHealingPattern;
|
||||
/** Treat an empty length-finished stream as a context-window error. Default: auto-detected. */
|
||||
emptyLengthFinishIsContextError?: boolean;
|
||||
/** Normalize tool call ids to OpenAI's 40-character limit. Default: auto-detected. */
|
||||
usesOpenAIToolCallIdLimit?: boolean;
|
||||
/**
|
||||
* Compat deltas applied when a request actually engages thinking mode
|
||||
* (reasoning requested and not disabled, model reasoning-capable, and not
|
||||
@@ -361,59 +397,135 @@ export interface VercelGatewayRouting {
|
||||
|
||||
type ResolvedToolStrictMode = NonNullable<OpenAICompat["toolStrictMode"]> | "mixed";
|
||||
|
||||
/**
|
||||
* Fields whose meaning is identical across chat-completions and Responses surfaces.
|
||||
* Each builder still computes its own per-surface value when defaults diverge.
|
||||
*/
|
||||
export interface ResolvedOpenAISharedCompat {
|
||||
supportsDeveloperRole: boolean;
|
||||
supportsStrictMode: boolean;
|
||||
supportsReasoningEffort: boolean;
|
||||
reasoningEffortMap: Partial<Record<Effort, string>>;
|
||||
supportsReasoningParams: boolean;
|
||||
thinkingFormat: OpenAIReasoningFormat;
|
||||
reasoningDisableMode: OpenAIReasoningDisableMode;
|
||||
omitReasoningEffort: boolean;
|
||||
includeEncryptedReasoning: boolean;
|
||||
filterReasoningHistory: boolean;
|
||||
disableReasoningOnForcedToolChoice: boolean;
|
||||
disableReasoningOnToolChoice: boolean;
|
||||
supportsToolChoice: boolean;
|
||||
supportsForcedToolChoice: boolean;
|
||||
reasoningContentField?: OpenAICompat["reasoningContentField"];
|
||||
requiresReasoningContentForToolCalls: boolean;
|
||||
requiresReasoningContentForAllAssistantTurns: boolean;
|
||||
allowsSyntheticReasoningContentForToolCalls: boolean;
|
||||
requiresThinkingAsText: boolean;
|
||||
requiresMistralToolIds: boolean;
|
||||
requiresToolResultName: boolean;
|
||||
requiresAssistantAfterToolResult: boolean;
|
||||
requiresAssistantContentForToolCalls: boolean;
|
||||
stripDeepseekSpecialTokens: boolean;
|
||||
streamMarkupHealingPattern?: OpenAIStreamMarkupHealingPattern;
|
||||
reasoningDeltasMayBeCumulative: boolean;
|
||||
emptyLengthFinishIsContextError: boolean;
|
||||
usesOpenAIToolCallIdLimit: boolean;
|
||||
promptCacheSessionHeader?: OpenAICompat["promptCacheSessionHeader"];
|
||||
/** The model sits behind OpenRouter (routing prefs and max-token omission apply). */
|
||||
isOpenRouterHost: boolean;
|
||||
/** Whether this endpoint needs a max-token field even when caller did not set one. */
|
||||
alwaysSendMaxTokens: boolean;
|
||||
/** See {@link OpenAICompat.enableGeminiThinkingLoopGuard}. Set by the builder from the family classifier. */
|
||||
enableGeminiThinkingLoopGuard?: boolean;
|
||||
openRouterRouting?: OpenAICompat["openRouterRouting"];
|
||||
/** Provider-specific wire model-id transform applied to the base id. */
|
||||
wireModelIdMode: "raw" | "firepass" | "fireworks" | "openrouter";
|
||||
}
|
||||
|
||||
/**
|
||||
* Fully-resolved chat-completions compat view: every detected default
|
||||
* materialized and user overrides applied. Built once per model by
|
||||
* `buildModel`; request handlers read fields and never detect, resolve, or
|
||||
* allocate.
|
||||
*/
|
||||
export type ResolvedOpenAICompat = Required<
|
||||
Omit<
|
||||
OpenAICompat,
|
||||
| "openRouterRouting"
|
||||
| "vercelGatewayRouting"
|
||||
| "extraBody"
|
||||
| "toolStrictMode"
|
||||
| "streamIdleTimeoutMs"
|
||||
| "supportsLongPromptCacheRetention"
|
||||
| "cacheControlFormat"
|
||||
| "thinkingKeep"
|
||||
| "strictResponsesPairing"
|
||||
| "requiresJuiceZeroHack"
|
||||
| "enableGeminiThinkingLoopGuard"
|
||||
| "whenThinking"
|
||||
>
|
||||
> & {
|
||||
openRouterRouting?: OpenAICompat["openRouterRouting"];
|
||||
vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"];
|
||||
extraBody?: OpenAICompat["extraBody"];
|
||||
cacheControlFormat?: OpenAICompat["cacheControlFormat"];
|
||||
thinkingKeep?: OpenAICompat["thinkingKeep"];
|
||||
streamIdleTimeoutMs?: number;
|
||||
toolStrictMode: ResolvedToolStrictMode;
|
||||
/** The model sits behind OpenRouter (routing prefs and max-token omission apply). */
|
||||
isOpenRouterHost: boolean;
|
||||
/** The model sits behind Vercel AI Gateway. */
|
||||
isVercelGatewayHost: boolean;
|
||||
/** See {@link OpenAICompat.enableGeminiThinkingLoopGuard}. Set by the builder from the family classifier. */
|
||||
enableGeminiThinkingLoopGuard?: boolean;
|
||||
/** Complete alternate view for thinking-engaged requests; swap pointers, never spread. */
|
||||
whenThinking?: ResolvedOpenAICompat;
|
||||
};
|
||||
export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
|
||||
Required<
|
||||
Omit<
|
||||
OpenAICompat,
|
||||
| "supportsDeveloperRole"
|
||||
| "supportsReasoningEffort"
|
||||
| "reasoningEffortMap"
|
||||
| "supportsReasoningParams"
|
||||
| "thinkingFormat"
|
||||
| "reasoningDisableMode"
|
||||
| "omitReasoningEffort"
|
||||
| "includeEncryptedReasoning"
|
||||
| "filterReasoningHistory"
|
||||
| "disableReasoningOnForcedToolChoice"
|
||||
| "disableReasoningOnToolChoice"
|
||||
| "supportsToolChoice"
|
||||
| "supportsForcedToolChoice"
|
||||
| "reasoningContentField"
|
||||
| "requiresReasoningContentForToolCalls"
|
||||
| "requiresReasoningContentForAllAssistantTurns"
|
||||
| "allowsSyntheticReasoningContentForToolCalls"
|
||||
| "requiresThinkingAsText"
|
||||
| "requiresMistralToolIds"
|
||||
| "requiresToolResultName"
|
||||
| "requiresAssistantAfterToolResult"
|
||||
| "requiresAssistantContentForToolCalls"
|
||||
| "stripDeepseekSpecialTokens"
|
||||
| "streamMarkupHealingPattern"
|
||||
| "reasoningDeltasMayBeCumulative"
|
||||
| "emptyLengthFinishIsContextError"
|
||||
| "usesOpenAIToolCallIdLimit"
|
||||
| "promptCacheSessionHeader"
|
||||
| "openRouterRouting"
|
||||
| "isOpenRouterHost"
|
||||
| "supportsStrictMode"
|
||||
| "supportsLongPromptCacheRetention"
|
||||
| "alwaysSendMaxTokens"
|
||||
| "wireModelIdMode"
|
||||
| "vercelGatewayRouting"
|
||||
| "extraBody"
|
||||
| "toolStrictMode"
|
||||
| "streamIdleTimeoutMs"
|
||||
| "cacheControlFormat"
|
||||
| "thinkingKeep"
|
||||
| "strictResponsesPairing"
|
||||
| "requiresJuiceZeroHack"
|
||||
| "enableGeminiThinkingLoopGuard"
|
||||
| "whenThinking"
|
||||
>
|
||||
> & {
|
||||
vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"];
|
||||
extraBody?: OpenAICompat["extraBody"];
|
||||
cacheControlFormat?: OpenAICompat["cacheControlFormat"];
|
||||
thinkingKeep?: OpenAICompat["thinkingKeep"];
|
||||
streamIdleTimeoutMs?: number;
|
||||
toolStrictMode: ResolvedToolStrictMode;
|
||||
/** The model sits behind Vercel AI Gateway. */
|
||||
isVercelGatewayHost: boolean;
|
||||
dropThinkingWhenReasoningEffort: boolean;
|
||||
/** Complete alternate view for thinking-engaged requests; swap pointers, never spread. */
|
||||
whenThinking?: ResolvedOpenAICompat;
|
||||
};
|
||||
|
||||
/** Fully-resolved Responses-API compat view (same contract as `ResolvedOpenAICompat`). */
|
||||
export interface ResolvedOpenAIResponsesCompat {
|
||||
supportsDeveloperRole: boolean;
|
||||
supportsStrictMode: boolean;
|
||||
supportsReasoningEffort: boolean;
|
||||
export interface ResolvedOpenAIResponsesCompat extends ResolvedOpenAISharedCompat {
|
||||
supportsLongPromptCacheRetention: boolean;
|
||||
strictResponsesPairing: boolean;
|
||||
requiresJuiceZeroHack: boolean;
|
||||
reasoningEffortMap: Partial<Record<Effort, string>>;
|
||||
/** See {@link OpenAICompat.enableGeminiThinkingLoopGuard}. */
|
||||
enableGeminiThinkingLoopGuard?: boolean;
|
||||
supportsObfuscationOptOut: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* OpenRouter is a pseudo API: runtime dispatch can use either Responses
|
||||
* (default) or Chat Completions (`PI_OPENROUTER_RESPONSES=0`) with the same
|
||||
* model object, so its resolved compat must satisfy both handlers.
|
||||
*/
|
||||
export type ResolvedOpenRouterCompat = ResolvedOpenAICompat & ResolvedOpenAIResponsesCompat;
|
||||
|
||||
/** Fully-resolved anthropic-messages compat view (same contract as `ResolvedOpenAICompat`). */
|
||||
export type ResolvedAnthropicCompat = Required<AnthropicCompat> & {
|
||||
/**
|
||||
@@ -428,6 +540,7 @@ export type ResolvedAnthropicCompat = Required<AnthropicCompat> & {
|
||||
/** Sparse, user-authored compat overrides for a given API (models.json / config vocabulary). */
|
||||
export type CompatConfigOf<TApi extends Api> = TApi extends
|
||||
| "openai-completions"
|
||||
| "openrouter"
|
||||
| "openai-responses"
|
||||
| "azure-openai-responses"
|
||||
| "openai-codex-responses"
|
||||
@@ -437,13 +550,15 @@ export type CompatConfigOf<TApi extends Api> = TApi extends
|
||||
: undefined;
|
||||
|
||||
/** Resolved compat for a given API: complete record, materialized once by `buildModel`. */
|
||||
export type CompatOf<TApi extends Api> = TApi extends "openai-completions"
|
||||
? ResolvedOpenAICompat
|
||||
: TApi extends "openai-responses" | "azure-openai-responses" | "openai-codex-responses"
|
||||
? ResolvedOpenAIResponsesCompat
|
||||
: TApi extends "anthropic-messages"
|
||||
? ResolvedAnthropicCompat
|
||||
: undefined;
|
||||
export type CompatOf<TApi extends Api> = TApi extends "openrouter"
|
||||
? ResolvedOpenRouterCompat
|
||||
: TApi extends "openai-completions"
|
||||
? ResolvedOpenAICompat
|
||||
: TApi extends "openai-responses" | "azure-openai-responses" | "openai-codex-responses"
|
||||
? ResolvedOpenAIResponsesCompat
|
||||
: TApi extends "anthropic-messages"
|
||||
? ResolvedAnthropicCompat
|
||||
: undefined;
|
||||
|
||||
// Model interface for the unified model system
|
||||
export interface Model<TApi extends Api = Api> {
|
||||
|
||||
@@ -5,7 +5,9 @@ import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { isOfficialAnthropicApiUrl } from "@oh-my-pi/pi-catalog/compat/anthropic";
|
||||
import { buildOpenAICompat, buildOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
|
||||
import { resolveProviderModels } from "@oh-my-pi/pi-catalog/model-manager";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import type { ModelSpec } from "@oh-my-pi/pi-catalog/types";
|
||||
|
||||
function completionsSpec(overrides: Partial<ModelSpec<"openai-completions">> = {}): ModelSpec<"openai-completions"> {
|
||||
@@ -24,6 +26,22 @@ function completionsSpec(overrides: Partial<ModelSpec<"openai-completions">> = {
|
||||
};
|
||||
}
|
||||
|
||||
function openrouterSpec(overrides: Partial<ModelSpec<"openrouter">> = {}): ModelSpec<"openrouter"> {
|
||||
return {
|
||||
id: "anthropic/claude-sonnet-4",
|
||||
name: "Claude Sonnet 4",
|
||||
api: "openrouter",
|
||||
provider: "openrouter",
|
||||
baseUrl: "https://openrouter.ai/api/v1",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 64_000,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
describe("buildModel", () => {
|
||||
it("resolves a complete compat record for an openai-completions spec with no compat", () => {
|
||||
const model = buildModel(completionsSpec());
|
||||
@@ -75,6 +93,29 @@ describe("buildModel", () => {
|
||||
expect(model.compat.whenThinking).toBeUndefined();
|
||||
});
|
||||
|
||||
it("builds OpenRouter pseudo-API models with shared chat and Responses compat", () => {
|
||||
const model = buildModel(
|
||||
openrouterSpec({
|
||||
compat: { openRouterRouting: { only: ["anthropic"], order: ["anthropic"] } },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(model.compat).toBeDefined();
|
||||
expect(model.compat.isOpenRouterHost).toBe(true);
|
||||
expect(model.compat.thinkingFormat).toBe("openrouter");
|
||||
expect(model.compat.supportsStrictMode).toBe(true);
|
||||
expect(model.compat.strictResponsesPairing).toBe(false);
|
||||
expect(model.compat.openRouterRouting).toEqual({ only: ["anthropic"], order: ["anthropic"] });
|
||||
});
|
||||
|
||||
it("loads bundled OpenRouter models with resolved compat", () => {
|
||||
const model = getBundledModel<"openrouter">("openrouter", "anthropic/claude-sonnet-4");
|
||||
|
||||
expect(model.compat).toBeDefined();
|
||||
expect(model.compat?.isOpenRouterHost).toBe(true);
|
||||
expect(model.compat?.supportsStrictMode).toBe(true);
|
||||
});
|
||||
|
||||
it("strips gateway author prefixes and extrinsic tags from display names", () => {
|
||||
const cases: [string, string][] = [
|
||||
["Anthropic: Claude Opus 4.6 (Fast) ($$$$)", "Claude Opus 4.6 (Fast)"],
|
||||
@@ -103,6 +144,103 @@ describe("buildModel", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("openai-completions wire-quirk compat detection", () => {
|
||||
it("derives wireModelIdMode from provider/host", () => {
|
||||
expect(buildOpenAICompat(completionsSpec({ provider: "firepass" })).wireModelIdMode).toBe("firepass");
|
||||
expect(
|
||||
buildOpenAICompat(completionsSpec({ provider: "fireworks", baseUrl: "https://api.fireworks.ai/inference/v1" }))
|
||||
.wireModelIdMode,
|
||||
).toBe("fireworks");
|
||||
expect(
|
||||
buildOpenAICompat(completionsSpec({ provider: "openrouter", baseUrl: "https://openrouter.ai/api/v1" }))
|
||||
.wireModelIdMode,
|
||||
).toBe("openrouter");
|
||||
expect(buildOpenAICompat(completionsSpec()).wireModelIdMode).toBe("raw");
|
||||
});
|
||||
|
||||
it("strips DeepSeek special tokens only for deepseek ids on nvidia/deepseek providers", () => {
|
||||
expect(
|
||||
buildOpenAICompat(
|
||||
completionsSpec({
|
||||
provider: "nvidia",
|
||||
id: "deepseek-ai/deepseek-v3.1",
|
||||
baseUrl: "https://integrate.api.nvidia.com/v1",
|
||||
}),
|
||||
).stripDeepseekSpecialTokens,
|
||||
).toBe(true);
|
||||
expect(
|
||||
buildOpenAICompat(
|
||||
completionsSpec({ provider: "deepseek", id: "deepseek-chat", baseUrl: "https://api.deepseek.com/v1" }),
|
||||
).stripDeepseekSpecialTokens,
|
||||
).toBe(true);
|
||||
// DeepSeek id behind another host must NOT strip (only nvidia/deepseek hosts emit the raw tokens).
|
||||
expect(
|
||||
buildOpenAICompat(
|
||||
completionsSpec({
|
||||
provider: "openrouter",
|
||||
id: "deepseek/deepseek-v3.1",
|
||||
baseUrl: "https://openrouter.ai/api/v1",
|
||||
}),
|
||||
).stripDeepseekSpecialTokens,
|
||||
).toBe(false);
|
||||
// Non-deepseek id on nvidia must NOT strip.
|
||||
expect(
|
||||
buildOpenAICompat(
|
||||
completionsSpec({
|
||||
provider: "nvidia",
|
||||
id: "meta/llama-3.1",
|
||||
baseUrl: "https://integrate.api.nvidia.com/v1",
|
||||
}),
|
||||
).stripDeepseekSpecialTokens,
|
||||
).toBe(false);
|
||||
});
|
||||
|
||||
it("flags cumulative reasoning deltas for MiniMax provider or id", () => {
|
||||
expect(buildOpenAICompat(completionsSpec({ provider: "minimax" })).reasoningDeltasMayBeCumulative).toBe(true);
|
||||
expect(buildOpenAICompat(completionsSpec({ id: "MiniMax-M2" })).reasoningDeltasMayBeCumulative).toBe(true);
|
||||
expect(buildOpenAICompat(completionsSpec()).reasoningDeltasMayBeCumulative).toBe(false);
|
||||
});
|
||||
|
||||
it("maps the remaining provider-keyed wire quirks", () => {
|
||||
expect(buildOpenAICompat(completionsSpec({ provider: "ollama" })).emptyLengthFinishIsContextError).toBe(true);
|
||||
expect(buildOpenAICompat(completionsSpec()).emptyLengthFinishIsContextError).toBe(false);
|
||||
expect(
|
||||
buildOpenAICompat(completionsSpec({ provider: "openai", baseUrl: "https://api.openai.com/v1" }))
|
||||
.usesOpenAIToolCallIdLimit,
|
||||
).toBe(true);
|
||||
expect(buildOpenAICompat(completionsSpec()).usesOpenAIToolCallIdLimit).toBe(false);
|
||||
expect(
|
||||
buildOpenAICompat(completionsSpec({ provider: "fireworks", baseUrl: "https://api.fireworks.ai/inference/v1" }))
|
||||
.dropThinkingWhenReasoningEffort,
|
||||
).toBe(true);
|
||||
expect(buildOpenAICompat(completionsSpec()).dropThinkingWhenReasoningEffort).toBe(false);
|
||||
});
|
||||
|
||||
it("derives Responses obfuscation opt-out and wire mode per surface", () => {
|
||||
expect(
|
||||
buildOpenAIResponsesCompat({
|
||||
id: "gpt-5",
|
||||
provider: "openai",
|
||||
name: "GPT 5",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
}).supportsObfuscationOptOut,
|
||||
).toBe(true);
|
||||
// Azure mirrors the schema but is NOT the OpenAI host: no obfuscation opt-out.
|
||||
expect(
|
||||
buildOpenAIResponsesCompat({ id: "gpt-5", provider: "azure", name: "gpt-5", baseUrl: "" })
|
||||
.supportsObfuscationOptOut,
|
||||
).toBe(false);
|
||||
const openrouterResponses = buildOpenAIResponsesCompat({
|
||||
id: "anthropic/claude-sonnet-4",
|
||||
provider: "openrouter",
|
||||
name: "Claude Sonnet 4",
|
||||
baseUrl: "https://openrouter.ai/api/v1",
|
||||
});
|
||||
expect(openrouterResponses.supportsObfuscationOptOut).toBe(false);
|
||||
expect(openrouterResponses.wireModelIdMode).toBe("openrouter");
|
||||
});
|
||||
});
|
||||
|
||||
describe("model cache spec round trip", () => {
|
||||
it("persists sparse specs and rebuilds resolved models on cache reads", async () => {
|
||||
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-model-cache-"));
|
||||
|
||||
@@ -12,6 +12,14 @@ import type { FetchImpl, Model } from "@oh-my-pi/pi-catalog/types";
|
||||
const TP_KEY = "tp-ci1p8t1w4e1sbxgyc8v65tnrjbzro287igmvyf25van9mt76";
|
||||
const SGP_BASE_URL = "https://token-plan-sgp.xiaomimimo.com/v1";
|
||||
|
||||
interface MessageWithReasoningContent {
|
||||
reasoning_content?: unknown;
|
||||
}
|
||||
|
||||
function isMessageWithReasoningContent(value: unknown): value is MessageWithReasoningContent {
|
||||
return value !== null && typeof value === "object";
|
||||
}
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
@@ -129,6 +137,8 @@ describe("issue #1846: Xiaomi Token Plan provider support", () => {
|
||||
expect(compat.thinkingFormat).toBe("zai");
|
||||
expect(compat.requiresReasoningContentForToolCalls).toBe(true);
|
||||
expect(compat.allowsSyntheticReasoningContentForToolCalls).toBe(false);
|
||||
expect(Reflect.get(assistant ?? {}, "reasoning_content")).toBe("I need to inspect the file before answering.");
|
||||
expect(isMessageWithReasoningContent(assistant) ? assistant.reasoning_content : undefined).toBe(
|
||||
"I need to inspect the file before answering.",
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,23 +1,31 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added support for OpenRouter fallback in Perplexity web search when direct Perplexity API keys fail or are unavailable
|
||||
- Added support for streaming the Perplexity Responses API (`/v1/responses`) via the `PI_PERPLEXITY_RESPONSES=1` environment variable
|
||||
- Added `omp ttsr` top-level CLI command to inspect and test Time-Traveling Stream Rules
|
||||
- Added `omp ttsr list` to enumerate all project/user-loaded TTSR rules with their conditions, scope, and source metadata
|
||||
- Added `omp ttsr test` to run snippets through the real TTSR matching pipeline with inline text, `--file <path>`, or stdin via `--file -`
|
||||
- Added `--json` output to `omp ttsr test` and `omp ttsr list` for machine-readable reporting
|
||||
- Added `--rule`, `--source`, `--tool`, `--path`, and `--verbose` options to `omp ttsr test` to control matching context and inspection details
|
||||
- Added `omp ttsr` subcommand for inspecting and testing Time-Traveling Stream Rules: `omp ttsr list` shows every TTSR-registered rule the current project/user config would load, and `omp ttsr test` feeds a snippet (inline, `--file`, or stdin) through the real TTSR matching pipeline (`TtsrManager.checkSnapshot`/`checkAstSnapshot`) and reports which rules would trigger. A positional that resolves to a file defaults to tool/edit context; `--source`, `--tool`, and `--path` override the inferred match context so glob/AST/scope-scoped rules evaluate the same way they do in a live session. `--rule` tests a single rule markdown file in isolation.
|
||||
- Added support for reading embedded PDF images via `read <pdf>:<image>.png` and listing available image members with `read <pdf>:`
|
||||
- Added a built-in `ts-no-inline-cast-access` TTSR rule that interrupts inline object-type assertions read immediately as a property (`(x as { y: T }).y`, including `?.` and bracket access), steering toward schema validation, `in`/`typeof` narrowing, or a validated named type
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed advisor model calls and overflow-compaction tasks to inherit and propagate primary telemetry spans, usage, and cost tracking
|
||||
- Changed PDF read output to replace `<!-- image: ... -->` placeholders with clickable `read <pdf>:<image>.png` handles, including line-range and multi-range reads
|
||||
- Changed the built-in `ts-no-any` rule to recommend a schema parse at trust boundaries and `in`-narrowing (instead of an inline `as`-cast) when reading fields off `unknown`
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Perplexity web search to use shared OpenAI streaming transports while preserving streamed sources, citations, and related questions
|
||||
- Fixed `read <pdf>:<member>` errors for unknown PDF images to surface available extracted image names
|
||||
- Fixed puppeteer stealth scripts to use cached Reflect methods (`Reflect_get`, `Reflect_apply`) and `Reflect.apply` instead of live `Reflect`/`Function.prototype.apply` calls, preventing page tampering from leaking through proxy traps.
|
||||
- Fixed Perplexity API-key web search to use shared OpenAI streaming transports while preserving streamed sources and OpenRouter fallback.
|
||||
|
||||
### Security
|
||||
|
||||
|
||||
@@ -67,6 +67,7 @@
|
||||
"@puppeteer/browsers": "catalog:",
|
||||
"@types/turndown": "catalog:",
|
||||
"@xterm/headless": "catalog:",
|
||||
"arktype": "catalog:",
|
||||
"chalk": "catalog:",
|
||||
"diff": "catalog:",
|
||||
"fflate": "catalog:",
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { describe, expect, it, vi } from "bun:test";
|
||||
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
|
||||
import type { AgentMessage, AgentTelemetryConfig } from "@oh-my-pi/pi-agent-core";
|
||||
import { createAdvisorMessageCard } from "../../modes/components/advisor-message";
|
||||
import { getThemeByName } from "../../modes/theme/theme";
|
||||
import { formatSessionHistoryMarkdown } from "../../session/session-history-format";
|
||||
@@ -11,6 +11,7 @@ import {
|
||||
type AdvisorNote,
|
||||
AdvisorRuntime,
|
||||
type AdvisorRuntimeHost,
|
||||
deriveAdvisorTelemetry,
|
||||
formatAdvisorBatchContent,
|
||||
isAdvisorInterruptImmuneTurnActive,
|
||||
isInterruptingSeverity,
|
||||
@@ -181,6 +182,37 @@ describe("advisor", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("deriveAdvisorTelemetry", () => {
|
||||
it("returns undefined when the primary has no telemetry so the advisor stays a no-op", () => {
|
||||
expect(deriveAdvisorTelemetry(undefined, { id: "s-advisor", name: "Advisor" })).toBeUndefined();
|
||||
});
|
||||
|
||||
it("inherits the primary's usage/cost hooks but restamps identity and clears the conversation", () => {
|
||||
const onChatUsage = vi.fn();
|
||||
const costEstimator = vi.fn();
|
||||
const primary: AgentTelemetryConfig = {
|
||||
agent: { id: "main", name: "Main" },
|
||||
conversationId: "session-1",
|
||||
attributes: { "deployment.id": "prod" },
|
||||
onChatUsage,
|
||||
costEstimator,
|
||||
};
|
||||
const identity = { id: "session-1-advisor", name: "Advisor", description: "anthropic/claude-sonnet-4-5" };
|
||||
|
||||
const derived = deriveAdvisorTelemetry(primary, identity);
|
||||
|
||||
// Usage/cost hooks are inherited so the advisor model's calls report through
|
||||
// the same pipeline as the primary — the whole point of the fix.
|
||||
expect(derived?.onChatUsage).toBe(onChatUsage);
|
||||
expect(derived?.costEstimator).toBe(costEstimator);
|
||||
expect(derived?.attributes).toEqual({ "deployment.id": "prod" });
|
||||
// Advisor identity replaces the primary's so spans are attributable to the advisor.
|
||||
expect(derived?.agent).toEqual(identity);
|
||||
// Conversation cleared so the advisor loop falls back to its own `-advisor` session id.
|
||||
expect(derived?.conversationId).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("AdvisorRuntime", () => {
|
||||
function makeAgent(promptInputs: string[]): AdvisorAgent {
|
||||
return {
|
||||
|
||||
@@ -1,4 +1,11 @@
|
||||
import type { AgentTool, AgentToolContext, AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core";
|
||||
import type {
|
||||
AgentIdentity,
|
||||
AgentTelemetryConfig,
|
||||
AgentTool,
|
||||
AgentToolContext,
|
||||
AgentToolResult,
|
||||
AgentToolUpdateCallback,
|
||||
} from "@oh-my-pi/pi-agent-core";
|
||||
import { escapeXmlText } from "@oh-my-pi/pi-utils";
|
||||
import { z } from "zod/v4";
|
||||
import adviseDescription from "../prompts/advisor/advise-tool.md" with { type: "text" };
|
||||
@@ -109,6 +116,25 @@ export function resolveAdvisorDeliveryChannel(opts: {
|
||||
return "steer";
|
||||
}
|
||||
|
||||
/**
|
||||
* Derive the advisor loop's telemetry from the primary session's config so the
|
||||
* advisor model's GenAI spans and usage/cost hooks (onChatUsage, onCostDelta,
|
||||
* costEstimator) fire under the same pipeline as every other model call —
|
||||
* stamped with the advisor's own agent identity. `conversationId` is cleared so
|
||||
* the advisor loop falls back to its own `-advisor` session id for
|
||||
* `gen_ai.conversation.id` instead of inheriting the primary's conversation.
|
||||
*
|
||||
* Returns undefined when the primary has no telemetry (instrumentation off), so
|
||||
* the advisor `Agent` stays a zero-overhead no-op as well.
|
||||
*/
|
||||
export function deriveAdvisorTelemetry(
|
||||
primaryTelemetry: AgentTelemetryConfig | undefined,
|
||||
identity: AgentIdentity,
|
||||
): AgentTelemetryConfig | undefined {
|
||||
if (!primaryTelemetry) return undefined;
|
||||
return { ...primaryTelemetry, agent: identity, conversationId: undefined };
|
||||
}
|
||||
|
||||
/**
|
||||
* Side-effect-free investigation tools handed to the advisor agent so it can
|
||||
* inspect the workspace before weighing in. Names match the primary session's
|
||||
|
||||
@@ -126,6 +126,7 @@ import {
|
||||
type AdvisorNote,
|
||||
AdvisorRuntime,
|
||||
type AdvisorSeverity,
|
||||
deriveAdvisorTelemetry,
|
||||
formatAdvisorBatchContent,
|
||||
isAdvisorInterruptImmuneTurnActive,
|
||||
isInterruptingSeverity,
|
||||
@@ -150,7 +151,7 @@ import {
|
||||
resolveModelRoleValue,
|
||||
resolveRoleSelection,
|
||||
} from "../config/model-resolver";
|
||||
import { MODEL_ROLE_IDS } from "../config/model-roles";
|
||||
import { MODEL_ROLE_IDS, MODEL_ROLES } from "../config/model-roles";
|
||||
import { expandPromptTemplate, type PromptTemplate } from "../config/prompt-templates";
|
||||
import type { Settings, SkillsSettings } from "../config/settings";
|
||||
import { onAppendOnlyModeChanged } from "../config/settings";
|
||||
@@ -1755,6 +1756,15 @@ export class AgentSession {
|
||||
systemPrompt.push(this.#advisorWatchdogPrompt);
|
||||
}
|
||||
const advisorSessionId = this.sessionId ? `${this.sessionId}-advisor` : undefined;
|
||||
|
||||
// Thread the primary's telemetry into the advisor loop so the advisor
|
||||
// model's GenAI spans + usage/cost hooks fire like every other model call,
|
||||
// stamped with the advisor's own identity (see deriveAdvisorTelemetry).
|
||||
const advisorTelemetry = deriveAdvisorTelemetry(this.agent.telemetry, {
|
||||
id: advisorSessionId,
|
||||
name: MODEL_ROLES.advisor.name,
|
||||
description: formatModelString(advisorSel.model),
|
||||
});
|
||||
const advisorAgent = new Agent({
|
||||
initialState: {
|
||||
systemPrompt,
|
||||
@@ -1766,6 +1776,7 @@ export class AgentSession {
|
||||
sessionId: advisorSessionId,
|
||||
getApiKey: requestModel => this.#modelRegistry.resolver(requestModel, advisorSessionId),
|
||||
intentTracing: false,
|
||||
telemetry: advisorTelemetry,
|
||||
});
|
||||
advisorAgent.setDisableReasoning(shouldDisableReasoning(advisorThinkingLevel));
|
||||
|
||||
@@ -1941,24 +1952,26 @@ export class AgentSession {
|
||||
|
||||
let compactResult: CompactionResult | undefined;
|
||||
let lastError: unknown;
|
||||
const advisorSessionId = this.sessionId ? `${this.sessionId}-advisor` : undefined;
|
||||
// Instrument the advisor's overflow-compaction one-shot like the primary
|
||||
// compaction path so the advisor model's maintenance call also emits spans.
|
||||
const telemetry = resolveTelemetry(advisor.telemetry, advisorSessionId);
|
||||
|
||||
for (const candidate of candidates) {
|
||||
const apiKey = await this.#modelRegistry.getApiKey(
|
||||
candidate,
|
||||
this.sessionId ? `${this.sessionId}-advisor` : undefined,
|
||||
);
|
||||
const apiKey = await this.#modelRegistry.getApiKey(candidate, advisorSessionId);
|
||||
if (!apiKey) continue;
|
||||
|
||||
try {
|
||||
compactResult = await compact(
|
||||
preparation,
|
||||
candidate,
|
||||
this.#modelRegistry.resolver(candidate, this.sessionId ? `${this.sessionId}-advisor` : undefined),
|
||||
this.#modelRegistry.resolver(candidate, advisorSessionId),
|
||||
undefined,
|
||||
undefined,
|
||||
{
|
||||
thinkingLevel: advisorCompactionThinkingLevel,
|
||||
convertToLlm: messages => this.#convertToLlmForSideRequest(messages),
|
||||
telemetry,
|
||||
},
|
||||
);
|
||||
break;
|
||||
|
||||
@@ -1,6 +1,14 @@
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { type ApiKey, type FetchImpl, getEnvApiKey, type Model, ProviderHttpError, withAuth } from "@oh-my-pi/pi-ai";
|
||||
import {
|
||||
type ApiKey,
|
||||
type FetchImpl,
|
||||
getEnvApiKey,
|
||||
getOpenRouterHeaders,
|
||||
type Model,
|
||||
ProviderHttpError,
|
||||
withAuth,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
import {
|
||||
CODEX_BASE_URL,
|
||||
getCodexAccountId,
|
||||
@@ -20,7 +28,6 @@ import {
|
||||
untilAborted,
|
||||
} from "@oh-my-pi/pi-utils";
|
||||
import { z } from "zod/v4";
|
||||
import packageJson from "../../package.json" with { type: "json" };
|
||||
import { isAuthenticated, type ModelRegistry } from "../config/model-registry";
|
||||
import { settings } from "../config/settings";
|
||||
import type { CustomTool } from "../extensibility/custom-tools/types";
|
||||
@@ -1407,9 +1414,7 @@ export const imageGenTool: CustomTool<typeof imageGenSchema, ImageGenToolDetails
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
Authorization: `Bearer ${key}`,
|
||||
"HTTP-Referer": "https://omp.sh/",
|
||||
"X-OpenRouter-Title": "Oh-My-Pi",
|
||||
"X-OpenRouter-Categories": "cli-agent",
|
||||
...getOpenRouterHeaders(),
|
||||
},
|
||||
body: JSON.stringify(requestBody),
|
||||
signal: requestSignal,
|
||||
|
||||
@@ -8,12 +8,24 @@
|
||||
* - Anonymous via `www.perplexity.ai/rest/sse/perplexity_ask`
|
||||
*/
|
||||
|
||||
import { type AuthStorage, type FetchImpl, getEnvApiKey, type OAuthAccess, withOAuthAccess } from "@oh-my-pi/pi-ai";
|
||||
import {
|
||||
type AssistantMessage,
|
||||
type AssistantMessageEventStream,
|
||||
type AuthStorage,
|
||||
type Context,
|
||||
type FetchImpl,
|
||||
type OAuthAccess,
|
||||
type Usage,
|
||||
withOAuthAccess,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import type { Model, ModelSpec } from "@oh-my-pi/pi-catalog/types";
|
||||
import { $env, readSseJson } from "@oh-my-pi/pi-utils";
|
||||
import type {
|
||||
PerplexityMessageOutput,
|
||||
PerplexityRequest,
|
||||
PerplexityResponse,
|
||||
PerplexitySearchResult,
|
||||
SearchCitation,
|
||||
SearchResponse,
|
||||
SearchSource,
|
||||
@@ -24,7 +36,9 @@ import type { SearchParams } from "./base";
|
||||
import { SearchProvider } from "./base";
|
||||
import { classifyProviderHttpError, withHardTimeout } from "./utils";
|
||||
|
||||
const PERPLEXITY_API_URL = "https://api.perplexity.ai/chat/completions";
|
||||
const PERPLEXITY_CHAT_BASE_URL = "https://api.perplexity.ai";
|
||||
const PERPLEXITY_RESPONSES_BASE_URL = "https://api.perplexity.ai/v1";
|
||||
const OPENROUTER_BASE_URL = "https://openrouter.io/api/v1";
|
||||
const PERPLEXITY_OAUTH_ASK_URL = "https://www.perplexity.ai/rest/sse/perplexity_ask";
|
||||
|
||||
const DEFAULT_MAX_TOKENS = 8192;
|
||||
@@ -37,10 +51,7 @@ const ANONYMOUS_USER_AGENT =
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36";
|
||||
|
||||
type PerplexityAuth =
|
||||
| {
|
||||
type: "api_key";
|
||||
token: string;
|
||||
}
|
||||
| ApiConfig
|
||||
| {
|
||||
type: "oauth";
|
||||
access: OAuthAccess;
|
||||
@@ -278,9 +289,52 @@ export interface PerplexitySearchParams {
|
||||
fetch?: FetchImpl;
|
||||
}
|
||||
|
||||
/** Find PERPLEXITY_API_KEY from environment or .env files (also checks PPLX_API_KEY) */
|
||||
export function findApiKey(): string | null {
|
||||
return getEnvApiKey("perplexity") ?? null;
|
||||
interface ApiConfig {
|
||||
type: "api_key";
|
||||
apiKey: string;
|
||||
provider: "perplexity" | "openrouter";
|
||||
chatBaseUrl: string;
|
||||
responsesBaseUrl: string;
|
||||
modelPrefix: string;
|
||||
useResponses: boolean;
|
||||
}
|
||||
|
||||
/** Detect API-key endpoints to try in priority order (Perplexity direct, then OpenRouter). */
|
||||
async function getApiConfigs(
|
||||
authStorage: AuthStorage,
|
||||
sessionId: string | undefined,
|
||||
signal: AbortSignal | undefined,
|
||||
): Promise<ApiConfig[]> {
|
||||
const useResponses = $env.PI_PERPLEXITY_RESPONSES === "1";
|
||||
const configs: ApiConfig[] = [];
|
||||
|
||||
const perplexityKey = await authStorage.getApiKey("perplexity", sessionId, { signal });
|
||||
if (perplexityKey) {
|
||||
configs.push({
|
||||
type: "api_key",
|
||||
apiKey: perplexityKey,
|
||||
provider: "perplexity",
|
||||
chatBaseUrl: PERPLEXITY_CHAT_BASE_URL,
|
||||
responsesBaseUrl: PERPLEXITY_RESPONSES_BASE_URL,
|
||||
modelPrefix: "",
|
||||
useResponses,
|
||||
});
|
||||
}
|
||||
|
||||
const openrouterKey = await authStorage.getApiKey("openrouter", sessionId, { signal });
|
||||
if (openrouterKey) {
|
||||
configs.push({
|
||||
type: "api_key",
|
||||
apiKey: openrouterKey,
|
||||
provider: "openrouter",
|
||||
chatBaseUrl: OPENROUTER_BASE_URL,
|
||||
responsesBaseUrl: OPENROUTER_BASE_URL,
|
||||
modelPrefix: "perplexity/",
|
||||
useResponses,
|
||||
});
|
||||
}
|
||||
|
||||
return configs;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -302,86 +356,236 @@ function jwtExpiryMs(token: string): number | undefined {
|
||||
}
|
||||
}
|
||||
|
||||
async function findOAuthAccess(
|
||||
/** Collect all available auth methods to try in priority order */
|
||||
async function getAvailableAuthMethods(
|
||||
authStorage: AuthStorage,
|
||||
sessionId: string | undefined,
|
||||
signal: AbortSignal | undefined,
|
||||
): Promise<OAuthAccess | null> {
|
||||
): Promise<PerplexityAuth[]> {
|
||||
const methods: PerplexityAuth[] = [];
|
||||
|
||||
// 1. Perplexity OAuth & Cookies (same priority - highest)
|
||||
try {
|
||||
// `getOAuthAccess` returns the raw OAuth bearer only — runtime/config
|
||||
// api_key overrides and stored api_key credentials are intentionally
|
||||
// suppressed so we don't POST an `api.perplexity.ai` key to the
|
||||
// `www.perplexity.ai` session/SSE endpoint.
|
||||
const access = await authStorage.getOAuthAccess("perplexity", sessionId, { signal });
|
||||
const token = access?.accessToken;
|
||||
if (!access || !token) return null;
|
||||
// Trust the JWT's own `exp` claim if it has one; otherwise treat as
|
||||
// non-expiring. Perplexity session JWTs commonly omit `exp`.
|
||||
const jwtExpiry = jwtExpiryMs(token);
|
||||
if (jwtExpiry !== undefined && jwtExpiry <= Date.now() + OAUTH_EXPIRY_BUFFER_MS) return null;
|
||||
return access;
|
||||
if (access && token) {
|
||||
const jwtExpiry = jwtExpiryMs(token);
|
||||
if (jwtExpiry === undefined || jwtExpiry > Date.now() + OAUTH_EXPIRY_BUFFER_MS) {
|
||||
methods.push({ type: "oauth", access });
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
return null;
|
||||
// ignored
|
||||
}
|
||||
}
|
||||
|
||||
async function findPerplexityAuth(
|
||||
authStorage: AuthStorage,
|
||||
sessionId: string | undefined,
|
||||
signal: AbortSignal | undefined,
|
||||
): Promise<PerplexityAuth> {
|
||||
// 1. PERPLEXITY_COOKIES env var
|
||||
const cookies = $env.PERPLEXITY_COOKIES?.trim();
|
||||
if (cookies) {
|
||||
return { type: "cookies", cookies };
|
||||
methods.push({ type: "cookies", cookies });
|
||||
}
|
||||
|
||||
const apiKey = findApiKey();
|
||||
const apiConfigs = await getApiConfigs(authStorage, sessionId, signal);
|
||||
methods.push(...apiConfigs);
|
||||
|
||||
// 2. OAuth/session bearer from AuthStorage.
|
||||
const oauthAccess = await findOAuthAccess(authStorage, sessionId, signal);
|
||||
if (oauthAccess) {
|
||||
return { type: "oauth", access: oauthAccess };
|
||||
// 5. Fallback to Perplexity free (anonymous)
|
||||
if (methods.length === 0) {
|
||||
methods.push({ type: "anonymous" });
|
||||
}
|
||||
|
||||
// 3. PERPLEXITY_API_KEY env var
|
||||
if (apiKey) {
|
||||
return { type: "api_key", token: apiKey };
|
||||
}
|
||||
|
||||
// 4. The consumer ask endpoint currently accepts unauthenticated browser-style requests.
|
||||
return { type: "anonymous" };
|
||||
return methods;
|
||||
}
|
||||
|
||||
/** Call Perplexity API-key endpoint. */
|
||||
interface PerplexityApiStreamMetadata {
|
||||
id?: string;
|
||||
model?: string;
|
||||
citations?: unknown;
|
||||
search_results?: unknown;
|
||||
related_questions?: unknown;
|
||||
}
|
||||
|
||||
function buildPerplexityCompletionsModel(config: ApiConfig, request: PerplexityRequest): Model<"openai-completions"> {
|
||||
const model = config.modelPrefix ? `${config.modelPrefix}${request.model}` : request.model;
|
||||
const spec: ModelSpec<"openai-completions"> = {
|
||||
id: model,
|
||||
name: model,
|
||||
api: "openai-completions",
|
||||
provider: config.provider,
|
||||
baseUrl: config.chatBaseUrl,
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
supportsTools: false,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: null,
|
||||
maxTokens: null,
|
||||
compat: {
|
||||
supportsStore: false,
|
||||
supportsMultipleSystemMessages: true,
|
||||
supportsReasoningParams: false,
|
||||
supportsUsageInStreaming: true,
|
||||
maxTokensField: "max_tokens",
|
||||
},
|
||||
};
|
||||
return buildModel(spec);
|
||||
}
|
||||
|
||||
function buildPerplexityResponsesModel(config: ApiConfig, request: PerplexityRequest): Model<"openai-responses"> {
|
||||
const model = config.modelPrefix ? `${config.modelPrefix}${request.model}` : request.model;
|
||||
const spec: ModelSpec<"openai-responses"> = {
|
||||
id: model,
|
||||
name: model,
|
||||
api: "openai-responses",
|
||||
provider: config.provider,
|
||||
baseUrl: config.responsesBaseUrl,
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
supportsTools: false,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: null,
|
||||
maxTokens: null,
|
||||
compat: {
|
||||
alwaysSendMaxTokens: true,
|
||||
supportsReasoningParams: false,
|
||||
},
|
||||
};
|
||||
return buildModel(spec);
|
||||
}
|
||||
|
||||
function buildPerplexityContext(request: PerplexityRequest): Context {
|
||||
const systemPrompt: string[] = [];
|
||||
const messages: Context["messages"] = [];
|
||||
for (const message of request.messages) {
|
||||
if (typeof message.content !== "string" || message.content.length === 0) continue;
|
||||
if (message.role === "system") {
|
||||
systemPrompt.push(message.content);
|
||||
continue;
|
||||
}
|
||||
if (message.role === "user") {
|
||||
messages.push({ role: "user", content: message.content, timestamp: 0 });
|
||||
}
|
||||
}
|
||||
return { systemPrompt: systemPrompt.length > 0 ? systemPrompt : undefined, messages };
|
||||
}
|
||||
|
||||
function buildPerplexityExtraBody(request: PerplexityRequest): Record<string, unknown> {
|
||||
return {
|
||||
search_mode: request.search_mode,
|
||||
num_search_results: request.num_search_results,
|
||||
web_search_options: request.web_search_options,
|
||||
enable_search_classifier: request.enable_search_classifier,
|
||||
reasoning_effort: request.reasoning_effort,
|
||||
language_preference: request.language_preference,
|
||||
return_related_questions: request.return_related_questions,
|
||||
search_recency_filter: request.search_recency_filter,
|
||||
};
|
||||
}
|
||||
|
||||
function applyPerplexityExtraBody(payload: unknown, request: PerplexityRequest): void {
|
||||
const record = asRecord(payload);
|
||||
if (!record) return;
|
||||
Object.assign(record, buildPerplexityExtraBody(request));
|
||||
}
|
||||
|
||||
function collectPerplexityOutputMetadata(metadata: PerplexityApiStreamMetadata, output: unknown): void {
|
||||
if (!Array.isArray(output)) return;
|
||||
for (const item of output) {
|
||||
const record = asRecord(item);
|
||||
if (!record) continue;
|
||||
if (Array.isArray(record.search_results)) metadata.search_results = record.search_results;
|
||||
if (Array.isArray(record.results)) metadata.search_results = record.results;
|
||||
if (Array.isArray(record.citations)) metadata.citations = record.citations;
|
||||
if (Array.isArray(record.related_questions)) metadata.related_questions = record.related_questions;
|
||||
collectPerplexityOutputMetadata(metadata, record.content);
|
||||
}
|
||||
}
|
||||
|
||||
function collectPerplexityMetadataFromRecord(
|
||||
metadata: PerplexityApiStreamMetadata,
|
||||
record: Record<string, unknown>,
|
||||
): void {
|
||||
const id = record.id;
|
||||
if (typeof id === "string" && id.length > 0) metadata.id = id;
|
||||
const model = record.model;
|
||||
if (typeof model === "string" && model.length > 0) metadata.model = model;
|
||||
if (Array.isArray(record.citations)) metadata.citations = record.citations;
|
||||
if (Array.isArray(record.search_results)) metadata.search_results = record.search_results;
|
||||
if (Array.isArray(record.related_questions)) metadata.related_questions = record.related_questions;
|
||||
if (Array.isArray(record.results)) metadata.search_results = record.results;
|
||||
collectPerplexityOutputMetadata(metadata, record.output);
|
||||
const response = asRecord(record.response);
|
||||
if (response) {
|
||||
collectPerplexityOutputMetadata(metadata, response.output);
|
||||
collectPerplexityMetadataFromRecord(metadata, response);
|
||||
}
|
||||
}
|
||||
|
||||
function collectPerplexityMetadata(metadata: PerplexityApiStreamMetadata, data: string): void {
|
||||
if (data === "[DONE]") return;
|
||||
const record = asRecord(parseJson(data));
|
||||
if (record) collectPerplexityMetadataFromRecord(metadata, record);
|
||||
}
|
||||
|
||||
async function drainAssistantStream(stream: AssistantMessageEventStream): Promise<AssistantMessage> {
|
||||
let finalMessage: AssistantMessage | undefined;
|
||||
for await (const event of stream) {
|
||||
if (event.type === "done") {
|
||||
finalMessage = event.message;
|
||||
} else if (event.type === "error") {
|
||||
finalMessage = event.error;
|
||||
}
|
||||
}
|
||||
return finalMessage ?? stream.result();
|
||||
}
|
||||
|
||||
function throwPerplexityStreamError(message: AssistantMessage): never {
|
||||
const status = message.errorStatus ?? 500;
|
||||
const details = message.errorMessage ?? "Perplexity API stream failed";
|
||||
const classified = classifyProviderHttpError("perplexity", status, details);
|
||||
if (classified) throw classified;
|
||||
throw new SearchProviderError("perplexity", `Perplexity API error (${status}): ${details}`, status);
|
||||
}
|
||||
|
||||
/** Call Perplexity API-key endpoint (or OpenRouter) through the shared OpenAI streaming providers. */
|
||||
async function callPerplexityApi(
|
||||
apiKey: string,
|
||||
config: ApiConfig,
|
||||
request: PerplexityRequest,
|
||||
fetchImpl: FetchImpl | undefined,
|
||||
signal?: AbortSignal,
|
||||
): Promise<PerplexityResponse> {
|
||||
const response = await (fetchImpl ?? fetch)(PERPLEXITY_API_URL, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
Authorization: `Bearer ${apiKey}`,
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
body: JSON.stringify(request),
|
||||
signal: withHardTimeout(signal),
|
||||
});
|
||||
): Promise<SearchResponse> {
|
||||
const metadata: PerplexityApiStreamMetadata = {};
|
||||
const context = buildPerplexityContext(request);
|
||||
const requestSignal = withHardTimeout(signal);
|
||||
const onSseEvent = (event: { data: string }): void => {
|
||||
collectPerplexityMetadata(metadata, event.data);
|
||||
};
|
||||
|
||||
if (!response.ok) {
|
||||
const errorText = await response.text();
|
||||
const classified = classifyProviderHttpError("perplexity", response.status, errorText);
|
||||
if (classified) throw classified;
|
||||
throw new SearchProviderError(
|
||||
"perplexity",
|
||||
`Perplexity API error (${response.status}): ${errorText}`,
|
||||
response.status,
|
||||
);
|
||||
const message = config.useResponses
|
||||
? await drainAssistantStream(
|
||||
streamOpenAIResponses(buildPerplexityResponsesModel(config, request), context, {
|
||||
apiKey: config.apiKey,
|
||||
maxTokens: request.max_tokens ?? undefined,
|
||||
temperature: request.temperature ?? undefined,
|
||||
signal: requestSignal,
|
||||
fetch: fetchImpl,
|
||||
extraBody: buildPerplexityExtraBody(request),
|
||||
onSseEvent,
|
||||
}),
|
||||
)
|
||||
: await drainAssistantStream(
|
||||
streamOpenAICompletions(buildPerplexityCompletionsModel(config, request), context, {
|
||||
apiKey: config.apiKey,
|
||||
maxTokens: request.max_tokens ?? undefined,
|
||||
temperature: request.temperature ?? undefined,
|
||||
signal: requestSignal,
|
||||
fetch: fetchImpl,
|
||||
onPayload: payload => applyPerplexityExtraBody(payload, request),
|
||||
onSseEvent,
|
||||
}),
|
||||
);
|
||||
|
||||
if (message.stopReason === "error" || message.stopReason === "aborted") {
|
||||
throwPerplexityStreamError(message);
|
||||
}
|
||||
|
||||
return response.json() as Promise<PerplexityResponse>;
|
||||
return parseStreamedApiResponse(message, metadata);
|
||||
}
|
||||
|
||||
function buildOAuthSources(event: PerplexityOAuthStreamEvent): SearchSource[] {
|
||||
@@ -570,26 +774,49 @@ async function callPerplexityAsk(
|
||||
};
|
||||
}
|
||||
|
||||
function messageContentToText(content: PerplexityMessageOutput["content"]): string {
|
||||
if (!content) return "";
|
||||
if (typeof content === "string") return content;
|
||||
return content.map(chunk => (chunk.type === "text" ? chunk.text : "")).join("");
|
||||
function assistantText(message: AssistantMessage): string {
|
||||
let text = "";
|
||||
for (const block of message.content) {
|
||||
if (block.type === "text") text += block.text;
|
||||
}
|
||||
return text;
|
||||
}
|
||||
|
||||
/** Parse API response into unified SearchResponse */
|
||||
function parseResponse(response: PerplexityResponse): SearchResponse {
|
||||
const messageContent = response.choices[0]?.message?.content ?? null;
|
||||
const answer = messageContentToText(messageContent);
|
||||
function isPerplexitySearchResult(value: unknown): value is PerplexitySearchResult {
|
||||
const record = asRecord(value);
|
||||
return typeof record?.url === "string" && record.url.length > 0;
|
||||
}
|
||||
|
||||
function searchResultsFromMetadata(metadata: PerplexityApiStreamMetadata): PerplexitySearchResult[] {
|
||||
return Array.isArray(metadata.search_results) ? metadata.search_results.filter(isPerplexitySearchResult) : [];
|
||||
}
|
||||
|
||||
function citationUrlsFromMetadata(metadata: PerplexityApiStreamMetadata): string[] {
|
||||
return Array.isArray(metadata.citations)
|
||||
? metadata.citations.filter((url): url is string => typeof url === "string" && url.length > 0)
|
||||
: [];
|
||||
}
|
||||
|
||||
function relatedQuestionsFromMetadata(metadata: PerplexityApiStreamMetadata): string[] {
|
||||
return Array.isArray(metadata.related_questions)
|
||||
? metadata.related_questions.filter(
|
||||
(question): question is string => typeof question === "string" && question.trim().length > 0,
|
||||
)
|
||||
: [];
|
||||
}
|
||||
|
||||
function buildApiSources(metadata: PerplexityApiStreamMetadata): {
|
||||
sources: SearchSource[];
|
||||
citations: SearchCitation[];
|
||||
} {
|
||||
const sources: SearchSource[] = [];
|
||||
const citations: SearchCitation[] = [];
|
||||
|
||||
const citationUrls = response.citations ?? [];
|
||||
const searchResults = response.search_results ?? [];
|
||||
const searchResults = searchResultsFromMetadata(metadata);
|
||||
const citationUrls = citationUrlsFromMetadata(metadata);
|
||||
|
||||
if (citationUrls.length > 0) {
|
||||
for (const url of citationUrls) {
|
||||
const searchResult = searchResults.find(r => r.url === url);
|
||||
const searchResult = searchResults.find(result => result.url === url);
|
||||
sources.push({
|
||||
title: searchResult?.title ?? url,
|
||||
url,
|
||||
@@ -597,10 +824,7 @@ function parseResponse(response: PerplexityResponse): SearchResponse {
|
||||
publishedDate: searchResult?.date ?? undefined,
|
||||
ageSeconds: dateToAgeSeconds(searchResult?.date),
|
||||
});
|
||||
citations.push({
|
||||
url,
|
||||
title: searchResult?.title ?? url,
|
||||
});
|
||||
citations.push({ url, title: searchResult?.title ?? url });
|
||||
}
|
||||
} else {
|
||||
for (const searchResult of searchResults) {
|
||||
@@ -614,7 +838,22 @@ function parseResponse(response: PerplexityResponse): SearchResponse {
|
||||
}
|
||||
}
|
||||
|
||||
const relatedQuestions = (response.related_questions ?? []).filter(q => q.trim().length > 0);
|
||||
return { sources, citations };
|
||||
}
|
||||
|
||||
function usageFromAssistant(usage: Usage): SearchResponse["usage"] | undefined {
|
||||
if (usage.input === 0 && usage.output === 0 && usage.totalTokens === 0) return undefined;
|
||||
return {
|
||||
inputTokens: usage.input,
|
||||
outputTokens: usage.output,
|
||||
totalTokens: usage.totalTokens,
|
||||
};
|
||||
}
|
||||
|
||||
function parseStreamedApiResponse(message: AssistantMessage, metadata: PerplexityApiStreamMetadata): SearchResponse {
|
||||
const { sources, citations } = buildApiSources(metadata);
|
||||
const relatedQuestions = relatedQuestionsFromMetadata(metadata);
|
||||
const answer = assistantText(message);
|
||||
|
||||
return {
|
||||
provider: "perplexity",
|
||||
@@ -622,15 +861,9 @@ function parseResponse(response: PerplexityResponse): SearchResponse {
|
||||
sources,
|
||||
citations: citations.length > 0 ? citations : undefined,
|
||||
relatedQuestions: relatedQuestions.length > 0 ? relatedQuestions : undefined,
|
||||
usage: response.usage
|
||||
? {
|
||||
inputTokens: response.usage.prompt_tokens,
|
||||
outputTokens: response.usage.completion_tokens,
|
||||
totalTokens: response.usage.total_tokens,
|
||||
}
|
||||
: undefined,
|
||||
model: response.model,
|
||||
requestId: response.id,
|
||||
usage: usageFromAssistant(message.usage),
|
||||
model: metadata.model ?? message.model,
|
||||
requestId: metadata.id ?? message.responseId,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -643,35 +876,6 @@ function applySourceLimit(result: SearchResponse, limit?: number): SearchRespons
|
||||
|
||||
/** Execute Perplexity web search */
|
||||
export async function searchPerplexity(params: PerplexitySearchParams): Promise<SearchResponse> {
|
||||
const auth = await findPerplexityAuth(params.authStorage, params.sessionId, params.signal);
|
||||
|
||||
if (auth.type !== "api_key") {
|
||||
// OAuth bearer mode routes the whole authenticated unit (the ask
|
||||
// session/SSE request) through the central auth-retry policy so a 401 or
|
||||
// usage-limit force-refreshes, then rotates to a sibling credential.
|
||||
// Cookie/env/anonymous modes have no rotatable credential — untouched.
|
||||
const askResult =
|
||||
auth.type === "oauth"
|
||||
? await withOAuthAccess(
|
||||
params.authStorage,
|
||||
"perplexity",
|
||||
access => callPerplexityAsk({ type: "oauth", token: access.accessToken }, params),
|
||||
{ sessionId: params.sessionId, signal: params.signal, seed: auth.access },
|
||||
)
|
||||
: await callPerplexityAsk(auth, params);
|
||||
return applySourceLimit(
|
||||
{
|
||||
provider: "perplexity",
|
||||
answer: askResult.answer || undefined,
|
||||
sources: askResult.sources,
|
||||
model: askResult.model,
|
||||
requestId: askResult.requestId,
|
||||
authMode: auth.type === "anonymous" ? "anonymous" : "oauth",
|
||||
},
|
||||
params.num_results,
|
||||
);
|
||||
}
|
||||
|
||||
const systemPrompt = params.system_prompt;
|
||||
const messages: PerplexityRequest["messages"] = [];
|
||||
if (systemPrompt) {
|
||||
@@ -700,10 +904,48 @@ export async function searchPerplexity(params: PerplexitySearchParams): Promise<
|
||||
request.search_recency_filter = params.search_recency_filter;
|
||||
}
|
||||
|
||||
const response = await callPerplexityApi(auth.token, request, params.fetch, params.signal);
|
||||
const result = parseResponse(response);
|
||||
result.authMode = "api_key";
|
||||
return applySourceLimit(result, params.num_results);
|
||||
const authMethods = await getAvailableAuthMethods(params.authStorage, params.sessionId, params.signal);
|
||||
|
||||
for (const auth of authMethods) {
|
||||
if (auth.type === "api_key") {
|
||||
try {
|
||||
const result = await callPerplexityApi(auth, request, params.fetch, params.signal);
|
||||
result.authMode = "api_key";
|
||||
return applySourceLimit(result, params.num_results);
|
||||
} catch (error) {
|
||||
if (params.signal?.aborted) throw error;
|
||||
// Try next method
|
||||
}
|
||||
} else {
|
||||
// Use OAuth/cookies/anonymous path
|
||||
try {
|
||||
const askResult =
|
||||
auth.type === "oauth"
|
||||
? await withOAuthAccess(
|
||||
params.authStorage,
|
||||
"perplexity",
|
||||
access => callPerplexityAsk({ type: "oauth", token: access.accessToken }, params),
|
||||
{ sessionId: params.sessionId, signal: params.signal, seed: auth.access },
|
||||
)
|
||||
: await callPerplexityAsk(auth, params);
|
||||
return applySourceLimit(
|
||||
{
|
||||
provider: "perplexity",
|
||||
answer: askResult.answer || undefined,
|
||||
sources: askResult.sources,
|
||||
model: askResult.model,
|
||||
requestId: askResult.requestId,
|
||||
authMode: auth.type === "anonymous" ? "anonymous" : "oauth",
|
||||
},
|
||||
params.num_results,
|
||||
);
|
||||
} catch {
|
||||
// Try next method
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
throw new SearchProviderError("perplexity", "No authentication method available.", 401);
|
||||
}
|
||||
|
||||
/** Search provider for Perplexity. */
|
||||
@@ -712,7 +954,9 @@ export class PerplexityProvider extends SearchProvider {
|
||||
readonly label = "Perplexity";
|
||||
|
||||
isAvailable(authStorage: AuthStorage): boolean {
|
||||
return !!$env.PERPLEXITY_COOKIES?.trim() || authStorage.hasAuth("perplexity") || !!findApiKey();
|
||||
return (
|
||||
!!$env.PERPLEXITY_COOKIES?.trim() || authStorage.hasAuth("perplexity") || authStorage.hasAuth("openrouter")
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -12,7 +12,7 @@ import {
|
||||
shouldCompact,
|
||||
} from "@oh-my-pi/pi-agent-core/compaction/compaction";
|
||||
import * as ai from "@oh-my-pi/pi-ai";
|
||||
import { encodeTextSignatureV1 } from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
|
||||
import { encodeTextSignatureV1 } from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
import type { AssistantMessage, Model, ProviderPayload, Usage } from "@oh-my-pi/pi-ai/types";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import { buildSessionContext } from "@oh-my-pi/pi-coding-agent/session/session-context";
|
||||
|
||||
@@ -10,7 +10,7 @@ import { BUILTIN_DEFAULTS_PROVIDER_ID, type Rule, ruleCapability } from "@oh-my-
|
||||
import type { LoadContext } from "@oh-my-pi/pi-coding-agent/capability/types";
|
||||
// Register all discovery providers as a side effect.
|
||||
import "@oh-my-pi/pi-coding-agent/discovery";
|
||||
import { TtsrManager } from "@oh-my-pi/pi-coding-agent/export/ttsr";
|
||||
import { TtsrManager, type TtsrMatchContext } from "@oh-my-pi/pi-coding-agent/export/ttsr";
|
||||
|
||||
function ruleProvider() {
|
||||
const cap = getCapability(ruleCapability.id);
|
||||
@@ -97,6 +97,56 @@ describe("builtin-defaults rule provider", () => {
|
||||
).toEqual([]);
|
||||
});
|
||||
|
||||
it("fires ts-no-inline-cast-access on inline cast-and-access but not named-type casts", async () => {
|
||||
const rules = await loadBuiltinRules();
|
||||
const rule = rules.find(r => r.name === "ts-no-inline-cast-access");
|
||||
if (!rule) throw new Error("ts-no-inline-cast-access rule missing");
|
||||
|
||||
const manager = new TtsrManager();
|
||||
expect(manager.addRule(rule)).toBe(true);
|
||||
|
||||
// AST conditions only run on edit/write streams, with the language inferred from the path.
|
||||
const ctx: TtsrMatchContext = { source: "tool", toolName: "edit", filePaths: ["src/foo.ts"] };
|
||||
|
||||
// Inline object-type assertion immediately read — every access form is flagged.
|
||||
const violations = [
|
||||
"const a = (value as { content: unknown }).content;",
|
||||
"const b = (value as { content: unknown })?.content;",
|
||||
'const c = (opts as { enabled: boolean })["enabled"];',
|
||||
"const d = (value as unknown as { content: unknown }).content;",
|
||||
];
|
||||
for (const snippet of violations) {
|
||||
manager.resetBuffer();
|
||||
const matches = await manager.checkAstSnapshot(snippet, ctx);
|
||||
expect(
|
||||
matches.map(r => r.name),
|
||||
snippet,
|
||||
).toEqual(["ts-no-inline-cast-access"]);
|
||||
}
|
||||
|
||||
// A cast to a named type, plain member access, and a bare cast (no read) are all left alone.
|
||||
const allowed = [
|
||||
"const e = (value as Foo).bar;",
|
||||
"const f = obj.content;",
|
||||
"const g = value as { content: unknown };",
|
||||
];
|
||||
for (const snippet of allowed) {
|
||||
manager.resetBuffer();
|
||||
const matches = await manager.checkAstSnapshot(snippet, ctx);
|
||||
expect(matches, snippet).toEqual([]);
|
||||
}
|
||||
|
||||
// Out of scope: the same violation in a non-TS file never reaches the matcher.
|
||||
manager.resetBuffer();
|
||||
expect(
|
||||
await manager.checkAstSnapshot("const h = (value as { content: unknown }).content;", {
|
||||
source: "tool",
|
||||
toolName: "edit",
|
||||
filePaths: ["src/foo.js"],
|
||||
}),
|
||||
).toEqual([]);
|
||||
});
|
||||
|
||||
it("is the lowest-priority rule provider so user/project rules override defaults", () => {
|
||||
const { cap, provider } = ruleProvider();
|
||||
const others = cap.providers.filter(p => p.id !== BUILTIN_DEFAULTS_PROVIDER_ID);
|
||||
|
||||
@@ -93,7 +93,17 @@ describe("browser stealth bootstrap", () => {
|
||||
it("uses documentElement as the iframe container when document.head is unavailable", () => {
|
||||
type NativeWindow = Pick<
|
||||
typeof globalThis,
|
||||
"Function" | "Object" | "setTimeout" | "Math" | "Event" | "Promise" | "Blob" | "Proxy" | "Intl" | "Date"
|
||||
| "Function"
|
||||
| "Object"
|
||||
| "setTimeout"
|
||||
| "Math"
|
||||
| "Event"
|
||||
| "Promise"
|
||||
| "Blob"
|
||||
| "Proxy"
|
||||
| "Intl"
|
||||
| "Date"
|
||||
| "Reflect"
|
||||
>;
|
||||
type FakeContainer = {
|
||||
appendChild(node: FakeIframe): void;
|
||||
@@ -128,6 +138,7 @@ describe("browser stealth bootstrap", () => {
|
||||
Proxy,
|
||||
Intl,
|
||||
Date,
|
||||
Reflect,
|
||||
};
|
||||
const iframe: FakeIframe = { contentWindow: nativeWindow, parentNode: null, style: {} };
|
||||
const createdTags: string[] = [];
|
||||
@@ -149,6 +160,14 @@ describe("browser stealth bootstrap", () => {
|
||||
expect(removed).toEqual([iframe]);
|
||||
expect(iframe.parentNode).toBeNull();
|
||||
});
|
||||
|
||||
it("caches Reflect methods before any stealth scripts run", () => {
|
||||
const script = buildStealthInjectionScriptForTest();
|
||||
expect(script).toContain("const Reflect_get = nativeWindow.Reflect.get");
|
||||
expect(script).toContain("const Reflect_apply = nativeWindow.Reflect.apply");
|
||||
expect(script).not.toContain("Reflect" + ".get(");
|
||||
expect(script).not.toContain("Reflect.apply(");
|
||||
});
|
||||
});
|
||||
describe("browser stealth target setup", () => {
|
||||
it("attempts user-agent override for page-like existing targets and skips non-page worker/browser targets", async () => {
|
||||
|
||||
@@ -3,6 +3,8 @@ import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai";
|
||||
import { PerplexityProvider, searchPerplexity } from "@oh-my-pi/pi-coding-agent/web/search/providers/perplexity";
|
||||
|
||||
const API_URL = "https://api.perplexity.ai/chat/completions";
|
||||
const OPENROUTER_API_URL = "https://openrouter.io/api/v1/chat/completions";
|
||||
const RESPONSES_URL = "https://api.perplexity.ai/v1/responses";
|
||||
|
||||
// API-key path only: getOAuthAccess returns undefined so findPerplexityAuth
|
||||
// falls through to PERPLEXITY_API_KEY (set per-test, restored in afterEach).
|
||||
@@ -10,51 +12,79 @@ const apiKeyAuthStorage = {
|
||||
async getOAuthAccess() {
|
||||
return undefined;
|
||||
},
|
||||
async getApiKey(provider: string) {
|
||||
if (provider === "perplexity") return process.env.PERPLEXITY_API_KEY;
|
||||
if (provider === "openrouter") return process.env.OPENROUTER_API_KEY;
|
||||
return undefined;
|
||||
},
|
||||
hasAuth() {
|
||||
return false;
|
||||
},
|
||||
} as unknown as AuthStorage;
|
||||
|
||||
function mockApi(capture: (body: Record<string, unknown>) => void, response: Record<string, unknown>): FetchImpl {
|
||||
function sseResponse(events: Record<string, unknown>[]): Response {
|
||||
return new Response(`${events.map(event => `data: ${JSON.stringify(event)}\n\n`).join("")}data: [DONE]\n\n`, {
|
||||
status: 200,
|
||||
headers: { "Content-Type": "text/event-stream" },
|
||||
});
|
||||
}
|
||||
|
||||
function mockApi(capture: (body: Record<string, unknown>) => void, events: Record<string, unknown>[]): FetchImpl {
|
||||
return async (input, init) => {
|
||||
const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
|
||||
if (url === API_URL) {
|
||||
capture(JSON.parse(init?.body as string));
|
||||
return new Response(JSON.stringify(response), {
|
||||
status: 200,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
return sseResponse(events);
|
||||
}
|
||||
return new Response("not mocked", { status: 500 });
|
||||
};
|
||||
}
|
||||
|
||||
function baseResponse(extra: Record<string, unknown> = {}) {
|
||||
return {
|
||||
id: "req-1",
|
||||
model: "sonar-pro",
|
||||
created: 0,
|
||||
choices: [{ index: 0, message: { role: "assistant", content: "answer" }, delta: {} }],
|
||||
search_results: [{ title: "T", url: "https://example.com", snippet: "s" }],
|
||||
...extra,
|
||||
};
|
||||
function baseResponse(extra: Record<string, unknown> = {}): Record<string, unknown>[] {
|
||||
return [
|
||||
{
|
||||
id: "req-1",
|
||||
model: "sonar-pro",
|
||||
choices: [{ index: 0, delta: { role: "assistant", content: "answer" }, finish_reason: null }],
|
||||
},
|
||||
{
|
||||
id: "req-1",
|
||||
model: "sonar-pro",
|
||||
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
|
||||
search_results: [{ title: "T", url: "https://example.com", snippet: "s" }],
|
||||
...extra,
|
||||
},
|
||||
{
|
||||
id: "req-1",
|
||||
model: "sonar-pro",
|
||||
choices: [],
|
||||
usage: { prompt_tokens: 1, completion_tokens: 2, total_tokens: 3 },
|
||||
},
|
||||
];
|
||||
}
|
||||
|
||||
describe("Perplexity API-key request shape", () => {
|
||||
const savedKey = process.env.PERPLEXITY_API_KEY;
|
||||
const savedOpenRouterKey = process.env.OPENROUTER_API_KEY;
|
||||
const savedCookies = process.env.PERPLEXITY_COOKIES;
|
||||
const savedResponsesMode = process.env.PI_PERPLEXITY_RESPONSES;
|
||||
|
||||
beforeEach(() => {
|
||||
process.env.PERPLEXITY_API_KEY = "test-key";
|
||||
delete process.env.PERPLEXITY_COOKIES;
|
||||
delete process.env.PI_PERPLEXITY_RESPONSES;
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
if (savedKey === undefined) delete process.env.PERPLEXITY_API_KEY;
|
||||
else process.env.PERPLEXITY_API_KEY = savedKey;
|
||||
if (savedOpenRouterKey === undefined) delete process.env.OPENROUTER_API_KEY;
|
||||
else process.env.OPENROUTER_API_KEY = savedOpenRouterKey;
|
||||
if (savedCookies === undefined) delete process.env.PERPLEXITY_COOKIES;
|
||||
else process.env.PERPLEXITY_COOKIES = savedCookies;
|
||||
if (savedResponsesMode === undefined) delete process.env.PI_PERPLEXITY_RESPONSES;
|
||||
else process.env.PI_PERPLEXITY_RESPONSES = savedResponsesMode;
|
||||
});
|
||||
|
||||
it("requests comprehensive defaults: 20 results, high context, related questions", async () => {
|
||||
@@ -107,6 +137,99 @@ describe("Perplexity API-key request shape", () => {
|
||||
|
||||
expect(response.relatedQuestions).toBeUndefined();
|
||||
});
|
||||
it("falls back to OpenRouter with the selected API-key config after direct Perplexity fails", async () => {
|
||||
process.env.OPENROUTER_API_KEY = "openrouter-test-key";
|
||||
const urls: string[] = [];
|
||||
const bodies: Record<string, unknown>[] = [];
|
||||
const fetchMock: FetchImpl = async (input, init) => {
|
||||
const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
|
||||
urls.push(url);
|
||||
bodies.push(JSON.parse(init?.body as string));
|
||||
if (url === API_URL) return new Response("direct failed", { status: 500 });
|
||||
if (url === OPENROUTER_API_URL) return sseResponse(baseResponse());
|
||||
return new Response("not mocked", { status: 500 });
|
||||
};
|
||||
|
||||
const response = await searchPerplexity({
|
||||
query: "quic vs tcp",
|
||||
authStorage: apiKeyAuthStorage,
|
||||
fetch: fetchMock,
|
||||
});
|
||||
|
||||
expect(urls).toEqual([API_URL, OPENROUTER_API_URL]);
|
||||
expect(bodies[0]?.model).toBe("sonar-pro");
|
||||
expect(bodies[1]?.model).toBe("perplexity/sonar-pro");
|
||||
expect(response.authMode).toBe("api_key");
|
||||
expect(response.answer).toBe("answer");
|
||||
});
|
||||
it("streams the Responses API and captures Perplexity search result events", async () => {
|
||||
process.env.PI_PERPLEXITY_RESPONSES = "1";
|
||||
let body: Record<string, unknown> | undefined;
|
||||
const fetchMock: FetchImpl = async (input, init) => {
|
||||
const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
|
||||
if (url !== RESPONSES_URL) return new Response("not mocked", { status: 500 });
|
||||
body = JSON.parse(init?.body as string);
|
||||
return sseResponse([
|
||||
{
|
||||
type: "response.created",
|
||||
response: { id: "resp-1", status: "in_progress" },
|
||||
},
|
||||
{
|
||||
type: "response.reasoning.search_results",
|
||||
results: [{ title: "Responses source", url: "https://example.org", snippet: "rs" }],
|
||||
},
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
output_index: 0,
|
||||
item: { id: "msg-1", type: "message", role: "assistant", content: [], status: "in_progress" },
|
||||
},
|
||||
{
|
||||
type: "response.output_text.delta",
|
||||
output_index: 0,
|
||||
item_id: "msg-1",
|
||||
content_index: 0,
|
||||
delta: "answer",
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: {
|
||||
id: "msg-1",
|
||||
type: "message",
|
||||
role: "assistant",
|
||||
status: "completed",
|
||||
content: [{ type: "output_text", text: "answer", annotations: [] }],
|
||||
},
|
||||
},
|
||||
{
|
||||
type: "response.completed",
|
||||
response: {
|
||||
id: "resp-1",
|
||||
model: "sonar-pro",
|
||||
status: "completed",
|
||||
output: [],
|
||||
usage: { input_tokens: 1, output_tokens: 2, total_tokens: 3 },
|
||||
},
|
||||
},
|
||||
]);
|
||||
};
|
||||
|
||||
const response = await searchPerplexity({
|
||||
query: "quic vs tcp",
|
||||
authStorage: apiKeyAuthStorage,
|
||||
fetch: fetchMock,
|
||||
});
|
||||
|
||||
expect(body?.num_search_results).toBe(20);
|
||||
expect(body?.max_output_tokens).toBe(8192);
|
||||
expect(response.answer).toBe("answer");
|
||||
expect(response.sources[0]).toMatchObject({
|
||||
title: "Responses source",
|
||||
url: "https://example.org",
|
||||
snippet: "rs",
|
||||
});
|
||||
expect(response.requestId).toBe("resp-1");
|
||||
});
|
||||
});
|
||||
|
||||
const OAUTH_ASK_URL = "https://www.perplexity.ai/rest/sse/perplexity_ask";
|
||||
@@ -117,6 +240,9 @@ const oauthAuthStorage = {
|
||||
async getOAuthAccess() {
|
||||
return { accessToken: "test-oauth-token" };
|
||||
},
|
||||
async getApiKey() {
|
||||
return undefined;
|
||||
},
|
||||
hasAuth() {
|
||||
return true;
|
||||
},
|
||||
@@ -126,6 +252,9 @@ const anonymousAuthStorage = {
|
||||
async getOAuthAccess() {
|
||||
return undefined;
|
||||
},
|
||||
async getApiKey() {
|
||||
return undefined;
|
||||
},
|
||||
hasAuth() {
|
||||
return false;
|
||||
},
|
||||
|
||||
@@ -1,6 +1,9 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Changed
|
||||
|
||||
- Updated OpenRouter request headers to use standard shared headers from the pi-ai package
|
||||
|
||||
## [16.0.5] - 2026-06-17
|
||||
|
||||
@@ -129,4 +132,4 @@
|
||||
- Fixed `rememberBatch(..., { extract: true })` to run background fact extraction for batch uploads (including per-item `extract` flags) so extracted facts are generated and recallable after extraction
|
||||
- Fixed `extract: true` fact extraction to continue safely when no LLM is configured by turning extraction failures into no-op background tasks
|
||||
- Fixed configured LLM fact extraction by using temperature 0 so re-ingesting the same text is deterministic and avoids near-duplicate extractions
|
||||
- Fixed `remember(..., { extract: true })` silently dropping the flag: it now schedules the LLM fact extractor (`extractFactsSafe`) over the stored content and persists the extracted facts so they become recallable. Previously the LLM extractor had no production callers and `extract` was dead.
|
||||
- Fixed `remember(..., { extract: true })` silently dropping the flag: it now schedules the LLM fact extractor (`extractFactsSafe`) over the stored content and persists the extracted facts so they become recallable. Previously the LLM extractor had no production callers and `extract` was dead.
|
||||
@@ -1,5 +1,5 @@
|
||||
import { mkdirSync } from "node:fs";
|
||||
import { type ApiKey, ProviderHttpError, withAuth } from "@oh-my-pi/pi-ai";
|
||||
import { type ApiKey, getOpenRouterHeaders, ProviderHttpError, withAuth } from "@oh-my-pi/pi-ai";
|
||||
import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts";
|
||||
import {
|
||||
$env,
|
||||
@@ -11,7 +11,6 @@ import {
|
||||
} from "@oh-my-pi/pi-utils";
|
||||
import type { EmbeddingModel } from "fastembed";
|
||||
import { LRUCache } from "lru-cache/raw";
|
||||
import packageJson from "../../package.json" with { type: "json" };
|
||||
import { loadFastembed } from "./fastembed-runtime";
|
||||
import {
|
||||
type EmbeddingOutput,
|
||||
@@ -268,10 +267,7 @@ async function embedApi(texts: readonly string[]): Promise<EmbeddingMatrix | nul
|
||||
const response = await withAuth(apiKey, async key => {
|
||||
const headers: Record<string, string> = {
|
||||
"Content-Type": "application/json",
|
||||
"User-Agent": `Oh-My-Pi/${packageJson.version}`,
|
||||
"HTTP-Referer": "https://omp.sh/",
|
||||
"X-OpenRouter-Title": "Oh-My-Pi",
|
||||
"X-OpenRouter-Categories": "cli-agent",
|
||||
...getOpenRouterHeaders(),
|
||||
};
|
||||
if (key !== "") {
|
||||
headers.Authorization = `Bearer ${key}`;
|
||||
|
||||
@@ -0,0 +1,546 @@
|
||||
import { describe, expect, test } from "bun:test";
|
||||
import { IndentationText, Project } from "ts-morph";
|
||||
import { inlineFile, type Options } from "./inline-functions";
|
||||
|
||||
function opts(overrides: Partial<Options> = {}): Options {
|
||||
return {
|
||||
maxStatements: overrides.maxStatements ?? 3,
|
||||
nameFilter: overrides.nameFilter,
|
||||
verbose: false,
|
||||
strictEffects: overrides.strictEffects ?? false,
|
||||
};
|
||||
}
|
||||
|
||||
/** Run the inliner on an in-memory source and return { text, inlined names }. */
|
||||
function run(src: string, overrides: Partial<Options> = {}): { text: string; inlined: string[] } {
|
||||
const project = new Project({
|
||||
useInMemoryFileSystem: true,
|
||||
manipulationSettings: { indentationText: IndentationText.Tab },
|
||||
});
|
||||
const sf = project.createSourceFile("input.ts", src);
|
||||
const inlined = inlineFile(sf, opts(overrides));
|
||||
return { text: sf.getFullText(), inlined };
|
||||
}
|
||||
|
||||
/** Collapse whitespace so assertions ignore the inliner's pre-format indentation. */
|
||||
function norm(s: string): string {
|
||||
return s.replace(/\s+/g, " ").trim();
|
||||
}
|
||||
|
||||
describe("inline-functions: guard inversion", () => {
|
||||
test("single `!==` guard becomes a positive `===` wrapper with params substituted", () => {
|
||||
const { text, inlined } = run(`
|
||||
function handlePart(currentItem: Item | null, rawEvent: Record<string, unknown>): void {
|
||||
if (currentItem?.type !== "reasoning") return;
|
||||
appendPart(currentItem, (rawEvent as { part: P }).part);
|
||||
}
|
||||
function dispatch(runtime: Runtime, rawEvent: Record<string, unknown>): void {
|
||||
handlePart(runtime.currentItem, rawEvent);
|
||||
}
|
||||
`);
|
||||
expect(inlined).toEqual(["handlePart"]);
|
||||
expect(text).not.toContain("function handlePart");
|
||||
const n = norm(text);
|
||||
expect(n).toContain('if (runtime.currentItem?.type === "reasoning") {');
|
||||
expect(n).toContain("appendPart(runtime.currentItem, (rawEvent as { part: P }).part);");
|
||||
});
|
||||
|
||||
test("`||` guard with two comparisons becomes `&&` via De Morgan", () => {
|
||||
const { text } = run(`
|
||||
function handleDelta(currentItem: Item | null, currentBlock: Block | null, rawEvent: Record<string, unknown>): void {
|
||||
if (currentItem?.type !== "reasoning" || currentBlock?.type !== "thinking") return;
|
||||
const delta = (rawEvent as { delta?: string }).delta || "";
|
||||
appendDelta(currentItem, currentBlock, delta);
|
||||
}
|
||||
function dispatch(runtime: Runtime, rawEvent: Record<string, unknown>): void {
|
||||
handleDelta(runtime.currentItem, runtime.currentBlock, rawEvent);
|
||||
}
|
||||
`);
|
||||
const n = norm(text);
|
||||
expect(n).toContain(
|
||||
'if (runtime.currentItem?.type === "reasoning" && runtime.currentBlock?.type === "thinking") {',
|
||||
);
|
||||
expect(n).toContain('const delta = (rawEvent as { delta?: string }).delta || "";');
|
||||
expect(n).toContain("appendDelta(runtime.currentItem, runtime.currentBlock, delta);");
|
||||
});
|
||||
|
||||
test("comparator flips: `=== null` guard becomes `!== null`", () => {
|
||||
const { text } = run(`
|
||||
function h(x: string | null): void {
|
||||
if (x === null) return;
|
||||
use(x);
|
||||
}
|
||||
function call(value: string | null): void {
|
||||
h(value);
|
||||
}
|
||||
`);
|
||||
expect(norm(text)).toContain("if (value !== null) {");
|
||||
});
|
||||
|
||||
test("multiple guards combine with `&&`, wrapping disjunctions", () => {
|
||||
const { text } = run(`
|
||||
function h(a: A, b: B): void {
|
||||
if (a.x !== 1) return;
|
||||
if (b.y === 2 || b.z === 3) return;
|
||||
done(a, b);
|
||||
}
|
||||
function call(p: A, q: B): void {
|
||||
h(p, q);
|
||||
}
|
||||
`);
|
||||
// !(a.x !== 1) -> p.x === 1 ; !(b.y === 2 || b.z === 3) -> q.y !== 2 && q.z !== 3.
|
||||
// The disjunction-negation is an && itself, so it nests flat with no parens.
|
||||
expect(norm(text)).toContain("if (p.x === 1 && q.y !== 2 && q.z !== 3) {");
|
||||
});
|
||||
|
||||
test("non-comparison guard uses a `!(...)` fallback", () => {
|
||||
const { text } = run(`
|
||||
function h(x: string): void {
|
||||
if (isBad(x)) return;
|
||||
use(x);
|
||||
}
|
||||
function call(v: string): void {
|
||||
h(v);
|
||||
}
|
||||
`);
|
||||
expect(norm(text)).toContain("if (!(isBad(v))) {");
|
||||
});
|
||||
});
|
||||
|
||||
describe("inline-functions: argument handling", () => {
|
||||
test("guardless helper splices its statements straight into the caller", () => {
|
||||
const { text, inlined } = run(`
|
||||
function h(a: number, b: string): void {
|
||||
first(a);
|
||||
second(b);
|
||||
}
|
||||
function call(x: string): void {
|
||||
h(1, x);
|
||||
}
|
||||
`);
|
||||
expect(inlined).toEqual(["h"]);
|
||||
const n = norm(text);
|
||||
expect(n).not.toContain("function h");
|
||||
expect(n).toContain("function call(x: string): void { first(1); second(x); }");
|
||||
});
|
||||
|
||||
test("impure argument is hoisted to a const evaluated unconditionally before the guard", () => {
|
||||
const { text } = run(`
|
||||
function h(x: number): void {
|
||||
if (bad(x)) return;
|
||||
use(x);
|
||||
}
|
||||
function call(): void {
|
||||
h(make());
|
||||
}
|
||||
`);
|
||||
const n = norm(text);
|
||||
// make() has a side effect and is used twice (guard + tail) -> hoist once, unconditionally.
|
||||
expect(n).toContain("const __inl_x = make();");
|
||||
expect(n.indexOf("const __inl_x = make();")).toBeLessThan(n.indexOf("if (!(bad(__inl_x)))"));
|
||||
expect(n).toContain("use(__inl_x);");
|
||||
// the call expression must appear exactly once (single evaluation)
|
||||
expect(n.match(/make\(\)/g)?.length).toBe(1);
|
||||
});
|
||||
|
||||
test("pure but non-trivial argument used twice is hoisted, not duplicated", () => {
|
||||
const { text } = run(`
|
||||
function h(x: number): void {
|
||||
if (x > 0) return;
|
||||
use(x);
|
||||
}
|
||||
function call(a: number, b: number): void {
|
||||
h(a + b);
|
||||
}
|
||||
`);
|
||||
const n = norm(text);
|
||||
expect(n).toContain("const __inl_x = a + b;");
|
||||
expect(n.match(/a \+ b/g)?.length).toBe(1);
|
||||
});
|
||||
|
||||
test("default mode treats a member chain as pure and inlines it (the pretty result)", () => {
|
||||
const { text } = run(`
|
||||
function h(x: number): void {
|
||||
if (x > 0) return;
|
||||
use(x);
|
||||
}
|
||||
function call(runtime: Runtime): void {
|
||||
h(runtime.count);
|
||||
}
|
||||
`);
|
||||
const n = norm(text);
|
||||
expect(n).not.toContain("const __inl_x");
|
||||
expect(n).toContain("if (runtime.count <= 0) {");
|
||||
expect(n).toContain("use(runtime.count);");
|
||||
});
|
||||
|
||||
test("low-precedence inline argument gets wrapped in parens at member-access use sites", () => {
|
||||
const { text } = run(`
|
||||
function h(x: number): void {
|
||||
take(x.toFixed());
|
||||
}
|
||||
function call(a: number, b: number): void {
|
||||
h(a ? b : a);
|
||||
}
|
||||
`);
|
||||
// conditional arg used once -> inlined, parenthesized because it feeds `.toFixed()`
|
||||
expect(norm(text)).toContain("take((a ? b : a).toFixed());");
|
||||
});
|
||||
|
||||
test("unused impure argument is still evaluated for its side effects", () => {
|
||||
const { text } = run(`
|
||||
function h(unusedArg: number): void {
|
||||
sideEffectFree();
|
||||
}
|
||||
function call(): void {
|
||||
h(makeNoise());
|
||||
}
|
||||
`);
|
||||
const n = norm(text);
|
||||
expect(n).toContain("makeNoise();");
|
||||
expect(n).toContain("sideEffectFree();");
|
||||
});
|
||||
});
|
||||
|
||||
describe("inline-functions: --strict-effects", () => {
|
||||
test("hoists a member-chain argument so it evaluates eagerly, exactly once", () => {
|
||||
const { text } = run(
|
||||
`
|
||||
function h(x: number): void {
|
||||
if (x > 0) return;
|
||||
use(x);
|
||||
}
|
||||
function call(runtime: Runtime): void {
|
||||
h(runtime.count);
|
||||
}
|
||||
`,
|
||||
{ strictEffects: true },
|
||||
);
|
||||
const n = norm(text);
|
||||
expect(n).toContain("const __inl_x = runtime.count;");
|
||||
expect(n).toContain("if (__inl_x <= 0) {");
|
||||
expect(n).toContain("use(__inl_x);");
|
||||
expect(n.match(/runtime\.count/g)?.length).toBe(1);
|
||||
});
|
||||
|
||||
test("snapshots every used argument left-to-right before the body", () => {
|
||||
const { text } = run(
|
||||
`
|
||||
function h(a: number, b: number): void {
|
||||
use(a);
|
||||
use2(b);
|
||||
}
|
||||
function call(x: number): void {
|
||||
h(x, (x = 2));
|
||||
}
|
||||
`,
|
||||
{ strictEffects: true },
|
||||
);
|
||||
const n = norm(text);
|
||||
// param a is read BEFORE the second argument mutates x.
|
||||
expect(n).toContain("const __inl_a = x;");
|
||||
expect(n).toContain("const __inl_b = (x = 2);");
|
||||
expect(n.indexOf("const __inl_a = x;")).toBeLessThan(n.indexOf("const __inl_b = (x = 2);"));
|
||||
expect(n).toContain("use(__inl_a);");
|
||||
expect(n).toContain("use2(__inl_b);");
|
||||
});
|
||||
});
|
||||
|
||||
describe("inline-functions: multiple call sites", () => {
|
||||
test("every call site is inlined and the declaration removed", () => {
|
||||
const { text, inlined } = run(`
|
||||
function h(item: Item | null): void {
|
||||
if (item?.type !== "x") return;
|
||||
touch(item);
|
||||
}
|
||||
function a(r: Runtime): void { h(r.currentItem); }
|
||||
function b(r: Runtime): void { h(r.other); }
|
||||
`);
|
||||
expect(inlined).toEqual(["h"]);
|
||||
const n = norm(text);
|
||||
expect(n).not.toContain("function h(");
|
||||
expect(n).toContain('if (r.currentItem?.type === "x") {');
|
||||
expect(n).toContain('if (r.other?.type === "x") {');
|
||||
});
|
||||
});
|
||||
|
||||
describe("inline-functions: safety skips", () => {
|
||||
const cases: Array<{ name: string; src: string }> = [
|
||||
{
|
||||
name: "exported helper",
|
||||
src: `export function h(x: number): void { if (x > 0) return; use(x); }
|
||||
function call(): void { h(1); }`,
|
||||
},
|
||||
{
|
||||
name: "recursive helper",
|
||||
src: `function h(x: number): void { if (x > 0) return; h(x - 1); }
|
||||
function call(): void { h(3); }`,
|
||||
},
|
||||
{
|
||||
name: "call result is used",
|
||||
src: `function h(x: number): boolean { if (x > 0) return false; return true; }
|
||||
function call(): void { const r = h(1); use(r); }`,
|
||||
},
|
||||
{
|
||||
name: "parameter is written",
|
||||
src: `function h(x: number): void { x = x + 1; use(x); }
|
||||
function call(): void { h(1); }`,
|
||||
},
|
||||
{
|
||||
name: "tail contains a return",
|
||||
src: `function h(x: number): void { use(x); if (x > 0) return; use2(x); }
|
||||
function call(): void { h(1); }`,
|
||||
},
|
||||
{
|
||||
name: "async helper",
|
||||
src: `async function h(x: number): Promise<void> { use(x); }
|
||||
function call(): void { void h(1); }`,
|
||||
},
|
||||
{
|
||||
name: "default parameter value",
|
||||
src: `function h(x: number = 5): void { use(x); }
|
||||
function call(): void { h(); }`,
|
||||
},
|
||||
{
|
||||
name: "tail too large for --max-statements",
|
||||
src: `function h(x: number): void { if (x > 0) return; one(x); two(x); three(x); four(x); }
|
||||
function call(): void { h(1); }`,
|
||||
},
|
||||
{
|
||||
name: "function-scoped var in tail",
|
||||
src: `function h(x: number): void { if (x > 0) return; var y = x; use(y); }
|
||||
function call(): void { h(1); }`,
|
||||
},
|
||||
{
|
||||
name: "for (var ...) in tail",
|
||||
src: `function h(x: number): void { for (var i = 0; i < x; i++) use(i); }
|
||||
function call(): void { h(2); }`,
|
||||
},
|
||||
{
|
||||
name: "tail declares a local function",
|
||||
src: `function h(x: number): void { if (x > 0) return; function inner() { return x; } use(inner()); }
|
||||
function call(): void { h(1); }`,
|
||||
},
|
||||
];
|
||||
|
||||
for (const c of cases) {
|
||||
test(`skips: ${c.name}`, () => {
|
||||
const { inlined } = run(c.src);
|
||||
expect(inlined).toEqual([]);
|
||||
});
|
||||
}
|
||||
|
||||
test("skips when a free body identifier is shadowed at a call site", () => {
|
||||
const { inlined, text } = run(`
|
||||
function h(x: number): void { if (x > 0) return; helper(x); }
|
||||
function call(): void {
|
||||
const helper = (n: number) => n;
|
||||
h(1);
|
||||
}
|
||||
`);
|
||||
expect(inlined).toEqual([]);
|
||||
expect(text).toContain("function h(");
|
||||
});
|
||||
|
||||
test("inlines when the same name lives only at module scope (no shadow)", () => {
|
||||
const { inlined } = run(`
|
||||
function helper(n: number): number { return n; }
|
||||
function h(x: number): void { if (x > 0) return; helper(x); }
|
||||
function call(): void { h(1); }
|
||||
`);
|
||||
expect(inlined).toEqual(["h"]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("inline-functions: guardless local-name collision", () => {
|
||||
test("renames a tail local that would redeclare a name already live in the target block", () => {
|
||||
const { text } = run(`
|
||||
function h(x: number): void {
|
||||
const delta = x + 1;
|
||||
use(delta);
|
||||
}
|
||||
function call(value: number): void {
|
||||
const delta = compute();
|
||||
h(value);
|
||||
log(delta);
|
||||
}
|
||||
`);
|
||||
const n = norm(text);
|
||||
// caller's `delta` is untouched; the inlined local is renamed.
|
||||
expect(n).toContain("const delta = compute();");
|
||||
expect(n).toContain("const delta_2 = value + 1;");
|
||||
expect(n).toContain("use(delta_2);");
|
||||
expect(n).toContain("log(delta);");
|
||||
});
|
||||
});
|
||||
|
||||
describe("inline-functions: type soundness", () => {
|
||||
test("a fully-typed inline introduces no new diagnostics", () => {
|
||||
const project = new Project({ useInMemoryFileSystem: true, compilerOptions: { strict: true } });
|
||||
const sf = project.createSourceFile(
|
||||
"typed.ts",
|
||||
`
|
||||
interface Block { type: "thinking" | "text"; }
|
||||
interface Item { type: "reasoning" | "message"; }
|
||||
interface Runtime { currentItem: Item | null; currentBlock: Block | null; }
|
||||
declare function appendDelta(item: Item, block: Block, delta: string, n: number): void;
|
||||
|
||||
function handleDelta(
|
||||
currentItem: Item | null,
|
||||
currentBlock: Block | null,
|
||||
rawEvent: Record<string, unknown>,
|
||||
n: number,
|
||||
): void {
|
||||
if (currentItem?.type !== "reasoning" || currentBlock?.type !== "thinking") return;
|
||||
const delta = (rawEvent as { delta?: string }).delta || "";
|
||||
appendDelta(currentItem, currentBlock, delta, n);
|
||||
}
|
||||
|
||||
export function dispatch(runtime: Runtime, rawEvent: Record<string, unknown>): void {
|
||||
handleDelta(runtime.currentItem, runtime.currentBlock, rawEvent, 0);
|
||||
}
|
||||
`,
|
||||
);
|
||||
expect(sf.getPreEmitDiagnostics()).toHaveLength(0);
|
||||
const inlined = inlineFile(sf, opts());
|
||||
expect(inlined).toEqual(["handleDelta"]);
|
||||
expect(sf.getPreEmitDiagnostics()).toHaveLength(0);
|
||||
expect(norm(sf.getFullText())).toContain(
|
||||
'if (runtime.currentItem?.type === "reasoning" && runtime.currentBlock?.type === "thinking") {',
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
describe("inline-functions: formatting safety", () => {
|
||||
test("wraps the tail exactly one indent level deeper than the inverted guard", () => {
|
||||
const { text } = run(
|
||||
'function dispatch(r: Runtime): void {\n\thandle(r.item);\n}\n' +
|
||||
'function handle(item: Item | null): void {\n\tif (item?.type !== "x") return;\n\ttouch(item);\n}\n',
|
||||
);
|
||||
// `if` sits at the function-body indent (one tab); its body is one deeper.
|
||||
expect(text).toContain('\tif (r.item?.type === "x") {\n\t\ttouch(r.item);\n\t}');
|
||||
});
|
||||
|
||||
test("never indents the interior of a multi-line template literal in the tail", () => {
|
||||
const { text } = run(
|
||||
"function h(x: number): void {\n\tif (x > 0) return;\n\tconst s = `line1\nline2`;\n\tuse(s);\n}\n" +
|
||||
"function call(v: number): void {\n\th(v);\n}\n",
|
||||
);
|
||||
// `line2` must stay at column 0 — no tab injected into the string contents.
|
||||
expect(text).toContain("`line1\nline2`");
|
||||
});
|
||||
});
|
||||
|
||||
describe("inline-functions: hoisted temp names", () => {
|
||||
test("distinct temp names for multiple impure calls in the same block", () => {
|
||||
const { text } = run(`
|
||||
function h(x: number): void { if (bad(x)) return; use(x); }
|
||||
function call(): void {
|
||||
h(make1());
|
||||
h(make2());
|
||||
}
|
||||
`);
|
||||
const n = norm(text);
|
||||
expect(n).toContain("const __inl_x = make1();");
|
||||
expect(n).toContain("const __inl_x_2 = make2();");
|
||||
});
|
||||
|
||||
test("temp names may be reused across separate, non-overlapping blocks", () => {
|
||||
const { text } = run(`
|
||||
function h(x: number): void { if (bad(x)) return; use(x); }
|
||||
function call(flag: boolean): void {
|
||||
if (flag) {
|
||||
h(make1());
|
||||
} else {
|
||||
h(make2());
|
||||
}
|
||||
}
|
||||
`);
|
||||
const n = norm(text);
|
||||
expect((n.match(/const __inl_x =/g) ?? []).length).toBe(2);
|
||||
expect(n).not.toContain("__inl_x_2");
|
||||
});
|
||||
|
||||
test("switch cases share one scope, so temp names stay distinct across cases", () => {
|
||||
const { text } = run(`
|
||||
function h(x: number): void { if (bad(x)) return; use(x); }
|
||||
function call(k: number): void {
|
||||
switch (k) {
|
||||
case 1:
|
||||
h(make1());
|
||||
break;
|
||||
case 2:
|
||||
h(make2());
|
||||
break;
|
||||
}
|
||||
}
|
||||
`);
|
||||
const n = norm(text);
|
||||
expect(n).toContain("const __inl_x = make1();");
|
||||
expect(n).toContain("const __inl_x_2 = make2();");
|
||||
});
|
||||
});
|
||||
|
||||
describe("inline-functions: object shorthand safety", () => {
|
||||
test("expands shorthand when a substituted parameter is not a bare identifier", () => {
|
||||
const { text } = run(`
|
||||
function warn(sourceType: string): void {
|
||||
logger.warn("msg", { sourceType });
|
||||
}
|
||||
function call(source: { type: string }): void {
|
||||
warn(source.type);
|
||||
}
|
||||
`);
|
||||
expect(norm(text)).toContain('logger.warn("msg", { sourceType: source.type });');
|
||||
});
|
||||
|
||||
test("expands shorthand when a colliding tail local is renamed", () => {
|
||||
const { text } = run(`
|
||||
function h(): void {
|
||||
const id = compute();
|
||||
use({ id });
|
||||
}
|
||||
function call(): void {
|
||||
const id = 1;
|
||||
h();
|
||||
log(id);
|
||||
}
|
||||
`);
|
||||
const n = norm(text);
|
||||
expect(n).toContain("const id_2 = compute();");
|
||||
expect(n).toContain("use({ id: id_2 });");
|
||||
});
|
||||
|
||||
test("does not rename a tail local that only matches a property name in the block", () => {
|
||||
const { text } = run(`
|
||||
function h(): void {
|
||||
const index = compute();
|
||||
use(index);
|
||||
}
|
||||
function call(obj: { index: number }): void {
|
||||
read(obj.index);
|
||||
h();
|
||||
}
|
||||
`);
|
||||
const n = norm(text);
|
||||
expect(n).toContain("const index = compute();");
|
||||
expect(n).not.toContain("index_2");
|
||||
});
|
||||
|
||||
test("does not rename a tail local that only matches a type name in the block", () => {
|
||||
const { text } = run(`
|
||||
interface Thing { id: number }
|
||||
function h(): void {
|
||||
const Thing = makeThing();
|
||||
use(Thing);
|
||||
}
|
||||
function call(): void {
|
||||
const x: Thing = { id: 1 };
|
||||
h();
|
||||
}
|
||||
`);
|
||||
const n = norm(text);
|
||||
expect(n).toContain("const Thing = makeThing();");
|
||||
expect(n).not.toContain("Thing_2");
|
||||
});
|
||||
});
|
||||
Executable
+1051
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user