refactor(catalog): baked thinking metadata into buildModel pipeline

- Replaced minLevel/maxLevel range with explicit efforts array plus baked effortMap/supportsDisplay wire facts.
- Removed runtime enrichment layer and modelOmitsReasoningEffort; providers now read baked fields.
- Fixed dotted Opus 4.7/4.8 ids missing adaptive display via classifier-based predicates (#1373).
- Bumped model cache schema to v4 to invalidate pre-efforts rows.
This commit is contained in:
can1357
2026-06-10 07:20:58 +02:00
parent 8c04c5576a
commit a25d521cab
46 changed files with 9618 additions and 3699 deletions
+5 -4
View File
@@ -539,10 +539,11 @@ function effortFromThinkingLevel(level: ThinkingLevel): Effort {
* - Explicit effort → respect user choice → clamped per model.
*
* The clamp routes through `clampThinkingLevelForModel`, which returns
* `undefined` for models with `compat.supportsReasoningEffort: false`
* (e.g. `xai-oauth/grok-build`). That `undefined` then flows through to the
* openai-responses mapper where `modelOmitsReasoningEffort` short-circuits
* the wire param — no `requireSupportedEffort` throw.
* `undefined` for reasoning models without a thinking config — the build-time
* encoding of `compat.supportsReasoningEffort: false` (e.g.
* `xai-oauth/grok-build`). That `undefined` then flows through to the
* openai-responses mapper, which omits the wire param — no
* `requireSupportedEffort` throw.
*/
function resolveCompactionEffort(model: Model, level: ThinkingLevel | undefined): Effort | undefined {
if (level === ThinkingLevel.Off) return undefined;
+2
View File
@@ -15,6 +15,7 @@
- Auth storage no longer issues per-boot no-op writes: the schema-version row is only rewritten when the recorded version actually changes, and the credential identity-key backfill skips rows whose derived identity is null — reopening a current-schema database now performs zero write transactions
- Plain provider env-var names moved to the catalog table: registry defs dropped their 48 `envKeys` literals (including the pure `$pickenv` pickers for `huggingface`/`qwen-portal`/`xai-oauth`), `getEnvApiKey` now derives those fallbacks from `CATALOG_PROVIDERS[].envVars`, and `envKeys` remains only for computed resolvers (Anthropic Foundry, Vertex ADC, Bedrock credential chains) and non-catalog providers (`kagi`, `tavily`, `parallel`, `perplexity`)
- Protocol handlers are now pure `model.compat` readers — the per-request `resolve*Compat`/`detect*Compat` calls (anthropic ×11, responses ×3, completions wrappers), inline `strictResponsesPairing` host detection, the OpenCode `reasoning_content` mutation block, and all `resolvedBaseUrl` threading are gone. Compat is materialized once at model build time (`@oh-my-pi/pi-catalog` `buildModel`); the OpenCode thinking-mode quirk is a precomputed `compat.whenThinking` pointer swap, and request-time base-URL overrides only feed the HTTP client. Behavior is unchanged (the Anthropic `supportsLongCacheRetention` official-endpoint gate is folded into detection).
- Providers now read baked thinking/wire metadata instead of re-parsing model ids per request: the Anthropic handler gates sampling params on `model.compat.supportsSamplingParams` and adaptive `display` on `model.thinking.supportsDisplay` (Bedrock too), adaptive effort tiers come from the baked `thinking.effortMap`, the Google `thinkingLevel` map is static, and effort-dial-less reasoners (`thinking: undefined`, e.g. `xai-oauth/grok-build`) short-circuit `resolveOpenAiReasoningEffort` without the removed `modelOmitsReasoningEffort` predicate.
### Fixed
@@ -26,6 +27,7 @@
- Fixed in-stream Anthropic SSE `error` events being thrown as raw JSON envelopes; the structured `error.type`/`message` is parsed out, keeping retry classification on the typed token instead of accidental regex hits.
- Fixed transparent-reconnect tolerance duplicating content behind replaying proxies: after a duplicate `message_start`, replayed `content_block_start` events for already-closed indexes are now consumed silently instead of appending duplicate text/tool calls.
- Fixed the Anthropic gateway accepting malformed known-type content blocks (e.g. `{type:"text", text:123}`) through the unknown-block catch-all, corrupting history and surfacing later as an opaque TypeError — they now fail validation with a clean 400. The gateway's encode stream also emits `ping` keepalives every 15s and a complete `message_start`/`message_delta`/`message_stop` envelope when the inner stream ends without a terminal event, so strict clients no longer classify slow or empty streams as protocol errors.
- Fixed dotted-version Claude ids (`claude-opus-4.7`/`4.8` on GitHub Copilot, Vercel AI Gateway, Zenmux) missing adaptive thinking `display` support — streamed reasoning stayed hidden on those entries because the display predicate only matched dash-form ids (same failure class as #1373).
- Fixed the Mistral `requiresThinkingAsText` replay path calling `.unshift()` on string assistant content — an unconditional TypeError that failed any same-model history turn carrying both thinking and text.
- Fixed the Responses gateway stripping `encrypted_content` from inbound reasoning items (strip-mode schema), which broke codex-style stateless replay; the schema is now loose, restoring the symmetry the outbound encoder already preserved. Composite internal `callId|itemId` ids are also split before hitting the wire so third-party clients that validate `call_id` charsets no longer reject them.
- Ported the shared unfinished-tool-call sweep to the codex `response.completed` handler, so a lost `output_item.done` can no longer persist a tool call with stale `{}` arguments and transient parser fields into session history.
+1 -2
View File
@@ -8,7 +8,6 @@
*/
import type { Effort } from "@oh-my-pi/pi-catalog/effort";
import { supportsAdaptiveThinkingDisplay } from "@oh-my-pi/pi-catalog/identity";
import { mapEffortToAnthropicAdaptiveEffort, requireSupportedEffort } from "@oh-my-pi/pi-catalog/model-thinking";
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
import { $env, $flag, extractHttpStatusFromError, fetchWithRetry } from "@oh-my-pi/pi-utils";
@@ -819,7 +818,7 @@ function buildAdditionalModelRequestFields(
// runs (issue #1373). Opt back into "summarized" by default on models that
// accept the field.
const adaptive: { type: "adaptive"; display?: BedrockThinkingDisplay } = { type: "adaptive" };
if (supportsAdaptiveThinkingDisplay(model.id)) {
if (model.thinking?.supportsDisplay) {
adaptive.display = options.thinkingDisplay ?? "summarized";
}
return {
+4 -5
View File
@@ -3,8 +3,7 @@ import * as fs from "node:fs";
import { scheduler } from "node:timers/promises";
import * as tls from "node:tls";
import { isOfficialAnthropicApiUrl } from "@oh-my-pi/pi-catalog/compat/anthropic";
import { supportsAdaptiveThinkingDisplay } from "@oh-my-pi/pi-catalog/identity";
import { hasOpus47ApiRestrictions, mapEffortToAnthropicAdaptiveEffort } from "@oh-my-pi/pi-catalog/model-thinking";
import { mapEffortToAnthropicAdaptiveEffort } from "@oh-my-pi/pi-catalog/model-thinking";
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
import { isAnthropicOAuthToken } from "@oh-my-pi/pi-catalog/utils";
import { parseGitHubCopilotApiKey } from "@oh-my-pi/pi-catalog/wire/github-copilot";
@@ -2279,7 +2278,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
claudeCodeSessionId,
} = args;
const compat = model.compat;
const needsInterleavedBeta = interleavedThinking && !supportsAdaptiveThinkingDisplay(model.id);
const needsInterleavedBeta = interleavedThinking && !model.thinking?.supportsDisplay;
const needsFineGrainedToolStreamingBeta = hasTools && !compat.supportsEagerToolInputStreaming;
const oauthToken = isOAuth ?? isAnthropicOAuthToken(apiKey);
const baseUrl = resolveAnthropicBaseUrl(model, apiKey);
@@ -2754,7 +2753,7 @@ function buildParams(
// callers that rely on it. The `display` field is gated strictly on model
// support: Opus 4.6 / Sonnet 4.6+ reject it with a 400, so an explicit
// `thinkingDisplay` MUST NOT force it onto a model that can't accept it.
if (supportsAdaptiveThinkingDisplay(model.id)) {
if (model.thinking?.supportsDisplay) {
adaptive.display = options.thinkingDisplay ?? "summarized";
}
thinking = adaptive;
@@ -2817,7 +2816,7 @@ function buildParams(
// Opus 4.7+ and Fable/Mythos 5 reject non-default sampling parameters with 400 error.
const thinkingType = params.thinking?.type;
const allowSamplingParams =
!hasOpus47ApiRestrictions(model.id) && (thinkingType === undefined || thinkingType === "disabled");
model.compat.supportsSamplingParams && (thinkingType === undefined || thinkingType === "disabled");
if (allowSamplingParams && options?.temperature !== undefined) {
params.temperature = options.temperature;
}
+12 -12
View File
@@ -3,7 +3,6 @@ import { isVertexExpressOpenAIUrl, isVertexRawPredictUrl } from "@oh-my-pi/pi-ca
import {
mapEffortToAnthropicAdaptiveEffort,
mapEffortToGoogleThinkingLevel,
modelOmitsReasoningEffort,
requireSupportedEffort,
} from "@oh-my-pi/pi-catalog/model-thinking";
import { CATALOG_PROVIDERS, type ProviderCatalogEntry } from "@oh-my-pi/pi-catalog/provider-models";
@@ -654,14 +653,15 @@ function resolveOpenAiReasoningEffort<TApi extends Api>(
): Effort | undefined {
const reasoning = options?.reasoning;
if (!reasoning || !model.reasoning) return undefined;
// Models with compat.supportsReasoningEffort: false reason natively but
// reject the wire effort param. The wire-side omitReasoningEffort gate
// (providers/xai-responses.ts:78) is the actual strip; returning
// undefined here avoids a redundant requireSupportedEffort throw that
// would defeat the gate and surface a confusing
// "Compaction failed: Thinking effort high is not supported by..." to
// the user.
if (modelOmitsReasoningEffort(model)) return undefined;
// Models that reason natively but expose no effort dial carry
// `thinking: undefined` (baked at build time from
// `compat.supportsReasoningEffort: false` on openai-responses*). The
// wire-side omitReasoningEffort gate (providers/xai-responses.ts:78) is the
// actual strip; returning undefined here avoids a redundant
// requireSupportedEffort throw that would defeat the gate and surface a
// confusing "Compaction failed: Thinking effort high is not supported
// by..." to the user.
if (!model.thinking) return undefined;
return requireSupportedEffort(model, reasoning);
}
@@ -869,7 +869,7 @@ function mapOptionsForApi<TApi extends Api>(
...base,
thinking: {
enabled: true,
level: mapEffortToGoogleThinkingLevel(googleModel, effort),
level: mapEffortToGoogleThinkingLevel(effort),
},
toolChoice: mapGoogleToolChoice(options?.toolChoice),
});
@@ -903,7 +903,7 @@ function mapOptionsForApi<TApi extends Api>(
...base,
thinking: {
enabled: true,
level: mapEffortToGoogleThinkingLevel(model, effort),
level: mapEffortToGoogleThinkingLevel(effort),
},
toolChoice: mapGoogleToolChoice(options?.toolChoice),
});
@@ -956,7 +956,7 @@ function mapOptionsForApi<TApi extends Api>(
...base,
thinking: {
enabled: true,
level: mapEffortToGoogleThinkingLevel(geminiModel, effort),
level: mapEffortToGoogleThinkingLevel(effort),
},
toolChoice: mapGoogleToolChoice(options?.toolChoice),
});
+10 -13
View File
@@ -376,7 +376,10 @@ describe("Anthropic request fingerprint alignment", () => {
...ANTHROPIC_MODEL_SPEC,
id: "claude-opus-4-8-20260528",
name: "Claude Opus 4.8",
thinking: { mode: "anthropic-adaptive", minLevel: Effort.Minimal, maxLevel: Effort.XHigh },
thinking: {
mode: "anthropic-adaptive",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
});
await streamAnthropic(
@@ -1677,8 +1680,7 @@ describe("Anthropic request fingerprint alignment", () => {
name: "Claude Opus 4.7",
thinking: {
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.XHigh,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
}),
{
@@ -1717,8 +1719,7 @@ describe("Anthropic request fingerprint alignment", () => {
name: "Claude Opus 4.7",
thinking: {
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.XHigh,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
}),
{
@@ -1745,8 +1746,7 @@ describe("Anthropic request fingerprint alignment", () => {
name: "Claude Opus 4.7",
thinking: {
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.XHigh,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
}),
{
@@ -1777,8 +1777,7 @@ describe("Anthropic request fingerprint alignment", () => {
name: "Claude Opus 4.7",
thinking: {
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.XHigh,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
}),
{
@@ -1811,8 +1810,7 @@ describe("Anthropic request fingerprint alignment", () => {
name: "Claude Opus 4.7",
thinking: {
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.XHigh,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
}),
{
@@ -1857,8 +1855,7 @@ describe("Anthropic request fingerprint alignment", () => {
maxTokens: 128_000,
thinking: {
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.XHigh,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
}),
{
@@ -24,7 +24,10 @@ function adaptiveModel(id: string): Model<"anthropic-messages"> {
const base = makeAnthropicModel(id);
return buildModel({
...base,
thinking: { mode: "anthropic-adaptive", minLevel: Effort.Minimal, maxLevel: Effort.XHigh },
thinking: {
mode: "anthropic-adaptive",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
compat: base.compatConfig,
} as ModelSpec<"anthropic-messages">);
}
+5 -2
View File
@@ -27,7 +27,10 @@ function adaptiveModel(id: string): Model<"bedrock-converse-stream"> {
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 1_000_000,
maxTokens: 128_000,
thinking: { mode: "anthropic-adaptive", minLevel: Effort.Minimal, maxLevel: Effort.XHigh },
thinking: {
mode: "anthropic-adaptive",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
});
}
@@ -43,7 +46,7 @@ function budgetModel(id: string): Model<"bedrock-converse-stream"> {
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 64_000,
thinking: { mode: "budget", minLevel: Effort.Minimal, maxLevel: Effort.High },
thinking: { mode: "budget", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
});
}
+4 -1
View File
@@ -36,7 +36,10 @@ const OPUS_46_OAUTH: Model<"anthropic-messages"> = buildModel({
cost: { input: 5, output: 25, cacheRead: 0.5, cacheWrite: 6.25 },
contextWindow: 1_000_000,
maxTokens: 128_000,
thinking: { mode: "anthropic-adaptive", minLevel: Effort.Minimal, maxLevel: Effort.XHigh },
thinking: {
mode: "anthropic-adaptive",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
});
const todoTool: Tool = {
+2 -4
View File
@@ -83,8 +83,7 @@ describe("issue #826: Anthropic strict-tools opt-out for Vertex-style proxies",
reasoning: true,
thinking: {
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.XHigh,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
compat: baseModel.compatConfig,
} as ModelSpec<"anthropic-messages">);
@@ -110,8 +109,7 @@ describe("issue #826: Anthropic strict-tools opt-out for Vertex-style proxies",
reasoning: true,
thinking: {
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.XHigh,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
compat: { ...baseModel.compatConfig, disableAdaptiveThinking: true },
} as ModelSpec<"anthropic-messages">);
+1 -2
View File
@@ -27,8 +27,7 @@ function customOpenAICompatModel(): Model<"openai-completions"> {
reasoning: true,
thinking: {
mode: "effort",
minLevel: Effort.Low,
maxLevel: Effort.XHigh,
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
@@ -26,8 +26,7 @@ function createReasoningEffortModel(): Model<"openai-completions"> {
reasoning: true,
thinking: {
mode: "effort",
minLevel: Effort.Minimal,
maxLevel: Effort.High,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
},
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
+24 -27
View File
@@ -1,46 +1,43 @@
import { describe, expect, test } from "bun:test";
import { modelOmitsReasoningEffort } from "@oh-my-pi/pi-catalog/model-thinking";
import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
// Pins fix #2 of the compaction effort-override bug. Before this fix,
// `resolveOpenAiReasoningEffort` called `requireSupportedEffort` which threw
// for any model with `compat.supportsReasoningEffort: false` (e.g.
// `xai-oauth/grok-build`) — producing the user-visible "Compaction failed:
// Thinking effort high is not supported by xai-oauth/grok-build. Supported
// efforts:" (empty list). The fix routes through the explicit
// `modelOmitsReasoningEffort` predicate, which lets the wire-side
// `omitReasoningEffort` gate (providers/xai-responses.ts:78) remain the
// single source of truth for the actual strip.
describe("modelOmitsReasoningEffort (regression)", () => {
test("returns true for xai-oauth/grok-build (supportsReasoningEffort: false)", () => {
// Pins fix #2 of the compaction effort-override bug. Models that reason
// natively but reject the wire `reasoning.effort` param (e.g.
// `xai-oauth/grok-build`, `compat.supportsReasoningEffort: false` on
// openai-responses*) are encoded at build time as `thinking: undefined` —
// "thinks, but exposes no control surface". `resolveOpenAiReasoningEffort`
// returns undefined for them instead of tripping `requireSupportedEffort`
// (the old user-visible "Compaction failed: Thinking effort high is not
// supported by xai-oauth/grok-build. Supported efforts:" with an empty list),
// and the wire-side `omitReasoningEffort` gate (providers/xai-responses.ts)
// remains the single source of truth for the actual strip.
describe("effort-dial-less reasoner encoding (regression)", () => {
test("xai-oauth/grok-build reasons but carries no thinking config", () => {
const grokBuild = getBundledModel("xai-oauth", "grok-build");
if (!grokBuild) throw new Error("xai-oauth/grok-build must be in bundled models.json");
expect(modelOmitsReasoningEffort(grokBuild)).toBe(true);
expect(grokBuild.reasoning).toBe(true);
expect(grokBuild.thinking).toBeUndefined();
expect(getSupportedEfforts(grokBuild)).toEqual([]);
});
test("returns false for xai-oauth/grok-4.3 (effort-capable)", () => {
test("xai-oauth/grok-4.3 keeps its effort dial", () => {
const grok43 = getBundledModel("xai-oauth", "grok-4.3");
if (!grok43) throw new Error("xai-oauth/grok-4.3 must be in bundled models.json");
expect(modelOmitsReasoningEffort(grok43)).toBe(false);
expect(grok43.thinking).toBeDefined();
expect(getSupportedEfforts(grok43).length).toBeGreaterThan(0);
});
test("returns true for xai-oauth/grok-4.20-0309-reasoning (supportsReasoningEffort: false)", () => {
test("xai-oauth/grok-4.20-0309-reasoning reasons but carries no thinking config", () => {
const grokR = getBundledModel("xai-oauth", "grok-4.20-0309-reasoning");
if (!grokR) throw new Error("xai-oauth/grok-4.20-0309-reasoning must be in bundled models.json");
expect(modelOmitsReasoningEffort(grokR)).toBe(true);
expect(grokR.reasoning).toBe(true);
expect(grokR.thinking).toBeUndefined();
});
test("returns false for an Anthropic model (different api surface)", () => {
test("the no-dial encoding stays scoped to openai-responses*", () => {
const claude = getBundledModel("anthropic", "claude-sonnet-4-6");
if (!claude) throw new Error("anthropic/claude-sonnet-4-6 must be in bundled models.json");
expect(modelOmitsReasoningEffort(claude)).toBe(false);
});
test("returns false for an openai-completions model (out of scope)", () => {
const openai = getBundledModel("openai", "gpt-4o-mini");
if (!openai) throw new Error("openai/gpt-4o-mini must be in bundled models.json");
// gpt-4o-mini is openai-completions, not openai-responses* — predicate
// must return false even if compat had supportsReasoningEffort: false.
expect(modelOmitsReasoningEffort(openai)).toBe(false);
expect(claude.thinking).toBeDefined();
});
});
+11 -1
View File
@@ -5,7 +5,8 @@
### Added
- Added `hostMatchesUrl`, `modelMatchesHost`, and endpoint-shape helpers in the new `hosts` module for consistent provider/baseUrl matching
- `buildModel(spec)` (`build.ts`) is now the single Model constructor: it runs thinking enrichment and materializes the fully-resolved compat record exactly once, so `Model.compat` is a required, complete `CompatOf<TApi>` (`ResolvedOpenAICompat`/`ResolvedOpenAIResponsesCompat`/`ResolvedAnthropicCompat`) and request-path code reads fields with zero URL parsing and zero per-request allocation. Sparse user/config overrides live on the new `ModelSpec<TApi>` input shape and survive on `Model.compatConfig` for introspection.
- `buildModel(spec)` (`build.ts`) is now the single Model constructor: it materializes the fully-resolved compat record and canonical thinking metadata exactly once (compat first, thinking derived from identity + resolved compat), so `Model.compat` is a required, complete `CompatOf<TApi>` (`ResolvedOpenAICompat`/`ResolvedOpenAIResponsesCompat`/`ResolvedAnthropicCompat`) and request-path code reads fields with zero URL parsing and zero per-request allocation. Sparse user/config overrides live on the new `ModelSpec<TApi>` input shape and survive on `Model.compatConfig` for introspection.
- Added `ResolvedAnthropicCompat.supportsSamplingParams` (Opus 4.7+/Fable/Mythos reject `temperature`/`top_p`/`top_k` with a 400), baked at build time from model identity so the request path stops re-parsing model ids.
- Compat detection gained model-time flags so handlers stop sniffing baseUrl: completions `supportsReasoningParams`, `alwaysSendMaxTokens`, `isOpenRouterHost`, `isVercelGatewayHost`, `streamIdleTimeoutMs`, and a precomputed `whenThinking` alternate view (OpenCode `reasoning_content` gating, #1071/#1484); responses `strictResponsesPairing`, `supportsLongPromptCacheRetention`, `supportsReasoningEffort`; anthropic `officialEndpoint`, `requiresToolResultId`, `replayUnsignedThinking`.
- New `@oh-my-pi/pi-catalog` package: the model catalog extracted from `@oh-my-pi/pi-ai`. Owns the bundled `models.json` and its generation pipeline (`scripts/generate-models.ts`), the core model data types (`Model`, `Api`, `ThinkingConfig`, `Effort`, `Usage`, compat interfaces), thinking metadata enrichment and generated policies (`model-thinking.ts`), the SQLite model cache and model manager, per-provider discovery factories (`provider-models/`), the discovery protocol clients (`discovery/`), and the new `CATALOG_PROVIDERS` table — the single source of truth for provider ids, default models, and discovery wiring (`KnownProvider`, `PROVIDER_DESCRIPTORS`, and `DEFAULT_MODEL_PER_PROVIDER` are derived from it).
- New `identity/` module centralizing model-identity concerns that were previously duplicated across packages: family classification and version parsing (`identity/classify.ts`, extracted from pi-ai's `model-thinking` internals), canonical model equivalence with injected reference data (`identity/equivalence.ts`, from coding-agent's `model-equivalence`), proxy/reseller reference lookup (`identity/reference.ts`, from coding-agent's `model-registry`), bracket-affix and id-segment helpers (`identity/id.ts`), a single trailing-marker vocabulary with canonical vs reference flavors (`identity/markers.ts` — `search` stays reference-only so Perplexity's `sonar-pro-search` remains canonical-distinct), and provider priority ordering (`identity/priority.ts`).
@@ -18,9 +19,18 @@
- Changed `hostMatchesUrl`/`modelMatchesHost` usage in compatibility detection to reduce mismatches across case variants and provider alias hosts
- Provider catalog entries now carry the runtime API-key env fallback as an ordered `envVars` list; `catalogDiscovery.envVars` became an optional generation-time override (only `cursor` and `vercel-ai-gateway` differ) and `PROVIDER_DESCRIPTORS` materializes the resolved list for `generate-models.ts`.
- `Model`'s api parameter now defaults to `Api` instead of `any` (`Model<TApi extends Api = Api>`), so bare `Model` no longer behaves as `Model<any>` at call sites.
- `ThinkingConfig` is now explicit and total: an ordered `efforts` array replaces the `minLevel`/`maxLevel`/`levels` range encoding, and the wire facts are baked alongside it — `effortMap` (anthropic-adaptive 4-tier vs 5-tier scale, shared with the OpenRouter completions remap) and `supportsDisplay` (adaptive `display` field support). Explicit spec thinking owns the capability surface (`mode`/`efforts`/`defaultLevel`) and wins over inference; missing wire facts are backfilled from identity so configs never need to know Anthropic's tier tables. Reasoning models that reject the wire effort param (`compat.supportsReasoningEffort: false` on openai-responses*) are encoded as `thinking: undefined` ("thinks, no control surface") instead of the removed `modelOmitsReasoningEffort` special case. `models.json` was re-baked in the new vocabulary behind a 3196-model behavioral parity gate, and the model cache schema bumped to v4 to invalidate old-shape rows.
- `mapEffortToGoogleThinkingLevel(effort)` is now a static map (model parameter dropped — validation stays at the `requireSupportedEffort` call sites), and `mapEffortToAnthropicAdaptiveEffort` reads the baked `thinking.effortMap` instead of re-classifying the model id per request.
- Generator-only policy code moved out of the runtime bundle into `scripts/generated-policies.ts`: `applyGeneratedModelPolicies` (now policy fixups + thinking re-bake via the shared deriver), `linkOpenAIPromotionTargets`, the Copilot context-window table, minimax/opencode-go compat fixups, and `CLOUDFLARE_FALLBACK_MODEL`. The anthropic id predicates (`hasOpus47ApiRestrictions`, `supportsMidConversationSystemMessages`, `isAnthropicFableOrMythosModel`) moved to `identity/family` for build-time use by the compat/thinking derivers only.
### Fixed
- Fixed Anthropic official-endpoint detection to require strict HTTPS hostname matching so non-official or lookalike URLs are no longer treated as official Anthropic hosts
- Fixed Ollama Cloud dynamic discovery so same-id matches from other providers no longer supply context-window or max-output-token limits for discovered models.
- Wired `@oh-my-pi/pi-catalog` into the release publish package list, tarball install smoke test, and root `bun generate-models` script.
- Fixed `supportsAdaptiveThinkingDisplay` only matching dash-form version ids: dotted ids (`claude-opus-4.7`) now classify through `identity/classify` like every other anthropic predicate, so six bundled dotted Opus 4.7/4.8 entries (github-copilot, vercel-ai-gateway, zenmux) regain adaptive `display` support; bare dated ids (`claude-opus-4-20250514` = Opus 4.0) stay excluded.
- Fixed the OpenRouter anthropic adaptive-effort map misclassifying bare dated Opus ids (`claude-opus-4-20250514` parsed as version 4.20 → wrongly adaptive); the map now derives from the shared classifier and the shared 4-/5-tier tables.
### Removed
- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads.
+9 -6
View File
@@ -17,11 +17,6 @@ import { $env } from "@oh-my-pi/pi-utils";
import { fetchAntigravityDiscoveryModels } from "../src/discovery/antigravity";
import { fetchCodexModels } from "../src/discovery/codex";
import { createModelManager } from "../src/model-manager";
import {
applyGeneratedModelPolicies,
CLOUDFLARE_FALLBACK_MODEL,
linkOpenAIPromotionTargets,
} from "../src/model-thinking";
import prevModelsJson from "../src/models.json" with { type: "json" };
import { toModelSpec } from "../src/provider-models/bundled-references";
import {
@@ -44,6 +39,11 @@ import {
} from "../src/provider-models/openai-compat";
import type { ModelSpec } from "../src/types";
import { JWT_CLAIM_PATH } from "../src/wire/codex";
import {
applyGeneratedModelPolicies,
CLOUDFLARE_FALLBACK_MODEL,
linkOpenAIPromotionTargets,
} from "./generated-policies";
const packageRoot = path.join(import.meta.dir, "..");
@@ -429,7 +429,10 @@ async function generateModels() {
// Discovery-only providers (local inference servers) — never bundle static models.
const fetchedKeys = new Set(allModels.map(model => `${model.provider}/${model.id}`));
for (const models of Object.values(prevModelsJson as Record<string, Record<string, ModelSpec>>)) {
// Previous-snapshot entries may carry an older ThinkingConfig vocabulary;
// applyGeneratedModelPolicies re-bakes `thinking` for every model, so the
// inbound shape is irrelevant beyond identity/pricing/compat fields.
for (const models of Object.values(prevModelsJson as unknown as Record<string, Record<string, ModelSpec>>)) {
for (const model of Object.values(models)) {
if (
!fetchedKeys.has(`${model.provider}/${model.id}`) &&
@@ -0,0 +1,223 @@
/**
* Generation-time catalog policies: upstream metadata corrections, derived
* field baking, and promotion-target linking. Runs only from
* `generate-models.ts` — none of this ships in the runtime bundle.
*/
import { buildCompat } from "../src/build";
import {
type AnthropicModel,
isFableOrMythos,
type OpenAIModel,
type OpenAIVariant,
type ParsedModel,
parseKnownModel,
semverEqual,
} from "../src/identity/classify";
import { resolveModelThinking } from "../src/model-thinking";
import type { Api, ModelSpec } from "../src/types";
const CLOUDFLARE_AI_GATEWAY_BASE_URL = "https://gateway.ai.cloudflare.com/v1/<account>/<gateway>/anthropic";
/**
* Static fallback model injected when Cloudflare AI Gateway discovery
* returns no results. Ensures the provider always has at least one usable
* model entry in the catalog.
*/
export const CLOUDFLARE_FALLBACK_MODEL: ModelSpec<"anthropic-messages"> = {
id: "claude-sonnet-4-5",
name: "Claude Sonnet 4.5",
api: "anthropic-messages",
provider: "cloudflare-ai-gateway",
baseUrl: CLOUDFLARE_AI_GATEWAY_BASE_URL,
reasoning: true,
input: ["text", "image"],
cost: {
input: 3,
output: 15,
cacheRead: 0.3,
cacheWrite: 3.75,
},
contextWindow: 200000,
maxTokens: 64000,
};
const CODEX_GPT_5_4_PRIORITY_BY_VARIANT: Partial<Record<OpenAIVariant, number>> = {
base: 0,
mini: 1,
nano: 2,
};
const COPILOT_GENERATED_LIMITS: Record<string, { contextWindow: number; maxTokens: number }> = {
"claude-opus-4.6": { contextWindow: 168000, maxTokens: 32000 },
"gpt-5.2": { contextWindow: 272000, maxTokens: 128000 },
"gpt-5.4": { contextWindow: 272000, maxTokens: 128000 },
"gpt-5.4-mini": { contextWindow: 272000, maxTokens: 128000 },
"grok-code-fast-1": { contextWindow: 192000, maxTokens: 64000 },
};
/**
* Apply upstream metadata corrections to a mutable array of models, then
* re-bake canonical thinking metadata so generated catalogs always carry the
* deriver's output for the post-policy spec.
*/
export function applyGeneratedModelPolicies(models: ModelSpec<Api>[]): void {
for (const model of models) {
applyGeneratedModelPolicy(model);
rebakeModelThinking(model);
}
}
/**
* Recompute `thinking` from the canonical deriver, replacing any baked value.
* Mirrors `buildModel`'s trust-or-derive resolution with trust disabled: the
* generator is the authority that produces the trusted values.
*/
export function rebakeModelThinking(model: ModelSpec<Api>): void {
const thinking = resolveModelThinking({ ...model, thinking: undefined }, buildCompat(model));
if (thinking) {
model.thinking = thinking;
} else {
delete model.thinking;
}
}
/**
* Link OpenAI model variants to their context promotion targets.
*
* When a model's context is exhausted, the agent can promote to a sibling
* model with a larger context window on the same provider:
* - `codex-spark` variants promote to `gpt-5.5`.
* - `gpt-5.5` (270K input) promotes to `gpt-5.4` (1M input).
*/
export function linkOpenAIPromotionTargets(models: ModelSpec<Api>[]): void {
for (const candidate of models) {
const parsedCandidate = parseKnownModel(candidate.id);
if (parsedCandidate.family !== "openai") continue;
let targetId: string | undefined;
if (parsedCandidate.variant === "codex-spark") {
targetId = "gpt-5.5";
} else if (parsedCandidate.variant === "base" && semverEqual(parsedCandidate.version, "5.5")) {
targetId = "gpt-5.4";
} else {
continue;
}
const fallback = models.find(
model => model.provider === candidate.provider && model.api === candidate.api && model.id === targetId,
);
if (!fallback) continue;
candidate.contextPromotionTarget = `${fallback.provider}/${fallback.id}`;
}
}
function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
const copilotLimits = model.provider === "github-copilot" ? COPILOT_GENERATED_LIMITS[model.id] : undefined;
if (copilotLimits) {
model.contextWindow = copilotLimits.contextWindow;
model.maxTokens = copilotLimits.maxTokens;
}
if (
model.api === "openai-completions" &&
(model.provider === "minimax-code" || model.provider === "minimax-code-cn")
) {
model.compat = {
...(model.compat ?? {}),
supportsStore: false,
supportsDeveloperRole: false,
supportsReasoningEffort: false,
reasoningContentField: "reasoning_content",
};
delete model.compat.thinkingFormat;
}
if (
model.api === "openai-completions" &&
model.provider === "opencode-go" &&
(model.id === "deepseek-v4-flash" || model.id === "deepseek-v4-pro")
) {
model.compat = {
...(model.compat ?? {}),
supportsToolChoice: false,
reasoningContentField: "reasoning_content",
requiresReasoningContentForToolCalls: true,
};
}
const parsedModel = parseKnownModel(model.id);
const applyPatchToolType = inferGeneratedApplyPatchToolType(model, parsedModel);
if (applyPatchToolType) {
model.applyPatchToolType = applyPatchToolType;
} else {
delete model.applyPatchToolType;
}
if (parsedModel.family === "anthropic") {
applyAnthropicCatalogPolicy(model, parsedModel);
}
if (parsedModel.family === "openai") {
applyOpenAICatalogPolicy(model, parsedModel);
}
}
function applyAnthropicCatalogPolicy(model: ModelSpec<Api>, parsedModel: AnthropicModel): void {
// Claude Opus 4.5: models.dev reports 3x the correct cache pricing.
if (model.provider === "anthropic" && parsedModel.kind === "opus" && semverEqual(parsedModel.version, "4.5")) {
model.cost.cacheRead = 0.5;
model.cost.cacheWrite = 6.25;
}
// Bedrock Opus 4.6: upstream metadata is stale for cache pricing and context.
if (model.provider === "amazon-bedrock" && parsedModel.kind === "opus" && semverEqual(parsedModel.version, "4.6")) {
model.cost.cacheRead = 0.5;
model.cost.cacheWrite = 6.25;
model.contextWindow = 1000000;
model.maxTokens = 128000;
}
// Claude Fable/Mythos 5: Anthropic's /v1/models omits token limits and
// pricing, and models.dev lags new releases. Pin authoritative values from
// the model card (1M context / 128k output) and pricing docs ($10 in / $50
// out per MTok).
if (model.provider === "anthropic" && isFableOrMythos(parsedModel.kind)) {
model.contextWindow = 1_000_000;
model.maxTokens = 128_000;
model.cost.input = 10;
model.cost.output = 50;
model.cost.cacheRead = 1;
model.cost.cacheWrite = 12.5;
}
}
function inferGeneratedApplyPatchToolType(
model: ModelSpec<Api>,
parsedModel: ParsedModel,
): ModelSpec<Api>["applyPatchToolType"] {
if (parsedModel.family !== "openai" || parsedModel.version.major !== 5) {
return undefined;
}
if (model.provider === "openai" && model.api === "openai-responses") {
return "freeform";
}
if (model.provider === "openai-codex" && model.api === "openai-codex-responses") {
return "freeform";
}
return undefined;
}
function applyOpenAICatalogPolicy(model: ModelSpec<Api>, parsedModel: OpenAIModel): void {
// Codex models: 400K figure includes output budget; input window is 272K.
if (parsedModel.variant.startsWith("codex") && parsedModel.variant !== "codex-spark") {
model.contextWindow = 272000;
return;
}
// GPT-5.4 mini/nano use plain OpenAI IDs on the Codex transport, but Codex still
// enforces the lower prompt budget for these variants. Codex discovery can also
// report inconsistent priorities for the GPT-5.4 family, so normalize by parsed
// variant instead of special-casing raw model ids.
if (model.api === "openai-codex-responses" && semverEqual(parsedModel.version, "5.4")) {
const normalizedPriority = CODEX_GPT_5_4_PRIORITY_BY_VARIANT[parsedModel.variant];
if (normalizedPriority !== undefined) {
model.priority = normalizedPriority;
}
if (parsedModel.variant === "mini" || parsedModel.variant === "nano") {
model.contextWindow = 272000;
}
}
}
+16 -10
View File
@@ -1,24 +1,30 @@
/**
* The single Model constructor. Thinking metadata and the resolved compat
* record are materialized here, exactly once per spec — request handlers read
* `model.compat` fields and perform zero URL parsing and zero compat
* allocation per request.
* The single Model constructor. Resolution order is a dependency chain, each
* step materialized exactly once per spec:
*
* 1. compat — URL/provider/id detection resolved into a complete record;
* 2. thinking — derived from identity + resolved compat (or trusted verbatim
* when the spec carries explicit metadata);
*
* Request handlers read fields — they never detect, parse ids, or allocate
* compat per request.
*/
import { buildAnthropicCompat } from "./compat/anthropic";
import { buildOpenAICompat, buildOpenAIResponsesCompat } from "./compat/openai";
import { enrichModelThinking } from "./model-thinking";
import { resolveModelThinking } from "./model-thinking";
import type { Api, CompatOf, Model, ModelSpec } from "./types";
export function buildModel<TApi extends Api>(spec: ModelSpec<TApi>): Model<TApi> {
const enriched = enrichModelThinking(spec);
const compat = buildCompat(spec) as CompatOf<TApi>;
return {
...enriched,
compat: buildCompat(enriched) as CompatOf<TApi>,
compatConfig: enriched.compat,
...spec,
thinking: resolveModelThinking(spec, compat),
compat,
compatConfig: spec.compat,
} as Model<TApi>;
}
function buildCompat(spec: ModelSpec<Api>): CompatOf<Api> {
export function buildCompat(spec: ModelSpec<Api>): CompatOf<Api> {
switch (spec.api) {
case "openai-completions":
return buildOpenAICompat(spec as ModelSpec<"openai-completions">);
+7 -1
View File
@@ -5,7 +5,11 @@
* classification, with explicit spec overrides assigned on top.
*/
import { modelMatchesHost } from "../hosts";
import { isAnthropicFableOrMythosModel, supportsMidConversationSystemMessages } from "../model-thinking";
import {
hasOpus47ApiRestrictions,
isAnthropicFableOrMythosModel,
supportsMidConversationSystemMessages,
} from "../identity/family";
import type { ModelSpec, ResolvedAnthropicCompat } from "../types";
import { applyCompatOverrides } from "./apply";
@@ -44,6 +48,8 @@ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): Res
// supported model id.
supportsMidConversationSystem: official && supportsMidConversationSystemMessages(spec.id),
supportsForcedToolChoice: !isAnthropicFableOrMythosModel(spec.id),
// Opus 4.7+ and Fable/Mythos reject temperature/top_p/top_k with a 400.
supportsSamplingParams: !hasOpus47ApiRestrictions(spec.id),
// Z.AI workaround (issue #814): its proxy deserializes tool_result blocks
// into a class that reads `.id`.
requiresToolResultId: isZai,
+12 -23
View File
@@ -8,6 +8,7 @@
* never detect, resolve, or allocate.
*/
import { hostMatchesUrl, modelMatchesHost } from "../hosts";
import { bareModelId, isFableOrMythos, parseAnthropicModel, semverGte } from "../identity/classify";
import {
isAnthropicNamespacedModelId,
isClaudeModelId,
@@ -17,6 +18,7 @@ import {
isMimoModelIdOrName,
isQwenModelId,
} from "../identity/family";
import { ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER, ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER } from "../model-thinking";
import type { ModelSpec, OpenAICompat, ResolvedOpenAICompat, ResolvedOpenAIResponsesCompat } from "../types";
import { applyCompatOverrides } from "./apply";
@@ -73,30 +75,17 @@ function detectStrictModeSupport(provider: string, baseUrl: string): boolean {
function getOpenRouterAnthropicReasoningEffortMap(
modelId: string,
): Partial<Record<OpenAIReasoningEffort, string>> | undefined {
const match = /(?:^|\/)claude-(opus|fable|mythos)-(\d{1,2})(?:[.-](\d{1,2}))?/.exec(modelId);
if (!match) return undefined;
const parsed = parseAnthropicModel(bareModelId(modelId));
if (!parsed) return undefined;
// Adaptive efforts on OpenRouter's completions front: Fable/Mythos and
// Opus 4.6+ only — Sonnet stays on the plain effort vocabulary there.
const isOpusAdaptive = parsed.kind === "opus" && semverGte(parsed.version, "4.6");
if (!isFableOrMythos(parsed.kind) && !isOpusAdaptive) return undefined;
const kind = match[1];
const major = Number(match[2]);
const minor = Number(match[3] ?? 0);
const isFableOrMythos = kind === "fable" || kind === "mythos";
const isOpusAdaptive = kind === "opus" && (major > 4 || (major === 4 && minor >= 6));
if (!isFableOrMythos && !isOpusAdaptive) return undefined;
const hasRealXHigh = isFableOrMythos || major > 4 || (major === 4 && minor >= 7);
if (hasRealXHigh) {
return {
minimal: "low",
low: "medium",
medium: "high",
high: "xhigh",
xhigh: "max",
};
}
return {
minimal: "low",
xhigh: "max",
};
const hasRealXHigh = isFableOrMythos(parsed.kind) || semverGte(parsed.version, "4.7");
return (hasRealXHigh ? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER) as Partial<
Record<OpenAIReasoningEffort, string>
>;
}
/**
+39 -10
View File
@@ -7,6 +7,8 @@
* here.
*/
import { bareModelId, isFableOrMythos, parseAnthropicModel, semverGte } from "./classify";
/** Kimi family ids in any namespace form (`moonshotai/kimi-*`, `kimi-k2.6`, `vendor/kimi.x`). */
export function isKimiModelId(modelId: string): boolean {
return modelId.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(modelId);
@@ -44,16 +46,43 @@ export function isMimoModelIdOrName(value: string): boolean {
/**
* Adaptive thinking `display` is supported starting with Claude Opus 4.7 and
* Claude Fable/Mythos 5. Older adaptive-thinking models (Opus 4.6, Sonnet
* 4.6+) reject the field.
* the Claude Fable/Mythos 5 generation. Older adaptive-thinking models
* (Opus 4.6, Sonnet 4.6+) reject the field. Classifier-based, so dotted and
* dashed version forms both match while bare dated ids
* (`claude-opus-4-20250514` = Opus 4.0) stay excluded.
*/
export function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
if (/claude-(?:fable|mythos)-5\b/.test(modelId)) return true;
// Bound the minor to non-date digits: bare dated ids like
// `claude-opus-4-20250514` (Opus 4.0) must not parse as minor=20250514.
const match = /claude-opus-(\d+)-(\d{1,2})(?!\d)/.exec(modelId);
if (!match) return false;
const major = Number(match[1]);
const minor = Number(match[2]);
return major > 4 || (major === 4 && minor >= 7);
const parsed = parseAnthropicModel(bareModelId(modelId));
if (!parsed) return false;
if (isFableOrMythos(parsed.kind)) return semverGte(parsed.version, "5");
return parsed.kind === "opus" && semverGte(parsed.version, "4.7");
}
/**
* Returns true for Anthropic models with Opus 4.7+/Fable/Mythos API restrictions:
* - Sampling parameters (temperature/top_p/top_k) return 400 error
* - Thinking content is omitted by default (needs display: "summarized")
*/
export function hasOpus47ApiRestrictions(modelId: string): boolean {
const parsed = parseAnthropicModel(bareModelId(modelId));
if (!parsed) return false;
return (parsed.kind === "opus" && semverGte(parsed.version, "4.7")) || isFableOrMythos(parsed.kind);
}
/**
* Mid-conversation `role: "system"` messages (system instructions appended at
* non-first positions in the `messages` array) are supported starting with
* Claude Opus 4.8 and the Claude Fable/Mythos 5 generation. Earlier Claude
* models reject the role.
* @see https://platform.claude.com/docs/en/build-with-claude/mid-conversation-system-messages
*/
export function supportsMidConversationSystemMessages(modelId: string): boolean {
const parsed = parseAnthropicModel(bareModelId(modelId));
if (!parsed) return false;
return (parsed.kind === "opus" && semverGte(parsed.version, "4.8")) || isFableOrMythos(parsed.kind);
}
export function isAnthropicFableOrMythosModel(modelId: string): boolean {
const parsed = parseAnthropicModel(bareModelId(modelId));
return parsed !== null && isFableOrMythos(parsed.kind);
}
+3 -3
View File
@@ -7,9 +7,9 @@ import { getModelDbPath } from "@oh-my-pi/pi-utils";
import type { Api, Model, ModelSpec } from "./types";
// Rows persist ModelSpec JSON (sparse `compat`, never the resolved record);
// the model manager rebuilds via `buildModel` on load. v3 rows predating the
// resolved-compat redesign already carried sparse compat, so they stay valid.
const CACHE_SCHEMA_VERSION = 3;
// the model manager rebuilds via `buildModel` on load. v4 invalidates rows
// carrying the pre-efforts ThinkingConfig shape (minLevel/maxLevel/levels).
const CACHE_SCHEMA_VERSION = 4;
interface CacheRow {
provider_id: string;
+270 -521
View File
@@ -1,24 +1,37 @@
import { buildOpenAICompat } from "./compat/openai";
/**
* Thinking metadata: build-time derivation and runtime field-read helpers.
*
* Derivation (`resolveModelThinking`) runs exactly once per model — from
* `buildModel` for dynamic specs and from the catalog generator for bundled
* entries. Everything below the "runtime helpers" divider reads baked fields
* only: no id parsing, no host matching, no compat detection per request.
*/
import { Effort, THINKING_EFFORTS } from "./effort";
import { modelMatchesHost } from "./hosts";
import {
type AnthropicModel,
bareModelId,
type GeminiModel,
isFableOrMythos,
type OpenAIModel,
type OpenAIVariant,
type ParsedModel,
parseAnthropicModel,
parseKnownModel,
semverEqual,
semverGte,
} from "./identity/classify";
import type { Api, Model, ModelSpec, ThinkingConfig } from "./types";
import { supportsAdaptiveThinkingDisplay } from "./identity/family";
import type {
Api,
CompatOf,
Model,
ModelSpec,
ResolvedOpenAICompat,
ResolvedOpenAIResponsesCompat,
ThinkingConfig,
} from "./types";
/**
* Thinking inference reads identity fields plus sparse compat intent, so it
* accepts both pre-build specs and built models.
* Runtime helpers read baked metadata only, so they accept both pre-build
* specs and built models.
*/
type ApiModel<TApi extends Api = Api> = ModelSpec<TApi> | Model<TApi>;
@@ -34,190 +47,286 @@ const GEMINI_3_PRO_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High];
const GEMINI_3_FLASH_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High];
const GPT_5_2_PLUS_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh];
const GPT_5_1_CODEX_MINI_EFFORTS: readonly Effort[] = [Effort.Medium, Effort.High];
const CLOUDFLARE_AI_GATEWAY_BASE_URL = "https://gateway.ai.cloudflare.com/v1/<account>/<gateway>/anthropic";
const CODEX_GPT_5_4_PRIORITY_BY_VARIANT: Partial<Record<OpenAIVariant, number>> = {
base: 0,
mini: 1,
nano: 2,
};
const COPILOT_GENERATED_LIMITS: Record<string, { contextWindow: number; maxTokens: number }> = {
"claude-opus-4.6": { contextWindow: 168000, maxTokens: 32000 },
"gpt-5.2": { contextWindow: 272000, maxTokens: 128000 },
"gpt-5.4": { contextWindow: 272000, maxTokens: 128000 },
"gpt-5.4-mini": { contextWindow: 272000, maxTokens: 128000 },
"grok-code-fast-1": { contextWindow: 192000, maxTokens: 64000 },
/**
* Effort → wire-value map for the 5-tier adaptive scale (Opus 4.7+ and
* Fable/Mythos 5 on the Messages API). User-facing efforts shift up one notch
* so the top tier reaches the genuine "max" and "high" lands on Anthropic's
* recommended "xhigh" coding/agentic default.
*/
export const ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER: Readonly<Partial<Record<Effort, string>>> = {
[Effort.Minimal]: "low",
[Effort.Low]: "medium",
[Effort.Medium]: "high",
[Effort.High]: "xhigh",
[Effort.XHigh]: "max",
};
/**
* Static fallback model injected when Cloudflare AI Gateway discovery
* returns no results. Ensures the provider always has at least one usable
* model entry in the catalog.
* Effort → wire-value map for the legacy 4-tier adaptive scale (Opus 4.6,
* Sonnet 4.6+, and every adaptive model on Bedrock Converse). `low..high` pass
* through verbatim; there is no real "xhigh", so it aliases the top "max" tier.
*/
export const CLOUDFLARE_FALLBACK_MODEL: ApiModel<"anthropic-messages"> = {
id: "claude-sonnet-4-5",
name: "Claude Sonnet 4.5",
api: "anthropic-messages",
provider: "cloudflare-ai-gateway",
baseUrl: CLOUDFLARE_AI_GATEWAY_BASE_URL,
reasoning: true,
input: ["text", "image"],
cost: {
input: 3,
output: 15,
cacheRead: 0.3,
cacheWrite: 3.75,
},
contextWindow: 200000,
maxTokens: 64000,
export const ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER: Readonly<Partial<Record<Effort, string>>> = {
[Effort.Minimal]: "low",
[Effort.XHigh]: "max",
};
const kEnrichedModel = Symbol("model-thinking.enrichedModel");
type ModelWithEnriched = ApiModel<Api> & { [kEnrichedModel]?: ApiModel<Api> };
// ---------------------------------------------------------------------------
// Build-time derivation (buildModel + catalog generator only)
// ---------------------------------------------------------------------------
/**
* Returns a copy of the model with canonical thinking metadata attached.
* Resolve the canonical thinking metadata for a spec. Called exactly once per
* model by `buildModel`, after compat resolution.
*
* This helper belongs to catalog enrichment only. Runtime consumers should
* trust `model.thinking` and avoid inferring capabilities on demand.
* - Non-reasoning models never carry thinking.
* - Models that reason natively but reject the wire effort param
* (`compat.supportsReasoningEffort: false` on openai-responses*) carry no
* thinking either: `reasoning: true, thinking: undefined` IS the encoding
* for "thinks, but exposes no control surface".
* - Explicit spec thinking (generator-baked or user-authored) owns the
* capability surface (`mode`, `efforts`, `defaultLevel`); the wire facts
* (`effortMap`, `supportsDisplay`) are backfilled from identity when not
* explicitly set, so configs never need to know Anthropic's tier tables.
* - Sparse specs go through full inference.
*/
export function enrichModelThinking<TApi extends Api>(model: ModelSpec<TApi>): ModelSpec<TApi>;
export function enrichModelThinking<TApi extends Api>(model: Model<TApi>): Model<TApi>;
export function enrichModelThinking<TApi extends Api>(model: ApiModel<TApi>): ApiModel<TApi> {
const tagged = model as ModelWithEnriched;
const cached = tagged[kEnrichedModel];
if (cached !== undefined) {
return cached as ApiModel<TApi>;
export function resolveModelThinking<TApi extends Api>(
spec: ModelSpec<TApi>,
compat: CompatOf<TApi>,
): ThinkingConfig | undefined {
if (!spec.reasoning) return undefined;
if (omitsWireReasoningEffort(spec.api, compat)) return undefined;
if (spec.thinking && spec.thinking.efforts.length > 0) {
return fillThinkingWireDefaults(spec, spec.thinking);
}
const normalizedThinking = normalizeThinkingConfig(model.thinking);
let result: ApiModel<TApi>;
if (!model.reasoning) {
result =
normalizedThinking === undefined && model.thinking === undefined ? model : { ...model, thinking: undefined };
} else {
const thinking = normalizedThinking ?? inferModelThinking(model);
result = thinkingsEqual(normalizedThinking, thinking) ? model : { ...model, thinking };
}
// Stash the enriched copy on a non-enumerable slot so callers that hand us
// the same reference twice skip the work. `enumerable: false` is critical:
// many call sites build derived models via `{ ...model, ...overrides }`,
// which would otherwise copy this cache slot and trick us into returning
// the *original* enriched model — silently discarding the overrides.
Object.defineProperty(tagged, kEnrichedModel, {
value: result,
enumerable: false,
configurable: true,
writable: true,
});
return result;
// Empty/malformed explicit metadata is treated as absent — infer instead.
return deriveThinking(spec, compat);
}
/**
* Returns a copy of the model with thinking metadata recomputed from the
* canonical rules, replacing any existing `thinking`.
* Backfill identity-derived wire facts onto explicit thinking metadata.
* Explicit `effortMap` / `supportsDisplay` (including `false`) always win;
* untouched configs are returned as-is with zero allocation.
*/
export function refreshModelThinking<TApi extends Api>(model: ApiModel<TApi>): ApiModel<TApi> {
if (!model.reasoning) {
const normalizedThinking = normalizeThinkingConfig(model.thinking);
return normalizedThinking === undefined && model.thinking === undefined
? model
: { ...model, thinking: undefined };
function fillThinkingWireDefaults<TApi extends Api>(spec: ModelSpec<TApi>, thinking: ThinkingConfig): ThinkingConfig {
const needsEffortMap = thinking.mode === "anthropic-adaptive" && thinking.effortMap === undefined;
const needsDisplay =
thinking.supportsDisplay === undefined &&
(spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") &&
supportsAdaptiveThinkingDisplay(spec.id);
if (!needsEffortMap && !needsDisplay) {
return thinking;
}
return { ...model, thinking: inferModelThinking(model) };
const filled: ThinkingConfig = { ...thinking };
if (needsEffortMap) {
filled.effortMap = anthropicModelHasRealXHighEffort(spec, parseKnownModel(spec.id))
? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER
: ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER;
}
if (needsDisplay) {
filled.supportsDisplay = true;
}
return filled;
}
/**
* Apply upstream metadata corrections to a mutable array of models.
*
* Each model is first normalized through `refreshModelThinking()` so generated
* catalogs keep canonical thinking metadata and policy fixes in one pass.
*/
export function applyGeneratedModelPolicies(models: ApiModel<Api>[]): void {
for (let index = 0; index < models.length; index++) {
const model = refreshModelThinking(models[index]!);
applyGeneratedModelPolicy(model);
models[index] = model;
/** Derive thinking from identity + resolved compat, ignoring any baked value. Generator-side entry. */
export function deriveThinking<TApi extends Api>(spec: ModelSpec<TApi>, compat: CompatOf<TApi>): ThinkingConfig {
const parsed = parseKnownModel(spec.id);
const efforts = inferSupportedEfforts(parsed, spec, compat);
if (efforts.length === 0) {
throw new Error(`Model ${spec.provider}/${spec.id} resolved to an empty thinking range`);
}
}
/**
* Link OpenAI model variants to their context promotion targets.
*
* When a model's context is exhausted, the agent can promote to a sibling
* model with a larger context window on the same provider:
* - `codex-spark` variants promote to `gpt-5.5`.
* - `gpt-5.5` (270K input) promotes to `gpt-5.4` (1M input).
*/
export function linkOpenAIPromotionTargets(models: ApiModel<Api>[]): void {
for (const candidate of models) {
const parsedCandidate = parseKnownModel(candidate.id);
if (parsedCandidate.family !== "openai") continue;
let targetId: string | undefined;
if (parsedCandidate.variant === "codex-spark") {
targetId = "gpt-5.5";
} else if (parsedCandidate.variant === "base" && semverEqual(parsedCandidate.version, "5.5")) {
targetId = "gpt-5.4";
} else {
continue;
}
const fallback = models.find(
model => model.provider === candidate.provider && model.api === candidate.api && model.id === targetId,
);
if (!fallback) continue;
candidate.contextPromotionTarget = `${fallback.provider}/${fallback.id}`;
const config: ThinkingConfig = {
mode: inferThinkingControlMode(spec, parsed),
efforts,
};
if (config.mode === "anthropic-adaptive") {
config.effortMap = anthropicModelHasRealXHighEffort(spec, parsed)
? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER
: ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER;
}
if (
(spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") &&
supportsAdaptiveThinkingDisplay(spec.id)
) {
config.supportsDisplay = true;
}
return config;
}
/**
* True when the model reasons natively but rejects the wire `reasoning.effort`
* param (compat.supportsReasoningEffort: false on openai-responses*). Callers
* are expected to omit the effort field; the wire-side omitReasoningEffort
* gate (providers/xai-responses.ts:78) is the actual strip, and this
* predicate is the upstream check that prevents a redundant
* requireSupportedEffort throw from defeating that gate.
*
* Scoped to openai-responses* because that's the only API surface where
* `compat.supportsReasoningEffort: false` is meaningful today. The
* `in`-narrowed access is necessary because Model.compat is
* `AnthropicCompat | OpenAICompat` and the api gate doesn't narrow the
* union for TS.
* param. Scoped to openai-responses* because that's the only API surface where
* `compat.supportsReasoningEffort: false` means "omit the field entirely"
* (xAI Grok off the GROK_EFFORT_CAPABLE_PREFIXES allowlist: grok-build,
* grok-4.20-0309-reasoning). openai-completions keeps its thinking config even
* without effort support — binary thinking formats (zai/qwen) drive reasoning
* through other request fields.
*/
export function modelOmitsReasoningEffort<TApi extends Api>(model: ApiModel<TApi>): boolean {
if (model.api !== "openai-responses" && model.api !== "openai-codex-responses") {
function omitsWireReasoningEffort(api: Api, compat: CompatOf<Api>): boolean {
if (api !== "openai-responses" && api !== "openai-codex-responses") {
return false;
}
const compat = model.compat;
return Boolean(compat && "supportsReasoningEffort" in compat && compat.supportsReasoningEffort === false);
return (compat as ResolvedOpenAIResponsesCompat | undefined)?.supportsReasoningEffort === false;
}
function inferSupportedEfforts<TApi extends Api>(
parsedModel: ParsedModel,
spec: ModelSpec<TApi>,
compat: CompatOf<TApi>,
): readonly Effort[] {
switch (parsedModel.family) {
case "openai":
return inferOpenAISupportedEfforts(parsedModel);
case "gemini":
return inferGeminiSupportedEfforts(parsedModel);
case "anthropic":
return inferAnthropicSupportedEfforts(parsedModel, spec, compat);
case "unknown":
return inferFallbackEfforts(spec, compat);
}
}
function inferOpenAISupportedEfforts(model: OpenAIModel): readonly Effort[] {
if (model.variant === "codex-mini" && semverEqual(model.version, "5.1")) {
return GPT_5_1_CODEX_MINI_EFFORTS;
}
if (semverGte(model.version, "5.2")) {
return GPT_5_2_PLUS_EFFORTS;
}
return DEFAULT_REASONING_EFFORTS;
}
function inferGeminiSupportedEfforts(model: GeminiModel): readonly Effort[] {
if (!semverGte(model.version, "3.0")) {
return DEFAULT_REASONING_EFFORTS;
}
return model.kind === "pro" ? GEMINI_3_PRO_EFFORTS : GEMINI_3_FLASH_EFFORTS;
}
function inferAnthropicSupportedEfforts<TApi extends Api>(
parsedModel: AnthropicModel,
spec: ModelSpec<TApi>,
compat: CompatOf<TApi>,
): readonly Effort[] {
if (
(spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") &&
semverGte(parsedModel.version, "4.6")
) {
return parsedModel.kind === "opus" || isFableOrMythos(parsedModel.kind)
? DEFAULT_REASONING_EFFORTS_WITH_XHIGH
: DEFAULT_REASONING_EFFORTS;
}
if (isOpenRouterAnthropicAdaptiveReasoningModel(parsedModel, spec)) {
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
}
return inferFallbackEfforts(spec, compat);
}
function inferFallbackEfforts<TApi extends Api>(spec: ModelSpec<TApi>, compat: CompatOf<TApi>): readonly Effort[] {
if (spec.api === "anthropic-messages") {
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
}
if (spec.name.includes("deepseek-v4")) {
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
}
if (spec.api === "bedrock-converse-stream") {
return DEFAULT_REASONING_EFFORTS;
}
if (spec.api === "openai-completions") {
const resolved = compat as ResolvedOpenAICompat;
if (resolved.thinkingFormat === "openai" && resolved.supportsReasoningEffort) {
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
}
return DEFAULT_REASONING_EFFORTS;
}
// OpenAI Responses APIs encode discrete effort levels, including xhigh.
if (spec.api === "openai-responses" || spec.api === "openai-codex-responses") {
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
}
return DEFAULT_REASONING_EFFORTS;
}
function inferThinkingControlMode<TApi extends Api>(
spec: ModelSpec<TApi>,
parsedModel: ParsedModel,
): ThinkingConfig["mode"] {
switch (spec.api) {
case "google-generative-ai":
case "google-gemini-cli":
case "google-vertex":
return parsedModel.family === "gemini" &&
semverGte(parsedModel.version, "3.0") &&
parsedModel.version.major === 3
? "google-level"
: "budget";
case "anthropic-messages":
if (parsedModel.family === "anthropic") {
if (semverGte(parsedModel.version, "4.6")) {
return "anthropic-adaptive";
}
if (semverGte(parsedModel.version, "4.5")) {
return "anthropic-budget-effort";
}
}
return "budget";
case "bedrock-converse-stream":
if (parsedModel.family === "anthropic") {
if (
semverGte(parsedModel.version, "4.6") &&
(parsedModel.kind === "opus" || isFableOrMythos(parsedModel.kind))
) {
return "anthropic-adaptive";
}
if (semverGte(parsedModel.version, "4.5")) {
return "anthropic-budget-effort";
}
}
return "budget";
default:
return "effort";
}
}
function isOpenRouterAnthropicAdaptiveReasoningModel<TApi extends Api>(
parsedModel: AnthropicModel,
spec: ModelSpec<TApi>,
): boolean {
if (spec.api !== "openai-completions") return false;
if (!modelMatchesHost(spec, "openrouter")) return false;
return isFableOrMythos(parsedModel.kind) || (parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.6"));
}
/**
* Opus 4.7+ and Fable/Mythos on the Messages API expose the full five-tier
* adaptive scale (low/medium/high/xhigh/max). Bedrock Converse stays on the
* four-tier scale regardless of model version.
*/
function anthropicModelHasRealXHighEffort<TApi extends Api>(spec: ModelSpec<TApi>, parsedModel: ParsedModel): boolean {
if (spec.api !== "anthropic-messages") return false;
if (parsedModel.family !== "anthropic") return false;
if (isFableOrMythos(parsedModel.kind)) return true;
return parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.7");
}
// ---------------------------------------------------------------------------
// Runtime helpers (field reads only — safe per request)
// ---------------------------------------------------------------------------
/**
* Returns the supported thinking efforts declared on the model metadata.
*
* Catalog enrichment is responsible for normalizing bundled model metadata up front.
* Runtime callers must treat explicit `model.thinking` on custom models as authoritative
* so proxy-specific overrides from `models.yml` survive request construction.
*
* @throws Error when a reasoning-capable model is missing thinking metadata
* Empty for non-reasoning models and for reasoning models without a
* controllable effort surface (`thinking: undefined`).
*/
export function getSupportedEfforts<TApi extends Api>(model: ApiModel<TApi>): readonly Effort[] {
if (!model.reasoning) {
return [];
}
// Models that reason natively but reject the `reasoning.effort` wire param
// (xAI Grok off the GROK_EFFORT_CAPABLE_PREFIXES allowlist in
// providers/xai-responses.ts: grok-build, grok-4.20-0309-reasoning) hide the
// picker's effort dial. Scoped to openai-responses* by
// `modelOmitsReasoningEffort` — openai-completions has its own
// supportsReasoningEffort consultation at inferFallbackEfforts L536 and
// changing that path's semantics is out-of-scope.
if (modelOmitsReasoningEffort(model)) {
return [];
}
if (!model.thinking) {
throw new Error(`Model ${model.provider}/${model.id} is missing thinking metadata`);
}
return expandEffortRange(model.thinking);
return model.thinking?.efforts ?? [];
}
/**
@@ -271,11 +380,8 @@ export function requireSupportedEffort<TApi extends Api>(model: ApiModel<TApi>,
}
/** Maps a normalized thinking effort to Google's `thinkingLevel` enum values. */
export function mapEffortToGoogleThinkingLevel<TApi extends Api>(
model: ApiModel<TApi>,
effort: Effort,
): "MINIMAL" | "LOW" | "MEDIUM" | "HIGH" {
switch (requireSupportedEffort(model, effort)) {
export function mapEffortToGoogleThinkingLevel(effort: Effort): "MINIMAL" | "LOW" | "MEDIUM" | "HIGH" {
switch (effort) {
case Effort.Minimal:
return "MINIMAL";
case Effort.Low:
@@ -288,371 +394,14 @@ export function mapEffortToGoogleThinkingLevel<TApi extends Api>(
}
}
/** Maps a normalized thinking effort to Anthropic adaptive effort values. */
/**
* Maps a normalized thinking effort to Anthropic adaptive effort values via
* the model's baked `thinking.effortMap` (identity for unmapped efforts).
*/
export function mapEffortToAnthropicAdaptiveEffort<TApi extends Api>(
model: ApiModel<TApi>,
effort: Effort,
): "low" | "medium" | "high" | "xhigh" | "max" {
const supported = requireSupportedEffort(model, effort);
if (anthropicModelHasRealXHighEffort(model)) {
// Opus 4.7+ and Fable/Mythos 5 on the Messages API expose the full
// five-tier adaptive scale
// (low/medium/high/xhigh/max). Shift our user-facing efforts up one notch so
// the top tier reaches the genuine "max" and "high" lands on Anthropic's
// recommended "xhigh" coding/agentic default.
switch (supported) {
case Effort.Minimal:
return "low";
case Effort.Low:
return "medium";
case Effort.Medium:
return "high";
case Effort.High:
return "xhigh";
case Effort.XHigh:
return "max";
}
}
// Older adaptive models (Opus 4.6) and Bedrock Converse expose only four tiers
// with no real "xhigh"; XHigh is a legacy alias for the top "max" tier there.
switch (supported) {
case Effort.Minimal:
case Effort.Low:
return "low";
case Effort.Medium:
return "medium";
case Effort.High:
return "high";
case Effort.XHigh:
return "max";
}
}
/**
* Returns true for Anthropic models with Opus 4.7+/Fable/Mythos API restrictions:
* - Sampling parameters (temperature/top_p/top_k) return 400 error
* - Thinking content is omitted by default (needs display: "summarized")
*/
export function hasOpus47ApiRestrictions(modelId: string): boolean {
const parsed = parseAnthropicModel(bareModelId(modelId));
if (!parsed) return false;
return (parsed.kind === "opus" && semverGte(parsed.version, "4.7")) || isFableOrMythos(parsed.kind);
}
/**
* Mid-conversation `role: "system"` messages (system instructions appended at
* non-first positions in the `messages` array) are supported starting with
* Claude Opus 4.8 and the Claude Fable/Mythos 5 generation. Earlier Claude
* models reject the role.
* @see https://platform.claude.com/docs/en/build-with-claude/mid-conversation-system-messages
*/
export function supportsMidConversationSystemMessages(modelId: string): boolean {
const parsed = parseAnthropicModel(bareModelId(modelId));
if (!parsed) return false;
return (parsed.kind === "opus" && semverGte(parsed.version, "4.8")) || isFableOrMythos(parsed.kind);
}
export function isAnthropicFableOrMythosModel(modelId: string): boolean {
const parsed = parseAnthropicModel(bareModelId(modelId));
return parsed !== null && isFableOrMythos(parsed.kind);
}
function isOpenRouterAnthropicAdaptiveReasoningModel<TApi extends Api>(
parsedModel: AnthropicModel,
model: ApiModel<TApi>,
): boolean {
if (model.api !== "openai-completions") return false;
if (!modelMatchesHost(model, "openrouter")) return false;
return isFableOrMythos(parsedModel.kind) || (parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.6"));
}
function anthropicModelHasRealXHighEffort<TApi extends Api>(model: ApiModel<TApi>): boolean {
if (model.api !== "anthropic-messages") return false;
const parsedModel = parseKnownModel(model.id);
if (parsedModel.family !== "anthropic") return false;
if (isFableOrMythos(parsedModel.kind)) return true;
return parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.7");
}
function applyGeneratedModelPolicy(model: ApiModel<Api>): void {
const copilotLimits = model.provider === "github-copilot" ? COPILOT_GENERATED_LIMITS[model.id] : undefined;
if (copilotLimits) {
model.contextWindow = copilotLimits.contextWindow;
model.maxTokens = copilotLimits.maxTokens;
}
if (
model.api === "openai-completions" &&
(model.provider === "minimax-code" || model.provider === "minimax-code-cn")
) {
model.compat = {
...(model.compat ?? {}),
supportsStore: false,
supportsDeveloperRole: false,
supportsReasoningEffort: false,
reasoningContentField: "reasoning_content",
};
delete model.compat.thinkingFormat;
}
if (
model.api === "openai-completions" &&
model.provider === "opencode-go" &&
(model.id === "deepseek-v4-flash" || model.id === "deepseek-v4-pro")
) {
model.compat = {
...(model.compat ?? {}),
supportsToolChoice: false,
reasoningContentField: "reasoning_content",
requiresReasoningContentForToolCalls: true,
};
}
const parsedModel = parseKnownModel(model.id);
const applyPatchToolType = inferGeneratedApplyPatchToolType(model, parsedModel);
if (applyPatchToolType) {
model.applyPatchToolType = applyPatchToolType;
} else {
delete model.applyPatchToolType;
}
if (parsedModel.family === "anthropic") {
applyAnthropicCatalogPolicy(model, parsedModel);
}
if (parsedModel.family === "openai") {
applyOpenAICatalogPolicy(model, parsedModel);
}
}
function applyAnthropicCatalogPolicy(model: ApiModel<Api>, parsedModel: AnthropicModel): void {
// Claude Opus 4.5: models.dev reports 3x the correct cache pricing.
if (model.provider === "anthropic" && parsedModel.kind === "opus" && semverEqual(parsedModel.version, "4.5")) {
model.cost.cacheRead = 0.5;
model.cost.cacheWrite = 6.25;
}
// Bedrock Opus 4.6: upstream metadata is stale for cache pricing and context.
if (model.provider === "amazon-bedrock" && parsedModel.kind === "opus" && semverEqual(parsedModel.version, "4.6")) {
model.cost.cacheRead = 0.5;
model.cost.cacheWrite = 6.25;
model.contextWindow = 1000000;
model.maxTokens = 128000;
}
// Claude Fable/Mythos 5: Anthropic's /v1/models omits token limits and
// pricing, and models.dev lags new releases. Pin authoritative values from
// the model card (1M context / 128k output) and pricing docs ($10 in / $50
// out per MTok).
if (model.provider === "anthropic" && isFableOrMythos(parsedModel.kind)) {
model.contextWindow = 1_000_000;
model.maxTokens = 128_000;
model.cost.input = 10;
model.cost.output = 50;
model.cost.cacheRead = 1;
model.cost.cacheWrite = 12.5;
}
}
function inferGeneratedApplyPatchToolType(
model: ApiModel<Api>,
parsedModel: ParsedModel,
): ApiModel<Api>["applyPatchToolType"] {
if (parsedModel.family !== "openai" || parsedModel.version.major !== 5) {
return undefined;
}
if (model.provider === "openai" && model.api === "openai-responses") {
return "freeform";
}
if (model.provider === "openai-codex" && model.api === "openai-codex-responses") {
return "freeform";
}
return undefined;
}
function applyOpenAICatalogPolicy(model: ApiModel<Api>, parsedModel: OpenAIModel): void {
// Codex models: 400K figure includes output budget; input window is 272K.
if (parsedModel.variant.startsWith("codex") && parsedModel.variant !== "codex-spark") {
model.contextWindow = 272000;
return;
}
// GPT-5.4 mini/nano use plain OpenAI IDs on the Codex transport, but Codex still
// enforces the lower prompt budget for these variants. Codex discovery can also
// report inconsistent priorities for the GPT-5.4 family, so normalize by parsed
// variant instead of special-casing raw model ids.
if (model.api === "openai-codex-responses" && semverEqual(parsedModel.version, "5.4")) {
const normalizedPriority = CODEX_GPT_5_4_PRIORITY_BY_VARIANT[parsedModel.variant];
if (normalizedPriority !== undefined) {
model.priority = normalizedPriority;
}
if (parsedModel.variant === "mini" || parsedModel.variant === "nano") {
model.contextWindow = 272000;
}
}
}
function inferModelThinking<TApi extends Api>(model: ApiModel<TApi>): ThinkingConfig {
const parsedModel = parseKnownModel(model.id);
const efforts = inferSupportedEfforts(parsedModel, model);
const minLevel = efforts[0];
const maxLevel = efforts.at(-1);
if (!minLevel || !maxLevel) {
throw new Error(`Model ${model.provider}/${model.id} resolved to an empty thinking range`);
}
const config: ThinkingConfig = {
mode: inferThinkingControlMode(model, parsedModel),
minLevel,
maxLevel,
};
// Encode explicit levels only when the inferred set has gaps the min..max range cannot represent.
const minIndex = THINKING_EFFORTS.indexOf(minLevel);
const maxIndex = THINKING_EFFORTS.indexOf(maxLevel);
const expandedRange = THINKING_EFFORTS.slice(minIndex, maxIndex + 1);
if (expandedRange.length !== efforts.length) {
config.levels = efforts;
}
return config;
}
function normalizeThinkingConfig(thinking: ThinkingConfig | undefined): ThinkingConfig | undefined {
if (!thinking || expandEffortRange(thinking).length === 0) {
return undefined;
}
return thinking;
}
function thinkingsEqual(left: ThinkingConfig | undefined, right: ThinkingConfig | undefined): boolean {
if (left === right) return true;
if (!left || !right) return false;
if (left.mode !== right.mode || left.minLevel !== right.minLevel || left.maxLevel !== right.maxLevel) return false;
const leftLevels = left.levels;
const rightLevels = right.levels;
if (leftLevels === rightLevels) return true;
if (!leftLevels || !rightLevels) return false;
if (leftLevels.length !== rightLevels.length) return false;
return leftLevels.every((level, index) => level === rightLevels[index]);
}
function expandEffortRange(thinking: ThinkingConfig): readonly Effort[] {
if (thinking.levels && thinking.levels.length > 0) {
return thinking.levels;
}
const minIndex = THINKING_EFFORTS.indexOf(thinking.minLevel);
const maxIndex = THINKING_EFFORTS.indexOf(thinking.maxLevel);
if (minIndex === -1 || maxIndex === -1 || minIndex > maxIndex) {
return [];
}
return THINKING_EFFORTS.slice(minIndex, maxIndex + 1);
}
function inferSupportedEfforts<TApi extends Api>(parsedModel: ParsedModel, model: ApiModel<TApi>): readonly Effort[] {
switch (parsedModel.family) {
case "openai":
return inferOpenAISupportedEfforts(parsedModel);
case "gemini":
return inferGeminiSupportedEfforts(parsedModel);
case "anthropic":
return inferAnthropicSupportedEfforts(parsedModel, model);
case "unknown":
return inferFallbackEfforts(model);
}
}
function inferOpenAISupportedEfforts(model: OpenAIModel): readonly Effort[] {
if (model.variant === "codex-mini" && semverEqual(model.version, "5.1")) {
return GPT_5_1_CODEX_MINI_EFFORTS;
}
if (semverGte(model.version, "5.2")) {
return GPT_5_2_PLUS_EFFORTS;
}
return DEFAULT_REASONING_EFFORTS;
}
function inferGeminiSupportedEfforts(model: GeminiModel): readonly Effort[] {
if (!semverGte(model.version, "3.0")) {
return DEFAULT_REASONING_EFFORTS;
}
return model.kind === "pro" ? GEMINI_3_PRO_EFFORTS : GEMINI_3_FLASH_EFFORTS;
}
function inferAnthropicSupportedEfforts<TApi extends Api>(
parsedModel: AnthropicModel,
model: ApiModel<TApi>,
): readonly Effort[] {
if (
(model.api === "anthropic-messages" || model.api === "bedrock-converse-stream") &&
semverGte(parsedModel.version, "4.6")
) {
return parsedModel.kind === "opus" || isFableOrMythos(parsedModel.kind)
? DEFAULT_REASONING_EFFORTS_WITH_XHIGH
: DEFAULT_REASONING_EFFORTS;
}
if (isOpenRouterAnthropicAdaptiveReasoningModel(parsedModel, model)) {
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
}
return inferFallbackEfforts(model);
}
function inferFallbackEfforts<TApi extends Api>(model: ApiModel<TApi>): readonly Effort[] {
if (model.api === "anthropic-messages") {
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
}
if (model.name.includes("deepseek-v4")) {
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
}
if (model.api === "bedrock-converse-stream") {
return DEFAULT_REASONING_EFFORTS;
}
if (model.api === "openai-completions") {
const compat = buildOpenAICompat(model as ModelSpec<"openai-completions">);
if (compat.thinkingFormat === "openai" && compat.supportsReasoningEffort) {
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
}
return DEFAULT_REASONING_EFFORTS;
}
// OpenAI Responses APIs encode discrete effort levels, including xhigh.
if (model.api === "openai-responses" || model.api === "openai-codex-responses") {
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
}
return DEFAULT_REASONING_EFFORTS;
}
function inferThinkingControlMode<TApi extends Api>(
model: ApiModel<TApi>,
parsedModel: ParsedModel,
): ThinkingConfig["mode"] {
switch (model.api) {
case "google-generative-ai":
case "google-gemini-cli":
case "google-vertex":
return parsedModel.family === "gemini" &&
semverGte(parsedModel.version, "3.0") &&
parsedModel.version.major === 3
? "google-level"
: "budget";
case "anthropic-messages":
if (parsedModel.family === "anthropic") {
if (semverGte(parsedModel.version, "4.6")) {
return "anthropic-adaptive";
}
if (semverGte(parsedModel.version, "4.5")) {
return "anthropic-budget-effort";
}
}
return "budget";
case "bedrock-converse-stream":
if (parsedModel.family === "anthropic") {
if (
semverGte(parsedModel.version, "4.6") &&
(parsedModel.kind === "opus" || isFableOrMythos(parsedModel.kind))
) {
return "anthropic-adaptive";
}
if (semverGte(parsedModel.version, "4.5")) {
return "anthropic-budget-effort";
}
}
return "budget";
default:
return "effort";
}
return (model.thinking?.effortMap?.[supported] ?? supported) as "low" | "medium" | "high" | "xhigh" | "max";
}
File diff suppressed because it is too large Load Diff
@@ -60,11 +60,7 @@ function getThinkingConfig(capabilities: string[] | undefined): ThinkingConfig |
if (!capabilities?.includes("thinking")) {
return undefined;
}
return {
mode: "effort",
minLevel: Effort.Minimal,
maxLevel: Effort.High,
};
return { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] };
}
async function fetchShowMetadata(
baseUrl: string,
@@ -151,9 +151,8 @@ function buildAnthropicReferenceMap(
* Seeded into model generation so the bundled catalog is never gated on
* models.dev's update cadence; deduped behind upstream catalog / models.dev
* entries once those appear. Token limits and pricing are pinned
* authoritatively in
* `applyAnthropicCatalogPolicy`, and `thinking` is derived by
* `refreshModelThinking` during generation.
* authoritatively in `applyAnthropicCatalogPolicy`, and `thinking` is re-baked
* by the generator's policy pass (scripts/generated-policies.ts).
*/
export const ANTHROPIC_CURATED_FALLBACK_MODELS: readonly ModelSpec<"anthropic-messages">[] = [
{
@@ -343,11 +342,7 @@ function getOllamaThinkingConfig(capabilities: string[] | undefined): ThinkingCo
if (!capabilities?.includes("thinking")) {
return undefined;
}
return {
mode: "effort",
minLevel: Effort.Minimal,
maxLevel: Effort.High,
};
return { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] };
}
/**
@@ -2076,7 +2071,7 @@ export function moonshotModelManagerOptions(
thinking:
model.thinking ??
(isKimiK2Reasoning
? { mode: "effort", minLevel: Effort.Minimal, maxLevel: Effort.High }
? { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] }
: undefined),
};
},
+25 -12
View File
@@ -26,20 +26,27 @@ export type ThinkingControlMode =
/** Per-model thinking capabilities used to clamp and map user-facing effort levels. */
export interface ThinkingConfig {
/** Least intensive supported user-facing effort level. */
minLevel: Effort;
/** Most intensive supported user-facing effort level. */
maxLevel: Effort;
/**
* Optional explicit list of supported levels. When present, takes precedence over
* the `minLevel`..`maxLevel` range — used to encode discrete sets with gaps
* (e.g. Gemini 3 Pro supports `low` and `high` but not `medium`).
*/
levels?: readonly Effort[];
/** Optional default effort applied when this model is selected. Falls back to global default if absent. */
defaultLevel?: Effort;
/** Provider-specific transport used to encode the selected effort. */
mode: ThinkingControlMode;
/**
* Supported user-facing efforts, ordered least → most intensive. Never
* empty: a reasoning model without a controllable effort surface carries
* `thinking: undefined` instead of an empty list.
*/
efforts: readonly Effort[];
/** Optional default effort applied when this model is selected. Falls back to global default if absent. */
defaultLevel?: Effort;
/**
* Effort → wire-value remap for `anthropic-adaptive` transports, baked at
* build time (4-tier legacy scale vs the 5-tier Opus 4.7+/Fable/Mythos
* scale). Identity for efforts the map omits.
*/
effortMap?: Partial<Record<Effort, string>>;
/**
* Adaptive thinking accepts the `display` field (Opus 4.7+, Fable/Mythos
* 5). Also implies native interleaved thinking — no beta header needed.
*/
supportsDisplay?: boolean;
}
// `Provider` is any provider-id string; `KnownProvider` (re-exported above) enumerates
@@ -246,6 +253,12 @@ export interface AnthropicCompat {
* When unset, auto-detected from the model id. Default: true.
*/
supportsForcedToolChoice?: boolean;
/**
* Whether the model accepts sampling parameters (`temperature`, `top_p`,
* `top_k`). Opus 4.7+ and Fable/Mythos reject them with a 400. When unset,
* auto-detected from the model id. Default: true.
*/
supportsSamplingParams?: boolean;
/**
* Include a non-standard `id` field (aliasing `tool_use_id`) on
* `tool_result` blocks. Z.AI's Anthropic-compatible proxy deserializes
@@ -0,0 +1,185 @@
import { describe, expect, it } from "bun:test";
import { Effort } from "@oh-my-pi/pi-catalog/effort";
import type { Api, ModelSpec, Provider } from "@oh-my-pi/pi-catalog/types";
import { applyGeneratedModelPolicies, linkOpenAIPromotionTargets } from "../scripts/generated-policies";
function createSpec<TApi extends Api>(overrides: {
id: string;
api: TApi;
provider: Provider;
reasoning?: boolean;
contextWindow?: number;
maxTokens?: number;
priority?: number;
applyPatchToolType?: "freeform" | "function";
cost?: ModelSpec<TApi>["cost"];
thinking?: ModelSpec<TApi>["thinking"];
}): ModelSpec<TApi> {
return {
id: overrides.id,
name: overrides.id,
api: overrides.api,
provider: overrides.provider,
baseUrl: "https://example.com",
reasoning: overrides.reasoning ?? true,
thinking: overrides.thinking,
input: ["text"],
cost: overrides.cost ?? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: overrides.contextWindow ?? 200000,
maxTokens: overrides.maxTokens ?? 32000,
priority: overrides.priority,
applyPatchToolType: overrides.applyPatchToolType,
};
}
describe("generated model policies", () => {
it("re-bakes thinking metadata and applies parsed catalog corrections", () => {
const models: ModelSpec<Api>[] = [
createSpec({
id: "claude-opus-4-5",
api: "anthropic-messages",
provider: "anthropic",
// Stale baked metadata must be replaced by the deriver's output.
thinking: { mode: "budget", efforts: [Effort.High] },
cost: { input: 0, output: 0, cacheRead: 1.5, cacheWrite: 18.75 },
contextWindow: 1000000,
}),
createSpec({
id: "anthropic.claude-opus-4-6-v1:0",
api: "bedrock-converse-stream",
provider: "amazon-bedrock",
cost: { input: 0, output: 0, cacheRead: 1.5, cacheWrite: 18.75 },
contextWindow: 1000000,
}),
createSpec({
id: "gpt-5.2-codex",
api: "openai-codex-responses",
provider: "openai-codex",
contextWindow: 400000,
}),
createSpec({
id: "gpt-5.4-mini",
api: "openai-codex-responses",
provider: "openai-codex",
contextWindow: 400000,
priority: 2,
}),
];
applyGeneratedModelPolicies(models);
expect(models[0]?.thinking).toEqual({
mode: "anthropic-budget-effort",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
});
expect(models[0]?.cost.cacheRead).toBe(0.5);
expect(models[0]?.cost.cacheWrite).toBe(6.25);
expect(models[1]?.thinking).toEqual({
mode: "anthropic-adaptive",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
effortMap: { minimal: "low", xhigh: "max" },
});
expect(models[1]?.cost.cacheRead).toBe(0.5);
expect(models[1]?.cost.cacheWrite).toBe(6.25);
expect(models[1]?.contextWindow).toBe(1000000);
expect(models[2]?.contextWindow).toBe(272000);
expect(models[3]?.contextWindow).toBe(272000);
expect(models[3]?.priority).toBe(1);
});
it("pins Claude Mythos 5 first-party Anthropic catalog metadata", () => {
const models: ModelSpec<Api>[] = [
createSpec({
id: "claude-mythos-5",
api: "anthropic-messages",
provider: "anthropic",
}),
];
applyGeneratedModelPolicies(models);
expect(models[0]?.contextWindow).toBe(1_000_000);
expect(models[0]?.maxTokens).toBe(128_000);
expect(models[0]?.cost).toEqual({ input: 10, output: 50, cacheRead: 1, cacheWrite: 12.5 });
expect(models[0]?.thinking).toEqual({
mode: "anthropic-adaptive",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
effortMap: { minimal: "low", low: "medium", medium: "high", high: "xhigh", xhigh: "max" },
supportsDisplay: true,
});
});
it("normalizes Copilot generated fallback limits", () => {
const models: ModelSpec<Api>[] = [
createSpec({
id: "claude-opus-4.6",
api: "anthropic-messages",
provider: "github-copilot",
contextWindow: 144000,
maxTokens: 64000,
}),
createSpec({
id: "gpt-5.4-mini",
api: "openai-responses",
provider: "github-copilot",
contextWindow: 400000,
maxTokens: 128000,
}),
createSpec({
id: "grok-code-fast-1",
api: "openai-completions",
provider: "github-copilot",
contextWindow: 128000,
maxTokens: 64000,
}),
];
applyGeneratedModelPolicies(models);
expect(models[0]?.contextWindow).toBe(168000);
expect(models[0]?.maxTokens).toBe(32000);
expect(models[1]?.contextWindow).toBe(272000);
expect(models[1]?.maxTokens).toBe(128000);
expect(models[2]?.contextWindow).toBe(192000);
expect(models[2]?.maxTokens).toBe(64000);
});
it("links spark variants and gpt-5.5 to their context promotion targets", () => {
const models = [
createSpec({ id: "gpt-5.3-codex-spark", api: "openai-codex-responses", provider: "openai-codex" }),
createSpec({ id: "gpt-5.5", api: "openai-codex-responses", provider: "openai-codex" }),
createSpec({ id: "gpt-5.4", api: "openai-codex-responses", provider: "openai-codex" }),
];
linkOpenAIPromotionTargets(models);
expect(models[0]?.contextPromotionTarget).toBe("openai-codex/gpt-5.5");
expect(models[1]?.contextPromotionTarget).toBe("openai-codex/gpt-5.4");
});
it("sets freeform apply_patch metadata for first-party GPT-5 Responses models", () => {
const models: ModelSpec<Api>[] = [
createSpec({ id: "gpt-5.4", api: "openai-responses", provider: "openai" }),
createSpec({ id: "gpt-5.3-codex-spark", api: "openai-codex-responses", provider: "openai-codex" }),
createSpec({
id: "gpt-5.3-codex-spark",
api: "openai-responses",
provider: "opencode",
applyPatchToolType: "freeform",
}),
createSpec({
id: "gpt-5.4",
api: "openai-completions",
provider: "litellm",
applyPatchToolType: "freeform",
}),
];
applyGeneratedModelPolicies(models);
expect(models[0]?.applyPatchToolType).toBe("freeform");
expect(models[1]?.applyPatchToolType).toBe("freeform");
expect(models[2]?.applyPatchToolType).toBeUndefined();
expect(models[3]?.applyPatchToolType).toBeUndefined();
});
});
@@ -198,8 +198,7 @@ describe("github copilot model limits mapping", () => {
expect(model?.premiumMultiplier).toBe(0.33);
expect(model?.thinking).toEqual({
mode: "effort",
minLevel: Effort.Low,
maxLevel: Effort.XHigh,
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
});
});
@@ -37,7 +37,11 @@ describe("supportsAdaptiveThinkingDisplay", () => {
expect(supportsAdaptiveThinkingDisplay("claude-fable-5")).toBe(true);
expect(supportsAdaptiveThinkingDisplay("claude-opus-4-7")).toBe(true);
expect(supportsAdaptiveThinkingDisplay("claude-opus-5-0")).toBe(true);
// Dotted and dashed version separators are equivalent.
expect(supportsAdaptiveThinkingDisplay("claude-opus-4.7")).toBe(true);
expect(supportsAdaptiveThinkingDisplay("anthropic/claude-opus-4.8")).toBe(true);
expect(supportsAdaptiveThinkingDisplay("claude-opus-4-6")).toBe(false);
expect(supportsAdaptiveThinkingDisplay("claude-opus-4.6")).toBe(false);
expect(supportsAdaptiveThinkingDisplay("claude-opus-4-20250514")).toBe(false);
expect(supportsAdaptiveThinkingDisplay("claude-sonnet-4-6")).toBe(false);
});
@@ -131,7 +131,10 @@ describe("issue #2113 — moonshot kimi-k2.6 discovery and wire format", () => {
const k26 = byId.get("kimi-k2.6");
expect(k26?.reasoning).toBe(true);
expect(k26?.input).toEqual(["text", "image"]);
expect(k26?.thinking).toEqual({ mode: "effort", minLevel: Effort.Minimal, maxLevel: Effort.High });
expect(k26?.thinking).toEqual({
mode: "effort",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
});
const thinkingOnly = byId.get("kimi-k2-thinking");
expect(thinkingOnly?.reasoning).toBe(true);
+145 -342
View File
@@ -1,29 +1,33 @@
import { describe, expect, it } from "bun:test";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { Effort } from "@oh-my-pi/pi-catalog/effort";
import {
applyGeneratedModelPolicies,
clampThinkingLevelForModel,
enrichModelThinking,
linkOpenAIPromotionTargets,
getSupportedEfforts,
mapEffortToAnthropicAdaptiveEffort,
mapEffortToGoogleThinkingLevel,
requireSupportedEffort,
} from "@oh-my-pi/pi-catalog/model-thinking";
import type { Api, ModelSpec, Provider } from "@oh-my-pi/pi-catalog/types";
import type { Api, Model, ModelSpec, Provider } from "@oh-my-pi/pi-catalog/types";
function createModel<TApi extends Api>(overrides: {
id: string;
api: TApi;
provider: Provider;
reasoning?: boolean;
}): ModelSpec<TApi> {
return enrichModelThinking({
baseUrl?: string;
compat?: ModelSpec<TApi>["compat"];
thinking?: ModelSpec<TApi>["thinking"];
}): Model<TApi> {
return buildModel({
id: overrides.id,
name: overrides.id,
api: overrides.api,
provider: overrides.provider,
baseUrl: "",
baseUrl: overrides.baseUrl ?? "",
reasoning: overrides.reasoning ?? true,
compat: overrides.compat,
thinking: overrides.thinking,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200000,
@@ -31,7 +35,7 @@ function createModel<TApi extends Api>(overrides: {
});
}
describe("model thinking metadata", () => {
describe("model thinking derivation", () => {
it("stores supported efforts for Codex mini in model metadata", () => {
const model = createModel({
id: "gpt-5.1-codex-mini",
@@ -41,8 +45,7 @@ describe("model thinking metadata", () => {
expect(model.thinking).toEqual({
mode: "effort",
minLevel: Effort.Medium,
maxLevel: Effort.High,
efforts: [Effort.Medium, Effort.High],
});
expect(() => requireSupportedEffort(model, Effort.Low)).toThrow(/Supported efforts: medium, high/);
expect(() => requireSupportedEffort(model, Effort.XHigh)).toThrow(/Supported efforts: medium, high/);
@@ -57,13 +60,12 @@ describe("model thinking metadata", () => {
expect(model.thinking).toEqual({
mode: "effort",
minLevel: Effort.Low,
maxLevel: Effort.XHigh,
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
});
expect(requireSupportedEffort(model, Effort.XHigh)).toBe(Effort.XHigh);
});
it("maps Gemini 3 Pro only for supported levels", () => {
it("encodes the Gemini 3 Pro effort gap directly in efforts", () => {
const model = createModel({
id: "gemini-3-pro-preview",
api: "google-generative-ai",
@@ -72,46 +74,25 @@ describe("model thinking metadata", () => {
expect(model.thinking).toEqual({
mode: "google-level",
minLevel: Effort.Low,
maxLevel: Effort.High,
levels: [Effort.Low, Effort.High],
efforts: [Effort.Low, Effort.High],
});
expect(mapEffortToGoogleThinkingLevel(model, Effort.Low)).toBe("LOW");
expect(mapEffortToGoogleThinkingLevel(model, Effort.High)).toBe("HIGH");
expect(() => mapEffortToGoogleThinkingLevel(model, Effort.Medium)).toThrow(/not supported/);
expect(mapEffortToGoogleThinkingLevel(Effort.Low)).toBe("LOW");
expect(mapEffortToGoogleThinkingLevel(Effort.High)).toBe("HIGH");
expect(mapEffortToGoogleThinkingLevel(Effort.XHigh)).toBe("HIGH");
expect(() => requireSupportedEffort(model, Effort.Medium)).toThrow(/not supported/);
});
it("encodes anthropic transport mode in metadata", () => {
const opus45 = createModel({
id: "claude-opus-4-5",
api: "anthropic-messages",
provider: "anthropic",
});
const opus46 = createModel({
id: "claude-opus-4.6",
api: "anthropic-messages",
provider: "anthropic",
});
const opus47 = createModel({
id: "claude-opus-4.7",
api: "anthropic-messages",
provider: "anthropic",
});
it("encodes anthropic transport mode and adaptive wire maps in metadata", () => {
const opus45 = createModel({ id: "claude-opus-4-5", api: "anthropic-messages", provider: "anthropic" });
const opus46 = createModel({ id: "claude-opus-4.6", api: "anthropic-messages", provider: "anthropic" });
const opus47 = createModel({ id: "claude-opus-4.7", api: "anthropic-messages", provider: "anthropic" });
const opus47Bedrock = createModel({
id: "us.anthropic.claude-opus-4-7",
api: "bedrock-converse-stream",
provider: "amazon-bedrock",
});
const sonnet46 = createModel({
id: "claude-sonnet-4.6",
api: "anthropic-messages",
provider: "anthropic",
});
const mythos = createModel({
id: "claude-mythos-5",
api: "anthropic-messages",
provider: "anthropic",
});
const sonnet46 = createModel({ id: "claude-sonnet-4.6", api: "anthropic-messages", provider: "anthropic" });
const mythos = createModel({ id: "claude-mythos-5", api: "anthropic-messages", provider: "anthropic" });
const mythosBedrock = createModel({
id: "global.anthropic.claude-mythos-5",
api: "bedrock-converse-stream",
@@ -121,275 +102,124 @@ describe("model thinking metadata", () => {
expect(opus45.thinking?.mode).toBe("anthropic-budget-effort");
expect(opus46.thinking?.mode).toBe("anthropic-adaptive");
expect(sonnet46.thinking?.mode).toBe("anthropic-adaptive");
expect(opus46.thinking).toEqual({
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.XHigh,
});
expect(sonnet46.thinking).toEqual({
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.High,
});
expect(mythos.thinking).toEqual({
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.XHigh,
});
expect(mythosBedrock.thinking?.mode).toBe("anthropic-adaptive");
// Opus 4.6 has no real xhigh level — pi-ai aliases XHigh to Anthropic's "max".
// Opus 4.6 has no real xhigh level — the baked 4-tier map aliases XHigh to "max".
expect(opus46.thinking?.effortMap).toEqual({ minimal: "low", xhigh: "max" });
expect(mapEffortToAnthropicAdaptiveEffort(opus46, Effort.XHigh)).toBe("max");
// Opus 4.7+ on the Messages API exposes the full five-tier scale, so pi-ai
// shifts each user-facing effort up one notch and the top tier reaches "max".
// Opus 4.7+ on the Messages API exposes the full five-tier scale: the baked
// map shifts each user-facing effort up one notch so the top tier reaches "max".
expect(opus47.thinking?.effortMap).toEqual({
minimal: "low",
low: "medium",
medium: "high",
high: "xhigh",
xhigh: "max",
});
expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.Minimal)).toBe("low");
expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.Low)).toBe("medium");
expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.Medium)).toBe("high");
expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.High)).toBe("xhigh");
expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.XHigh)).toBe("max");
expect(mapEffortToAnthropicAdaptiveEffort(mythos, Effort.High)).toBe("xhigh");
expect(mapEffortToAnthropicAdaptiveEffort(mythos, Effort.XHigh)).toBe("max");
expect(mapEffortToAnthropicAdaptiveEffort(mythosBedrock, Effort.XHigh)).toBe("max");
// Bedrock Converse keeps the four-tier legacy mapping; xhigh aliases to "max".
expect(opus47Bedrock.thinking?.effortMap).toEqual({ minimal: "low", xhigh: "max" });
expect(mapEffortToAnthropicAdaptiveEffort(opus47Bedrock, Effort.High)).toBe("high");
expect(mapEffortToAnthropicAdaptiveEffort(opus47Bedrock, Effort.XHigh)).toBe("max");
expect(() => mapEffortToAnthropicAdaptiveEffort(sonnet46, Effort.XHigh)).toThrow(/not supported/);
});
});
describe("generated model policies", () => {
it("refreshes thinking metadata and applies parsed catalog corrections", () => {
const models: ModelSpec<Api>[] = [
{
id: "claude-opus-4-5",
name: "Claude Opus 4.5",
api: "anthropic-messages",
provider: "anthropic",
baseUrl: "https://example.com",
reasoning: true,
thinking: {
mode: "budget",
minLevel: Effort.High,
maxLevel: Effort.High,
},
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 1.5, cacheWrite: 18.75 },
contextWindow: 1000000,
maxTokens: 32000,
},
{
id: "anthropic.claude-opus-4-6-v1:0",
name: "Claude Opus 4.6",
api: "bedrock-converse-stream",
provider: "amazon-bedrock",
baseUrl: "https://example.com",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 1.5, cacheWrite: 18.75 },
contextWindow: 1000000,
maxTokens: 32000,
},
{
id: "gpt-5.2-codex",
name: "GPT-5.2 Codex",
api: "openai-codex-responses",
provider: "openai-codex",
baseUrl: "https://example.com",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 400000,
maxTokens: 32000,
},
{
id: "gpt-5.4-mini",
name: "GPT-5.4 mini",
api: "openai-codex-responses",
provider: "openai-codex",
baseUrl: "https://example.com",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 400000,
maxTokens: 32000,
priority: 2,
},
];
applyGeneratedModelPolicies(models);
expect(models[0]?.thinking).toEqual({
mode: "anthropic-budget-effort",
minLevel: Effort.Minimal,
maxLevel: Effort.XHigh,
it("bakes adaptive display support for Opus 4.7+ and Fable/Mythos 5", () => {
const opus46 = createModel({ id: "claude-opus-4.6", api: "anthropic-messages", provider: "anthropic" });
const opus47 = createModel({ id: "claude-opus-4-7", api: "anthropic-messages", provider: "anthropic" });
// Dotted and dashed version forms are equivalent; bare dated ids stay Opus 4.0.
const opus47Dotted = createModel({ id: "claude-opus-4.7", api: "anthropic-messages", provider: "anthropic" });
const opus4Dated = createModel({
id: "claude-opus-4-20250514",
api: "anthropic-messages",
provider: "anthropic",
});
expect(models[0]?.cost.cacheRead).toBe(0.5);
expect(models[0]?.cost.cacheWrite).toBe(6.25);
expect(models[1]?.thinking).toEqual({
const fable = createModel({ id: "claude-fable-5", api: "anthropic-messages", provider: "anthropic" });
const fableBedrock = createModel({
id: "global.anthropic.claude-fable-5",
api: "bedrock-converse-stream",
provider: "amazon-bedrock",
});
expect(opus46.thinking?.supportsDisplay).toBeUndefined();
expect(opus47.thinking?.supportsDisplay).toBe(true);
expect(opus47Dotted.thinking?.supportsDisplay).toBe(true);
expect(opus4Dated.thinking?.supportsDisplay).toBeUndefined();
expect(fable.thinking?.supportsDisplay).toBe(true);
expect(fableBedrock.thinking?.supportsDisplay).toBe(true);
});
it("backfills wire facts onto explicit thinking, explicit values winning", () => {
// Authored capability surface (mode/efforts) keeps identity-derived wire
// facts: configs never need to know Anthropic's tier tables.
const filled = createModel({
id: "claude-opus-4-8",
api: "anthropic-messages",
provider: "anthropic",
thinking: { mode: "anthropic-adaptive", efforts: [Effort.Low, Effort.High] },
});
expect(filled.thinking).toEqual({
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.XHigh,
efforts: [Effort.Low, Effort.High],
effortMap: { minimal: "low", low: "medium", medium: "high", high: "xhigh", xhigh: "max" },
supportsDisplay: true,
});
expect(models[1]?.cost.cacheRead).toBe(0.5);
expect(models[1]?.cost.cacheWrite).toBe(6.25);
expect(models[1]?.contextWindow).toBe(1000000);
expect(models[2]?.contextWindow).toBe(272000);
expect(models[3]?.contextWindow).toBe(272000);
expect(models[3]?.priority).toBe(1);
});
it("pins Claude Mythos 5 first-party Anthropic catalog metadata", () => {
const models: ModelSpec<Api>[] = [
{
id: "claude-mythos-5",
name: "Claude Mythos 5",
api: "anthropic-messages",
provider: "anthropic",
baseUrl: "https://example.com",
reasoning: true,
input: ["text", "image"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200000,
maxTokens: 32000,
// Explicit wire facts are authoritative — including `false`.
const pinned = createModel({
id: "claude-opus-4-8",
api: "anthropic-messages",
provider: "anthropic",
thinking: {
mode: "anthropic-adaptive",
efforts: [Effort.Low, Effort.High],
effortMap: { xhigh: "max" },
supportsDisplay: false,
},
];
applyGeneratedModelPolicies(models);
expect(models[0]?.contextWindow).toBe(1_000_000);
expect(models[0]?.maxTokens).toBe(128_000);
expect(models[0]?.cost).toEqual({ input: 10, output: 50, cacheRead: 1, cacheWrite: 12.5 });
expect(models[0]?.thinking).toEqual({
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.XHigh,
});
expect(pinned.thinking?.effortMap).toEqual({ xhigh: "max" });
expect(pinned.thinking?.supportsDisplay).toBe(false);
});
it("normalizes Copilot generated fallback limits", () => {
const models: ModelSpec<Api>[] = [
{
...createModel({
id: "claude-opus-4.6",
api: "anthropic-messages",
provider: "github-copilot",
}),
contextWindow: 144000,
maxTokens: 64000,
},
{
...createModel({
id: "gpt-5.4-mini",
api: "openai-responses",
provider: "github-copilot",
}),
contextWindow: 400000,
maxTokens: 128000,
},
{
...createModel({
id: "grok-code-fast-1",
api: "openai-completions",
provider: "github-copilot",
}),
contextWindow: 128000,
maxTokens: 64000,
},
];
it("bakes sampling-param rejection into anthropic compat", () => {
const sonnet45 = createModel({ id: "claude-sonnet-4-5", api: "anthropic-messages", provider: "anthropic" });
const opus47 = createModel({ id: "claude-opus-4.7", api: "anthropic-messages", provider: "anthropic" });
const fable = createModel({ id: "claude-fable-5", api: "anthropic-messages", provider: "anthropic" });
applyGeneratedModelPolicies(models);
expect(models[0]?.contextWindow).toBe(168000);
expect(models[0]?.maxTokens).toBe(32000);
expect(models[1]?.contextWindow).toBe(272000);
expect(models[1]?.maxTokens).toBe(128000);
expect(models[2]?.contextWindow).toBe(192000);
expect(models[2]?.maxTokens).toBe(64000);
expect(sonnet45.compat.supportsSamplingParams).toBe(true);
expect(opus47.compat.supportsSamplingParams).toBe(false);
expect(fable.compat.supportsSamplingParams).toBe(false);
});
it("links spark variants and gpt-5.5 to their context promotion targets", () => {
const models = [
createModel({
id: "gpt-5.3-codex-spark",
api: "openai-codex-responses",
provider: "openai-codex",
}),
createModel({
id: "gpt-5.5",
api: "openai-codex-responses",
provider: "openai-codex",
}),
createModel({
id: "gpt-5.4",
api: "openai-codex-responses",
provider: "openai-codex",
}),
];
it("encodes effort-dial-less reasoners as thinking: undefined", () => {
const model = createModel({
id: "grok-build",
api: "openai-responses",
provider: "xai-oauth",
compat: { supportsReasoningEffort: false },
});
linkOpenAIPromotionTargets(models);
expect(models[0]?.contextPromotionTarget).toBe("openai-codex/gpt-5.5");
expect(models[1]?.contextPromotionTarget).toBe("openai-codex/gpt-5.4");
});
it("sets freeform apply_patch metadata for first-party GPT-5 Responses models", () => {
const models: ModelSpec<Api>[] = [
createModel({
id: "gpt-5.4",
api: "openai-responses",
provider: "openai",
}),
createModel({
id: "gpt-5.3-codex-spark",
api: "openai-codex-responses",
provider: "openai-codex",
}),
{
...createModel({
id: "gpt-5.3-codex-spark",
api: "openai-responses",
provider: "opencode",
}),
applyPatchToolType: "freeform",
},
{
...createModel({
id: "gpt-5.4",
api: "openai-completions",
provider: "litellm",
}),
applyPatchToolType: "freeform",
},
];
applyGeneratedModelPolicies(models);
expect(models[0]?.applyPatchToolType).toBe("freeform");
expect(models[1]?.applyPatchToolType).toBe("freeform");
expect(models[2]?.applyPatchToolType).toBeUndefined();
expect(models[3]?.applyPatchToolType).toBeUndefined();
expect(model.reasoning).toBe(true);
expect(model.thinking).toBeUndefined();
expect(getSupportedEfforts(model)).toEqual([]);
expect(clampThinkingLevelForModel(model, Effort.High)).toBeUndefined();
});
});
describe("model thinking runtime helpers", () => {
it("clamps from explicit metadata instead of inferring from model id", () => {
const model: ModelSpec<"openai-codex-responses"> = {
const model = createModel({
id: "custom-reasoner",
name: "Custom Reasoner",
api: "openai-codex-responses",
provider: "custom",
baseUrl: "https://example.com",
reasoning: true,
thinking: {
mode: "effort",
minLevel: Effort.Medium,
maxLevel: Effort.High,
},
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200000,
maxTokens: 32000,
};
thinking: { mode: "effort", efforts: [Effort.Medium, Effort.High] },
});
expect(model.thinking).toEqual({ mode: "effort", efforts: [Effort.Medium, Effort.High] });
expect(clampThinkingLevelForModel(model, Effort.Minimal)).toBe(Effort.Medium);
expect(clampThinkingLevelForModel(model, Effort.XHigh)).toBe(Effort.High);
expect(clampThinkingLevelForModel(model, Effort.High)).toBe(Effort.High);
@@ -413,32 +243,22 @@ describe("model thinking runtime helpers", () => {
provider: "custom",
});
// openai-completions should support xhigh by default
expect(model.thinking?.maxLevel).toBe(Effort.XHigh);
expect(model.thinking?.efforts.at(-1)).toBe(Effort.XHigh);
expect(requireSupportedEffort(model, Effort.XHigh)).toBe(Effort.XHigh);
});
it("does not expose xhigh for binary-thinking openai-compat transports", () => {
const model = enrichModelThinking({
const model = createModel({
id: "glm-4.7",
name: "GLM-4.7",
api: "openai-completions",
provider: "zai",
baseUrl: "https://api.z.ai/v1",
reasoning: true,
compat: {
thinkingFormat: "zai",
},
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128000,
maxTokens: 32000,
} satisfies ModelSpec<"openai-completions">);
compat: { thinkingFormat: "zai" },
});
expect(model.thinking).toEqual({
mode: "effort",
minLevel: Effort.Minimal,
maxLevel: Effort.High,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
});
expect(requireSupportedEffort(model, Effort.High)).toBe(Effort.High);
expect(() => requireSupportedEffort(model, Effort.XHigh)).toThrow(
@@ -447,26 +267,17 @@ describe("model thinking runtime helpers", () => {
});
it("derives binary-thinking fallback from resolved compat when catalog compat is partial", () => {
const model = enrichModelThinking({
const model = createModel({
id: "qwen/qwen3-32b",
name: "Qwen 3 32B",
api: "openai-completions",
provider: "openrouter",
baseUrl: "https://openrouter.ai/api/v1",
reasoning: true,
compat: {
supportsToolChoice: true,
},
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128000,
maxTokens: 32000,
} satisfies ModelSpec<"openai-completions">);
compat: { supportsToolChoice: true },
});
expect(model.thinking).toEqual({
mode: "effort",
minLevel: Effort.Minimal,
maxLevel: Effort.High,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
});
expect(requireSupportedEffort(model, Effort.High)).toBe(Effort.High);
expect(() => requireSupportedEffort(model, Effort.XHigh)).toThrow(
@@ -491,34 +302,24 @@ describe("model thinking runtime helpers", () => {
provider: "openrouter",
});
expect(fable.thinking?.maxLevel).toBe(Effort.XHigh);
expect(opus46.thinking?.maxLevel).toBe(Effort.XHigh);
expect(sonnet46.thinking?.maxLevel).toBe(Effort.High);
expect(fable.thinking?.efforts.at(-1)).toBe(Effort.XHigh);
expect(opus46.thinking?.efforts.at(-1)).toBe(Effort.XHigh);
expect(sonnet46.thinking?.efforts.at(-1)).toBe(Effort.High);
expect(requireSupportedEffort(fable, Effort.XHigh)).toBe(Effort.XHigh);
});
it("enables xhigh for openai-responses and openai-codex-responses APIs", () => {
const responsesModel = createModel({
id: "custom-responses",
api: "openai-responses",
provider: "custom",
});
const responsesModel = createModel({ id: "custom-responses", api: "openai-responses", provider: "custom" });
const codexModel = createModel({ id: "custom-codex", api: "openai-codex-responses", provider: "custom" });
const codexModel = createModel({
id: "custom-codex",
api: "openai-codex-responses",
provider: "custom",
});
// Both should support xhigh
expect(responsesModel.thinking?.maxLevel).toBe(Effort.XHigh);
expect(codexModel.thinking?.maxLevel).toBe(Effort.XHigh);
expect(responsesModel.thinking?.efforts.at(-1)).toBe(Effort.XHigh);
expect(codexModel.thinking?.efforts.at(-1)).toBe(Effort.XHigh);
expect(requireSupportedEffort(responsesModel, Effort.XHigh)).toBe(Effort.XHigh);
expect(requireSupportedEffort(codexModel, Effort.XHigh)).toBe(Effort.XHigh);
});
it("rejects reasoning models that are missing thinking metadata at runtime", () => {
const model = {
it("rejects effort requests against un-built reasoning specs", () => {
const spec = {
id: "broken-reasoner",
name: "Broken Reasoner",
api: "openai-responses",
@@ -531,28 +332,30 @@ describe("model thinking runtime helpers", () => {
maxTokens: 32000,
} as ModelSpec<"openai-responses">;
expect(() => requireSupportedEffort(model, Effort.High)).toThrow(/missing thinking metadata/);
expect(() => requireSupportedEffort(spec, Effort.High)).toThrow(/not supported/);
});
it("drops empty thinking metadata so presence checks stay meaningful", () => {
const model = enrichModelThinking({
it("drops authored thinking on non-reasoning models and re-derives empty efforts", () => {
const nonReasoning = createModel({
id: "plain-model",
name: "Plain Model",
api: "openai-responses",
provider: "custom",
baseUrl: "https://example.com",
reasoning: false,
thinking: {
mode: "effort",
minLevel: Effort.High,
maxLevel: Effort.Low,
},
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200000,
maxTokens: 32000,
} satisfies ModelSpec<"openai-responses">);
thinking: { mode: "effort", efforts: [Effort.High] },
});
expect(nonReasoning.thinking).toBeUndefined();
expect(model.thinking).toBeUndefined();
// Empty explicit efforts are treated as absent metadata: infer instead.
const emptyEfforts = createModel({
id: "gpt-5.2-codex",
api: "openai-codex-responses",
provider: "openai-codex",
thinking: { mode: "effort", efforts: [] },
});
expect(emptyEfforts.thinking).toEqual({
mode: "effort",
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
});
});
});
@@ -50,8 +50,7 @@ describe("nanogpt model limits mapping", () => {
expect(model?.premiumMultiplier).toBeUndefined();
expect(model?.thinking).toEqual({
mode: "effort",
minLevel: Effort.Low,
maxLevel: Effort.XHigh,
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
});
expect(fetchMock).toHaveBeenCalledTimes(1);
});
@@ -14,6 +14,7 @@ const cloudModel: Model<"ollama-chat"> = {
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128_000,
maxTokens: 8_192,
compat: undefined,
};
function createNdjsonResponse(lines: unknown[]): Response {
@@ -45,7 +45,10 @@ describe("ollama local provider discovery", () => {
expect(model?.api).toBe("openai-responses");
expect(model?.contextWindow).toBe(1048576);
expect(model?.reasoning).toBe(true);
expect(model?.thinking).toEqual({ mode: "effort", minLevel: Effort.Minimal, maxLevel: Effort.High });
expect(model?.thinking).toEqual({
mode: "effort",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
});
expect(model?.input).toEqual(["text", "image"]);
});
+1
View File
@@ -4,6 +4,7 @@
### Added
- Added `supportsReasoningParams`, `alwaysSendMaxTokens`, `strictResponsesPairing`, and a recursive `whenThinking` overlay (alongside `streamIdleTimeoutMs`/`supportsLongPromptCacheRetention`/`requiresToolResultId`/`replayUnsignedThinking`) to the OpenAI/Anthropic `compat` schema so custom model entries can configure those provider-specific capabilities
- Custom model `thinking` config now uses the catalog's explicit vocabulary: `efforts` (ordered list) plus optional `defaultLevel`, `effortMap`, and `supportsDisplay` overrides; the legacy `minLevel`/`maxLevel`/`levels` range shape is still accepted and normalized at parse time. Wire facts (`effortMap`/`supportsDisplay`) are backfilled from model identity when not set, so existing claude-proxy configs keep the 5-tier adaptive scale and summarized display without changes.
- New `omp usage` command: a detailed per-account breakdown of provider usage limits (bars, windows, reset times, plan metadata) covering every stored credential — accounts with no usage endpoint are listed as "no usage data" rows. Each provider section ends with per-window capacity stats ("capacity: 5h → 2.40/5 accounts used (2.60× quota left)"). Flags: `--provider` to filter, `--json` for the broker-shaped report payload, and `--redact` to mask account emails/ids down to a two-char anchor plus a minimal middle-out differentiator (`ca*9*`) for screenshot-safe sharing.
- Startup hangs are now self-diagnosing (speculative fix for the "zero output, hangs even on `omp -h`" report class): a watchdog prints a stderr line every 10s naming the deepest in-flight startup phase (via `logger.openSpanPath()`) until a mode runner takes over, pausing around legitimate interactive waits (fork/move prompts, the `--resume` session picker); `PI_DEBUG_STARTUP` is restored as streaming synchronous `[startup]` phase markers covering command-module imports and the native addon load, which the post-startup `PI_TIMING` tree structurally cannot show for a hang; and waiting on piped-stdin EOF announces itself after 1s instead of blocking silently.
- npm installs now execute a prebundled single-file entry: the published `bin.omp` points at `dist/cli.js` (built by `scripts/bundle-dist.ts` during `prepack`, ~18MB minified, natives/transformers/mupdf external), cutting npm-install cold start by roughly 3x versus transpiling the raw TypeScript graph per launch; `src/**` stays published for SDK consumers and worker fallbacks. The on-repo manifest keeps `bin.omp` at `src/cli.ts` — release rewrites it via the `publishBin` override in `scripts/ci-release-publish.ts` — so source installs (`bun link`, `install.sh --source`) keep working without a build step
@@ -68,13 +68,50 @@ const ThinkingControlModeSchema = z.enum([
"anthropic-budget-effort",
]);
const ModelThinkingSchema = z.object({
minLevel: EffortSchema,
maxLevel: EffortSchema,
mode: ThinkingControlModeSchema,
defaultLevel: EffortSchema.optional(),
levels: z.array(EffortSchema).optional(),
});
const EFFORT_ORDER = ["minimal", "low", "medium", "high", "xhigh"] as const;
/**
* Accepts the canonical `efforts` vocabulary plus the legacy
* `minLevel`/`maxLevel`/`levels` range shape, normalizing both to
* `ThinkingConfig` (ordered `efforts`, never empty). Precedence mirrors the
* old runtime: explicit `levels` beat the min..max range; `efforts` beats both.
*/
const ModelThinkingSchema = z
.object({
mode: ThinkingControlModeSchema,
efforts: z.array(EffortSchema).min(1).optional(),
defaultLevel: EffortSchema.optional(),
effortMap: ReasoningEffortMapSchema.optional(),
supportsDisplay: z.boolean().optional(),
// Legacy range vocabulary (pre-efforts configs).
minLevel: EffortSchema.optional(),
maxLevel: EffortSchema.optional(),
levels: z.array(EffortSchema).min(1).optional(),
})
.refine(
value =>
value.efforts !== undefined ||
value.levels !== undefined ||
(value.minLevel !== undefined && value.maxLevel !== undefined),
{
message: "thinking requires `efforts` (or legacy `levels`/`minLevel`+`maxLevel`)",
},
)
.transform(({ efforts, levels, minLevel, maxLevel, mode, defaultLevel, effortMap, supportsDisplay }) => {
let resolved = efforts ?? levels;
if (!resolved) {
const minIndex = EFFORT_ORDER.indexOf(minLevel!);
const maxIndex = EFFORT_ORDER.indexOf(maxLevel!);
resolved = EFFORT_ORDER.slice(minIndex, Math.max(minIndex, maxIndex) + 1);
}
return {
mode,
efforts: resolved,
...(defaultLevel !== undefined && { defaultLevel }),
...(effortMap !== undefined && { effortMap }),
...(supportsDisplay !== undefined && { supportsDisplay }),
};
});
const ModelDefinitionSchema = z.object({
id: z.string().min(1),
@@ -38,7 +38,7 @@ const SLOW = makeModel("p", "slow");
const REASONING_SLOW = makeModel("p", "slow", {
api: "anthropic-messages",
reasoning: true,
thinking: { minLevel: Effort.Low, maxLevel: Effort.High, mode: "anthropic-adaptive" },
thinking: { efforts: [Effort.Low, Effort.Medium, Effort.High], mode: "anthropic-adaptive" },
});
interface SessionOptions {
@@ -1082,7 +1082,7 @@ export interface ExtensionAPI {
* id: "claude-sonnet-4@20250514",
* name: "Claude Sonnet 4 (Vertex)",
* reasoning: true,
* thinking: { mode: "anthropic-adaptive", minLevel: "minimal", maxLevel: "high" },
* thinking: { mode: "anthropic-adaptive", efforts: ["minimal", "low", "medium", "high"] },
* input: ["text", "image"],
* cost: { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
* contextWindow: 200000,
@@ -699,7 +699,7 @@ describe("AgentSession MCP discovery", () => {
const reasoningModel: Model<"openai-responses"> = {
...createModel(),
reasoning: true,
thinking: { mode: "effort", minLevel: Effort.Medium, maxLevel: Effort.Medium },
thinking: { mode: "effort", efforts: [Effort.Medium] },
};
const agent = new Agent({
@@ -71,8 +71,7 @@ describe("issue #775: per-model defaultLevel", () => {
...opus,
thinking: {
mode: "anthropic-adaptive",
minLevel: Effort.Low,
maxLevel: Effort.XHigh,
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
defaultLevel: Effort.XHigh,
},
};
@@ -238,7 +238,7 @@ describe("memories runtime", () => {
const constrainedModel: Model = {
...fx.model,
reasoning: true,
thinking: { mode: "effort", minLevel: Effort.High, maxLevel: Effort.XHigh },
thinking: { mode: "effort", efforts: [Effort.High, Effort.XHigh] },
};
fx.session.model = constrainedModel;
fx.modelRegistry.find = vi.fn(() => constrainedModel);
@@ -351,8 +351,7 @@ describe("ModelRegistry runtime discovery", () => {
expect(qwen?.reasoning).toBe(true);
expect(qwen?.thinking).toEqual({
mode: "effort",
minLevel: Effort.Minimal,
maxLevel: Effort.High,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
});
const llama = registry.find("ollama", "llama3.2:3b");
@@ -141,7 +141,7 @@ describe("ModelRegistry runtime provider registration", () => {
expectProviderHeader(registry, providerName, "Authorization", undefined);
});
test("registerProvider preserves explicit thinking on runtime models", () => {
test("registerProvider preserves explicit thinking and backfills wire facts", () => {
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const config: ProviderConfigInput = {
baseUrl: "https://runtime.example.com/v1",
@@ -154,8 +154,7 @@ describe("ModelRegistry runtime provider registration", () => {
reasoning: true,
thinking: {
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.High,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
},
},
],
@@ -166,8 +165,10 @@ describe("ModelRegistry runtime provider registration", () => {
expect(model?.thinking).toEqual({
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.High,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
// Wire facts are backfilled from identity; non-claude ids get the
// 4-tier adaptive map.
effortMap: { minimal: "low", xhigh: "max" },
});
});
@@ -284,7 +284,7 @@ describe("ModelRegistry", () => {
const variants = registry.getCanonicalVariants("deepseek-v4-pro");
expect(model?.cost.cacheRead).toBeGreaterThan(0);
expect(model?.thinking?.maxLevel).toBe(Effort.XHigh);
expect(model?.thinking?.efforts.at(-1)).toBe(Effort.XHigh);
expect(variants.some(variant => variant.selector === "ollama/deepseek-v4-pro:cloud")).toBe(true);
});
@@ -1132,12 +1132,10 @@ describe("ModelRegistry", () => {
});
describe("thinking metadata normalization", () => {
test("custom models preserve explicit thinking", () => {
test("custom models preserve explicit thinking and gain backfilled wire facts", () => {
const thinking: ThinkingConfig = {
mode: "anthropic-adaptive",
minLevel: Effort.Minimal,
maxLevel: Effort.High,
levels: [Effort.Minimal, Effort.High],
efforts: [Effort.Minimal, Effort.High],
};
writeModelsJson({
@@ -1149,7 +1147,11 @@ describe("ModelRegistry", () => {
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const model = getModelsForProvider(registry, "anthropic").find(m => m.id === "claude-custom");
expect(model?.thinking).toEqual(thinking);
expect(model?.thinking).toEqual({
...thinking,
// Versionless claude ids resolve to the 4-tier adaptive wire map.
effortMap: { minimal: "low", xhigh: "max" },
});
});
test("model overrides can replace canonical thinking metadata", () => {
@@ -1157,7 +1159,7 @@ describe("ModelRegistry", () => {
openrouter: {
modelOverrides: {
"anthropic/claude-sonnet-4": {
thinking: { mode: "budget", minLevel: Effort.Low, maxLevel: Effort.Medium },
thinking: { mode: "budget", efforts: [Effort.Low, Effort.Medium] },
},
},
},
@@ -1168,8 +1170,7 @@ describe("ModelRegistry", () => {
expect(model?.thinking).toEqual({
mode: "budget",
minLevel: Effort.Low,
maxLevel: Effort.Medium,
efforts: [Effort.Low, Effort.Medium],
});
});
});
@@ -25,8 +25,7 @@ const mockModels: Model<"anthropic-messages">[] = [
reasoning: true,
thinking: {
mode: "budget",
minLevel: Effort.Minimal,
maxLevel: Effort.High,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
},
input: ["text", "image"],
cost: { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
@@ -58,8 +57,7 @@ const mockOpenRouterModels: Model<Api>[] = [
reasoning: true,
thinking: {
mode: "budget",
minLevel: Effort.Minimal,
maxLevel: Effort.High,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
},
input: ["text"],
cost: { input: 1, output: 2, cacheRead: 0.1, cacheWrite: 1 },
@@ -87,8 +85,7 @@ const mockOpenRouterModels: Model<Api>[] = [
reasoning: true,
thinking: {
mode: "budget",
minLevel: Effort.Minimal,
maxLevel: Effort.High,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
},
input: ["text"],
cost: { input: 1, output: 2, cacheRead: 0.1, cacheWrite: 1 },
@@ -134,8 +131,7 @@ const mockCodexOverlapModels: Model<"anthropic-messages">[] = [
reasoning: true,
thinking: {
mode: "effort",
minLevel: Effort.Low,
maxLevel: Effort.XHigh,
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
input: ["text"],
cost: { input: 1.5, output: 6, cacheRead: 0.15, cacheWrite: 1.5 },
@@ -151,8 +147,7 @@ const mockCodexOverlapModels: Model<"anthropic-messages">[] = [
reasoning: true,
thinking: {
mode: "effort",
minLevel: Effort.Low,
maxLevel: Effort.XHigh,
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
input: ["text"],
cost: { input: 1, output: 4, cacheRead: 0.1, cacheWrite: 1 },
@@ -171,8 +166,7 @@ function createOpusModel(provider: string, id: string, name: string): Model<"ant
reasoning: true,
thinking: {
mode: "budget",
minLevel: Effort.Minimal,
maxLevel: Effort.XHigh,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
},
input: ["text", "image"],
cost: { input: 15, output: 75, cacheRead: 1.5, cacheWrite: 18.75 },
@@ -191,8 +185,7 @@ const canonicalVariantModels: Model<"anthropic-messages">[] = [
reasoning: true,
thinking: {
mode: "budget",
minLevel: Effort.Minimal,
maxLevel: Effort.High,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
},
input: ["text", "image"],
cost: { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
@@ -208,8 +201,7 @@ const canonicalVariantModels: Model<"anthropic-messages">[] = [
reasoning: true,
thinking: {
mode: "budget",
minLevel: Effort.Minimal,
maxLevel: Effort.High,
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
},
input: ["text", "image"],
cost: { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
@@ -37,7 +37,7 @@ function createReasoningModel(): Model<"openai-responses"> {
provider: "openai",
baseUrl: "https://example.invalid",
reasoning: true,
thinking: { mode: "effort", minLevel: Effort.Medium, maxLevel: Effort.High },
thinking: { mode: "effort", efforts: [Effort.Medium, Effort.High] },
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 8192,