refactor(catalog): baked thinking metadata into buildModel pipeline
- Replaced minLevel/maxLevel range with explicit efforts array plus baked effortMap/supportsDisplay wire facts. - Removed runtime enrichment layer and modelOmitsReasoningEffort; providers now read baked fields. - Fixed dotted Opus 4.7/4.8 ids missing adaptive display via classifier-based predicates (#1373). - Bumped model cache schema to v4 to invalidate pre-efforts rows.
This commit is contained in:
@@ -539,10 +539,11 @@ function effortFromThinkingLevel(level: ThinkingLevel): Effort {
|
||||
* - Explicit effort → respect user choice → clamped per model.
|
||||
*
|
||||
* The clamp routes through `clampThinkingLevelForModel`, which returns
|
||||
* `undefined` for models with `compat.supportsReasoningEffort: false`
|
||||
* (e.g. `xai-oauth/grok-build`). That `undefined` then flows through to the
|
||||
* openai-responses mapper where `modelOmitsReasoningEffort` short-circuits
|
||||
* the wire param — no `requireSupportedEffort` throw.
|
||||
* `undefined` for reasoning models without a thinking config — the build-time
|
||||
* encoding of `compat.supportsReasoningEffort: false` (e.g.
|
||||
* `xai-oauth/grok-build`). That `undefined` then flows through to the
|
||||
* openai-responses mapper, which omits the wire param — no
|
||||
* `requireSupportedEffort` throw.
|
||||
*/
|
||||
function resolveCompactionEffort(model: Model, level: ThinkingLevel | undefined): Effort | undefined {
|
||||
if (level === ThinkingLevel.Off) return undefined;
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
- Auth storage no longer issues per-boot no-op writes: the schema-version row is only rewritten when the recorded version actually changes, and the credential identity-key backfill skips rows whose derived identity is null — reopening a current-schema database now performs zero write transactions
|
||||
- Plain provider env-var names moved to the catalog table: registry defs dropped their 48 `envKeys` literals (including the pure `$pickenv` pickers for `huggingface`/`qwen-portal`/`xai-oauth`), `getEnvApiKey` now derives those fallbacks from `CATALOG_PROVIDERS[].envVars`, and `envKeys` remains only for computed resolvers (Anthropic Foundry, Vertex ADC, Bedrock credential chains) and non-catalog providers (`kagi`, `tavily`, `parallel`, `perplexity`)
|
||||
- Protocol handlers are now pure `model.compat` readers — the per-request `resolve*Compat`/`detect*Compat` calls (anthropic ×11, responses ×3, completions wrappers), inline `strictResponsesPairing` host detection, the OpenCode `reasoning_content` mutation block, and all `resolvedBaseUrl` threading are gone. Compat is materialized once at model build time (`@oh-my-pi/pi-catalog` `buildModel`); the OpenCode thinking-mode quirk is a precomputed `compat.whenThinking` pointer swap, and request-time base-URL overrides only feed the HTTP client. Behavior is unchanged (the Anthropic `supportsLongCacheRetention` official-endpoint gate is folded into detection).
|
||||
- Providers now read baked thinking/wire metadata instead of re-parsing model ids per request: the Anthropic handler gates sampling params on `model.compat.supportsSamplingParams` and adaptive `display` on `model.thinking.supportsDisplay` (Bedrock too), adaptive effort tiers come from the baked `thinking.effortMap`, the Google `thinkingLevel` map is static, and effort-dial-less reasoners (`thinking: undefined`, e.g. `xai-oauth/grok-build`) short-circuit `resolveOpenAiReasoningEffort` without the removed `modelOmitsReasoningEffort` predicate.
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -26,6 +27,7 @@
|
||||
- Fixed in-stream Anthropic SSE `error` events being thrown as raw JSON envelopes; the structured `error.type`/`message` is parsed out, keeping retry classification on the typed token instead of accidental regex hits.
|
||||
- Fixed transparent-reconnect tolerance duplicating content behind replaying proxies: after a duplicate `message_start`, replayed `content_block_start` events for already-closed indexes are now consumed silently instead of appending duplicate text/tool calls.
|
||||
- Fixed the Anthropic gateway accepting malformed known-type content blocks (e.g. `{type:"text", text:123}`) through the unknown-block catch-all, corrupting history and surfacing later as an opaque TypeError — they now fail validation with a clean 400. The gateway's encode stream also emits `ping` keepalives every 15s and a complete `message_start`/`message_delta`/`message_stop` envelope when the inner stream ends without a terminal event, so strict clients no longer classify slow or empty streams as protocol errors.
|
||||
- Fixed dotted-version Claude ids (`claude-opus-4.7`/`4.8` on GitHub Copilot, Vercel AI Gateway, Zenmux) missing adaptive thinking `display` support — streamed reasoning stayed hidden on those entries because the display predicate only matched dash-form ids (same failure class as #1373).
|
||||
- Fixed the Mistral `requiresThinkingAsText` replay path calling `.unshift()` on string assistant content — an unconditional TypeError that failed any same-model history turn carrying both thinking and text.
|
||||
- Fixed the Responses gateway stripping `encrypted_content` from inbound reasoning items (strip-mode schema), which broke codex-style stateless replay; the schema is now loose, restoring the symmetry the outbound encoder already preserved. Composite internal `callId|itemId` ids are also split before hitting the wire so third-party clients that validate `call_id` charsets no longer reject them.
|
||||
- Ported the shared unfinished-tool-call sweep to the codex `response.completed` handler, so a lost `output_item.done` can no longer persist a tool call with stale `{}` arguments and transient parser fields into session history.
|
||||
|
||||
@@ -8,7 +8,6 @@
|
||||
*/
|
||||
|
||||
import type { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import { supportsAdaptiveThinkingDisplay } from "@oh-my-pi/pi-catalog/identity";
|
||||
import { mapEffortToAnthropicAdaptiveEffort, requireSupportedEffort } from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
||||
import { $env, $flag, extractHttpStatusFromError, fetchWithRetry } from "@oh-my-pi/pi-utils";
|
||||
@@ -819,7 +818,7 @@ function buildAdditionalModelRequestFields(
|
||||
// runs (issue #1373). Opt back into "summarized" by default on models that
|
||||
// accept the field.
|
||||
const adaptive: { type: "adaptive"; display?: BedrockThinkingDisplay } = { type: "adaptive" };
|
||||
if (supportsAdaptiveThinkingDisplay(model.id)) {
|
||||
if (model.thinking?.supportsDisplay) {
|
||||
adaptive.display = options.thinkingDisplay ?? "summarized";
|
||||
}
|
||||
return {
|
||||
|
||||
@@ -3,8 +3,7 @@ import * as fs from "node:fs";
|
||||
import { scheduler } from "node:timers/promises";
|
||||
import * as tls from "node:tls";
|
||||
import { isOfficialAnthropicApiUrl } from "@oh-my-pi/pi-catalog/compat/anthropic";
|
||||
import { supportsAdaptiveThinkingDisplay } from "@oh-my-pi/pi-catalog/identity";
|
||||
import { hasOpus47ApiRestrictions, mapEffortToAnthropicAdaptiveEffort } from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { mapEffortToAnthropicAdaptiveEffort } from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
||||
import { isAnthropicOAuthToken } from "@oh-my-pi/pi-catalog/utils";
|
||||
import { parseGitHubCopilotApiKey } from "@oh-my-pi/pi-catalog/wire/github-copilot";
|
||||
@@ -2279,7 +2278,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
||||
claudeCodeSessionId,
|
||||
} = args;
|
||||
const compat = model.compat;
|
||||
const needsInterleavedBeta = interleavedThinking && !supportsAdaptiveThinkingDisplay(model.id);
|
||||
const needsInterleavedBeta = interleavedThinking && !model.thinking?.supportsDisplay;
|
||||
const needsFineGrainedToolStreamingBeta = hasTools && !compat.supportsEagerToolInputStreaming;
|
||||
const oauthToken = isOAuth ?? isAnthropicOAuthToken(apiKey);
|
||||
const baseUrl = resolveAnthropicBaseUrl(model, apiKey);
|
||||
@@ -2754,7 +2753,7 @@ function buildParams(
|
||||
// callers that rely on it. The `display` field is gated strictly on model
|
||||
// support: Opus 4.6 / Sonnet 4.6+ reject it with a 400, so an explicit
|
||||
// `thinkingDisplay` MUST NOT force it onto a model that can't accept it.
|
||||
if (supportsAdaptiveThinkingDisplay(model.id)) {
|
||||
if (model.thinking?.supportsDisplay) {
|
||||
adaptive.display = options.thinkingDisplay ?? "summarized";
|
||||
}
|
||||
thinking = adaptive;
|
||||
@@ -2817,7 +2816,7 @@ function buildParams(
|
||||
// Opus 4.7+ and Fable/Mythos 5 reject non-default sampling parameters with 400 error.
|
||||
const thinkingType = params.thinking?.type;
|
||||
const allowSamplingParams =
|
||||
!hasOpus47ApiRestrictions(model.id) && (thinkingType === undefined || thinkingType === "disabled");
|
||||
model.compat.supportsSamplingParams && (thinkingType === undefined || thinkingType === "disabled");
|
||||
if (allowSamplingParams && options?.temperature !== undefined) {
|
||||
params.temperature = options.temperature;
|
||||
}
|
||||
|
||||
+12
-12
@@ -3,7 +3,6 @@ import { isVertexExpressOpenAIUrl, isVertexRawPredictUrl } from "@oh-my-pi/pi-ca
|
||||
import {
|
||||
mapEffortToAnthropicAdaptiveEffort,
|
||||
mapEffortToGoogleThinkingLevel,
|
||||
modelOmitsReasoningEffort,
|
||||
requireSupportedEffort,
|
||||
} from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { CATALOG_PROVIDERS, type ProviderCatalogEntry } from "@oh-my-pi/pi-catalog/provider-models";
|
||||
@@ -654,14 +653,15 @@ function resolveOpenAiReasoningEffort<TApi extends Api>(
|
||||
): Effort | undefined {
|
||||
const reasoning = options?.reasoning;
|
||||
if (!reasoning || !model.reasoning) return undefined;
|
||||
// Models with compat.supportsReasoningEffort: false reason natively but
|
||||
// reject the wire effort param. The wire-side omitReasoningEffort gate
|
||||
// (providers/xai-responses.ts:78) is the actual strip; returning
|
||||
// undefined here avoids a redundant requireSupportedEffort throw that
|
||||
// would defeat the gate and surface a confusing
|
||||
// "Compaction failed: Thinking effort high is not supported by..." to
|
||||
// the user.
|
||||
if (modelOmitsReasoningEffort(model)) return undefined;
|
||||
// Models that reason natively but expose no effort dial carry
|
||||
// `thinking: undefined` (baked at build time from
|
||||
// `compat.supportsReasoningEffort: false` on openai-responses*). The
|
||||
// wire-side omitReasoningEffort gate (providers/xai-responses.ts:78) is the
|
||||
// actual strip; returning undefined here avoids a redundant
|
||||
// requireSupportedEffort throw that would defeat the gate and surface a
|
||||
// confusing "Compaction failed: Thinking effort high is not supported
|
||||
// by..." to the user.
|
||||
if (!model.thinking) return undefined;
|
||||
return requireSupportedEffort(model, reasoning);
|
||||
}
|
||||
|
||||
@@ -869,7 +869,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
...base,
|
||||
thinking: {
|
||||
enabled: true,
|
||||
level: mapEffortToGoogleThinkingLevel(googleModel, effort),
|
||||
level: mapEffortToGoogleThinkingLevel(effort),
|
||||
},
|
||||
toolChoice: mapGoogleToolChoice(options?.toolChoice),
|
||||
});
|
||||
@@ -903,7 +903,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
...base,
|
||||
thinking: {
|
||||
enabled: true,
|
||||
level: mapEffortToGoogleThinkingLevel(model, effort),
|
||||
level: mapEffortToGoogleThinkingLevel(effort),
|
||||
},
|
||||
toolChoice: mapGoogleToolChoice(options?.toolChoice),
|
||||
});
|
||||
@@ -956,7 +956,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
...base,
|
||||
thinking: {
|
||||
enabled: true,
|
||||
level: mapEffortToGoogleThinkingLevel(geminiModel, effort),
|
||||
level: mapEffortToGoogleThinkingLevel(effort),
|
||||
},
|
||||
toolChoice: mapGoogleToolChoice(options?.toolChoice),
|
||||
});
|
||||
|
||||
@@ -376,7 +376,10 @@ describe("Anthropic request fingerprint alignment", () => {
|
||||
...ANTHROPIC_MODEL_SPEC,
|
||||
id: "claude-opus-4-8-20260528",
|
||||
name: "Claude Opus 4.8",
|
||||
thinking: { mode: "anthropic-adaptive", minLevel: Effort.Minimal, maxLevel: Effort.XHigh },
|
||||
thinking: {
|
||||
mode: "anthropic-adaptive",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
});
|
||||
|
||||
await streamAnthropic(
|
||||
@@ -1677,8 +1680,7 @@ describe("Anthropic request fingerprint alignment", () => {
|
||||
name: "Claude Opus 4.7",
|
||||
thinking: {
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
}),
|
||||
{
|
||||
@@ -1717,8 +1719,7 @@ describe("Anthropic request fingerprint alignment", () => {
|
||||
name: "Claude Opus 4.7",
|
||||
thinking: {
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
}),
|
||||
{
|
||||
@@ -1745,8 +1746,7 @@ describe("Anthropic request fingerprint alignment", () => {
|
||||
name: "Claude Opus 4.7",
|
||||
thinking: {
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
}),
|
||||
{
|
||||
@@ -1777,8 +1777,7 @@ describe("Anthropic request fingerprint alignment", () => {
|
||||
name: "Claude Opus 4.7",
|
||||
thinking: {
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
}),
|
||||
{
|
||||
@@ -1811,8 +1810,7 @@ describe("Anthropic request fingerprint alignment", () => {
|
||||
name: "Claude Opus 4.7",
|
||||
thinking: {
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
}),
|
||||
{
|
||||
@@ -1857,8 +1855,7 @@ describe("Anthropic request fingerprint alignment", () => {
|
||||
maxTokens: 128_000,
|
||||
thinking: {
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
}),
|
||||
{
|
||||
|
||||
@@ -24,7 +24,10 @@ function adaptiveModel(id: string): Model<"anthropic-messages"> {
|
||||
const base = makeAnthropicModel(id);
|
||||
return buildModel({
|
||||
...base,
|
||||
thinking: { mode: "anthropic-adaptive", minLevel: Effort.Minimal, maxLevel: Effort.XHigh },
|
||||
thinking: {
|
||||
mode: "anthropic-adaptive",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
compat: base.compatConfig,
|
||||
} as ModelSpec<"anthropic-messages">);
|
||||
}
|
||||
|
||||
@@ -27,7 +27,10 @@ function adaptiveModel(id: string): Model<"bedrock-converse-stream"> {
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 1_000_000,
|
||||
maxTokens: 128_000,
|
||||
thinking: { mode: "anthropic-adaptive", minLevel: Effort.Minimal, maxLevel: Effort.XHigh },
|
||||
thinking: {
|
||||
mode: "anthropic-adaptive",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
@@ -43,7 +46,7 @@ function budgetModel(id: string): Model<"bedrock-converse-stream"> {
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 64_000,
|
||||
thinking: { mode: "budget", minLevel: Effort.Minimal, maxLevel: Effort.High },
|
||||
thinking: { mode: "budget", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -36,7 +36,10 @@ const OPUS_46_OAUTH: Model<"anthropic-messages"> = buildModel({
|
||||
cost: { input: 5, output: 25, cacheRead: 0.5, cacheWrite: 6.25 },
|
||||
contextWindow: 1_000_000,
|
||||
maxTokens: 128_000,
|
||||
thinking: { mode: "anthropic-adaptive", minLevel: Effort.Minimal, maxLevel: Effort.XHigh },
|
||||
thinking: {
|
||||
mode: "anthropic-adaptive",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
});
|
||||
|
||||
const todoTool: Tool = {
|
||||
|
||||
@@ -83,8 +83,7 @@ describe("issue #826: Anthropic strict-tools opt-out for Vertex-style proxies",
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
compat: baseModel.compatConfig,
|
||||
} as ModelSpec<"anthropic-messages">);
|
||||
@@ -110,8 +109,7 @@ describe("issue #826: Anthropic strict-tools opt-out for Vertex-style proxies",
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
compat: { ...baseModel.compatConfig, disableAdaptiveThinking: true },
|
||||
} as ModelSpec<"anthropic-messages">);
|
||||
|
||||
@@ -27,8 +27,7 @@ function customOpenAICompatModel(): Model<"openai-completions"> {
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
minLevel: Effort.Low,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
|
||||
@@ -26,8 +26,7 @@ function createReasoningEffortModel(): Model<"openai-completions"> {
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.High,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
|
||||
@@ -1,46 +1,43 @@
|
||||
import { describe, expect, test } from "bun:test";
|
||||
import { modelOmitsReasoningEffort } from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
|
||||
// Pins fix #2 of the compaction effort-override bug. Before this fix,
|
||||
// `resolveOpenAiReasoningEffort` called `requireSupportedEffort` which threw
|
||||
// for any model with `compat.supportsReasoningEffort: false` (e.g.
|
||||
// `xai-oauth/grok-build`) — producing the user-visible "Compaction failed:
|
||||
// Thinking effort high is not supported by xai-oauth/grok-build. Supported
|
||||
// efforts:" (empty list). The fix routes through the explicit
|
||||
// `modelOmitsReasoningEffort` predicate, which lets the wire-side
|
||||
// `omitReasoningEffort` gate (providers/xai-responses.ts:78) remain the
|
||||
// single source of truth for the actual strip.
|
||||
describe("modelOmitsReasoningEffort (regression)", () => {
|
||||
test("returns true for xai-oauth/grok-build (supportsReasoningEffort: false)", () => {
|
||||
// Pins fix #2 of the compaction effort-override bug. Models that reason
|
||||
// natively but reject the wire `reasoning.effort` param (e.g.
|
||||
// `xai-oauth/grok-build`, `compat.supportsReasoningEffort: false` on
|
||||
// openai-responses*) are encoded at build time as `thinking: undefined` —
|
||||
// "thinks, but exposes no control surface". `resolveOpenAiReasoningEffort`
|
||||
// returns undefined for them instead of tripping `requireSupportedEffort`
|
||||
// (the old user-visible "Compaction failed: Thinking effort high is not
|
||||
// supported by xai-oauth/grok-build. Supported efforts:" with an empty list),
|
||||
// and the wire-side `omitReasoningEffort` gate (providers/xai-responses.ts)
|
||||
// remains the single source of truth for the actual strip.
|
||||
describe("effort-dial-less reasoner encoding (regression)", () => {
|
||||
test("xai-oauth/grok-build reasons but carries no thinking config", () => {
|
||||
const grokBuild = getBundledModel("xai-oauth", "grok-build");
|
||||
if (!grokBuild) throw new Error("xai-oauth/grok-build must be in bundled models.json");
|
||||
expect(modelOmitsReasoningEffort(grokBuild)).toBe(true);
|
||||
expect(grokBuild.reasoning).toBe(true);
|
||||
expect(grokBuild.thinking).toBeUndefined();
|
||||
expect(getSupportedEfforts(grokBuild)).toEqual([]);
|
||||
});
|
||||
|
||||
test("returns false for xai-oauth/grok-4.3 (effort-capable)", () => {
|
||||
test("xai-oauth/grok-4.3 keeps its effort dial", () => {
|
||||
const grok43 = getBundledModel("xai-oauth", "grok-4.3");
|
||||
if (!grok43) throw new Error("xai-oauth/grok-4.3 must be in bundled models.json");
|
||||
expect(modelOmitsReasoningEffort(grok43)).toBe(false);
|
||||
expect(grok43.thinking).toBeDefined();
|
||||
expect(getSupportedEfforts(grok43).length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
test("returns true for xai-oauth/grok-4.20-0309-reasoning (supportsReasoningEffort: false)", () => {
|
||||
test("xai-oauth/grok-4.20-0309-reasoning reasons but carries no thinking config", () => {
|
||||
const grokR = getBundledModel("xai-oauth", "grok-4.20-0309-reasoning");
|
||||
if (!grokR) throw new Error("xai-oauth/grok-4.20-0309-reasoning must be in bundled models.json");
|
||||
expect(modelOmitsReasoningEffort(grokR)).toBe(true);
|
||||
expect(grokR.reasoning).toBe(true);
|
||||
expect(grokR.thinking).toBeUndefined();
|
||||
});
|
||||
|
||||
test("returns false for an Anthropic model (different api surface)", () => {
|
||||
test("the no-dial encoding stays scoped to openai-responses*", () => {
|
||||
const claude = getBundledModel("anthropic", "claude-sonnet-4-6");
|
||||
if (!claude) throw new Error("anthropic/claude-sonnet-4-6 must be in bundled models.json");
|
||||
expect(modelOmitsReasoningEffort(claude)).toBe(false);
|
||||
});
|
||||
|
||||
test("returns false for an openai-completions model (out of scope)", () => {
|
||||
const openai = getBundledModel("openai", "gpt-4o-mini");
|
||||
if (!openai) throw new Error("openai/gpt-4o-mini must be in bundled models.json");
|
||||
// gpt-4o-mini is openai-completions, not openai-responses* — predicate
|
||||
// must return false even if compat had supportsReasoningEffort: false.
|
||||
expect(modelOmitsReasoningEffort(openai)).toBe(false);
|
||||
expect(claude.thinking).toBeDefined();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -5,7 +5,8 @@
|
||||
### Added
|
||||
|
||||
- Added `hostMatchesUrl`, `modelMatchesHost`, and endpoint-shape helpers in the new `hosts` module for consistent provider/baseUrl matching
|
||||
- `buildModel(spec)` (`build.ts`) is now the single Model constructor: it runs thinking enrichment and materializes the fully-resolved compat record exactly once, so `Model.compat` is a required, complete `CompatOf<TApi>` (`ResolvedOpenAICompat`/`ResolvedOpenAIResponsesCompat`/`ResolvedAnthropicCompat`) and request-path code reads fields with zero URL parsing and zero per-request allocation. Sparse user/config overrides live on the new `ModelSpec<TApi>` input shape and survive on `Model.compatConfig` for introspection.
|
||||
- `buildModel(spec)` (`build.ts`) is now the single Model constructor: it materializes the fully-resolved compat record and canonical thinking metadata exactly once (compat first, thinking derived from identity + resolved compat), so `Model.compat` is a required, complete `CompatOf<TApi>` (`ResolvedOpenAICompat`/`ResolvedOpenAIResponsesCompat`/`ResolvedAnthropicCompat`) and request-path code reads fields with zero URL parsing and zero per-request allocation. Sparse user/config overrides live on the new `ModelSpec<TApi>` input shape and survive on `Model.compatConfig` for introspection.
|
||||
- Added `ResolvedAnthropicCompat.supportsSamplingParams` (Opus 4.7+/Fable/Mythos reject `temperature`/`top_p`/`top_k` with a 400), baked at build time from model identity so the request path stops re-parsing model ids.
|
||||
- Compat detection gained model-time flags so handlers stop sniffing baseUrl: completions `supportsReasoningParams`, `alwaysSendMaxTokens`, `isOpenRouterHost`, `isVercelGatewayHost`, `streamIdleTimeoutMs`, and a precomputed `whenThinking` alternate view (OpenCode `reasoning_content` gating, #1071/#1484); responses `strictResponsesPairing`, `supportsLongPromptCacheRetention`, `supportsReasoningEffort`; anthropic `officialEndpoint`, `requiresToolResultId`, `replayUnsignedThinking`.
|
||||
- New `@oh-my-pi/pi-catalog` package: the model catalog extracted from `@oh-my-pi/pi-ai`. Owns the bundled `models.json` and its generation pipeline (`scripts/generate-models.ts`), the core model data types (`Model`, `Api`, `ThinkingConfig`, `Effort`, `Usage`, compat interfaces), thinking metadata enrichment and generated policies (`model-thinking.ts`), the SQLite model cache and model manager, per-provider discovery factories (`provider-models/`), the discovery protocol clients (`discovery/`), and the new `CATALOG_PROVIDERS` table — the single source of truth for provider ids, default models, and discovery wiring (`KnownProvider`, `PROVIDER_DESCRIPTORS`, and `DEFAULT_MODEL_PER_PROVIDER` are derived from it).
|
||||
- New `identity/` module centralizing model-identity concerns that were previously duplicated across packages: family classification and version parsing (`identity/classify.ts`, extracted from pi-ai's `model-thinking` internals), canonical model equivalence with injected reference data (`identity/equivalence.ts`, from coding-agent's `model-equivalence`), proxy/reseller reference lookup (`identity/reference.ts`, from coding-agent's `model-registry`), bracket-affix and id-segment helpers (`identity/id.ts`), a single trailing-marker vocabulary with canonical vs reference flavors (`identity/markers.ts` — `search` stays reference-only so Perplexity's `sonar-pro-search` remains canonical-distinct), and provider priority ordering (`identity/priority.ts`).
|
||||
@@ -18,9 +19,18 @@
|
||||
- Changed `hostMatchesUrl`/`modelMatchesHost` usage in compatibility detection to reduce mismatches across case variants and provider alias hosts
|
||||
- Provider catalog entries now carry the runtime API-key env fallback as an ordered `envVars` list; `catalogDiscovery.envVars` became an optional generation-time override (only `cursor` and `vercel-ai-gateway` differ) and `PROVIDER_DESCRIPTORS` materializes the resolved list for `generate-models.ts`.
|
||||
- `Model`'s api parameter now defaults to `Api` instead of `any` (`Model<TApi extends Api = Api>`), so bare `Model` no longer behaves as `Model<any>` at call sites.
|
||||
- `ThinkingConfig` is now explicit and total: an ordered `efforts` array replaces the `minLevel`/`maxLevel`/`levels` range encoding, and the wire facts are baked alongside it — `effortMap` (anthropic-adaptive 4-tier vs 5-tier scale, shared with the OpenRouter completions remap) and `supportsDisplay` (adaptive `display` field support). Explicit spec thinking owns the capability surface (`mode`/`efforts`/`defaultLevel`) and wins over inference; missing wire facts are backfilled from identity so configs never need to know Anthropic's tier tables. Reasoning models that reject the wire effort param (`compat.supportsReasoningEffort: false` on openai-responses*) are encoded as `thinking: undefined` ("thinks, no control surface") instead of the removed `modelOmitsReasoningEffort` special case. `models.json` was re-baked in the new vocabulary behind a 3196-model behavioral parity gate, and the model cache schema bumped to v4 to invalidate old-shape rows.
|
||||
- `mapEffortToGoogleThinkingLevel(effort)` is now a static map (model parameter dropped — validation stays at the `requireSupportedEffort` call sites), and `mapEffortToAnthropicAdaptiveEffort` reads the baked `thinking.effortMap` instead of re-classifying the model id per request.
|
||||
- Generator-only policy code moved out of the runtime bundle into `scripts/generated-policies.ts`: `applyGeneratedModelPolicies` (now policy fixups + thinking re-bake via the shared deriver), `linkOpenAIPromotionTargets`, the Copilot context-window table, minimax/opencode-go compat fixups, and `CLOUDFLARE_FALLBACK_MODEL`. The anthropic id predicates (`hasOpus47ApiRestrictions`, `supportsMidConversationSystemMessages`, `isAnthropicFableOrMythosModel`) moved to `identity/family` for build-time use by the compat/thinking derivers only.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Anthropic official-endpoint detection to require strict HTTPS hostname matching so non-official or lookalike URLs are no longer treated as official Anthropic hosts
|
||||
- Fixed Ollama Cloud dynamic discovery so same-id matches from other providers no longer supply context-window or max-output-token limits for discovered models.
|
||||
- Wired `@oh-my-pi/pi-catalog` into the release publish package list, tarball install smoke test, and root `bun generate-models` script.
|
||||
- Fixed `supportsAdaptiveThinkingDisplay` only matching dash-form version ids: dotted ids (`claude-opus-4.7`) now classify through `identity/classify` like every other anthropic predicate, so six bundled dotted Opus 4.7/4.8 entries (github-copilot, vercel-ai-gateway, zenmux) regain adaptive `display` support; bare dated ids (`claude-opus-4-20250514` = Opus 4.0) stay excluded.
|
||||
- Fixed the OpenRouter anthropic adaptive-effort map misclassifying bare dated Opus ids (`claude-opus-4-20250514` parsed as version 4.20 → wrongly adaptive); the map now derives from the shared classifier and the shared 4-/5-tier tables.
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads.
|
||||
|
||||
@@ -17,11 +17,6 @@ import { $env } from "@oh-my-pi/pi-utils";
|
||||
import { fetchAntigravityDiscoveryModels } from "../src/discovery/antigravity";
|
||||
import { fetchCodexModels } from "../src/discovery/codex";
|
||||
import { createModelManager } from "../src/model-manager";
|
||||
import {
|
||||
applyGeneratedModelPolicies,
|
||||
CLOUDFLARE_FALLBACK_MODEL,
|
||||
linkOpenAIPromotionTargets,
|
||||
} from "../src/model-thinking";
|
||||
import prevModelsJson from "../src/models.json" with { type: "json" };
|
||||
import { toModelSpec } from "../src/provider-models/bundled-references";
|
||||
import {
|
||||
@@ -44,6 +39,11 @@ import {
|
||||
} from "../src/provider-models/openai-compat";
|
||||
import type { ModelSpec } from "../src/types";
|
||||
import { JWT_CLAIM_PATH } from "../src/wire/codex";
|
||||
import {
|
||||
applyGeneratedModelPolicies,
|
||||
CLOUDFLARE_FALLBACK_MODEL,
|
||||
linkOpenAIPromotionTargets,
|
||||
} from "./generated-policies";
|
||||
|
||||
const packageRoot = path.join(import.meta.dir, "..");
|
||||
|
||||
@@ -429,7 +429,10 @@ async function generateModels() {
|
||||
// Discovery-only providers (local inference servers) — never bundle static models.
|
||||
const fetchedKeys = new Set(allModels.map(model => `${model.provider}/${model.id}`));
|
||||
|
||||
for (const models of Object.values(prevModelsJson as Record<string, Record<string, ModelSpec>>)) {
|
||||
// Previous-snapshot entries may carry an older ThinkingConfig vocabulary;
|
||||
// applyGeneratedModelPolicies re-bakes `thinking` for every model, so the
|
||||
// inbound shape is irrelevant beyond identity/pricing/compat fields.
|
||||
for (const models of Object.values(prevModelsJson as unknown as Record<string, Record<string, ModelSpec>>)) {
|
||||
for (const model of Object.values(models)) {
|
||||
if (
|
||||
!fetchedKeys.has(`${model.provider}/${model.id}`) &&
|
||||
|
||||
@@ -0,0 +1,223 @@
|
||||
/**
|
||||
* Generation-time catalog policies: upstream metadata corrections, derived
|
||||
* field baking, and promotion-target linking. Runs only from
|
||||
* `generate-models.ts` — none of this ships in the runtime bundle.
|
||||
*/
|
||||
import { buildCompat } from "../src/build";
|
||||
import {
|
||||
type AnthropicModel,
|
||||
isFableOrMythos,
|
||||
type OpenAIModel,
|
||||
type OpenAIVariant,
|
||||
type ParsedModel,
|
||||
parseKnownModel,
|
||||
semverEqual,
|
||||
} from "../src/identity/classify";
|
||||
import { resolveModelThinking } from "../src/model-thinking";
|
||||
import type { Api, ModelSpec } from "../src/types";
|
||||
|
||||
const CLOUDFLARE_AI_GATEWAY_BASE_URL = "https://gateway.ai.cloudflare.com/v1/<account>/<gateway>/anthropic";
|
||||
|
||||
/**
|
||||
* Static fallback model injected when Cloudflare AI Gateway discovery
|
||||
* returns no results. Ensures the provider always has at least one usable
|
||||
* model entry in the catalog.
|
||||
*/
|
||||
export const CLOUDFLARE_FALLBACK_MODEL: ModelSpec<"anthropic-messages"> = {
|
||||
id: "claude-sonnet-4-5",
|
||||
name: "Claude Sonnet 4.5",
|
||||
api: "anthropic-messages",
|
||||
provider: "cloudflare-ai-gateway",
|
||||
baseUrl: CLOUDFLARE_AI_GATEWAY_BASE_URL,
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: {
|
||||
input: 3,
|
||||
output: 15,
|
||||
cacheRead: 0.3,
|
||||
cacheWrite: 3.75,
|
||||
},
|
||||
contextWindow: 200000,
|
||||
maxTokens: 64000,
|
||||
};
|
||||
|
||||
const CODEX_GPT_5_4_PRIORITY_BY_VARIANT: Partial<Record<OpenAIVariant, number>> = {
|
||||
base: 0,
|
||||
mini: 1,
|
||||
nano: 2,
|
||||
};
|
||||
|
||||
const COPILOT_GENERATED_LIMITS: Record<string, { contextWindow: number; maxTokens: number }> = {
|
||||
"claude-opus-4.6": { contextWindow: 168000, maxTokens: 32000 },
|
||||
"gpt-5.2": { contextWindow: 272000, maxTokens: 128000 },
|
||||
"gpt-5.4": { contextWindow: 272000, maxTokens: 128000 },
|
||||
"gpt-5.4-mini": { contextWindow: 272000, maxTokens: 128000 },
|
||||
"grok-code-fast-1": { contextWindow: 192000, maxTokens: 64000 },
|
||||
};
|
||||
|
||||
/**
|
||||
* Apply upstream metadata corrections to a mutable array of models, then
|
||||
* re-bake canonical thinking metadata so generated catalogs always carry the
|
||||
* deriver's output for the post-policy spec.
|
||||
*/
|
||||
export function applyGeneratedModelPolicies(models: ModelSpec<Api>[]): void {
|
||||
for (const model of models) {
|
||||
applyGeneratedModelPolicy(model);
|
||||
rebakeModelThinking(model);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Recompute `thinking` from the canonical deriver, replacing any baked value.
|
||||
* Mirrors `buildModel`'s trust-or-derive resolution with trust disabled: the
|
||||
* generator is the authority that produces the trusted values.
|
||||
*/
|
||||
export function rebakeModelThinking(model: ModelSpec<Api>): void {
|
||||
const thinking = resolveModelThinking({ ...model, thinking: undefined }, buildCompat(model));
|
||||
if (thinking) {
|
||||
model.thinking = thinking;
|
||||
} else {
|
||||
delete model.thinking;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Link OpenAI model variants to their context promotion targets.
|
||||
*
|
||||
* When a model's context is exhausted, the agent can promote to a sibling
|
||||
* model with a larger context window on the same provider:
|
||||
* - `codex-spark` variants promote to `gpt-5.5`.
|
||||
* - `gpt-5.5` (270K input) promotes to `gpt-5.4` (1M input).
|
||||
*/
|
||||
export function linkOpenAIPromotionTargets(models: ModelSpec<Api>[]): void {
|
||||
for (const candidate of models) {
|
||||
const parsedCandidate = parseKnownModel(candidate.id);
|
||||
if (parsedCandidate.family !== "openai") continue;
|
||||
let targetId: string | undefined;
|
||||
if (parsedCandidate.variant === "codex-spark") {
|
||||
targetId = "gpt-5.5";
|
||||
} else if (parsedCandidate.variant === "base" && semverEqual(parsedCandidate.version, "5.5")) {
|
||||
targetId = "gpt-5.4";
|
||||
} else {
|
||||
continue;
|
||||
}
|
||||
const fallback = models.find(
|
||||
model => model.provider === candidate.provider && model.api === candidate.api && model.id === targetId,
|
||||
);
|
||||
if (!fallback) continue;
|
||||
candidate.contextPromotionTarget = `${fallback.provider}/${fallback.id}`;
|
||||
}
|
||||
}
|
||||
|
||||
function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
|
||||
const copilotLimits = model.provider === "github-copilot" ? COPILOT_GENERATED_LIMITS[model.id] : undefined;
|
||||
if (copilotLimits) {
|
||||
model.contextWindow = copilotLimits.contextWindow;
|
||||
model.maxTokens = copilotLimits.maxTokens;
|
||||
}
|
||||
|
||||
if (
|
||||
model.api === "openai-completions" &&
|
||||
(model.provider === "minimax-code" || model.provider === "minimax-code-cn")
|
||||
) {
|
||||
model.compat = {
|
||||
...(model.compat ?? {}),
|
||||
supportsStore: false,
|
||||
supportsDeveloperRole: false,
|
||||
supportsReasoningEffort: false,
|
||||
reasoningContentField: "reasoning_content",
|
||||
};
|
||||
delete model.compat.thinkingFormat;
|
||||
}
|
||||
if (
|
||||
model.api === "openai-completions" &&
|
||||
model.provider === "opencode-go" &&
|
||||
(model.id === "deepseek-v4-flash" || model.id === "deepseek-v4-pro")
|
||||
) {
|
||||
model.compat = {
|
||||
...(model.compat ?? {}),
|
||||
supportsToolChoice: false,
|
||||
reasoningContentField: "reasoning_content",
|
||||
requiresReasoningContentForToolCalls: true,
|
||||
};
|
||||
}
|
||||
const parsedModel = parseKnownModel(model.id);
|
||||
const applyPatchToolType = inferGeneratedApplyPatchToolType(model, parsedModel);
|
||||
if (applyPatchToolType) {
|
||||
model.applyPatchToolType = applyPatchToolType;
|
||||
} else {
|
||||
delete model.applyPatchToolType;
|
||||
}
|
||||
if (parsedModel.family === "anthropic") {
|
||||
applyAnthropicCatalogPolicy(model, parsedModel);
|
||||
}
|
||||
if (parsedModel.family === "openai") {
|
||||
applyOpenAICatalogPolicy(model, parsedModel);
|
||||
}
|
||||
}
|
||||
|
||||
function applyAnthropicCatalogPolicy(model: ModelSpec<Api>, parsedModel: AnthropicModel): void {
|
||||
// Claude Opus 4.5: models.dev reports 3x the correct cache pricing.
|
||||
if (model.provider === "anthropic" && parsedModel.kind === "opus" && semverEqual(parsedModel.version, "4.5")) {
|
||||
model.cost.cacheRead = 0.5;
|
||||
model.cost.cacheWrite = 6.25;
|
||||
}
|
||||
|
||||
// Bedrock Opus 4.6: upstream metadata is stale for cache pricing and context.
|
||||
if (model.provider === "amazon-bedrock" && parsedModel.kind === "opus" && semverEqual(parsedModel.version, "4.6")) {
|
||||
model.cost.cacheRead = 0.5;
|
||||
model.cost.cacheWrite = 6.25;
|
||||
model.contextWindow = 1000000;
|
||||
model.maxTokens = 128000;
|
||||
}
|
||||
|
||||
// Claude Fable/Mythos 5: Anthropic's /v1/models omits token limits and
|
||||
// pricing, and models.dev lags new releases. Pin authoritative values from
|
||||
// the model card (1M context / 128k output) and pricing docs ($10 in / $50
|
||||
// out per MTok).
|
||||
if (model.provider === "anthropic" && isFableOrMythos(parsedModel.kind)) {
|
||||
model.contextWindow = 1_000_000;
|
||||
model.maxTokens = 128_000;
|
||||
model.cost.input = 10;
|
||||
model.cost.output = 50;
|
||||
model.cost.cacheRead = 1;
|
||||
model.cost.cacheWrite = 12.5;
|
||||
}
|
||||
}
|
||||
|
||||
function inferGeneratedApplyPatchToolType(
|
||||
model: ModelSpec<Api>,
|
||||
parsedModel: ParsedModel,
|
||||
): ModelSpec<Api>["applyPatchToolType"] {
|
||||
if (parsedModel.family !== "openai" || parsedModel.version.major !== 5) {
|
||||
return undefined;
|
||||
}
|
||||
if (model.provider === "openai" && model.api === "openai-responses") {
|
||||
return "freeform";
|
||||
}
|
||||
if (model.provider === "openai-codex" && model.api === "openai-codex-responses") {
|
||||
return "freeform";
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function applyOpenAICatalogPolicy(model: ModelSpec<Api>, parsedModel: OpenAIModel): void {
|
||||
// Codex models: 400K figure includes output budget; input window is 272K.
|
||||
if (parsedModel.variant.startsWith("codex") && parsedModel.variant !== "codex-spark") {
|
||||
model.contextWindow = 272000;
|
||||
return;
|
||||
}
|
||||
// GPT-5.4 mini/nano use plain OpenAI IDs on the Codex transport, but Codex still
|
||||
// enforces the lower prompt budget for these variants. Codex discovery can also
|
||||
// report inconsistent priorities for the GPT-5.4 family, so normalize by parsed
|
||||
// variant instead of special-casing raw model ids.
|
||||
if (model.api === "openai-codex-responses" && semverEqual(parsedModel.version, "5.4")) {
|
||||
const normalizedPriority = CODEX_GPT_5_4_PRIORITY_BY_VARIANT[parsedModel.variant];
|
||||
if (normalizedPriority !== undefined) {
|
||||
model.priority = normalizedPriority;
|
||||
}
|
||||
if (parsedModel.variant === "mini" || parsedModel.variant === "nano") {
|
||||
model.contextWindow = 272000;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,24 +1,30 @@
|
||||
/**
|
||||
* The single Model constructor. Thinking metadata and the resolved compat
|
||||
* record are materialized here, exactly once per spec — request handlers read
|
||||
* `model.compat` fields and perform zero URL parsing and zero compat
|
||||
* allocation per request.
|
||||
* The single Model constructor. Resolution order is a dependency chain, each
|
||||
* step materialized exactly once per spec:
|
||||
*
|
||||
* 1. compat — URL/provider/id detection resolved into a complete record;
|
||||
* 2. thinking — derived from identity + resolved compat (or trusted verbatim
|
||||
* when the spec carries explicit metadata);
|
||||
*
|
||||
* Request handlers read fields — they never detect, parse ids, or allocate
|
||||
* compat per request.
|
||||
*/
|
||||
import { buildAnthropicCompat } from "./compat/anthropic";
|
||||
import { buildOpenAICompat, buildOpenAIResponsesCompat } from "./compat/openai";
|
||||
import { enrichModelThinking } from "./model-thinking";
|
||||
import { resolveModelThinking } from "./model-thinking";
|
||||
import type { Api, CompatOf, Model, ModelSpec } from "./types";
|
||||
|
||||
export function buildModel<TApi extends Api>(spec: ModelSpec<TApi>): Model<TApi> {
|
||||
const enriched = enrichModelThinking(spec);
|
||||
const compat = buildCompat(spec) as CompatOf<TApi>;
|
||||
return {
|
||||
...enriched,
|
||||
compat: buildCompat(enriched) as CompatOf<TApi>,
|
||||
compatConfig: enriched.compat,
|
||||
...spec,
|
||||
thinking: resolveModelThinking(spec, compat),
|
||||
compat,
|
||||
compatConfig: spec.compat,
|
||||
} as Model<TApi>;
|
||||
}
|
||||
|
||||
function buildCompat(spec: ModelSpec<Api>): CompatOf<Api> {
|
||||
export function buildCompat(spec: ModelSpec<Api>): CompatOf<Api> {
|
||||
switch (spec.api) {
|
||||
case "openai-completions":
|
||||
return buildOpenAICompat(spec as ModelSpec<"openai-completions">);
|
||||
|
||||
@@ -5,7 +5,11 @@
|
||||
* classification, with explicit spec overrides assigned on top.
|
||||
*/
|
||||
import { modelMatchesHost } from "../hosts";
|
||||
import { isAnthropicFableOrMythosModel, supportsMidConversationSystemMessages } from "../model-thinking";
|
||||
import {
|
||||
hasOpus47ApiRestrictions,
|
||||
isAnthropicFableOrMythosModel,
|
||||
supportsMidConversationSystemMessages,
|
||||
} from "../identity/family";
|
||||
import type { ModelSpec, ResolvedAnthropicCompat } from "../types";
|
||||
import { applyCompatOverrides } from "./apply";
|
||||
|
||||
@@ -44,6 +48,8 @@ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): Res
|
||||
// supported model id.
|
||||
supportsMidConversationSystem: official && supportsMidConversationSystemMessages(spec.id),
|
||||
supportsForcedToolChoice: !isAnthropicFableOrMythosModel(spec.id),
|
||||
// Opus 4.7+ and Fable/Mythos reject temperature/top_p/top_k with a 400.
|
||||
supportsSamplingParams: !hasOpus47ApiRestrictions(spec.id),
|
||||
// Z.AI workaround (issue #814): its proxy deserializes tool_result blocks
|
||||
// into a class that reads `.id`.
|
||||
requiresToolResultId: isZai,
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
* never detect, resolve, or allocate.
|
||||
*/
|
||||
import { hostMatchesUrl, modelMatchesHost } from "../hosts";
|
||||
import { bareModelId, isFableOrMythos, parseAnthropicModel, semverGte } from "../identity/classify";
|
||||
import {
|
||||
isAnthropicNamespacedModelId,
|
||||
isClaudeModelId,
|
||||
@@ -17,6 +18,7 @@ import {
|
||||
isMimoModelIdOrName,
|
||||
isQwenModelId,
|
||||
} from "../identity/family";
|
||||
import { ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER, ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER } from "../model-thinking";
|
||||
import type { ModelSpec, OpenAICompat, ResolvedOpenAICompat, ResolvedOpenAIResponsesCompat } from "../types";
|
||||
import { applyCompatOverrides } from "./apply";
|
||||
|
||||
@@ -73,30 +75,17 @@ function detectStrictModeSupport(provider: string, baseUrl: string): boolean {
|
||||
function getOpenRouterAnthropicReasoningEffortMap(
|
||||
modelId: string,
|
||||
): Partial<Record<OpenAIReasoningEffort, string>> | undefined {
|
||||
const match = /(?:^|\/)claude-(opus|fable|mythos)-(\d{1,2})(?:[.-](\d{1,2}))?/.exec(modelId);
|
||||
if (!match) return undefined;
|
||||
const parsed = parseAnthropicModel(bareModelId(modelId));
|
||||
if (!parsed) return undefined;
|
||||
// Adaptive efforts on OpenRouter's completions front: Fable/Mythos and
|
||||
// Opus 4.6+ only — Sonnet stays on the plain effort vocabulary there.
|
||||
const isOpusAdaptive = parsed.kind === "opus" && semverGte(parsed.version, "4.6");
|
||||
if (!isFableOrMythos(parsed.kind) && !isOpusAdaptive) return undefined;
|
||||
|
||||
const kind = match[1];
|
||||
const major = Number(match[2]);
|
||||
const minor = Number(match[3] ?? 0);
|
||||
const isFableOrMythos = kind === "fable" || kind === "mythos";
|
||||
const isOpusAdaptive = kind === "opus" && (major > 4 || (major === 4 && minor >= 6));
|
||||
if (!isFableOrMythos && !isOpusAdaptive) return undefined;
|
||||
|
||||
const hasRealXHigh = isFableOrMythos || major > 4 || (major === 4 && minor >= 7);
|
||||
if (hasRealXHigh) {
|
||||
return {
|
||||
minimal: "low",
|
||||
low: "medium",
|
||||
medium: "high",
|
||||
high: "xhigh",
|
||||
xhigh: "max",
|
||||
};
|
||||
}
|
||||
return {
|
||||
minimal: "low",
|
||||
xhigh: "max",
|
||||
};
|
||||
const hasRealXHigh = isFableOrMythos(parsed.kind) || semverGte(parsed.version, "4.7");
|
||||
return (hasRealXHigh ? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER) as Partial<
|
||||
Record<OpenAIReasoningEffort, string>
|
||||
>;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -7,6 +7,8 @@
|
||||
* here.
|
||||
*/
|
||||
|
||||
import { bareModelId, isFableOrMythos, parseAnthropicModel, semverGte } from "./classify";
|
||||
|
||||
/** Kimi family ids in any namespace form (`moonshotai/kimi-*`, `kimi-k2.6`, `vendor/kimi.x`). */
|
||||
export function isKimiModelId(modelId: string): boolean {
|
||||
return modelId.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(modelId);
|
||||
@@ -44,16 +46,43 @@ export function isMimoModelIdOrName(value: string): boolean {
|
||||
|
||||
/**
|
||||
* Adaptive thinking `display` is supported starting with Claude Opus 4.7 and
|
||||
* Claude Fable/Mythos 5. Older adaptive-thinking models (Opus 4.6, Sonnet
|
||||
* 4.6+) reject the field.
|
||||
* the Claude Fable/Mythos 5 generation. Older adaptive-thinking models
|
||||
* (Opus 4.6, Sonnet 4.6+) reject the field. Classifier-based, so dotted and
|
||||
* dashed version forms both match while bare dated ids
|
||||
* (`claude-opus-4-20250514` = Opus 4.0) stay excluded.
|
||||
*/
|
||||
export function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
|
||||
if (/claude-(?:fable|mythos)-5\b/.test(modelId)) return true;
|
||||
// Bound the minor to non-date digits: bare dated ids like
|
||||
// `claude-opus-4-20250514` (Opus 4.0) must not parse as minor=20250514.
|
||||
const match = /claude-opus-(\d+)-(\d{1,2})(?!\d)/.exec(modelId);
|
||||
if (!match) return false;
|
||||
const major = Number(match[1]);
|
||||
const minor = Number(match[2]);
|
||||
return major > 4 || (major === 4 && minor >= 7);
|
||||
const parsed = parseAnthropicModel(bareModelId(modelId));
|
||||
if (!parsed) return false;
|
||||
if (isFableOrMythos(parsed.kind)) return semverGte(parsed.version, "5");
|
||||
return parsed.kind === "opus" && semverGte(parsed.version, "4.7");
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns true for Anthropic models with Opus 4.7+/Fable/Mythos API restrictions:
|
||||
* - Sampling parameters (temperature/top_p/top_k) return 400 error
|
||||
* - Thinking content is omitted by default (needs display: "summarized")
|
||||
*/
|
||||
export function hasOpus47ApiRestrictions(modelId: string): boolean {
|
||||
const parsed = parseAnthropicModel(bareModelId(modelId));
|
||||
if (!parsed) return false;
|
||||
return (parsed.kind === "opus" && semverGte(parsed.version, "4.7")) || isFableOrMythos(parsed.kind);
|
||||
}
|
||||
|
||||
/**
|
||||
* Mid-conversation `role: "system"` messages (system instructions appended at
|
||||
* non-first positions in the `messages` array) are supported starting with
|
||||
* Claude Opus 4.8 and the Claude Fable/Mythos 5 generation. Earlier Claude
|
||||
* models reject the role.
|
||||
* @see https://platform.claude.com/docs/en/build-with-claude/mid-conversation-system-messages
|
||||
*/
|
||||
export function supportsMidConversationSystemMessages(modelId: string): boolean {
|
||||
const parsed = parseAnthropicModel(bareModelId(modelId));
|
||||
if (!parsed) return false;
|
||||
return (parsed.kind === "opus" && semverGte(parsed.version, "4.8")) || isFableOrMythos(parsed.kind);
|
||||
}
|
||||
|
||||
export function isAnthropicFableOrMythosModel(modelId: string): boolean {
|
||||
const parsed = parseAnthropicModel(bareModelId(modelId));
|
||||
return parsed !== null && isFableOrMythos(parsed.kind);
|
||||
}
|
||||
|
||||
@@ -7,9 +7,9 @@ import { getModelDbPath } from "@oh-my-pi/pi-utils";
|
||||
import type { Api, Model, ModelSpec } from "./types";
|
||||
|
||||
// Rows persist ModelSpec JSON (sparse `compat`, never the resolved record);
|
||||
// the model manager rebuilds via `buildModel` on load. v3 rows predating the
|
||||
// resolved-compat redesign already carried sparse compat, so they stay valid.
|
||||
const CACHE_SCHEMA_VERSION = 3;
|
||||
// the model manager rebuilds via `buildModel` on load. v4 invalidates rows
|
||||
// carrying the pre-efforts ThinkingConfig shape (minLevel/maxLevel/levels).
|
||||
const CACHE_SCHEMA_VERSION = 4;
|
||||
|
||||
interface CacheRow {
|
||||
provider_id: string;
|
||||
|
||||
@@ -1,24 +1,37 @@
|
||||
import { buildOpenAICompat } from "./compat/openai";
|
||||
/**
|
||||
* Thinking metadata: build-time derivation and runtime field-read helpers.
|
||||
*
|
||||
* Derivation (`resolveModelThinking`) runs exactly once per model — from
|
||||
* `buildModel` for dynamic specs and from the catalog generator for bundled
|
||||
* entries. Everything below the "runtime helpers" divider reads baked fields
|
||||
* only: no id parsing, no host matching, no compat detection per request.
|
||||
*/
|
||||
import { Effort, THINKING_EFFORTS } from "./effort";
|
||||
import { modelMatchesHost } from "./hosts";
|
||||
import {
|
||||
type AnthropicModel,
|
||||
bareModelId,
|
||||
type GeminiModel,
|
||||
isFableOrMythos,
|
||||
type OpenAIModel,
|
||||
type OpenAIVariant,
|
||||
type ParsedModel,
|
||||
parseAnthropicModel,
|
||||
parseKnownModel,
|
||||
semverEqual,
|
||||
semverGte,
|
||||
} from "./identity/classify";
|
||||
import type { Api, Model, ModelSpec, ThinkingConfig } from "./types";
|
||||
import { supportsAdaptiveThinkingDisplay } from "./identity/family";
|
||||
import type {
|
||||
Api,
|
||||
CompatOf,
|
||||
Model,
|
||||
ModelSpec,
|
||||
ResolvedOpenAICompat,
|
||||
ResolvedOpenAIResponsesCompat,
|
||||
ThinkingConfig,
|
||||
} from "./types";
|
||||
|
||||
/**
|
||||
* Thinking inference reads identity fields plus sparse compat intent, so it
|
||||
* accepts both pre-build specs and built models.
|
||||
* Runtime helpers read baked metadata only, so they accept both pre-build
|
||||
* specs and built models.
|
||||
*/
|
||||
type ApiModel<TApi extends Api = Api> = ModelSpec<TApi> | Model<TApi>;
|
||||
|
||||
@@ -34,190 +47,286 @@ const GEMINI_3_PRO_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High];
|
||||
const GEMINI_3_FLASH_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High];
|
||||
const GPT_5_2_PLUS_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh];
|
||||
const GPT_5_1_CODEX_MINI_EFFORTS: readonly Effort[] = [Effort.Medium, Effort.High];
|
||||
const CLOUDFLARE_AI_GATEWAY_BASE_URL = "https://gateway.ai.cloudflare.com/v1/<account>/<gateway>/anthropic";
|
||||
|
||||
const CODEX_GPT_5_4_PRIORITY_BY_VARIANT: Partial<Record<OpenAIVariant, number>> = {
|
||||
base: 0,
|
||||
mini: 1,
|
||||
nano: 2,
|
||||
};
|
||||
|
||||
const COPILOT_GENERATED_LIMITS: Record<string, { contextWindow: number; maxTokens: number }> = {
|
||||
"claude-opus-4.6": { contextWindow: 168000, maxTokens: 32000 },
|
||||
"gpt-5.2": { contextWindow: 272000, maxTokens: 128000 },
|
||||
"gpt-5.4": { contextWindow: 272000, maxTokens: 128000 },
|
||||
"gpt-5.4-mini": { contextWindow: 272000, maxTokens: 128000 },
|
||||
"grok-code-fast-1": { contextWindow: 192000, maxTokens: 64000 },
|
||||
/**
|
||||
* Effort → wire-value map for the 5-tier adaptive scale (Opus 4.7+ and
|
||||
* Fable/Mythos 5 on the Messages API). User-facing efforts shift up one notch
|
||||
* so the top tier reaches the genuine "max" and "high" lands on Anthropic's
|
||||
* recommended "xhigh" coding/agentic default.
|
||||
*/
|
||||
export const ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER: Readonly<Partial<Record<Effort, string>>> = {
|
||||
[Effort.Minimal]: "low",
|
||||
[Effort.Low]: "medium",
|
||||
[Effort.Medium]: "high",
|
||||
[Effort.High]: "xhigh",
|
||||
[Effort.XHigh]: "max",
|
||||
};
|
||||
|
||||
/**
|
||||
* Static fallback model injected when Cloudflare AI Gateway discovery
|
||||
* returns no results. Ensures the provider always has at least one usable
|
||||
* model entry in the catalog.
|
||||
* Effort → wire-value map for the legacy 4-tier adaptive scale (Opus 4.6,
|
||||
* Sonnet 4.6+, and every adaptive model on Bedrock Converse). `low..high` pass
|
||||
* through verbatim; there is no real "xhigh", so it aliases the top "max" tier.
|
||||
*/
|
||||
export const CLOUDFLARE_FALLBACK_MODEL: ApiModel<"anthropic-messages"> = {
|
||||
id: "claude-sonnet-4-5",
|
||||
name: "Claude Sonnet 4.5",
|
||||
api: "anthropic-messages",
|
||||
provider: "cloudflare-ai-gateway",
|
||||
baseUrl: CLOUDFLARE_AI_GATEWAY_BASE_URL,
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: {
|
||||
input: 3,
|
||||
output: 15,
|
||||
cacheRead: 0.3,
|
||||
cacheWrite: 3.75,
|
||||
},
|
||||
contextWindow: 200000,
|
||||
maxTokens: 64000,
|
||||
export const ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER: Readonly<Partial<Record<Effort, string>>> = {
|
||||
[Effort.Minimal]: "low",
|
||||
[Effort.XHigh]: "max",
|
||||
};
|
||||
|
||||
const kEnrichedModel = Symbol("model-thinking.enrichedModel");
|
||||
type ModelWithEnriched = ApiModel<Api> & { [kEnrichedModel]?: ApiModel<Api> };
|
||||
// ---------------------------------------------------------------------------
|
||||
// Build-time derivation (buildModel + catalog generator only)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Returns a copy of the model with canonical thinking metadata attached.
|
||||
* Resolve the canonical thinking metadata for a spec. Called exactly once per
|
||||
* model by `buildModel`, after compat resolution.
|
||||
*
|
||||
* This helper belongs to catalog enrichment only. Runtime consumers should
|
||||
* trust `model.thinking` and avoid inferring capabilities on demand.
|
||||
* - Non-reasoning models never carry thinking.
|
||||
* - Models that reason natively but reject the wire effort param
|
||||
* (`compat.supportsReasoningEffort: false` on openai-responses*) carry no
|
||||
* thinking either: `reasoning: true, thinking: undefined` IS the encoding
|
||||
* for "thinks, but exposes no control surface".
|
||||
* - Explicit spec thinking (generator-baked or user-authored) owns the
|
||||
* capability surface (`mode`, `efforts`, `defaultLevel`); the wire facts
|
||||
* (`effortMap`, `supportsDisplay`) are backfilled from identity when not
|
||||
* explicitly set, so configs never need to know Anthropic's tier tables.
|
||||
* - Sparse specs go through full inference.
|
||||
*/
|
||||
export function enrichModelThinking<TApi extends Api>(model: ModelSpec<TApi>): ModelSpec<TApi>;
|
||||
export function enrichModelThinking<TApi extends Api>(model: Model<TApi>): Model<TApi>;
|
||||
export function enrichModelThinking<TApi extends Api>(model: ApiModel<TApi>): ApiModel<TApi> {
|
||||
const tagged = model as ModelWithEnriched;
|
||||
const cached = tagged[kEnrichedModel];
|
||||
if (cached !== undefined) {
|
||||
return cached as ApiModel<TApi>;
|
||||
export function resolveModelThinking<TApi extends Api>(
|
||||
spec: ModelSpec<TApi>,
|
||||
compat: CompatOf<TApi>,
|
||||
): ThinkingConfig | undefined {
|
||||
if (!spec.reasoning) return undefined;
|
||||
if (omitsWireReasoningEffort(spec.api, compat)) return undefined;
|
||||
if (spec.thinking && spec.thinking.efforts.length > 0) {
|
||||
return fillThinkingWireDefaults(spec, spec.thinking);
|
||||
}
|
||||
const normalizedThinking = normalizeThinkingConfig(model.thinking);
|
||||
let result: ApiModel<TApi>;
|
||||
if (!model.reasoning) {
|
||||
result =
|
||||
normalizedThinking === undefined && model.thinking === undefined ? model : { ...model, thinking: undefined };
|
||||
} else {
|
||||
const thinking = normalizedThinking ?? inferModelThinking(model);
|
||||
result = thinkingsEqual(normalizedThinking, thinking) ? model : { ...model, thinking };
|
||||
}
|
||||
// Stash the enriched copy on a non-enumerable slot so callers that hand us
|
||||
// the same reference twice skip the work. `enumerable: false` is critical:
|
||||
// many call sites build derived models via `{ ...model, ...overrides }`,
|
||||
// which would otherwise copy this cache slot and trick us into returning
|
||||
// the *original* enriched model — silently discarding the overrides.
|
||||
Object.defineProperty(tagged, kEnrichedModel, {
|
||||
value: result,
|
||||
enumerable: false,
|
||||
configurable: true,
|
||||
writable: true,
|
||||
});
|
||||
return result;
|
||||
// Empty/malformed explicit metadata is treated as absent — infer instead.
|
||||
return deriveThinking(spec, compat);
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns a copy of the model with thinking metadata recomputed from the
|
||||
* canonical rules, replacing any existing `thinking`.
|
||||
* Backfill identity-derived wire facts onto explicit thinking metadata.
|
||||
* Explicit `effortMap` / `supportsDisplay` (including `false`) always win;
|
||||
* untouched configs are returned as-is with zero allocation.
|
||||
*/
|
||||
export function refreshModelThinking<TApi extends Api>(model: ApiModel<TApi>): ApiModel<TApi> {
|
||||
if (!model.reasoning) {
|
||||
const normalizedThinking = normalizeThinkingConfig(model.thinking);
|
||||
return normalizedThinking === undefined && model.thinking === undefined
|
||||
? model
|
||||
: { ...model, thinking: undefined };
|
||||
function fillThinkingWireDefaults<TApi extends Api>(spec: ModelSpec<TApi>, thinking: ThinkingConfig): ThinkingConfig {
|
||||
const needsEffortMap = thinking.mode === "anthropic-adaptive" && thinking.effortMap === undefined;
|
||||
const needsDisplay =
|
||||
thinking.supportsDisplay === undefined &&
|
||||
(spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") &&
|
||||
supportsAdaptiveThinkingDisplay(spec.id);
|
||||
if (!needsEffortMap && !needsDisplay) {
|
||||
return thinking;
|
||||
}
|
||||
return { ...model, thinking: inferModelThinking(model) };
|
||||
const filled: ThinkingConfig = { ...thinking };
|
||||
if (needsEffortMap) {
|
||||
filled.effortMap = anthropicModelHasRealXHighEffort(spec, parseKnownModel(spec.id))
|
||||
? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER
|
||||
: ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER;
|
||||
}
|
||||
if (needsDisplay) {
|
||||
filled.supportsDisplay = true;
|
||||
}
|
||||
return filled;
|
||||
}
|
||||
|
||||
/**
|
||||
* Apply upstream metadata corrections to a mutable array of models.
|
||||
*
|
||||
* Each model is first normalized through `refreshModelThinking()` so generated
|
||||
* catalogs keep canonical thinking metadata and policy fixes in one pass.
|
||||
*/
|
||||
export function applyGeneratedModelPolicies(models: ApiModel<Api>[]): void {
|
||||
for (let index = 0; index < models.length; index++) {
|
||||
const model = refreshModelThinking(models[index]!);
|
||||
applyGeneratedModelPolicy(model);
|
||||
models[index] = model;
|
||||
/** Derive thinking from identity + resolved compat, ignoring any baked value. Generator-side entry. */
|
||||
export function deriveThinking<TApi extends Api>(spec: ModelSpec<TApi>, compat: CompatOf<TApi>): ThinkingConfig {
|
||||
const parsed = parseKnownModel(spec.id);
|
||||
const efforts = inferSupportedEfforts(parsed, spec, compat);
|
||||
if (efforts.length === 0) {
|
||||
throw new Error(`Model ${spec.provider}/${spec.id} resolved to an empty thinking range`);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Link OpenAI model variants to their context promotion targets.
|
||||
*
|
||||
* When a model's context is exhausted, the agent can promote to a sibling
|
||||
* model with a larger context window on the same provider:
|
||||
* - `codex-spark` variants promote to `gpt-5.5`.
|
||||
* - `gpt-5.5` (270K input) promotes to `gpt-5.4` (1M input).
|
||||
*/
|
||||
export function linkOpenAIPromotionTargets(models: ApiModel<Api>[]): void {
|
||||
for (const candidate of models) {
|
||||
const parsedCandidate = parseKnownModel(candidate.id);
|
||||
if (parsedCandidate.family !== "openai") continue;
|
||||
let targetId: string | undefined;
|
||||
if (parsedCandidate.variant === "codex-spark") {
|
||||
targetId = "gpt-5.5";
|
||||
} else if (parsedCandidate.variant === "base" && semverEqual(parsedCandidate.version, "5.5")) {
|
||||
targetId = "gpt-5.4";
|
||||
} else {
|
||||
continue;
|
||||
}
|
||||
const fallback = models.find(
|
||||
model => model.provider === candidate.provider && model.api === candidate.api && model.id === targetId,
|
||||
);
|
||||
if (!fallback) continue;
|
||||
candidate.contextPromotionTarget = `${fallback.provider}/${fallback.id}`;
|
||||
const config: ThinkingConfig = {
|
||||
mode: inferThinkingControlMode(spec, parsed),
|
||||
efforts,
|
||||
};
|
||||
if (config.mode === "anthropic-adaptive") {
|
||||
config.effortMap = anthropicModelHasRealXHighEffort(spec, parsed)
|
||||
? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER
|
||||
: ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER;
|
||||
}
|
||||
if (
|
||||
(spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") &&
|
||||
supportsAdaptiveThinkingDisplay(spec.id)
|
||||
) {
|
||||
config.supportsDisplay = true;
|
||||
}
|
||||
return config;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when the model reasons natively but rejects the wire `reasoning.effort`
|
||||
* param (compat.supportsReasoningEffort: false on openai-responses*). Callers
|
||||
* are expected to omit the effort field; the wire-side omitReasoningEffort
|
||||
* gate (providers/xai-responses.ts:78) is the actual strip, and this
|
||||
* predicate is the upstream check that prevents a redundant
|
||||
* requireSupportedEffort throw from defeating that gate.
|
||||
*
|
||||
* Scoped to openai-responses* because that's the only API surface where
|
||||
* `compat.supportsReasoningEffort: false` is meaningful today. The
|
||||
* `in`-narrowed access is necessary because Model.compat is
|
||||
* `AnthropicCompat | OpenAICompat` and the api gate doesn't narrow the
|
||||
* union for TS.
|
||||
* param. Scoped to openai-responses* because that's the only API surface where
|
||||
* `compat.supportsReasoningEffort: false` means "omit the field entirely"
|
||||
* (xAI Grok off the GROK_EFFORT_CAPABLE_PREFIXES allowlist: grok-build,
|
||||
* grok-4.20-0309-reasoning). openai-completions keeps its thinking config even
|
||||
* without effort support — binary thinking formats (zai/qwen) drive reasoning
|
||||
* through other request fields.
|
||||
*/
|
||||
export function modelOmitsReasoningEffort<TApi extends Api>(model: ApiModel<TApi>): boolean {
|
||||
if (model.api !== "openai-responses" && model.api !== "openai-codex-responses") {
|
||||
function omitsWireReasoningEffort(api: Api, compat: CompatOf<Api>): boolean {
|
||||
if (api !== "openai-responses" && api !== "openai-codex-responses") {
|
||||
return false;
|
||||
}
|
||||
const compat = model.compat;
|
||||
return Boolean(compat && "supportsReasoningEffort" in compat && compat.supportsReasoningEffort === false);
|
||||
return (compat as ResolvedOpenAIResponsesCompat | undefined)?.supportsReasoningEffort === false;
|
||||
}
|
||||
|
||||
function inferSupportedEfforts<TApi extends Api>(
|
||||
parsedModel: ParsedModel,
|
||||
spec: ModelSpec<TApi>,
|
||||
compat: CompatOf<TApi>,
|
||||
): readonly Effort[] {
|
||||
switch (parsedModel.family) {
|
||||
case "openai":
|
||||
return inferOpenAISupportedEfforts(parsedModel);
|
||||
case "gemini":
|
||||
return inferGeminiSupportedEfforts(parsedModel);
|
||||
case "anthropic":
|
||||
return inferAnthropicSupportedEfforts(parsedModel, spec, compat);
|
||||
case "unknown":
|
||||
return inferFallbackEfforts(spec, compat);
|
||||
}
|
||||
}
|
||||
|
||||
function inferOpenAISupportedEfforts(model: OpenAIModel): readonly Effort[] {
|
||||
if (model.variant === "codex-mini" && semverEqual(model.version, "5.1")) {
|
||||
return GPT_5_1_CODEX_MINI_EFFORTS;
|
||||
}
|
||||
if (semverGte(model.version, "5.2")) {
|
||||
return GPT_5_2_PLUS_EFFORTS;
|
||||
}
|
||||
return DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
|
||||
function inferGeminiSupportedEfforts(model: GeminiModel): readonly Effort[] {
|
||||
if (!semverGte(model.version, "3.0")) {
|
||||
return DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
return model.kind === "pro" ? GEMINI_3_PRO_EFFORTS : GEMINI_3_FLASH_EFFORTS;
|
||||
}
|
||||
|
||||
function inferAnthropicSupportedEfforts<TApi extends Api>(
|
||||
parsedModel: AnthropicModel,
|
||||
spec: ModelSpec<TApi>,
|
||||
compat: CompatOf<TApi>,
|
||||
): readonly Effort[] {
|
||||
if (
|
||||
(spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") &&
|
||||
semverGte(parsedModel.version, "4.6")
|
||||
) {
|
||||
return parsedModel.kind === "opus" || isFableOrMythos(parsedModel.kind)
|
||||
? DEFAULT_REASONING_EFFORTS_WITH_XHIGH
|
||||
: DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
if (isOpenRouterAnthropicAdaptiveReasoningModel(parsedModel, spec)) {
|
||||
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
||||
}
|
||||
return inferFallbackEfforts(spec, compat);
|
||||
}
|
||||
|
||||
function inferFallbackEfforts<TApi extends Api>(spec: ModelSpec<TApi>, compat: CompatOf<TApi>): readonly Effort[] {
|
||||
if (spec.api === "anthropic-messages") {
|
||||
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
||||
}
|
||||
if (spec.name.includes("deepseek-v4")) {
|
||||
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
||||
}
|
||||
if (spec.api === "bedrock-converse-stream") {
|
||||
return DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
if (spec.api === "openai-completions") {
|
||||
const resolved = compat as ResolvedOpenAICompat;
|
||||
if (resolved.thinkingFormat === "openai" && resolved.supportsReasoningEffort) {
|
||||
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
||||
}
|
||||
return DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
// OpenAI Responses APIs encode discrete effort levels, including xhigh.
|
||||
if (spec.api === "openai-responses" || spec.api === "openai-codex-responses") {
|
||||
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
||||
}
|
||||
return DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
|
||||
function inferThinkingControlMode<TApi extends Api>(
|
||||
spec: ModelSpec<TApi>,
|
||||
parsedModel: ParsedModel,
|
||||
): ThinkingConfig["mode"] {
|
||||
switch (spec.api) {
|
||||
case "google-generative-ai":
|
||||
case "google-gemini-cli":
|
||||
case "google-vertex":
|
||||
return parsedModel.family === "gemini" &&
|
||||
semverGte(parsedModel.version, "3.0") &&
|
||||
parsedModel.version.major === 3
|
||||
? "google-level"
|
||||
: "budget";
|
||||
|
||||
case "anthropic-messages":
|
||||
if (parsedModel.family === "anthropic") {
|
||||
if (semverGte(parsedModel.version, "4.6")) {
|
||||
return "anthropic-adaptive";
|
||||
}
|
||||
if (semverGte(parsedModel.version, "4.5")) {
|
||||
return "anthropic-budget-effort";
|
||||
}
|
||||
}
|
||||
return "budget";
|
||||
|
||||
case "bedrock-converse-stream":
|
||||
if (parsedModel.family === "anthropic") {
|
||||
if (
|
||||
semverGte(parsedModel.version, "4.6") &&
|
||||
(parsedModel.kind === "opus" || isFableOrMythos(parsedModel.kind))
|
||||
) {
|
||||
return "anthropic-adaptive";
|
||||
}
|
||||
if (semverGte(parsedModel.version, "4.5")) {
|
||||
return "anthropic-budget-effort";
|
||||
}
|
||||
}
|
||||
return "budget";
|
||||
|
||||
default:
|
||||
return "effort";
|
||||
}
|
||||
}
|
||||
|
||||
function isOpenRouterAnthropicAdaptiveReasoningModel<TApi extends Api>(
|
||||
parsedModel: AnthropicModel,
|
||||
spec: ModelSpec<TApi>,
|
||||
): boolean {
|
||||
if (spec.api !== "openai-completions") return false;
|
||||
if (!modelMatchesHost(spec, "openrouter")) return false;
|
||||
return isFableOrMythos(parsedModel.kind) || (parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.6"));
|
||||
}
|
||||
|
||||
/**
|
||||
* Opus 4.7+ and Fable/Mythos on the Messages API expose the full five-tier
|
||||
* adaptive scale (low/medium/high/xhigh/max). Bedrock Converse stays on the
|
||||
* four-tier scale regardless of model version.
|
||||
*/
|
||||
function anthropicModelHasRealXHighEffort<TApi extends Api>(spec: ModelSpec<TApi>, parsedModel: ParsedModel): boolean {
|
||||
if (spec.api !== "anthropic-messages") return false;
|
||||
if (parsedModel.family !== "anthropic") return false;
|
||||
if (isFableOrMythos(parsedModel.kind)) return true;
|
||||
return parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.7");
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Runtime helpers (field reads only — safe per request)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Returns the supported thinking efforts declared on the model metadata.
|
||||
*
|
||||
* Catalog enrichment is responsible for normalizing bundled model metadata up front.
|
||||
* Runtime callers must treat explicit `model.thinking` on custom models as authoritative
|
||||
* so proxy-specific overrides from `models.yml` survive request construction.
|
||||
*
|
||||
* @throws Error when a reasoning-capable model is missing thinking metadata
|
||||
* Empty for non-reasoning models and for reasoning models without a
|
||||
* controllable effort surface (`thinking: undefined`).
|
||||
*/
|
||||
export function getSupportedEfforts<TApi extends Api>(model: ApiModel<TApi>): readonly Effort[] {
|
||||
if (!model.reasoning) {
|
||||
return [];
|
||||
}
|
||||
// Models that reason natively but reject the `reasoning.effort` wire param
|
||||
// (xAI Grok off the GROK_EFFORT_CAPABLE_PREFIXES allowlist in
|
||||
// providers/xai-responses.ts: grok-build, grok-4.20-0309-reasoning) hide the
|
||||
// picker's effort dial. Scoped to openai-responses* by
|
||||
// `modelOmitsReasoningEffort` — openai-completions has its own
|
||||
// supportsReasoningEffort consultation at inferFallbackEfforts L536 and
|
||||
// changing that path's semantics is out-of-scope.
|
||||
if (modelOmitsReasoningEffort(model)) {
|
||||
return [];
|
||||
}
|
||||
if (!model.thinking) {
|
||||
throw new Error(`Model ${model.provider}/${model.id} is missing thinking metadata`);
|
||||
}
|
||||
return expandEffortRange(model.thinking);
|
||||
return model.thinking?.efforts ?? [];
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -271,11 +380,8 @@ export function requireSupportedEffort<TApi extends Api>(model: ApiModel<TApi>,
|
||||
}
|
||||
|
||||
/** Maps a normalized thinking effort to Google's `thinkingLevel` enum values. */
|
||||
export function mapEffortToGoogleThinkingLevel<TApi extends Api>(
|
||||
model: ApiModel<TApi>,
|
||||
effort: Effort,
|
||||
): "MINIMAL" | "LOW" | "MEDIUM" | "HIGH" {
|
||||
switch (requireSupportedEffort(model, effort)) {
|
||||
export function mapEffortToGoogleThinkingLevel(effort: Effort): "MINIMAL" | "LOW" | "MEDIUM" | "HIGH" {
|
||||
switch (effort) {
|
||||
case Effort.Minimal:
|
||||
return "MINIMAL";
|
||||
case Effort.Low:
|
||||
@@ -288,371 +394,14 @@ export function mapEffortToGoogleThinkingLevel<TApi extends Api>(
|
||||
}
|
||||
}
|
||||
|
||||
/** Maps a normalized thinking effort to Anthropic adaptive effort values. */
|
||||
/**
|
||||
* Maps a normalized thinking effort to Anthropic adaptive effort values via
|
||||
* the model's baked `thinking.effortMap` (identity for unmapped efforts).
|
||||
*/
|
||||
export function mapEffortToAnthropicAdaptiveEffort<TApi extends Api>(
|
||||
model: ApiModel<TApi>,
|
||||
effort: Effort,
|
||||
): "low" | "medium" | "high" | "xhigh" | "max" {
|
||||
const supported = requireSupportedEffort(model, effort);
|
||||
if (anthropicModelHasRealXHighEffort(model)) {
|
||||
// Opus 4.7+ and Fable/Mythos 5 on the Messages API expose the full
|
||||
// five-tier adaptive scale
|
||||
// (low/medium/high/xhigh/max). Shift our user-facing efforts up one notch so
|
||||
// the top tier reaches the genuine "max" and "high" lands on Anthropic's
|
||||
// recommended "xhigh" coding/agentic default.
|
||||
switch (supported) {
|
||||
case Effort.Minimal:
|
||||
return "low";
|
||||
case Effort.Low:
|
||||
return "medium";
|
||||
case Effort.Medium:
|
||||
return "high";
|
||||
case Effort.High:
|
||||
return "xhigh";
|
||||
case Effort.XHigh:
|
||||
return "max";
|
||||
}
|
||||
}
|
||||
// Older adaptive models (Opus 4.6) and Bedrock Converse expose only four tiers
|
||||
// with no real "xhigh"; XHigh is a legacy alias for the top "max" tier there.
|
||||
switch (supported) {
|
||||
case Effort.Minimal:
|
||||
case Effort.Low:
|
||||
return "low";
|
||||
case Effort.Medium:
|
||||
return "medium";
|
||||
case Effort.High:
|
||||
return "high";
|
||||
case Effort.XHigh:
|
||||
return "max";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns true for Anthropic models with Opus 4.7+/Fable/Mythos API restrictions:
|
||||
* - Sampling parameters (temperature/top_p/top_k) return 400 error
|
||||
* - Thinking content is omitted by default (needs display: "summarized")
|
||||
*/
|
||||
export function hasOpus47ApiRestrictions(modelId: string): boolean {
|
||||
const parsed = parseAnthropicModel(bareModelId(modelId));
|
||||
if (!parsed) return false;
|
||||
return (parsed.kind === "opus" && semverGte(parsed.version, "4.7")) || isFableOrMythos(parsed.kind);
|
||||
}
|
||||
|
||||
/**
|
||||
* Mid-conversation `role: "system"` messages (system instructions appended at
|
||||
* non-first positions in the `messages` array) are supported starting with
|
||||
* Claude Opus 4.8 and the Claude Fable/Mythos 5 generation. Earlier Claude
|
||||
* models reject the role.
|
||||
* @see https://platform.claude.com/docs/en/build-with-claude/mid-conversation-system-messages
|
||||
*/
|
||||
export function supportsMidConversationSystemMessages(modelId: string): boolean {
|
||||
const parsed = parseAnthropicModel(bareModelId(modelId));
|
||||
if (!parsed) return false;
|
||||
return (parsed.kind === "opus" && semverGte(parsed.version, "4.8")) || isFableOrMythos(parsed.kind);
|
||||
}
|
||||
|
||||
export function isAnthropicFableOrMythosModel(modelId: string): boolean {
|
||||
const parsed = parseAnthropicModel(bareModelId(modelId));
|
||||
return parsed !== null && isFableOrMythos(parsed.kind);
|
||||
}
|
||||
|
||||
function isOpenRouterAnthropicAdaptiveReasoningModel<TApi extends Api>(
|
||||
parsedModel: AnthropicModel,
|
||||
model: ApiModel<TApi>,
|
||||
): boolean {
|
||||
if (model.api !== "openai-completions") return false;
|
||||
if (!modelMatchesHost(model, "openrouter")) return false;
|
||||
return isFableOrMythos(parsedModel.kind) || (parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.6"));
|
||||
}
|
||||
|
||||
function anthropicModelHasRealXHighEffort<TApi extends Api>(model: ApiModel<TApi>): boolean {
|
||||
if (model.api !== "anthropic-messages") return false;
|
||||
const parsedModel = parseKnownModel(model.id);
|
||||
if (parsedModel.family !== "anthropic") return false;
|
||||
if (isFableOrMythos(parsedModel.kind)) return true;
|
||||
return parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.7");
|
||||
}
|
||||
|
||||
function applyGeneratedModelPolicy(model: ApiModel<Api>): void {
|
||||
const copilotLimits = model.provider === "github-copilot" ? COPILOT_GENERATED_LIMITS[model.id] : undefined;
|
||||
if (copilotLimits) {
|
||||
model.contextWindow = copilotLimits.contextWindow;
|
||||
model.maxTokens = copilotLimits.maxTokens;
|
||||
}
|
||||
|
||||
if (
|
||||
model.api === "openai-completions" &&
|
||||
(model.provider === "minimax-code" || model.provider === "minimax-code-cn")
|
||||
) {
|
||||
model.compat = {
|
||||
...(model.compat ?? {}),
|
||||
supportsStore: false,
|
||||
supportsDeveloperRole: false,
|
||||
supportsReasoningEffort: false,
|
||||
reasoningContentField: "reasoning_content",
|
||||
};
|
||||
delete model.compat.thinkingFormat;
|
||||
}
|
||||
if (
|
||||
model.api === "openai-completions" &&
|
||||
model.provider === "opencode-go" &&
|
||||
(model.id === "deepseek-v4-flash" || model.id === "deepseek-v4-pro")
|
||||
) {
|
||||
model.compat = {
|
||||
...(model.compat ?? {}),
|
||||
supportsToolChoice: false,
|
||||
reasoningContentField: "reasoning_content",
|
||||
requiresReasoningContentForToolCalls: true,
|
||||
};
|
||||
}
|
||||
const parsedModel = parseKnownModel(model.id);
|
||||
const applyPatchToolType = inferGeneratedApplyPatchToolType(model, parsedModel);
|
||||
if (applyPatchToolType) {
|
||||
model.applyPatchToolType = applyPatchToolType;
|
||||
} else {
|
||||
delete model.applyPatchToolType;
|
||||
}
|
||||
if (parsedModel.family === "anthropic") {
|
||||
applyAnthropicCatalogPolicy(model, parsedModel);
|
||||
}
|
||||
if (parsedModel.family === "openai") {
|
||||
applyOpenAICatalogPolicy(model, parsedModel);
|
||||
}
|
||||
}
|
||||
|
||||
function applyAnthropicCatalogPolicy(model: ApiModel<Api>, parsedModel: AnthropicModel): void {
|
||||
// Claude Opus 4.5: models.dev reports 3x the correct cache pricing.
|
||||
if (model.provider === "anthropic" && parsedModel.kind === "opus" && semverEqual(parsedModel.version, "4.5")) {
|
||||
model.cost.cacheRead = 0.5;
|
||||
model.cost.cacheWrite = 6.25;
|
||||
}
|
||||
|
||||
// Bedrock Opus 4.6: upstream metadata is stale for cache pricing and context.
|
||||
if (model.provider === "amazon-bedrock" && parsedModel.kind === "opus" && semverEqual(parsedModel.version, "4.6")) {
|
||||
model.cost.cacheRead = 0.5;
|
||||
model.cost.cacheWrite = 6.25;
|
||||
model.contextWindow = 1000000;
|
||||
model.maxTokens = 128000;
|
||||
}
|
||||
|
||||
// Claude Fable/Mythos 5: Anthropic's /v1/models omits token limits and
|
||||
// pricing, and models.dev lags new releases. Pin authoritative values from
|
||||
// the model card (1M context / 128k output) and pricing docs ($10 in / $50
|
||||
// out per MTok).
|
||||
if (model.provider === "anthropic" && isFableOrMythos(parsedModel.kind)) {
|
||||
model.contextWindow = 1_000_000;
|
||||
model.maxTokens = 128_000;
|
||||
model.cost.input = 10;
|
||||
model.cost.output = 50;
|
||||
model.cost.cacheRead = 1;
|
||||
model.cost.cacheWrite = 12.5;
|
||||
}
|
||||
}
|
||||
|
||||
function inferGeneratedApplyPatchToolType(
|
||||
model: ApiModel<Api>,
|
||||
parsedModel: ParsedModel,
|
||||
): ApiModel<Api>["applyPatchToolType"] {
|
||||
if (parsedModel.family !== "openai" || parsedModel.version.major !== 5) {
|
||||
return undefined;
|
||||
}
|
||||
if (model.provider === "openai" && model.api === "openai-responses") {
|
||||
return "freeform";
|
||||
}
|
||||
if (model.provider === "openai-codex" && model.api === "openai-codex-responses") {
|
||||
return "freeform";
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function applyOpenAICatalogPolicy(model: ApiModel<Api>, parsedModel: OpenAIModel): void {
|
||||
// Codex models: 400K figure includes output budget; input window is 272K.
|
||||
if (parsedModel.variant.startsWith("codex") && parsedModel.variant !== "codex-spark") {
|
||||
model.contextWindow = 272000;
|
||||
return;
|
||||
}
|
||||
// GPT-5.4 mini/nano use plain OpenAI IDs on the Codex transport, but Codex still
|
||||
// enforces the lower prompt budget for these variants. Codex discovery can also
|
||||
// report inconsistent priorities for the GPT-5.4 family, so normalize by parsed
|
||||
// variant instead of special-casing raw model ids.
|
||||
if (model.api === "openai-codex-responses" && semverEqual(parsedModel.version, "5.4")) {
|
||||
const normalizedPriority = CODEX_GPT_5_4_PRIORITY_BY_VARIANT[parsedModel.variant];
|
||||
if (normalizedPriority !== undefined) {
|
||||
model.priority = normalizedPriority;
|
||||
}
|
||||
if (parsedModel.variant === "mini" || parsedModel.variant === "nano") {
|
||||
model.contextWindow = 272000;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function inferModelThinking<TApi extends Api>(model: ApiModel<TApi>): ThinkingConfig {
|
||||
const parsedModel = parseKnownModel(model.id);
|
||||
const efforts = inferSupportedEfforts(parsedModel, model);
|
||||
const minLevel = efforts[0];
|
||||
const maxLevel = efforts.at(-1);
|
||||
if (!minLevel || !maxLevel) {
|
||||
throw new Error(`Model ${model.provider}/${model.id} resolved to an empty thinking range`);
|
||||
}
|
||||
const config: ThinkingConfig = {
|
||||
mode: inferThinkingControlMode(model, parsedModel),
|
||||
minLevel,
|
||||
maxLevel,
|
||||
};
|
||||
// Encode explicit levels only when the inferred set has gaps the min..max range cannot represent.
|
||||
const minIndex = THINKING_EFFORTS.indexOf(minLevel);
|
||||
const maxIndex = THINKING_EFFORTS.indexOf(maxLevel);
|
||||
const expandedRange = THINKING_EFFORTS.slice(minIndex, maxIndex + 1);
|
||||
if (expandedRange.length !== efforts.length) {
|
||||
config.levels = efforts;
|
||||
}
|
||||
return config;
|
||||
}
|
||||
|
||||
function normalizeThinkingConfig(thinking: ThinkingConfig | undefined): ThinkingConfig | undefined {
|
||||
if (!thinking || expandEffortRange(thinking).length === 0) {
|
||||
return undefined;
|
||||
}
|
||||
return thinking;
|
||||
}
|
||||
|
||||
function thinkingsEqual(left: ThinkingConfig | undefined, right: ThinkingConfig | undefined): boolean {
|
||||
if (left === right) return true;
|
||||
if (!left || !right) return false;
|
||||
if (left.mode !== right.mode || left.minLevel !== right.minLevel || left.maxLevel !== right.maxLevel) return false;
|
||||
const leftLevels = left.levels;
|
||||
const rightLevels = right.levels;
|
||||
if (leftLevels === rightLevels) return true;
|
||||
if (!leftLevels || !rightLevels) return false;
|
||||
if (leftLevels.length !== rightLevels.length) return false;
|
||||
return leftLevels.every((level, index) => level === rightLevels[index]);
|
||||
}
|
||||
|
||||
function expandEffortRange(thinking: ThinkingConfig): readonly Effort[] {
|
||||
if (thinking.levels && thinking.levels.length > 0) {
|
||||
return thinking.levels;
|
||||
}
|
||||
const minIndex = THINKING_EFFORTS.indexOf(thinking.minLevel);
|
||||
const maxIndex = THINKING_EFFORTS.indexOf(thinking.maxLevel);
|
||||
if (minIndex === -1 || maxIndex === -1 || minIndex > maxIndex) {
|
||||
return [];
|
||||
}
|
||||
return THINKING_EFFORTS.slice(minIndex, maxIndex + 1);
|
||||
}
|
||||
|
||||
function inferSupportedEfforts<TApi extends Api>(parsedModel: ParsedModel, model: ApiModel<TApi>): readonly Effort[] {
|
||||
switch (parsedModel.family) {
|
||||
case "openai":
|
||||
return inferOpenAISupportedEfforts(parsedModel);
|
||||
case "gemini":
|
||||
return inferGeminiSupportedEfforts(parsedModel);
|
||||
case "anthropic":
|
||||
return inferAnthropicSupportedEfforts(parsedModel, model);
|
||||
case "unknown":
|
||||
return inferFallbackEfforts(model);
|
||||
}
|
||||
}
|
||||
|
||||
function inferOpenAISupportedEfforts(model: OpenAIModel): readonly Effort[] {
|
||||
if (model.variant === "codex-mini" && semverEqual(model.version, "5.1")) {
|
||||
return GPT_5_1_CODEX_MINI_EFFORTS;
|
||||
}
|
||||
if (semverGte(model.version, "5.2")) {
|
||||
return GPT_5_2_PLUS_EFFORTS;
|
||||
}
|
||||
return DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
|
||||
function inferGeminiSupportedEfforts(model: GeminiModel): readonly Effort[] {
|
||||
if (!semverGte(model.version, "3.0")) {
|
||||
return DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
return model.kind === "pro" ? GEMINI_3_PRO_EFFORTS : GEMINI_3_FLASH_EFFORTS;
|
||||
}
|
||||
|
||||
function inferAnthropicSupportedEfforts<TApi extends Api>(
|
||||
parsedModel: AnthropicModel,
|
||||
model: ApiModel<TApi>,
|
||||
): readonly Effort[] {
|
||||
if (
|
||||
(model.api === "anthropic-messages" || model.api === "bedrock-converse-stream") &&
|
||||
semverGte(parsedModel.version, "4.6")
|
||||
) {
|
||||
return parsedModel.kind === "opus" || isFableOrMythos(parsedModel.kind)
|
||||
? DEFAULT_REASONING_EFFORTS_WITH_XHIGH
|
||||
: DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
if (isOpenRouterAnthropicAdaptiveReasoningModel(parsedModel, model)) {
|
||||
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
||||
}
|
||||
return inferFallbackEfforts(model);
|
||||
}
|
||||
|
||||
function inferFallbackEfforts<TApi extends Api>(model: ApiModel<TApi>): readonly Effort[] {
|
||||
if (model.api === "anthropic-messages") {
|
||||
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
||||
}
|
||||
if (model.name.includes("deepseek-v4")) {
|
||||
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
||||
}
|
||||
if (model.api === "bedrock-converse-stream") {
|
||||
return DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
if (model.api === "openai-completions") {
|
||||
const compat = buildOpenAICompat(model as ModelSpec<"openai-completions">);
|
||||
if (compat.thinkingFormat === "openai" && compat.supportsReasoningEffort) {
|
||||
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
||||
}
|
||||
return DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
// OpenAI Responses APIs encode discrete effort levels, including xhigh.
|
||||
if (model.api === "openai-responses" || model.api === "openai-codex-responses") {
|
||||
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
||||
}
|
||||
return DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
|
||||
function inferThinkingControlMode<TApi extends Api>(
|
||||
model: ApiModel<TApi>,
|
||||
parsedModel: ParsedModel,
|
||||
): ThinkingConfig["mode"] {
|
||||
switch (model.api) {
|
||||
case "google-generative-ai":
|
||||
case "google-gemini-cli":
|
||||
case "google-vertex":
|
||||
return parsedModel.family === "gemini" &&
|
||||
semverGte(parsedModel.version, "3.0") &&
|
||||
parsedModel.version.major === 3
|
||||
? "google-level"
|
||||
: "budget";
|
||||
|
||||
case "anthropic-messages":
|
||||
if (parsedModel.family === "anthropic") {
|
||||
if (semverGte(parsedModel.version, "4.6")) {
|
||||
return "anthropic-adaptive";
|
||||
}
|
||||
if (semverGte(parsedModel.version, "4.5")) {
|
||||
return "anthropic-budget-effort";
|
||||
}
|
||||
}
|
||||
return "budget";
|
||||
|
||||
case "bedrock-converse-stream":
|
||||
if (parsedModel.family === "anthropic") {
|
||||
if (
|
||||
semverGte(parsedModel.version, "4.6") &&
|
||||
(parsedModel.kind === "opus" || isFableOrMythos(parsedModel.kind))
|
||||
) {
|
||||
return "anthropic-adaptive";
|
||||
}
|
||||
if (semverGte(parsedModel.version, "4.5")) {
|
||||
return "anthropic-budget-effort";
|
||||
}
|
||||
}
|
||||
return "budget";
|
||||
|
||||
default:
|
||||
return "effort";
|
||||
}
|
||||
return (model.thinking?.effortMap?.[supported] ?? supported) as "low" | "medium" | "high" | "xhigh" | "max";
|
||||
}
|
||||
|
||||
+8502
-2629
File diff suppressed because it is too large
Load Diff
@@ -60,11 +60,7 @@ function getThinkingConfig(capabilities: string[] | undefined): ThinkingConfig |
|
||||
if (!capabilities?.includes("thinking")) {
|
||||
return undefined;
|
||||
}
|
||||
return {
|
||||
mode: "effort",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.High,
|
||||
};
|
||||
return { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] };
|
||||
}
|
||||
async function fetchShowMetadata(
|
||||
baseUrl: string,
|
||||
|
||||
@@ -151,9 +151,8 @@ function buildAnthropicReferenceMap(
|
||||
* Seeded into model generation so the bundled catalog is never gated on
|
||||
* models.dev's update cadence; deduped behind upstream catalog / models.dev
|
||||
* entries once those appear. Token limits and pricing are pinned
|
||||
* authoritatively in
|
||||
* `applyAnthropicCatalogPolicy`, and `thinking` is derived by
|
||||
* `refreshModelThinking` during generation.
|
||||
* authoritatively in `applyAnthropicCatalogPolicy`, and `thinking` is re-baked
|
||||
* by the generator's policy pass (scripts/generated-policies.ts).
|
||||
*/
|
||||
export const ANTHROPIC_CURATED_FALLBACK_MODELS: readonly ModelSpec<"anthropic-messages">[] = [
|
||||
{
|
||||
@@ -343,11 +342,7 @@ function getOllamaThinkingConfig(capabilities: string[] | undefined): ThinkingCo
|
||||
if (!capabilities?.includes("thinking")) {
|
||||
return undefined;
|
||||
}
|
||||
return {
|
||||
mode: "effort",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.High,
|
||||
};
|
||||
return { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] };
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -2076,7 +2071,7 @@ export function moonshotModelManagerOptions(
|
||||
thinking:
|
||||
model.thinking ??
|
||||
(isKimiK2Reasoning
|
||||
? { mode: "effort", minLevel: Effort.Minimal, maxLevel: Effort.High }
|
||||
? { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] }
|
||||
: undefined),
|
||||
};
|
||||
},
|
||||
|
||||
@@ -26,20 +26,27 @@ export type ThinkingControlMode =
|
||||
|
||||
/** Per-model thinking capabilities used to clamp and map user-facing effort levels. */
|
||||
export interface ThinkingConfig {
|
||||
/** Least intensive supported user-facing effort level. */
|
||||
minLevel: Effort;
|
||||
/** Most intensive supported user-facing effort level. */
|
||||
maxLevel: Effort;
|
||||
/**
|
||||
* Optional explicit list of supported levels. When present, takes precedence over
|
||||
* the `minLevel`..`maxLevel` range — used to encode discrete sets with gaps
|
||||
* (e.g. Gemini 3 Pro supports `low` and `high` but not `medium`).
|
||||
*/
|
||||
levels?: readonly Effort[];
|
||||
/** Optional default effort applied when this model is selected. Falls back to global default if absent. */
|
||||
defaultLevel?: Effort;
|
||||
/** Provider-specific transport used to encode the selected effort. */
|
||||
mode: ThinkingControlMode;
|
||||
/**
|
||||
* Supported user-facing efforts, ordered least → most intensive. Never
|
||||
* empty: a reasoning model without a controllable effort surface carries
|
||||
* `thinking: undefined` instead of an empty list.
|
||||
*/
|
||||
efforts: readonly Effort[];
|
||||
/** Optional default effort applied when this model is selected. Falls back to global default if absent. */
|
||||
defaultLevel?: Effort;
|
||||
/**
|
||||
* Effort → wire-value remap for `anthropic-adaptive` transports, baked at
|
||||
* build time (4-tier legacy scale vs the 5-tier Opus 4.7+/Fable/Mythos
|
||||
* scale). Identity for efforts the map omits.
|
||||
*/
|
||||
effortMap?: Partial<Record<Effort, string>>;
|
||||
/**
|
||||
* Adaptive thinking accepts the `display` field (Opus 4.7+, Fable/Mythos
|
||||
* 5). Also implies native interleaved thinking — no beta header needed.
|
||||
*/
|
||||
supportsDisplay?: boolean;
|
||||
}
|
||||
|
||||
// `Provider` is any provider-id string; `KnownProvider` (re-exported above) enumerates
|
||||
@@ -246,6 +253,12 @@ export interface AnthropicCompat {
|
||||
* When unset, auto-detected from the model id. Default: true.
|
||||
*/
|
||||
supportsForcedToolChoice?: boolean;
|
||||
/**
|
||||
* Whether the model accepts sampling parameters (`temperature`, `top_p`,
|
||||
* `top_k`). Opus 4.7+ and Fable/Mythos reject them with a 400. When unset,
|
||||
* auto-detected from the model id. Default: true.
|
||||
*/
|
||||
supportsSamplingParams?: boolean;
|
||||
/**
|
||||
* Include a non-standard `id` field (aliasing `tool_use_id`) on
|
||||
* `tool_result` blocks. Z.AI's Anthropic-compatible proxy deserializes
|
||||
|
||||
@@ -0,0 +1,185 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import type { Api, ModelSpec, Provider } from "@oh-my-pi/pi-catalog/types";
|
||||
import { applyGeneratedModelPolicies, linkOpenAIPromotionTargets } from "../scripts/generated-policies";
|
||||
|
||||
function createSpec<TApi extends Api>(overrides: {
|
||||
id: string;
|
||||
api: TApi;
|
||||
provider: Provider;
|
||||
reasoning?: boolean;
|
||||
contextWindow?: number;
|
||||
maxTokens?: number;
|
||||
priority?: number;
|
||||
applyPatchToolType?: "freeform" | "function";
|
||||
cost?: ModelSpec<TApi>["cost"];
|
||||
thinking?: ModelSpec<TApi>["thinking"];
|
||||
}): ModelSpec<TApi> {
|
||||
return {
|
||||
id: overrides.id,
|
||||
name: overrides.id,
|
||||
api: overrides.api,
|
||||
provider: overrides.provider,
|
||||
baseUrl: "https://example.com",
|
||||
reasoning: overrides.reasoning ?? true,
|
||||
thinking: overrides.thinking,
|
||||
input: ["text"],
|
||||
cost: overrides.cost ?? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: overrides.contextWindow ?? 200000,
|
||||
maxTokens: overrides.maxTokens ?? 32000,
|
||||
priority: overrides.priority,
|
||||
applyPatchToolType: overrides.applyPatchToolType,
|
||||
};
|
||||
}
|
||||
|
||||
describe("generated model policies", () => {
|
||||
it("re-bakes thinking metadata and applies parsed catalog corrections", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
createSpec({
|
||||
id: "claude-opus-4-5",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
// Stale baked metadata must be replaced by the deriver's output.
|
||||
thinking: { mode: "budget", efforts: [Effort.High] },
|
||||
cost: { input: 0, output: 0, cacheRead: 1.5, cacheWrite: 18.75 },
|
||||
contextWindow: 1000000,
|
||||
}),
|
||||
createSpec({
|
||||
id: "anthropic.claude-opus-4-6-v1:0",
|
||||
api: "bedrock-converse-stream",
|
||||
provider: "amazon-bedrock",
|
||||
cost: { input: 0, output: 0, cacheRead: 1.5, cacheWrite: 18.75 },
|
||||
contextWindow: 1000000,
|
||||
}),
|
||||
createSpec({
|
||||
id: "gpt-5.2-codex",
|
||||
api: "openai-codex-responses",
|
||||
provider: "openai-codex",
|
||||
contextWindow: 400000,
|
||||
}),
|
||||
createSpec({
|
||||
id: "gpt-5.4-mini",
|
||||
api: "openai-codex-responses",
|
||||
provider: "openai-codex",
|
||||
contextWindow: 400000,
|
||||
priority: 2,
|
||||
}),
|
||||
];
|
||||
|
||||
applyGeneratedModelPolicies(models);
|
||||
|
||||
expect(models[0]?.thinking).toEqual({
|
||||
mode: "anthropic-budget-effort",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
});
|
||||
expect(models[0]?.cost.cacheRead).toBe(0.5);
|
||||
expect(models[0]?.cost.cacheWrite).toBe(6.25);
|
||||
expect(models[1]?.thinking).toEqual({
|
||||
mode: "anthropic-adaptive",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
effortMap: { minimal: "low", xhigh: "max" },
|
||||
});
|
||||
expect(models[1]?.cost.cacheRead).toBe(0.5);
|
||||
expect(models[1]?.cost.cacheWrite).toBe(6.25);
|
||||
expect(models[1]?.contextWindow).toBe(1000000);
|
||||
expect(models[2]?.contextWindow).toBe(272000);
|
||||
expect(models[3]?.contextWindow).toBe(272000);
|
||||
expect(models[3]?.priority).toBe(1);
|
||||
});
|
||||
|
||||
it("pins Claude Mythos 5 first-party Anthropic catalog metadata", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
createSpec({
|
||||
id: "claude-mythos-5",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
}),
|
||||
];
|
||||
|
||||
applyGeneratedModelPolicies(models);
|
||||
|
||||
expect(models[0]?.contextWindow).toBe(1_000_000);
|
||||
expect(models[0]?.maxTokens).toBe(128_000);
|
||||
expect(models[0]?.cost).toEqual({ input: 10, output: 50, cacheRead: 1, cacheWrite: 12.5 });
|
||||
expect(models[0]?.thinking).toEqual({
|
||||
mode: "anthropic-adaptive",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
effortMap: { minimal: "low", low: "medium", medium: "high", high: "xhigh", xhigh: "max" },
|
||||
supportsDisplay: true,
|
||||
});
|
||||
});
|
||||
|
||||
it("normalizes Copilot generated fallback limits", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
createSpec({
|
||||
id: "claude-opus-4.6",
|
||||
api: "anthropic-messages",
|
||||
provider: "github-copilot",
|
||||
contextWindow: 144000,
|
||||
maxTokens: 64000,
|
||||
}),
|
||||
createSpec({
|
||||
id: "gpt-5.4-mini",
|
||||
api: "openai-responses",
|
||||
provider: "github-copilot",
|
||||
contextWindow: 400000,
|
||||
maxTokens: 128000,
|
||||
}),
|
||||
createSpec({
|
||||
id: "grok-code-fast-1",
|
||||
api: "openai-completions",
|
||||
provider: "github-copilot",
|
||||
contextWindow: 128000,
|
||||
maxTokens: 64000,
|
||||
}),
|
||||
];
|
||||
|
||||
applyGeneratedModelPolicies(models);
|
||||
|
||||
expect(models[0]?.contextWindow).toBe(168000);
|
||||
expect(models[0]?.maxTokens).toBe(32000);
|
||||
expect(models[1]?.contextWindow).toBe(272000);
|
||||
expect(models[1]?.maxTokens).toBe(128000);
|
||||
expect(models[2]?.contextWindow).toBe(192000);
|
||||
expect(models[2]?.maxTokens).toBe(64000);
|
||||
});
|
||||
|
||||
it("links spark variants and gpt-5.5 to their context promotion targets", () => {
|
||||
const models = [
|
||||
createSpec({ id: "gpt-5.3-codex-spark", api: "openai-codex-responses", provider: "openai-codex" }),
|
||||
createSpec({ id: "gpt-5.5", api: "openai-codex-responses", provider: "openai-codex" }),
|
||||
createSpec({ id: "gpt-5.4", api: "openai-codex-responses", provider: "openai-codex" }),
|
||||
];
|
||||
|
||||
linkOpenAIPromotionTargets(models);
|
||||
|
||||
expect(models[0]?.contextPromotionTarget).toBe("openai-codex/gpt-5.5");
|
||||
expect(models[1]?.contextPromotionTarget).toBe("openai-codex/gpt-5.4");
|
||||
});
|
||||
|
||||
it("sets freeform apply_patch metadata for first-party GPT-5 Responses models", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
createSpec({ id: "gpt-5.4", api: "openai-responses", provider: "openai" }),
|
||||
createSpec({ id: "gpt-5.3-codex-spark", api: "openai-codex-responses", provider: "openai-codex" }),
|
||||
createSpec({
|
||||
id: "gpt-5.3-codex-spark",
|
||||
api: "openai-responses",
|
||||
provider: "opencode",
|
||||
applyPatchToolType: "freeform",
|
||||
}),
|
||||
createSpec({
|
||||
id: "gpt-5.4",
|
||||
api: "openai-completions",
|
||||
provider: "litellm",
|
||||
applyPatchToolType: "freeform",
|
||||
}),
|
||||
];
|
||||
|
||||
applyGeneratedModelPolicies(models);
|
||||
|
||||
expect(models[0]?.applyPatchToolType).toBe("freeform");
|
||||
expect(models[1]?.applyPatchToolType).toBe("freeform");
|
||||
expect(models[2]?.applyPatchToolType).toBeUndefined();
|
||||
expect(models[3]?.applyPatchToolType).toBeUndefined();
|
||||
});
|
||||
});
|
||||
@@ -198,8 +198,7 @@ describe("github copilot model limits mapping", () => {
|
||||
expect(model?.premiumMultiplier).toBe(0.33);
|
||||
expect(model?.thinking).toEqual({
|
||||
mode: "effort",
|
||||
minLevel: Effort.Low,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
@@ -37,7 +37,11 @@ describe("supportsAdaptiveThinkingDisplay", () => {
|
||||
expect(supportsAdaptiveThinkingDisplay("claude-fable-5")).toBe(true);
|
||||
expect(supportsAdaptiveThinkingDisplay("claude-opus-4-7")).toBe(true);
|
||||
expect(supportsAdaptiveThinkingDisplay("claude-opus-5-0")).toBe(true);
|
||||
// Dotted and dashed version separators are equivalent.
|
||||
expect(supportsAdaptiveThinkingDisplay("claude-opus-4.7")).toBe(true);
|
||||
expect(supportsAdaptiveThinkingDisplay("anthropic/claude-opus-4.8")).toBe(true);
|
||||
expect(supportsAdaptiveThinkingDisplay("claude-opus-4-6")).toBe(false);
|
||||
expect(supportsAdaptiveThinkingDisplay("claude-opus-4.6")).toBe(false);
|
||||
expect(supportsAdaptiveThinkingDisplay("claude-opus-4-20250514")).toBe(false);
|
||||
expect(supportsAdaptiveThinkingDisplay("claude-sonnet-4-6")).toBe(false);
|
||||
});
|
||||
|
||||
@@ -131,7 +131,10 @@ describe("issue #2113 — moonshot kimi-k2.6 discovery and wire format", () => {
|
||||
const k26 = byId.get("kimi-k2.6");
|
||||
expect(k26?.reasoning).toBe(true);
|
||||
expect(k26?.input).toEqual(["text", "image"]);
|
||||
expect(k26?.thinking).toEqual({ mode: "effort", minLevel: Effort.Minimal, maxLevel: Effort.High });
|
||||
expect(k26?.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
});
|
||||
|
||||
const thinkingOnly = byId.get("kimi-k2-thinking");
|
||||
expect(thinkingOnly?.reasoning).toBe(true);
|
||||
|
||||
@@ -1,29 +1,33 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import {
|
||||
applyGeneratedModelPolicies,
|
||||
clampThinkingLevelForModel,
|
||||
enrichModelThinking,
|
||||
linkOpenAIPromotionTargets,
|
||||
getSupportedEfforts,
|
||||
mapEffortToAnthropicAdaptiveEffort,
|
||||
mapEffortToGoogleThinkingLevel,
|
||||
requireSupportedEffort,
|
||||
} from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import type { Api, ModelSpec, Provider } from "@oh-my-pi/pi-catalog/types";
|
||||
import type { Api, Model, ModelSpec, Provider } from "@oh-my-pi/pi-catalog/types";
|
||||
|
||||
function createModel<TApi extends Api>(overrides: {
|
||||
id: string;
|
||||
api: TApi;
|
||||
provider: Provider;
|
||||
reasoning?: boolean;
|
||||
}): ModelSpec<TApi> {
|
||||
return enrichModelThinking({
|
||||
baseUrl?: string;
|
||||
compat?: ModelSpec<TApi>["compat"];
|
||||
thinking?: ModelSpec<TApi>["thinking"];
|
||||
}): Model<TApi> {
|
||||
return buildModel({
|
||||
id: overrides.id,
|
||||
name: overrides.id,
|
||||
api: overrides.api,
|
||||
provider: overrides.provider,
|
||||
baseUrl: "",
|
||||
baseUrl: overrides.baseUrl ?? "",
|
||||
reasoning: overrides.reasoning ?? true,
|
||||
compat: overrides.compat,
|
||||
thinking: overrides.thinking,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200000,
|
||||
@@ -31,7 +35,7 @@ function createModel<TApi extends Api>(overrides: {
|
||||
});
|
||||
}
|
||||
|
||||
describe("model thinking metadata", () => {
|
||||
describe("model thinking derivation", () => {
|
||||
it("stores supported efforts for Codex mini in model metadata", () => {
|
||||
const model = createModel({
|
||||
id: "gpt-5.1-codex-mini",
|
||||
@@ -41,8 +45,7 @@ describe("model thinking metadata", () => {
|
||||
|
||||
expect(model.thinking).toEqual({
|
||||
mode: "effort",
|
||||
minLevel: Effort.Medium,
|
||||
maxLevel: Effort.High,
|
||||
efforts: [Effort.Medium, Effort.High],
|
||||
});
|
||||
expect(() => requireSupportedEffort(model, Effort.Low)).toThrow(/Supported efforts: medium, high/);
|
||||
expect(() => requireSupportedEffort(model, Effort.XHigh)).toThrow(/Supported efforts: medium, high/);
|
||||
@@ -57,13 +60,12 @@ describe("model thinking metadata", () => {
|
||||
|
||||
expect(model.thinking).toEqual({
|
||||
mode: "effort",
|
||||
minLevel: Effort.Low,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
});
|
||||
expect(requireSupportedEffort(model, Effort.XHigh)).toBe(Effort.XHigh);
|
||||
});
|
||||
|
||||
it("maps Gemini 3 Pro only for supported levels", () => {
|
||||
it("encodes the Gemini 3 Pro effort gap directly in efforts", () => {
|
||||
const model = createModel({
|
||||
id: "gemini-3-pro-preview",
|
||||
api: "google-generative-ai",
|
||||
@@ -72,46 +74,25 @@ describe("model thinking metadata", () => {
|
||||
|
||||
expect(model.thinking).toEqual({
|
||||
mode: "google-level",
|
||||
minLevel: Effort.Low,
|
||||
maxLevel: Effort.High,
|
||||
levels: [Effort.Low, Effort.High],
|
||||
efforts: [Effort.Low, Effort.High],
|
||||
});
|
||||
expect(mapEffortToGoogleThinkingLevel(model, Effort.Low)).toBe("LOW");
|
||||
expect(mapEffortToGoogleThinkingLevel(model, Effort.High)).toBe("HIGH");
|
||||
expect(() => mapEffortToGoogleThinkingLevel(model, Effort.Medium)).toThrow(/not supported/);
|
||||
expect(mapEffortToGoogleThinkingLevel(Effort.Low)).toBe("LOW");
|
||||
expect(mapEffortToGoogleThinkingLevel(Effort.High)).toBe("HIGH");
|
||||
expect(mapEffortToGoogleThinkingLevel(Effort.XHigh)).toBe("HIGH");
|
||||
expect(() => requireSupportedEffort(model, Effort.Medium)).toThrow(/not supported/);
|
||||
});
|
||||
|
||||
it("encodes anthropic transport mode in metadata", () => {
|
||||
const opus45 = createModel({
|
||||
id: "claude-opus-4-5",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
});
|
||||
const opus46 = createModel({
|
||||
id: "claude-opus-4.6",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
});
|
||||
const opus47 = createModel({
|
||||
id: "claude-opus-4.7",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
});
|
||||
it("encodes anthropic transport mode and adaptive wire maps in metadata", () => {
|
||||
const opus45 = createModel({ id: "claude-opus-4-5", api: "anthropic-messages", provider: "anthropic" });
|
||||
const opus46 = createModel({ id: "claude-opus-4.6", api: "anthropic-messages", provider: "anthropic" });
|
||||
const opus47 = createModel({ id: "claude-opus-4.7", api: "anthropic-messages", provider: "anthropic" });
|
||||
const opus47Bedrock = createModel({
|
||||
id: "us.anthropic.claude-opus-4-7",
|
||||
api: "bedrock-converse-stream",
|
||||
provider: "amazon-bedrock",
|
||||
});
|
||||
const sonnet46 = createModel({
|
||||
id: "claude-sonnet-4.6",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
});
|
||||
const mythos = createModel({
|
||||
id: "claude-mythos-5",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
});
|
||||
const sonnet46 = createModel({ id: "claude-sonnet-4.6", api: "anthropic-messages", provider: "anthropic" });
|
||||
const mythos = createModel({ id: "claude-mythos-5", api: "anthropic-messages", provider: "anthropic" });
|
||||
const mythosBedrock = createModel({
|
||||
id: "global.anthropic.claude-mythos-5",
|
||||
api: "bedrock-converse-stream",
|
||||
@@ -121,275 +102,124 @@ describe("model thinking metadata", () => {
|
||||
expect(opus45.thinking?.mode).toBe("anthropic-budget-effort");
|
||||
expect(opus46.thinking?.mode).toBe("anthropic-adaptive");
|
||||
expect(sonnet46.thinking?.mode).toBe("anthropic-adaptive");
|
||||
expect(opus46.thinking).toEqual({
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.XHigh,
|
||||
});
|
||||
expect(sonnet46.thinking).toEqual({
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.High,
|
||||
});
|
||||
expect(mythos.thinking).toEqual({
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.XHigh,
|
||||
});
|
||||
expect(mythosBedrock.thinking?.mode).toBe("anthropic-adaptive");
|
||||
// Opus 4.6 has no real xhigh level — pi-ai aliases XHigh to Anthropic's "max".
|
||||
|
||||
// Opus 4.6 has no real xhigh level — the baked 4-tier map aliases XHigh to "max".
|
||||
expect(opus46.thinking?.effortMap).toEqual({ minimal: "low", xhigh: "max" });
|
||||
expect(mapEffortToAnthropicAdaptiveEffort(opus46, Effort.XHigh)).toBe("max");
|
||||
// Opus 4.7+ on the Messages API exposes the full five-tier scale, so pi-ai
|
||||
// shifts each user-facing effort up one notch and the top tier reaches "max".
|
||||
// Opus 4.7+ on the Messages API exposes the full five-tier scale: the baked
|
||||
// map shifts each user-facing effort up one notch so the top tier reaches "max".
|
||||
expect(opus47.thinking?.effortMap).toEqual({
|
||||
minimal: "low",
|
||||
low: "medium",
|
||||
medium: "high",
|
||||
high: "xhigh",
|
||||
xhigh: "max",
|
||||
});
|
||||
expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.Minimal)).toBe("low");
|
||||
expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.Low)).toBe("medium");
|
||||
expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.Medium)).toBe("high");
|
||||
expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.High)).toBe("xhigh");
|
||||
expect(mapEffortToAnthropicAdaptiveEffort(opus47, Effort.XHigh)).toBe("max");
|
||||
expect(mapEffortToAnthropicAdaptiveEffort(mythos, Effort.High)).toBe("xhigh");
|
||||
expect(mapEffortToAnthropicAdaptiveEffort(mythos, Effort.XHigh)).toBe("max");
|
||||
expect(mapEffortToAnthropicAdaptiveEffort(mythosBedrock, Effort.XHigh)).toBe("max");
|
||||
// Bedrock Converse keeps the four-tier legacy mapping; xhigh aliases to "max".
|
||||
expect(opus47Bedrock.thinking?.effortMap).toEqual({ minimal: "low", xhigh: "max" });
|
||||
expect(mapEffortToAnthropicAdaptiveEffort(opus47Bedrock, Effort.High)).toBe("high");
|
||||
expect(mapEffortToAnthropicAdaptiveEffort(opus47Bedrock, Effort.XHigh)).toBe("max");
|
||||
expect(() => mapEffortToAnthropicAdaptiveEffort(sonnet46, Effort.XHigh)).toThrow(/not supported/);
|
||||
});
|
||||
});
|
||||
|
||||
describe("generated model policies", () => {
|
||||
it("refreshes thinking metadata and applies parsed catalog corrections", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
{
|
||||
id: "claude-opus-4-5",
|
||||
name: "Claude Opus 4.5",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
baseUrl: "https://example.com",
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "budget",
|
||||
minLevel: Effort.High,
|
||||
maxLevel: Effort.High,
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 1.5, cacheWrite: 18.75 },
|
||||
contextWindow: 1000000,
|
||||
maxTokens: 32000,
|
||||
},
|
||||
{
|
||||
id: "anthropic.claude-opus-4-6-v1:0",
|
||||
name: "Claude Opus 4.6",
|
||||
api: "bedrock-converse-stream",
|
||||
provider: "amazon-bedrock",
|
||||
baseUrl: "https://example.com",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 1.5, cacheWrite: 18.75 },
|
||||
contextWindow: 1000000,
|
||||
maxTokens: 32000,
|
||||
},
|
||||
{
|
||||
id: "gpt-5.2-codex",
|
||||
name: "GPT-5.2 Codex",
|
||||
api: "openai-codex-responses",
|
||||
provider: "openai-codex",
|
||||
baseUrl: "https://example.com",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 400000,
|
||||
maxTokens: 32000,
|
||||
},
|
||||
{
|
||||
id: "gpt-5.4-mini",
|
||||
name: "GPT-5.4 mini",
|
||||
api: "openai-codex-responses",
|
||||
provider: "openai-codex",
|
||||
baseUrl: "https://example.com",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 400000,
|
||||
maxTokens: 32000,
|
||||
priority: 2,
|
||||
},
|
||||
];
|
||||
|
||||
applyGeneratedModelPolicies(models);
|
||||
|
||||
expect(models[0]?.thinking).toEqual({
|
||||
mode: "anthropic-budget-effort",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.XHigh,
|
||||
it("bakes adaptive display support for Opus 4.7+ and Fable/Mythos 5", () => {
|
||||
const opus46 = createModel({ id: "claude-opus-4.6", api: "anthropic-messages", provider: "anthropic" });
|
||||
const opus47 = createModel({ id: "claude-opus-4-7", api: "anthropic-messages", provider: "anthropic" });
|
||||
// Dotted and dashed version forms are equivalent; bare dated ids stay Opus 4.0.
|
||||
const opus47Dotted = createModel({ id: "claude-opus-4.7", api: "anthropic-messages", provider: "anthropic" });
|
||||
const opus4Dated = createModel({
|
||||
id: "claude-opus-4-20250514",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
});
|
||||
expect(models[0]?.cost.cacheRead).toBe(0.5);
|
||||
expect(models[0]?.cost.cacheWrite).toBe(6.25);
|
||||
expect(models[1]?.thinking).toEqual({
|
||||
const fable = createModel({ id: "claude-fable-5", api: "anthropic-messages", provider: "anthropic" });
|
||||
const fableBedrock = createModel({
|
||||
id: "global.anthropic.claude-fable-5",
|
||||
api: "bedrock-converse-stream",
|
||||
provider: "amazon-bedrock",
|
||||
});
|
||||
|
||||
expect(opus46.thinking?.supportsDisplay).toBeUndefined();
|
||||
expect(opus47.thinking?.supportsDisplay).toBe(true);
|
||||
expect(opus47Dotted.thinking?.supportsDisplay).toBe(true);
|
||||
expect(opus4Dated.thinking?.supportsDisplay).toBeUndefined();
|
||||
expect(fable.thinking?.supportsDisplay).toBe(true);
|
||||
expect(fableBedrock.thinking?.supportsDisplay).toBe(true);
|
||||
});
|
||||
|
||||
it("backfills wire facts onto explicit thinking, explicit values winning", () => {
|
||||
// Authored capability surface (mode/efforts) keeps identity-derived wire
|
||||
// facts: configs never need to know Anthropic's tier tables.
|
||||
const filled = createModel({
|
||||
id: "claude-opus-4-8",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
thinking: { mode: "anthropic-adaptive", efforts: [Effort.Low, Effort.High] },
|
||||
});
|
||||
expect(filled.thinking).toEqual({
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Low, Effort.High],
|
||||
effortMap: { minimal: "low", low: "medium", medium: "high", high: "xhigh", xhigh: "max" },
|
||||
supportsDisplay: true,
|
||||
});
|
||||
expect(models[1]?.cost.cacheRead).toBe(0.5);
|
||||
expect(models[1]?.cost.cacheWrite).toBe(6.25);
|
||||
expect(models[1]?.contextWindow).toBe(1000000);
|
||||
expect(models[2]?.contextWindow).toBe(272000);
|
||||
expect(models[3]?.contextWindow).toBe(272000);
|
||||
expect(models[3]?.priority).toBe(1);
|
||||
});
|
||||
|
||||
it("pins Claude Mythos 5 first-party Anthropic catalog metadata", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
{
|
||||
id: "claude-mythos-5",
|
||||
name: "Claude Mythos 5",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
baseUrl: "https://example.com",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200000,
|
||||
maxTokens: 32000,
|
||||
// Explicit wire facts are authoritative — including `false`.
|
||||
const pinned = createModel({
|
||||
id: "claude-opus-4-8",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
thinking: {
|
||||
mode: "anthropic-adaptive",
|
||||
efforts: [Effort.Low, Effort.High],
|
||||
effortMap: { xhigh: "max" },
|
||||
supportsDisplay: false,
|
||||
},
|
||||
];
|
||||
|
||||
applyGeneratedModelPolicies(models);
|
||||
|
||||
expect(models[0]?.contextWindow).toBe(1_000_000);
|
||||
expect(models[0]?.maxTokens).toBe(128_000);
|
||||
expect(models[0]?.cost).toEqual({ input: 10, output: 50, cacheRead: 1, cacheWrite: 12.5 });
|
||||
expect(models[0]?.thinking).toEqual({
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.XHigh,
|
||||
});
|
||||
expect(pinned.thinking?.effortMap).toEqual({ xhigh: "max" });
|
||||
expect(pinned.thinking?.supportsDisplay).toBe(false);
|
||||
});
|
||||
|
||||
it("normalizes Copilot generated fallback limits", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
{
|
||||
...createModel({
|
||||
id: "claude-opus-4.6",
|
||||
api: "anthropic-messages",
|
||||
provider: "github-copilot",
|
||||
}),
|
||||
contextWindow: 144000,
|
||||
maxTokens: 64000,
|
||||
},
|
||||
{
|
||||
...createModel({
|
||||
id: "gpt-5.4-mini",
|
||||
api: "openai-responses",
|
||||
provider: "github-copilot",
|
||||
}),
|
||||
contextWindow: 400000,
|
||||
maxTokens: 128000,
|
||||
},
|
||||
{
|
||||
...createModel({
|
||||
id: "grok-code-fast-1",
|
||||
api: "openai-completions",
|
||||
provider: "github-copilot",
|
||||
}),
|
||||
contextWindow: 128000,
|
||||
maxTokens: 64000,
|
||||
},
|
||||
];
|
||||
it("bakes sampling-param rejection into anthropic compat", () => {
|
||||
const sonnet45 = createModel({ id: "claude-sonnet-4-5", api: "anthropic-messages", provider: "anthropic" });
|
||||
const opus47 = createModel({ id: "claude-opus-4.7", api: "anthropic-messages", provider: "anthropic" });
|
||||
const fable = createModel({ id: "claude-fable-5", api: "anthropic-messages", provider: "anthropic" });
|
||||
|
||||
applyGeneratedModelPolicies(models);
|
||||
|
||||
expect(models[0]?.contextWindow).toBe(168000);
|
||||
expect(models[0]?.maxTokens).toBe(32000);
|
||||
expect(models[1]?.contextWindow).toBe(272000);
|
||||
expect(models[1]?.maxTokens).toBe(128000);
|
||||
expect(models[2]?.contextWindow).toBe(192000);
|
||||
expect(models[2]?.maxTokens).toBe(64000);
|
||||
expect(sonnet45.compat.supportsSamplingParams).toBe(true);
|
||||
expect(opus47.compat.supportsSamplingParams).toBe(false);
|
||||
expect(fable.compat.supportsSamplingParams).toBe(false);
|
||||
});
|
||||
|
||||
it("links spark variants and gpt-5.5 to their context promotion targets", () => {
|
||||
const models = [
|
||||
createModel({
|
||||
id: "gpt-5.3-codex-spark",
|
||||
api: "openai-codex-responses",
|
||||
provider: "openai-codex",
|
||||
}),
|
||||
createModel({
|
||||
id: "gpt-5.5",
|
||||
api: "openai-codex-responses",
|
||||
provider: "openai-codex",
|
||||
}),
|
||||
createModel({
|
||||
id: "gpt-5.4",
|
||||
api: "openai-codex-responses",
|
||||
provider: "openai-codex",
|
||||
}),
|
||||
];
|
||||
it("encodes effort-dial-less reasoners as thinking: undefined", () => {
|
||||
const model = createModel({
|
||||
id: "grok-build",
|
||||
api: "openai-responses",
|
||||
provider: "xai-oauth",
|
||||
compat: { supportsReasoningEffort: false },
|
||||
});
|
||||
|
||||
linkOpenAIPromotionTargets(models);
|
||||
|
||||
expect(models[0]?.contextPromotionTarget).toBe("openai-codex/gpt-5.5");
|
||||
expect(models[1]?.contextPromotionTarget).toBe("openai-codex/gpt-5.4");
|
||||
});
|
||||
|
||||
it("sets freeform apply_patch metadata for first-party GPT-5 Responses models", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
createModel({
|
||||
id: "gpt-5.4",
|
||||
api: "openai-responses",
|
||||
provider: "openai",
|
||||
}),
|
||||
createModel({
|
||||
id: "gpt-5.3-codex-spark",
|
||||
api: "openai-codex-responses",
|
||||
provider: "openai-codex",
|
||||
}),
|
||||
{
|
||||
...createModel({
|
||||
id: "gpt-5.3-codex-spark",
|
||||
api: "openai-responses",
|
||||
provider: "opencode",
|
||||
}),
|
||||
applyPatchToolType: "freeform",
|
||||
},
|
||||
{
|
||||
...createModel({
|
||||
id: "gpt-5.4",
|
||||
api: "openai-completions",
|
||||
provider: "litellm",
|
||||
}),
|
||||
applyPatchToolType: "freeform",
|
||||
},
|
||||
];
|
||||
|
||||
applyGeneratedModelPolicies(models);
|
||||
|
||||
expect(models[0]?.applyPatchToolType).toBe("freeform");
|
||||
expect(models[1]?.applyPatchToolType).toBe("freeform");
|
||||
expect(models[2]?.applyPatchToolType).toBeUndefined();
|
||||
expect(models[3]?.applyPatchToolType).toBeUndefined();
|
||||
expect(model.reasoning).toBe(true);
|
||||
expect(model.thinking).toBeUndefined();
|
||||
expect(getSupportedEfforts(model)).toEqual([]);
|
||||
expect(clampThinkingLevelForModel(model, Effort.High)).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("model thinking runtime helpers", () => {
|
||||
it("clamps from explicit metadata instead of inferring from model id", () => {
|
||||
const model: ModelSpec<"openai-codex-responses"> = {
|
||||
const model = createModel({
|
||||
id: "custom-reasoner",
|
||||
name: "Custom Reasoner",
|
||||
api: "openai-codex-responses",
|
||||
provider: "custom",
|
||||
baseUrl: "https://example.com",
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
minLevel: Effort.Medium,
|
||||
maxLevel: Effort.High,
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200000,
|
||||
maxTokens: 32000,
|
||||
};
|
||||
thinking: { mode: "effort", efforts: [Effort.Medium, Effort.High] },
|
||||
});
|
||||
|
||||
expect(model.thinking).toEqual({ mode: "effort", efforts: [Effort.Medium, Effort.High] });
|
||||
expect(clampThinkingLevelForModel(model, Effort.Minimal)).toBe(Effort.Medium);
|
||||
expect(clampThinkingLevelForModel(model, Effort.XHigh)).toBe(Effort.High);
|
||||
expect(clampThinkingLevelForModel(model, Effort.High)).toBe(Effort.High);
|
||||
@@ -413,32 +243,22 @@ describe("model thinking runtime helpers", () => {
|
||||
provider: "custom",
|
||||
});
|
||||
|
||||
// openai-completions should support xhigh by default
|
||||
expect(model.thinking?.maxLevel).toBe(Effort.XHigh);
|
||||
expect(model.thinking?.efforts.at(-1)).toBe(Effort.XHigh);
|
||||
expect(requireSupportedEffort(model, Effort.XHigh)).toBe(Effort.XHigh);
|
||||
});
|
||||
|
||||
it("does not expose xhigh for binary-thinking openai-compat transports", () => {
|
||||
const model = enrichModelThinking({
|
||||
const model = createModel({
|
||||
id: "glm-4.7",
|
||||
name: "GLM-4.7",
|
||||
api: "openai-completions",
|
||||
provider: "zai",
|
||||
baseUrl: "https://api.z.ai/v1",
|
||||
reasoning: true,
|
||||
compat: {
|
||||
thinkingFormat: "zai",
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128000,
|
||||
maxTokens: 32000,
|
||||
} satisfies ModelSpec<"openai-completions">);
|
||||
compat: { thinkingFormat: "zai" },
|
||||
});
|
||||
|
||||
expect(model.thinking).toEqual({
|
||||
mode: "effort",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.High,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
});
|
||||
expect(requireSupportedEffort(model, Effort.High)).toBe(Effort.High);
|
||||
expect(() => requireSupportedEffort(model, Effort.XHigh)).toThrow(
|
||||
@@ -447,26 +267,17 @@ describe("model thinking runtime helpers", () => {
|
||||
});
|
||||
|
||||
it("derives binary-thinking fallback from resolved compat when catalog compat is partial", () => {
|
||||
const model = enrichModelThinking({
|
||||
const model = createModel({
|
||||
id: "qwen/qwen3-32b",
|
||||
name: "Qwen 3 32B",
|
||||
api: "openai-completions",
|
||||
provider: "openrouter",
|
||||
baseUrl: "https://openrouter.ai/api/v1",
|
||||
reasoning: true,
|
||||
compat: {
|
||||
supportsToolChoice: true,
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128000,
|
||||
maxTokens: 32000,
|
||||
} satisfies ModelSpec<"openai-completions">);
|
||||
compat: { supportsToolChoice: true },
|
||||
});
|
||||
|
||||
expect(model.thinking).toEqual({
|
||||
mode: "effort",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.High,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
});
|
||||
expect(requireSupportedEffort(model, Effort.High)).toBe(Effort.High);
|
||||
expect(() => requireSupportedEffort(model, Effort.XHigh)).toThrow(
|
||||
@@ -491,34 +302,24 @@ describe("model thinking runtime helpers", () => {
|
||||
provider: "openrouter",
|
||||
});
|
||||
|
||||
expect(fable.thinking?.maxLevel).toBe(Effort.XHigh);
|
||||
expect(opus46.thinking?.maxLevel).toBe(Effort.XHigh);
|
||||
expect(sonnet46.thinking?.maxLevel).toBe(Effort.High);
|
||||
expect(fable.thinking?.efforts.at(-1)).toBe(Effort.XHigh);
|
||||
expect(opus46.thinking?.efforts.at(-1)).toBe(Effort.XHigh);
|
||||
expect(sonnet46.thinking?.efforts.at(-1)).toBe(Effort.High);
|
||||
expect(requireSupportedEffort(fable, Effort.XHigh)).toBe(Effort.XHigh);
|
||||
});
|
||||
|
||||
it("enables xhigh for openai-responses and openai-codex-responses APIs", () => {
|
||||
const responsesModel = createModel({
|
||||
id: "custom-responses",
|
||||
api: "openai-responses",
|
||||
provider: "custom",
|
||||
});
|
||||
const responsesModel = createModel({ id: "custom-responses", api: "openai-responses", provider: "custom" });
|
||||
const codexModel = createModel({ id: "custom-codex", api: "openai-codex-responses", provider: "custom" });
|
||||
|
||||
const codexModel = createModel({
|
||||
id: "custom-codex",
|
||||
api: "openai-codex-responses",
|
||||
provider: "custom",
|
||||
});
|
||||
|
||||
// Both should support xhigh
|
||||
expect(responsesModel.thinking?.maxLevel).toBe(Effort.XHigh);
|
||||
expect(codexModel.thinking?.maxLevel).toBe(Effort.XHigh);
|
||||
expect(responsesModel.thinking?.efforts.at(-1)).toBe(Effort.XHigh);
|
||||
expect(codexModel.thinking?.efforts.at(-1)).toBe(Effort.XHigh);
|
||||
expect(requireSupportedEffort(responsesModel, Effort.XHigh)).toBe(Effort.XHigh);
|
||||
expect(requireSupportedEffort(codexModel, Effort.XHigh)).toBe(Effort.XHigh);
|
||||
});
|
||||
|
||||
it("rejects reasoning models that are missing thinking metadata at runtime", () => {
|
||||
const model = {
|
||||
it("rejects effort requests against un-built reasoning specs", () => {
|
||||
const spec = {
|
||||
id: "broken-reasoner",
|
||||
name: "Broken Reasoner",
|
||||
api: "openai-responses",
|
||||
@@ -531,28 +332,30 @@ describe("model thinking runtime helpers", () => {
|
||||
maxTokens: 32000,
|
||||
} as ModelSpec<"openai-responses">;
|
||||
|
||||
expect(() => requireSupportedEffort(model, Effort.High)).toThrow(/missing thinking metadata/);
|
||||
expect(() => requireSupportedEffort(spec, Effort.High)).toThrow(/not supported/);
|
||||
});
|
||||
|
||||
it("drops empty thinking metadata so presence checks stay meaningful", () => {
|
||||
const model = enrichModelThinking({
|
||||
it("drops authored thinking on non-reasoning models and re-derives empty efforts", () => {
|
||||
const nonReasoning = createModel({
|
||||
id: "plain-model",
|
||||
name: "Plain Model",
|
||||
api: "openai-responses",
|
||||
provider: "custom",
|
||||
baseUrl: "https://example.com",
|
||||
reasoning: false,
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
minLevel: Effort.High,
|
||||
maxLevel: Effort.Low,
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200000,
|
||||
maxTokens: 32000,
|
||||
} satisfies ModelSpec<"openai-responses">);
|
||||
thinking: { mode: "effort", efforts: [Effort.High] },
|
||||
});
|
||||
expect(nonReasoning.thinking).toBeUndefined();
|
||||
|
||||
expect(model.thinking).toBeUndefined();
|
||||
// Empty explicit efforts are treated as absent metadata: infer instead.
|
||||
const emptyEfforts = createModel({
|
||||
id: "gpt-5.2-codex",
|
||||
api: "openai-codex-responses",
|
||||
provider: "openai-codex",
|
||||
thinking: { mode: "effort", efforts: [] },
|
||||
});
|
||||
expect(emptyEfforts.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -50,8 +50,7 @@ describe("nanogpt model limits mapping", () => {
|
||||
expect(model?.premiumMultiplier).toBeUndefined();
|
||||
expect(model?.thinking).toEqual({
|
||||
mode: "effort",
|
||||
minLevel: Effort.Low,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
});
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
@@ -14,6 +14,7 @@ const cloudModel: Model<"ollama-chat"> = {
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128_000,
|
||||
maxTokens: 8_192,
|
||||
compat: undefined,
|
||||
};
|
||||
|
||||
function createNdjsonResponse(lines: unknown[]): Response {
|
||||
|
||||
@@ -45,7 +45,10 @@ describe("ollama local provider discovery", () => {
|
||||
expect(model?.api).toBe("openai-responses");
|
||||
expect(model?.contextWindow).toBe(1048576);
|
||||
expect(model?.reasoning).toBe(true);
|
||||
expect(model?.thinking).toEqual({ mode: "effort", minLevel: Effort.Minimal, maxLevel: Effort.High });
|
||||
expect(model?.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
});
|
||||
expect(model?.input).toEqual(["text", "image"]);
|
||||
});
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
### Added
|
||||
|
||||
- Added `supportsReasoningParams`, `alwaysSendMaxTokens`, `strictResponsesPairing`, and a recursive `whenThinking` overlay (alongside `streamIdleTimeoutMs`/`supportsLongPromptCacheRetention`/`requiresToolResultId`/`replayUnsignedThinking`) to the OpenAI/Anthropic `compat` schema so custom model entries can configure those provider-specific capabilities
|
||||
- Custom model `thinking` config now uses the catalog's explicit vocabulary: `efforts` (ordered list) plus optional `defaultLevel`, `effortMap`, and `supportsDisplay` overrides; the legacy `minLevel`/`maxLevel`/`levels` range shape is still accepted and normalized at parse time. Wire facts (`effortMap`/`supportsDisplay`) are backfilled from model identity when not set, so existing claude-proxy configs keep the 5-tier adaptive scale and summarized display without changes.
|
||||
- New `omp usage` command: a detailed per-account breakdown of provider usage limits (bars, windows, reset times, plan metadata) covering every stored credential — accounts with no usage endpoint are listed as "no usage data" rows. Each provider section ends with per-window capacity stats ("capacity: 5h → 2.40/5 accounts used (2.60× quota left)"). Flags: `--provider` to filter, `--json` for the broker-shaped report payload, and `--redact` to mask account emails/ids down to a two-char anchor plus a minimal middle-out differentiator (`ca*9*`) for screenshot-safe sharing.
|
||||
- Startup hangs are now self-diagnosing (speculative fix for the "zero output, hangs even on `omp -h`" report class): a watchdog prints a stderr line every 10s naming the deepest in-flight startup phase (via `logger.openSpanPath()`) until a mode runner takes over, pausing around legitimate interactive waits (fork/move prompts, the `--resume` session picker); `PI_DEBUG_STARTUP` is restored as streaming synchronous `[startup]` phase markers covering command-module imports and the native addon load, which the post-startup `PI_TIMING` tree structurally cannot show for a hang; and waiting on piped-stdin EOF announces itself after 1s instead of blocking silently.
|
||||
- npm installs now execute a prebundled single-file entry: the published `bin.omp` points at `dist/cli.js` (built by `scripts/bundle-dist.ts` during `prepack`, ~18MB minified, natives/transformers/mupdf external), cutting npm-install cold start by roughly 3x versus transpiling the raw TypeScript graph per launch; `src/**` stays published for SDK consumers and worker fallbacks. The on-repo manifest keeps `bin.omp` at `src/cli.ts` — release rewrites it via the `publishBin` override in `scripts/ci-release-publish.ts` — so source installs (`bun link`, `install.sh --source`) keep working without a build step
|
||||
|
||||
@@ -68,13 +68,50 @@ const ThinkingControlModeSchema = z.enum([
|
||||
"anthropic-budget-effort",
|
||||
]);
|
||||
|
||||
const ModelThinkingSchema = z.object({
|
||||
minLevel: EffortSchema,
|
||||
maxLevel: EffortSchema,
|
||||
mode: ThinkingControlModeSchema,
|
||||
defaultLevel: EffortSchema.optional(),
|
||||
levels: z.array(EffortSchema).optional(),
|
||||
});
|
||||
const EFFORT_ORDER = ["minimal", "low", "medium", "high", "xhigh"] as const;
|
||||
|
||||
/**
|
||||
* Accepts the canonical `efforts` vocabulary plus the legacy
|
||||
* `minLevel`/`maxLevel`/`levels` range shape, normalizing both to
|
||||
* `ThinkingConfig` (ordered `efforts`, never empty). Precedence mirrors the
|
||||
* old runtime: explicit `levels` beat the min..max range; `efforts` beats both.
|
||||
*/
|
||||
const ModelThinkingSchema = z
|
||||
.object({
|
||||
mode: ThinkingControlModeSchema,
|
||||
efforts: z.array(EffortSchema).min(1).optional(),
|
||||
defaultLevel: EffortSchema.optional(),
|
||||
effortMap: ReasoningEffortMapSchema.optional(),
|
||||
supportsDisplay: z.boolean().optional(),
|
||||
// Legacy range vocabulary (pre-efforts configs).
|
||||
minLevel: EffortSchema.optional(),
|
||||
maxLevel: EffortSchema.optional(),
|
||||
levels: z.array(EffortSchema).min(1).optional(),
|
||||
})
|
||||
.refine(
|
||||
value =>
|
||||
value.efforts !== undefined ||
|
||||
value.levels !== undefined ||
|
||||
(value.minLevel !== undefined && value.maxLevel !== undefined),
|
||||
{
|
||||
message: "thinking requires `efforts` (or legacy `levels`/`minLevel`+`maxLevel`)",
|
||||
},
|
||||
)
|
||||
.transform(({ efforts, levels, minLevel, maxLevel, mode, defaultLevel, effortMap, supportsDisplay }) => {
|
||||
let resolved = efforts ?? levels;
|
||||
if (!resolved) {
|
||||
const minIndex = EFFORT_ORDER.indexOf(minLevel!);
|
||||
const maxIndex = EFFORT_ORDER.indexOf(maxLevel!);
|
||||
resolved = EFFORT_ORDER.slice(minIndex, Math.max(minIndex, maxIndex) + 1);
|
||||
}
|
||||
return {
|
||||
mode,
|
||||
efforts: resolved,
|
||||
...(defaultLevel !== undefined && { defaultLevel }),
|
||||
...(effortMap !== undefined && { effortMap }),
|
||||
...(supportsDisplay !== undefined && { supportsDisplay }),
|
||||
};
|
||||
});
|
||||
|
||||
const ModelDefinitionSchema = z.object({
|
||||
id: z.string().min(1),
|
||||
|
||||
@@ -38,7 +38,7 @@ const SLOW = makeModel("p", "slow");
|
||||
const REASONING_SLOW = makeModel("p", "slow", {
|
||||
api: "anthropic-messages",
|
||||
reasoning: true,
|
||||
thinking: { minLevel: Effort.Low, maxLevel: Effort.High, mode: "anthropic-adaptive" },
|
||||
thinking: { efforts: [Effort.Low, Effort.Medium, Effort.High], mode: "anthropic-adaptive" },
|
||||
});
|
||||
|
||||
interface SessionOptions {
|
||||
|
||||
@@ -1082,7 +1082,7 @@ export interface ExtensionAPI {
|
||||
* id: "claude-sonnet-4@20250514",
|
||||
* name: "Claude Sonnet 4 (Vertex)",
|
||||
* reasoning: true,
|
||||
* thinking: { mode: "anthropic-adaptive", minLevel: "minimal", maxLevel: "high" },
|
||||
* thinking: { mode: "anthropic-adaptive", efforts: ["minimal", "low", "medium", "high"] },
|
||||
* input: ["text", "image"],
|
||||
* cost: { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
|
||||
* contextWindow: 200000,
|
||||
|
||||
@@ -699,7 +699,7 @@ describe("AgentSession MCP discovery", () => {
|
||||
const reasoningModel: Model<"openai-responses"> = {
|
||||
...createModel(),
|
||||
reasoning: true,
|
||||
thinking: { mode: "effort", minLevel: Effort.Medium, maxLevel: Effort.Medium },
|
||||
thinking: { mode: "effort", efforts: [Effort.Medium] },
|
||||
};
|
||||
|
||||
const agent = new Agent({
|
||||
|
||||
@@ -71,8 +71,7 @@ describe("issue #775: per-model defaultLevel", () => {
|
||||
...opus,
|
||||
thinking: {
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Low,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
defaultLevel: Effort.XHigh,
|
||||
},
|
||||
};
|
||||
|
||||
@@ -238,7 +238,7 @@ describe("memories runtime", () => {
|
||||
const constrainedModel: Model = {
|
||||
...fx.model,
|
||||
reasoning: true,
|
||||
thinking: { mode: "effort", minLevel: Effort.High, maxLevel: Effort.XHigh },
|
||||
thinking: { mode: "effort", efforts: [Effort.High, Effort.XHigh] },
|
||||
};
|
||||
fx.session.model = constrainedModel;
|
||||
fx.modelRegistry.find = vi.fn(() => constrainedModel);
|
||||
|
||||
@@ -351,8 +351,7 @@ describe("ModelRegistry runtime discovery", () => {
|
||||
expect(qwen?.reasoning).toBe(true);
|
||||
expect(qwen?.thinking).toEqual({
|
||||
mode: "effort",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.High,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
});
|
||||
|
||||
const llama = registry.find("ollama", "llama3.2:3b");
|
||||
|
||||
@@ -141,7 +141,7 @@ describe("ModelRegistry runtime provider registration", () => {
|
||||
expectProviderHeader(registry, providerName, "Authorization", undefined);
|
||||
});
|
||||
|
||||
test("registerProvider preserves explicit thinking on runtime models", () => {
|
||||
test("registerProvider preserves explicit thinking and backfills wire facts", () => {
|
||||
const registry = new ModelRegistry(authStorage, modelsJsonPath);
|
||||
const config: ProviderConfigInput = {
|
||||
baseUrl: "https://runtime.example.com/v1",
|
||||
@@ -154,8 +154,7 @@ describe("ModelRegistry runtime provider registration", () => {
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.High,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
},
|
||||
},
|
||||
],
|
||||
@@ -166,8 +165,10 @@ describe("ModelRegistry runtime provider registration", () => {
|
||||
|
||||
expect(model?.thinking).toEqual({
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.High,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
// Wire facts are backfilled from identity; non-claude ids get the
|
||||
// 4-tier adaptive map.
|
||||
effortMap: { minimal: "low", xhigh: "max" },
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
@@ -284,7 +284,7 @@ describe("ModelRegistry", () => {
|
||||
const variants = registry.getCanonicalVariants("deepseek-v4-pro");
|
||||
|
||||
expect(model?.cost.cacheRead).toBeGreaterThan(0);
|
||||
expect(model?.thinking?.maxLevel).toBe(Effort.XHigh);
|
||||
expect(model?.thinking?.efforts.at(-1)).toBe(Effort.XHigh);
|
||||
expect(variants.some(variant => variant.selector === "ollama/deepseek-v4-pro:cloud")).toBe(true);
|
||||
});
|
||||
|
||||
@@ -1132,12 +1132,10 @@ describe("ModelRegistry", () => {
|
||||
});
|
||||
|
||||
describe("thinking metadata normalization", () => {
|
||||
test("custom models preserve explicit thinking", () => {
|
||||
test("custom models preserve explicit thinking and gain backfilled wire facts", () => {
|
||||
const thinking: ThinkingConfig = {
|
||||
mode: "anthropic-adaptive",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.High,
|
||||
levels: [Effort.Minimal, Effort.High],
|
||||
efforts: [Effort.Minimal, Effort.High],
|
||||
};
|
||||
|
||||
writeModelsJson({
|
||||
@@ -1149,7 +1147,11 @@ describe("ModelRegistry", () => {
|
||||
const registry = new ModelRegistry(authStorage, modelsJsonPath);
|
||||
const model = getModelsForProvider(registry, "anthropic").find(m => m.id === "claude-custom");
|
||||
|
||||
expect(model?.thinking).toEqual(thinking);
|
||||
expect(model?.thinking).toEqual({
|
||||
...thinking,
|
||||
// Versionless claude ids resolve to the 4-tier adaptive wire map.
|
||||
effortMap: { minimal: "low", xhigh: "max" },
|
||||
});
|
||||
});
|
||||
|
||||
test("model overrides can replace canonical thinking metadata", () => {
|
||||
@@ -1157,7 +1159,7 @@ describe("ModelRegistry", () => {
|
||||
openrouter: {
|
||||
modelOverrides: {
|
||||
"anthropic/claude-sonnet-4": {
|
||||
thinking: { mode: "budget", minLevel: Effort.Low, maxLevel: Effort.Medium },
|
||||
thinking: { mode: "budget", efforts: [Effort.Low, Effort.Medium] },
|
||||
},
|
||||
},
|
||||
},
|
||||
@@ -1168,8 +1170,7 @@ describe("ModelRegistry", () => {
|
||||
|
||||
expect(model?.thinking).toEqual({
|
||||
mode: "budget",
|
||||
minLevel: Effort.Low,
|
||||
maxLevel: Effort.Medium,
|
||||
efforts: [Effort.Low, Effort.Medium],
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -25,8 +25,7 @@ const mockModels: Model<"anthropic-messages">[] = [
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "budget",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.High,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
},
|
||||
input: ["text", "image"],
|
||||
cost: { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
|
||||
@@ -58,8 +57,7 @@ const mockOpenRouterModels: Model<Api>[] = [
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "budget",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.High,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 1, output: 2, cacheRead: 0.1, cacheWrite: 1 },
|
||||
@@ -87,8 +85,7 @@ const mockOpenRouterModels: Model<Api>[] = [
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "budget",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.High,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 1, output: 2, cacheRead: 0.1, cacheWrite: 1 },
|
||||
@@ -134,8 +131,7 @@ const mockCodexOverlapModels: Model<"anthropic-messages">[] = [
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
minLevel: Effort.Low,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 1.5, output: 6, cacheRead: 0.15, cacheWrite: 1.5 },
|
||||
@@ -151,8 +147,7 @@ const mockCodexOverlapModels: Model<"anthropic-messages">[] = [
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
minLevel: Effort.Low,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 1, output: 4, cacheRead: 0.1, cacheWrite: 1 },
|
||||
@@ -171,8 +166,7 @@ function createOpusModel(provider: string, id: string, name: string): Model<"ant
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "budget",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.XHigh,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
||||
},
|
||||
input: ["text", "image"],
|
||||
cost: { input: 15, output: 75, cacheRead: 1.5, cacheWrite: 18.75 },
|
||||
@@ -191,8 +185,7 @@ const canonicalVariantModels: Model<"anthropic-messages">[] = [
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "budget",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.High,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
},
|
||||
input: ["text", "image"],
|
||||
cost: { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
|
||||
@@ -208,8 +201,7 @@ const canonicalVariantModels: Model<"anthropic-messages">[] = [
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "budget",
|
||||
minLevel: Effort.Minimal,
|
||||
maxLevel: Effort.High,
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
},
|
||||
input: ["text", "image"],
|
||||
cost: { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
|
||||
|
||||
@@ -37,7 +37,7 @@ function createReasoningModel(): Model<"openai-responses"> {
|
||||
provider: "openai",
|
||||
baseUrl: "https://example.invalid",
|
||||
reasoning: true,
|
||||
thinking: { mode: "effort", minLevel: Effort.Medium, maxLevel: Effort.High },
|
||||
thinking: { mode: "effort", efforts: [Effort.Medium, Effort.High] },
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 8192,
|
||||
|
||||
Reference in New Issue
Block a user