From d8c5659d9af3178633ffac492dd94a647b60f2e9 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 17 Aug 2026 10:42:16 +0300 Subject: [PATCH] feat(catalog): updated context window floor and pricing parameters for gpt models - Updated GPT-5.6 context window floor to 1,000,000 tokens across discovery, policies, and tests. - Updated model configurations and pricing parameters in catalog models JSON. --- packages/catalog/CHANGELOG.md | 4 + .../catalog/scripts/generated-policies.ts | 16 +- packages/catalog/src/discovery/codex.ts | 26 +- packages/catalog/src/models.json | 890 ++++++++++++++---- packages/catalog/test/codex-discovery.test.ts | 22 +- .../catalog/test/generated-policies.test.ts | 10 +- packages/catalog/test/issue-887-repro.test.ts | 65 -- 7 files changed, 735 insertions(+), 298 deletions(-) delete mode 100644 packages/catalog/test/issue-887-repro.test.ts diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 9d2da00bc..9822535f3 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -1,6 +1,10 @@ # Changelog ## [Unreleased] +### Fixed + +- Raised the GPT-5.6 Sol/Terra/Luna context window on the Codex transport (openai-codex) from 372K to 1M tokens: OpenAI enabled the 1M window for subscription Codex on 2026-08-16, but the Codex model registry still reports the stale 272,000, so discovery now floors these SKUs at 1,000,000 instead of trusting the reported value ([openai/codex#38917](https://github.com/openai/codex/issues/38917)). + ## [17.3.5] - 2026-08-16 diff --git a/packages/catalog/scripts/generated-policies.ts b/packages/catalog/scripts/generated-policies.ts index caaeee24e..2ca015337 100644 --- a/packages/catalog/scripts/generated-policies.ts +++ b/packages/catalog/scripts/generated-policies.ts @@ -145,7 +145,7 @@ const CODEX_GPT_5_4_PRIORITY_BY_VARIANT: Partial> nano: 2, }; -const CODEX_GPT_5_6_372K_MODEL_IDS: Record = { +const CODEX_GPT_5_6_1M_MODEL_IDS: Record = { "gpt-5.6-luna": true, "gpt-5.6-sol": true, "gpt-5.6-terra": true, @@ -536,12 +536,12 @@ function applyOpenAICatalogPolicy(model: ModelSpec, parsedModel: OpenAIMode model.contextWindow = 272000; } } - // GPT-5.6 luna/sol/terra on the Codex transport: OpenAI's Codex model - // registry declares context_window = max_context_window = 372000, but Codex - // discovery omits `context_window` for these SKUs and falls back to - // DEFAULT_CONTEXT_WINDOW (272000, src/discovery/codex.ts), which regressed - // the bundled hard capacity (#5705). Pin the true 372K input window. - if (model.api === "openai-codex-responses" && CODEX_GPT_5_6_372K_MODEL_IDS[model.id]) { - model.contextWindow = 372000; + // GPT-5.6 luna/sol/terra on the Codex transport: OpenAI enabled a 1M-token + // window for subscription Codex (2026-08-16), but the Codex model registry + // still reports the stale 272000 (openai/codex#38917), so floor the bundled + // window at 1,000,000. Daybreak aliases are excluded — the registry actively + // reports their true window. + if (model.api === "openai-codex-responses" && CODEX_GPT_5_6_1M_MODEL_IDS[model.id]) { + model.contextWindow = Math.max(model.contextWindow ?? 0, 1_000_000); } } diff --git a/packages/catalog/src/discovery/codex.ts b/packages/catalog/src/discovery/codex.ts index 6cbc32293..9c950ee14 100644 --- a/packages/catalog/src/discovery/codex.ts +++ b/packages/catalog/src/discovery/codex.ts @@ -9,13 +9,19 @@ const DEFAULT_MODEL_LIST_PATHS = ["/codex/models", "/models"] as const; const DEFAULT_CONTEXT_WINDOW = 272_000; const DEFAULT_MAX_TOKENS = 128_000; /** - * GPT-5.6 luna/sol/terra hard context capacity. Codex discovery omits - * `context_window` for these SKUs, so the generic {@link DEFAULT_CONTEXT_WINDOW} - * (272000) would understate the real window — OpenAI's Codex model registry - * declares context_window = max_context_window = 372000 (#5705). Used as the - * fallback only when upstream reports no value. + * Fallback for GPT-5.6-family SKUs when upstream omits `context_window`: the + * generic {@link DEFAULT_CONTEXT_WINDOW} (272000) understates the registry's + * former 372000 hard capacity (#5705). */ const GPT_5_6_CONTEXT_WINDOW = 372_000; +/** + * OpenAI enabled a 1M-token window for subscription Codex on GPT-5.6 + * luna/sol/terra (2026-08-16), but the Codex model registry still reports the + * stale 272000 — so the reported value must be floored, not just defaulted + * (openai/codex#38917; Codex CLI override `model_context_window = 1000000`). + */ +const GPT_5_6_1M_CONTEXT_WINDOW = 1_000_000; +const CODEX_GPT_5_6_1M_SLUGS: ReadonlySet = new Set(["gpt-5.6-luna", "gpt-5.6-sol", "gpt-5.6-terra"]); const CODEX_REMOTE_COMPACTION = { enabled: true, api: "openai-codex-responses", @@ -224,14 +230,18 @@ function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCo } const name = toNonEmptyString(payload.display_name) ?? slug; - // Codex discovery omits `context_window` for GPT-5.6 luna/sol/terra; the - // generic 272000 fallback understates their real 372000 window (#5705). + // Codex discovery historically omitted `context_window` for GPT-5.6-family + // SKUs (#5705); luna/sol/terra additionally floor the reported value because + // the registry still declares the pre-1M 272000 window. const parsed = parseKnownModel(slug); const fallbackContextWindow = parsed.family === "openai" && semverEqual(parsed.version, "5.6") ? GPT_5_6_CONTEXT_WINDOW : DEFAULT_CONTEXT_WINDOW; - const contextWindow = toPositiveInt(payload.context_window) ?? fallbackContextWindow; + const reportedContextWindow = toPositiveInt(payload.context_window) ?? fallbackContextWindow; + const contextWindow = CODEX_GPT_5_6_1M_SLUGS.has(slug) + ? Math.max(reportedContextWindow, GPT_5_6_1M_CONTEXT_WINDOW) + : reportedContextWindow; const maxTokens = Math.min(DEFAULT_MAX_TOKENS, contextWindow); const reasoning = supportsReasoning(payload.default_reasoning_level, payload.supported_reasoning_levels); const input = normalizeInputModalities(payload.input_modalities); diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index fb3fdbce0..e7215b13a 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -9533,6 +9533,96 @@ "supportsDisplay": true } }, + "global.openai.gpt-5.6-luna": { + "id": "global.openai.gpt-5.6-luna", + "name": "GPT-5.6 Luna (Global)", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.22, + "output": 1.32, + "cacheRead": 0.022, + "cacheWrite": 0.275 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, + "global.openai.gpt-5.6-sol": { + "id": "global.openai.gpt-5.6-sol", + "name": "GPT-5.6 Sol (Global)", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5.5, + "output": 33, + "cacheRead": 0.55, + "cacheWrite": 6.875 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, + "global.openai.gpt-5.6-terra": { + "id": "global.openai.gpt-5.6-terra", + "name": "GPT-5.6 Terra (Global)", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.2, + "output": 13.2, + "cacheRead": 0.22, + "cacheWrite": 2.75 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, "google.gemma-3-27b-it": { "id": "google.gemma-3-27b-it", "name": "Google Gemma 3 27B Instruct", @@ -17022,7 +17112,7 @@ "cursor": { "claude-4-sonnet": { "id": "claude-4-sonnet", - "name": "Sonnet 4", + "name": "Claude Sonnet 4", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17061,7 +17151,7 @@ }, "claude-4.5-opus-high": { "id": "claude-4.5-opus-high", - "name": "Opus 4.5", + "name": "Claude Opus 4.5", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17100,7 +17190,7 @@ }, "claude-4.5-sonnet": { "id": "claude-4.5-sonnet", - "name": "Sonnet 4.5", + "name": "Claude Sonnet 4.5", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17139,7 +17229,7 @@ }, "claude-4.6-opus-high": { "id": "claude-4.6-opus-high", - "name": "Opus 4.6 1M", + "name": "Claude Opus 4.6 1M", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17177,7 +17267,7 @@ }, "claude-4.6-opus-max": { "id": "claude-4.6-opus-max", - "name": "Opus 4.6 1M Max", + "name": "Claude Opus 4.6 1M Max", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17216,7 +17306,7 @@ }, "claude-4.6-sonnet-medium": { "id": "claude-4.6-sonnet-medium", - "name": "Sonnet 4.6 1M", + "name": "Claude Sonnet 4.6 1M", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17254,7 +17344,7 @@ }, "claude-fable-5-high": { "id": "claude-fable-5-high", - "name": "Fable 5 1M (NO ZDR)", + "name": "Claude Fable 5 1M (NO ZDR)", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17293,7 +17383,7 @@ }, "claude-fable-5-low": { "id": "claude-fable-5-low", - "name": "Fable 5 1M Low (NO ZDR)", + "name": "Claude Fable 5 1M Low (NO ZDR)", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17332,7 +17422,7 @@ }, "claude-fable-5-max": { "id": "claude-fable-5-max", - "name": "Fable 5 1M Max (NO ZDR)", + "name": "Claude Fable 5 1M Max (NO ZDR)", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17371,7 +17461,7 @@ }, "claude-fable-5-medium": { "id": "claude-fable-5-medium", - "name": "Fable 5 1M Medium (NO ZDR)", + "name": "Claude Fable 5 1M Medium (NO ZDR)", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17410,7 +17500,7 @@ }, "claude-fable-5-xhigh": { "id": "claude-fable-5-xhigh", - "name": "Fable 5 1M Extra High (NO ZDR)", + "name": "Claude Fable 5 1M Extra High (NO ZDR)", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17449,7 +17539,7 @@ }, "claude-opus-4-7-high": { "id": "claude-opus-4-7-high", - "name": "Opus 4.7 1M High", + "name": "Claude Opus 4.7 1M High", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17488,7 +17578,7 @@ }, "claude-opus-4-7-high-fast": { "id": "claude-opus-4-7-high-fast", - "name": "Opus 4.7 1M High Fast", + "name": "Claude Opus 4.7 1M High Fast", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17527,7 +17617,7 @@ }, "claude-opus-4-7-low": { "id": "claude-opus-4-7-low", - "name": "Opus 4.7 1M Low", + "name": "Claude Opus 4.7 1M Low", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17566,7 +17656,7 @@ }, "claude-opus-4-7-low-fast": { "id": "claude-opus-4-7-low-fast", - "name": "Opus 4.7 1M Low Fast", + "name": "Claude Opus 4.7 1M Low Fast", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17605,7 +17695,7 @@ }, "claude-opus-4-7-max": { "id": "claude-opus-4-7-max", - "name": "Opus 4.7 1M Max", + "name": "Claude Opus 4.7 1M Max", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17644,7 +17734,7 @@ }, "claude-opus-4-7-max-fast": { "id": "claude-opus-4-7-max-fast", - "name": "Opus 4.7 1M Max Fast", + "name": "Claude Opus 4.7 1M Max Fast", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17683,7 +17773,7 @@ }, "claude-opus-4-7-medium": { "id": "claude-opus-4-7-medium", - "name": "Opus 4.7 1M Medium", + "name": "Claude Opus 4.7 1M Medium", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17722,7 +17812,7 @@ }, "claude-opus-4-7-medium-fast": { "id": "claude-opus-4-7-medium-fast", - "name": "Opus 4.7 1M Medium Fast", + "name": "Claude Opus 4.7 1M Medium Fast", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17761,7 +17851,7 @@ }, "claude-opus-4-7-xhigh": { "id": "claude-opus-4-7-xhigh", - "name": "Opus 4.7 1M", + "name": "Claude Opus 4.7 1M", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17800,7 +17890,7 @@ }, "claude-opus-4-7-xhigh-fast": { "id": "claude-opus-4-7-xhigh-fast", - "name": "Opus 4.7 1M Fast", + "name": "Claude Opus 4.7 1M Fast", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17839,7 +17929,7 @@ }, "claude-opus-4-8-high": { "id": "claude-opus-4-8-high", - "name": "Opus 4.8 1M", + "name": "Claude Opus 4.8 1M", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17878,7 +17968,7 @@ }, "claude-opus-4-8-high-fast": { "id": "claude-opus-4-8-high-fast", - "name": "Opus 4.8 1M Fast", + "name": "Claude Opus 4.8 1M Fast", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17917,7 +18007,7 @@ }, "claude-opus-4-8-low": { "id": "claude-opus-4-8-low", - "name": "Opus 4.8 1M Low", + "name": "Claude Opus 4.8 1M Low", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17956,7 +18046,7 @@ }, "claude-opus-4-8-low-fast": { "id": "claude-opus-4-8-low-fast", - "name": "Opus 4.8 1M Low Fast", + "name": "Claude Opus 4.8 1M Low Fast", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -17995,7 +18085,7 @@ }, "claude-opus-4-8-max": { "id": "claude-opus-4-8-max", - "name": "Opus 4.8 1M Max", + "name": "Claude Opus 4.8 1M Max", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18034,7 +18124,7 @@ }, "claude-opus-4-8-max-fast": { "id": "claude-opus-4-8-max-fast", - "name": "Opus 4.8 1M Max Fast", + "name": "Claude Opus 4.8 1M Max Fast", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18073,7 +18163,7 @@ }, "claude-opus-4-8-medium": { "id": "claude-opus-4-8-medium", - "name": "Opus 4.8 1M Medium", + "name": "Claude Opus 4.8 1M Medium", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18112,7 +18202,7 @@ }, "claude-opus-4-8-medium-fast": { "id": "claude-opus-4-8-medium-fast", - "name": "Opus 4.8 1M Medium Fast", + "name": "Claude Opus 4.8 1M Medium Fast", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18151,7 +18241,7 @@ }, "claude-opus-4-8-xhigh": { "id": "claude-opus-4-8-xhigh", - "name": "Opus 4.8 1M Extra High", + "name": "Claude Opus 4.8 1M Extra High", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18190,7 +18280,7 @@ }, "claude-opus-4-8-xhigh-fast": { "id": "claude-opus-4-8-xhigh-fast", - "name": "Opus 4.8 1M Extra High Fast", + "name": "Claude Opus 4.8 1M Extra High Fast", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18229,7 +18319,7 @@ }, "claude-opus-5-high": { "id": "claude-opus-5-high", - "name": "Opus 5 1M", + "name": "Claude Opus 5 1M", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18268,7 +18358,7 @@ }, "claude-opus-5-high-fast": { "id": "claude-opus-5-high-fast", - "name": "Opus 5 1M Fast", + "name": "Claude Opus 5 1M Fast", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18307,7 +18397,7 @@ }, "claude-opus-5-low": { "id": "claude-opus-5-low", - "name": "Opus 5 1M Low", + "name": "Claude Opus 5 1M Low", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18346,7 +18436,7 @@ }, "claude-opus-5-low-fast": { "id": "claude-opus-5-low-fast", - "name": "Opus 5 1M Low Fast", + "name": "Claude Opus 5 1M Low Fast", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18385,7 +18475,7 @@ }, "claude-opus-5-medium": { "id": "claude-opus-5-medium", - "name": "Opus 5 1M Medium", + "name": "Claude Opus 5 1M Medium", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18424,7 +18514,7 @@ }, "claude-opus-5-medium-fast": { "id": "claude-opus-5-medium-fast", - "name": "Opus 5 1M Medium Fast", + "name": "Claude Opus 5 1M Medium Fast", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18463,7 +18553,7 @@ }, "claude-opus-5-thinking-max": { "id": "claude-opus-5-thinking-max", - "name": "Opus 5 1M Max Thinking", + "name": "Claude Opus 5 1M Max Thinking", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18486,7 +18576,7 @@ }, "claude-opus-5-thinking-max-fast": { "id": "claude-opus-5-thinking-max-fast", - "name": "Opus 5 1M Max Thinking Fast", + "name": "Claude Opus 5 1M Max Thinking Fast", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18509,7 +18599,7 @@ }, "claude-opus-5-thinking-xhigh": { "id": "claude-opus-5-thinking-xhigh", - "name": "Opus 5 1M Extra High Thinking", + "name": "Claude Opus 5 1M Extra High Thinking", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18532,7 +18622,7 @@ }, "claude-opus-5-thinking-xhigh-fast": { "id": "claude-opus-5-thinking-xhigh-fast", - "name": "Opus 5 1M Extra High Thinking Fast", + "name": "Claude Opus 5 1M Extra High Thinking Fast", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18555,7 +18645,7 @@ }, "claude-sonnet-5-high": { "id": "claude-sonnet-5-high", - "name": "Sonnet 5 1M", + "name": "Claude Sonnet 5 1M", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18594,7 +18684,7 @@ }, "claude-sonnet-5-low": { "id": "claude-sonnet-5-low", - "name": "Sonnet 5 1M Low", + "name": "Claude Sonnet 5 1M Low", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18633,7 +18723,7 @@ }, "claude-sonnet-5-max": { "id": "claude-sonnet-5-max", - "name": "Sonnet 5 1M Max", + "name": "Claude Sonnet 5 1M Max", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18672,7 +18762,7 @@ }, "claude-sonnet-5-medium": { "id": "claude-sonnet-5-medium", - "name": "Sonnet 5 1M Medium", + "name": "Claude Sonnet 5 1M Medium", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -18711,7 +18801,7 @@ }, "claude-sonnet-5-xhigh": { "id": "claude-sonnet-5-xhigh", - "name": "Sonnet 5 1M Extra High", + "name": "Claude Sonnet 5 1M Extra High", "api": "cursor-agent", "provider": "cursor", "baseUrl": "https://api2.cursor.sh", @@ -29507,6 +29597,34 @@ ] } }, + "Qwen/Qwen3.8-2.4T-A95B": { + "id": "Qwen/Qwen3.8-2.4T-A95B", + "name": "Qwen3.8 2.4T A95B", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 2.5, + "output": 6.25, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "stepfun-ai/Step-3.5-Flash": { "id": "stepfun-ai/Step-3.5-Flash", "name": "Step 3.5 Flash", @@ -30131,9 +30249,9 @@ "text" ], "cost": { - "input": 0.0675, - "output": 0.135, - "cacheRead": 0.0135, + "input": 0.0603, + "output": 0.1206, + "cacheRead": 0.01206, "cacheWrite": 0 }, "contextWindow": 262144, @@ -30219,13 +30337,13 @@ "image" ], "cost": { - "input": 2.8, - "output": 14, + "input": 2.6, + "output": 13, "cacheRead": 0.29, "cacheWrite": 0 }, - "contextWindow": 1048576, - "maxTokens": 1048576, + "contextWindow": 974842, + "maxTokens": 974842, "thinking": { "mode": "effort", "efforts": [ @@ -32620,7 +32738,7 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 393216, + "maxTokens": 384000, "thinking": { "mode": "effort", "efforts": [ @@ -33592,7 +33710,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, + "maxTokens": 16384, "thinking": { "mode": "effort", "efforts": [ @@ -36503,7 +36621,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 228000, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -36699,7 +36817,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -39669,8 +39787,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 40960, - "maxTokens": 16384, + "contextWindow": 131072, + "maxTokens": 8192, "thinking": { "mode": "effort", "efforts": [ @@ -40416,8 +40534,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 262144, - "maxTokens": 65536, + "contextWindow": 131072, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -40440,7 +40558,7 @@ "image" ], "cost": { - "input": 0.15, + "input": 0.14, "output": 1, "cacheRead": 0.05, "cacheWrite": 0 @@ -40734,7 +40852,7 @@ "cost": { "input": 0.45, "output": 3.2, - "cacheRead": 0, + "cacheRead": 0.05, "cacheWrite": 0 }, "contextWindow": 262144, @@ -41422,9 +41540,9 @@ "text" ], "cost": { - "input": 0.132, - "output": 0.528, - "cacheRead": 0.033, + "input": 0.14, + "output": 0.58, + "cacheRead": 0.035, "cacheWrite": 0 }, "contextWindow": 262144, @@ -42768,7 +42886,7 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 131072, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -42780,6 +42898,26 @@ ] } }, + "z-ai/glm-5.2:free": { + "id": "z-ai/glm-5.2:free", + "name": "GLM 5.2 (free)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "supportsComputerUse": false + }, "z-ai/glm-5v-turbo": { "id": "z-ai/glm-5v-turbo", "name": "GLM-5V-Turbo", @@ -45222,7 +45360,8 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": null + "maxTokens": null, + "supportsComputerUse": false }, "moonshot-v1-128k-vision-preview": { "id": "moonshot-v1-128k-vision-preview", @@ -45242,7 +45381,8 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": null + "maxTokens": null, + "supportsComputerUse": false }, "moonshot-v1-32k": { "id": "moonshot-v1-32k", @@ -45261,7 +45401,8 @@ "cacheWrite": 0 }, "contextWindow": 32768, - "maxTokens": null + "maxTokens": null, + "supportsComputerUse": false }, "moonshot-v1-32k-vision-preview": { "id": "moonshot-v1-32k-vision-preview", @@ -45281,7 +45422,8 @@ "cacheWrite": 0 }, "contextWindow": 32768, - "maxTokens": null + "maxTokens": null, + "supportsComputerUse": false }, "moonshot-v1-8k": { "id": "moonshot-v1-8k", @@ -45300,7 +45442,8 @@ "cacheWrite": 0 }, "contextWindow": 8192, - "maxTokens": null + "maxTokens": null, + "supportsComputerUse": false }, "moonshot-v1-8k-vision-preview": { "id": "moonshot-v1-8k-vision-preview", @@ -45320,7 +45463,8 @@ "cacheWrite": 0 }, "contextWindow": 8192, - "maxTokens": null + "maxTokens": null, + "supportsComputerUse": false }, "moonshot-v1-auto": { "id": "moonshot-v1-auto", @@ -45339,7 +45483,8 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": null + "maxTokens": null, + "supportsComputerUse": false } }, "nanogpt": { @@ -48816,7 +48961,8 @@ "high", "max" ] - } + }, + "supportsComputerUse": false }, "deepseek/deepseek-v4-flash-0731-cheaper:thinking": { "id": "deepseek/deepseek-v4-flash-0731-cheaper:thinking", @@ -48843,7 +48989,8 @@ "high", "max" ] - } + }, + "supportsComputerUse": false }, "deepseek/deepseek-v4-flash-0731:thinking": { "id": "deepseek/deepseek-v4-flash-0731:thinking", @@ -49032,7 +49179,8 @@ "high", "max" ] - } + }, + "supportsComputerUse": false }, "deepseek/deepseek-v4-pro-cheaper:thinking": { "id": "deepseek/deepseek-v4-pro-cheaper:thinking", @@ -49059,7 +49207,8 @@ "high", "max" ] - } + }, + "supportsComputerUse": false }, "deepseek/deepseek-v4-pro:thinking": { "id": "deepseek/deepseek-v4-pro:thinking", @@ -49151,9 +49300,39 @@ "supportsComputerUse": false, "supportsComputerUseConfig": false }, + "dots-studio/dots-3-note-preview": { + "id": "dots-studio/dots-3-note-preview", + "name": "Dots3-Note Preview", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.1, + "output": 0.2, + "cacheRead": 0.05, + "cacheWrite": 0 + }, + "contextWindow": 393216, + "maxTokens": 393216, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "dots-studio/dots-3-note-preview:free": { "id": "dots-studio/dots-3-note-preview:free", - "name": "Dots3-Note Preview", + "name": "Dots3-Note Preview (free)", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", @@ -49179,7 +49358,8 @@ "high", "xhigh" ] - } + }, + "supportsComputerUse": false }, "doubao-1-5-thinking-pro-250415": { "id": "doubao-1-5-thinking-pro-250415", @@ -60271,6 +60451,35 @@ "supportsComputerUse": false, "supportsComputerUseConfig": false }, + "qwen3.5-0.8b": { + "id": "qwen3.5-0.8b", + "name": "Qwen3.5 0.8B", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.06, + "output": 0.12, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "qwen3.5-122b-a10b": { "id": "qwen3.5-122b-a10b", "name": "Qwen3.5 122B A10B", @@ -61328,6 +61537,35 @@ ] } }, + "qwen3.5-2b": { + "id": "qwen3.5-2b", + "name": "Qwen3.5 2B", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.08, + "output": 0.16, + "cacheRead": 0.04, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "qwen3.5-35b-a3b": { "id": "qwen3.5-35b-a3b", "name": "Qwen3.5 35B A3B", @@ -61377,6 +61615,35 @@ ] } }, + "qwen3.5-4b": { + "id": "qwen3.5-4b", + "name": "Qwen3.5 4B", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.1, + "output": 0.2, + "cacheRead": 0.05, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "qwen3.5-flash": { "id": "qwen3.5-flash", "name": "qwen3.5-flash", @@ -61642,6 +61909,64 @@ ] } }, + "qwen3.8-27b": { + "id": "qwen3.8-27b", + "name": "Qwen3.8 27B", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.4, + "output": 3, + "cacheRead": 0.2, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "qwen3.8-27b:thinking": { + "id": "qwen3.8-27b:thinking", + "name": "Qwen3.8 27B Thinking", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.4, + "output": 3, + "cacheRead": 0.2, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "qwen3.8-max": { "id": "qwen3.8-max", "name": "Qwen3.8 Max", @@ -62730,6 +63055,60 @@ "supportsComputerUse": false, "supportsComputerUseConfig": false }, + "TEE/deepseek-v4-pro-0813": { + "id": "TEE/deepseek-v4-pro-0813", + "name": "DeepSeek V4 Pro 0813 TEE", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.32, + "output": 3.96, + "cacheRead": 0.132, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 1048576, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ] + } + }, + "TEE/deepseek-v4-pro-0813:thinking": { + "id": "TEE/deepseek-v4-pro-0813:thinking", + "name": "DeepSeek V4 Pro 0813 Thinking TEE", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.32, + "output": 3.96, + "cacheRead": 0.132, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 1048576, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ] + } + }, "TEE/deepseek-v4-pro:thinking": { "id": "TEE/deepseek-v4-pro:thinking", "name": "DeepSeek V4 Pro Thinking TEE", @@ -67022,7 +67401,7 @@ }, "google/gemma-4-26b-a4b-it": { "id": "google/gemma-4-26b-a4b-it", - "name": "Gemma 4 26B A4B IT", + "name": "Gemma 4 26B A4B", "api": "openai-completions", "provider": "novita", "baseUrl": "https://api.novita.ai/openai/v1", @@ -76174,7 +76553,7 @@ "api": "openai-codex-responses", "v2StreamingEnabled": true }, - "contextWindow": 372000, + "contextWindow": 1000000, "maxTokens": 128000, "preferWebsockets": true, "useResponsesLite": true, @@ -76213,7 +76592,7 @@ "api": "openai-codex-responses", "v2StreamingEnabled": true }, - "contextWindow": 372000, + "contextWindow": 1000000, "maxTokens": 128000, "preferWebsockets": true, "useResponsesLite": true, @@ -76252,7 +76631,7 @@ "api": "openai-codex-responses", "v2StreamingEnabled": true }, - "contextWindow": 372000, + "contextWindow": 1000000, "maxTokens": 128000, "preferWebsockets": true, "useResponsesLite": true, @@ -76459,7 +76838,7 @@ "opencode-go": { "deepseek-v4-flash": { "id": "deepseek-v4-flash", - "name": "DeepSeek V4 Flash (2x usage)", + "name": "DeepSeek V4 Flash", "api": "openai-responses", "provider": "opencode-go", "baseUrl": "https://opencode.ai/zen/go/v1", @@ -76468,9 +76847,9 @@ "text" ], "cost": { - "input": 0.07, - "output": 0.14, - "cacheRead": 0.0014, + "input": 0.22, + "output": 0.66, + "cacheRead": 0.007, "cacheWrite": 0 }, "contextWindow": 1000000, @@ -76503,9 +76882,9 @@ "text" ], "cost": { - "input": 0.435, - "output": 0.87, - "cacheRead": 0.003625, + "input": 0.66, + "output": 1.98, + "cacheRead": 0.022, "cacheWrite": 0 }, "contextWindow": 1000000, @@ -77130,9 +77509,9 @@ "qwen3.7-max": { "id": "qwen3.7-max", "name": "Qwen3.7 Max", - "api": "anthropic-messages", + "api": "openai-completions", "provider": "opencode-go", - "baseUrl": "https://opencode.ai/zen/go", + "baseUrl": "https://opencode.ai/zen/go/v1", "reasoning": true, "input": [ "text" @@ -77146,22 +77525,21 @@ "contextWindow": 1000000, "maxTokens": 65536, "thinking": { - "mode": "budget", + "mode": "effort", "efforts": [ "minimal", "low", "medium", - "high", - "xhigh" + "high" ] } }, "qwen3.7-plus": { "id": "qwen3.7-plus", "name": "Qwen3.7 Plus", - "api": "anthropic-messages", + "api": "openai-completions", "provider": "opencode-go", - "baseUrl": "https://opencode.ai/zen/go", + "baseUrl": "https://opencode.ai/zen/go/v1", "reasoning": true, "input": [ "text", @@ -77176,22 +77554,21 @@ "contextWindow": 1000000, "maxTokens": 65536, "thinking": { - "mode": "budget", + "mode": "effort", "efforts": [ "minimal", "low", "medium", - "high", - "xhigh" + "high" ] } }, "qwen3.8-max": { "id": "qwen3.8-max", "name": "Qwen3.8 Max", - "api": "anthropic-messages", + "api": "openai-completions", "provider": "opencode-go", - "baseUrl": "https://opencode.ai/zen/go", + "baseUrl": "https://opencode.ai/zen/go/v1", "reasoning": true, "input": [ "text", @@ -77206,13 +77583,12 @@ "contextWindow": 1000000, "maxTokens": 131072, "thinking": { - "mode": "budget", + "mode": "effort", "efforts": [ "minimal", "low", "medium", - "high", - "xhigh" + "high" ] } } @@ -79790,12 +80166,12 @@ "text" ], "cost": { - "input": 0.0675, - "output": 0.135, - "cacheRead": 0.0135, + "input": 0.060300000000000006, + "output": 0.12060000000000001, + "cacheRead": 0.01206, "cacheWrite": 0 }, - "contextWindow": 1048576, + "contextWindow": 1310720, "maxTokens": 262144, "thinking": { "mode": "effort", @@ -79882,13 +80258,13 @@ "image" ], "cost": { - "input": 2.8, - "output": 14, + "input": 2.6, + "output": 13, "cacheRead": 0.29, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 1048576, + "maxTokens": 974842, "thinking": { "mode": "effort", "efforts": [ @@ -82102,9 +82478,9 @@ "text" ], "cost": { - "input": 0.06426, - "output": 0.12852, - "cacheRead": 0.012852, + "input": 0.08259999999999999, + "output": 0.16519999999999999, + "cacheRead": 0.01652, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -82136,7 +82512,7 @@ "cacheRead": 0.028, "cacheWrite": 0 }, - "contextWindow": 1048576, + "contextWindow": 1310720, "maxTokens": 393216, "thinking": { "mode": "effort", @@ -82189,13 +82565,13 @@ "text" ], "cost": { - "input": 1.1680000000000001, - "output": 2.3360000000000003, - "cacheRead": 0.09855000000000001, + "input": 1.32, + "output": 3.9600000000000004, + "cacheRead": 0.044, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 393216, + "maxTokens": 384000, "thinking": { "mode": "effort", "efforts": [ @@ -82216,9 +82592,9 @@ "text" ], "cost": { - "input": 0.435, - "output": 0.87, - "cacheRead": 0.003625, + "input": 1.32, + "output": 3.9600000000000004, + "cacheRead": 0.044, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -83268,7 +83644,7 @@ }, "google/gemma-4-26b-a4b-it": { "id": "google/gemma-4-26b-a4b-it", - "name": "Gemma 4 26B A4B IT", + "name": "Gemma 4 26B A4B", "api": "openrouter", "provider": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", @@ -83278,13 +83654,13 @@ "image" ], "cost": { - "input": 0.12, - "output": 0.39999999999999997, + "input": 0.07, + "output": 0.33999999999999997, "cacheRead": 0.049999999999999996, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, + "maxTokens": 16384, "thinking": { "mode": "effort", "efforts": [ @@ -85134,9 +85510,9 @@ "image" ], "cost": { - "input": 0.65, - "output": 3.41, - "cacheRead": 0.15, + "input": 0.95, + "output": 4, + "cacheRead": 0.16, "cacheWrite": 0 }, "contextWindow": 262144, @@ -85519,11 +85895,11 @@ "cost": { "input": 0.049999999999999996, "output": 0.19999999999999998, - "cacheRead": 0.024999999999999998, + "cacheRead": 0.03, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 228000, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -85759,13 +86135,13 @@ "text" ], "cost": { - "input": 0.09999999999999999, - "output": 0.25, - "cacheRead": 0.049999999999999996, + "input": 0.08, + "output": 0.19999999999999998, + "cacheRead": 0.04, "cacheWrite": 0 }, "contextWindow": 1000000, - "maxTokens": 262144, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -89562,13 +89938,13 @@ "text" ], "cost": { - "input": 0.12, - "output": 0.5, + "input": 0.13, + "output": 0.52, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 16384, + "maxTokens": 8192, "thinking": { "mode": "effort", "efforts": [ @@ -90028,8 +90404,8 @@ "image" ], "cost": { - "input": 0.26, - "output": 1.04, + "input": 0.21, + "output": 1.9, "cacheRead": 0.09999999999999999, "cacheWrite": 0 }, @@ -90460,13 +90836,13 @@ "image" ], "cost": { - "input": 0.3, - "output": 2, + "input": 0.28900000000000003, + "output": 2.4, "cacheRead": 0.03, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -90491,7 +90867,7 @@ "image" ], "cost": { - "input": 0.15, + "input": 0.14, "output": 1, "cacheRead": 0.049999999999999996, "cacheWrite": 0 @@ -90799,7 +91175,7 @@ "cost": { "input": 0.44999999999999996, "output": 3.1999999999999997, - "cacheRead": 0, + "cacheRead": 0.049999999999999996, "cacheWrite": 0 }, "contextWindow": 262144, @@ -92491,13 +92867,13 @@ "text" ], "cost": { - "input": 0.46199999999999997, - "output": 1.452, - "cacheRead": 0.0858, + "input": 1.19, + "output": 3.74, + "cacheRead": 0.221, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 131072, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -93187,6 +93563,33 @@ ] } }, + "deepseek-ai/DeepSeek-V4-Pro-0813": { + "id": "deepseek-ai/DeepSeek-V4-Pro-0813", + "name": "DeepSeek V4 Pro 0813", + "api": "openai-completions", + "provider": "together", + "baseUrl": "https://api.together.xyz/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.32, + "output": 3.96, + "cacheRead": 0.13, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 384000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ] + } + }, "essentialai/Rnj-1-Instruct": { "id": "essentialai/Rnj-1-Instruct", "name": "Rnj-1 Instruct", @@ -94025,9 +94428,9 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.32, + "output": 3.96, + "cacheRead": 0.044, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -98784,7 +99187,7 @@ "cost": { "input": 2, "output": 6, - "cacheRead": 0.25, + "cacheRead": 0.19999999999999998, "cacheWrite": 0 }, "contextWindow": 262144, @@ -98801,6 +99204,37 @@ }, "supportsComputerUse": false }, + "alibaba/qwen3.8-27b": { + "id": "alibaba/qwen3.8-27b", + "name": "Qwen3.8 27B", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262133, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + }, + "supportsComputerUse": false + }, "alibaba/qwen3.8-max": { "id": "alibaba/qwen3.8-max", "name": "Qwen 3.8 Max", @@ -99894,9 +100328,9 @@ "text" ], "cost": { - "input": 0.19999999999999998, - "output": 0.39999999999999997, - "cacheRead": 0.04, + "input": 0.13, + "output": 0.26, + "cacheRead": 0.028, "cacheWrite": 0 }, "contextWindow": 1000000, @@ -99924,9 +100358,9 @@ "text" ], "cost": { - "input": 0.19999999999999998, - "output": 0.39999999999999997, - "cacheRead": 0.04, + "input": 0.13, + "output": 0.26, + "cacheRead": 0.028, "cacheWrite": 0 }, "contextWindow": 1000000, @@ -100469,7 +100903,7 @@ }, "google/gemma-4-26b-a4b-it": { "id": "google/gemma-4-26b-a4b-it", - "name": "Gemma 4 26B A4B IT", + "name": "Gemma 4 26B A4B", "api": "anthropic-messages", "provider": "vercel-ai-gateway", "baseUrl": "https://ai-gateway.vercel.sh", @@ -102165,6 +102599,36 @@ }, "supportsComputerUse": false }, + "nvidia/nemotron-3.5-lightning": { + "id": "nvidia/nemotron-3.5-lightning", + "name": "Nvidia Nemotron 3.5 Lightning", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 131072, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + }, + "supportsComputerUse": false + }, "nvidia/nemotron-nano-12b-v2-vl": { "id": "nvidia/nemotron-nano-12b-v2-vl", "name": "Nemotron Nano 12B v2 VL", @@ -106420,6 +106884,7 @@ }, "contextWindow": 2000000, "maxTokens": 2000000, + "supportsComputerUse": false, "compat": { "includeEncryptedReasoning": true, "filterReasoningHistory": false, @@ -106447,6 +106912,7 @@ }, "contextWindow": 2000000, "maxTokens": 2000000, + "supportsComputerUse": false, "compat": { "includeEncryptedReasoning": true, "filterReasoningHistory": false, @@ -106473,16 +106939,6 @@ }, "contextWindow": 2000000, "maxTokens": 2000000, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "includeEncryptedReasoning": true, - "filterReasoningHistory": false, - "supportsImageDetailOriginal": false, - "omitReasoningEffort": false, - "supportsReasoningEffort": true - }, "thinking": { "mode": "effort", "efforts": [ @@ -106495,6 +106951,17 @@ "effortMap": { "minimal": "low" } + }, + "supportsComputerUse": false, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": true, + "filterReasoningHistory": false, + "supportsImageDetailOriginal": false, + "omitReasoningEffort": false, + "supportsReasoningEffort": true } }, "grok-4.3": { @@ -106516,18 +106983,6 @@ }, "contextWindow": 1000000, "maxTokens": 1000000, - "compat": { - "reasoningEffortMap": { - "minimal": "low", - "xhigh": "high", - "max": "high" - }, - "includeEncryptedReasoning": true, - "filterReasoningHistory": false, - "supportsImageDetailOriginal": false, - "omitReasoningEffort": false, - "supportsReasoningEffort": true - }, "thinking": { "mode": "effort", "efforts": [ @@ -106539,6 +106994,19 @@ "effortMap": { "minimal": "low" } + }, + "supportsComputerUse": false, + "compat": { + "reasoningEffortMap": { + "minimal": "low", + "xhigh": "high", + "max": "high" + }, + "includeEncryptedReasoning": true, + "filterReasoningHistory": false, + "supportsImageDetailOriginal": false, + "omitReasoningEffort": false, + "supportsReasoningEffort": true } }, "grok-4.5": { @@ -106560,18 +107028,6 @@ }, "contextWindow": 500000, "maxTokens": 500000, - "compat": { - "reasoningEffortMap": { - "minimal": "low", - "xhigh": "high", - "max": "high" - }, - "includeEncryptedReasoning": true, - "filterReasoningHistory": false, - "supportsImageDetailOriginal": false, - "omitReasoningEffort": false, - "supportsReasoningEffort": true - }, "thinking": { "mode": "effort", "efforts": [ @@ -106583,6 +107039,19 @@ "effortMap": { "minimal": "low" } + }, + "supportsComputerUse": false, + "compat": { + "reasoningEffortMap": { + "minimal": "low", + "xhigh": "high", + "max": "high" + }, + "includeEncryptedReasoning": true, + "filterReasoningHistory": false, + "supportsImageDetailOriginal": false, + "omitReasoningEffort": false, + "supportsReasoningEffort": true } }, "grok-4.6": { @@ -106604,16 +107073,6 @@ }, "contextWindow": 500000, "maxTokens": 500000, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "includeEncryptedReasoning": true, - "filterReasoningHistory": false, - "supportsImageDetailOriginal": false, - "omitReasoningEffort": false, - "supportsReasoningEffort": true - }, "thinking": { "mode": "effort", "efforts": [ @@ -106626,6 +107085,17 @@ "effortMap": { "minimal": "low" } + }, + "supportsComputerUse": false, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": true, + "filterReasoningHistory": false, + "supportsImageDetailOriginal": false, + "omitReasoningEffort": false, + "supportsReasoningEffort": true } }, "grok-build": { @@ -106647,6 +107117,7 @@ }, "contextWindow": 512000, "maxTokens": 512000, + "supportsComputerUse": false, "compat": { "includeEncryptedReasoning": true, "filterReasoningHistory": false, @@ -106674,6 +107145,7 @@ }, "contextWindow": 256000, "maxTokens": 256000, + "supportsComputerUse": false, "compat": { "includeEncryptedReasoning": true, "filterReasoningHistory": false, @@ -106700,6 +107172,7 @@ }, "contextWindow": 200000, "maxTokens": 200000, + "supportsComputerUse": false, "compat": { "includeEncryptedReasoning": true, "filterReasoningHistory": false, @@ -107676,7 +108149,8 @@ "defaultLevel": "max", "requiresEffort": true }, - "supportsComputerUse": false + "supportsComputerUse": false, + "supportsComputerUseConfig": false }, "glm-5v-turbo": { "id": "glm-5v-turbo", @@ -109391,7 +109865,7 @@ }, "google/gemma-4-26b-a4b-it": { "id": "google/gemma-4-26b-a4b-it", - "name": "Gemma 4 26B A4B IT", + "name": "Gemma 4 26B A4B", "api": "openai-completions", "provider": "zenmux", "baseUrl": "https://zenmux.ai/api/v1", diff --git a/packages/catalog/test/codex-discovery.test.ts b/packages/catalog/test/codex-discovery.test.ts index 17fa6cb64..7cf08217e 100644 --- a/packages/catalog/test/codex-discovery.test.ts +++ b/packages/catalog/test/codex-discovery.test.ts @@ -103,7 +103,7 @@ describe("Codex model discovery", () => { expect(legacy?.useResponsesLite).toBeUndefined(); }); - it("falls back to the 372K window for GPT-5.6 SKUs when upstream omits context_window (#5705)", async () => { + it("floors GPT-5.6 luna/sol/terra at the 1M window when upstream omits context_window (#5705)", async () => { const fetchFn: typeof fetch = Object.assign( async () => new Response( @@ -138,7 +138,7 @@ describe("Codex model discovery", () => { }); const sol = result?.models.find(model => model.id === "gpt-5.6-sol"); - expect(sol?.contextWindow).toBe(372_000); + expect(sol?.contextWindow).toBe(1_000_000); const legacy = result?.models.find(model => model.id === "gpt-5.5"); expect(legacy?.contextWindow).toBe(272_000); }); @@ -195,7 +195,7 @@ describe("Codex model discovery", () => { expect(red.cost).toEqual({ input: 12.5, output: 75, cacheRead: 1.25, cacheWrite: 15.625 }); }); - it("honors context_window when upstream actively reports it for GPT-5.6 SKUs", async () => { + it("floors stale reported windows for GPT-5.6 luna/sol/terra and honors reports above the floor", async () => { const fetchFn: typeof fetch = Object.assign( async () => new Response( @@ -210,6 +210,15 @@ describe("Codex model discovery", () => { input_modalities: ["text", "image"], supported_in_api: true, }, + { + slug: "gpt-5.6-terra", + display_name: "GPT-5.6-Terra", + context_window: 1_050_000, + default_reasoning_level: "medium", + supported_reasoning_levels: ["low", "medium", "high"], + input_modalities: ["text", "image"], + supported_in_api: true, + }, { slug: "gpt-5.5", display_name: "GPT-5.5", @@ -231,8 +240,13 @@ describe("Codex model discovery", () => { fetchFn, }); + // Registry still reports the pre-1M 272000 for sol; the floor must win. const sol = result?.models.find(model => model.id === "gpt-5.6-sol"); - expect(sol?.contextWindow).toBe(272_000); + expect(sol?.contextWindow).toBe(1_000_000); + // Reports above the floor are honored as-is. + const terra = result?.models.find(model => model.id === "gpt-5.6-terra"); + expect(terra?.contextWindow).toBe(1_050_000); + // Non-floored SKUs keep the actively reported value. const legacy = result?.models.find(model => model.id === "gpt-5.5"); expect(legacy?.contextWindow).toBe(272_000); }); diff --git a/packages/catalog/test/generated-policies.test.ts b/packages/catalog/test/generated-policies.test.ts index 88f6fd27a..45674a517 100644 --- a/packages/catalog/test/generated-policies.test.ts +++ b/packages/catalog/test/generated-policies.test.ts @@ -120,9 +120,9 @@ describe("generated model policies", () => { expect(models[4]?.cost.longContext).toBeUndefined(); }); - it("pins GPT-5.6 Codex-transport context window to the 372K hard capacity (#5705)", () => { + it("floors GPT-5.6 Codex-transport context windows at 1M (openai/codex#38917)", () => { const models: ModelSpec[] = [ - // Codex discovery underreports these via DEFAULT_CONTEXT_WINDOW=272000. + // Codex discovery/registry still reports the stale 272000 for these. createSpec({ id: "gpt-5.6-luna", api: "openai-codex-responses", @@ -155,9 +155,9 @@ describe("generated model policies", () => { applyGeneratedModelPolicies(models); - expect(models[0]?.contextWindow).toBe(372000); - expect(models[1]?.contextWindow).toBe(372000); - expect(models[2]?.contextWindow).toBe(372000); + expect(models[0]?.contextWindow).toBe(1_000_000); + expect(models[1]?.contextWindow).toBe(1_000_000); + expect(models[2]?.contextWindow).toBe(1_000_000); expect(models[3]?.contextWindow).toBe(1050000); expect(models[4]?.contextWindow).toBe(272000); }); diff --git a/packages/catalog/test/issue-887-repro.test.ts b/packages/catalog/test/issue-887-repro.test.ts deleted file mode 100644 index 2fac1dbee..000000000 --- a/packages/catalog/test/issue-887-repro.test.ts +++ /dev/null @@ -1,65 +0,0 @@ -/** - * Repro for #887 — OpenCode Go: Minimax M2.7 (and Qwen3.5/3.6 Plus) return 404 - * because the resolver routes them to anthropic-messages /v1/messages while - * the OpenCode Go gateway only serves them at /v1/chat/completions. - * - * stencil.so declares these ids with `provider.npm = "@ai-sdk/anthropic"`, - * which by default would resolve to anthropic-messages on opencode-go. The - * descriptor must override these specific ids to openai-completions so that - * regenerated models.json keeps the correct routing. - */ -import { describe, expect, test } from "bun:test"; -import { - MODELS_DEV_PROVIDER_DESCRIPTORS, - type ModelsDevModel, - opencodeGoModelManagerOptions, -} from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; - -const OPENCODE_GO_BASE = "https://opencode.ai/zen/go/v1"; - -describe("opencode-go resolver routes 404-ing ids to openai-completions (issue #887)", () => { - const descriptor = MODELS_DEV_PROVIDER_DESCRIPTORS.find(d => d.providerId === "opencode-go"); - - // Per upstream stencil.so (verified 2026-05-02 against - // https://stencil.so/api.json["opencode-go"].models), these three ids carry - // `provider.npm = "@ai-sdk/anthropic"`. The naive @ai-sdk/anthropic rule - // would route them to /v1/messages on opencode.ai/zen/go which 404s. - const npmAnthropic: ModelsDevModel = { provider: { npm: "@ai-sdk/anthropic" }, tool_call: true }; - - test.each([["minimax-m2.7"], ["qwen3.5-plus"], ["qwen3.6-plus"]])( - "%s resolves to openai-completions on /v1/chat/completions", - modelId => { - const resolved = descriptor?.resolveApi?.(modelId, npmAnthropic); - expect(resolved).toEqual({ api: "openai-completions", baseUrl: OPENCODE_GO_BASE }); - }, - ); - - test("minimax-m2.5 (control: works empirically) also resolves to openai-completions", () => { - // stencil.so currently lists minimax-m2.5 without an explicit provider.npm, - // so it falls through to the default openai-completions resolution. - const m25: ModelsDevModel = { tool_call: true }; - const resolved = descriptor?.resolveApi?.("minimax-m2.5", m25); - expect(resolved).toEqual({ api: "openai-completions", baseUrl: OPENCODE_GO_BASE }); - }); - - test("runtime /v1/models refresh preserves qwen3.7-max Anthropic transport", async () => { - let requestedUrl = ""; - const fetchMock = (async (input: string | Request | URL): Promise => { - requestedUrl = input instanceof Request ? input.url : String(input); - return new Response( - JSON.stringify({ - data: [{ id: "qwen3.7-max", name: "Qwen3.7 Max", context_length: 1000000 }], - }), - { headers: { "content-type": "application/json" } }, - ); - }) as typeof fetch; - - const options = opencodeGoModelManagerOptions({ apiKey: "opencode-test-key", fetch: fetchMock }); - const models = await options.fetchDynamicModels?.(); - const qwenMax = models?.find(model => model.id === "qwen3.7-max"); - - expect(requestedUrl).toBe("https://opencode.ai/zen/go/v1/models"); - expect(qwenMax?.api).toBe("anthropic-messages"); - expect(qwenMax?.baseUrl).toBe("https://opencode.ai/zen/go"); - }); -});