From faa70100eaaccf999d3f141ea49f37124c17cd3f Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 22:20:32 +0200 Subject: [PATCH] feat: enabled openai reasoning mode and integrated new model catalog - Enabled OpenAI pro reasoning mode by integrating reasoning aliases and parameter injection. - Expanded the model catalog with GPT-5.6 Luna, Sol, Terra, and Meta Muse Spark 1.1. - Updated model type definitions and provider request transformers to support reasoning configurations. - Refined model generation scripts to include new pro-reasoning aliases for OpenAI providers. --- packages/ai/CHANGELOG.md | 4 + .../src/providers/openai-codex-responses.ts | 2 +- .../openai-codex/request-transformer.ts | 8 + .../ai/src/providers/openai-responses-wire.ts | 7 + packages/ai/src/providers/openai-responses.ts | 7 + packages/catalog/CHANGELOG.md | 8 + packages/catalog/scripts/generate-models.ts | 5 + packages/catalog/src/models.json | 733 +++++++++++++++++- .../src/provider-models/openai-compat.ts | 56 ++ packages/catalog/src/types.ts | 7 + 10 files changed, 813 insertions(+), 24 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index f159a1eb6..d15331c39 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added OpenAI pro reasoning mode support: models carrying the catalog `reasoningMode: "pro"` marker (GPT-5.6 Pro aliases) send `reasoning: { mode: "pro" }` on OpenAI Responses and Codex Responses requests, alongside the configured effort. The Codex request body now honors `requestModelId` so catalog aliases request the base upstream model id. + ### Changed - Updated xAI OAuth to use a dedicated device-code flow instead of redirect/loopback server diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 0bde541d0..05654f0e0 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -949,7 +949,7 @@ export async function buildTransformedCodexRequestBody( promptCacheKey = normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId), ): Promise { const params: RequestBody = { - model: model.id, + model: model.requestModelId ?? model.id, input: convertMessages(model, context), stream: true, prompt_cache_key: promptCacheKey, diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index 67af8f232..968226016 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -23,6 +23,8 @@ export interface ReasoningConfig { effort: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; summary?: "auto" | "concise" | "detailed"; context?: CodexReasoningContext; + /** Pro reasoning serving mode (gpt-5.6+ catalog pro aliases). */ + mode?: "pro"; } export interface CodexRequestOptions { @@ -367,6 +369,12 @@ export async function transformRequestBody( } else { delete body.reasoning; } + // Catalog pro aliases (`gpt-5.6-*-pro`): applied after the effort branch so + // the mode is sent even when no effort is set (the branch above deletes + // `body.reasoning` in that case) — mode and effort are independent fields. + if (model.reasoningMode) { + body.reasoning = { ...body.reasoning, mode: model.reasoningMode }; + } body.text = { ...body.text, diff --git a/packages/ai/src/providers/openai-responses-wire.ts b/packages/ai/src/providers/openai-responses-wire.ts index 5246b5eaf..7992ec319 100644 --- a/packages/ai/src/providers/openai-responses-wire.ts +++ b/packages/ai/src/providers/openai-responses-wire.ts @@ -6318,6 +6318,13 @@ export interface Reasoning { * - `xhigh` is supported for all models after `gpt-5.1-codex-max`. */ effort?: ReasoningEffort | null; + /** + * **gpt-5.6 and later models only** + * + * Reasoning serving mode. `pro` routes the request to the pro reasoning + * path (more compute per response); omit for the standard path. + */ + mode?: "pro" | null; /** * @deprecated **Deprecated:** use `summary` instead. * diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index 759fdbfdc..ca0f89aa9 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -904,6 +904,13 @@ export function buildParams( model.thinking?.effortMap?.[effort as NonNullable] ?? effort, }); + // Catalog pro aliases (`gpt-5.6-*-pro`): merge AFTER the compat policy so the + // mode survives every policy branch (disabled/omitted effort included) while + // keeping whatever effort/summary the policy produced — mode and effort are + // independent wire fields. + if (model.reasoningMode) { + params.reasoning = { ...params.reasoning, mode: model.reasoningMode }; + } applyOpenAIGatewayRouting(params, model.compat); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index bece2a3cd..937e3ed7f 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,14 @@ ## [Unreleased] +### Added + +- Added `gpt-5.6` base models and `gpt-5.6-{luna,sol,terra}-pro` variants +- Added `meta/muse-spark-1.1` model support +- Added support for thinking modes on `poolside/laguna` models + +- Added generated GPT-5.6 Pro aliases (`gpt-5.6-{luna,sol,terra}-pro`) on the `openai` and `openai-codex` providers: each alias sends the base model id on the wire (`requestModelId`) with the new `reasoningMode: "pro"` marker, and re-derives from the current base rows on every catalog regeneration. + ## [16.3.14] - 2026-07-09 ### Added diff --git a/packages/catalog/scripts/generate-models.ts b/packages/catalog/scripts/generate-models.ts index 09b9548a9..1d9df8b78 100644 --- a/packages/catalog/scripts/generate-models.ts +++ b/packages/catalog/scripts/generate-models.ts @@ -38,6 +38,7 @@ import { isKimiK27CodeModelId, MODELS_DEV_PROVIDER_DESCRIPTORS, mapModelsDevToModels, + projectOpenAIProReasoningAliases, SAKANA_FUGU_STATIC_MODELS, stripFireworksDeepSeekThinkingToggle, } from "../src/provider-models/openai-compat"; @@ -585,6 +586,10 @@ async function generateModels() { const name = cleanModelName(model.name); return name === model.name ? model : { ...model, name }; }); + // Re-derive the first-party gpt-5.6 pro-reasoning aliases from the current + // base rows (stale previous-snapshot aliases are dropped inside), before the + // policy re-bake so the aliases get the same baked thinking metadata. + allModels = projectOpenAIProReasoningAliases(allModels); applyGeneratedModelPolicies(allModels); linkOpenAIPromotionTargets(allModels); // Collapse effort-tier variants AFTER the policy re-bake: live-discovery diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 55d87779c..67ed8c8e0 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -32288,7 +32288,7 @@ "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -32299,7 +32299,17 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768 + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "poolside/laguna-m.1:free": { "id": "poolside/laguna-m.1:free", @@ -32374,7 +32384,7 @@ "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -32385,7 +32395,17 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768 + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "poolside/laguna-xs.2:free": { "id": "poolside/laguna-xs.2:free", @@ -45679,6 +45699,25 @@ "contextWindow": 328000, "maxTokens": 65536 }, + "meta/muse-spark-1.1": { + "id": "meta/muse-spark-1.1", + "name": "meta/muse-spark-1.1", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 1048576 + }, "microsoft/MAI-DS-R1-FP8": { "id": "microsoft/MAI-DS-R1-FP8", "name": "microsoft/MAI-DS-R1-FP8", @@ -48497,6 +48536,174 @@ }, "contextPromotionTarget": "nanogpt/openai/gpt-5.4" }, + "openai/gpt-5.6-luna": { + "id": "openai/gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.09999999999999999, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-luna-pro": { + "id": "openai/gpt-5.6-luna-pro", + "name": "GPT-5.6 Luna Pro", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-sol": { + "id": "openai/gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-sol-pro": { + "id": "openai/gpt-5.6-sol-pro", + "name": "GPT-5.6 Sol Pro", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-terra": { + "id": "openai/gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "openai/gpt-5.6-terra-pro": { + "id": "openai/gpt-5.6-terra-pro", + "name": "GPT-5.6 Terra Pro", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, "openai/gpt-chat-latest": { "id": "openai/gpt-chat-latest", "name": "GPT Chat Latest", @@ -49105,41 +49312,61 @@ }, "poolside/laguna-m.1": { "id": "poolside/laguna-m.1", - "name": "poolside/laguna-m.1", + "name": "Laguna M.1", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], "cost": { - "input": 0, - "output": 0, + "input": 0.2, + "output": 0.4, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768 + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "poolside/laguna-xs.2": { "id": "poolside/laguna-xs.2", - "name": "poolside/laguna-xs.2", + "name": "Laguna XS.2", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], "cost": { - "input": 0, - "output": 0, + "input": 0.2, + "output": 0.4, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768 + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "qvq-max": { "id": "qvq-max", @@ -60096,6 +60323,44 @@ }, "contextPromotionTarget": "openai/gpt-5.4" }, + "gpt-5.6": { + "id": "gpt-5.6", + "name": "GPT-5.6", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "gpt-5.6-luna": { "id": "gpt-5.6-luna", "name": "GPT-5.6 Luna", @@ -60134,6 +60399,46 @@ } } }, + "gpt-5.6-luna-pro": { + "id": "gpt-5.6-luna-pro", + "name": "GPT-5.6 Luna Pro", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 1.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "requestModelId": "gpt-5.6-luna", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", "name": "GPT-5.6 Sol", @@ -60172,6 +60477,46 @@ } } }, + "gpt-5.6-sol-pro": { + "id": "gpt-5.6-sol-pro", + "name": "GPT-5.6 Sol Pro", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "requestModelId": "gpt-5.6-sol", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", "name": "GPT-5.6 Terra", @@ -60210,6 +60555,46 @@ } } }, + "gpt-5.6-terra-pro": { + "id": "gpt-5.6-terra-pro", + "name": "GPT-5.6 Terra Pro", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "requestModelId": "gpt-5.6-terra", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "o1": { "id": "o1", "name": "o1", @@ -61034,6 +61419,53 @@ } } }, + "gpt-5.6-luna-pro": { + "id": "gpt-5.6-luna-pro", + "name": "GPT-5.6 Luna Pro", + "api": "openai-codex-responses", + "provider": "openai-codex", + "baseUrl": "https://chatgpt.com/backend-api", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 1.25 + }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, + "contextWindow": 372000, + "maxTokens": 128000, + "preferWebsockets": true, + "priority": 3, + "requestModelId": "gpt-5.6-luna", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", "name": "GPT-5.6 Sol", @@ -61079,6 +61511,53 @@ } } }, + "gpt-5.6-sol-pro": { + "id": "gpt-5.6-sol-pro", + "name": "GPT-5.6 Sol Pro", + "api": "openai-codex-responses", + "provider": "openai-codex", + "baseUrl": "https://chatgpt.com/backend-api", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, + "contextWindow": 372000, + "maxTokens": 128000, + "preferWebsockets": true, + "priority": 1, + "requestModelId": "gpt-5.6-sol", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", "name": "GPT-5.6 Terra", @@ -61123,6 +61602,53 @@ "xhigh": "max" } } + }, + "gpt-5.6-terra-pro": { + "id": "gpt-5.6-terra-pro", + "name": "GPT-5.6 Terra Pro", + "api": "openai-codex-responses", + "provider": "openai-codex", + "baseUrl": "https://chatgpt.com/backend-api", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, + "contextWindow": 372000, + "maxTokens": 128000, + "preferWebsockets": true, + "priority": 2, + "requestModelId": "gpt-5.6-terra", + "reasoningMode": "pro", + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } } }, "opencode": { @@ -64155,7 +64681,7 @@ "input": 5, "output": 30, "cacheRead": 0.5, - "cacheWrite": 0 + "cacheWrite": 6.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -67983,8 +68509,8 @@ ], "cost": { "input": 0.72, - "output": 3.5, - "cacheRead": 0.15, + "output": 3.49, + "cacheRead": 0.159, "cacheWrite": 0 }, "contextWindow": 262144, @@ -69547,7 +70073,7 @@ "input": 1, "output": 6, "cacheRead": 0.09999999999999999, - "cacheWrite": 0 + "cacheWrite": 1.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -69584,7 +70110,7 @@ "input": 1, "output": 6, "cacheRead": 0.09999999999999999, - "cacheWrite": 0 + "cacheWrite": 1.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -69621,7 +70147,7 @@ "input": 5, "output": 30, "cacheRead": 0.5, - "cacheWrite": 0 + "cacheWrite": 6.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -69658,7 +70184,7 @@ "input": 5, "output": 30, "cacheRead": 0.5, - "cacheWrite": 0 + "cacheWrite": 6.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -69695,7 +70221,7 @@ "input": 2.5, "output": 15, "cacheRead": 0.25, - "cacheWrite": 0 + "cacheWrite": 3.125 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -69732,7 +70258,7 @@ "input": 2.5, "output": 15, "cacheRead": 0.25, - "cacheWrite": 0 + "cacheWrite": 3.125 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -77243,6 +77769,138 @@ ] } }, + "openai-gpt-56-luna": { + "id": "openai-gpt-56-luna", + "name": "openai-gpt-56-luna", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "compat": { + "supportsUsageInStreaming": false + } + }, + "openai-gpt-56-luna-pro": { + "id": "openai-gpt-56-luna-pro", + "name": "openai-gpt-56-luna-pro", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "compat": { + "supportsUsageInStreaming": false + } + }, + "openai-gpt-56-sol": { + "id": "openai-gpt-56-sol", + "name": "openai-gpt-56-sol", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "compat": { + "supportsUsageInStreaming": false + } + }, + "openai-gpt-56-sol-pro": { + "id": "openai-gpt-56-sol-pro", + "name": "openai-gpt-56-sol-pro", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "compat": { + "supportsUsageInStreaming": false + } + }, + "openai-gpt-56-terra": { + "id": "openai-gpt-56-terra", + "name": "openai-gpt-56-terra", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "compat": { + "supportsUsageInStreaming": false + } + }, + "openai-gpt-56-terra-pro": { + "id": "openai-gpt-56-terra-pro", + "name": "openai-gpt-56-terra-pro", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "compat": { + "supportsUsageInStreaming": false + } + }, "openai-gpt-oss-120b": { "id": "openai-gpt-oss-120b", "name": "OpenAI GPT OSS 120B", @@ -80352,6 +81010,35 @@ "contextWindow": 128000, "maxTokens": 8192 }, + "meta/muse-spark-1.1": { + "id": "meta/muse-spark-1.1", + "name": "Muse Spark 1.1", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.25, + "output": 4.25, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 1048576, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "minimax/minimax-m2": { "id": "minimax/minimax-m2", "name": "MiniMax M2", diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 83cd58136..1877fa7d2 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -821,6 +821,62 @@ export function openaiModelManagerOptions(config?: OpenAIModelManagerConfig): Mo }; } +/** First-party gpt-5.6 SKUs that accept `reasoning: { mode: "pro" }` on the Responses APIs. */ +const OPENAI_PRO_REASONING_BASE_IDS: Record = { + "gpt-5.6-luna": true, + "gpt-5.6-sol": true, + "gpt-5.6-terra": true, +}; +const OPENAI_PRO_REASONING_PROVIDERS: Record = { openai: true, "openai-codex": true }; + +/** + * A row this generator pass owns: one of the derived `gpt-5.6-*-pro` alias ids + * on `openai`/`openai-codex` that carries the generated `reasoningMode` marker. + * A real upstream model occupying the same id has no `reasoningMode` and is + * never touched. + */ +function isGeneratedOpenAIProReasoningAlias(model: ModelSpec): boolean { + return ( + OPENAI_PRO_REASONING_PROVIDERS[model.provider] === true && + model.reasoningMode !== undefined && + model.id.endsWith("-pro") && + OPENAI_PRO_REASONING_BASE_IDS[model.id.slice(0, -"-pro".length)] === true + ); +} + +/** + * Re-derive the generated pro-reasoning aliases (`gpt-5.6-*-pro`) for the + * first-party `openai`/`openai-codex` gpt-5.6 rows. Each alias inherits the + * base row's metadata, requests the base wire id via `requestModelId`, and + * sets `reasoningMode: "pro"` so Responses-family request builders emit + * `reasoning: { mode: "pro" }`. Called by the models.json generator after all + * sources merge: stale copies of the owned aliases (previous snapshot) are + * dropped and re-projected from the current base rows so alias metadata always + * tracks the base, while a real upstream model that occupies an alias id wins + * and suppresses the projection. + */ +export function projectOpenAIProReasoningAliases(models: readonly ModelSpec[]): ModelSpec[] { + const kept = models.filter(model => !isGeneratedOpenAIProReasoningAlias(model)); + const ids = new Set(kept.map(model => `${model.provider}/${model.id}`)); + const out = [...kept]; + for (const model of kept) { + if (!OPENAI_PRO_REASONING_PROVIDERS[model.provider]) continue; + if (!OPENAI_PRO_REASONING_BASE_IDS[model.id]) continue; + const aliasId = `${model.id}-pro`; + const aliasKey = `${model.provider}/${aliasId}`; + if (ids.has(aliasKey)) continue; + ids.add(aliasKey); + out.push({ + ...model, + id: aliasId, + name: `${model.name} Pro`, + requestModelId: model.id, + reasoningMode: "pro", + }); + } + return out; +} + // --------------------------------------------------------------------------- // 2. Groq // --------------------------------------------------------------------------- diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts index 103549d0f..c0f4d7620 100644 --- a/packages/catalog/src/types.ts +++ b/packages/catalog/src/types.ts @@ -691,6 +691,13 @@ export interface Model { * everything local (selection, caching, usage attribution) keys on `id`. */ requestModelId?: string; + /** + * `reasoning.mode` to send on OpenAI Responses-family requests. Set on + * generated pro aliases (`gpt-5.6-*-pro` on `openai`/`openai-codex`) that + * pair a base wire id (`requestModelId`) with OpenAI's pro reasoning + * serving path. Absent everywhere else; providers omit the wire field. + */ + reasoningMode?: "pro"; name: string; api: TApi; provider: Provider;