diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 4758480c4..ec13caf7f 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -11,6 +11,7 @@ - Fixed `gen:models` Codex discovery to union models across every stored OAuth account and fail closed on partial resolution, matching runtime discovery ([#6265](https://github.com/can1357/oh-my-pi/issues/6265)); restored the bundled `gpt-5.4`, `gpt-5.6-sol`, and `gpt-5.3-codex-spark` entries a single-account regen had dropped. - Fixed `google-antigravity` models always reporting $0 cost: Antigravity discovery carries no pricing, so the generator now back-fills each model with its Google list price (Gemini ids from the `google` provider, including `-preview` id aliases; Claude ids from `google-vertex`, falling back to `anthropic`). +- Fixed OpenRouter `deepseek/deepseek-v4-flash-0731` exposing only `high` thinking effort by consuming the live `reasoning.supported_efforts` and `default_effort` metadata and bundling its `low`/`high`/`max` ladder. ([#7307](https://github.com/can1357/oh-my-pi/issues/7307)) ## [17.2.3] - 2026-08-01 diff --git a/packages/catalog/src/model-thinking.ts b/packages/catalog/src/model-thinking.ts index 1783d1508..2cde43081 100644 --- a/packages/catalog/src/model-thinking.ts +++ b/packages/catalog/src/model-thinking.ts @@ -62,8 +62,8 @@ const GEMINI_3_FLASH_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, E const GPT_5_2_PLUS_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]; const GPT_5_1_CODEX_MINI_EFFORTS: readonly Effort[] = [Effort.Medium, Effort.High]; const LOW_MEDIUM_HIGH_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High]; -/** Kimi K3's wire-exact mandatory reasoning scale. */ -const KIMI_K3_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High, Effort.Max]; +/** Wire-exact `low`/`high`/`max` scale used by Kimi K3 and OpenRouter DeepSeek V4 Flash 0731. */ +const LOW_HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High, Effort.Max]; /** Wire-exact two-tier scale (`high`/`max`): GLM-5.2 on Z.ai/Umans/Ollama Cloud/Baseten, Sakana Fugu, DeepSeek. */ const HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.Max]; /** OpenRouter's DeepSeek route accepts only `high`. */ @@ -339,7 +339,7 @@ function getModelDefinedEfforts( } } if (isKimiK3ModelId(spec.id)) { - return KIMI_K3_REASONING_EFFORTS; + return LOW_HIGH_MAX_REASONING_EFFORTS; } if (isSakanaFuguReasoningModel(spec)) { return HIGH_MAX_REASONING_EFFORTS; @@ -366,9 +366,14 @@ function getModelDefinedEfforts( return OLLAMA_REASONING_EFFORTS; } if (isOpenAICompatReasoningApi(spec.api) && isDeepseekReasoningModel(spec)) { - // DeepSeek's reasoning_effort accepts only high/max; OpenRouter's - // DeepSeek route tops out at high. - return isOpenRouterThinkingFormat(compat) ? HIGH_ONLY_REASONING_EFFORTS : HIGH_MAX_REASONING_EFFORTS; + // OpenRouter generally exposes only high for DeepSeek, but V4 Flash 0731 + // advertises and accepts the wire-exact low/high/max ladder. + if (isOpenRouterThinkingFormat(compat)) { + return bareModelId(spec.id) === "deepseek-v4-flash-0731" + ? LOW_HIGH_MAX_REASONING_EFFORTS + : HIGH_ONLY_REASONING_EFFORTS; + } + return HIGH_MAX_REASONING_EFFORTS; } if (spec.provider === "baseten" && isOpenAIGptOssModelId(spec.id)) { // Baseten's gpt-oss router mirrors its GLM route: high/max only. diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 490c55614..01217365b 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -11768,8 +11768,7 @@ ], "supportsDisplay": true }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false + "supportsComputerUse": false }, "claude-opus-4-0": { "id": "claude-opus-4-0", @@ -76433,7 +76432,9 @@ "thinking": { "mode": "effort", "efforts": [ - "high" + "low", + "high", + "max" ] }, "supportsComputerUse": false, @@ -85668,8 +85669,6 @@ }, "contextWindow": 524288, "maxTokens": 65536, - "supportsComputerUse": false, - "supportsComputerUseConfig": false, "thinking": { "mode": "effort", "efforts": [ @@ -85682,7 +85681,8 @@ "max": "max" }, "requiresEffort": true - } + }, + "supportsComputerUse": false }, "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4": { "id": "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4", @@ -85712,8 +85712,7 @@ "xhigh" ] }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false + "supportsComputerUse": false }, "hf:openai/gpt-oss-120b": { "id": "hf:openai/gpt-oss-120b", @@ -85741,8 +85740,7 @@ "high" ] }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false + "supportsComputerUse": false }, "hf:Qwen/Qwen3.6-27B": { "id": "hf:Qwen/Qwen3.6-27B", @@ -85772,8 +85770,7 @@ "high" ] }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false + "supportsComputerUse": false }, "hf:zai-org/GLM-4.7-Flash": { "id": "hf:zai-org/GLM-4.7-Flash", @@ -85803,8 +85800,7 @@ "xhigh" ] }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false + "supportsComputerUse": false }, "hf:zai-org/GLM-5.2": { "id": "hf:zai-org/GLM-5.2", @@ -85834,8 +85830,7 @@ "xhigh" ] }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false + "supportsComputerUse": false }, "syn:large:text": { "id": "syn:large:text", @@ -98552,7 +98547,6 @@ "contextWindow": 2000000, "maxTokens": 2000000, "supportsComputerUse": false, - "supportsComputerUseConfig": false, "compat": { "reasoningEffortMap": { "minimal": "low" @@ -98584,7 +98578,6 @@ "contextWindow": 2000000, "maxTokens": 2000000, "supportsComputerUse": false, - "supportsComputerUseConfig": false, "compat": { "reasoningEffortMap": { "minimal": "low" @@ -98628,7 +98621,6 @@ } }, "supportsComputerUse": false, - "supportsComputerUseConfig": false, "compat": { "reasoningEffortMap": { "minimal": "low" @@ -98673,7 +98665,6 @@ } }, "supportsComputerUse": false, - "supportsComputerUseConfig": false, "compat": { "reasoningEffortMap": { "minimal": "low" @@ -98718,7 +98709,6 @@ } }, "supportsComputerUse": false, - "supportsComputerUseConfig": false, "compat": { "reasoningEffortMap": { "minimal": "low" @@ -98750,7 +98740,6 @@ "contextWindow": 512000, "maxTokens": 512000, "supportsComputerUse": false, - "supportsComputerUseConfig": false, "compat": { "reasoningEffortMap": { "minimal": "low" @@ -98782,7 +98771,6 @@ "contextWindow": 256000, "maxTokens": 256000, "supportsComputerUse": false, - "supportsComputerUseConfig": false, "compat": { "reasoningEffortMap": { "minimal": "low" @@ -98813,7 +98801,6 @@ "contextWindow": 200000, "maxTokens": 200000, "supportsComputerUse": false, - "supportsComputerUseConfig": false, "compat": { "reasoningEffortMap": { "minimal": "low" diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 837cd1991..4fa7e1354 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -2465,6 +2465,24 @@ export interface OpenRouterModelManagerConfig { fetch?: FetchImpl; } +function mapOpenRouterThinking(entry: OpenAICompatibleModelRecord): ThinkingConfig | undefined { + const reasoning = entry.reasoning; + if (!isRecord(reasoning)) return undefined; + const supportedEfforts = reasoning.supported_efforts; + if (!Array.isArray(supportedEfforts)) return undefined; + const efforts = THINKING_EFFORTS.filter(effort => supportedEfforts.includes(effort)); + if (efforts.length === 0) return undefined; + const defaultLevel = + typeof reasoning.default_effort === "string" + ? THINKING_EFFORTS.find(effort => effort === reasoning.default_effort) + : undefined; + return { + mode: "effort", + efforts, + ...(defaultLevel !== undefined && efforts.includes(defaultLevel) ? { defaultLevel } : {}), + }; +} + export function openrouterModelManagerOptions( config?: OpenRouterModelManagerConfig, ): ModelManagerOptions<"openrouter"> { @@ -2496,6 +2514,7 @@ export function openrouterModelManagerOptions( const baseModel = mapWithBundledReference(entry, defaults, reference); const pricing = entry.pricing as Record | undefined; const params = Array.isArray(entry.supported_parameters) ? (entry.supported_parameters as string[]) : []; + const thinking = mapOpenRouterThinking(entry); const modality = String((entry.architecture as Record | undefined)?.modality ?? ""); const topProvider = entry.top_provider as Record | undefined; @@ -2504,6 +2523,7 @@ export function openrouterModelManagerOptions( return { ...baseModel, reasoning: params.includes("reasoning"), + ...(thinking !== undefined ? { thinking } : {}), input: modality.includes("image") ? ["text", "image"] : ["text"], cost: { input: parseFloat(String(pricing?.prompt ?? "0")) * 1_000_000, diff --git a/packages/catalog/test/build.test.ts b/packages/catalog/test/build.test.ts index 386bb23b6..4c617cf58 100644 --- a/packages/catalog/test/build.test.ts +++ b/packages/catalog/test/build.test.ts @@ -6,6 +6,7 @@ import * as path from "node:path"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { isOfficialAnthropicApiUrl } from "@oh-my-pi/pi-catalog/compat/anthropic"; import { buildOpenAICompat, buildOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { readModelCache, writeModelCache } from "@oh-my-pi/pi-catalog/model-cache"; import { resolveProviderModels } from "@oh-my-pi/pi-catalog/model-manager"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; @@ -610,6 +611,34 @@ describe("OpenRouter model discovery", () => { } }); + it("maps OpenRouter's advertised reasoning effort ladder and default", async () => { + const options = openrouterModelManagerOptions({ + fetch: async () => + Response.json({ + data: [ + { + id: "deepseek/deepseek-v4-flash-0731", + name: "DeepSeek V4 Flash 0731", + supported_parameters: ["tools", "reasoning", "reasoning_effort"], + reasoning: { + supported_efforts: ["max", "high", "low"], + default_effort: "high", + }, + }, + ], + }), + }); + const specs = await options.fetchDynamicModels?.(); + const spec = specs?.find(model => model.id === "deepseek/deepseek-v4-flash-0731"); + if (!spec) throw new Error("Expected discovered DeepSeek V4 Flash 0731 model"); + + expect(buildModel(spec).thinking).toEqual({ + mode: "effort", + efforts: [Effort.Low, Effort.High, Effort.Max], + defaultLevel: Effort.High, + }); + }); + it("ignores legacy OpenRouter chat-completions cache rows", async () => { const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-openrouter-legacy-cache-")); const dbPath = path.join(tempDir, "models.db"); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index a245b2b55..76a68b2ea 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -36,6 +36,9 @@ - Preserved explicit `-e`/`--extension` and `--hook` packages under `--no-extensions` while excluding ambient extension factories and sibling capabilities from settings or installed OMP packages. +### Fixed + +- Fixed explicit `thinking` metadata in `models.yml` custom definitions and `modelOverrides` being replaced by canonical catalog policy during model rebuilding. ([#7307](https://github.com/can1357/oh-my-pi/issues/7307)) ## [17.2.3] - 2026-08-01 diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index 5c5024988..99f65e8de 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -580,7 +580,13 @@ function applyModelPatch(base: Model, patch: ModelPatch, transport: ModelTr result.headers = patch.headers; compat = patch.compat; } - return buildModel({ ...result, compat } as ModelSpec); + const built = buildModel({ ...result, compat } as ModelSpec); + if (patch.thinking !== undefined && built.thinking !== undefined) { + // Config-authored capability metadata owns the explicit surface; build + // first so non-reasoning and wire-disabled models still suppress it. + built.thinking = patch.thinking; + } + return built; } function applyModelOverride(model: Model, override: ModelOverride): Model { diff --git a/packages/coding-agent/test/model-registry.test.ts b/packages/coding-agent/test/model-registry.test.ts index 6e3694f9b..87f1f7247 100644 --- a/packages/coding-agent/test/model-registry.test.ts +++ b/packages/coding-agent/test/model-registry.test.ts @@ -1174,6 +1174,7 @@ describe("ModelRegistry", () => { }; let thinkingCustom: ModelRegistry; let thinkingOverride: ModelRegistry; + let deepseekOverride: ModelRegistry; beforeAll(() => { thinkingCustom = readonlyRegistry({ providers: { @@ -1193,6 +1194,21 @@ describe("ModelRegistry", () => { }, }, }); + deepseekOverride = readonlyRegistry({ + providers: { + openrouter: { + modelOverrides: { + "deepseek/deepseek-v4-flash-0731": { + thinking: { + mode: "effort", + efforts: [Effort.Max, Effort.High, Effort.Low], + defaultLevel: Effort.High, + }, + }, + }, + }, + }, + }); }); test("custom models preserve explicit thinking verbatim", () => { @@ -1211,6 +1227,17 @@ describe("ModelRegistry", () => { efforts: [Effort.Low, Effort.Medium], }); }); + + test("model overrides preserve explicit OpenRouter DeepSeek thinking metadata", () => { + const model = getModelsForProvider(deepseekOverride, "openrouter").find( + m => m.id === "deepseek/deepseek-v4-flash-0731", + ); + expect(model?.thinking).toEqual({ + mode: "effort", + efforts: [Effort.Max, Effort.High, Effort.Low], + defaultLevel: Effort.High, + }); + }); }); describe("modelOverrides (per-model customization)", () => {