Merge PR #7311: fix(catalog): honor openrouter deepseek effort metadata (@roboomp)

This commit is contained in:
can1357
2026-08-01 21:29:35 +02:00
8 changed files with 109 additions and 31 deletions
+1
View File
@@ -11,6 +11,7 @@
- Fixed `gen:models` Codex discovery to union models across every stored OAuth account and fail closed on partial resolution, matching runtime discovery ([#6265](https://github.com/can1357/oh-my-pi/issues/6265)); restored the bundled `gpt-5.4`, `gpt-5.6-sol`, and `gpt-5.3-codex-spark` entries a single-account regen had dropped.
- Fixed `google-antigravity` models always reporting $0 cost: Antigravity discovery carries no pricing, so the generator now back-fills each model with its Google list price (Gemini ids from the `google` provider, including `-preview` id aliases; Claude ids from `google-vertex`, falling back to `anthropic`).
- Fixed OpenRouter `deepseek/deepseek-v4-flash-0731` exposing only `high` thinking effort by consuming the live `reasoning.supported_efforts` and `default_effort` metadata and bundling its `low`/`high`/`max` ladder. ([#7307](https://github.com/can1357/oh-my-pi/issues/7307))
## [17.2.3] - 2026-08-01
+11 -6
View File
@@ -62,8 +62,8 @@ const GEMINI_3_FLASH_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, E
const GPT_5_2_PLUS_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh];
const GPT_5_1_CODEX_MINI_EFFORTS: readonly Effort[] = [Effort.Medium, Effort.High];
const LOW_MEDIUM_HIGH_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High];
/** Kimi K3's wire-exact mandatory reasoning scale. */
const KIMI_K3_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High, Effort.Max];
/** Wire-exact `low`/`high`/`max` scale used by Kimi K3 and OpenRouter DeepSeek V4 Flash 0731. */
const LOW_HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High, Effort.Max];
/** Wire-exact two-tier scale (`high`/`max`): GLM-5.2 on Z.ai/Umans/Ollama Cloud/Baseten, Sakana Fugu, DeepSeek. */
const HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.Max];
/** OpenRouter's DeepSeek route accepts only `high`. */
@@ -339,7 +339,7 @@ function getModelDefinedEfforts<TApi extends Api>(
}
}
if (isKimiK3ModelId(spec.id)) {
return KIMI_K3_REASONING_EFFORTS;
return LOW_HIGH_MAX_REASONING_EFFORTS;
}
if (isSakanaFuguReasoningModel(spec)) {
return HIGH_MAX_REASONING_EFFORTS;
@@ -366,9 +366,14 @@ function getModelDefinedEfforts<TApi extends Api>(
return OLLAMA_REASONING_EFFORTS;
}
if (isOpenAICompatReasoningApi(spec.api) && isDeepseekReasoningModel(spec)) {
// DeepSeek's reasoning_effort accepts only high/max; OpenRouter's
// DeepSeek route tops out at high.
return isOpenRouterThinkingFormat(compat) ? HIGH_ONLY_REASONING_EFFORTS : HIGH_MAX_REASONING_EFFORTS;
// OpenRouter generally exposes only high for DeepSeek, but V4 Flash 0731
// advertises and accepts the wire-exact low/high/max ladder.
if (isOpenRouterThinkingFormat(compat)) {
return bareModelId(spec.id) === "deepseek-v4-flash-0731"
? LOW_HIGH_MAX_REASONING_EFFORTS
: HIGH_ONLY_REASONING_EFFORTS;
}
return HIGH_MAX_REASONING_EFFORTS;
}
if (spec.provider === "baseten" && isOpenAIGptOssModelId(spec.id)) {
// Baseten's gpt-oss router mirrors its GLM route: high/max only.
+11 -24
View File
@@ -11768,8 +11768,7 @@
],
"supportsDisplay": true
},
"supportsComputerUse": false,
"supportsComputerUseConfig": false
"supportsComputerUse": false
},
"claude-opus-4-0": {
"id": "claude-opus-4-0",
@@ -76433,7 +76432,9 @@
"thinking": {
"mode": "effort",
"efforts": [
"high"
"low",
"high",
"max"
]
},
"supportsComputerUse": false,
@@ -85668,8 +85669,6 @@
},
"contextWindow": 524288,
"maxTokens": 65536,
"supportsComputerUse": false,
"supportsComputerUseConfig": false,
"thinking": {
"mode": "effort",
"efforts": [
@@ -85682,7 +85681,8 @@
"max": "max"
},
"requiresEffort": true
}
},
"supportsComputerUse": false
},
"hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4": {
"id": "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4",
@@ -85712,8 +85712,7 @@
"xhigh"
]
},
"supportsComputerUse": false,
"supportsComputerUseConfig": false
"supportsComputerUse": false
},
"hf:openai/gpt-oss-120b": {
"id": "hf:openai/gpt-oss-120b",
@@ -85741,8 +85740,7 @@
"high"
]
},
"supportsComputerUse": false,
"supportsComputerUseConfig": false
"supportsComputerUse": false
},
"hf:Qwen/Qwen3.6-27B": {
"id": "hf:Qwen/Qwen3.6-27B",
@@ -85772,8 +85770,7 @@
"high"
]
},
"supportsComputerUse": false,
"supportsComputerUseConfig": false
"supportsComputerUse": false
},
"hf:zai-org/GLM-4.7-Flash": {
"id": "hf:zai-org/GLM-4.7-Flash",
@@ -85803,8 +85800,7 @@
"xhigh"
]
},
"supportsComputerUse": false,
"supportsComputerUseConfig": false
"supportsComputerUse": false
},
"hf:zai-org/GLM-5.2": {
"id": "hf:zai-org/GLM-5.2",
@@ -85834,8 +85830,7 @@
"xhigh"
]
},
"supportsComputerUse": false,
"supportsComputerUseConfig": false
"supportsComputerUse": false
},
"syn:large:text": {
"id": "syn:large:text",
@@ -98552,7 +98547,6 @@
"contextWindow": 2000000,
"maxTokens": 2000000,
"supportsComputerUse": false,
"supportsComputerUseConfig": false,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
@@ -98584,7 +98578,6 @@
"contextWindow": 2000000,
"maxTokens": 2000000,
"supportsComputerUse": false,
"supportsComputerUseConfig": false,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
@@ -98628,7 +98621,6 @@
}
},
"supportsComputerUse": false,
"supportsComputerUseConfig": false,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
@@ -98673,7 +98665,6 @@
}
},
"supportsComputerUse": false,
"supportsComputerUseConfig": false,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
@@ -98718,7 +98709,6 @@
}
},
"supportsComputerUse": false,
"supportsComputerUseConfig": false,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
@@ -98750,7 +98740,6 @@
"contextWindow": 512000,
"maxTokens": 512000,
"supportsComputerUse": false,
"supportsComputerUseConfig": false,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
@@ -98782,7 +98771,6 @@
"contextWindow": 256000,
"maxTokens": 256000,
"supportsComputerUse": false,
"supportsComputerUseConfig": false,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
@@ -98813,7 +98801,6 @@
"contextWindow": 200000,
"maxTokens": 200000,
"supportsComputerUse": false,
"supportsComputerUseConfig": false,
"compat": {
"reasoningEffortMap": {
"minimal": "low"
@@ -2465,6 +2465,24 @@ export interface OpenRouterModelManagerConfig {
fetch?: FetchImpl;
}
function mapOpenRouterThinking(entry: OpenAICompatibleModelRecord): ThinkingConfig | undefined {
const reasoning = entry.reasoning;
if (!isRecord(reasoning)) return undefined;
const supportedEfforts = reasoning.supported_efforts;
if (!Array.isArray(supportedEfforts)) return undefined;
const efforts = THINKING_EFFORTS.filter(effort => supportedEfforts.includes(effort));
if (efforts.length === 0) return undefined;
const defaultLevel =
typeof reasoning.default_effort === "string"
? THINKING_EFFORTS.find(effort => effort === reasoning.default_effort)
: undefined;
return {
mode: "effort",
efforts,
...(defaultLevel !== undefined && efforts.includes(defaultLevel) ? { defaultLevel } : {}),
};
}
export function openrouterModelManagerOptions(
config?: OpenRouterModelManagerConfig,
): ModelManagerOptions<"openrouter"> {
@@ -2496,6 +2514,7 @@ export function openrouterModelManagerOptions(
const baseModel = mapWithBundledReference(entry, defaults, reference);
const pricing = entry.pricing as Record<string, unknown> | undefined;
const params = Array.isArray(entry.supported_parameters) ? (entry.supported_parameters as string[]) : [];
const thinking = mapOpenRouterThinking(entry);
const modality = String((entry.architecture as Record<string, unknown> | undefined)?.modality ?? "");
const topProvider = entry.top_provider as Record<string, unknown> | undefined;
@@ -2504,6 +2523,7 @@ export function openrouterModelManagerOptions(
return {
...baseModel,
reasoning: params.includes("reasoning"),
...(thinking !== undefined ? { thinking } : {}),
input: modality.includes("image") ? ["text", "image"] : ["text"],
cost: {
input: parseFloat(String(pricing?.prompt ?? "0")) * 1_000_000,
+29
View File
@@ -6,6 +6,7 @@ import * as path from "node:path";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { isOfficialAnthropicApiUrl } from "@oh-my-pi/pi-catalog/compat/anthropic";
import { buildOpenAICompat, buildOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
import { Effort } from "@oh-my-pi/pi-catalog/effort";
import { readModelCache, writeModelCache } from "@oh-my-pi/pi-catalog/model-cache";
import { resolveProviderModels } from "@oh-my-pi/pi-catalog/model-manager";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
@@ -610,6 +611,34 @@ describe("OpenRouter model discovery", () => {
}
});
it("maps OpenRouter's advertised reasoning effort ladder and default", async () => {
const options = openrouterModelManagerOptions({
fetch: async () =>
Response.json({
data: [
{
id: "deepseek/deepseek-v4-flash-0731",
name: "DeepSeek V4 Flash 0731",
supported_parameters: ["tools", "reasoning", "reasoning_effort"],
reasoning: {
supported_efforts: ["max", "high", "low"],
default_effort: "high",
},
},
],
}),
});
const specs = await options.fetchDynamicModels?.();
const spec = specs?.find(model => model.id === "deepseek/deepseek-v4-flash-0731");
if (!spec) throw new Error("Expected discovered DeepSeek V4 Flash 0731 model");
expect(buildModel(spec).thinking).toEqual({
mode: "effort",
efforts: [Effort.Low, Effort.High, Effort.Max],
defaultLevel: Effort.High,
});
});
it("ignores legacy OpenRouter chat-completions cache rows", async () => {
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-openrouter-legacy-cache-"));
const dbPath = path.join(tempDir, "models.db");
+3
View File
@@ -36,6 +36,9 @@
- Preserved explicit `-e`/`--extension` and `--hook` packages under
`--no-extensions` while excluding ambient extension factories and sibling
capabilities from settings or installed OMP packages.
### Fixed
- Fixed explicit `thinking` metadata in `models.yml` custom definitions and `modelOverrides` being replaced by canonical catalog policy during model rebuilding. ([#7307](https://github.com/can1357/oh-my-pi/issues/7307))
## [17.2.3] - 2026-08-01
@@ -580,7 +580,13 @@ function applyModelPatch(base: Model<Api>, patch: ModelPatch, transport: ModelTr
result.headers = patch.headers;
compat = patch.compat;
}
return buildModel({ ...result, compat } as ModelSpec<Api>);
const built = buildModel({ ...result, compat } as ModelSpec<Api>);
if (patch.thinking !== undefined && built.thinking !== undefined) {
// Config-authored capability metadata owns the explicit surface; build
// first so non-reasoning and wire-disabled models still suppress it.
built.thinking = patch.thinking;
}
return built;
}
function applyModelOverride(model: Model<Api>, override: ModelOverride): Model<Api> {
@@ -1174,6 +1174,7 @@ describe("ModelRegistry", () => {
};
let thinkingCustom: ModelRegistry;
let thinkingOverride: ModelRegistry;
let deepseekOverride: ModelRegistry;
beforeAll(() => {
thinkingCustom = readonlyRegistry({
providers: {
@@ -1193,6 +1194,21 @@ describe("ModelRegistry", () => {
},
},
});
deepseekOverride = readonlyRegistry({
providers: {
openrouter: {
modelOverrides: {
"deepseek/deepseek-v4-flash-0731": {
thinking: {
mode: "effort",
efforts: [Effort.Max, Effort.High, Effort.Low],
defaultLevel: Effort.High,
},
},
},
},
},
});
});
test("custom models preserve explicit thinking verbatim", () => {
@@ -1211,6 +1227,17 @@ describe("ModelRegistry", () => {
efforts: [Effort.Low, Effort.Medium],
});
});
test("model overrides preserve explicit OpenRouter DeepSeek thinking metadata", () => {
const model = getModelsForProvider(deepseekOverride, "openrouter").find(
m => m.id === "deepseek/deepseek-v4-flash-0731",
);
expect(model?.thinking).toEqual({
mode: "effort",
efforts: [Effort.Max, Effort.High, Effort.Low],
defaultLevel: Effort.High,
});
});
});
describe("modelOverrides (per-model customization)", () => {