Merge PR #7311: fix(catalog): honor openrouter deepseek effort metadata (@roboomp)
This commit is contained in:
@@ -11,6 +11,7 @@
|
||||
|
||||
- Fixed `gen:models` Codex discovery to union models across every stored OAuth account and fail closed on partial resolution, matching runtime discovery ([#6265](https://github.com/can1357/oh-my-pi/issues/6265)); restored the bundled `gpt-5.4`, `gpt-5.6-sol`, and `gpt-5.3-codex-spark` entries a single-account regen had dropped.
|
||||
- Fixed `google-antigravity` models always reporting $0 cost: Antigravity discovery carries no pricing, so the generator now back-fills each model with its Google list price (Gemini ids from the `google` provider, including `-preview` id aliases; Claude ids from `google-vertex`, falling back to `anthropic`).
|
||||
- Fixed OpenRouter `deepseek/deepseek-v4-flash-0731` exposing only `high` thinking effort by consuming the live `reasoning.supported_efforts` and `default_effort` metadata and bundling its `low`/`high`/`max` ladder. ([#7307](https://github.com/can1357/oh-my-pi/issues/7307))
|
||||
|
||||
## [17.2.3] - 2026-08-01
|
||||
|
||||
|
||||
@@ -62,8 +62,8 @@ const GEMINI_3_FLASH_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, E
|
||||
const GPT_5_2_PLUS_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh];
|
||||
const GPT_5_1_CODEX_MINI_EFFORTS: readonly Effort[] = [Effort.Medium, Effort.High];
|
||||
const LOW_MEDIUM_HIGH_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High];
|
||||
/** Kimi K3's wire-exact mandatory reasoning scale. */
|
||||
const KIMI_K3_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High, Effort.Max];
|
||||
/** Wire-exact `low`/`high`/`max` scale used by Kimi K3 and OpenRouter DeepSeek V4 Flash 0731. */
|
||||
const LOW_HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High, Effort.Max];
|
||||
/** Wire-exact two-tier scale (`high`/`max`): GLM-5.2 on Z.ai/Umans/Ollama Cloud/Baseten, Sakana Fugu, DeepSeek. */
|
||||
const HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.Max];
|
||||
/** OpenRouter's DeepSeek route accepts only `high`. */
|
||||
@@ -339,7 +339,7 @@ function getModelDefinedEfforts<TApi extends Api>(
|
||||
}
|
||||
}
|
||||
if (isKimiK3ModelId(spec.id)) {
|
||||
return KIMI_K3_REASONING_EFFORTS;
|
||||
return LOW_HIGH_MAX_REASONING_EFFORTS;
|
||||
}
|
||||
if (isSakanaFuguReasoningModel(spec)) {
|
||||
return HIGH_MAX_REASONING_EFFORTS;
|
||||
@@ -366,9 +366,14 @@ function getModelDefinedEfforts<TApi extends Api>(
|
||||
return OLLAMA_REASONING_EFFORTS;
|
||||
}
|
||||
if (isOpenAICompatReasoningApi(spec.api) && isDeepseekReasoningModel(spec)) {
|
||||
// DeepSeek's reasoning_effort accepts only high/max; OpenRouter's
|
||||
// DeepSeek route tops out at high.
|
||||
return isOpenRouterThinkingFormat(compat) ? HIGH_ONLY_REASONING_EFFORTS : HIGH_MAX_REASONING_EFFORTS;
|
||||
// OpenRouter generally exposes only high for DeepSeek, but V4 Flash 0731
|
||||
// advertises and accepts the wire-exact low/high/max ladder.
|
||||
if (isOpenRouterThinkingFormat(compat)) {
|
||||
return bareModelId(spec.id) === "deepseek-v4-flash-0731"
|
||||
? LOW_HIGH_MAX_REASONING_EFFORTS
|
||||
: HIGH_ONLY_REASONING_EFFORTS;
|
||||
}
|
||||
return HIGH_MAX_REASONING_EFFORTS;
|
||||
}
|
||||
if (spec.provider === "baseten" && isOpenAIGptOssModelId(spec.id)) {
|
||||
// Baseten's gpt-oss router mirrors its GLM route: high/max only.
|
||||
|
||||
@@ -11768,8 +11768,7 @@
|
||||
],
|
||||
"supportsDisplay": true
|
||||
},
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false
|
||||
"supportsComputerUse": false
|
||||
},
|
||||
"claude-opus-4-0": {
|
||||
"id": "claude-opus-4-0",
|
||||
@@ -76433,7 +76432,9 @@
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"high"
|
||||
"low",
|
||||
"high",
|
||||
"max"
|
||||
]
|
||||
},
|
||||
"supportsComputerUse": false,
|
||||
@@ -85668,8 +85669,6 @@
|
||||
},
|
||||
"contextWindow": 524288,
|
||||
"maxTokens": 65536,
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
@@ -85682,7 +85681,8 @@
|
||||
"max": "max"
|
||||
},
|
||||
"requiresEffort": true
|
||||
}
|
||||
},
|
||||
"supportsComputerUse": false
|
||||
},
|
||||
"hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4": {
|
||||
"id": "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4",
|
||||
@@ -85712,8 +85712,7 @@
|
||||
"xhigh"
|
||||
]
|
||||
},
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false
|
||||
"supportsComputerUse": false
|
||||
},
|
||||
"hf:openai/gpt-oss-120b": {
|
||||
"id": "hf:openai/gpt-oss-120b",
|
||||
@@ -85741,8 +85740,7 @@
|
||||
"high"
|
||||
]
|
||||
},
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false
|
||||
"supportsComputerUse": false
|
||||
},
|
||||
"hf:Qwen/Qwen3.6-27B": {
|
||||
"id": "hf:Qwen/Qwen3.6-27B",
|
||||
@@ -85772,8 +85770,7 @@
|
||||
"high"
|
||||
]
|
||||
},
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false
|
||||
"supportsComputerUse": false
|
||||
},
|
||||
"hf:zai-org/GLM-4.7-Flash": {
|
||||
"id": "hf:zai-org/GLM-4.7-Flash",
|
||||
@@ -85803,8 +85800,7 @@
|
||||
"xhigh"
|
||||
]
|
||||
},
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false
|
||||
"supportsComputerUse": false
|
||||
},
|
||||
"hf:zai-org/GLM-5.2": {
|
||||
"id": "hf:zai-org/GLM-5.2",
|
||||
@@ -85834,8 +85830,7 @@
|
||||
"xhigh"
|
||||
]
|
||||
},
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false
|
||||
"supportsComputerUse": false
|
||||
},
|
||||
"syn:large:text": {
|
||||
"id": "syn:large:text",
|
||||
@@ -98552,7 +98547,6 @@
|
||||
"contextWindow": 2000000,
|
||||
"maxTokens": 2000000,
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
@@ -98584,7 +98578,6 @@
|
||||
"contextWindow": 2000000,
|
||||
"maxTokens": 2000000,
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
@@ -98628,7 +98621,6 @@
|
||||
}
|
||||
},
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
@@ -98673,7 +98665,6 @@
|
||||
}
|
||||
},
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
@@ -98718,7 +98709,6 @@
|
||||
}
|
||||
},
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
@@ -98750,7 +98740,6 @@
|
||||
"contextWindow": 512000,
|
||||
"maxTokens": 512000,
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
@@ -98782,7 +98771,6 @@
|
||||
"contextWindow": 256000,
|
||||
"maxTokens": 256000,
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
@@ -98813,7 +98801,6 @@
|
||||
"contextWindow": 200000,
|
||||
"maxTokens": 200000,
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false,
|
||||
"compat": {
|
||||
"reasoningEffortMap": {
|
||||
"minimal": "low"
|
||||
|
||||
@@ -2465,6 +2465,24 @@ export interface OpenRouterModelManagerConfig {
|
||||
fetch?: FetchImpl;
|
||||
}
|
||||
|
||||
function mapOpenRouterThinking(entry: OpenAICompatibleModelRecord): ThinkingConfig | undefined {
|
||||
const reasoning = entry.reasoning;
|
||||
if (!isRecord(reasoning)) return undefined;
|
||||
const supportedEfforts = reasoning.supported_efforts;
|
||||
if (!Array.isArray(supportedEfforts)) return undefined;
|
||||
const efforts = THINKING_EFFORTS.filter(effort => supportedEfforts.includes(effort));
|
||||
if (efforts.length === 0) return undefined;
|
||||
const defaultLevel =
|
||||
typeof reasoning.default_effort === "string"
|
||||
? THINKING_EFFORTS.find(effort => effort === reasoning.default_effort)
|
||||
: undefined;
|
||||
return {
|
||||
mode: "effort",
|
||||
efforts,
|
||||
...(defaultLevel !== undefined && efforts.includes(defaultLevel) ? { defaultLevel } : {}),
|
||||
};
|
||||
}
|
||||
|
||||
export function openrouterModelManagerOptions(
|
||||
config?: OpenRouterModelManagerConfig,
|
||||
): ModelManagerOptions<"openrouter"> {
|
||||
@@ -2496,6 +2514,7 @@ export function openrouterModelManagerOptions(
|
||||
const baseModel = mapWithBundledReference(entry, defaults, reference);
|
||||
const pricing = entry.pricing as Record<string, unknown> | undefined;
|
||||
const params = Array.isArray(entry.supported_parameters) ? (entry.supported_parameters as string[]) : [];
|
||||
const thinking = mapOpenRouterThinking(entry);
|
||||
const modality = String((entry.architecture as Record<string, unknown> | undefined)?.modality ?? "");
|
||||
const topProvider = entry.top_provider as Record<string, unknown> | undefined;
|
||||
|
||||
@@ -2504,6 +2523,7 @@ export function openrouterModelManagerOptions(
|
||||
return {
|
||||
...baseModel,
|
||||
reasoning: params.includes("reasoning"),
|
||||
...(thinking !== undefined ? { thinking } : {}),
|
||||
input: modality.includes("image") ? ["text", "image"] : ["text"],
|
||||
cost: {
|
||||
input: parseFloat(String(pricing?.prompt ?? "0")) * 1_000_000,
|
||||
|
||||
@@ -6,6 +6,7 @@ import * as path from "node:path";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { isOfficialAnthropicApiUrl } from "@oh-my-pi/pi-catalog/compat/anthropic";
|
||||
import { buildOpenAICompat, buildOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
|
||||
import { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import { readModelCache, writeModelCache } from "@oh-my-pi/pi-catalog/model-cache";
|
||||
import { resolveProviderModels } from "@oh-my-pi/pi-catalog/model-manager";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
@@ -610,6 +611,34 @@ describe("OpenRouter model discovery", () => {
|
||||
}
|
||||
});
|
||||
|
||||
it("maps OpenRouter's advertised reasoning effort ladder and default", async () => {
|
||||
const options = openrouterModelManagerOptions({
|
||||
fetch: async () =>
|
||||
Response.json({
|
||||
data: [
|
||||
{
|
||||
id: "deepseek/deepseek-v4-flash-0731",
|
||||
name: "DeepSeek V4 Flash 0731",
|
||||
supported_parameters: ["tools", "reasoning", "reasoning_effort"],
|
||||
reasoning: {
|
||||
supported_efforts: ["max", "high", "low"],
|
||||
default_effort: "high",
|
||||
},
|
||||
},
|
||||
],
|
||||
}),
|
||||
});
|
||||
const specs = await options.fetchDynamicModels?.();
|
||||
const spec = specs?.find(model => model.id === "deepseek/deepseek-v4-flash-0731");
|
||||
if (!spec) throw new Error("Expected discovered DeepSeek V4 Flash 0731 model");
|
||||
|
||||
expect(buildModel(spec).thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Low, Effort.High, Effort.Max],
|
||||
defaultLevel: Effort.High,
|
||||
});
|
||||
});
|
||||
|
||||
it("ignores legacy OpenRouter chat-completions cache rows", async () => {
|
||||
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-openrouter-legacy-cache-"));
|
||||
const dbPath = path.join(tempDir, "models.db");
|
||||
|
||||
@@ -36,6 +36,9 @@
|
||||
- Preserved explicit `-e`/`--extension` and `--hook` packages under
|
||||
`--no-extensions` while excluding ambient extension factories and sibling
|
||||
capabilities from settings or installed OMP packages.
|
||||
### Fixed
|
||||
|
||||
- Fixed explicit `thinking` metadata in `models.yml` custom definitions and `modelOverrides` being replaced by canonical catalog policy during model rebuilding. ([#7307](https://github.com/can1357/oh-my-pi/issues/7307))
|
||||
|
||||
## [17.2.3] - 2026-08-01
|
||||
|
||||
|
||||
@@ -580,7 +580,13 @@ function applyModelPatch(base: Model<Api>, patch: ModelPatch, transport: ModelTr
|
||||
result.headers = patch.headers;
|
||||
compat = patch.compat;
|
||||
}
|
||||
return buildModel({ ...result, compat } as ModelSpec<Api>);
|
||||
const built = buildModel({ ...result, compat } as ModelSpec<Api>);
|
||||
if (patch.thinking !== undefined && built.thinking !== undefined) {
|
||||
// Config-authored capability metadata owns the explicit surface; build
|
||||
// first so non-reasoning and wire-disabled models still suppress it.
|
||||
built.thinking = patch.thinking;
|
||||
}
|
||||
return built;
|
||||
}
|
||||
|
||||
function applyModelOverride(model: Model<Api>, override: ModelOverride): Model<Api> {
|
||||
|
||||
@@ -1174,6 +1174,7 @@ describe("ModelRegistry", () => {
|
||||
};
|
||||
let thinkingCustom: ModelRegistry;
|
||||
let thinkingOverride: ModelRegistry;
|
||||
let deepseekOverride: ModelRegistry;
|
||||
beforeAll(() => {
|
||||
thinkingCustom = readonlyRegistry({
|
||||
providers: {
|
||||
@@ -1193,6 +1194,21 @@ describe("ModelRegistry", () => {
|
||||
},
|
||||
},
|
||||
});
|
||||
deepseekOverride = readonlyRegistry({
|
||||
providers: {
|
||||
openrouter: {
|
||||
modelOverrides: {
|
||||
"deepseek/deepseek-v4-flash-0731": {
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
efforts: [Effort.Max, Effort.High, Effort.Low],
|
||||
defaultLevel: Effort.High,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
test("custom models preserve explicit thinking verbatim", () => {
|
||||
@@ -1211,6 +1227,17 @@ describe("ModelRegistry", () => {
|
||||
efforts: [Effort.Low, Effort.Medium],
|
||||
});
|
||||
});
|
||||
|
||||
test("model overrides preserve explicit OpenRouter DeepSeek thinking metadata", () => {
|
||||
const model = getModelsForProvider(deepseekOverride, "openrouter").find(
|
||||
m => m.id === "deepseek/deepseek-v4-flash-0731",
|
||||
);
|
||||
expect(model?.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Max, Effort.High, Effort.Low],
|
||||
defaultLevel: Effort.High,
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("modelOverrides (per-model customization)", () => {
|
||||
|
||||
Reference in New Issue
Block a user