diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index e40f2fe88..ea3051c54 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -16,6 +16,12 @@ - Fixed Anthropic cache-write pricing to correctly honor mixed 5-minute and 1-hour TTL usage instead of incorrectly charging all writes at the 5-minute rate. - Fixed Ollama Cloud DeepSeek V4 Flash and older reasoners to correctly apply the DeepSeek effort contract (e.g., low/high/max) instead of the generic effort ladder. - Added a default request timeout to OpenAI-compatible model discovery to prevent stalled provider endpoints from hanging startup indefinitely. +- Fixed Anthropic cache-write pricing to honor mixed 5-minute and 1-hour TTL usage instead of charging every write at the 5-minute rate. +- Fixed Ollama Cloud DeepSeek V4 Flash (including dated/preview tags like `deepseek-v4-flash:0731`) exposing the generic `minimal`/`low`/`medium`/`high`/`xhigh` effort ladder without `max`; the `ollama-chat` transport now applies the DeepSeek effort contract (Flash → `low`/`high`/`max`, older reasoners → `high`/`max`), matching the direct API and every other host ([#8334](https://github.com/can1357/oh-my-pi/issues/8334)). +- Exposed the `low` reasoning-effort tier for DeepSeek V4 Pro on the direct API and faithful aggregator routes, matching DeepSeek's updated API contract advertising `reasoning_effort` `low`/`high`/`max` for both V4 SKUs; OpenRouter's non-Flash route still exposes only `high`, and the older V3.x/R1 reasoners remain `high`/`max` ([#8405](https://github.com/can1357/oh-my-pi/issues/8405)). +- Bounded OpenAI-compatible model discovery with a default request timeout so a stalled provider `/models` endpoint can no longer hang startup indefinitely in `resolveModelDiscoveryFallback` ([#8315](https://github.com/can1357/oh-my-pi/issues/8315)). +- Fixed Codex-discovered `gpt-daybreak-*` aliases being treated as unknown models, restoring the GPT-5.6 `low`/`medium`/`high`/`xhigh`/`max` effort ladder and its 372K fallback only when the Codex registry omits `context_window`. +- Fixed first-party OpenAI GPT-5.6 aliases to preserve wire-level `off` through generated pro aliases and to price requests above 272K input at each SKU's documented long-context rates. ## [17.2.15] - 2026-08-12 diff --git a/packages/catalog/src/identity/family.ts b/packages/catalog/src/identity/family.ts index 59e0ae26e..289f80821 100644 --- a/packages/catalog/src/identity/family.ts +++ b/packages/catalog/src/identity/family.ts @@ -85,9 +85,11 @@ export const isDeepseekModelIdOrName = memo((value: string): boolean => { /** * DeepSeek V4 Flash SKU in any host/namespace form (`deepseek-v4-flash`, dated - * `deepseek-v4-flash-0731`, `deepseek-ai/DeepSeek-V4-Flash`). Flash is the only - * V4 model whose `reasoning_effort` accepts the `low` tier; V4 Pro tops out at - * `high`/`max`. See https://api-docs.deepseek.com/api/create-chat-completion. + * `deepseek-v4-flash-0731`, `deepseek-ai/DeepSeek-V4-Flash`). Both V4 SKUs + * (Flash and Pro) accept the `low` reasoning_effort tier; this predicate keeps + * Flash distinguishable from Pro where a host quirk splits them (e.g. + * OpenRouter exposes `low` on Flash but only `high` on non-Flash V4). + * See https://api-docs.deepseek.com/api/create-chat-completion. */ export const isDeepseekV4FlashModelId = memo((modelId: string): boolean => { return bareModelId(modelId).toLowerCase().includes("deepseek-v4-flash"); diff --git a/packages/catalog/src/model-thinking.ts b/packages/catalog/src/model-thinking.ts index e2299955c..f83237574 100644 --- a/packages/catalog/src/model-thinking.ts +++ b/packages/catalog/src/model-thinking.ts @@ -63,9 +63,9 @@ const GEMINI_3_FLASH_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, E const GPT_5_2_PLUS_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]; const GPT_5_1_CODEX_MINI_EFFORTS: readonly Effort[] = [Effort.Medium, Effort.High]; const LOW_MEDIUM_HIGH_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High]; -/** Wire-exact `low`/`high`/`max` scale used by Kimi K3 and DeepSeek V4 Flash (direct API and aggregators). */ +/** Wire-exact `low`/`high`/`max` scale used by Kimi K3 and DeepSeek V4 (Flash and Pro, direct API and aggregators). */ const LOW_HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High, Effort.Max]; -/** Wire-exact two-tier scale (`high`/`max`): GLM-5.2 on Z.ai/Umans/Ollama Cloud/Baseten, Sakana Fugu, DeepSeek V4 Pro. */ +/** Wire-exact two-tier scale (`high`/`max`): GLM-5.2 on Z.ai/Umans/Ollama Cloud/Baseten, Sakana Fugu, older DeepSeek reasoners (V3.x/R1). */ const HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.Max]; /** OpenRouter's DeepSeek route accepts only `high`. */ const HIGH_ONLY_REASONING_EFFORTS: readonly Effort[] = [Effort.High]; @@ -370,13 +370,18 @@ function getModelDefinedEfforts( (isOpenAICompatReasoningApi(spec.api) || (spec.api === "ollama-chat" && spec.provider === "ollama-cloud")) && isDeepseekReasoningModel(spec) ) { - // DeepSeek V4 Flash accepts the wire-exact low/high/max ladder on every - // host — the direct API, aggregators, and Ollama Cloud alike (medium/xhigh - // map to high). V4 Pro and the older reasoners top out at high/max, and - // OpenRouter's non-flash DeepSeek route exposes only high. + // DeepSeek V4 (Flash and Pro) accepts the wire-exact low/high/max ladder + // on every first-party/aggregator host — the direct API, aggregators, and + // Ollama Cloud alike (medium/xhigh fold into high, max is a real wire + // tier). See https://api-docs.deepseek.com/api/create-chat-completion. + // OpenRouter's non-Flash V4 route still exposes only high; the older + // reasoners (V3.x, R1, deepseek-reasoner) top out at high/max. if (isDeepseekV4FlashModelId(spec.id)) { return LOW_HIGH_MAX_REASONING_EFFORTS; } + if (bareModelId(spec.id).toLowerCase().includes("deepseek-v4")) { + return isOpenRouterThinkingFormat(compat) ? HIGH_ONLY_REASONING_EFFORTS : LOW_HIGH_MAX_REASONING_EFFORTS; + } return isOpenRouterThinkingFormat(compat) ? HIGH_ONLY_REASONING_EFFORTS : HIGH_MAX_REASONING_EFFORTS; } if (spec.provider === "baseten" && isOpenAIGptOssModelId(spec.id)) { diff --git a/packages/catalog/test/model-thinking.test.ts b/packages/catalog/test/model-thinking.test.ts index 035c51346..926ee93a9 100644 --- a/packages/catalog/test/model-thinking.test.ts +++ b/packages/catalog/test/model-thinking.test.ts @@ -322,9 +322,10 @@ describe("model thinking derivation", () => { expect(getSupportedEfforts(flash)).toEqual([Effort.Low, Effort.High, Effort.Max]); expect(getSupportedEfforts(flashDated)).toEqual([Effort.Low, Effort.High, Effort.Max]); expect(flash.thinking?.effortMap).toBeUndefined(); - // V4 Pro and the older reasoners top out at high/max, matching the - // direct DeepSeek API and every aggregator route. - expect(getSupportedEfforts(pro)).toEqual([Effort.High, Effort.Max]); + // V4 Pro shares Flash's low/high/max ladder on the direct API and every + // aggregator route (DeepSeek's docs advertise `low` for both V4 SKUs); + // the older V3.x reasoners still top out at high/max. + expect(getSupportedEfforts(pro)).toEqual([Effort.Low, Effort.High, Effort.Max]); expect(getSupportedEfforts(v32)).toEqual([Effort.High, Effort.Max]); });