fix(catalog): expose low reasoning effort for deepseek-v4-pro
DeepSeek's Chat Completions API now advertises reasoning_effort low/high/max for both deepseek-v4-flash and deepseek-v4-pro, but the catalog gated the low tier behind isDeepseekV4FlashModelId, so V4 Pro (and every non-Flash reasoner) fell through to high/max. Broaden the DeepSeek effort ladder to give any V4 SKU the low/high/max scale on the direct API and faithful aggregator routes, keeping OpenRouter's non-Flash route at high-only and the older V3.x/R1 reasoners at high/max. Fixes #8405
This commit is contained in:
@@ -16,6 +16,12 @@
|
||||
- Fixed Anthropic cache-write pricing to correctly honor mixed 5-minute and 1-hour TTL usage instead of incorrectly charging all writes at the 5-minute rate.
|
||||
- Fixed Ollama Cloud DeepSeek V4 Flash and older reasoners to correctly apply the DeepSeek effort contract (e.g., low/high/max) instead of the generic effort ladder.
|
||||
- Added a default request timeout to OpenAI-compatible model discovery to prevent stalled provider endpoints from hanging startup indefinitely.
|
||||
- Fixed Anthropic cache-write pricing to honor mixed 5-minute and 1-hour TTL usage instead of charging every write at the 5-minute rate.
|
||||
- Fixed Ollama Cloud DeepSeek V4 Flash (including dated/preview tags like `deepseek-v4-flash:0731`) exposing the generic `minimal`/`low`/`medium`/`high`/`xhigh` effort ladder without `max`; the `ollama-chat` transport now applies the DeepSeek effort contract (Flash → `low`/`high`/`max`, older reasoners → `high`/`max`), matching the direct API and every other host ([#8334](https://github.com/can1357/oh-my-pi/issues/8334)).
|
||||
- Exposed the `low` reasoning-effort tier for DeepSeek V4 Pro on the direct API and faithful aggregator routes, matching DeepSeek's updated API contract advertising `reasoning_effort` `low`/`high`/`max` for both V4 SKUs; OpenRouter's non-Flash route still exposes only `high`, and the older V3.x/R1 reasoners remain `high`/`max` ([#8405](https://github.com/can1357/oh-my-pi/issues/8405)).
|
||||
- Bounded OpenAI-compatible model discovery with a default request timeout so a stalled provider `/models` endpoint can no longer hang startup indefinitely in `resolveModelDiscoveryFallback` ([#8315](https://github.com/can1357/oh-my-pi/issues/8315)).
|
||||
- Fixed Codex-discovered `gpt-daybreak-*` aliases being treated as unknown models, restoring the GPT-5.6 `low`/`medium`/`high`/`xhigh`/`max` effort ladder and its 372K fallback only when the Codex registry omits `context_window`.
|
||||
- Fixed first-party OpenAI GPT-5.6 aliases to preserve wire-level `off` through generated pro aliases and to price requests above 272K input at each SKU's documented long-context rates.
|
||||
|
||||
## [17.2.15] - 2026-08-12
|
||||
|
||||
|
||||
@@ -85,9 +85,11 @@ export const isDeepseekModelIdOrName = memo((value: string): boolean => {
|
||||
|
||||
/**
|
||||
* DeepSeek V4 Flash SKU in any host/namespace form (`deepseek-v4-flash`, dated
|
||||
* `deepseek-v4-flash-0731`, `deepseek-ai/DeepSeek-V4-Flash`). Flash is the only
|
||||
* V4 model whose `reasoning_effort` accepts the `low` tier; V4 Pro tops out at
|
||||
* `high`/`max`. See https://api-docs.deepseek.com/api/create-chat-completion.
|
||||
* `deepseek-v4-flash-0731`, `deepseek-ai/DeepSeek-V4-Flash`). Both V4 SKUs
|
||||
* (Flash and Pro) accept the `low` reasoning_effort tier; this predicate keeps
|
||||
* Flash distinguishable from Pro where a host quirk splits them (e.g.
|
||||
* OpenRouter exposes `low` on Flash but only `high` on non-Flash V4).
|
||||
* See https://api-docs.deepseek.com/api/create-chat-completion.
|
||||
*/
|
||||
export const isDeepseekV4FlashModelId = memo((modelId: string): boolean => {
|
||||
return bareModelId(modelId).toLowerCase().includes("deepseek-v4-flash");
|
||||
|
||||
@@ -63,9 +63,9 @@ const GEMINI_3_FLASH_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, E
|
||||
const GPT_5_2_PLUS_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh];
|
||||
const GPT_5_1_CODEX_MINI_EFFORTS: readonly Effort[] = [Effort.Medium, Effort.High];
|
||||
const LOW_MEDIUM_HIGH_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High];
|
||||
/** Wire-exact `low`/`high`/`max` scale used by Kimi K3 and DeepSeek V4 Flash (direct API and aggregators). */
|
||||
/** Wire-exact `low`/`high`/`max` scale used by Kimi K3 and DeepSeek V4 (Flash and Pro, direct API and aggregators). */
|
||||
const LOW_HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High, Effort.Max];
|
||||
/** Wire-exact two-tier scale (`high`/`max`): GLM-5.2 on Z.ai/Umans/Ollama Cloud/Baseten, Sakana Fugu, DeepSeek V4 Pro. */
|
||||
/** Wire-exact two-tier scale (`high`/`max`): GLM-5.2 on Z.ai/Umans/Ollama Cloud/Baseten, Sakana Fugu, older DeepSeek reasoners (V3.x/R1). */
|
||||
const HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.Max];
|
||||
/** OpenRouter's DeepSeek route accepts only `high`. */
|
||||
const HIGH_ONLY_REASONING_EFFORTS: readonly Effort[] = [Effort.High];
|
||||
@@ -370,13 +370,18 @@ function getModelDefinedEfforts<TApi extends Api>(
|
||||
(isOpenAICompatReasoningApi(spec.api) || (spec.api === "ollama-chat" && spec.provider === "ollama-cloud")) &&
|
||||
isDeepseekReasoningModel(spec)
|
||||
) {
|
||||
// DeepSeek V4 Flash accepts the wire-exact low/high/max ladder on every
|
||||
// host — the direct API, aggregators, and Ollama Cloud alike (medium/xhigh
|
||||
// map to high). V4 Pro and the older reasoners top out at high/max, and
|
||||
// OpenRouter's non-flash DeepSeek route exposes only high.
|
||||
// DeepSeek V4 (Flash and Pro) accepts the wire-exact low/high/max ladder
|
||||
// on every first-party/aggregator host — the direct API, aggregators, and
|
||||
// Ollama Cloud alike (medium/xhigh fold into high, max is a real wire
|
||||
// tier). See https://api-docs.deepseek.com/api/create-chat-completion.
|
||||
// OpenRouter's non-Flash V4 route still exposes only high; the older
|
||||
// reasoners (V3.x, R1, deepseek-reasoner) top out at high/max.
|
||||
if (isDeepseekV4FlashModelId(spec.id)) {
|
||||
return LOW_HIGH_MAX_REASONING_EFFORTS;
|
||||
}
|
||||
if (bareModelId(spec.id).toLowerCase().includes("deepseek-v4")) {
|
||||
return isOpenRouterThinkingFormat(compat) ? HIGH_ONLY_REASONING_EFFORTS : LOW_HIGH_MAX_REASONING_EFFORTS;
|
||||
}
|
||||
return isOpenRouterThinkingFormat(compat) ? HIGH_ONLY_REASONING_EFFORTS : HIGH_MAX_REASONING_EFFORTS;
|
||||
}
|
||||
if (spec.provider === "baseten" && isOpenAIGptOssModelId(spec.id)) {
|
||||
|
||||
@@ -322,9 +322,10 @@ describe("model thinking derivation", () => {
|
||||
expect(getSupportedEfforts(flash)).toEqual([Effort.Low, Effort.High, Effort.Max]);
|
||||
expect(getSupportedEfforts(flashDated)).toEqual([Effort.Low, Effort.High, Effort.Max]);
|
||||
expect(flash.thinking?.effortMap).toBeUndefined();
|
||||
// V4 Pro and the older reasoners top out at high/max, matching the
|
||||
// direct DeepSeek API and every aggregator route.
|
||||
expect(getSupportedEfforts(pro)).toEqual([Effort.High, Effort.Max]);
|
||||
// V4 Pro shares Flash's low/high/max ladder on the direct API and every
|
||||
// aggregator route (DeepSeek's docs advertise `low` for both V4 SKUs);
|
||||
// the older V3.x reasoners still top out at high/max.
|
||||
expect(getSupportedEfforts(pro)).toEqual([Effort.Low, Effort.High, Effort.Max]);
|
||||
expect(getSupportedEfforts(v32)).toEqual([Effort.High, Effort.Max]);
|
||||
});
|
||||
|
||||
|
||||
Reference in New Issue
Block a user