Merge PR #8852: fix(catalog): add deepseek-v4-pro-0813 discovery limits (@tommyldev)
This commit is contained in:
@@ -5,6 +5,7 @@
|
||||
### Fixed
|
||||
|
||||
- Fixed local Qwen 3.8+ models (llama.cpp, vLLM, loopback custom providers) exposing the generic `minimal..high` thinking ladder instead of the chat template's real `low`/`medium`/`xhigh` `reasoning_effort` tiers. The derived metadata now marks thinking as mandatory (the official 3.8 template raises on `enable_thinking: false`), vLLM-served Qwen routes through the `chat_template_kwargs` dialect (top-level `enable_thinking` is ignored by vLLM), and vLLM discovery lights up the reasoning dial for Qwen 3.8+ ids its `/v1/models` endpoint reports as non-reasoning.
|
||||
- Fixed `deepseek-v4-pro-0813` surfacing from Alibaba Token Plan discovery with `contextWindow`/`maxTokens` of `null`. The dated DeepSeek V4 Pro snapshot was missing from `ALIBABA_TOKEN_PLAN_DISCOVERED_MODEL_LIMITS`, so unlike its `deepseek-v4-flash-0731` sibling it fell through to unknown limits ([#8847](https://github.com/can1357/oh-my-pi/issues/8847)).
|
||||
|
||||
## [17.3.6] - 2026-08-17
|
||||
|
||||
|
||||
@@ -2981,6 +2981,10 @@ export const ALIBABA_TOKEN_PLAN_DISCOVERED_MODEL_LIMITS: Readonly<Record<string,
|
||||
contextWindow: 1_000_000,
|
||||
maxTokens: 384_000,
|
||||
},
|
||||
"deepseek-v4-pro-0813": {
|
||||
contextWindow: 1_000_000,
|
||||
maxTokens: 384_000,
|
||||
},
|
||||
"deepseek-v3.2": {
|
||||
contextWindow: 131_072,
|
||||
maxTokens: 65_536,
|
||||
|
||||
@@ -75,6 +75,7 @@ describe("QwenCloud Token Plan provider", () => {
|
||||
},
|
||||
{ id: "deepseek-v4-flash", owned_by: "qwencloud" },
|
||||
{ id: "deepseek-v4-flash-0731", owned_by: "qwencloud" },
|
||||
{ id: "deepseek-v4-pro-0813", owned_by: "qwencloud" },
|
||||
{ id: "kimi-k2.7-code", owned_by: "qwencloud" },
|
||||
{ id: "MiniMax-M2.5", owned_by: "qwencloud" },
|
||||
{ id: "qwen3.6-plus", owned_by: "qwencloud" },
|
||||
@@ -106,6 +107,7 @@ describe("QwenCloud Token Plan provider", () => {
|
||||
"deepseek-v3.2",
|
||||
"deepseek-v4-flash",
|
||||
"deepseek-v4-flash-0731",
|
||||
"deepseek-v4-pro-0813",
|
||||
"future-chat-model",
|
||||
"glm-5",
|
||||
"glm-5.1",
|
||||
@@ -122,6 +124,7 @@ describe("QwenCloud Token Plan provider", () => {
|
||||
["qwen3.8-max", 1_000_000, 131_072],
|
||||
["deepseek-v4-flash", 1_000_000, 384_000],
|
||||
["deepseek-v4-flash-0731", 1_000_000, 384_000],
|
||||
["deepseek-v4-pro-0813", 1_000_000, 384_000],
|
||||
["deepseek-v3.2", 131_072, 65_536],
|
||||
["glm-5.1", 202_752, 128_000],
|
||||
["glm-5", 202_752, 16_384],
|
||||
@@ -133,7 +136,7 @@ describe("QwenCloud Token Plan provider", () => {
|
||||
for (const [id, contextWindow, maxTokens] of expectedLimits) {
|
||||
expect(models?.find(model => model.id === id)).toMatchObject({ contextWindow, maxTokens });
|
||||
}
|
||||
for (const id of ["deepseek-v4-flash", "deepseek-v4-flash-0731"]) {
|
||||
for (const id of ["deepseek-v4-flash", "deepseek-v4-flash-0731", "deepseek-v4-pro-0813"]) {
|
||||
expect(models?.find(model => model.id === id)).toMatchObject({
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
|
||||
Reference in New Issue
Block a user