From c00790fa3ab0be90cc7135aa8466933e374128cd Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 1 Aug 2026 13:09:47 +0000 Subject: [PATCH] fix(catalog): cap ollama cloud deepseek-v4 output at 65536 Ollama Cloud's deepseek-v4-pro and deepseek-v4-flash deployments reject any output budget above 65536 with HTTP 400, despite advertising a 1M context / 384K output (ollama/ollama#16890). Ollama's /api/show never reports this cap, so the catalog left the base models at the full context window and the dated tag deepseek-v4-flash:0731 at a stale 8192 fallback. Pin these ids (base plus tag variants) to min(contextWindow, 65536) at both runtime discovery and generation; other cloud models keep their discovered limits. Fixes #7266 --- packages/catalog/CHANGELOG.md | 4 + packages/catalog/scripts/generate-models.ts | 4 + .../catalog/scripts/generated-policies.ts | 21 ++++ packages/catalog/src/models.json | 6 +- .../catalog/src/provider-models/ollama.ts | 43 +++++++-- .../catalog/test/generated-policies.test.ts | 96 ++++++++++++++++++- .../test/ollama-cloud-output-caps.test.ts | 60 +++++++++++- 7 files changed, 221 insertions(+), 13 deletions(-) diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index c329a78e5..4b28cdd68 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Ollama Cloud DeepSeek V4 Pro/Flash models (including dated tag variants such as `deepseek-v4-flash:0731`) reporting an incorrect max-output-tokens figure by pinning it to the deployment's enforced 65536-token output ceiling ([#7266](https://github.com/can1357/oh-my-pi/issues/7266)). + ## [17.2.3] - 2026-08-01 ### Added diff --git a/packages/catalog/scripts/generate-models.ts b/packages/catalog/scripts/generate-models.ts index 8c6f86cc6..2ed40bc68 100644 --- a/packages/catalog/scripts/generate-models.ts +++ b/packages/catalog/scripts/generate-models.ts @@ -55,6 +55,7 @@ import { JWT_CLAIM_PATH } from "../src/wire/codex"; import { applyCanonicalLimitFallback, applyGeneratedModelPolicies, + applyOllamaCloudOutputCap, CLOUDFLARE_FALLBACK_MODEL, dropUnsupportedBedrockGeoIds, linkOpenAIPromotionTargets, @@ -678,6 +679,9 @@ async function generateModels() { // Fill remaining null endpoint limits from each model's canonical-family // reference. Runs last so canonical ids and explicit policy limits are final. applyCanonicalLimitFallback(allModels); + // Pin every Ollama Cloud model's max-output to the enforced ceiling; runs + // after canonical fallback so finalized context windows drive the cap. + applyOllamaCloudOutputCap(allModels); for (const model of allModels) { canonicalizeModelCompat(model); diff --git a/packages/catalog/scripts/generated-policies.ts b/packages/catalog/scripts/generated-policies.ts index 4e3d0c0fb..ad2171c27 100644 --- a/packages/catalog/scripts/generated-policies.ts +++ b/packages/catalog/scripts/generated-policies.ts @@ -18,6 +18,7 @@ import { isMimoModelIdOrName } from "../src/identity/family"; import { getLongestModelLikeIdSegment } from "../src/identity/id"; import { buildModelReferenceIndex, resolveModelReference } from "../src/identity/reference"; import { resolveModelThinking } from "../src/model-thinking"; +import { isOllamaCloudOutputCapped, OLLAMA_CLOUD_MAX_OUTPUT_TOKENS } from "../src/provider-models/ollama"; import { ALIBABA_TOKEN_PLAN_STATIC_MODELS, resolveWaferServerlessThinkingFormat, @@ -220,6 +221,26 @@ export function applyCanonicalLimitFallback(models: ModelSpec[]): void { } } +/** + * Pin the max-output figure for Ollama Cloud models whose deployment enforces a + * lower ceiling than their advertised window. + * + * Ollama's `/api/show` never reports a per-model output cap, so discovery and + * previous snapshots leave `maxTokens` at the full context window (or a stale + * conservative fallback, as with `deepseek-v4-flash:0731`). DeepSeek V4 + * Pro/Flash deployments actually reject any output budget above + * {@link OLLAMA_CLOUD_MAX_OUTPUT_TOKENS} (ollama/ollama#16890, #3392/#3394), so + * pin those ids to `min(contextWindow, ceiling)` — the true amount the endpoint + * accepts (#7266). Other cloud models keep their discovered limits. + */ +export function applyOllamaCloudOutputCap(models: ModelSpec[]): void { + for (const model of models) { + if (model.provider !== "ollama-cloud" || model.contextWindow === null) continue; + if (!isOllamaCloudOutputCapped(model.id)) continue; + model.maxTokens = Math.min(model.contextWindow, OLLAMA_CLOUD_MAX_OUTPUT_TOKENS); + } +} + function applyGeneratedModelPolicy(model: ModelSpec): void { const copilotLimits = model.provider === "github-copilot" ? COPILOT_GENERATED_LIMITS[model.id] : undefined; if (copilotLimits) { diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 9e9f0bb2f..4e50edacd 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -68099,7 +68099,7 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 1048576, + "maxTokens": 65536, "omitMaxOutputTokens": true, "thinking": { "mode": "effort", @@ -68139,7 +68139,7 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 8192, + "maxTokens": 65536, "omitMaxOutputTokens": true, "supportsComputerUse": false }, @@ -68160,7 +68160,7 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 1048576, + "maxTokens": 65536, "omitMaxOutputTokens": true, "thinking": { "mode": "effort", diff --git a/packages/catalog/src/provider-models/ollama.ts b/packages/catalog/src/provider-models/ollama.ts index 2e0598e11..693a8ad4d 100644 --- a/packages/catalog/src/provider-models/ollama.ts +++ b/packages/catalog/src/provider-models/ollama.ts @@ -23,6 +23,36 @@ type OllamaShowResponse = { }; const OLLAMA_RETRY_DELAYS_MS = [2_000, 5_000, 10_000]; +/** + * Output-token ceiling that Ollama Cloud enforces for the DeepSeek V4 Pro/Flash + * deployments: `/api/chat` rejects `num_predict` above it with HTTP 400 + * (`max_tokens (...) exceeds model's maximum output tokens (65536)`) even though + * the model pages advertise a 1M context / 384K output. Ollama's `/api/show` + * never reports this cap, so the catalog pins it for the affected models + * (ollama/ollama#16890, #7266). The wire layer clamps `num_predict` to the same + * value (`OLLAMA_CLOUD_NUM_PREDICT_CAP` in `packages/ai/src/providers/ollama.ts`, + * #3392/#3394). + */ +export const OLLAMA_CLOUD_MAX_OUTPUT_TOKENS = 65_536; + +/** + * Untagged base ids whose Ollama Cloud deployment enforces + * {@link OLLAMA_CLOUD_MAX_OUTPUT_TOKENS}. Only DeepSeek V4 Pro/Flash are known + * to cap output below their advertised window (ollama/ollama#16890); other cloud + * models keep their discovered limits. + */ +const OLLAMA_CLOUD_OUTPUT_CAPPED_BASE_IDS: Record = { + "deepseek-v4-flash": true, + "deepseek-v4-pro": true, +}; + +/** Whether an Ollama Cloud model id (tagged or not) enforces the 65536 output cap. */ +export function isOllamaCloudOutputCapped(id: string): boolean { + const separator = id.indexOf(":"); + const baseId = separator > 0 ? id.slice(0, separator) : id; + return OLLAMA_CLOUD_OUTPUT_CAPPED_BASE_IDS[baseId] === true; +} + const OLLAMA_CLOUD_GLM_52_THINKING: ThinkingConfig = { mode: "effort", efforts: [Effort.High, Effort.Max], @@ -133,10 +163,10 @@ export function ollamaCloudModelManagerOptions( } const capabilities = metadata?.capabilities; const discoveredContextWindow = getContextWindow(metadata?.model_info); - // `/api/show` is the only trustworthy Ollama-owned source for size caps. - // When it is unavailable (or returns only coarse capabilities), do NOT - // inherit giant budgets from bundled fallback metadata sourced from a - // different catalog; keep the historical safe fallback instead. + // `/api/show` reports the context length but never a per-model output + // cap. DeepSeek V4 Pro/Flash deployments enforce a 65536 output ceiling + // (ollama/ollama#16890, #7266); every other id keeps the trusted + // reference limit, falling back to the historical safe cap otherwise. const contextWindow = discoveredContextWindow ?? 128000; const reasoning = capabilities ? capabilities.includes("thinking") : (reference?.reasoning ?? false); const thinking = capabilities ? getThinkingConfig(id, capabilities) : reference?.thinking; @@ -157,8 +187,9 @@ export function ollamaCloudModelManagerOptions( input, cost: reference?.cost ?? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow, - maxTokens: - discoveredContextWindow !== null && discoveredContextWindow !== undefined + maxTokens: isOllamaCloudOutputCapped(id) + ? Math.min(contextWindow, OLLAMA_CLOUD_MAX_OUTPUT_TOKENS) + : discoveredContextWindow !== null && discoveredContextWindow !== undefined ? (providerReference?.maxTokens ?? Math.min(contextWindow, 8192)) : Math.min(contextWindow, 8192), omitMaxOutputTokens: true, diff --git a/packages/catalog/test/generated-policies.test.ts b/packages/catalog/test/generated-policies.test.ts index 54905c0ed..4d2ba9416 100644 --- a/packages/catalog/test/generated-policies.test.ts +++ b/packages/catalog/test/generated-policies.test.ts @@ -1,7 +1,11 @@ import { describe, expect, it } from "bun:test"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; import type { Api, ModelSpec, Provider } from "@oh-my-pi/pi-catalog/types"; -import { applyGeneratedModelPolicies, linkOpenAIPromotionTargets } from "../scripts/generated-policies"; +import { + applyGeneratedModelPolicies, + applyOllamaCloudOutputCap, + linkOpenAIPromotionTargets, +} from "../scripts/generated-policies"; function createSpec(overrides: { id: string; @@ -408,3 +412,93 @@ describe("generated model policies", () => { expect(models[3]?.applyPatchToolType).toBeUndefined(); }); }); + +describe("applyOllamaCloudOutputCap", () => { + it("pins DeepSeek V4 Pro/Flash (and their tag variants) to the enforced ceiling (#7266)", () => { + const models: ModelSpec[] = [ + createSpec({ + id: "deepseek-v4-flash", + api: "ollama-chat", + provider: "ollama-cloud", + contextWindow: 1048576, + maxTokens: 1048576, + }), + createSpec({ + id: "deepseek-v4-flash:0731", + api: "ollama-chat", + provider: "ollama-cloud", + contextWindow: 1048576, + maxTokens: 8192, + }), + createSpec({ + id: "deepseek-v4-pro", + api: "ollama-chat", + provider: "ollama-cloud", + contextWindow: 1048576, + maxTokens: 1048576, + }), + ]; + + applyOllamaCloudOutputCap(models); + + expect(models[0]?.maxTokens).toBe(65536); + expect(models[1]?.maxTokens).toBe(65536); + expect(models[2]?.maxTokens).toBe(65536); + }); + + it("leaves other Ollama Cloud models' discovered limits untouched", () => { + const models: ModelSpec[] = [ + createSpec({ + id: "kimi-k2.5", + api: "ollama-chat", + provider: "ollama-cloud", + contextWindow: 262144, + maxTokens: 262144, + }), + createSpec({ + id: "deepseek-v3.1:671b", + api: "ollama-chat", + provider: "ollama-cloud", + contextWindow: 163840, + maxTokens: 163840, + }), + ]; + + applyOllamaCloudOutputCap(models); + + expect(models[0]?.maxTokens).toBe(262144); + expect(models[1]?.maxTokens).toBe(163840); + }); + + it("caps by the context window when a capped model's window is below the ceiling", () => { + const models: ModelSpec[] = [ + createSpec({ + id: "deepseek-v4-flash", + api: "ollama-chat", + provider: "ollama-cloud", + contextWindow: 32768, + maxTokens: 32768, + }), + ]; + + applyOllamaCloudOutputCap(models); + + expect(models[0]?.maxTokens).toBe(32768); + }); + + it("does not touch other providers", () => { + const models: ModelSpec[] = [ + createSpec({ + id: "deepseek-v4-flash", + api: "openai-completions", + provider: "deepseek", + contextWindow: 1048576, + maxTokens: 1048576, + }), + ]; + + applyOllamaCloudOutputCap(models); + + expect(models[0]?.maxTokens).toBe(1048576); + }); +}); diff --git a/packages/catalog/test/ollama-cloud-output-caps.test.ts b/packages/catalog/test/ollama-cloud-output-caps.test.ts index 4a47ada4f..42703d253 100644 --- a/packages/catalog/test/ollama-cloud-output-caps.test.ts +++ b/packages/catalog/test/ollama-cloud-output-caps.test.ts @@ -26,7 +26,7 @@ test("ollama-cloud discovery does not inherit unsafe cross-provider maxTokens", const fetchMock: FetchImpl = vi.fn(async (input, _init) => { const url = String(input); if (url === "https://ollama.com/api/tags") { - return new Response(JSON.stringify({ models: [{ name: "deepseek-v4-flash" }] }), { + return new Response(JSON.stringify({ models: [{ name: "kimi-k2.5" }] }), { status: 200, headers: { "Content-Type": "application/json" }, }); @@ -42,12 +42,66 @@ test("ollama-cloud discovery does not inherit unsafe cross-provider maxTokens", const options = ollamaCloudModelManagerOptions({ apiKey: "cloud-test-key", fetch: fetchMock }); const models = await options.fetchDynamicModels?.(); - const model = models?.find(candidate => candidate.id === "deepseek-v4-flash"); + const model = models?.find(candidate => candidate.id === "kimi-k2.5"); expect(model?.contextWindow).toBe(128000); expect(model?.maxTokens).toBe(8192); }); +test("ollama-cloud discovery caps discovered max-output at the enforced ceiling (#7266)", async () => { + const fetchMock: FetchImpl = vi.fn(async (input, _init) => { + const url = String(input); + if (url === "https://ollama.com/api/tags") { + return new Response(JSON.stringify({ models: [{ name: "deepseek-v4-flash:0731" }] }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + } + if (url === "https://ollama.com/api/show") { + return new Response( + JSON.stringify({ capabilities: ["completion"], model_info: { "deepseek.context_length": 1048576 } }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + } + throw new Error(`Unexpected URL: ${url}`); + }); + + const options = ollamaCloudModelManagerOptions({ apiKey: "cloud-test-key", fetch: fetchMock }); + const models = await options.fetchDynamicModels?.(); + const model = models?.find(candidate => candidate.id === "deepseek-v4-flash:0731"); + + expect(model?.contextWindow).toBe(1048576); + // Ollama Cloud rejects output budgets above 65536, so the 1M context window + // must not surface as the max-output figure. + expect(model?.maxTokens).toBe(65536); +}); + +test("ollama-cloud discovery caps a capped model by its context window when below the ceiling", async () => { + const fetchMock: FetchImpl = vi.fn(async (input, _init) => { + const url = String(input); + if (url === "https://ollama.com/api/tags") { + return new Response(JSON.stringify({ models: [{ name: "deepseek-v4-flash:mini" }] }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + } + if (url === "https://ollama.com/api/show") { + return new Response( + JSON.stringify({ capabilities: ["completion"], model_info: { "deepseek.context_length": 32768 } }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + } + throw new Error(`Unexpected URL: ${url}`); + }); + + const options = ollamaCloudModelManagerOptions({ apiKey: "cloud-test-key", fetch: fetchMock }); + const models = await options.fetchDynamicModels?.(); + const model = models?.find(candidate => candidate.id === "deepseek-v4-flash:mini"); + + expect(model?.contextWindow).toBe(32768); + expect(model?.maxTokens).toBe(32768); +}); + test("ollama-cloud discovery always omits max output tokens", async () => { const fetchMock: FetchImpl = vi.fn(async (input, _init) => { const url = String(input); @@ -78,7 +132,7 @@ test("ollama-cloud discovery always omits max output tokens", async () => { expect(model?.provider).toBe("ollama-cloud"); expect(model?.contextWindow).toBe(1048576); - expect(model?.maxTokens).toBe(1048576); + expect(model?.maxTokens).toBe(65536); expect(model?.omitMaxOutputTokens).toBe(true); });