fix(catalog): cap ollama cloud deepseek-v4 output at 65536

Ollama Cloud's deepseek-v4-pro and deepseek-v4-flash deployments reject any
output budget above 65536 with HTTP 400, despite advertising a 1M context /
384K output (ollama/ollama#16890). Ollama's /api/show never reports this cap,
so the catalog left the base models at the full context window and the dated
tag deepseek-v4-flash:0731 at a stale 8192 fallback. Pin these ids (base plus
tag variants) to min(contextWindow, 65536) at both runtime discovery and
generation; other cloud models keep their discovered limits.

Fixes #7266
This commit is contained in:
roboomp
2026-08-01 13:09:47 +00:00
parent 8baa3300bc
commit c00790fa3a
7 changed files with 221 additions and 13 deletions
@@ -55,6 +55,7 @@ import { JWT_CLAIM_PATH } from "../src/wire/codex";
import {
applyCanonicalLimitFallback,
applyGeneratedModelPolicies,
applyOllamaCloudOutputCap,
CLOUDFLARE_FALLBACK_MODEL,
dropUnsupportedBedrockGeoIds,
linkOpenAIPromotionTargets,
@@ -678,6 +679,9 @@ async function generateModels() {
// Fill remaining null endpoint limits from each model's canonical-family
// reference. Runs last so canonical ids and explicit policy limits are final.
applyCanonicalLimitFallback(allModels);
// Pin every Ollama Cloud model's max-output to the enforced ceiling; runs
// after canonical fallback so finalized context windows drive the cap.
applyOllamaCloudOutputCap(allModels);
for (const model of allModels) {
canonicalizeModelCompat(model);
@@ -18,6 +18,7 @@ import { isMimoModelIdOrName } from "../src/identity/family";
import { getLongestModelLikeIdSegment } from "../src/identity/id";
import { buildModelReferenceIndex, resolveModelReference } from "../src/identity/reference";
import { resolveModelThinking } from "../src/model-thinking";
import { isOllamaCloudOutputCapped, OLLAMA_CLOUD_MAX_OUTPUT_TOKENS } from "../src/provider-models/ollama";
import {
ALIBABA_TOKEN_PLAN_STATIC_MODELS,
resolveWaferServerlessThinkingFormat,
@@ -220,6 +221,26 @@ export function applyCanonicalLimitFallback(models: ModelSpec<Api>[]): void {
}
}
/**
* Pin the max-output figure for Ollama Cloud models whose deployment enforces a
* lower ceiling than their advertised window.
*
* Ollama's `/api/show` never reports a per-model output cap, so discovery and
* previous snapshots leave `maxTokens` at the full context window (or a stale
* conservative fallback, as with `deepseek-v4-flash:0731`). DeepSeek V4
* Pro/Flash deployments actually reject any output budget above
* {@link OLLAMA_CLOUD_MAX_OUTPUT_TOKENS} (ollama/ollama#16890, #3392/#3394), so
* pin those ids to `min(contextWindow, ceiling)` — the true amount the endpoint
* accepts (#7266). Other cloud models keep their discovered limits.
*/
export function applyOllamaCloudOutputCap(models: ModelSpec<Api>[]): void {
for (const model of models) {
if (model.provider !== "ollama-cloud" || model.contextWindow === null) continue;
if (!isOllamaCloudOutputCapped(model.id)) continue;
model.maxTokens = Math.min(model.contextWindow, OLLAMA_CLOUD_MAX_OUTPUT_TOKENS);
}
}
function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
const copilotLimits = model.provider === "github-copilot" ? COPILOT_GENERATED_LIMITS[model.id] : undefined;
if (copilotLimits) {