fix(catalog): cap ollama cloud deepseek-v4 output at 65536
Ollama Cloud's deepseek-v4-pro and deepseek-v4-flash deployments reject any output budget above 65536 with HTTP 400, despite advertising a 1M context / 384K output (ollama/ollama#16890). Ollama's /api/show never reports this cap, so the catalog left the base models at the full context window and the dated tag deepseek-v4-flash:0731 at a stale 8192 fallback. Pin these ids (base plus tag variants) to min(contextWindow, 65536) at both runtime discovery and generation; other cloud models keep their discovered limits. Fixes #7266
This commit is contained in:
@@ -55,6 +55,7 @@ import { JWT_CLAIM_PATH } from "../src/wire/codex";
|
||||
import {
|
||||
applyCanonicalLimitFallback,
|
||||
applyGeneratedModelPolicies,
|
||||
applyOllamaCloudOutputCap,
|
||||
CLOUDFLARE_FALLBACK_MODEL,
|
||||
dropUnsupportedBedrockGeoIds,
|
||||
linkOpenAIPromotionTargets,
|
||||
@@ -678,6 +679,9 @@ async function generateModels() {
|
||||
// Fill remaining null endpoint limits from each model's canonical-family
|
||||
// reference. Runs last so canonical ids and explicit policy limits are final.
|
||||
applyCanonicalLimitFallback(allModels);
|
||||
// Pin every Ollama Cloud model's max-output to the enforced ceiling; runs
|
||||
// after canonical fallback so finalized context windows drive the cap.
|
||||
applyOllamaCloudOutputCap(allModels);
|
||||
|
||||
for (const model of allModels) {
|
||||
canonicalizeModelCompat(model);
|
||||
|
||||
@@ -18,6 +18,7 @@ import { isMimoModelIdOrName } from "../src/identity/family";
|
||||
import { getLongestModelLikeIdSegment } from "../src/identity/id";
|
||||
import { buildModelReferenceIndex, resolveModelReference } from "../src/identity/reference";
|
||||
import { resolveModelThinking } from "../src/model-thinking";
|
||||
import { isOllamaCloudOutputCapped, OLLAMA_CLOUD_MAX_OUTPUT_TOKENS } from "../src/provider-models/ollama";
|
||||
import {
|
||||
ALIBABA_TOKEN_PLAN_STATIC_MODELS,
|
||||
resolveWaferServerlessThinkingFormat,
|
||||
@@ -220,6 +221,26 @@ export function applyCanonicalLimitFallback(models: ModelSpec<Api>[]): void {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Pin the max-output figure for Ollama Cloud models whose deployment enforces a
|
||||
* lower ceiling than their advertised window.
|
||||
*
|
||||
* Ollama's `/api/show` never reports a per-model output cap, so discovery and
|
||||
* previous snapshots leave `maxTokens` at the full context window (or a stale
|
||||
* conservative fallback, as with `deepseek-v4-flash:0731`). DeepSeek V4
|
||||
* Pro/Flash deployments actually reject any output budget above
|
||||
* {@link OLLAMA_CLOUD_MAX_OUTPUT_TOKENS} (ollama/ollama#16890, #3392/#3394), so
|
||||
* pin those ids to `min(contextWindow, ceiling)` — the true amount the endpoint
|
||||
* accepts (#7266). Other cloud models keep their discovered limits.
|
||||
*/
|
||||
export function applyOllamaCloudOutputCap(models: ModelSpec<Api>[]): void {
|
||||
for (const model of models) {
|
||||
if (model.provider !== "ollama-cloud" || model.contextWindow === null) continue;
|
||||
if (!isOllamaCloudOutputCapped(model.id)) continue;
|
||||
model.maxTokens = Math.min(model.contextWindow, OLLAMA_CLOUD_MAX_OUTPUT_TOKENS);
|
||||
}
|
||||
}
|
||||
|
||||
function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
|
||||
const copilotLimits = model.provider === "github-copilot" ? COPILOT_GENERATED_LIMITS[model.id] : undefined;
|
||||
if (copilotLimits) {
|
||||
|
||||
Reference in New Issue
Block a user