fix(ai/ollama): clamped num_predict at the 65536 Ollama Cloud cap

Ollama Cloud rejects any chat request whose options.num_predict exceeds
65536 with HTTP 400, but the existing safety relied on the load-time
omitMaxOutputTokens policy in model-registry.ts. Stale models.db rows
predating that policy (and custom modelOverrides re-enabling output
caps) carried maxTokens: 1048576 forward to the wire layer untouched,
so every request to deepseek-v4-pro / deepseek-v4-flash 400'd with:

  max_tokens (1048576) exceeds model's maximum output tokens (65536)
  for model deepseek-v4-pro

createChatBody now resolves num_predict through resolveNumPredict,
which clamps every ollama-cloud request at the documented cap before
serialization — independent of the cached spec, on top of the existing
omitMaxOutputTokens path. Self-hosted ollama traffic is unaffected.

Fixes #3392
This commit is contained in:
roboomp
2026-06-24 17:29:28 +00:00
parent 958a923a42
commit 0edb3d32b8
3 changed files with 101 additions and 1 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed Ollama Cloud `num_predict` ignoring the provider's 65536 output-token cap so stale `models.db` rows (or custom `modelOverrides` re-enabling output caps) that carried `maxTokens: 1048576` from a pre-omitMaxOutputTokens catalog 400'd every request with `max_tokens (1048576) exceeds model's maximum output tokens (65536) for model deepseek-v4-pro`. The Ollama provider now clamps `num_predict` for any `ollama-cloud` request at the documented 65536 cap before sending, independent of the cached spec's `maxTokens` and on top of the existing `omitMaxOutputTokens` policy — so the request stays valid even when the load-time policy never normalized the spec. Self-hosted `ollama` traffic is unaffected. ([#3392](https://github.com/can1357/oh-my-pi/issues/3392))
## [16.1.16] - 2026-06-23
### Fixed
+21 -1
View File
@@ -282,6 +282,26 @@ function convertTools(tools: Tool[] | undefined): OllamaFunctionTool[] | undefin
}));
}
/**
* Ollama Cloud rejects `num_predict` above this value with HTTP 400
* (`max_tokens (...) exceeds model's maximum output tokens (65536)`).
* The cap currently applies uniformly to cloud-served models; the cloud-side
* limit was confirmed empirically against `deepseek-v4-pro`/`-flash` and is
* the same cap surfaced for every other Ollama Cloud model we've probed.
*
* Acts as a wire-level safety net so stale `models.db` rows (or custom
* `modelOverrides` re-enabling `num_predict`) cannot 400 the request — even
* when `model.omitMaxOutputTokens` was never applied. See #3392.
*/
const OLLAMA_CLOUD_NUM_PREDICT_CAP = 65_536;
function resolveNumPredict(model: Model<"ollama-chat">, requested: number): number {
if (model.provider === "ollama-cloud") {
return Math.min(requested, OLLAMA_CLOUD_NUM_PREDICT_CAP);
}
return requested;
}
function createChatBody(model: Model<"ollama-chat">, context: Context, options: OllamaChatOptions | undefined) {
const think = mapReasoning(model, options?.reasoning, options?.disableReasoning);
const toolChoice = mapToolChoice(options?.toolChoice);
@@ -294,7 +314,7 @@ function createChatBody(model: Model<"ollama-chat">, context: Context, options:
...(think !== undefined ? { think } : {}),
...(toolChoice !== undefined ? { tool_choice: toolChoice } : {}),
...(options?.maxTokens !== undefined && !model.omitMaxOutputTokens
? { options: { num_predict: options.maxTokens } }
? { options: { num_predict: resolveNumPredict(model, options.maxTokens) } }
: {}),
stream: true,
};
@@ -102,6 +102,82 @@ test("ollama-chat omits num_predict when model opts out of max output tokens", a
expect(requestBody).not.toHaveProperty("options");
});
test("ollama-chat clamps num_predict at the Ollama Cloud 65536 output-token cap (#3392)", async () => {
// Stale cached models.db row: maxTokens carried forward from a pre-fix
// catalog (or a user modelOverride re-enabling output caps), no
// `omitMaxOutputTokens` policy applied — so the model spec arrives at the
// wire with `maxTokens: 1048576`. Ollama Cloud rejects anything above
// 65536 with HTTP 400; the wire layer must clamp before sending.
const staleCachedSpec: Model<"ollama-chat"> = {
id: "deepseek-v4-pro",
name: "deepseek-v4-pro",
api: "ollama-chat",
provider: "ollama-cloud",
baseUrl: "https://ollama.com",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 524288,
maxTokens: 1_048_576,
compat: undefined,
};
let requestBody: Record<string, unknown> | undefined;
const fetchMock: FetchImpl = vi.fn(async (_input, init) => {
requestBody = JSON.parse(String(init?.body ?? "{}")) as Record<string, unknown>;
return createNdjsonResponse([
{ model: "deepseek-v4-pro", message: { role: "assistant", content: "ok" }, done: false },
{ model: "deepseek-v4-pro", done: true, done_reason: "stop", prompt_eval_count: 1, eval_count: 1 },
]);
});
await streamSimple(
staleCachedSpec,
{ messages: [{ role: "user", content: "Reply ok", timestamp: Date.now() }] },
{ apiKey: "cloud-test-key", fetch: fetchMock },
).result();
const options = requestBody?.options as { num_predict?: number } | undefined;
expect(options?.num_predict).toBe(65_536);
});
test("ollama-chat does not clamp num_predict for self-hosted ollama (#3392)", async () => {
// Sanity: the clamp is provider-scoped to `ollama-cloud`. A local Ollama
// host carries its own output-token semantics — the wire layer must not
// rewrite `num_predict` for non-cloud `ollama-chat` traffic.
const localSpec: Model<"ollama-chat"> = {
id: "deepseek-r1:70b",
name: "deepseek-r1:70b",
api: "ollama-chat",
provider: "ollama",
baseUrl: "http://127.0.0.1:11434",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 131_072,
maxTokens: 131_072,
compat: undefined,
};
let requestBody: Record<string, unknown> | undefined;
const fetchMock: FetchImpl = vi.fn(async (_input, init) => {
requestBody = JSON.parse(String(init?.body ?? "{}")) as Record<string, unknown>;
return createNdjsonResponse([
{ model: "deepseek-r1:70b", message: { role: "assistant", content: "ok" }, done: false },
{ model: "deepseek-r1:70b", done: true, done_reason: "stop", prompt_eval_count: 1, eval_count: 1 },
]);
});
await streamSimple(
localSpec,
{ messages: [{ role: "user", content: "Reply ok", timestamp: Date.now() }] },
{ apiKey: "local-key", fetch: fetchMock },
).result();
const options = requestBody?.options as { num_predict?: number } | undefined;
expect(options?.num_predict).toBe(131_072);
});
test("ollama-chat sends think false when reasoning is disabled", async () => {
let requestBody: Record<string, unknown> | undefined;
const fetchMock: FetchImpl = vi.fn(async (_input, init) => {