fix(ai/ollama): clamped num_predict at the 65536 Ollama Cloud cap
Ollama Cloud rejects any chat request whose options.num_predict exceeds 65536 with HTTP 400, but the existing safety relied on the load-time omitMaxOutputTokens policy in model-registry.ts. Stale models.db rows predating that policy (and custom modelOverrides re-enabling output caps) carried maxTokens: 1048576 forward to the wire layer untouched, so every request to deepseek-v4-pro / deepseek-v4-flash 400'd with: max_tokens (1048576) exceeds model's maximum output tokens (65536) for model deepseek-v4-pro createChatBody now resolves num_predict through resolveNumPredict, which clamps every ollama-cloud request at the documented cap before serialization — independent of the cached spec, on top of the existing omitMaxOutputTokens path. Self-hosted ollama traffic is unaffected. Fixes #3392
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Ollama Cloud `num_predict` ignoring the provider's 65536 output-token cap so stale `models.db` rows (or custom `modelOverrides` re-enabling output caps) that carried `maxTokens: 1048576` from a pre-omitMaxOutputTokens catalog 400'd every request with `max_tokens (1048576) exceeds model's maximum output tokens (65536) for model deepseek-v4-pro`. The Ollama provider now clamps `num_predict` for any `ollama-cloud` request at the documented 65536 cap before sending, independent of the cached spec's `maxTokens` and on top of the existing `omitMaxOutputTokens` policy — so the request stays valid even when the load-time policy never normalized the spec. Self-hosted `ollama` traffic is unaffected. ([#3392](https://github.com/can1357/oh-my-pi/issues/3392))
|
||||
|
||||
## [16.1.16] - 2026-06-23
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -282,6 +282,26 @@ function convertTools(tools: Tool[] | undefined): OllamaFunctionTool[] | undefin
|
||||
}));
|
||||
}
|
||||
|
||||
/**
|
||||
* Ollama Cloud rejects `num_predict` above this value with HTTP 400
|
||||
* (`max_tokens (...) exceeds model's maximum output tokens (65536)`).
|
||||
* The cap currently applies uniformly to cloud-served models; the cloud-side
|
||||
* limit was confirmed empirically against `deepseek-v4-pro`/`-flash` and is
|
||||
* the same cap surfaced for every other Ollama Cloud model we've probed.
|
||||
*
|
||||
* Acts as a wire-level safety net so stale `models.db` rows (or custom
|
||||
* `modelOverrides` re-enabling `num_predict`) cannot 400 the request — even
|
||||
* when `model.omitMaxOutputTokens` was never applied. See #3392.
|
||||
*/
|
||||
const OLLAMA_CLOUD_NUM_PREDICT_CAP = 65_536;
|
||||
|
||||
function resolveNumPredict(model: Model<"ollama-chat">, requested: number): number {
|
||||
if (model.provider === "ollama-cloud") {
|
||||
return Math.min(requested, OLLAMA_CLOUD_NUM_PREDICT_CAP);
|
||||
}
|
||||
return requested;
|
||||
}
|
||||
|
||||
function createChatBody(model: Model<"ollama-chat">, context: Context, options: OllamaChatOptions | undefined) {
|
||||
const think = mapReasoning(model, options?.reasoning, options?.disableReasoning);
|
||||
const toolChoice = mapToolChoice(options?.toolChoice);
|
||||
@@ -294,7 +314,7 @@ function createChatBody(model: Model<"ollama-chat">, context: Context, options:
|
||||
...(think !== undefined ? { think } : {}),
|
||||
...(toolChoice !== undefined ? { tool_choice: toolChoice } : {}),
|
||||
...(options?.maxTokens !== undefined && !model.omitMaxOutputTokens
|
||||
? { options: { num_predict: options.maxTokens } }
|
||||
? { options: { num_predict: resolveNumPredict(model, options.maxTokens) } }
|
||||
: {}),
|
||||
stream: true,
|
||||
};
|
||||
|
||||
@@ -102,6 +102,82 @@ test("ollama-chat omits num_predict when model opts out of max output tokens", a
|
||||
expect(requestBody).not.toHaveProperty("options");
|
||||
});
|
||||
|
||||
test("ollama-chat clamps num_predict at the Ollama Cloud 65536 output-token cap (#3392)", async () => {
|
||||
// Stale cached models.db row: maxTokens carried forward from a pre-fix
|
||||
// catalog (or a user modelOverride re-enabling output caps), no
|
||||
// `omitMaxOutputTokens` policy applied — so the model spec arrives at the
|
||||
// wire with `maxTokens: 1048576`. Ollama Cloud rejects anything above
|
||||
// 65536 with HTTP 400; the wire layer must clamp before sending.
|
||||
const staleCachedSpec: Model<"ollama-chat"> = {
|
||||
id: "deepseek-v4-pro",
|
||||
name: "deepseek-v4-pro",
|
||||
api: "ollama-chat",
|
||||
provider: "ollama-cloud",
|
||||
baseUrl: "https://ollama.com",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 524288,
|
||||
maxTokens: 1_048_576,
|
||||
compat: undefined,
|
||||
};
|
||||
|
||||
let requestBody: Record<string, unknown> | undefined;
|
||||
const fetchMock: FetchImpl = vi.fn(async (_input, init) => {
|
||||
requestBody = JSON.parse(String(init?.body ?? "{}")) as Record<string, unknown>;
|
||||
return createNdjsonResponse([
|
||||
{ model: "deepseek-v4-pro", message: { role: "assistant", content: "ok" }, done: false },
|
||||
{ model: "deepseek-v4-pro", done: true, done_reason: "stop", prompt_eval_count: 1, eval_count: 1 },
|
||||
]);
|
||||
});
|
||||
|
||||
await streamSimple(
|
||||
staleCachedSpec,
|
||||
{ messages: [{ role: "user", content: "Reply ok", timestamp: Date.now() }] },
|
||||
{ apiKey: "cloud-test-key", fetch: fetchMock },
|
||||
).result();
|
||||
|
||||
const options = requestBody?.options as { num_predict?: number } | undefined;
|
||||
expect(options?.num_predict).toBe(65_536);
|
||||
});
|
||||
|
||||
test("ollama-chat does not clamp num_predict for self-hosted ollama (#3392)", async () => {
|
||||
// Sanity: the clamp is provider-scoped to `ollama-cloud`. A local Ollama
|
||||
// host carries its own output-token semantics — the wire layer must not
|
||||
// rewrite `num_predict` for non-cloud `ollama-chat` traffic.
|
||||
const localSpec: Model<"ollama-chat"> = {
|
||||
id: "deepseek-r1:70b",
|
||||
name: "deepseek-r1:70b",
|
||||
api: "ollama-chat",
|
||||
provider: "ollama",
|
||||
baseUrl: "http://127.0.0.1:11434",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 131_072,
|
||||
maxTokens: 131_072,
|
||||
compat: undefined,
|
||||
};
|
||||
|
||||
let requestBody: Record<string, unknown> | undefined;
|
||||
const fetchMock: FetchImpl = vi.fn(async (_input, init) => {
|
||||
requestBody = JSON.parse(String(init?.body ?? "{}")) as Record<string, unknown>;
|
||||
return createNdjsonResponse([
|
||||
{ model: "deepseek-r1:70b", message: { role: "assistant", content: "ok" }, done: false },
|
||||
{ model: "deepseek-r1:70b", done: true, done_reason: "stop", prompt_eval_count: 1, eval_count: 1 },
|
||||
]);
|
||||
});
|
||||
|
||||
await streamSimple(
|
||||
localSpec,
|
||||
{ messages: [{ role: "user", content: "Reply ok", timestamp: Date.now() }] },
|
||||
{ apiKey: "local-key", fetch: fetchMock },
|
||||
).result();
|
||||
|
||||
const options = requestBody?.options as { num_predict?: number } | undefined;
|
||||
expect(options?.num_predict).toBe(131_072);
|
||||
});
|
||||
|
||||
test("ollama-chat sends think false when reasoning is disabled", async () => {
|
||||
let requestBody: Record<string, unknown> | undefined;
|
||||
const fetchMock: FetchImpl = vi.fn(async (_input, init) => {
|
||||
|
||||
Reference in New Issue
Block a user