From 00a41749e4ad0fc5c5117b428d0532cc84ecd3aa Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 29 Jun 2026 04:03:56 +0000 Subject: [PATCH] fix(providers): clamped llama.cpp refreshed output cap Resolved selected-model refresh maxTokens against the effective context window, including live contextWindow overrides, so unlimited llama.cpp caps cannot exceed the configured context. Fixes #3781 --- .../coding-agent/src/config/model-registry.ts | 11 +++- .../coding-agent/test/model-discovery.test.ts | 66 +++++++++++++++++++ 2 files changed, 75 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index c824afe65..630ac5993 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -889,12 +889,19 @@ export class ModelRegistry { ) { patch.contextWindow = contextWindow; } + const effectiveContextWindow = + override?.contextWindow ?? + customModel?.contextWindow ?? + patch.contextWindow ?? + current.contextWindow ?? + contextWindow; + const effectiveMaxTokens = Math.min(maxTokens, effectiveContextWindow); if ( override?.maxTokens === undefined && customModel?.maxTokens === undefined && - current.maxTokens !== maxTokens + current.maxTokens !== effectiveMaxTokens ) { - patch.maxTokens = maxTokens; + patch.maxTokens = effectiveMaxTokens; } if (patch.contextWindow === undefined && patch.maxTokens === undefined) { return current; diff --git a/packages/coding-agent/test/model-discovery.test.ts b/packages/coding-agent/test/model-discovery.test.ts index 53e641191..318c6d203 100644 --- a/packages/coding-agent/test/model-discovery.test.ts +++ b/packages/coding-agent/test/model-discovery.test.ts @@ -884,6 +884,72 @@ describe("ModelRegistry runtime discovery", () => { expect(refreshed.maxTokens).toBe(32768); }); + test("llama.cpp selected model refresh clamps unlimited output to overridden context", async () => { + writeRawModelsJson({ + "llama.cpp": { + baseUrl: "http://127.0.0.1:8080", + api: "openai-responses", + auth: "none", + discovery: { type: "llama.cpp" }, + modelOverrides: { + "bounded-context-model": { contextWindow: 128000 }, + }, + }, + }); + writeModelCache( + "llama.cpp", + Date.now(), + [ + buildModel({ + id: "bounded-context-model", + name: "bounded-context-model", + provider: "llama.cpp", + api: "openai-responses", + baseUrl: "http://127.0.0.1:8080", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 262144, + maxTokens: 32768, + }), + ], + true, + "", + cacheDbPath, + ); + const fetchMock: FetchImpl = async input => { + const url = String(input); + if (url === "http://127.0.0.1:8080/models") { + return new Response(JSON.stringify({ data: [{ id: "bounded-context-model", meta: { n_ctx: 262144 } }] }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + } + if (url === "http://127.0.0.1:8080/props") { + return new Response( + JSON.stringify({ + default_generation_settings: { + n_ctx: 262144, + params: { max_tokens: -1, n_predict: -1 }, + }, + }), + { + status: 200, + headers: { "Content-Type": "application/json" }, + }, + ); + } + throw new Error(`Unexpected URL: ${url}`); + }; + const registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: fetchMock }); + const bounded = registry.find("llama.cpp", "bounded-context-model"); + if (!bounded) throw new Error("cached llama.cpp model missing"); + expect(bounded.contextWindow).toBe(128000); + const refreshed = await registry.refreshSelectedModelMetadata(bounded); + expect(refreshed.contextWindow).toBe(128000); + expect(refreshed.maxTokens).toBe(128000); + }); + test("llama.cpp selected model refresh does not resolve command api keys", async () => { const commandLogPath = path.join(tempDir, "llama-cpp-key-command.log"); writeRawModelsJson({