diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index d231182bd..db68c9b81 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,6 +4,7 @@ ### Fixed +- Fixed llama.cpp discovery mapping unlimited `max_tokens = -1` / `n_predict = -1` output limits to the generic 32K discovery cap instead of the discovered runtime context window. ([#3781](https://github.com/can1357/oh-my-pi/issues/3781)) - Fixed the bash interceptor blocking `echo` / `printf` redirects to `/dev/null`, `/dev/tty`, `/dev/stdout`, and `/dev/stderr` device sinks while still directing real file writes to the write tool. ([#3763](https://github.com/can1357/oh-my-pi/issues/3763)) ## [16.2.5] - 2026-06-28 diff --git a/packages/coding-agent/src/config/model-discovery.ts b/packages/coding-agent/src/config/model-discovery.ts index 3268d9753..5e6188dad 100644 --- a/packages/coding-agent/src/config/model-discovery.ts +++ b/packages/coding-agent/src/config/model-discovery.ts @@ -122,6 +122,12 @@ type OllamaDiscoveredModelMetadata = { type LlamaCppDiscoveredServerMetadata = { contextWindow?: number; input?: ("text" | "image")[]; + maxTokens?: number | "contextWindow"; +}; + +type LlamaCppDiscoveredModelRuntimeMetadata = { + contextWindow: number; + maxTokens: number; }; type LlamaCppModelListEntry = { @@ -143,6 +149,52 @@ function toPositiveNumberOrUndefined(value: unknown): number | undefined { return undefined; } +function toFiniteNumberOrUndefined(value: unknown): number | undefined { + if (typeof value === "number" && Number.isFinite(value)) { + return value; + } + if (typeof value === "string" && value.trim()) { + const parsed = Number(value); + if (Number.isFinite(parsed)) { + return parsed; + } + } + return undefined; +} + +function extractLlamaCppMaxTokens(payload: Record): number | "contextWindow" | undefined { + const generationSettings = payload.default_generation_settings; + const params = isRecord(generationSettings) ? generationSettings.params : undefined; + const candidates = [ + isRecord(params) ? params.max_tokens : undefined, + isRecord(params) ? params.n_predict : undefined, + isRecord(generationSettings) ? generationSettings.max_tokens : undefined, + isRecord(generationSettings) ? generationSettings.n_predict : undefined, + payload.max_tokens, + payload.n_predict, + ]; + let hasContextBoundedLimit = false; + for (const candidate of candidates) { + const value = toFiniteNumberOrUndefined(candidate); + if (value === undefined) { + continue; + } + if (value > 0) { + return value; + } + if (value === -1) { + hasContextBoundedLimit = true; + } + } + return hasContextBoundedLimit ? "contextWindow" : undefined; +} + +function resolveLlamaCppMaxTokens(contextWindow: number, maxTokens: number | "contextWindow" | undefined): number { + return maxTokens === "contextWindow" + ? contextWindow + : Math.min(contextWindow, maxTokens ?? DISCOVERY_DEFAULT_MAX_TOKENS); +} + function extractOllamaRuntimeContextWindow(payload: Record): number | undefined { const parameters = payload.parameters; if (typeof parameters !== "string") { @@ -353,6 +405,7 @@ async function discoverLlamaCppServerMetadata( } return { contextWindow: extractLlamaCppContextWindow(payload), + maxTokens: extractLlamaCppMaxTokens(payload), input: extractLlamaCppInputCapabilities(payload), }; } catch { @@ -410,7 +463,7 @@ export async function discoverLlamaCppModels( imageInputDecoder: "stb", cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow, - maxTokens: Math.min(contextWindow, DISCOVERY_DEFAULT_MAX_TOKENS), + maxTokens: resolveLlamaCppMaxTokens(contextWindow, serverMetadata?.maxTokens), headers, compat: { supportsStore: false, @@ -423,24 +476,35 @@ export async function discoverLlamaCppModels( return discovered; } -export async function discoverLlamaCppModelContextWindow( +export async function discoverLlamaCppModelRuntimeMetadata( model: Pick, "provider" | "id" | "baseUrl" | "headers">, ctx: DiscoveryContext, -): Promise { +): Promise { const baseUrl = normalizeLlamaCppBaseUrl(model.baseUrl); const modelsUrl = `${baseUrl}/models`; const baseHeaders: Record = { ...(model.headers ?? {}) }; const attempt = async (headers: Record) => { - const response = await ctx.fetch(modelsUrl, { - headers, - signal: AbortSignal.timeout(250), - }); + const [response, serverMetadata] = await Promise.all([ + ctx.fetch(modelsUrl, { + headers, + signal: AbortSignal.timeout(250), + }), + discoverLlamaCppServerMetadata(ctx, baseUrl, headers), + ]); if (!response.ok) { return undefined; } const entries = parseLlamaCppModelList(await response.json()); const entry = entries.find(entry => entry.id === model.id); - return entry?.runtimeContextWindow ?? entry?.trainingContextWindow; + const contextWindow = + entry?.runtimeContextWindow ?? serverMetadata?.contextWindow ?? entry?.trainingContextWindow; + if (contextWindow === undefined) { + return undefined; + } + return { + contextWindow, + maxTokens: resolveLlamaCppMaxTokens(contextWindow, serverMetadata?.maxTokens), + }; }; try { const apiKey = await ctx.getBearerApiKeyResolver(model.provider); diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index c467e8422..c824afe65 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -75,7 +75,7 @@ import { DISCOVERY_DEFAULT_MAX_TOKENS, type DiscoveryContext, type DiscoveryProviderConfig, - discoverLlamaCppModelContextWindow, + discoverLlamaCppModelRuntimeMetadata, discoverModelsByProviderType, getImplicitOllamaBaseUrl, getOllamaContextLengthOverride, @@ -873,10 +873,11 @@ export class ModelRegistry { if (!isLlamaCppDiscovery) { return model; } - const contextWindow = await discoverLlamaCppModelContextWindow(model, this.#nonResolvingDiscoveryContext()); - if (contextWindow === undefined) { + const runtimeMetadata = await discoverLlamaCppModelRuntimeMetadata(model, this.#nonResolvingDiscoveryContext()); + if (runtimeMetadata === undefined) { return this.find(model.provider, model.id) ?? model; } + const { contextWindow, maxTokens } = runtimeMetadata; const current = this.find(model.provider, model.id) ?? model; const override = this.#resolveLiveModelOverride(current); const customModel = this.#resolveLiveCustomModelOverlay(current); @@ -888,7 +889,6 @@ export class ModelRegistry { ) { patch.contextWindow = contextWindow; } - const maxTokens = Math.min(contextWindow, DISCOVERY_DEFAULT_MAX_TOKENS); if ( override?.maxTokens === undefined && customModel?.maxTokens === undefined && diff --git a/packages/coding-agent/test/model-discovery.test.ts b/packages/coding-agent/test/model-discovery.test.ts index 59ee1c245..fcad1e0b4 100644 --- a/packages/coding-agent/test/model-discovery.test.ts +++ b/packages/coding-agent/test/model-discovery.test.ts @@ -648,7 +648,7 @@ describe("ModelRegistry runtime discovery", () => { const apiKey = await registry.getApiKey(llamaModels[0]); expect(apiKey).toBe(kNoAuth); }); - test("llama.cpp discovery reads context window from props n_ctx", async () => { + test("llama.cpp discovery maps unlimited output limits to the context window", async () => { const fetchMock: FetchImpl = async input => { const url = String(input); if (url === "http://127.0.0.1:8080/models") { @@ -662,6 +662,7 @@ describe("ModelRegistry runtime discovery", () => { JSON.stringify({ default_generation_settings: { n_ctx: 262144, + params: { max_tokens: -1, n_predict: -1 }, }, modalities: { vision: true, @@ -680,9 +681,41 @@ describe("ModelRegistry runtime discovery", () => { await registry.refresh(); const llama = registry.find("llama.cpp", "qwen35-35b-a3b"); expect(llama?.contextWindow).toBe(262144); - expect(llama?.maxTokens).toBe(32_768); + expect(llama?.maxTokens).toBe(262144); expect(llama?.input).toEqual(["text", "image"]); }); + + test("llama.cpp discovery honors positive output limits from props", async () => { + const fetchMock: FetchImpl = async input => { + const url = String(input); + if (url === "http://127.0.0.1:8080/models") { + return new Response(JSON.stringify({ data: [{ id: "bounded-output" }] }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + } + if (url === "http://127.0.0.1:8080/props") { + return new Response( + JSON.stringify({ + default_generation_settings: { + n_ctx: 262144, + params: { max_tokens: 65536, n_predict: 65536 }, + }, + }), + { + status: 200, + headers: { "Content-Type": "application/json" }, + }, + ); + } + throw new Error(`Unexpected URL: ${url}`); + }; + const registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: fetchMock }); + await registry.refresh(); + const llama = registry.find("llama.cpp", "bounded-output"); + expect(llama?.contextWindow).toBe(262144); + expect(llama?.maxTokens).toBe(65536); + }); test("llama.cpp discovery prefers runtime n_ctx over training context metadata", async () => { const fetchMock: FetchImpl = async input => { const url = String(input); @@ -741,7 +774,7 @@ describe("ModelRegistry runtime discovery", () => { expect(registry.find("llama.cpp", "unloaded")?.contextWindow).toBe(128000); }); - test("llama.cpp selected model refresh patches newly loaded meta n_ctx", async () => { + test("llama.cpp selected model refresh patches newly loaded meta n_ctx and unlimited output limit", async () => { writeModelCache( "llama.cpp", Date.now(), @@ -771,6 +804,20 @@ describe("ModelRegistry runtime discovery", () => { headers: { "Content-Type": "application/json" }, }); } + if (url === "http://127.0.0.1:8080/props") { + return new Response( + JSON.stringify({ + default_generation_settings: { + n_ctx: 239104, + params: { max_tokens: -1, n_predict: -1 }, + }, + }), + { + status: 200, + headers: { "Content-Type": "application/json" }, + }, + ); + } throw new Error(`Unexpected URL: ${url}`); }; const registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: fetchMock }); @@ -779,7 +826,7 @@ describe("ModelRegistry runtime discovery", () => { expect(stale.contextWindow).toBe(128000); const refreshed = await registry.refreshSelectedModelMetadata(stale); expect(refreshed.contextWindow).toBe(239104); - expect(refreshed.maxTokens).toBe(32768); + expect(refreshed.maxTokens).toBe(239104); expect(registry.find("llama.cpp", "sleeping-model")?.contextWindow).toBe(239104); });