fix(providers): honored llama.cpp unlimited output cap

Mapped llama.cpp -1 generation limits from /props to the discovered runtime context window instead of the generic discovery default, including selected-model metadata refresh.

Fixes #3781
This commit is contained in:
roboomp
2026-06-29 03:47:11 +00:00
parent ca9f2847e6
commit dca8a7afba
4 changed files with 128 additions and 16 deletions
+1
View File
@@ -4,6 +4,7 @@
### Fixed
- Fixed llama.cpp discovery mapping unlimited `max_tokens = -1` / `n_predict = -1` output limits to the generic 32K discovery cap instead of the discovered runtime context window. ([#3781](https://github.com/can1357/oh-my-pi/issues/3781))
- Fixed the bash interceptor blocking `echo` / `printf` redirects to `/dev/null`, `/dev/tty`, `/dev/stdout`, and `/dev/stderr` device sinks while still directing real file writes to the write tool. ([#3763](https://github.com/can1357/oh-my-pi/issues/3763))
## [16.2.5] - 2026-06-28
@@ -122,6 +122,12 @@ type OllamaDiscoveredModelMetadata = {
type LlamaCppDiscoveredServerMetadata = {
contextWindow?: number;
input?: ("text" | "image")[];
maxTokens?: number | "contextWindow";
};
type LlamaCppDiscoveredModelRuntimeMetadata = {
contextWindow: number;
maxTokens: number;
};
type LlamaCppModelListEntry = {
@@ -143,6 +149,52 @@ function toPositiveNumberOrUndefined(value: unknown): number | undefined {
return undefined;
}
function toFiniteNumberOrUndefined(value: unknown): number | undefined {
if (typeof value === "number" && Number.isFinite(value)) {
return value;
}
if (typeof value === "string" && value.trim()) {
const parsed = Number(value);
if (Number.isFinite(parsed)) {
return parsed;
}
}
return undefined;
}
function extractLlamaCppMaxTokens(payload: Record<string, unknown>): number | "contextWindow" | undefined {
const generationSettings = payload.default_generation_settings;
const params = isRecord(generationSettings) ? generationSettings.params : undefined;
const candidates = [
isRecord(params) ? params.max_tokens : undefined,
isRecord(params) ? params.n_predict : undefined,
isRecord(generationSettings) ? generationSettings.max_tokens : undefined,
isRecord(generationSettings) ? generationSettings.n_predict : undefined,
payload.max_tokens,
payload.n_predict,
];
let hasContextBoundedLimit = false;
for (const candidate of candidates) {
const value = toFiniteNumberOrUndefined(candidate);
if (value === undefined) {
continue;
}
if (value > 0) {
return value;
}
if (value === -1) {
hasContextBoundedLimit = true;
}
}
return hasContextBoundedLimit ? "contextWindow" : undefined;
}
function resolveLlamaCppMaxTokens(contextWindow: number, maxTokens: number | "contextWindow" | undefined): number {
return maxTokens === "contextWindow"
? contextWindow
: Math.min(contextWindow, maxTokens ?? DISCOVERY_DEFAULT_MAX_TOKENS);
}
function extractOllamaRuntimeContextWindow(payload: Record<string, unknown>): number | undefined {
const parameters = payload.parameters;
if (typeof parameters !== "string") {
@@ -353,6 +405,7 @@ async function discoverLlamaCppServerMetadata(
}
return {
contextWindow: extractLlamaCppContextWindow(payload),
maxTokens: extractLlamaCppMaxTokens(payload),
input: extractLlamaCppInputCapabilities(payload),
};
} catch {
@@ -410,7 +463,7 @@ export async function discoverLlamaCppModels(
imageInputDecoder: "stb",
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow,
maxTokens: Math.min(contextWindow, DISCOVERY_DEFAULT_MAX_TOKENS),
maxTokens: resolveLlamaCppMaxTokens(contextWindow, serverMetadata?.maxTokens),
headers,
compat: {
supportsStore: false,
@@ -423,24 +476,35 @@ export async function discoverLlamaCppModels(
return discovered;
}
export async function discoverLlamaCppModelContextWindow(
export async function discoverLlamaCppModelRuntimeMetadata(
model: Pick<Model<Api>, "provider" | "id" | "baseUrl" | "headers">,
ctx: DiscoveryContext,
): Promise<number | undefined> {
): Promise<LlamaCppDiscoveredModelRuntimeMetadata | undefined> {
const baseUrl = normalizeLlamaCppBaseUrl(model.baseUrl);
const modelsUrl = `${baseUrl}/models`;
const baseHeaders: Record<string, string> = { ...(model.headers ?? {}) };
const attempt = async (headers: Record<string, string>) => {
const response = await ctx.fetch(modelsUrl, {
headers,
signal: AbortSignal.timeout(250),
});
const [response, serverMetadata] = await Promise.all([
ctx.fetch(modelsUrl, {
headers,
signal: AbortSignal.timeout(250),
}),
discoverLlamaCppServerMetadata(ctx, baseUrl, headers),
]);
if (!response.ok) {
return undefined;
}
const entries = parseLlamaCppModelList(await response.json());
const entry = entries.find(entry => entry.id === model.id);
return entry?.runtimeContextWindow ?? entry?.trainingContextWindow;
const contextWindow =
entry?.runtimeContextWindow ?? serverMetadata?.contextWindow ?? entry?.trainingContextWindow;
if (contextWindow === undefined) {
return undefined;
}
return {
contextWindow,
maxTokens: resolveLlamaCppMaxTokens(contextWindow, serverMetadata?.maxTokens),
};
};
try {
const apiKey = await ctx.getBearerApiKeyResolver(model.provider);
@@ -75,7 +75,7 @@ import {
DISCOVERY_DEFAULT_MAX_TOKENS,
type DiscoveryContext,
type DiscoveryProviderConfig,
discoverLlamaCppModelContextWindow,
discoverLlamaCppModelRuntimeMetadata,
discoverModelsByProviderType,
getImplicitOllamaBaseUrl,
getOllamaContextLengthOverride,
@@ -873,10 +873,11 @@ export class ModelRegistry {
if (!isLlamaCppDiscovery) {
return model;
}
const contextWindow = await discoverLlamaCppModelContextWindow(model, this.#nonResolvingDiscoveryContext());
if (contextWindow === undefined) {
const runtimeMetadata = await discoverLlamaCppModelRuntimeMetadata(model, this.#nonResolvingDiscoveryContext());
if (runtimeMetadata === undefined) {
return this.find(model.provider, model.id) ?? model;
}
const { contextWindow, maxTokens } = runtimeMetadata;
const current = this.find(model.provider, model.id) ?? model;
const override = this.#resolveLiveModelOverride(current);
const customModel = this.#resolveLiveCustomModelOverlay(current);
@@ -888,7 +889,6 @@ export class ModelRegistry {
) {
patch.contextWindow = contextWindow;
}
const maxTokens = Math.min(contextWindow, DISCOVERY_DEFAULT_MAX_TOKENS);
if (
override?.maxTokens === undefined &&
customModel?.maxTokens === undefined &&
@@ -648,7 +648,7 @@ describe("ModelRegistry runtime discovery", () => {
const apiKey = await registry.getApiKey(llamaModels[0]);
expect(apiKey).toBe(kNoAuth);
});
test("llama.cpp discovery reads context window from props n_ctx", async () => {
test("llama.cpp discovery maps unlimited output limits to the context window", async () => {
const fetchMock: FetchImpl = async input => {
const url = String(input);
if (url === "http://127.0.0.1:8080/models") {
@@ -662,6 +662,7 @@ describe("ModelRegistry runtime discovery", () => {
JSON.stringify({
default_generation_settings: {
n_ctx: 262144,
params: { max_tokens: -1, n_predict: -1 },
},
modalities: {
vision: true,
@@ -680,9 +681,41 @@ describe("ModelRegistry runtime discovery", () => {
await registry.refresh();
const llama = registry.find("llama.cpp", "qwen35-35b-a3b");
expect(llama?.contextWindow).toBe(262144);
expect(llama?.maxTokens).toBe(32_768);
expect(llama?.maxTokens).toBe(262144);
expect(llama?.input).toEqual(["text", "image"]);
});
test("llama.cpp discovery honors positive output limits from props", async () => {
const fetchMock: FetchImpl = async input => {
const url = String(input);
if (url === "http://127.0.0.1:8080/models") {
return new Response(JSON.stringify({ data: [{ id: "bounded-output" }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
}
if (url === "http://127.0.0.1:8080/props") {
return new Response(
JSON.stringify({
default_generation_settings: {
n_ctx: 262144,
params: { max_tokens: 65536, n_predict: 65536 },
},
}),
{
status: 200,
headers: { "Content-Type": "application/json" },
},
);
}
throw new Error(`Unexpected URL: ${url}`);
};
const registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: fetchMock });
await registry.refresh();
const llama = registry.find("llama.cpp", "bounded-output");
expect(llama?.contextWindow).toBe(262144);
expect(llama?.maxTokens).toBe(65536);
});
test("llama.cpp discovery prefers runtime n_ctx over training context metadata", async () => {
const fetchMock: FetchImpl = async input => {
const url = String(input);
@@ -741,7 +774,7 @@ describe("ModelRegistry runtime discovery", () => {
expect(registry.find("llama.cpp", "unloaded")?.contextWindow).toBe(128000);
});
test("llama.cpp selected model refresh patches newly loaded meta n_ctx", async () => {
test("llama.cpp selected model refresh patches newly loaded meta n_ctx and unlimited output limit", async () => {
writeModelCache(
"llama.cpp",
Date.now(),
@@ -771,6 +804,20 @@ describe("ModelRegistry runtime discovery", () => {
headers: { "Content-Type": "application/json" },
});
}
if (url === "http://127.0.0.1:8080/props") {
return new Response(
JSON.stringify({
default_generation_settings: {
n_ctx: 239104,
params: { max_tokens: -1, n_predict: -1 },
},
}),
{
status: 200,
headers: { "Content-Type": "application/json" },
},
);
}
throw new Error(`Unexpected URL: ${url}`);
};
const registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: fetchMock });
@@ -779,7 +826,7 @@ describe("ModelRegistry runtime discovery", () => {
expect(stale.contextWindow).toBe(128000);
const refreshed = await registry.refreshSelectedModelMetadata(stale);
expect(refreshed.contextWindow).toBe(239104);
expect(refreshed.maxTokens).toBe(32768);
expect(refreshed.maxTokens).toBe(239104);
expect(registry.find("llama.cpp", "sleeping-model")?.contextWindow).toBe(239104);
});