fix(providers): honored llama.cpp unlimited output cap
Mapped llama.cpp -1 generation limits from /props to the discovered runtime context window instead of the generic discovery default, including selected-model metadata refresh. Fixes #3781
This commit is contained in:
@@ -4,6 +4,7 @@
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed llama.cpp discovery mapping unlimited `max_tokens = -1` / `n_predict = -1` output limits to the generic 32K discovery cap instead of the discovered runtime context window. ([#3781](https://github.com/can1357/oh-my-pi/issues/3781))
|
||||
- Fixed the bash interceptor blocking `echo` / `printf` redirects to `/dev/null`, `/dev/tty`, `/dev/stdout`, and `/dev/stderr` device sinks while still directing real file writes to the write tool. ([#3763](https://github.com/can1357/oh-my-pi/issues/3763))
|
||||
|
||||
## [16.2.5] - 2026-06-28
|
||||
|
||||
@@ -122,6 +122,12 @@ type OllamaDiscoveredModelMetadata = {
|
||||
type LlamaCppDiscoveredServerMetadata = {
|
||||
contextWindow?: number;
|
||||
input?: ("text" | "image")[];
|
||||
maxTokens?: number | "contextWindow";
|
||||
};
|
||||
|
||||
type LlamaCppDiscoveredModelRuntimeMetadata = {
|
||||
contextWindow: number;
|
||||
maxTokens: number;
|
||||
};
|
||||
|
||||
type LlamaCppModelListEntry = {
|
||||
@@ -143,6 +149,52 @@ function toPositiveNumberOrUndefined(value: unknown): number | undefined {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function toFiniteNumberOrUndefined(value: unknown): number | undefined {
|
||||
if (typeof value === "number" && Number.isFinite(value)) {
|
||||
return value;
|
||||
}
|
||||
if (typeof value === "string" && value.trim()) {
|
||||
const parsed = Number(value);
|
||||
if (Number.isFinite(parsed)) {
|
||||
return parsed;
|
||||
}
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function extractLlamaCppMaxTokens(payload: Record<string, unknown>): number | "contextWindow" | undefined {
|
||||
const generationSettings = payload.default_generation_settings;
|
||||
const params = isRecord(generationSettings) ? generationSettings.params : undefined;
|
||||
const candidates = [
|
||||
isRecord(params) ? params.max_tokens : undefined,
|
||||
isRecord(params) ? params.n_predict : undefined,
|
||||
isRecord(generationSettings) ? generationSettings.max_tokens : undefined,
|
||||
isRecord(generationSettings) ? generationSettings.n_predict : undefined,
|
||||
payload.max_tokens,
|
||||
payload.n_predict,
|
||||
];
|
||||
let hasContextBoundedLimit = false;
|
||||
for (const candidate of candidates) {
|
||||
const value = toFiniteNumberOrUndefined(candidate);
|
||||
if (value === undefined) {
|
||||
continue;
|
||||
}
|
||||
if (value > 0) {
|
||||
return value;
|
||||
}
|
||||
if (value === -1) {
|
||||
hasContextBoundedLimit = true;
|
||||
}
|
||||
}
|
||||
return hasContextBoundedLimit ? "contextWindow" : undefined;
|
||||
}
|
||||
|
||||
function resolveLlamaCppMaxTokens(contextWindow: number, maxTokens: number | "contextWindow" | undefined): number {
|
||||
return maxTokens === "contextWindow"
|
||||
? contextWindow
|
||||
: Math.min(contextWindow, maxTokens ?? DISCOVERY_DEFAULT_MAX_TOKENS);
|
||||
}
|
||||
|
||||
function extractOllamaRuntimeContextWindow(payload: Record<string, unknown>): number | undefined {
|
||||
const parameters = payload.parameters;
|
||||
if (typeof parameters !== "string") {
|
||||
@@ -353,6 +405,7 @@ async function discoverLlamaCppServerMetadata(
|
||||
}
|
||||
return {
|
||||
contextWindow: extractLlamaCppContextWindow(payload),
|
||||
maxTokens: extractLlamaCppMaxTokens(payload),
|
||||
input: extractLlamaCppInputCapabilities(payload),
|
||||
};
|
||||
} catch {
|
||||
@@ -410,7 +463,7 @@ export async function discoverLlamaCppModels(
|
||||
imageInputDecoder: "stb",
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow,
|
||||
maxTokens: Math.min(contextWindow, DISCOVERY_DEFAULT_MAX_TOKENS),
|
||||
maxTokens: resolveLlamaCppMaxTokens(contextWindow, serverMetadata?.maxTokens),
|
||||
headers,
|
||||
compat: {
|
||||
supportsStore: false,
|
||||
@@ -423,24 +476,35 @@ export async function discoverLlamaCppModels(
|
||||
return discovered;
|
||||
}
|
||||
|
||||
export async function discoverLlamaCppModelContextWindow(
|
||||
export async function discoverLlamaCppModelRuntimeMetadata(
|
||||
model: Pick<Model<Api>, "provider" | "id" | "baseUrl" | "headers">,
|
||||
ctx: DiscoveryContext,
|
||||
): Promise<number | undefined> {
|
||||
): Promise<LlamaCppDiscoveredModelRuntimeMetadata | undefined> {
|
||||
const baseUrl = normalizeLlamaCppBaseUrl(model.baseUrl);
|
||||
const modelsUrl = `${baseUrl}/models`;
|
||||
const baseHeaders: Record<string, string> = { ...(model.headers ?? {}) };
|
||||
const attempt = async (headers: Record<string, string>) => {
|
||||
const response = await ctx.fetch(modelsUrl, {
|
||||
headers,
|
||||
signal: AbortSignal.timeout(250),
|
||||
});
|
||||
const [response, serverMetadata] = await Promise.all([
|
||||
ctx.fetch(modelsUrl, {
|
||||
headers,
|
||||
signal: AbortSignal.timeout(250),
|
||||
}),
|
||||
discoverLlamaCppServerMetadata(ctx, baseUrl, headers),
|
||||
]);
|
||||
if (!response.ok) {
|
||||
return undefined;
|
||||
}
|
||||
const entries = parseLlamaCppModelList(await response.json());
|
||||
const entry = entries.find(entry => entry.id === model.id);
|
||||
return entry?.runtimeContextWindow ?? entry?.trainingContextWindow;
|
||||
const contextWindow =
|
||||
entry?.runtimeContextWindow ?? serverMetadata?.contextWindow ?? entry?.trainingContextWindow;
|
||||
if (contextWindow === undefined) {
|
||||
return undefined;
|
||||
}
|
||||
return {
|
||||
contextWindow,
|
||||
maxTokens: resolveLlamaCppMaxTokens(contextWindow, serverMetadata?.maxTokens),
|
||||
};
|
||||
};
|
||||
try {
|
||||
const apiKey = await ctx.getBearerApiKeyResolver(model.provider);
|
||||
|
||||
@@ -75,7 +75,7 @@ import {
|
||||
DISCOVERY_DEFAULT_MAX_TOKENS,
|
||||
type DiscoveryContext,
|
||||
type DiscoveryProviderConfig,
|
||||
discoverLlamaCppModelContextWindow,
|
||||
discoverLlamaCppModelRuntimeMetadata,
|
||||
discoverModelsByProviderType,
|
||||
getImplicitOllamaBaseUrl,
|
||||
getOllamaContextLengthOverride,
|
||||
@@ -873,10 +873,11 @@ export class ModelRegistry {
|
||||
if (!isLlamaCppDiscovery) {
|
||||
return model;
|
||||
}
|
||||
const contextWindow = await discoverLlamaCppModelContextWindow(model, this.#nonResolvingDiscoveryContext());
|
||||
if (contextWindow === undefined) {
|
||||
const runtimeMetadata = await discoverLlamaCppModelRuntimeMetadata(model, this.#nonResolvingDiscoveryContext());
|
||||
if (runtimeMetadata === undefined) {
|
||||
return this.find(model.provider, model.id) ?? model;
|
||||
}
|
||||
const { contextWindow, maxTokens } = runtimeMetadata;
|
||||
const current = this.find(model.provider, model.id) ?? model;
|
||||
const override = this.#resolveLiveModelOverride(current);
|
||||
const customModel = this.#resolveLiveCustomModelOverlay(current);
|
||||
@@ -888,7 +889,6 @@ export class ModelRegistry {
|
||||
) {
|
||||
patch.contextWindow = contextWindow;
|
||||
}
|
||||
const maxTokens = Math.min(contextWindow, DISCOVERY_DEFAULT_MAX_TOKENS);
|
||||
if (
|
||||
override?.maxTokens === undefined &&
|
||||
customModel?.maxTokens === undefined &&
|
||||
|
||||
@@ -648,7 +648,7 @@ describe("ModelRegistry runtime discovery", () => {
|
||||
const apiKey = await registry.getApiKey(llamaModels[0]);
|
||||
expect(apiKey).toBe(kNoAuth);
|
||||
});
|
||||
test("llama.cpp discovery reads context window from props n_ctx", async () => {
|
||||
test("llama.cpp discovery maps unlimited output limits to the context window", async () => {
|
||||
const fetchMock: FetchImpl = async input => {
|
||||
const url = String(input);
|
||||
if (url === "http://127.0.0.1:8080/models") {
|
||||
@@ -662,6 +662,7 @@ describe("ModelRegistry runtime discovery", () => {
|
||||
JSON.stringify({
|
||||
default_generation_settings: {
|
||||
n_ctx: 262144,
|
||||
params: { max_tokens: -1, n_predict: -1 },
|
||||
},
|
||||
modalities: {
|
||||
vision: true,
|
||||
@@ -680,9 +681,41 @@ describe("ModelRegistry runtime discovery", () => {
|
||||
await registry.refresh();
|
||||
const llama = registry.find("llama.cpp", "qwen35-35b-a3b");
|
||||
expect(llama?.contextWindow).toBe(262144);
|
||||
expect(llama?.maxTokens).toBe(32_768);
|
||||
expect(llama?.maxTokens).toBe(262144);
|
||||
expect(llama?.input).toEqual(["text", "image"]);
|
||||
});
|
||||
|
||||
test("llama.cpp discovery honors positive output limits from props", async () => {
|
||||
const fetchMock: FetchImpl = async input => {
|
||||
const url = String(input);
|
||||
if (url === "http://127.0.0.1:8080/models") {
|
||||
return new Response(JSON.stringify({ data: [{ id: "bounded-output" }] }), {
|
||||
status: 200,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
if (url === "http://127.0.0.1:8080/props") {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
default_generation_settings: {
|
||||
n_ctx: 262144,
|
||||
params: { max_tokens: 65536, n_predict: 65536 },
|
||||
},
|
||||
}),
|
||||
{
|
||||
status: 200,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
},
|
||||
);
|
||||
}
|
||||
throw new Error(`Unexpected URL: ${url}`);
|
||||
};
|
||||
const registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: fetchMock });
|
||||
await registry.refresh();
|
||||
const llama = registry.find("llama.cpp", "bounded-output");
|
||||
expect(llama?.contextWindow).toBe(262144);
|
||||
expect(llama?.maxTokens).toBe(65536);
|
||||
});
|
||||
test("llama.cpp discovery prefers runtime n_ctx over training context metadata", async () => {
|
||||
const fetchMock: FetchImpl = async input => {
|
||||
const url = String(input);
|
||||
@@ -741,7 +774,7 @@ describe("ModelRegistry runtime discovery", () => {
|
||||
expect(registry.find("llama.cpp", "unloaded")?.contextWindow).toBe(128000);
|
||||
});
|
||||
|
||||
test("llama.cpp selected model refresh patches newly loaded meta n_ctx", async () => {
|
||||
test("llama.cpp selected model refresh patches newly loaded meta n_ctx and unlimited output limit", async () => {
|
||||
writeModelCache(
|
||||
"llama.cpp",
|
||||
Date.now(),
|
||||
@@ -771,6 +804,20 @@ describe("ModelRegistry runtime discovery", () => {
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
if (url === "http://127.0.0.1:8080/props") {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
default_generation_settings: {
|
||||
n_ctx: 239104,
|
||||
params: { max_tokens: -1, n_predict: -1 },
|
||||
},
|
||||
}),
|
||||
{
|
||||
status: 200,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
},
|
||||
);
|
||||
}
|
||||
throw new Error(`Unexpected URL: ${url}`);
|
||||
};
|
||||
const registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: fetchMock });
|
||||
@@ -779,7 +826,7 @@ describe("ModelRegistry runtime discovery", () => {
|
||||
expect(stale.contextWindow).toBe(128000);
|
||||
const refreshed = await registry.refreshSelectedModelMetadata(stale);
|
||||
expect(refreshed.contextWindow).toBe(239104);
|
||||
expect(refreshed.maxTokens).toBe(32768);
|
||||
expect(refreshed.maxTokens).toBe(239104);
|
||||
expect(registry.find("llama.cpp", "sleeping-model")?.contextWindow).toBe(239104);
|
||||
});
|
||||
|
||||
|
||||
Reference in New Issue
Block a user