fix(providers): preserved llama train context fallback

Kept llama.cpp n_ctx_train as a bounded fallback only after runtime n_ctx and server props are unavailable.
This commit is contained in:
roboomp
2026-06-28 21:10:35 +00:00
parent 3600dc0cc4
commit 89305f724f
2 changed files with 43 additions and 7 deletions
@@ -126,7 +126,8 @@ type LlamaCppDiscoveredServerMetadata = {
type LlamaCppModelListEntry = {
id: string;
contextWindow?: number;
runtimeContextWindow?: number;
trainingContextWindow?: number;
};
function toPositiveNumberOrUndefined(value: unknown): number | undefined {
@@ -183,12 +184,17 @@ function extractLlamaCppContextWindow(payload: Record<string, unknown>): number
return toPositiveNumberOrUndefined(payload.n_ctx);
}
function extractLlamaCppModelContextWindow(item: Record<string, unknown>): number | undefined {
function extractLlamaCppModelContextWindows(
item: Record<string, unknown>,
): Pick<LlamaCppModelListEntry, "runtimeContextWindow" | "trainingContextWindow"> {
const meta = item.meta;
if (!isRecord(meta)) {
return undefined;
return {};
}
return toPositiveNumberOrUndefined(meta.n_ctx);
return {
runtimeContextWindow: toPositiveNumberOrUndefined(meta.n_ctx),
trainingContextWindow: toPositiveNumberOrUndefined(meta.n_ctx_train),
};
}
function parseLlamaCppModelList(payload: unknown): LlamaCppModelListEntry[] {
@@ -199,7 +205,7 @@ function parseLlamaCppModelList(payload: unknown): LlamaCppModelListEntry[] {
if (!isRecord(item) || typeof item.id !== "string" || !item.id) {
return [];
}
return [{ id: item.id, contextWindow: extractLlamaCppModelContextWindow(item) }];
return [{ id: item.id, ...extractLlamaCppModelContextWindows(item) }];
});
}
@@ -387,7 +393,11 @@ export async function discoverLlamaCppModels(
for (const item of models) {
const { id } = item;
if (!id) continue;
const contextWindow = item.contextWindow ?? serverMetadata?.contextWindow ?? DISCOVERY_DEFAULT_CONTEXT_WINDOW;
const contextWindow =
item.runtimeContextWindow ??
serverMetadata?.contextWindow ??
item.trainingContextWindow ??
DISCOVERY_DEFAULT_CONTEXT_WINDOW;
discovered.push(
buildModel({
id,
@@ -429,7 +439,8 @@ export async function discoverLlamaCppModelContextWindow(
return undefined;
}
const entries = parseLlamaCppModelList(await response.json());
return entries.find(entry => entry.id === model.id)?.contextWindow;
const entry = entries.find(entry => entry.id === model.id);
return entry?.runtimeContextWindow ?? entry?.trainingContextWindow;
};
try {
const apiKey = await ctx.getBearerApiKeyResolver(model.provider);
@@ -716,6 +716,31 @@ describe("ModelRegistry runtime discovery", () => {
expect(registry.find("llama.cpp", "unloaded")?.contextWindow).toBe(128000);
});
test("llama.cpp discovery falls back to n_ctx_train before the global default", async () => {
const fetchMock: FetchImpl = async input => {
const url = String(input);
if (url === "http://127.0.0.1:8080/models") {
return new Response(
JSON.stringify({
data: [{ id: "ctx-train", meta: { n_ctx_train: 65536 } }, { id: "unloaded" }],
}),
{ status: 200, headers: { "Content-Type": "application/json" } },
);
}
if (url === "http://127.0.0.1:8080/props") {
return new Response(JSON.stringify({ default_generation_settings: {} }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
}
throw new Error(`Unexpected URL: ${url}`);
};
const registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: fetchMock });
await registry.refresh();
expect(registry.find("llama.cpp", "ctx-train")?.contextWindow).toBe(65536);
expect(registry.find("llama.cpp", "unloaded")?.contextWindow).toBe(128000);
});
test("llama.cpp selected model refresh patches newly loaded meta n_ctx", async () => {
writeModelCache(
"llama.cpp",