fix(providers): preserved llama train context fallback
Kept llama.cpp n_ctx_train as a bounded fallback only after runtime n_ctx and server props are unavailable.
This commit is contained in:
@@ -126,7 +126,8 @@ type LlamaCppDiscoveredServerMetadata = {
|
||||
|
||||
type LlamaCppModelListEntry = {
|
||||
id: string;
|
||||
contextWindow?: number;
|
||||
runtimeContextWindow?: number;
|
||||
trainingContextWindow?: number;
|
||||
};
|
||||
|
||||
function toPositiveNumberOrUndefined(value: unknown): number | undefined {
|
||||
@@ -183,12 +184,17 @@ function extractLlamaCppContextWindow(payload: Record<string, unknown>): number
|
||||
return toPositiveNumberOrUndefined(payload.n_ctx);
|
||||
}
|
||||
|
||||
function extractLlamaCppModelContextWindow(item: Record<string, unknown>): number | undefined {
|
||||
function extractLlamaCppModelContextWindows(
|
||||
item: Record<string, unknown>,
|
||||
): Pick<LlamaCppModelListEntry, "runtimeContextWindow" | "trainingContextWindow"> {
|
||||
const meta = item.meta;
|
||||
if (!isRecord(meta)) {
|
||||
return undefined;
|
||||
return {};
|
||||
}
|
||||
return toPositiveNumberOrUndefined(meta.n_ctx);
|
||||
return {
|
||||
runtimeContextWindow: toPositiveNumberOrUndefined(meta.n_ctx),
|
||||
trainingContextWindow: toPositiveNumberOrUndefined(meta.n_ctx_train),
|
||||
};
|
||||
}
|
||||
|
||||
function parseLlamaCppModelList(payload: unknown): LlamaCppModelListEntry[] {
|
||||
@@ -199,7 +205,7 @@ function parseLlamaCppModelList(payload: unknown): LlamaCppModelListEntry[] {
|
||||
if (!isRecord(item) || typeof item.id !== "string" || !item.id) {
|
||||
return [];
|
||||
}
|
||||
return [{ id: item.id, contextWindow: extractLlamaCppModelContextWindow(item) }];
|
||||
return [{ id: item.id, ...extractLlamaCppModelContextWindows(item) }];
|
||||
});
|
||||
}
|
||||
|
||||
@@ -387,7 +393,11 @@ export async function discoverLlamaCppModels(
|
||||
for (const item of models) {
|
||||
const { id } = item;
|
||||
if (!id) continue;
|
||||
const contextWindow = item.contextWindow ?? serverMetadata?.contextWindow ?? DISCOVERY_DEFAULT_CONTEXT_WINDOW;
|
||||
const contextWindow =
|
||||
item.runtimeContextWindow ??
|
||||
serverMetadata?.contextWindow ??
|
||||
item.trainingContextWindow ??
|
||||
DISCOVERY_DEFAULT_CONTEXT_WINDOW;
|
||||
discovered.push(
|
||||
buildModel({
|
||||
id,
|
||||
@@ -429,7 +439,8 @@ export async function discoverLlamaCppModelContextWindow(
|
||||
return undefined;
|
||||
}
|
||||
const entries = parseLlamaCppModelList(await response.json());
|
||||
return entries.find(entry => entry.id === model.id)?.contextWindow;
|
||||
const entry = entries.find(entry => entry.id === model.id);
|
||||
return entry?.runtimeContextWindow ?? entry?.trainingContextWindow;
|
||||
};
|
||||
try {
|
||||
const apiKey = await ctx.getBearerApiKeyResolver(model.provider);
|
||||
|
||||
@@ -716,6 +716,31 @@ describe("ModelRegistry runtime discovery", () => {
|
||||
expect(registry.find("llama.cpp", "unloaded")?.contextWindow).toBe(128000);
|
||||
});
|
||||
|
||||
test("llama.cpp discovery falls back to n_ctx_train before the global default", async () => {
|
||||
const fetchMock: FetchImpl = async input => {
|
||||
const url = String(input);
|
||||
if (url === "http://127.0.0.1:8080/models") {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
data: [{ id: "ctx-train", meta: { n_ctx_train: 65536 } }, { id: "unloaded" }],
|
||||
}),
|
||||
{ status: 200, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
}
|
||||
if (url === "http://127.0.0.1:8080/props") {
|
||||
return new Response(JSON.stringify({ default_generation_settings: {} }), {
|
||||
status: 200,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
throw new Error(`Unexpected URL: ${url}`);
|
||||
};
|
||||
const registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: fetchMock });
|
||||
await registry.refresh();
|
||||
expect(registry.find("llama.cpp", "ctx-train")?.contextWindow).toBe(65536);
|
||||
expect(registry.find("llama.cpp", "unloaded")?.contextWindow).toBe(128000);
|
||||
});
|
||||
|
||||
test("llama.cpp selected model refresh patches newly loaded meta n_ctx", async () => {
|
||||
writeModelCache(
|
||||
"llama.cpp",
|
||||
|
||||
Reference in New Issue
Block a user