fix(agent): refreshed startup llama.cpp vision metadata

Refresh cached llama.cpp runtime metadata before exposing the initial session model so local vision defaults are not treated as text-only.

Fixes #4670
This commit is contained in:
roboomp
2026-07-06 04:03:00 +00:00
parent 674cf01762
commit 17080bef3c
3 changed files with 96 additions and 1 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed startup of cached llama.cpp vision models so the initial default/restored model refreshes `/props` metadata before the session exposes it as text-only.
## [16.3.9] - 2026-07-06
### Added
+18
View File
@@ -2132,6 +2132,24 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
}
}
if (model) {
const selectedModel = model;
const refreshedModel = await logger.time("refreshInitialModelMetadata", () =>
modelRegistry.refreshSelectedModelMetadata(selectedModel),
);
if (refreshedModel !== selectedModel) {
model = refreshedModel;
thinkingLevel = pickInitialThinkingLevel(refreshedModel);
autoThinking = thinkingLevel === AUTO_THINKING;
effectiveThinkingLevel = concreteThinkingLevel(thinkingLevel);
effectiveThinkingLevel = logger.time("resolveThinkingLevelForModel", () =>
autoThinking
? resolveProvisionalAutoLevel(refreshedModel)
: resolveThinkingLevelForModel(refreshedModel, effectiveThinkingLevel),
);
}
}
// Discover custom commands (TypeScript slash commands)
const customCommandsResult: CustomCommandsLoadResult = options.disableExtensionDiscovery
? { commands: [], errors: [] }
@@ -2,7 +2,9 @@ import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { Effort } from "@oh-my-pi/pi-ai";
import { Effort, type FetchImpl } from "@oh-my-pi/pi-ai";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import { ModelRegistry, type ProviderConfigInput } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
@@ -334,6 +336,77 @@ describe("createAgentSession deferred model pattern resolution", () => {
}
});
test("refreshes cached llama.cpp vision metadata for the startup default model", async () => {
const authStorage = await AuthStorage.create(path.join(tempDir, "llama-vision-auth.db"));
authStoragesToClose.push(authStorage);
const modelsPath = path.join(tempDir, "llama-vision-models.yml");
const cacheDbPath = path.join(tempDir, "models.db");
const cachedModel = buildModel({
id: "vision-model",
name: "vision-model",
provider: "llama.cpp",
api: "openai-responses",
baseUrl: "http://127.0.0.1:8080",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128000,
maxTokens: 32768,
});
writeModelCache("llama.cpp", Date.now(), [cachedModel], true, "", cacheDbPath);
const fetchMock: FetchImpl = async input => {
const url = String(input);
if (url === "http://127.0.0.1:8080/models") {
return new Response(
JSON.stringify({ data: [{ id: "vision-model", object: "model", meta: { n_ctx: 239104 } }] }),
{ status: 200, headers: { "Content-Type": "application/json" } },
);
}
if (url === "http://127.0.0.1:8080/props") {
return new Response(
JSON.stringify({
default_generation_settings: {
n_ctx: 239104,
params: { max_tokens: -1, n_predict: -1 },
},
modalities: { vision: true },
}),
{ status: 200, headers: { "Content-Type": "application/json" } },
);
}
throw new Error(`Unexpected URL: ${url}`);
};
const modelRegistry = new ModelRegistry(authStorage, modelsPath, { fetch: fetchMock });
const settings = Settings.isolated();
settings.setModelRole("default", "llama.cpp/vision-model");
expect(modelRegistry.find("llama.cpp", "vision-model")?.input).toEqual(["text"]);
const { session } = await createAgentSession({
cwd: tempDir,
agentDir: tempDir,
authStorage,
modelRegistry,
settings,
sessionManager: SessionManager.inMemory(),
disableExtensionDiscovery: true,
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
skipPythonPreflight: true,
});
try {
expect(session.model?.input).toEqual(["text", "image"]);
expect(modelRegistry.find("llama.cpp", "vision-model")?.input).toEqual(["text", "image"]);
} finally {
await session.dispose();
}
});
test("restores the saved session model without resolving auth over the network", async () => {
// Regression: `restoreSessionModel` probed each saved-model candidate with
// the async `getApiKey`, which refreshes OAuth tokens and hits the auth