From 17080bef3ca5adaab6a075bb28834607c45994d6 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 6 Jul 2026 04:03:00 +0000 Subject: [PATCH] fix(agent): refreshed startup llama.cpp vision metadata Refresh cached llama.cpp runtime metadata before exposing the initial session model so local vision defaults are not treated as text-only. Fixes #4670 --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/sdk.ts | 18 +++++ .../test/sdk-model-selection.test.ts | 75 ++++++++++++++++++- 3 files changed, 96 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c5595a893..3893bbb84 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed startup of cached llama.cpp vision models so the initial default/restored model refreshes `/props` metadata before the session exposes it as text-only. + ## [16.3.9] - 2026-07-06 ### Added diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index bca73c40e..afbbfa42d 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -2132,6 +2132,24 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} } } + if (model) { + const selectedModel = model; + const refreshedModel = await logger.time("refreshInitialModelMetadata", () => + modelRegistry.refreshSelectedModelMetadata(selectedModel), + ); + if (refreshedModel !== selectedModel) { + model = refreshedModel; + thinkingLevel = pickInitialThinkingLevel(refreshedModel); + autoThinking = thinkingLevel === AUTO_THINKING; + effectiveThinkingLevel = concreteThinkingLevel(thinkingLevel); + effectiveThinkingLevel = logger.time("resolveThinkingLevelForModel", () => + autoThinking + ? resolveProvisionalAutoLevel(refreshedModel) + : resolveThinkingLevelForModel(refreshedModel, effectiveThinkingLevel), + ); + } + } + // Discover custom commands (TypeScript slash commands) const customCommandsResult: CustomCommandsLoadResult = options.disableExtensionDiscovery ? { commands: [], errors: [] } diff --git a/packages/coding-agent/test/sdk-model-selection.test.ts b/packages/coding-agent/test/sdk-model-selection.test.ts index 3f73fda19..ab3c0396c 100644 --- a/packages/coding-agent/test/sdk-model-selection.test.ts +++ b/packages/coding-agent/test/sdk-model-selection.test.ts @@ -2,7 +2,9 @@ import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; -import { Effort } from "@oh-my-pi/pi-ai"; +import { Effort, type FetchImpl } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { ModelRegistry, type ProviderConfigInput } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; @@ -334,6 +336,77 @@ describe("createAgentSession deferred model pattern resolution", () => { } }); + test("refreshes cached llama.cpp vision metadata for the startup default model", async () => { + const authStorage = await AuthStorage.create(path.join(tempDir, "llama-vision-auth.db")); + authStoragesToClose.push(authStorage); + const modelsPath = path.join(tempDir, "llama-vision-models.yml"); + const cacheDbPath = path.join(tempDir, "models.db"); + const cachedModel = buildModel({ + id: "vision-model", + name: "vision-model", + provider: "llama.cpp", + api: "openai-responses", + baseUrl: "http://127.0.0.1:8080", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128000, + maxTokens: 32768, + }); + writeModelCache("llama.cpp", Date.now(), [cachedModel], true, "", cacheDbPath); + + const fetchMock: FetchImpl = async input => { + const url = String(input); + if (url === "http://127.0.0.1:8080/models") { + return new Response( + JSON.stringify({ data: [{ id: "vision-model", object: "model", meta: { n_ctx: 239104 } }] }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + } + if (url === "http://127.0.0.1:8080/props") { + return new Response( + JSON.stringify({ + default_generation_settings: { + n_ctx: 239104, + params: { max_tokens: -1, n_predict: -1 }, + }, + modalities: { vision: true }, + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + } + throw new Error(`Unexpected URL: ${url}`); + }; + const modelRegistry = new ModelRegistry(authStorage, modelsPath, { fetch: fetchMock }); + const settings = Settings.isolated(); + settings.setModelRole("default", "llama.cpp/vision-model"); + + expect(modelRegistry.find("llama.cpp", "vision-model")?.input).toEqual(["text"]); + const { session } = await createAgentSession({ + cwd: tempDir, + agentDir: tempDir, + authStorage, + modelRegistry, + settings, + sessionManager: SessionManager.inMemory(), + disableExtensionDiscovery: true, + skills: [], + contextFiles: [], + promptTemplates: [], + slashCommands: [], + enableMCP: false, + enableLsp: false, + skipPythonPreflight: true, + }); + + try { + expect(session.model?.input).toEqual(["text", "image"]); + expect(modelRegistry.find("llama.cpp", "vision-model")?.input).toEqual(["text", "image"]); + } finally { + await session.dispose(); + } + }); + test("restores the saved session model without resolving auth over the network", async () => { // Regression: `restoreSessionModel` probed each saved-model candidate with // the async `getApiKey`, which refreshes OAuth tokens and hits the auth