diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8ed4a7ab7..f950dd0da 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,14 @@ # Changelog ## [Unreleased] +### Added + +- Added `images.describeForTextModels` option (default `true`) to control automatic image description for attachments sent to models without vision input +- Added automatic vision fallback prompts to describe images for text-only models + +### Fixed + +- Fixed image attachment handling for text-only models by saving attachments to `local://` and injecting generated descriptions so they are no longer lost when the target model cannot process images ## [16.0.4] - 2026-06-17 @@ -11843,4 +11851,4 @@ Initial public release. ## [0.7.6] - 2025-11-13 -Previous releases did not maintain a changelog. +Previous releases did not maintain a changelog. \ No newline at end of file diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 99163fa67..e181fc170 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -712,6 +712,18 @@ export const SETTINGS_SCHEMA = { }, }, + "images.describeForTextModels": { + type: "boolean", + default: true, + ui: { + tab: "model", + group: "Vision", + label: "Describe Images for Text Models", + description: + "When an image is attached to a model without vision support, save it under local:// and inject a description from a vision-capable model instead of dropping it", + }, + }, + "tui.maxInlineImageColumns": { type: "number", default: 100, diff --git a/packages/coding-agent/src/prompts/tools/image-attachment-describe-system.md b/packages/coding-agent/src/prompts/tools/image-attachment-describe-system.md new file mode 100644 index 000000000..3f8c1d950 --- /dev/null +++ b/packages/coding-agent/src/prompts/tools/image-attachment-describe-system.md @@ -0,0 +1,8 @@ +You are an image-analysis assistant. The user attached an image to a model that cannot see images, so your description is injected into that model's context in place of the image. The downstream model relies entirely on your text — it never sees the pixels. + +Core behavior: +- Be faithful and evidence-first: distinguish direct observations from inferences. +- Transcribe ALL visible text verbatim, preserving casing, punctuation, and layout order. Mark unreadable segments explicitly rather than guessing. +- NEVER fabricate occluded, blurry, or uncertain details — say what is uncertain. +- Be thorough but compact: prefer dense, information-rich prose over filler. +- Do not add meta commentary, preambles ("This image shows…"), or closing remarks. Output only the description. diff --git a/packages/coding-agent/src/prompts/tools/image-attachment-describe.md b/packages/coding-agent/src/prompts/tools/image-attachment-describe.md new file mode 100644 index 000000000..cbe63fc38 --- /dev/null +++ b/packages/coding-agent/src/prompts/tools/image-attachment-describe.md @@ -0,0 +1,10 @@ +Describe this image in enough detail that a model which cannot see it can reason about its content. + +Cover, where present: +- The overall scene, subject, and what is happening. +- People, objects, and their relationships, positions, colors, and counts. +- All visible text, transcribed verbatim (OCR). +- UI/screenshot elements: labels, buttons, inputs, states, errors, highlighted or disabled controls. +- Diagrams, charts, tables: structure, axes, series, and the values they encode. + +Flag anything ambiguous or unreadable. Output the description as plain prose only. diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 3df41823d..32981a76b 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -261,6 +261,7 @@ import { type EditMode, resolveEditMode } from "../utils/edit-mode"; import { resolveFileDisplayMode } from "../utils/file-display-mode"; import { extractFileMentions, generateFileMentionMessages } from "../utils/file-mentions"; import { normalizeModelContextImages } from "../utils/image-loading"; +import { describeAttachedImagesForTextModel } from "../utils/image-vision-fallback"; import { buildNamedToolChoice, isToolChoiceActive } from "../utils/tool-choice"; import type { AuthStorage } from "./auth-storage"; import type { ClientBridge, ClientBridgePermissionOption, ClientBridgePermissionOutcome } from "./client-bridge"; @@ -976,6 +977,10 @@ const MAGIC_KEYWORD_NOTICE_TYPES: ReadonlySet = new Set([ "workflow-notice", ]); +/** Custom-message type of the hidden companion carrying vision descriptions of image + * attachments sent to a text-only model (see `#buildImageDescriptionNotice`). */ +const IMAGE_ATTACHMENT_DESCRIPTION_TYPE = "image-attachment-description"; + /** * A hidden, user-attributed companion of a queued user prompt: the magic-keyword * notices (`ultrathink`/`orchestrate`/`workflow`) enqueued alongside the user @@ -989,7 +994,7 @@ function isHiddenUserCompanion(message: AgentMessage): boolean { message.role === "custom" && message.attribution === "user" && message.display === false && - MAGIC_KEYWORD_NOTICE_TYPES.has(message.customType) + (MAGIC_KEYWORD_NOTICE_TYPES.has(message.customType) || message.customType === IMAGE_ATTACHMENT_DESCRIPTION_TYPE) ); } @@ -5114,6 +5119,62 @@ export class AgentSession { return normalizeModelContextImages(images, { model: this.model }); } + /** + * Build a hidden companion message describing image attachments for a text-only + * model. Each image is saved under local:// and a vision-capable model describes + * it; the descriptions are returned as a `display: false` custom message (so the + * model reads them but the TUI does not render the blob) carrying one + * `…` block per image. Returns `undefined` when + * the active model already accepts images, the feature is disabled, or no + * description could be produced. Never throws. + */ + async #buildImageDescriptionNotice( + normalizedImages: ImageContent[], + signal?: AbortSignal, + ): Promise { + const model = this.model; + const shouldDescribe = + !!model && + !model.input.includes("image") && + !this.settings.get("images.blockImages") && + this.settings.get("images.describeForTextModels"); + if (!shouldDescribe || !model) { + return undefined; + } + let blocks: TextContent[]; + try { + blocks = await describeAttachedImagesForTextModel( + normalizedImages, + { + activeModel: model, + modelRegistry: this.#modelRegistry, + settings: this.settings, + localProtocolOptions: this.#localProtocolOptions(), + activeModelString: formatModelString(model), + telemetryConfig: this.agent.telemetry, + sessionId: this.sessionId, + }, + signal, + ); + } catch (err) { + logger.warn("image attachment vision fallback failed; image left undescribed", { + error: err instanceof Error ? err.message : String(err), + }); + return undefined; + } + if (blocks.length === 0) { + return undefined; + } + return { + role: "custom", + customType: IMAGE_ATTACHMENT_DESCRIPTION_TYPE, + content: blocks, + display: false, + attribution: "user", + timestamp: Date.now(), + }; + } + async #normalizeMessageContentImages( content: string | (TextContent | ImageContent)[], ): Promise { @@ -5261,9 +5322,14 @@ export class AgentSession { const normalizedImages = await this.#normalizeImagesForModel(options?.images); const userContent: (TextContent | ImageContent)[] = [{ type: "text", text: expandedText }]; - if (normalizedImages) { + if (normalizedImages?.length) { userContent.push(...normalizedImages); } + // Text-only model + image attachment: describe via a vision model and inject the + // description as a hidden companion (the image stays in the visible user message). + const imageDescriptionNotice = normalizedImages?.length + ? await this.#buildImageDescriptionNotice(normalizedImages) + : undefined; const promptAttribution = options?.attribution ?? (options?.synthetic ? "agent" : "user"); const message = options?.synthetic @@ -5288,8 +5354,8 @@ export class AgentSession { ...options, images: normalizedImages, prependMessages: - preludeMessages.length > 0 || keywordNotices.length > 0 - ? [...preludeMessages, ...keywordNotices] + preludeMessages.length > 0 || keywordNotices.length > 0 || imageDescriptionNotice + ? [...preludeMessages, ...keywordNotices, ...(imageDescriptionNotice ? [imageDescriptionNotice] : [])] : undefined, }); } finally { @@ -5699,7 +5765,13 @@ export class AgentSession { if (normalizedImages?.length) { content.push(...normalizedImages); } + // Text-only model + image attachment: describe via a vision model and enqueue the + // description as a hidden companion immediately before the user message. + const imageDescriptionNotice = normalizedImages?.length + ? await this.#buildImageDescriptionNotice(normalizedImages) + : undefined; if (mode === "followUp") { + if (imageDescriptionNotice) this.agent.followUp(imageDescriptionNotice); this.agent.followUp({ role: "user", content, @@ -5707,6 +5779,7 @@ export class AgentSession { timestamp: Date.now(), }); } else { + if (imageDescriptionNotice) this.agent.steer(imageDescriptionNotice); this.agent.steer({ role: "user", content, diff --git a/packages/coding-agent/src/utils/image-vision-fallback.ts b/packages/coding-agent/src/utils/image-vision-fallback.ts new file mode 100644 index 000000000..acf0975b1 --- /dev/null +++ b/packages/coding-agent/src/utils/image-vision-fallback.ts @@ -0,0 +1,197 @@ +/** + * Vision fallback for text-only models. When a user attaches an image to a model + * that cannot accept image input, this: + * 1. saves each image under the session `local://` root (for later analysis), and + * 2. asks a vision-capable model to describe it and injects that description as + * a text block in place of the image: + * + * + * + * + * + * Without this the provider layer drops the image entirely (NON_VISION_IMAGE_PLACEHOLDER). + */ +import * as path from "node:path"; +import { + type AgentTelemetry, + type AgentTelemetryConfig, + instrumentedCompleteSimple, + resolveTelemetry, +} from "@oh-my-pi/pi-agent-core"; +import type { Api, completeSimple, ImageContent, Model, TextContent } from "@oh-my-pi/pi-ai"; +import { logger, prompt, toError } from "@oh-my-pi/pi-utils"; +import { extractTextContent } from "../commit/utils"; +import type { ModelRegistry } from "../config/model-registry"; +import { expandRoleAlias, getModelMatchPreferences, resolveModelFromString } from "../config/model-resolver"; +import type { Settings } from "../config/settings"; +import { type LocalProtocolOptions, resolveLocalRoot } from "../internal-urls"; +import describeUserPrompt from "../prompts/tools/image-attachment-describe.md" with { type: "text" }; +import describeSystemPrompt from "../prompts/tools/image-attachment-describe-system.md" with { type: "text" }; + +/** Telemetry tag for the oneshot vision-description calls. */ +const ONESHOT_KIND = "image_attachment_describe"; + +const NO_VISION_MODEL_NOTE = + "[No vision-capable model is configured, so this image could not be described automatically. " + + "The image was saved; configure a vision model role (modelRoles.vision) and use the inspect_image tool to analyze it.]"; + +const DESCRIPTION_UNAVAILABLE_NOTE = + "[Image description unavailable: the vision model returned no usable text. The image was saved for further analysis.]"; + +/** Registry surface needed to resolve a vision model and authorize requests. */ +export type VisionFallbackRegistry = Pick & + Partial>; + +export interface DescribeAttachedImagesDeps { + /** Active (text-only) model the prompt is destined for. */ + activeModel: Model; + modelRegistry: VisionFallbackRegistry; + settings: Settings; + /** Inputs for resolving the session-scoped `local://` root. */ + localProtocolOptions: LocalProtocolOptions; + /** `provider/id` of the active model; a last-resort vision-model candidate (filtered to image-capable). */ + activeModelString?: string; + telemetryConfig?: AgentTelemetryConfig; + sessionId?: string; + /** Test seam: overrides the underlying completeSimple call. */ + completeImpl?: typeof completeSimple; +} + +/** Map an image MIME type to a file extension for the saved artifact. */ +function extensionForMime(mimeType: string): string { + const subtype = mimeType.split("/")[1]?.toLowerCase() ?? ""; + switch (subtype) { + case "jpeg": + case "jpg": + return "jpg"; + case "png": + return "png"; + case "gif": + return "gif"; + case "webp": + return "webp"; + default: { + const sanitized = subtype.replace(/[^a-z0-9]/g, ""); + return sanitized || "png"; + } + } +} + +/** Content-addressed file name so re-pasting the same image reuses one artifact. */ +function imageFileName(image: ImageContent): string { + const hash = Bun.hash(image.data).toString(16); + return `image-${hash}.${extensionForMime(image.mimeType)}`; +} + +/** Persist an image under the local root; returns its `local://` URL. */ +async function saveImage(image: ImageContent, localRoot: string): Promise { + const fileName = imageFileName(image); + const filePath = path.join(localRoot, fileName); + // Content-addressed: identical bytes overwrite themselves harmlessly. Bun.write creates parent dirs. + await Bun.write(filePath, Buffer.from(image.data, "base64")); + return `local://${fileName}`; +} + +function formatImageBlock(localUrl: string, description: string): string { + return `\n${description}\n`; +} + +/** + * Resolve a vision-capable model, mirroring the inspect_image priority + * (`pi/vision` → `pi/default` → active → first image-capable available), but + * never returning a text-only model. + */ +function resolveVisionModel(deps: DescribeAttachedImagesDeps): Model | undefined { + const available = deps.modelRegistry.getAvailable(); + if (available.length === 0) return undefined; + const preferences = getModelMatchPreferences(deps.settings); + const resolvePattern = (pattern: string | undefined): Model | undefined => { + if (!pattern) return undefined; + const expanded = expandRoleAlias(pattern, deps.settings); + const model = resolveModelFromString(expanded, available, preferences, deps.modelRegistry); + return model?.input.includes("image") ? model : undefined; + }; + return ( + resolvePattern("pi/vision") ?? + resolvePattern("pi/default") ?? + resolvePattern(deps.activeModelString) ?? + available.find(model => model.input.includes("image")) + ); +} + +/** Run one vision-description round-trip; returns trimmed text or `null` on any failure. */ +async function describeImage( + image: ImageContent, + visionModel: Model, + deps: DescribeAttachedImagesDeps, + telemetry: AgentTelemetry | undefined, + signal: AbortSignal | undefined, +): Promise { + try { + const response = await instrumentedCompleteSimple( + visionModel, + { + systemPrompt: [prompt.render(describeSystemPrompt)], + messages: [ + { + role: "user", + content: [ + { type: "image", data: image.data, mimeType: image.mimeType }, + { type: "text", text: prompt.render(describeUserPrompt) }, + ], + timestamp: Date.now(), + }, + ], + }, + { apiKey: deps.modelRegistry.resolver(visionModel, deps.sessionId), signal }, + { telemetry, oneshotKind: ONESHOT_KIND, completeImpl: deps.completeImpl }, + ); + if (response.stopReason === "error" || response.stopReason === "aborted") { + logger.warn("image attachment description did not complete", { + stopReason: response.stopReason, + model: `${visionModel.provider}/${visionModel.id}`, + }); + return null; + } + const text = extractTextContent(response).trim(); + return text.length > 0 ? text : null; + } catch (err) { + logger.warn("image attachment description failed", { + error: toError(err).message, + model: `${visionModel.provider}/${visionModel.id}`, + }); + return null; + } +} + +/** + * Save each attached image under `local://` and replace it with a descriptive + * text block. Returns one {@link TextContent} per input image, in order. Never + * throws for an individual image: a failed description falls back to a note while + * the saved-path block is still emitted. + */ +export async function describeAttachedImagesForTextModel( + images: readonly ImageContent[], + deps: DescribeAttachedImagesDeps, + signal?: AbortSignal, +): Promise { + const localRoot = resolveLocalRoot(deps.localProtocolOptions); + const visionModel = resolveVisionModel(deps); + const apiKey = visionModel ? await deps.modelRegistry.getApiKey(visionModel, deps.sessionId) : undefined; + const canDescribe = Boolean(visionModel && apiKey); + const telemetry = resolveTelemetry(deps.telemetryConfig, deps.sessionId); + + return Promise.all( + images.map(async (image): Promise => { + const localUrl = await saveImage(image, localRoot); + let description: string; + if (canDescribe && visionModel) { + description = + (await describeImage(image, visionModel, deps, telemetry, signal)) ?? DESCRIPTION_UNAVAILABLE_NOTE; + } else { + description = NO_VISION_MODEL_NOTE; + } + return { type: "text", text: formatImageBlock(localUrl, description) }; + }), + ); +} diff --git a/packages/coding-agent/test/agent-session-context-promotion.test.ts b/packages/coding-agent/test/agent-session-context-promotion.test.ts index 06347689b..b6b5081d5 100644 --- a/packages/coding-agent/test/agent-session-context-promotion.test.ts +++ b/packages/coding-agent/test/agent-session-context-promotion.test.ts @@ -416,6 +416,38 @@ describe("AgentSession context promotion", () => { expect(session.providerSessionState.size).toBe(1); }); + it("does not promote by default", async () => { + const sparkModel = modelRegistry.find("openai-codex", "gpt-5.3-codex-spark"); + if (!sparkModel) { + throw new Error("Expected codex spark model to exist"); + } + + const agent = new Agent({ + initialState: { + model: sparkModel, + systemPrompt: ["Test"], + tools: [], + messages: [], + }, + }); + + session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings: Settings.isolated({ "compaction.enabled": false }), + modelRegistry, + }); + + const overflowMessage = createOverflowMessage(sparkModel); + session.agent.emitExternalEvent({ type: "message_end", message: overflowMessage }); + session.agent.emitExternalEvent({ type: "agent_end", messages: [overflowMessage] }); + + await settle(); + + expect(session.model?.provider).toBe(sparkModel.provider); + expect(session.model?.id).toBe(sparkModel.id); + }); + it("promotes to a larger-context model on response.incomplete (length stop)", async () => { const sparkModel = modelRegistry.find("openai-codex", "gpt-5.3-codex-spark"); const codexModel = modelRegistry.find("openai-codex", "gpt-5.5"); diff --git a/packages/coding-agent/test/utils/image-vision-fallback.test.ts b/packages/coding-agent/test/utils/image-vision-fallback.test.ts new file mode 100644 index 000000000..fe8f67470 --- /dev/null +++ b/packages/coding-agent/test/utils/image-vision-fallback.test.ts @@ -0,0 +1,151 @@ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import type { AssistantMessage, completeSimple, Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { + type DescribeAttachedImagesDeps, + describeAttachedImagesForTextModel, +} from "@oh-my-pi/pi-coding-agent/utils/image-vision-fallback"; + +// 1x1 transparent PNG. +const TINY_PNG_BASE64 = + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8DwHwAFBQIAX8jx0gAAAABJRU5ErkJggg=="; + +const visionModel: Model<"openai-responses"> = buildModel({ + id: "gpt-4o", + name: "GPT-4o", + api: "openai-responses", + provider: "openai", + baseUrl: "https://api.openai.com/v1", + reasoning: false, + input: ["text", "image"], + cost: { input: 5, output: 15, cacheRead: 0.5, cacheWrite: 5 }, + contextWindow: 128000, + maxTokens: 4096, +}); + +const textModel: Model<"openai-responses"> = { ...visionModel, id: "gpt-4.1-mini", input: ["text"] }; + +function makeCompleteStub(text: string): { calls: unknown[][]; fn: typeof completeSimple } { + const calls: unknown[][] = []; + const fn = (async (...args: unknown[]) => { + calls.push(args); + return { + role: "assistant", + api: visionModel.api, + provider: visionModel.provider, + model: visionModel.id, + usage: { + input: 1, + output: 1, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 2, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp: Date.now(), + content: [{ type: "text", text }], + } satisfies AssistantMessage; + }) as typeof completeSimple; + return { calls, fn }; +} + +function makeDeps( + artifactsDir: string, + available: Model<"openai-responses">[], + completeImpl: typeof completeSimple, + apiKey: string | undefined = "test-key", +): DescribeAttachedImagesDeps { + return { + activeModel: textModel, + modelRegistry: { + getAvailable: () => available, + getApiKey: async () => apiKey, + resolver: () => async () => apiKey, + } as unknown as DescribeAttachedImagesDeps["modelRegistry"], + settings: Settings.isolated(), + localProtocolOptions: { getArtifactsDir: () => artifactsDir, getSessionId: () => "test-session" }, + activeModelString: `${textModel.provider}/${textModel.id}`, + completeImpl, + }; +} + +describe("describeAttachedImagesForTextModel", () => { + let testDir: string; + + beforeEach(async () => { + testDir = await fs.mkdtemp(path.join(os.tmpdir(), "vision-fallback-")); + }); + + afterEach(async () => { + await fs.rm(testDir, { recursive: true, force: true }); + }); + + it("saves the image under local:// and injects a vision description block", async () => { + const stub = makeCompleteStub("A man holding a red balloon."); + const blocks = await describeAttachedImagesForTextModel( + [{ type: "image", data: TINY_PNG_BASE64, mimeType: "image/png" }], + makeDeps(testDir, [textModel, visionModel], stub.fn), + ); + + expect(blocks).toHaveLength(1); + const text = blocks[0]!.text; + // Block wraps the local:// path and the description. + const match = text.match(/^\n([\s\S]*)\n<\/image>$/); + expect(match).not.toBeNull(); + expect(match![2]).toBe("A man holding a red balloon."); + + // The vision model was actually consulted. + expect(stub.calls).toHaveLength(1); + + // The image was persisted under /local and round-trips. + const fileName = match![1].slice("local://".length); + const saved = await fs.readFile(path.join(testDir, "local", fileName)); + expect(saved.toString("base64")).toBe(TINY_PNG_BASE64); + }); + + it("saves the image but emits a no-vision note when no vision model is available", async () => { + const stub = makeCompleteStub("should not be used"); + const blocks = await describeAttachedImagesForTextModel( + [{ type: "image", data: TINY_PNG_BASE64, mimeType: "image/png" }], + makeDeps(testDir, [textModel], stub.fn), + ); + + expect(blocks).toHaveLength(1); + // No vision-capable model -> no model call, but a note + saved artifact. + expect(stub.calls).toHaveLength(0); + expect(blocks[0]!.text).toContain("No vision-capable model"); + + const match = blocks[0]!.text.match(/path="local:\/\/([^"]+)"/); + expect(match).not.toBeNull(); + const saved = await fs.readFile(path.join(testDir, "local", match![1])); + expect(saved.toString("base64")).toBe(TINY_PNG_BASE64); + }); + + it("falls back to a note when the vision model returns no text", async () => { + const stub = makeCompleteStub(" "); + const blocks = await describeAttachedImagesForTextModel( + [{ type: "image", data: TINY_PNG_BASE64, mimeType: "image/png" }], + makeDeps(testDir, [textModel, visionModel], stub.fn), + ); + + expect(stub.calls).toHaveLength(1); + expect(blocks[0]!.text).toContain("Image description unavailable"); + }); + + it("is content-addressed: identical images reuse one saved file path", async () => { + const image = { type: "image", data: TINY_PNG_BASE64, mimeType: "image/png" } as const; + const stub = makeCompleteStub("desc"); + const blocks = await describeAttachedImagesForTextModel( + [image, image], + makeDeps(testDir, [textModel, visionModel], stub.fn), + ); + + const paths = blocks.map(b => b.text.match(/path="(local:\/\/[^"]+)"/)![1]); + expect(paths[0]).toBe(paths[1]); + }); +});