fix(coding-agent): added vision fallback for text-only model image attachments

- Added `images.describeForTextModels` configuration defaulting to true for text models.
- Added `describeAttachedImagesForTextModel` to persist images and generate local:// descriptions.
- Added image-description notices to session flow with hidden typing and pre-user insertion.
- Added fallback behavior that returns notes when vision is unavailable or output is empty.
This commit is contained in:
can1357
2026-06-17 10:35:48 +02:00
parent 4060be5b44
commit 0b04fda921
8 changed files with 496 additions and 5 deletions
+9 -1
View File
@@ -1,6 +1,14 @@
# Changelog
## [Unreleased]
### Added
- Added `images.describeForTextModels` option (default `true`) to control automatic image description for attachments sent to models without vision input
- Added automatic vision fallback prompts to describe images for text-only models
### Fixed
- Fixed image attachment handling for text-only models by saving attachments to `local://` and injecting generated descriptions so they are no longer lost when the target model cannot process images
## [16.0.4] - 2026-06-17
@@ -11843,4 +11851,4 @@ Initial public release.
## [0.7.6] - 2025-11-13
Previous releases did not maintain a changelog.
Previous releases did not maintain a changelog.
@@ -712,6 +712,18 @@ export const SETTINGS_SCHEMA = {
},
},
"images.describeForTextModels": {
type: "boolean",
default: true,
ui: {
tab: "model",
group: "Vision",
label: "Describe Images for Text Models",
description:
"When an image is attached to a model without vision support, save it under local:// and inject a description from a vision-capable model instead of dropping it",
},
},
"tui.maxInlineImageColumns": {
type: "number",
default: 100,
@@ -0,0 +1,8 @@
You are an image-analysis assistant. The user attached an image to a model that cannot see images, so your description is injected into that model's context in place of the image. The downstream model relies entirely on your text — it never sees the pixels.
Core behavior:
- Be faithful and evidence-first: distinguish direct observations from inferences.
- Transcribe ALL visible text verbatim, preserving casing, punctuation, and layout order. Mark unreadable segments explicitly rather than guessing.
- NEVER fabricate occluded, blurry, or uncertain details — say what is uncertain.
- Be thorough but compact: prefer dense, information-rich prose over filler.
- Do not add meta commentary, preambles ("This image shows…"), or closing remarks. Output only the description.
@@ -0,0 +1,10 @@
Describe this image in enough detail that a model which cannot see it can reason about its content.
Cover, where present:
- The overall scene, subject, and what is happening.
- People, objects, and their relationships, positions, colors, and counts.
- All visible text, transcribed verbatim (OCR).
- UI/screenshot elements: labels, buttons, inputs, states, errors, highlighted or disabled controls.
- Diagrams, charts, tables: structure, axes, series, and the values they encode.
Flag anything ambiguous or unreadable. Output the description as plain prose only.
@@ -261,6 +261,7 @@ import { type EditMode, resolveEditMode } from "../utils/edit-mode";
import { resolveFileDisplayMode } from "../utils/file-display-mode";
import { extractFileMentions, generateFileMentionMessages } from "../utils/file-mentions";
import { normalizeModelContextImages } from "../utils/image-loading";
import { describeAttachedImagesForTextModel } from "../utils/image-vision-fallback";
import { buildNamedToolChoice, isToolChoiceActive } from "../utils/tool-choice";
import type { AuthStorage } from "./auth-storage";
import type { ClientBridge, ClientBridgePermissionOption, ClientBridgePermissionOutcome } from "./client-bridge";
@@ -976,6 +977,10 @@ const MAGIC_KEYWORD_NOTICE_TYPES: ReadonlySet<string> = new Set([
"workflow-notice",
]);
/** Custom-message type of the hidden companion carrying vision descriptions of image
* attachments sent to a text-only model (see `#buildImageDescriptionNotice`). */
const IMAGE_ATTACHMENT_DESCRIPTION_TYPE = "image-attachment-description";
/**
* A hidden, user-attributed companion of a queued user prompt: the magic-keyword
* notices (`ultrathink`/`orchestrate`/`workflow`) enqueued alongside the user
@@ -989,7 +994,7 @@ function isHiddenUserCompanion(message: AgentMessage): boolean {
message.role === "custom" &&
message.attribution === "user" &&
message.display === false &&
MAGIC_KEYWORD_NOTICE_TYPES.has(message.customType)
(MAGIC_KEYWORD_NOTICE_TYPES.has(message.customType) || message.customType === IMAGE_ATTACHMENT_DESCRIPTION_TYPE)
);
}
@@ -5114,6 +5119,62 @@ export class AgentSession {
return normalizeModelContextImages(images, { model: this.model });
}
/**
* Build a hidden companion message describing image attachments for a text-only
* model. Each image is saved under local:// and a vision-capable model describes
* it; the descriptions are returned as a `display: false` custom message (so the
* model reads them but the TUI does not render the blob) carrying one
* `<image path="local://…">…</image>` block per image. Returns `undefined` when
* the active model already accepts images, the feature is disabled, or no
* description could be produced. Never throws.
*/
async #buildImageDescriptionNotice(
normalizedImages: ImageContent[],
signal?: AbortSignal,
): Promise<CustomMessage | undefined> {
const model = this.model;
const shouldDescribe =
!!model &&
!model.input.includes("image") &&
!this.settings.get("images.blockImages") &&
this.settings.get("images.describeForTextModels");
if (!shouldDescribe || !model) {
return undefined;
}
let blocks: TextContent[];
try {
blocks = await describeAttachedImagesForTextModel(
normalizedImages,
{
activeModel: model,
modelRegistry: this.#modelRegistry,
settings: this.settings,
localProtocolOptions: this.#localProtocolOptions(),
activeModelString: formatModelString(model),
telemetryConfig: this.agent.telemetry,
sessionId: this.sessionId,
},
signal,
);
} catch (err) {
logger.warn("image attachment vision fallback failed; image left undescribed", {
error: err instanceof Error ? err.message : String(err),
});
return undefined;
}
if (blocks.length === 0) {
return undefined;
}
return {
role: "custom",
customType: IMAGE_ATTACHMENT_DESCRIPTION_TYPE,
content: blocks,
display: false,
attribution: "user",
timestamp: Date.now(),
};
}
async #normalizeMessageContentImages(
content: string | (TextContent | ImageContent)[],
): Promise<string | (TextContent | ImageContent)[]> {
@@ -5261,9 +5322,14 @@ export class AgentSession {
const normalizedImages = await this.#normalizeImagesForModel(options?.images);
const userContent: (TextContent | ImageContent)[] = [{ type: "text", text: expandedText }];
if (normalizedImages) {
if (normalizedImages?.length) {
userContent.push(...normalizedImages);
}
// Text-only model + image attachment: describe via a vision model and inject the
// description as a hidden companion (the image stays in the visible user message).
const imageDescriptionNotice = normalizedImages?.length
? await this.#buildImageDescriptionNotice(normalizedImages)
: undefined;
const promptAttribution = options?.attribution ?? (options?.synthetic ? "agent" : "user");
const message = options?.synthetic
@@ -5288,8 +5354,8 @@ export class AgentSession {
...options,
images: normalizedImages,
prependMessages:
preludeMessages.length > 0 || keywordNotices.length > 0
? [...preludeMessages, ...keywordNotices]
preludeMessages.length > 0 || keywordNotices.length > 0 || imageDescriptionNotice
? [...preludeMessages, ...keywordNotices, ...(imageDescriptionNotice ? [imageDescriptionNotice] : [])]
: undefined,
});
} finally {
@@ -5699,7 +5765,13 @@ export class AgentSession {
if (normalizedImages?.length) {
content.push(...normalizedImages);
}
// Text-only model + image attachment: describe via a vision model and enqueue the
// description as a hidden companion immediately before the user message.
const imageDescriptionNotice = normalizedImages?.length
? await this.#buildImageDescriptionNotice(normalizedImages)
: undefined;
if (mode === "followUp") {
if (imageDescriptionNotice) this.agent.followUp(imageDescriptionNotice);
this.agent.followUp({
role: "user",
content,
@@ -5707,6 +5779,7 @@ export class AgentSession {
timestamp: Date.now(),
});
} else {
if (imageDescriptionNotice) this.agent.steer(imageDescriptionNotice);
this.agent.steer({
role: "user",
content,
@@ -0,0 +1,197 @@
/**
* Vision fallback for text-only models. When a user attaches an image to a model
* that cannot accept image input, this:
* 1. saves each image under the session `local://` root (for later analysis), and
* 2. asks a vision-capable model to describe it and injects that description as
* a text block in place of the image:
*
* <image path="local://image-<hash>.png">
* <description>
* </image>
*
* Without this the provider layer drops the image entirely (NON_VISION_IMAGE_PLACEHOLDER).
*/
import * as path from "node:path";
import {
type AgentTelemetry,
type AgentTelemetryConfig,
instrumentedCompleteSimple,
resolveTelemetry,
} from "@oh-my-pi/pi-agent-core";
import type { Api, completeSimple, ImageContent, Model, TextContent } from "@oh-my-pi/pi-ai";
import { logger, prompt, toError } from "@oh-my-pi/pi-utils";
import { extractTextContent } from "../commit/utils";
import type { ModelRegistry } from "../config/model-registry";
import { expandRoleAlias, getModelMatchPreferences, resolveModelFromString } from "../config/model-resolver";
import type { Settings } from "../config/settings";
import { type LocalProtocolOptions, resolveLocalRoot } from "../internal-urls";
import describeUserPrompt from "../prompts/tools/image-attachment-describe.md" with { type: "text" };
import describeSystemPrompt from "../prompts/tools/image-attachment-describe-system.md" with { type: "text" };
/** Telemetry tag for the oneshot vision-description calls. */
const ONESHOT_KIND = "image_attachment_describe";
const NO_VISION_MODEL_NOTE =
"[No vision-capable model is configured, so this image could not be described automatically. " +
"The image was saved; configure a vision model role (modelRoles.vision) and use the inspect_image tool to analyze it.]";
const DESCRIPTION_UNAVAILABLE_NOTE =
"[Image description unavailable: the vision model returned no usable text. The image was saved for further analysis.]";
/** Registry surface needed to resolve a vision model and authorize requests. */
export type VisionFallbackRegistry = Pick<ModelRegistry, "getAvailable" | "getApiKey" | "resolver"> &
Partial<Pick<ModelRegistry, "resolveCanonicalModel" | "getCanonicalVariants" | "getCanonicalId">>;
export interface DescribeAttachedImagesDeps {
/** Active (text-only) model the prompt is destined for. */
activeModel: Model<Api>;
modelRegistry: VisionFallbackRegistry;
settings: Settings;
/** Inputs for resolving the session-scoped `local://` root. */
localProtocolOptions: LocalProtocolOptions;
/** `provider/id` of the active model; a last-resort vision-model candidate (filtered to image-capable). */
activeModelString?: string;
telemetryConfig?: AgentTelemetryConfig;
sessionId?: string;
/** Test seam: overrides the underlying completeSimple call. */
completeImpl?: typeof completeSimple;
}
/** Map an image MIME type to a file extension for the saved artifact. */
function extensionForMime(mimeType: string): string {
const subtype = mimeType.split("/")[1]?.toLowerCase() ?? "";
switch (subtype) {
case "jpeg":
case "jpg":
return "jpg";
case "png":
return "png";
case "gif":
return "gif";
case "webp":
return "webp";
default: {
const sanitized = subtype.replace(/[^a-z0-9]/g, "");
return sanitized || "png";
}
}
}
/** Content-addressed file name so re-pasting the same image reuses one artifact. */
function imageFileName(image: ImageContent): string {
const hash = Bun.hash(image.data).toString(16);
return `image-${hash}.${extensionForMime(image.mimeType)}`;
}
/** Persist an image under the local root; returns its `local://` URL. */
async function saveImage(image: ImageContent, localRoot: string): Promise<string> {
const fileName = imageFileName(image);
const filePath = path.join(localRoot, fileName);
// Content-addressed: identical bytes overwrite themselves harmlessly. Bun.write creates parent dirs.
await Bun.write(filePath, Buffer.from(image.data, "base64"));
return `local://${fileName}`;
}
function formatImageBlock(localUrl: string, description: string): string {
return `<image path="${localUrl}">\n${description}\n</image>`;
}
/**
* Resolve a vision-capable model, mirroring the inspect_image priority
* (`pi/vision` → `pi/default` → active → first image-capable available), but
* never returning a text-only model.
*/
function resolveVisionModel(deps: DescribeAttachedImagesDeps): Model<Api> | undefined {
const available = deps.modelRegistry.getAvailable();
if (available.length === 0) return undefined;
const preferences = getModelMatchPreferences(deps.settings);
const resolvePattern = (pattern: string | undefined): Model<Api> | undefined => {
if (!pattern) return undefined;
const expanded = expandRoleAlias(pattern, deps.settings);
const model = resolveModelFromString(expanded, available, preferences, deps.modelRegistry);
return model?.input.includes("image") ? model : undefined;
};
return (
resolvePattern("pi/vision") ??
resolvePattern("pi/default") ??
resolvePattern(deps.activeModelString) ??
available.find(model => model.input.includes("image"))
);
}
/** Run one vision-description round-trip; returns trimmed text or `null` on any failure. */
async function describeImage(
image: ImageContent,
visionModel: Model<Api>,
deps: DescribeAttachedImagesDeps,
telemetry: AgentTelemetry | undefined,
signal: AbortSignal | undefined,
): Promise<string | null> {
try {
const response = await instrumentedCompleteSimple(
visionModel,
{
systemPrompt: [prompt.render(describeSystemPrompt)],
messages: [
{
role: "user",
content: [
{ type: "image", data: image.data, mimeType: image.mimeType },
{ type: "text", text: prompt.render(describeUserPrompt) },
],
timestamp: Date.now(),
},
],
},
{ apiKey: deps.modelRegistry.resolver(visionModel, deps.sessionId), signal },
{ telemetry, oneshotKind: ONESHOT_KIND, completeImpl: deps.completeImpl },
);
if (response.stopReason === "error" || response.stopReason === "aborted") {
logger.warn("image attachment description did not complete", {
stopReason: response.stopReason,
model: `${visionModel.provider}/${visionModel.id}`,
});
return null;
}
const text = extractTextContent(response).trim();
return text.length > 0 ? text : null;
} catch (err) {
logger.warn("image attachment description failed", {
error: toError(err).message,
model: `${visionModel.provider}/${visionModel.id}`,
});
return null;
}
}
/**
* Save each attached image under `local://` and replace it with a descriptive
* text block. Returns one {@link TextContent} per input image, in order. Never
* throws for an individual image: a failed description falls back to a note while
* the saved-path block is still emitted.
*/
export async function describeAttachedImagesForTextModel(
images: readonly ImageContent[],
deps: DescribeAttachedImagesDeps,
signal?: AbortSignal,
): Promise<TextContent[]> {
const localRoot = resolveLocalRoot(deps.localProtocolOptions);
const visionModel = resolveVisionModel(deps);
const apiKey = visionModel ? await deps.modelRegistry.getApiKey(visionModel, deps.sessionId) : undefined;
const canDescribe = Boolean(visionModel && apiKey);
const telemetry = resolveTelemetry(deps.telemetryConfig, deps.sessionId);
return Promise.all(
images.map(async (image): Promise<TextContent> => {
const localUrl = await saveImage(image, localRoot);
let description: string;
if (canDescribe && visionModel) {
description =
(await describeImage(image, visionModel, deps, telemetry, signal)) ?? DESCRIPTION_UNAVAILABLE_NOTE;
} else {
description = NO_VISION_MODEL_NOTE;
}
return { type: "text", text: formatImageBlock(localUrl, description) };
}),
);
}
@@ -416,6 +416,38 @@ describe("AgentSession context promotion", () => {
expect(session.providerSessionState.size).toBe(1);
});
it("does not promote by default", async () => {
const sparkModel = modelRegistry.find("openai-codex", "gpt-5.3-codex-spark");
if (!sparkModel) {
throw new Error("Expected codex spark model to exist");
}
const agent = new Agent({
initialState: {
model: sparkModel,
systemPrompt: ["Test"],
tools: [],
messages: [],
},
});
session = new AgentSession({
agent,
sessionManager: SessionManager.inMemory(),
settings: Settings.isolated({ "compaction.enabled": false }),
modelRegistry,
});
const overflowMessage = createOverflowMessage(sparkModel);
session.agent.emitExternalEvent({ type: "message_end", message: overflowMessage });
session.agent.emitExternalEvent({ type: "agent_end", messages: [overflowMessage] });
await settle();
expect(session.model?.provider).toBe(sparkModel.provider);
expect(session.model?.id).toBe(sparkModel.id);
});
it("promotes to a larger-context model on response.incomplete (length stop)", async () => {
const sparkModel = modelRegistry.find("openai-codex", "gpt-5.3-codex-spark");
const codexModel = modelRegistry.find("openai-codex", "gpt-5.5");
@@ -0,0 +1,151 @@
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import type { AssistantMessage, completeSimple, Model } from "@oh-my-pi/pi-ai";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import {
type DescribeAttachedImagesDeps,
describeAttachedImagesForTextModel,
} from "@oh-my-pi/pi-coding-agent/utils/image-vision-fallback";
// 1x1 transparent PNG.
const TINY_PNG_BASE64 =
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8DwHwAFBQIAX8jx0gAAAABJRU5ErkJggg==";
const visionModel: Model<"openai-responses"> = buildModel({
id: "gpt-4o",
name: "GPT-4o",
api: "openai-responses",
provider: "openai",
baseUrl: "https://api.openai.com/v1",
reasoning: false,
input: ["text", "image"],
cost: { input: 5, output: 15, cacheRead: 0.5, cacheWrite: 5 },
contextWindow: 128000,
maxTokens: 4096,
});
const textModel: Model<"openai-responses"> = { ...visionModel, id: "gpt-4.1-mini", input: ["text"] };
function makeCompleteStub(text: string): { calls: unknown[][]; fn: typeof completeSimple } {
const calls: unknown[][] = [];
const fn = (async (...args: unknown[]) => {
calls.push(args);
return {
role: "assistant",
api: visionModel.api,
provider: visionModel.provider,
model: visionModel.id,
usage: {
input: 1,
output: 1,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 2,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "stop",
timestamp: Date.now(),
content: [{ type: "text", text }],
} satisfies AssistantMessage;
}) as typeof completeSimple;
return { calls, fn };
}
function makeDeps(
artifactsDir: string,
available: Model<"openai-responses">[],
completeImpl: typeof completeSimple,
apiKey: string | undefined = "test-key",
): DescribeAttachedImagesDeps {
return {
activeModel: textModel,
modelRegistry: {
getAvailable: () => available,
getApiKey: async () => apiKey,
resolver: () => async () => apiKey,
} as unknown as DescribeAttachedImagesDeps["modelRegistry"],
settings: Settings.isolated(),
localProtocolOptions: { getArtifactsDir: () => artifactsDir, getSessionId: () => "test-session" },
activeModelString: `${textModel.provider}/${textModel.id}`,
completeImpl,
};
}
describe("describeAttachedImagesForTextModel", () => {
let testDir: string;
beforeEach(async () => {
testDir = await fs.mkdtemp(path.join(os.tmpdir(), "vision-fallback-"));
});
afterEach(async () => {
await fs.rm(testDir, { recursive: true, force: true });
});
it("saves the image under local:// and injects a vision description block", async () => {
const stub = makeCompleteStub("A man holding a red balloon.");
const blocks = await describeAttachedImagesForTextModel(
[{ type: "image", data: TINY_PNG_BASE64, mimeType: "image/png" }],
makeDeps(testDir, [textModel, visionModel], stub.fn),
);
expect(blocks).toHaveLength(1);
const text = blocks[0]!.text;
// Block wraps the local:// path and the description.
const match = text.match(/^<image path="(local:\/\/[^"]+\.png)">\n([\s\S]*)\n<\/image>$/);
expect(match).not.toBeNull();
expect(match![2]).toBe("A man holding a red balloon.");
// The vision model was actually consulted.
expect(stub.calls).toHaveLength(1);
// The image was persisted under <artifactsDir>/local and round-trips.
const fileName = match![1].slice("local://".length);
const saved = await fs.readFile(path.join(testDir, "local", fileName));
expect(saved.toString("base64")).toBe(TINY_PNG_BASE64);
});
it("saves the image but emits a no-vision note when no vision model is available", async () => {
const stub = makeCompleteStub("should not be used");
const blocks = await describeAttachedImagesForTextModel(
[{ type: "image", data: TINY_PNG_BASE64, mimeType: "image/png" }],
makeDeps(testDir, [textModel], stub.fn),
);
expect(blocks).toHaveLength(1);
// No vision-capable model -> no model call, but a note + saved artifact.
expect(stub.calls).toHaveLength(0);
expect(blocks[0]!.text).toContain("No vision-capable model");
const match = blocks[0]!.text.match(/path="local:\/\/([^"]+)"/);
expect(match).not.toBeNull();
const saved = await fs.readFile(path.join(testDir, "local", match![1]));
expect(saved.toString("base64")).toBe(TINY_PNG_BASE64);
});
it("falls back to a note when the vision model returns no text", async () => {
const stub = makeCompleteStub(" ");
const blocks = await describeAttachedImagesForTextModel(
[{ type: "image", data: TINY_PNG_BASE64, mimeType: "image/png" }],
makeDeps(testDir, [textModel, visionModel], stub.fn),
);
expect(stub.calls).toHaveLength(1);
expect(blocks[0]!.text).toContain("Image description unavailable");
});
it("is content-addressed: identical images reuse one saved file path", async () => {
const image = { type: "image", data: TINY_PNG_BASE64, mimeType: "image/png" } as const;
const stub = makeCompleteStub("desc");
const blocks = await describeAttachedImagesForTextModel(
[image, image],
makeDeps(testDir, [textModel, visionModel], stub.fn),
);
const paths = blocks.map(b => b.text.match(/path="(local:\/\/[^"]+)"/)![1]);
expect(paths[0]).toBe(paths[1]);
});
});