fix(coding-agent): added vision fallback for text-only model image attachments
- Added `images.describeForTextModels` configuration defaulting to true for text models. - Added `describeAttachedImagesForTextModel` to persist images and generate local:// descriptions. - Added image-description notices to session flow with hidden typing and pre-user insertion. - Added fallback behavior that returns notes when vision is unavailable or output is empty.
This commit is contained in:
@@ -1,6 +1,14 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Added
|
||||
|
||||
- Added `images.describeForTextModels` option (default `true`) to control automatic image description for attachments sent to models without vision input
|
||||
- Added automatic vision fallback prompts to describe images for text-only models
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed image attachment handling for text-only models by saving attachments to `local://` and injecting generated descriptions so they are no longer lost when the target model cannot process images
|
||||
|
||||
## [16.0.4] - 2026-06-17
|
||||
|
||||
@@ -11843,4 +11851,4 @@ Initial public release.
|
||||
|
||||
## [0.7.6] - 2025-11-13
|
||||
|
||||
Previous releases did not maintain a changelog.
|
||||
Previous releases did not maintain a changelog.
|
||||
@@ -712,6 +712,18 @@ export const SETTINGS_SCHEMA = {
|
||||
},
|
||||
},
|
||||
|
||||
"images.describeForTextModels": {
|
||||
type: "boolean",
|
||||
default: true,
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Vision",
|
||||
label: "Describe Images for Text Models",
|
||||
description:
|
||||
"When an image is attached to a model without vision support, save it under local:// and inject a description from a vision-capable model instead of dropping it",
|
||||
},
|
||||
},
|
||||
|
||||
"tui.maxInlineImageColumns": {
|
||||
type: "number",
|
||||
default: 100,
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
You are an image-analysis assistant. The user attached an image to a model that cannot see images, so your description is injected into that model's context in place of the image. The downstream model relies entirely on your text — it never sees the pixels.
|
||||
|
||||
Core behavior:
|
||||
- Be faithful and evidence-first: distinguish direct observations from inferences.
|
||||
- Transcribe ALL visible text verbatim, preserving casing, punctuation, and layout order. Mark unreadable segments explicitly rather than guessing.
|
||||
- NEVER fabricate occluded, blurry, or uncertain details — say what is uncertain.
|
||||
- Be thorough but compact: prefer dense, information-rich prose over filler.
|
||||
- Do not add meta commentary, preambles ("This image shows…"), or closing remarks. Output only the description.
|
||||
@@ -0,0 +1,10 @@
|
||||
Describe this image in enough detail that a model which cannot see it can reason about its content.
|
||||
|
||||
Cover, where present:
|
||||
- The overall scene, subject, and what is happening.
|
||||
- People, objects, and their relationships, positions, colors, and counts.
|
||||
- All visible text, transcribed verbatim (OCR).
|
||||
- UI/screenshot elements: labels, buttons, inputs, states, errors, highlighted or disabled controls.
|
||||
- Diagrams, charts, tables: structure, axes, series, and the values they encode.
|
||||
|
||||
Flag anything ambiguous or unreadable. Output the description as plain prose only.
|
||||
@@ -261,6 +261,7 @@ import { type EditMode, resolveEditMode } from "../utils/edit-mode";
|
||||
import { resolveFileDisplayMode } from "../utils/file-display-mode";
|
||||
import { extractFileMentions, generateFileMentionMessages } from "../utils/file-mentions";
|
||||
import { normalizeModelContextImages } from "../utils/image-loading";
|
||||
import { describeAttachedImagesForTextModel } from "../utils/image-vision-fallback";
|
||||
import { buildNamedToolChoice, isToolChoiceActive } from "../utils/tool-choice";
|
||||
import type { AuthStorage } from "./auth-storage";
|
||||
import type { ClientBridge, ClientBridgePermissionOption, ClientBridgePermissionOutcome } from "./client-bridge";
|
||||
@@ -976,6 +977,10 @@ const MAGIC_KEYWORD_NOTICE_TYPES: ReadonlySet<string> = new Set([
|
||||
"workflow-notice",
|
||||
]);
|
||||
|
||||
/** Custom-message type of the hidden companion carrying vision descriptions of image
|
||||
* attachments sent to a text-only model (see `#buildImageDescriptionNotice`). */
|
||||
const IMAGE_ATTACHMENT_DESCRIPTION_TYPE = "image-attachment-description";
|
||||
|
||||
/**
|
||||
* A hidden, user-attributed companion of a queued user prompt: the magic-keyword
|
||||
* notices (`ultrathink`/`orchestrate`/`workflow`) enqueued alongside the user
|
||||
@@ -989,7 +994,7 @@ function isHiddenUserCompanion(message: AgentMessage): boolean {
|
||||
message.role === "custom" &&
|
||||
message.attribution === "user" &&
|
||||
message.display === false &&
|
||||
MAGIC_KEYWORD_NOTICE_TYPES.has(message.customType)
|
||||
(MAGIC_KEYWORD_NOTICE_TYPES.has(message.customType) || message.customType === IMAGE_ATTACHMENT_DESCRIPTION_TYPE)
|
||||
);
|
||||
}
|
||||
|
||||
@@ -5114,6 +5119,62 @@ export class AgentSession {
|
||||
return normalizeModelContextImages(images, { model: this.model });
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a hidden companion message describing image attachments for a text-only
|
||||
* model. Each image is saved under local:// and a vision-capable model describes
|
||||
* it; the descriptions are returned as a `display: false` custom message (so the
|
||||
* model reads them but the TUI does not render the blob) carrying one
|
||||
* `<image path="local://…">…</image>` block per image. Returns `undefined` when
|
||||
* the active model already accepts images, the feature is disabled, or no
|
||||
* description could be produced. Never throws.
|
||||
*/
|
||||
async #buildImageDescriptionNotice(
|
||||
normalizedImages: ImageContent[],
|
||||
signal?: AbortSignal,
|
||||
): Promise<CustomMessage | undefined> {
|
||||
const model = this.model;
|
||||
const shouldDescribe =
|
||||
!!model &&
|
||||
!model.input.includes("image") &&
|
||||
!this.settings.get("images.blockImages") &&
|
||||
this.settings.get("images.describeForTextModels");
|
||||
if (!shouldDescribe || !model) {
|
||||
return undefined;
|
||||
}
|
||||
let blocks: TextContent[];
|
||||
try {
|
||||
blocks = await describeAttachedImagesForTextModel(
|
||||
normalizedImages,
|
||||
{
|
||||
activeModel: model,
|
||||
modelRegistry: this.#modelRegistry,
|
||||
settings: this.settings,
|
||||
localProtocolOptions: this.#localProtocolOptions(),
|
||||
activeModelString: formatModelString(model),
|
||||
telemetryConfig: this.agent.telemetry,
|
||||
sessionId: this.sessionId,
|
||||
},
|
||||
signal,
|
||||
);
|
||||
} catch (err) {
|
||||
logger.warn("image attachment vision fallback failed; image left undescribed", {
|
||||
error: err instanceof Error ? err.message : String(err),
|
||||
});
|
||||
return undefined;
|
||||
}
|
||||
if (blocks.length === 0) {
|
||||
return undefined;
|
||||
}
|
||||
return {
|
||||
role: "custom",
|
||||
customType: IMAGE_ATTACHMENT_DESCRIPTION_TYPE,
|
||||
content: blocks,
|
||||
display: false,
|
||||
attribution: "user",
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
}
|
||||
|
||||
async #normalizeMessageContentImages(
|
||||
content: string | (TextContent | ImageContent)[],
|
||||
): Promise<string | (TextContent | ImageContent)[]> {
|
||||
@@ -5261,9 +5322,14 @@ export class AgentSession {
|
||||
const normalizedImages = await this.#normalizeImagesForModel(options?.images);
|
||||
|
||||
const userContent: (TextContent | ImageContent)[] = [{ type: "text", text: expandedText }];
|
||||
if (normalizedImages) {
|
||||
if (normalizedImages?.length) {
|
||||
userContent.push(...normalizedImages);
|
||||
}
|
||||
// Text-only model + image attachment: describe via a vision model and inject the
|
||||
// description as a hidden companion (the image stays in the visible user message).
|
||||
const imageDescriptionNotice = normalizedImages?.length
|
||||
? await this.#buildImageDescriptionNotice(normalizedImages)
|
||||
: undefined;
|
||||
|
||||
const promptAttribution = options?.attribution ?? (options?.synthetic ? "agent" : "user");
|
||||
const message = options?.synthetic
|
||||
@@ -5288,8 +5354,8 @@ export class AgentSession {
|
||||
...options,
|
||||
images: normalizedImages,
|
||||
prependMessages:
|
||||
preludeMessages.length > 0 || keywordNotices.length > 0
|
||||
? [...preludeMessages, ...keywordNotices]
|
||||
preludeMessages.length > 0 || keywordNotices.length > 0 || imageDescriptionNotice
|
||||
? [...preludeMessages, ...keywordNotices, ...(imageDescriptionNotice ? [imageDescriptionNotice] : [])]
|
||||
: undefined,
|
||||
});
|
||||
} finally {
|
||||
@@ -5699,7 +5765,13 @@ export class AgentSession {
|
||||
if (normalizedImages?.length) {
|
||||
content.push(...normalizedImages);
|
||||
}
|
||||
// Text-only model + image attachment: describe via a vision model and enqueue the
|
||||
// description as a hidden companion immediately before the user message.
|
||||
const imageDescriptionNotice = normalizedImages?.length
|
||||
? await this.#buildImageDescriptionNotice(normalizedImages)
|
||||
: undefined;
|
||||
if (mode === "followUp") {
|
||||
if (imageDescriptionNotice) this.agent.followUp(imageDescriptionNotice);
|
||||
this.agent.followUp({
|
||||
role: "user",
|
||||
content,
|
||||
@@ -5707,6 +5779,7 @@ export class AgentSession {
|
||||
timestamp: Date.now(),
|
||||
});
|
||||
} else {
|
||||
if (imageDescriptionNotice) this.agent.steer(imageDescriptionNotice);
|
||||
this.agent.steer({
|
||||
role: "user",
|
||||
content,
|
||||
|
||||
@@ -0,0 +1,197 @@
|
||||
/**
|
||||
* Vision fallback for text-only models. When a user attaches an image to a model
|
||||
* that cannot accept image input, this:
|
||||
* 1. saves each image under the session `local://` root (for later analysis), and
|
||||
* 2. asks a vision-capable model to describe it and injects that description as
|
||||
* a text block in place of the image:
|
||||
*
|
||||
* <image path="local://image-<hash>.png">
|
||||
* <description>
|
||||
* </image>
|
||||
*
|
||||
* Without this the provider layer drops the image entirely (NON_VISION_IMAGE_PLACEHOLDER).
|
||||
*/
|
||||
import * as path from "node:path";
|
||||
import {
|
||||
type AgentTelemetry,
|
||||
type AgentTelemetryConfig,
|
||||
instrumentedCompleteSimple,
|
||||
resolveTelemetry,
|
||||
} from "@oh-my-pi/pi-agent-core";
|
||||
import type { Api, completeSimple, ImageContent, Model, TextContent } from "@oh-my-pi/pi-ai";
|
||||
import { logger, prompt, toError } from "@oh-my-pi/pi-utils";
|
||||
import { extractTextContent } from "../commit/utils";
|
||||
import type { ModelRegistry } from "../config/model-registry";
|
||||
import { expandRoleAlias, getModelMatchPreferences, resolveModelFromString } from "../config/model-resolver";
|
||||
import type { Settings } from "../config/settings";
|
||||
import { type LocalProtocolOptions, resolveLocalRoot } from "../internal-urls";
|
||||
import describeUserPrompt from "../prompts/tools/image-attachment-describe.md" with { type: "text" };
|
||||
import describeSystemPrompt from "../prompts/tools/image-attachment-describe-system.md" with { type: "text" };
|
||||
|
||||
/** Telemetry tag for the oneshot vision-description calls. */
|
||||
const ONESHOT_KIND = "image_attachment_describe";
|
||||
|
||||
const NO_VISION_MODEL_NOTE =
|
||||
"[No vision-capable model is configured, so this image could not be described automatically. " +
|
||||
"The image was saved; configure a vision model role (modelRoles.vision) and use the inspect_image tool to analyze it.]";
|
||||
|
||||
const DESCRIPTION_UNAVAILABLE_NOTE =
|
||||
"[Image description unavailable: the vision model returned no usable text. The image was saved for further analysis.]";
|
||||
|
||||
/** Registry surface needed to resolve a vision model and authorize requests. */
|
||||
export type VisionFallbackRegistry = Pick<ModelRegistry, "getAvailable" | "getApiKey" | "resolver"> &
|
||||
Partial<Pick<ModelRegistry, "resolveCanonicalModel" | "getCanonicalVariants" | "getCanonicalId">>;
|
||||
|
||||
export interface DescribeAttachedImagesDeps {
|
||||
/** Active (text-only) model the prompt is destined for. */
|
||||
activeModel: Model<Api>;
|
||||
modelRegistry: VisionFallbackRegistry;
|
||||
settings: Settings;
|
||||
/** Inputs for resolving the session-scoped `local://` root. */
|
||||
localProtocolOptions: LocalProtocolOptions;
|
||||
/** `provider/id` of the active model; a last-resort vision-model candidate (filtered to image-capable). */
|
||||
activeModelString?: string;
|
||||
telemetryConfig?: AgentTelemetryConfig;
|
||||
sessionId?: string;
|
||||
/** Test seam: overrides the underlying completeSimple call. */
|
||||
completeImpl?: typeof completeSimple;
|
||||
}
|
||||
|
||||
/** Map an image MIME type to a file extension for the saved artifact. */
|
||||
function extensionForMime(mimeType: string): string {
|
||||
const subtype = mimeType.split("/")[1]?.toLowerCase() ?? "";
|
||||
switch (subtype) {
|
||||
case "jpeg":
|
||||
case "jpg":
|
||||
return "jpg";
|
||||
case "png":
|
||||
return "png";
|
||||
case "gif":
|
||||
return "gif";
|
||||
case "webp":
|
||||
return "webp";
|
||||
default: {
|
||||
const sanitized = subtype.replace(/[^a-z0-9]/g, "");
|
||||
return sanitized || "png";
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Content-addressed file name so re-pasting the same image reuses one artifact. */
|
||||
function imageFileName(image: ImageContent): string {
|
||||
const hash = Bun.hash(image.data).toString(16);
|
||||
return `image-${hash}.${extensionForMime(image.mimeType)}`;
|
||||
}
|
||||
|
||||
/** Persist an image under the local root; returns its `local://` URL. */
|
||||
async function saveImage(image: ImageContent, localRoot: string): Promise<string> {
|
||||
const fileName = imageFileName(image);
|
||||
const filePath = path.join(localRoot, fileName);
|
||||
// Content-addressed: identical bytes overwrite themselves harmlessly. Bun.write creates parent dirs.
|
||||
await Bun.write(filePath, Buffer.from(image.data, "base64"));
|
||||
return `local://${fileName}`;
|
||||
}
|
||||
|
||||
function formatImageBlock(localUrl: string, description: string): string {
|
||||
return `<image path="${localUrl}">\n${description}\n</image>`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a vision-capable model, mirroring the inspect_image priority
|
||||
* (`pi/vision` → `pi/default` → active → first image-capable available), but
|
||||
* never returning a text-only model.
|
||||
*/
|
||||
function resolveVisionModel(deps: DescribeAttachedImagesDeps): Model<Api> | undefined {
|
||||
const available = deps.modelRegistry.getAvailable();
|
||||
if (available.length === 0) return undefined;
|
||||
const preferences = getModelMatchPreferences(deps.settings);
|
||||
const resolvePattern = (pattern: string | undefined): Model<Api> | undefined => {
|
||||
if (!pattern) return undefined;
|
||||
const expanded = expandRoleAlias(pattern, deps.settings);
|
||||
const model = resolveModelFromString(expanded, available, preferences, deps.modelRegistry);
|
||||
return model?.input.includes("image") ? model : undefined;
|
||||
};
|
||||
return (
|
||||
resolvePattern("pi/vision") ??
|
||||
resolvePattern("pi/default") ??
|
||||
resolvePattern(deps.activeModelString) ??
|
||||
available.find(model => model.input.includes("image"))
|
||||
);
|
||||
}
|
||||
|
||||
/** Run one vision-description round-trip; returns trimmed text or `null` on any failure. */
|
||||
async function describeImage(
|
||||
image: ImageContent,
|
||||
visionModel: Model<Api>,
|
||||
deps: DescribeAttachedImagesDeps,
|
||||
telemetry: AgentTelemetry | undefined,
|
||||
signal: AbortSignal | undefined,
|
||||
): Promise<string | null> {
|
||||
try {
|
||||
const response = await instrumentedCompleteSimple(
|
||||
visionModel,
|
||||
{
|
||||
systemPrompt: [prompt.render(describeSystemPrompt)],
|
||||
messages: [
|
||||
{
|
||||
role: "user",
|
||||
content: [
|
||||
{ type: "image", data: image.data, mimeType: image.mimeType },
|
||||
{ type: "text", text: prompt.render(describeUserPrompt) },
|
||||
],
|
||||
timestamp: Date.now(),
|
||||
},
|
||||
],
|
||||
},
|
||||
{ apiKey: deps.modelRegistry.resolver(visionModel, deps.sessionId), signal },
|
||||
{ telemetry, oneshotKind: ONESHOT_KIND, completeImpl: deps.completeImpl },
|
||||
);
|
||||
if (response.stopReason === "error" || response.stopReason === "aborted") {
|
||||
logger.warn("image attachment description did not complete", {
|
||||
stopReason: response.stopReason,
|
||||
model: `${visionModel.provider}/${visionModel.id}`,
|
||||
});
|
||||
return null;
|
||||
}
|
||||
const text = extractTextContent(response).trim();
|
||||
return text.length > 0 ? text : null;
|
||||
} catch (err) {
|
||||
logger.warn("image attachment description failed", {
|
||||
error: toError(err).message,
|
||||
model: `${visionModel.provider}/${visionModel.id}`,
|
||||
});
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Save each attached image under `local://` and replace it with a descriptive
|
||||
* text block. Returns one {@link TextContent} per input image, in order. Never
|
||||
* throws for an individual image: a failed description falls back to a note while
|
||||
* the saved-path block is still emitted.
|
||||
*/
|
||||
export async function describeAttachedImagesForTextModel(
|
||||
images: readonly ImageContent[],
|
||||
deps: DescribeAttachedImagesDeps,
|
||||
signal?: AbortSignal,
|
||||
): Promise<TextContent[]> {
|
||||
const localRoot = resolveLocalRoot(deps.localProtocolOptions);
|
||||
const visionModel = resolveVisionModel(deps);
|
||||
const apiKey = visionModel ? await deps.modelRegistry.getApiKey(visionModel, deps.sessionId) : undefined;
|
||||
const canDescribe = Boolean(visionModel && apiKey);
|
||||
const telemetry = resolveTelemetry(deps.telemetryConfig, deps.sessionId);
|
||||
|
||||
return Promise.all(
|
||||
images.map(async (image): Promise<TextContent> => {
|
||||
const localUrl = await saveImage(image, localRoot);
|
||||
let description: string;
|
||||
if (canDescribe && visionModel) {
|
||||
description =
|
||||
(await describeImage(image, visionModel, deps, telemetry, signal)) ?? DESCRIPTION_UNAVAILABLE_NOTE;
|
||||
} else {
|
||||
description = NO_VISION_MODEL_NOTE;
|
||||
}
|
||||
return { type: "text", text: formatImageBlock(localUrl, description) };
|
||||
}),
|
||||
);
|
||||
}
|
||||
@@ -416,6 +416,38 @@ describe("AgentSession context promotion", () => {
|
||||
expect(session.providerSessionState.size).toBe(1);
|
||||
});
|
||||
|
||||
it("does not promote by default", async () => {
|
||||
const sparkModel = modelRegistry.find("openai-codex", "gpt-5.3-codex-spark");
|
||||
if (!sparkModel) {
|
||||
throw new Error("Expected codex spark model to exist");
|
||||
}
|
||||
|
||||
const agent = new Agent({
|
||||
initialState: {
|
||||
model: sparkModel,
|
||||
systemPrompt: ["Test"],
|
||||
tools: [],
|
||||
messages: [],
|
||||
},
|
||||
});
|
||||
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
sessionManager: SessionManager.inMemory(),
|
||||
settings: Settings.isolated({ "compaction.enabled": false }),
|
||||
modelRegistry,
|
||||
});
|
||||
|
||||
const overflowMessage = createOverflowMessage(sparkModel);
|
||||
session.agent.emitExternalEvent({ type: "message_end", message: overflowMessage });
|
||||
session.agent.emitExternalEvent({ type: "agent_end", messages: [overflowMessage] });
|
||||
|
||||
await settle();
|
||||
|
||||
expect(session.model?.provider).toBe(sparkModel.provider);
|
||||
expect(session.model?.id).toBe(sparkModel.id);
|
||||
});
|
||||
|
||||
it("promotes to a larger-context model on response.incomplete (length stop)", async () => {
|
||||
const sparkModel = modelRegistry.find("openai-codex", "gpt-5.3-codex-spark");
|
||||
const codexModel = modelRegistry.find("openai-codex", "gpt-5.5");
|
||||
|
||||
@@ -0,0 +1,151 @@
|
||||
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import type { AssistantMessage, completeSimple, Model } from "@oh-my-pi/pi-ai";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import {
|
||||
type DescribeAttachedImagesDeps,
|
||||
describeAttachedImagesForTextModel,
|
||||
} from "@oh-my-pi/pi-coding-agent/utils/image-vision-fallback";
|
||||
|
||||
// 1x1 transparent PNG.
|
||||
const TINY_PNG_BASE64 =
|
||||
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8DwHwAFBQIAX8jx0gAAAABJRU5ErkJggg==";
|
||||
|
||||
const visionModel: Model<"openai-responses"> = buildModel({
|
||||
id: "gpt-4o",
|
||||
name: "GPT-4o",
|
||||
api: "openai-responses",
|
||||
provider: "openai",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
reasoning: false,
|
||||
input: ["text", "image"],
|
||||
cost: { input: 5, output: 15, cacheRead: 0.5, cacheWrite: 5 },
|
||||
contextWindow: 128000,
|
||||
maxTokens: 4096,
|
||||
});
|
||||
|
||||
const textModel: Model<"openai-responses"> = { ...visionModel, id: "gpt-4.1-mini", input: ["text"] };
|
||||
|
||||
function makeCompleteStub(text: string): { calls: unknown[][]; fn: typeof completeSimple } {
|
||||
const calls: unknown[][] = [];
|
||||
const fn = (async (...args: unknown[]) => {
|
||||
calls.push(args);
|
||||
return {
|
||||
role: "assistant",
|
||||
api: visionModel.api,
|
||||
provider: visionModel.provider,
|
||||
model: visionModel.id,
|
||||
usage: {
|
||||
input: 1,
|
||||
output: 1,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 2,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "stop",
|
||||
timestamp: Date.now(),
|
||||
content: [{ type: "text", text }],
|
||||
} satisfies AssistantMessage;
|
||||
}) as typeof completeSimple;
|
||||
return { calls, fn };
|
||||
}
|
||||
|
||||
function makeDeps(
|
||||
artifactsDir: string,
|
||||
available: Model<"openai-responses">[],
|
||||
completeImpl: typeof completeSimple,
|
||||
apiKey: string | undefined = "test-key",
|
||||
): DescribeAttachedImagesDeps {
|
||||
return {
|
||||
activeModel: textModel,
|
||||
modelRegistry: {
|
||||
getAvailable: () => available,
|
||||
getApiKey: async () => apiKey,
|
||||
resolver: () => async () => apiKey,
|
||||
} as unknown as DescribeAttachedImagesDeps["modelRegistry"],
|
||||
settings: Settings.isolated(),
|
||||
localProtocolOptions: { getArtifactsDir: () => artifactsDir, getSessionId: () => "test-session" },
|
||||
activeModelString: `${textModel.provider}/${textModel.id}`,
|
||||
completeImpl,
|
||||
};
|
||||
}
|
||||
|
||||
describe("describeAttachedImagesForTextModel", () => {
|
||||
let testDir: string;
|
||||
|
||||
beforeEach(async () => {
|
||||
testDir = await fs.mkdtemp(path.join(os.tmpdir(), "vision-fallback-"));
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
await fs.rm(testDir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
it("saves the image under local:// and injects a vision description block", async () => {
|
||||
const stub = makeCompleteStub("A man holding a red balloon.");
|
||||
const blocks = await describeAttachedImagesForTextModel(
|
||||
[{ type: "image", data: TINY_PNG_BASE64, mimeType: "image/png" }],
|
||||
makeDeps(testDir, [textModel, visionModel], stub.fn),
|
||||
);
|
||||
|
||||
expect(blocks).toHaveLength(1);
|
||||
const text = blocks[0]!.text;
|
||||
// Block wraps the local:// path and the description.
|
||||
const match = text.match(/^<image path="(local:\/\/[^"]+\.png)">\n([\s\S]*)\n<\/image>$/);
|
||||
expect(match).not.toBeNull();
|
||||
expect(match![2]).toBe("A man holding a red balloon.");
|
||||
|
||||
// The vision model was actually consulted.
|
||||
expect(stub.calls).toHaveLength(1);
|
||||
|
||||
// The image was persisted under <artifactsDir>/local and round-trips.
|
||||
const fileName = match![1].slice("local://".length);
|
||||
const saved = await fs.readFile(path.join(testDir, "local", fileName));
|
||||
expect(saved.toString("base64")).toBe(TINY_PNG_BASE64);
|
||||
});
|
||||
|
||||
it("saves the image but emits a no-vision note when no vision model is available", async () => {
|
||||
const stub = makeCompleteStub("should not be used");
|
||||
const blocks = await describeAttachedImagesForTextModel(
|
||||
[{ type: "image", data: TINY_PNG_BASE64, mimeType: "image/png" }],
|
||||
makeDeps(testDir, [textModel], stub.fn),
|
||||
);
|
||||
|
||||
expect(blocks).toHaveLength(1);
|
||||
// No vision-capable model -> no model call, but a note + saved artifact.
|
||||
expect(stub.calls).toHaveLength(0);
|
||||
expect(blocks[0]!.text).toContain("No vision-capable model");
|
||||
|
||||
const match = blocks[0]!.text.match(/path="local:\/\/([^"]+)"/);
|
||||
expect(match).not.toBeNull();
|
||||
const saved = await fs.readFile(path.join(testDir, "local", match![1]));
|
||||
expect(saved.toString("base64")).toBe(TINY_PNG_BASE64);
|
||||
});
|
||||
|
||||
it("falls back to a note when the vision model returns no text", async () => {
|
||||
const stub = makeCompleteStub(" ");
|
||||
const blocks = await describeAttachedImagesForTextModel(
|
||||
[{ type: "image", data: TINY_PNG_BASE64, mimeType: "image/png" }],
|
||||
makeDeps(testDir, [textModel, visionModel], stub.fn),
|
||||
);
|
||||
|
||||
expect(stub.calls).toHaveLength(1);
|
||||
expect(blocks[0]!.text).toContain("Image description unavailable");
|
||||
});
|
||||
|
||||
it("is content-addressed: identical images reuse one saved file path", async () => {
|
||||
const image = { type: "image", data: TINY_PNG_BASE64, mimeType: "image/png" } as const;
|
||||
const stub = makeCompleteStub("desc");
|
||||
const blocks = await describeAttachedImagesForTextModel(
|
||||
[image, image],
|
||||
makeDeps(testDir, [textModel, visionModel], stub.fn),
|
||||
);
|
||||
|
||||
const paths = blocks.map(b => b.text.match(/path="(local:\/\/[^"]+)"/)![1]);
|
||||
expect(paths[0]).toBe(paths[1]);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user