From 0507cf1ba3dc79d7d13c08c746eeb3a66aaff1b0 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 10 May 2026 05:01:52 +0200 Subject: [PATCH] fix(ai): replaced silent image drop with placeholder for non-vision models - Image blocks sent to models without vision support now emit `[image omitted: model does not support vision]` instead of being silently dropped, preventing opaque 404 errors. - Fix applies to user messages and tool-result payloads across Anthropic, OpenAI completions/responses, Codex, and Google providers. - Extracted shared `vision-guard.ts` with `partitionVisionContent`, `joinTextWithImagePlaceholder`, and `NON_VISION_IMAGE_PLACEHOLDER`. - Added regression tests covering all five provider paths. Fixes #967 Fixes #968 --- packages/ai/CHANGELOG.md | 3 + packages/ai/src/providers/anthropic.ts | 85 +++--- packages/ai/src/providers/google-shared.ts | 49 +-- .../src/providers/openai-codex-responses.ts | 33 +-- .../ai/src/providers/openai-completions.ts | 38 ++- .../src/providers/openai-responses-shared.ts | 53 ++-- packages/ai/src/providers/vision-guard.ts | 31 ++ .../ai/test/issue-967-vision-guard.test.ts | 278 ++++++++++++++++++ 8 files changed, 455 insertions(+), 115 deletions(-) create mode 100644 packages/ai/src/providers/vision-guard.ts create mode 100644 packages/ai/test/issue-967-vision-guard.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 499896653..e4ed18054 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,6 +4,9 @@ ### Added +### Fixed +- Fixed silent forwarding of image content (for example Python plot output rendered in the terminal) to models without vision support, which produced opaque 404 errors from upstream. Image blocks are now stripped and replaced with a `[image omitted: model does not support vision]` placeholder for non-vision models, including tool-result payloads ([#967](https://github.com/can1357/oh-my-pi/issues/967), [#968](https://github.com/can1357/oh-my-pi/issues/968)). + - Added `AuthStorage` `onCredentialDisabled` callback (sync or async) so embedders can react when a credential is automatically disabled (e.g. OAuth refresh fails with `invalid_grant`) — useful for surfacing a banner or auto-launching a re-login flow instead of letting the credential silently disappear. Sync throws and async rejections are both caught and logged so a misbehaving subscriber cannot break the disable path. - Added Anthropic OAuth `account.uuid` and `account.email_address` extraction from the `/v1/oauth/token` exchange and refresh responses; both `AnthropicOAuthFlow.exchangeToken()` and `refreshAnthropicToken()` now populate `OAuthCredentials.{accountId, email}` so downstream consumers can attribute requests to the authenticated account without a separate `/api/oauth/profile` round-trip. - Added `onSseEvent` stream diagnostics so HTTP SSE providers can expose raw SSE frames without changing parsed model output. diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 9d36f07e0..9a425e731 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -57,6 +57,7 @@ import { resolveGitHubCopilotBaseUrl, } from "./github-copilot-headers"; import { transformMessages } from "./transform-messages"; +import { NON_VISION_IMAGE_PLACEHOLDER } from "./vision-guard"; export type AnthropicHeaderOptions = { apiKey: string; @@ -418,7 +419,10 @@ export const stripClaudeToolPrefix = (name: string, prefixOverride: string = cla /** * Convert content blocks to Anthropic API format */ -function convertContentBlocks(content: (TextContent | ImageContent)[]): +function convertContentBlocks( + content: (TextContent | ImageContent)[], + supportsImages = true, +): | string | Array< | { type: "text"; text: string } @@ -431,36 +435,35 @@ function convertContentBlocks(content: (TextContent | ImageContent)[]): }; } > { - // If only text blocks, return as concatenated string for simplicity - const hasImages = content.some(c => c.type === "image"); - if (!hasImages) { - return content - .map(c => (c as TextContent).text) - .join("\n") - .toWellFormed(); + const textBlocks = content + .filter((block): block is TextContent => block.type === "text") + .map(block => block.text.toWellFormed()) + .filter(text => text.trim().length > 0); + const imageBlocks = content.filter((block): block is ImageContent => block.type === "image"); + const omittedImages = !supportsImages && imageBlocks.length > 0; + if (imageBlocks.length === 0 || !supportsImages) { + if (omittedImages) { + textBlocks.push(NON_VISION_IMAGE_PLACEHOLDER); + } + return textBlocks.join("\n").toWellFormed(); } - // If we have images, convert to content block array - const blocks = content.map(block => { - if (block.type === "text") { - return { - type: "text" as const, - text: block.text.toWellFormed(), - }; - } - return { + const blocks = [ + ...textBlocks.map(text => ({ + type: "text" as const, + text, + })), + ...imageBlocks.map(block => ({ type: "image" as const, source: { type: "base64" as const, media_type: block.mimeType as "image/jpeg" | "image/png" | "image/gif" | "image/webp", data: block.data, }, - }; - }); + })), + ]; - // If only images (no text), add placeholder text block - const hasText = blocks.some(b => b.type === "text"); - if (!hasText) { + if (!textBlocks.length) { blocks.unshift({ type: "text" as const, text: "(see attached image)", @@ -1890,7 +1893,7 @@ function buildToolResultBlock(model: Model<"anthropic-messages">, msg: ToolResul const block: ContentBlockParam = { type: "tool_result", tool_use_id: msg.toolCallId, - content: convertContentBlocks(msg.content), + content: convertContentBlocks(msg.content, model.input.includes("image")), is_error: msg.isError, }; if (isZaiAnthropicEndpoint(model)) { @@ -1923,33 +1926,19 @@ export function convertAnthropicMessages( }); } } else { - const blocks: ContentBlockParam[] = msg.content.map(item => { - if (item.type === "text") { - return { - type: "text", - text: item.text.toWellFormed(), - }; - } - return { - type: "image", - source: { - type: "base64", - media_type: item.mimeType as "image/jpeg" | "image/png" | "image/gif" | "image/webp", - data: item.data, - }, - }; - }); - let filteredBlocks = !model?.input.includes("image") ? blocks.filter(b => b.type !== "image") : blocks; - filteredBlocks = filteredBlocks.filter(b => { - if (b.type === "text") { - return b.text.trim().length > 0; - } - return true; - }); - if (filteredBlocks.length === 0) continue; + const contentBlocks = convertContentBlocks(msg.content, model.input.includes("image")); + if (typeof contentBlocks === "string") { + if (contentBlocks.trim().length === 0) continue; + params.push({ + role: "user", + content: contentBlocks, + }); + continue; + } + if (contentBlocks.length === 0) continue; params.push({ role: "user", - content: filteredBlocks, + content: contentBlocks, }); } } else if (msg.role === "assistant") { diff --git a/packages/ai/src/providers/google-shared.ts b/packages/ai/src/providers/google-shared.ts index 595aa29f0..39501744c 100644 --- a/packages/ai/src/providers/google-shared.ts +++ b/packages/ai/src/providers/google-shared.ts @@ -5,6 +5,7 @@ import { type Content, FinishReason, FunctionCallingConfigMode, type Part } from import type { Context, ImageContent, Model, StopReason, TextContent, Tool } from "../types"; import { prepareSchemaForCCA, sanitizeSchemaForGoogle } from "../utils/schema"; import { transformMessages } from "./transform-messages"; +import { NON_VISION_IMAGE_PLACEHOLDER } from "./vision-guard"; export { sanitizeSchemaForGoogle }; @@ -108,30 +109,32 @@ export function convertMessages(model: Model, contex parts: [{ text: msg.content.toWellFormed() }], }); } else { - const parts: Part[] = msg.content.map(item => { + const supportsImages = model.input.includes("image"); + const parts: Part[] = []; + let omittedImages = false; + for (const item of msg.content) { if (item.type === "text") { - return { text: item.text.toWellFormed() }; - } else { - return { + const text = item.text.toWellFormed(); + if (text.trim().length === 0) continue; + parts.push({ text }); + } else if (supportsImages) { + parts.push({ inlineData: { mimeType: item.mimeType, data: item.data, }, - }; + }); + } else { + omittedImages = true; } - }); - // Filter out images if model doesn't support them, and empty text blocks - let filteredParts = !model.input.includes("image") ? parts.filter(p => p.text !== undefined) : parts; - filteredParts = filteredParts.filter(p => { - if (p.text !== undefined) { - return p.text.trim().length > 0; - } - return true; // Keep non-text parts (images) - }); - if (filteredParts.length === 0) continue; + } + if (omittedImages) { + parts.push({ text: NON_VISION_IMAGE_PLACEHOLDER }); + } + if (parts.length === 0) continue; contents.push({ role: "user", - parts: filteredParts, + parts, }); } } else if (msg.role === "assistant") { @@ -194,11 +197,11 @@ export function convertMessages(model: Model, contex }); } else if (msg.role === "toolResult") { // Extract text and image content + const supportsImages = model.input.includes("image"); const textContent = msg.content.filter((c): c is TextContent => c.type === "text"); const textResult = textContent.map(c => c.text).join("\n"); - const imageContent = model.input.includes("image") - ? msg.content.filter((c): c is ImageContent => c.type === "image") - : []; + const imageContent = supportsImages ? msg.content.filter((c): c is ImageContent => c.type === "image") : []; + const omittedImages = !supportsImages && msg.content.some((c): c is ImageContent => c.type === "image"); const hasText = textResult.length > 0; const hasImages = imageContent.length > 0; @@ -209,7 +212,13 @@ export function convertMessages(model: Model, contex const modelSupportsMultimodalFunctionResponse = supportsMultimodalFunctionResponse(model.id); // Use "output" key for success, "error" key for errors as per SDK documentation - const responseValue = hasText ? textResult.toWellFormed() : hasImages ? "(see attached image)" : ""; + const responseValue = omittedImages + ? [hasText ? textResult.toWellFormed() : "", NON_VISION_IMAGE_PLACEHOLDER].filter(Boolean).join("\n") + : hasText + ? textResult.toWellFormed() + : hasImages + ? "(see attached image)" + : ""; const imageParts: Part[] = imageContent.map(imageBlock => ({ inlineData: { diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index ed05c6fe4..24a169db7 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -54,12 +54,14 @@ import { import { parseCodexError } from "./openai-codex/response-handler"; import { normalizeOpenAIResponsesPromptCacheKey } from "./openai-responses"; import { + convertResponsesInputContent, encodeResponsesToolCallId, encodeTextSignatureV1, mapOpenAIResponsesStopReason, parseTextSignature, } from "./openai-responses-shared"; import { transformMessages } from "./transform-messages"; +import { joinTextWithImagePlaceholder } from "./vision-guard"; export interface OpenAICodexResponsesOptions extends StreamOptions { reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh"; @@ -2527,13 +2529,21 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex } if (msg.role === "toolResult") { + const supportsImages = model.input.includes("image"); const textResult = msg.content .filter(content => content.type === "text") .map(content => content.text) .join("\n"); const hasImages = msg.content.some(content => content.type === "image"); + const omittedImages = hasImages && !supportsImages; const normalized = normalizeResponsesToolCallId(msg.toolCallId); - const output = (textResult.length > 0 ? textResult : "(see attached image)").toWellFormed(); + const output = ( + omittedImages + ? joinTextWithImagePlaceholder(textResult, true) + : textResult.length > 0 + ? textResult + : "(see attached image)" + ).toWellFormed(); if (customCallIds.has(normalized.callId)) { messages.push({ type: "custom_tool_call_output", @@ -2547,7 +2557,7 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex output, }); } - if (hasImages && model.input.includes("image")) { + if (hasImages && supportsImages) { const contentParts: ResponseInputContent[] = [ { type: "input_text", text: "Attached image(s) from tool result:" } satisfies ResponseInputText, ]; @@ -2579,23 +2589,12 @@ function normalizeInputMessageContent( return [{ type: "input_text", text: content.toWellFormed() }]; } - const normalizedContent: ResponseInputContent[] = content.map(item => { - if (item.type === "text") { - return { type: "input_text", text: item.text.toWellFormed() } satisfies ResponseInputText; - } - return { - type: "input_image", - detail: "auto", - image_url: `data:${item.mimeType};base64,${item.data}`, - } satisfies ResponseInputImage; - }); - - const maybeWithoutImages = model.input.includes("image") - ? normalizedContent - : normalizedContent.filter(item => item.type !== "input_image"); - return maybeWithoutImages.filter(item => item.type !== "input_text" || item.text.trim().length > 0); + return convertResponsesInputContent(content, model.input.includes("image")) ?? []; } +/** @internal Exported for tests. */ +export { convertMessages as convertCodexResponsesMessages }; + /** * Whether this Codex-backend model should get the custom-tool grammar * variant for `apply_patch`. codex-rs uses a single serializer for both diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index dc841c400..5d12ede1f 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -64,6 +64,7 @@ import { } from "./github-copilot-headers"; import { detectOpenAICompat, type ResolvedOpenAICompat, resolveOpenAICompat } from "./openai-completions-compat"; import { transformMessages } from "./transform-messages"; +import { joinTextWithImagePlaceholder, NON_VISION_IMAGE_PLACEHOLDER } from "./vision-guard"; /** * Normalize tool call ID for Mistral. @@ -1254,7 +1255,9 @@ export function convertMessages( content: text, }); } else { + const supportsImages = model.input.includes("image"); const content: ChatCompletionContentPart[] = []; + let omittedImages = false; for (const item of msg.content) { if (item.type === "text") { const text = item.text.toWellFormed(); @@ -1263,22 +1266,27 @@ export function convertMessages( type: "text", text, } satisfies ChatCompletionContentPartText); - } else { + } else if (supportsImages) { content.push({ type: "image_url", image_url: { url: `data:${item.mimeType};base64,${item.data}`, }, } satisfies ChatCompletionContentPartImage); + } else { + omittedImages = true; } } - const filteredContent = !model.input.includes("image") - ? content.filter(c => c.type !== "image_url") - : content; - if (filteredContent.length === 0) continue; + if (omittedImages) { + content.push({ + type: "text", + text: NON_VISION_IMAGE_PLACEHOLDER, + } satisfies ChatCompletionContentPartText); + } + if (content.length === 0) continue; params.push({ role: "user", - content: filteredContent, + content, }); } } else if (msg.role === "assistant") { @@ -1473,19 +1481,27 @@ export function convertMessages( // Extract text and image content const textResult = toolMsg.content .filter(c => c.type === "text") - .map(c => (c as any).text) + .map(c => (c as TextContent).text) .join("\n"); + const supportsImages = model.input.includes("image"); const hasImages = toolMsg.content.some(c => c.type === "image"); + const omittedImages = hasImages && !supportsImages; // Always send tool result with text (or placeholder if only images) const hasText = textResult.length > 0; - // Some providers (e.g. Mistral) require the 'name' field in tool results const remappedToolCallId = consumeToolCallId(toolMsg.toolCallId); const resolvedToolCallId = remappedToolCallId ?? ensureToolCallId(toolMsg.toolCallId, `${j}:${toolMsg.toolName ?? "tool"}`); + const toolResultContent = omittedImages + ? joinTextWithImagePlaceholder(textResult, true) + : hasText + ? textResult + : hasImages + ? "(see attached image)" + : ""; const toolResultMsg: ChatCompletionToolMessageParam = { role: "tool", - content: (hasText ? textResult : "(see attached image)").toWellFormed(), + content: toolResultContent.toWellFormed(), tool_call_id: normalizeMistralToolId(resolvedToolCallId, compat.requiresMistralToolIds), }; if (compat.requiresToolResultName && toolMsg.toolName) { @@ -1493,13 +1509,13 @@ export function convertMessages( } params.push(toolResultMsg); - if (hasImages && model.input.includes("image")) { + if (hasImages && supportsImages) { for (const block of toolMsg.content) { if (block.type === "image") { imageBlocks.push({ type: "image_url", image_url: { - url: `data:${(block as any).mimeType};base64,${(block as any).data}`, + url: `data:${block.mimeType};base64,${block.data}`, }, }); } diff --git a/packages/ai/src/providers/openai-responses-shared.ts b/packages/ai/src/providers/openai-responses-shared.ts index 21e944f90..3cf01120d 100644 --- a/packages/ai/src/providers/openai-responses-shared.ts +++ b/packages/ai/src/providers/openai-responses-shared.ts @@ -27,6 +27,7 @@ import type { import { normalizeResponsesToolCallId } from "../utils"; import type { AssistantMessageEventStream } from "../utils/event-stream"; import { parseStreamingJson } from "../utils/json-parse"; +import { joinTextWithImagePlaceholder, NON_VISION_IMAGE_PLACEHOLDER, partitionVisionContent } from "./vision-guard"; export function encodeTextSignatureV1(id: string, phase?: TextSignatureV1["phase"]): string { const payload: TextSignatureV1 = { v: 1, id }; @@ -121,23 +122,29 @@ export function convertResponsesInputContent( return [{ type: "input_text", text: content.toWellFormed() } satisfies ResponseInputText]; } - const normalizedContent = content - .map((item): ResponseInputContent => { - if (item.type === "text") { - return { - type: "input_text", - text: item.text.toWellFormed(), - } satisfies ResponseInputText; - } - return { - type: "input_image", - detail: "auto", - image_url: `data:${item.mimeType};base64,${item.data}`, - } satisfies ResponseInputImage; - }) - .filter(item => supportsImages || item.type !== "input_image") - .filter(item => item.type !== "input_text" || item.text.trim().length > 0); - + const { textBlocks, imageBlocks, omittedImages } = partitionVisionContent(content, supportsImages); + const normalizedContent: ResponseInputContent[] = []; + for (const item of textBlocks) { + const text = item.text.toWellFormed(); + if (text.trim().length === 0) continue; + normalizedContent.push({ + type: "input_text", + text, + } satisfies ResponseInputText); + } + for (const item of imageBlocks) { + normalizedContent.push({ + type: "input_image", + detail: "auto", + image_url: `data:${item.mimeType};base64,${item.data}`, + } satisfies ResponseInputImage); + } + if (omittedImages) { + normalizedContent.push({ + type: "input_text", + text: NON_VISION_IMAGE_PLACEHOLDER, + } satisfies ResponseInputText); + } return normalizedContent.length > 0 ? normalizedContent : undefined; } @@ -225,17 +232,25 @@ export function appendResponsesToolResultMessages( knownCallIds: ReadonlySet, customCallIds?: ReadonlySet, ): void { + const supportsImages = model.input.includes("image"); const textResult = toolResult.content .filter((block): block is TextContent => block.type === "text") .map(block => block.text) .join("\n"); const hasImages = toolResult.content.some((block): block is ImageContent => block.type === "image"); + const omittedImages = hasImages && !supportsImages; const normalized = normalizeResponsesToolCallId(toolResult.toolCallId); if (strictResponsesPairing && !knownCallIds.has(normalized.callId)) { return; } - const output = (textResult.length > 0 ? textResult : "(see attached image)").toWellFormed(); + const output = ( + omittedImages + ? joinTextWithImagePlaceholder(textResult, true) + : textResult.length > 0 + ? textResult + : "(see attached image)" + ).toWellFormed(); if (customCallIds?.has(normalized.callId)) { messages.push({ type: "custom_tool_call_output", @@ -250,7 +265,7 @@ export function appendResponsesToolResultMessages( }); } - if (!hasImages || !model.input.includes("image")) { + if (!hasImages || !supportsImages) { return; } diff --git a/packages/ai/src/providers/vision-guard.ts b/packages/ai/src/providers/vision-guard.ts new file mode 100644 index 000000000..5e12d7892 --- /dev/null +++ b/packages/ai/src/providers/vision-guard.ts @@ -0,0 +1,31 @@ +import type { ImageContent, TextContent } from "../types"; + +export const NON_VISION_IMAGE_PLACEHOLDER = "[image omitted: model does not support vision]"; + +export function partitionVisionContent( + content: ReadonlyArray, + supportsImages: boolean, +): { + textBlocks: TextContent[]; + imageBlocks: ImageContent[]; + omittedImages: boolean; +} { + const textBlocks = content.filter((block): block is TextContent => block.type === "text"); + const imageBlocks = content.filter((block): block is ImageContent => block.type === "image"); + return { + textBlocks, + imageBlocks: supportsImages ? imageBlocks : [], + omittedImages: !supportsImages && imageBlocks.length > 0, + }; +} + +export function joinTextWithImagePlaceholder(text: string, omittedImages: boolean): string { + const parts: string[] = []; + if (text.length > 0) { + parts.push(text); + } + if (omittedImages) { + parts.push(NON_VISION_IMAGE_PLACEHOLDER); + } + return parts.join("\n"); +} diff --git a/packages/ai/test/issue-967-vision-guard.test.ts b/packages/ai/test/issue-967-vision-guard.test.ts new file mode 100644 index 000000000..f2b3399ce --- /dev/null +++ b/packages/ai/test/issue-967-vision-guard.test.ts @@ -0,0 +1,278 @@ +import { describe, expect, it } from "bun:test"; +import { convertAnthropicMessages } from "../src/providers/anthropic"; +import { convertMessages as convertGoogleMessages } from "../src/providers/google-shared"; +import { convertCodexResponsesMessages } from "../src/providers/openai-codex-responses"; +import { convertMessages as convertOpenAICompletionsMessages } from "../src/providers/openai-completions"; +import { + appendResponsesToolResultMessages, + convertResponsesInputContent, +} from "../src/providers/openai-responses-shared"; +import { NON_VISION_IMAGE_PLACEHOLDER } from "../src/providers/vision-guard"; +import type { Api, AssistantMessage, Context, Model, OpenAICompat, ToolResultMessage, Usage } from "../src/types"; + +const emptyUsage: Usage = { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, +}; + +const compat: Required = { + supportsStore: true, + supportsDeveloperRole: true, + supportsMultipleSystemMessages: true, + supportsReasoningEffort: true, + reasoningEffortMap: {}, + supportsUsageInStreaming: true, + supportsToolChoice: true, + disableReasoningOnForcedToolChoice: false, + disableReasoningOnToolChoice: false, + maxTokensField: "max_completion_tokens", + requiresToolResultName: false, + requiresAssistantAfterToolResult: false, + requiresThinkingAsText: false, + requiresMistralToolIds: false, + thinkingFormat: "openai", + reasoningContentField: "reasoning_content", + requiresReasoningContentForToolCalls: false, + allowsSyntheticReasoningContentForToolCalls: true, + requiresAssistantContentForToolCalls: false, + openRouterRouting: {}, + vercelGatewayRouting: {}, + extraBody: {}, + supportsStrictMode: true, + toolStrictMode: "none", +}; + +function makeModel(api: TApi, provider: Model["provider"]): Model { + return { + id: `${provider}-${api}-text-only`, + name: `${provider} ${api}`, + api, + provider, + baseUrl: "https://example.com", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128_000, + maxTokens: 8_192, + }; +} + +function makeAssistant(api: Model["api"], provider: Model["provider"], modelId: string): AssistantMessage { + return { + role: "assistant", + content: [{ type: "toolCall", id: "call_1", name: "python", arguments: { code: "plot()" } }], + api, + provider, + model: modelId, + usage: emptyUsage, + stopReason: "toolUse", + timestamp: 2, + }; +} + +function makeToolResult(content: ToolResultMessage["content"]): ToolResultMessage { + return { + role: "toolResult", + toolCallId: "call_1", + toolName: "python", + content, + isError: false, + timestamp: 3, + }; +} + +function countTaggedValues(value: unknown, tag: string): number { + if (Array.isArray(value)) { + return value.reduce((sum, item) => sum + countTaggedValues(item, tag), 0); + } + if (!value || typeof value !== "object") { + return 0; + } + const record = value as Record; + const own = record.type === tag ? 1 : 0; + return Object.values(record).reduce((sum, item) => sum + countTaggedValues(item, tag), own); +} + +function countObjectKeys(value: unknown, key: string): number { + if (Array.isArray(value)) { + return value.reduce((sum, item) => sum + countObjectKeys(item, key), 0); + } + if (!value || typeof value !== "object") { + return 0; + } + const record = value as Record; + const own = Object.hasOwn(record, key) ? 1 : 0; + return Object.values(record).reduce((sum, item) => sum + countObjectKeys(item, key), own); +} + +describe("issue #967 vision guard", () => { + it("strips non-vision images from OpenAI chat-completions user and tool-result payloads", () => { + const model = makeModel("openai-completions", "openrouter"); + const context: Context = { + messages: [ + { + role: "user", + content: [ + { type: "text", text: "plot summary" }, + { type: "image", mimeType: "image/png", data: "ZmFrZQ==" }, + ], + timestamp: 1, + }, + makeAssistant(model.api, model.provider, model.id), + makeToolResult([ + { type: "text", text: "saved plot to /tmp/plot.png" }, + { type: "image", mimeType: "image/png", data: "ZmFrZQ==" }, + ]), + ], + }; + + const messages = convertOpenAICompletionsMessages(model, context, compat); + expect(countTaggedValues(messages, "image_url")).toBe(0); + expect(messages.filter(message => message.role === "user")).toHaveLength(1); + expect(messages[0]).toMatchObject({ + role: "user", + content: [ + { type: "text", text: "plot summary" }, + { type: "text", text: NON_VISION_IMAGE_PLACEHOLDER }, + ], + }); + expect(messages.find(message => message.role === "tool")).toMatchObject({ + content: `saved plot to /tmp/plot.png\n${NON_VISION_IMAGE_PLACEHOLDER}`, + }); + }); + + it("strips non-vision images from OpenAI responses payload builders", () => { + const model = makeModel("openai-responses", "openrouter"); + const userContent = convertResponsesInputContent( + [ + { type: "text", text: "plot summary" }, + { type: "image", mimeType: "image/png", data: "ZmFrZQ==" }, + ], + false, + ); + expect(countTaggedValues(userContent, "input_image")).toBe(0); + expect(userContent).toEqual([ + { type: "input_text", text: "plot summary" }, + { type: "input_text", text: NON_VISION_IMAGE_PLACEHOLDER }, + ]); + + const payload: unknown[] = []; + appendResponsesToolResultMessages( + payload as never, + makeToolResult([ + { type: "text", text: "saved plot to /tmp/plot.png" }, + { type: "image", mimeType: "image/png", data: "ZmFrZQ==" }, + ]), + model, + true, + new Set(["call_1"]), + ); + expect(countTaggedValues(payload, "input_image")).toBe(0); + expect(payload).toEqual([ + { + type: "function_call_output", + call_id: "call_1", + output: `saved plot to /tmp/plot.png\n${NON_VISION_IMAGE_PLACEHOLDER}`, + }, + ]); + }); + + it("strips non-vision images from Codex responses user and tool-result payloads", () => { + const model = makeModel("openai-codex-responses", "openai-codex"); + const context: Context = { + messages: [ + { + role: "user", + content: [ + { type: "text", text: "plot summary" }, + { type: "image", mimeType: "image/png", data: "ZmFrZQ==" }, + ], + timestamp: 1, + }, + makeAssistant(model.api, model.provider, model.id), + makeToolResult([{ type: "image", mimeType: "image/png", data: "ZmFrZQ==" }]), + ], + }; + + const messages = convertCodexResponsesMessages(model, context); + expect(countTaggedValues(messages, "input_image")).toBe(0); + expect(messages.filter(item => (item as { role?: string }).role === "user")).toHaveLength(1); + expect(messages[0]).toMatchObject({ + role: "user", + content: [ + { type: "input_text", text: "plot summary" }, + { type: "input_text", text: NON_VISION_IMAGE_PLACEHOLDER }, + ], + }); + expect(messages.find(item => (item as { type?: string }).type === "function_call_output")).toMatchObject({ + output: NON_VISION_IMAGE_PLACEHOLDER, + }); + }); + + it("strips non-vision images from Anthropic payloads", () => { + const model = makeModel("anthropic-messages", "anthropic"); + const messages = convertAnthropicMessages( + [ + { + role: "user", + content: [ + { type: "text", text: "plot summary" }, + { type: "image", mimeType: "image/png", data: "ZmFrZQ==" }, + ], + timestamp: 1, + }, + makeAssistant(model.api, model.provider, model.id), + makeToolResult([{ type: "image", mimeType: "image/png", data: "ZmFrZQ==" }]), + ], + model, + false, + ); + expect(countTaggedValues(messages, "image")).toBe(0); + expect(messages[0]).toMatchObject({ role: "user", content: `plot summary\n${NON_VISION_IMAGE_PLACEHOLDER}` }); + const toolResult = messages.at(-1) as { role: string; content: Array<{ type: string; content: unknown }> }; + expect(toolResult.role).toBe("user"); + expect(toolResult.content[0]).toMatchObject({ + type: "tool_result", + content: NON_VISION_IMAGE_PLACEHOLDER, + }); + }); + + it("strips non-vision images from Google payloads", () => { + const model = makeModel("google-generative-ai", "google"); + const context: Context = { + messages: [ + { + role: "user", + content: [ + { type: "text", text: "plot summary" }, + { type: "image", mimeType: "image/png", data: "ZmFrZQ==" }, + ], + timestamp: 1, + }, + makeAssistant(model.api, model.provider, model.id), + makeToolResult([{ type: "image", mimeType: "image/png", data: "ZmFrZQ==" }]), + ], + }; + + const messages = convertGoogleMessages(model, context); + expect(countObjectKeys(messages, "inlineData")).toBe(0); + expect(messages[0]).toMatchObject({ + role: "user", + parts: [{ text: "plot summary" }, { text: NON_VISION_IMAGE_PLACEHOLDER }], + }); + expect(messages.at(-1)).toMatchObject({ + role: "user", + parts: [ + { + functionResponse: { + response: { output: NON_VISION_IMAGE_PLACEHOLDER }, + }, + }, + ], + }); + }); +});