fix(ai): replaced silent image drop with placeholder for non-vision models
- Image blocks sent to models without vision support now emit `[image omitted: model does not support vision]` instead of being silently dropped, preventing opaque 404 errors. - Fix applies to user messages and tool-result payloads across Anthropic, OpenAI completions/responses, Codex, and Google providers. - Extracted shared `vision-guard.ts` with `partitionVisionContent`, `joinTextWithImagePlaceholder`, and `NON_VISION_IMAGE_PLACEHOLDER`. - Added regression tests covering all five provider paths. Fixes #967 Fixes #968
This commit is contained in:
@@ -4,6 +4,9 @@
|
||||
|
||||
### Added
|
||||
|
||||
### Fixed
|
||||
- Fixed silent forwarding of image content (for example Python plot output rendered in the terminal) to models without vision support, which produced opaque 404 errors from upstream. Image blocks are now stripped and replaced with a `[image omitted: model does not support vision]` placeholder for non-vision models, including tool-result payloads ([#967](https://github.com/can1357/oh-my-pi/issues/967), [#968](https://github.com/can1357/oh-my-pi/issues/968)).
|
||||
|
||||
- Added `AuthStorage` `onCredentialDisabled` callback (sync or async) so embedders can react when a credential is automatically disabled (e.g. OAuth refresh fails with `invalid_grant`) — useful for surfacing a banner or auto-launching a re-login flow instead of letting the credential silently disappear. Sync throws and async rejections are both caught and logged so a misbehaving subscriber cannot break the disable path.
|
||||
- Added Anthropic OAuth `account.uuid` and `account.email_address` extraction from the `/v1/oauth/token` exchange and refresh responses; both `AnthropicOAuthFlow.exchangeToken()` and `refreshAnthropicToken()` now populate `OAuthCredentials.{accountId, email}` so downstream consumers can attribute requests to the authenticated account without a separate `/api/oauth/profile` round-trip.
|
||||
- Added `onSseEvent` stream diagnostics so HTTP SSE providers can expose raw SSE frames without changing parsed model output.
|
||||
|
||||
@@ -57,6 +57,7 @@ import {
|
||||
resolveGitHubCopilotBaseUrl,
|
||||
} from "./github-copilot-headers";
|
||||
import { transformMessages } from "./transform-messages";
|
||||
import { NON_VISION_IMAGE_PLACEHOLDER } from "./vision-guard";
|
||||
|
||||
export type AnthropicHeaderOptions = {
|
||||
apiKey: string;
|
||||
@@ -418,7 +419,10 @@ export const stripClaudeToolPrefix = (name: string, prefixOverride: string = cla
|
||||
/**
|
||||
* Convert content blocks to Anthropic API format
|
||||
*/
|
||||
function convertContentBlocks(content: (TextContent | ImageContent)[]):
|
||||
function convertContentBlocks(
|
||||
content: (TextContent | ImageContent)[],
|
||||
supportsImages = true,
|
||||
):
|
||||
| string
|
||||
| Array<
|
||||
| { type: "text"; text: string }
|
||||
@@ -431,36 +435,35 @@ function convertContentBlocks(content: (TextContent | ImageContent)[]):
|
||||
};
|
||||
}
|
||||
> {
|
||||
// If only text blocks, return as concatenated string for simplicity
|
||||
const hasImages = content.some(c => c.type === "image");
|
||||
if (!hasImages) {
|
||||
return content
|
||||
.map(c => (c as TextContent).text)
|
||||
.join("\n")
|
||||
.toWellFormed();
|
||||
const textBlocks = content
|
||||
.filter((block): block is TextContent => block.type === "text")
|
||||
.map(block => block.text.toWellFormed())
|
||||
.filter(text => text.trim().length > 0);
|
||||
const imageBlocks = content.filter((block): block is ImageContent => block.type === "image");
|
||||
const omittedImages = !supportsImages && imageBlocks.length > 0;
|
||||
if (imageBlocks.length === 0 || !supportsImages) {
|
||||
if (omittedImages) {
|
||||
textBlocks.push(NON_VISION_IMAGE_PLACEHOLDER);
|
||||
}
|
||||
return textBlocks.join("\n").toWellFormed();
|
||||
}
|
||||
|
||||
// If we have images, convert to content block array
|
||||
const blocks = content.map(block => {
|
||||
if (block.type === "text") {
|
||||
return {
|
||||
type: "text" as const,
|
||||
text: block.text.toWellFormed(),
|
||||
};
|
||||
}
|
||||
return {
|
||||
const blocks = [
|
||||
...textBlocks.map(text => ({
|
||||
type: "text" as const,
|
||||
text,
|
||||
})),
|
||||
...imageBlocks.map(block => ({
|
||||
type: "image" as const,
|
||||
source: {
|
||||
type: "base64" as const,
|
||||
media_type: block.mimeType as "image/jpeg" | "image/png" | "image/gif" | "image/webp",
|
||||
data: block.data,
|
||||
},
|
||||
};
|
||||
});
|
||||
})),
|
||||
];
|
||||
|
||||
// If only images (no text), add placeholder text block
|
||||
const hasText = blocks.some(b => b.type === "text");
|
||||
if (!hasText) {
|
||||
if (!textBlocks.length) {
|
||||
blocks.unshift({
|
||||
type: "text" as const,
|
||||
text: "(see attached image)",
|
||||
@@ -1890,7 +1893,7 @@ function buildToolResultBlock(model: Model<"anthropic-messages">, msg: ToolResul
|
||||
const block: ContentBlockParam = {
|
||||
type: "tool_result",
|
||||
tool_use_id: msg.toolCallId,
|
||||
content: convertContentBlocks(msg.content),
|
||||
content: convertContentBlocks(msg.content, model.input.includes("image")),
|
||||
is_error: msg.isError,
|
||||
};
|
||||
if (isZaiAnthropicEndpoint(model)) {
|
||||
@@ -1923,33 +1926,19 @@ export function convertAnthropicMessages(
|
||||
});
|
||||
}
|
||||
} else {
|
||||
const blocks: ContentBlockParam[] = msg.content.map(item => {
|
||||
if (item.type === "text") {
|
||||
return {
|
||||
type: "text",
|
||||
text: item.text.toWellFormed(),
|
||||
};
|
||||
}
|
||||
return {
|
||||
type: "image",
|
||||
source: {
|
||||
type: "base64",
|
||||
media_type: item.mimeType as "image/jpeg" | "image/png" | "image/gif" | "image/webp",
|
||||
data: item.data,
|
||||
},
|
||||
};
|
||||
});
|
||||
let filteredBlocks = !model?.input.includes("image") ? blocks.filter(b => b.type !== "image") : blocks;
|
||||
filteredBlocks = filteredBlocks.filter(b => {
|
||||
if (b.type === "text") {
|
||||
return b.text.trim().length > 0;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
if (filteredBlocks.length === 0) continue;
|
||||
const contentBlocks = convertContentBlocks(msg.content, model.input.includes("image"));
|
||||
if (typeof contentBlocks === "string") {
|
||||
if (contentBlocks.trim().length === 0) continue;
|
||||
params.push({
|
||||
role: "user",
|
||||
content: contentBlocks,
|
||||
});
|
||||
continue;
|
||||
}
|
||||
if (contentBlocks.length === 0) continue;
|
||||
params.push({
|
||||
role: "user",
|
||||
content: filteredBlocks,
|
||||
content: contentBlocks,
|
||||
});
|
||||
}
|
||||
} else if (msg.role === "assistant") {
|
||||
|
||||
@@ -5,6 +5,7 @@ import { type Content, FinishReason, FunctionCallingConfigMode, type Part } from
|
||||
import type { Context, ImageContent, Model, StopReason, TextContent, Tool } from "../types";
|
||||
import { prepareSchemaForCCA, sanitizeSchemaForGoogle } from "../utils/schema";
|
||||
import { transformMessages } from "./transform-messages";
|
||||
import { NON_VISION_IMAGE_PLACEHOLDER } from "./vision-guard";
|
||||
|
||||
export { sanitizeSchemaForGoogle };
|
||||
|
||||
@@ -108,30 +109,32 @@ export function convertMessages<T extends GoogleApiType>(model: Model<T>, contex
|
||||
parts: [{ text: msg.content.toWellFormed() }],
|
||||
});
|
||||
} else {
|
||||
const parts: Part[] = msg.content.map(item => {
|
||||
const supportsImages = model.input.includes("image");
|
||||
const parts: Part[] = [];
|
||||
let omittedImages = false;
|
||||
for (const item of msg.content) {
|
||||
if (item.type === "text") {
|
||||
return { text: item.text.toWellFormed() };
|
||||
} else {
|
||||
return {
|
||||
const text = item.text.toWellFormed();
|
||||
if (text.trim().length === 0) continue;
|
||||
parts.push({ text });
|
||||
} else if (supportsImages) {
|
||||
parts.push({
|
||||
inlineData: {
|
||||
mimeType: item.mimeType,
|
||||
data: item.data,
|
||||
},
|
||||
};
|
||||
});
|
||||
} else {
|
||||
omittedImages = true;
|
||||
}
|
||||
});
|
||||
// Filter out images if model doesn't support them, and empty text blocks
|
||||
let filteredParts = !model.input.includes("image") ? parts.filter(p => p.text !== undefined) : parts;
|
||||
filteredParts = filteredParts.filter(p => {
|
||||
if (p.text !== undefined) {
|
||||
return p.text.trim().length > 0;
|
||||
}
|
||||
return true; // Keep non-text parts (images)
|
||||
});
|
||||
if (filteredParts.length === 0) continue;
|
||||
}
|
||||
if (omittedImages) {
|
||||
parts.push({ text: NON_VISION_IMAGE_PLACEHOLDER });
|
||||
}
|
||||
if (parts.length === 0) continue;
|
||||
contents.push({
|
||||
role: "user",
|
||||
parts: filteredParts,
|
||||
parts,
|
||||
});
|
||||
}
|
||||
} else if (msg.role === "assistant") {
|
||||
@@ -194,11 +197,11 @@ export function convertMessages<T extends GoogleApiType>(model: Model<T>, contex
|
||||
});
|
||||
} else if (msg.role === "toolResult") {
|
||||
// Extract text and image content
|
||||
const supportsImages = model.input.includes("image");
|
||||
const textContent = msg.content.filter((c): c is TextContent => c.type === "text");
|
||||
const textResult = textContent.map(c => c.text).join("\n");
|
||||
const imageContent = model.input.includes("image")
|
||||
? msg.content.filter((c): c is ImageContent => c.type === "image")
|
||||
: [];
|
||||
const imageContent = supportsImages ? msg.content.filter((c): c is ImageContent => c.type === "image") : [];
|
||||
const omittedImages = !supportsImages && msg.content.some((c): c is ImageContent => c.type === "image");
|
||||
|
||||
const hasText = textResult.length > 0;
|
||||
const hasImages = imageContent.length > 0;
|
||||
@@ -209,7 +212,13 @@ export function convertMessages<T extends GoogleApiType>(model: Model<T>, contex
|
||||
const modelSupportsMultimodalFunctionResponse = supportsMultimodalFunctionResponse(model.id);
|
||||
|
||||
// Use "output" key for success, "error" key for errors as per SDK documentation
|
||||
const responseValue = hasText ? textResult.toWellFormed() : hasImages ? "(see attached image)" : "";
|
||||
const responseValue = omittedImages
|
||||
? [hasText ? textResult.toWellFormed() : "", NON_VISION_IMAGE_PLACEHOLDER].filter(Boolean).join("\n")
|
||||
: hasText
|
||||
? textResult.toWellFormed()
|
||||
: hasImages
|
||||
? "(see attached image)"
|
||||
: "";
|
||||
|
||||
const imageParts: Part[] = imageContent.map(imageBlock => ({
|
||||
inlineData: {
|
||||
|
||||
@@ -54,12 +54,14 @@ import {
|
||||
import { parseCodexError } from "./openai-codex/response-handler";
|
||||
import { normalizeOpenAIResponsesPromptCacheKey } from "./openai-responses";
|
||||
import {
|
||||
convertResponsesInputContent,
|
||||
encodeResponsesToolCallId,
|
||||
encodeTextSignatureV1,
|
||||
mapOpenAIResponsesStopReason,
|
||||
parseTextSignature,
|
||||
} from "./openai-responses-shared";
|
||||
import { transformMessages } from "./transform-messages";
|
||||
import { joinTextWithImagePlaceholder } from "./vision-guard";
|
||||
|
||||
export interface OpenAICodexResponsesOptions extends StreamOptions {
|
||||
reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh";
|
||||
@@ -2527,13 +2529,21 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex
|
||||
}
|
||||
|
||||
if (msg.role === "toolResult") {
|
||||
const supportsImages = model.input.includes("image");
|
||||
const textResult = msg.content
|
||||
.filter(content => content.type === "text")
|
||||
.map(content => content.text)
|
||||
.join("\n");
|
||||
const hasImages = msg.content.some(content => content.type === "image");
|
||||
const omittedImages = hasImages && !supportsImages;
|
||||
const normalized = normalizeResponsesToolCallId(msg.toolCallId);
|
||||
const output = (textResult.length > 0 ? textResult : "(see attached image)").toWellFormed();
|
||||
const output = (
|
||||
omittedImages
|
||||
? joinTextWithImagePlaceholder(textResult, true)
|
||||
: textResult.length > 0
|
||||
? textResult
|
||||
: "(see attached image)"
|
||||
).toWellFormed();
|
||||
if (customCallIds.has(normalized.callId)) {
|
||||
messages.push({
|
||||
type: "custom_tool_call_output",
|
||||
@@ -2547,7 +2557,7 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex
|
||||
output,
|
||||
});
|
||||
}
|
||||
if (hasImages && model.input.includes("image")) {
|
||||
if (hasImages && supportsImages) {
|
||||
const contentParts: ResponseInputContent[] = [
|
||||
{ type: "input_text", text: "Attached image(s) from tool result:" } satisfies ResponseInputText,
|
||||
];
|
||||
@@ -2579,23 +2589,12 @@ function normalizeInputMessageContent(
|
||||
return [{ type: "input_text", text: content.toWellFormed() }];
|
||||
}
|
||||
|
||||
const normalizedContent: ResponseInputContent[] = content.map(item => {
|
||||
if (item.type === "text") {
|
||||
return { type: "input_text", text: item.text.toWellFormed() } satisfies ResponseInputText;
|
||||
}
|
||||
return {
|
||||
type: "input_image",
|
||||
detail: "auto",
|
||||
image_url: `data:${item.mimeType};base64,${item.data}`,
|
||||
} satisfies ResponseInputImage;
|
||||
});
|
||||
|
||||
const maybeWithoutImages = model.input.includes("image")
|
||||
? normalizedContent
|
||||
: normalizedContent.filter(item => item.type !== "input_image");
|
||||
return maybeWithoutImages.filter(item => item.type !== "input_text" || item.text.trim().length > 0);
|
||||
return convertResponsesInputContent(content, model.input.includes("image")) ?? [];
|
||||
}
|
||||
|
||||
/** @internal Exported for tests. */
|
||||
export { convertMessages as convertCodexResponsesMessages };
|
||||
|
||||
/**
|
||||
* Whether this Codex-backend model should get the custom-tool grammar
|
||||
* variant for `apply_patch`. codex-rs uses a single serializer for both
|
||||
|
||||
@@ -64,6 +64,7 @@ import {
|
||||
} from "./github-copilot-headers";
|
||||
import { detectOpenAICompat, type ResolvedOpenAICompat, resolveOpenAICompat } from "./openai-completions-compat";
|
||||
import { transformMessages } from "./transform-messages";
|
||||
import { joinTextWithImagePlaceholder, NON_VISION_IMAGE_PLACEHOLDER } from "./vision-guard";
|
||||
|
||||
/**
|
||||
* Normalize tool call ID for Mistral.
|
||||
@@ -1254,7 +1255,9 @@ export function convertMessages(
|
||||
content: text,
|
||||
});
|
||||
} else {
|
||||
const supportsImages = model.input.includes("image");
|
||||
const content: ChatCompletionContentPart[] = [];
|
||||
let omittedImages = false;
|
||||
for (const item of msg.content) {
|
||||
if (item.type === "text") {
|
||||
const text = item.text.toWellFormed();
|
||||
@@ -1263,22 +1266,27 @@ export function convertMessages(
|
||||
type: "text",
|
||||
text,
|
||||
} satisfies ChatCompletionContentPartText);
|
||||
} else {
|
||||
} else if (supportsImages) {
|
||||
content.push({
|
||||
type: "image_url",
|
||||
image_url: {
|
||||
url: `data:${item.mimeType};base64,${item.data}`,
|
||||
},
|
||||
} satisfies ChatCompletionContentPartImage);
|
||||
} else {
|
||||
omittedImages = true;
|
||||
}
|
||||
}
|
||||
const filteredContent = !model.input.includes("image")
|
||||
? content.filter(c => c.type !== "image_url")
|
||||
: content;
|
||||
if (filteredContent.length === 0) continue;
|
||||
if (omittedImages) {
|
||||
content.push({
|
||||
type: "text",
|
||||
text: NON_VISION_IMAGE_PLACEHOLDER,
|
||||
} satisfies ChatCompletionContentPartText);
|
||||
}
|
||||
if (content.length === 0) continue;
|
||||
params.push({
|
||||
role: "user",
|
||||
content: filteredContent,
|
||||
content,
|
||||
});
|
||||
}
|
||||
} else if (msg.role === "assistant") {
|
||||
@@ -1473,19 +1481,27 @@ export function convertMessages(
|
||||
// Extract text and image content
|
||||
const textResult = toolMsg.content
|
||||
.filter(c => c.type === "text")
|
||||
.map(c => (c as any).text)
|
||||
.map(c => (c as TextContent).text)
|
||||
.join("\n");
|
||||
const supportsImages = model.input.includes("image");
|
||||
const hasImages = toolMsg.content.some(c => c.type === "image");
|
||||
const omittedImages = hasImages && !supportsImages;
|
||||
|
||||
// Always send tool result with text (or placeholder if only images)
|
||||
const hasText = textResult.length > 0;
|
||||
// Some providers (e.g. Mistral) require the 'name' field in tool results
|
||||
const remappedToolCallId = consumeToolCallId(toolMsg.toolCallId);
|
||||
const resolvedToolCallId =
|
||||
remappedToolCallId ?? ensureToolCallId(toolMsg.toolCallId, `${j}:${toolMsg.toolName ?? "tool"}`);
|
||||
const toolResultContent = omittedImages
|
||||
? joinTextWithImagePlaceholder(textResult, true)
|
||||
: hasText
|
||||
? textResult
|
||||
: hasImages
|
||||
? "(see attached image)"
|
||||
: "";
|
||||
const toolResultMsg: ChatCompletionToolMessageParam = {
|
||||
role: "tool",
|
||||
content: (hasText ? textResult : "(see attached image)").toWellFormed(),
|
||||
content: toolResultContent.toWellFormed(),
|
||||
tool_call_id: normalizeMistralToolId(resolvedToolCallId, compat.requiresMistralToolIds),
|
||||
};
|
||||
if (compat.requiresToolResultName && toolMsg.toolName) {
|
||||
@@ -1493,13 +1509,13 @@ export function convertMessages(
|
||||
}
|
||||
params.push(toolResultMsg);
|
||||
|
||||
if (hasImages && model.input.includes("image")) {
|
||||
if (hasImages && supportsImages) {
|
||||
for (const block of toolMsg.content) {
|
||||
if (block.type === "image") {
|
||||
imageBlocks.push({
|
||||
type: "image_url",
|
||||
image_url: {
|
||||
url: `data:${(block as any).mimeType};base64,${(block as any).data}`,
|
||||
url: `data:${block.mimeType};base64,${block.data}`,
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
@@ -27,6 +27,7 @@ import type {
|
||||
import { normalizeResponsesToolCallId } from "../utils";
|
||||
import type { AssistantMessageEventStream } from "../utils/event-stream";
|
||||
import { parseStreamingJson } from "../utils/json-parse";
|
||||
import { joinTextWithImagePlaceholder, NON_VISION_IMAGE_PLACEHOLDER, partitionVisionContent } from "./vision-guard";
|
||||
|
||||
export function encodeTextSignatureV1(id: string, phase?: TextSignatureV1["phase"]): string {
|
||||
const payload: TextSignatureV1 = { v: 1, id };
|
||||
@@ -121,23 +122,29 @@ export function convertResponsesInputContent(
|
||||
return [{ type: "input_text", text: content.toWellFormed() } satisfies ResponseInputText];
|
||||
}
|
||||
|
||||
const normalizedContent = content
|
||||
.map((item): ResponseInputContent => {
|
||||
if (item.type === "text") {
|
||||
return {
|
||||
type: "input_text",
|
||||
text: item.text.toWellFormed(),
|
||||
} satisfies ResponseInputText;
|
||||
}
|
||||
return {
|
||||
type: "input_image",
|
||||
detail: "auto",
|
||||
image_url: `data:${item.mimeType};base64,${item.data}`,
|
||||
} satisfies ResponseInputImage;
|
||||
})
|
||||
.filter(item => supportsImages || item.type !== "input_image")
|
||||
.filter(item => item.type !== "input_text" || item.text.trim().length > 0);
|
||||
|
||||
const { textBlocks, imageBlocks, omittedImages } = partitionVisionContent(content, supportsImages);
|
||||
const normalizedContent: ResponseInputContent[] = [];
|
||||
for (const item of textBlocks) {
|
||||
const text = item.text.toWellFormed();
|
||||
if (text.trim().length === 0) continue;
|
||||
normalizedContent.push({
|
||||
type: "input_text",
|
||||
text,
|
||||
} satisfies ResponseInputText);
|
||||
}
|
||||
for (const item of imageBlocks) {
|
||||
normalizedContent.push({
|
||||
type: "input_image",
|
||||
detail: "auto",
|
||||
image_url: `data:${item.mimeType};base64,${item.data}`,
|
||||
} satisfies ResponseInputImage);
|
||||
}
|
||||
if (omittedImages) {
|
||||
normalizedContent.push({
|
||||
type: "input_text",
|
||||
text: NON_VISION_IMAGE_PLACEHOLDER,
|
||||
} satisfies ResponseInputText);
|
||||
}
|
||||
return normalizedContent.length > 0 ? normalizedContent : undefined;
|
||||
}
|
||||
|
||||
@@ -225,17 +232,25 @@ export function appendResponsesToolResultMessages<TApi extends Api>(
|
||||
knownCallIds: ReadonlySet<string>,
|
||||
customCallIds?: ReadonlySet<string>,
|
||||
): void {
|
||||
const supportsImages = model.input.includes("image");
|
||||
const textResult = toolResult.content
|
||||
.filter((block): block is TextContent => block.type === "text")
|
||||
.map(block => block.text)
|
||||
.join("\n");
|
||||
const hasImages = toolResult.content.some((block): block is ImageContent => block.type === "image");
|
||||
const omittedImages = hasImages && !supportsImages;
|
||||
const normalized = normalizeResponsesToolCallId(toolResult.toolCallId);
|
||||
if (strictResponsesPairing && !knownCallIds.has(normalized.callId)) {
|
||||
return;
|
||||
}
|
||||
|
||||
const output = (textResult.length > 0 ? textResult : "(see attached image)").toWellFormed();
|
||||
const output = (
|
||||
omittedImages
|
||||
? joinTextWithImagePlaceholder(textResult, true)
|
||||
: textResult.length > 0
|
||||
? textResult
|
||||
: "(see attached image)"
|
||||
).toWellFormed();
|
||||
if (customCallIds?.has(normalized.callId)) {
|
||||
messages.push({
|
||||
type: "custom_tool_call_output",
|
||||
@@ -250,7 +265,7 @@ export function appendResponsesToolResultMessages<TApi extends Api>(
|
||||
});
|
||||
}
|
||||
|
||||
if (!hasImages || !model.input.includes("image")) {
|
||||
if (!hasImages || !supportsImages) {
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
import type { ImageContent, TextContent } from "../types";
|
||||
|
||||
export const NON_VISION_IMAGE_PLACEHOLDER = "[image omitted: model does not support vision]";
|
||||
|
||||
export function partitionVisionContent(
|
||||
content: ReadonlyArray<TextContent | ImageContent>,
|
||||
supportsImages: boolean,
|
||||
): {
|
||||
textBlocks: TextContent[];
|
||||
imageBlocks: ImageContent[];
|
||||
omittedImages: boolean;
|
||||
} {
|
||||
const textBlocks = content.filter((block): block is TextContent => block.type === "text");
|
||||
const imageBlocks = content.filter((block): block is ImageContent => block.type === "image");
|
||||
return {
|
||||
textBlocks,
|
||||
imageBlocks: supportsImages ? imageBlocks : [],
|
||||
omittedImages: !supportsImages && imageBlocks.length > 0,
|
||||
};
|
||||
}
|
||||
|
||||
export function joinTextWithImagePlaceholder(text: string, omittedImages: boolean): string {
|
||||
const parts: string[] = [];
|
||||
if (text.length > 0) {
|
||||
parts.push(text);
|
||||
}
|
||||
if (omittedImages) {
|
||||
parts.push(NON_VISION_IMAGE_PLACEHOLDER);
|
||||
}
|
||||
return parts.join("\n");
|
||||
}
|
||||
@@ -0,0 +1,278 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { convertAnthropicMessages } from "../src/providers/anthropic";
|
||||
import { convertMessages as convertGoogleMessages } from "../src/providers/google-shared";
|
||||
import { convertCodexResponsesMessages } from "../src/providers/openai-codex-responses";
|
||||
import { convertMessages as convertOpenAICompletionsMessages } from "../src/providers/openai-completions";
|
||||
import {
|
||||
appendResponsesToolResultMessages,
|
||||
convertResponsesInputContent,
|
||||
} from "../src/providers/openai-responses-shared";
|
||||
import { NON_VISION_IMAGE_PLACEHOLDER } from "../src/providers/vision-guard";
|
||||
import type { Api, AssistantMessage, Context, Model, OpenAICompat, ToolResultMessage, Usage } from "../src/types";
|
||||
|
||||
const emptyUsage: Usage = {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
};
|
||||
|
||||
const compat: Required<OpenAICompat> = {
|
||||
supportsStore: true,
|
||||
supportsDeveloperRole: true,
|
||||
supportsMultipleSystemMessages: true,
|
||||
supportsReasoningEffort: true,
|
||||
reasoningEffortMap: {},
|
||||
supportsUsageInStreaming: true,
|
||||
supportsToolChoice: true,
|
||||
disableReasoningOnForcedToolChoice: false,
|
||||
disableReasoningOnToolChoice: false,
|
||||
maxTokensField: "max_completion_tokens",
|
||||
requiresToolResultName: false,
|
||||
requiresAssistantAfterToolResult: false,
|
||||
requiresThinkingAsText: false,
|
||||
requiresMistralToolIds: false,
|
||||
thinkingFormat: "openai",
|
||||
reasoningContentField: "reasoning_content",
|
||||
requiresReasoningContentForToolCalls: false,
|
||||
allowsSyntheticReasoningContentForToolCalls: true,
|
||||
requiresAssistantContentForToolCalls: false,
|
||||
openRouterRouting: {},
|
||||
vercelGatewayRouting: {},
|
||||
extraBody: {},
|
||||
supportsStrictMode: true,
|
||||
toolStrictMode: "none",
|
||||
};
|
||||
|
||||
function makeModel<TApi extends Api>(api: TApi, provider: Model["provider"]): Model<TApi> {
|
||||
return {
|
||||
id: `${provider}-${api}-text-only`,
|
||||
name: `${provider} ${api}`,
|
||||
api,
|
||||
provider,
|
||||
baseUrl: "https://example.com",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128_000,
|
||||
maxTokens: 8_192,
|
||||
};
|
||||
}
|
||||
|
||||
function makeAssistant(api: Model["api"], provider: Model["provider"], modelId: string): AssistantMessage {
|
||||
return {
|
||||
role: "assistant",
|
||||
content: [{ type: "toolCall", id: "call_1", name: "python", arguments: { code: "plot()" } }],
|
||||
api,
|
||||
provider,
|
||||
model: modelId,
|
||||
usage: emptyUsage,
|
||||
stopReason: "toolUse",
|
||||
timestamp: 2,
|
||||
};
|
||||
}
|
||||
|
||||
function makeToolResult(content: ToolResultMessage["content"]): ToolResultMessage {
|
||||
return {
|
||||
role: "toolResult",
|
||||
toolCallId: "call_1",
|
||||
toolName: "python",
|
||||
content,
|
||||
isError: false,
|
||||
timestamp: 3,
|
||||
};
|
||||
}
|
||||
|
||||
function countTaggedValues(value: unknown, tag: string): number {
|
||||
if (Array.isArray(value)) {
|
||||
return value.reduce((sum, item) => sum + countTaggedValues(item, tag), 0);
|
||||
}
|
||||
if (!value || typeof value !== "object") {
|
||||
return 0;
|
||||
}
|
||||
const record = value as Record<string, unknown>;
|
||||
const own = record.type === tag ? 1 : 0;
|
||||
return Object.values(record).reduce<number>((sum, item) => sum + countTaggedValues(item, tag), own);
|
||||
}
|
||||
|
||||
function countObjectKeys(value: unknown, key: string): number {
|
||||
if (Array.isArray(value)) {
|
||||
return value.reduce((sum, item) => sum + countObjectKeys(item, key), 0);
|
||||
}
|
||||
if (!value || typeof value !== "object") {
|
||||
return 0;
|
||||
}
|
||||
const record = value as Record<string, unknown>;
|
||||
const own = Object.hasOwn(record, key) ? 1 : 0;
|
||||
return Object.values(record).reduce<number>((sum, item) => sum + countObjectKeys(item, key), own);
|
||||
}
|
||||
|
||||
describe("issue #967 vision guard", () => {
|
||||
it("strips non-vision images from OpenAI chat-completions user and tool-result payloads", () => {
|
||||
const model = makeModel("openai-completions", "openrouter");
|
||||
const context: Context = {
|
||||
messages: [
|
||||
{
|
||||
role: "user",
|
||||
content: [
|
||||
{ type: "text", text: "plot summary" },
|
||||
{ type: "image", mimeType: "image/png", data: "ZmFrZQ==" },
|
||||
],
|
||||
timestamp: 1,
|
||||
},
|
||||
makeAssistant(model.api, model.provider, model.id),
|
||||
makeToolResult([
|
||||
{ type: "text", text: "saved plot to /tmp/plot.png" },
|
||||
{ type: "image", mimeType: "image/png", data: "ZmFrZQ==" },
|
||||
]),
|
||||
],
|
||||
};
|
||||
|
||||
const messages = convertOpenAICompletionsMessages(model, context, compat);
|
||||
expect(countTaggedValues(messages, "image_url")).toBe(0);
|
||||
expect(messages.filter(message => message.role === "user")).toHaveLength(1);
|
||||
expect(messages[0]).toMatchObject({
|
||||
role: "user",
|
||||
content: [
|
||||
{ type: "text", text: "plot summary" },
|
||||
{ type: "text", text: NON_VISION_IMAGE_PLACEHOLDER },
|
||||
],
|
||||
});
|
||||
expect(messages.find(message => message.role === "tool")).toMatchObject({
|
||||
content: `saved plot to /tmp/plot.png\n${NON_VISION_IMAGE_PLACEHOLDER}`,
|
||||
});
|
||||
});
|
||||
|
||||
it("strips non-vision images from OpenAI responses payload builders", () => {
|
||||
const model = makeModel("openai-responses", "openrouter");
|
||||
const userContent = convertResponsesInputContent(
|
||||
[
|
||||
{ type: "text", text: "plot summary" },
|
||||
{ type: "image", mimeType: "image/png", data: "ZmFrZQ==" },
|
||||
],
|
||||
false,
|
||||
);
|
||||
expect(countTaggedValues(userContent, "input_image")).toBe(0);
|
||||
expect(userContent).toEqual([
|
||||
{ type: "input_text", text: "plot summary" },
|
||||
{ type: "input_text", text: NON_VISION_IMAGE_PLACEHOLDER },
|
||||
]);
|
||||
|
||||
const payload: unknown[] = [];
|
||||
appendResponsesToolResultMessages(
|
||||
payload as never,
|
||||
makeToolResult([
|
||||
{ type: "text", text: "saved plot to /tmp/plot.png" },
|
||||
{ type: "image", mimeType: "image/png", data: "ZmFrZQ==" },
|
||||
]),
|
||||
model,
|
||||
true,
|
||||
new Set(["call_1"]),
|
||||
);
|
||||
expect(countTaggedValues(payload, "input_image")).toBe(0);
|
||||
expect(payload).toEqual([
|
||||
{
|
||||
type: "function_call_output",
|
||||
call_id: "call_1",
|
||||
output: `saved plot to /tmp/plot.png\n${NON_VISION_IMAGE_PLACEHOLDER}`,
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
it("strips non-vision images from Codex responses user and tool-result payloads", () => {
|
||||
const model = makeModel("openai-codex-responses", "openai-codex");
|
||||
const context: Context = {
|
||||
messages: [
|
||||
{
|
||||
role: "user",
|
||||
content: [
|
||||
{ type: "text", text: "plot summary" },
|
||||
{ type: "image", mimeType: "image/png", data: "ZmFrZQ==" },
|
||||
],
|
||||
timestamp: 1,
|
||||
},
|
||||
makeAssistant(model.api, model.provider, model.id),
|
||||
makeToolResult([{ type: "image", mimeType: "image/png", data: "ZmFrZQ==" }]),
|
||||
],
|
||||
};
|
||||
|
||||
const messages = convertCodexResponsesMessages(model, context);
|
||||
expect(countTaggedValues(messages, "input_image")).toBe(0);
|
||||
expect(messages.filter(item => (item as { role?: string }).role === "user")).toHaveLength(1);
|
||||
expect(messages[0]).toMatchObject({
|
||||
role: "user",
|
||||
content: [
|
||||
{ type: "input_text", text: "plot summary" },
|
||||
{ type: "input_text", text: NON_VISION_IMAGE_PLACEHOLDER },
|
||||
],
|
||||
});
|
||||
expect(messages.find(item => (item as { type?: string }).type === "function_call_output")).toMatchObject({
|
||||
output: NON_VISION_IMAGE_PLACEHOLDER,
|
||||
});
|
||||
});
|
||||
|
||||
it("strips non-vision images from Anthropic payloads", () => {
|
||||
const model = makeModel("anthropic-messages", "anthropic");
|
||||
const messages = convertAnthropicMessages(
|
||||
[
|
||||
{
|
||||
role: "user",
|
||||
content: [
|
||||
{ type: "text", text: "plot summary" },
|
||||
{ type: "image", mimeType: "image/png", data: "ZmFrZQ==" },
|
||||
],
|
||||
timestamp: 1,
|
||||
},
|
||||
makeAssistant(model.api, model.provider, model.id),
|
||||
makeToolResult([{ type: "image", mimeType: "image/png", data: "ZmFrZQ==" }]),
|
||||
],
|
||||
model,
|
||||
false,
|
||||
);
|
||||
expect(countTaggedValues(messages, "image")).toBe(0);
|
||||
expect(messages[0]).toMatchObject({ role: "user", content: `plot summary\n${NON_VISION_IMAGE_PLACEHOLDER}` });
|
||||
const toolResult = messages.at(-1) as { role: string; content: Array<{ type: string; content: unknown }> };
|
||||
expect(toolResult.role).toBe("user");
|
||||
expect(toolResult.content[0]).toMatchObject({
|
||||
type: "tool_result",
|
||||
content: NON_VISION_IMAGE_PLACEHOLDER,
|
||||
});
|
||||
});
|
||||
|
||||
it("strips non-vision images from Google payloads", () => {
|
||||
const model = makeModel("google-generative-ai", "google");
|
||||
const context: Context = {
|
||||
messages: [
|
||||
{
|
||||
role: "user",
|
||||
content: [
|
||||
{ type: "text", text: "plot summary" },
|
||||
{ type: "image", mimeType: "image/png", data: "ZmFrZQ==" },
|
||||
],
|
||||
timestamp: 1,
|
||||
},
|
||||
makeAssistant(model.api, model.provider, model.id),
|
||||
makeToolResult([{ type: "image", mimeType: "image/png", data: "ZmFrZQ==" }]),
|
||||
],
|
||||
};
|
||||
|
||||
const messages = convertGoogleMessages(model, context);
|
||||
expect(countObjectKeys(messages, "inlineData")).toBe(0);
|
||||
expect(messages[0]).toMatchObject({
|
||||
role: "user",
|
||||
parts: [{ text: "plot summary" }, { text: NON_VISION_IMAGE_PLACEHOLDER }],
|
||||
});
|
||||
expect(messages.at(-1)).toMatchObject({
|
||||
role: "user",
|
||||
parts: [
|
||||
{
|
||||
functionResponse: {
|
||||
response: { output: NON_VISION_IMAGE_PLACEHOLDER },
|
||||
},
|
||||
},
|
||||
],
|
||||
});
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user