From a3d1f35099fa459f8490c605af22eeb8acd85e2c Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 3 Aug 2026 02:40:04 +0000 Subject: [PATCH] fix(openai): preserved Codex native image results - Normalized result-bearing Codex image items on terminal output events and emitted standard image content. - Preserved result-bearing image calls during full Responses history replay despite stale provider status. - Added stream and replay regressions for the Codex path. Fixes #7445 --- packages/ai/CHANGELOG.md | 1 + .../src/providers/openai-codex-responses.ts | 12 +++++- packages/ai/src/providers/openai-shared.ts | 33 +++++++++++------ packages/ai/src/utils.ts | 2 +- packages/ai/test/openai-codex-stream.test.ts | 24 +++++++++++- .../openai-responses-history-payload.test.ts | 37 +++++++++++-------- 6 files changed, 79 insertions(+), 30 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index d6aa6f6c1..abbb23e0e 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,6 +4,7 @@ ### Fixed +- Fixed Codex Responses dropping native image-generation results from assistant content and replay when terminal output items retained a stale `generating` status ([#7445](https://github.com/can1357/oh-my-pi/issues/7445)). - Fixed Anthropic streams truncated mid-generation (connection closed with neither a `message_delta` stop_reason nor a `message_stop` frame) finalizing the partial message as a clean `stop`, which made the agent loop treat a truncated turn as complete and halt silently mid-sentence. Such streams raise the stream-envelope error again: transparently retried before replay-unsafe content streams; afterwards the turn surfaces as an error whose complete tool calls the agent loop still runs (`recoverTransientErrorToolTurn` now recognizes the envelope-error text after `retainCompletedToolCalls` drops half-streamed calls). Streams that delivered a `stop_reason` (or `message_stop`) keep degrading to best-effort content when the other terminal frame is missing. - Fixed Anthropic prompt caching writing a fresh entry for the entire system prefix whenever the project footer (cwd, date, workspace tree) changed. `applyPromptCaching` placed its only system breakpoint on the last block — normally the volatile footer — so starting omp in a new directory or crossing midnight re-wrote the whole cached system prefix instead of reusing it (issue [#7324](https://github.com/can1357/oh-my-pi/issues/7324)). System caching now marks up to the last three eligible blocks, covering both `[stable prefix, project footer]` and `[stable prefix, project footer, active-repo context]` layouts while skipping the OAuth cloak blocks (billing header + Claude Code identity). Message caching also skips the synthetic trailing `Continue.` pad and anchors on the preceding real assistant turn when the four-breakpoint budget is tight. This does not address open-weight chat templates that render tool schemas after the system block; keeping those cached requires relocating the per-request footer out of the system message. diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 6cd1435d1..f4107d17a 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -84,6 +84,7 @@ import type { ResponseFunctionToolCall, ResponseInput, ResponseInputContent, + ResponseOutputItem, ResponseOutputMessage, ResponseReasoningItem, ResponseStatus, @@ -96,6 +97,7 @@ import { appendReasoningSummaryPart, appendReasoningSummaryPartDone, appendReasoningSummaryTextDelta, + appendResponsesImageResult, appendResponsesToolResultMessages, applyOpenAIServiceTier, applyReasoningSummaryDone, @@ -371,7 +373,8 @@ type CodexEventItem = | ResponseOutputMessage | ResponseFunctionToolCall | ResponseCustomToolCall - | ResponseComputerToolCall; + | ResponseComputerToolCall + | ResponseOutputItem.ImageGenerationCall; type CodexOutputBlock = | ThinkingContent | TextContent @@ -2289,6 +2292,7 @@ class CodexStreamProcessor { const rawItem = rawEvent.item; if (!rawItem || typeof rawItem !== "object") return; const item = structuredCloneJSON(rawItem) as CodexEventItem; + if (item.type === "image_generation_call" && item.result) item.status = "completed"; runtime.nativeOutputItems.push(item as unknown as Record); // Match the finalization to the OPEN ITEM that started this block, not the @@ -2301,6 +2305,12 @@ class CodexStreamProcessor { const block = entry?.block ?? null; const contentIndex = entry?.contentIndex ?? output.content.length - 1; + if (item.type === "image_generation_call" && item.result) { + appendResponsesImageResult(output, stream, item.result); + runtime.closeOpenItem(entry); + return; + } + if (item.type === "reasoning" && block?.type === "thinking") { this.#flushSummaryDeltas(entry); block.thinking = finalizeReasoningThinking( diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index 3c3eab8ab..000bbfbca 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -2417,6 +2417,26 @@ export function computerCallMetadata(item: ResponseComputerToolCall): ComputerTo }; } +/** Append a native Responses image result and emit its completion event. */ +export function appendResponsesImageResult( + output: AssistantMessage, + stream: AssistantMessageEventStream, + result: string, +): void { + const image: ImageContent = { + type: "image", + data: result, + mimeType: parseImageMetadata(Buffer.from(result, "base64"))?.mimeType ?? "image/png", + }; + output.content.push(image); + stream.push({ + type: "image_end", + contentIndex: output.content.length - 1, + content: image, + partial: output, + }); +} + export async function processResponsesStream( openaiStream: AsyncIterable, output: AssistantMessage, @@ -2930,18 +2950,7 @@ export async function processResponsesStream( closeOpenItem(event.output_index, item.id, entry, item.call_id, prefixedFunctionCallItemKey(item.call_id)); stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); } else if (item.type === "image_generation_call" && item.status === "completed" && item.result) { - const image: ImageContent = { - type: "image", - data: item.result, - mimeType: parseImageMetadata(Buffer.from(item.result, "base64"))?.mimeType ?? "image/png", - }; - output.content.push(image); - stream.push({ - type: "image_end", - contentIndex: output.content.length - 1, - content: image, - partial: output, - }); + appendResponsesImageResult(output, stream, item.result); } } else if (terminalEvent) { const response = terminalEvent.response; diff --git a/packages/ai/src/utils.ts b/packages/ai/src/utils.ts index 5890400d9..177c9c63a 100644 --- a/packages/ai/src/utils.ts +++ b/packages/ai/src/utils.ts @@ -403,7 +403,7 @@ function sanitizeOpenAIResponsesReasoningItemForReplay( function sanitizeOpenAIResponsesImageGenerationCallForReplay( item: Record, ): ResponseInputItem.ImageGenerationCall | undefined { - if (typeof item.id !== "string" || item.status !== "completed" || typeof item.result !== "string") { + if (typeof item.id !== "string" || typeof item.result !== "string" || item.result.length === 0) { return undefined; } return { diff --git a/packages/ai/test/openai-codex-stream.test.ts b/packages/ai/test/openai-codex-stream.test.ts index 2890f76bb..4e4c543dd 100644 --- a/packages/ai/test/openai-codex-stream.test.ts +++ b/packages/ai/test/openai-codex-stream.test.ts @@ -509,6 +509,7 @@ describe("openai-codex streaming", () => { const fetchMock: FetchImpl = async () => new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } }); const textEndContents: string[] = []; + const eventTypes: string[] = []; const stream = streamOpenAICodexResponses(model, context, { apiKey: token, @@ -516,15 +517,36 @@ describe("openai-codex streaming", () => { }); const readPromise = (async () => { for await (const event of stream) { + eventTypes.push(event.type); if (event.type === "text_end") textEndContents.push(event.content); } })(); const result = await stream.result(); await readPromise; - return { result, textEndContents }; + return { result, textEndContents, eventTypes }; } + it("surfaces result-bearing native images with stale generating status", async () => { + const data = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+A8AAQUBAScY42YAAAAASUVORK5CYII="; + const { result, eventTypes } = await runCodexSseEvents([ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "image_generation_call", id: "ig_1", status: "generating", result: null }, + }, + { + type: "response.output_item.done", + output_index: 0, + item: { type: "image_generation_call", id: "ig_1", status: "generating", result: data }, + }, + { type: "response.completed", response: { id: "resp_image", status: "completed" } }, + ]); + + expect(result.content).toEqual([{ type: "image", data, mimeType: "image/png" }]); + expect(eventTypes).toContain("image_end"); + }); + for (const testCase of [ { name: "absent terminal content preserves streamed text", diff --git a/packages/ai/test/openai-responses-history-payload.test.ts b/packages/ai/test/openai-responses-history-payload.test.ts index 86e7d85c3..58ebd7a64 100644 --- a/packages/ai/test/openai-responses-history-payload.test.ts +++ b/packages/ai/test/openai-responses-history-payload.test.ts @@ -726,8 +726,8 @@ describe("OpenAI responses history payload", () => { ]); }); - it("drops unfinished image generation calls from replayed native history", async () => { - const model = getOpenAIReasoningModel("openai", "gpt-5-mini"); + it("normalizes result-bearing native images for full Codex replay", () => { + const model = getBundledModel<"openai-codex-responses">("openai-codex", "gpt-5.5"); const context: Context = { messages: [ { role: "user", content: "first user", timestamp: Date.now() }, @@ -742,36 +742,43 @@ describe("OpenAI responses history payload", () => { id: "ig_generating", type: "image_generation_call", status: "generating", - action: "generate", + }, + { + id: "ig_stale_result", + type: "image_generation_call", + status: "generating", + result: "stale-result-image", }, { id: "ig_completed", type: "image_generation_call", status: "completed", - result: "base64-image", - action: "generate", - background: "opaque", - output_format: "png", - quality: "medium", + result: "completed-image", }, ], - true, + false, + "openai-codex", + model.id, ), { role: "user", content: "follow-up user", timestamp: Date.now() }, ], }; - const payload = (await captureResponsesPayload(model, context)) as { input?: unknown[] }; - const imageGenerationItems = payload.input?.filter(item => { - if (!item || typeof item !== "object") return false; - return (item as { type?: unknown }).type === "image_generation_call"; - }); + const imageGenerationItems = convertCodexResponsesMessages(model, context).filter( + item => item.type === "image_generation_call", + ); expect(imageGenerationItems).toEqual([ + { + id: "ig_stale_result", + type: "image_generation_call", + status: "completed", + result: "stale-result-image", + }, { id: "ig_completed", type: "image_generation_call", status: "completed", - result: "base64-image", + result: "completed-image", }, ]); });