fix(openai): preserved Codex native image results
- Normalized result-bearing Codex image items on terminal output events and emitted standard image content. - Preserved result-bearing image calls during full Responses history replay despite stale provider status. - Added stream and replay regressions for the Codex path. Fixes #7445
This commit is contained in:
@@ -4,6 +4,7 @@
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Codex Responses dropping native image-generation results from assistant content and replay when terminal output items retained a stale `generating` status ([#7445](https://github.com/can1357/oh-my-pi/issues/7445)).
|
||||
- Fixed Anthropic streams truncated mid-generation (connection closed with neither a `message_delta` stop_reason nor a `message_stop` frame) finalizing the partial message as a clean `stop`, which made the agent loop treat a truncated turn as complete and halt silently mid-sentence. Such streams raise the stream-envelope error again: transparently retried before replay-unsafe content streams; afterwards the turn surfaces as an error whose complete tool calls the agent loop still runs (`recoverTransientErrorToolTurn` now recognizes the envelope-error text after `retainCompletedToolCalls` drops half-streamed calls). Streams that delivered a `stop_reason` (or `message_stop`) keep degrading to best-effort content when the other terminal frame is missing.
|
||||
- Fixed Anthropic prompt caching writing a fresh entry for the entire system prefix whenever the project footer (cwd, date, workspace tree) changed. `applyPromptCaching` placed its only system breakpoint on the last block — normally the volatile footer — so starting omp in a new directory or crossing midnight re-wrote the whole cached system prefix instead of reusing it (issue [#7324](https://github.com/can1357/oh-my-pi/issues/7324)). System caching now marks up to the last three eligible blocks, covering both `[stable prefix, project footer]` and `[stable prefix, project footer, active-repo context]` layouts while skipping the OAuth cloak blocks (billing header + Claude Code identity). Message caching also skips the synthetic trailing `Continue.` pad and anchors on the preceding real assistant turn when the four-breakpoint budget is tight. This does not address open-weight chat templates that render tool schemas after the system block; keeping those cached requires relocating the per-request footer out of the system message.
|
||||
|
||||
|
||||
@@ -84,6 +84,7 @@ import type {
|
||||
ResponseFunctionToolCall,
|
||||
ResponseInput,
|
||||
ResponseInputContent,
|
||||
ResponseOutputItem,
|
||||
ResponseOutputMessage,
|
||||
ResponseReasoningItem,
|
||||
ResponseStatus,
|
||||
@@ -96,6 +97,7 @@ import {
|
||||
appendReasoningSummaryPart,
|
||||
appendReasoningSummaryPartDone,
|
||||
appendReasoningSummaryTextDelta,
|
||||
appendResponsesImageResult,
|
||||
appendResponsesToolResultMessages,
|
||||
applyOpenAIServiceTier,
|
||||
applyReasoningSummaryDone,
|
||||
@@ -371,7 +373,8 @@ type CodexEventItem =
|
||||
| ResponseOutputMessage
|
||||
| ResponseFunctionToolCall
|
||||
| ResponseCustomToolCall
|
||||
| ResponseComputerToolCall;
|
||||
| ResponseComputerToolCall
|
||||
| ResponseOutputItem.ImageGenerationCall;
|
||||
type CodexOutputBlock =
|
||||
| ThinkingContent
|
||||
| TextContent
|
||||
@@ -2289,6 +2292,7 @@ class CodexStreamProcessor {
|
||||
const rawItem = rawEvent.item;
|
||||
if (!rawItem || typeof rawItem !== "object") return;
|
||||
const item = structuredCloneJSON(rawItem) as CodexEventItem;
|
||||
if (item.type === "image_generation_call" && item.result) item.status = "completed";
|
||||
runtime.nativeOutputItems.push(item as unknown as Record<string, unknown>);
|
||||
|
||||
// Match the finalization to the OPEN ITEM that started this block, not the
|
||||
@@ -2301,6 +2305,12 @@ class CodexStreamProcessor {
|
||||
const block = entry?.block ?? null;
|
||||
const contentIndex = entry?.contentIndex ?? output.content.length - 1;
|
||||
|
||||
if (item.type === "image_generation_call" && item.result) {
|
||||
appendResponsesImageResult(output, stream, item.result);
|
||||
runtime.closeOpenItem(entry);
|
||||
return;
|
||||
}
|
||||
|
||||
if (item.type === "reasoning" && block?.type === "thinking") {
|
||||
this.#flushSummaryDeltas(entry);
|
||||
block.thinking = finalizeReasoningThinking(
|
||||
|
||||
@@ -2417,6 +2417,26 @@ export function computerCallMetadata(item: ResponseComputerToolCall): ComputerTo
|
||||
};
|
||||
}
|
||||
|
||||
/** Append a native Responses image result and emit its completion event. */
|
||||
export function appendResponsesImageResult(
|
||||
output: AssistantMessage,
|
||||
stream: AssistantMessageEventStream,
|
||||
result: string,
|
||||
): void {
|
||||
const image: ImageContent = {
|
||||
type: "image",
|
||||
data: result,
|
||||
mimeType: parseImageMetadata(Buffer.from(result, "base64"))?.mimeType ?? "image/png",
|
||||
};
|
||||
output.content.push(image);
|
||||
stream.push({
|
||||
type: "image_end",
|
||||
contentIndex: output.content.length - 1,
|
||||
content: image,
|
||||
partial: output,
|
||||
});
|
||||
}
|
||||
|
||||
export async function processResponsesStream<TApi extends Api>(
|
||||
openaiStream: AsyncIterable<ResponseStreamEvent>,
|
||||
output: AssistantMessage,
|
||||
@@ -2930,18 +2950,7 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
closeOpenItem(event.output_index, item.id, entry, item.call_id, prefixedFunctionCallItemKey(item.call_id));
|
||||
stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
|
||||
} else if (item.type === "image_generation_call" && item.status === "completed" && item.result) {
|
||||
const image: ImageContent = {
|
||||
type: "image",
|
||||
data: item.result,
|
||||
mimeType: parseImageMetadata(Buffer.from(item.result, "base64"))?.mimeType ?? "image/png",
|
||||
};
|
||||
output.content.push(image);
|
||||
stream.push({
|
||||
type: "image_end",
|
||||
contentIndex: output.content.length - 1,
|
||||
content: image,
|
||||
partial: output,
|
||||
});
|
||||
appendResponsesImageResult(output, stream, item.result);
|
||||
}
|
||||
} else if (terminalEvent) {
|
||||
const response = terminalEvent.response;
|
||||
|
||||
@@ -403,7 +403,7 @@ function sanitizeOpenAIResponsesReasoningItemForReplay(
|
||||
function sanitizeOpenAIResponsesImageGenerationCallForReplay(
|
||||
item: Record<string, unknown>,
|
||||
): ResponseInputItem.ImageGenerationCall | undefined {
|
||||
if (typeof item.id !== "string" || item.status !== "completed" || typeof item.result !== "string") {
|
||||
if (typeof item.id !== "string" || typeof item.result !== "string" || item.result.length === 0) {
|
||||
return undefined;
|
||||
}
|
||||
return {
|
||||
|
||||
@@ -509,6 +509,7 @@ describe("openai-codex streaming", () => {
|
||||
const fetchMock: FetchImpl = async () =>
|
||||
new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } });
|
||||
const textEndContents: string[] = [];
|
||||
const eventTypes: string[] = [];
|
||||
|
||||
const stream = streamOpenAICodexResponses(model, context, {
|
||||
apiKey: token,
|
||||
@@ -516,15 +517,36 @@ describe("openai-codex streaming", () => {
|
||||
});
|
||||
const readPromise = (async () => {
|
||||
for await (const event of stream) {
|
||||
eventTypes.push(event.type);
|
||||
if (event.type === "text_end") textEndContents.push(event.content);
|
||||
}
|
||||
})();
|
||||
const result = await stream.result();
|
||||
await readPromise;
|
||||
|
||||
return { result, textEndContents };
|
||||
return { result, textEndContents, eventTypes };
|
||||
}
|
||||
|
||||
it("surfaces result-bearing native images with stale generating status", async () => {
|
||||
const data = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+A8AAQUBAScY42YAAAAASUVORK5CYII=";
|
||||
const { result, eventTypes } = await runCodexSseEvents([
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
output_index: 0,
|
||||
item: { type: "image_generation_call", id: "ig_1", status: "generating", result: null },
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: { type: "image_generation_call", id: "ig_1", status: "generating", result: data },
|
||||
},
|
||||
{ type: "response.completed", response: { id: "resp_image", status: "completed" } },
|
||||
]);
|
||||
|
||||
expect(result.content).toEqual([{ type: "image", data, mimeType: "image/png" }]);
|
||||
expect(eventTypes).toContain("image_end");
|
||||
});
|
||||
|
||||
for (const testCase of [
|
||||
{
|
||||
name: "absent terminal content preserves streamed text",
|
||||
|
||||
@@ -726,8 +726,8 @@ describe("OpenAI responses history payload", () => {
|
||||
]);
|
||||
});
|
||||
|
||||
it("drops unfinished image generation calls from replayed native history", async () => {
|
||||
const model = getOpenAIReasoningModel("openai", "gpt-5-mini");
|
||||
it("normalizes result-bearing native images for full Codex replay", () => {
|
||||
const model = getBundledModel<"openai-codex-responses">("openai-codex", "gpt-5.5");
|
||||
const context: Context = {
|
||||
messages: [
|
||||
{ role: "user", content: "first user", timestamp: Date.now() },
|
||||
@@ -742,36 +742,43 @@ describe("OpenAI responses history payload", () => {
|
||||
id: "ig_generating",
|
||||
type: "image_generation_call",
|
||||
status: "generating",
|
||||
action: "generate",
|
||||
},
|
||||
{
|
||||
id: "ig_stale_result",
|
||||
type: "image_generation_call",
|
||||
status: "generating",
|
||||
result: "stale-result-image",
|
||||
},
|
||||
{
|
||||
id: "ig_completed",
|
||||
type: "image_generation_call",
|
||||
status: "completed",
|
||||
result: "base64-image",
|
||||
action: "generate",
|
||||
background: "opaque",
|
||||
output_format: "png",
|
||||
quality: "medium",
|
||||
result: "completed-image",
|
||||
},
|
||||
],
|
||||
true,
|
||||
false,
|
||||
"openai-codex",
|
||||
model.id,
|
||||
),
|
||||
{ role: "user", content: "follow-up user", timestamp: Date.now() },
|
||||
],
|
||||
};
|
||||
const payload = (await captureResponsesPayload(model, context)) as { input?: unknown[] };
|
||||
const imageGenerationItems = payload.input?.filter(item => {
|
||||
if (!item || typeof item !== "object") return false;
|
||||
return (item as { type?: unknown }).type === "image_generation_call";
|
||||
});
|
||||
const imageGenerationItems = convertCodexResponsesMessages(model, context).filter(
|
||||
item => item.type === "image_generation_call",
|
||||
);
|
||||
|
||||
expect(imageGenerationItems).toEqual([
|
||||
{
|
||||
id: "ig_stale_result",
|
||||
type: "image_generation_call",
|
||||
status: "completed",
|
||||
result: "stale-result-image",
|
||||
},
|
||||
{
|
||||
id: "ig_completed",
|
||||
type: "image_generation_call",
|
||||
status: "completed",
|
||||
result: "base64-image",
|
||||
result: "completed-image",
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user