diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index e5d6c5da5..933e44250 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -7,6 +7,7 @@ - Fixed `/compact` (and automatic compaction) resurrecting pre-`/clear` conversation turns: `prepareCompaction` now honors the latest `reset_boundary`, so a compaction after an in-place `/clear` only summarizes messages created after the reset ([#8718](https://github.com/can1357/oh-my-pi/issues/8718)). - Hardened compaction summarization against prompt injection: conversation history and previous summaries are now treated as untrusted, and embedded ``/`` boundary tags are neutralized before prompt assembly ([#8727](https://github.com/can1357/oh-my-pi/pull/8727) by [@koopmannleon19977-cmyk](https://github.com/koopmannleon19977-cmyk)). - Compaction summarization input is now bounded to the summary model's context (windowed fold for oversized spans) and deterministic context-overflow 400s are no longer retried up to the full retry budget; artifact ids containing `503` no longer misclassify hard 400s as transient. +- Fixed remote compaction mirroring the #8789 Responses shape: `buildOpenAiNativeHistory` now hoists an assistant `message` wedged between a tool-call batch and its outputs ahead of the batch, so compaction requests to strict opencode-go gateways match the canonical `message(s) → calls → outputs` order ([#8789](https://github.com/can1357/oh-my-pi/issues/8789)). ## [17.3.5] - 2026-08-16 diff --git a/packages/agent/src/compaction/openai.ts b/packages/agent/src/compaction/openai.ts index 004c7c492..86ecbbfb3 100644 --- a/packages/agent/src/compaction/openai.ts +++ b/packages/agent/src/compaction/openai.ts @@ -22,7 +22,11 @@ import { createOpenAICodexCompatibilityMetadata, getCodexAttestationHeader, } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; -import { parseAzureDeploymentNameMap, parseTextSignature } from "@oh-my-pi/pi-ai/providers/openai-shared"; +import { + hoistInterleavedResponsesToolBatchMessages, + parseAzureDeploymentNameMap, + parseTextSignature, +} from "@oh-my-pi/pi-ai/providers/openai-shared"; import { transformMessages } from "@oh-my-pi/pi-ai/providers/transform-messages"; import type { Api, @@ -740,7 +744,7 @@ export function buildOpenAiNativeHistory( msgIndex++; } - return stripOpenAIResponsesOutputOnlyStatusesForReplay(input); + return stripOpenAIResponsesOutputOnlyStatusesForReplay(hoistInterleavedResponsesToolBatchMessages(input)); } // ============================================================================ diff --git a/packages/agent/test/remote-compaction.test.ts b/packages/agent/test/remote-compaction.test.ts index 702856a4b..a92f99b36 100644 --- a/packages/agent/test/remote-compaction.test.ts +++ b/packages/agent/test/remote-compaction.test.ts @@ -365,6 +365,45 @@ function toolResultFor(callId: string, custom = false): ToolResultMessage { }; } +describe("buildOpenAiNativeHistory interleaved assistant message (#8789)", () => { + test("hoists a trailing text block before its tool-call batch", () => { + // deepseek-v4-flash on opencode-go streamed [thinking, 2 tool calls, + // trailing "\n" }, + ], + timestamp: Date.now(), + provider: "opencode-go", + model: "deepseek-v4-flash", + api: "openai-responses", + usage: ZERO_USAGE, + stopReason: "toolUse", + }; + + const items = buildOpenAiNativeHistory([assistant, toolResultFor("call_a"), toolResultFor("call_b")], model); + + expect(items.map(item => item.type)).toEqual([ + "message", + "function_call", + "function_call", + "function_call_output", + "function_call_output", + ]); + expect(JSON.stringify(items[0]?.content)).toContain(" { test("registers function_call ids carried in providerPayload so later tool results are emitted", () => { const items = buildOpenAiNativeHistory( diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index cb40dada2..ab3567700 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -19,6 +19,7 @@ - Cloud Code Assist Gemini 3.6/3.7 Flash requests at `minimal` now send `thinkingLevel: LOW` on the aliased `-low` SKU instead of `MINIMAL`, which the API rejects with HTTP 400. - Answer Cursor `interaction_query` permission gates (hosted web search, Exa, unnamed field-9 WebFetch) so the Run RPC continues instead of sitting silent until the 300s idle watchdog. - Fixed provider tool calls arriving with flattened array argument paths (e.g. Gemini's `questions[0].id`) being stripped and rejected by argument validation; well-formed flattened paths are now rebuilt into the nested arrays the tool schema expects ([#8886](https://github.com/can1357/oh-my-pi/issues/8886)). +- Fixed opencode-go (Console Go) rejecting Responses turns with `400 No tool output found for tool call …` (naming a random call of the batch on each retry) when a model streamed a trailing text/thinking block after its tool calls: `buildResponsesInput` emitted that block as an assistant `message` item wedged between the `function_call` batch and its `function_call_output` items. Such interleaved messages are now hoisted ahead of their call batch (canonical `message(s) → calls → outputs`), which the strict gateway validator accepts; content is unchanged ([#8789](https://github.com/can1357/oh-my-pi/issues/8789)). ## [17.3.7] - 2026-08-17 diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index 5e7e96321..5452e1156 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -1522,6 +1522,77 @@ export function repairOrphanResponsesToolCalls(input: ResponseInput): ResponseIn return repaired; } +type ResponsesBatchItemKind = "call" | "output" | "assistant-message" | "other"; + +/** Classify a Responses input item for tool-call/output batch normalization. */ +function classifyResponsesBatchItem(item: object): ResponsesBatchItemKind { + const type = "type" in item ? item.type : undefined; + if (responsesToolCallKind(type) !== undefined) return "call"; + if (responsesToolOutputKind(type) !== undefined) return "output"; + const role = "role" in item ? item.role : undefined; + if (type === "message" && role === "assistant") return "assistant-message"; + return "other"; +} + +/** + * Relocate assistant `message` items wedged inside a tool-call → tool-output + * batch to before the batch, yielding canonical `message(s) → calls → outputs` + * order. Idempotent; returns the same array reference when nothing moves. + * + * OpenAI's Responses API pairs tool outputs by `call_id` and tolerates any item + * order, but stricter gateways (notably opencode-go's "Console Go") reject a + * shape where an assistant message interrupts a `function_call` → + * `function_call_output` run, 400ing with `No tool output found for tool call …` + * (naming a random call of the batch on each retry). This arises whenever a + * model streams a trailing text / demoted-thinking block *after* its tool calls: + * the block-encode path preserves stream order, emitting the message between the + * calls and the outputs appended afterward. Moving the already-model-owned + * message ahead of its call batch keeps content identical while satisfying the + * strict validator. See #8789. + */ +export function hoistInterleavedResponsesToolBatchMessages(items: readonly T[]): T[] { + const moved = new Set(); + const insertBefore = new Map(); + for (let index = 0; index < items.length; index++) { + if (classifyResponsesBatchItem(items[index]) !== "output") continue; + // Only anchor on the first output of a run. + if (index > 0 && classifyResponsesBatchItem(items[index - 1]) === "output") continue; + // Walk back over the batch body (calls interleaved with assistant messages). + let start = index; + let sawCall = false; + const messageIndexes: number[] = []; + while (start > 0) { + const kind = classifyResponsesBatchItem(items[start - 1]); + if (kind === "call") { + sawCall = true; + } else if (kind === "assistant-message") { + messageIndexes.push(start - 1); + } else { + break; + } + start -= 1; + } + // Nothing to hoist unless a message actually sits among the calls. + if (!sawCall || messageIndexes.length === 0) continue; + messageIndexes.reverse(); + const target = insertBefore.get(start) ?? []; + for (const messageIndex of messageIndexes) { + moved.add(messageIndex); + target.push(messageIndex); + } + insertBefore.set(start, target); + } + if (moved.size === 0) return items.slice(); + const result: T[] = []; + for (let index = 0; index < items.length; index++) { + const pending = insertBefore.get(index); + if (pending) for (const messageIndex of pending) result.push(items[messageIndex]); + if (moved.has(index)) continue; + result.push(items[index]); + } + return result; +} + /** * Some Responses backends (notably GitHub Copilot) reject the OpenAI image * `detail: "original"` value with a 400. When the model does not advertise @@ -1923,7 +1994,8 @@ export function buildResponsesInput(options: BuildResponsesInp msgIndex++; } - const withRepairedOutputs = options.repairOrphanOutputs ? repairOrphanResponsesToolOutputs(messages) : messages; + const hoisted = hoistInterleavedResponsesToolBatchMessages(messages); + const withRepairedOutputs = options.repairOrphanOutputs ? repairOrphanResponsesToolOutputs(hoisted) : hoisted; const withRepairedCalls = repairOrphanResponsesToolCalls(withRepairedOutputs); return stripUnpairedOpenAIResponsesComputerReasoningIdsForReplay(withRepairedCalls); } diff --git a/packages/ai/test/issue-8789-responses-interleaved-message.test.ts b/packages/ai/test/issue-8789-responses-interleaved-message.test.ts new file mode 100644 index 000000000..83463774f --- /dev/null +++ b/packages/ai/test/issue-8789-responses-interleaved-message.test.ts @@ -0,0 +1,142 @@ +import { describe, expect, it } from "bun:test"; +import type { ResponseInput } from "@oh-my-pi/pi-ai/providers/openai-responses-wire"; +import { buildResponsesInput } from "@oh-my-pi/pi-ai/providers/openai-shared"; +import type { AssistantMessage, Context, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +// deepseek-v4-flash on opencode-go (Console Go) — the gateway from #8789 that +// rejects an assistant message interleaved between a function_call batch and +// its function_call_output items. +const model = buildModel({ + id: "deepseek-v4-flash", + name: "DeepSeek V4 Flash", + api: "openai-responses", + provider: "opencode-go", + baseUrl: "https://opencode.ai/zen/go/v1", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128_000, + maxTokens: 16_000, +}); + +const zeroUsage = { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, +}; + +function wireType(item: ResponseInput[number]): string { + if ("type" in item && typeof item.type === "string") return item.type; + if ("role" in item && typeof item.role === "string") return `message:${item.role}`; + return "unknown"; +} + +function toolResult(callId: string, text: string): ToolResultMessage { + return { + role: "toolResult", + toolCallId: callId, + toolName: "read", + content: [{ type: "text", text }], + isError: false, + timestamp: 2, + }; +} + +describe("buildResponsesInput #8789 interleaved assistant message", () => { + it("hoists a trailing text block before its tool-call batch", () => { + // Model streamed [thinking, 3 tool calls, trailing "\n" }, + ], + api: model.api, + provider: model.provider, + model: model.id, + usage: zeroUsage, + stopReason: "toolUse", + timestamp: 1, + }; + const context: Context = { + messages: [ + assistant, + toolResult("call_a", "out a"), + toolResult("call_b", "out b"), + toolResult("call_c", "out c"), + { role: "user", content: "continue", timestamp: 5 }, + ], + }; + + const items = buildResponsesInput({ + model, + context, + strictResponsesPairing: false, + supportsImageDetailOriginal: false, + }); + + expect(items.map(wireType)).toEqual([ + "message", + "function_call", + "function_call", + "function_call", + "function_call_output", + "function_call_output", + "function_call_output", + "message:user", + ]); + + // The demoted-thinking text is preserved verbatim as the hoisted message. + const hoisted = items[0]; + expect(JSON.stringify(hoisted)).toContain(" { + const assistant: AssistantMessage = { + role: "assistant", + content: [ + { type: "text", text: "calling read on two files" }, + { type: "toolCall", id: "call_a", name: "read", arguments: { path: "a" } }, + { type: "toolCall", id: "call_b", name: "read", arguments: { path: "b" } }, + ], + api: model.api, + provider: model.provider, + model: model.id, + usage: zeroUsage, + stopReason: "toolUse", + timestamp: 1, + }; + const context: Context = { + messages: [assistant, toolResult("call_a", "out a"), toolResult("call_b", "out b")], + }; + + const items = buildResponsesInput({ + model, + context, + strictResponsesPairing: false, + supportsImageDetailOriginal: false, + }); + + expect(items.map(wireType)).toEqual([ + "message", + "function_call", + "function_call", + "function_call_output", + "function_call_output", + ]); + }); +});