diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index fb069b25c..6cd7f72a1 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -17,6 +17,9 @@ - Allowed passive Google callers to accept empty or thinking-only `STOP` responses as successful silence instead of exhausting the provider's empty-response retry budget. ([#8223](https://github.com/can1357/oh-my-pi/issues/8223)) - Fixed the AWS credential resolver ignoring `role_arn` profiles: shared-config role chaining (`source_profile` recursion, `web_identity_token_file`, `credential_source`) now resolves via STS `AssumeRole`/`AssumeRoleWithWebIdentity`, honoring `role_session_name`/`duration_seconds`/`external_id`, so Bedrock is detected on EKS/IRSA and multi-account setups instead of reporting "No models available" ([#8209](https://github.com/can1357/oh-my-pi/issues/8209)). - Fixed Bedrock availability being under-detected on Nitro/EKS hosts: the EC2 metadata probe now recognizes Nitro DMI markers (`board_asset_tag` instance ids, `Amazon EC2` vendor fields) in addition to the Xen `ec2` UUID prefix ([#8209](https://github.com/can1357/oh-my-pi/issues/8209)). +### Fixed + +- Fixed DeepSeek Responses targets (opencode-go) rejecting a thinking-mode continuation with `400 The reasoning_text in the thinking mode must be passed back to the API` after a prewalk hand-off plus mid-run compaction: the Responses input builder re-encoded replayed assistant turns without a reasoning item, so the request enabled reasoning but shipped no `reasoning_text`. The encoder now synthesizes a `reasoning_text` reasoning item for every replayed assistant turn when the target requires reasoning replay in thinking mode (`requiresReasoningContentForAllAssistantTurns` / `requiresReasoningContentForToolCalls`), mirroring the chat-completions `reasoning_content` safety net ([#8248](https://github.com/can1357/oh-my-pi/issues/8248)). ## [17.2.12] - 2026-08-08 diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index ed659a49a..518ec0d95 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -1113,6 +1113,10 @@ export function buildParams( filterReasoning: policy.reasoning.filterReasoningHistory, }, includeThinkingSignatures: shouldReplayNativeHistory && !policy.reasoning.filterReasoningHistory, + requiresReasoningReplayForAllTurns: + policy.reasoning.enabled && policy.reasoning.requiresReasoningContentForAllAssistantTurns, + requiresReasoningReplayForToolCalls: + policy.reasoning.enabled && policy.reasoning.requiresReasoningContentForToolCalls, repairOrphanOutputs: true, }); diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index b964e5643..e49a1ed0e 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -1635,6 +1635,14 @@ export interface BuildResponsesInputOptions { repairOrphanOutputs?: boolean; /** Preserve assistant message item IDs from text signatures during fallback replay. */ preserveAssistantMessageIds?: boolean; + /** + * Synthesize a reasoning item for every replayed assistant turn that carries + * content but no reasoning item. Set for DeepSeek-family Responses targets + * that reject a thinking-mode continuation lacking `reasoning_text`. + */ + requiresReasoningReplayForAllTurns?: boolean; + /** As {@link requiresReasoningReplayForAllTurns}, but only for turns that contain a tool call. */ + requiresReasoningReplayForToolCalls?: boolean; } /** @@ -1861,6 +1869,8 @@ export function buildResponsesInput(options: BuildResponsesInp supportsCustomToolCalls, customToolWireNameMap, computerCallIds, + options.requiresReasoningReplayForAllTurns ?? false, + options.requiresReasoningReplayForToolCalls ?? false, ); const outputItems = suppressHiddenEmptyFallback ? sanitizeOpenAIResponsesAssistantFallbackItemsForReplay(convertedOutputItems) @@ -1914,6 +1924,8 @@ export function convertResponsesAssistantMessage( supportsCustomToolCalls = true, customToolWireNameMap?: ReadonlyMap, computerCallIds?: Set, + requiresReasoningReplayForAllTurns = false, + requiresReasoningReplayForToolCalls = false, ): ResponseInput { const outputItems: ResponseInput = []; let unsignedTextBlocks = 0; @@ -1925,14 +1937,36 @@ export function convertResponsesAssistantMessage( ); const isDifferentModel = assistantMsg.model !== model.id && assistantMsg.provider === model.provider && assistantMsg.api === model.api; + // DeepSeek-family Responses targets (e.g. opencode-go) reject a thinking-mode + // continuation whose replayed assistant turns carry no reasoning item: "The + // reasoning_text in the thinking mode must be passed back to the API." After a + // cross-model prewalk hand-off or a compaction that drops the native replay + // payload, the block re-encode below demotes reasoning to text and emits no + // reasoning item. Track reasoning emission so a placeholder can be synthesized, + // mirroring the chat-completions `requiresReasoningContentForAllAssistantTurns` + // empty-`reasoning_content` safety net. + const requiresReasoningItem = + assistantMsg.stopReason !== "error" && + (requiresReasoningReplayForAllTurns || + (requiresReasoningReplayForToolCalls && assistantMsg.content.some(block => block.type === "toolCall"))); + let reasoningItemEmitted = false; + const carriedReasoningTexts: string[] = []; + let synthesizedReasoningItemId: string | undefined; for (const block of assistantMsg.content) { if (block.type === "thinking" && assistantMsg.stopReason !== "error") { + if (requiresReasoningItem) { + if (block.itemId) synthesizedReasoningItemId ??= block.itemId; + if (block.thinking.trim().length > 0) carriedReasoningTexts.push(block.thinking); + } if (!includeThinkingSignatures) { continue; } const reasoningItem = parseResponseReasoningReplayItem(block.thinkingSignature); - if (reasoningItem) outputItems.push(reasoningItem); + if (reasoningItem) { + outputItems.push(reasoningItem); + reasoningItemEmitted = true; + } continue; } @@ -2033,6 +2067,26 @@ export function convertResponsesAssistantMessage( }); } + if (requiresReasoningItem && !reasoningItemEmitted && outputItems.length > 0) { + // Replay the demoted reasoning (already present in `content` as visible + // text) as a structured reasoning item so the thinking-mode continuation + // carries the `reasoning_text` the provider requires. The text may be empty + // when the source turn was minted by another model and its reasoning is + // already folded into the message text; the item's presence is what + // satisfies the provider contract, mirroring the empty `reasoning_content` + // placeholder used on the chat-completions path. + const reasoningText = carriedReasoningTexts.join("\n"); + const reasoningId = + synthesizedReasoningItemId ?? `rs_${Bun.hash(`${model.id}:${msgIndex}:${reasoningText}`).toString(36)}`; + const reasoningItem: ResponseReasoningItem = { + type: "reasoning", + id: reasoningId, + summary: [], + content: [{ type: "reasoning_text", text: reasoningText }], + }; + outputItems.unshift(reasoningItem); + } + return outputItems; } diff --git a/packages/ai/test/issue-8248-repro.test.ts b/packages/ai/test/issue-8248-repro.test.ts new file mode 100644 index 000000000..052b30f41 --- /dev/null +++ b/packages/ai/test/issue-8248-repro.test.ts @@ -0,0 +1,216 @@ +import { describe, expect, it } from "bun:test"; +import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses"; +import type { AssistantMessage, Context, Model } from "@oh-my-pi/pi-ai/types"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; + +// Issue #8248: with prewalk enabled, OMP switches into a DeepSeek Responses +// target (opencode-go) after mid-run compaction. The replayed assistant turns +// were minted by the previous model, so the Responses input builder re-encodes +// them and demotes their reasoning to plain text, emitting no reasoning item. +// DeepSeek then rejects the thinking-mode continuation: +// 400 The reasoning_text in the thinking mode must be passed back to the API. +// The encoder must synthesize a reasoning item carrying `reasoning_text` for +// each replayed assistant turn when the target requires it in thinking mode. + +interface ReasoningTextPart { + type: string; + text: string; +} + +interface ResponsesInputItem { + type?: string; + role?: string; + content?: unknown; +} + +interface ResponsesPayload { + reasoning?: { effort?: string }; + input?: ResponsesInputItem[]; +} + +function abortedSignal(): AbortSignal { + const controller = new AbortController(); + controller.abort(); + return controller.signal; +} + +function capture( + model: Model<"openai-responses">, + context: Context, + overrides: { reasoning?: Effort; disableReasoning?: boolean } = {}, +): Promise { + const { promise, resolve } = Promise.withResolvers(); + streamOpenAIResponses(model, context, { + apiKey: "sk-test", + reasoning: "reasoning" in overrides ? overrides.reasoning : Effort.XHigh, + disableReasoning: overrides.disableReasoning, + signal: abortedSignal(), + onPayload: payload => resolve(payload as ResponsesPayload), + }); + return promise; +} + +function reasoningItems(payload: ResponsesPayload): ResponsesInputItem[] { + return (payload.input ?? []).filter(item => item.type === "reasoning"); +} + +function reasoningTextOf(item: ResponsesInputItem): string { + const content = Array.isArray(item.content) ? (item.content as ReasoningTextPart[]) : []; + return content + .filter(part => part.type === "reasoning_text") + .map(part => part.text) + .join(""); +} + +const deepseek = getBundledModel("opencode-go", "deepseek-v4-flash") as Model<"openai-responses">; + +const usage = { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, +} as const; + +describe("issue #8248: DeepSeek Responses reasoning replay after prewalk/compaction", () => { + it("targets a reasoning Responses model that requires reasoning replay", () => { + expect(deepseek.api).toBe("openai-responses"); + expect(deepseek.compat.requiresReasoningContentForAllAssistantTurns).toBe(true); + }); + + it("synthesizes a reasoning item for a foreign assistant turn replayed after a prewalk switch", async () => { + // Kept-tail turn minted by the previous model (prewalk hopped gpt-5.6-sol + // -> deepseek). Same api, different provider+model -> block re-encode. + const prior: AssistantMessage = { + role: "assistant", + api: "openai-responses", + provider: "github-copilot", + model: "gpt-5.6-sol", + stopReason: "stop", + usage, + content: [ + { + type: "thinking", + thinking: "Refactor plan for foo.", + thinkingSignature: JSON.stringify({ type: "reasoning", id: "rs_prev" }), + }, + { type: "text", text: "Refactored bar.ts." }, + ], + timestamp: Date.now(), + }; + const context: Context = { + messages: [ + { role: "user", content: "Refactor foo", timestamp: Date.now() }, + prior, + { role: "user", content: "Now update the tests", timestamp: Date.now() }, + ], + }; + + const payload = await capture(deepseek, context); + expect(payload.reasoning?.effort).toBeDefined(); + + const input = payload.input ?? []; + const reasoning = reasoningItems(payload); + expect(reasoning).toHaveLength(1); + // The reasoning item must precede the assistant message it belongs to. + const reasoningIdx = input.findIndex(item => item.type === "reasoning"); + const assistantIdx = input.findIndex(item => item.type === "message" && item.role === "assistant"); + expect(reasoningIdx).toBeGreaterThanOrEqual(0); + expect(reasoningIdx).toBeLessThan(assistantIdx); + // It carries a reasoning_text content part (the field DeepSeek requires). + const content = Array.isArray(reasoning[0]!.content) ? (reasoning[0]!.content as ReasoningTextPart[]) : []; + expect(content.some(part => part.type === "reasoning_text")).toBe(true); + }); + + it("carries the actual reasoning text when a same-model thinking block survives replay", async () => { + // Same provider/model (deepseek) but no native providerPayload (dropped by + // compaction). The thinking block survives transform with no native + // Responses signature, so its text must ride in the synthesized item. + const prior: AssistantMessage = { + role: "assistant", + api: "openai-responses", + provider: "opencode-go", + model: "deepseek-v4-flash", + stopReason: "stop", + usage, + content: [ + { type: "thinking", thinking: "Inspect bar.ts before editing." }, + { type: "text", text: "Edited bar.ts." }, + ], + timestamp: Date.now(), + }; + const context: Context = { + messages: [ + { role: "user", content: "Edit bar", timestamp: Date.now() }, + prior, + { role: "user", content: "Run the tests", timestamp: Date.now() }, + ], + }; + + const payload = await capture(deepseek, context); + const reasoning = reasoningItems(payload); + expect(reasoning).toHaveLength(1); + expect(reasoningTextOf(reasoning[0]!)).toBe("Inspect bar.ts before editing."); + }); + + it("does not synthesize a reasoning item when reasoning is disabled for the turn", async () => { + const prior: AssistantMessage = { + role: "assistant", + api: "openai-responses", + provider: "github-copilot", + model: "gpt-5.6-sol", + stopReason: "stop", + usage, + content: [ + { + type: "thinking", + thinking: "Refactor plan for foo.", + thinkingSignature: JSON.stringify({ type: "reasoning", id: "rs_prev" }), + }, + { type: "text", text: "Refactored bar.ts." }, + ], + timestamp: Date.now(), + }; + const context: Context = { + messages: [ + { role: "user", content: "Refactor foo", timestamp: Date.now() }, + prior, + { role: "user", content: "Now update the tests", timestamp: Date.now() }, + ], + }; + + const payload = await capture(deepseek, context, { reasoning: undefined, disableReasoning: true }); + expect(reasoningItems(payload)).toHaveLength(0); + }); + + it("does not synthesize reasoning items for non-DeepSeek Responses targets", async () => { + const openai = getBundledModel("openai", "gpt-5-mini") as Model<"openai-responses">; + expect(openai.compat.requiresReasoningContentForAllAssistantTurns).toBe(false); + + const prior: AssistantMessage = { + role: "assistant", + api: "openai-responses", + provider: "anthropic", + model: "claude-sonnet-4-5", + stopReason: "stop", + usage, + content: [ + { type: "thinking", thinking: "Cross-provider reasoning." }, + { type: "text", text: "Answer." }, + ], + timestamp: Date.now(), + }; + const context: Context = { + messages: [ + { role: "user", content: "Question", timestamp: Date.now() }, + prior, + { role: "user", content: "Follow up", timestamp: Date.now() }, + ], + }; + + const payload = await capture(openai, context); + expect(reasoningItems(payload)).toHaveLength(0); + }); +}); diff --git a/packages/coding-agent/test/advisor/advisor.test.ts b/packages/coding-agent/test/advisor/advisor.test.ts index 44063d3f4..b4f1a6f3b 100644 --- a/packages/coding-agent/test/advisor/advisor.test.ts +++ b/packages/coding-agent/test/advisor/advisor.test.ts @@ -2613,8 +2613,24 @@ describe("advisor", () => { const state: { messages: AgentMessage[]; error?: string } = { messages: [] }; let promptCalls = 0; const agent: AdvisorAgent = { - prompt: async () => { + prompt: async input => { promptCalls++; + const content = + typeof input === "string" + ? input + : input + .map(message => { + if (!("content" in message)) return ""; + if (typeof message.content === "string") return message.content; + const textParts: string[] = []; + for (const block of message.content) { + if (block.type === "text") textParts.push(block.text); + } + return textParts.join(""); + }) + .filter(Boolean) + .join("\n\n"); + state.messages.push({ role: "user", content, timestamp: Date.now() } as AgentMessage); if (promptCalls === 2) { firstOverflowPromptStarted.resolve(); await releaseOverflowPrompt.promise;