fix(ai): synthesized reasoning_text on DeepSeek Responses replay

After a prewalk hand-off plus mid-run compaction, the openai-responses input
builder re-encoded replayed assistant turns via convertResponsesAssistantMessage,
demoting their reasoning to <think> output_text and emitting no reasoning item.
DeepSeek (opencode-go) then rejected the thinking-mode continuation with
400 "The reasoning_text in the thinking mode must be passed back to the API"
and OMP looped on the unchanged request.

The encoder now synthesizes a reasoning_text reasoning item for every replayed
assistant turn when the target requires reasoning replay in thinking mode
(requiresReasoningContentForAllAssistantTurns / requiresReasoningContentForToolCalls),
carrying surviving thinking text when present, mirroring the chat-completions
reasoning_content safety net. Gated on reasoning being active for the request,
so non-DeepSeek Responses targets and reasoning-disabled turns are unaffected.

Fixes #8248
This commit is contained in:
roboomp
2026-08-11 13:27:21 +00:00
committed by can1357
parent b524dfe36f
commit 54ce9e47fc
5 changed files with 295 additions and 2 deletions
+3
View File
@@ -17,6 +17,9 @@
- Allowed passive Google callers to accept empty or thinking-only `STOP` responses as successful silence instead of exhausting the provider's empty-response retry budget. ([#8223](https://github.com/can1357/oh-my-pi/issues/8223))
- Fixed the AWS credential resolver ignoring `role_arn` profiles: shared-config role chaining (`source_profile` recursion, `web_identity_token_file`, `credential_source`) now resolves via STS `AssumeRole`/`AssumeRoleWithWebIdentity`, honoring `role_session_name`/`duration_seconds`/`external_id`, so Bedrock is detected on EKS/IRSA and multi-account setups instead of reporting "No models available" ([#8209](https://github.com/can1357/oh-my-pi/issues/8209)).
- Fixed Bedrock availability being under-detected on Nitro/EKS hosts: the EC2 metadata probe now recognizes Nitro DMI markers (`board_asset_tag` instance ids, `Amazon EC2` vendor fields) in addition to the Xen `ec2` UUID prefix ([#8209](https://github.com/can1357/oh-my-pi/issues/8209)).
### Fixed
- Fixed DeepSeek Responses targets (opencode-go) rejecting a thinking-mode continuation with `400 The reasoning_text in the thinking mode must be passed back to the API` after a prewalk hand-off plus mid-run compaction: the Responses input builder re-encoded replayed assistant turns without a reasoning item, so the request enabled reasoning but shipped no `reasoning_text`. The encoder now synthesizes a `reasoning_text` reasoning item for every replayed assistant turn when the target requires reasoning replay in thinking mode (`requiresReasoningContentForAllAssistantTurns` / `requiresReasoningContentForToolCalls`), mirroring the chat-completions `reasoning_content` safety net ([#8248](https://github.com/can1357/oh-my-pi/issues/8248)).
## [17.2.12] - 2026-08-08
@@ -1113,6 +1113,10 @@ export function buildParams(
filterReasoning: policy.reasoning.filterReasoningHistory,
},
includeThinkingSignatures: shouldReplayNativeHistory && !policy.reasoning.filterReasoningHistory,
requiresReasoningReplayForAllTurns:
policy.reasoning.enabled && policy.reasoning.requiresReasoningContentForAllAssistantTurns,
requiresReasoningReplayForToolCalls:
policy.reasoning.enabled && policy.reasoning.requiresReasoningContentForToolCalls,
repairOrphanOutputs: true,
});
+55 -1
View File
@@ -1635,6 +1635,14 @@ export interface BuildResponsesInputOptions<TApi extends Api> {
repairOrphanOutputs?: boolean;
/** Preserve assistant message item IDs from text signatures during fallback replay. */
preserveAssistantMessageIds?: boolean;
/**
* Synthesize a reasoning item for every replayed assistant turn that carries
* content but no reasoning item. Set for DeepSeek-family Responses targets
* that reject a thinking-mode continuation lacking `reasoning_text`.
*/
requiresReasoningReplayForAllTurns?: boolean;
/** As {@link requiresReasoningReplayForAllTurns}, but only for turns that contain a tool call. */
requiresReasoningReplayForToolCalls?: boolean;
}
/**
@@ -1861,6 +1869,8 @@ export function buildResponsesInput<TApi extends Api>(options: BuildResponsesInp
supportsCustomToolCalls,
customToolWireNameMap,
computerCallIds,
options.requiresReasoningReplayForAllTurns ?? false,
options.requiresReasoningReplayForToolCalls ?? false,
);
const outputItems = suppressHiddenEmptyFallback
? sanitizeOpenAIResponsesAssistantFallbackItemsForReplay(convertedOutputItems)
@@ -1914,6 +1924,8 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
supportsCustomToolCalls = true,
customToolWireNameMap?: ReadonlyMap<string, string>,
computerCallIds?: Set<string>,
requiresReasoningReplayForAllTurns = false,
requiresReasoningReplayForToolCalls = false,
): ResponseInput {
const outputItems: ResponseInput = [];
let unsignedTextBlocks = 0;
@@ -1925,14 +1937,36 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
);
const isDifferentModel =
assistantMsg.model !== model.id && assistantMsg.provider === model.provider && assistantMsg.api === model.api;
// DeepSeek-family Responses targets (e.g. opencode-go) reject a thinking-mode
// continuation whose replayed assistant turns carry no reasoning item: "The
// reasoning_text in the thinking mode must be passed back to the API." After a
// cross-model prewalk hand-off or a compaction that drops the native replay
// payload, the block re-encode below demotes reasoning to text and emits no
// reasoning item. Track reasoning emission so a placeholder can be synthesized,
// mirroring the chat-completions `requiresReasoningContentForAllAssistantTurns`
// empty-`reasoning_content` safety net.
const requiresReasoningItem =
assistantMsg.stopReason !== "error" &&
(requiresReasoningReplayForAllTurns ||
(requiresReasoningReplayForToolCalls && assistantMsg.content.some(block => block.type === "toolCall")));
let reasoningItemEmitted = false;
const carriedReasoningTexts: string[] = [];
let synthesizedReasoningItemId: string | undefined;
for (const block of assistantMsg.content) {
if (block.type === "thinking" && assistantMsg.stopReason !== "error") {
if (requiresReasoningItem) {
if (block.itemId) synthesizedReasoningItemId ??= block.itemId;
if (block.thinking.trim().length > 0) carriedReasoningTexts.push(block.thinking);
}
if (!includeThinkingSignatures) {
continue;
}
const reasoningItem = parseResponseReasoningReplayItem(block.thinkingSignature);
if (reasoningItem) outputItems.push(reasoningItem);
if (reasoningItem) {
outputItems.push(reasoningItem);
reasoningItemEmitted = true;
}
continue;
}
@@ -2033,6 +2067,26 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
});
}
if (requiresReasoningItem && !reasoningItemEmitted && outputItems.length > 0) {
// Replay the demoted reasoning (already present in `content` as visible
// text) as a structured reasoning item so the thinking-mode continuation
// carries the `reasoning_text` the provider requires. The text may be empty
// when the source turn was minted by another model and its reasoning is
// already folded into the message text; the item's presence is what
// satisfies the provider contract, mirroring the empty `reasoning_content`
// placeholder used on the chat-completions path.
const reasoningText = carriedReasoningTexts.join("\n");
const reasoningId =
synthesizedReasoningItemId ?? `rs_${Bun.hash(`${model.id}:${msgIndex}:${reasoningText}`).toString(36)}`;
const reasoningItem: ResponseReasoningItem = {
type: "reasoning",
id: reasoningId,
summary: [],
content: [{ type: "reasoning_text", text: reasoningText }],
};
outputItems.unshift(reasoningItem);
}
return outputItems;
}
+216
View File
@@ -0,0 +1,216 @@
import { describe, expect, it } from "bun:test";
import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
import type { AssistantMessage, Context, Model } from "@oh-my-pi/pi-ai/types";
import { Effort } from "@oh-my-pi/pi-catalog/effort";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
// Issue #8248: with prewalk enabled, OMP switches into a DeepSeek Responses
// target (opencode-go) after mid-run compaction. The replayed assistant turns
// were minted by the previous model, so the Responses input builder re-encodes
// them and demotes their reasoning to plain text, emitting no reasoning item.
// DeepSeek then rejects the thinking-mode continuation:
// 400 The reasoning_text in the thinking mode must be passed back to the API.
// The encoder must synthesize a reasoning item carrying `reasoning_text` for
// each replayed assistant turn when the target requires it in thinking mode.
interface ReasoningTextPart {
type: string;
text: string;
}
interface ResponsesInputItem {
type?: string;
role?: string;
content?: unknown;
}
interface ResponsesPayload {
reasoning?: { effort?: string };
input?: ResponsesInputItem[];
}
function abortedSignal(): AbortSignal {
const controller = new AbortController();
controller.abort();
return controller.signal;
}
function capture(
model: Model<"openai-responses">,
context: Context,
overrides: { reasoning?: Effort; disableReasoning?: boolean } = {},
): Promise<ResponsesPayload> {
const { promise, resolve } = Promise.withResolvers<ResponsesPayload>();
streamOpenAIResponses(model, context, {
apiKey: "sk-test",
reasoning: "reasoning" in overrides ? overrides.reasoning : Effort.XHigh,
disableReasoning: overrides.disableReasoning,
signal: abortedSignal(),
onPayload: payload => resolve(payload as ResponsesPayload),
});
return promise;
}
function reasoningItems(payload: ResponsesPayload): ResponsesInputItem[] {
return (payload.input ?? []).filter(item => item.type === "reasoning");
}
function reasoningTextOf(item: ResponsesInputItem): string {
const content = Array.isArray(item.content) ? (item.content as ReasoningTextPart[]) : [];
return content
.filter(part => part.type === "reasoning_text")
.map(part => part.text)
.join("");
}
const deepseek = getBundledModel("opencode-go", "deepseek-v4-flash") as Model<"openai-responses">;
const usage = {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
} as const;
describe("issue #8248: DeepSeek Responses reasoning replay after prewalk/compaction", () => {
it("targets a reasoning Responses model that requires reasoning replay", () => {
expect(deepseek.api).toBe("openai-responses");
expect(deepseek.compat.requiresReasoningContentForAllAssistantTurns).toBe(true);
});
it("synthesizes a reasoning item for a foreign assistant turn replayed after a prewalk switch", async () => {
// Kept-tail turn minted by the previous model (prewalk hopped gpt-5.6-sol
// -> deepseek). Same api, different provider+model -> block re-encode.
const prior: AssistantMessage = {
role: "assistant",
api: "openai-responses",
provider: "github-copilot",
model: "gpt-5.6-sol",
stopReason: "stop",
usage,
content: [
{
type: "thinking",
thinking: "Refactor plan for foo.",
thinkingSignature: JSON.stringify({ type: "reasoning", id: "rs_prev" }),
},
{ type: "text", text: "Refactored bar.ts." },
],
timestamp: Date.now(),
};
const context: Context = {
messages: [
{ role: "user", content: "Refactor foo", timestamp: Date.now() },
prior,
{ role: "user", content: "Now update the tests", timestamp: Date.now() },
],
};
const payload = await capture(deepseek, context);
expect(payload.reasoning?.effort).toBeDefined();
const input = payload.input ?? [];
const reasoning = reasoningItems(payload);
expect(reasoning).toHaveLength(1);
// The reasoning item must precede the assistant message it belongs to.
const reasoningIdx = input.findIndex(item => item.type === "reasoning");
const assistantIdx = input.findIndex(item => item.type === "message" && item.role === "assistant");
expect(reasoningIdx).toBeGreaterThanOrEqual(0);
expect(reasoningIdx).toBeLessThan(assistantIdx);
// It carries a reasoning_text content part (the field DeepSeek requires).
const content = Array.isArray(reasoning[0]!.content) ? (reasoning[0]!.content as ReasoningTextPart[]) : [];
expect(content.some(part => part.type === "reasoning_text")).toBe(true);
});
it("carries the actual reasoning text when a same-model thinking block survives replay", async () => {
// Same provider/model (deepseek) but no native providerPayload (dropped by
// compaction). The thinking block survives transform with no native
// Responses signature, so its text must ride in the synthesized item.
const prior: AssistantMessage = {
role: "assistant",
api: "openai-responses",
provider: "opencode-go",
model: "deepseek-v4-flash",
stopReason: "stop",
usage,
content: [
{ type: "thinking", thinking: "Inspect bar.ts before editing." },
{ type: "text", text: "Edited bar.ts." },
],
timestamp: Date.now(),
};
const context: Context = {
messages: [
{ role: "user", content: "Edit bar", timestamp: Date.now() },
prior,
{ role: "user", content: "Run the tests", timestamp: Date.now() },
],
};
const payload = await capture(deepseek, context);
const reasoning = reasoningItems(payload);
expect(reasoning).toHaveLength(1);
expect(reasoningTextOf(reasoning[0]!)).toBe("Inspect bar.ts before editing.");
});
it("does not synthesize a reasoning item when reasoning is disabled for the turn", async () => {
const prior: AssistantMessage = {
role: "assistant",
api: "openai-responses",
provider: "github-copilot",
model: "gpt-5.6-sol",
stopReason: "stop",
usage,
content: [
{
type: "thinking",
thinking: "Refactor plan for foo.",
thinkingSignature: JSON.stringify({ type: "reasoning", id: "rs_prev" }),
},
{ type: "text", text: "Refactored bar.ts." },
],
timestamp: Date.now(),
};
const context: Context = {
messages: [
{ role: "user", content: "Refactor foo", timestamp: Date.now() },
prior,
{ role: "user", content: "Now update the tests", timestamp: Date.now() },
],
};
const payload = await capture(deepseek, context, { reasoning: undefined, disableReasoning: true });
expect(reasoningItems(payload)).toHaveLength(0);
});
it("does not synthesize reasoning items for non-DeepSeek Responses targets", async () => {
const openai = getBundledModel("openai", "gpt-5-mini") as Model<"openai-responses">;
expect(openai.compat.requiresReasoningContentForAllAssistantTurns).toBe(false);
const prior: AssistantMessage = {
role: "assistant",
api: "openai-responses",
provider: "anthropic",
model: "claude-sonnet-4-5",
stopReason: "stop",
usage,
content: [
{ type: "thinking", thinking: "Cross-provider reasoning." },
{ type: "text", text: "Answer." },
],
timestamp: Date.now(),
};
const context: Context = {
messages: [
{ role: "user", content: "Question", timestamp: Date.now() },
prior,
{ role: "user", content: "Follow up", timestamp: Date.now() },
],
};
const payload = await capture(openai, context);
expect(reasoningItems(payload)).toHaveLength(0);
});
});
@@ -2613,8 +2613,24 @@ describe("advisor", () => {
const state: { messages: AgentMessage[]; error?: string } = { messages: [] };
let promptCalls = 0;
const agent: AdvisorAgent = {
prompt: async () => {
prompt: async input => {
promptCalls++;
const content =
typeof input === "string"
? input
: input
.map(message => {
if (!("content" in message)) return "";
if (typeof message.content === "string") return message.content;
const textParts: string[] = [];
for (const block of message.content) {
if (block.type === "text") textParts.push(block.text);
}
return textParts.join("");
})
.filter(Boolean)
.join("\n\n");
state.messages.push({ role: "user", content, timestamp: Date.now() } as AgentMessage);
if (promptCalls === 2) {
firstOverflowPromptStarted.resolve();
await releaseOverflowPrompt.promise;