fix(ai): synthesized reasoning_text on DeepSeek Responses replay
After a prewalk hand-off plus mid-run compaction, the openai-responses input builder re-encoded replayed assistant turns via convertResponsesAssistantMessage, demoting their reasoning to <think> output_text and emitting no reasoning item. DeepSeek (opencode-go) then rejected the thinking-mode continuation with 400 "The reasoning_text in the thinking mode must be passed back to the API" and OMP looped on the unchanged request. The encoder now synthesizes a reasoning_text reasoning item for every replayed assistant turn when the target requires reasoning replay in thinking mode (requiresReasoningContentForAllAssistantTurns / requiresReasoningContentForToolCalls), carrying surviving thinking text when present, mirroring the chat-completions reasoning_content safety net. Gated on reasoning being active for the request, so non-DeepSeek Responses targets and reasoning-disabled turns are unaffected. Fixes #8248
This commit is contained in:
@@ -17,6 +17,9 @@
|
||||
- Allowed passive Google callers to accept empty or thinking-only `STOP` responses as successful silence instead of exhausting the provider's empty-response retry budget. ([#8223](https://github.com/can1357/oh-my-pi/issues/8223))
|
||||
- Fixed the AWS credential resolver ignoring `role_arn` profiles: shared-config role chaining (`source_profile` recursion, `web_identity_token_file`, `credential_source`) now resolves via STS `AssumeRole`/`AssumeRoleWithWebIdentity`, honoring `role_session_name`/`duration_seconds`/`external_id`, so Bedrock is detected on EKS/IRSA and multi-account setups instead of reporting "No models available" ([#8209](https://github.com/can1357/oh-my-pi/issues/8209)).
|
||||
- Fixed Bedrock availability being under-detected on Nitro/EKS hosts: the EC2 metadata probe now recognizes Nitro DMI markers (`board_asset_tag` instance ids, `Amazon EC2` vendor fields) in addition to the Xen `ec2` UUID prefix ([#8209](https://github.com/can1357/oh-my-pi/issues/8209)).
|
||||
### Fixed
|
||||
|
||||
- Fixed DeepSeek Responses targets (opencode-go) rejecting a thinking-mode continuation with `400 The reasoning_text in the thinking mode must be passed back to the API` after a prewalk hand-off plus mid-run compaction: the Responses input builder re-encoded replayed assistant turns without a reasoning item, so the request enabled reasoning but shipped no `reasoning_text`. The encoder now synthesizes a `reasoning_text` reasoning item for every replayed assistant turn when the target requires reasoning replay in thinking mode (`requiresReasoningContentForAllAssistantTurns` / `requiresReasoningContentForToolCalls`), mirroring the chat-completions `reasoning_content` safety net ([#8248](https://github.com/can1357/oh-my-pi/issues/8248)).
|
||||
|
||||
## [17.2.12] - 2026-08-08
|
||||
|
||||
|
||||
@@ -1113,6 +1113,10 @@ export function buildParams(
|
||||
filterReasoning: policy.reasoning.filterReasoningHistory,
|
||||
},
|
||||
includeThinkingSignatures: shouldReplayNativeHistory && !policy.reasoning.filterReasoningHistory,
|
||||
requiresReasoningReplayForAllTurns:
|
||||
policy.reasoning.enabled && policy.reasoning.requiresReasoningContentForAllAssistantTurns,
|
||||
requiresReasoningReplayForToolCalls:
|
||||
policy.reasoning.enabled && policy.reasoning.requiresReasoningContentForToolCalls,
|
||||
repairOrphanOutputs: true,
|
||||
});
|
||||
|
||||
|
||||
@@ -1635,6 +1635,14 @@ export interface BuildResponsesInputOptions<TApi extends Api> {
|
||||
repairOrphanOutputs?: boolean;
|
||||
/** Preserve assistant message item IDs from text signatures during fallback replay. */
|
||||
preserveAssistantMessageIds?: boolean;
|
||||
/**
|
||||
* Synthesize a reasoning item for every replayed assistant turn that carries
|
||||
* content but no reasoning item. Set for DeepSeek-family Responses targets
|
||||
* that reject a thinking-mode continuation lacking `reasoning_text`.
|
||||
*/
|
||||
requiresReasoningReplayForAllTurns?: boolean;
|
||||
/** As {@link requiresReasoningReplayForAllTurns}, but only for turns that contain a tool call. */
|
||||
requiresReasoningReplayForToolCalls?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1861,6 +1869,8 @@ export function buildResponsesInput<TApi extends Api>(options: BuildResponsesInp
|
||||
supportsCustomToolCalls,
|
||||
customToolWireNameMap,
|
||||
computerCallIds,
|
||||
options.requiresReasoningReplayForAllTurns ?? false,
|
||||
options.requiresReasoningReplayForToolCalls ?? false,
|
||||
);
|
||||
const outputItems = suppressHiddenEmptyFallback
|
||||
? sanitizeOpenAIResponsesAssistantFallbackItemsForReplay(convertedOutputItems)
|
||||
@@ -1914,6 +1924,8 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
|
||||
supportsCustomToolCalls = true,
|
||||
customToolWireNameMap?: ReadonlyMap<string, string>,
|
||||
computerCallIds?: Set<string>,
|
||||
requiresReasoningReplayForAllTurns = false,
|
||||
requiresReasoningReplayForToolCalls = false,
|
||||
): ResponseInput {
|
||||
const outputItems: ResponseInput = [];
|
||||
let unsignedTextBlocks = 0;
|
||||
@@ -1925,14 +1937,36 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
|
||||
);
|
||||
const isDifferentModel =
|
||||
assistantMsg.model !== model.id && assistantMsg.provider === model.provider && assistantMsg.api === model.api;
|
||||
// DeepSeek-family Responses targets (e.g. opencode-go) reject a thinking-mode
|
||||
// continuation whose replayed assistant turns carry no reasoning item: "The
|
||||
// reasoning_text in the thinking mode must be passed back to the API." After a
|
||||
// cross-model prewalk hand-off or a compaction that drops the native replay
|
||||
// payload, the block re-encode below demotes reasoning to text and emits no
|
||||
// reasoning item. Track reasoning emission so a placeholder can be synthesized,
|
||||
// mirroring the chat-completions `requiresReasoningContentForAllAssistantTurns`
|
||||
// empty-`reasoning_content` safety net.
|
||||
const requiresReasoningItem =
|
||||
assistantMsg.stopReason !== "error" &&
|
||||
(requiresReasoningReplayForAllTurns ||
|
||||
(requiresReasoningReplayForToolCalls && assistantMsg.content.some(block => block.type === "toolCall")));
|
||||
let reasoningItemEmitted = false;
|
||||
const carriedReasoningTexts: string[] = [];
|
||||
let synthesizedReasoningItemId: string | undefined;
|
||||
|
||||
for (const block of assistantMsg.content) {
|
||||
if (block.type === "thinking" && assistantMsg.stopReason !== "error") {
|
||||
if (requiresReasoningItem) {
|
||||
if (block.itemId) synthesizedReasoningItemId ??= block.itemId;
|
||||
if (block.thinking.trim().length > 0) carriedReasoningTexts.push(block.thinking);
|
||||
}
|
||||
if (!includeThinkingSignatures) {
|
||||
continue;
|
||||
}
|
||||
const reasoningItem = parseResponseReasoningReplayItem(block.thinkingSignature);
|
||||
if (reasoningItem) outputItems.push(reasoningItem);
|
||||
if (reasoningItem) {
|
||||
outputItems.push(reasoningItem);
|
||||
reasoningItemEmitted = true;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -2033,6 +2067,26 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
|
||||
});
|
||||
}
|
||||
|
||||
if (requiresReasoningItem && !reasoningItemEmitted && outputItems.length > 0) {
|
||||
// Replay the demoted reasoning (already present in `content` as visible
|
||||
// text) as a structured reasoning item so the thinking-mode continuation
|
||||
// carries the `reasoning_text` the provider requires. The text may be empty
|
||||
// when the source turn was minted by another model and its reasoning is
|
||||
// already folded into the message text; the item's presence is what
|
||||
// satisfies the provider contract, mirroring the empty `reasoning_content`
|
||||
// placeholder used on the chat-completions path.
|
||||
const reasoningText = carriedReasoningTexts.join("\n");
|
||||
const reasoningId =
|
||||
synthesizedReasoningItemId ?? `rs_${Bun.hash(`${model.id}:${msgIndex}:${reasoningText}`).toString(36)}`;
|
||||
const reasoningItem: ResponseReasoningItem = {
|
||||
type: "reasoning",
|
||||
id: reasoningId,
|
||||
summary: [],
|
||||
content: [{ type: "reasoning_text", text: reasoningText }],
|
||||
};
|
||||
outputItems.unshift(reasoningItem);
|
||||
}
|
||||
|
||||
return outputItems;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,216 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
|
||||
import type { AssistantMessage, Context, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
|
||||
// Issue #8248: with prewalk enabled, OMP switches into a DeepSeek Responses
|
||||
// target (opencode-go) after mid-run compaction. The replayed assistant turns
|
||||
// were minted by the previous model, so the Responses input builder re-encodes
|
||||
// them and demotes their reasoning to plain text, emitting no reasoning item.
|
||||
// DeepSeek then rejects the thinking-mode continuation:
|
||||
// 400 The reasoning_text in the thinking mode must be passed back to the API.
|
||||
// The encoder must synthesize a reasoning item carrying `reasoning_text` for
|
||||
// each replayed assistant turn when the target requires it in thinking mode.
|
||||
|
||||
interface ReasoningTextPart {
|
||||
type: string;
|
||||
text: string;
|
||||
}
|
||||
|
||||
interface ResponsesInputItem {
|
||||
type?: string;
|
||||
role?: string;
|
||||
content?: unknown;
|
||||
}
|
||||
|
||||
interface ResponsesPayload {
|
||||
reasoning?: { effort?: string };
|
||||
input?: ResponsesInputItem[];
|
||||
}
|
||||
|
||||
function abortedSignal(): AbortSignal {
|
||||
const controller = new AbortController();
|
||||
controller.abort();
|
||||
return controller.signal;
|
||||
}
|
||||
|
||||
function capture(
|
||||
model: Model<"openai-responses">,
|
||||
context: Context,
|
||||
overrides: { reasoning?: Effort; disableReasoning?: boolean } = {},
|
||||
): Promise<ResponsesPayload> {
|
||||
const { promise, resolve } = Promise.withResolvers<ResponsesPayload>();
|
||||
streamOpenAIResponses(model, context, {
|
||||
apiKey: "sk-test",
|
||||
reasoning: "reasoning" in overrides ? overrides.reasoning : Effort.XHigh,
|
||||
disableReasoning: overrides.disableReasoning,
|
||||
signal: abortedSignal(),
|
||||
onPayload: payload => resolve(payload as ResponsesPayload),
|
||||
});
|
||||
return promise;
|
||||
}
|
||||
|
||||
function reasoningItems(payload: ResponsesPayload): ResponsesInputItem[] {
|
||||
return (payload.input ?? []).filter(item => item.type === "reasoning");
|
||||
}
|
||||
|
||||
function reasoningTextOf(item: ResponsesInputItem): string {
|
||||
const content = Array.isArray(item.content) ? (item.content as ReasoningTextPart[]) : [];
|
||||
return content
|
||||
.filter(part => part.type === "reasoning_text")
|
||||
.map(part => part.text)
|
||||
.join("");
|
||||
}
|
||||
|
||||
const deepseek = getBundledModel("opencode-go", "deepseek-v4-flash") as Model<"openai-responses">;
|
||||
|
||||
const usage = {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
} as const;
|
||||
|
||||
describe("issue #8248: DeepSeek Responses reasoning replay after prewalk/compaction", () => {
|
||||
it("targets a reasoning Responses model that requires reasoning replay", () => {
|
||||
expect(deepseek.api).toBe("openai-responses");
|
||||
expect(deepseek.compat.requiresReasoningContentForAllAssistantTurns).toBe(true);
|
||||
});
|
||||
|
||||
it("synthesizes a reasoning item for a foreign assistant turn replayed after a prewalk switch", async () => {
|
||||
// Kept-tail turn minted by the previous model (prewalk hopped gpt-5.6-sol
|
||||
// -> deepseek). Same api, different provider+model -> block re-encode.
|
||||
const prior: AssistantMessage = {
|
||||
role: "assistant",
|
||||
api: "openai-responses",
|
||||
provider: "github-copilot",
|
||||
model: "gpt-5.6-sol",
|
||||
stopReason: "stop",
|
||||
usage,
|
||||
content: [
|
||||
{
|
||||
type: "thinking",
|
||||
thinking: "Refactor plan for foo.",
|
||||
thinkingSignature: JSON.stringify({ type: "reasoning", id: "rs_prev" }),
|
||||
},
|
||||
{ type: "text", text: "Refactored bar.ts." },
|
||||
],
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const context: Context = {
|
||||
messages: [
|
||||
{ role: "user", content: "Refactor foo", timestamp: Date.now() },
|
||||
prior,
|
||||
{ role: "user", content: "Now update the tests", timestamp: Date.now() },
|
||||
],
|
||||
};
|
||||
|
||||
const payload = await capture(deepseek, context);
|
||||
expect(payload.reasoning?.effort).toBeDefined();
|
||||
|
||||
const input = payload.input ?? [];
|
||||
const reasoning = reasoningItems(payload);
|
||||
expect(reasoning).toHaveLength(1);
|
||||
// The reasoning item must precede the assistant message it belongs to.
|
||||
const reasoningIdx = input.findIndex(item => item.type === "reasoning");
|
||||
const assistantIdx = input.findIndex(item => item.type === "message" && item.role === "assistant");
|
||||
expect(reasoningIdx).toBeGreaterThanOrEqual(0);
|
||||
expect(reasoningIdx).toBeLessThan(assistantIdx);
|
||||
// It carries a reasoning_text content part (the field DeepSeek requires).
|
||||
const content = Array.isArray(reasoning[0]!.content) ? (reasoning[0]!.content as ReasoningTextPart[]) : [];
|
||||
expect(content.some(part => part.type === "reasoning_text")).toBe(true);
|
||||
});
|
||||
|
||||
it("carries the actual reasoning text when a same-model thinking block survives replay", async () => {
|
||||
// Same provider/model (deepseek) but no native providerPayload (dropped by
|
||||
// compaction). The thinking block survives transform with no native
|
||||
// Responses signature, so its text must ride in the synthesized item.
|
||||
const prior: AssistantMessage = {
|
||||
role: "assistant",
|
||||
api: "openai-responses",
|
||||
provider: "opencode-go",
|
||||
model: "deepseek-v4-flash",
|
||||
stopReason: "stop",
|
||||
usage,
|
||||
content: [
|
||||
{ type: "thinking", thinking: "Inspect bar.ts before editing." },
|
||||
{ type: "text", text: "Edited bar.ts." },
|
||||
],
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const context: Context = {
|
||||
messages: [
|
||||
{ role: "user", content: "Edit bar", timestamp: Date.now() },
|
||||
prior,
|
||||
{ role: "user", content: "Run the tests", timestamp: Date.now() },
|
||||
],
|
||||
};
|
||||
|
||||
const payload = await capture(deepseek, context);
|
||||
const reasoning = reasoningItems(payload);
|
||||
expect(reasoning).toHaveLength(1);
|
||||
expect(reasoningTextOf(reasoning[0]!)).toBe("Inspect bar.ts before editing.");
|
||||
});
|
||||
|
||||
it("does not synthesize a reasoning item when reasoning is disabled for the turn", async () => {
|
||||
const prior: AssistantMessage = {
|
||||
role: "assistant",
|
||||
api: "openai-responses",
|
||||
provider: "github-copilot",
|
||||
model: "gpt-5.6-sol",
|
||||
stopReason: "stop",
|
||||
usage,
|
||||
content: [
|
||||
{
|
||||
type: "thinking",
|
||||
thinking: "Refactor plan for foo.",
|
||||
thinkingSignature: JSON.stringify({ type: "reasoning", id: "rs_prev" }),
|
||||
},
|
||||
{ type: "text", text: "Refactored bar.ts." },
|
||||
],
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const context: Context = {
|
||||
messages: [
|
||||
{ role: "user", content: "Refactor foo", timestamp: Date.now() },
|
||||
prior,
|
||||
{ role: "user", content: "Now update the tests", timestamp: Date.now() },
|
||||
],
|
||||
};
|
||||
|
||||
const payload = await capture(deepseek, context, { reasoning: undefined, disableReasoning: true });
|
||||
expect(reasoningItems(payload)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it("does not synthesize reasoning items for non-DeepSeek Responses targets", async () => {
|
||||
const openai = getBundledModel("openai", "gpt-5-mini") as Model<"openai-responses">;
|
||||
expect(openai.compat.requiresReasoningContentForAllAssistantTurns).toBe(false);
|
||||
|
||||
const prior: AssistantMessage = {
|
||||
role: "assistant",
|
||||
api: "openai-responses",
|
||||
provider: "anthropic",
|
||||
model: "claude-sonnet-4-5",
|
||||
stopReason: "stop",
|
||||
usage,
|
||||
content: [
|
||||
{ type: "thinking", thinking: "Cross-provider reasoning." },
|
||||
{ type: "text", text: "Answer." },
|
||||
],
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const context: Context = {
|
||||
messages: [
|
||||
{ role: "user", content: "Question", timestamp: Date.now() },
|
||||
prior,
|
||||
{ role: "user", content: "Follow up", timestamp: Date.now() },
|
||||
],
|
||||
};
|
||||
|
||||
const payload = await capture(openai, context);
|
||||
expect(reasoningItems(payload)).toHaveLength(0);
|
||||
});
|
||||
});
|
||||
@@ -2613,8 +2613,24 @@ describe("advisor", () => {
|
||||
const state: { messages: AgentMessage[]; error?: string } = { messages: [] };
|
||||
let promptCalls = 0;
|
||||
const agent: AdvisorAgent = {
|
||||
prompt: async () => {
|
||||
prompt: async input => {
|
||||
promptCalls++;
|
||||
const content =
|
||||
typeof input === "string"
|
||||
? input
|
||||
: input
|
||||
.map(message => {
|
||||
if (!("content" in message)) return "";
|
||||
if (typeof message.content === "string") return message.content;
|
||||
const textParts: string[] = [];
|
||||
for (const block of message.content) {
|
||||
if (block.type === "text") textParts.push(block.text);
|
||||
}
|
||||
return textParts.join("");
|
||||
})
|
||||
.filter(Boolean)
|
||||
.join("\n\n");
|
||||
state.messages.push({ role: "user", content, timestamp: Date.now() } as AgentMessage);
|
||||
if (promptCalls === 2) {
|
||||
firstOverflowPromptStarted.resolve();
|
||||
await releaseOverflowPrompt.promise;
|
||||
|
||||
Reference in New Issue
Block a user