feat(ai): standardized thinking block demotion across model providers

- Centralized thinking block demotion logic using `renderDemotedThinking` across all model providers.
- Updated `transformMessages` to default to text-based demotion for foreign thinking content.
- Restricted foreign thinking preservation to explicitly supported targets with `zai` thinking formats.
- Refactored message transformations and updated test suites to validate standardized demotion outcomes.
This commit is contained in:
can1357
2026-06-27 08:19:11 +02:00
parent 98b7db0c84
commit f8badc10ff
14 changed files with 126 additions and 187 deletions
+3 -1
View File
@@ -4,8 +4,10 @@
### Changed ### Changed
- Rendered demoted cross-model reasoning blocks in the target model's canonical thinking dialect - Demote cross-vendor reasoning to plain text when the target does not natively support it
- Refine cross-model reasoning preservation to prevent leaking inert context into structured fields
- Rendered demoted cross-model reasoning blocks in the target model's canonical thinking dialect
- Improved reliability of AI model responses by implementing automatic retry logic for detected thinking-loop stalls - Improved reliability of AI model responses by implementing automatic retry logic for detected thinking-loop stalls
- Changed cross-provider/cross-model thinking demotion to render the prior turn's reasoning in the target model's canonical inline thinking dialect (a ```` ```thinking ```` fence for Gemini, `<think>`/`<thinking>` tags for others) instead of bare prose, with a neutral `<think>` fallback for control-token dialects (Harmony, Gemma) so chat-template tokens never leak into history. Replaying it as a native `thought` block was ruled out: end-to-end testing against Gemini 3 confirmed an unsigned `thought` part is schema-accepted but silently discarded — neither recalled nor influencing generation. - Changed cross-provider/cross-model thinking demotion to render the prior turn's reasoning in the target model's canonical inline thinking dialect (a ```` ```thinking ```` fence for Gemini, `<think>`/`<thinking>` tags for others) instead of bare prose, with a neutral `<think>` fallback for control-token dialects (Harmony, Gemma) so chat-template tokens never leak into history. Replaying it as a native `thought` block was ruled out: end-to-end testing against Gemini 3 confirmed an unsigned `thought` part is schema-accepted but silently discarded — neither recalled nor influencing generation.
+1
View File
@@ -24,6 +24,7 @@ import { getDialectDefinition } from "./factory";
*/ */
export function renderDemotedThinking(modelId: string, text: string): string { export function renderDemotedThinking(modelId: string, text: string): string {
if (!text) return ""; if (!text) return "";
text = text.toWellFormed();
const dialect = preferredDialect(modelId); const dialect = preferredDialect(modelId);
if (dialect === "harmony" || dialect === "gemma") return `<think>\n${text}\n</think>\n`; if (dialect === "harmony" || dialect === "gemma") return `<think>\n${text}\n</think>\n`;
return `${getDialectDefinition(dialect).renderThinking(text)}\n`; return `${getDialectDefinition(dialect).renderThinking(text)}\n`;
+2 -1
View File
@@ -18,6 +18,7 @@ import {
parseStreamingJson, parseStreamingJson,
parseStreamingJsonThrottled, parseStreamingJsonThrottled,
} from "@oh-my-pi/pi-utils"; } from "@oh-my-pi/pi-utils";
import { renderDemotedThinking } from "../dialect/demotion";
import { ProviderHttpError } from "../errors"; import { ProviderHttpError } from "../errors";
import type { import type {
Api, Api,
@@ -820,7 +821,7 @@ function convertMessages(
}); });
} else { } else {
// Model requires signature but we don't have one — demote to text // Model requires signature but we don't have one — demote to text
contentBlocks.push({ text: `[Thinking]: ${c.thinking.toWellFormed()}` }); contentBlocks.push({ text: renderDemotedThinking(model.id, c.thinking) });
} }
break; break;
default: default:
+3 -2
View File
@@ -19,6 +19,7 @@ import {
parseStreamingJsonThrottled, parseStreamingJsonThrottled,
readSseEvents, readSseEvents,
} from "@oh-my-pi/pi-utils"; } from "@oh-my-pi/pi-utils";
import { renderDemotedThinking } from "../dialect/demotion";
import { isUsageLimitError } from "../rate-limit-utils"; import { isUsageLimitError } from "../rate-limit-utils";
import { getEnvApiKey, OUTPUT_FALLBACK_BUFFER } from "../stream"; import { getEnvApiKey, OUTPUT_FALLBACK_BUFFER } from "../stream";
import type { import type {
@@ -3238,7 +3239,7 @@ export function convertAnthropicMessages(
if (block.thinking.trim().length === 0) continue; if (block.thinking.trim().length === 0) continue;
blocks.push({ blocks.push({
type: "text", type: "text",
text: block.thinking.toWellFormed(), text: renderDemotedThinking(model.id, block.thinking),
}); });
continue; continue;
} }
@@ -3260,7 +3261,7 @@ export function convertAnthropicMessages(
} else { } else {
blocks.push({ blocks.push({
type: "text", type: "text",
text: block.thinking.toWellFormed(), text: renderDemotedThinking(model.id, block.thinking),
}); });
} }
} else { } else {
+5 -6
View File
@@ -5,6 +5,7 @@
import { scheduler } from "node:timers/promises"; import { scheduler } from "node:timers/promises";
import { calculateCost } from "@oh-my-pi/pi-catalog/models"; import { calculateCost } from "@oh-my-pi/pi-catalog/models";
import { extractHttpStatusFromError, readSseJson } from "@oh-my-pi/pi-utils"; import { extractHttpStatusFromError, readSseJson } from "@oh-my-pi/pi-utils";
import { renderDemotedThinking } from "../dialect/demotion";
import { ProviderHttpError } from "../errors"; import { ProviderHttpError } from "../errors";
import type { import type {
Api, Api,
@@ -234,18 +235,16 @@ export function convertMessages<T extends GoogleApiType>(model: Model<T>, contex
} else if (block.type === "thinking") { } else if (block.type === "thinking") {
// Skip empty thinking blocks // Skip empty thinking blocks
if (!block.thinking || block.thinking.trim() === "") continue; if (!block.thinking || block.thinking.trim() === "") continue;
// Only keep as thinking block if same provider AND same model const thoughtSignature = resolveThoughtSignature(isSameProviderAndModel, block.thinkingSignature);
// Otherwise convert to plain text (no tags to avoid model mimicking them) if (thoughtSignature) {
if (isSameProviderAndModel) {
const thoughtSignature = resolveThoughtSignature(isSameProviderAndModel, block.thinkingSignature);
parts.push({ parts.push({
thought: true, thought: true,
text: block.thinking.toWellFormed(), text: block.thinking.toWellFormed(),
...(thoughtSignature && { thoughtSignature }), thoughtSignature,
}); });
} else { } else {
parts.push({ parts.push({
text: block.thinking.toWellFormed(), text: renderDemotedThinking(model.id, block.thinking),
}); });
} }
} else if (block.type === "toolCall") { } else if (block.type === "toolCall") {
@@ -4,6 +4,7 @@ import { resolveWireModelId } from "@oh-my-pi/pi-catalog/model-thinking";
import { calculateCost } from "@oh-my-pi/pi-catalog/models"; import { calculateCost } from "@oh-my-pi/pi-catalog/models";
import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types"; import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types";
import { $env, extractHttpStatusFromError, parseStreamingJson, parseStreamingJsonThrottled } from "@oh-my-pi/pi-utils"; import { $env, extractHttpStatusFromError, parseStreamingJson, parseStreamingJsonThrottled } from "@oh-my-pi/pi-utils";
import { renderDemotedThinking } from "../dialect/demotion";
import { getKimiCommonHeaders } from "../registry/oauth/kimi"; import { getKimiCommonHeaders } from "../registry/oauth/kimi";
import { getEnvApiKey } from "../stream"; import { getEnvApiKey } from "../stream";
import type { import type {
@@ -1775,13 +1776,14 @@ export function convertMessages(
const nonEmptyThinkingBlocks = thinkingBlocks.filter(b => b.thinking && b.thinking.trim().length > 0); const nonEmptyThinkingBlocks = thinkingBlocks.filter(b => b.thinking && b.thinking.trim().length > 0);
if (nonEmptyThinkingBlocks.length > 0) { if (nonEmptyThinkingBlocks.length > 0) {
if (compat.requiresThinkingAsText) { if (compat.requiresThinkingAsText) {
// Convert thinking blocks to plain text (no tags to avoid model mimicking them) const thinkingText = nonEmptyThinkingBlocks
const thinkingText = nonEmptyThinkingBlocks.map(b => b.thinking).join("\n\n"); .map(b => renderDemotedThinking(model.id, b.thinking))
.join("");
// `content` is a plain string at this point (set above) or null — // `content` is a plain string at this point (set above) or null —
// never an array. Prepend the thinking text to the string form. // never an array. Prepend the demoted thinking to the string form.
assistantMsg.content = assistantMsg.content =
typeof assistantMsg.content === "string" && assistantMsg.content.length > 0 typeof assistantMsg.content === "string" && assistantMsg.content.length > 0
? `${thinkingText}\n\n${assistantMsg.content}` ? `${thinkingText}${assistantMsg.content}`
: thinkingText; : thinkingText;
} else if (compat.requiresReasoningContentForToolCalls) { } else if (compat.requiresReasoningContentForToolCalls) {
// Use the streamed signature when the backend accepts whichever // Use the streamed signature when the backend accepts whichever
+19 -56
View File
@@ -227,52 +227,20 @@ function isAnthropicMessagesModel(model: Model): model is Model<"anthropic-messa
} }
/** /**
* Cross-API `openai-completions` targets that can replay a prior turn's * Targets that have proven they read unsigned foreign thinking when replayed
* reasoning as a native, signature-stripped `thinking` block on the wire. * natively. This is a semantic-carry allowlist only: OpenAI-compatible
* Anthropic's same-API path (`replayUnsignedThinking`) covers * `reasoning_content` schema requirements and llama.cpp cache-prefix replay are
* `anthropic-messages` targets directly; this is the analogue for the * handled by their encoders and MUST NOT make foreign thinking look meaningful.
* `openai-completions` branch of the cross-API path (#3433/#3434). 3p ↔ 3p
* replays between an Anthropic-compatible source (Z.AI Anthropic, Kimi
* Anthropic, …) and an OpenAI-compat reasoning target on the same vendor must
* keep reasoning as structured `reasoning_content` instead of degrading it to
* conversation text.
*
* `compat` MUST be the request-time RESOLVED compat that `convertMessages`
* threads into `transformMessages`, not `model.compat`. OpenCode-hosted
* reasoning models (`opencode-go`/`opencode-zen`) keep
* `requiresReasoningContentForToolCalls` off on the base compat to dodge the
* thinking-off `Extra inputs are not permitted` 400 (#1071) and reactivate it
* on `compat.whenThinking` for thinking-engaged requests to dodge the
* `thinking is enabled but reasoning_content is missing` 400 (#1484).
* `resolveOpenAICompatPolicy` already swaps in `whenThinking` for thinking-on
* requests, so basing this decision on the resolved compat keeps the predicate
* and the encoder in lockstep; reading `model.compat` would re-open #1484 for
* every cross-API switch into an OpenCode reasoning model.
*
* The downstream encoder MUST then surface the preserved block on the wire via
* `reasoningContentField` — see `openai-completions.ts` for the matching
* branch.
*/ */
function openAICompletionsReplaysUnsignedThinking(model: Model, compat: Model["compat"]): boolean { function targetReadsForeignThinking(model: Model, compat: Model["compat"]): boolean {
if (compat === undefined) return false;
if (model.api === "anthropic-messages") {
return "replayUnsignedThinking" in compat && compat.replayUnsignedThinking === true;
}
if (model.api !== "openai-completions") return false; if (model.api !== "openai-completions") return false;
if (compat === undefined || !("requiresReasoningContentForToolCalls" in compat)) return false; if (!("thinkingFormat" in compat)) return false;
if (compat.requiresThinkingAsText) return false; if (compat.requiresThinkingAsText) return false;
// Local llama.cpp-style servers (`replayReasoningContent`) need the replay return model.reasoning && compat.thinkingFormat === "zai";
// for KV-cache prefix reuse — Qwen3 / DeepSeek-R1 / GLM chat templates
// reconstruct the prior turn's `<think>` block from `reasoning_content`
// (#3528). Checked BEFORE the `model.reasoning` gate: the runtime discovery
// paths for `llama.cpp` / `lm-studio` / `openai-models-list` hardcode
// `reasoning: false` even when the upstream actually emits reasoning, so
// gating on the spec flag here would let a cross-API switch into such a
// target demote the prior `thinking` block to text and lose the
// cache-stable prefix `replayReasoningContent` is meant to preserve.
if (compat.replayReasoningContent) return true;
if (!model.reasoning) return false;
// Hosts that REQUIRE `reasoning_content` on tool-call turns (DeepSeek
// reasoning, Kimi, OpenRouter reasoning, OpenCode thinking-on) already
// accept the replay; Z.AI-format hosts (Z.AI, Zhipu, Moonshot Kimi native,
// Xiaomi MiMo) advertise `reasoning_content` as a continuation hint.
return compat.requiresReasoningContentForToolCalls || compat.thinkingFormat === "zai";
} }
const ANTHROPIC_TOOL_CALL_ID_PATTERN = /^[a-zA-Z0-9_-]{1,64}$/; const ANTHROPIC_TOOL_CALL_ID_PATTERN = /^[a-zA-Z0-9_-]{1,64}$/;
@@ -462,19 +430,14 @@ export function transformMessages<TApi extends Api>(
// thinking blocks before the cross-model paths. // thinking blocks before the cross-model paths.
if (!sanitized.thinking || sanitized.thinking.trim() === "") return []; if (!sanitized.thinking || sanitized.thinking.trim() === "") return [];
if (isSameModel) return sanitized; if (isSameModel) return sanitized;
// Cross-model + cross-API: preserve as a native, signature-stripped // Cross-model + cross-API: preserve native thinking only for
// `thinking` block whenever the target encoder can re-emit it on the // targets proven to read unsigned foreign reasoning (Z.AI-format
// wire (today: `openai-completions` reasoning targets that accept // OpenAI-compatible targets, plus Anthropic-compatible
// `reasoning_content` as a continuation hint — Z.AI, Zhipu, DeepSeek // `replayUnsignedThinking`). Tool-call schema requirements and
// reasoning, Kimi native, MiMo, OpenRouter reasoning, …). The source // llama.cpp cache-prefix replay are orthogonal encoder concerns;
// signature is always dropped because it is bound to the source // keeping inert foreign CoT native for those flags loses the
// wire-format (Anthropic crypto sig / OpenAI Responses encrypted // canonical visible-text fallback without adding model context.
// blob) and would be rejected by the target. Without this branch if (targetReadsForeignThinking(model, targetCompat)) {
// every cross-API 3p ↔ 3p switch (Z.AI Anthropic → Z.AI OpenAI,
// Kimi Anthropic → Kimi OpenAI, etc.) demoted prior reasoning to
// conversation text and lost it as structured reasoning context
// (#3433/#3434).
if (openAICompletionsReplaysUnsignedThinking(model, targetCompat)) {
return sanitized.thinkingSignature ? { ...sanitized, thinkingSignature: undefined } : sanitized; return sanitized.thinkingSignature ? { ...sanitized, thinkingSignature: undefined } : sanitized;
} }
// Other cross-API targets (openai-responses encrypted blobs, google // Other cross-API targets (openai-responses encrypted blobs, google
@@ -260,7 +260,7 @@ describe("Anthropic prior-turn thinking preservation (#2257, #2265)", () => {
const assistants = params.filter(p => p.role === "assistant"); const assistants = params.filter(p => p.role === "assistant");
const priorBlocks = assistants[0].content as WireBlock[]; const priorBlocks = assistants[0].content as WireBlock[];
const text = priorBlocks.find(b => b.type === "text") as WireTextBlock | undefined; const text = priorBlocks.find(b => b.type === "text") as WireTextBlock | undefined;
expect(text?.text).toBe("visible reasoning"); expect(text?.text).toBe(renderDemotedThinking(target.id, "visible reasoning"));
expect(priorBlocks.find(b => b.type === "thinking")).toBeUndefined(); expect(priorBlocks.find(b => b.type === "thinking")).toBeUndefined();
expect(priorBlocks.find(b => b.type === "redacted_thinking")).toBeUndefined(); expect(priorBlocks.find(b => b.type === "redacted_thinking")).toBeUndefined();
}); });
@@ -301,13 +301,9 @@ describe("Anthropic prior-turn thinking preservation (#2257, #2265)", () => {
expect(thinking?.signature).toBe(""); expect(thinking?.signature).toBe("");
}); });
it("demotes prior unsigned thinking from non-anthropic sources to canonical-dialect text, not native blocks", () => { it("preserves prior unsigned thinking from non-anthropic sources on unsigned-replay targets", () => {
// Cross-API replay: the prior turn came from OpenAI-responses with no // Anthropic-compatible targets that advertise `replayUnsignedThinking`
// Anthropic signature, so it can't wire as a native `thinking` block // accept unsigned native thinking as their semantic-carry analogue.
// (Anthropic rejects a foreign/missing signature). It is demoted to a
// text block wrapped in the TARGET's canonical thinking dialect
// (Anthropic → `<thinking>`) so the reasoning survives as recognizable
// reasoning context rather than bare prose.
const target = makeAnthropicModel(); const target = makeAnthropicModel();
const messages: Message[] = [ const messages: Message[] = [
makeUser("Summarize README"), makeUser("Summarize README"),
@@ -336,10 +332,8 @@ describe("Anthropic prior-turn thinking preservation (#2257, #2265)", () => {
const params = convertAnthropicMessages(messages, target, false); const params = convertAnthropicMessages(messages, target, false);
const assistants = params.filter(p => p.role === "assistant"); const assistants = params.filter(p => p.role === "assistant");
const priorBlocks = assistants[0].content as WireBlock[]; const priorBlocks = assistants[0].content as WireBlock[];
expect(priorBlocks.find(b => b.type === "thinking")).toBeUndefined(); const thinking = priorBlocks.find(b => b.type === "thinking") as WireThinkingBlock | undefined;
// Reasoning survives on the wire as text, wrapped in the target's canonical expect(thinking?.thinking).toBe("openai chain-of-thought");
// thinking dialect rather than emitted as a native (signature-bound) block. expect(thinking?.signature).toBe("");
const text = priorBlocks.find(b => b.type === "text") as WireTextBlock | undefined;
expect(text?.text?.trimEnd()).toBe(renderDemotedThinking(target.id, "openai chain-of-thought").trimEnd());
}); });
}); });
@@ -1,4 +1,5 @@
import { describe, expect, it } from "bun:test"; import { describe, expect, it } from "bun:test";
import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect";
import { convertAnthropicMessages, streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; import { convertAnthropicMessages, streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic";
import type { import type {
AssistantMessage, AssistantMessage,
@@ -234,7 +235,7 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => {
const model = makeModel({ provider: "anthropic", baseUrl: "https://api.anthropic.com" }); const model = makeModel({ provider: "anthropic", baseUrl: "https://api.anthropic.com" });
const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model); const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model);
expect(blocks[0]?.type).toBe("text"); expect(blocks[0]?.type).toBe("text");
expect((blocks[0] as WireTextBlock).text).toBe("internal scratch"); expect((blocks[0] as WireTextBlock).text).toBe(renderDemotedThinking(model.id, "internal scratch"));
}); });
it("treats a missing baseUrl as official Anthropic (resolveAnthropicBaseUrl default)", () => { it("treats a missing baseUrl as official Anthropic (resolveAnthropicBaseUrl default)", () => {
@@ -245,14 +246,14 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => {
const model = makeModel({ provider: "anthropic", baseUrl: "" }); const model = makeModel({ provider: "anthropic", baseUrl: "" });
const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model); const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model);
expect(blocks[0]?.type).toBe("text"); expect(blocks[0]?.type).toBe("text");
expect((blocks[0] as WireTextBlock).text).toBe("internal scratch"); expect((blocks[0] as WireTextBlock).text).toBe(renderDemotedThinking(model.id, "internal scratch"));
}); });
it("still degrades unsigned thinking to text for non-reasoning unknown endpoints", () => { it("still degrades unsigned thinking to text for non-reasoning unknown endpoints", () => {
const model = makeModel({ reasoning: false, baseUrl: "https://plain.example.com/anthropic" }); const model = makeModel({ reasoning: false, baseUrl: "https://plain.example.com/anthropic" });
const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("scratch")], model); const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("scratch")], model);
expect(blocks[0]?.type).toBe("text"); expect(blocks[0]?.type).toBe("text");
expect((blocks[0] as WireTextBlock).text).toBe("scratch"); expect((blocks[0] as WireTextBlock).text).toBe(renderDemotedThinking(model.id, "scratch"));
}); });
it("keeps thinking → tool_use pairing intact across continuation conversion", () => { it("keeps thinking → tool_use pairing intact across continuation conversion", () => {
@@ -1,4 +1,5 @@
import { describe, expect, it } from "bun:test"; import { describe, expect, it } from "bun:test";
import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect";
import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions";
import type { AssistantMessage, Model, ModelSpec, ThinkingContent, ToolCall } from "@oh-my-pi/pi-ai/types"; import type { AssistantMessage, Model, ModelSpec, ThinkingContent, ToolCall } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { buildModel } from "@oh-my-pi/pi-catalog/build";
@@ -327,7 +328,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
// Should have set reasoning_content from the thinking text via the openai path. // Should have set reasoning_content from the thinking text via the openai path.
expect(assistant?.reasoning_content).toBe("some reasoning"); expect(assistant?.reasoning_content).toBe("some reasoning");
}); });
it("replays cross-api thinking with stripped signature through reasoning_content", () => { it("demotes cross-api foreign thinking while satisfying tool-call reasoning_content schema", () => {
const model = deepseekModel({ const model = deepseekModel({
provider: "opencode-go", provider: "opencode-go",
baseUrl: "https://opencode.ai/zen/go/v1", baseUrl: "https://opencode.ai/zen/go/v1",
@@ -354,8 +355,8 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
const messages = convertMessages(model, { messages: [msg] }, compat); const messages = convertMessages(model, { messages: [msg] }, compat);
const assistant = findOpenAICompletionAssistantWireMessage(messages); const assistant = findOpenAICompletionAssistantWireMessage(messages);
expect(assistant).toBeDefined(); expect(assistant).toBeDefined();
expect(assistant?.reasoning_content).toBe("Need to preserve cross-api reasoning."); expect(assistant?.reasoning_content).toBe("");
expect(assistant?.content).toBe(""); expect(assistant?.content).toBe(renderDemotedThinking(model.id, "Need to preserve cross-api reasoning."));
}); });
it("falls through to empty-string when thinking block has opaque signature and empty text", () => { it("falls through to empty-string when thinking block has opaque signature and empty text", () => {
const model = deepseekModel({ const model = deepseekModel({
@@ -1,4 +1,5 @@
import { describe, expect, it } from "bun:test"; import { describe, expect, it } from "bun:test";
import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect";
import { streamGoogle } from "@oh-my-pi/pi-ai/providers/google"; import { streamGoogle } from "@oh-my-pi/pi-ai/providers/google";
import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { buildModel } from "@oh-my-pi/pi-catalog/build";
@@ -78,4 +79,33 @@ describe("Google provider system prompts", () => {
}); });
expect(payload.contents).toHaveLength(1); expect(payload.contents).toHaveLength(1);
}); });
it("demotes same-model unsigned thinking instead of emitting an unsigned thought part", async () => {
const payload = await captureGooglePayload({
messages: [
{
role: "assistant",
api: "google-generative-ai",
provider: "google",
model: model.id,
content: [{ type: "thinking", thinking: "unsigned prior thought" }],
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "stop",
timestamp: 1,
},
],
});
expect(payload.contents[0]).toEqual({
role: "model",
parts: [{ text: renderDemotedThinking(model.id, "unsigned prior thought") }],
});
});
}); });
+26 -78
View File
@@ -1,32 +1,18 @@
/** /**
* Regression guard for cross-API 3p ↔ 3p thinking-block preservation (#3434). * Regression guard for cross-API 3p ↔ 3p thinking-block handling (#3434).
* *
* Mid-session switches between an Anthropic-compatible 3p provider and an * Mid-session switches can replay a prior assistant turn whose native reasoning
* OpenAI-compatible 3p provider on the same vendor (Z.AI Anthropic → Z.AI * slot was authored by a different provider. Live provider probes showed that
* OpenAI, Kimi Anthropic → Kimi OpenAI, …) used to demote every prior * unsigned foreign reasoning is only semantically carried by Z.AI-format
* `thinking` block to plain text on the cross-API path of `transformMessages`: * OpenAI-compatible targets; schema requirements such as
* * `requiresReasoningContentForToolCalls` and local llama.cpp cache-prefix replay
* // Cross-API target: keep the existing text-demotion fallback. * do not make the reasoning meaningful. Non-allowlisted targets demote the
* return { type: "text", text: sanitized.thinking }; * reasoning into canonical visible text so the next model can still read it.
*
* The next request shipped the reasoning chain as conversation text instead
* of structured `reasoning_content`, so the target model lost the prior
* reasoning context and the user paid twice — once to generate the thinking
* on the source endpoint, once again to re-derive it on the target.
*
* The fix has two halves:
*
* 1. `transformMessages` preserves the prior thinking text as a native,
* signature-stripped `thinking` block whenever the target encoder can
* re-emit it on the wire (today: `openai-completions` reasoning targets
* that accept `reasoning_content` as a continuation hint).
* 2. The `openai-completions` encoder surfaces those preserved blocks via
* `reasoningContentField` even for hosts that don't strictly require
* `reasoning_content` — specifically `thinkingFormat: "zai"` targets.
* *
* This file pins the wire output for the canonical scenarios. * This file pins the wire output for the canonical scenarios.
*/ */
import { describe, expect, it } from "bun:test"; import { describe, expect, it } from "bun:test";
import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect";
import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions";
import type { AssistantMessage, Message, Model, ModelSpec, UserMessage } from "@oh-my-pi/pi-ai/types"; import type { AssistantMessage, Message, Model, ModelSpec, UserMessage } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { buildModel } from "@oh-my-pi/pi-catalog/build";
@@ -98,9 +84,10 @@ function zaiOpenAITarget(): Model<"openai-completions"> {
} }
function deepseekReasoningTarget(): Model<"openai-completions"> { function deepseekReasoningTarget(): Model<"openai-completions"> {
// DeepSeek-family reasoning target: requiresReasoningContentForToolCalls is // DeepSeek-family reasoning targets require `reasoning_content` for schema
// true here, so the preserved block reaches reasoning_content via the // validity, but measured foreign reasoning in that slot is inert. Cross-API
// existing recovery branch. Guards the other half of the fix from regressing. // foreign thinking must demote to text; the encoder may still emit an empty
// schema placeholder where required.
return buildModel({ return buildModel({
id: "deepseek-v4-flash", id: "deepseek-v4-flash",
name: "DeepSeek V4 Flash", name: "DeepSeek V4 Flash",
@@ -119,12 +106,8 @@ function opencodeGoKimiTarget(): Model<"openai-completions"> {
// OpenCode Go's reasoning-enabled Kimi. Base compat keeps // OpenCode Go's reasoning-enabled Kimi. Base compat keeps
// `requiresReasoningContentForToolCalls: false` to dodge the // `requiresReasoningContentForToolCalls: false` to dodge the
// `Extra inputs are not permitted` 400 (#1071); only the resolved // `Extra inputs are not permitted` 400 (#1071); only the resolved
// `whenThinking` policy reactivates it (#1484). `convertMessages` threads // `whenThinking` policy reactivates it (#1484). That schema requirement must
// that request-time resolved compat into `transformMessages`, so a // not preserve foreign non-tool-call reasoning as native semantic context.
// thinking-on request preserves the prior reasoning; without the resolved
// compat the predicate would read base compat, demote to text, and the
// next thinking-on request would 400 with `thinking is enabled but
// reasoning_content is missing in assistant tool call message at index N`.
return buildModel({ return buildModel({
id: "kimi-k2.6", id: "kimi-k2.6",
name: "Kimi K2.6", name: "Kimi K2.6",
@@ -197,11 +180,7 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => {
expect(assistant.reasoning_content).toBe("opaque continuation metadata payload"); expect(assistant.reasoning_content).toBe("opaque continuation metadata payload");
}); });
it("emits reasoning_content on Anthropic 3p → DeepSeek cross-API switch", () => { it("demotes Anthropic 3p → DeepSeek cross-API thinking instead of semantic replay", () => {
// DeepSeek-family reasoning targets reach reasoning_content via the
// existing `requiresReasoningContentForToolCalls` recovery branch. This
// pin guards against a regression in either fix half that would drop
// the preserved block before recovery runs.
const target = deepseekReasoningTarget(); const target = deepseekReasoningTarget();
const messages: Message[] = [ const messages: Message[] = [
userMessage("Inspect README"), userMessage("Inspect README"),
@@ -214,15 +193,11 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => {
expect(assistant).toBeDefined(); expect(assistant).toBeDefined();
if (!assistant) throw new Error("assistant message missing"); if (!assistant) throw new Error("assistant message missing");
expect(assistant.reasoning_content).toBe("Read README and answer."); expect(assistant.reasoning_content).toBe("");
expect(assistant.content).toBe(`${renderDemotedThinking(target.id, "Read README and answer.")}Done.`);
}); });
it("demotes thinking to text when the target cannot replay reasoning_content", () => { it("demotes thinking to canonical text when the target cannot replay it semantically", () => {
// Anthropic 3p → official OpenAI non-reasoning model: the encoder
// cannot emit `reasoning_content` here (the field would be ignored and
// strict OpenAI-compat shims would reject it). Reasoning must survive
// at minimum as visible conversation text so the next turn still sees
// the prior plan.
const target = openAIGpt4oTarget(); const target = openAIGpt4oTarget();
const messages: Message[] = [ const messages: Message[] = [
userMessage("Plan it."), userMessage("Plan it."),
@@ -236,23 +211,10 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => {
if (!assistant) throw new Error("assistant message missing"); if (!assistant) throw new Error("assistant message missing");
expect(assistant.reasoning_content).toBeUndefined(); expect(assistant.reasoning_content).toBeUndefined();
const content = assistant.content; expect(assistant.content).toBe(`${renderDemotedThinking(target.id, "Explore the repo, then patch it.")}Done.`);
expect(typeof content).toBe("string");
if (typeof content !== "string") throw new Error("content not a string");
expect(content).toContain("Explore the repo, then patch it.");
expect(content).toContain("Done.");
}); });
it("preserves cross-API thinking for OpenCode reasoning targets that gate replay via compat.whenThinking", () => { it("demotes cross-API thinking for OpenCode reasoning targets with whenThinking schema", () => {
// OpenCode (`opencode-go`, `opencode-zen`) reasoning models keep
// `requiresReasoningContentForToolCalls: false` on the base compat
// (dodges the thinking-off `Extra inputs are not permitted` 400 — #1071)
// and reactivate the flag on `compat.whenThinking` for thinking-engaged
// requests (dodges the `thinking is enabled but reasoning_content is
// missing` 400 — #1484). The cross-API preservation predicate must run
// against the resolved compat that `convertMessages` threads in (the
// `whenThinking` view here); reading base compat would demote the prior
// thinking to text and re-trigger #1484 on the next thinking-on request.
const target = opencodeGoKimiTarget(); const target = opencodeGoKimiTarget();
const messages: Message[] = [ const messages: Message[] = [
userMessage("Plan it."), userMessage("Plan it."),
@@ -261,9 +223,7 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => {
]; ];
// Resolve the thinking-engaged compat the way `streamOpenAICompletions` // Resolve the thinking-engaged compat the way `streamOpenAICompletions`
// does for a request with reasoning effort set, then hand it to // does for a request with reasoning effort set.
// `convertMessages` directly so the test exercises the same encoder
// configuration the live wire would.
const compat = target.compat.whenThinking ?? target.compat; const compat = target.compat.whenThinking ?? target.compat;
expect(compat.requiresReasoningContentForToolCalls).toBe(true); expect(compat.requiresReasoningContentForToolCalls).toBe(true);
@@ -272,19 +232,11 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => {
expect(assistant).toBeDefined(); expect(assistant).toBeDefined();
if (!assistant) throw new Error("assistant message missing"); if (!assistant) throw new Error("assistant message missing");
expect(assistant.reasoning_content).toBe("Read README and answer."); expect(assistant.reasoning_content).toBeUndefined();
expect(assistant.content).toBe(`${renderDemotedThinking(target.id, "Read README and answer.")}Done.`);
}); });
it("demotes prior thinking to content when the OpenCode base compat (thinking off) cannot surface reasoning_content", () => { it("demotes prior thinking to content when the OpenCode base compat runs with thinking off", () => {
// Companion of the prior test: same OpenCode target, but the request
// runs against the BASE compat (thinking disabled, the path that bars
// `reasoning_content` per #1071). The cross-API preservation predicate
// reads this resolved base compat — which neither requires
// `reasoning_content` nor is a Z.AI-format host — so it preserves no
// native thinking block the encoder couldn't surface; the cross-API path
// instead text-demotes the prior reasoning into visible content. The
// reasoning still survives as conversation context, with no
// `reasoning_content` on the wire and no #1071 regression.
const target = opencodeGoKimiTarget(); const target = opencodeGoKimiTarget();
const compat = target.compat; const compat = target.compat;
expect(compat.requiresReasoningContentForToolCalls).toBe(false); expect(compat.requiresReasoningContentForToolCalls).toBe(false);
@@ -301,11 +253,7 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => {
if (!assistant) throw new Error("assistant message missing"); if (!assistant) throw new Error("assistant message missing");
expect(assistant.reasoning_content).toBeUndefined(); expect(assistant.reasoning_content).toBeUndefined();
const content = assistant.content; expect(assistant.content).toBe(`${renderDemotedThinking(target.id, "Read README and answer.")}Done.`);
expect(typeof content).toBe("string");
if (typeof content !== "string") throw new Error("content not a string");
expect(content).toContain("Read README and answer.");
expect(content).toContain("Done.");
}); });
it("does not promote markup-healed same-model thinking into visible content", () => { it("does not promote markup-healed same-model thinking into visible content", () => {
+10 -15
View File
@@ -40,6 +40,7 @@
* This file pins the wire output across the relevant axes. * This file pins the wire output across the relevant axes.
*/ */
import { describe, expect, it } from "bun:test"; import { describe, expect, it } from "bun:test";
import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect";
import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions";
import { import {
applyChatCompletionsReasoningParams, applyChatCompletionsReasoningParams,
@@ -239,17 +240,11 @@ describe("llama.cpp warm-prefix preservation (#3528)", () => {
expect(assistant.reasoning_content).toBe("Trace the call graph through service.ts and the registry."); expect(assistant.reasoning_content).toBe("Trace the call graph through service.ts and the registry.");
}); });
it("preserves cross-API thinking into a discovered local target (reasoning: false on the spec)", () => { it("demotes cross-API thinking into a discovered local target", () => {
// Cross-API/model switch into a discovered llama.cpp target: an // `replayReasoningContent` is a same-wire cache concern. It keeps
// Anthropic-source thinking block (opaque continuation signature, foreign // llama.cpp turns cache-stable when the prior assistant already emitted an
// to the openai-completions wire) must NOT be demoted to text just because // OpenAI-compatible reasoning field, but it must not preserve foreign
// the discovery path stamped `reasoning: false` on the spec. The // Anthropic reasoning as native semantic context.
// `replayReasoningContent` flag has to bypass the `model.reasoning` gate
// in `transform-messages.ts` for the cross-API replay branch to fire, so
// the encoder receives a signature-stripped thinking block to surface as
// `reasoning_content` on the wire. Without the bypass the prior turn's
// reasoning rides as plain conversation text and the local server still
// loses the cache-stable `<think>` prefix.
const target = llamaCppQwenModel({ reasoning: false }); const target = llamaCppQwenModel({ reasoning: false });
const anthropicSourceTurn: AssistantMessage = { const anthropicSourceTurn: AssistantMessage = {
role: "assistant", role: "assistant",
@@ -287,10 +282,10 @@ describe("llama.cpp warm-prefix preservation (#3528)", () => {
target.compat, target.compat,
); );
const found = findAssistantMessage(wire) as Record<string, unknown> | undefined; const found = findAssistantMessage(wire) as Record<string, unknown> | undefined;
expect(found?.reasoning_content).toBe("Cross-vendor reasoning chain that must survive the switch."); expect(found?.reasoning_content).toBeUndefined();
expect(found?.content).toBe("Switched-in answer."); expect(found?.content).toBe(
// The Anthropic continuation signature is bound to the source wire and `${renderDemotedThinking(target.id, "Cross-vendor reasoning chain that must survive the switch.")}Switched-in answer.`,
// must NEVER leak as a stray field name on the openai-completions target. );
expect("EvAnthropicOpaqueContinuationBlob==" in (found ?? {})).toBe(false); expect("EvAnthropicOpaqueContinuationBlob==" in (found ?? {})).toBe(false);
}); });
@@ -1,4 +1,5 @@
import { describe, expect, it } from "bun:test"; import { describe, expect, it } from "bun:test";
import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect";
import { import {
applyOpenRouterRoutingVariant, applyOpenRouterRoutingVariant,
convertMessages, convertMessages,
@@ -273,7 +274,7 @@ describe("openai-completions compatibility", () => {
// Regression: thinking+text replay used to call `.unshift` on the string // Regression: thinking+text replay used to call `.unshift` on the string
// content set above (TypeError). Both blocks must survive as one string. // content set above (TypeError). Both blocks must survive as one string.
expect(typeof assistant.content).toBe("string"); expect(typeof assistant.content).toBe("string");
expect(assistant.content).toBe("chain of thought\n\nfinal answer"); expect(assistant.content).toBe(`${renderDemotedThinking(model.id, "chain of thought")}final answer`);
}); });
it("emits thinking-only assistant content as a plain string when requiresThinkingAsText is set", () => { it("emits thinking-only assistant content as a plain string when requiresThinkingAsText is set", () => {
@@ -309,7 +310,7 @@ describe("openai-completions compatibility", () => {
const assistant = messages.find(message => message.role === "assistant"); const assistant = messages.find(message => message.role === "assistant");
expect(assistant).toBeDefined(); expect(assistant).toBeDefined();
if (assistant?.role !== "assistant") throw new Error("assistant message missing"); if (assistant?.role !== "assistant") throw new Error("assistant message missing");
expect(assistant.content).toBe("only thoughts"); expect(assistant.content).toBe(renderDemotedThinking(model.id, "only thoughts"));
}); });
it("preserves multiple system prompts as leading system messages for chat completions", () => { it("preserves multiple system prompts as leading system messages for chat completions", () => {
@@ -1227,7 +1228,7 @@ describe("kimi model detection via detectCompat", () => {
expect(assistant?.reasoning).toBeUndefined(); expect(assistant?.reasoning).toBeUndefined();
}); });
it("uses thinking-enabled compat when replaying cross-api reasoning on kimi opencode-go", async () => { it("demotes cross-api reasoning while keeping thinking-enabled tool-call schema on kimi opencode-go", async () => {
const model = kimiOpenCodeModel("kimi-k2.6"); const model = kimiOpenCodeModel("kimi-k2.6");
expect(model.compat.requiresReasoningContentForToolCalls).toBe(false); expect(model.compat.requiresReasoningContentForToolCalls).toBe(false);
const priorAssistant: AssistantMessage = { const priorAssistant: AssistantMessage = {
@@ -1290,8 +1291,8 @@ describe("kimi model detection via detectCompat", () => {
const payload = (await promise) as { messages: Array<Record<string, unknown>> }; const payload = (await promise) as { messages: Array<Record<string, unknown>> };
const assistant = payload.messages.find(m => m.role === "assistant"); const assistant = payload.messages.find(m => m.role === "assistant");
expect(assistant).toBeDefined(); expect(assistant).toBeDefined();
expect(assistant?.content).toBe("."); expect(assistant?.content).toBe(renderDemotedThinking(model.id, "Need to preserve cross-api reasoning."));
expect(assistant?.reasoning_content).toBe("Need to preserve cross-api reasoning."); expect(assistant?.reasoning_content).toBe("");
expect(assistant?.reasoning).toBeUndefined(); expect(assistant?.reasoning).toBeUndefined();
expect(assistant?.reasoning_text).toBeUndefined(); expect(assistant?.reasoning_text).toBeUndefined();
}); });