diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index a7e7bef38..3f2cb961a 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,8 +4,10 @@ ### Changed -- Rendered demoted cross-model reasoning blocks in the target model's canonical thinking dialect +- Demote cross-vendor reasoning to plain text when the target does not natively support it +- Refine cross-model reasoning preservation to prevent leaking inert context into structured fields +- Rendered demoted cross-model reasoning blocks in the target model's canonical thinking dialect - Improved reliability of AI model responses by implementing automatic retry logic for detected thinking-loop stalls - Changed cross-provider/cross-model thinking demotion to render the prior turn's reasoning in the target model's canonical inline thinking dialect (a ```` ```thinking ```` fence for Gemini, ``/`` tags for others) instead of bare prose, with a neutral `` fallback for control-token dialects (Harmony, Gemma) so chat-template tokens never leak into history. Replaying it as a native `thought` block was ruled out: end-to-end testing against Gemini 3 confirmed an unsigned `thought` part is schema-accepted but silently discarded — neither recalled nor influencing generation. diff --git a/packages/ai/src/dialect/demotion.ts b/packages/ai/src/dialect/demotion.ts index 065d0dd4e..b554b2185 100644 --- a/packages/ai/src/dialect/demotion.ts +++ b/packages/ai/src/dialect/demotion.ts @@ -24,6 +24,7 @@ import { getDialectDefinition } from "./factory"; */ export function renderDemotedThinking(modelId: string, text: string): string { if (!text) return ""; + text = text.toWellFormed(); const dialect = preferredDialect(modelId); if (dialect === "harmony" || dialect === "gemma") return `\n${text}\n\n`; return `${getDialectDefinition(dialect).renderThinking(text)}\n`; diff --git a/packages/ai/src/providers/amazon-bedrock.ts b/packages/ai/src/providers/amazon-bedrock.ts index 9433d5d4d..9928aead3 100644 --- a/packages/ai/src/providers/amazon-bedrock.ts +++ b/packages/ai/src/providers/amazon-bedrock.ts @@ -18,6 +18,7 @@ import { parseStreamingJson, parseStreamingJsonThrottled, } from "@oh-my-pi/pi-utils"; +import { renderDemotedThinking } from "../dialect/demotion"; import { ProviderHttpError } from "../errors"; import type { Api, @@ -820,7 +821,7 @@ function convertMessages( }); } else { // Model requires signature but we don't have one — demote to text - contentBlocks.push({ text: `[Thinking]: ${c.thinking.toWellFormed()}` }); + contentBlocks.push({ text: renderDemotedThinking(model.id, c.thinking) }); } break; default: diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index c48537a5c..e50b09197 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -19,6 +19,7 @@ import { parseStreamingJsonThrottled, readSseEvents, } from "@oh-my-pi/pi-utils"; +import { renderDemotedThinking } from "../dialect/demotion"; import { isUsageLimitError } from "../rate-limit-utils"; import { getEnvApiKey, OUTPUT_FALLBACK_BUFFER } from "../stream"; import type { @@ -3238,7 +3239,7 @@ export function convertAnthropicMessages( if (block.thinking.trim().length === 0) continue; blocks.push({ type: "text", - text: block.thinking.toWellFormed(), + text: renderDemotedThinking(model.id, block.thinking), }); continue; } @@ -3260,7 +3261,7 @@ export function convertAnthropicMessages( } else { blocks.push({ type: "text", - text: block.thinking.toWellFormed(), + text: renderDemotedThinking(model.id, block.thinking), }); } } else { diff --git a/packages/ai/src/providers/google-shared.ts b/packages/ai/src/providers/google-shared.ts index fed3d0865..b61f2c08c 100644 --- a/packages/ai/src/providers/google-shared.ts +++ b/packages/ai/src/providers/google-shared.ts @@ -5,6 +5,7 @@ import { scheduler } from "node:timers/promises"; import { calculateCost } from "@oh-my-pi/pi-catalog/models"; import { extractHttpStatusFromError, readSseJson } from "@oh-my-pi/pi-utils"; +import { renderDemotedThinking } from "../dialect/demotion"; import { ProviderHttpError } from "../errors"; import type { Api, @@ -234,18 +235,16 @@ export function convertMessages(model: Model, contex } else if (block.type === "thinking") { // Skip empty thinking blocks if (!block.thinking || block.thinking.trim() === "") continue; - // Only keep as thinking block if same provider AND same model - // Otherwise convert to plain text (no tags to avoid model mimicking them) - if (isSameProviderAndModel) { - const thoughtSignature = resolveThoughtSignature(isSameProviderAndModel, block.thinkingSignature); + const thoughtSignature = resolveThoughtSignature(isSameProviderAndModel, block.thinkingSignature); + if (thoughtSignature) { parts.push({ thought: true, text: block.thinking.toWellFormed(), - ...(thoughtSignature && { thoughtSignature }), + thoughtSignature, }); } else { parts.push({ - text: block.thinking.toWellFormed(), + text: renderDemotedThinking(model.id, block.thinking), }); } } else if (block.type === "toolCall") { diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index ca3c6c32d..b875c7bf8 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -4,6 +4,7 @@ import { resolveWireModelId } from "@oh-my-pi/pi-catalog/model-thinking"; import { calculateCost } from "@oh-my-pi/pi-catalog/models"; import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types"; import { $env, extractHttpStatusFromError, parseStreamingJson, parseStreamingJsonThrottled } from "@oh-my-pi/pi-utils"; +import { renderDemotedThinking } from "../dialect/demotion"; import { getKimiCommonHeaders } from "../registry/oauth/kimi"; import { getEnvApiKey } from "../stream"; import type { @@ -1775,13 +1776,14 @@ export function convertMessages( const nonEmptyThinkingBlocks = thinkingBlocks.filter(b => b.thinking && b.thinking.trim().length > 0); if (nonEmptyThinkingBlocks.length > 0) { if (compat.requiresThinkingAsText) { - // Convert thinking blocks to plain text (no tags to avoid model mimicking them) - const thinkingText = nonEmptyThinkingBlocks.map(b => b.thinking).join("\n\n"); + const thinkingText = nonEmptyThinkingBlocks + .map(b => renderDemotedThinking(model.id, b.thinking)) + .join(""); // `content` is a plain string at this point (set above) or null — - // never an array. Prepend the thinking text to the string form. + // never an array. Prepend the demoted thinking to the string form. assistantMsg.content = typeof assistantMsg.content === "string" && assistantMsg.content.length > 0 - ? `${thinkingText}\n\n${assistantMsg.content}` + ? `${thinkingText}${assistantMsg.content}` : thinkingText; } else if (compat.requiresReasoningContentForToolCalls) { // Use the streamed signature when the backend accepts whichever diff --git a/packages/ai/src/providers/transform-messages.ts b/packages/ai/src/providers/transform-messages.ts index b72b1dbaa..6c75c0b63 100644 --- a/packages/ai/src/providers/transform-messages.ts +++ b/packages/ai/src/providers/transform-messages.ts @@ -227,52 +227,20 @@ function isAnthropicMessagesModel(model: Model): model is Model<"anthropic-messa } /** - * Cross-API `openai-completions` targets that can replay a prior turn's - * reasoning as a native, signature-stripped `thinking` block on the wire. - * Anthropic's same-API path (`replayUnsignedThinking`) covers - * `anthropic-messages` targets directly; this is the analogue for the - * `openai-completions` branch of the cross-API path (#3433/#3434). 3p ↔ 3p - * replays between an Anthropic-compatible source (Z.AI Anthropic, Kimi - * Anthropic, …) and an OpenAI-compat reasoning target on the same vendor must - * keep reasoning as structured `reasoning_content` instead of degrading it to - * conversation text. - * - * `compat` MUST be the request-time RESOLVED compat that `convertMessages` - * threads into `transformMessages`, not `model.compat`. OpenCode-hosted - * reasoning models (`opencode-go`/`opencode-zen`) keep - * `requiresReasoningContentForToolCalls` off on the base compat to dodge the - * thinking-off `Extra inputs are not permitted` 400 (#1071) and reactivate it - * on `compat.whenThinking` for thinking-engaged requests to dodge the - * `thinking is enabled but reasoning_content is missing` 400 (#1484). - * `resolveOpenAICompatPolicy` already swaps in `whenThinking` for thinking-on - * requests, so basing this decision on the resolved compat keeps the predicate - * and the encoder in lockstep; reading `model.compat` would re-open #1484 for - * every cross-API switch into an OpenCode reasoning model. - * - * The downstream encoder MUST then surface the preserved block on the wire via - * `reasoningContentField` — see `openai-completions.ts` for the matching - * branch. + * Targets that have proven they read unsigned foreign thinking when replayed + * natively. This is a semantic-carry allowlist only: OpenAI-compatible + * `reasoning_content` schema requirements and llama.cpp cache-prefix replay are + * handled by their encoders and MUST NOT make foreign thinking look meaningful. */ -function openAICompletionsReplaysUnsignedThinking(model: Model, compat: Model["compat"]): boolean { +function targetReadsForeignThinking(model: Model, compat: Model["compat"]): boolean { + if (compat === undefined) return false; + if (model.api === "anthropic-messages") { + return "replayUnsignedThinking" in compat && compat.replayUnsignedThinking === true; + } if (model.api !== "openai-completions") return false; - if (compat === undefined || !("requiresReasoningContentForToolCalls" in compat)) return false; + if (!("thinkingFormat" in compat)) return false; if (compat.requiresThinkingAsText) return false; - // Local llama.cpp-style servers (`replayReasoningContent`) need the replay - // for KV-cache prefix reuse — Qwen3 / DeepSeek-R1 / GLM chat templates - // reconstruct the prior turn's `` block from `reasoning_content` - // (#3528). Checked BEFORE the `model.reasoning` gate: the runtime discovery - // paths for `llama.cpp` / `lm-studio` / `openai-models-list` hardcode - // `reasoning: false` even when the upstream actually emits reasoning, so - // gating on the spec flag here would let a cross-API switch into such a - // target demote the prior `thinking` block to text and lose the - // cache-stable prefix `replayReasoningContent` is meant to preserve. - if (compat.replayReasoningContent) return true; - if (!model.reasoning) return false; - // Hosts that REQUIRE `reasoning_content` on tool-call turns (DeepSeek - // reasoning, Kimi, OpenRouter reasoning, OpenCode thinking-on) already - // accept the replay; Z.AI-format hosts (Z.AI, Zhipu, Moonshot Kimi native, - // Xiaomi MiMo) advertise `reasoning_content` as a continuation hint. - return compat.requiresReasoningContentForToolCalls || compat.thinkingFormat === "zai"; + return model.reasoning && compat.thinkingFormat === "zai"; } const ANTHROPIC_TOOL_CALL_ID_PATTERN = /^[a-zA-Z0-9_-]{1,64}$/; @@ -462,19 +430,14 @@ export function transformMessages( // thinking blocks before the cross-model paths. if (!sanitized.thinking || sanitized.thinking.trim() === "") return []; if (isSameModel) return sanitized; - // Cross-model + cross-API: preserve as a native, signature-stripped - // `thinking` block whenever the target encoder can re-emit it on the - // wire (today: `openai-completions` reasoning targets that accept - // `reasoning_content` as a continuation hint — Z.AI, Zhipu, DeepSeek - // reasoning, Kimi native, MiMo, OpenRouter reasoning, …). The source - // signature is always dropped because it is bound to the source - // wire-format (Anthropic crypto sig / OpenAI Responses encrypted - // blob) and would be rejected by the target. Without this branch - // every cross-API 3p ↔ 3p switch (Z.AI Anthropic → Z.AI OpenAI, - // Kimi Anthropic → Kimi OpenAI, etc.) demoted prior reasoning to - // conversation text and lost it as structured reasoning context - // (#3433/#3434). - if (openAICompletionsReplaysUnsignedThinking(model, targetCompat)) { + // Cross-model + cross-API: preserve native thinking only for + // targets proven to read unsigned foreign reasoning (Z.AI-format + // OpenAI-compatible targets, plus Anthropic-compatible + // `replayUnsignedThinking`). Tool-call schema requirements and + // llama.cpp cache-prefix replay are orthogonal encoder concerns; + // keeping inert foreign CoT native for those flags loses the + // canonical visible-text fallback without adding model context. + if (targetReadsForeignThinking(model, targetCompat)) { return sanitized.thinkingSignature ? { ...sanitized, thinkingSignature: undefined } : sanitized; } // Other cross-API targets (openai-responses encrypted blobs, google diff --git a/packages/ai/test/anthropic-prior-turn-thinking.test.ts b/packages/ai/test/anthropic-prior-turn-thinking.test.ts index 48285135e..8b242a193 100644 --- a/packages/ai/test/anthropic-prior-turn-thinking.test.ts +++ b/packages/ai/test/anthropic-prior-turn-thinking.test.ts @@ -260,7 +260,7 @@ describe("Anthropic prior-turn thinking preservation (#2257, #2265)", () => { const assistants = params.filter(p => p.role === "assistant"); const priorBlocks = assistants[0].content as WireBlock[]; const text = priorBlocks.find(b => b.type === "text") as WireTextBlock | undefined; - expect(text?.text).toBe("visible reasoning"); + expect(text?.text).toBe(renderDemotedThinking(target.id, "visible reasoning")); expect(priorBlocks.find(b => b.type === "thinking")).toBeUndefined(); expect(priorBlocks.find(b => b.type === "redacted_thinking")).toBeUndefined(); }); @@ -301,13 +301,9 @@ describe("Anthropic prior-turn thinking preservation (#2257, #2265)", () => { expect(thinking?.signature).toBe(""); }); - it("demotes prior unsigned thinking from non-anthropic sources to canonical-dialect text, not native blocks", () => { - // Cross-API replay: the prior turn came from OpenAI-responses with no - // Anthropic signature, so it can't wire as a native `thinking` block - // (Anthropic rejects a foreign/missing signature). It is demoted to a - // text block wrapped in the TARGET's canonical thinking dialect - // (Anthropic → ``) so the reasoning survives as recognizable - // reasoning context rather than bare prose. + it("preserves prior unsigned thinking from non-anthropic sources on unsigned-replay targets", () => { + // Anthropic-compatible targets that advertise `replayUnsignedThinking` + // accept unsigned native thinking as their semantic-carry analogue. const target = makeAnthropicModel(); const messages: Message[] = [ makeUser("Summarize README"), @@ -336,10 +332,8 @@ describe("Anthropic prior-turn thinking preservation (#2257, #2265)", () => { const params = convertAnthropicMessages(messages, target, false); const assistants = params.filter(p => p.role === "assistant"); const priorBlocks = assistants[0].content as WireBlock[]; - expect(priorBlocks.find(b => b.type === "thinking")).toBeUndefined(); - // Reasoning survives on the wire as text, wrapped in the target's canonical - // thinking dialect rather than emitted as a native (signature-bound) block. - const text = priorBlocks.find(b => b.type === "text") as WireTextBlock | undefined; - expect(text?.text?.trimEnd()).toBe(renderDemotedThinking(target.id, "openai chain-of-thought").trimEnd()); + const thinking = priorBlocks.find(b => b.type === "thinking") as WireThinkingBlock | undefined; + expect(thinking?.thinking).toBe("openai chain-of-thought"); + expect(thinking?.signature).toBe(""); }); }); diff --git a/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts index 6cd7369b2..5238f582c 100644 --- a/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts +++ b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts @@ -1,4 +1,5 @@ import { describe, expect, it } from "bun:test"; +import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect"; import { convertAnthropicMessages, streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; import type { AssistantMessage, @@ -234,7 +235,7 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => { const model = makeModel({ provider: "anthropic", baseUrl: "https://api.anthropic.com" }); const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model); expect(blocks[0]?.type).toBe("text"); - expect((blocks[0] as WireTextBlock).text).toBe("internal scratch"); + expect((blocks[0] as WireTextBlock).text).toBe(renderDemotedThinking(model.id, "internal scratch")); }); it("treats a missing baseUrl as official Anthropic (resolveAnthropicBaseUrl default)", () => { @@ -245,14 +246,14 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => { const model = makeModel({ provider: "anthropic", baseUrl: "" }); const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model); expect(blocks[0]?.type).toBe("text"); - expect((blocks[0] as WireTextBlock).text).toBe("internal scratch"); + expect((blocks[0] as WireTextBlock).text).toBe(renderDemotedThinking(model.id, "internal scratch")); }); it("still degrades unsigned thinking to text for non-reasoning unknown endpoints", () => { const model = makeModel({ reasoning: false, baseUrl: "https://plain.example.com/anthropic" }); const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("scratch")], model); expect(blocks[0]?.type).toBe("text"); - expect((blocks[0] as WireTextBlock).text).toBe("scratch"); + expect((blocks[0] as WireTextBlock).text).toBe(renderDemotedThinking(model.id, "scratch")); }); it("keeps thinking → tool_use pairing intact across continuation conversion", () => { diff --git a/packages/ai/test/deepseek-reasoning-content.test.ts b/packages/ai/test/deepseek-reasoning-content.test.ts index 16bae42cf..3cbdfd349 100644 --- a/packages/ai/test/deepseek-reasoning-content.test.ts +++ b/packages/ai/test/deepseek-reasoning-content.test.ts @@ -1,4 +1,5 @@ import { describe, expect, it } from "bun:test"; +import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect"; import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; import type { AssistantMessage, Model, ModelSpec, ThinkingContent, ToolCall } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; @@ -327,7 +328,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => { // Should have set reasoning_content from the thinking text via the openai path. expect(assistant?.reasoning_content).toBe("some reasoning"); }); - it("replays cross-api thinking with stripped signature through reasoning_content", () => { + it("demotes cross-api foreign thinking while satisfying tool-call reasoning_content schema", () => { const model = deepseekModel({ provider: "opencode-go", baseUrl: "https://opencode.ai/zen/go/v1", @@ -354,8 +355,8 @@ describe("DeepSeek reasoning_content tool-call replay", () => { const messages = convertMessages(model, { messages: [msg] }, compat); const assistant = findOpenAICompletionAssistantWireMessage(messages); expect(assistant).toBeDefined(); - expect(assistant?.reasoning_content).toBe("Need to preserve cross-api reasoning."); - expect(assistant?.content).toBe(""); + expect(assistant?.reasoning_content).toBe(""); + expect(assistant?.content).toBe(renderDemotedThinking(model.id, "Need to preserve cross-api reasoning.")); }); it("falls through to empty-string when thinking block has opaque signature and empty text", () => { const model = deepseekModel({ diff --git a/packages/ai/test/google-system-prompt.test.ts b/packages/ai/test/google-system-prompt.test.ts index 3298fa389..1ca718188 100644 --- a/packages/ai/test/google-system-prompt.test.ts +++ b/packages/ai/test/google-system-prompt.test.ts @@ -1,4 +1,5 @@ import { describe, expect, it } from "bun:test"; +import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect"; import { streamGoogle } from "@oh-my-pi/pi-ai/providers/google"; import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; @@ -78,4 +79,33 @@ describe("Google provider system prompts", () => { }); expect(payload.contents).toHaveLength(1); }); + + it("demotes same-model unsigned thinking instead of emitting an unsigned thought part", async () => { + const payload = await captureGooglePayload({ + messages: [ + { + role: "assistant", + api: "google-generative-ai", + provider: "google", + model: model.id, + content: [{ type: "thinking", thinking: "unsigned prior thought" }], + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp: 1, + }, + ], + }); + + expect(payload.contents[0]).toEqual({ + role: "model", + parts: [{ text: renderDemotedThinking(model.id, "unsigned prior thought") }], + }); + }); }); diff --git a/packages/ai/test/issue-3434-repro.test.ts b/packages/ai/test/issue-3434-repro.test.ts index f909a07c4..29c55130b 100644 --- a/packages/ai/test/issue-3434-repro.test.ts +++ b/packages/ai/test/issue-3434-repro.test.ts @@ -1,32 +1,18 @@ /** - * Regression guard for cross-API 3p ↔ 3p thinking-block preservation (#3434). + * Regression guard for cross-API 3p ↔ 3p thinking-block handling (#3434). * - * Mid-session switches between an Anthropic-compatible 3p provider and an - * OpenAI-compatible 3p provider on the same vendor (Z.AI Anthropic → Z.AI - * OpenAI, Kimi Anthropic → Kimi OpenAI, …) used to demote every prior - * `thinking` block to plain text on the cross-API path of `transformMessages`: - * - * // Cross-API target: keep the existing text-demotion fallback. - * return { type: "text", text: sanitized.thinking }; - * - * The next request shipped the reasoning chain as conversation text instead - * of structured `reasoning_content`, so the target model lost the prior - * reasoning context and the user paid twice — once to generate the thinking - * on the source endpoint, once again to re-derive it on the target. - * - * The fix has two halves: - * - * 1. `transformMessages` preserves the prior thinking text as a native, - * signature-stripped `thinking` block whenever the target encoder can - * re-emit it on the wire (today: `openai-completions` reasoning targets - * that accept `reasoning_content` as a continuation hint). - * 2. The `openai-completions` encoder surfaces those preserved blocks via - * `reasoningContentField` even for hosts that don't strictly require - * `reasoning_content` — specifically `thinkingFormat: "zai"` targets. + * Mid-session switches can replay a prior assistant turn whose native reasoning + * slot was authored by a different provider. Live provider probes showed that + * unsigned foreign reasoning is only semantically carried by Z.AI-format + * OpenAI-compatible targets; schema requirements such as + * `requiresReasoningContentForToolCalls` and local llama.cpp cache-prefix replay + * do not make the reasoning meaningful. Non-allowlisted targets demote the + * reasoning into canonical visible text so the next model can still read it. * * This file pins the wire output for the canonical scenarios. */ import { describe, expect, it } from "bun:test"; +import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect"; import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; import type { AssistantMessage, Message, Model, ModelSpec, UserMessage } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; @@ -98,9 +84,10 @@ function zaiOpenAITarget(): Model<"openai-completions"> { } function deepseekReasoningTarget(): Model<"openai-completions"> { - // DeepSeek-family reasoning target: requiresReasoningContentForToolCalls is - // true here, so the preserved block reaches reasoning_content via the - // existing recovery branch. Guards the other half of the fix from regressing. + // DeepSeek-family reasoning targets require `reasoning_content` for schema + // validity, but measured foreign reasoning in that slot is inert. Cross-API + // foreign thinking must demote to text; the encoder may still emit an empty + // schema placeholder where required. return buildModel({ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", @@ -119,12 +106,8 @@ function opencodeGoKimiTarget(): Model<"openai-completions"> { // OpenCode Go's reasoning-enabled Kimi. Base compat keeps // `requiresReasoningContentForToolCalls: false` to dodge the // `Extra inputs are not permitted` 400 (#1071); only the resolved - // `whenThinking` policy reactivates it (#1484). `convertMessages` threads - // that request-time resolved compat into `transformMessages`, so a - // thinking-on request preserves the prior reasoning; without the resolved - // compat the predicate would read base compat, demote to text, and the - // next thinking-on request would 400 with `thinking is enabled but - // reasoning_content is missing in assistant tool call message at index N`. + // `whenThinking` policy reactivates it (#1484). That schema requirement must + // not preserve foreign non-tool-call reasoning as native semantic context. return buildModel({ id: "kimi-k2.6", name: "Kimi K2.6", @@ -197,11 +180,7 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => { expect(assistant.reasoning_content).toBe("opaque continuation metadata payload"); }); - it("emits reasoning_content on Anthropic 3p → DeepSeek cross-API switch", () => { - // DeepSeek-family reasoning targets reach reasoning_content via the - // existing `requiresReasoningContentForToolCalls` recovery branch. This - // pin guards against a regression in either fix half that would drop - // the preserved block before recovery runs. + it("demotes Anthropic 3p → DeepSeek cross-API thinking instead of semantic replay", () => { const target = deepseekReasoningTarget(); const messages: Message[] = [ userMessage("Inspect README"), @@ -214,15 +193,11 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => { expect(assistant).toBeDefined(); if (!assistant) throw new Error("assistant message missing"); - expect(assistant.reasoning_content).toBe("Read README and answer."); + expect(assistant.reasoning_content).toBe(""); + expect(assistant.content).toBe(`${renderDemotedThinking(target.id, "Read README and answer.")}Done.`); }); - it("demotes thinking to text when the target cannot replay reasoning_content", () => { - // Anthropic 3p → official OpenAI non-reasoning model: the encoder - // cannot emit `reasoning_content` here (the field would be ignored and - // strict OpenAI-compat shims would reject it). Reasoning must survive - // at minimum as visible conversation text so the next turn still sees - // the prior plan. + it("demotes thinking to canonical text when the target cannot replay it semantically", () => { const target = openAIGpt4oTarget(); const messages: Message[] = [ userMessage("Plan it."), @@ -236,23 +211,10 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => { if (!assistant) throw new Error("assistant message missing"); expect(assistant.reasoning_content).toBeUndefined(); - const content = assistant.content; - expect(typeof content).toBe("string"); - if (typeof content !== "string") throw new Error("content not a string"); - expect(content).toContain("Explore the repo, then patch it."); - expect(content).toContain("Done."); + expect(assistant.content).toBe(`${renderDemotedThinking(target.id, "Explore the repo, then patch it.")}Done.`); }); - it("preserves cross-API thinking for OpenCode reasoning targets that gate replay via compat.whenThinking", () => { - // OpenCode (`opencode-go`, `opencode-zen`) reasoning models keep - // `requiresReasoningContentForToolCalls: false` on the base compat - // (dodges the thinking-off `Extra inputs are not permitted` 400 — #1071) - // and reactivate the flag on `compat.whenThinking` for thinking-engaged - // requests (dodges the `thinking is enabled but reasoning_content is - // missing` 400 — #1484). The cross-API preservation predicate must run - // against the resolved compat that `convertMessages` threads in (the - // `whenThinking` view here); reading base compat would demote the prior - // thinking to text and re-trigger #1484 on the next thinking-on request. + it("demotes cross-API thinking for OpenCode reasoning targets with whenThinking schema", () => { const target = opencodeGoKimiTarget(); const messages: Message[] = [ userMessage("Plan it."), @@ -261,9 +223,7 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => { ]; // Resolve the thinking-engaged compat the way `streamOpenAICompletions` - // does for a request with reasoning effort set, then hand it to - // `convertMessages` directly so the test exercises the same encoder - // configuration the live wire would. + // does for a request with reasoning effort set. const compat = target.compat.whenThinking ?? target.compat; expect(compat.requiresReasoningContentForToolCalls).toBe(true); @@ -272,19 +232,11 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => { expect(assistant).toBeDefined(); if (!assistant) throw new Error("assistant message missing"); - expect(assistant.reasoning_content).toBe("Read README and answer."); + expect(assistant.reasoning_content).toBeUndefined(); + expect(assistant.content).toBe(`${renderDemotedThinking(target.id, "Read README and answer.")}Done.`); }); - it("demotes prior thinking to content when the OpenCode base compat (thinking off) cannot surface reasoning_content", () => { - // Companion of the prior test: same OpenCode target, but the request - // runs against the BASE compat (thinking disabled, the path that bars - // `reasoning_content` per #1071). The cross-API preservation predicate - // reads this resolved base compat — which neither requires - // `reasoning_content` nor is a Z.AI-format host — so it preserves no - // native thinking block the encoder couldn't surface; the cross-API path - // instead text-demotes the prior reasoning into visible content. The - // reasoning still survives as conversation context, with no - // `reasoning_content` on the wire and no #1071 regression. + it("demotes prior thinking to content when the OpenCode base compat runs with thinking off", () => { const target = opencodeGoKimiTarget(); const compat = target.compat; expect(compat.requiresReasoningContentForToolCalls).toBe(false); @@ -301,11 +253,7 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => { if (!assistant) throw new Error("assistant message missing"); expect(assistant.reasoning_content).toBeUndefined(); - const content = assistant.content; - expect(typeof content).toBe("string"); - if (typeof content !== "string") throw new Error("content not a string"); - expect(content).toContain("Read README and answer."); - expect(content).toContain("Done."); + expect(assistant.content).toBe(`${renderDemotedThinking(target.id, "Read README and answer.")}Done.`); }); it("does not promote markup-healed same-model thinking into visible content", () => { diff --git a/packages/ai/test/issue-3528-repro.test.ts b/packages/ai/test/issue-3528-repro.test.ts index df4e87490..77ebeae09 100644 --- a/packages/ai/test/issue-3528-repro.test.ts +++ b/packages/ai/test/issue-3528-repro.test.ts @@ -40,6 +40,7 @@ * This file pins the wire output across the relevant axes. */ import { describe, expect, it } from "bun:test"; +import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect"; import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { applyChatCompletionsReasoningParams, @@ -239,17 +240,11 @@ describe("llama.cpp warm-prefix preservation (#3528)", () => { expect(assistant.reasoning_content).toBe("Trace the call graph through service.ts and the registry."); }); - it("preserves cross-API thinking into a discovered local target (reasoning: false on the spec)", () => { - // Cross-API/model switch into a discovered llama.cpp target: an - // Anthropic-source thinking block (opaque continuation signature, foreign - // to the openai-completions wire) must NOT be demoted to text just because - // the discovery path stamped `reasoning: false` on the spec. The - // `replayReasoningContent` flag has to bypass the `model.reasoning` gate - // in `transform-messages.ts` for the cross-API replay branch to fire, so - // the encoder receives a signature-stripped thinking block to surface as - // `reasoning_content` on the wire. Without the bypass the prior turn's - // reasoning rides as plain conversation text and the local server still - // loses the cache-stable `` prefix. + it("demotes cross-API thinking into a discovered local target", () => { + // `replayReasoningContent` is a same-wire cache concern. It keeps + // llama.cpp turns cache-stable when the prior assistant already emitted an + // OpenAI-compatible reasoning field, but it must not preserve foreign + // Anthropic reasoning as native semantic context. const target = llamaCppQwenModel({ reasoning: false }); const anthropicSourceTurn: AssistantMessage = { role: "assistant", @@ -287,10 +282,10 @@ describe("llama.cpp warm-prefix preservation (#3528)", () => { target.compat, ); const found = findAssistantMessage(wire) as Record | undefined; - expect(found?.reasoning_content).toBe("Cross-vendor reasoning chain that must survive the switch."); - expect(found?.content).toBe("Switched-in answer."); - // The Anthropic continuation signature is bound to the source wire and - // must NEVER leak as a stray field name on the openai-completions target. + expect(found?.reasoning_content).toBeUndefined(); + expect(found?.content).toBe( + `${renderDemotedThinking(target.id, "Cross-vendor reasoning chain that must survive the switch.")}Switched-in answer.`, + ); expect("EvAnthropicOpaqueContinuationBlob==" in (found ?? {})).toBe(false); }); diff --git a/packages/ai/test/openai-completions-compat.test.ts b/packages/ai/test/openai-completions-compat.test.ts index 669b15737..6eb74af66 100644 --- a/packages/ai/test/openai-completions-compat.test.ts +++ b/packages/ai/test/openai-completions-compat.test.ts @@ -1,4 +1,5 @@ import { describe, expect, it } from "bun:test"; +import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect"; import { applyOpenRouterRoutingVariant, convertMessages, @@ -273,7 +274,7 @@ describe("openai-completions compatibility", () => { // Regression: thinking+text replay used to call `.unshift` on the string // content set above (TypeError). Both blocks must survive as one string. expect(typeof assistant.content).toBe("string"); - expect(assistant.content).toBe("chain of thought\n\nfinal answer"); + expect(assistant.content).toBe(`${renderDemotedThinking(model.id, "chain of thought")}final answer`); }); it("emits thinking-only assistant content as a plain string when requiresThinkingAsText is set", () => { @@ -309,7 +310,7 @@ describe("openai-completions compatibility", () => { const assistant = messages.find(message => message.role === "assistant"); expect(assistant).toBeDefined(); if (assistant?.role !== "assistant") throw new Error("assistant message missing"); - expect(assistant.content).toBe("only thoughts"); + expect(assistant.content).toBe(renderDemotedThinking(model.id, "only thoughts")); }); it("preserves multiple system prompts as leading system messages for chat completions", () => { @@ -1227,7 +1228,7 @@ describe("kimi model detection via detectCompat", () => { expect(assistant?.reasoning).toBeUndefined(); }); - it("uses thinking-enabled compat when replaying cross-api reasoning on kimi opencode-go", async () => { + it("demotes cross-api reasoning while keeping thinking-enabled tool-call schema on kimi opencode-go", async () => { const model = kimiOpenCodeModel("kimi-k2.6"); expect(model.compat.requiresReasoningContentForToolCalls).toBe(false); const priorAssistant: AssistantMessage = { @@ -1290,8 +1291,8 @@ describe("kimi model detection via detectCompat", () => { const payload = (await promise) as { messages: Array> }; const assistant = payload.messages.find(m => m.role === "assistant"); expect(assistant).toBeDefined(); - expect(assistant?.content).toBe("."); - expect(assistant?.reasoning_content).toBe("Need to preserve cross-api reasoning."); + expect(assistant?.content).toBe(renderDemotedThinking(model.id, "Need to preserve cross-api reasoning.")); + expect(assistant?.reasoning_content).toBe(""); expect(assistant?.reasoning).toBeUndefined(); expect(assistant?.reasoning_text).toBeUndefined(); });