feat(ai): standardized thinking block demotion across model providers
- Centralized thinking block demotion logic using `renderDemotedThinking` across all model providers. - Updated `transformMessages` to default to text-based demotion for foreign thinking content. - Restricted foreign thinking preservation to explicitly supported targets with `zai` thinking formats. - Refactored message transformations and updated test suites to validate standardized demotion outcomes.
This commit is contained in:
@@ -4,8 +4,10 @@
|
||||
|
||||
### Changed
|
||||
|
||||
- Rendered demoted cross-model reasoning blocks in the target model's canonical thinking dialect
|
||||
- Demote cross-vendor reasoning to plain text when the target does not natively support it
|
||||
- Refine cross-model reasoning preservation to prevent leaking inert context into structured fields
|
||||
|
||||
- Rendered demoted cross-model reasoning blocks in the target model's canonical thinking dialect
|
||||
- Improved reliability of AI model responses by implementing automatic retry logic for detected thinking-loop stalls
|
||||
- Changed cross-provider/cross-model thinking demotion to render the prior turn's reasoning in the target model's canonical inline thinking dialect (a ```` ```thinking ```` fence for Gemini, `<think>`/`<thinking>` tags for others) instead of bare prose, with a neutral `<think>` fallback for control-token dialects (Harmony, Gemma) so chat-template tokens never leak into history. Replaying it as a native `thought` block was ruled out: end-to-end testing against Gemini 3 confirmed an unsigned `thought` part is schema-accepted but silently discarded — neither recalled nor influencing generation.
|
||||
|
||||
|
||||
@@ -24,6 +24,7 @@ import { getDialectDefinition } from "./factory";
|
||||
*/
|
||||
export function renderDemotedThinking(modelId: string, text: string): string {
|
||||
if (!text) return "";
|
||||
text = text.toWellFormed();
|
||||
const dialect = preferredDialect(modelId);
|
||||
if (dialect === "harmony" || dialect === "gemma") return `<think>\n${text}\n</think>\n`;
|
||||
return `${getDialectDefinition(dialect).renderThinking(text)}\n`;
|
||||
|
||||
@@ -18,6 +18,7 @@ import {
|
||||
parseStreamingJson,
|
||||
parseStreamingJsonThrottled,
|
||||
} from "@oh-my-pi/pi-utils";
|
||||
import { renderDemotedThinking } from "../dialect/demotion";
|
||||
import { ProviderHttpError } from "../errors";
|
||||
import type {
|
||||
Api,
|
||||
@@ -820,7 +821,7 @@ function convertMessages(
|
||||
});
|
||||
} else {
|
||||
// Model requires signature but we don't have one — demote to text
|
||||
contentBlocks.push({ text: `[Thinking]: ${c.thinking.toWellFormed()}` });
|
||||
contentBlocks.push({ text: renderDemotedThinking(model.id, c.thinking) });
|
||||
}
|
||||
break;
|
||||
default:
|
||||
|
||||
@@ -19,6 +19,7 @@ import {
|
||||
parseStreamingJsonThrottled,
|
||||
readSseEvents,
|
||||
} from "@oh-my-pi/pi-utils";
|
||||
import { renderDemotedThinking } from "../dialect/demotion";
|
||||
import { isUsageLimitError } from "../rate-limit-utils";
|
||||
import { getEnvApiKey, OUTPUT_FALLBACK_BUFFER } from "../stream";
|
||||
import type {
|
||||
@@ -3238,7 +3239,7 @@ export function convertAnthropicMessages(
|
||||
if (block.thinking.trim().length === 0) continue;
|
||||
blocks.push({
|
||||
type: "text",
|
||||
text: block.thinking.toWellFormed(),
|
||||
text: renderDemotedThinking(model.id, block.thinking),
|
||||
});
|
||||
continue;
|
||||
}
|
||||
@@ -3260,7 +3261,7 @@ export function convertAnthropicMessages(
|
||||
} else {
|
||||
blocks.push({
|
||||
type: "text",
|
||||
text: block.thinking.toWellFormed(),
|
||||
text: renderDemotedThinking(model.id, block.thinking),
|
||||
});
|
||||
}
|
||||
} else {
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
import { scheduler } from "node:timers/promises";
|
||||
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
||||
import { extractHttpStatusFromError, readSseJson } from "@oh-my-pi/pi-utils";
|
||||
import { renderDemotedThinking } from "../dialect/demotion";
|
||||
import { ProviderHttpError } from "../errors";
|
||||
import type {
|
||||
Api,
|
||||
@@ -234,18 +235,16 @@ export function convertMessages<T extends GoogleApiType>(model: Model<T>, contex
|
||||
} else if (block.type === "thinking") {
|
||||
// Skip empty thinking blocks
|
||||
if (!block.thinking || block.thinking.trim() === "") continue;
|
||||
// Only keep as thinking block if same provider AND same model
|
||||
// Otherwise convert to plain text (no tags to avoid model mimicking them)
|
||||
if (isSameProviderAndModel) {
|
||||
const thoughtSignature = resolveThoughtSignature(isSameProviderAndModel, block.thinkingSignature);
|
||||
const thoughtSignature = resolveThoughtSignature(isSameProviderAndModel, block.thinkingSignature);
|
||||
if (thoughtSignature) {
|
||||
parts.push({
|
||||
thought: true,
|
||||
text: block.thinking.toWellFormed(),
|
||||
...(thoughtSignature && { thoughtSignature }),
|
||||
thoughtSignature,
|
||||
});
|
||||
} else {
|
||||
parts.push({
|
||||
text: block.thinking.toWellFormed(),
|
||||
text: renderDemotedThinking(model.id, block.thinking),
|
||||
});
|
||||
}
|
||||
} else if (block.type === "toolCall") {
|
||||
|
||||
@@ -4,6 +4,7 @@ import { resolveWireModelId } from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
||||
import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types";
|
||||
import { $env, extractHttpStatusFromError, parseStreamingJson, parseStreamingJsonThrottled } from "@oh-my-pi/pi-utils";
|
||||
import { renderDemotedThinking } from "../dialect/demotion";
|
||||
import { getKimiCommonHeaders } from "../registry/oauth/kimi";
|
||||
import { getEnvApiKey } from "../stream";
|
||||
import type {
|
||||
@@ -1775,13 +1776,14 @@ export function convertMessages(
|
||||
const nonEmptyThinkingBlocks = thinkingBlocks.filter(b => b.thinking && b.thinking.trim().length > 0);
|
||||
if (nonEmptyThinkingBlocks.length > 0) {
|
||||
if (compat.requiresThinkingAsText) {
|
||||
// Convert thinking blocks to plain text (no tags to avoid model mimicking them)
|
||||
const thinkingText = nonEmptyThinkingBlocks.map(b => b.thinking).join("\n\n");
|
||||
const thinkingText = nonEmptyThinkingBlocks
|
||||
.map(b => renderDemotedThinking(model.id, b.thinking))
|
||||
.join("");
|
||||
// `content` is a plain string at this point (set above) or null —
|
||||
// never an array. Prepend the thinking text to the string form.
|
||||
// never an array. Prepend the demoted thinking to the string form.
|
||||
assistantMsg.content =
|
||||
typeof assistantMsg.content === "string" && assistantMsg.content.length > 0
|
||||
? `${thinkingText}\n\n${assistantMsg.content}`
|
||||
? `${thinkingText}${assistantMsg.content}`
|
||||
: thinkingText;
|
||||
} else if (compat.requiresReasoningContentForToolCalls) {
|
||||
// Use the streamed signature when the backend accepts whichever
|
||||
|
||||
@@ -227,52 +227,20 @@ function isAnthropicMessagesModel(model: Model): model is Model<"anthropic-messa
|
||||
}
|
||||
|
||||
/**
|
||||
* Cross-API `openai-completions` targets that can replay a prior turn's
|
||||
* reasoning as a native, signature-stripped `thinking` block on the wire.
|
||||
* Anthropic's same-API path (`replayUnsignedThinking`) covers
|
||||
* `anthropic-messages` targets directly; this is the analogue for the
|
||||
* `openai-completions` branch of the cross-API path (#3433/#3434). 3p ↔ 3p
|
||||
* replays between an Anthropic-compatible source (Z.AI Anthropic, Kimi
|
||||
* Anthropic, …) and an OpenAI-compat reasoning target on the same vendor must
|
||||
* keep reasoning as structured `reasoning_content` instead of degrading it to
|
||||
* conversation text.
|
||||
*
|
||||
* `compat` MUST be the request-time RESOLVED compat that `convertMessages`
|
||||
* threads into `transformMessages`, not `model.compat`. OpenCode-hosted
|
||||
* reasoning models (`opencode-go`/`opencode-zen`) keep
|
||||
* `requiresReasoningContentForToolCalls` off on the base compat to dodge the
|
||||
* thinking-off `Extra inputs are not permitted` 400 (#1071) and reactivate it
|
||||
* on `compat.whenThinking` for thinking-engaged requests to dodge the
|
||||
* `thinking is enabled but reasoning_content is missing` 400 (#1484).
|
||||
* `resolveOpenAICompatPolicy` already swaps in `whenThinking` for thinking-on
|
||||
* requests, so basing this decision on the resolved compat keeps the predicate
|
||||
* and the encoder in lockstep; reading `model.compat` would re-open #1484 for
|
||||
* every cross-API switch into an OpenCode reasoning model.
|
||||
*
|
||||
* The downstream encoder MUST then surface the preserved block on the wire via
|
||||
* `reasoningContentField` — see `openai-completions.ts` for the matching
|
||||
* branch.
|
||||
* Targets that have proven they read unsigned foreign thinking when replayed
|
||||
* natively. This is a semantic-carry allowlist only: OpenAI-compatible
|
||||
* `reasoning_content` schema requirements and llama.cpp cache-prefix replay are
|
||||
* handled by their encoders and MUST NOT make foreign thinking look meaningful.
|
||||
*/
|
||||
function openAICompletionsReplaysUnsignedThinking(model: Model, compat: Model["compat"]): boolean {
|
||||
function targetReadsForeignThinking(model: Model, compat: Model["compat"]): boolean {
|
||||
if (compat === undefined) return false;
|
||||
if (model.api === "anthropic-messages") {
|
||||
return "replayUnsignedThinking" in compat && compat.replayUnsignedThinking === true;
|
||||
}
|
||||
if (model.api !== "openai-completions") return false;
|
||||
if (compat === undefined || !("requiresReasoningContentForToolCalls" in compat)) return false;
|
||||
if (!("thinkingFormat" in compat)) return false;
|
||||
if (compat.requiresThinkingAsText) return false;
|
||||
// Local llama.cpp-style servers (`replayReasoningContent`) need the replay
|
||||
// for KV-cache prefix reuse — Qwen3 / DeepSeek-R1 / GLM chat templates
|
||||
// reconstruct the prior turn's `<think>` block from `reasoning_content`
|
||||
// (#3528). Checked BEFORE the `model.reasoning` gate: the runtime discovery
|
||||
// paths for `llama.cpp` / `lm-studio` / `openai-models-list` hardcode
|
||||
// `reasoning: false` even when the upstream actually emits reasoning, so
|
||||
// gating on the spec flag here would let a cross-API switch into such a
|
||||
// target demote the prior `thinking` block to text and lose the
|
||||
// cache-stable prefix `replayReasoningContent` is meant to preserve.
|
||||
if (compat.replayReasoningContent) return true;
|
||||
if (!model.reasoning) return false;
|
||||
// Hosts that REQUIRE `reasoning_content` on tool-call turns (DeepSeek
|
||||
// reasoning, Kimi, OpenRouter reasoning, OpenCode thinking-on) already
|
||||
// accept the replay; Z.AI-format hosts (Z.AI, Zhipu, Moonshot Kimi native,
|
||||
// Xiaomi MiMo) advertise `reasoning_content` as a continuation hint.
|
||||
return compat.requiresReasoningContentForToolCalls || compat.thinkingFormat === "zai";
|
||||
return model.reasoning && compat.thinkingFormat === "zai";
|
||||
}
|
||||
|
||||
const ANTHROPIC_TOOL_CALL_ID_PATTERN = /^[a-zA-Z0-9_-]{1,64}$/;
|
||||
@@ -462,19 +430,14 @@ export function transformMessages<TApi extends Api>(
|
||||
// thinking blocks before the cross-model paths.
|
||||
if (!sanitized.thinking || sanitized.thinking.trim() === "") return [];
|
||||
if (isSameModel) return sanitized;
|
||||
// Cross-model + cross-API: preserve as a native, signature-stripped
|
||||
// `thinking` block whenever the target encoder can re-emit it on the
|
||||
// wire (today: `openai-completions` reasoning targets that accept
|
||||
// `reasoning_content` as a continuation hint — Z.AI, Zhipu, DeepSeek
|
||||
// reasoning, Kimi native, MiMo, OpenRouter reasoning, …). The source
|
||||
// signature is always dropped because it is bound to the source
|
||||
// wire-format (Anthropic crypto sig / OpenAI Responses encrypted
|
||||
// blob) and would be rejected by the target. Without this branch
|
||||
// every cross-API 3p ↔ 3p switch (Z.AI Anthropic → Z.AI OpenAI,
|
||||
// Kimi Anthropic → Kimi OpenAI, etc.) demoted prior reasoning to
|
||||
// conversation text and lost it as structured reasoning context
|
||||
// (#3433/#3434).
|
||||
if (openAICompletionsReplaysUnsignedThinking(model, targetCompat)) {
|
||||
// Cross-model + cross-API: preserve native thinking only for
|
||||
// targets proven to read unsigned foreign reasoning (Z.AI-format
|
||||
// OpenAI-compatible targets, plus Anthropic-compatible
|
||||
// `replayUnsignedThinking`). Tool-call schema requirements and
|
||||
// llama.cpp cache-prefix replay are orthogonal encoder concerns;
|
||||
// keeping inert foreign CoT native for those flags loses the
|
||||
// canonical visible-text fallback without adding model context.
|
||||
if (targetReadsForeignThinking(model, targetCompat)) {
|
||||
return sanitized.thinkingSignature ? { ...sanitized, thinkingSignature: undefined } : sanitized;
|
||||
}
|
||||
// Other cross-API targets (openai-responses encrypted blobs, google
|
||||
|
||||
@@ -260,7 +260,7 @@ describe("Anthropic prior-turn thinking preservation (#2257, #2265)", () => {
|
||||
const assistants = params.filter(p => p.role === "assistant");
|
||||
const priorBlocks = assistants[0].content as WireBlock[];
|
||||
const text = priorBlocks.find(b => b.type === "text") as WireTextBlock | undefined;
|
||||
expect(text?.text).toBe("visible reasoning");
|
||||
expect(text?.text).toBe(renderDemotedThinking(target.id, "visible reasoning"));
|
||||
expect(priorBlocks.find(b => b.type === "thinking")).toBeUndefined();
|
||||
expect(priorBlocks.find(b => b.type === "redacted_thinking")).toBeUndefined();
|
||||
});
|
||||
@@ -301,13 +301,9 @@ describe("Anthropic prior-turn thinking preservation (#2257, #2265)", () => {
|
||||
expect(thinking?.signature).toBe("");
|
||||
});
|
||||
|
||||
it("demotes prior unsigned thinking from non-anthropic sources to canonical-dialect text, not native blocks", () => {
|
||||
// Cross-API replay: the prior turn came from OpenAI-responses with no
|
||||
// Anthropic signature, so it can't wire as a native `thinking` block
|
||||
// (Anthropic rejects a foreign/missing signature). It is demoted to a
|
||||
// text block wrapped in the TARGET's canonical thinking dialect
|
||||
// (Anthropic → `<thinking>`) so the reasoning survives as recognizable
|
||||
// reasoning context rather than bare prose.
|
||||
it("preserves prior unsigned thinking from non-anthropic sources on unsigned-replay targets", () => {
|
||||
// Anthropic-compatible targets that advertise `replayUnsignedThinking`
|
||||
// accept unsigned native thinking as their semantic-carry analogue.
|
||||
const target = makeAnthropicModel();
|
||||
const messages: Message[] = [
|
||||
makeUser("Summarize README"),
|
||||
@@ -336,10 +332,8 @@ describe("Anthropic prior-turn thinking preservation (#2257, #2265)", () => {
|
||||
const params = convertAnthropicMessages(messages, target, false);
|
||||
const assistants = params.filter(p => p.role === "assistant");
|
||||
const priorBlocks = assistants[0].content as WireBlock[];
|
||||
expect(priorBlocks.find(b => b.type === "thinking")).toBeUndefined();
|
||||
// Reasoning survives on the wire as text, wrapped in the target's canonical
|
||||
// thinking dialect rather than emitted as a native (signature-bound) block.
|
||||
const text = priorBlocks.find(b => b.type === "text") as WireTextBlock | undefined;
|
||||
expect(text?.text?.trimEnd()).toBe(renderDemotedThinking(target.id, "openai chain-of-thought").trimEnd());
|
||||
const thinking = priorBlocks.find(b => b.type === "thinking") as WireThinkingBlock | undefined;
|
||||
expect(thinking?.thinking).toBe("openai chain-of-thought");
|
||||
expect(thinking?.signature).toBe("");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect";
|
||||
import { convertAnthropicMessages, streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic";
|
||||
import type {
|
||||
AssistantMessage,
|
||||
@@ -234,7 +235,7 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => {
|
||||
const model = makeModel({ provider: "anthropic", baseUrl: "https://api.anthropic.com" });
|
||||
const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model);
|
||||
expect(blocks[0]?.type).toBe("text");
|
||||
expect((blocks[0] as WireTextBlock).text).toBe("internal scratch");
|
||||
expect((blocks[0] as WireTextBlock).text).toBe(renderDemotedThinking(model.id, "internal scratch"));
|
||||
});
|
||||
|
||||
it("treats a missing baseUrl as official Anthropic (resolveAnthropicBaseUrl default)", () => {
|
||||
@@ -245,14 +246,14 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => {
|
||||
const model = makeModel({ provider: "anthropic", baseUrl: "" });
|
||||
const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model);
|
||||
expect(blocks[0]?.type).toBe("text");
|
||||
expect((blocks[0] as WireTextBlock).text).toBe("internal scratch");
|
||||
expect((blocks[0] as WireTextBlock).text).toBe(renderDemotedThinking(model.id, "internal scratch"));
|
||||
});
|
||||
|
||||
it("still degrades unsigned thinking to text for non-reasoning unknown endpoints", () => {
|
||||
const model = makeModel({ reasoning: false, baseUrl: "https://plain.example.com/anthropic" });
|
||||
const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("scratch")], model);
|
||||
expect(blocks[0]?.type).toBe("text");
|
||||
expect((blocks[0] as WireTextBlock).text).toBe("scratch");
|
||||
expect((blocks[0] as WireTextBlock).text).toBe(renderDemotedThinking(model.id, "scratch"));
|
||||
});
|
||||
|
||||
it("keeps thinking → tool_use pairing intact across continuation conversion", () => {
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect";
|
||||
import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import type { AssistantMessage, Model, ModelSpec, ThinkingContent, ToolCall } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
@@ -327,7 +328,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
// Should have set reasoning_content from the thinking text via the openai path.
|
||||
expect(assistant?.reasoning_content).toBe("some reasoning");
|
||||
});
|
||||
it("replays cross-api thinking with stripped signature through reasoning_content", () => {
|
||||
it("demotes cross-api foreign thinking while satisfying tool-call reasoning_content schema", () => {
|
||||
const model = deepseekModel({
|
||||
provider: "opencode-go",
|
||||
baseUrl: "https://opencode.ai/zen/go/v1",
|
||||
@@ -354,8 +355,8 @@ describe("DeepSeek reasoning_content tool-call replay", () => {
|
||||
const messages = convertMessages(model, { messages: [msg] }, compat);
|
||||
const assistant = findOpenAICompletionAssistantWireMessage(messages);
|
||||
expect(assistant).toBeDefined();
|
||||
expect(assistant?.reasoning_content).toBe("Need to preserve cross-api reasoning.");
|
||||
expect(assistant?.content).toBe("");
|
||||
expect(assistant?.reasoning_content).toBe("");
|
||||
expect(assistant?.content).toBe(renderDemotedThinking(model.id, "Need to preserve cross-api reasoning."));
|
||||
});
|
||||
it("falls through to empty-string when thinking block has opaque signature and empty text", () => {
|
||||
const model = deepseekModel({
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect";
|
||||
import { streamGoogle } from "@oh-my-pi/pi-ai/providers/google";
|
||||
import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
@@ -78,4 +79,33 @@ describe("Google provider system prompts", () => {
|
||||
});
|
||||
expect(payload.contents).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("demotes same-model unsigned thinking instead of emitting an unsigned thought part", async () => {
|
||||
const payload = await captureGooglePayload({
|
||||
messages: [
|
||||
{
|
||||
role: "assistant",
|
||||
api: "google-generative-ai",
|
||||
provider: "google",
|
||||
model: model.id,
|
||||
content: [{ type: "thinking", thinking: "unsigned prior thought" }],
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "stop",
|
||||
timestamp: 1,
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
expect(payload.contents[0]).toEqual({
|
||||
role: "model",
|
||||
parts: [{ text: renderDemotedThinking(model.id, "unsigned prior thought") }],
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,32 +1,18 @@
|
||||
/**
|
||||
* Regression guard for cross-API 3p ↔ 3p thinking-block preservation (#3434).
|
||||
* Regression guard for cross-API 3p ↔ 3p thinking-block handling (#3434).
|
||||
*
|
||||
* Mid-session switches between an Anthropic-compatible 3p provider and an
|
||||
* OpenAI-compatible 3p provider on the same vendor (Z.AI Anthropic → Z.AI
|
||||
* OpenAI, Kimi Anthropic → Kimi OpenAI, …) used to demote every prior
|
||||
* `thinking` block to plain text on the cross-API path of `transformMessages`:
|
||||
*
|
||||
* // Cross-API target: keep the existing text-demotion fallback.
|
||||
* return { type: "text", text: sanitized.thinking };
|
||||
*
|
||||
* The next request shipped the reasoning chain as conversation text instead
|
||||
* of structured `reasoning_content`, so the target model lost the prior
|
||||
* reasoning context and the user paid twice — once to generate the thinking
|
||||
* on the source endpoint, once again to re-derive it on the target.
|
||||
*
|
||||
* The fix has two halves:
|
||||
*
|
||||
* 1. `transformMessages` preserves the prior thinking text as a native,
|
||||
* signature-stripped `thinking` block whenever the target encoder can
|
||||
* re-emit it on the wire (today: `openai-completions` reasoning targets
|
||||
* that accept `reasoning_content` as a continuation hint).
|
||||
* 2. The `openai-completions` encoder surfaces those preserved blocks via
|
||||
* `reasoningContentField` even for hosts that don't strictly require
|
||||
* `reasoning_content` — specifically `thinkingFormat: "zai"` targets.
|
||||
* Mid-session switches can replay a prior assistant turn whose native reasoning
|
||||
* slot was authored by a different provider. Live provider probes showed that
|
||||
* unsigned foreign reasoning is only semantically carried by Z.AI-format
|
||||
* OpenAI-compatible targets; schema requirements such as
|
||||
* `requiresReasoningContentForToolCalls` and local llama.cpp cache-prefix replay
|
||||
* do not make the reasoning meaningful. Non-allowlisted targets demote the
|
||||
* reasoning into canonical visible text so the next model can still read it.
|
||||
*
|
||||
* This file pins the wire output for the canonical scenarios.
|
||||
*/
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect";
|
||||
import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import type { AssistantMessage, Message, Model, ModelSpec, UserMessage } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
@@ -98,9 +84,10 @@ function zaiOpenAITarget(): Model<"openai-completions"> {
|
||||
}
|
||||
|
||||
function deepseekReasoningTarget(): Model<"openai-completions"> {
|
||||
// DeepSeek-family reasoning target: requiresReasoningContentForToolCalls is
|
||||
// true here, so the preserved block reaches reasoning_content via the
|
||||
// existing recovery branch. Guards the other half of the fix from regressing.
|
||||
// DeepSeek-family reasoning targets require `reasoning_content` for schema
|
||||
// validity, but measured foreign reasoning in that slot is inert. Cross-API
|
||||
// foreign thinking must demote to text; the encoder may still emit an empty
|
||||
// schema placeholder where required.
|
||||
return buildModel({
|
||||
id: "deepseek-v4-flash",
|
||||
name: "DeepSeek V4 Flash",
|
||||
@@ -119,12 +106,8 @@ function opencodeGoKimiTarget(): Model<"openai-completions"> {
|
||||
// OpenCode Go's reasoning-enabled Kimi. Base compat keeps
|
||||
// `requiresReasoningContentForToolCalls: false` to dodge the
|
||||
// `Extra inputs are not permitted` 400 (#1071); only the resolved
|
||||
// `whenThinking` policy reactivates it (#1484). `convertMessages` threads
|
||||
// that request-time resolved compat into `transformMessages`, so a
|
||||
// thinking-on request preserves the prior reasoning; without the resolved
|
||||
// compat the predicate would read base compat, demote to text, and the
|
||||
// next thinking-on request would 400 with `thinking is enabled but
|
||||
// reasoning_content is missing in assistant tool call message at index N`.
|
||||
// `whenThinking` policy reactivates it (#1484). That schema requirement must
|
||||
// not preserve foreign non-tool-call reasoning as native semantic context.
|
||||
return buildModel({
|
||||
id: "kimi-k2.6",
|
||||
name: "Kimi K2.6",
|
||||
@@ -197,11 +180,7 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => {
|
||||
expect(assistant.reasoning_content).toBe("opaque continuation metadata payload");
|
||||
});
|
||||
|
||||
it("emits reasoning_content on Anthropic 3p → DeepSeek cross-API switch", () => {
|
||||
// DeepSeek-family reasoning targets reach reasoning_content via the
|
||||
// existing `requiresReasoningContentForToolCalls` recovery branch. This
|
||||
// pin guards against a regression in either fix half that would drop
|
||||
// the preserved block before recovery runs.
|
||||
it("demotes Anthropic 3p → DeepSeek cross-API thinking instead of semantic replay", () => {
|
||||
const target = deepseekReasoningTarget();
|
||||
const messages: Message[] = [
|
||||
userMessage("Inspect README"),
|
||||
@@ -214,15 +193,11 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => {
|
||||
expect(assistant).toBeDefined();
|
||||
if (!assistant) throw new Error("assistant message missing");
|
||||
|
||||
expect(assistant.reasoning_content).toBe("Read README and answer.");
|
||||
expect(assistant.reasoning_content).toBe("");
|
||||
expect(assistant.content).toBe(`${renderDemotedThinking(target.id, "Read README and answer.")}Done.`);
|
||||
});
|
||||
|
||||
it("demotes thinking to text when the target cannot replay reasoning_content", () => {
|
||||
// Anthropic 3p → official OpenAI non-reasoning model: the encoder
|
||||
// cannot emit `reasoning_content` here (the field would be ignored and
|
||||
// strict OpenAI-compat shims would reject it). Reasoning must survive
|
||||
// at minimum as visible conversation text so the next turn still sees
|
||||
// the prior plan.
|
||||
it("demotes thinking to canonical text when the target cannot replay it semantically", () => {
|
||||
const target = openAIGpt4oTarget();
|
||||
const messages: Message[] = [
|
||||
userMessage("Plan it."),
|
||||
@@ -236,23 +211,10 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => {
|
||||
if (!assistant) throw new Error("assistant message missing");
|
||||
|
||||
expect(assistant.reasoning_content).toBeUndefined();
|
||||
const content = assistant.content;
|
||||
expect(typeof content).toBe("string");
|
||||
if (typeof content !== "string") throw new Error("content not a string");
|
||||
expect(content).toContain("Explore the repo, then patch it.");
|
||||
expect(content).toContain("Done.");
|
||||
expect(assistant.content).toBe(`${renderDemotedThinking(target.id, "Explore the repo, then patch it.")}Done.`);
|
||||
});
|
||||
|
||||
it("preserves cross-API thinking for OpenCode reasoning targets that gate replay via compat.whenThinking", () => {
|
||||
// OpenCode (`opencode-go`, `opencode-zen`) reasoning models keep
|
||||
// `requiresReasoningContentForToolCalls: false` on the base compat
|
||||
// (dodges the thinking-off `Extra inputs are not permitted` 400 — #1071)
|
||||
// and reactivate the flag on `compat.whenThinking` for thinking-engaged
|
||||
// requests (dodges the `thinking is enabled but reasoning_content is
|
||||
// missing` 400 — #1484). The cross-API preservation predicate must run
|
||||
// against the resolved compat that `convertMessages` threads in (the
|
||||
// `whenThinking` view here); reading base compat would demote the prior
|
||||
// thinking to text and re-trigger #1484 on the next thinking-on request.
|
||||
it("demotes cross-API thinking for OpenCode reasoning targets with whenThinking schema", () => {
|
||||
const target = opencodeGoKimiTarget();
|
||||
const messages: Message[] = [
|
||||
userMessage("Plan it."),
|
||||
@@ -261,9 +223,7 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => {
|
||||
];
|
||||
|
||||
// Resolve the thinking-engaged compat the way `streamOpenAICompletions`
|
||||
// does for a request with reasoning effort set, then hand it to
|
||||
// `convertMessages` directly so the test exercises the same encoder
|
||||
// configuration the live wire would.
|
||||
// does for a request with reasoning effort set.
|
||||
const compat = target.compat.whenThinking ?? target.compat;
|
||||
expect(compat.requiresReasoningContentForToolCalls).toBe(true);
|
||||
|
||||
@@ -272,19 +232,11 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => {
|
||||
expect(assistant).toBeDefined();
|
||||
if (!assistant) throw new Error("assistant message missing");
|
||||
|
||||
expect(assistant.reasoning_content).toBe("Read README and answer.");
|
||||
expect(assistant.reasoning_content).toBeUndefined();
|
||||
expect(assistant.content).toBe(`${renderDemotedThinking(target.id, "Read README and answer.")}Done.`);
|
||||
});
|
||||
|
||||
it("demotes prior thinking to content when the OpenCode base compat (thinking off) cannot surface reasoning_content", () => {
|
||||
// Companion of the prior test: same OpenCode target, but the request
|
||||
// runs against the BASE compat (thinking disabled, the path that bars
|
||||
// `reasoning_content` per #1071). The cross-API preservation predicate
|
||||
// reads this resolved base compat — which neither requires
|
||||
// `reasoning_content` nor is a Z.AI-format host — so it preserves no
|
||||
// native thinking block the encoder couldn't surface; the cross-API path
|
||||
// instead text-demotes the prior reasoning into visible content. The
|
||||
// reasoning still survives as conversation context, with no
|
||||
// `reasoning_content` on the wire and no #1071 regression.
|
||||
it("demotes prior thinking to content when the OpenCode base compat runs with thinking off", () => {
|
||||
const target = opencodeGoKimiTarget();
|
||||
const compat = target.compat;
|
||||
expect(compat.requiresReasoningContentForToolCalls).toBe(false);
|
||||
@@ -301,11 +253,7 @@ describe("cross-API thinking-block preservation (#3433/#3434)", () => {
|
||||
if (!assistant) throw new Error("assistant message missing");
|
||||
|
||||
expect(assistant.reasoning_content).toBeUndefined();
|
||||
const content = assistant.content;
|
||||
expect(typeof content).toBe("string");
|
||||
if (typeof content !== "string") throw new Error("content not a string");
|
||||
expect(content).toContain("Read README and answer.");
|
||||
expect(content).toContain("Done.");
|
||||
expect(assistant.content).toBe(`${renderDemotedThinking(target.id, "Read README and answer.")}Done.`);
|
||||
});
|
||||
|
||||
it("does not promote markup-healed same-model thinking into visible content", () => {
|
||||
|
||||
@@ -40,6 +40,7 @@
|
||||
* This file pins the wire output across the relevant axes.
|
||||
*/
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect";
|
||||
import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import {
|
||||
applyChatCompletionsReasoningParams,
|
||||
@@ -239,17 +240,11 @@ describe("llama.cpp warm-prefix preservation (#3528)", () => {
|
||||
expect(assistant.reasoning_content).toBe("Trace the call graph through service.ts and the registry.");
|
||||
});
|
||||
|
||||
it("preserves cross-API thinking into a discovered local target (reasoning: false on the spec)", () => {
|
||||
// Cross-API/model switch into a discovered llama.cpp target: an
|
||||
// Anthropic-source thinking block (opaque continuation signature, foreign
|
||||
// to the openai-completions wire) must NOT be demoted to text just because
|
||||
// the discovery path stamped `reasoning: false` on the spec. The
|
||||
// `replayReasoningContent` flag has to bypass the `model.reasoning` gate
|
||||
// in `transform-messages.ts` for the cross-API replay branch to fire, so
|
||||
// the encoder receives a signature-stripped thinking block to surface as
|
||||
// `reasoning_content` on the wire. Without the bypass the prior turn's
|
||||
// reasoning rides as plain conversation text and the local server still
|
||||
// loses the cache-stable `<think>` prefix.
|
||||
it("demotes cross-API thinking into a discovered local target", () => {
|
||||
// `replayReasoningContent` is a same-wire cache concern. It keeps
|
||||
// llama.cpp turns cache-stable when the prior assistant already emitted an
|
||||
// OpenAI-compatible reasoning field, but it must not preserve foreign
|
||||
// Anthropic reasoning as native semantic context.
|
||||
const target = llamaCppQwenModel({ reasoning: false });
|
||||
const anthropicSourceTurn: AssistantMessage = {
|
||||
role: "assistant",
|
||||
@@ -287,10 +282,10 @@ describe("llama.cpp warm-prefix preservation (#3528)", () => {
|
||||
target.compat,
|
||||
);
|
||||
const found = findAssistantMessage(wire) as Record<string, unknown> | undefined;
|
||||
expect(found?.reasoning_content).toBe("Cross-vendor reasoning chain that must survive the switch.");
|
||||
expect(found?.content).toBe("Switched-in answer.");
|
||||
// The Anthropic continuation signature is bound to the source wire and
|
||||
// must NEVER leak as a stray field name on the openai-completions target.
|
||||
expect(found?.reasoning_content).toBeUndefined();
|
||||
expect(found?.content).toBe(
|
||||
`${renderDemotedThinking(target.id, "Cross-vendor reasoning chain that must survive the switch.")}Switched-in answer.`,
|
||||
);
|
||||
expect("EvAnthropicOpaqueContinuationBlob==" in (found ?? {})).toBe(false);
|
||||
});
|
||||
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { renderDemotedThinking } from "@oh-my-pi/pi-ai/dialect";
|
||||
import {
|
||||
applyOpenRouterRoutingVariant,
|
||||
convertMessages,
|
||||
@@ -273,7 +274,7 @@ describe("openai-completions compatibility", () => {
|
||||
// Regression: thinking+text replay used to call `.unshift` on the string
|
||||
// content set above (TypeError). Both blocks must survive as one string.
|
||||
expect(typeof assistant.content).toBe("string");
|
||||
expect(assistant.content).toBe("chain of thought\n\nfinal answer");
|
||||
expect(assistant.content).toBe(`${renderDemotedThinking(model.id, "chain of thought")}final answer`);
|
||||
});
|
||||
|
||||
it("emits thinking-only assistant content as a plain string when requiresThinkingAsText is set", () => {
|
||||
@@ -309,7 +310,7 @@ describe("openai-completions compatibility", () => {
|
||||
const assistant = messages.find(message => message.role === "assistant");
|
||||
expect(assistant).toBeDefined();
|
||||
if (assistant?.role !== "assistant") throw new Error("assistant message missing");
|
||||
expect(assistant.content).toBe("only thoughts");
|
||||
expect(assistant.content).toBe(renderDemotedThinking(model.id, "only thoughts"));
|
||||
});
|
||||
|
||||
it("preserves multiple system prompts as leading system messages for chat completions", () => {
|
||||
@@ -1227,7 +1228,7 @@ describe("kimi model detection via detectCompat", () => {
|
||||
expect(assistant?.reasoning).toBeUndefined();
|
||||
});
|
||||
|
||||
it("uses thinking-enabled compat when replaying cross-api reasoning on kimi opencode-go", async () => {
|
||||
it("demotes cross-api reasoning while keeping thinking-enabled tool-call schema on kimi opencode-go", async () => {
|
||||
const model = kimiOpenCodeModel("kimi-k2.6");
|
||||
expect(model.compat.requiresReasoningContentForToolCalls).toBe(false);
|
||||
const priorAssistant: AssistantMessage = {
|
||||
@@ -1290,8 +1291,8 @@ describe("kimi model detection via detectCompat", () => {
|
||||
const payload = (await promise) as { messages: Array<Record<string, unknown>> };
|
||||
const assistant = payload.messages.find(m => m.role === "assistant");
|
||||
expect(assistant).toBeDefined();
|
||||
expect(assistant?.content).toBe(".");
|
||||
expect(assistant?.reasoning_content).toBe("Need to preserve cross-api reasoning.");
|
||||
expect(assistant?.content).toBe(renderDemotedThinking(model.id, "Need to preserve cross-api reasoning."));
|
||||
expect(assistant?.reasoning_content).toBe("");
|
||||
expect(assistant?.reasoning).toBeUndefined();
|
||||
expect(assistant?.reasoning_text).toBeUndefined();
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user