From cd840303a1fbb03da359debbb773f2c83bbd8604 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 2 Jul 2026 02:30:43 +0200 Subject: [PATCH] fix(ai): formatted demoted thinking as italic prose for Claude Fable - Avoid using `. --- packages/ai/CHANGELOG.md | 1 + packages/ai/src/dialect/demotion.ts | 21 +++--- .../anthropic-prior-turn-thinking.test.ts | 44 ++++++++++++ ...ransform-messages-thinking-dialect.test.ts | 67 +++++++++++++++++-- 4 files changed, 121 insertions(+), 12 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 847a2ec45..03bad4b4f 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -15,6 +15,7 @@ - Fixed an issue where broker usage fetch failures were not cached, causing sequential ranking passes to repeatedly hit the broker when it is down. - Fixed Xiaomi MiMo API key validation to use the supported `mimo-v2.5` model. - Fixed certificate verification errors for custom gateways behind private CA bundles by applying `NODE_EXTRA_CA_CERTS` to all provider fetches (including OpenAI-compatible, Codex, Ollama, Azure, and Google). +- Fixed Claude Fable demoted-thinking replay to use markdown-italic assistant prose instead of `` tags, avoiding reasoning-extraction-shaped context after model switches. ### Fixed - Fixed OpenAI Responses replay emitting locally rebuilt assistant item IDs without their required reasoning items, preventing `function_call` / `message` replay 400s from poisoned history. ([#4173](https://github.com/can1357/oh-my-pi/issues/4173)) diff --git a/packages/ai/src/dialect/demotion.ts b/packages/ai/src/dialect/demotion.ts index b554b2185..f89b2218f 100644 --- a/packages/ai/src/dialect/demotion.ts +++ b/packages/ai/src/dialect/demotion.ts @@ -1,6 +1,8 @@ -import { preferredDialect } from "@oh-my-pi/pi-catalog/identity"; +import { bareModelId, preferredDialect } from "@oh-my-pi/pi-catalog/identity"; import { getDialectDefinition } from "./factory"; +const CLAUDE_FABLE_ID = /(?:^|[./])claude[-.]fable(?:[-.]|$)/i; + /** * Wrap a prior-turn reasoning string for demotion into native conversation * history — the cross-provider / cross-model case where the target cannot replay @@ -8,13 +10,14 @@ import { getDialectDefinition } from "./factory"; * replayed unsigned `thought` part is schema-accepted but silently discarded — * neither recalled nor influencing generation). * - * The reasoning is rendered in the TARGET model's canonical inline thinking - * delimiters so it reads as reasoning in that model's own idiom instead of bare - * prose the model might continue. Harmony and Gemma are the exception: their - * `renderThinking` emits chat-template control tokens (`<|channel|>analysis`, - * `<|channel>thought`) that must not appear inside a structured native message, - * so they fall back to a plain `` block. Every other dialect's thinking - * form is inline-safe XML tags or a markdown fence. + * Fable is the exception: replaying prior reasoning inside `` / + * `antml:thinking`-style assistant text is treated as a reasoning-extraction + * attempt and can train the next turn to leak thoughts, so Fable receives the + * reasoning as markdown-italic assistant prose instead. Harmony and Gemma are + * also exceptions: their `renderThinking` emits chat-template control tokens + * (`<|channel|>analysis`, `<|channel>thought`) that must not appear inside a + * structured native message, so they fall back to a plain `` block. Every + * other dialect's thinking form is inline-safe XML tags or a markdown fence. * * The result ends with a trailing newline so the block stays separated from the * turn's reply text when the wire encoder concatenates parts. @@ -25,7 +28,9 @@ import { getDialectDefinition } from "./factory"; export function renderDemotedThinking(modelId: string, text: string): string { if (!text) return ""; text = text.toWellFormed(); + const canonicalId = bareModelId(modelId); const dialect = preferredDialect(modelId); + if (CLAUDE_FABLE_ID.test(canonicalId)) return `_Hmm. ${text}_\n`; if (dialect === "harmony" || dialect === "gemma") return `\n${text}\n\n`; return `${getDialectDefinition(dialect).renderThinking(text)}\n`; } diff --git a/packages/ai/test/anthropic-prior-turn-thinking.test.ts b/packages/ai/test/anthropic-prior-turn-thinking.test.ts index 8b242a193..abe54af25 100644 --- a/packages/ai/test/anthropic-prior-turn-thinking.test.ts +++ b/packages/ai/test/anthropic-prior-turn-thinking.test.ts @@ -265,6 +265,50 @@ describe("Anthropic prior-turn thinking preservation (#2257, #2265)", () => { expect(priorBlocks.find(b => b.type === "redacted_thinking")).toBeUndefined(); }); + it("demotes invalid official Anthropic prior signatures to Fable markdown prose after a model switch", () => { + // official Anthropic → official Fable, with the signed turn no longer + // latest. The source signature is bound to the issuing Anthropic model, + // so replaying it after the switch must not emit native thinking or + // Anthropic/Kimi-style thinking tags that Fable treats as visible text. + const target = makeAnthropicModel({ + provider: "anthropic", + id: "claude-fable-5", + name: "Claude Fable 5", + baseUrl: "https://api.anthropic.com", + }); + const reasoning = "Need to preserve the plan while switching models."; + const messages: Message[] = [ + makeUser("Read the project notes"), + makeAssistant( + [ + { type: "thinking", thinking: reasoning, thinkingSignature: "sig_sonnet" }, + { type: "toolCall", id: "toolu_prior", name: "read", arguments: { path: "NOTES.md" } }, + ], + { provider: "anthropic", model: "claude-sonnet-4-6" }, + ), + toolResult("toolu_prior", "notes body"), + makeAssistant([{ type: "text", text: "I found the relevant notes." }], { + provider: "anthropic", + model: "claude-fable-5", + stopReason: "stop", + }), + makeUser("Continue from those notes."), + ]; + + const params = convertAnthropicMessages(messages, target, false); + const assistants = params.filter(p => p.role === "assistant"); + expect(assistants).toHaveLength(2); + const priorBlocks = assistants[0].content as WireBlock[]; + const text = priorBlocks.find(b => b.type === "text") as WireTextBlock | undefined; + expect(text?.text).toBe(renderDemotedThinking("claude-fable-5", reasoning)); + expect(text?.text).toBe(`_Hmm. ${reasoning}_\n`); + expect(text?.text).not.toContain(""); + expect(text?.text).not.toContain(""); + expect(text?.text).not.toContain(""); + expect(text?.text).not.toContain(""); + expect(priorBlocks.find(b => b.type === "thinking")).toBeUndefined(); + }); + it("strips official Anthropic source signatures on cross-model replay to a 3p target", () => { // official Anthropic → 3p. Anthropic's signature is bound to the // issuing model+session, so the 3p target cannot reverify or diff --git a/packages/ai/test/transform-messages-thinking-dialect.test.ts b/packages/ai/test/transform-messages-thinking-dialect.test.ts index ea6b6f65e..ef5dca300 100644 --- a/packages/ai/test/transform-messages-thinking-dialect.test.ts +++ b/packages/ai/test/transform-messages-thinking-dialect.test.ts @@ -11,10 +11,10 @@ import { buildModel } from "@oh-my-pi/pi-catalog/build"; * silently discards unsigned thought content (a replayed `thought:true` part is * neither recalled nor influences generation). `transformMessages` therefore * demotes the reasoning to a `text` block so it survives as conversation - * context, wrapping it in the TARGET model's own canonical thinking-block - * dialect (e.g. a ```thinking fence for Gemini) so it reads as reasoning in - * that model's idiom instead of bare prose the model might mimic. - * + * context, usually wrapping it in the TARGET model's own canonical + * thinking-block dialect (e.g. a ```thinking fence for Gemini). Claude Fable is + * the exception: it receives markdown-italic assistant prose so replayed + * reasoning does not look like an extraction request it should continue. * Same-model continuations keep the native `thinking` block untouched. */ const REASONING = "The user wants the Paris weather; I will call get_weather with city=Paris."; @@ -62,6 +62,20 @@ function anthropicThinkingTurn(): AssistantMessage { }; } +/** A prior assistant turn authored by a Gemini model: foreign thinking + a text reply. */ +function geminiThinkingTurn(): AssistantMessage { + return { + ...anthropicThinkingTurn(), + api: "google-generative-ai", + provider: "google", + model: "gemini-3-pro-preview", + content: [ + { type: "thinking", thinking: REASONING, thinkingSignature: "google-sig" }, + { type: "text", text: "Checking the forecast." }, + ], + }; +} + function transformedAssistant(messages: Message[], target: Model): AssistantMessage { const out = transformMessages(messages, target); const assistant = out.find((m): m is AssistantMessage => m.role === "assistant"); @@ -114,6 +128,51 @@ describe("transformMessages cross-provider thinking demotion → canonical diale expect(text).toContain(REASONING); }); + it("renders demoted foreign reasoning for Claude Fable as markdown italic assistant text", () => { + const fable = makeModel("anthropic-messages", "anthropic", "claude-fable-5"); + const assistant = transformedAssistant([user("weather in Paris?"), geminiThinkingTurn()], fable); + + expect(assistant.content.some(b => b.type === "thinking")).toBe(false); + + const first = assistant.content[0]; + expect(first?.type).toBe("text"); + const text = first && first.type === "text" ? first.text : ""; + expect(text).toBe(`_Hmm. ${REASONING}_\n`); + expect(text).not.toContain(""); + expect(text).not.toContain(""); + expect(text).not.toContain(""); + expect(text).not.toContain(""); + + const reply = assistant.content[1]; + expect(reply?.type).toBe("text"); + expect(reply && reply.type === "text" ? reply.text : "").toBe("Checking the forecast."); + }); + + it("keeps canonical Anthropic thinking tags for non-Fable Anthropic targets", () => { + const targets = [ + { name: "Claude Opus", id: "claude-opus-4-8" }, + { name: "Claude Mythos", id: "claude-mythos-5" }, + ] as const; + + for (const target of targets) { + const model = makeModel("anthropic-messages", "anthropic", target.id); + const assistant = transformedAssistant( + [user(`weather in Paris for ${target.name}?`), geminiThinkingTurn()], + model, + ); + + expect(assistant.content.some(b => b.type === "thinking")).toBe(false); + + const first = assistant.content[0]; + expect(first?.type).toBe("text"); + const text = first && first.type === "text" ? first.text : ""; + expect(text).toBe(`${getDialectDefinition("anthropic").renderThinking(REASONING)}\n`); + expect(text).toContain(""); + expect(text).toContain(""); + expect(text).not.toContain("_Hmm."); + } + }); + it("keeps the native thinking block for a same-provider/same-model continuation", () => { const gemini = makeModel("google-generative-ai", "google", "gemini-3-pro-preview"); const sameModelTurn: AssistantMessage = {