fix(ai): formatted demoted thinking as italic prose for Claude Fable

- Avoid using `.
This commit is contained in:
can1357
2026-07-02 02:30:43 +02:00
parent e8a8c2402c
commit cd840303a1
4 changed files with 121 additions and 12 deletions
+1
View File
@@ -15,6 +15,7 @@
- Fixed an issue where broker usage fetch failures were not cached, causing sequential ranking passes to repeatedly hit the broker when it is down.
- Fixed Xiaomi MiMo API key validation to use the supported `mimo-v2.5` model.
- Fixed certificate verification errors for custom gateways behind private CA bundles by applying `NODE_EXTRA_CA_CERTS` to all provider fetches (including OpenAI-compatible, Codex, Ollama, Azure, and Google).
- Fixed Claude Fable demoted-thinking replay to use markdown-italic assistant prose instead of `<thinking>` tags, avoiding reasoning-extraction-shaped context after model switches.
### Fixed
- Fixed OpenAI Responses replay emitting locally rebuilt assistant item IDs without their required reasoning items, preventing `function_call` / `message` replay 400s from poisoned history. ([#4173](https://github.com/can1357/oh-my-pi/issues/4173))
+13 -8
View File
@@ -1,6 +1,8 @@
import { preferredDialect } from "@oh-my-pi/pi-catalog/identity";
import { bareModelId, preferredDialect } from "@oh-my-pi/pi-catalog/identity";
import { getDialectDefinition } from "./factory";
const CLAUDE_FABLE_ID = /(?:^|[./])claude[-.]fable(?:[-.]|$)/i;
/**
* Wrap a prior-turn reasoning string for demotion into native conversation
* history — the cross-provider / cross-model case where the target cannot replay
@@ -8,13 +10,14 @@ import { getDialectDefinition } from "./factory";
* replayed unsigned `thought` part is schema-accepted but silently discarded —
* neither recalled nor influencing generation).
*
* The reasoning is rendered in the TARGET model's canonical inline thinking
* delimiters so it reads as reasoning in that model's own idiom instead of bare
* prose the model might continue. Harmony and Gemma are the exception: their
* `renderThinking` emits chat-template control tokens (`<|channel|>analysis`,
* `<|channel>thought`) that must not appear inside a structured native message,
* so they fall back to a plain `<think>` block. Every other dialect's thinking
* form is inline-safe XML tags or a markdown fence.
* Fable is the exception: replaying prior reasoning inside `<thinking>` /
* `antml:thinking`-style assistant text is treated as a reasoning-extraction
* attempt and can train the next turn to leak thoughts, so Fable receives the
* reasoning as markdown-italic assistant prose instead. Harmony and Gemma are
* also exceptions: their `renderThinking` emits chat-template control tokens
* (`<|channel|>analysis`, `<|channel>thought`) that must not appear inside a
* structured native message, so they fall back to a plain `<think>` block. Every
* other dialect's thinking form is inline-safe XML tags or a markdown fence.
*
* The result ends with a trailing newline so the block stays separated from the
* turn's reply text when the wire encoder concatenates parts.
@@ -25,7 +28,9 @@ import { getDialectDefinition } from "./factory";
export function renderDemotedThinking(modelId: string, text: string): string {
if (!text) return "";
text = text.toWellFormed();
const canonicalId = bareModelId(modelId);
const dialect = preferredDialect(modelId);
if (CLAUDE_FABLE_ID.test(canonicalId)) return `_Hmm. ${text}_\n`;
if (dialect === "harmony" || dialect === "gemma") return `<think>\n${text}\n</think>\n`;
return `${getDialectDefinition(dialect).renderThinking(text)}\n`;
}
@@ -265,6 +265,50 @@ describe("Anthropic prior-turn thinking preservation (#2257, #2265)", () => {
expect(priorBlocks.find(b => b.type === "redacted_thinking")).toBeUndefined();
});
it("demotes invalid official Anthropic prior signatures to Fable markdown prose after a model switch", () => {
// official Anthropic → official Fable, with the signed turn no longer
// latest. The source signature is bound to the issuing Anthropic model,
// so replaying it after the switch must not emit native thinking or
// Anthropic/Kimi-style thinking tags that Fable treats as visible text.
const target = makeAnthropicModel({
provider: "anthropic",
id: "claude-fable-5",
name: "Claude Fable 5",
baseUrl: "https://api.anthropic.com",
});
const reasoning = "Need to preserve the plan while switching models.";
const messages: Message[] = [
makeUser("Read the project notes"),
makeAssistant(
[
{ type: "thinking", thinking: reasoning, thinkingSignature: "sig_sonnet" },
{ type: "toolCall", id: "toolu_prior", name: "read", arguments: { path: "NOTES.md" } },
],
{ provider: "anthropic", model: "claude-sonnet-4-6" },
),
toolResult("toolu_prior", "notes body"),
makeAssistant([{ type: "text", text: "I found the relevant notes." }], {
provider: "anthropic",
model: "claude-fable-5",
stopReason: "stop",
}),
makeUser("Continue from those notes."),
];
const params = convertAnthropicMessages(messages, target, false);
const assistants = params.filter(p => p.role === "assistant");
expect(assistants).toHaveLength(2);
const priorBlocks = assistants[0].content as WireBlock[];
const text = priorBlocks.find(b => b.type === "text") as WireTextBlock | undefined;
expect(text?.text).toBe(renderDemotedThinking("claude-fable-5", reasoning));
expect(text?.text).toBe(`_Hmm. ${reasoning}_\n`);
expect(text?.text).not.toContain("<thinking>");
expect(text?.text).not.toContain("</thinking>");
expect(text?.text).not.toContain("<think>");
expect(text?.text).not.toContain("</think>");
expect(priorBlocks.find(b => b.type === "thinking")).toBeUndefined();
});
it("strips official Anthropic source signatures on cross-model replay to a 3p target", () => {
// official Anthropic → 3p. Anthropic's signature is bound to the
// issuing model+session, so the 3p target cannot reverify or
@@ -11,10 +11,10 @@ import { buildModel } from "@oh-my-pi/pi-catalog/build";
* silently discards unsigned thought content (a replayed `thought:true` part is
* neither recalled nor influences generation). `transformMessages` therefore
* demotes the reasoning to a `text` block so it survives as conversation
* context, wrapping it in the TARGET model's own canonical thinking-block
* dialect (e.g. a ```thinking fence for Gemini) so it reads as reasoning in
* that model's idiom instead of bare prose the model might mimic.
*
* context, usually wrapping it in the TARGET model's own canonical
* thinking-block dialect (e.g. a ```thinking fence for Gemini). Claude Fable is
* the exception: it receives markdown-italic assistant prose so replayed
* reasoning does not look like an extraction request it should continue.
* Same-model continuations keep the native `thinking` block untouched.
*/
const REASONING = "The user wants the Paris weather; I will call get_weather with city=Paris.";
@@ -62,6 +62,20 @@ function anthropicThinkingTurn(): AssistantMessage {
};
}
/** A prior assistant turn authored by a Gemini model: foreign thinking + a text reply. */
function geminiThinkingTurn(): AssistantMessage {
return {
...anthropicThinkingTurn(),
api: "google-generative-ai",
provider: "google",
model: "gemini-3-pro-preview",
content: [
{ type: "thinking", thinking: REASONING, thinkingSignature: "google-sig" },
{ type: "text", text: "Checking the forecast." },
],
};
}
function transformedAssistant(messages: Message[], target: Model<Api>): AssistantMessage {
const out = transformMessages(messages, target);
const assistant = out.find((m): m is AssistantMessage => m.role === "assistant");
@@ -114,6 +128,51 @@ describe("transformMessages cross-provider thinking demotion → canonical diale
expect(text).toContain(REASONING);
});
it("renders demoted foreign reasoning for Claude Fable as markdown italic assistant text", () => {
const fable = makeModel("anthropic-messages", "anthropic", "claude-fable-5");
const assistant = transformedAssistant([user("weather in Paris?"), geminiThinkingTurn()], fable);
expect(assistant.content.some(b => b.type === "thinking")).toBe(false);
const first = assistant.content[0];
expect(first?.type).toBe("text");
const text = first && first.type === "text" ? first.text : "";
expect(text).toBe(`_Hmm. ${REASONING}_\n`);
expect(text).not.toContain("<thinking>");
expect(text).not.toContain("</thinking>");
expect(text).not.toContain("<think>");
expect(text).not.toContain("</think>");
const reply = assistant.content[1];
expect(reply?.type).toBe("text");
expect(reply && reply.type === "text" ? reply.text : "").toBe("Checking the forecast.");
});
it("keeps canonical Anthropic thinking tags for non-Fable Anthropic targets", () => {
const targets = [
{ name: "Claude Opus", id: "claude-opus-4-8" },
{ name: "Claude Mythos", id: "claude-mythos-5" },
] as const;
for (const target of targets) {
const model = makeModel("anthropic-messages", "anthropic", target.id);
const assistant = transformedAssistant(
[user(`weather in Paris for ${target.name}?`), geminiThinkingTurn()],
model,
);
expect(assistant.content.some(b => b.type === "thinking")).toBe(false);
const first = assistant.content[0];
expect(first?.type).toBe("text");
const text = first && first.type === "text" ? first.text : "";
expect(text).toBe(`${getDialectDefinition("anthropic").renderThinking(REASONING)}\n`);
expect(text).toContain("<thinking>");
expect(text).toContain("</thinking>");
expect(text).not.toContain("_Hmm.");
}
});
it("keeps the native thinking block for a same-provider/same-model continuation", () => {
const gemini = makeModel("google-generative-ai", "google", "gemini-3-pro-preview");
const sameModelTurn: AssistantMessage = {