diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 5add5603b..34e94678e 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed compaction/summarization serializing prior assistant reasoning back to Claude as text (rendered verbatim inside `` tags for the `anthropic` dialect), which tripped Anthropic's `reasoning_extraction` refusal and blocked compaction on Fable 5 sessions; `serializeConversation` now drops `thinking` blocks for Anthropic-dialect summary targets while other dialects (e.g. Harmony) keep their native reasoning ([#6093](https://github.com/can1357/oh-my-pi/issues/6093)). + ## [17.0.5] - 2026-07-18 ### Added diff --git a/packages/agent/src/compaction/utils.ts b/packages/agent/src/compaction/utils.ts index 286f4e7f9..e38290bdd 100644 --- a/packages/agent/src/compaction/utils.ts +++ b/packages/agent/src/compaction/utils.ts @@ -234,10 +234,21 @@ export function serializeConversation(messages: Message[], dialect?: Dialect): s } } if (dialect) { + // Claude's classifier refuses inputs that reproduce the model's own + // reasoning as text ("reasoning_extraction"), and the anthropic dialect + // otherwise renders thinking verbatim inside tags. Reasoning is + // ephemeral and low-signal for a summary, so drop it from Anthropic-target + // summary input. Other dialects (e.g. Harmony) carry reasoning natively in + // their transcript format and keep it. + const dropThinking = dialect === "anthropic"; const processed: Message[] = []; for (const msg of messages) { if (msg.role === "assistant") { - const content = msg.content.filter(block => block.type !== "toolCall" || !uselessCallIds.has(block.id)); + const content = msg.content.filter( + block => + (block.type !== "toolCall" || !uselessCallIds.has(block.id)) && + (!dropThinking || block.type !== "thinking"), + ); if (content.length > 0) processed.push(content.length === msg.content.length ? msg : { ...msg, content }); continue; } diff --git a/packages/agent/test/serialize-conversation.test.ts b/packages/agent/test/serialize-conversation.test.ts index 0c7b119e4..04928e7b6 100644 --- a/packages/agent/test/serialize-conversation.test.ts +++ b/packages/agent/test/serialize-conversation.test.ts @@ -130,4 +130,39 @@ describe("serializeConversation — useless pairs", () => { expect(out).toBe(""); }); + + test("strips assistant reasoning from Anthropic-dialect summary input but keeps text and tool calls", () => { + const reasoning = "PRIVATE chain of thought that must not be replayed to Claude"; + const out = serializeConversation( + [ + assistantMessage([ + { type: "thinking", thinking: reasoning }, + { type: "text", text: "The visible answer." }, + { type: "toolCall", id: "c1", name: "search", arguments: { pattern: "delta" } }, + ]), + ], + "anthropic", + ); + + expect(out).not.toContain(reasoning); + expect(out).not.toContain(""); + expect(out).toContain("The visible answer."); + expect(out).toContain(""); + }); + + test("keeps assistant reasoning for non-Anthropic dialects", () => { + const reasoning = "reasoning kept for the XML transcript"; + const out = serializeConversation( + [ + assistantMessage([ + { type: "thinking", thinking: reasoning }, + { type: "text", text: "answer" }, + ]), + ], + "xml", + ); + + expect(out).toContain(reasoning); + expect(out).toContain(""); + }); }); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6f32dc885..fa41be92e 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed snapcompact archiving reproduced assistant reasoning (`¶think:` sections) into frames replayed to the model on every subsequent request, wedging Fable 5 sessions on `reasoning_extraction` refusals; snapcompact serialization now excludes reasoning when the session model uses the Anthropic dialect ([#6093](https://github.com/can1357/oh-my-pi/issues/6093)). + ## [17.0.5] - 2026-07-18 ### Added diff --git a/packages/coding-agent/src/prompts/system/workflow-notice.md b/packages/coding-agent/src/prompts/system/workflow-notice.md index d15ad9127..09a849bef 100644 --- a/packages/coding-agent/src/prompts/system/workflow-notice.md +++ b/packages/coding-agent/src/prompts/system/workflow-notice.md @@ -47,7 +47,7 @@ For independent per-item chains (review → verify, fetch → extract → score) schema: FINDINGS_SCHEMA, }); return await parallel(found.findings.map((f) => async () => ({ - ...f, + …f, verdict: await agent( `Refute if you can (default refuted when unsure): ${f.title}`, { label: `verify:${f.file}`, schema: VERDICT_SCHEMA }, @@ -57,8 +57,6 @@ For independent per-item chains (review → verify, fetch → extract → score) phase("Review"); const results = await parallel(DIMENSIONS.map((d) => async () => reviewAndVerify(d))); const confirmed = results.flat().filter((f) => f.verdict.is_real); - - Reach for `pipeline()` only when a stage genuinely needs ALL of the previous stage first — dedup/merge across the whole set, early-exit on zero, or "compare against the other findings" — because its inter-stage barrier makes every item wait for the slowest peer: **Python (`eval`, Python backend):** @@ -80,8 +78,6 @@ Reach for `pipeline()` only when a stage genuinely needs ALL of the previous sta const verdicts = await parallel(findings.map((f) => async () => await agent(verifyPrompt(f), { schema: VERDICT_SCHEMA }), )); - - Use ordinary code between calls to flatten/map/filter; don't add a barrier just for that. Nested `parallel()` pools each cap independently, so keep total fan-out sane. diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index e39ca813d..fae7de0ec 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -127,6 +127,7 @@ import { toolWireSchema } from "@oh-my-pi/pi-ai/utils/schema"; import { GeminiHeaderRunDetector, isGeminiThinkingModel } from "@oh-my-pi/pi-ai/utils/thinking-loop"; import { type RepeatedToolCallDetection, ToolCallLoopGuard } from "@oh-my-pi/pi-ai/utils/tool-call-loop-guard"; import { isFireworksFastModelId, toFireworksBaseModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id"; +import { preferredDialect } from "@oh-my-pi/pi-catalog/identity"; import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking"; import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models"; import { MacOSPowerAssertion } from "@oh-my-pi/pi-natives"; @@ -10962,6 +10963,11 @@ export class AgentSession { let snapcompactReady = wantsSnapcompact; const snapcompactShapeSetting = this.settings.get("snapcompact.shape"); let snapcompactShape: snapcompact.Shape | undefined; + // Claude refuses inputs that reproduce its own reasoning as text + // ("reasoning_extraction"), and the snapcompact archive is replayed as + // text into every later request; drop `¶think:` sections for + // Anthropic-dialect targets (issue #6093). + const snapcompactIncludeThinking = preferredDialect(this.model.id) !== "anthropic"; if (wantsSnapcompact && !this.model.input.includes("image")) { if (explicitSnapcompact) { this.emitNotice( @@ -10980,6 +10986,7 @@ export class AgentSession { } else if (snapcompactReady) { const text = snapcompact.serializeConversation( convertToLlm(preparation.messagesToSummarize.concat(preparation.turnPrefixMessages)), + { includeThinking: snapcompactIncludeThinking }, ); const probeText = snapcompact.renderabilityProbeText( text, @@ -11035,6 +11042,7 @@ export class AgentSession { model: this.model, ...(snapcompactShapeSetting === "auto" ? {} : { shape }), maxFrames, + includeThinking: snapcompactIncludeThinking, }); const framePayloadBytes = this.#snapcompactFramePayloadBytes(snapcompactResult); if (framePayloadBytes > snapcompact.FRAME_DATA_BYTES_BUDGET) { @@ -14021,8 +14029,13 @@ export class AgentSession { let snapcompactResult: snapcompact.CompactionResult | undefined; let snapcompactBlocker: string | undefined; if (action === "snapcompact" && compactionPrep.kind !== "fromHook") { + // Drop `¶think:` sections for Anthropic-dialect targets: the archive + // is replayed as text and Claude refuses reproduced reasoning + // ("reasoning_extraction", issue #6093). + const snapcompactIncludeThinking = preferredDialect(this.model.id) !== "anthropic"; const text = snapcompact.serializeConversation( convertToLlm(preparation.messagesToSummarize.concat(preparation.turnPrefixMessages)), + { includeThinking: snapcompactIncludeThinking }, ); const probeText = snapcompact.renderabilityProbeText( text, @@ -14053,6 +14066,7 @@ export class AgentSession { model: this.model, ...(shapeSetting === "auto" ? {} : { shape }), maxFrames, + includeThinking: snapcompactIncludeThinking, }); const framePayloadBytes = this.#snapcompactFramePayloadBytes(snapcompactResult); if (framePayloadBytes > snapcompact.FRAME_DATA_BYTES_BUDGET) { diff --git a/packages/snapcompact/CHANGELOG.md b/packages/snapcompact/CHANGELOG.md index 35b58d674..6777da3ef 100644 --- a/packages/snapcompact/CHANGELOG.md +++ b/packages/snapcompact/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added an `includeThinking` serialize option (default `true`) so callers can exclude assistant reasoning (`¶think:` sections) from archived transcripts; used to keep reproduced reasoning out of frames replayed to Anthropic-dialect models ([#6093](https://github.com/can1357/oh-my-pi/issues/6093)). + ## [16.5.0] - 2026-07-13 ### Changed diff --git a/packages/snapcompact/src/snapcompact.ts b/packages/snapcompact/src/snapcompact.ts index 443fe93f3..771b43b9a 100644 --- a/packages/snapcompact/src/snapcompact.ts +++ b/packages/snapcompact/src/snapcompact.ts @@ -735,6 +735,12 @@ export interface SerializeOptions { /** Print tool-result text in dim gray ink so archived conversation reads * louder than archived tool noise. Defaults to `true`. */ dimToolResults?: boolean; + /** Serialize assistant reasoning as `¶think:` sections. Defaults to `true`. + * Callers archiving for a Claude/Anthropic-dialect model set this `false`: + * the archive frames are replayed as text into every later request, and + * reasoning rendered back to Claude trips its `reasoning_extraction` + * classifier (issue #6093). */ + includeThinking?: boolean; } /** Keep the head and tail of `text`, eliding the middle beyond `maxChars`. */ @@ -773,6 +779,7 @@ export function serializeConversation(messages: Message[], options?: SerializeOp const toolCallMaxChars = options?.toolCallMaxChars ?? TOOL_CALL_MAX_CHARS; const headRatio = options?.truncateHeadRatio ?? TRUNCATE_HEAD_RATIO; const dimToolResults = options?.dimToolResults !== false; + const includeThinking = options?.includeThinking !== false; const parts: string[] = []; let lastPrefix: string | null = null; @@ -845,6 +852,7 @@ export function serializeConversation(messages: Message[], options?: SerializeOp const text = stripDimMarkers(block.text); if (text.trim()) pendingText.push(text); } else if (block.type === "thinking") { + if (!includeThinking) continue; const thinking = stripDimMarkers(block.thinking); if (thinking.trim()) pendingThinking.push(thinking); } else if (block.type === "toolCall") { diff --git a/packages/snapcompact/test/snapcompact.test.ts b/packages/snapcompact/test/snapcompact.test.ts index e49dcb573..e83ddde9c 100644 --- a/packages/snapcompact/test/snapcompact.test.ts +++ b/packages/snapcompact/test/snapcompact.test.ts @@ -717,6 +717,21 @@ describe("serializeConversation", () => { expect(out).toBe("¶think:weigh options\n\n¶ai:the answer"); }); + it("drops ¶think reasoning sections when includeThinking is false but keeps the reply", () => { + const out = snapcompact.serializeConversation( + [ + createAssistantMessage([ + { type: "thinking", thinking: "private chain of thought" }, + { type: "text", text: "the answer" }, + ]), + ], + { includeThinking: false }, + ); + expect(out).not.toContain("¶think:"); + expect(out).not.toContain("private chain of thought"); + expect(out).toBe("¶ai:the answer"); + }); + it("gives a thinking-only turn its own heading before the tool calls", () => { const out = snapcompact.serializeConversation([ createAssistantMessage([