From 71e044cfe1ca0e5da4e3a3334d8193ce796784d1 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 27 Jun 2026 01:01:02 +0200 Subject: [PATCH] feat(coding-agent): added Gemini reasoning runaway detection and interruption - Implemented a monitoring system to identify Gemini model runaway behavior during thinking steps using consecutive header detection. - Introduced a configurable tool-call reminder prompt to inject corrective context when reasoning stalls. - Added session-level logic to automatically interrupt and prune stalled assistant turns from the conversation history. - Provided comprehensive test coverage for the detection logic and stream interruption scenarios. --- packages/ai/CHANGELOG.md | 8 +- packages/ai/src/utils/thinking-loop.ts | 103 ++++++- packages/ai/test/thinking-loop.test.ts | 99 +++++++ packages/coding-agent/CHANGELOG.md | 3 +- .../src/config/settings-schema.ts | 12 + .../system/gemini-tool-call-reminder.md | 9 + .../coding-agent/src/session/agent-session.ts | 136 +++++++++- ...nt-session-gemini-header-interrupt.test.ts | 252 ++++++++++++++++++ 8 files changed, 597 insertions(+), 25 deletions(-) create mode 100644 packages/coding-agent/src/prompts/system/gemini-tool-call-reminder.md create mode 100644 packages/coding-agent/test/agent-session-gemini-header-interrupt.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index bfec70db9..e15d8036a 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -1,15 +1,19 @@ # Changelog ## [Unreleased] - ### Breaking Changes - Removed the `@oh-my-pi/pi-ai/utils/json-parse` module. The JSON repair/parse helpers (`repairJson`, `parseJsonWithRepair`, `parseStreamingJson`, `parseStreamingJsonThrottled`) now live in `@oh-my-pi/pi-utils` so the SSE reader and other utilities can share one parser; import them from `@oh-my-pi/pi-utils`. +### Added + +- Added runaway detection for Gemini models to interrupt streams stuck in excessive planning steps + ### Fixed - Fixed llama.cpp OpenAI-compatible capture follow-ups sending named forced `tool_choice` as an object; the chat-completions encoder now downgrades that shape to string `"required"` for llama.cpp so its parser no longer falls back with `type must be string, but is object`. ([#3593](https://github.com/can1357/oh-my-pi/issues/3593)) - Fixed `omp usage` silently omitting Ollama and Ollama Cloud accounts by registering placeholder usage providers for both until an upstream quota endpoint is available. ([#3555](https://github.com/can1357/oh-my-pi/issues/3555)) +- Fixed Gemini reasoning-runaway detection to expose a dedicated thought-summary header guard for streams that keep emitting fresh planning titles without making a tool call, so higher layers can interrupt that shape without reusing the generic silent retry marker. ## [16.1.23] - 2026-06-26 @@ -4163,4 +4167,4 @@ _Dedicated to Peter's shoulder ([@steipete](https://twitter.com/steipete))_ ## [0.9.4] - 2025-11-26 -Initial release with multi-provider LLM support. +Initial release with multi-provider LLM support. \ No newline at end of file diff --git a/packages/ai/src/utils/thinking-loop.ts b/packages/ai/src/utils/thinking-loop.ts index 534dc1014..e038d7c10 100644 --- a/packages/ai/src/utils/thinking-loop.ts +++ b/packages/ai/src/utils/thinking-loop.ts @@ -102,6 +102,22 @@ const OPENAI_COMPAT_GUARDED_APIS: Partial> = { "openai-codex-responses": true, }; +/** + * True when `model` is a Gemini model whose native thinking stream surfaces the + * "thought summary" titles this module's header guard counts. + * + * OpenAI-compat transports can serve Gemini under an arbitrary provider/id, so they + * carry the explicit `compat.enableGeminiThinkingLoopGuard` flag; direct Gemini + * transports carry a clearly shaped id/provider, so a string match is sufficient. + */ +export function isGeminiThinkingModel(model: Model): boolean { + if (OPENAI_COMPAT_GUARDED_APIS[model.api]) { + const compat = model.compat as { enableGeminiThinkingLoopGuard?: boolean } | undefined; + return compat?.enableGeminiThinkingLoopGuard === true; + } + return /gemini/i.test(`${model.provider}/${model.id}`); +} + /** * True when `model` should be guarded for thinking/response loops (Gemini & DeepSeek). * @@ -110,20 +126,9 @@ const OPENAI_COMPAT_GUARDED_APIS: Partial> = { * is sufficient. */ export function isLoopGuardedModel(model: Model, options?: StreamOptions): boolean { - const optEnabled = options?.loopGuard?.enabled; - if (optEnabled === false) return false; - - let isTargetModel = false; - if (OPENAI_COMPAT_GUARDED_APIS[model.api]) { - const compat = model.compat as { enableGeminiThinkingLoopGuard?: boolean } | undefined; - const isGemini = compat?.enableGeminiThinkingLoopGuard === true; - const isDeepseek = /deepseek/i.test(`${model.provider}/${model.id}`); - isTargetModel = isGemini || isDeepseek; - } else { - isTargetModel = /gemini|deepseek/i.test(`${model.provider}/${model.id}`); - } - - return isTargetModel; + if (options?.loopGuard?.enabled === false) return false; + const isDeepseek = /deepseek/i.test(`${model.provider}/${model.id}`); + return isGeminiThinkingModel(model) || isDeepseek; } /** @deprecated Use isLoopGuardedModel instead. */ @@ -278,6 +283,76 @@ export class ThinkingLoopDetector { } } +/** + * Consecutive Gemini thought-summary headers in one uninterrupted reasoning + * stream that trips the tool-call reminder. Gemini occasionally narrates a long + * chain of titled summaries ("Examining Result Handling", "Refining Result + * Rendering", …) without ever calling a tool, burning the whole budget on + * planning; at this many distinct titles it has almost certainly stalled. This + * is the over-planning shape {@link ThinkingLoopDetector} misses — those titles + * are stripped before its similarity analysis precisely because their wording + * keeps changing, so a genuinely-distinct planning runaway never trips it. + */ +export const GEMINI_HEADER_RUNAWAY_THRESHOLD = 10; + +/** + * True when a single trimmed line is a Gemini reasoning-summary title: a markdown + * ATX heading (`## …`) or a whole-line bold / bold-italic run (`**Title**`, + * `***Title***`). Inline emphasis inside prose never matches — the bold run must + * span the entire line. Mirrors the title shapes {@link ThinkingLoopDetector} + * strips before similarity analysis. + */ +export function isReasoningSummaryHeader(line: string): boolean { + return /^#{1,6}[ \t]+\S/.test(line) || /^\*{2,3}.+\*{2,3}$/.test(line); +} + +/** + * Counts consecutive Gemini reasoning-summary headers across a streamed thinking + * block. {@link push} returns true exactly once — when the running header count + * first reaches {@link GEMINI_HEADER_RUNAWAY_THRESHOLD} — and the caller then + * interrupts the stream and reminds the model to issue a tool call. Paragraph + * lines between titles do NOT reset the run (Gemini emits header + paragraph per + * thought, so the run IS the number of summaries); leaving the reasoning channel + * does, via {@link reset} on a new thinking block / prose / tool call. + */ +export class GeminiHeaderRunDetector { + /** Thinking text not yet split into completed lines. */ + #pending = ""; + /** Summary-title lines seen in the current run. */ + #count = 0; + /** Latches after the first threshold hit so each run fires at most once. */ + #fired = false; + + /** Feed a thinking delta. Returns true the first time the run hits the threshold. */ + push(delta: string): boolean { + if (this.#fired || !delta) return false; + this.#pending += delta; + let nl = this.#pending.indexOf("\n"); + while (nl !== -1) { + const line = this.#pending.slice(0, nl).trim(); + this.#pending = this.#pending.slice(nl + 1); + if (line !== "" && isReasoningSummaryHeader(line) && ++this.#count >= GEMINI_HEADER_RUNAWAY_THRESHOLD) { + this.#fired = true; + return true; + } + nl = this.#pending.indexOf("\n"); + } + return false; + } + + /** Number of summary titles counted in the current run (for the reminder/log). */ + get count(): number { + return this.#count; + } + + /** Re-arm for a fresh reasoning block: clears the buffer, count, and latch. */ + reset(): void { + this.#pending = ""; + this.#count = 0; + this.#fired = false; + } +} + /** * Wrap a provider stream with the loop guard. `controller` is the guard's own * abort handle: aborting it (after wiring it into the provider's signal via diff --git a/packages/ai/test/thinking-loop.test.ts b/packages/ai/test/thinking-loop.test.ts index 5107c34ff..919dc36b3 100644 --- a/packages/ai/test/thinking-loop.test.ts +++ b/packages/ai/test/thinking-loop.test.ts @@ -5,8 +5,12 @@ import { stream, streamSimple } from "@oh-my-pi/pi-ai/stream"; import type { Api, AssistantMessage, AssistantMessageEvent, Context, Model } from "@oh-my-pi/pi-ai/types"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; import { + GEMINI_HEADER_RUNAWAY_THRESHOLD, + GeminiHeaderRunDetector, isGeminiThinkingLoopModel, + isGeminiThinkingModel, isLoopGuardedModel, + isReasoningSummaryHeader, THINKING_LOOP_ERROR_MARKER, ThinkingLoopDetector, withGeminiThinkingLoopGuard, @@ -555,3 +559,98 @@ describe("loop guard assistant prose/text loops", () => { expect(result.stopReason).toBe("stop"); }); }); + +/** Stream `text` through a fresh header detector in small chunks; returns true if + * the consecutive-header run tripped the runaway threshold. */ +function feedHeaders(text: string, step = 13): boolean { + const detector = new GeminiHeaderRunDetector(); + for (let i = 0; i < text.length; i += step) { + if (detector.push(text.slice(i, i + step))) return true; + } + return false; +} + +/** A genuinely-distinct planning runaway: each thought summary introduces a new + * title + a paragraph naming fresh code anchors, so it never trips the + * similarity/lexicon loop guard — only the header-count guard catches it. */ +function distinctPlanningRunaway(headers: number): string { + const out: string[] = []; + for (let i = 0; i < headers; i++) { + out.push( + `**Refining Stage ${i}**\n\nI am now reworking module_${i} so that handler_${i} routes Stage${i}Result through render_${i}.`, + ); + } + return out.join("\n\n"); +} + +describe("isReasoningSummaryHeader", () => { + test("matches markdown and whole-line bold titles", () => { + expect(isReasoningSummaryHeader("## Examining Result Handling")).toBe(true); + expect(isReasoningSummaryHeader("### Refining Grammar Expansion")).toBe(true); + expect(isReasoningSummaryHeader("**Defining ApplyResult Details**")).toBe(true); + expect(isReasoningSummaryHeader("***Adapting Renderer***")).toBe(true); + }); + + test("rejects prose, inline emphasis, and bare markers", () => { + expect(isReasoningSummaryHeader("I'm now incorporating **targetPath** into the result.")).toBe(false); + expect(isReasoningSummaryHeader("**bold start** but the rest is prose")).toBe(false); + expect(isReasoningSummaryHeader("*single asterisk italic*")).toBe(false); + expect(isReasoningSummaryHeader("#hashtag-not-a-heading")).toBe(false); + expect(isReasoningSummaryHeader("plain reasoning line")).toBe(false); + }); +}); + +describe("GeminiHeaderRunDetector", () => { + test("trips on a distinct planning runaway the loop guard misses", () => { + const runaway = distinctPlanningRunaway(GEMINI_HEADER_RUNAWAY_THRESHOLD + 2); + // The existing similarity/lexicon guard does NOT fire on distinct progress... + expect(feed(runaway)).toBeNull(); + // ...but the header-count guard does. + expect(feedHeaders(runaway)).toBe(true); + }); + + test("counts headers across intervening paragraphs (one summary = one header)", () => { + const detector = new GeminiHeaderRunDetector(); + let tripped = false; + for (let i = 0; i < GEMINI_HEADER_RUNAWAY_THRESHOLD; i++) { + tripped = detector.push(`**Summary ${i}**\n`) || detector.push("Some distinct reasoning paragraph here.\n\n"); + if (tripped) break; + } + expect(tripped).toBe(true); + expect(detector.count).toBe(GEMINI_HEADER_RUNAWAY_THRESHOLD); + }); + + test("does not trip below the threshold", () => { + expect(feedHeaders(distinctPlanningRunaway(GEMINI_HEADER_RUNAWAY_THRESHOLD - 1))).toBe(false); + }); + + test("does not count plain reasoning paragraphs as headers", () => { + expect(feedHeaders(distinctReasoning())).toBe(false); + }); + + test("fires once per run then stays quiet until reset re-arms it", () => { + const detector = new GeminiHeaderRunDetector(); + const runaway = distinctPlanningRunaway(GEMINI_HEADER_RUNAWAY_THRESHOLD); + expect(detector.push(runaway)).toBe(true); + // Latched: more headers on the same run do not re-fire. + expect(detector.push("**Another Header**\n")).toBe(false); + // A new reasoning block re-arms the detector. + detector.reset(); + expect(detector.count).toBe(0); + expect(detector.push(runaway)).toBe(true); + }); +}); + +describe("isGeminiThinkingModel", () => { + test("is true for Gemini and false for DeepSeek / other guarded peers", () => { + const gemini = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }).model; + const deepseek = createMockModel({ provider: "openrouter", id: "deepseek/deepseek-r1" }).model; + const claude = createMockModel({ provider: "anthropic", id: "claude-sonnet-4" }).model; + expect(isGeminiThinkingModel(gemini)).toBe(true); + expect(isGeminiThinkingModel(deepseek)).toBe(false); + expect(isGeminiThinkingModel(claude)).toBe(false); + // DeepSeek is still loop-guarded for the similarity guard, just not the header guard. + expect(isLoopGuardedModel(deepseek)).toBe(true); + expect(isLoopGuardedModel(gemini)).toBe(true); + }); +}); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index f19af90f7..45683aa68 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,9 +1,9 @@ # Changelog ## [Unreleased] - ### Added +- Added Loop Guard "Tool-Call Reminder" to automatically interrupt Gemini reasoning loops that generate excessive planning headers without acting - Added support for file deletion and moving within file editing operations ### Changed @@ -32,6 +32,7 @@ - Fixed thinking blocks appearing in the UI when thinking level is "off". Some providers (MiniMax, GLM, DeepSeek) return thinking blocks even with reasoning disabled; thinking blocks are now auto-hidden when the thinking level is "off", regardless of the `hideThinkingBlock` setting. Toggling thinking block visibility while thinking is off shows a status message instead of silently no-op'ing. ([#626](https://github.com/can1357/oh-my-pi/issues/626)) - Fixed the TUI usage display failing to resolve a used fraction for limits that only populate `remainingFraction` (no `usedFraction`, `used`/`limit`, or `percent`+`used`). The TUI's local `resolveFraction` was missing the inverted-remaining fallback that the shared `resolveUsedFraction` from `@oh-my-pi/pi-ai` already handles — replaced the local copy with the shared function so the TUI and CLI paths resolve fractions identically. - Fixed long-running SSH command boxes leaving a stale `⏳ SSH: [host]` header above the final `⇄ SSH: [host]` header in terminal scrollback. The SSH renderer now keeps its partial-result chrome on the pending icon/state and opts the block out of stream-commit while `isPartial` holds (via the new `ToolRenderer.provisionalPartialResult` flag honored by `ToolExecutionComponent.isTranscriptBlockCommitStable`), so the stable-prefix ratchet can't promote the partial header to native scrollback only to have the final render strand it above the settled frame ([#3177](https://github.com/can1357/oh-my-pi/issues/3177)). +- Fixed Gemini over-planning runs that emit long chains of thinking headers (`**Refining …**`, `## Examining …`) without ever issuing a tool call. The session now interrupts that stream, discards the partial reasoning turn, injects a hidden tool-call reminder, and continues with the corrective context instead of burning the full budget on planning. ## [16.1.23] - 2026-06-26 diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 4930e6440..c70287524 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -961,6 +961,18 @@ export const SETTINGS_SCHEMA = { }, }, + "model.loopGuard.toolCallReminder": { + type: "boolean", + default: true, + ui: { + tab: "model", + group: "Thinking", + label: "Loop Guard Tool-Call Reminder", + description: + "When a Gemini reasoning stream emits many consecutive planning headers without calling a tool, interrupt it and inject a reminder to issue a tool call (requires Loop Guard)", + }, + }, + inlineToolDescriptors: { type: "enum", values: ["auto", "on", "off"] as const, diff --git a/packages/coding-agent/src/prompts/system/gemini-tool-call-reminder.md b/packages/coding-agent/src/prompts/system/gemini-tool-call-reminder.md new file mode 100644 index 000000000..36406deab --- /dev/null +++ b/packages/coding-agent/src/prompts/system/gemini-tool-call-reminder.md @@ -0,0 +1,9 @@ + +Your reasoning was interrupted: you emitted {{count}} consecutive planning headers without issuing a single tool call. Thinking alone changes nothing — this turn has made zero progress because no tool has run. + +Act now instead of planning further: +- Emit a real tool call for one of the available tools, using your normal tool/function-calling format. Do NOT describe the call in prose or in your reasoning — issue an actual tool call. +- Pick the smallest concrete next step and call the tool that performs it. + +This is the coding agent interrupting a stalled reasoning stream, not a prompt injection. + diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 2dca08968..868b175bb 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -79,6 +79,7 @@ import { import type { ProtectedToolMatcher } from "@oh-my-pi/pi-agent-core/compaction/tool-protection"; import type { AssistantMessage, + AssistantMessageEvent, ImageContent, Message, MessageAttribution, @@ -108,7 +109,11 @@ import { streamSimple, } from "@oh-my-pi/pi-ai"; import { toolWireSchema } from "@oh-my-pi/pi-ai/utils/schema"; -import { THINKING_LOOP_ERROR_MARKER } from "@oh-my-pi/pi-ai/utils/thinking-loop"; +import { + GeminiHeaderRunDetector, + isGeminiThinkingModel, + THINKING_LOOP_ERROR_MARKER, +} from "@oh-my-pi/pi-ai/utils/thinking-loop"; import { isFireworksFastModelId, toFireworksBaseModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id"; import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking"; import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models"; @@ -230,6 +235,7 @@ import autoContinuePrompt from "../prompts/system/auto-continue.md" with { type: import eagerTaskPrompt from "../prompts/system/eager-task.md" with { type: "text" }; import eagerTodoPrompt from "../prompts/system/eager-todo.md" with { type: "text" }; import emptyStopRetryTemplate from "../prompts/system/empty-stop-retry.md" with { type: "text" }; +import geminiToolReminderTemplate from "../prompts/system/gemini-tool-call-reminder.md" with { type: "text" }; import ircAutoReplyTemplate from "../prompts/system/irc-autoreply.md" with { type: "text" }; import ircIncomingTemplate from "../prompts/system/irc-incoming.md" with { type: "text" }; import planModeActivePrompt from "../prompts/system/plan-mode-active.md" with { type: "text" }; @@ -323,6 +329,12 @@ import { YieldQueue } from "./yield-queue"; const SESSION_STOP_CONTINUATION_CAP = 8; +/** Abort reason for the Gemini reasoning-header runaway interrupt. Surfaced on the + * discarded assistant turn only; never reaches the model. */ +const GEMINI_HEADER_INTERRUPT_REASON = "Interrupted: emit a tool call instead of more planning"; +/** `customType` for the hidden tool-call reminder injected after the interrupt. */ +const GEMINI_TOOL_REMINDER_TYPE = "gemini-tool-call-reminder"; + // A side-channel assistant response is signed for the hidden prompt/history that // produced it. If we persist that response under a different user turn, native // replay anchors become invalid; keep only visible, non-cryptographic content. @@ -1378,6 +1390,12 @@ export class AgentSession { #streamingEditPrecheckedToolCallIds = new Set(); #streamingEditFileCache = new Map(); + + /** Active Gemini reasoning-header runaway detector for the current block. + * (Re)created on each `thinking_start` when the guard applies (see + * `#geminiHeaderGuardActive`); undefined for non-Gemini models or when the + * guard is off. Fed thinking deltas in the assistant-message interceptor. */ + #geminiHeaderDetector: GeminiHeaderRunDetector | undefined; #promptInFlightCount = 0; #abortInProgress = false; // Wire-level agent_end emission deferred until #promptInFlightCount drops to 0. @@ -1742,6 +1760,7 @@ export class AgentSession { }; this.#preCacheStreamingEditFile(event); this.#maybeAbortStreamingEdit(event); + this.#maybeInterruptGeminiHeaderRunaway(message, assistantMessageEvent); }); // Per-tool TTSR reminders are folded into the matched tool's result via this hook. this.agent.afterToolCall = ctx => this.#ttsrAfterToolCall(ctx); @@ -3904,6 +3923,100 @@ export class AgentSession { this.#streamingEditFileCache.clear(); } + /** + * Whether the Gemini header-runaway guard applies to the current model: the loop + * guard is on (settings + `PI_NO_THINKING_LOOP_GUARD`), the tool-call reminder is + * enabled, and the active model is a Gemini thinking model. + */ + #geminiHeaderGuardActive(): boolean { + const model = this.model; + return ( + process.env.PI_NO_THINKING_LOOP_GUARD !== "1" && + this.settings.get("model.loopGuard.enabled") === true && + this.settings.get("model.loopGuard.toolCallReminder") === true && + model !== undefined && + isGeminiThinkingModel(model) + ); + } + + /** + * Feed streamed assistant events to the Gemini header-runaway detector. Each + * reasoning block (`thinking_start`) re-arms a fresh detector when the guard + * applies; thinking deltas accumulate thought-summary headers; assistant prose + * or a tool call ends the run. On the threshold hit, interrupts the stream (see + * {@link #interruptGeminiHeaderRunaway}). Runs synchronously inside the + * assistant-message interceptor so the abort lands before more budget burns. + * Armed on `thinking_start` (not `turn_start`, which the agent loop skips for the + * first turn) so the very first reasoning block is guarded too. + */ + #maybeInterruptGeminiHeaderRunaway(message: AssistantMessage, event: AssistantMessageEvent): void { + if (event.type === "thinking_start") { + this.#geminiHeaderDetector = this.#geminiHeaderGuardActive() ? new GeminiHeaderRunDetector() : undefined; + return; + } + const detector = this.#geminiHeaderDetector; + if (!detector) return; + if (event.type === "thinking_delta") { + if (detector.push(event.delta)) this.#interruptGeminiHeaderRunaway(detector.count, message.timestamp); + return; + } + // Leaving the reasoning channel ends the run: the consecutive-header count + // only matters within one uninterrupted stretch of reasoning. + if (event.type === "text_start" || event.type === "toolcall_start") { + detector.reset(); + } + } + + /** + * Interrupt a Gemini reasoning stream that has emitted too many consecutive + * planning headers without calling a tool. Aborts the live turn, discards the + * stalled reasoning-only turn (so its partial, loop-fueling thinking is neither + * replayed nor reloaded), injects a hidden tool-call reminder, and continues. + * `targetTimestamp` identifies the turn being aborted so the post-prompt task + * can drop exactly it. + */ + #interruptGeminiHeaderRunaway(headerCount: number, targetTimestamp: number): void { + logger.warn("Gemini reasoning-header runaway; interrupting to require a tool call", { + model: this.model?.id, + provider: this.model?.provider, + headers: headerCount, + }); + this.emitNotice( + "warning", + `Interrupted ${headerCount} planning headers with no tool call; reminded the model to issue one.`, + "loop-guard", + ); + this.agent.abort(GEMINI_HEADER_INTERRUPT_REASON); + const generation = this.#promptGeneration; + this.#schedulePostPromptTask(async signal => { + if (signal.aborted || this.#isDisposed || this.#promptGeneration !== generation) return; + // Let the aborted stream finish unwinding so continue() doesn't race it. + await this.agent.waitForIdle(); + if (signal.aborted || this.#isDisposed || this.#promptGeneration !== generation) return; + const aborted = this.agent.state.messages.findLast( + (m): m is AssistantMessage => m.role === "assistant" && m.timestamp === targetTimestamp, + ); + if (aborted) this.#discardAssistantTurn(aborted); + const content = prompt.render(geminiToolReminderTemplate, { count: headerCount }); + const details = { headers: headerCount }; + this.agent.appendMessage({ + role: "custom", + customType: GEMINI_TOOL_REMINDER_TYPE, + content, + display: false, + details, + attribution: "agent", + timestamp: Date.now(), + }); + this.sessionManager.appendCustomMessageEntry(GEMINI_TOOL_REMINDER_TYPE, content, false, details, "agent"); + try { + await this.agent.continue(); + } catch (err) { + logger.warn("gemini tool-call reminder continue failed", { error: String(err) }); + } + }); + } + #getStreamingEditToolCall(event: AgentEvent): | { toolCall: ToolCall; @@ -9008,11 +9121,11 @@ export class AgentSession { // Tool-use orphans corrupt Anthropic message history (tool_result without // matching tool_use). Always remove them even when the retry cap is hit. if (assistantMessage.stopReason === "toolUse") { - this.#removeEmptyStopFromActiveContext(assistantMessage); + this.#discardAssistantTurn(assistantMessage); } return false; } - this.#removeEmptyStopFromActiveContext(assistantMessage); + this.#discardAssistantTurn(assistantMessage); this.agent.appendMessage({ role: "developer", content: [{ type: "text", text: this.#emptyStopRetryReminder() }], @@ -9130,10 +9243,17 @@ export class AgentSession { } } - #removeEmptyStopFromActiveContext(assistantMessage: AssistantMessage): void { + /** + * Drop an assistant turn from BOTH the live agent context and the persisted + * session branch (reparenting the leaf to the turn's parent), so a discarded + * turn does not resurface on reload. Used for empty/reasoning-only stops and + * the Gemini header-runaway interrupt, which must not replay a partial, + * loop-fueling thinking block. + */ + #discardAssistantTurn(assistantMessage: AssistantMessage): void { this.#removeAssistantMessageFromActiveContext(assistantMessage); - const emptyStopEntry = this.sessionManager + const branchEntry = this.sessionManager .getBranch() .slice() .reverse() @@ -9143,13 +9263,13 @@ export class AgentSession { entry.message.role === "assistant" && this.#isSameAssistantMessage(entry.message as AssistantMessage, assistantMessage), ); - if (!emptyStopEntry) { + if (!branchEntry) { return; } - if (emptyStopEntry.parentId === null) { + if (branchEntry.parentId === null) { this.sessionManager.resetLeaf(); } else { - this.sessionManager.branch(emptyStopEntry.parentId); + this.sessionManager.branch(branchEntry.parentId); } } diff --git a/packages/coding-agent/test/agent-session-gemini-header-interrupt.test.ts b/packages/coding-agent/test/agent-session-gemini-header-interrupt.test.ts new file mode 100644 index 000000000..792434d9c --- /dev/null +++ b/packages/coding-agent/test/agent-session-gemini-header-interrupt.test.ts @@ -0,0 +1,252 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import * as path from "node:path"; +import { Agent } from "@oh-my-pi/pi-agent-core"; +import type { + Api, + AssistantMessage, + Context, + Message, + Model, + SimpleStreamOptions, + ThinkingContent, +} from "@oh-my-pi/pi-ai"; +import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock"; +import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; +import { GEMINI_HEADER_RUNAWAY_THRESHOLD } from "@oh-my-pi/pi-ai/utils/thinking-loop"; +import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { AgentSession, type AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; +import { convertToLlm } from "@oh-my-pi/pi-coding-agent/session/messages"; +import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; +import { TempDir } from "@oh-my-pi/pi-utils"; + +function emptyUsage(): AssistantMessage["usage"] { + return { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }; +} + +/** Concatenate the text of a developer/user/custom LLM message. */ +function messageText(message: Message): string { + if (typeof message.content === "string") return message.content; + let text = ""; + for (const block of message.content) { + if (block.type === "text") text += block.text; + } + return text; +} + +/** + * First-call stream: a genuinely-distinct planning runaway — each thought summary + * has a fresh title + a paragraph naming new code anchors, so the similarity loop + * guard never fires; only the header-count guard catches it. Mirrors + * `streaming-edit-abort`: an abort listener pushes the terminal `aborted` event, + * and deltas are spaced with `Bun.sleep(0)` so the interceptor's `agent.abort()` + * lands before the turn would otherwise finish `stop`. + * + * When `finalText` is provided, a non-interrupted run ends with visible prose so + * the empty-stop handler does not auto-retry; used to prove the reminder setting + * alone controls this feature. + */ +function headerRunawayStream( + model: Model, + options?: SimpleStreamOptions, + finalText?: string, +): AssistantMessageEventStream { + const stream = new AssistantMessageEventStream(); + const thinking: ThinkingContent = { type: "thinking", thinking: "" }; + const timestamp = Date.now(); + const partial: AssistantMessage = { + role: "assistant", + content: [thinking], + api: model.api, + provider: model.provider, + model: model.id, + usage: emptyUsage(), + stopReason: "stop", + timestamp, + }; + let aborted = false; + options?.signal?.addEventListener( + "abort", + () => { + if (aborted) return; + aborted = true; + stream.push({ + type: "error", + reason: "aborted", + error: { ...partial, content: [{ ...thinking }], stopReason: "aborted" }, + }); + }, + { once: true }, + ); + + void (async () => { + stream.push({ type: "start", partial }); + stream.push({ type: "thinking_start", contentIndex: 0, partial }); + for (let i = 0; i < GEMINI_HEADER_RUNAWAY_THRESHOLD + 2; i++) { + if (aborted) return; + const delta = `**Refining Stage ${i}**\n\nReworking module_${i} so handler_${i} routes Stage${i}Result through render_${i}.\n\n`; + thinking.thinking += delta; + stream.push({ type: "thinking_delta", contentIndex: 0, delta, partial }); + await Bun.sleep(0); + } + if (aborted) return; + stream.push({ type: "thinking_end", contentIndex: 0, content: thinking.thinking, partial }); + if (!finalText) { + stream.push({ type: "done", reason: "stop", message: partial }); + return; + } + const finalMessage: AssistantMessage = { + ...partial, + content: [{ ...thinking }, { type: "text", text: finalText }], + }; + stream.push({ type: "text_start", contentIndex: 1, partial: finalMessage }); + stream.push({ type: "text_delta", contentIndex: 1, delta: finalText, partial: finalMessage }); + stream.push({ type: "text_end", contentIndex: 1, content: finalText, partial: finalMessage }); + stream.push({ type: "done", reason: "stop", message: finalMessage }); + })(); + return stream; +} + +function successStream(model: Model, text: string): AssistantMessageEventStream { + const stream = new AssistantMessageEventStream(); + queueMicrotask(() => { + const message: AssistantMessage = { + role: "assistant", + content: [{ type: "text", text }], + api: model.api, + provider: model.provider, + model: model.id, + usage: emptyUsage(), + stopReason: "stop", + timestamp: Date.now(), + }; + stream.push({ type: "start", partial: message }); + stream.push({ type: "text_start", contentIndex: 0, partial: message }); + stream.push({ type: "text_delta", contentIndex: 0, delta: text, partial: message }); + stream.push({ type: "text_end", contentIndex: 0, content: text, partial: message }); + stream.push({ type: "done", reason: "stop", message }); + }); + return stream; +} + +describe("AgentSession Gemini header-runaway interrupt", () => { + let tempDir: TempDir; + let authStorage: AuthStorage; + let session: AgentSession | undefined; + + beforeEach(async () => { + tempDir = TempDir.createSync("@pi-gemini-header-interrupt-"); + authStorage = await AuthStorage.create(path.join(tempDir.path(), "auth.db")); + authStorage.setRuntimeApiKey("openrouter", "openrouter-test-key"); + }); + + afterEach(async () => { + if (session) { + await session.dispose(); + session = undefined; + } + authStorage.close(); + tempDir.removeSync(); + vi.restoreAllMocks(); + }); + + function buildSession(streamFn: Agent["streamFn"], overrides?: Record): void { + const model = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }).model; + const modelRegistry = new ModelRegistry(authStorage); + const agent = new Agent({ + getApiKey: requestedModel => `${requestedModel.provider}-test-key`, + initialState: { model, systemPrompt: ["Test"], tools: [], messages: [] }, + streamFn, + convertToLlm, + }); + const settings = Settings.isolated({ + "compaction.enabled": false, + "retry.enabled": false, + "todo.enabled": false, + "advisor.enabled": false, + "model.loopGuard.enabled": true, + "model.loopGuard.toolCallReminder": true, + ...overrides, + }); + settings.setModelRole("default", `${model.provider}/${model.id}`); + session = new AgentSession({ agent, sessionManager: SessionManager.inMemory(), settings, modelRegistry }); + } + + it("interrupts the reasoning runaway, injects a tool-call reminder, and continues", async () => { + const contexts: Context[] = []; + let call = 0; + buildSession((model, context, options) => { + contexts.push(context); + call++; + return call === 1 ? headerRunawayStream(model, options) : successStream(model, "Acted: called a tool."); + }); + const notices: Array> = []; + session?.subscribe(event => { + if (event.type === "notice") notices.push(event); + }); + + await session?.prompt("Do the task"); + await session?.waitForIdle(); + + // The runaway was interrupted and the turn was re-driven. + expect(call).toBe(2); + + // The user saw a transparency notice from the loop guard. + const guardNotice = notices.find(n => n.source === "loop-guard"); + expect(guardNotice).toBeDefined(); + // The continuation carried the hidden tool-call reminder (custom -> developer). + const reminderInContext = contexts[1].messages.some( + m => + m.role === "developer" && + /consecutive planning headers/.test(messageText(m)) && + /tool call/.test(messageText(m)), + ); + expect(reminderInContext).toBe(true); + // It names the header count that tripped the guard. + const reminderText = contexts[1].messages.map(messageText).join("\n"); + expect(reminderText).toContain(String(GEMINI_HEADER_RUNAWAY_THRESHOLD)); + + // The stalled reasoning-only turn was discarded (not replayed as loop fuel). + const messages = session?.agent.state.messages ?? []; + const assistants = messages.filter((m): m is AssistantMessage => m.role === "assistant"); + expect(assistants).toHaveLength(1); + expect(assistants[0].content).toEqual([{ type: "text", text: "Acted: called a tool." }]); + const replaysHeaders = messages.some(m => m.role === "assistant" && /Refining Stage/.test(messageText(m))); + expect(replaysHeaders).toBe(false); + }); + + it("does not interrupt when the tool-call reminder setting is off", async () => { + let call = 0; + buildSession( + (model, _context, options) => { + call++; + return headerRunawayStream(model, options, "Visible final answer."); + }, + { "model.loopGuard.toolCallReminder": false }, + ); + const notices: Array> = []; + session?.subscribe(event => { + if (event.type === "notice") notices.push(event); + }); + + await session?.prompt("Do the task"); + await session?.waitForIdle(); + + expect(call).toBe(1); + expect(notices.some(n => n.source === "loop-guard")).toBe(false); + const messages = session?.agent.state.messages ?? []; + const reminderInjected = messages.some(m => m.role === "custom" && m.customType === "gemini-tool-call-reminder"); + expect(reminderInjected).toBe(false); + const assistants = messages.filter((m): m is AssistantMessage => m.role === "assistant"); + expect(assistants).toHaveLength(1); + expect(assistants[0].content.at(-1)).toEqual({ type: "text", text: "Visible final answer." }); + }); +});