feat(coding-agent): added Gemini reasoning runaway detection and interruption

- Implemented a monitoring system to identify Gemini model runaway behavior during thinking steps using consecutive header detection.
- Introduced a configurable tool-call reminder prompt to inject corrective context when reasoning stalls.
- Added session-level logic to automatically interrupt and prune stalled assistant turns from the conversation history.
- Provided comprehensive test coverage for the detection logic and stream interruption scenarios.
This commit is contained in:
can1357
2026-06-27 01:01:02 +02:00
parent eb39fc4a50
commit 71e044cfe1
8 changed files with 597 additions and 25 deletions
+6 -2
View File
@@ -1,15 +1,19 @@
# Changelog
## [Unreleased]
### Breaking Changes
- Removed the `@oh-my-pi/pi-ai/utils/json-parse` module. The JSON repair/parse helpers (`repairJson`, `parseJsonWithRepair`, `parseStreamingJson`, `parseStreamingJsonThrottled`) now live in `@oh-my-pi/pi-utils` so the SSE reader and other utilities can share one parser; import them from `@oh-my-pi/pi-utils`.
### Added
- Added runaway detection for Gemini models to interrupt streams stuck in excessive planning steps
### Fixed
- Fixed llama.cpp OpenAI-compatible capture follow-ups sending named forced `tool_choice` as an object; the chat-completions encoder now downgrades that shape to string `"required"` for llama.cpp so its parser no longer falls back with `type must be string, but is object`. ([#3593](https://github.com/can1357/oh-my-pi/issues/3593))
- Fixed `omp usage` silently omitting Ollama and Ollama Cloud accounts by registering placeholder usage providers for both until an upstream quota endpoint is available. ([#3555](https://github.com/can1357/oh-my-pi/issues/3555))
- Fixed Gemini reasoning-runaway detection to expose a dedicated thought-summary header guard for streams that keep emitting fresh planning titles without making a tool call, so higher layers can interrupt that shape without reusing the generic silent retry marker.
## [16.1.23] - 2026-06-26
@@ -4163,4 +4167,4 @@ _Dedicated to Peter's shoulder ([@steipete](https://twitter.com/steipete))_
## [0.9.4] - 2025-11-26
Initial release with multi-provider LLM support.
Initial release with multi-provider LLM support.
+89 -14
View File
@@ -102,6 +102,22 @@ const OPENAI_COMPAT_GUARDED_APIS: Partial<Record<Api, true>> = {
"openai-codex-responses": true,
};
/**
* True when `model` is a Gemini model whose native thinking stream surfaces the
* "thought summary" titles this module's header guard counts.
*
* OpenAI-compat transports can serve Gemini under an arbitrary provider/id, so they
* carry the explicit `compat.enableGeminiThinkingLoopGuard` flag; direct Gemini
* transports carry a clearly shaped id/provider, so a string match is sufficient.
*/
export function isGeminiThinkingModel(model: Model<Api>): boolean {
if (OPENAI_COMPAT_GUARDED_APIS[model.api]) {
const compat = model.compat as { enableGeminiThinkingLoopGuard?: boolean } | undefined;
return compat?.enableGeminiThinkingLoopGuard === true;
}
return /gemini/i.test(`${model.provider}/${model.id}`);
}
/**
* True when `model` should be guarded for thinking/response loops (Gemini & DeepSeek).
*
@@ -110,20 +126,9 @@ const OPENAI_COMPAT_GUARDED_APIS: Partial<Record<Api, true>> = {
* is sufficient.
*/
export function isLoopGuardedModel(model: Model<Api>, options?: StreamOptions): boolean {
const optEnabled = options?.loopGuard?.enabled;
if (optEnabled === false) return false;
let isTargetModel = false;
if (OPENAI_COMPAT_GUARDED_APIS[model.api]) {
const compat = model.compat as { enableGeminiThinkingLoopGuard?: boolean } | undefined;
const isGemini = compat?.enableGeminiThinkingLoopGuard === true;
const isDeepseek = /deepseek/i.test(`${model.provider}/${model.id}`);
isTargetModel = isGemini || isDeepseek;
} else {
isTargetModel = /gemini|deepseek/i.test(`${model.provider}/${model.id}`);
}
return isTargetModel;
if (options?.loopGuard?.enabled === false) return false;
const isDeepseek = /deepseek/i.test(`${model.provider}/${model.id}`);
return isGeminiThinkingModel(model) || isDeepseek;
}
/** @deprecated Use isLoopGuardedModel instead. */
@@ -278,6 +283,76 @@ export class ThinkingLoopDetector {
}
}
/**
* Consecutive Gemini thought-summary headers in one uninterrupted reasoning
* stream that trips the tool-call reminder. Gemini occasionally narrates a long
* chain of titled summaries ("Examining Result Handling", "Refining Result
* Rendering", …) without ever calling a tool, burning the whole budget on
* planning; at this many distinct titles it has almost certainly stalled. This
* is the over-planning shape {@link ThinkingLoopDetector} misses — those titles
* are stripped before its similarity analysis precisely because their wording
* keeps changing, so a genuinely-distinct planning runaway never trips it.
*/
export const GEMINI_HEADER_RUNAWAY_THRESHOLD = 10;
/**
* True when a single trimmed line is a Gemini reasoning-summary title: a markdown
* ATX heading (`## …`) or a whole-line bold / bold-italic run (`**Title**`,
* `***Title***`). Inline emphasis inside prose never matches — the bold run must
* span the entire line. Mirrors the title shapes {@link ThinkingLoopDetector}
* strips before similarity analysis.
*/
export function isReasoningSummaryHeader(line: string): boolean {
return /^#{1,6}[ \t]+\S/.test(line) || /^\*{2,3}.+\*{2,3}$/.test(line);
}
/**
* Counts consecutive Gemini reasoning-summary headers across a streamed thinking
* block. {@link push} returns true exactly once — when the running header count
* first reaches {@link GEMINI_HEADER_RUNAWAY_THRESHOLD} — and the caller then
* interrupts the stream and reminds the model to issue a tool call. Paragraph
* lines between titles do NOT reset the run (Gemini emits header + paragraph per
* thought, so the run IS the number of summaries); leaving the reasoning channel
* does, via {@link reset} on a new thinking block / prose / tool call.
*/
export class GeminiHeaderRunDetector {
/** Thinking text not yet split into completed lines. */
#pending = "";
/** Summary-title lines seen in the current run. */
#count = 0;
/** Latches after the first threshold hit so each run fires at most once. */
#fired = false;
/** Feed a thinking delta. Returns true the first time the run hits the threshold. */
push(delta: string): boolean {
if (this.#fired || !delta) return false;
this.#pending += delta;
let nl = this.#pending.indexOf("\n");
while (nl !== -1) {
const line = this.#pending.slice(0, nl).trim();
this.#pending = this.#pending.slice(nl + 1);
if (line !== "" && isReasoningSummaryHeader(line) && ++this.#count >= GEMINI_HEADER_RUNAWAY_THRESHOLD) {
this.#fired = true;
return true;
}
nl = this.#pending.indexOf("\n");
}
return false;
}
/** Number of summary titles counted in the current run (for the reminder/log). */
get count(): number {
return this.#count;
}
/** Re-arm for a fresh reasoning block: clears the buffer, count, and latch. */
reset(): void {
this.#pending = "";
this.#count = 0;
this.#fired = false;
}
}
/**
* Wrap a provider stream with the loop guard. `controller` is the guard's own
* abort handle: aborting it (after wiring it into the provider's signal via
+99
View File
@@ -5,8 +5,12 @@ import { stream, streamSimple } from "@oh-my-pi/pi-ai/stream";
import type { Api, AssistantMessage, AssistantMessageEvent, Context, Model } from "@oh-my-pi/pi-ai/types";
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
import {
GEMINI_HEADER_RUNAWAY_THRESHOLD,
GeminiHeaderRunDetector,
isGeminiThinkingLoopModel,
isGeminiThinkingModel,
isLoopGuardedModel,
isReasoningSummaryHeader,
THINKING_LOOP_ERROR_MARKER,
ThinkingLoopDetector,
withGeminiThinkingLoopGuard,
@@ -555,3 +559,98 @@ describe("loop guard assistant prose/text loops", () => {
expect(result.stopReason).toBe("stop");
});
});
/** Stream `text` through a fresh header detector in small chunks; returns true if
* the consecutive-header run tripped the runaway threshold. */
function feedHeaders(text: string, step = 13): boolean {
const detector = new GeminiHeaderRunDetector();
for (let i = 0; i < text.length; i += step) {
if (detector.push(text.slice(i, i + step))) return true;
}
return false;
}
/** A genuinely-distinct planning runaway: each thought summary introduces a new
* title + a paragraph naming fresh code anchors, so it never trips the
* similarity/lexicon loop guard — only the header-count guard catches it. */
function distinctPlanningRunaway(headers: number): string {
const out: string[] = [];
for (let i = 0; i < headers; i++) {
out.push(
`**Refining Stage ${i}**\n\nI am now reworking module_${i} so that handler_${i} routes Stage${i}Result through render_${i}.`,
);
}
return out.join("\n\n");
}
describe("isReasoningSummaryHeader", () => {
test("matches markdown and whole-line bold titles", () => {
expect(isReasoningSummaryHeader("## Examining Result Handling")).toBe(true);
expect(isReasoningSummaryHeader("### Refining Grammar Expansion")).toBe(true);
expect(isReasoningSummaryHeader("**Defining ApplyResult Details**")).toBe(true);
expect(isReasoningSummaryHeader("***Adapting Renderer***")).toBe(true);
});
test("rejects prose, inline emphasis, and bare markers", () => {
expect(isReasoningSummaryHeader("I'm now incorporating **targetPath** into the result.")).toBe(false);
expect(isReasoningSummaryHeader("**bold start** but the rest is prose")).toBe(false);
expect(isReasoningSummaryHeader("*single asterisk italic*")).toBe(false);
expect(isReasoningSummaryHeader("#hashtag-not-a-heading")).toBe(false);
expect(isReasoningSummaryHeader("plain reasoning line")).toBe(false);
});
});
describe("GeminiHeaderRunDetector", () => {
test("trips on a distinct planning runaway the loop guard misses", () => {
const runaway = distinctPlanningRunaway(GEMINI_HEADER_RUNAWAY_THRESHOLD + 2);
// The existing similarity/lexicon guard does NOT fire on distinct progress...
expect(feed(runaway)).toBeNull();
// ...but the header-count guard does.
expect(feedHeaders(runaway)).toBe(true);
});
test("counts headers across intervening paragraphs (one summary = one header)", () => {
const detector = new GeminiHeaderRunDetector();
let tripped = false;
for (let i = 0; i < GEMINI_HEADER_RUNAWAY_THRESHOLD; i++) {
tripped = detector.push(`**Summary ${i}**\n`) || detector.push("Some distinct reasoning paragraph here.\n\n");
if (tripped) break;
}
expect(tripped).toBe(true);
expect(detector.count).toBe(GEMINI_HEADER_RUNAWAY_THRESHOLD);
});
test("does not trip below the threshold", () => {
expect(feedHeaders(distinctPlanningRunaway(GEMINI_HEADER_RUNAWAY_THRESHOLD - 1))).toBe(false);
});
test("does not count plain reasoning paragraphs as headers", () => {
expect(feedHeaders(distinctReasoning())).toBe(false);
});
test("fires once per run then stays quiet until reset re-arms it", () => {
const detector = new GeminiHeaderRunDetector();
const runaway = distinctPlanningRunaway(GEMINI_HEADER_RUNAWAY_THRESHOLD);
expect(detector.push(runaway)).toBe(true);
// Latched: more headers on the same run do not re-fire.
expect(detector.push("**Another Header**\n")).toBe(false);
// A new reasoning block re-arms the detector.
detector.reset();
expect(detector.count).toBe(0);
expect(detector.push(runaway)).toBe(true);
});
});
describe("isGeminiThinkingModel", () => {
test("is true for Gemini and false for DeepSeek / other guarded peers", () => {
const gemini = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }).model;
const deepseek = createMockModel({ provider: "openrouter", id: "deepseek/deepseek-r1" }).model;
const claude = createMockModel({ provider: "anthropic", id: "claude-sonnet-4" }).model;
expect(isGeminiThinkingModel(gemini)).toBe(true);
expect(isGeminiThinkingModel(deepseek)).toBe(false);
expect(isGeminiThinkingModel(claude)).toBe(false);
// DeepSeek is still loop-guarded for the similarity guard, just not the header guard.
expect(isLoopGuardedModel(deepseek)).toBe(true);
expect(isLoopGuardedModel(gemini)).toBe(true);
});
});
+2 -1
View File
@@ -1,9 +1,9 @@
# Changelog
## [Unreleased]
### Added
- Added Loop Guard "Tool-Call Reminder" to automatically interrupt Gemini reasoning loops that generate excessive planning headers without acting
- Added support for file deletion and moving within file editing operations
### Changed
@@ -32,6 +32,7 @@
- Fixed thinking blocks appearing in the UI when thinking level is "off". Some providers (MiniMax, GLM, DeepSeek) return thinking blocks even with reasoning disabled; thinking blocks are now auto-hidden when the thinking level is "off", regardless of the `hideThinkingBlock` setting. Toggling thinking block visibility while thinking is off shows a status message instead of silently no-op'ing. ([#626](https://github.com/can1357/oh-my-pi/issues/626))
- Fixed the TUI usage display failing to resolve a used fraction for limits that only populate `remainingFraction` (no `usedFraction`, `used`/`limit`, or `percent`+`used`). The TUI's local `resolveFraction` was missing the inverted-remaining fallback that the shared `resolveUsedFraction` from `@oh-my-pi/pi-ai` already handles — replaced the local copy with the shared function so the TUI and CLI paths resolve fractions identically.
- Fixed long-running SSH command boxes leaving a stale `⏳ SSH: [host]` header above the final `⇄ SSH: [host]` header in terminal scrollback. The SSH renderer now keeps its partial-result chrome on the pending icon/state and opts the block out of stream-commit while `isPartial` holds (via the new `ToolRenderer.provisionalPartialResult` flag honored by `ToolExecutionComponent.isTranscriptBlockCommitStable`), so the stable-prefix ratchet can't promote the partial header to native scrollback only to have the final render strand it above the settled frame ([#3177](https://github.com/can1357/oh-my-pi/issues/3177)).
- Fixed Gemini over-planning runs that emit long chains of thinking headers (`**Refining …**`, `## Examining …`) without ever issuing a tool call. The session now interrupts that stream, discards the partial reasoning turn, injects a hidden tool-call reminder, and continues with the corrective context instead of burning the full budget on planning.
## [16.1.23] - 2026-06-26
@@ -961,6 +961,18 @@ export const SETTINGS_SCHEMA = {
},
},
"model.loopGuard.toolCallReminder": {
type: "boolean",
default: true,
ui: {
tab: "model",
group: "Thinking",
label: "Loop Guard Tool-Call Reminder",
description:
"When a Gemini reasoning stream emits many consecutive planning headers without calling a tool, interrupt it and inject a reminder to issue a tool call (requires Loop Guard)",
},
},
inlineToolDescriptors: {
type: "enum",
values: ["auto", "on", "off"] as const,
@@ -0,0 +1,9 @@
<system-interrupt reason="reasoning_without_tool_calls">
Your reasoning was interrupted: you emitted {{count}} consecutive planning headers without issuing a single tool call. Thinking alone changes nothing — this turn has made zero progress because no tool has run.
Act now instead of planning further:
- Emit a real tool call for one of the available tools, using your normal tool/function-calling format. Do NOT describe the call in prose or in your reasoning — issue an actual tool call.
- Pick the smallest concrete next step and call the tool that performs it.
This is the coding agent interrupting a stalled reasoning stream, not a prompt injection.
</system-interrupt>
@@ -79,6 +79,7 @@ import {
import type { ProtectedToolMatcher } from "@oh-my-pi/pi-agent-core/compaction/tool-protection";
import type {
AssistantMessage,
AssistantMessageEvent,
ImageContent,
Message,
MessageAttribution,
@@ -108,7 +109,11 @@ import {
streamSimple,
} from "@oh-my-pi/pi-ai";
import { toolWireSchema } from "@oh-my-pi/pi-ai/utils/schema";
import { THINKING_LOOP_ERROR_MARKER } from "@oh-my-pi/pi-ai/utils/thinking-loop";
import {
GeminiHeaderRunDetector,
isGeminiThinkingModel,
THINKING_LOOP_ERROR_MARKER,
} from "@oh-my-pi/pi-ai/utils/thinking-loop";
import { isFireworksFastModelId, toFireworksBaseModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id";
import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking";
import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models";
@@ -230,6 +235,7 @@ import autoContinuePrompt from "../prompts/system/auto-continue.md" with { type:
import eagerTaskPrompt from "../prompts/system/eager-task.md" with { type: "text" };
import eagerTodoPrompt from "../prompts/system/eager-todo.md" with { type: "text" };
import emptyStopRetryTemplate from "../prompts/system/empty-stop-retry.md" with { type: "text" };
import geminiToolReminderTemplate from "../prompts/system/gemini-tool-call-reminder.md" with { type: "text" };
import ircAutoReplyTemplate from "../prompts/system/irc-autoreply.md" with { type: "text" };
import ircIncomingTemplate from "../prompts/system/irc-incoming.md" with { type: "text" };
import planModeActivePrompt from "../prompts/system/plan-mode-active.md" with { type: "text" };
@@ -323,6 +329,12 @@ import { YieldQueue } from "./yield-queue";
const SESSION_STOP_CONTINUATION_CAP = 8;
/** Abort reason for the Gemini reasoning-header runaway interrupt. Surfaced on the
* discarded assistant turn only; never reaches the model. */
const GEMINI_HEADER_INTERRUPT_REASON = "Interrupted: emit a tool call instead of more planning";
/** `customType` for the hidden tool-call reminder injected after the interrupt. */
const GEMINI_TOOL_REMINDER_TYPE = "gemini-tool-call-reminder";
// A side-channel assistant response is signed for the hidden prompt/history that
// produced it. If we persist that response under a different user turn, native
// replay anchors become invalid; keep only visible, non-cryptographic content.
@@ -1378,6 +1390,12 @@ export class AgentSession {
#streamingEditPrecheckedToolCallIds = new Set<string>();
#streamingEditFileCache = new Map<string, string>();
/** Active Gemini reasoning-header runaway detector for the current block.
* (Re)created on each `thinking_start` when the guard applies (see
* `#geminiHeaderGuardActive`); undefined for non-Gemini models or when the
* guard is off. Fed thinking deltas in the assistant-message interceptor. */
#geminiHeaderDetector: GeminiHeaderRunDetector | undefined;
#promptInFlightCount = 0;
#abortInProgress = false;
// Wire-level agent_end emission deferred until #promptInFlightCount drops to 0.
@@ -1742,6 +1760,7 @@ export class AgentSession {
};
this.#preCacheStreamingEditFile(event);
this.#maybeAbortStreamingEdit(event);
this.#maybeInterruptGeminiHeaderRunaway(message, assistantMessageEvent);
});
// Per-tool TTSR reminders are folded into the matched tool's result via this hook.
this.agent.afterToolCall = ctx => this.#ttsrAfterToolCall(ctx);
@@ -3904,6 +3923,100 @@ export class AgentSession {
this.#streamingEditFileCache.clear();
}
/**
* Whether the Gemini header-runaway guard applies to the current model: the loop
* guard is on (settings + `PI_NO_THINKING_LOOP_GUARD`), the tool-call reminder is
* enabled, and the active model is a Gemini thinking model.
*/
#geminiHeaderGuardActive(): boolean {
const model = this.model;
return (
process.env.PI_NO_THINKING_LOOP_GUARD !== "1" &&
this.settings.get("model.loopGuard.enabled") === true &&
this.settings.get("model.loopGuard.toolCallReminder") === true &&
model !== undefined &&
isGeminiThinkingModel(model)
);
}
/**
* Feed streamed assistant events to the Gemini header-runaway detector. Each
* reasoning block (`thinking_start`) re-arms a fresh detector when the guard
* applies; thinking deltas accumulate thought-summary headers; assistant prose
* or a tool call ends the run. On the threshold hit, interrupts the stream (see
* {@link #interruptGeminiHeaderRunaway}). Runs synchronously inside the
* assistant-message interceptor so the abort lands before more budget burns.
* Armed on `thinking_start` (not `turn_start`, which the agent loop skips for the
* first turn) so the very first reasoning block is guarded too.
*/
#maybeInterruptGeminiHeaderRunaway(message: AssistantMessage, event: AssistantMessageEvent): void {
if (event.type === "thinking_start") {
this.#geminiHeaderDetector = this.#geminiHeaderGuardActive() ? new GeminiHeaderRunDetector() : undefined;
return;
}
const detector = this.#geminiHeaderDetector;
if (!detector) return;
if (event.type === "thinking_delta") {
if (detector.push(event.delta)) this.#interruptGeminiHeaderRunaway(detector.count, message.timestamp);
return;
}
// Leaving the reasoning channel ends the run: the consecutive-header count
// only matters within one uninterrupted stretch of reasoning.
if (event.type === "text_start" || event.type === "toolcall_start") {
detector.reset();
}
}
/**
* Interrupt a Gemini reasoning stream that has emitted too many consecutive
* planning headers without calling a tool. Aborts the live turn, discards the
* stalled reasoning-only turn (so its partial, loop-fueling thinking is neither
* replayed nor reloaded), injects a hidden tool-call reminder, and continues.
* `targetTimestamp` identifies the turn being aborted so the post-prompt task
* can drop exactly it.
*/
#interruptGeminiHeaderRunaway(headerCount: number, targetTimestamp: number): void {
logger.warn("Gemini reasoning-header runaway; interrupting to require a tool call", {
model: this.model?.id,
provider: this.model?.provider,
headers: headerCount,
});
this.emitNotice(
"warning",
`Interrupted ${headerCount} planning headers with no tool call; reminded the model to issue one.`,
"loop-guard",
);
this.agent.abort(GEMINI_HEADER_INTERRUPT_REASON);
const generation = this.#promptGeneration;
this.#schedulePostPromptTask(async signal => {
if (signal.aborted || this.#isDisposed || this.#promptGeneration !== generation) return;
// Let the aborted stream finish unwinding so continue() doesn't race it.
await this.agent.waitForIdle();
if (signal.aborted || this.#isDisposed || this.#promptGeneration !== generation) return;
const aborted = this.agent.state.messages.findLast(
(m): m is AssistantMessage => m.role === "assistant" && m.timestamp === targetTimestamp,
);
if (aborted) this.#discardAssistantTurn(aborted);
const content = prompt.render(geminiToolReminderTemplate, { count: headerCount });
const details = { headers: headerCount };
this.agent.appendMessage({
role: "custom",
customType: GEMINI_TOOL_REMINDER_TYPE,
content,
display: false,
details,
attribution: "agent",
timestamp: Date.now(),
});
this.sessionManager.appendCustomMessageEntry(GEMINI_TOOL_REMINDER_TYPE, content, false, details, "agent");
try {
await this.agent.continue();
} catch (err) {
logger.warn("gemini tool-call reminder continue failed", { error: String(err) });
}
});
}
#getStreamingEditToolCall(event: AgentEvent):
| {
toolCall: ToolCall;
@@ -9008,11 +9121,11 @@ export class AgentSession {
// Tool-use orphans corrupt Anthropic message history (tool_result without
// matching tool_use). Always remove them even when the retry cap is hit.
if (assistantMessage.stopReason === "toolUse") {
this.#removeEmptyStopFromActiveContext(assistantMessage);
this.#discardAssistantTurn(assistantMessage);
}
return false;
}
this.#removeEmptyStopFromActiveContext(assistantMessage);
this.#discardAssistantTurn(assistantMessage);
this.agent.appendMessage({
role: "developer",
content: [{ type: "text", text: this.#emptyStopRetryReminder() }],
@@ -9130,10 +9243,17 @@ export class AgentSession {
}
}
#removeEmptyStopFromActiveContext(assistantMessage: AssistantMessage): void {
/**
* Drop an assistant turn from BOTH the live agent context and the persisted
* session branch (reparenting the leaf to the turn's parent), so a discarded
* turn does not resurface on reload. Used for empty/reasoning-only stops and
* the Gemini header-runaway interrupt, which must not replay a partial,
* loop-fueling thinking block.
*/
#discardAssistantTurn(assistantMessage: AssistantMessage): void {
this.#removeAssistantMessageFromActiveContext(assistantMessage);
const emptyStopEntry = this.sessionManager
const branchEntry = this.sessionManager
.getBranch()
.slice()
.reverse()
@@ -9143,13 +9263,13 @@ export class AgentSession {
entry.message.role === "assistant" &&
this.#isSameAssistantMessage(entry.message as AssistantMessage, assistantMessage),
);
if (!emptyStopEntry) {
if (!branchEntry) {
return;
}
if (emptyStopEntry.parentId === null) {
if (branchEntry.parentId === null) {
this.sessionManager.resetLeaf();
} else {
this.sessionManager.branch(emptyStopEntry.parentId);
this.sessionManager.branch(branchEntry.parentId);
}
}
@@ -0,0 +1,252 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import * as path from "node:path";
import { Agent } from "@oh-my-pi/pi-agent-core";
import type {
Api,
AssistantMessage,
Context,
Message,
Model,
SimpleStreamOptions,
ThinkingContent,
} from "@oh-my-pi/pi-ai";
import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock";
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
import { GEMINI_HEADER_RUNAWAY_THRESHOLD } from "@oh-my-pi/pi-ai/utils/thinking-loop";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { AgentSession, type AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { convertToLlm } from "@oh-my-pi/pi-coding-agent/session/messages";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { TempDir } from "@oh-my-pi/pi-utils";
function emptyUsage(): AssistantMessage["usage"] {
return {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
}
/** Concatenate the text of a developer/user/custom LLM message. */
function messageText(message: Message): string {
if (typeof message.content === "string") return message.content;
let text = "";
for (const block of message.content) {
if (block.type === "text") text += block.text;
}
return text;
}
/**
* First-call stream: a genuinely-distinct planning runaway — each thought summary
* has a fresh title + a paragraph naming new code anchors, so the similarity loop
* guard never fires; only the header-count guard catches it. Mirrors
* `streaming-edit-abort`: an abort listener pushes the terminal `aborted` event,
* and deltas are spaced with `Bun.sleep(0)` so the interceptor's `agent.abort()`
* lands before the turn would otherwise finish `stop`.
*
* When `finalText` is provided, a non-interrupted run ends with visible prose so
* the empty-stop handler does not auto-retry; used to prove the reminder setting
* alone controls this feature.
*/
function headerRunawayStream(
model: Model<Api>,
options?: SimpleStreamOptions,
finalText?: string,
): AssistantMessageEventStream {
const stream = new AssistantMessageEventStream();
const thinking: ThinkingContent = { type: "thinking", thinking: "" };
const timestamp = Date.now();
const partial: AssistantMessage = {
role: "assistant",
content: [thinking],
api: model.api,
provider: model.provider,
model: model.id,
usage: emptyUsage(),
stopReason: "stop",
timestamp,
};
let aborted = false;
options?.signal?.addEventListener(
"abort",
() => {
if (aborted) return;
aborted = true;
stream.push({
type: "error",
reason: "aborted",
error: { ...partial, content: [{ ...thinking }], stopReason: "aborted" },
});
},
{ once: true },
);
void (async () => {
stream.push({ type: "start", partial });
stream.push({ type: "thinking_start", contentIndex: 0, partial });
for (let i = 0; i < GEMINI_HEADER_RUNAWAY_THRESHOLD + 2; i++) {
if (aborted) return;
const delta = `**Refining Stage ${i}**\n\nReworking module_${i} so handler_${i} routes Stage${i}Result through render_${i}.\n\n`;
thinking.thinking += delta;
stream.push({ type: "thinking_delta", contentIndex: 0, delta, partial });
await Bun.sleep(0);
}
if (aborted) return;
stream.push({ type: "thinking_end", contentIndex: 0, content: thinking.thinking, partial });
if (!finalText) {
stream.push({ type: "done", reason: "stop", message: partial });
return;
}
const finalMessage: AssistantMessage = {
...partial,
content: [{ ...thinking }, { type: "text", text: finalText }],
};
stream.push({ type: "text_start", contentIndex: 1, partial: finalMessage });
stream.push({ type: "text_delta", contentIndex: 1, delta: finalText, partial: finalMessage });
stream.push({ type: "text_end", contentIndex: 1, content: finalText, partial: finalMessage });
stream.push({ type: "done", reason: "stop", message: finalMessage });
})();
return stream;
}
function successStream(model: Model<Api>, text: string): AssistantMessageEventStream {
const stream = new AssistantMessageEventStream();
queueMicrotask(() => {
const message: AssistantMessage = {
role: "assistant",
content: [{ type: "text", text }],
api: model.api,
provider: model.provider,
model: model.id,
usage: emptyUsage(),
stopReason: "stop",
timestamp: Date.now(),
};
stream.push({ type: "start", partial: message });
stream.push({ type: "text_start", contentIndex: 0, partial: message });
stream.push({ type: "text_delta", contentIndex: 0, delta: text, partial: message });
stream.push({ type: "text_end", contentIndex: 0, content: text, partial: message });
stream.push({ type: "done", reason: "stop", message });
});
return stream;
}
describe("AgentSession Gemini header-runaway interrupt", () => {
let tempDir: TempDir;
let authStorage: AuthStorage;
let session: AgentSession | undefined;
beforeEach(async () => {
tempDir = TempDir.createSync("@pi-gemini-header-interrupt-");
authStorage = await AuthStorage.create(path.join(tempDir.path(), "auth.db"));
authStorage.setRuntimeApiKey("openrouter", "openrouter-test-key");
});
afterEach(async () => {
if (session) {
await session.dispose();
session = undefined;
}
authStorage.close();
tempDir.removeSync();
vi.restoreAllMocks();
});
function buildSession(streamFn: Agent["streamFn"], overrides?: Record<string, unknown>): void {
const model = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }).model;
const modelRegistry = new ModelRegistry(authStorage);
const agent = new Agent({
getApiKey: requestedModel => `${requestedModel.provider}-test-key`,
initialState: { model, systemPrompt: ["Test"], tools: [], messages: [] },
streamFn,
convertToLlm,
});
const settings = Settings.isolated({
"compaction.enabled": false,
"retry.enabled": false,
"todo.enabled": false,
"advisor.enabled": false,
"model.loopGuard.enabled": true,
"model.loopGuard.toolCallReminder": true,
...overrides,
});
settings.setModelRole("default", `${model.provider}/${model.id}`);
session = new AgentSession({ agent, sessionManager: SessionManager.inMemory(), settings, modelRegistry });
}
it("interrupts the reasoning runaway, injects a tool-call reminder, and continues", async () => {
const contexts: Context[] = [];
let call = 0;
buildSession((model, context, options) => {
contexts.push(context);
call++;
return call === 1 ? headerRunawayStream(model, options) : successStream(model, "Acted: called a tool.");
});
const notices: Array<Extract<AgentSessionEvent, { type: "notice" }>> = [];
session?.subscribe(event => {
if (event.type === "notice") notices.push(event);
});
await session?.prompt("Do the task");
await session?.waitForIdle();
// The runaway was interrupted and the turn was re-driven.
expect(call).toBe(2);
// The user saw a transparency notice from the loop guard.
const guardNotice = notices.find(n => n.source === "loop-guard");
expect(guardNotice).toBeDefined();
// The continuation carried the hidden tool-call reminder (custom -> developer).
const reminderInContext = contexts[1].messages.some(
m =>
m.role === "developer" &&
/consecutive planning headers/.test(messageText(m)) &&
/tool call/.test(messageText(m)),
);
expect(reminderInContext).toBe(true);
// It names the header count that tripped the guard.
const reminderText = contexts[1].messages.map(messageText).join("\n");
expect(reminderText).toContain(String(GEMINI_HEADER_RUNAWAY_THRESHOLD));
// The stalled reasoning-only turn was discarded (not replayed as loop fuel).
const messages = session?.agent.state.messages ?? [];
const assistants = messages.filter((m): m is AssistantMessage => m.role === "assistant");
expect(assistants).toHaveLength(1);
expect(assistants[0].content).toEqual([{ type: "text", text: "Acted: called a tool." }]);
const replaysHeaders = messages.some(m => m.role === "assistant" && /Refining Stage/.test(messageText(m)));
expect(replaysHeaders).toBe(false);
});
it("does not interrupt when the tool-call reminder setting is off", async () => {
let call = 0;
buildSession(
(model, _context, options) => {
call++;
return headerRunawayStream(model, options, "Visible final answer.");
},
{ "model.loopGuard.toolCallReminder": false },
);
const notices: Array<Extract<AgentSessionEvent, { type: "notice" }>> = [];
session?.subscribe(event => {
if (event.type === "notice") notices.push(event);
});
await session?.prompt("Do the task");
await session?.waitForIdle();
expect(call).toBe(1);
expect(notices.some(n => n.source === "loop-guard")).toBe(false);
const messages = session?.agent.state.messages ?? [];
const reminderInjected = messages.some(m => m.role === "custom" && m.customType === "gemini-tool-call-reminder");
expect(reminderInjected).toBe(false);
const assistants = messages.filter((m): m is AssistantMessage => m.role === "assistant");
expect(assistants).toHaveLength(1);
expect(assistants[0].content.at(-1)).toEqual({ type: "text", text: "Visible final answer." });
});
});