feat: enabled auto-retry for AI thinking loops

- Improved thinking loop detection logic by refining text normalization and treating stalls as retryable errors.
- Instrumented the agent session to recognize thinking loop markers within retryable error conditions.
- Automated the clearing of stale error banners upon successful auto-retry execution.
- Added comprehensive test coverage for chunked thinking loop errors and banner management.
This commit is contained in:
can1357
2026-06-19 17:51:53 +02:00
parent 8351536641
commit b2dc706ee7
8 changed files with 397 additions and 29 deletions
+6 -2
View File
@@ -1,11 +1,15 @@
# Changelog
## [Unreleased]
### Added
- Added regression test pinning that `openai-completions` emits a `thinking` block for `reasoning_content` deltas even when `delta.content` is explicitly JSON `null` (the DeepSeek-format dual-key pattern used by custom GLM/Qwen reasoning providers). See [#2996](https://github.com/can1357/oh-my-pi/issues/2996).
### Changed
- Improved the thinking loop guard to treat assistant text loops as retryable errors
- Refined text normalization logic to reduce false positives in the thinking loop detector
### Fixed
- Fixed Ollama chat requests sending image payloads to text-only models. Image blocks are now omitted and replaced with the standard non-vision placeholder for models without vision support, while vision-capable Ollama models continue to receive images. ([#3009](https://github.com/can1357/oh-my-pi/pull/3009) by [@serverinspector](https://github.com/serverinspector))
@@ -3943,4 +3947,4 @@ _Dedicated to Peter's shoulder ([@steipete](https://twitter.com/steipete))_
## [0.9.4] - 2025-11-26
Initial release with multi-provider LLM support.
Initial release with multi-provider LLM support.
+18 -19
View File
@@ -25,11 +25,11 @@
* 13.5k non-loop thinking blocks (zero false positives; hardest negative
* scored 3 against the trigger of 4).
*
* Scope is deliberately narrow: **thinking only**. Answer text is left
* untouched so the guard can never discard already-streamed visible output. The
* guard is gated to Gemini models and wraps the provider stream, so it works
* across every Gemini transport (OpenRouter `openai-completions`, direct
* `google-generative-ai` / `google-gemini-cli`, Vertex). Disable with
* Scope is narrow: guarded Gemini/DeepSeek streams before any tool call. Native
* thinking is checked first; assistant text can also be checked for providers
* that surface reasoning as visible prose. On a hit, the failed turn is emitted
* as an empty retryable stream-stall error so the session drops and re-samples
* it instead of committing the runaway transcript. Disable with
* `PI_NO_THINKING_LOOP_GUARD=1`.
*/
import { logger } from "@oh-my-pi/pi-utils";
@@ -208,7 +208,6 @@ export function guardThinkingLoopStream(
void (async () => {
let thinkingArmed = true;
let textArmed = checkAssistantContent;
let accumulatedText = "";
try {
for await (const event of inner) {
let detail: string | null = null;
@@ -221,7 +220,6 @@ export function guardThinkingLoopStream(
thinkingArmed = false;
if (textArmed && event.type === "text_delta") {
detail = textDetector.push(event.delta);
accumulatedText += event.delta;
}
} else if (event.type === "toolcall_start" || event.type === "toolcall_delta") {
thinkingArmed = false;
@@ -244,7 +242,7 @@ export function guardThinkingLoopStream(
outer.push({
type: "error",
reason: "error",
error: buildThinkingLoopError(model, detail, accumulatedText),
error: buildThinkingLoopError(model, detail),
});
return;
}
@@ -289,13 +287,13 @@ export function withGeminiThinkingLoopGuard<
return guardThinkingLoopStream(dispatch(merged), model, controller, options);
}
function buildThinkingLoopError(model: Model<Api>, detail: string, accumulatedText?: string): AssistantMessage {
const hasText = Boolean(accumulatedText);
function buildThinkingLoopError(model: Model<Api>, detail: string): AssistantMessage {
return {
role: "assistant",
// Empty content is load-bearing: a contentful error stop is replay-unsafe
// and would NOT be auto-retried by the session.
content: hasText ? [{ type: "text", text: accumulatedText! }] : [],
// Empty content is load-bearing: loop-guard output is replay garbage, even
// when it arrived as assistant text instead of native thinking. Keeping it
// would persist the failed attempt before AgentSession retries.
content: [],
api: model.api,
provider: model.provider,
model: model.id,
@@ -310,7 +308,7 @@ function buildThinkingLoopError(model: Model<Api>, detail: string, accumulatedTe
stopReason: "error",
// "stream stall" makes the transport/session retry classifiers treat this
// as a transient (retryable) failure with no bespoke rule.
errorMessage: `${THINKING_LOOP_ERROR_MARKER}: the model repeated near-identical content (${detail}).${hasText ? " Non-retryable because output was already streamed." : " Treating as a stream stall and retrying."}`,
errorMessage: `${THINKING_LOOP_ERROR_MARKER}: the model repeated near-identical content (${detail}). Treating as a stream stall and retrying.`,
timestamp: Date.now(),
};
}
@@ -345,14 +343,15 @@ function detectVerbatimRepetition(text: string): [unit: string, count: number] |
return null;
}
/** Lowercase, drop code spans / paths / digits, keep only letter words. */
/** Lowercase and tokenize prose plus code/path payloads, dropping pure numbers. */
function normalizeSegment(segment: string): string {
return segment
.toLowerCase()
.replace(/`[^`]*`/g, " ")
.replace(/\/[^\s`]+/g, " ")
.replace(/\d+/g, " ")
.replace(/[^a-z]+/g, " ")
.replace(/`([^`]*)`/g, " $1 ")
.replace(/[^a-z0-9]+/g, " ")
.split(/\s+/)
.filter(token => /[a-z]/.test(token))
.join(" ")
.trim();
}
+75 -7
View File
@@ -2,7 +2,7 @@ import { describe, expect, test } from "bun:test";
import { clearCustomApis } from "@oh-my-pi/pi-ai/api-registry";
import { createMockModel, type MockContent, registerMockApi } from "@oh-my-pi/pi-ai/providers/mock";
import { stream, streamSimple } from "@oh-my-pi/pi-ai/stream";
import type { Api, AssistantMessage, AssistantMessageEvent, Context, Model, TextContent } from "@oh-my-pi/pi-ai/types";
import type { Api, AssistantMessage, AssistantMessageEvent, Context, Model } from "@oh-my-pi/pi-ai/types";
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
import {
isGeminiThinkingLoopModel,
@@ -120,6 +120,75 @@ describe("ThinkingLoopDetector", () => {
expect(detail).toBeNull();
});
test("does not collapse distinct per-file assignment templates into a loop", () => {
const detector = new ThinkingLoopDetector();
const text = [
`1. Subagent ApprovalModeTest:
- target: packages/coding-agent/test/tools/approval-mode.test.ts
- role: "Test-file refactoring specialist"
- assignment:
\`\`\`markdown
# Target
packages/coding-agent/test/tools/approval-mode.test.ts
# Change
Replace the type annotation:
\`let session: Awaited<ReturnType<typeof createAgentSession>>["session"];\`
with \`AgentSession\`.
Verify where \`AgentSession\` is imported from.
Run \`biome check --write --unsafe\` on the file.
\`\`\``,
`2. Subagent GhTest:
- target: packages/coding-agent/test/tools/gh.test.ts
- role: "Test-file refactoring specialist"
- assignment:
\`\`\`markdown
# Target
packages/coding-agent/test/tools/gh.test.ts
# Change
Replace the type annotations:
\`let tempHome: Awaited<ReturnType<typeof setupTempHome>>;\`
with \`TempDir\`.
Verify where \`TempDir\` is imported from.
Run \`biome check --write --unsafe\` on the file.
\`\`\``,
`3. Subagent TodoTest:
- target: packages/coding-agent/test/tools/todo.test.ts
- role: "Test-file refactoring specialist"
- assignment:
\`\`\`markdown
# Target
packages/coding-agent/test/tools/todo.test.ts
# Change
Locate line 438 containing:
\`function innerLines(component: ReturnType<typeof todoToolRenderer.renderResult>): string[] {\`
Replace \`ReturnType<typeof todoToolRenderer.renderResult>\` with the explicit return type.
Run \`biome check --write --unsafe\` on the file.
\`\`\``,
`4. Subagent HookEditorTest:
- target: packages/coding-agent/test/hook-editor.test.ts
- role: "Test-file refactoring specialist"
- assignment:
\`\`\`markdown
# Target
packages/coding-agent/test/hook-editor.test.ts
# Change
Replace \`setFocus: ReturnType<typeof vi.fn>;\` and \`requestRender: ReturnType<typeof vi.fn>;\`
with Bun's explicit mock type from \`bun:test\`.
Run \`biome check --write --unsafe\` on the file.
\`\`\``,
].join("\n\n");
let detail: string | null = null;
for (let i = 0; i < text.length && !detail; i += 37) {
detail = detector.push(text.slice(i, i + 37));
}
detail ??= detector.flush();
expect(detail).toBeNull();
});
test("flush() catches a final unterminated duplicate paragraph", () => {
const detector = new ThinkingLoopDetector();
// Seven blank-line-separated dupes leave the eighth (cluster-completing)
@@ -326,13 +395,12 @@ describe("loop guard assistant prose/text loops", () => {
const result = await guarded.result();
expect(result.stopReason).toBe("error");
// Content must hold the text streamed BEFORE the loop detector tripped.
expect(result.content.length).toBe(1);
expect(result.content[0].type).toBe("text");
expect((result.content[0] as TextContent).text).toContain("First healthy text");
// Loop-guard output is replay garbage even when it came through text_delta:
// drop it so AgentSession can retry with a clean assistant turn.
expect(result.content).toEqual([]);
expect(result.errorMessage).toContain(THINKING_LOOP_ERROR_MARKER);
// Since some text was forwarded, it is replay-unsafe, so isRetryableError should return false.
expect(isRetryableError(new Error(result.errorMessage))).toBe(false);
expect(result.errorMessage).toContain("stream stall");
expect(isRetryableError(new Error(result.errorMessage))).toBe(true);
});
test("does not trip on assistant text loop when checkAssistantContent is false", async () => {
+3
View File
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
### Changed
- Refactored Perplexity authentication logic to prioritize cookies over OAuth in search operations
@@ -8,6 +9,8 @@
### Fixed
- Enabled auto-retry for AI "thinking loop" errors encountered during model inference
- Cleared stale error banners automatically when triggered by an auto-retry recovery phase
- Preserved bundled `omitMaxOutputTokens` policy when fresh cached provider discovery rows replace Ollama Cloud catalog models, so stale `models.db` entries cannot re-enable context-window-sized `num_predict` values. ([#2984](https://github.com/can1357/oh-my-pi/issues/2984))
- Normalized cached-only Ollama Cloud discovery rows to omit on-the-wire output-token caps even when the cached model id has no bundled catalog entry. ([#2984](https://github.com/can1357/oh-my-pi/issues/2984))
- Fixed Ollama, LM Studio, and llama.cpp (plus loopback vLLM / sglang servers) reprocessing the full prompt on every turn because `provider.appendOnlyContext: auto` only recognized DeepSeek and Xiaomi as prefix-cache providers. The auto-detect now enables append-only mode for `ollama`, `ollama-cloud`, `lm-studio`, `llama.cpp`, and any baseUrl resolving to a loopback/RFC1918/`.local` host, so the system prompt + tool catalogue + prior-turn message bytes stay byte-stable across turns and llama.cpp's KV-cache prefix reuse can hit ([#3033](https://github.com/can1357/oh-my-pi/issues/3033)).
@@ -1,4 +1,5 @@
import type { ImageContent } from "@oh-my-pi/pi-ai";
import { THINKING_LOOP_ERROR_MARKER } from "@oh-my-pi/pi-ai/utils/thinking-loop";
import { type Component, Loader, TERMINAL } from "@oh-my-pi/pi-tui";
import { INTENT_FIELD } from "@oh-my-pi/pi-wire";
import { extractTextContent } from "../../commit/utils";
@@ -1014,6 +1015,13 @@ export class EventController {
async #handleAutoRetryStart(event: Extract<AgentSessionEvent, { type: "auto_retry_start" }>): Promise<void> {
this.#stopWorkingLoader();
this.ctx.statusContainer.clear();
if (event.errorMessage?.includes(THINKING_LOOP_ERROR_MARKER)) {
// The retry path drops the failed assistant from runtime context. Do not
// restore its inline Error row; just unpin the fixed-region banner so the
// retry UI is the visible state.
this.#pinnedErrorComponent = undefined;
this.ctx.clearPinnedError();
}
const delaySeconds = Math.round(event.delayMs / 1000);
this.ctx.retryLoader = new Loader(
this.ctx.ui,
@@ -104,6 +104,7 @@ import {
streamSimple,
} from "@oh-my-pi/pi-ai";
import { stripToolDescriptions } from "@oh-my-pi/pi-ai/utils/schema";
import { THINKING_LOOP_ERROR_MARKER } from "@oh-my-pi/pi-ai/utils/thinking-loop";
import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking";
import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models";
import { MacOSPowerAssertion } from "@oh-my-pi/pi-natives";
@@ -9983,6 +9984,7 @@ export class AgentSession {
if (this.#isProviderErrorFinishReasonBeforeToolUse(message)) return true;
if (this.#isMalformedFunctionCallError(message)) return true;
if (this.#hasReplayUnsafeToolOutput(message)) return false;
if (message.errorMessage.includes(THINKING_LOOP_ERROR_MARKER)) return true;
if (this.#isStaleOpenAIResponsesReplayError(message)) return true;
const err = message.errorMessage;
@@ -0,0 +1,247 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import * as path from "node:path";
import { scheduler } from "node:timers/promises";
import { Agent } from "@oh-my-pi/pi-agent-core";
import type {
Api,
AssistantMessage,
Context,
Model,
SimpleStreamOptions,
TextContent,
ThinkingContent,
} from "@oh-my-pi/pi-ai";
import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock";
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
import { THINKING_LOOP_ERROR_MARKER, withGeminiThinkingLoopGuard } from "@oh-my-pi/pi-ai/utils/thinking-loop";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { AgentSession, type AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { TempDir } from "@oh-my-pi/pi-utils";
const LOOP_PARAGRAPHS = [
"I am now verifying the test module to guarantee there are no compile errors and the code is completely safe.",
"I am now verifying the test module once more to ensure there are no compile errors and the code stays completely safe.",
"I am now re-verifying the test module to confirm there are no compile errors and the code remains completely safe.",
];
function emptyUsage(): AssistantMessage["usage"] {
return {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
}
function chunkedThinkingLoopStream(model: Model<Api>, options?: SimpleStreamOptions): AssistantMessageEventStream {
const inner = new AssistantMessageEventStream();
queueMicrotask(() => {
const thinking: ThinkingContent = { type: "thinking", thinking: "" };
const partial: AssistantMessage = {
role: "assistant",
content: [thinking],
api: model.api,
provider: model.provider,
model: model.id,
usage: emptyUsage(),
stopReason: "stop",
timestamp: Date.now(),
};
inner.push({ type: "start", partial });
inner.push({ type: "thinking_start", contentIndex: 0, partial });
for (let index = 0; index < 12; index++) {
if (options?.signal?.aborted) return;
const delta = `**Confirming Safety ${index}**\n\n${LOOP_PARAGRAPHS[index % LOOP_PARAGRAPHS.length]}\n\n\n`;
thinking.thinking += delta;
inner.push({ type: "thinking_delta", contentIndex: 0, delta, partial });
}
inner.push({ type: "thinking_end", contentIndex: 0, content: thinking.thinking, partial });
inner.push({ type: "done", reason: "stop", message: partial });
});
return withGeminiThinkingLoopGuard(model, options, () => inner);
}
function successStream(model: Model<Api>): AssistantMessageEventStream {
const stream = new AssistantMessageEventStream();
queueMicrotask(() => {
const text: TextContent = { type: "text", text: "Recovered after retry." };
const partial: AssistantMessage = {
role: "assistant",
content: [text],
api: model.api,
provider: model.provider,
model: model.id,
usage: emptyUsage(),
stopReason: "stop",
timestamp: Date.now(),
};
stream.push({ type: "start", partial });
stream.push({ type: "text_start", contentIndex: 0, partial });
stream.push({ type: "text_delta", contentIndex: 0, delta: text.text, partial });
stream.push({ type: "text_end", contentIndex: 0, content: text.text, partial });
stream.push({ type: "done", reason: "stop", message: partial });
});
return stream;
}
function legacyContentfulLoopErrorStream(model: Model<Api>): AssistantMessageEventStream {
const stream = new AssistantMessageEventStream();
queueMicrotask(() => {
const text: TextContent = { type: "text", text: "Looping visible reasoning garbage." };
const partial: AssistantMessage = {
role: "assistant",
content: [text],
api: model.api,
provider: model.provider,
model: model.id,
usage: emptyUsage(),
stopReason: "error",
errorMessage: `${THINKING_LOOP_ERROR_MARKER}: the model repeated near-identical content. Non-retryable because output was already streamed.`,
timestamp: Date.now(),
};
stream.push({ type: "start", partial });
stream.push({ type: "text_start", contentIndex: 0, partial });
stream.push({ type: "text_delta", contentIndex: 0, delta: text.text, partial });
stream.push({ type: "text_end", contentIndex: 0, content: text.text, partial });
stream.push({ type: "error", reason: "error", error: partial });
});
return stream;
}
describe("AgentSession thinking-loop retry", () => {
let tempDir: TempDir;
let authStorage: AuthStorage;
let session: AgentSession | undefined;
beforeEach(async () => {
tempDir = TempDir.createSync("@pi-thinking-loop-retry-");
authStorage = await AuthStorage.create(path.join(tempDir.path(), "auth.db"));
authStorage.setRuntimeApiKey("openrouter", "openrouter-test-key");
});
afterEach(async () => {
if (session) {
await session.dispose();
session = undefined;
}
authStorage.close();
tempDir.removeSync();
vi.restoreAllMocks();
});
it("drops a chunked thinking-loop error and retries the turn", async () => {
const model = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }).model;
const modelRegistry = new ModelRegistry(authStorage);
const calls: string[] = [];
const agent = new Agent({
getApiKey: requestedModel => `${requestedModel.provider}-test-key`,
initialState: {
model,
systemPrompt: ["Test"],
tools: [],
messages: [],
},
streamFn: (requestedModel, _context: Context, options?: SimpleStreamOptions) => {
calls.push(`${requestedModel.provider}/${requestedModel.id}`);
return calls.length === 1
? chunkedThinkingLoopStream(requestedModel, options)
: successStream(requestedModel);
},
});
const settings = Settings.isolated({
"compaction.enabled": false,
"retry.enabled": true,
"retry.baseDelayMs": 0,
"retry.maxDelayMs": 5_000,
"retry.maxRetries": 1,
"retry.modelFallback": false,
"todo.enabled": false,
});
settings.setModelRole("default", `${model.provider}/${model.id}`);
session = new AgentSession({
agent,
sessionManager: SessionManager.inMemory(),
settings,
modelRegistry,
});
vi.spyOn(scheduler, "wait").mockResolvedValue(undefined);
const retryStartEvents: Array<Extract<AgentSessionEvent, { type: "auto_retry_start" }>> = [];
const retryEndEvents: Array<Extract<AgentSessionEvent, { type: "auto_retry_end" }>> = [];
session.subscribe(event => {
if (event.type === "auto_retry_start") retryStartEvents.push(event);
if (event.type === "auto_retry_end") retryEndEvents.push(event);
});
await session.prompt("Trigger thinking loop once");
await session.waitForIdle();
expect(calls).toEqual(["openrouter/google/gemini-3.5-flash", "openrouter/google/gemini-3.5-flash"]);
expect(retryStartEvents).toHaveLength(1);
expect(retryStartEvents[0].errorMessage).toContain(THINKING_LOOP_ERROR_MARKER);
expect(retryEndEvents).toEqual([{ type: "auto_retry_end", success: true, attempt: 1 }]);
const assistants = session.agent.state.messages.filter(
(message): message is AssistantMessage => message.role === "assistant",
);
expect(assistants).toHaveLength(1);
expect(assistants[0].stopReason).toBe("stop");
expect(assistants[0].content).toEqual([{ type: "text", text: "Recovered after retry." }]);
expect(assistants[0].errorMessage).toBeUndefined();
});
it("starts retry for loop-marker errors even without transient wording", async () => {
const model = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }).model;
const modelRegistry = new ModelRegistry(authStorage);
const calls: string[] = [];
const agent = new Agent({
getApiKey: requestedModel => `${requestedModel.provider}-test-key`,
initialState: {
model,
systemPrompt: ["Test"],
tools: [],
messages: [],
},
streamFn: requestedModel => {
calls.push(`${requestedModel.provider}/${requestedModel.id}`);
return calls.length === 1 ? legacyContentfulLoopErrorStream(requestedModel) : successStream(requestedModel);
},
});
const settings = Settings.isolated({
"compaction.enabled": false,
"retry.enabled": true,
"retry.baseDelayMs": 0,
"retry.maxDelayMs": 5_000,
"retry.maxRetries": 1,
"retry.modelFallback": false,
"todo.enabled": false,
});
settings.setModelRole("default", `${model.provider}/${model.id}`);
session = new AgentSession({
agent,
sessionManager: SessionManager.inMemory(),
settings,
modelRegistry,
});
vi.spyOn(scheduler, "wait").mockResolvedValue(undefined);
const retryStartEvents: Array<Extract<AgentSessionEvent, { type: "auto_retry_start" }>> = [];
session.subscribe(event => {
if (event.type === "auto_retry_start") retryStartEvents.push(event);
});
await session.prompt("Trigger legacy loop marker once");
await session.waitForIdle();
expect(calls).toEqual(["openrouter/google/gemini-3.5-flash", "openrouter/google/gemini-3.5-flash"]);
expect(retryStartEvents).toHaveLength(1);
expect(retryStartEvents[0].errorMessage).toContain("Non-retryable because output was already streamed");
const assistants = session.agent.state.messages.filter(
(message): message is AssistantMessage => message.role === "assistant",
);
expect(assistants).toHaveLength(1);
expect(assistants[0].content).toEqual([{ type: "text", text: "Recovered after retry." }]);
});
});
@@ -9,6 +9,7 @@
*/
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import type { AssistantMessage } from "@oh-my-pi/pi-ai";
import { THINKING_LOOP_ERROR_MARKER } from "@oh-my-pi/pi-ai/utils/thinking-loop";
import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { AssistantMessageComponent } from "@oh-my-pi/pi-coding-agent/modes/components/assistant-message";
import { ErrorBannerComponent } from "@oh-my-pi/pi-coding-agent/modes/components/error-banner";
@@ -60,15 +61,22 @@ function createFixture(streamingMessage?: AssistantMessage) {
};
const showPinnedError = vi.fn();
const clearPinnedError = vi.fn();
const statusContainer = {
clear: vi.fn(),
addChild: vi.fn(),
};
const session = { isTtsrAbortPending: false, retryAttempt: 0 };
const ctx = {
isInitialized: true,
init: vi.fn(async () => {}),
ui: { requestRender: vi.fn() },
ui: { requestRender: vi.fn(), requestComponentRender: vi.fn() },
statusLine: { invalidate: vi.fn() },
updateEditorTopBorder: vi.fn(),
ensureLoadingAnimation: vi.fn(),
statusContainer,
loadingAnimation: undefined,
retryLoader: undefined,
editor: {},
streamingComponent: streamingMessage ? streamingComponent : undefined,
streamingMessage,
@@ -121,6 +129,35 @@ describe("EventController error banner", () => {
expect(streamingComponent.setErrorPinned).toHaveBeenCalledWith(false);
});
it("clears retryable thinking-loop banners without restoring the dropped inline error", async () => {
const errorMessage = `${THINKING_LOOP_ERROR_MARKER}: the model repeated near-identical content. Treating as a stream stall and retrying.`;
const message = makeAssistantMessage({ stopReason: "error", errorMessage });
const { controller, clearPinnedError, streamingComponent } = createFixture(message);
await controller.handleEvent({ type: "message_end", message } as Extract<
AgentSessionEvent,
{ type: "message_end" }
>);
clearPinnedError.mockClear();
streamingComponent.setErrorPinned.mockClear();
await controller.handleEvent({
type: "auto_retry_start",
attempt: 1,
maxAttempts: 2,
delayMs: 0,
errorMessage,
} as Extract<AgentSessionEvent, { type: "auto_retry_start" }>);
expect(clearPinnedError).toHaveBeenCalledTimes(1);
expect(streamingComponent.setErrorPinned).not.toHaveBeenCalledWith(false);
await controller.handleEvent({
type: "auto_retry_end",
success: true,
attempt: 1,
} as Extract<AgentSessionEvent, { type: "auto_retry_end" }>);
});
it("does not pin a banner for a normal assistant stop", async () => {
const message = makeAssistantMessage({ stopReason: "stop" });
const { controller, showPinnedError } = createFixture(message);