fix: fixed Gemini thinking-loop handling across streaming dispatch paths
- Added Gemini thinking-loop detection helpers for near-duplicate and verbatim output checks. - Wrapped `stream`, `streamPiNative`, and `streamSimple` dispatches with the loop guard. - Emitted retryable empty-content loop errors and stopped completion events on loop hits. - Added `enableGeminiThinkingLoopGuard` options for OpenAI compatibility with Gemini defaults and overrides.
This commit is contained in:
@@ -14,6 +14,10 @@
|
||||
|
||||
- Added agent-loop deadline support for graceful wall-clock session stops.
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed Gemini repetition-loop detection to live in the pi-ai stream layer instead of the agent loop. The agent no longer runs its own Gemini-gated verbatim repetition check (`detectRepetition`/`truncateRepetition`); loops now surface as a retryable transient stream error that the standard auto-retry path discards and re-samples, rather than a committed contentful error message.
|
||||
|
||||
## [16.0.1] - 2026-06-15
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -35,7 +35,7 @@ import {
|
||||
signalListLabel,
|
||||
} from "@oh-my-pi/pi-ai/utils/harmony-leak";
|
||||
import { preferredDialect } from "@oh-my-pi/pi-catalog/identity";
|
||||
import { logger, sanitizeText } from "@oh-my-pi/pi-utils";
|
||||
import { sanitizeText } from "@oh-my-pi/pi-utils";
|
||||
import { type AgentRunCoverage, type AgentRunSummary, ToolCallBlockedError } from "./run-collector";
|
||||
import {
|
||||
type AgentTelemetry,
|
||||
@@ -1054,7 +1054,6 @@ async function streamAssistantResponse(
|
||||
? AbortSignal.any([signal, harmonyAbortController.signal])
|
||||
: harmonyAbortController.signal
|
||||
: signal;
|
||||
const repetitionAbortController = new AbortController();
|
||||
// Owned tool calling: aborted by the stream wrapper when the model starts
|
||||
// fabricating a `<tool_response>`, so the provider stops generating the rest of
|
||||
// the hallucinated turn. Merged into the provider signal ONLY (not
|
||||
@@ -1063,10 +1062,13 @@ async function streamAssistantResponse(
|
||||
const promptToolAbortController = ownedDialect ? new AbortController() : undefined;
|
||||
const providerAbortSignals: AbortSignal[] = [];
|
||||
if (requestSignal) providerAbortSignals.push(requestSignal);
|
||||
providerAbortSignals.push(repetitionAbortController.signal);
|
||||
if (promptToolAbortController) providerAbortSignals.push(promptToolAbortController.signal);
|
||||
const finalRequestSignal =
|
||||
providerAbortSignals.length === 1 ? providerAbortSignals[0]! : AbortSignal.any(providerAbortSignals);
|
||||
providerAbortSignals.length === 0
|
||||
? undefined
|
||||
: providerAbortSignals.length === 1
|
||||
? providerAbortSignals[0]!
|
||||
: AbortSignal.any(providerAbortSignals);
|
||||
const requestApiKey = (config.getApiKey ? await config.getApiKey(config.model) : undefined) ?? config.apiKey;
|
||||
const resolvedApiKey = await resolveApiKeyOnce(requestApiKey, finalRequestSignal);
|
||||
const apiKey = isApiKeyResolver(requestApiKey) ? seedApiKeyResolver(resolvedApiKey, requestApiKey) : requestApiKey;
|
||||
@@ -1173,56 +1175,6 @@ async function streamAssistantResponse(
|
||||
return aborted;
|
||||
};
|
||||
|
||||
const finishRepetitionStream = async (
|
||||
kind: "text" | "thinking",
|
||||
pattern: string,
|
||||
count: number,
|
||||
): Promise<AssistantMessage> => {
|
||||
repetitionAbortController.abort();
|
||||
try {
|
||||
const cleanup = responseIterator.return?.();
|
||||
if (cleanup) void cleanup.catch(() => {});
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
if (partialMessage) {
|
||||
truncateRepetition(partialMessage, kind, pattern);
|
||||
partialMessage.stopReason = "error";
|
||||
partialMessage.errorMessage = `Repetition loop detected: assistant repeated "${pattern.trim()}" ${count} times consecutively.`;
|
||||
}
|
||||
const finalMsg = snapshotAssistantMessage(
|
||||
partialMessage ?? {
|
||||
role: "assistant",
|
||||
content: [],
|
||||
api: config.model.api,
|
||||
provider: config.model.provider,
|
||||
model: config.model.id,
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "error",
|
||||
errorMessage: `Repetition loop detected.`,
|
||||
timestamp: Date.now(),
|
||||
},
|
||||
);
|
||||
if (addedPartial) {
|
||||
context.messages[context.messages.length - 1] = finalMsg;
|
||||
} else {
|
||||
context.messages.push(finalMsg);
|
||||
}
|
||||
if (!addedPartial) {
|
||||
stream.push({ type: "message_start", message: snapshotAssistantMessage(finalMsg) });
|
||||
}
|
||||
stream.push({ type: "message_end", message: snapshotAssistantMessage(finalMsg) });
|
||||
await finishChat(finalMsg);
|
||||
return finalMsg;
|
||||
};
|
||||
|
||||
// Set up a single abort race: register the abort listener once for the whole
|
||||
// stream and reuse the same race promise for every iterator.next() instead of
|
||||
// allocating Promise.withResolvers and add/removeEventListener per event.
|
||||
@@ -1239,14 +1191,6 @@ async function streamAssistantResponse(
|
||||
detachAbortListener = () => requestSignal.removeEventListener("abort", onAbort);
|
||||
}
|
||||
|
||||
// Rolling tail of streamed text/thinking used for repetition-loop detection.
|
||||
// Bounded to REPETITION_WINDOW chars and reset when the active block kind
|
||||
// switches (text <-> thinking) so detection stays O(1) per delta and never
|
||||
// miscounts a repeated unit across a thinking/answer boundary.
|
||||
let repetitionTail = "";
|
||||
let repetitionKind: "text" | "thinking" | undefined;
|
||||
const isGeminiModel = config.model.provider.includes("google") || config.model.provider.includes("gemini");
|
||||
|
||||
try {
|
||||
while (true) {
|
||||
let next: IteratorResult<AssistantMessageEvent>;
|
||||
@@ -1331,27 +1275,6 @@ async function streamAssistantResponse(
|
||||
assistantMessageEvent: snapshotAssistantMessageEvent(event),
|
||||
message: snapshotAssistantMessage(partialMessage),
|
||||
});
|
||||
|
||||
if (isGeminiModel && (event.type === "text_delta" || event.type === "thinking_delta")) {
|
||||
const kind = event.type === "text_delta" ? "text" : "thinking";
|
||||
if (repetitionKind !== kind) {
|
||||
repetitionKind = kind;
|
||||
repetitionTail = "";
|
||||
}
|
||||
repetitionTail += event.delta;
|
||||
if (repetitionTail.length > REPETITION_WINDOW) {
|
||||
repetitionTail = repetitionTail.slice(-REPETITION_WINDOW);
|
||||
}
|
||||
const repetition = detectRepetition(repetitionTail);
|
||||
if (repetition) {
|
||||
const [pattern, count] = repetition;
|
||||
logger.warn("Repetition loop detected during assistant stream, aborting.", {
|
||||
pattern,
|
||||
count,
|
||||
});
|
||||
return await finishRepetitionStream(kind, pattern, count);
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -1987,97 +1910,3 @@ function createSkippedToolResult(): AgentToolResult<any> {
|
||||
details: {},
|
||||
};
|
||||
}
|
||||
|
||||
const REPETITION_WINDOW = 250;
|
||||
const REPETITION_MIN_REPEATED_CHARS = 180;
|
||||
|
||||
function detectRepetition(text: string): [pattern: string, count: number] | null {
|
||||
if (text.length < REPETITION_MIN_REPEATED_CHARS) return null;
|
||||
|
||||
const windowSize = Math.min(text.length, REPETITION_WINDOW);
|
||||
const searchSpace = text.slice(-windowSize);
|
||||
|
||||
for (let len = 2; len <= 60; len++) {
|
||||
if (searchSpace.length < len * 4) continue;
|
||||
|
||||
const pattern = searchSpace.slice(-len);
|
||||
// Only treat a repeated unit as a pathological loop when it carries real
|
||||
// linguistic content (a letter or a pictographic emoji). Runs made purely of
|
||||
// digits, whitespace or punctuation are legitimate in tabular / hex / numeric
|
||||
// output (e.g. "00 00 00", "0, 0, 0", "| -- | -- |") and must not trip.
|
||||
if (!/[\p{L}\p{Extended_Pictographic}]/u.test(pattern)) continue;
|
||||
|
||||
let count = 0;
|
||||
let pos = searchSpace.length;
|
||||
while (pos >= len) {
|
||||
const chunk = searchSpace.slice(pos - len, pos);
|
||||
if (chunk === pattern) {
|
||||
count++;
|
||||
pos -= len;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (count >= 4 && len * count >= REPETITION_MIN_REPEATED_CHARS) {
|
||||
return [pattern, count];
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function truncateRepetition(message: AssistantMessage, kind: "text" | "thinking", pattern: string): void {
|
||||
// A repetition loop streams into a single growing block (real providers) or a run
|
||||
// of same-kind blocks (some transports), always at the tail of the message. Gather
|
||||
// that trailing contiguous run and collapse its repeated copies down to one, so the
|
||||
// committed transcript keeps a representative sample instead of the full runaway.
|
||||
const matches = (block: AssistantContentBlock): boolean =>
|
||||
kind === "text" ? block.type === "text" : block.type === "thinking";
|
||||
const readBlock = (block: AssistantContentBlock): string =>
|
||||
block.type === "text" ? block.text : block.type === "thinking" ? block.thinking : "";
|
||||
const clearThinkingReplayAnchors = (block: AssistantContentBlock): void => {
|
||||
if (block.type !== "thinking") return;
|
||||
block.thinkingSignature = undefined;
|
||||
block.itemId = undefined;
|
||||
};
|
||||
const writeBlock = (block: AssistantContentBlock, value: string): void => {
|
||||
if (block.type === "text") {
|
||||
block.text = value;
|
||||
} else if (block.type === "thinking") {
|
||||
block.thinking = value;
|
||||
clearThinkingReplayAnchors(block);
|
||||
}
|
||||
};
|
||||
|
||||
const trailing: AssistantContentBlock[] = [];
|
||||
for (let i = message.content.length - 1; i >= 0; i--) {
|
||||
const block = message.content[i];
|
||||
if (!matches(block)) break;
|
||||
trailing.unshift(block);
|
||||
}
|
||||
if (trailing.length === 0) return;
|
||||
if (kind === "thinking") {
|
||||
for (const block of trailing) clearThinkingReplayAnchors(block);
|
||||
}
|
||||
|
||||
let joined = "";
|
||||
for (const block of trailing) joined += readBlock(block);
|
||||
|
||||
let kept = joined;
|
||||
while (kept.length >= pattern.length * 2 && kept.slice(kept.length - pattern.length * 2) === pattern + pattern) {
|
||||
kept = kept.slice(0, kept.length - pattern.length);
|
||||
}
|
||||
|
||||
let remainingToRemove = joined.length - kept.length;
|
||||
for (let i = trailing.length - 1; i >= 0 && remainingToRemove > 0; i--) {
|
||||
const block = trailing[i];
|
||||
const value = readBlock(block);
|
||||
if (value.length <= remainingToRemove) {
|
||||
remainingToRemove -= value.length;
|
||||
writeBlock(block, "");
|
||||
} else {
|
||||
writeBlock(block, value.slice(0, value.length - remainingToRemove));
|
||||
remainingToRemove = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1879,126 +1879,6 @@ describe("agentLoopContinue with AgentMessage", () => {
|
||||
}
|
||||
});
|
||||
|
||||
it("should detect repetition loops during assistant stream and abort gracefully", async () => {
|
||||
const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] };
|
||||
const mock = createMockModel({
|
||||
provider: "google-gemini-cli",
|
||||
responses: [
|
||||
{
|
||||
content: Array.from({ length: 80 }, () => "🌊 "),
|
||||
},
|
||||
],
|
||||
});
|
||||
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter };
|
||||
|
||||
const stream = agentLoop([createUserMessage("Hello")], context, config, undefined, mock.stream);
|
||||
for await (const _event of stream) {
|
||||
// drain the stream to completion
|
||||
}
|
||||
|
||||
const messages = await stream.result();
|
||||
expect(messages.length).toBe(2);
|
||||
expect(messages[1].role).toBe("assistant");
|
||||
|
||||
const assistantMsg = messages[1] as AssistantMessage;
|
||||
expect(assistantMsg.stopReason).toBe("error");
|
||||
expect(assistantMsg.errorMessage).toContain("Repetition loop detected");
|
||||
|
||||
let text = "";
|
||||
for (const block of assistantMsg.content) {
|
||||
if (block.type === "text") text += block.text;
|
||||
}
|
||||
expect(text).toBe("🌊 ");
|
||||
});
|
||||
|
||||
it("detects and truncates repetition loops inside a thinking stream", async () => {
|
||||
const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] };
|
||||
const mock = createMockModel({
|
||||
provider: "google-gemini-cli",
|
||||
responses: [
|
||||
{
|
||||
content: Array.from({ length: 80 }, (_, index) => ({
|
||||
type: "thinking" as const,
|
||||
thinking: "🌊 ",
|
||||
thinkingSignature: `signature-${index}`,
|
||||
itemId: `rs_${index}`,
|
||||
})),
|
||||
},
|
||||
],
|
||||
});
|
||||
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter };
|
||||
|
||||
const stream = agentLoop([createUserMessage("Hello")], context, config, undefined, mock.stream);
|
||||
for await (const _event of stream) {
|
||||
// drain the stream to completion
|
||||
}
|
||||
|
||||
const assistantMsg = (await stream.result())[1] as AssistantMessage;
|
||||
expect(assistantMsg.stopReason).toBe("error");
|
||||
expect(assistantMsg.errorMessage).toContain("Repetition loop detected");
|
||||
|
||||
// A looping thinking stream must be both detected AND collapsed to a single
|
||||
// representative copy — not committed to the transcript in full.
|
||||
let thinking = "";
|
||||
for (const block of assistantMsg.content) {
|
||||
if (block.type === "thinking") thinking += block.thinking;
|
||||
}
|
||||
expect(thinking).toBe("🌊 ");
|
||||
for (const block of assistantMsg.content) {
|
||||
if (block.type === "thinking") {
|
||||
expect(block.thinkingSignature).toBeUndefined();
|
||||
expect(block.itemId).toBeUndefined();
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
it("does not flag short requested repetitive text as a loop", async () => {
|
||||
const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] };
|
||||
const repeated = "🌊 ".repeat(26);
|
||||
const mock = createMockModel({
|
||||
provider: "google-gemini-cli",
|
||||
responses: [{ content: [repeated] }],
|
||||
});
|
||||
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter };
|
||||
|
||||
const stream = agentLoop([createUserMessage("print 26 wave emoji")], context, config, undefined, mock.stream);
|
||||
for await (const _event of stream) {
|
||||
// drain the stream to completion
|
||||
}
|
||||
|
||||
const assistantMsg = (await stream.result())[1] as AssistantMessage;
|
||||
expect(assistantMsg.stopReason).not.toBe("error");
|
||||
let text = "";
|
||||
for (const block of assistantMsg.content) {
|
||||
if (block.type === "text") text += block.text;
|
||||
}
|
||||
expect(text).toBe(repeated);
|
||||
});
|
||||
|
||||
it("does not flag legitimate repetitive numeric output as a loop", async () => {
|
||||
const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] };
|
||||
// A hexdump of zero-filled memory is highly repetitive but legitimate; the
|
||||
// detector must not classify pure digit/whitespace runs as a loop.
|
||||
const hexdump = "00 ".repeat(80);
|
||||
const mock = createMockModel({
|
||||
provider: "google-gemini-cli",
|
||||
responses: [{ content: [hexdump] }],
|
||||
});
|
||||
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter };
|
||||
|
||||
const stream = agentLoop([createUserMessage("dump the zero page")], context, config, undefined, mock.stream);
|
||||
for await (const _event of stream) {
|
||||
// drain the stream to completion
|
||||
}
|
||||
|
||||
const assistantMsg = (await stream.result())[1] as AssistantMessage;
|
||||
expect(assistantMsg.stopReason).not.toBe("error");
|
||||
let text = "";
|
||||
for (const block of assistantMsg.content) {
|
||||
if (block.type === "text") text += block.text;
|
||||
}
|
||||
expect(text).toBe(hexdump);
|
||||
});
|
||||
it("aborts pending tool calls instead of running them when the deadline is crossed during the request", async () => {
|
||||
const context: AgentContext = {
|
||||
systemPrompt: ["You are helpful."],
|
||||
|
||||
@@ -1,16 +1,17 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added `antigravityEndpointMode` stream option with `auto`, `production`, and `sandbox` values to control Antigravity endpoint routing
|
||||
- Added `seedApiKeyResolver` for reusing a pre-resolved request key while preserving resolver-driven auth retry and credential rotation
|
||||
- Added optional `contextSnapshot` property to `AssistantMessage` with token usage metadata via new `ContextSnapshot` interface (`promptTokens`, `nonMessageTokens`, and optional `lastMessageTimestamp`)
|
||||
- Added `LITELLM_BASE_URL` guidance to the LiteLLM login prompt so non-default proxy endpoints are discoverable. ([#2726](https://github.com/can1357/oh-my-pi/issues/2726))
|
||||
- Added a Gemini thinking-loop guard that watches streamed `thinking` deltas for degenerate reasoning loops — verbatim tail repetition and near-duplicate paragraph cycling — and terminates the stream with a retryable, empty-content `error` message (worded as a transient stream stall) so the turn is discarded and re-sampled instead of committing a runaway transcript. Gated to Gemini models across every transport (OpenRouter, direct Google, Vertex) and disarmed once visible answer text or a tool call starts; disable with `PI_NO_THINKING_LOOP_GUARD=1`.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed `streamSimple()` Gemini streams to run through the thinking-loop guard for custom API and pi-native transports, so degenerate `thinking` loops now abort with the same retryable empty-content error path as other Gemini stream paths
|
||||
- Fixed Antigravity model streaming and usage fetch paths to retry on transient `429`/`5xx` errors by failing over to the alternate endpoint before surfacing an error
|
||||
- Fixed Antigravity endpoint tracking to prefer a previously successful endpoint in `auto` mode for subsequent requests
|
||||
- Fixed Antigravity and Gemini CLI model requests failing with an opaque error when Google requires account verification. Cloud Code Assist `403 VALIDATION_REQUIRED` responses now surface the `validation_url` and the signed-in account email when available, so users see an actionable account-verification message instead of the raw API error body.
|
||||
|
||||
@@ -41,4 +41,5 @@ export * from "./utils/event-stream";
|
||||
export * from "./utils/overflow";
|
||||
export * from "./utils/retry";
|
||||
export * from "./utils/schema";
|
||||
export * from "./utils/thinking-loop";
|
||||
export * from "./utils/validation";
|
||||
|
||||
@@ -64,6 +64,7 @@ import type {
|
||||
} from "./types";
|
||||
import { AssistantMessageEventStream } from "./utils/event-stream";
|
||||
import { withRequestDebugFetch } from "./utils/request-debug";
|
||||
import { withGeminiThinkingLoopGuard } from "./utils/thinking-loop";
|
||||
|
||||
function isGoogleVertexAuthenticatedModel(model: Model<Api>): boolean {
|
||||
return (
|
||||
@@ -227,6 +228,14 @@ export function stream<TApi extends Api>(
|
||||
model: Model<TApi>,
|
||||
context: Context,
|
||||
options?: OptionsForApi<TApi>,
|
||||
): AssistantMessageEventStream {
|
||||
return withGeminiThinkingLoopGuard(model, options, opts => streamDispatch(model, context, opts));
|
||||
}
|
||||
|
||||
function streamDispatch<TApi extends Api>(
|
||||
model: Model<TApi>,
|
||||
context: Context,
|
||||
options?: OptionsForApi<TApi>,
|
||||
): AssistantMessageEventStream {
|
||||
const requestOptions = withRequestDebugFetch(options as StreamOptions | undefined) as
|
||||
| OptionsForApi<TApi>
|
||||
@@ -489,13 +498,15 @@ export function streamSimple<TApi extends Api>(
|
||||
// extension-registered APIs can't accidentally override a configured
|
||||
// pi-native transport.
|
||||
if (model.transport === "pi-native") {
|
||||
return streamPiNative(model, context, requestOptions);
|
||||
return withGeminiThinkingLoopGuard(model, requestOptions, opts => streamPiNative(model, context, opts));
|
||||
}
|
||||
|
||||
// Check custom API registry (extension-provided APIs)
|
||||
const customApiProvider = getCustomApi(model.api);
|
||||
if (customApiProvider) {
|
||||
return customApiProvider.streamSimple(model, context, requestOptions);
|
||||
return withGeminiThinkingLoopGuard(model, requestOptions, opts =>
|
||||
customApiProvider.streamSimple(model, context, opts),
|
||||
);
|
||||
}
|
||||
|
||||
// Vertex AI uses Application Default Credentials, not API keys
|
||||
|
||||
@@ -0,0 +1,354 @@
|
||||
/**
|
||||
* Gemini thinking-loop guard.
|
||||
*
|
||||
* Gemini models (notably `gemini-3.5-flash` via OpenRouter) occasionally fall
|
||||
* into a degenerate reasoning loop: they re-emit the same paragraph intent over
|
||||
* and over with cosmetic wording drift ("Confirming Safety", "Verifying
|
||||
* Completion", …), burning the entire output budget without ever calling a tool
|
||||
* or answering. The runaway is *not* byte-identical, so a cheap verbatim
|
||||
* tail-repeat check alone misses it.
|
||||
*
|
||||
* This guard watches the streamed `thinking` deltas and, on a match, terminates
|
||||
* the stream with a synthetic `error` {@link AssistantMessage} that carries
|
||||
* **no observable content**. An empty-content `stopReason: "error"` whose
|
||||
* message hits the transient-transport pattern is what `AgentSession`
|
||||
* classifies as a *retryable* stop (a contentful error stop is treated as
|
||||
* replay-unsafe and is never retried), so the turn is discarded and re-sampled
|
||||
* instead of committing the garbage transcript.
|
||||
*
|
||||
* Two failure shapes are detected:
|
||||
* 1. **Verbatim tail repetition** — a short unit repeated back-to-back (e.g.
|
||||
* "🌊 🌊 🌊 …"). Caught from a rolling 250-char tail.
|
||||
* 2. **Near-duplicate segments** — paragraphs that normalize to the same
|
||||
* word-trigram fingerprint. Caught with a Jaccard window over recent
|
||||
* paragraphs. Thresholds were calibrated on a real loop transcript plus
|
||||
* 13.5k non-loop thinking blocks (zero false positives; hardest negative
|
||||
* scored 3 against the trigger of 4).
|
||||
*
|
||||
* Scope is deliberately narrow: **thinking only**. Answer text is left
|
||||
* untouched so the guard can never discard already-streamed visible output. The
|
||||
* guard is gated to Gemini models and wraps the provider stream, so it works
|
||||
* across every Gemini transport (OpenRouter `openai-completions`, direct
|
||||
* `google-generative-ai` / `google-gemini-cli`, Vertex). Disable with
|
||||
* `PI_NO_THINKING_LOOP_GUARD=1`.
|
||||
*/
|
||||
import { logger } from "@oh-my-pi/pi-utils";
|
||||
import type { Api, AssistantMessage, Model } from "../types";
|
||||
import { AssistantMessageEventStream } from "./event-stream";
|
||||
|
||||
/** Stable lead phrase of the guard's error message; exported for tests. The
|
||||
* message also carries "stream stall" so the session + transport retry
|
||||
* classifiers treat it as a transient (retryable) stop without bespoke rules. */
|
||||
export const THINKING_LOOP_ERROR_MARKER = "Thinking loop detected";
|
||||
|
||||
/** Rolling tail (chars) inspected for verbatim back-to-back repetition. */
|
||||
const VERBATIM_TAIL_WINDOW = 250;
|
||||
/** Minimum total repeated chars before a verbatim run counts as a loop. */
|
||||
const VERBATIM_MIN_REPEATED_CHARS = 180;
|
||||
/** Longest unit length probed for a verbatim repeat. */
|
||||
const VERBATIM_MAX_UNIT = 60;
|
||||
|
||||
/** Char cap for an unterminated segment; forces a flush so a wall-of-text loop
|
||||
* (no blank lines / headings) still segments. */
|
||||
const SEGMENT_CHAR_CAP = 700;
|
||||
/** Normalized-length floor below which a segment is ignored (too short to be a
|
||||
* meaningful paragraph; bare headings must not trip detection). */
|
||||
const SEGMENT_MIN_NORM_CHARS = 60;
|
||||
/** How many recent substantial segments are kept for similarity comparison. */
|
||||
const SEGMENT_WINDOW = 16;
|
||||
/** Word-trigram Jaccard at/above which two segments count as near-duplicates. */
|
||||
const SEGMENT_SIMILARITY = 0.8;
|
||||
/** Substantial segments required before detection may fire (warm-up). */
|
||||
const SEGMENT_MIN_COUNT = 8;
|
||||
/** Near-duplicate cluster size (current + matches) that trips the loop. */
|
||||
const SEGMENT_MIN_CLUSTER = 4;
|
||||
|
||||
const OPENAI_COMPAT_GUARDED_APIS: Partial<Record<Api, true>> = {
|
||||
"openai-completions": true,
|
||||
"openai-responses": true,
|
||||
"azure-openai-responses": true,
|
||||
"openai-codex-responses": true,
|
||||
};
|
||||
|
||||
/**
|
||||
* True when `model` should be guarded for thinking loops (Gemini only).
|
||||
*
|
||||
* OpenAI-compat transports can serve Gemini under an arbitrary provider/id, so
|
||||
* for those we trust the family-derived (and user-overridable)
|
||||
* `compat.enableGeminiThinkingLoopGuard` flag set by the catalog. Direct Google
|
||||
* transports always carry a clearly gemini-shaped id/provider, so a string
|
||||
* match is sufficient (and works for hand-built models without a compat record).
|
||||
*/
|
||||
export function isGeminiThinkingLoopModel(model: Model<Api>): boolean {
|
||||
if (OPENAI_COMPAT_GUARDED_APIS[model.api]) {
|
||||
return (
|
||||
(model.compat as { enableGeminiThinkingLoopGuard?: boolean } | undefined)?.enableGeminiThinkingLoopGuard ===
|
||||
true
|
||||
);
|
||||
}
|
||||
return /gemini/i.test(`${model.provider}/${model.id}`);
|
||||
}
|
||||
|
||||
/**
|
||||
* Stateful detector fed the streamed thinking deltas. `push` returns a
|
||||
* human-readable reason the first time a loop shape is recognized; the caller
|
||||
* is responsible for stopping after the first hit.
|
||||
*/
|
||||
export class ThinkingLoopDetector {
|
||||
/** Rolling char tail for verbatim repeat detection. */
|
||||
#tail = "";
|
||||
/** Pending thinking text not yet split into completed segments. */
|
||||
#pending = "";
|
||||
/** Fingerprints of the most recent substantial segments (≤ SEGMENT_WINDOW). */
|
||||
#window: Set<string>[] = [];
|
||||
/** Count of substantial segments seen so far (warm-up gate). */
|
||||
#count = 0;
|
||||
|
||||
push(delta: string): string | null {
|
||||
if (!delta) return null;
|
||||
|
||||
// 1. Verbatim back-to-back repetition over the rolling tail.
|
||||
this.#tail += delta;
|
||||
if (this.#tail.length > VERBATIM_TAIL_WINDOW) this.#tail = this.#tail.slice(-VERBATIM_TAIL_WINDOW);
|
||||
const verbatim = detectVerbatimRepetition(this.#tail);
|
||||
if (verbatim) {
|
||||
const [unit, times] = verbatim;
|
||||
return `repeated "${unit.trim()}" ${times}× back-to-back`;
|
||||
}
|
||||
|
||||
// 2. Near-duplicate paragraph loop. Append, then drain completed segments.
|
||||
this.#pending += delta;
|
||||
while (true) {
|
||||
const boundary = /\n\s*\n/.exec(this.#pending);
|
||||
let raw: string;
|
||||
if (boundary) {
|
||||
raw = this.#pending.slice(0, boundary.index);
|
||||
this.#pending = this.#pending.slice(boundary.index + boundary[0].length);
|
||||
} else if (this.#pending.length > SEGMENT_CHAR_CAP) {
|
||||
// No boundary yet but the segment is runaway-long: force a flush.
|
||||
raw = this.#pending.slice(0, SEGMENT_CHAR_CAP);
|
||||
this.#pending = this.#pending.slice(SEGMENT_CHAR_CAP);
|
||||
} else {
|
||||
return null;
|
||||
}
|
||||
// An over-long segment is chunked so each piece stays comparable.
|
||||
for (let rest = raw; rest.length > 0; ) {
|
||||
const chunk = rest.length > SEGMENT_CHAR_CAP ? rest.slice(0, SEGMENT_CHAR_CAP) : rest;
|
||||
rest = rest.slice(chunk.length);
|
||||
const hit = this.#consumeSegment(chunk);
|
||||
if (hit) return hit;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Process the buffered trailing paragraph (one with no blank-line / heading
|
||||
* terminator). Called when the thinking block ends so the final segment —
|
||||
* which may be the one that completes a duplicate cluster — is not dropped. */
|
||||
flush(): string | null {
|
||||
if (!this.#pending) return null;
|
||||
let rest = this.#pending;
|
||||
this.#pending = "";
|
||||
while (rest.length > 0) {
|
||||
const chunk = rest.length > SEGMENT_CHAR_CAP ? rest.slice(0, SEGMENT_CHAR_CAP) : rest;
|
||||
rest = rest.slice(chunk.length);
|
||||
const hit = this.#consumeSegment(chunk);
|
||||
if (hit) return hit;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
#consumeSegment(segment: string): string | null {
|
||||
const normalized = normalizeSegment(segment);
|
||||
if (normalized.length < SEGMENT_MIN_NORM_CHARS) return null;
|
||||
|
||||
const fingerprint = trigramShingles(normalized);
|
||||
let cluster = 1;
|
||||
for (const prev of this.#window) {
|
||||
if (jaccard(fingerprint, prev) >= SEGMENT_SIMILARITY) cluster++;
|
||||
}
|
||||
|
||||
this.#window.push(fingerprint);
|
||||
if (this.#window.length > SEGMENT_WINDOW) this.#window.shift();
|
||||
this.#count++;
|
||||
|
||||
if (this.#count >= SEGMENT_MIN_COUNT && cluster >= SEGMENT_MIN_CLUSTER) {
|
||||
return `${cluster} near-identical segments within the last ${SEGMENT_WINDOW}`;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Wrap a provider stream with the loop guard. `controller` is the guard's own
|
||||
* abort handle: aborting it (after wiring it into the provider's signal via
|
||||
* {@link withGeminiThinkingLoopGuard}) tears down the upstream once a loop
|
||||
* trips.
|
||||
*/
|
||||
export function guardThinkingLoopStream(
|
||||
inner: AssistantMessageEventStream,
|
||||
model: Model<Api>,
|
||||
controller: AbortController,
|
||||
): AssistantMessageEventStream {
|
||||
const outer = new AssistantMessageEventStream();
|
||||
const detector = new ThinkingLoopDetector();
|
||||
|
||||
void (async () => {
|
||||
// Once any visible answer text or tool call starts, disarm: some providers
|
||||
// (e.g. openai-completions with cumulative `reasoning_content`) re-emit
|
||||
// fresh thinking deltas after `</think>`, and tripping the retriable path
|
||||
// then would discard already-streamed observable output.
|
||||
let armed = true;
|
||||
try {
|
||||
for await (const event of inner) {
|
||||
let detail: string | null = null;
|
||||
if (armed && event.type === "thinking_delta") {
|
||||
detail = detector.push(event.delta);
|
||||
} else if (armed && (event.type === "thinking_end" || event.type === "done")) {
|
||||
// Final paragraph of the block has no trailing blank line; flush it
|
||||
// so the segment that completes a cluster is not dropped. `done`
|
||||
// is the backstop for providers that omit a trailing thinking_end.
|
||||
detail = detector.flush();
|
||||
} else if (
|
||||
event.type === "text_start" ||
|
||||
event.type === "text_delta" ||
|
||||
event.type === "toolcall_start" ||
|
||||
event.type === "toolcall_delta"
|
||||
) {
|
||||
armed = false;
|
||||
}
|
||||
if (detail) {
|
||||
logger.warn("Gemini thinking loop detected; aborting stream for retry.", {
|
||||
model: model.id,
|
||||
provider: model.provider,
|
||||
detail,
|
||||
});
|
||||
controller.abort(new Error(THINKING_LOOP_ERROR_MARKER));
|
||||
outer.push({ type: "error", reason: "error", error: buildThinkingLoopError(model, detail) });
|
||||
return;
|
||||
}
|
||||
outer.push(event);
|
||||
if (outer.done) return;
|
||||
}
|
||||
if (!outer.done) {
|
||||
try {
|
||||
outer.end(await inner.result());
|
||||
} catch (err) {
|
||||
outer.fail(err);
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
if (!outer.done) outer.fail(err);
|
||||
}
|
||||
})();
|
||||
|
||||
return outer;
|
||||
}
|
||||
|
||||
/**
|
||||
* Apply the Gemini loop guard around a provider dispatch. For non-Gemini models
|
||||
* (or when disabled) this is a transparent pass-through. For Gemini it injects a
|
||||
* guard abort signal into the provider call so a detected loop tears down the
|
||||
* upstream, then wraps the returned stream.
|
||||
*/
|
||||
export function withGeminiThinkingLoopGuard<O extends { signal?: AbortSignal }>(
|
||||
model: Model<Api>,
|
||||
options: O | undefined,
|
||||
dispatch: (options: O | undefined) => AssistantMessageEventStream,
|
||||
): AssistantMessageEventStream {
|
||||
if (process.env.PI_NO_THINKING_LOOP_GUARD === "1" || !isGeminiThinkingLoopModel(model)) {
|
||||
return dispatch(options);
|
||||
}
|
||||
const controller = new AbortController();
|
||||
const caller = options?.signal;
|
||||
const signal = caller ? AbortSignal.any([caller, controller.signal]) : controller.signal;
|
||||
const merged = { ...(options ?? {}), signal } as O;
|
||||
return guardThinkingLoopStream(dispatch(merged), model, controller);
|
||||
}
|
||||
|
||||
function buildThinkingLoopError(model: Model<Api>, detail: string): AssistantMessage {
|
||||
return {
|
||||
role: "assistant",
|
||||
// Empty content is load-bearing: a contentful error stop is replay-unsafe
|
||||
// and would NOT be auto-retried by the session.
|
||||
content: [],
|
||||
api: model.api,
|
||||
provider: model.provider,
|
||||
model: model.id,
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "error",
|
||||
// "stream stall" makes the transport/session retry classifiers treat this
|
||||
// as a transient (retryable) failure with no bespoke rule.
|
||||
errorMessage: `${THINKING_LOOP_ERROR_MARKER}: the model repeated near-identical reasoning (${detail}). Treating as a stream stall and retrying.`,
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Detect a short unit repeated back-to-back at the tail (verbatim loop). Only a
|
||||
* unit carrying a letter or pictographic emoji counts — runs of digits,
|
||||
* whitespace or punctuation are legitimate in tabular / hex / numeric output.
|
||||
*/
|
||||
function detectVerbatimRepetition(text: string): [unit: string, count: number] | null {
|
||||
if (text.length < VERBATIM_MIN_REPEATED_CHARS) return null;
|
||||
const windowSize = Math.min(text.length, VERBATIM_TAIL_WINDOW);
|
||||
const searchSpace = text.slice(-windowSize);
|
||||
|
||||
for (let len = 2; len <= VERBATIM_MAX_UNIT; len++) {
|
||||
if (searchSpace.length < len * 4) continue;
|
||||
const unit = searchSpace.slice(-len);
|
||||
if (!/[\p{L}\p{Extended_Pictographic}]/u.test(unit)) continue;
|
||||
|
||||
let count = 0;
|
||||
let pos = searchSpace.length;
|
||||
while (pos >= len) {
|
||||
if (searchSpace.slice(pos - len, pos) === unit) {
|
||||
count++;
|
||||
pos -= len;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (count >= 4 && len * count >= VERBATIM_MIN_REPEATED_CHARS) return [unit, count];
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Lowercase, drop code spans / paths / digits, keep only letter words. */
|
||||
function normalizeSegment(segment: string): string {
|
||||
return segment
|
||||
.toLowerCase()
|
||||
.replace(/`[^`]*`/g, " ")
|
||||
.replace(/\/[^\s`]+/g, " ")
|
||||
.replace(/\d+/g, " ")
|
||||
.replace(/[^a-z]+/g, " ")
|
||||
.trim();
|
||||
}
|
||||
|
||||
/** Word-trigram shingle set of a normalized segment. */
|
||||
function trigramShingles(normalized: string): Set<string> {
|
||||
const words = normalized.split(" ").filter(Boolean);
|
||||
if (words.length < 3) return new Set(words.length > 0 ? [words.join(" ")] : []);
|
||||
const shingles = new Set<string>();
|
||||
for (let i = 0; i + 3 <= words.length; i++) {
|
||||
shingles.add(`${words[i]} ${words[i + 1]} ${words[i + 2]}`);
|
||||
}
|
||||
return shingles;
|
||||
}
|
||||
|
||||
function jaccard(a: Set<string>, b: Set<string>): number {
|
||||
if (a.size === 0 || b.size === 0) return 0;
|
||||
const [small, large] = a.size < b.size ? [a, b] : [b, a];
|
||||
let intersection = 0;
|
||||
for (const x of small) {
|
||||
if (large.has(x)) intersection++;
|
||||
}
|
||||
const union = a.size + b.size - intersection;
|
||||
return union === 0 ? 0 : intersection / union;
|
||||
}
|
||||
@@ -0,0 +1,282 @@
|
||||
import { describe, expect, test } from "bun:test";
|
||||
import { clearCustomApis } from "@oh-my-pi/pi-ai/api-registry";
|
||||
import { createMockModel, type MockContent, registerMockApi } from "@oh-my-pi/pi-ai/providers/mock";
|
||||
import { stream, streamSimple } from "@oh-my-pi/pi-ai/stream";
|
||||
import type { Api, AssistantMessage, AssistantMessageEvent, Context, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
|
||||
import {
|
||||
isGeminiThinkingLoopModel,
|
||||
THINKING_LOOP_ERROR_MARKER,
|
||||
ThinkingLoopDetector,
|
||||
withGeminiThinkingLoopGuard,
|
||||
} from "@oh-my-pi/pi-ai/utils/thinking-loop";
|
||||
import { isRetryableError } from "@oh-my-pi/pi-utils";
|
||||
|
||||
function context(): Context {
|
||||
return { systemPrompt: [], messages: [{ role: "user", content: "go", timestamp: 0 }] };
|
||||
}
|
||||
|
||||
async function collect(events: AsyncIterable<AssistantMessageEvent>): Promise<AssistantMessageEvent[]> {
|
||||
const out: AssistantMessageEvent[] = [];
|
||||
for await (const event of events) out.push(event);
|
||||
return out;
|
||||
}
|
||||
|
||||
/** A degenerate near-duplicate reasoning loop: the same paragraph intent with
|
||||
* cosmetic wording drift, blank-line separated (the gemini-3.5-flash shape). */
|
||||
function nearDuplicateLoop(paragraphs: number): string {
|
||||
const variants = [
|
||||
"I am now verifying the test module to guarantee there are no compile errors and the code is completely safe.",
|
||||
"I am now verifying the test module once more to ensure there are no compile errors and the code stays completely safe.",
|
||||
"I am now re-verifying the test module to confirm there are no compile errors and the code remains completely safe.",
|
||||
];
|
||||
const out: string[] = [];
|
||||
for (let i = 0; i < paragraphs; i++) {
|
||||
out.push(`**Confirming Safety ${i}**\n\n${variants[i % variants.length]}`);
|
||||
}
|
||||
return out.join("\n\n\n");
|
||||
}
|
||||
|
||||
/** Genuinely distinct reasoning paragraphs — must never trip the detector. */
|
||||
function distinctReasoning(): string {
|
||||
return [
|
||||
"First I read the agent loop to understand how the stream wrapper commits the final assistant message.",
|
||||
"Next I traced the retry classifier to see which error shapes are treated as transient by the session.",
|
||||
"The Vertex transport needs ADC, so its credential path differs from the OpenRouter completions route entirely.",
|
||||
"I should add a regression covering the empty-content terminal so the auto-retry gate stays satisfied later.",
|
||||
"The tokenizer counts code points, which matters for the wide-emoji case in the truncation helper above here.",
|
||||
"Compaction journals are per-request, so they never persist into the on-disk session transcript at all ever.",
|
||||
"The mock provider emits one delta per block, so streaming nuances need a direct detector unit test instead.",
|
||||
"Finally I will run the focused package suite and the type checker before touching the changelog entries now.",
|
||||
"Telemetry spans wrap each provider call, so failing one must still close its span in the catch branch too.",
|
||||
"A device default of CPU keeps the tiny model worker from crashing Bun on teardown across every platform now.",
|
||||
].join("\n\n\n");
|
||||
}
|
||||
|
||||
describe("isGeminiThinkingLoopModel", () => {
|
||||
test("matches direct and aggregator-routed gemini ids, not lookalikes", () => {
|
||||
const gate = (provider: string, id: string) => isGeminiThinkingLoopModel(createMockModel({ provider, id }).model);
|
||||
expect(gate("google", "gemini-3-pro-preview")).toBe(true);
|
||||
expect(gate("openrouter", "google/gemini-3.5-flash")).toBe(true);
|
||||
expect(gate("google-gemini-cli", "gemini-3-flash")).toBe(true);
|
||||
expect(gate("openai", "gpt-5.5")).toBe(false);
|
||||
expect(gate("google", "gemma-3-1b")).toBe(false);
|
||||
});
|
||||
|
||||
test("trusts the compat flag over the id regex for every OpenAI-compat API", () => {
|
||||
const gate = (api: string, id: string, enableGeminiThinkingLoopGuard: boolean) =>
|
||||
isGeminiThinkingLoopModel({
|
||||
api,
|
||||
provider: "openrouter",
|
||||
id,
|
||||
compat: { enableGeminiThinkingLoopGuard },
|
||||
} as unknown as Model<Api>);
|
||||
// Opaque proxy alias opted in despite a non-gemini id (completions + responses).
|
||||
expect(gate("openai-completions", "my-fast-model", true)).toBe(true);
|
||||
expect(gate("openai-responses", "my-fast-model", true)).toBe(true);
|
||||
// Gemini-shaped id explicitly opted out stays off — the flag wins over the regex.
|
||||
expect(gate("openai-completions", "gemini-3.5-flash", false)).toBe(false);
|
||||
expect(gate("openai-responses", "gemini-3.5-flash", false)).toBe(false);
|
||||
});
|
||||
|
||||
test("guards non-compat Gemini transports (Vertex, direct Google) via id", () => {
|
||||
const gate = (api: string, provider: string, id: string) =>
|
||||
isGeminiThinkingLoopModel({ api, provider, id } as unknown as Model<Api>);
|
||||
// Vertex has no OpenAICompat record; its canonical ids are gemini-shaped.
|
||||
expect(gate("google-vertex", "google-vertex", "gemini-2.5-pro")).toBe(true);
|
||||
expect(gate("google-generative-ai", "google", "gemini-3-pro")).toBe(true);
|
||||
// Non-Gemini models on the same transports (e.g. Claude on Vertex) stay unguarded.
|
||||
expect(gate("google-vertex", "google-vertex", "claude-sonnet-4")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("ThinkingLoopDetector", () => {
|
||||
test("trips on near-duplicate segments fed as small streamed chunks", () => {
|
||||
const detector = new ThinkingLoopDetector();
|
||||
const text = nearDuplicateLoop(12);
|
||||
let detail: string | null = null;
|
||||
for (let i = 0; i < text.length && !detail; i += 17) {
|
||||
detail = detector.push(text.slice(i, i + 17));
|
||||
}
|
||||
expect(detail).toContain("near-identical segments");
|
||||
});
|
||||
|
||||
test("trips on verbatim back-to-back repetition", () => {
|
||||
const detector = new ThinkingLoopDetector();
|
||||
const detail = detector.push("🌊 ".repeat(120));
|
||||
expect(detail).toContain("back-to-back");
|
||||
});
|
||||
|
||||
test("does not trip on genuinely distinct reasoning paragraphs", () => {
|
||||
const detector = new ThinkingLoopDetector();
|
||||
let detail: string | null = null;
|
||||
const text = distinctReasoning();
|
||||
for (let i = 0; i < text.length && !detail; i += 23) {
|
||||
detail = detector.push(text.slice(i, i + 23));
|
||||
}
|
||||
// flush the trailing paragraph too
|
||||
detail ??= detector.flush();
|
||||
expect(detail).toBeNull();
|
||||
});
|
||||
|
||||
test("flush() catches a final unterminated duplicate paragraph", () => {
|
||||
const detector = new ThinkingLoopDetector();
|
||||
// Seven blank-line-separated dupes leave the eighth (cluster-completing)
|
||||
// paragraph in the buffer with no trailing blank line.
|
||||
const block = `${nearDuplicateLoop(7)}\n\n\nI am now verifying the test module to guarantee there are no compile errors and the code is completely safe.`;
|
||||
expect(detector.push(block)).toBeNull();
|
||||
expect(detector.flush()).toContain("near-identical segments");
|
||||
});
|
||||
|
||||
test("does not trip on legitimate repetitive numeric output", () => {
|
||||
// A zero-page hexdump is highly repetitive but legitimate: a unit with no
|
||||
// letter or pictograph must never count as a loop.
|
||||
const detector = new ThinkingLoopDetector();
|
||||
expect(detector.push("00 ".repeat(200))).toBeNull();
|
||||
});
|
||||
|
||||
test("does not trip on short requested repetitive text", () => {
|
||||
// Below the repeated-char floor: a brief on-purpose repeat is not a loop.
|
||||
const detector = new ThinkingLoopDetector();
|
||||
expect(detector.push("🌊 ".repeat(26))).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe("gemini thinking-loop guard (stream wrapper)", () => {
|
||||
function loopingThinkingResponse(): { content: MockContent[] } {
|
||||
return { content: [{ type: "thinking", thinking: nearDuplicateLoop(12) }] };
|
||||
}
|
||||
|
||||
test("terminates a gemini loop with a retryable empty-content error", async () => {
|
||||
registerMockApi();
|
||||
try {
|
||||
const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" });
|
||||
mock.push(loopingThinkingResponse());
|
||||
|
||||
const result = await stream(mock.model, context()).result();
|
||||
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.content).toEqual([]);
|
||||
expect(result.errorMessage).toContain(THINKING_LOOP_ERROR_MARKER);
|
||||
// Empty content + transient phrasing is what makes the turn auto-retry.
|
||||
expect(result.errorMessage).toContain("stream stall");
|
||||
expect(isRetryableError(new Error(result.errorMessage))).toBe(true);
|
||||
} finally {
|
||||
clearCustomApis();
|
||||
}
|
||||
});
|
||||
|
||||
test("emits no observable thinking/text content before the error terminal", async () => {
|
||||
registerMockApi();
|
||||
try {
|
||||
const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" });
|
||||
mock.push(loopingThinkingResponse());
|
||||
|
||||
const events = await collect(stream(mock.model, context()));
|
||||
const terminal = events.at(-1);
|
||||
expect(terminal?.type).toBe("error");
|
||||
// The guard must not forward the looping thinking_end / done.
|
||||
expect(events.some(e => e.type === "thinking_end")).toBe(false);
|
||||
expect(events.some(e => e.type === "done")).toBe(false);
|
||||
} finally {
|
||||
clearCustomApis();
|
||||
}
|
||||
});
|
||||
|
||||
test("passes a non-gemini model through untouched even when it loops", async () => {
|
||||
registerMockApi();
|
||||
try {
|
||||
const mock = createMockModel({ provider: "openai", id: "gpt-5.5" });
|
||||
mock.push(loopingThinkingResponse());
|
||||
|
||||
const result = await stream(mock.model, context()).result();
|
||||
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.content.some(b => b.type === "thinking")).toBe(true);
|
||||
} finally {
|
||||
clearCustomApis();
|
||||
}
|
||||
});
|
||||
|
||||
test("does not trip on a healthy gemini turn that reasons then answers", async () => {
|
||||
registerMockApi();
|
||||
try {
|
||||
const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" });
|
||||
mock.push({ content: [{ type: "thinking", thinking: distinctReasoning() }, "Here is the final answer."] });
|
||||
|
||||
const result = await stream(mock.model, context()).result();
|
||||
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.content.some(b => b.type === "text")).toBe(true);
|
||||
} finally {
|
||||
clearCustomApis();
|
||||
}
|
||||
});
|
||||
|
||||
test("does not retry once a loop re-emerges after visible answer text (armed latch)", async () => {
|
||||
registerMockApi();
|
||||
try {
|
||||
const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" });
|
||||
// Healthy reasoning, then visible answer text, then a runaway loop re-emitted
|
||||
// as cumulative reasoning after `</think>` (the openai-completions shape).
|
||||
mock.push({
|
||||
content: [
|
||||
{ type: "thinking", thinking: distinctReasoning() },
|
||||
"Here is the final answer.",
|
||||
{ type: "thinking", thinking: nearDuplicateLoop(12) },
|
||||
],
|
||||
});
|
||||
|
||||
const result = await stream(mock.model, context()).result();
|
||||
|
||||
// Visible text already streamed, so the loop must NOT hijack the turn.
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.content.some(b => b.type === "text")).toBe(true);
|
||||
} finally {
|
||||
clearCustomApis();
|
||||
}
|
||||
});
|
||||
|
||||
test("guards the streamSimple custom-api path (agent default entrypoint)", async () => {
|
||||
registerMockApi();
|
||||
try {
|
||||
const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" });
|
||||
mock.push(loopingThinkingResponse());
|
||||
|
||||
const result = await streamSimple(mock.model, context()).result();
|
||||
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.content).toEqual([]);
|
||||
expect(result.errorMessage).toContain(THINKING_LOOP_ERROR_MARKER);
|
||||
expect(isRetryableError(new Error(result.errorMessage))).toBe(true);
|
||||
} finally {
|
||||
clearCustomApis();
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe("withGeminiThinkingLoopGuard (Vertex transport)", () => {
|
||||
test("emits a retryable empty-content error for a looping Vertex Gemini stream", async () => {
|
||||
const model = { api: "google-vertex", provider: "google-vertex", id: "gemini-2.5-pro" } as unknown as Model<Api>;
|
||||
const partial = { role: "assistant", content: [] } as unknown as AssistantMessage;
|
||||
|
||||
const guarded = withGeminiThinkingLoopGuard(model, undefined, () => {
|
||||
const inner = new AssistantMessageEventStream();
|
||||
const events: AssistantMessageEvent[] = [
|
||||
{ type: "start", partial },
|
||||
{ type: "thinking_start", contentIndex: 0, partial },
|
||||
{ type: "thinking_delta", contentIndex: 0, delta: nearDuplicateLoop(12), partial },
|
||||
{ type: "thinking_end", contentIndex: 0, content: "", partial },
|
||||
{ type: "done", reason: "stop", message: partial },
|
||||
];
|
||||
for (const event of events) inner.push(event);
|
||||
return inner;
|
||||
});
|
||||
|
||||
const result = await guarded.result();
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.content.length).toBe(0);
|
||||
expect(result.errorMessage).toContain(THINKING_LOOP_ERROR_MARKER);
|
||||
expect(isRetryableError(new Error(result.errorMessage))).toBe(true);
|
||||
});
|
||||
});
|
||||
@@ -1,13 +1,16 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added `enableGeminiThinkingLoopGuard` to OpenAI compatibility options to allow explicit opt-in or opt-out of the Gemini thinking-loop guard for OpenAI-compatible model aliases
|
||||
- Added `LITELLM_BASE_URL` as the LiteLLM provider discovery base URL fallback, with discovery caches scoped by the resolved proxy URL and explicit provider `baseUrl` config kept at higher precedence. ([#2726](https://github.com/can1357/oh-my-pi/issues/2726))
|
||||
|
||||
### Changed
|
||||
|
||||
- Defaulted `enableGeminiThinkingLoopGuard` from Gemini family detection for both OpenAI completions and responses compatibility specs so Gemini models now enable the thinking-loop guard automatically
|
||||
- Updated the default Gemini CLI user-agent version fallback to 0.46.0.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Routed google-antigravity default baseUrl to the stable primary daily endpoint in the catalog generator and all fallback snapshots, resolving connection drops on heavy queries.
|
||||
@@ -247,4 +250,4 @@
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads.
|
||||
- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads.
|
||||
@@ -17,6 +17,7 @@ import {
|
||||
isKimiModelId,
|
||||
isMimoModelIdOrName,
|
||||
isQwenModelId,
|
||||
modelFamilyToken,
|
||||
} from "../identity/family";
|
||||
import type { ModelSpec, OpenAICompat, ResolvedOpenAICompat, ResolvedOpenAIResponsesCompat } from "../types";
|
||||
import { applyCompatOverrides } from "./apply";
|
||||
@@ -211,6 +212,10 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
supportsReasoningParams: provider !== "github-copilot",
|
||||
reasoningEffortMap: {},
|
||||
supportsUsageInStreaming: !isCerebras,
|
||||
// pi-ai's thinking-loop guard is gemini-only; default the flag from the
|
||||
// family classifier so OpenAI-compat proxies serving Gemini are covered.
|
||||
// An opaque alias can opt in via `compat.enableGeminiThinkingLoopGuard`.
|
||||
enableGeminiThinkingLoopGuard: modelFamilyToken(spec.id) === "gemini",
|
||||
// Kimi (including via OpenRouter and Fireworks router-form IDs such as
|
||||
// `accounts/fireworks/routers/kimi-*`) calculates TPM rate limits based on
|
||||
// max_tokens, not actual output. The official Kimi K2 model guidance
|
||||
@@ -295,6 +300,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
}
|
||||
|
||||
interface OpenAIResponsesSpecLike {
|
||||
id?: string;
|
||||
provider: string;
|
||||
name: string;
|
||||
baseUrl: string;
|
||||
@@ -329,6 +335,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
|
||||
strictResponsesPairing: isAzure || spec.provider === "github-copilot",
|
||||
requiresJuiceZeroHack: spec.name.toLowerCase().startsWith("gpt-5"),
|
||||
reasoningEffortMap: {},
|
||||
enableGeminiThinkingLoopGuard: modelFamilyToken(spec.id ?? "") === "gemini",
|
||||
};
|
||||
applyCompatOverrides(compat, spec.compat);
|
||||
return compat;
|
||||
|
||||
@@ -170,6 +170,13 @@ export interface OpenAICompat {
|
||||
reasoningEffortMap?: Partial<Record<Effort, string>>;
|
||||
/** Whether the provider supports `stream_options: { include_usage: true }` for token usage in streaming responses. Default: true. */
|
||||
supportsUsageInStreaming?: boolean;
|
||||
/**
|
||||
* Enable the Gemini thinking-loop guard (pi-ai stream layer) for this model.
|
||||
* Defaults to true when the model id classifies as the gemini family. Set
|
||||
* explicitly to cover an opaque OpenAI-compat proxy alias (e.g. `my-model`)
|
||||
* that routes to Gemini, or to false to opt a gemini-family id out.
|
||||
*/
|
||||
enableGeminiThinkingLoopGuard?: boolean;
|
||||
/** Which field to use for max tokens. Default: auto-detected from URL. */
|
||||
maxTokensField?: "max_completion_tokens" | "max_tokens";
|
||||
/** Whether tool results require the `name` field. Default: auto-detected from URL. */
|
||||
@@ -373,6 +380,7 @@ export type ResolvedOpenAICompat = Required<
|
||||
| "thinkingKeep"
|
||||
| "strictResponsesPairing"
|
||||
| "requiresJuiceZeroHack"
|
||||
| "enableGeminiThinkingLoopGuard"
|
||||
| "whenThinking"
|
||||
>
|
||||
> & {
|
||||
@@ -387,6 +395,8 @@ export type ResolvedOpenAICompat = Required<
|
||||
isOpenRouterHost: boolean;
|
||||
/** The model sits behind Vercel AI Gateway. */
|
||||
isVercelGatewayHost: boolean;
|
||||
/** See {@link OpenAICompat.enableGeminiThinkingLoopGuard}. Set by the builder from the family classifier. */
|
||||
enableGeminiThinkingLoopGuard?: boolean;
|
||||
/** Complete alternate view for thinking-engaged requests; swap pointers, never spread. */
|
||||
whenThinking?: ResolvedOpenAICompat;
|
||||
};
|
||||
@@ -400,6 +410,8 @@ export interface ResolvedOpenAIResponsesCompat {
|
||||
strictResponsesPairing: boolean;
|
||||
requiresJuiceZeroHack: boolean;
|
||||
reasoningEffortMap: Partial<Record<Effort, string>>;
|
||||
/** See {@link OpenAICompat.enableGeminiThinkingLoopGuard}. */
|
||||
enableGeminiThinkingLoopGuard?: boolean;
|
||||
}
|
||||
|
||||
/** Fully-resolved anthropic-messages compat view (same contract as `ResolvedOpenAICompat`). */
|
||||
|
||||
@@ -0,0 +1,69 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { buildOpenAICompat, buildOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
|
||||
import type { ModelSpec, OpenAICompat } from "@oh-my-pi/pi-catalog/types";
|
||||
|
||||
/**
|
||||
* The pi-ai thinking-loop guard is gemini-only and, for `openai-completions`
|
||||
* models, gates on `compat.enableGeminiThinkingLoopGuard`. `buildOpenAICompat`
|
||||
* must default that flag from the family classifier and honor explicit
|
||||
* overrides so an opaque OpenAI-compat proxy alias can opt in/out.
|
||||
*/
|
||||
function spec(id: string, compat?: OpenAICompat): ModelSpec<"openai-completions"> {
|
||||
return {
|
||||
api: "openai-completions",
|
||||
id,
|
||||
name: id,
|
||||
provider: "custom",
|
||||
baseUrl: "https://proxy.example.com/v1",
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
maxTokens: 32_000,
|
||||
contextWindow: 200_000,
|
||||
reasoning: true,
|
||||
...(compat ? { compat } : {}),
|
||||
};
|
||||
}
|
||||
|
||||
describe("buildOpenAICompat enableGeminiThinkingLoopGuard", () => {
|
||||
it("defaults on for gemini-family ids, including aggregator namespaces", () => {
|
||||
expect(buildOpenAICompat(spec("gemini-3.5-flash")).enableGeminiThinkingLoopGuard).toBe(true);
|
||||
expect(buildOpenAICompat(spec("google/gemini-3-pro")).enableGeminiThinkingLoopGuard).toBe(true);
|
||||
});
|
||||
|
||||
it("defaults off for non-gemini ids (incl. gemma lookalikes)", () => {
|
||||
expect(buildOpenAICompat(spec("gpt-5.5")).enableGeminiThinkingLoopGuard).toBe(false);
|
||||
expect(buildOpenAICompat(spec("gemma-3-1b")).enableGeminiThinkingLoopGuard).toBe(false);
|
||||
});
|
||||
|
||||
it("lets an opaque proxy alias opt in via explicit compat override", () => {
|
||||
const compat = buildOpenAICompat(spec("my-fast-model", { enableGeminiThinkingLoopGuard: true }));
|
||||
expect(compat.enableGeminiThinkingLoopGuard).toBe(true);
|
||||
});
|
||||
|
||||
it("lets a gemini-family id opt out via explicit compat override", () => {
|
||||
const compat = buildOpenAICompat(spec("gemini-3.5-flash", { enableGeminiThinkingLoopGuard: false }));
|
||||
expect(compat.enableGeminiThinkingLoopGuard).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("buildOpenAIResponsesCompat enableGeminiThinkingLoopGuard", () => {
|
||||
const responsesSpec = (id: string, compat?: OpenAICompat) => ({
|
||||
id,
|
||||
name: id,
|
||||
provider: "custom",
|
||||
baseUrl: "https://proxy.example.com/v1",
|
||||
...(compat ? { compat } : {}),
|
||||
});
|
||||
|
||||
it("defaults from the family classifier", () => {
|
||||
expect(buildOpenAIResponsesCompat(responsesSpec("gemini-3-pro")).enableGeminiThinkingLoopGuard).toBe(true);
|
||||
expect(buildOpenAIResponsesCompat(responsesSpec("gpt-5.5")).enableGeminiThinkingLoopGuard).toBe(false);
|
||||
});
|
||||
|
||||
it("honors an explicit override for an opaque proxy alias", () => {
|
||||
expect(
|
||||
buildOpenAIResponsesCompat(responsesSpec("my-fast-model", { enableGeminiThinkingLoopGuard: true }))
|
||||
.enableGeminiThinkingLoopGuard,
|
||||
).toBe(true);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user