From 0d9ec354e7c765d6fcb5812a0a488efe0a0b05f4 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 17 Jun 2026 13:37:57 +0200 Subject: [PATCH] fix: fixed Gemini thinking-loop handling across streaming dispatch paths - Added Gemini thinking-loop detection helpers for near-duplicate and verbatim output checks. - Wrapped `stream`, `streamPiNative`, and `streamSimple` dispatches with the loop guard. - Emitted retryable empty-content loop errors and stopped completion events on loop hits. - Added `enableGeminiThinkingLoopGuard` options for OpenAI compatibility with Gemini defaults and overrides. --- packages/agent/CHANGELOG.md | 4 + packages/agent/src/agent-loop.ts | 183 +-------- packages/agent/test/agent-loop.test.ts | 120 ------ packages/ai/CHANGELOG.md | 3 +- packages/ai/src/index.ts | 1 + packages/ai/src/stream.ts | 15 +- packages/ai/src/utils/thinking-loop.ts | 354 ++++++++++++++++++ packages/ai/test/thinking-loop.test.ts | 282 ++++++++++++++ packages/catalog/CHANGELOG.md | 7 +- packages/catalog/src/compat/openai.ts | 7 + packages/catalog/src/types.ts | 12 + .../test/gemini-thinking-loop-compat.test.ts | 69 ++++ 12 files changed, 755 insertions(+), 302 deletions(-) create mode 100644 packages/ai/src/utils/thinking-loop.ts create mode 100644 packages/ai/test/thinking-loop.test.ts create mode 100644 packages/catalog/test/gemini-thinking-loop-compat.test.ts diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 8c947d783..74b82660d 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -14,6 +14,10 @@ - Added agent-loop deadline support for graceful wall-clock session stops. +### Changed + +- Changed Gemini repetition-loop detection to live in the pi-ai stream layer instead of the agent loop. The agent no longer runs its own Gemini-gated verbatim repetition check (`detectRepetition`/`truncateRepetition`); loops now surface as a retryable transient stream error that the standard auto-retry path discards and re-samples, rather than a committed contentful error message. + ## [16.0.1] - 2026-06-15 ### Fixed diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index 2f6cf1ba4..6500c60a1 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -35,7 +35,7 @@ import { signalListLabel, } from "@oh-my-pi/pi-ai/utils/harmony-leak"; import { preferredDialect } from "@oh-my-pi/pi-catalog/identity"; -import { logger, sanitizeText } from "@oh-my-pi/pi-utils"; +import { sanitizeText } from "@oh-my-pi/pi-utils"; import { type AgentRunCoverage, type AgentRunSummary, ToolCallBlockedError } from "./run-collector"; import { type AgentTelemetry, @@ -1054,7 +1054,6 @@ async function streamAssistantResponse( ? AbortSignal.any([signal, harmonyAbortController.signal]) : harmonyAbortController.signal : signal; - const repetitionAbortController = new AbortController(); // Owned tool calling: aborted by the stream wrapper when the model starts // fabricating a ``, so the provider stops generating the rest of // the hallucinated turn. Merged into the provider signal ONLY (not @@ -1063,10 +1062,13 @@ async function streamAssistantResponse( const promptToolAbortController = ownedDialect ? new AbortController() : undefined; const providerAbortSignals: AbortSignal[] = []; if (requestSignal) providerAbortSignals.push(requestSignal); - providerAbortSignals.push(repetitionAbortController.signal); if (promptToolAbortController) providerAbortSignals.push(promptToolAbortController.signal); const finalRequestSignal = - providerAbortSignals.length === 1 ? providerAbortSignals[0]! : AbortSignal.any(providerAbortSignals); + providerAbortSignals.length === 0 + ? undefined + : providerAbortSignals.length === 1 + ? providerAbortSignals[0]! + : AbortSignal.any(providerAbortSignals); const requestApiKey = (config.getApiKey ? await config.getApiKey(config.model) : undefined) ?? config.apiKey; const resolvedApiKey = await resolveApiKeyOnce(requestApiKey, finalRequestSignal); const apiKey = isApiKeyResolver(requestApiKey) ? seedApiKeyResolver(resolvedApiKey, requestApiKey) : requestApiKey; @@ -1173,56 +1175,6 @@ async function streamAssistantResponse( return aborted; }; - const finishRepetitionStream = async ( - kind: "text" | "thinking", - pattern: string, - count: number, - ): Promise => { - repetitionAbortController.abort(); - try { - const cleanup = responseIterator.return?.(); - if (cleanup) void cleanup.catch(() => {}); - } catch { - // ignore - } - if (partialMessage) { - truncateRepetition(partialMessage, kind, pattern); - partialMessage.stopReason = "error"; - partialMessage.errorMessage = `Repetition loop detected: assistant repeated "${pattern.trim()}" ${count} times consecutively.`; - } - const finalMsg = snapshotAssistantMessage( - partialMessage ?? { - role: "assistant", - content: [], - api: config.model.api, - provider: config.model.provider, - model: config.model.id, - usage: { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - totalTokens: 0, - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - }, - stopReason: "error", - errorMessage: `Repetition loop detected.`, - timestamp: Date.now(), - }, - ); - if (addedPartial) { - context.messages[context.messages.length - 1] = finalMsg; - } else { - context.messages.push(finalMsg); - } - if (!addedPartial) { - stream.push({ type: "message_start", message: snapshotAssistantMessage(finalMsg) }); - } - stream.push({ type: "message_end", message: snapshotAssistantMessage(finalMsg) }); - await finishChat(finalMsg); - return finalMsg; - }; - // Set up a single abort race: register the abort listener once for the whole // stream and reuse the same race promise for every iterator.next() instead of // allocating Promise.withResolvers and add/removeEventListener per event. @@ -1239,14 +1191,6 @@ async function streamAssistantResponse( detachAbortListener = () => requestSignal.removeEventListener("abort", onAbort); } - // Rolling tail of streamed text/thinking used for repetition-loop detection. - // Bounded to REPETITION_WINDOW chars and reset when the active block kind - // switches (text <-> thinking) so detection stays O(1) per delta and never - // miscounts a repeated unit across a thinking/answer boundary. - let repetitionTail = ""; - let repetitionKind: "text" | "thinking" | undefined; - const isGeminiModel = config.model.provider.includes("google") || config.model.provider.includes("gemini"); - try { while (true) { let next: IteratorResult; @@ -1331,27 +1275,6 @@ async function streamAssistantResponse( assistantMessageEvent: snapshotAssistantMessageEvent(event), message: snapshotAssistantMessage(partialMessage), }); - - if (isGeminiModel && (event.type === "text_delta" || event.type === "thinking_delta")) { - const kind = event.type === "text_delta" ? "text" : "thinking"; - if (repetitionKind !== kind) { - repetitionKind = kind; - repetitionTail = ""; - } - repetitionTail += event.delta; - if (repetitionTail.length > REPETITION_WINDOW) { - repetitionTail = repetitionTail.slice(-REPETITION_WINDOW); - } - const repetition = detectRepetition(repetitionTail); - if (repetition) { - const [pattern, count] = repetition; - logger.warn("Repetition loop detected during assistant stream, aborting.", { - pattern, - count, - }); - return await finishRepetitionStream(kind, pattern, count); - } - } } break; } @@ -1987,97 +1910,3 @@ function createSkippedToolResult(): AgentToolResult { details: {}, }; } - -const REPETITION_WINDOW = 250; -const REPETITION_MIN_REPEATED_CHARS = 180; - -function detectRepetition(text: string): [pattern: string, count: number] | null { - if (text.length < REPETITION_MIN_REPEATED_CHARS) return null; - - const windowSize = Math.min(text.length, REPETITION_WINDOW); - const searchSpace = text.slice(-windowSize); - - for (let len = 2; len <= 60; len++) { - if (searchSpace.length < len * 4) continue; - - const pattern = searchSpace.slice(-len); - // Only treat a repeated unit as a pathological loop when it carries real - // linguistic content (a letter or a pictographic emoji). Runs made purely of - // digits, whitespace or punctuation are legitimate in tabular / hex / numeric - // output (e.g. "00 00 00", "0, 0, 0", "| -- | -- |") and must not trip. - if (!/[\p{L}\p{Extended_Pictographic}]/u.test(pattern)) continue; - - let count = 0; - let pos = searchSpace.length; - while (pos >= len) { - const chunk = searchSpace.slice(pos - len, pos); - if (chunk === pattern) { - count++; - pos -= len; - } else { - break; - } - } - - if (count >= 4 && len * count >= REPETITION_MIN_REPEATED_CHARS) { - return [pattern, count]; - } - } - return null; -} - -function truncateRepetition(message: AssistantMessage, kind: "text" | "thinking", pattern: string): void { - // A repetition loop streams into a single growing block (real providers) or a run - // of same-kind blocks (some transports), always at the tail of the message. Gather - // that trailing contiguous run and collapse its repeated copies down to one, so the - // committed transcript keeps a representative sample instead of the full runaway. - const matches = (block: AssistantContentBlock): boolean => - kind === "text" ? block.type === "text" : block.type === "thinking"; - const readBlock = (block: AssistantContentBlock): string => - block.type === "text" ? block.text : block.type === "thinking" ? block.thinking : ""; - const clearThinkingReplayAnchors = (block: AssistantContentBlock): void => { - if (block.type !== "thinking") return; - block.thinkingSignature = undefined; - block.itemId = undefined; - }; - const writeBlock = (block: AssistantContentBlock, value: string): void => { - if (block.type === "text") { - block.text = value; - } else if (block.type === "thinking") { - block.thinking = value; - clearThinkingReplayAnchors(block); - } - }; - - const trailing: AssistantContentBlock[] = []; - for (let i = message.content.length - 1; i >= 0; i--) { - const block = message.content[i]; - if (!matches(block)) break; - trailing.unshift(block); - } - if (trailing.length === 0) return; - if (kind === "thinking") { - for (const block of trailing) clearThinkingReplayAnchors(block); - } - - let joined = ""; - for (const block of trailing) joined += readBlock(block); - - let kept = joined; - while (kept.length >= pattern.length * 2 && kept.slice(kept.length - pattern.length * 2) === pattern + pattern) { - kept = kept.slice(0, kept.length - pattern.length); - } - - let remainingToRemove = joined.length - kept.length; - for (let i = trailing.length - 1; i >= 0 && remainingToRemove > 0; i--) { - const block = trailing[i]; - const value = readBlock(block); - if (value.length <= remainingToRemove) { - remainingToRemove -= value.length; - writeBlock(block, ""); - } else { - writeBlock(block, value.slice(0, value.length - remainingToRemove)); - remainingToRemove = 0; - } - } -} diff --git a/packages/agent/test/agent-loop.test.ts b/packages/agent/test/agent-loop.test.ts index ca8935d21..a988473c3 100644 --- a/packages/agent/test/agent-loop.test.ts +++ b/packages/agent/test/agent-loop.test.ts @@ -1879,126 +1879,6 @@ describe("agentLoopContinue with AgentMessage", () => { } }); - it("should detect repetition loops during assistant stream and abort gracefully", async () => { - const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] }; - const mock = createMockModel({ - provider: "google-gemini-cli", - responses: [ - { - content: Array.from({ length: 80 }, () => "🌊 "), - }, - ], - }); - const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter }; - - const stream = agentLoop([createUserMessage("Hello")], context, config, undefined, mock.stream); - for await (const _event of stream) { - // drain the stream to completion - } - - const messages = await stream.result(); - expect(messages.length).toBe(2); - expect(messages[1].role).toBe("assistant"); - - const assistantMsg = messages[1] as AssistantMessage; - expect(assistantMsg.stopReason).toBe("error"); - expect(assistantMsg.errorMessage).toContain("Repetition loop detected"); - - let text = ""; - for (const block of assistantMsg.content) { - if (block.type === "text") text += block.text; - } - expect(text).toBe("🌊 "); - }); - - it("detects and truncates repetition loops inside a thinking stream", async () => { - const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] }; - const mock = createMockModel({ - provider: "google-gemini-cli", - responses: [ - { - content: Array.from({ length: 80 }, (_, index) => ({ - type: "thinking" as const, - thinking: "🌊 ", - thinkingSignature: `signature-${index}`, - itemId: `rs_${index}`, - })), - }, - ], - }); - const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter }; - - const stream = agentLoop([createUserMessage("Hello")], context, config, undefined, mock.stream); - for await (const _event of stream) { - // drain the stream to completion - } - - const assistantMsg = (await stream.result())[1] as AssistantMessage; - expect(assistantMsg.stopReason).toBe("error"); - expect(assistantMsg.errorMessage).toContain("Repetition loop detected"); - - // A looping thinking stream must be both detected AND collapsed to a single - // representative copy — not committed to the transcript in full. - let thinking = ""; - for (const block of assistantMsg.content) { - if (block.type === "thinking") thinking += block.thinking; - } - expect(thinking).toBe("🌊 "); - for (const block of assistantMsg.content) { - if (block.type === "thinking") { - expect(block.thinkingSignature).toBeUndefined(); - expect(block.itemId).toBeUndefined(); - } - } - }); - - it("does not flag short requested repetitive text as a loop", async () => { - const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] }; - const repeated = "🌊 ".repeat(26); - const mock = createMockModel({ - provider: "google-gemini-cli", - responses: [{ content: [repeated] }], - }); - const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter }; - - const stream = agentLoop([createUserMessage("print 26 wave emoji")], context, config, undefined, mock.stream); - for await (const _event of stream) { - // drain the stream to completion - } - - const assistantMsg = (await stream.result())[1] as AssistantMessage; - expect(assistantMsg.stopReason).not.toBe("error"); - let text = ""; - for (const block of assistantMsg.content) { - if (block.type === "text") text += block.text; - } - expect(text).toBe(repeated); - }); - - it("does not flag legitimate repetitive numeric output as a loop", async () => { - const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] }; - // A hexdump of zero-filled memory is highly repetitive but legitimate; the - // detector must not classify pure digit/whitespace runs as a loop. - const hexdump = "00 ".repeat(80); - const mock = createMockModel({ - provider: "google-gemini-cli", - responses: [{ content: [hexdump] }], - }); - const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter }; - - const stream = agentLoop([createUserMessage("dump the zero page")], context, config, undefined, mock.stream); - for await (const _event of stream) { - // drain the stream to completion - } - - const assistantMsg = (await stream.result())[1] as AssistantMessage; - expect(assistantMsg.stopReason).not.toBe("error"); - let text = ""; - for (const block of assistantMsg.content) { - if (block.type === "text") text += block.text; - } - expect(text).toBe(hexdump); - }); it("aborts pending tool calls instead of running them when the deadline is crossed during the request", async () => { const context: AgentContext = { systemPrompt: ["You are helpful."], diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 0d94ab1db..f86e4a669 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -1,16 +1,17 @@ # Changelog ## [Unreleased] - ### Added - Added `antigravityEndpointMode` stream option with `auto`, `production`, and `sandbox` values to control Antigravity endpoint routing - Added `seedApiKeyResolver` for reusing a pre-resolved request key while preserving resolver-driven auth retry and credential rotation - Added optional `contextSnapshot` property to `AssistantMessage` with token usage metadata via new `ContextSnapshot` interface (`promptTokens`, `nonMessageTokens`, and optional `lastMessageTimestamp`) - Added `LITELLM_BASE_URL` guidance to the LiteLLM login prompt so non-default proxy endpoints are discoverable. ([#2726](https://github.com/can1357/oh-my-pi/issues/2726)) +- Added a Gemini thinking-loop guard that watches streamed `thinking` deltas for degenerate reasoning loops — verbatim tail repetition and near-duplicate paragraph cycling — and terminates the stream with a retryable, empty-content `error` message (worded as a transient stream stall) so the turn is discarded and re-sampled instead of committing a runaway transcript. Gated to Gemini models across every transport (OpenRouter, direct Google, Vertex) and disarmed once visible answer text or a tool call starts; disable with `PI_NO_THINKING_LOOP_GUARD=1`. ### Fixed +- Fixed `streamSimple()` Gemini streams to run through the thinking-loop guard for custom API and pi-native transports, so degenerate `thinking` loops now abort with the same retryable empty-content error path as other Gemini stream paths - Fixed Antigravity model streaming and usage fetch paths to retry on transient `429`/`5xx` errors by failing over to the alternate endpoint before surfacing an error - Fixed Antigravity endpoint tracking to prefer a previously successful endpoint in `auto` mode for subsequent requests - Fixed Antigravity and Gemini CLI model requests failing with an opaque error when Google requires account verification. Cloud Code Assist `403 VALIDATION_REQUIRED` responses now surface the `validation_url` and the signed-in account email when available, so users see an actionable account-verification message instead of the raw API error body. diff --git a/packages/ai/src/index.ts b/packages/ai/src/index.ts index 6c7e4eeb1..922c79419 100644 --- a/packages/ai/src/index.ts +++ b/packages/ai/src/index.ts @@ -41,4 +41,5 @@ export * from "./utils/event-stream"; export * from "./utils/overflow"; export * from "./utils/retry"; export * from "./utils/schema"; +export * from "./utils/thinking-loop"; export * from "./utils/validation"; diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 01825333e..3b1dd8ab5 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -64,6 +64,7 @@ import type { } from "./types"; import { AssistantMessageEventStream } from "./utils/event-stream"; import { withRequestDebugFetch } from "./utils/request-debug"; +import { withGeminiThinkingLoopGuard } from "./utils/thinking-loop"; function isGoogleVertexAuthenticatedModel(model: Model): boolean { return ( @@ -227,6 +228,14 @@ export function stream( model: Model, context: Context, options?: OptionsForApi, +): AssistantMessageEventStream { + return withGeminiThinkingLoopGuard(model, options, opts => streamDispatch(model, context, opts)); +} + +function streamDispatch( + model: Model, + context: Context, + options?: OptionsForApi, ): AssistantMessageEventStream { const requestOptions = withRequestDebugFetch(options as StreamOptions | undefined) as | OptionsForApi @@ -489,13 +498,15 @@ export function streamSimple( // extension-registered APIs can't accidentally override a configured // pi-native transport. if (model.transport === "pi-native") { - return streamPiNative(model, context, requestOptions); + return withGeminiThinkingLoopGuard(model, requestOptions, opts => streamPiNative(model, context, opts)); } // Check custom API registry (extension-provided APIs) const customApiProvider = getCustomApi(model.api); if (customApiProvider) { - return customApiProvider.streamSimple(model, context, requestOptions); + return withGeminiThinkingLoopGuard(model, requestOptions, opts => + customApiProvider.streamSimple(model, context, opts), + ); } // Vertex AI uses Application Default Credentials, not API keys diff --git a/packages/ai/src/utils/thinking-loop.ts b/packages/ai/src/utils/thinking-loop.ts new file mode 100644 index 000000000..8033531be --- /dev/null +++ b/packages/ai/src/utils/thinking-loop.ts @@ -0,0 +1,354 @@ +/** + * Gemini thinking-loop guard. + * + * Gemini models (notably `gemini-3.5-flash` via OpenRouter) occasionally fall + * into a degenerate reasoning loop: they re-emit the same paragraph intent over + * and over with cosmetic wording drift ("Confirming Safety", "Verifying + * Completion", …), burning the entire output budget without ever calling a tool + * or answering. The runaway is *not* byte-identical, so a cheap verbatim + * tail-repeat check alone misses it. + * + * This guard watches the streamed `thinking` deltas and, on a match, terminates + * the stream with a synthetic `error` {@link AssistantMessage} that carries + * **no observable content**. An empty-content `stopReason: "error"` whose + * message hits the transient-transport pattern is what `AgentSession` + * classifies as a *retryable* stop (a contentful error stop is treated as + * replay-unsafe and is never retried), so the turn is discarded and re-sampled + * instead of committing the garbage transcript. + * + * Two failure shapes are detected: + * 1. **Verbatim tail repetition** — a short unit repeated back-to-back (e.g. + * "🌊 🌊 🌊 …"). Caught from a rolling 250-char tail. + * 2. **Near-duplicate segments** — paragraphs that normalize to the same + * word-trigram fingerprint. Caught with a Jaccard window over recent + * paragraphs. Thresholds were calibrated on a real loop transcript plus + * 13.5k non-loop thinking blocks (zero false positives; hardest negative + * scored 3 against the trigger of 4). + * + * Scope is deliberately narrow: **thinking only**. Answer text is left + * untouched so the guard can never discard already-streamed visible output. The + * guard is gated to Gemini models and wraps the provider stream, so it works + * across every Gemini transport (OpenRouter `openai-completions`, direct + * `google-generative-ai` / `google-gemini-cli`, Vertex). Disable with + * `PI_NO_THINKING_LOOP_GUARD=1`. + */ +import { logger } from "@oh-my-pi/pi-utils"; +import type { Api, AssistantMessage, Model } from "../types"; +import { AssistantMessageEventStream } from "./event-stream"; + +/** Stable lead phrase of the guard's error message; exported for tests. The + * message also carries "stream stall" so the session + transport retry + * classifiers treat it as a transient (retryable) stop without bespoke rules. */ +export const THINKING_LOOP_ERROR_MARKER = "Thinking loop detected"; + +/** Rolling tail (chars) inspected for verbatim back-to-back repetition. */ +const VERBATIM_TAIL_WINDOW = 250; +/** Minimum total repeated chars before a verbatim run counts as a loop. */ +const VERBATIM_MIN_REPEATED_CHARS = 180; +/** Longest unit length probed for a verbatim repeat. */ +const VERBATIM_MAX_UNIT = 60; + +/** Char cap for an unterminated segment; forces a flush so a wall-of-text loop + * (no blank lines / headings) still segments. */ +const SEGMENT_CHAR_CAP = 700; +/** Normalized-length floor below which a segment is ignored (too short to be a + * meaningful paragraph; bare headings must not trip detection). */ +const SEGMENT_MIN_NORM_CHARS = 60; +/** How many recent substantial segments are kept for similarity comparison. */ +const SEGMENT_WINDOW = 16; +/** Word-trigram Jaccard at/above which two segments count as near-duplicates. */ +const SEGMENT_SIMILARITY = 0.8; +/** Substantial segments required before detection may fire (warm-up). */ +const SEGMENT_MIN_COUNT = 8; +/** Near-duplicate cluster size (current + matches) that trips the loop. */ +const SEGMENT_MIN_CLUSTER = 4; + +const OPENAI_COMPAT_GUARDED_APIS: Partial> = { + "openai-completions": true, + "openai-responses": true, + "azure-openai-responses": true, + "openai-codex-responses": true, +}; + +/** + * True when `model` should be guarded for thinking loops (Gemini only). + * + * OpenAI-compat transports can serve Gemini under an arbitrary provider/id, so + * for those we trust the family-derived (and user-overridable) + * `compat.enableGeminiThinkingLoopGuard` flag set by the catalog. Direct Google + * transports always carry a clearly gemini-shaped id/provider, so a string + * match is sufficient (and works for hand-built models without a compat record). + */ +export function isGeminiThinkingLoopModel(model: Model): boolean { + if (OPENAI_COMPAT_GUARDED_APIS[model.api]) { + return ( + (model.compat as { enableGeminiThinkingLoopGuard?: boolean } | undefined)?.enableGeminiThinkingLoopGuard === + true + ); + } + return /gemini/i.test(`${model.provider}/${model.id}`); +} + +/** + * Stateful detector fed the streamed thinking deltas. `push` returns a + * human-readable reason the first time a loop shape is recognized; the caller + * is responsible for stopping after the first hit. + */ +export class ThinkingLoopDetector { + /** Rolling char tail for verbatim repeat detection. */ + #tail = ""; + /** Pending thinking text not yet split into completed segments. */ + #pending = ""; + /** Fingerprints of the most recent substantial segments (≤ SEGMENT_WINDOW). */ + #window: Set[] = []; + /** Count of substantial segments seen so far (warm-up gate). */ + #count = 0; + + push(delta: string): string | null { + if (!delta) return null; + + // 1. Verbatim back-to-back repetition over the rolling tail. + this.#tail += delta; + if (this.#tail.length > VERBATIM_TAIL_WINDOW) this.#tail = this.#tail.slice(-VERBATIM_TAIL_WINDOW); + const verbatim = detectVerbatimRepetition(this.#tail); + if (verbatim) { + const [unit, times] = verbatim; + return `repeated "${unit.trim()}" ${times}× back-to-back`; + } + + // 2. Near-duplicate paragraph loop. Append, then drain completed segments. + this.#pending += delta; + while (true) { + const boundary = /\n\s*\n/.exec(this.#pending); + let raw: string; + if (boundary) { + raw = this.#pending.slice(0, boundary.index); + this.#pending = this.#pending.slice(boundary.index + boundary[0].length); + } else if (this.#pending.length > SEGMENT_CHAR_CAP) { + // No boundary yet but the segment is runaway-long: force a flush. + raw = this.#pending.slice(0, SEGMENT_CHAR_CAP); + this.#pending = this.#pending.slice(SEGMENT_CHAR_CAP); + } else { + return null; + } + // An over-long segment is chunked so each piece stays comparable. + for (let rest = raw; rest.length > 0; ) { + const chunk = rest.length > SEGMENT_CHAR_CAP ? rest.slice(0, SEGMENT_CHAR_CAP) : rest; + rest = rest.slice(chunk.length); + const hit = this.#consumeSegment(chunk); + if (hit) return hit; + } + } + } + + /** Process the buffered trailing paragraph (one with no blank-line / heading + * terminator). Called when the thinking block ends so the final segment — + * which may be the one that completes a duplicate cluster — is not dropped. */ + flush(): string | null { + if (!this.#pending) return null; + let rest = this.#pending; + this.#pending = ""; + while (rest.length > 0) { + const chunk = rest.length > SEGMENT_CHAR_CAP ? rest.slice(0, SEGMENT_CHAR_CAP) : rest; + rest = rest.slice(chunk.length); + const hit = this.#consumeSegment(chunk); + if (hit) return hit; + } + return null; + } + + #consumeSegment(segment: string): string | null { + const normalized = normalizeSegment(segment); + if (normalized.length < SEGMENT_MIN_NORM_CHARS) return null; + + const fingerprint = trigramShingles(normalized); + let cluster = 1; + for (const prev of this.#window) { + if (jaccard(fingerprint, prev) >= SEGMENT_SIMILARITY) cluster++; + } + + this.#window.push(fingerprint); + if (this.#window.length > SEGMENT_WINDOW) this.#window.shift(); + this.#count++; + + if (this.#count >= SEGMENT_MIN_COUNT && cluster >= SEGMENT_MIN_CLUSTER) { + return `${cluster} near-identical segments within the last ${SEGMENT_WINDOW}`; + } + return null; + } +} + +/** + * Wrap a provider stream with the loop guard. `controller` is the guard's own + * abort handle: aborting it (after wiring it into the provider's signal via + * {@link withGeminiThinkingLoopGuard}) tears down the upstream once a loop + * trips. + */ +export function guardThinkingLoopStream( + inner: AssistantMessageEventStream, + model: Model, + controller: AbortController, +): AssistantMessageEventStream { + const outer = new AssistantMessageEventStream(); + const detector = new ThinkingLoopDetector(); + + void (async () => { + // Once any visible answer text or tool call starts, disarm: some providers + // (e.g. openai-completions with cumulative `reasoning_content`) re-emit + // fresh thinking deltas after ``, and tripping the retriable path + // then would discard already-streamed observable output. + let armed = true; + try { + for await (const event of inner) { + let detail: string | null = null; + if (armed && event.type === "thinking_delta") { + detail = detector.push(event.delta); + } else if (armed && (event.type === "thinking_end" || event.type === "done")) { + // Final paragraph of the block has no trailing blank line; flush it + // so the segment that completes a cluster is not dropped. `done` + // is the backstop for providers that omit a trailing thinking_end. + detail = detector.flush(); + } else if ( + event.type === "text_start" || + event.type === "text_delta" || + event.type === "toolcall_start" || + event.type === "toolcall_delta" + ) { + armed = false; + } + if (detail) { + logger.warn("Gemini thinking loop detected; aborting stream for retry.", { + model: model.id, + provider: model.provider, + detail, + }); + controller.abort(new Error(THINKING_LOOP_ERROR_MARKER)); + outer.push({ type: "error", reason: "error", error: buildThinkingLoopError(model, detail) }); + return; + } + outer.push(event); + if (outer.done) return; + } + if (!outer.done) { + try { + outer.end(await inner.result()); + } catch (err) { + outer.fail(err); + } + } + } catch (err) { + if (!outer.done) outer.fail(err); + } + })(); + + return outer; +} + +/** + * Apply the Gemini loop guard around a provider dispatch. For non-Gemini models + * (or when disabled) this is a transparent pass-through. For Gemini it injects a + * guard abort signal into the provider call so a detected loop tears down the + * upstream, then wraps the returned stream. + */ +export function withGeminiThinkingLoopGuard( + model: Model, + options: O | undefined, + dispatch: (options: O | undefined) => AssistantMessageEventStream, +): AssistantMessageEventStream { + if (process.env.PI_NO_THINKING_LOOP_GUARD === "1" || !isGeminiThinkingLoopModel(model)) { + return dispatch(options); + } + const controller = new AbortController(); + const caller = options?.signal; + const signal = caller ? AbortSignal.any([caller, controller.signal]) : controller.signal; + const merged = { ...(options ?? {}), signal } as O; + return guardThinkingLoopStream(dispatch(merged), model, controller); +} + +function buildThinkingLoopError(model: Model, detail: string): AssistantMessage { + return { + role: "assistant", + // Empty content is load-bearing: a contentful error stop is replay-unsafe + // and would NOT be auto-retried by the session. + content: [], + api: model.api, + provider: model.provider, + model: model.id, + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "error", + // "stream stall" makes the transport/session retry classifiers treat this + // as a transient (retryable) failure with no bespoke rule. + errorMessage: `${THINKING_LOOP_ERROR_MARKER}: the model repeated near-identical reasoning (${detail}). Treating as a stream stall and retrying.`, + timestamp: Date.now(), + }; +} + +/** + * Detect a short unit repeated back-to-back at the tail (verbatim loop). Only a + * unit carrying a letter or pictographic emoji counts — runs of digits, + * whitespace or punctuation are legitimate in tabular / hex / numeric output. + */ +function detectVerbatimRepetition(text: string): [unit: string, count: number] | null { + if (text.length < VERBATIM_MIN_REPEATED_CHARS) return null; + const windowSize = Math.min(text.length, VERBATIM_TAIL_WINDOW); + const searchSpace = text.slice(-windowSize); + + for (let len = 2; len <= VERBATIM_MAX_UNIT; len++) { + if (searchSpace.length < len * 4) continue; + const unit = searchSpace.slice(-len); + if (!/[\p{L}\p{Extended_Pictographic}]/u.test(unit)) continue; + + let count = 0; + let pos = searchSpace.length; + while (pos >= len) { + if (searchSpace.slice(pos - len, pos) === unit) { + count++; + pos -= len; + } else { + break; + } + } + if (count >= 4 && len * count >= VERBATIM_MIN_REPEATED_CHARS) return [unit, count]; + } + return null; +} + +/** Lowercase, drop code spans / paths / digits, keep only letter words. */ +function normalizeSegment(segment: string): string { + return segment + .toLowerCase() + .replace(/`[^`]*`/g, " ") + .replace(/\/[^\s`]+/g, " ") + .replace(/\d+/g, " ") + .replace(/[^a-z]+/g, " ") + .trim(); +} + +/** Word-trigram shingle set of a normalized segment. */ +function trigramShingles(normalized: string): Set { + const words = normalized.split(" ").filter(Boolean); + if (words.length < 3) return new Set(words.length > 0 ? [words.join(" ")] : []); + const shingles = new Set(); + for (let i = 0; i + 3 <= words.length; i++) { + shingles.add(`${words[i]} ${words[i + 1]} ${words[i + 2]}`); + } + return shingles; +} + +function jaccard(a: Set, b: Set): number { + if (a.size === 0 || b.size === 0) return 0; + const [small, large] = a.size < b.size ? [a, b] : [b, a]; + let intersection = 0; + for (const x of small) { + if (large.has(x)) intersection++; + } + const union = a.size + b.size - intersection; + return union === 0 ? 0 : intersection / union; +} diff --git a/packages/ai/test/thinking-loop.test.ts b/packages/ai/test/thinking-loop.test.ts new file mode 100644 index 000000000..4e45c08d2 --- /dev/null +++ b/packages/ai/test/thinking-loop.test.ts @@ -0,0 +1,282 @@ +import { describe, expect, test } from "bun:test"; +import { clearCustomApis } from "@oh-my-pi/pi-ai/api-registry"; +import { createMockModel, type MockContent, registerMockApi } from "@oh-my-pi/pi-ai/providers/mock"; +import { stream, streamSimple } from "@oh-my-pi/pi-ai/stream"; +import type { Api, AssistantMessage, AssistantMessageEvent, Context, Model } from "@oh-my-pi/pi-ai/types"; +import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; +import { + isGeminiThinkingLoopModel, + THINKING_LOOP_ERROR_MARKER, + ThinkingLoopDetector, + withGeminiThinkingLoopGuard, +} from "@oh-my-pi/pi-ai/utils/thinking-loop"; +import { isRetryableError } from "@oh-my-pi/pi-utils"; + +function context(): Context { + return { systemPrompt: [], messages: [{ role: "user", content: "go", timestamp: 0 }] }; +} + +async function collect(events: AsyncIterable): Promise { + const out: AssistantMessageEvent[] = []; + for await (const event of events) out.push(event); + return out; +} + +/** A degenerate near-duplicate reasoning loop: the same paragraph intent with + * cosmetic wording drift, blank-line separated (the gemini-3.5-flash shape). */ +function nearDuplicateLoop(paragraphs: number): string { + const variants = [ + "I am now verifying the test module to guarantee there are no compile errors and the code is completely safe.", + "I am now verifying the test module once more to ensure there are no compile errors and the code stays completely safe.", + "I am now re-verifying the test module to confirm there are no compile errors and the code remains completely safe.", + ]; + const out: string[] = []; + for (let i = 0; i < paragraphs; i++) { + out.push(`**Confirming Safety ${i}**\n\n${variants[i % variants.length]}`); + } + return out.join("\n\n\n"); +} + +/** Genuinely distinct reasoning paragraphs — must never trip the detector. */ +function distinctReasoning(): string { + return [ + "First I read the agent loop to understand how the stream wrapper commits the final assistant message.", + "Next I traced the retry classifier to see which error shapes are treated as transient by the session.", + "The Vertex transport needs ADC, so its credential path differs from the OpenRouter completions route entirely.", + "I should add a regression covering the empty-content terminal so the auto-retry gate stays satisfied later.", + "The tokenizer counts code points, which matters for the wide-emoji case in the truncation helper above here.", + "Compaction journals are per-request, so they never persist into the on-disk session transcript at all ever.", + "The mock provider emits one delta per block, so streaming nuances need a direct detector unit test instead.", + "Finally I will run the focused package suite and the type checker before touching the changelog entries now.", + "Telemetry spans wrap each provider call, so failing one must still close its span in the catch branch too.", + "A device default of CPU keeps the tiny model worker from crashing Bun on teardown across every platform now.", + ].join("\n\n\n"); +} + +describe("isGeminiThinkingLoopModel", () => { + test("matches direct and aggregator-routed gemini ids, not lookalikes", () => { + const gate = (provider: string, id: string) => isGeminiThinkingLoopModel(createMockModel({ provider, id }).model); + expect(gate("google", "gemini-3-pro-preview")).toBe(true); + expect(gate("openrouter", "google/gemini-3.5-flash")).toBe(true); + expect(gate("google-gemini-cli", "gemini-3-flash")).toBe(true); + expect(gate("openai", "gpt-5.5")).toBe(false); + expect(gate("google", "gemma-3-1b")).toBe(false); + }); + + test("trusts the compat flag over the id regex for every OpenAI-compat API", () => { + const gate = (api: string, id: string, enableGeminiThinkingLoopGuard: boolean) => + isGeminiThinkingLoopModel({ + api, + provider: "openrouter", + id, + compat: { enableGeminiThinkingLoopGuard }, + } as unknown as Model); + // Opaque proxy alias opted in despite a non-gemini id (completions + responses). + expect(gate("openai-completions", "my-fast-model", true)).toBe(true); + expect(gate("openai-responses", "my-fast-model", true)).toBe(true); + // Gemini-shaped id explicitly opted out stays off — the flag wins over the regex. + expect(gate("openai-completions", "gemini-3.5-flash", false)).toBe(false); + expect(gate("openai-responses", "gemini-3.5-flash", false)).toBe(false); + }); + + test("guards non-compat Gemini transports (Vertex, direct Google) via id", () => { + const gate = (api: string, provider: string, id: string) => + isGeminiThinkingLoopModel({ api, provider, id } as unknown as Model); + // Vertex has no OpenAICompat record; its canonical ids are gemini-shaped. + expect(gate("google-vertex", "google-vertex", "gemini-2.5-pro")).toBe(true); + expect(gate("google-generative-ai", "google", "gemini-3-pro")).toBe(true); + // Non-Gemini models on the same transports (e.g. Claude on Vertex) stay unguarded. + expect(gate("google-vertex", "google-vertex", "claude-sonnet-4")).toBe(false); + }); +}); + +describe("ThinkingLoopDetector", () => { + test("trips on near-duplicate segments fed as small streamed chunks", () => { + const detector = new ThinkingLoopDetector(); + const text = nearDuplicateLoop(12); + let detail: string | null = null; + for (let i = 0; i < text.length && !detail; i += 17) { + detail = detector.push(text.slice(i, i + 17)); + } + expect(detail).toContain("near-identical segments"); + }); + + test("trips on verbatim back-to-back repetition", () => { + const detector = new ThinkingLoopDetector(); + const detail = detector.push("🌊 ".repeat(120)); + expect(detail).toContain("back-to-back"); + }); + + test("does not trip on genuinely distinct reasoning paragraphs", () => { + const detector = new ThinkingLoopDetector(); + let detail: string | null = null; + const text = distinctReasoning(); + for (let i = 0; i < text.length && !detail; i += 23) { + detail = detector.push(text.slice(i, i + 23)); + } + // flush the trailing paragraph too + detail ??= detector.flush(); + expect(detail).toBeNull(); + }); + + test("flush() catches a final unterminated duplicate paragraph", () => { + const detector = new ThinkingLoopDetector(); + // Seven blank-line-separated dupes leave the eighth (cluster-completing) + // paragraph in the buffer with no trailing blank line. + const block = `${nearDuplicateLoop(7)}\n\n\nI am now verifying the test module to guarantee there are no compile errors and the code is completely safe.`; + expect(detector.push(block)).toBeNull(); + expect(detector.flush()).toContain("near-identical segments"); + }); + + test("does not trip on legitimate repetitive numeric output", () => { + // A zero-page hexdump is highly repetitive but legitimate: a unit with no + // letter or pictograph must never count as a loop. + const detector = new ThinkingLoopDetector(); + expect(detector.push("00 ".repeat(200))).toBeNull(); + }); + + test("does not trip on short requested repetitive text", () => { + // Below the repeated-char floor: a brief on-purpose repeat is not a loop. + const detector = new ThinkingLoopDetector(); + expect(detector.push("🌊 ".repeat(26))).toBeNull(); + }); +}); + +describe("gemini thinking-loop guard (stream wrapper)", () => { + function loopingThinkingResponse(): { content: MockContent[] } { + return { content: [{ type: "thinking", thinking: nearDuplicateLoop(12) }] }; + } + + test("terminates a gemini loop with a retryable empty-content error", async () => { + registerMockApi(); + try { + const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }); + mock.push(loopingThinkingResponse()); + + const result = await stream(mock.model, context()).result(); + + expect(result.stopReason).toBe("error"); + expect(result.content).toEqual([]); + expect(result.errorMessage).toContain(THINKING_LOOP_ERROR_MARKER); + // Empty content + transient phrasing is what makes the turn auto-retry. + expect(result.errorMessage).toContain("stream stall"); + expect(isRetryableError(new Error(result.errorMessage))).toBe(true); + } finally { + clearCustomApis(); + } + }); + + test("emits no observable thinking/text content before the error terminal", async () => { + registerMockApi(); + try { + const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }); + mock.push(loopingThinkingResponse()); + + const events = await collect(stream(mock.model, context())); + const terminal = events.at(-1); + expect(terminal?.type).toBe("error"); + // The guard must not forward the looping thinking_end / done. + expect(events.some(e => e.type === "thinking_end")).toBe(false); + expect(events.some(e => e.type === "done")).toBe(false); + } finally { + clearCustomApis(); + } + }); + + test("passes a non-gemini model through untouched even when it loops", async () => { + registerMockApi(); + try { + const mock = createMockModel({ provider: "openai", id: "gpt-5.5" }); + mock.push(loopingThinkingResponse()); + + const result = await stream(mock.model, context()).result(); + + expect(result.stopReason).toBe("stop"); + expect(result.content.some(b => b.type === "thinking")).toBe(true); + } finally { + clearCustomApis(); + } + }); + + test("does not trip on a healthy gemini turn that reasons then answers", async () => { + registerMockApi(); + try { + const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }); + mock.push({ content: [{ type: "thinking", thinking: distinctReasoning() }, "Here is the final answer."] }); + + const result = await stream(mock.model, context()).result(); + + expect(result.stopReason).toBe("stop"); + expect(result.content.some(b => b.type === "text")).toBe(true); + } finally { + clearCustomApis(); + } + }); + + test("does not retry once a loop re-emerges after visible answer text (armed latch)", async () => { + registerMockApi(); + try { + const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }); + // Healthy reasoning, then visible answer text, then a runaway loop re-emitted + // as cumulative reasoning after `` (the openai-completions shape). + mock.push({ + content: [ + { type: "thinking", thinking: distinctReasoning() }, + "Here is the final answer.", + { type: "thinking", thinking: nearDuplicateLoop(12) }, + ], + }); + + const result = await stream(mock.model, context()).result(); + + // Visible text already streamed, so the loop must NOT hijack the turn. + expect(result.stopReason).toBe("stop"); + expect(result.content.some(b => b.type === "text")).toBe(true); + } finally { + clearCustomApis(); + } + }); + + test("guards the streamSimple custom-api path (agent default entrypoint)", async () => { + registerMockApi(); + try { + const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }); + mock.push(loopingThinkingResponse()); + + const result = await streamSimple(mock.model, context()).result(); + + expect(result.stopReason).toBe("error"); + expect(result.content).toEqual([]); + expect(result.errorMessage).toContain(THINKING_LOOP_ERROR_MARKER); + expect(isRetryableError(new Error(result.errorMessage))).toBe(true); + } finally { + clearCustomApis(); + } + }); +}); + +describe("withGeminiThinkingLoopGuard (Vertex transport)", () => { + test("emits a retryable empty-content error for a looping Vertex Gemini stream", async () => { + const model = { api: "google-vertex", provider: "google-vertex", id: "gemini-2.5-pro" } as unknown as Model; + const partial = { role: "assistant", content: [] } as unknown as AssistantMessage; + + const guarded = withGeminiThinkingLoopGuard(model, undefined, () => { + const inner = new AssistantMessageEventStream(); + const events: AssistantMessageEvent[] = [ + { type: "start", partial }, + { type: "thinking_start", contentIndex: 0, partial }, + { type: "thinking_delta", contentIndex: 0, delta: nearDuplicateLoop(12), partial }, + { type: "thinking_end", contentIndex: 0, content: "", partial }, + { type: "done", reason: "stop", message: partial }, + ]; + for (const event of events) inner.push(event); + return inner; + }); + + const result = await guarded.result(); + expect(result.stopReason).toBe("error"); + expect(result.content.length).toBe(0); + expect(result.errorMessage).toContain(THINKING_LOOP_ERROR_MARKER); + expect(isRetryableError(new Error(result.errorMessage))).toBe(true); + }); +}); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 47babe0ce..ccddd2ad0 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -1,13 +1,16 @@ # Changelog ## [Unreleased] - ### Added +- Added `enableGeminiThinkingLoopGuard` to OpenAI compatibility options to allow explicit opt-in or opt-out of the Gemini thinking-loop guard for OpenAI-compatible model aliases - Added `LITELLM_BASE_URL` as the LiteLLM provider discovery base URL fallback, with discovery caches scoped by the resolved proxy URL and explicit provider `baseUrl` config kept at higher precedence. ([#2726](https://github.com/can1357/oh-my-pi/issues/2726)) + ### Changed +- Defaulted `enableGeminiThinkingLoopGuard` from Gemini family detection for both OpenAI completions and responses compatibility specs so Gemini models now enable the thinking-loop guard automatically - Updated the default Gemini CLI user-agent version fallback to 0.46.0. + ### Fixed - Routed google-antigravity default baseUrl to the stable primary daily endpoint in the catalog generator and all fallback snapshots, resolving connection drops on heavy queries. @@ -247,4 +250,4 @@ ### Removed -- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads. +- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads. \ No newline at end of file diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index 3e2f3d72b..b594cfeed 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -17,6 +17,7 @@ import { isKimiModelId, isMimoModelIdOrName, isQwenModelId, + modelFamilyToken, } from "../identity/family"; import type { ModelSpec, OpenAICompat, ResolvedOpenAICompat, ResolvedOpenAIResponsesCompat } from "../types"; import { applyCompatOverrides } from "./apply"; @@ -211,6 +212,10 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv supportsReasoningParams: provider !== "github-copilot", reasoningEffortMap: {}, supportsUsageInStreaming: !isCerebras, + // pi-ai's thinking-loop guard is gemini-only; default the flag from the + // family classifier so OpenAI-compat proxies serving Gemini are covered. + // An opaque alias can opt in via `compat.enableGeminiThinkingLoopGuard`. + enableGeminiThinkingLoopGuard: modelFamilyToken(spec.id) === "gemini", // Kimi (including via OpenRouter and Fireworks router-form IDs such as // `accounts/fireworks/routers/kimi-*`) calculates TPM rate limits based on // max_tokens, not actual output. The official Kimi K2 model guidance @@ -295,6 +300,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv } interface OpenAIResponsesSpecLike { + id?: string; provider: string; name: string; baseUrl: string; @@ -329,6 +335,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol strictResponsesPairing: isAzure || spec.provider === "github-copilot", requiresJuiceZeroHack: spec.name.toLowerCase().startsWith("gpt-5"), reasoningEffortMap: {}, + enableGeminiThinkingLoopGuard: modelFamilyToken(spec.id ?? "") === "gemini", }; applyCompatOverrides(compat, spec.compat); return compat; diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts index 0131e32d0..00dc9bbb0 100644 --- a/packages/catalog/src/types.ts +++ b/packages/catalog/src/types.ts @@ -170,6 +170,13 @@ export interface OpenAICompat { reasoningEffortMap?: Partial>; /** Whether the provider supports `stream_options: { include_usage: true }` for token usage in streaming responses. Default: true. */ supportsUsageInStreaming?: boolean; + /** + * Enable the Gemini thinking-loop guard (pi-ai stream layer) for this model. + * Defaults to true when the model id classifies as the gemini family. Set + * explicitly to cover an opaque OpenAI-compat proxy alias (e.g. `my-model`) + * that routes to Gemini, or to false to opt a gemini-family id out. + */ + enableGeminiThinkingLoopGuard?: boolean; /** Which field to use for max tokens. Default: auto-detected from URL. */ maxTokensField?: "max_completion_tokens" | "max_tokens"; /** Whether tool results require the `name` field. Default: auto-detected from URL. */ @@ -373,6 +380,7 @@ export type ResolvedOpenAICompat = Required< | "thinkingKeep" | "strictResponsesPairing" | "requiresJuiceZeroHack" + | "enableGeminiThinkingLoopGuard" | "whenThinking" > > & { @@ -387,6 +395,8 @@ export type ResolvedOpenAICompat = Required< isOpenRouterHost: boolean; /** The model sits behind Vercel AI Gateway. */ isVercelGatewayHost: boolean; + /** See {@link OpenAICompat.enableGeminiThinkingLoopGuard}. Set by the builder from the family classifier. */ + enableGeminiThinkingLoopGuard?: boolean; /** Complete alternate view for thinking-engaged requests; swap pointers, never spread. */ whenThinking?: ResolvedOpenAICompat; }; @@ -400,6 +410,8 @@ export interface ResolvedOpenAIResponsesCompat { strictResponsesPairing: boolean; requiresJuiceZeroHack: boolean; reasoningEffortMap: Partial>; + /** See {@link OpenAICompat.enableGeminiThinkingLoopGuard}. */ + enableGeminiThinkingLoopGuard?: boolean; } /** Fully-resolved anthropic-messages compat view (same contract as `ResolvedOpenAICompat`). */ diff --git a/packages/catalog/test/gemini-thinking-loop-compat.test.ts b/packages/catalog/test/gemini-thinking-loop-compat.test.ts new file mode 100644 index 000000000..8eb2b40f1 --- /dev/null +++ b/packages/catalog/test/gemini-thinking-loop-compat.test.ts @@ -0,0 +1,69 @@ +import { describe, expect, it } from "bun:test"; +import { buildOpenAICompat, buildOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai"; +import type { ModelSpec, OpenAICompat } from "@oh-my-pi/pi-catalog/types"; + +/** + * The pi-ai thinking-loop guard is gemini-only and, for `openai-completions` + * models, gates on `compat.enableGeminiThinkingLoopGuard`. `buildOpenAICompat` + * must default that flag from the family classifier and honor explicit + * overrides so an opaque OpenAI-compat proxy alias can opt in/out. + */ +function spec(id: string, compat?: OpenAICompat): ModelSpec<"openai-completions"> { + return { + api: "openai-completions", + id, + name: id, + provider: "custom", + baseUrl: "https://proxy.example.com/v1", + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + maxTokens: 32_000, + contextWindow: 200_000, + reasoning: true, + ...(compat ? { compat } : {}), + }; +} + +describe("buildOpenAICompat enableGeminiThinkingLoopGuard", () => { + it("defaults on for gemini-family ids, including aggregator namespaces", () => { + expect(buildOpenAICompat(spec("gemini-3.5-flash")).enableGeminiThinkingLoopGuard).toBe(true); + expect(buildOpenAICompat(spec("google/gemini-3-pro")).enableGeminiThinkingLoopGuard).toBe(true); + }); + + it("defaults off for non-gemini ids (incl. gemma lookalikes)", () => { + expect(buildOpenAICompat(spec("gpt-5.5")).enableGeminiThinkingLoopGuard).toBe(false); + expect(buildOpenAICompat(spec("gemma-3-1b")).enableGeminiThinkingLoopGuard).toBe(false); + }); + + it("lets an opaque proxy alias opt in via explicit compat override", () => { + const compat = buildOpenAICompat(spec("my-fast-model", { enableGeminiThinkingLoopGuard: true })); + expect(compat.enableGeminiThinkingLoopGuard).toBe(true); + }); + + it("lets a gemini-family id opt out via explicit compat override", () => { + const compat = buildOpenAICompat(spec("gemini-3.5-flash", { enableGeminiThinkingLoopGuard: false })); + expect(compat.enableGeminiThinkingLoopGuard).toBe(false); + }); +}); + +describe("buildOpenAIResponsesCompat enableGeminiThinkingLoopGuard", () => { + const responsesSpec = (id: string, compat?: OpenAICompat) => ({ + id, + name: id, + provider: "custom", + baseUrl: "https://proxy.example.com/v1", + ...(compat ? { compat } : {}), + }); + + it("defaults from the family classifier", () => { + expect(buildOpenAIResponsesCompat(responsesSpec("gemini-3-pro")).enableGeminiThinkingLoopGuard).toBe(true); + expect(buildOpenAIResponsesCompat(responsesSpec("gpt-5.5")).enableGeminiThinkingLoopGuard).toBe(false); + }); + + it("honors an explicit override for an opaque proxy alias", () => { + expect( + buildOpenAIResponsesCompat(responsesSpec("my-fast-model", { enableGeminiThinkingLoopGuard: true })) + .enableGeminiThinkingLoopGuard, + ).toBe(true); + }); +});