fix: fixed Gemini thinking-loop handling across streaming dispatch paths

- Added Gemini thinking-loop detection helpers for near-duplicate and verbatim output checks.
- Wrapped `stream`, `streamPiNative`, and `streamSimple` dispatches with the loop guard.
- Emitted retryable empty-content loop errors and stopped completion events on loop hits.
- Added `enableGeminiThinkingLoopGuard` options for OpenAI compatibility with Gemini defaults and overrides.
This commit is contained in:
can1357
2026-06-17 13:37:57 +02:00
parent 54d212ec33
commit 0d9ec354e7
12 changed files with 755 additions and 302 deletions
+4
View File
@@ -14,6 +14,10 @@
- Added agent-loop deadline support for graceful wall-clock session stops.
### Changed
- Changed Gemini repetition-loop detection to live in the pi-ai stream layer instead of the agent loop. The agent no longer runs its own Gemini-gated verbatim repetition check (`detectRepetition`/`truncateRepetition`); loops now surface as a retryable transient stream error that the standard auto-retry path discards and re-samples, rather than a committed contentful error message.
## [16.0.1] - 2026-06-15
### Fixed
+6 -177
View File
@@ -35,7 +35,7 @@ import {
signalListLabel,
} from "@oh-my-pi/pi-ai/utils/harmony-leak";
import { preferredDialect } from "@oh-my-pi/pi-catalog/identity";
import { logger, sanitizeText } from "@oh-my-pi/pi-utils";
import { sanitizeText } from "@oh-my-pi/pi-utils";
import { type AgentRunCoverage, type AgentRunSummary, ToolCallBlockedError } from "./run-collector";
import {
type AgentTelemetry,
@@ -1054,7 +1054,6 @@ async function streamAssistantResponse(
? AbortSignal.any([signal, harmonyAbortController.signal])
: harmonyAbortController.signal
: signal;
const repetitionAbortController = new AbortController();
// Owned tool calling: aborted by the stream wrapper when the model starts
// fabricating a `<tool_response>`, so the provider stops generating the rest of
// the hallucinated turn. Merged into the provider signal ONLY (not
@@ -1063,10 +1062,13 @@ async function streamAssistantResponse(
const promptToolAbortController = ownedDialect ? new AbortController() : undefined;
const providerAbortSignals: AbortSignal[] = [];
if (requestSignal) providerAbortSignals.push(requestSignal);
providerAbortSignals.push(repetitionAbortController.signal);
if (promptToolAbortController) providerAbortSignals.push(promptToolAbortController.signal);
const finalRequestSignal =
providerAbortSignals.length === 1 ? providerAbortSignals[0]! : AbortSignal.any(providerAbortSignals);
providerAbortSignals.length === 0
? undefined
: providerAbortSignals.length === 1
? providerAbortSignals[0]!
: AbortSignal.any(providerAbortSignals);
const requestApiKey = (config.getApiKey ? await config.getApiKey(config.model) : undefined) ?? config.apiKey;
const resolvedApiKey = await resolveApiKeyOnce(requestApiKey, finalRequestSignal);
const apiKey = isApiKeyResolver(requestApiKey) ? seedApiKeyResolver(resolvedApiKey, requestApiKey) : requestApiKey;
@@ -1173,56 +1175,6 @@ async function streamAssistantResponse(
return aborted;
};
const finishRepetitionStream = async (
kind: "text" | "thinking",
pattern: string,
count: number,
): Promise<AssistantMessage> => {
repetitionAbortController.abort();
try {
const cleanup = responseIterator.return?.();
if (cleanup) void cleanup.catch(() => {});
} catch {
// ignore
}
if (partialMessage) {
truncateRepetition(partialMessage, kind, pattern);
partialMessage.stopReason = "error";
partialMessage.errorMessage = `Repetition loop detected: assistant repeated "${pattern.trim()}" ${count} times consecutively.`;
}
const finalMsg = snapshotAssistantMessage(
partialMessage ?? {
role: "assistant",
content: [],
api: config.model.api,
provider: config.model.provider,
model: config.model.id,
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "error",
errorMessage: `Repetition loop detected.`,
timestamp: Date.now(),
},
);
if (addedPartial) {
context.messages[context.messages.length - 1] = finalMsg;
} else {
context.messages.push(finalMsg);
}
if (!addedPartial) {
stream.push({ type: "message_start", message: snapshotAssistantMessage(finalMsg) });
}
stream.push({ type: "message_end", message: snapshotAssistantMessage(finalMsg) });
await finishChat(finalMsg);
return finalMsg;
};
// Set up a single abort race: register the abort listener once for the whole
// stream and reuse the same race promise for every iterator.next() instead of
// allocating Promise.withResolvers and add/removeEventListener per event.
@@ -1239,14 +1191,6 @@ async function streamAssistantResponse(
detachAbortListener = () => requestSignal.removeEventListener("abort", onAbort);
}
// Rolling tail of streamed text/thinking used for repetition-loop detection.
// Bounded to REPETITION_WINDOW chars and reset when the active block kind
// switches (text <-> thinking) so detection stays O(1) per delta and never
// miscounts a repeated unit across a thinking/answer boundary.
let repetitionTail = "";
let repetitionKind: "text" | "thinking" | undefined;
const isGeminiModel = config.model.provider.includes("google") || config.model.provider.includes("gemini");
try {
while (true) {
let next: IteratorResult<AssistantMessageEvent>;
@@ -1331,27 +1275,6 @@ async function streamAssistantResponse(
assistantMessageEvent: snapshotAssistantMessageEvent(event),
message: snapshotAssistantMessage(partialMessage),
});
if (isGeminiModel && (event.type === "text_delta" || event.type === "thinking_delta")) {
const kind = event.type === "text_delta" ? "text" : "thinking";
if (repetitionKind !== kind) {
repetitionKind = kind;
repetitionTail = "";
}
repetitionTail += event.delta;
if (repetitionTail.length > REPETITION_WINDOW) {
repetitionTail = repetitionTail.slice(-REPETITION_WINDOW);
}
const repetition = detectRepetition(repetitionTail);
if (repetition) {
const [pattern, count] = repetition;
logger.warn("Repetition loop detected during assistant stream, aborting.", {
pattern,
count,
});
return await finishRepetitionStream(kind, pattern, count);
}
}
}
break;
}
@@ -1987,97 +1910,3 @@ function createSkippedToolResult(): AgentToolResult<any> {
details: {},
};
}
const REPETITION_WINDOW = 250;
const REPETITION_MIN_REPEATED_CHARS = 180;
function detectRepetition(text: string): [pattern: string, count: number] | null {
if (text.length < REPETITION_MIN_REPEATED_CHARS) return null;
const windowSize = Math.min(text.length, REPETITION_WINDOW);
const searchSpace = text.slice(-windowSize);
for (let len = 2; len <= 60; len++) {
if (searchSpace.length < len * 4) continue;
const pattern = searchSpace.slice(-len);
// Only treat a repeated unit as a pathological loop when it carries real
// linguistic content (a letter or a pictographic emoji). Runs made purely of
// digits, whitespace or punctuation are legitimate in tabular / hex / numeric
// output (e.g. "00 00 00", "0, 0, 0", "| -- | -- |") and must not trip.
if (!/[\p{L}\p{Extended_Pictographic}]/u.test(pattern)) continue;
let count = 0;
let pos = searchSpace.length;
while (pos >= len) {
const chunk = searchSpace.slice(pos - len, pos);
if (chunk === pattern) {
count++;
pos -= len;
} else {
break;
}
}
if (count >= 4 && len * count >= REPETITION_MIN_REPEATED_CHARS) {
return [pattern, count];
}
}
return null;
}
function truncateRepetition(message: AssistantMessage, kind: "text" | "thinking", pattern: string): void {
// A repetition loop streams into a single growing block (real providers) or a run
// of same-kind blocks (some transports), always at the tail of the message. Gather
// that trailing contiguous run and collapse its repeated copies down to one, so the
// committed transcript keeps a representative sample instead of the full runaway.
const matches = (block: AssistantContentBlock): boolean =>
kind === "text" ? block.type === "text" : block.type === "thinking";
const readBlock = (block: AssistantContentBlock): string =>
block.type === "text" ? block.text : block.type === "thinking" ? block.thinking : "";
const clearThinkingReplayAnchors = (block: AssistantContentBlock): void => {
if (block.type !== "thinking") return;
block.thinkingSignature = undefined;
block.itemId = undefined;
};
const writeBlock = (block: AssistantContentBlock, value: string): void => {
if (block.type === "text") {
block.text = value;
} else if (block.type === "thinking") {
block.thinking = value;
clearThinkingReplayAnchors(block);
}
};
const trailing: AssistantContentBlock[] = [];
for (let i = message.content.length - 1; i >= 0; i--) {
const block = message.content[i];
if (!matches(block)) break;
trailing.unshift(block);
}
if (trailing.length === 0) return;
if (kind === "thinking") {
for (const block of trailing) clearThinkingReplayAnchors(block);
}
let joined = "";
for (const block of trailing) joined += readBlock(block);
let kept = joined;
while (kept.length >= pattern.length * 2 && kept.slice(kept.length - pattern.length * 2) === pattern + pattern) {
kept = kept.slice(0, kept.length - pattern.length);
}
let remainingToRemove = joined.length - kept.length;
for (let i = trailing.length - 1; i >= 0 && remainingToRemove > 0; i--) {
const block = trailing[i];
const value = readBlock(block);
if (value.length <= remainingToRemove) {
remainingToRemove -= value.length;
writeBlock(block, "");
} else {
writeBlock(block, value.slice(0, value.length - remainingToRemove));
remainingToRemove = 0;
}
}
}
-120
View File
@@ -1879,126 +1879,6 @@ describe("agentLoopContinue with AgentMessage", () => {
}
});
it("should detect repetition loops during assistant stream and abort gracefully", async () => {
const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] };
const mock = createMockModel({
provider: "google-gemini-cli",
responses: [
{
content: Array.from({ length: 80 }, () => "🌊 "),
},
],
});
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter };
const stream = agentLoop([createUserMessage("Hello")], context, config, undefined, mock.stream);
for await (const _event of stream) {
// drain the stream to completion
}
const messages = await stream.result();
expect(messages.length).toBe(2);
expect(messages[1].role).toBe("assistant");
const assistantMsg = messages[1] as AssistantMessage;
expect(assistantMsg.stopReason).toBe("error");
expect(assistantMsg.errorMessage).toContain("Repetition loop detected");
let text = "";
for (const block of assistantMsg.content) {
if (block.type === "text") text += block.text;
}
expect(text).toBe("🌊 ");
});
it("detects and truncates repetition loops inside a thinking stream", async () => {
const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] };
const mock = createMockModel({
provider: "google-gemini-cli",
responses: [
{
content: Array.from({ length: 80 }, (_, index) => ({
type: "thinking" as const,
thinking: "🌊 ",
thinkingSignature: `signature-${index}`,
itemId: `rs_${index}`,
})),
},
],
});
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter };
const stream = agentLoop([createUserMessage("Hello")], context, config, undefined, mock.stream);
for await (const _event of stream) {
// drain the stream to completion
}
const assistantMsg = (await stream.result())[1] as AssistantMessage;
expect(assistantMsg.stopReason).toBe("error");
expect(assistantMsg.errorMessage).toContain("Repetition loop detected");
// A looping thinking stream must be both detected AND collapsed to a single
// representative copy — not committed to the transcript in full.
let thinking = "";
for (const block of assistantMsg.content) {
if (block.type === "thinking") thinking += block.thinking;
}
expect(thinking).toBe("🌊 ");
for (const block of assistantMsg.content) {
if (block.type === "thinking") {
expect(block.thinkingSignature).toBeUndefined();
expect(block.itemId).toBeUndefined();
}
}
});
it("does not flag short requested repetitive text as a loop", async () => {
const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] };
const repeated = "🌊 ".repeat(26);
const mock = createMockModel({
provider: "google-gemini-cli",
responses: [{ content: [repeated] }],
});
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter };
const stream = agentLoop([createUserMessage("print 26 wave emoji")], context, config, undefined, mock.stream);
for await (const _event of stream) {
// drain the stream to completion
}
const assistantMsg = (await stream.result())[1] as AssistantMessage;
expect(assistantMsg.stopReason).not.toBe("error");
let text = "";
for (const block of assistantMsg.content) {
if (block.type === "text") text += block.text;
}
expect(text).toBe(repeated);
});
it("does not flag legitimate repetitive numeric output as a loop", async () => {
const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] };
// A hexdump of zero-filled memory is highly repetitive but legitimate; the
// detector must not classify pure digit/whitespace runs as a loop.
const hexdump = "00 ".repeat(80);
const mock = createMockModel({
provider: "google-gemini-cli",
responses: [{ content: [hexdump] }],
});
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter };
const stream = agentLoop([createUserMessage("dump the zero page")], context, config, undefined, mock.stream);
for await (const _event of stream) {
// drain the stream to completion
}
const assistantMsg = (await stream.result())[1] as AssistantMessage;
expect(assistantMsg.stopReason).not.toBe("error");
let text = "";
for (const block of assistantMsg.content) {
if (block.type === "text") text += block.text;
}
expect(text).toBe(hexdump);
});
it("aborts pending tool calls instead of running them when the deadline is crossed during the request", async () => {
const context: AgentContext = {
systemPrompt: ["You are helpful."],
+2 -1
View File
@@ -1,16 +1,17 @@
# Changelog
## [Unreleased]
### Added
- Added `antigravityEndpointMode` stream option with `auto`, `production`, and `sandbox` values to control Antigravity endpoint routing
- Added `seedApiKeyResolver` for reusing a pre-resolved request key while preserving resolver-driven auth retry and credential rotation
- Added optional `contextSnapshot` property to `AssistantMessage` with token usage metadata via new `ContextSnapshot` interface (`promptTokens`, `nonMessageTokens`, and optional `lastMessageTimestamp`)
- Added `LITELLM_BASE_URL` guidance to the LiteLLM login prompt so non-default proxy endpoints are discoverable. ([#2726](https://github.com/can1357/oh-my-pi/issues/2726))
- Added a Gemini thinking-loop guard that watches streamed `thinking` deltas for degenerate reasoning loops — verbatim tail repetition and near-duplicate paragraph cycling — and terminates the stream with a retryable, empty-content `error` message (worded as a transient stream stall) so the turn is discarded and re-sampled instead of committing a runaway transcript. Gated to Gemini models across every transport (OpenRouter, direct Google, Vertex) and disarmed once visible answer text or a tool call starts; disable with `PI_NO_THINKING_LOOP_GUARD=1`.
### Fixed
- Fixed `streamSimple()` Gemini streams to run through the thinking-loop guard for custom API and pi-native transports, so degenerate `thinking` loops now abort with the same retryable empty-content error path as other Gemini stream paths
- Fixed Antigravity model streaming and usage fetch paths to retry on transient `429`/`5xx` errors by failing over to the alternate endpoint before surfacing an error
- Fixed Antigravity endpoint tracking to prefer a previously successful endpoint in `auto` mode for subsequent requests
- Fixed Antigravity and Gemini CLI model requests failing with an opaque error when Google requires account verification. Cloud Code Assist `403 VALIDATION_REQUIRED` responses now surface the `validation_url` and the signed-in account email when available, so users see an actionable account-verification message instead of the raw API error body.
+1
View File
@@ -41,4 +41,5 @@ export * from "./utils/event-stream";
export * from "./utils/overflow";
export * from "./utils/retry";
export * from "./utils/schema";
export * from "./utils/thinking-loop";
export * from "./utils/validation";
+13 -2
View File
@@ -64,6 +64,7 @@ import type {
} from "./types";
import { AssistantMessageEventStream } from "./utils/event-stream";
import { withRequestDebugFetch } from "./utils/request-debug";
import { withGeminiThinkingLoopGuard } from "./utils/thinking-loop";
function isGoogleVertexAuthenticatedModel(model: Model<Api>): boolean {
return (
@@ -227,6 +228,14 @@ export function stream<TApi extends Api>(
model: Model<TApi>,
context: Context,
options?: OptionsForApi<TApi>,
): AssistantMessageEventStream {
return withGeminiThinkingLoopGuard(model, options, opts => streamDispatch(model, context, opts));
}
function streamDispatch<TApi extends Api>(
model: Model<TApi>,
context: Context,
options?: OptionsForApi<TApi>,
): AssistantMessageEventStream {
const requestOptions = withRequestDebugFetch(options as StreamOptions | undefined) as
| OptionsForApi<TApi>
@@ -489,13 +498,15 @@ export function streamSimple<TApi extends Api>(
// extension-registered APIs can't accidentally override a configured
// pi-native transport.
if (model.transport === "pi-native") {
return streamPiNative(model, context, requestOptions);
return withGeminiThinkingLoopGuard(model, requestOptions, opts => streamPiNative(model, context, opts));
}
// Check custom API registry (extension-provided APIs)
const customApiProvider = getCustomApi(model.api);
if (customApiProvider) {
return customApiProvider.streamSimple(model, context, requestOptions);
return withGeminiThinkingLoopGuard(model, requestOptions, opts =>
customApiProvider.streamSimple(model, context, opts),
);
}
// Vertex AI uses Application Default Credentials, not API keys
+354
View File
@@ -0,0 +1,354 @@
/**
* Gemini thinking-loop guard.
*
* Gemini models (notably `gemini-3.5-flash` via OpenRouter) occasionally fall
* into a degenerate reasoning loop: they re-emit the same paragraph intent over
* and over with cosmetic wording drift ("Confirming Safety", "Verifying
* Completion", …), burning the entire output budget without ever calling a tool
* or answering. The runaway is *not* byte-identical, so a cheap verbatim
* tail-repeat check alone misses it.
*
* This guard watches the streamed `thinking` deltas and, on a match, terminates
* the stream with a synthetic `error` {@link AssistantMessage} that carries
* **no observable content**. An empty-content `stopReason: "error"` whose
* message hits the transient-transport pattern is what `AgentSession`
* classifies as a *retryable* stop (a contentful error stop is treated as
* replay-unsafe and is never retried), so the turn is discarded and re-sampled
* instead of committing the garbage transcript.
*
* Two failure shapes are detected:
* 1. **Verbatim tail repetition** — a short unit repeated back-to-back (e.g.
* "🌊 🌊 🌊 …"). Caught from a rolling 250-char tail.
* 2. **Near-duplicate segments** — paragraphs that normalize to the same
* word-trigram fingerprint. Caught with a Jaccard window over recent
* paragraphs. Thresholds were calibrated on a real loop transcript plus
* 13.5k non-loop thinking blocks (zero false positives; hardest negative
* scored 3 against the trigger of 4).
*
* Scope is deliberately narrow: **thinking only**. Answer text is left
* untouched so the guard can never discard already-streamed visible output. The
* guard is gated to Gemini models and wraps the provider stream, so it works
* across every Gemini transport (OpenRouter `openai-completions`, direct
* `google-generative-ai` / `google-gemini-cli`, Vertex). Disable with
* `PI_NO_THINKING_LOOP_GUARD=1`.
*/
import { logger } from "@oh-my-pi/pi-utils";
import type { Api, AssistantMessage, Model } from "../types";
import { AssistantMessageEventStream } from "./event-stream";
/** Stable lead phrase of the guard's error message; exported for tests. The
* message also carries "stream stall" so the session + transport retry
* classifiers treat it as a transient (retryable) stop without bespoke rules. */
export const THINKING_LOOP_ERROR_MARKER = "Thinking loop detected";
/** Rolling tail (chars) inspected for verbatim back-to-back repetition. */
const VERBATIM_TAIL_WINDOW = 250;
/** Minimum total repeated chars before a verbatim run counts as a loop. */
const VERBATIM_MIN_REPEATED_CHARS = 180;
/** Longest unit length probed for a verbatim repeat. */
const VERBATIM_MAX_UNIT = 60;
/** Char cap for an unterminated segment; forces a flush so a wall-of-text loop
* (no blank lines / headings) still segments. */
const SEGMENT_CHAR_CAP = 700;
/** Normalized-length floor below which a segment is ignored (too short to be a
* meaningful paragraph; bare headings must not trip detection). */
const SEGMENT_MIN_NORM_CHARS = 60;
/** How many recent substantial segments are kept for similarity comparison. */
const SEGMENT_WINDOW = 16;
/** Word-trigram Jaccard at/above which two segments count as near-duplicates. */
const SEGMENT_SIMILARITY = 0.8;
/** Substantial segments required before detection may fire (warm-up). */
const SEGMENT_MIN_COUNT = 8;
/** Near-duplicate cluster size (current + matches) that trips the loop. */
const SEGMENT_MIN_CLUSTER = 4;
const OPENAI_COMPAT_GUARDED_APIS: Partial<Record<Api, true>> = {
"openai-completions": true,
"openai-responses": true,
"azure-openai-responses": true,
"openai-codex-responses": true,
};
/**
* True when `model` should be guarded for thinking loops (Gemini only).
*
* OpenAI-compat transports can serve Gemini under an arbitrary provider/id, so
* for those we trust the family-derived (and user-overridable)
* `compat.enableGeminiThinkingLoopGuard` flag set by the catalog. Direct Google
* transports always carry a clearly gemini-shaped id/provider, so a string
* match is sufficient (and works for hand-built models without a compat record).
*/
export function isGeminiThinkingLoopModel(model: Model<Api>): boolean {
if (OPENAI_COMPAT_GUARDED_APIS[model.api]) {
return (
(model.compat as { enableGeminiThinkingLoopGuard?: boolean } | undefined)?.enableGeminiThinkingLoopGuard ===
true
);
}
return /gemini/i.test(`${model.provider}/${model.id}`);
}
/**
* Stateful detector fed the streamed thinking deltas. `push` returns a
* human-readable reason the first time a loop shape is recognized; the caller
* is responsible for stopping after the first hit.
*/
export class ThinkingLoopDetector {
/** Rolling char tail for verbatim repeat detection. */
#tail = "";
/** Pending thinking text not yet split into completed segments. */
#pending = "";
/** Fingerprints of the most recent substantial segments (≤ SEGMENT_WINDOW). */
#window: Set<string>[] = [];
/** Count of substantial segments seen so far (warm-up gate). */
#count = 0;
push(delta: string): string | null {
if (!delta) return null;
// 1. Verbatim back-to-back repetition over the rolling tail.
this.#tail += delta;
if (this.#tail.length > VERBATIM_TAIL_WINDOW) this.#tail = this.#tail.slice(-VERBATIM_TAIL_WINDOW);
const verbatim = detectVerbatimRepetition(this.#tail);
if (verbatim) {
const [unit, times] = verbatim;
return `repeated "${unit.trim()}" ${times}× back-to-back`;
}
// 2. Near-duplicate paragraph loop. Append, then drain completed segments.
this.#pending += delta;
while (true) {
const boundary = /\n\s*\n/.exec(this.#pending);
let raw: string;
if (boundary) {
raw = this.#pending.slice(0, boundary.index);
this.#pending = this.#pending.slice(boundary.index + boundary[0].length);
} else if (this.#pending.length > SEGMENT_CHAR_CAP) {
// No boundary yet but the segment is runaway-long: force a flush.
raw = this.#pending.slice(0, SEGMENT_CHAR_CAP);
this.#pending = this.#pending.slice(SEGMENT_CHAR_CAP);
} else {
return null;
}
// An over-long segment is chunked so each piece stays comparable.
for (let rest = raw; rest.length > 0; ) {
const chunk = rest.length > SEGMENT_CHAR_CAP ? rest.slice(0, SEGMENT_CHAR_CAP) : rest;
rest = rest.slice(chunk.length);
const hit = this.#consumeSegment(chunk);
if (hit) return hit;
}
}
}
/** Process the buffered trailing paragraph (one with no blank-line / heading
* terminator). Called when the thinking block ends so the final segment —
* which may be the one that completes a duplicate cluster — is not dropped. */
flush(): string | null {
if (!this.#pending) return null;
let rest = this.#pending;
this.#pending = "";
while (rest.length > 0) {
const chunk = rest.length > SEGMENT_CHAR_CAP ? rest.slice(0, SEGMENT_CHAR_CAP) : rest;
rest = rest.slice(chunk.length);
const hit = this.#consumeSegment(chunk);
if (hit) return hit;
}
return null;
}
#consumeSegment(segment: string): string | null {
const normalized = normalizeSegment(segment);
if (normalized.length < SEGMENT_MIN_NORM_CHARS) return null;
const fingerprint = trigramShingles(normalized);
let cluster = 1;
for (const prev of this.#window) {
if (jaccard(fingerprint, prev) >= SEGMENT_SIMILARITY) cluster++;
}
this.#window.push(fingerprint);
if (this.#window.length > SEGMENT_WINDOW) this.#window.shift();
this.#count++;
if (this.#count >= SEGMENT_MIN_COUNT && cluster >= SEGMENT_MIN_CLUSTER) {
return `${cluster} near-identical segments within the last ${SEGMENT_WINDOW}`;
}
return null;
}
}
/**
* Wrap a provider stream with the loop guard. `controller` is the guard's own
* abort handle: aborting it (after wiring it into the provider's signal via
* {@link withGeminiThinkingLoopGuard}) tears down the upstream once a loop
* trips.
*/
export function guardThinkingLoopStream(
inner: AssistantMessageEventStream,
model: Model<Api>,
controller: AbortController,
): AssistantMessageEventStream {
const outer = new AssistantMessageEventStream();
const detector = new ThinkingLoopDetector();
void (async () => {
// Once any visible answer text or tool call starts, disarm: some providers
// (e.g. openai-completions with cumulative `reasoning_content`) re-emit
// fresh thinking deltas after `</think>`, and tripping the retriable path
// then would discard already-streamed observable output.
let armed = true;
try {
for await (const event of inner) {
let detail: string | null = null;
if (armed && event.type === "thinking_delta") {
detail = detector.push(event.delta);
} else if (armed && (event.type === "thinking_end" || event.type === "done")) {
// Final paragraph of the block has no trailing blank line; flush it
// so the segment that completes a cluster is not dropped. `done`
// is the backstop for providers that omit a trailing thinking_end.
detail = detector.flush();
} else if (
event.type === "text_start" ||
event.type === "text_delta" ||
event.type === "toolcall_start" ||
event.type === "toolcall_delta"
) {
armed = false;
}
if (detail) {
logger.warn("Gemini thinking loop detected; aborting stream for retry.", {
model: model.id,
provider: model.provider,
detail,
});
controller.abort(new Error(THINKING_LOOP_ERROR_MARKER));
outer.push({ type: "error", reason: "error", error: buildThinkingLoopError(model, detail) });
return;
}
outer.push(event);
if (outer.done) return;
}
if (!outer.done) {
try {
outer.end(await inner.result());
} catch (err) {
outer.fail(err);
}
}
} catch (err) {
if (!outer.done) outer.fail(err);
}
})();
return outer;
}
/**
* Apply the Gemini loop guard around a provider dispatch. For non-Gemini models
* (or when disabled) this is a transparent pass-through. For Gemini it injects a
* guard abort signal into the provider call so a detected loop tears down the
* upstream, then wraps the returned stream.
*/
export function withGeminiThinkingLoopGuard<O extends { signal?: AbortSignal }>(
model: Model<Api>,
options: O | undefined,
dispatch: (options: O | undefined) => AssistantMessageEventStream,
): AssistantMessageEventStream {
if (process.env.PI_NO_THINKING_LOOP_GUARD === "1" || !isGeminiThinkingLoopModel(model)) {
return dispatch(options);
}
const controller = new AbortController();
const caller = options?.signal;
const signal = caller ? AbortSignal.any([caller, controller.signal]) : controller.signal;
const merged = { ...(options ?? {}), signal } as O;
return guardThinkingLoopStream(dispatch(merged), model, controller);
}
function buildThinkingLoopError(model: Model<Api>, detail: string): AssistantMessage {
return {
role: "assistant",
// Empty content is load-bearing: a contentful error stop is replay-unsafe
// and would NOT be auto-retried by the session.
content: [],
api: model.api,
provider: model.provider,
model: model.id,
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "error",
// "stream stall" makes the transport/session retry classifiers treat this
// as a transient (retryable) failure with no bespoke rule.
errorMessage: `${THINKING_LOOP_ERROR_MARKER}: the model repeated near-identical reasoning (${detail}). Treating as a stream stall and retrying.`,
timestamp: Date.now(),
};
}
/**
* Detect a short unit repeated back-to-back at the tail (verbatim loop). Only a
* unit carrying a letter or pictographic emoji counts — runs of digits,
* whitespace or punctuation are legitimate in tabular / hex / numeric output.
*/
function detectVerbatimRepetition(text: string): [unit: string, count: number] | null {
if (text.length < VERBATIM_MIN_REPEATED_CHARS) return null;
const windowSize = Math.min(text.length, VERBATIM_TAIL_WINDOW);
const searchSpace = text.slice(-windowSize);
for (let len = 2; len <= VERBATIM_MAX_UNIT; len++) {
if (searchSpace.length < len * 4) continue;
const unit = searchSpace.slice(-len);
if (!/[\p{L}\p{Extended_Pictographic}]/u.test(unit)) continue;
let count = 0;
let pos = searchSpace.length;
while (pos >= len) {
if (searchSpace.slice(pos - len, pos) === unit) {
count++;
pos -= len;
} else {
break;
}
}
if (count >= 4 && len * count >= VERBATIM_MIN_REPEATED_CHARS) return [unit, count];
}
return null;
}
/** Lowercase, drop code spans / paths / digits, keep only letter words. */
function normalizeSegment(segment: string): string {
return segment
.toLowerCase()
.replace(/`[^`]*`/g, " ")
.replace(/\/[^\s`]+/g, " ")
.replace(/\d+/g, " ")
.replace(/[^a-z]+/g, " ")
.trim();
}
/** Word-trigram shingle set of a normalized segment. */
function trigramShingles(normalized: string): Set<string> {
const words = normalized.split(" ").filter(Boolean);
if (words.length < 3) return new Set(words.length > 0 ? [words.join(" ")] : []);
const shingles = new Set<string>();
for (let i = 0; i + 3 <= words.length; i++) {
shingles.add(`${words[i]} ${words[i + 1]} ${words[i + 2]}`);
}
return shingles;
}
function jaccard(a: Set<string>, b: Set<string>): number {
if (a.size === 0 || b.size === 0) return 0;
const [small, large] = a.size < b.size ? [a, b] : [b, a];
let intersection = 0;
for (const x of small) {
if (large.has(x)) intersection++;
}
const union = a.size + b.size - intersection;
return union === 0 ? 0 : intersection / union;
}
+282
View File
@@ -0,0 +1,282 @@
import { describe, expect, test } from "bun:test";
import { clearCustomApis } from "@oh-my-pi/pi-ai/api-registry";
import { createMockModel, type MockContent, registerMockApi } from "@oh-my-pi/pi-ai/providers/mock";
import { stream, streamSimple } from "@oh-my-pi/pi-ai/stream";
import type { Api, AssistantMessage, AssistantMessageEvent, Context, Model } from "@oh-my-pi/pi-ai/types";
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
import {
isGeminiThinkingLoopModel,
THINKING_LOOP_ERROR_MARKER,
ThinkingLoopDetector,
withGeminiThinkingLoopGuard,
} from "@oh-my-pi/pi-ai/utils/thinking-loop";
import { isRetryableError } from "@oh-my-pi/pi-utils";
function context(): Context {
return { systemPrompt: [], messages: [{ role: "user", content: "go", timestamp: 0 }] };
}
async function collect(events: AsyncIterable<AssistantMessageEvent>): Promise<AssistantMessageEvent[]> {
const out: AssistantMessageEvent[] = [];
for await (const event of events) out.push(event);
return out;
}
/** A degenerate near-duplicate reasoning loop: the same paragraph intent with
* cosmetic wording drift, blank-line separated (the gemini-3.5-flash shape). */
function nearDuplicateLoop(paragraphs: number): string {
const variants = [
"I am now verifying the test module to guarantee there are no compile errors and the code is completely safe.",
"I am now verifying the test module once more to ensure there are no compile errors and the code stays completely safe.",
"I am now re-verifying the test module to confirm there are no compile errors and the code remains completely safe.",
];
const out: string[] = [];
for (let i = 0; i < paragraphs; i++) {
out.push(`**Confirming Safety ${i}**\n\n${variants[i % variants.length]}`);
}
return out.join("\n\n\n");
}
/** Genuinely distinct reasoning paragraphs — must never trip the detector. */
function distinctReasoning(): string {
return [
"First I read the agent loop to understand how the stream wrapper commits the final assistant message.",
"Next I traced the retry classifier to see which error shapes are treated as transient by the session.",
"The Vertex transport needs ADC, so its credential path differs from the OpenRouter completions route entirely.",
"I should add a regression covering the empty-content terminal so the auto-retry gate stays satisfied later.",
"The tokenizer counts code points, which matters for the wide-emoji case in the truncation helper above here.",
"Compaction journals are per-request, so they never persist into the on-disk session transcript at all ever.",
"The mock provider emits one delta per block, so streaming nuances need a direct detector unit test instead.",
"Finally I will run the focused package suite and the type checker before touching the changelog entries now.",
"Telemetry spans wrap each provider call, so failing one must still close its span in the catch branch too.",
"A device default of CPU keeps the tiny model worker from crashing Bun on teardown across every platform now.",
].join("\n\n\n");
}
describe("isGeminiThinkingLoopModel", () => {
test("matches direct and aggregator-routed gemini ids, not lookalikes", () => {
const gate = (provider: string, id: string) => isGeminiThinkingLoopModel(createMockModel({ provider, id }).model);
expect(gate("google", "gemini-3-pro-preview")).toBe(true);
expect(gate("openrouter", "google/gemini-3.5-flash")).toBe(true);
expect(gate("google-gemini-cli", "gemini-3-flash")).toBe(true);
expect(gate("openai", "gpt-5.5")).toBe(false);
expect(gate("google", "gemma-3-1b")).toBe(false);
});
test("trusts the compat flag over the id regex for every OpenAI-compat API", () => {
const gate = (api: string, id: string, enableGeminiThinkingLoopGuard: boolean) =>
isGeminiThinkingLoopModel({
api,
provider: "openrouter",
id,
compat: { enableGeminiThinkingLoopGuard },
} as unknown as Model<Api>);
// Opaque proxy alias opted in despite a non-gemini id (completions + responses).
expect(gate("openai-completions", "my-fast-model", true)).toBe(true);
expect(gate("openai-responses", "my-fast-model", true)).toBe(true);
// Gemini-shaped id explicitly opted out stays off — the flag wins over the regex.
expect(gate("openai-completions", "gemini-3.5-flash", false)).toBe(false);
expect(gate("openai-responses", "gemini-3.5-flash", false)).toBe(false);
});
test("guards non-compat Gemini transports (Vertex, direct Google) via id", () => {
const gate = (api: string, provider: string, id: string) =>
isGeminiThinkingLoopModel({ api, provider, id } as unknown as Model<Api>);
// Vertex has no OpenAICompat record; its canonical ids are gemini-shaped.
expect(gate("google-vertex", "google-vertex", "gemini-2.5-pro")).toBe(true);
expect(gate("google-generative-ai", "google", "gemini-3-pro")).toBe(true);
// Non-Gemini models on the same transports (e.g. Claude on Vertex) stay unguarded.
expect(gate("google-vertex", "google-vertex", "claude-sonnet-4")).toBe(false);
});
});
describe("ThinkingLoopDetector", () => {
test("trips on near-duplicate segments fed as small streamed chunks", () => {
const detector = new ThinkingLoopDetector();
const text = nearDuplicateLoop(12);
let detail: string | null = null;
for (let i = 0; i < text.length && !detail; i += 17) {
detail = detector.push(text.slice(i, i + 17));
}
expect(detail).toContain("near-identical segments");
});
test("trips on verbatim back-to-back repetition", () => {
const detector = new ThinkingLoopDetector();
const detail = detector.push("🌊 ".repeat(120));
expect(detail).toContain("back-to-back");
});
test("does not trip on genuinely distinct reasoning paragraphs", () => {
const detector = new ThinkingLoopDetector();
let detail: string | null = null;
const text = distinctReasoning();
for (let i = 0; i < text.length && !detail; i += 23) {
detail = detector.push(text.slice(i, i + 23));
}
// flush the trailing paragraph too
detail ??= detector.flush();
expect(detail).toBeNull();
});
test("flush() catches a final unterminated duplicate paragraph", () => {
const detector = new ThinkingLoopDetector();
// Seven blank-line-separated dupes leave the eighth (cluster-completing)
// paragraph in the buffer with no trailing blank line.
const block = `${nearDuplicateLoop(7)}\n\n\nI am now verifying the test module to guarantee there are no compile errors and the code is completely safe.`;
expect(detector.push(block)).toBeNull();
expect(detector.flush()).toContain("near-identical segments");
});
test("does not trip on legitimate repetitive numeric output", () => {
// A zero-page hexdump is highly repetitive but legitimate: a unit with no
// letter or pictograph must never count as a loop.
const detector = new ThinkingLoopDetector();
expect(detector.push("00 ".repeat(200))).toBeNull();
});
test("does not trip on short requested repetitive text", () => {
// Below the repeated-char floor: a brief on-purpose repeat is not a loop.
const detector = new ThinkingLoopDetector();
expect(detector.push("🌊 ".repeat(26))).toBeNull();
});
});
describe("gemini thinking-loop guard (stream wrapper)", () => {
function loopingThinkingResponse(): { content: MockContent[] } {
return { content: [{ type: "thinking", thinking: nearDuplicateLoop(12) }] };
}
test("terminates a gemini loop with a retryable empty-content error", async () => {
registerMockApi();
try {
const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" });
mock.push(loopingThinkingResponse());
const result = await stream(mock.model, context()).result();
expect(result.stopReason).toBe("error");
expect(result.content).toEqual([]);
expect(result.errorMessage).toContain(THINKING_LOOP_ERROR_MARKER);
// Empty content + transient phrasing is what makes the turn auto-retry.
expect(result.errorMessage).toContain("stream stall");
expect(isRetryableError(new Error(result.errorMessage))).toBe(true);
} finally {
clearCustomApis();
}
});
test("emits no observable thinking/text content before the error terminal", async () => {
registerMockApi();
try {
const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" });
mock.push(loopingThinkingResponse());
const events = await collect(stream(mock.model, context()));
const terminal = events.at(-1);
expect(terminal?.type).toBe("error");
// The guard must not forward the looping thinking_end / done.
expect(events.some(e => e.type === "thinking_end")).toBe(false);
expect(events.some(e => e.type === "done")).toBe(false);
} finally {
clearCustomApis();
}
});
test("passes a non-gemini model through untouched even when it loops", async () => {
registerMockApi();
try {
const mock = createMockModel({ provider: "openai", id: "gpt-5.5" });
mock.push(loopingThinkingResponse());
const result = await stream(mock.model, context()).result();
expect(result.stopReason).toBe("stop");
expect(result.content.some(b => b.type === "thinking")).toBe(true);
} finally {
clearCustomApis();
}
});
test("does not trip on a healthy gemini turn that reasons then answers", async () => {
registerMockApi();
try {
const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" });
mock.push({ content: [{ type: "thinking", thinking: distinctReasoning() }, "Here is the final answer."] });
const result = await stream(mock.model, context()).result();
expect(result.stopReason).toBe("stop");
expect(result.content.some(b => b.type === "text")).toBe(true);
} finally {
clearCustomApis();
}
});
test("does not retry once a loop re-emerges after visible answer text (armed latch)", async () => {
registerMockApi();
try {
const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" });
// Healthy reasoning, then visible answer text, then a runaway loop re-emitted
// as cumulative reasoning after `</think>` (the openai-completions shape).
mock.push({
content: [
{ type: "thinking", thinking: distinctReasoning() },
"Here is the final answer.",
{ type: "thinking", thinking: nearDuplicateLoop(12) },
],
});
const result = await stream(mock.model, context()).result();
// Visible text already streamed, so the loop must NOT hijack the turn.
expect(result.stopReason).toBe("stop");
expect(result.content.some(b => b.type === "text")).toBe(true);
} finally {
clearCustomApis();
}
});
test("guards the streamSimple custom-api path (agent default entrypoint)", async () => {
registerMockApi();
try {
const mock = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" });
mock.push(loopingThinkingResponse());
const result = await streamSimple(mock.model, context()).result();
expect(result.stopReason).toBe("error");
expect(result.content).toEqual([]);
expect(result.errorMessage).toContain(THINKING_LOOP_ERROR_MARKER);
expect(isRetryableError(new Error(result.errorMessage))).toBe(true);
} finally {
clearCustomApis();
}
});
});
describe("withGeminiThinkingLoopGuard (Vertex transport)", () => {
test("emits a retryable empty-content error for a looping Vertex Gemini stream", async () => {
const model = { api: "google-vertex", provider: "google-vertex", id: "gemini-2.5-pro" } as unknown as Model<Api>;
const partial = { role: "assistant", content: [] } as unknown as AssistantMessage;
const guarded = withGeminiThinkingLoopGuard(model, undefined, () => {
const inner = new AssistantMessageEventStream();
const events: AssistantMessageEvent[] = [
{ type: "start", partial },
{ type: "thinking_start", contentIndex: 0, partial },
{ type: "thinking_delta", contentIndex: 0, delta: nearDuplicateLoop(12), partial },
{ type: "thinking_end", contentIndex: 0, content: "", partial },
{ type: "done", reason: "stop", message: partial },
];
for (const event of events) inner.push(event);
return inner;
});
const result = await guarded.result();
expect(result.stopReason).toBe("error");
expect(result.content.length).toBe(0);
expect(result.errorMessage).toContain(THINKING_LOOP_ERROR_MARKER);
expect(isRetryableError(new Error(result.errorMessage))).toBe(true);
});
});
+5 -2
View File
@@ -1,13 +1,16 @@
# Changelog
## [Unreleased]
### Added
- Added `enableGeminiThinkingLoopGuard` to OpenAI compatibility options to allow explicit opt-in or opt-out of the Gemini thinking-loop guard for OpenAI-compatible model aliases
- Added `LITELLM_BASE_URL` as the LiteLLM provider discovery base URL fallback, with discovery caches scoped by the resolved proxy URL and explicit provider `baseUrl` config kept at higher precedence. ([#2726](https://github.com/can1357/oh-my-pi/issues/2726))
### Changed
- Defaulted `enableGeminiThinkingLoopGuard` from Gemini family detection for both OpenAI completions and responses compatibility specs so Gemini models now enable the thinking-loop guard automatically
- Updated the default Gemini CLI user-agent version fallback to 0.46.0.
### Fixed
- Routed google-antigravity default baseUrl to the stable primary daily endpoint in the catalog generator and all fallback snapshots, resolving connection drops on heavy queries.
@@ -247,4 +250,4 @@
### Removed
- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads.
- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads.
+7
View File
@@ -17,6 +17,7 @@ import {
isKimiModelId,
isMimoModelIdOrName,
isQwenModelId,
modelFamilyToken,
} from "../identity/family";
import type { ModelSpec, OpenAICompat, ResolvedOpenAICompat, ResolvedOpenAIResponsesCompat } from "../types";
import { applyCompatOverrides } from "./apply";
@@ -211,6 +212,10 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
supportsReasoningParams: provider !== "github-copilot",
reasoningEffortMap: {},
supportsUsageInStreaming: !isCerebras,
// pi-ai's thinking-loop guard is gemini-only; default the flag from the
// family classifier so OpenAI-compat proxies serving Gemini are covered.
// An opaque alias can opt in via `compat.enableGeminiThinkingLoopGuard`.
enableGeminiThinkingLoopGuard: modelFamilyToken(spec.id) === "gemini",
// Kimi (including via OpenRouter and Fireworks router-form IDs such as
// `accounts/fireworks/routers/kimi-*`) calculates TPM rate limits based on
// max_tokens, not actual output. The official Kimi K2 model guidance
@@ -295,6 +300,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
}
interface OpenAIResponsesSpecLike {
id?: string;
provider: string;
name: string;
baseUrl: string;
@@ -329,6 +335,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
strictResponsesPairing: isAzure || spec.provider === "github-copilot",
requiresJuiceZeroHack: spec.name.toLowerCase().startsWith("gpt-5"),
reasoningEffortMap: {},
enableGeminiThinkingLoopGuard: modelFamilyToken(spec.id ?? "") === "gemini",
};
applyCompatOverrides(compat, spec.compat);
return compat;
+12
View File
@@ -170,6 +170,13 @@ export interface OpenAICompat {
reasoningEffortMap?: Partial<Record<Effort, string>>;
/** Whether the provider supports `stream_options: { include_usage: true }` for token usage in streaming responses. Default: true. */
supportsUsageInStreaming?: boolean;
/**
* Enable the Gemini thinking-loop guard (pi-ai stream layer) for this model.
* Defaults to true when the model id classifies as the gemini family. Set
* explicitly to cover an opaque OpenAI-compat proxy alias (e.g. `my-model`)
* that routes to Gemini, or to false to opt a gemini-family id out.
*/
enableGeminiThinkingLoopGuard?: boolean;
/** Which field to use for max tokens. Default: auto-detected from URL. */
maxTokensField?: "max_completion_tokens" | "max_tokens";
/** Whether tool results require the `name` field. Default: auto-detected from URL. */
@@ -373,6 +380,7 @@ export type ResolvedOpenAICompat = Required<
| "thinkingKeep"
| "strictResponsesPairing"
| "requiresJuiceZeroHack"
| "enableGeminiThinkingLoopGuard"
| "whenThinking"
>
> & {
@@ -387,6 +395,8 @@ export type ResolvedOpenAICompat = Required<
isOpenRouterHost: boolean;
/** The model sits behind Vercel AI Gateway. */
isVercelGatewayHost: boolean;
/** See {@link OpenAICompat.enableGeminiThinkingLoopGuard}. Set by the builder from the family classifier. */
enableGeminiThinkingLoopGuard?: boolean;
/** Complete alternate view for thinking-engaged requests; swap pointers, never spread. */
whenThinking?: ResolvedOpenAICompat;
};
@@ -400,6 +410,8 @@ export interface ResolvedOpenAIResponsesCompat {
strictResponsesPairing: boolean;
requiresJuiceZeroHack: boolean;
reasoningEffortMap: Partial<Record<Effort, string>>;
/** See {@link OpenAICompat.enableGeminiThinkingLoopGuard}. */
enableGeminiThinkingLoopGuard?: boolean;
}
/** Fully-resolved anthropic-messages compat view (same contract as `ResolvedOpenAICompat`). */
@@ -0,0 +1,69 @@
import { describe, expect, it } from "bun:test";
import { buildOpenAICompat, buildOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
import type { ModelSpec, OpenAICompat } from "@oh-my-pi/pi-catalog/types";
/**
* The pi-ai thinking-loop guard is gemini-only and, for `openai-completions`
* models, gates on `compat.enableGeminiThinkingLoopGuard`. `buildOpenAICompat`
* must default that flag from the family classifier and honor explicit
* overrides so an opaque OpenAI-compat proxy alias can opt in/out.
*/
function spec(id: string, compat?: OpenAICompat): ModelSpec<"openai-completions"> {
return {
api: "openai-completions",
id,
name: id,
provider: "custom",
baseUrl: "https://proxy.example.com/v1",
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
maxTokens: 32_000,
contextWindow: 200_000,
reasoning: true,
...(compat ? { compat } : {}),
};
}
describe("buildOpenAICompat enableGeminiThinkingLoopGuard", () => {
it("defaults on for gemini-family ids, including aggregator namespaces", () => {
expect(buildOpenAICompat(spec("gemini-3.5-flash")).enableGeminiThinkingLoopGuard).toBe(true);
expect(buildOpenAICompat(spec("google/gemini-3-pro")).enableGeminiThinkingLoopGuard).toBe(true);
});
it("defaults off for non-gemini ids (incl. gemma lookalikes)", () => {
expect(buildOpenAICompat(spec("gpt-5.5")).enableGeminiThinkingLoopGuard).toBe(false);
expect(buildOpenAICompat(spec("gemma-3-1b")).enableGeminiThinkingLoopGuard).toBe(false);
});
it("lets an opaque proxy alias opt in via explicit compat override", () => {
const compat = buildOpenAICompat(spec("my-fast-model", { enableGeminiThinkingLoopGuard: true }));
expect(compat.enableGeminiThinkingLoopGuard).toBe(true);
});
it("lets a gemini-family id opt out via explicit compat override", () => {
const compat = buildOpenAICompat(spec("gemini-3.5-flash", { enableGeminiThinkingLoopGuard: false }));
expect(compat.enableGeminiThinkingLoopGuard).toBe(false);
});
});
describe("buildOpenAIResponsesCompat enableGeminiThinkingLoopGuard", () => {
const responsesSpec = (id: string, compat?: OpenAICompat) => ({
id,
name: id,
provider: "custom",
baseUrl: "https://proxy.example.com/v1",
...(compat ? { compat } : {}),
});
it("defaults from the family classifier", () => {
expect(buildOpenAIResponsesCompat(responsesSpec("gemini-3-pro")).enableGeminiThinkingLoopGuard).toBe(true);
expect(buildOpenAIResponsesCompat(responsesSpec("gpt-5.5")).enableGeminiThinkingLoopGuard).toBe(false);
});
it("honors an explicit override for an opaque proxy alias", () => {
expect(
buildOpenAIResponsesCompat(responsesSpec("my-fast-model", { enableGeminiThinkingLoopGuard: true }))
.enableGeminiThinkingLoopGuard,
).toBe(true);
});
});