diff --git a/docs/provider-compat-reference.md b/docs/provider-compat-reference.md index d3c020a73..67fc7a9de 100644 --- a/docs/provider-compat-reference.md +++ b/docs/provider-compat-reference.md @@ -102,7 +102,6 @@ Types: `OpenAICompat` / `ResolvedOpenAISharedCompat` in `packages/catalog/src/ty | `stripDeepseekSpecialTokens` | DeepSeek on NVIDIA NIM or direct API | Strips leaked chat-template tokens (`<|User|>`, …) from visible text | | `streamMarkupHealingPattern` | `"kimi"` (Kimi/Moonshot), `"dsml"` (DeepSeek DSML hosts), `"thinking"` (generic compat hosts), unset for official OpenAI | Selects the `StreamMarkupHealing` pattern for leaked template markup | | `emptyLengthFinishIsContextError` | Ollama | Empty completion with `finish_reason: "length"` → context-overflow error | -| `enableGeminiThinkingLoopGuard` | Gemini-family model ids | Activates the thinking-loop guard on OpenAI-compat streams (`utils/thinking-loop.ts`) | | `streamFirstEventTimeoutMs` | `0` for local backends | First-event watchdog hint (`0` = unbounded prefill/model-load time) | | `streamIdleTimeoutMs` | GLM/Alibaba coding plans 600 s; MiMo, Kimi reasoning, DeepSeek reasoning, local backends 300 s | Inter-event idle watchdog floor (`stream.ts`) | @@ -174,7 +173,7 @@ If a host rejects the emitted effort with 400/422, `resolveOpenAIReasoningEffort - **Structured deltas**: providers emit `thinking_start` / `thinking_delta` / `thinking_end` stream events. - **History replay**: prior thinking is replayed via `reasoningContentField` on assistant messages (KV-cache preservation on DeepSeek/Z.AI/Qwen/local backends); models that demand reasoning content on tool-call turns get real content or a `"."` placeholder per `allowsSyntheticReasoningContentForToolCalls`. - **Leaked thinking healing**: `wrapLeakedThinkingStream` (`utils/leaked-thinking-stream.ts`) converts in-band ` ```thinking ` / `` fences from misbehaving hosts into structured thinking blocks live. -- **Loop guard**: `withGeminiThinkingLoopGuard` (`utils/thinking-loop.ts`) detects runaway reasoning (verbatim repeats, near-duplicate trigram clusters, progress-lexicon stalls) and kills the stream with a retryable `AIError.Flag.ThinkingLoop`. +- **Loop guard**: `withThinkingLoopGuard` (`utils/thinking-loop.ts`) detects runaway reasoning (verbatim repeats, near-duplicate trigram clusters, progress-lexicon stalls) and kills the stream with a retryable `AIError.Flag.ThinkingLoop`. ### Interactions diff --git a/docs/provider-quirks.md b/docs/provider-quirks.md index d55905ecf..c3d4722eb 100644 --- a/docs/provider-quirks.md +++ b/docs/provider-quirks.md @@ -188,11 +188,11 @@ Google Gemini integrations use REST/SSE over HTTP (`POST https://generativelangu - **`streamGenerateContent` SSE protocol**: Streams are consumed via `readSseJson` in `streamGoogleGenAI`. - **Thought parts & signature retention**: `isThinkingPart` identifies reasoning text when `part.thought === true`. Encrypted `part.thoughtSignature` fields are preserved across deltas using `retainThoughtSignature`. In `convertMessages`, thought signatures are retained only when message provider/model match the target (`msg.provider === model.provider && msg.model === model.id`) and pass `isValidThoughtSignature` (base64 check). Gemini 3 tool calls lacking a signature fall back to `SKIP_THOUGHT_SIGNATURE` (`"skip_thought_signature_validator"`). - **Empty response retry loop**: `streamGoogleGenAI` guards against Gemini returning `finishReason: STOP` with blank content without calling tools. `hasMeaningfulGoogleContent` validates output; if empty, `streamGoogleGenAI` retries up to `MAX_EMPTY_STREAM_RETRIES` (2 retries, 3 total attempts) with exponential backoff (`EMPTY_STREAM_BASE_DELAY_MS * 2^attempt`) after resetting stream output via `resetGoogleStreamOutputForRetry`. -- **Thinking loop guard**: Implemented in `packages/ai/src/utils/thinking-loop.ts` (`ThinkingLoopDetector`, `isGeminiThinkingModel`). Streams before tool calls are monitored for three runaway shapes: +- **Thinking loop guard**: Implemented in `packages/ai/src/utils/thinking-loop.ts` (`ThinkingLoopDetector`). Gemini, DeepSeek, and Grok model-id families are monitored before tool calls for three runaway shapes: 1. *Verbatim tail repetition* (`VERBATIM_TAIL_WINDOW = 250`, >= 180 repeated chars). 2. *Near-duplicate segments* (trigram Jaccard similarity >= 0.8 across last 16 segments). 3. *Progress-lexicon stall* (novelty <= 0.2 without new concrete reference anchors over 8 consecutive segments). - Additionally, `GEMINI_HEADER_RUNAWAY_THRESHOLD = 24` halts streams emitting excessive titled reasoning summaries without acting. Triggers emit a synthetic retryable `error` tagged with `AIError.Flag.ThinkingLoop`. + 4. Gemini's `GEMINI_HEADER_RUNAWAY_THRESHOLD = 24` halts streams emitting excessive titled reasoning summaries without acting. Triggers emit a synthetic retryable `error` tagged with `AIError.Flag.ThinkingLoop`. - **Finish reason mapping & incomplete streams**: `candidate.finishReason` is mapped via `mapStopReason`; `stop`/`length` reasons upgrade to `toolUse` if output contains tool calls. Drops without `finishReason` throw `ProviderResponseError` with `kind: "incomplete-stream"`. - **UsageMetadata accounting**: Attached to trailing chunks in `consumeGoogleStream`. `input` is calculated as `promptTokenCount - (cachedContentTokenCount || 0)`; `output` as `candidatesTokenCount + (thoughtsTokenCount || 0)`; `cacheRead` as `cachedContentTokenCount || 0`; and `reasoningTokens` as `thoughtsTokenCount`. Token costs are computed via `calculateCost(model, output.usage)`. @@ -570,7 +570,7 @@ Pi Native is a lossless internal server/client transport protocol used when a pi - **Idle & First-Event Watchdogs**: Client wraps SSE streams with `iterateWithIdleTimeout` using `PI_STREAM_FIRST_EVENT_TIMEOUT_MS` and `PI_STREAM_IDLE_TIMEOUT_MS`. `isPiNativeProgressEvent` in `packages/ai/src/providers/pi-native-client.ts` ignores `type: "start"` events so initial setup does not reset the idle timeout. - **Synthetic Terminal Boundaries**: If the SSE stream closes without a `done` or `error` event, client's `streamPiNative` constructs a synthetic assistant message via `makeSyntheticAssistant`. It pushes `{ type: "error", reason: "aborted", error: { ..., stopReason: "aborted", errorMessage: "stream closed without terminal event" } }` if caller aborted, or `{ type: "done", reason: "stop", message: { ..., stopReason: "stop" } }` on ungraceful clean close. - **Server Iterator Exception Fallback**: If the server's `encodeStream` event iterator throws, it enqueues `data: {"type":"error","reason":"error","errorMessage":"..."}\n\n` followed by `data: [DONE]\n\n` so client iterators resolve instead of hanging. -- **Gemini Thinking Loop Guard**: `packages/ai/src/stream.ts` `streamSimple` wraps `streamPiNative` with `withGeminiThinkingLoopGuard` and `withProviderInFlightLimit`, ensuring degenerate Gemini thinking loops abort with empty-content retryable errors. +- **Thinking loop guard**: `packages/ai/src/stream.ts` `streamSimple` wraps `streamPiNative` with `withThinkingLoopGuard` and `withProviderInFlightLimit`, ensuring Gemini, DeepSeek, and Grok runaway thinking streams abort with empty-content retryable errors. ### Auth & usage - **Bearer Token Authorization**: Client (`packages/ai/src/providers/pi-native-client.ts` `buildHeaders`) passes `options.apiKey` (the gateway bearer token) in `Authorization: Bearer `, unless `model.headers.Authorization` is explicitly provided. diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index e4c8dd760..37a57ca5a 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Breaking Changes + +- Renamed `withGeminiThinkingLoopGuard` to `withThinkingLoopGuard`; the guard applies to Gemini, DeepSeek, and Grok model-id families. + ### Changed - Updated OpenCode Go integration to use the official usage endpoint, removing hardcoded caps, enabling real-time credential validation, and routing multi-key pools based on rolling and weekly headroom. diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index cd022b0db..0092ebac8 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -82,7 +82,7 @@ import { isFoundryEnabled } from "./utils/foundry"; import { wrapLeakedThinkingStream } from "./utils/leaked-thinking-stream"; import { wrapFetchForProxy } from "./utils/proxy"; import { withRequestDebugFetch } from "./utils/request-debug"; -import { withGeminiThinkingLoopGuard } from "./utils/thinking-loop"; +import { withThinkingLoopGuard } from "./utils/thinking-loop"; function defaultFetchForModel(model: Model): FetchImpl { if (model.provider === "anthropic" && model.api === "anthropic-messages") return coworkFetch; @@ -868,7 +868,7 @@ export function stream( context: Context, options?: OptionsForApi, ): AssistantMessageEventStream { - return withGeminiThinkingLoopGuard(model, options, opts => + return withThinkingLoopGuard(model, options, opts => withProviderInFlightLimit(model, opts, () => streamDispatch(model, context, opts)), ); } @@ -1533,7 +1533,7 @@ function streamSimpleRequest( // extension-registered APIs can't accidentally override a configured // pi-native transport. if (model.transport === "pi-native") { - return withGeminiThinkingLoopGuard(model, requestOptions, opts => + return withThinkingLoopGuard(model, requestOptions, opts => withProviderInFlightLimit(model, opts, () => streamPiNative(model, context, opts)), ); } @@ -1541,7 +1541,7 @@ function streamSimpleRequest( // Check custom API registry (extension-provided APIs) const customApiProvider = getCustomApi(model.api); if (customApiProvider) { - return withGeminiThinkingLoopGuard(model, requestOptions, opts => + return withThinkingLoopGuard(model, requestOptions, opts => withProviderInFlightLimit(model, opts, () => customApiProvider.streamSimple(model, context, opts)), ); } diff --git a/packages/ai/src/utils/thinking-loop.ts b/packages/ai/src/utils/thinking-loop.ts index 86d13aa63..25e8a493b 100644 --- a/packages/ai/src/utils/thinking-loop.ts +++ b/packages/ai/src/utils/thinking-loop.ts @@ -29,14 +29,14 @@ * anchor-free segments; a segment naming a path/identifier resets the run, so * genuine but vocabulary-repetitive work (per-file templates) is spared. * - * Scope is narrow: guarded Gemini, DeepSeek, and Grok 4.6 streams before any tool call. Native + * Scope is narrow: guarded Gemini, DeepSeek, and Grok family streams before any tool call. Native * thinking is checked first; assistant text can also be checked for providers * that surface reasoning as visible prose. On a hit the failed turn is emitted as * an empty retryable stream-stall error; result-awaiting callers (`complete`, * `completeSimple`) re-sample it a few times and then let a stubborn loop cook * through one unguarded pass. Disable detection with `PI_NO_THINKING_LOOP_GUARD=1`. */ -import { isGrok46ModelId } from "@oh-my-pi/pi-catalog/identity"; +import { modelFamilyToken } from "@oh-my-pi/pi-catalog/identity"; import { logger } from "@oh-my-pi/pi-utils"; import * as AIError from "../error"; import type { Api, AssistantMessage, Model, StreamOptions } from "../types"; @@ -95,45 +95,23 @@ const LEX_STALL_MIN_RUN = 8; const CONCRETE_ANCHOR = /`[^`]+`|\b\w{2,}\.[a-zA-Z]\w{0,4}\b|[\w-]+(?:\/[\w-]+){2,}|\b\w+_\w+\b|\b[a-z]+[A-Z]\w*\b|\b[A-Z][a-z]+[A-Z]\w*\b/g; -const OPENAI_COMPAT_GUARDED_APIS: Partial> = { - "openai-completions": true, - "openai-responses": true, - "azure-openai-responses": true, - "openai-codex-responses": true, -}; - /** - * True when `model` is a Gemini model whose native thinking stream surfaces the - * "thought summary" titles this module's header guard counts. + * True when `model.id` belongs to a family guarded for thinking/response loops: + * Gemini, DeepSeek, or Grok. * - * OpenAI-compat transports can serve Gemini under an arbitrary provider/id, so they - * carry the explicit `compat.enableGeminiThinkingLoopGuard` flag; direct Gemini - * transports carry a clearly shaped id/provider, so a string match is sufficient. - */ -export function isGeminiThinkingModel(model: Model): boolean { - if (OPENAI_COMPAT_GUARDED_APIS[model.api]) { - const compat = model.compat as { enableGeminiThinkingLoopGuard?: boolean } | undefined; - return compat?.enableGeminiThinkingLoopGuard === true; - } - return /gemini/i.test(`${model.provider}/${model.id}`); -} - -/** - * True when `model` should be guarded for thinking/response loops (Gemini, DeepSeek, and Grok 4.6). - * - * OpenAI-compat transports can serve Gemini or DeepSeek under an arbitrary provider/id. Grok 4.6 - * is recognized by {@link isGrok46ModelId} across transports; direct Gemini/DeepSeek transports - * carry a clearly shaped id/provider, so a string match is sufficient. + * Model identity is derived only from its id; provider and compatibility metadata + * do not opt opaque aliases into the guard. */ export function isLoopGuardedModel(model: Model, options?: StreamOptions): boolean { if (options?.loopGuard?.enabled === false) return false; - const isDeepseek = /deepseek/i.test(`${model.provider}/${model.id}`); - return isGeminiThinkingModel(model) || isDeepseek || isGrok46ModelId(model.id); -} - -/** @deprecated Use isLoopGuardedModel instead. */ -export function isGeminiThinkingLoopModel(model: Model): boolean { - return isLoopGuardedModel(model); + switch (modelFamilyToken(model.id)) { + case "gemini": + case "deepseek": + case "grok": + return true; + default: + return false; + } } /** @@ -361,7 +339,7 @@ export class GeminiHeaderRunDetector { /** * Wrap a provider stream with the loop guard. `controller` is the guard's own * abort handle: aborting it (after wiring it into the provider's signal via - * {@link withGeminiThinkingLoopGuard}) tears down the upstream once a loop + * {@link withThinkingLoopGuard}) tears down the upstream once a loop * trips. */ export function guardThinkingLoopStream( @@ -444,7 +422,7 @@ export function guardThinkingLoopStream( * stall; bounding the re-samples and the final cook pass lives in the * result-awaiting caller. */ -export function withGeminiThinkingLoopGuard< +export function withThinkingLoopGuard< O extends { signal?: AbortSignal; loopGuard?: { enabled?: boolean; checkAssistantContent?: boolean } }, >( model: Model, diff --git a/packages/ai/test/thinking-loop.test.ts b/packages/ai/test/thinking-loop.test.ts index a80630cd6..4736a7866 100644 --- a/packages/ai/test/thinking-loop.test.ts +++ b/packages/ai/test/thinking-loop.test.ts @@ -9,13 +9,11 @@ import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream" import { GEMINI_HEADER_RUNAWAY_THRESHOLD, GeminiHeaderRunDetector, - isGeminiThinkingLoopModel, - isGeminiThinkingModel, isLoopGuardedModel, isReasoningSummaryHeader, THINKING_LOOP_ERROR_MARKER, ThinkingLoopDetector, - withGeminiThinkingLoopGuard, + withThinkingLoopGuard, } from "@oh-my-pi/pi-ai/utils/thinking-loop"; import { isRetryableError } from "@oh-my-pi/pi-utils"; @@ -226,43 +224,6 @@ function perFileTemplates(): string { .join("\n\n"); } -describe("isGeminiThinkingLoopModel", () => { - test("matches direct and aggregator-routed gemini ids, not lookalikes", () => { - const gate = (provider: string, id: string) => isGeminiThinkingLoopModel(createMockModel({ provider, id }).model); - expect(gate("google", "gemini-3-pro-preview")).toBe(true); - expect(gate("openrouter", "google/gemini-3.5-flash")).toBe(true); - expect(gate("google-gemini-cli", "gemini-3-flash")).toBe(true); - expect(gate("openai", "gpt-5.5")).toBe(false); - expect(gate("google", "gemma-3-1b")).toBe(false); - }); - - test("trusts the compat flag over the id regex for every OpenAI-compat API", () => { - const gate = (api: string, id: string, enableGeminiThinkingLoopGuard: boolean) => - isGeminiThinkingLoopModel({ - api, - provider: "openrouter", - id, - compat: { enableGeminiThinkingLoopGuard }, - } as unknown as Model); - // Opaque proxy alias opted in despite a non-gemini id (completions + responses). - expect(gate("openai-completions", "my-fast-model", true)).toBe(true); - expect(gate("openai-responses", "my-fast-model", true)).toBe(true); - // Gemini-shaped id explicitly opted out stays off — the flag wins over the regex. - expect(gate("openai-completions", "gemini-3.5-flash", false)).toBe(false); - expect(gate("openai-responses", "gemini-3.5-flash", false)).toBe(false); - }); - - test("guards non-compat Gemini transports (Vertex, direct Google) via id", () => { - const gate = (api: string, provider: string, id: string) => - isGeminiThinkingLoopModel({ api, provider, id } as unknown as Model); - // Vertex has no OpenAICompat record; its canonical ids are gemini-shaped. - expect(gate("google-vertex", "google-vertex", "gemini-2.5-pro")).toBe(true); - expect(gate("google-generative-ai", "google", "gemini-3-pro")).toBe(true); - // Non-Gemini models on the same transports (e.g. Claude on Vertex) stay unguarded. - expect(gate("google-vertex", "google-vertex", "claude-sonnet-4")).toBe(false); - }); -}); - describe("ThinkingLoopDetector", () => { test("trips on a tight near-duplicate paragraph loop via the trigram path", () => { // High word-trigram overlap: the cluster check claims it before the lexical @@ -464,12 +425,12 @@ describe("thinking-loop guard (stream wrapper)", () => { }); }); -describe("withGeminiThinkingLoopGuard (Vertex transport)", () => { +describe("withThinkingLoopGuard (Vertex transport)", () => { test("emits a retryable empty-content error for a looping Vertex Gemini stream", async () => { const model = { api: "google-vertex", provider: "google-vertex", id: "gemini-2.5-pro" } as unknown as Model; const partial = { role: "assistant", content: [] } as unknown as AssistantMessage; - const guarded = withGeminiThinkingLoopGuard(model, undefined, () => { + const guarded = withThinkingLoopGuard(model, undefined, () => { const inner = new AssistantMessageEventStream(); const events: AssistantMessageEvent[] = [ { type: "start", partial }, @@ -491,29 +452,31 @@ describe("withGeminiThinkingLoopGuard (Vertex transport)", () => { }); }); describe("isLoopGuardedModel", () => { - test("guards Gemini, DeepSeek, and Grok 4.6 models by default, respects overrides", () => { + test("guards Gemini, DeepSeek, and Grok model-id families only", () => { const gemini = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }).model; const deepseek = createMockModel({ provider: "deepseek", id: "deepseek-reasoner" }).model; const grok46 = createMockModel({ provider: "venice", id: "grok-4-6" }).model; const cursorGrok46 = createMockModel({ provider: "cursor", id: "cursor-grok-4.6-high" }).model; const grok460 = createMockModel({ provider: "venice", id: "grok-4.60" }).model; const grok45 = createMockModel({ provider: "cursor", id: "cursor-grok-4.5-high" }).model; + const opaqueDeepseek = createMockModel({ provider: "deepseek", id: "opaque-model" }).model; const other = createMockModel({ provider: "openai", id: "gpt-4o" }).model; expect(isLoopGuardedModel(gemini)).toBe(true); expect(isLoopGuardedModel(deepseek)).toBe(true); expect(isLoopGuardedModel(grok46)).toBe(true); expect(isLoopGuardedModel(cursorGrok46)).toBe(true); - expect(isLoopGuardedModel(grok460)).toBe(false); - expect(isLoopGuardedModel(grok45)).toBe(false); + expect(isLoopGuardedModel(grok460)).toBe(true); + expect(isLoopGuardedModel(grok45)).toBe(true); + expect(isLoopGuardedModel(opaqueDeepseek)).toBe(false); expect(isLoopGuardedModel(other)).toBe(false); - // enabled: false disables even for target models + // enabled: false disables every guarded family. expect(isLoopGuardedModel(gemini, { loopGuard: { enabled: false } })).toBe(false); expect(isLoopGuardedModel(deepseek, { loopGuard: { enabled: false } })).toBe(false); - expect(isLoopGuardedModel(grok46, { loopGuard: { enabled: false } })).toBe(false); + expect(isLoopGuardedModel(grok45, { loopGuard: { enabled: false } })).toBe(false); - // force enabled for other models — but disabled overall unless it is Gemini, DeepSeek, or Grok 4.6 + // enabled: true does not opt unrelated models into the guard. expect(isLoopGuardedModel(other, { loopGuard: { enabled: true } })).toBe(false); }); }); @@ -528,7 +491,7 @@ describe("loop guard assistant prose/text loops", () => { const partial = { role: "assistant", content: [], stopReason: "stop" } as unknown as AssistantMessage; const options = { loopGuard: { checkAssistantContent: true } }; - const guarded = withGeminiThinkingLoopGuard(model, options, () => { + const guarded = withThinkingLoopGuard(model, options, () => { const inner = new AssistantMessageEventStream(); const events: AssistantMessageEvent[] = [ { type: "start", partial }, @@ -562,7 +525,7 @@ describe("loop guard assistant prose/text loops", () => { const partial = { role: "assistant", content: [], stopReason: "stop" } as unknown as AssistantMessage; const options = { loopGuard: { checkAssistantContent: false } }; - const guarded = withGeminiThinkingLoopGuard(model, options, () => { + const guarded = withThinkingLoopGuard(model, options, () => { const inner = new AssistantMessageEventStream(); const events: AssistantMessageEvent[] = [ { type: "start", partial }, @@ -667,20 +630,6 @@ describe("GeminiHeaderRunDetector", () => { }); }); -describe("isGeminiThinkingModel", () => { - test("is true for Gemini and false for DeepSeek / other guarded peers", () => { - const gemini = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }).model; - const deepseek = createMockModel({ provider: "openrouter", id: "deepseek/deepseek-r1" }).model; - const claude = createMockModel({ provider: "anthropic", id: "claude-sonnet-4" }).model; - expect(isGeminiThinkingModel(gemini)).toBe(true); - expect(isGeminiThinkingModel(deepseek)).toBe(false); - expect(isGeminiThinkingModel(claude)).toBe(false); - // DeepSeek is still loop-guarded for the similarity guard, just not the header guard. - expect(isLoopGuardedModel(deepseek)).toBe(true); - expect(isLoopGuardedModel(gemini)).toBe(true); - }); -}); - describe("thinking-loop cook fallback (result path)", () => { function loopResponse(): { content: MockContent[] } { return { content: [{ type: "thinking", thinking: nearDuplicateLoop(12) }] }; diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 5d9901854..e40f2fe88 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Breaking Changes + +- Removed `OpenAICompat.enableGeminiThinkingLoopGuard`; thinking-loop eligibility is derived solely from the `model.id` family. + ### Added - Added first-party OpenAI Daybreak Blue, Daybreak Red, and GPT-5.6 Cyber models with full support for their documented API pricing (including long-context rates above 272K input), token limits, tools, and reasoning effort controls (off/low/medium/high/xhigh/max). diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index c7f97b4e1..ad7ba06b0 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -22,7 +22,6 @@ import { isMimoModelIdOrName, isOpenAISamplingRestrictedModelId, isQwenModelId, - modelFamilyToken, } from "../identity/family"; import type { ModelSpec, @@ -474,10 +473,6 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv supportsSamplingParams: !isOpenAISamplingRestrictedModelId(spec.id), reasoningEffortMap: {}, supportsUsageInStreaming: !isCerebras, - // pi-ai's thinking-loop guard is gemini-only; default the flag from the - // family classifier so OpenAI-compat proxies serving Gemini are covered. - // An opaque alias can opt in via `compat.enableGeminiThinkingLoopGuard`. - enableGeminiThinkingLoopGuard: modelFamilyToken(spec.id) === "gemini", // Kimi (including via OpenRouter and Fireworks router-form IDs such as // `accounts/fireworks/routers/kimi-*`) calculates TPM rate limits based on // max_tokens, not actual output. The official Kimi K2 model guidance @@ -749,7 +744,6 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol // lands on Moonshot's MFJS validator. toolSchemaFlavor: isKimiModel ? "moonshot-mfjs" : undefined, alwaysSendMaxTokens: spec.id ? isKimiModelId(spec.id) : false, - enableGeminiThinkingLoopGuard: modelFamilyToken(spec.id ?? "") === "gemini", supportsObfuscationOptOut: isOpenAIUrl || spec.provider === "openai", stripDeepseekSpecialTokens: Boolean(id) && isDeepseekModelIdOrName(id) && (spec.provider === "nvidia" || spec.provider === "deepseek"), diff --git a/packages/catalog/src/identity/family.ts b/packages/catalog/src/identity/family.ts index 0ee25317e..0080c0aea 100644 --- a/packages/catalog/src/identity/family.ts +++ b/packages/catalog/src/identity/family.ts @@ -98,12 +98,14 @@ export const isMimoModelIdOrName = memo((value: string): boolean => { return value.toLowerCase().includes("mimo"); }); -/** - * Grok 4.6 model IDs, including canonical dashed and Cursor dotted variants. - * Adjacent versions such as `grok-4.60` are deliberately excluded. - */ -export const isGrok46ModelId = memo((modelId: string): boolean => { - return /(?:^|[./_-])grok-4[.-]6(?:$|[-_:])/i.test(bareModelId(modelId)); +/** Gemini family ids in any namespace form (`gemini-*`, `google/gemini-*`, `openrouter/google/gemini-…`). */ +export const isGeminiModelId = memo((modelId: string): boolean => { + return /(^|\/)gemini[-.]?/i.test(modelId); +}); + +/** Grok family ids across namespace and delimiter forms (`grok-*`, `cursor-grok-*`, `xai/grok-*`). */ +export const isGrokModelId = memo((modelId: string): boolean => { + return /(?:^|[./_-])grok(?:[-.]|$)/i.test(modelId); }); const GROK_EFFORT_CAPABLE_PREFIXES = ["grok-3-mini", "grok-4.20-multi-agent", "grok-4.3", "grok-4.5"] as const; @@ -268,6 +270,8 @@ export const modelFamilyToken = memo((modelId: string): string => { if (parsed.family !== "unknown") return parsed.family; if (isClaudeModelId(modelId) || isAnthropicNamespacedModelId(modelId)) return "anthropic"; if (isOpenAIModelId(modelId)) return "openai"; + if (isGeminiModelId(modelId)) return "gemini"; + if (isGrokModelId(modelId)) return "grok"; if (isKimiModelId(modelId)) return "kimi"; if (isQwenModelId(modelId)) return "qwen"; if (isMinimaxM2FamilyModelId(modelId) || isMinimaxM3FamilyModelId(modelId)) return "minimax"; diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts index 8dfa0dcdd..729ab4d89 100644 --- a/packages/catalog/src/types.ts +++ b/packages/catalog/src/types.ts @@ -191,13 +191,6 @@ export interface OpenAICompat { reasoningEffortMap?: Partial>; /** Whether the provider supports `stream_options: { include_usage: true }` for token usage in streaming responses. Default: true. */ supportsUsageInStreaming?: boolean; - /** - * Enable the Gemini thinking-loop guard (pi-ai stream layer) for this model. - * Defaults to true when the model id classifies as the gemini family. Set - * explicitly to cover an opaque OpenAI-compat proxy alias (e.g. `my-model`) - * that routes to Gemini, or to false to opt a gemini-family id out. - */ - enableGeminiThinkingLoopGuard?: boolean; /** Which field to use for max tokens. Default: auto-detected from URL. */ maxTokensField?: "max_completion_tokens" | "max_tokens"; /** Whether tool results require the `name` field. Default: auto-detected from URL. */ @@ -628,8 +621,6 @@ export interface ResolvedOpenAISharedCompat { isOpenRouterHost: boolean; /** Whether this endpoint needs a max-token field even when caller did not set one. */ alwaysSendMaxTokens: boolean; - /** See {@link OpenAICompat.enableGeminiThinkingLoopGuard}. Set by the builder from the family classifier. */ - enableGeminiThinkingLoopGuard?: boolean; openRouterRouting?: OpenAICompat["openRouterRouting"]; /** Provider-specific wire model-id transform applied to the base id. */ wireModelIdMode: "raw" | "firepass" | "fireworks" | "openrouter"; @@ -698,7 +689,6 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat & | "thinkingKeep" | "strictResponsesPairing" | "supportsImageDetailOriginal" - | "enableGeminiThinkingLoopGuard" | "whenThinking" > > & { diff --git a/packages/catalog/test/gemini-thinking-loop-compat.test.ts b/packages/catalog/test/gemini-thinking-loop-compat.test.ts deleted file mode 100644 index 8eb2b40f1..000000000 --- a/packages/catalog/test/gemini-thinking-loop-compat.test.ts +++ /dev/null @@ -1,69 +0,0 @@ -import { describe, expect, it } from "bun:test"; -import { buildOpenAICompat, buildOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai"; -import type { ModelSpec, OpenAICompat } from "@oh-my-pi/pi-catalog/types"; - -/** - * The pi-ai thinking-loop guard is gemini-only and, for `openai-completions` - * models, gates on `compat.enableGeminiThinkingLoopGuard`. `buildOpenAICompat` - * must default that flag from the family classifier and honor explicit - * overrides so an opaque OpenAI-compat proxy alias can opt in/out. - */ -function spec(id: string, compat?: OpenAICompat): ModelSpec<"openai-completions"> { - return { - api: "openai-completions", - id, - name: id, - provider: "custom", - baseUrl: "https://proxy.example.com/v1", - input: ["text"], - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, - maxTokens: 32_000, - contextWindow: 200_000, - reasoning: true, - ...(compat ? { compat } : {}), - }; -} - -describe("buildOpenAICompat enableGeminiThinkingLoopGuard", () => { - it("defaults on for gemini-family ids, including aggregator namespaces", () => { - expect(buildOpenAICompat(spec("gemini-3.5-flash")).enableGeminiThinkingLoopGuard).toBe(true); - expect(buildOpenAICompat(spec("google/gemini-3-pro")).enableGeminiThinkingLoopGuard).toBe(true); - }); - - it("defaults off for non-gemini ids (incl. gemma lookalikes)", () => { - expect(buildOpenAICompat(spec("gpt-5.5")).enableGeminiThinkingLoopGuard).toBe(false); - expect(buildOpenAICompat(spec("gemma-3-1b")).enableGeminiThinkingLoopGuard).toBe(false); - }); - - it("lets an opaque proxy alias opt in via explicit compat override", () => { - const compat = buildOpenAICompat(spec("my-fast-model", { enableGeminiThinkingLoopGuard: true })); - expect(compat.enableGeminiThinkingLoopGuard).toBe(true); - }); - - it("lets a gemini-family id opt out via explicit compat override", () => { - const compat = buildOpenAICompat(spec("gemini-3.5-flash", { enableGeminiThinkingLoopGuard: false })); - expect(compat.enableGeminiThinkingLoopGuard).toBe(false); - }); -}); - -describe("buildOpenAIResponsesCompat enableGeminiThinkingLoopGuard", () => { - const responsesSpec = (id: string, compat?: OpenAICompat) => ({ - id, - name: id, - provider: "custom", - baseUrl: "https://proxy.example.com/v1", - ...(compat ? { compat } : {}), - }); - - it("defaults from the family classifier", () => { - expect(buildOpenAIResponsesCompat(responsesSpec("gemini-3-pro")).enableGeminiThinkingLoopGuard).toBe(true); - expect(buildOpenAIResponsesCompat(responsesSpec("gpt-5.5")).enableGeminiThinkingLoopGuard).toBe(false); - }); - - it("honors an explicit override for an opaque proxy alias", () => { - expect( - buildOpenAIResponsesCompat(responsesSpec("my-fast-model", { enableGeminiThinkingLoopGuard: true })) - .enableGeminiThinkingLoopGuard, - ).toBe(true); - }); -}); diff --git a/packages/catalog/test/identity-family.test.ts b/packages/catalog/test/identity-family.test.ts index 71450a586..a6d256e8e 100644 --- a/packages/catalog/test/identity-family.test.ts +++ b/packages/catalog/test/identity-family.test.ts @@ -2,8 +2,9 @@ import { describe, expect, test } from "bun:test"; import { hasOpus47ApiRestrictions, isClaudeModelId, + isGeminiModelId, isGlmVisionModelId, - isGrok46ModelId, + isGrokModelId, isGrokReasoningEffortCapable, isKimiK26ModelId, isKimiModelId, @@ -295,8 +296,9 @@ describe("modelFamilyToken", () => { test("classifies non-first-party families", () => { expect(modelFamilyToken("moonshotai/kimi-k2")).toBe("kimi"); expect(modelFamilyToken("qwen/qwen3-coder")).toBe("qwen"); + expect(modelFamilyToken("google/gemini-2.5-flash")).toBe("gemini"); + expect(modelFamilyToken("xai/grok-4.6")).toBe("grok"); }); - test("classifies GLM across provider mirrors so same-lineage SKUs fold together", () => { expect(modelFamilyToken("glm-5.2")).toBe("glm"); expect(modelFamilyToken("zai/glm-5.2")).toBe(modelFamilyToken("zhipu-coding-plan/glm-5.2")); @@ -307,19 +309,23 @@ describe("modelFamilyToken", () => { expect(modelFamilyToken("some-unknown-model")).toBe(""); }); }); - -describe("isGrok46ModelId", () => { - test("matches canonical dashed and Cursor dotted identifiers", () => { - expect(isGrok46ModelId("grok-4-6")).toBe(true); - expect(isGrok46ModelId("venice/grok-4-6")).toBe(true); - expect(isGrok46ModelId("cursor-grok-4.6-high")).toBe(true); +describe("isGeminiModelId", () => { + test("matches gemini ids across namespaces", () => { + expect(isGeminiModelId("gemini-3.5-flash")).toBe(true); + expect(isGeminiModelId("google/gemini-3-pro")).toBe(true); + expect(isGeminiModelId("openrouter/google/gemini-2.5-flash")).toBe(true); + expect(isGeminiModelId("gpt-4o")).toBe(false); }); +}); - test("rejects adjacent versions and lookalikes", () => { - expect(isGrok46ModelId("grok-4.60")).toBe(false); - expect(isGrok46ModelId("grok-4.6.0")).toBe(false); - expect(isGrok46ModelId("grok-4-5")).toBe(false); - expect(isGrok46ModelId("notgrok-4.6")).toBe(false); +describe("isGrokModelId", () => { + test("matches grok ids across namespaces and delimiters", () => { + expect(isGrokModelId("grok-4-6")).toBe(true); + expect(isGrokModelId("xai/grok-3")).toBe(true); + expect(isGrokModelId("venice/grok-4.5")).toBe(true); + expect(isGrokModelId("cursor-grok-4.5-high")).toBe(true); + expect(isGrokModelId("notgrok-4.6")).toBe(false); + expect(isGrokModelId("gpt-4o")).toBe(false); }); }); diff --git a/packages/coding-agent/src/session/stream-guards.ts b/packages/coding-agent/src/session/stream-guards.ts index 83e75af64..978be9680 100644 --- a/packages/coding-agent/src/session/stream-guards.ts +++ b/packages/coding-agent/src/session/stream-guards.ts @@ -1,8 +1,9 @@ import * as fs from "node:fs"; import type { Agent, AgentEvent, AgentMessage, AgentTurnEndContext } from "@oh-my-pi/pi-agent-core"; import type { AssistantMessage, AssistantMessageEvent, Model, ToolCall } from "@oh-my-pi/pi-ai"; -import { GeminiHeaderRunDetector, isGeminiThinkingModel } from "@oh-my-pi/pi-ai/utils/thinking-loop"; +import { GeminiHeaderRunDetector } from "@oh-my-pi/pi-ai/utils/thinking-loop"; import { type RepeatedToolCallDetection, ToolCallLoopGuard } from "@oh-my-pi/pi-ai/utils/tool-call-loop-guard"; +import { modelFamilyToken } from "@oh-my-pi/pi-catalog/identity"; import { isEnoent, logger, prompt } from "@oh-my-pi/pi-utils"; import type { Settings } from "../config/settings"; import { normalizeDiff, normalizeToLF, ParseError, previewPatch, stripBom } from "../edit"; @@ -362,7 +363,7 @@ export class LoopGuards { this.#host.settings.get("model.loopGuard.enabled") === true && this.#host.settings.get("model.loopGuard.toolCallReminder") === true && model !== undefined && - isGeminiThinkingModel(model) + modelFamilyToken(model.id) === "gemini" ); } diff --git a/packages/coding-agent/test/agent-session-gemini-header-interrupt.test.ts b/packages/coding-agent/test/agent-session-gemini-header-interrupt.test.ts index 792434d9c..d2eed8c2e 100644 --- a/packages/coding-agent/test/agent-session-gemini-header-interrupt.test.ts +++ b/packages/coding-agent/test/agent-session-gemini-header-interrupt.test.ts @@ -158,8 +158,12 @@ describe("AgentSession Gemini header-runaway interrupt", () => { vi.restoreAllMocks(); }); - function buildSession(streamFn: Agent["streamFn"], overrides?: Record): void { - const model = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }).model; + function buildSession( + streamFn: Agent["streamFn"], + overrides?: Record, + modelId = "google/gemini-3.5-flash", + ): void { + const model = createMockModel({ provider: "openrouter", id: modelId }).model; const modelRegistry = new ModelRegistry(authStorage); const agent = new Agent({ getApiKey: requestedModel => `${requestedModel.provider}-test-key`, @@ -249,4 +253,25 @@ describe("AgentSession Gemini header-runaway interrupt", () => { expect(assistants).toHaveLength(1); expect(assistants[0].content.at(-1)).toEqual({ type: "text", text: "Visible final answer." }); }); + + it("does not interrupt a DeepSeek header run", async () => { + let call = 0; + buildSession( + (model, _context, options) => { + call++; + return headerRunawayStream(model, options, "Visible DeepSeek answer."); + }, + undefined, + "deepseek-reasoner", + ); + + await session?.prompt("Do the task"); + await session?.waitForIdle(); + + expect(call).toBe(1); + const messages = session?.agent.state.messages ?? []; + expect( + messages.some(message => message.role === "custom" && message.customType === "gemini-tool-call-reminder"), + ).toBe(false); + }); }); diff --git a/packages/coding-agent/test/agent-session-thinking-loop-retry.test.ts b/packages/coding-agent/test/agent-session-thinking-loop-retry.test.ts index fe7d58224..23ae7f48d 100644 --- a/packages/coding-agent/test/agent-session-thinking-loop-retry.test.ts +++ b/packages/coding-agent/test/agent-session-thinking-loop-retry.test.ts @@ -14,7 +14,7 @@ import type { import * as AIError from "@oh-my-pi/pi-ai/error"; import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; -import { withGeminiThinkingLoopGuard } from "@oh-my-pi/pi-ai/utils/thinking-loop"; +import { withThinkingLoopGuard } from "@oh-my-pi/pi-ai/utils/thinking-loop"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { AgentSession, type AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session"; @@ -65,7 +65,7 @@ function chunkedThinkingLoopStream(model: Model, options?: SimpleStreamOpti inner.push({ type: "thinking_end", contentIndex: 0, content: thinking.thinking, partial }); inner.push({ type: "done", reason: "stop", message: partial }); }); - return withGeminiThinkingLoopGuard(model, options, () => inner); + return withThinkingLoopGuard(model, options, () => inner); } function successStream(model: Model): AssistantMessageEventStream {