diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index ac44e9698..5bec2012c 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -10,6 +10,10 @@ - Fixed Codex requests failing outright when the signed-in ChatGPT account is not entitled to the requested model; the exact model denial is now classified as an account-policy error so credential rotation can reach an entitled sibling account - Fixed Perplexity email-OTP login after its verification response renamed the encrypted session token from `token` to `challenge_token`. +### Fixed + +- Cloud Code Assist Gemini 3.6/3.7 Flash requests at `minimal` now send `thinkingLevel: LOW` on the aliased `-low` SKU instead of `MINIMAL`, which the API rejects with HTTP 400. + ## [17.3.7] - 2026-08-17 ### Changed diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 206de1fbc..8a5029767 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -2174,7 +2174,7 @@ function mapOptionsForApi( serviceTier: options?.serviceTier, thinking: { enabled: true, - level: mapEffortToGoogleThinkingLevel(effort), + level: mapEffortToGoogleThinkingLevel(effort, googleModel), }, hideThinkingSummary: options?.hideThinkingSummary, toolChoice: mapGoogleToolChoice(options?.toolChoice), @@ -2207,7 +2207,7 @@ function mapOptionsForApi( requestModelId: resolveWireModelId(model, effort), thinking: { enabled: true, - level: mapEffortToGoogleThinkingLevel(effort), + level: mapEffortToGoogleThinkingLevel(effort, model), }, hideThinkingSummary: options?.hideThinkingSummary, toolChoice, @@ -2279,7 +2279,7 @@ function mapOptionsForApi( serviceTier: options?.serviceTier, thinking: { enabled: true, - level: mapEffortToGoogleThinkingLevel(effort), + level: mapEffortToGoogleThinkingLevel(effort, model), }, hideThinkingSummary: options?.hideThinkingSummary, toolChoice: mapGoogleToolChoice(options?.toolChoice), diff --git a/packages/ai/test/google-gemini-cli-3x-thinking.test.ts b/packages/ai/test/google-gemini-cli-3x-thinking.test.ts index 32de52174..bb993720f 100644 --- a/packages/ai/test/google-gemini-cli-3x-thinking.test.ts +++ b/packages/ai/test/google-gemini-cli-3x-thinking.test.ts @@ -11,6 +11,7 @@ interface GeminiCliThinkingConfig { } interface CapturedRequestBody { + model?: string; request?: { generationConfig?: { thinkingConfig?: GeminiCliThinkingConfig; @@ -138,4 +139,47 @@ describe("google-gemini-cli Gemini 3.x thinking mapping", () => { expect(thinking?.thinkingLevel).toBeUndefined(); expect(thinking?.thinkingBudget).toBeDefined(); }); + + it("sends LOW not MINIMAL when gemini-3.7-flash minimal aliases the -low SKU", async () => { + let requestBody: string | undefined; + const fetchMock = createFetchMock(body => { + requestBody = body; + }); + + const model = buildModel({ + id: "gemini-3.7-flash", + name: "Gemini 3.7 Flash", + api: "google-gemini-cli", + provider: "google-antigravity", + baseUrl: "https://daily-cloudcode-pa.googleapis.com", + reasoning: true, + thinking: { + mode: "google-level", + efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High], + requiresEffort: true, + effortRouting: { + [Effort.Minimal]: "gemini-3.7-flash-low", + [Effort.Low]: "gemini-3.7-flash-low", + [Effort.Medium]: "gemini-3.7-flash-medium", + [Effort.High]: "gemini-3.7-flash-high", + }, + }, + input: ["text", "image"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 1_048_576, + maxTokens: 65_536, + }); + + const stream = streamSimple(model, context, { + apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }), + reasoning: Effort.Minimal, + fetch: fetchMock, + }); + await stream.result(); + + const parsed = JSON.parse(requestBody ?? "{}") as CapturedRequestBody; + expect(parsed.model).toBe("gemini-3.7-flash-low"); + expect(parsed.request?.generationConfig?.thinkingConfig?.thinkingLevel).toBe("LOW"); + }); + }); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 9476dd519..9d7d82908 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -7,6 +7,10 @@ - Fixed local Qwen 3.8+ models (llama.cpp, vLLM, loopback custom providers) exposing the generic `minimal..high` thinking ladder instead of the chat template's real `low`/`medium`/`xhigh` `reasoning_effort` tiers. The derived metadata now marks thinking as mandatory (the official 3.8 template raises on `enable_thinking: false`), vLLM-served Qwen routes through the `chat_template_kwargs` dialect (top-level `enable_thinking` is ignored by vLLM), and vLLM discovery lights up the reasoning dial for Qwen 3.8+ ids its `/v1/models` endpoint reports as non-reasoning. - Fixed `deepseek-v4-pro-0813` surfacing from Alibaba Token Plan discovery with `contextWindow`/`maxTokens` of `null`. The dated DeepSeek V4 Pro snapshot was missing from `ALIBABA_TOKEN_PLAN_DISCOVERED_MODEL_LIMITS`, so unlike its `deepseek-v4-flash-0731` sibling it fell through to unknown limits ([#8847](https://github.com/can1357/oh-my-pi/issues/8847)). +### Fixed + +- Cloud Code Assist Gemini 3.6/3.7 Flash no longer maps user `minimal` to wire `thinkingLevel: MINIMAL` when that effort is aliased onto the `-low` SKU. The request now sends `LOW`, which those SKUs accept. + ## [17.3.6] - 2026-08-17 ### Changed diff --git a/packages/catalog/src/model-thinking.ts b/packages/catalog/src/model-thinking.ts index 3f89ae195..05e263346 100644 --- a/packages/catalog/src/model-thinking.ts +++ b/packages/catalog/src/model-thinking.ts @@ -816,11 +816,22 @@ export function requireSupportedEffort(model: ApiModel, return effort; } -/** Maps a normalized thinking effort to Google's `thinkingLevel` enum values. */ -export function mapEffortToGoogleThinkingLevel(effort: Effort): "MINIMAL" | "LOW" | "MEDIUM" | "HIGH" { +/** Maps a normalized thinking effort to Google's `thinkingLevel` enum values. + * When a collapsed family routes `minimal` onto the same wire id as `low` + * (Antigravity Gemini 3.6/3.7 Flash), emit `LOW` — Cloud Code Assist rejects + * `MINIMAL` on those `-low` SKUs. + */ + effort: Effort, + model?: ApiModel, +): "MINIMAL" | "LOW" | "MEDIUM" | "HIGH" { + if (effort === Effort.Minimal) { + const routing = model?.thinking?.effortRouting; + if (routing?.[Effort.Minimal] && routing[Effort.Minimal] === routing[Effort.Low]) { + return "LOW"; + } + return "MINIMAL"; + } switch (effort) { - case Effort.Minimal: - return "MINIMAL"; case Effort.Low: return "LOW"; case Effort.Medium: diff --git a/packages/catalog/test/model-thinking.test.ts b/packages/catalog/test/model-thinking.test.ts index 41b15e7b2..874c63202 100644 --- a/packages/catalog/test/model-thinking.test.ts +++ b/packages/catalog/test/model-thinking.test.ts @@ -360,6 +360,29 @@ describe("model thinking derivation", () => { expect(() => requireSupportedEffort(model, Effort.Medium)).toThrow(/not supported/); }); + it("maps minimal to LOW when a collapsed family aliases it onto the low wire id", () => { + const model = createModel({ + id: "gemini-3.7-flash", + api: "google-gemini-cli", + provider: "google-antigravity", + thinking: { + mode: "google-level", + efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High], + requiresEffort: true, + effortRouting: { + [Effort.Minimal]: "gemini-3.7-flash-low", + [Effort.Low]: "gemini-3.7-flash-low", + [Effort.Medium]: "gemini-3.7-flash-medium", + [Effort.High]: "gemini-3.7-flash-high", + }, + }, + }); + expect(mapEffortToGoogleThinkingLevel(Effort.Minimal, model)).toBe("LOW"); + expect(mapEffortToGoogleThinkingLevel(Effort.Low, model)).toBe("LOW"); + expect(mapEffortToGoogleThinkingLevel(Effort.Minimal)).toBe("MINIMAL"); + }); + + it("bakes requiresEffort for Gemini 3.x on any provider and backfills explicit metadata", () => { // Derivation: aggregator-hosted Gemini 3.5 gets the flag, 2.5 does not. const openRouterFlash = createModel({