Merge PR #8871: fix(catalog): map aliased Gemini Flash minimal to LOW on Cloud Code Assist (@audreyt)

This commit is contained in:
can1357
2026-08-19 01:37:00 +02:00
6 changed files with 93 additions and 7 deletions
+4
View File
@@ -10,6 +10,10 @@
- Fixed Codex requests failing outright when the signed-in ChatGPT account is not entitled to the requested model; the exact model denial is now classified as an account-policy error so credential rotation can reach an entitled sibling account
- Fixed Perplexity email-OTP login after its verification response renamed the encrypted session token from `token` to `challenge_token`.
### Fixed
- Cloud Code Assist Gemini 3.6/3.7 Flash requests at `minimal` now send `thinkingLevel: LOW` on the aliased `-low` SKU instead of `MINIMAL`, which the API rejects with HTTP 400.
## [17.3.7] - 2026-08-17
### Changed
+3 -3
View File
@@ -2174,7 +2174,7 @@ function mapOptionsForApi<TApi extends Api>(
serviceTier: options?.serviceTier,
thinking: {
enabled: true,
level: mapEffortToGoogleThinkingLevel(effort),
level: mapEffortToGoogleThinkingLevel(effort, googleModel),
},
hideThinkingSummary: options?.hideThinkingSummary,
toolChoice: mapGoogleToolChoice(options?.toolChoice),
@@ -2207,7 +2207,7 @@ function mapOptionsForApi<TApi extends Api>(
requestModelId: resolveWireModelId(model, effort),
thinking: {
enabled: true,
level: mapEffortToGoogleThinkingLevel(effort),
level: mapEffortToGoogleThinkingLevel(effort, model),
},
hideThinkingSummary: options?.hideThinkingSummary,
toolChoice,
@@ -2279,7 +2279,7 @@ function mapOptionsForApi<TApi extends Api>(
serviceTier: options?.serviceTier,
thinking: {
enabled: true,
level: mapEffortToGoogleThinkingLevel(effort),
level: mapEffortToGoogleThinkingLevel(effort, model),
},
hideThinkingSummary: options?.hideThinkingSummary,
toolChoice: mapGoogleToolChoice(options?.toolChoice),
@@ -11,6 +11,7 @@ interface GeminiCliThinkingConfig {
}
interface CapturedRequestBody {
model?: string;
request?: {
generationConfig?: {
thinkingConfig?: GeminiCliThinkingConfig;
@@ -138,4 +139,47 @@ describe("google-gemini-cli Gemini 3.x thinking mapping", () => {
expect(thinking?.thinkingLevel).toBeUndefined();
expect(thinking?.thinkingBudget).toBeDefined();
});
it("sends LOW not MINIMAL when gemini-3.7-flash minimal aliases the -low SKU", async () => {
let requestBody: string | undefined;
const fetchMock = createFetchMock(body => {
requestBody = body;
});
const model = buildModel({
id: "gemini-3.7-flash",
name: "Gemini 3.7 Flash",
api: "google-gemini-cli",
provider: "google-antigravity",
baseUrl: "https://daily-cloudcode-pa.googleapis.com",
reasoning: true,
thinking: {
mode: "google-level",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
requiresEffort: true,
effortRouting: {
[Effort.Minimal]: "gemini-3.7-flash-low",
[Effort.Low]: "gemini-3.7-flash-low",
[Effort.Medium]: "gemini-3.7-flash-medium",
[Effort.High]: "gemini-3.7-flash-high",
},
},
input: ["text", "image"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 1_048_576,
maxTokens: 65_536,
});
const stream = streamSimple(model, context, {
apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }),
reasoning: Effort.Minimal,
fetch: fetchMock,
});
await stream.result();
const parsed = JSON.parse(requestBody ?? "{}") as CapturedRequestBody;
expect(parsed.model).toBe("gemini-3.7-flash-low");
expect(parsed.request?.generationConfig?.thinkingConfig?.thinkingLevel).toBe("LOW");
});
});
+4
View File
@@ -7,6 +7,10 @@
- Fixed local Qwen 3.8+ models (llama.cpp, vLLM, loopback custom providers) exposing the generic `minimal..high` thinking ladder instead of the chat template's real `low`/`medium`/`xhigh` `reasoning_effort` tiers. The derived metadata now marks thinking as mandatory (the official 3.8 template raises on `enable_thinking: false`), vLLM-served Qwen routes through the `chat_template_kwargs` dialect (top-level `enable_thinking` is ignored by vLLM), and vLLM discovery lights up the reasoning dial for Qwen 3.8+ ids its `/v1/models` endpoint reports as non-reasoning.
- Fixed `deepseek-v4-pro-0813` surfacing from Alibaba Token Plan discovery with `contextWindow`/`maxTokens` of `null`. The dated DeepSeek V4 Pro snapshot was missing from `ALIBABA_TOKEN_PLAN_DISCOVERED_MODEL_LIMITS`, so unlike its `deepseek-v4-flash-0731` sibling it fell through to unknown limits ([#8847](https://github.com/can1357/oh-my-pi/issues/8847)).
### Fixed
- Cloud Code Assist Gemini 3.6/3.7 Flash no longer maps user `minimal` to wire `thinkingLevel: MINIMAL` when that effort is aliased onto the `-low` SKU. The request now sends `LOW`, which those SKUs accept.
## [17.3.6] - 2026-08-17
### Changed
+15 -4
View File
@@ -816,11 +816,22 @@ export function requireSupportedEffort<TApi extends Api>(model: ApiModel<TApi>,
return effort;
}
/** Maps a normalized thinking effort to Google's `thinkingLevel` enum values. */
export function mapEffortToGoogleThinkingLevel(effort: Effort): "MINIMAL" | "LOW" | "MEDIUM" | "HIGH" {
/** Maps a normalized thinking effort to Google's `thinkingLevel` enum values.
* When a collapsed family routes `minimal` onto the same wire id as `low`
* (Antigravity Gemini 3.6/3.7 Flash), emit `LOW` — Cloud Code Assist rejects
* `MINIMAL` on those `-low` SKUs.
*/
effort: Effort,
model?: ApiModel<TApi>,
): "MINIMAL" | "LOW" | "MEDIUM" | "HIGH" {
if (effort === Effort.Minimal) {
const routing = model?.thinking?.effortRouting;
if (routing?.[Effort.Minimal] && routing[Effort.Minimal] === routing[Effort.Low]) {
return "LOW";
}
return "MINIMAL";
}
switch (effort) {
case Effort.Minimal:
return "MINIMAL";
case Effort.Low:
return "LOW";
case Effort.Medium:
@@ -360,6 +360,29 @@ describe("model thinking derivation", () => {
expect(() => requireSupportedEffort(model, Effort.Medium)).toThrow(/not supported/);
});
it("maps minimal to LOW when a collapsed family aliases it onto the low wire id", () => {
const model = createModel({
id: "gemini-3.7-flash",
api: "google-gemini-cli",
provider: "google-antigravity",
thinking: {
mode: "google-level",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
requiresEffort: true,
effortRouting: {
[Effort.Minimal]: "gemini-3.7-flash-low",
[Effort.Low]: "gemini-3.7-flash-low",
[Effort.Medium]: "gemini-3.7-flash-medium",
[Effort.High]: "gemini-3.7-flash-high",
},
},
});
expect(mapEffortToGoogleThinkingLevel(Effort.Minimal, model)).toBe("LOW");
expect(mapEffortToGoogleThinkingLevel(Effort.Low, model)).toBe("LOW");
expect(mapEffortToGoogleThinkingLevel(Effort.Minimal)).toBe("MINIMAL");
});
it("bakes requiresEffort for Gemini 3.x on any provider and backfills explicit metadata", () => {
// Derivation: aggregator-hosted Gemini 3.5 gets the flag, 2.5 does not.
const openRouterFlash = createModel({