fix(gemini): suppress CCA thinking summaries by default

This commit is contained in:
cagedbird043
2026-07-03 13:28:30 +08:00
parent fef9b62334
commit 1c56a9f315
9 changed files with 91 additions and 2 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed Google Gemini hidden-thinking-summary requests so direct Google and Cloud Code Assist providers keep the requested reasoning tier while sending `includeThoughts: false`.
## [16.3.5] - 2026-07-04
### Added
@@ -263,6 +263,8 @@ export interface GoogleGeminiCliOptions extends StreamOptions {
*/
suppress?: { level: GoogleThinkingLevel } | { budget: number };
};
/** Request that Cloud Code Assist omit human-readable thought summaries while still allowing internal reasoning. */
hideThinkingSummary?: boolean;
/**
* Upstream wire model id override for collapsed effort-tier variants.
* Serialized as `requestModelId ?? model.requestModelId ?? model.id`.
@@ -1242,7 +1244,7 @@ export function buildRequest(
// Thinking config
if (options.thinking?.enabled && model.reasoning) {
generationConfig.thinkingConfig = {
includeThoughts: true,
includeThoughts: !options.hideThinkingSummary,
};
// Gemini 3 models use thinkingLevel, older models use thinkingBudget
if (options.thinking.level !== undefined) {
+3 -1
View File
@@ -75,6 +75,8 @@ export interface GoogleSharedStreamOptions extends StreamOptions {
budgetTokens?: number;
level?: GoogleThinkingLevel;
};
/** Request that Google omit human-readable thought summaries while still allowing internal reasoning. */
hideThinkingSummary?: boolean;
/** Gemini/Vertex serving tier (`flex`/`priority`); other values are omitted. */
serviceTier?: ServiceTier;
/**
@@ -859,7 +861,7 @@ export function buildGoogleGenerateContentParams<T extends "google-generative-ai
}
if (options.thinking?.enabled && model.reasoning) {
const cfg: ThinkingConfig = { includeThoughts: true };
const cfg: ThinkingConfig = { includeThoughts: !options.hideThinkingSummary };
if (options.thinking.level !== undefined) {
// GoogleThinkingLevel mirrors the SDK's `ThinkingLevel` string enum values 1:1.
cfg.thinkingLevel = options.thinking.level as ThinkingLevel;
+6
View File
@@ -1656,6 +1656,7 @@ function mapOptionsForApi<TApi extends Api>(
enabled: true,
level: mapEffortToGoogleThinkingLevel(effort),
},
hideThinkingSummary: options?.hideThinkingSummary,
toolChoice: mapGoogleToolChoice(options?.toolChoice),
});
}
@@ -1666,6 +1667,7 @@ function mapOptionsForApi<TApi extends Api>(
enabled: true,
budgetTokens: getGoogleBudget(googleModel, effort, options?.thinkingBudgets),
},
hideThinkingSummary: options?.hideThinkingSummary,
toolChoice: mapGoogleToolChoice(options?.toolChoice),
});
}
@@ -1685,6 +1687,7 @@ function mapOptionsForApi<TApi extends Api>(
enabled: true,
level: mapEffortToGoogleThinkingLevel(effort),
},
hideThinkingSummary: options?.hideThinkingSummary,
toolChoice,
antigravityEndpointMode: options?.antigravityEndpointMode,
});
@@ -1707,6 +1710,7 @@ function mapOptionsForApi<TApi extends Api>(
maxTokens,
requestModelId: resolveWireModelId(model, effort),
thinking: { enabled: true, budgetTokens: thinkingBudget },
hideThinkingSummary: options?.hideThinkingSummary,
toolChoice,
antigravityEndpointMode: options?.antigravityEndpointMode,
});
@@ -1753,6 +1757,7 @@ function mapOptionsForApi<TApi extends Api>(
enabled: true,
level: mapEffortToGoogleThinkingLevel(effort),
},
hideThinkingSummary: options?.hideThinkingSummary,
toolChoice: mapGoogleToolChoice(options?.toolChoice),
});
}
@@ -1764,6 +1769,7 @@ function mapOptionsForApi<TApi extends Api>(
enabled: true,
budgetTokens: getGoogleBudget(geminiModel, effort, options?.thinkingBudgets),
},
hideThinkingSummary: options?.hideThinkingSummary,
toolChoice: mapGoogleToolChoice(options?.toolChoice),
});
}
@@ -7,6 +7,7 @@ import { buildModel } from "@oh-my-pi/pi-catalog/build";
interface GeminiCliThinkingConfig {
thinkingLevel?: string;
thinkingBudget?: number;
includeThoughts?: boolean;
}
interface CapturedRequestBody {
@@ -67,6 +68,25 @@ describe("google-gemini-cli Gemini 3.x thinking mapping", () => {
expect(thinking?.thinkingBudget).toBeUndefined();
});
it("keeps Cloud Code Assist reasoning enabled when only summaries are hidden", async () => {
let requestBody: string | undefined;
const fetchMock = createFetchMock(body => {
requestBody = body;
});
const stream = streamSimple(createModel("gemini-3.1-pro-preview"), context, {
apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }),
reasoning: Effort.High,
hideThinkingSummary: true,
fetch: fetchMock,
});
await stream.result();
const thinking = extractThinking(requestBody);
expect(thinking?.includeThoughts).toBe(false);
expect(thinking?.thinkingLevel).toBe("HIGH");
});
it("rejects unsupported gemini-3.1-pro-preview efforts instead of promoting them", () => {
let requestBody: string | undefined;
const fetchMock = createFetchMock(body => {
@@ -83,6 +83,24 @@ describe("Google service tier wire encoding", () => {
expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBeNull();
});
it("Gemini API omits human-readable thought summaries when requested", async () => {
const { fetch, captured } = capturingFetch();
await drain(
streamGoogle(geminiModel, context, {
apiKey: "k",
fetch,
useInteractionsApi: false,
thinking: { enabled: true, level: "HIGH" },
hideThinkingSummary: true,
}),
);
expect((captured().body.generationConfig as { thinkingConfig?: unknown } | undefined)?.thinkingConfig).toEqual({
includeThoughts: false,
thinkingLevel: "HIGH",
});
});
it("Vertex sends priority via header and omits the body tier field", async () => {
const { fetch, captured } = capturingFetch();
await drain(streamGoogleVertex(vertexModel, context, { apiKey: "k", serviceTier: "priority", fetch }));
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed `omitThinking` settings propagation so settings-aware streams request hidden thinking summaries when users explicitly enable the option.
## [16.3.5] - 2026-07-04
### Fixed
@@ -66,6 +66,7 @@ export function createSettingsAwareStreamFn(settings: Settings, base: StreamFn =
checkAssistantContent: settings.get("model.loopGuard.checkAssistantContent"),
...streamOptions?.loopGuard,
},
hideThinkingSummary: streamOptions?.hideThinkingSummary ?? settings.get("omitThinking"),
...(fallbacks !== undefined ? { fallbacks } : {}),
};
return base(model, context, merged);
@@ -51,6 +51,36 @@ describe("createSettingsAwareStreamFn", () => {
expect(options?.apiKey).toBe("k");
});
it("keeps assistant prose loop scanning at its configured default", () => {
const settings = Settings.isolated({});
const { fn: base, calls } = captureBase();
const wrapped = createSettingsAwareStreamFn(settings, base);
wrapped(stubModel, stubContext, undefined);
expect(calls[0]?.options?.loopGuard).toEqual({ enabled: true, checkAssistantContent: true });
});
it("keeps thinking summaries visible unless configured otherwise", () => {
const settings = Settings.isolated({});
const { fn: base, calls } = captureBase();
const wrapped = createSettingsAwareStreamFn(settings, base);
wrapped(stubModel, stubContext, undefined);
expect(calls[0]?.options?.hideThinkingSummary).toBe(false);
});
it("forwards configured hidden thinking summaries", () => {
const settings = Settings.isolated({ omitThinking: true });
const { fn: base, calls } = captureBase();
const wrapped = createSettingsAwareStreamFn(settings, base);
wrapped(stubModel, stubContext, undefined);
expect(calls[0]?.options?.hideThinkingSummary).toBe(true);
});
it("applies Responses-family text verbosity from settings while preserving caller overrides", () => {
const settings = Settings.isolated({ textVerbosity: "low" });
const { fn: base, calls } = captureBase();
@@ -110,6 +140,7 @@ describe("createSettingsAwareStreamFn", () => {
antigravityEndpointMode: "production",
maxInFlightRequests: { openrouter: 1 },
loopGuard: { enabled: false },
hideThinkingSummary: false,
});
const options = calls[0]?.options;
@@ -120,6 +151,7 @@ describe("createSettingsAwareStreamFn", () => {
// the rest (the inline closure the main agent used has the same shape).
expect(options?.loopGuard?.enabled).toBe(false);
expect(options?.loopGuard?.checkAssistantContent).toBe(true);
expect(options?.hideThinkingSummary).toBe(false);
});
describe("providers.anthropic.serverSideFallback (opt-in)", () => {
const stubFableModel = {