From 1c56a9f3153c564280e78f64f4e31c50f1e37c62 Mon Sep 17 00:00:00 2001 From: cagedbird043 Date: Fri, 3 Jul 2026 13:28:30 +0800 Subject: [PATCH] fix(gemini): suppress CCA thinking summaries by default --- packages/ai/CHANGELOG.md | 4 +++ .../ai/src/providers/google-gemini-cli.ts | 4 ++- packages/ai/src/providers/google-shared.ts | 4 ++- packages/ai/src/stream.ts | 6 ++++ .../google-gemini-cli-3x-thinking.test.ts | 20 ++++++++++++ packages/ai/test/google-service-tier.test.ts | 18 +++++++++++ packages/coding-agent/CHANGELOG.md | 4 +++ .../src/session/settings-stream-fn.ts | 1 + .../test/settings-stream-fn.test.ts | 32 +++++++++++++++++++ 9 files changed, 91 insertions(+), 2 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 9cbda6ba9..c03e5e3c8 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Google Gemini hidden-thinking-summary requests so direct Google and Cloud Code Assist providers keep the requested reasoning tier while sending `includeThoughts: false`. + ## [16.3.5] - 2026-07-04 ### Added diff --git a/packages/ai/src/providers/google-gemini-cli.ts b/packages/ai/src/providers/google-gemini-cli.ts index adbe02a87..b5f07bf83 100644 --- a/packages/ai/src/providers/google-gemini-cli.ts +++ b/packages/ai/src/providers/google-gemini-cli.ts @@ -263,6 +263,8 @@ export interface GoogleGeminiCliOptions extends StreamOptions { */ suppress?: { level: GoogleThinkingLevel } | { budget: number }; }; + /** Request that Cloud Code Assist omit human-readable thought summaries while still allowing internal reasoning. */ + hideThinkingSummary?: boolean; /** * Upstream wire model id override for collapsed effort-tier variants. * Serialized as `requestModelId ?? model.requestModelId ?? model.id`. @@ -1242,7 +1244,7 @@ export function buildRequest( // Thinking config if (options.thinking?.enabled && model.reasoning) { generationConfig.thinkingConfig = { - includeThoughts: true, + includeThoughts: !options.hideThinkingSummary, }; // Gemini 3 models use thinkingLevel, older models use thinkingBudget if (options.thinking.level !== undefined) { diff --git a/packages/ai/src/providers/google-shared.ts b/packages/ai/src/providers/google-shared.ts index 8f9b70cad..28c805b41 100644 --- a/packages/ai/src/providers/google-shared.ts +++ b/packages/ai/src/providers/google-shared.ts @@ -75,6 +75,8 @@ export interface GoogleSharedStreamOptions extends StreamOptions { budgetTokens?: number; level?: GoogleThinkingLevel; }; + /** Request that Google omit human-readable thought summaries while still allowing internal reasoning. */ + hideThinkingSummary?: boolean; /** Gemini/Vertex serving tier (`flex`/`priority`); other values are omitted. */ serviceTier?: ServiceTier; /** @@ -859,7 +861,7 @@ export function buildGoogleGenerateContentParams( enabled: true, level: mapEffortToGoogleThinkingLevel(effort), }, + hideThinkingSummary: options?.hideThinkingSummary, toolChoice: mapGoogleToolChoice(options?.toolChoice), }); } @@ -1666,6 +1667,7 @@ function mapOptionsForApi( enabled: true, budgetTokens: getGoogleBudget(googleModel, effort, options?.thinkingBudgets), }, + hideThinkingSummary: options?.hideThinkingSummary, toolChoice: mapGoogleToolChoice(options?.toolChoice), }); } @@ -1685,6 +1687,7 @@ function mapOptionsForApi( enabled: true, level: mapEffortToGoogleThinkingLevel(effort), }, + hideThinkingSummary: options?.hideThinkingSummary, toolChoice, antigravityEndpointMode: options?.antigravityEndpointMode, }); @@ -1707,6 +1710,7 @@ function mapOptionsForApi( maxTokens, requestModelId: resolveWireModelId(model, effort), thinking: { enabled: true, budgetTokens: thinkingBudget }, + hideThinkingSummary: options?.hideThinkingSummary, toolChoice, antigravityEndpointMode: options?.antigravityEndpointMode, }); @@ -1753,6 +1757,7 @@ function mapOptionsForApi( enabled: true, level: mapEffortToGoogleThinkingLevel(effort), }, + hideThinkingSummary: options?.hideThinkingSummary, toolChoice: mapGoogleToolChoice(options?.toolChoice), }); } @@ -1764,6 +1769,7 @@ function mapOptionsForApi( enabled: true, budgetTokens: getGoogleBudget(geminiModel, effort, options?.thinkingBudgets), }, + hideThinkingSummary: options?.hideThinkingSummary, toolChoice: mapGoogleToolChoice(options?.toolChoice), }); } diff --git a/packages/ai/test/google-gemini-cli-3x-thinking.test.ts b/packages/ai/test/google-gemini-cli-3x-thinking.test.ts index 3d068cafc..32de52174 100644 --- a/packages/ai/test/google-gemini-cli-3x-thinking.test.ts +++ b/packages/ai/test/google-gemini-cli-3x-thinking.test.ts @@ -7,6 +7,7 @@ import { buildModel } from "@oh-my-pi/pi-catalog/build"; interface GeminiCliThinkingConfig { thinkingLevel?: string; thinkingBudget?: number; + includeThoughts?: boolean; } interface CapturedRequestBody { @@ -67,6 +68,25 @@ describe("google-gemini-cli Gemini 3.x thinking mapping", () => { expect(thinking?.thinkingBudget).toBeUndefined(); }); + it("keeps Cloud Code Assist reasoning enabled when only summaries are hidden", async () => { + let requestBody: string | undefined; + const fetchMock = createFetchMock(body => { + requestBody = body; + }); + + const stream = streamSimple(createModel("gemini-3.1-pro-preview"), context, { + apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }), + reasoning: Effort.High, + hideThinkingSummary: true, + fetch: fetchMock, + }); + await stream.result(); + + const thinking = extractThinking(requestBody); + expect(thinking?.includeThoughts).toBe(false); + expect(thinking?.thinkingLevel).toBe("HIGH"); + }); + it("rejects unsupported gemini-3.1-pro-preview efforts instead of promoting them", () => { let requestBody: string | undefined; const fetchMock = createFetchMock(body => { diff --git a/packages/ai/test/google-service-tier.test.ts b/packages/ai/test/google-service-tier.test.ts index ec541d847..74066c206 100644 --- a/packages/ai/test/google-service-tier.test.ts +++ b/packages/ai/test/google-service-tier.test.ts @@ -83,6 +83,24 @@ describe("Google service tier wire encoding", () => { expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBeNull(); }); + it("Gemini API omits human-readable thought summaries when requested", async () => { + const { fetch, captured } = capturingFetch(); + await drain( + streamGoogle(geminiModel, context, { + apiKey: "k", + fetch, + useInteractionsApi: false, + thinking: { enabled: true, level: "HIGH" }, + hideThinkingSummary: true, + }), + ); + + expect((captured().body.generationConfig as { thinkingConfig?: unknown } | undefined)?.thinkingConfig).toEqual({ + includeThoughts: false, + thinkingLevel: "HIGH", + }); + }); + it("Vertex sends priority via header and omits the body tier field", async () => { const { fetch, captured } = capturingFetch(); await drain(streamGoogleVertex(vertexModel, context, { apiKey: "k", serviceTier: "priority", fetch })); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8de3614a0..5d999cf66 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `omitThinking` settings propagation so settings-aware streams request hidden thinking summaries when users explicitly enable the option. + ## [16.3.5] - 2026-07-04 ### Fixed diff --git a/packages/coding-agent/src/session/settings-stream-fn.ts b/packages/coding-agent/src/session/settings-stream-fn.ts index eb5ccf10c..5df2c33f2 100644 --- a/packages/coding-agent/src/session/settings-stream-fn.ts +++ b/packages/coding-agent/src/session/settings-stream-fn.ts @@ -66,6 +66,7 @@ export function createSettingsAwareStreamFn(settings: Settings, base: StreamFn = checkAssistantContent: settings.get("model.loopGuard.checkAssistantContent"), ...streamOptions?.loopGuard, }, + hideThinkingSummary: streamOptions?.hideThinkingSummary ?? settings.get("omitThinking"), ...(fallbacks !== undefined ? { fallbacks } : {}), }; return base(model, context, merged); diff --git a/packages/coding-agent/test/settings-stream-fn.test.ts b/packages/coding-agent/test/settings-stream-fn.test.ts index 2a64fd154..4abdc0d6d 100644 --- a/packages/coding-agent/test/settings-stream-fn.test.ts +++ b/packages/coding-agent/test/settings-stream-fn.test.ts @@ -51,6 +51,36 @@ describe("createSettingsAwareStreamFn", () => { expect(options?.apiKey).toBe("k"); }); + it("keeps assistant prose loop scanning at its configured default", () => { + const settings = Settings.isolated({}); + const { fn: base, calls } = captureBase(); + const wrapped = createSettingsAwareStreamFn(settings, base); + + wrapped(stubModel, stubContext, undefined); + + expect(calls[0]?.options?.loopGuard).toEqual({ enabled: true, checkAssistantContent: true }); + }); + + it("keeps thinking summaries visible unless configured otherwise", () => { + const settings = Settings.isolated({}); + const { fn: base, calls } = captureBase(); + const wrapped = createSettingsAwareStreamFn(settings, base); + + wrapped(stubModel, stubContext, undefined); + + expect(calls[0]?.options?.hideThinkingSummary).toBe(false); + }); + + it("forwards configured hidden thinking summaries", () => { + const settings = Settings.isolated({ omitThinking: true }); + const { fn: base, calls } = captureBase(); + const wrapped = createSettingsAwareStreamFn(settings, base); + + wrapped(stubModel, stubContext, undefined); + + expect(calls[0]?.options?.hideThinkingSummary).toBe(true); + }); + it("applies Responses-family text verbosity from settings while preserving caller overrides", () => { const settings = Settings.isolated({ textVerbosity: "low" }); const { fn: base, calls } = captureBase(); @@ -110,6 +140,7 @@ describe("createSettingsAwareStreamFn", () => { antigravityEndpointMode: "production", maxInFlightRequests: { openrouter: 1 }, loopGuard: { enabled: false }, + hideThinkingSummary: false, }); const options = calls[0]?.options; @@ -120,6 +151,7 @@ describe("createSettingsAwareStreamFn", () => { // the rest (the inline closure the main agent used has the same shape). expect(options?.loopGuard?.enabled).toBe(false); expect(options?.loopGuard?.checkAssistantContent).toBe(true); + expect(options?.hideThinkingSummary).toBe(false); }); describe("providers.anthropic.serverSideFallback (opt-in)", () => { const stubFableModel = {