fix(gemini): suppress CCA thinking summaries by default
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Google Gemini hidden-thinking-summary requests so direct Google and Cloud Code Assist providers keep the requested reasoning tier while sending `includeThoughts: false`.
|
||||
|
||||
## [16.3.5] - 2026-07-04
|
||||
|
||||
### Added
|
||||
|
||||
@@ -263,6 +263,8 @@ export interface GoogleGeminiCliOptions extends StreamOptions {
|
||||
*/
|
||||
suppress?: { level: GoogleThinkingLevel } | { budget: number };
|
||||
};
|
||||
/** Request that Cloud Code Assist omit human-readable thought summaries while still allowing internal reasoning. */
|
||||
hideThinkingSummary?: boolean;
|
||||
/**
|
||||
* Upstream wire model id override for collapsed effort-tier variants.
|
||||
* Serialized as `requestModelId ?? model.requestModelId ?? model.id`.
|
||||
@@ -1242,7 +1244,7 @@ export function buildRequest(
|
||||
// Thinking config
|
||||
if (options.thinking?.enabled && model.reasoning) {
|
||||
generationConfig.thinkingConfig = {
|
||||
includeThoughts: true,
|
||||
includeThoughts: !options.hideThinkingSummary,
|
||||
};
|
||||
// Gemini 3 models use thinkingLevel, older models use thinkingBudget
|
||||
if (options.thinking.level !== undefined) {
|
||||
|
||||
@@ -75,6 +75,8 @@ export interface GoogleSharedStreamOptions extends StreamOptions {
|
||||
budgetTokens?: number;
|
||||
level?: GoogleThinkingLevel;
|
||||
};
|
||||
/** Request that Google omit human-readable thought summaries while still allowing internal reasoning. */
|
||||
hideThinkingSummary?: boolean;
|
||||
/** Gemini/Vertex serving tier (`flex`/`priority`); other values are omitted. */
|
||||
serviceTier?: ServiceTier;
|
||||
/**
|
||||
@@ -859,7 +861,7 @@ export function buildGoogleGenerateContentParams<T extends "google-generative-ai
|
||||
}
|
||||
|
||||
if (options.thinking?.enabled && model.reasoning) {
|
||||
const cfg: ThinkingConfig = { includeThoughts: true };
|
||||
const cfg: ThinkingConfig = { includeThoughts: !options.hideThinkingSummary };
|
||||
if (options.thinking.level !== undefined) {
|
||||
// GoogleThinkingLevel mirrors the SDK's `ThinkingLevel` string enum values 1:1.
|
||||
cfg.thinkingLevel = options.thinking.level as ThinkingLevel;
|
||||
|
||||
@@ -1656,6 +1656,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
enabled: true,
|
||||
level: mapEffortToGoogleThinkingLevel(effort),
|
||||
},
|
||||
hideThinkingSummary: options?.hideThinkingSummary,
|
||||
toolChoice: mapGoogleToolChoice(options?.toolChoice),
|
||||
});
|
||||
}
|
||||
@@ -1666,6 +1667,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
enabled: true,
|
||||
budgetTokens: getGoogleBudget(googleModel, effort, options?.thinkingBudgets),
|
||||
},
|
||||
hideThinkingSummary: options?.hideThinkingSummary,
|
||||
toolChoice: mapGoogleToolChoice(options?.toolChoice),
|
||||
});
|
||||
}
|
||||
@@ -1685,6 +1687,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
enabled: true,
|
||||
level: mapEffortToGoogleThinkingLevel(effort),
|
||||
},
|
||||
hideThinkingSummary: options?.hideThinkingSummary,
|
||||
toolChoice,
|
||||
antigravityEndpointMode: options?.antigravityEndpointMode,
|
||||
});
|
||||
@@ -1707,6 +1710,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
maxTokens,
|
||||
requestModelId: resolveWireModelId(model, effort),
|
||||
thinking: { enabled: true, budgetTokens: thinkingBudget },
|
||||
hideThinkingSummary: options?.hideThinkingSummary,
|
||||
toolChoice,
|
||||
antigravityEndpointMode: options?.antigravityEndpointMode,
|
||||
});
|
||||
@@ -1753,6 +1757,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
enabled: true,
|
||||
level: mapEffortToGoogleThinkingLevel(effort),
|
||||
},
|
||||
hideThinkingSummary: options?.hideThinkingSummary,
|
||||
toolChoice: mapGoogleToolChoice(options?.toolChoice),
|
||||
});
|
||||
}
|
||||
@@ -1764,6 +1769,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
enabled: true,
|
||||
budgetTokens: getGoogleBudget(geminiModel, effort, options?.thinkingBudgets),
|
||||
},
|
||||
hideThinkingSummary: options?.hideThinkingSummary,
|
||||
toolChoice: mapGoogleToolChoice(options?.toolChoice),
|
||||
});
|
||||
}
|
||||
|
||||
@@ -7,6 +7,7 @@ import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
interface GeminiCliThinkingConfig {
|
||||
thinkingLevel?: string;
|
||||
thinkingBudget?: number;
|
||||
includeThoughts?: boolean;
|
||||
}
|
||||
|
||||
interface CapturedRequestBody {
|
||||
@@ -67,6 +68,25 @@ describe("google-gemini-cli Gemini 3.x thinking mapping", () => {
|
||||
expect(thinking?.thinkingBudget).toBeUndefined();
|
||||
});
|
||||
|
||||
it("keeps Cloud Code Assist reasoning enabled when only summaries are hidden", async () => {
|
||||
let requestBody: string | undefined;
|
||||
const fetchMock = createFetchMock(body => {
|
||||
requestBody = body;
|
||||
});
|
||||
|
||||
const stream = streamSimple(createModel("gemini-3.1-pro-preview"), context, {
|
||||
apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }),
|
||||
reasoning: Effort.High,
|
||||
hideThinkingSummary: true,
|
||||
fetch: fetchMock,
|
||||
});
|
||||
await stream.result();
|
||||
|
||||
const thinking = extractThinking(requestBody);
|
||||
expect(thinking?.includeThoughts).toBe(false);
|
||||
expect(thinking?.thinkingLevel).toBe("HIGH");
|
||||
});
|
||||
|
||||
it("rejects unsupported gemini-3.1-pro-preview efforts instead of promoting them", () => {
|
||||
let requestBody: string | undefined;
|
||||
const fetchMock = createFetchMock(body => {
|
||||
|
||||
@@ -83,6 +83,24 @@ describe("Google service tier wire encoding", () => {
|
||||
expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBeNull();
|
||||
});
|
||||
|
||||
it("Gemini API omits human-readable thought summaries when requested", async () => {
|
||||
const { fetch, captured } = capturingFetch();
|
||||
await drain(
|
||||
streamGoogle(geminiModel, context, {
|
||||
apiKey: "k",
|
||||
fetch,
|
||||
useInteractionsApi: false,
|
||||
thinking: { enabled: true, level: "HIGH" },
|
||||
hideThinkingSummary: true,
|
||||
}),
|
||||
);
|
||||
|
||||
expect((captured().body.generationConfig as { thinkingConfig?: unknown } | undefined)?.thinkingConfig).toEqual({
|
||||
includeThoughts: false,
|
||||
thinkingLevel: "HIGH",
|
||||
});
|
||||
});
|
||||
|
||||
it("Vertex sends priority via header and omits the body tier field", async () => {
|
||||
const { fetch, captured } = capturingFetch();
|
||||
await drain(streamGoogleVertex(vertexModel, context, { apiKey: "k", serviceTier: "priority", fetch }));
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed `omitThinking` settings propagation so settings-aware streams request hidden thinking summaries when users explicitly enable the option.
|
||||
|
||||
## [16.3.5] - 2026-07-04
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -66,6 +66,7 @@ export function createSettingsAwareStreamFn(settings: Settings, base: StreamFn =
|
||||
checkAssistantContent: settings.get("model.loopGuard.checkAssistantContent"),
|
||||
...streamOptions?.loopGuard,
|
||||
},
|
||||
hideThinkingSummary: streamOptions?.hideThinkingSummary ?? settings.get("omitThinking"),
|
||||
...(fallbacks !== undefined ? { fallbacks } : {}),
|
||||
};
|
||||
return base(model, context, merged);
|
||||
|
||||
@@ -51,6 +51,36 @@ describe("createSettingsAwareStreamFn", () => {
|
||||
expect(options?.apiKey).toBe("k");
|
||||
});
|
||||
|
||||
it("keeps assistant prose loop scanning at its configured default", () => {
|
||||
const settings = Settings.isolated({});
|
||||
const { fn: base, calls } = captureBase();
|
||||
const wrapped = createSettingsAwareStreamFn(settings, base);
|
||||
|
||||
wrapped(stubModel, stubContext, undefined);
|
||||
|
||||
expect(calls[0]?.options?.loopGuard).toEqual({ enabled: true, checkAssistantContent: true });
|
||||
});
|
||||
|
||||
it("keeps thinking summaries visible unless configured otherwise", () => {
|
||||
const settings = Settings.isolated({});
|
||||
const { fn: base, calls } = captureBase();
|
||||
const wrapped = createSettingsAwareStreamFn(settings, base);
|
||||
|
||||
wrapped(stubModel, stubContext, undefined);
|
||||
|
||||
expect(calls[0]?.options?.hideThinkingSummary).toBe(false);
|
||||
});
|
||||
|
||||
it("forwards configured hidden thinking summaries", () => {
|
||||
const settings = Settings.isolated({ omitThinking: true });
|
||||
const { fn: base, calls } = captureBase();
|
||||
const wrapped = createSettingsAwareStreamFn(settings, base);
|
||||
|
||||
wrapped(stubModel, stubContext, undefined);
|
||||
|
||||
expect(calls[0]?.options?.hideThinkingSummary).toBe(true);
|
||||
});
|
||||
|
||||
it("applies Responses-family text verbosity from settings while preserving caller overrides", () => {
|
||||
const settings = Settings.isolated({ textVerbosity: "low" });
|
||||
const { fn: base, calls } = captureBase();
|
||||
@@ -110,6 +140,7 @@ describe("createSettingsAwareStreamFn", () => {
|
||||
antigravityEndpointMode: "production",
|
||||
maxInFlightRequests: { openrouter: 1 },
|
||||
loopGuard: { enabled: false },
|
||||
hideThinkingSummary: false,
|
||||
});
|
||||
|
||||
const options = calls[0]?.options;
|
||||
@@ -120,6 +151,7 @@ describe("createSettingsAwareStreamFn", () => {
|
||||
// the rest (the inline closure the main agent used has the same shape).
|
||||
expect(options?.loopGuard?.enabled).toBe(false);
|
||||
expect(options?.loopGuard?.checkAssistantContent).toBe(true);
|
||||
expect(options?.hideThinkingSummary).toBe(false);
|
||||
});
|
||||
describe("providers.anthropic.serverSideFallback (opt-in)", () => {
|
||||
const stubFableModel = {
|
||||
|
||||
Reference in New Issue
Block a user