feat(catalog): enabled budget-aware Antigravity routing and request envelope tracking
- Enabled Antigravity request envelopes to include requestId/labels and per-attempt responseId state. - Enabled budget transport to send `thinkingBudget` and capped `maxOutputTokens` 65_536 for Gemini calls. - Added budget-aware model profiles for Gemini flash/pro variants and updated effort routing expectations. - Refreshed variant collapse to separate Gemini and Antigravity tables and heal stale snapshots.
This commit is contained in:
@@ -1,6 +1,7 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added `antigravityEndpointMode` stream option with `auto`, `production`, and `sandbox` values to control Antigravity endpoint routing
|
||||
@@ -9,8 +10,14 @@
|
||||
- Added `LITELLM_BASE_URL` guidance to the LiteLLM login prompt so non-default proxy endpoints are discoverable. ([#2726](https://github.com/can1357/oh-my-pi/issues/2726))
|
||||
- Added a Gemini thinking-loop guard that watches streamed `thinking` deltas for degenerate reasoning loops — verbatim tail repetition and near-duplicate paragraph cycling — and terminates the stream with a retryable, empty-content `error` message (worded as a transient stream stall) so the turn is discarded and re-sampled instead of committing a runaway transcript. Gated to Gemini models across every transport (OpenRouter, direct Google, Vertex) and disarmed once visible answer text or a tool call starts; disable with `PI_NO_THINKING_LOOP_GUARD=1`.
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed the Antigravity (`google-antigravity`) request builder to mirror the captured `antigravity/hub` client: gemini-3.x send `thinkingConfig.thinkingBudget` per tier, a fixed per-model `maxOutputTokens`, a default `functionCallingConfig.mode: "VALIDATED"` tool mode (auto/unset tool choice only), a `role: "user"` system instruction, a structured `requestId` (`agent/<id>/<ts>/<trajectoryId>/<step>`), and `labels` (`model_enum`, `trajectory_id`, `last_step_index`, `last_execution_id`, `used_claude*`) tracked across the conversation via provider session state.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Gemini usage-tier mapping so `gemini-3.5-flash` is treated as `Flash` and `gemini-3.1-pro` plus `gemini-pro-agent` are treated as `Pro` in usage accounting
|
||||
- Fixed Antigravity stream state handling so a request’s `last_execution_id` is committed only after a successful completion and cleared between retry attempts
|
||||
- Fixed `streamSimple()` Gemini streams to run through the thinking-loop guard for custom API and pi-native transports, so degenerate `thinking` loops now abort with the same retryable empty-content error path as other Gemini stream paths
|
||||
- Fixed Antigravity model streaming and usage fetch paths to retry on transient `429`/`5xx` errors by failing over to the alternate endpoint before surfacing an error
|
||||
- Fixed Antigravity endpoint tracking to prefer a previously successful endpoint in `auto` mode for subsequent requests
|
||||
|
||||
@@ -9,6 +9,7 @@ import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
||||
import {
|
||||
ANTIGRAVITY_NO_PREAMBLE_INSTRUCTION,
|
||||
ANTIGRAVITY_SYSTEM_INSTRUCTION,
|
||||
getAntigravityModelWireProfile,
|
||||
getAntigravityUserAgent,
|
||||
getGeminiCliHeaders,
|
||||
} from "@oh-my-pi/pi-catalog/wire/gemini-headers";
|
||||
@@ -104,6 +105,18 @@ export interface GoogleGeminiCliOptions extends StreamOptions {
|
||||
|
||||
export interface AntigravityProviderSessionState extends ProviderSessionState {
|
||||
lastGoodEndpoint?: string;
|
||||
/**
|
||||
* Per-conversation request-envelope identity that mirrors the real
|
||||
* Antigravity client. `sessionId` is the signed-decimal session id;
|
||||
* `agentId`/`trajectoryId` are UUIDs; `stepIndex` is the monotonic step
|
||||
* counter; `lastExecutionId` is the prior response id echoed as
|
||||
* `labels.last_execution_id`.
|
||||
*/
|
||||
agentId?: string;
|
||||
trajectoryId?: string;
|
||||
sessionId?: string;
|
||||
stepIndex?: number;
|
||||
lastExecutionId?: string;
|
||||
}
|
||||
|
||||
const ANTIGRAVITY_PROVIDER_SESSION_STATE_KEY = "google-antigravity-session-state";
|
||||
@@ -277,6 +290,7 @@ interface CloudCodeAssistRequest {
|
||||
allowedFunctionNames?: string[];
|
||||
};
|
||||
};
|
||||
labels?: Record<string, string>;
|
||||
};
|
||||
requestType?: string;
|
||||
userAgent?: string;
|
||||
@@ -458,6 +472,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
||||
|
||||
let started = false;
|
||||
let sawFinishReason = false;
|
||||
let lastResponseId: string | undefined;
|
||||
const ensureStarted = () => {
|
||||
if (!started) {
|
||||
if (!firstTokenTime) firstTokenTime = Date.now();
|
||||
@@ -487,6 +502,10 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
||||
throw new Error("No response body");
|
||||
}
|
||||
|
||||
// Scoped per attempt so a failed/empty retry cannot leak its
|
||||
// response id into the next request's last_execution_id.
|
||||
lastResponseId = undefined;
|
||||
|
||||
let currentBlock: TextContent | ThinkingContent | null = null;
|
||||
const blocks = output.content;
|
||||
const blockIndex = () => blocks.length - 1;
|
||||
@@ -505,6 +524,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
||||
}
|
||||
const responseData = chunk.response;
|
||||
if (!responseData) continue;
|
||||
if (responseData.responseId) lastResponseId = responseData.responseId;
|
||||
if (!responseData.candidates?.length && responseData.promptFeedback?.blockReason) {
|
||||
const detail = responseData.promptFeedback.blockReasonMessage;
|
||||
throw new Error(
|
||||
@@ -750,6 +770,12 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
||||
) {
|
||||
providerState.lastGoodEndpoint = endpoint;
|
||||
}
|
||||
// Commit after a fully successful attempt (content + finish reason);
|
||||
// used as the next request's last_execution_id. Overwrite even when
|
||||
// undefined so a response without an id can't leave a stale value.
|
||||
if (providerState) {
|
||||
providerState.lastExecutionId = lastResponseId;
|
||||
}
|
||||
break;
|
||||
} catch (error) {
|
||||
const status = extractHttpStatusFromError(error);
|
||||
@@ -876,6 +902,48 @@ function normalizeAntigravityTools(
|
||||
}));
|
||||
}
|
||||
|
||||
interface AntigravityRequestEnvelope {
|
||||
sessionId: string;
|
||||
requestId: string;
|
||||
labels: Record<string, string>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the Antigravity request envelope (sessionId, structured requestId,
|
||||
* labels) advancing the per-conversation session state. Mirrors the real
|
||||
* `antigravity/hub` client: `requestId` is `agent/<agentId>/<ts>/<trajectoryId>/<step>`
|
||||
* and `labels.last_step_index` trails the requestId step by one. Without session
|
||||
* state (direct callers/tests) it falls back to ephemeral ids.
|
||||
*/
|
||||
function buildAntigravityRequestEnvelope(
|
||||
model: Model<"google-gemini-cli">,
|
||||
context: Context,
|
||||
wireModelId: string,
|
||||
state: AntigravityProviderSessionState | undefined,
|
||||
): AntigravityRequestEnvelope {
|
||||
if (state) {
|
||||
state.agentId ??= randomUUID();
|
||||
state.trajectoryId ??= randomUUID();
|
||||
state.sessionId ??= randomSignedDecimalSessionId();
|
||||
state.stepIndex = (state.stepIndex ?? 1) + 1;
|
||||
}
|
||||
const agentId = state?.agentId ?? randomUUID();
|
||||
const trajectoryId = state?.trajectoryId ?? randomUUID();
|
||||
const sessionId = state?.sessionId ?? deriveAntigravitySessionId(context);
|
||||
const step = state?.stepIndex ?? 2;
|
||||
const requestId = `agent/${agentId}/${Date.now()}/${trajectoryId}/${step}`;
|
||||
const isClaude = isClaudeModel(model.id);
|
||||
const profile = getAntigravityModelWireProfile(wireModelId);
|
||||
const labels: Record<string, string> = {};
|
||||
if (state?.lastExecutionId) labels.last_execution_id = state.lastExecutionId;
|
||||
labels.last_step_index = String(step - 1);
|
||||
if (profile) labels.model_enum = profile.modelEnum;
|
||||
labels.trajectory_id = trajectoryId;
|
||||
labels.used_claude = String(isClaude);
|
||||
labels.used_claude_conservative = String(isClaude);
|
||||
return { sessionId, requestId, labels };
|
||||
}
|
||||
|
||||
export function buildRequest(
|
||||
model: Model<"google-gemini-cli">,
|
||||
context: Context,
|
||||
@@ -937,19 +1005,26 @@ export function buildRequest(
|
||||
contents,
|
||||
};
|
||||
|
||||
if (isAntigravity) {
|
||||
request.sessionId = deriveAntigravitySessionId(context);
|
||||
}
|
||||
|
||||
// System instruction must be object with parts, not plain string
|
||||
// System instruction is an object with parts, not a plain string. Antigravity
|
||||
// tags it with role "user" to mirror the real client.
|
||||
if (systemPrompts.length > 0) {
|
||||
request.systemInstruction = {
|
||||
...(isAntigravity ? { role: "user" } : {}),
|
||||
parts: systemPrompts.map(text => ({ text })),
|
||||
};
|
||||
}
|
||||
|
||||
if (Object.keys(generationConfig).length > 0) {
|
||||
request.generationConfig = generationConfig;
|
||||
if (isAntigravity && shouldInjectAntigravitySystemInstruction(model.id)) {
|
||||
const existingParts = request.systemInstruction?.parts ?? [];
|
||||
request.systemInstruction = {
|
||||
role: "user",
|
||||
parts: [
|
||||
{ text: ANTIGRAVITY_SYSTEM_INSTRUCTION },
|
||||
{ text: `Please ignore following [ignore]${ANTIGRAVITY_SYSTEM_INSTRUCTION}[/ignore]` },
|
||||
{ text: ANTIGRAVITY_NO_PREAMBLE_INSTRUCTION },
|
||||
...existingParts,
|
||||
],
|
||||
};
|
||||
}
|
||||
|
||||
if (context.tools && context.tools.length > 0) {
|
||||
@@ -973,15 +1048,16 @@ export function buildRequest(
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (isAntigravity && !isClaudeModel(model.id) && request.generationConfig?.maxOutputTokens !== undefined) {
|
||||
delete request.generationConfig.maxOutputTokens;
|
||||
if (Object.keys(request.generationConfig).length === 0) {
|
||||
delete request.generationConfig;
|
||||
// Antigravity's default tool mode is VALIDATED (verified for Gemini and
|
||||
// Claude); an explicit non-auto tool choice above wins.
|
||||
if (isAntigravity && !request.toolConfig) {
|
||||
request.toolConfig = {
|
||||
functionCallingConfig: { mode: "VALIDATED" as FunctionCallingConfigMode },
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// Claude on Antigravity always forces VALIDATED, even with no tools declared.
|
||||
if (isAntigravity && isClaudeModel(model.id)) {
|
||||
request.toolConfig = {
|
||||
functionCallingConfig: {
|
||||
@@ -990,28 +1066,39 @@ export function buildRequest(
|
||||
};
|
||||
}
|
||||
|
||||
if (isAntigravity && shouldInjectAntigravitySystemInstruction(model.id)) {
|
||||
const existingParts = request.systemInstruction?.parts ?? [];
|
||||
request.systemInstruction = {
|
||||
parts: [
|
||||
{ text: ANTIGRAVITY_SYSTEM_INSTRUCTION },
|
||||
{ text: `Please ignore following [ignore]${ANTIGRAVITY_SYSTEM_INSTRUCTION}[/ignore]` },
|
||||
{ text: ANTIGRAVITY_NO_PREAMBLE_INSTRUCTION },
|
||||
...existingParts,
|
||||
],
|
||||
const wireModelId = options.requestModelId ?? model.requestModelId ?? model.id;
|
||||
|
||||
if (isAntigravity) {
|
||||
// The real client sends a fixed per-model output cap independent of the
|
||||
// thinking budget; reassign so it keeps its slot ahead of thinkingConfig.
|
||||
const profile = getAntigravityModelWireProfile(wireModelId);
|
||||
if (profile) {
|
||||
generationConfig.maxOutputTokens = profile.maxOutputTokens;
|
||||
}
|
||||
const state = getAntigravityProviderSessionState(options.providerSessionState);
|
||||
const envelope = buildAntigravityRequestEnvelope(model, context, wireModelId, state);
|
||||
request.labels = envelope.labels;
|
||||
if (Object.keys(generationConfig).length > 0) {
|
||||
request.generationConfig = generationConfig;
|
||||
}
|
||||
request.sessionId = envelope.sessionId;
|
||||
return {
|
||||
project: projectId,
|
||||
requestId: envelope.requestId,
|
||||
request,
|
||||
model: wireModelId,
|
||||
userAgent: "antigravity",
|
||||
requestType: "agent",
|
||||
};
|
||||
}
|
||||
|
||||
if (Object.keys(generationConfig).length > 0) {
|
||||
request.generationConfig = generationConfig;
|
||||
}
|
||||
|
||||
return {
|
||||
project: projectId,
|
||||
model: options.requestModelId ?? model.requestModelId ?? model.id,
|
||||
model: wireModelId,
|
||||
request,
|
||||
...(isAntigravity
|
||||
? {
|
||||
requestType: "agent",
|
||||
userAgent: "antigravity",
|
||||
requestId: `agent-${randomUUID()}`,
|
||||
}
|
||||
: {}),
|
||||
};
|
||||
}
|
||||
|
||||
@@ -16,7 +16,7 @@ const DEFAULT_ENDPOINT = "https://cloudcode-pa.googleapis.com";
|
||||
const GEMINI_TIER_MAP: Array<{ tier: string; models: string[] }> = [
|
||||
{
|
||||
tier: "3-Flash",
|
||||
models: ["gemini-3-flash-preview", "gemini-3-flash"],
|
||||
models: ["gemini-3-flash-preview", "gemini-3-flash", "gemini-3.5-flash"],
|
||||
},
|
||||
{
|
||||
tier: "Flash",
|
||||
@@ -24,7 +24,15 @@ const GEMINI_TIER_MAP: Array<{ tier: string; models: string[] }> = [
|
||||
},
|
||||
{
|
||||
tier: "Pro",
|
||||
models: ["gemini-2.5-pro", "gemini-3-pro-preview", "gemini-3.1-pro-preview", "gemini-3-pro", "gemini-1.5-pro"],
|
||||
models: [
|
||||
"gemini-2.5-pro",
|
||||
"gemini-3-pro-preview",
|
||||
"gemini-3.1-pro-preview",
|
||||
"gemini-3-pro",
|
||||
"gemini-3.1-pro",
|
||||
"gemini-pro-agent",
|
||||
"gemini-1.5-pro",
|
||||
],
|
||||
},
|
||||
];
|
||||
|
||||
|
||||
@@ -180,7 +180,11 @@ describe("Google Gemini CLI alignment", () => {
|
||||
it("keeps antigravity metadata in antigravity request payloads", () => {
|
||||
const model = createModel("google-antigravity");
|
||||
const payload = buildRequest(model, createContext(), "proj-123", {}, true) as {
|
||||
request: { sessionId?: string };
|
||||
request: {
|
||||
sessionId?: string;
|
||||
labels?: Record<string, string>;
|
||||
systemInstruction?: { role?: string };
|
||||
};
|
||||
requestType?: string;
|
||||
userAgent?: string;
|
||||
requestId?: string;
|
||||
@@ -189,38 +193,74 @@ describe("Google Gemini CLI alignment", () => {
|
||||
expect(payload.request.sessionId).toMatch(/^-[0-9]+$/);
|
||||
expect(payload.requestType).toBe("agent");
|
||||
expect(payload.userAgent).toBe("antigravity");
|
||||
expect(payload.requestId).toMatch(/^agent-/);
|
||||
// Structured requestId: agent/<agentId>/<ts>/<trajectoryId>/<step>.
|
||||
expect(payload.requestId).toMatch(/^agent\/[0-9a-f-]+\/\d+\/[0-9a-f-]+\/\d+$/);
|
||||
// Antigravity tags its system instruction with role "user".
|
||||
expect(payload.request.systemInstruction?.role).toBe("user");
|
||||
const labels = payload.request.labels;
|
||||
expect(labels?.trajectory_id).toMatch(/^[0-9a-f-]+$/);
|
||||
expect(labels?.last_step_index).toBe("1");
|
||||
expect(labels?.used_claude).toBe("false");
|
||||
expect(labels?.used_claude_conservative).toBe("false");
|
||||
});
|
||||
it("omits AUTO toolConfig for Google Gemini CLI tool calls", () => {
|
||||
for (const provider of ["google-gemini-cli", "google-antigravity"] as const) {
|
||||
const model = createModel(provider);
|
||||
const context: Context = {
|
||||
messages: [{ role: "user", content: "inspect repo", timestamp: Date.now() }],
|
||||
tools: [
|
||||
{
|
||||
name: "read_file",
|
||||
description: "Read a file",
|
||||
parameters: {
|
||||
type: "object",
|
||||
properties: { path: { type: "string" } },
|
||||
required: ["path"],
|
||||
} as TJsonSchema,
|
||||
},
|
||||
],
|
||||
};
|
||||
const payload = buildRequest(
|
||||
model,
|
||||
context,
|
||||
"proj-123",
|
||||
{ toolChoice: "auto" },
|
||||
provider === "google-antigravity",
|
||||
) as {
|
||||
request: { tools?: unknown; toolConfig?: unknown };
|
||||
};
|
||||
|
||||
expect(payload.request.tools).toBeDefined();
|
||||
expect(payload.request.toolConfig).toBeUndefined();
|
||||
}
|
||||
it("stamps the antigravity wire profile (maxOutputTokens + model_enum) by routed wire id", () => {
|
||||
const model = createModel("google-antigravity");
|
||||
const payload = buildRequest(
|
||||
model,
|
||||
createContext(),
|
||||
"proj-123",
|
||||
{ requestModelId: "gemini-3.5-flash-low" },
|
||||
true,
|
||||
) as {
|
||||
model?: string;
|
||||
request: { generationConfig?: { maxOutputTokens?: number }; labels?: Record<string, string> };
|
||||
};
|
||||
|
||||
expect(payload.model).toBe("gemini-3.5-flash-low");
|
||||
expect(payload.request.generationConfig?.maxOutputTokens).toBe(65536);
|
||||
expect(payload.request.labels?.model_enum).toBe("MODEL_PLACEHOLDER_M20");
|
||||
});
|
||||
|
||||
it("defaults antigravity tools to VALIDATED but omits AUTO toolConfig for plain gemini-cli", () => {
|
||||
const context: Context = {
|
||||
messages: [{ role: "user", content: "inspect repo", timestamp: Date.now() }],
|
||||
tools: [
|
||||
{
|
||||
name: "read_file",
|
||||
description: "Read a file",
|
||||
parameters: {
|
||||
type: "object",
|
||||
properties: { path: { type: "string" } },
|
||||
required: ["path"],
|
||||
} as TJsonSchema,
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
const cli = buildRequest(
|
||||
createModel("google-gemini-cli"),
|
||||
context,
|
||||
"proj-123",
|
||||
{ toolChoice: "auto" },
|
||||
false,
|
||||
) as {
|
||||
request: { tools?: unknown; toolConfig?: unknown };
|
||||
};
|
||||
expect(cli.request.tools).toBeDefined();
|
||||
expect(cli.request.toolConfig).toBeUndefined();
|
||||
|
||||
const antigravity = buildRequest(
|
||||
createModel("google-antigravity"),
|
||||
context,
|
||||
"proj-123",
|
||||
{ toolChoice: "auto" },
|
||||
true,
|
||||
) as {
|
||||
request: { tools?: unknown; toolConfig?: { functionCallingConfig: { mode: string } } };
|
||||
};
|
||||
expect(antigravity.request.tools).toBeDefined();
|
||||
expect(antigravity.request.toolConfig).toEqual({ functionCallingConfig: { mode: "VALIDATED" } });
|
||||
});
|
||||
|
||||
it("strips patternProperties when antigravity rewrites tools to legacy parameters", () => {
|
||||
@@ -306,6 +346,23 @@ describe("Google Gemini CLI alignment", () => {
|
||||
expect(requestHeaders!.get("Client-Metadata")).toBeNull();
|
||||
});
|
||||
|
||||
it("sends the antigravity/hub User-Agent header on the Antigravity transport", async () => {
|
||||
let requestHeaders: Headers | undefined;
|
||||
const fetchMock: FetchImpl = async (_url, init) => {
|
||||
requestHeaders = new Headers(init?.headers);
|
||||
return new Response('{"error":{"message":"bad request"}}', { status: 400 });
|
||||
};
|
||||
|
||||
const model = createModel("google-antigravity");
|
||||
await streamGoogleGeminiCli(model, createContext(), {
|
||||
apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }),
|
||||
fetch: fetchMock,
|
||||
}).result();
|
||||
|
||||
expect(requestHeaders).toBeDefined();
|
||||
expect(requestHeaders!.get("User-Agent")).toMatch(/^antigravity\/hub\/[0-9.]+ /);
|
||||
});
|
||||
|
||||
it("filters out empty text parts at stream end but preserves terminal thought signatures", async () => {
|
||||
const sseChunks = [
|
||||
'data: {"response":{"candidates":[{"content":{"role":"model","parts":[{"text":"Hello"}]}}]}}\n\n',
|
||||
|
||||
@@ -9,6 +9,7 @@ interface CapturedRequestBody {
|
||||
model?: string;
|
||||
request?: {
|
||||
generationConfig?: {
|
||||
maxOutputTokens?: number;
|
||||
thinkingConfig?: {
|
||||
includeThoughts?: boolean;
|
||||
thinkingLevel?: string;
|
||||
@@ -32,21 +33,27 @@ function collapsedFlashModel(): Model<"google-gemini-cli"> {
|
||||
baseUrl: "https://daily-cloudcode-pa.googleapis.com",
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "google-level",
|
||||
mode: "budget",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
effortBudgets: {
|
||||
[Effort.Minimal]: 1000,
|
||||
[Effort.Low]: 1000,
|
||||
[Effort.Medium]: 4000,
|
||||
[Effort.High]: 10000,
|
||||
},
|
||||
effortRouting: {
|
||||
off: "gemini-3.5-flash-extra-low",
|
||||
[Effort.Minimal]: "gemini-3-flash-agent",
|
||||
[Effort.Minimal]: "gemini-3.5-flash-extra-low",
|
||||
[Effort.Low]: "gemini-3.5-flash-extra-low",
|
||||
[Effort.Medium]: "gemini-3.5-flash-extra-low",
|
||||
[Effort.High]: "gemini-3.5-flash-low",
|
||||
[Effort.Medium]: "gemini-3.5-flash-low",
|
||||
[Effort.High]: "gemini-3-flash-agent",
|
||||
},
|
||||
suppressWhenOff: true,
|
||||
},
|
||||
input: ["text", "image"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 1_048_576,
|
||||
maxTokens: 65_535,
|
||||
maxTokens: 65_536,
|
||||
} satisfies ModelSpec<"google-gemini-cli">);
|
||||
}
|
||||
|
||||
@@ -113,26 +120,31 @@ async function captureRequest(
|
||||
}
|
||||
|
||||
describe("google-gemini-cli effort-tier variant routing", () => {
|
||||
it("routes each effort to its backing wire id and attributes usage to the logical id", async () => {
|
||||
it("routes each effort to its backing wire id with the per-tier budget and attributes usage to the logical id", async () => {
|
||||
const high = await captureRequest(collapsedFlashModel(), Effort.High);
|
||||
expect(high.body.model).toBe("gemini-3.5-flash-low");
|
||||
expect(high.body.model).toBe("gemini-3-flash-agent");
|
||||
expect(high.body.request?.generationConfig?.thinkingConfig).toEqual({
|
||||
includeThoughts: true,
|
||||
thinkingLevel: "HIGH",
|
||||
thinkingBudget: 10000,
|
||||
});
|
||||
expect(high.body.request?.generationConfig?.maxOutputTokens).toBe(65536);
|
||||
expect(high.attributedModel).toBe("gemini-3.5-flash");
|
||||
|
||||
const minimal = await captureRequest(collapsedFlashModel(), Effort.Minimal);
|
||||
expect(minimal.body.model).toBe("gemini-3-flash-agent");
|
||||
expect(minimal.body.request?.generationConfig?.thinkingConfig?.thinkingLevel).toBe("MINIMAL");
|
||||
const medium = await captureRequest(collapsedFlashModel(), Effort.Medium);
|
||||
expect(medium.body.model).toBe("gemini-3.5-flash-low");
|
||||
expect(medium.body.request?.generationConfig?.thinkingConfig?.thinkingBudget).toBe(4000);
|
||||
|
||||
const low = await captureRequest(collapsedFlashModel(), Effort.Low);
|
||||
expect(low.body.model).toBe("gemini-3.5-flash-extra-low");
|
||||
expect(low.body.request?.generationConfig?.thinkingConfig?.thinkingBudget).toBe(1000);
|
||||
});
|
||||
|
||||
it("suppresses thinking explicitly on the wire when off and suppressWhenOff is set", async () => {
|
||||
it("suppresses thinking with a zero budget on the wire when off and suppressWhenOff is set", async () => {
|
||||
const off = await captureRequest(collapsedFlashModel(), undefined);
|
||||
expect(off.body.model).toBe("gemini-3.5-flash-extra-low");
|
||||
expect(off.body.request?.generationConfig?.thinkingConfig).toEqual({
|
||||
includeThoughts: false,
|
||||
thinkingLevel: "MINIMAL",
|
||||
thinkingBudget: 0,
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
@@ -1,18 +1,25 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added `enableGeminiThinkingLoopGuard` to OpenAI compatibility options to allow explicit opt-in or opt-out of the Gemini thinking-loop guard for OpenAI-compatible model aliases
|
||||
- Added `LITELLM_BASE_URL` as the LiteLLM provider discovery base URL fallback, with discovery caches scoped by the resolved proxy URL and explicit provider `baseUrl` config kept at higher precedence. ([#2726](https://github.com/can1357/oh-my-pi/issues/2726))
|
||||
- Added `ThinkingConfig.effortBudgets` (per-effort thinking-budget contract baked into collapsed variants) and `ANTIGRAVITY_MODEL_WIRE_PROFILES` (`maxOutputTokens` + `model_enum` per Antigravity wire id) to mirror the captured Antigravity Cloud Code Assist client request shape.
|
||||
|
||||
### Changed
|
||||
|
||||
- Defaulted `enableGeminiThinkingLoopGuard` from Gemini family detection for both OpenAI completions and responses compatibility specs so Gemini models now enable the thinking-loop guard automatically
|
||||
- Updated the default Gemini CLI user-agent version fallback to 0.46.0.
|
||||
- Changed the Antigravity (`google-antigravity`, daily-cloudcode-pa) gemini-3.x collapse families to the `budget` thinking transport with the client's per-tier `thinkingBudget` (3.5 Flash low/medium/high = 1000/4000/10000, 3.1 Pro low/high = 1001/10001) and corrected 3.5 Flash effort→wire routing (medium → `gemini-3.5-flash-low`, high → `gemini-3-flash-agent`). Split the shared CCA collapse table so `google-gemini-cli` (cloudcode-pa) keeps the `google-level` `thinkingLevel` transport for official Gemini CLI parity. Stale collapsed snapshots (bundled catalog, recycled `gemini-3-flash` alias) self-heal from the hand table at collapse time, and the model cache schema is bumped to v7 to invalidate pre-budget Antigravity rows.
|
||||
- Changed the Antigravity user-agent to the `antigravity/hub/<version>` format (default `2.1.4`) to match the captured client.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed `off` effort routing for `claude-opus-4-5` and `claude-opus-4-6` to use their base model IDs when thinking is disabled
|
||||
- Fixed `gemini-2.5-flash` effort routing so all non-off effort levels resolve to `gemini-2.5-flash-thinking`
|
||||
- Fixed shared variant alias provider resolution so `resolveBareVariantAlias` reports all matching providers when model aliases are present in both CCA collapse tables
|
||||
- Routed google-antigravity default baseUrl to the stable primary daily endpoint in the catalog generator and all fallback snapshots, resolving connection drops on heavy queries.
|
||||
|
||||
## [16.0.4] - 2026-06-17
|
||||
|
||||
@@ -1,7 +1,11 @@
|
||||
import { z } from "zod/v4";
|
||||
import type { ModelSpec } from "../types";
|
||||
import { toPositiveNumber } from "../utils";
|
||||
import { ANTIGRAVITY_VARIANT_COLLAPSE_TABLE, collapseEffortVariants } from "../variant-collapse";
|
||||
import {
|
||||
ANTIGRAVITY_VARIANT_COLLAPSE_TABLE,
|
||||
collapseEffortVariants,
|
||||
type VariantCollapseTable,
|
||||
} from "../variant-collapse";
|
||||
import { getAntigravityUserAgent } from "../wire/gemini-headers";
|
||||
|
||||
export const ANTIGRAVITY_PRIMARY_ENDPOINT = "https://daily-cloudcode-pa.googleapis.com";
|
||||
@@ -156,6 +160,12 @@ export interface FetchAntigravityDiscoveryModelsOptions {
|
||||
signal?: AbortSignal;
|
||||
/** Optional fetch implementation override for tests. */
|
||||
fetcher?: typeof fetch;
|
||||
/**
|
||||
* Hand collapse table to apply to the discovered list. Defaults to the
|
||||
* Antigravity (budget-transport) table; `googleGeminiCli` passes the
|
||||
* level-transport table so cloudcode-pa keeps `thinkingLevel`.
|
||||
*/
|
||||
collapseTable?: VariantCollapseTable;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -238,7 +248,7 @@ export async function fetchAntigravityDiscoveryModels(
|
||||
// Collapse effort-tier variants at the source so runtime discovery,
|
||||
// the gemini-cli re-provision, and the catalog generator all see
|
||||
// logical ids only.
|
||||
const collapsed = collapseEffortVariants(models, ANTIGRAVITY_VARIANT_COLLAPSE_TABLE);
|
||||
const collapsed = collapseEffortVariants(models, options.collapseTable ?? ANTIGRAVITY_VARIANT_COLLAPSE_TABLE);
|
||||
collapsed.sort((a, b) => a.name.localeCompare(b.name) || a.id.localeCompare(b.id));
|
||||
return collapsed;
|
||||
}
|
||||
|
||||
@@ -7,12 +7,14 @@ import { getModelDbPath } from "@oh-my-pi/pi-utils";
|
||||
import type { Api, Model, ModelSpec } from "./types";
|
||||
|
||||
// Rows persist ModelSpec JSON (sparse `compat`, never the resolved record);
|
||||
// the model manager rebuilds via `buildModel` on load. v6 invalidates rows
|
||||
// that may contain the retired unknown-limit sentinels (222222/8888); v5
|
||||
// invalidated rows predating effort-tier variant collapsing (raw
|
||||
// `-low`/`-high`/`-thinking` member ids); v4 dropped the pre-efforts
|
||||
// ThinkingConfig shape.
|
||||
const CACHE_SCHEMA_VERSION = 6;
|
||||
// the model manager rebuilds via `buildModel` on load. v7 invalidates rows
|
||||
// predating the Antigravity Gemini budget-mode migration (cached specs still
|
||||
// carrying `thinking.mode: "google-level"` and the old 3.5-flash effort
|
||||
// routing); v6 invalidates rows that may contain the retired unknown-limit
|
||||
// sentinels (222222/8888); v5 invalidated rows predating effort-tier variant
|
||||
// collapsing (raw `-low`/`-high`/`-thinking` member ids); v4 dropped the
|
||||
// pre-efforts ThinkingConfig shape.
|
||||
const CACHE_SCHEMA_VERSION = 7;
|
||||
|
||||
interface CacheRow {
|
||||
provider_id: string;
|
||||
|
||||
@@ -17749,6 +17749,7 @@
|
||||
"high"
|
||||
],
|
||||
"effortRouting": {
|
||||
"off": "claude-opus-4-5",
|
||||
"minimal": "claude-opus-4-5-thinking",
|
||||
"low": "claude-opus-4-5-thinking",
|
||||
"medium": "claude-opus-4-5-thinking",
|
||||
@@ -17785,6 +17786,7 @@
|
||||
"high"
|
||||
],
|
||||
"effortRouting": {
|
||||
"off": "claude-opus-4-6",
|
||||
"minimal": "claude-opus-4-6-thinking",
|
||||
"low": "claude-opus-4-6-thinking",
|
||||
"medium": "claude-opus-4-6-thinking",
|
||||
@@ -17953,15 +17955,29 @@
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 65536,
|
||||
"thinking": {
|
||||
"mode": "google-level",
|
||||
"mode": "budget",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"requiresEffort": true
|
||||
}
|
||||
"effortBudgets": {
|
||||
"minimal": 1000,
|
||||
"low": 1000,
|
||||
"medium": 4000,
|
||||
"high": 10000
|
||||
},
|
||||
"effortRouting": {
|
||||
"off": "gemini-3.5-flash-extra-low",
|
||||
"minimal": "gemini-3.5-flash-extra-low",
|
||||
"low": "gemini-3.5-flash-extra-low",
|
||||
"medium": "gemini-3.5-flash-low",
|
||||
"high": "gemini-3-flash-agent"
|
||||
},
|
||||
"suppressWhenOff": true
|
||||
},
|
||||
"requestModelId": "gemini-3.5-flash-extra-low"
|
||||
},
|
||||
"gemini-3-pro": {
|
||||
"id": "gemini-3-pro",
|
||||
@@ -18017,11 +18033,15 @@
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 65535,
|
||||
"thinking": {
|
||||
"mode": "google-level",
|
||||
"mode": "budget",
|
||||
"efforts": [
|
||||
"low",
|
||||
"high"
|
||||
],
|
||||
"effortBudgets": {
|
||||
"low": 1001,
|
||||
"high": 10001
|
||||
},
|
||||
"effortRouting": {
|
||||
"off": "gemini-3.1-pro-low",
|
||||
"low": "gemini-3.1-pro-low",
|
||||
@@ -18110,7 +18130,11 @@
|
||||
"high"
|
||||
],
|
||||
"effortRouting": {
|
||||
"off": "gemini-2.5-flash"
|
||||
"off": "gemini-2.5-flash",
|
||||
"minimal": "gemini-2.5-flash-thinking",
|
||||
"low": "gemini-2.5-flash-thinking",
|
||||
"medium": "gemini-2.5-flash-thinking",
|
||||
"high": "gemini-2.5-flash-thinking"
|
||||
}
|
||||
}
|
||||
},
|
||||
|
||||
@@ -2,6 +2,7 @@ import { fetchAntigravityDiscoveryModels } from "../discovery/antigravity";
|
||||
import { fetchGeminiModels } from "../discovery/gemini";
|
||||
import type { ModelManagerOptions } from "../model-manager";
|
||||
import type { FetchImpl } from "../types";
|
||||
import { GEMINI_CLI_VARIANT_COLLAPSE_TABLE } from "../variant-collapse";
|
||||
|
||||
export interface GoogleModelManagerConfig {
|
||||
apiKey?: string;
|
||||
@@ -89,6 +90,7 @@ export function googleGeminiCliModelManagerOptions(
|
||||
token,
|
||||
endpoint,
|
||||
fetcher: toDiscoveryFetch(config?.fetch),
|
||||
collapseTable: GEMINI_CLI_VARIANT_COLLAPSE_TABLE,
|
||||
});
|
||||
if (models === null) {
|
||||
return null;
|
||||
|
||||
@@ -124,86 +124,115 @@ const GEMINI_3_PRO_FAMILY_BUDGETS: Readonly<Partial<Record<Effort, number>>> = {
|
||||
};
|
||||
|
||||
/**
|
||||
* Shared by `google-antigravity` and `google-gemini-cli` — both serve the
|
||||
* Antigravity discovery list (`fetchAntigravityDiscoveryModels`).
|
||||
* The two Cloud Code Assist providers share the same Antigravity discovery list
|
||||
* but disagree on the thinking transport: `google-antigravity` (daily-cloudcode-pa)
|
||||
* sends an explicit `thinkingBudget` (verified against captured requests), while
|
||||
* `google-gemini-cli` (cloudcode-pa) follows the official Gemini CLI and uses
|
||||
* `thinkingLevel`. The Gemini 3.x families therefore differ only in thinking
|
||||
* transport (and, for Flash, the per-tier wire-id routing); everything else is
|
||||
* shared verbatim.
|
||||
*/
|
||||
function geminiFlashFamily(mode: "budget" | "google-level"): EffortVariantFamily {
|
||||
const budget = mode === "budget";
|
||||
return {
|
||||
id: "gemini-3.5-flash",
|
||||
name: "Gemini 3.5 Flash",
|
||||
members: ["gemini-3.5-flash-extra-low", "gemini-3.5-flash-low", "gemini-3-flash-agent"],
|
||||
routing: budget
|
||||
? {
|
||||
off: "gemini-3.5-flash-extra-low",
|
||||
[Effort.Minimal]: "gemini-3.5-flash-extra-low",
|
||||
[Effort.Low]: "gemini-3.5-flash-extra-low",
|
||||
[Effort.Medium]: "gemini-3.5-flash-low",
|
||||
[Effort.High]: "gemini-3-flash-agent",
|
||||
}
|
||||
: {
|
||||
off: "gemini-3.5-flash-extra-low",
|
||||
[Effort.Minimal]: "gemini-3-flash-agent",
|
||||
[Effort.Low]: "gemini-3.5-flash-extra-low",
|
||||
[Effort.Medium]: "gemini-3.5-flash-extra-low",
|
||||
[Effort.High]: "gemini-3.5-flash-low",
|
||||
},
|
||||
thinking: budget
|
||||
? { mode: "budget", efforts: GEMINI_3_FLASH_FAMILY_EFFORTS, effortBudgets: GEMINI_3_FLASH_FAMILY_BUDGETS }
|
||||
: { mode: "google-level", efforts: GEMINI_3_FLASH_FAMILY_EFFORTS },
|
||||
suppressWhenOff: true,
|
||||
// Retired bare id; the alias only fires when no live model holds it
|
||||
// (exact match wins in every resolver).
|
||||
extraAliases: ["gemini-3-flash"],
|
||||
};
|
||||
}
|
||||
|
||||
function geminiProFamily(mode: "budget" | "google-level"): EffortVariantFamily {
|
||||
const budget = mode === "budget";
|
||||
return {
|
||||
id: "gemini-3.1-pro",
|
||||
name: "Gemini 3.1 Pro",
|
||||
// High routes to `gemini-pro-agent` — the upstream `gemini-3.1-pro-high`
|
||||
// deployment returns INVALID_ARGUMENT on every streamGenerateContent
|
||||
// request (both CCA endpoints) while discovery still lists it;
|
||||
// `gemini-pro-agent` is the same model ("Gemini 3.1 Pro (High)", same
|
||||
// thinking budget/caps) and accepts the identical request body.
|
||||
// `gemini-3.1-pro-high` stays a member so the dead raw id is consumed.
|
||||
members: ["gemini-3.1-pro-low", "gemini-pro-agent", "gemini-3.1-pro-high"],
|
||||
retiredMembers: ["gemini-3.1-pro-high"],
|
||||
routing: {
|
||||
off: "gemini-3.1-pro-low",
|
||||
[Effort.Low]: "gemini-3.1-pro-low",
|
||||
[Effort.High]: "gemini-pro-agent",
|
||||
},
|
||||
thinking: budget
|
||||
? { mode: "budget", efforts: GEMINI_3_PRO_FAMILY_EFFORTS, effortBudgets: GEMINI_3_PRO_FAMILY_BUDGETS }
|
||||
: { mode: "google-level", efforts: GEMINI_3_PRO_FAMILY_EFFORTS },
|
||||
suppressWhenOff: true,
|
||||
};
|
||||
}
|
||||
|
||||
/** CCA families shared verbatim by both providers (transport-agnostic). */
|
||||
const SHARED_CCA_FAMILIES: readonly EffortVariantFamily[] = [
|
||||
{
|
||||
// Legacy static family — covers stale snapshots and caches. Stale ids are
|
||||
// unverified against the budget-mode CCA contract; keep them on level.
|
||||
id: "gemini-3-pro",
|
||||
name: "Gemini 3 Pro",
|
||||
members: ["gemini-3-pro-low", "gemini-3-pro-high"],
|
||||
routing: {
|
||||
off: "gemini-3-pro-low",
|
||||
[Effort.Low]: "gemini-3-pro-low",
|
||||
[Effort.High]: "gemini-3-pro-high",
|
||||
},
|
||||
thinking: { mode: "google-level", efforts: GEMINI_3_PRO_FAMILY_EFFORTS },
|
||||
suppressWhenOff: true,
|
||||
},
|
||||
{
|
||||
// Rename-only collapse: every effort and off fall back to the wire id.
|
||||
id: "gpt-oss-120b",
|
||||
name: "GPT-OSS 120B",
|
||||
members: ["gpt-oss-120b-medium"],
|
||||
routing: {},
|
||||
thinking: { mode: "budget", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
|
||||
},
|
||||
thinkingPair("claude-sonnet-4-6", "Claude Sonnet 4.6"),
|
||||
thinkingPair("claude-opus-4-6", "Claude Opus 4.6"),
|
||||
thinkingPair("claude-sonnet-4-5", "Claude Sonnet 4.5"),
|
||||
thinkingPair("claude-opus-4-5", "Claude Opus 4.5"),
|
||||
thinkingPair("gemini-2.5-flash", "Gemini 2.5 Flash"),
|
||||
];
|
||||
|
||||
/** `google-antigravity` (daily-cloudcode-pa): Gemini 3.x on the budget transport. */
|
||||
export const ANTIGRAVITY_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
|
||||
families: [
|
||||
{
|
||||
id: "gemini-3.5-flash",
|
||||
name: "Gemini 3.5 Flash",
|
||||
members: ["gemini-3.5-flash-extra-low", "gemini-3.5-flash-low", "gemini-3-flash-agent"],
|
||||
routing: {
|
||||
off: "gemini-3.5-flash-extra-low",
|
||||
[Effort.Minimal]: "gemini-3.5-flash-extra-low",
|
||||
[Effort.Low]: "gemini-3.5-flash-extra-low",
|
||||
[Effort.Medium]: "gemini-3.5-flash-low",
|
||||
[Effort.High]: "gemini-3-flash-agent",
|
||||
},
|
||||
thinking: {
|
||||
mode: "budget",
|
||||
efforts: GEMINI_3_FLASH_FAMILY_EFFORTS,
|
||||
effortBudgets: GEMINI_3_FLASH_FAMILY_BUDGETS,
|
||||
},
|
||||
suppressWhenOff: true,
|
||||
// Retired bare id; the alias only fires when no live model holds it
|
||||
// (exact match wins in every resolver).
|
||||
extraAliases: ["gemini-3-flash"],
|
||||
},
|
||||
{
|
||||
id: "gemini-3.1-pro",
|
||||
name: "Gemini 3.1 Pro",
|
||||
// High routes to `gemini-pro-agent` — the upstream `gemini-3.1-pro-high`
|
||||
// deployment returns INVALID_ARGUMENT on every streamGenerateContent
|
||||
// request (both CCA endpoints) while discovery still lists it;
|
||||
// `gemini-pro-agent` is the same model ("Gemini 3.1 Pro (High)", same
|
||||
// thinking budget/caps) and accepts the identical request body.
|
||||
// `gemini-3.1-pro-high` stays a member so the dead raw id is consumed.
|
||||
members: ["gemini-3.1-pro-low", "gemini-pro-agent", "gemini-3.1-pro-high"],
|
||||
retiredMembers: ["gemini-3.1-pro-high"],
|
||||
routing: {
|
||||
off: "gemini-3.1-pro-low",
|
||||
[Effort.Low]: "gemini-3.1-pro-low",
|
||||
[Effort.High]: "gemini-pro-agent",
|
||||
},
|
||||
thinking: { mode: "budget", efforts: GEMINI_3_PRO_FAMILY_EFFORTS, effortBudgets: GEMINI_3_PRO_FAMILY_BUDGETS },
|
||||
suppressWhenOff: true,
|
||||
},
|
||||
{
|
||||
// Legacy static family — covers stale snapshots and caches.
|
||||
id: "gemini-3-pro",
|
||||
name: "Gemini 3 Pro",
|
||||
members: ["gemini-3-pro-low", "gemini-3-pro-high"],
|
||||
routing: {
|
||||
off: "gemini-3-pro-low",
|
||||
[Effort.Low]: "gemini-3-pro-low",
|
||||
[Effort.High]: "gemini-3-pro-high",
|
||||
},
|
||||
// Legacy ids are stale-cache only and unverified against the budget-mode
|
||||
// CCA contract; keep them on the inferred level transport.
|
||||
thinking: { mode: "google-level", efforts: GEMINI_3_PRO_FAMILY_EFFORTS },
|
||||
suppressWhenOff: true,
|
||||
},
|
||||
{
|
||||
// Rename-only collapse: every effort and off fall back to the wire id.
|
||||
id: "gpt-oss-120b",
|
||||
name: "GPT-OSS 120B",
|
||||
members: ["gpt-oss-120b-medium"],
|
||||
routing: {},
|
||||
thinking: { mode: "budget", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
|
||||
},
|
||||
thinkingPair("claude-sonnet-4-6", "Claude Sonnet 4.6"),
|
||||
thinkingPair("claude-opus-4-6", "Claude Opus 4.6"),
|
||||
thinkingPair("claude-sonnet-4-5", "Claude Sonnet 4.5"),
|
||||
thinkingPair("claude-opus-4-5", "Claude Opus 4.5"),
|
||||
thinkingPair("gemini-2.5-flash", "Gemini 2.5 Flash"),
|
||||
],
|
||||
families: [geminiFlashFamily("budget"), geminiProFamily("budget"), ...SHARED_CCA_FAMILIES],
|
||||
};
|
||||
|
||||
/** Provider id → hand collapse table. Both CCA providers share one table. */
|
||||
/** `google-gemini-cli` (cloudcode-pa): Gemini 3.x on the level transport (official CLI parity). */
|
||||
export const GEMINI_CLI_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
|
||||
families: [geminiFlashFamily("google-level"), geminiProFamily("google-level"), ...SHARED_CCA_FAMILIES],
|
||||
};
|
||||
|
||||
/** Provider id → hand collapse table. The CCA providers diverge on thinking transport. */
|
||||
export const VARIANT_COLLAPSE_TABLES: Readonly<Record<string, VariantCollapseTable>> = {
|
||||
"google-antigravity": ANTIGRAVITY_VARIANT_COLLAPSE_TABLE,
|
||||
"google-gemini-cli": ANTIGRAVITY_VARIANT_COLLAPSE_TABLE,
|
||||
"google-gemini-cli": GEMINI_CLI_VARIANT_COLLAPSE_TABLE,
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -383,6 +412,47 @@ function reconcileRetiredRouting<TSpec extends VariantSpecLike>(
|
||||
return next;
|
||||
}
|
||||
|
||||
/**
|
||||
* Refresh a collapsed snapshot's thinking surface in place. Bundled catalog and
|
||||
* prev-generation snapshots freeze a family's transport, budgets, and routing;
|
||||
* discovery emits the canonical id but the exact-id merge never overwrites a
|
||||
* stale `family.id` row (e.g. `gemini-3.1-pro`) nor a recycled `extraAliases`
|
||||
* row (e.g. `gemini-3-flash`). This re-applies the hand-table family's thinking,
|
||||
* routing, and default wire id while keeping the spec id (load-bearing for exact
|
||||
* selectors and bundled lookups). Returns `spec` by reference when unchanged.
|
||||
*/
|
||||
function refreshCollapsedThinking<TSpec extends VariantSpecLike>(
|
||||
spec: TSpec,
|
||||
family: EffortVariantFamily,
|
||||
retired: ReadonlySet<string> | undefined,
|
||||
): TSpec {
|
||||
// Scope snapshot self-heal to families carrying a curated per-effort budget
|
||||
// contract (Antigravity gemini-3.x). Their routing targets are all verified
|
||||
// live, so rebuilding routing here is safe; families without `effortBudgets`
|
||||
// (derived `X`/`X-thinking` pairs, claude pairs) keep their presence-filtered
|
||||
// snapshot routing untouched.
|
||||
if (!spec.reasoning || family.thinking.effortBudgets === undefined) return spec;
|
||||
const routing: Partial<Record<Effort | "off", string>> = {};
|
||||
let hasRouting = false;
|
||||
for (const effortKey in family.routing) {
|
||||
const target = family.routing[effortKey as Effort | "off"];
|
||||
if (target !== undefined && !retired?.has(target)) {
|
||||
routing[effortKey as Effort | "off"] = target;
|
||||
hasRouting = true;
|
||||
}
|
||||
}
|
||||
const thinking: ThinkingConfig = { ...family.thinking };
|
||||
if (hasRouting) thinking.effortRouting = routing;
|
||||
if (family.suppressWhenOff) thinking.suppressWhenOff = true;
|
||||
const offTarget = family.routing.off;
|
||||
const requestModelId =
|
||||
offTarget !== undefined && !retired?.has(offTarget) && offTarget !== spec.id ? offTarget : spec.requestModelId;
|
||||
if (Bun.deepEquals(thinking, spec.thinking) && requestModelId === spec.requestModelId) {
|
||||
return spec;
|
||||
}
|
||||
return { ...spec, thinking, ...(requestModelId !== undefined ? { requestModelId } : {}) };
|
||||
}
|
||||
|
||||
/**
|
||||
* Collapse every family in `table` found in `specs`. Non-member specs pass
|
||||
* through verbatim (by reference), order preserved; the collapsed spec
|
||||
@@ -417,11 +487,17 @@ export function collapseEffortVariants<TSpec extends VariantSpecLike>(
|
||||
: existing;
|
||||
const rawPresent = family.members.filter(id => byId.has(id) && !(id === family.id && existingCollapsed));
|
||||
if (rawPresent.length === 0) {
|
||||
// Inert (no members) or already collapsed (pass-through) — idempotence.
|
||||
// A stale collapsed entry still gets retired routing re-pointed.
|
||||
if (reconciled !== undefined && reconciled !== existing) {
|
||||
// Inert (no members) or already collapsed (pass-through). A stale
|
||||
// family.id-keyed snapshot is refreshed in place from the current
|
||||
// hand-table family (transport/budgets/routing); retired targets drop.
|
||||
// Recycled extraAliases rows are healed in a later pass.
|
||||
const refreshed =
|
||||
existing !== undefined && existingCollapsed
|
||||
? refreshCollapsedThinking(reconciled ?? existing, family, retired)
|
||||
: reconciled;
|
||||
if (refreshed !== undefined && refreshed !== existing) {
|
||||
familyIdBySpecId.set(family.id, family.id);
|
||||
replacement.set(family.id, reconciled);
|
||||
replacement.set(family.id, refreshed);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
@@ -487,6 +563,27 @@ export function collapseEffortVariants<TSpec extends VariantSpecLike>(
|
||||
replacement.set(family.id, collapsed);
|
||||
}
|
||||
|
||||
// Refresh stale alias-keyed snapshots in place (recycled bare ids). Runs even
|
||||
// when the canonical family.id row is also present, since the exact-id merge
|
||||
// keeps the stale alias row alongside the discovered canonical one.
|
||||
for (const family of table.families) {
|
||||
if (family.extraAliases === undefined) continue;
|
||||
const retired =
|
||||
family.retiredMembers !== undefined && family.retiredMembers.length > 0
|
||||
? new Set(family.retiredMembers)
|
||||
: undefined;
|
||||
for (const alias of family.extraAliases) {
|
||||
if (alias === family.id || familyIdBySpecId.has(alias)) continue;
|
||||
const aliasSpec = byId.get(alias);
|
||||
if (aliasSpec === undefined) continue;
|
||||
const refreshed = refreshCollapsedThinking(aliasSpec, family, retired);
|
||||
if (refreshed !== aliasSpec) {
|
||||
familyIdBySpecId.set(alias, alias);
|
||||
replacement.set(alias, refreshed);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (replacement.size === 0) return [...specs];
|
||||
|
||||
const emitted = new Set<string>();
|
||||
@@ -625,7 +722,13 @@ export function resolveBareVariantAlias(modelId: string): BareVariantAliasHit |
|
||||
if (hit === undefined) continue;
|
||||
const providers: Provider[] = [];
|
||||
for (const candidate in VARIANT_COLLAPSE_TABLES) {
|
||||
if (VARIANT_COLLAPSE_TABLES[candidate] === table) providers.push(candidate);
|
||||
// Match by resolved alias target, not table identity: the CCA providers
|
||||
// now hold distinct table objects that still share these aliases.
|
||||
if (
|
||||
getAliasIndex(VARIANT_COLLAPSE_TABLES[candidate] as VariantCollapseTable).forward.get(normalized) === hit
|
||||
) {
|
||||
providers.push(candidate);
|
||||
}
|
||||
}
|
||||
return { id: hit, providers };
|
||||
}
|
||||
|
||||
@@ -18,6 +18,7 @@ import {
|
||||
collapseEffortVariants,
|
||||
collapseEffortVariantsAcrossProviders,
|
||||
deriveThinkingPairFamilies,
|
||||
GEMINI_CLI_VARIANT_COLLAPSE_TABLE,
|
||||
getVariantAliasSources,
|
||||
isVariantCollapsedSpec,
|
||||
resolveBareVariantAlias,
|
||||
@@ -89,15 +90,21 @@ describe("collapseEffortVariants", () => {
|
||||
// Capability union: max caps, image support from any member.
|
||||
expect(flash?.maxTokens).toBe(65_535);
|
||||
expect(flash?.input).toEqual(["text", "image"]);
|
||||
expect(flash?.thinking?.mode).toBe("google-level");
|
||||
expect(flash?.thinking?.mode).toBe("budget");
|
||||
expect(flash?.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]);
|
||||
expect(flash?.thinking?.effortBudgets).toEqual({
|
||||
minimal: 1000,
|
||||
low: 1000,
|
||||
medium: 4000,
|
||||
high: 10000,
|
||||
});
|
||||
expect(flash?.thinking?.suppressWhenOff).toBe(true);
|
||||
expect(flash?.thinking?.effortRouting).toEqual({
|
||||
off: "gemini-3.5-flash-extra-low",
|
||||
minimal: "gemini-3-flash-agent",
|
||||
minimal: "gemini-3.5-flash-extra-low",
|
||||
low: "gemini-3.5-flash-extra-low",
|
||||
medium: "gemini-3.5-flash-extra-low",
|
||||
high: "gemini-3.5-flash-low",
|
||||
medium: "gemini-3.5-flash-low",
|
||||
high: "gemini-3-flash-agent",
|
||||
});
|
||||
});
|
||||
|
||||
@@ -110,11 +117,12 @@ describe("collapseEffortVariants", () => {
|
||||
expect(out).toHaveLength(1);
|
||||
expect(out[0]?.id).toBe("gemini-3.5-flash");
|
||||
expect(out[0]?.requestModelId).toBe("gemini-3.5-flash-extra-low");
|
||||
// minimal (gemini-3-flash-agent) and high (gemini-3.5-flash-low) targets are absent.
|
||||
// minimal+low route to extra-low (present); medium (flash-low) and high
|
||||
// (flash-agent) targets are absent and drop.
|
||||
expect(out[0]?.thinking?.effortRouting).toEqual({
|
||||
off: "gemini-3.5-flash-extra-low",
|
||||
minimal: "gemini-3.5-flash-extra-low",
|
||||
low: "gemini-3.5-flash-extra-low",
|
||||
medium: "gemini-3.5-flash-extra-low",
|
||||
});
|
||||
});
|
||||
|
||||
@@ -176,6 +184,95 @@ describe("collapseEffortVariants", () => {
|
||||
const deduped = collapseEffortVariants(mixed, ANTIGRAVITY_VARIANT_COLLAPSE_TABLE);
|
||||
expect(deduped).toEqual(once);
|
||||
});
|
||||
|
||||
it("keeps gemini-cli flash on the level transport with the original routing", () => {
|
||||
const out = collapseEffortVariants(FLASH_TRIPLET(), GEMINI_CLI_VARIANT_COLLAPSE_TABLE);
|
||||
const flash = out.find(m => m.id === "gemini-3.5-flash");
|
||||
expect(flash?.thinking?.mode).toBe("google-level");
|
||||
expect(flash?.thinking?.effortBudgets).toBeUndefined();
|
||||
expect(flash?.thinking?.effortRouting).toEqual({
|
||||
off: "gemini-3.5-flash-extra-low",
|
||||
minimal: "gemini-3-flash-agent",
|
||||
low: "gemini-3.5-flash-extra-low",
|
||||
medium: "gemini-3.5-flash-extra-low",
|
||||
high: "gemini-3.5-flash-low",
|
||||
});
|
||||
});
|
||||
|
||||
it("collapses the 3.1-pro family on the budget transport with the +1 budgets", () => {
|
||||
const out = collapseEffortVariants(
|
||||
[memberSpec("gemini-3.1-pro-low"), memberSpec("gemini-pro-agent")],
|
||||
ANTIGRAVITY_VARIANT_COLLAPSE_TABLE,
|
||||
);
|
||||
const pro = out.find(m => m.id === "gemini-3.1-pro");
|
||||
expect(pro?.thinking?.mode).toBe("budget");
|
||||
expect(pro?.thinking?.effortBudgets).toEqual({ low: 1001, high: 10001 });
|
||||
expect(pro?.thinking?.effortRouting).toEqual({
|
||||
off: "gemini-3.1-pro-low",
|
||||
low: "gemini-3.1-pro-low",
|
||||
high: "gemini-pro-agent",
|
||||
});
|
||||
});
|
||||
|
||||
it("refreshes a stale alias-keyed flash snapshot in place to the budget contract", () => {
|
||||
// Bundled snapshots key the flash family under the recycled `gemini-3-flash`
|
||||
// id on the old level transport. That exact id is load-bearing, so it is
|
||||
// refreshed in place (same id) rather than re-keyed to `gemini-3.5-flash`.
|
||||
const stale: ModelSpec<"google-gemini-cli"> = {
|
||||
...memberSpec("gemini-3-flash"),
|
||||
reasoning: true,
|
||||
thinking: { mode: "google-level", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
|
||||
};
|
||||
const out = collapseEffortVariants([stale], ANTIGRAVITY_VARIANT_COLLAPSE_TABLE);
|
||||
const flash = out.find(m => m.id === "gemini-3-flash");
|
||||
expect(flash).toBeDefined();
|
||||
expect(flash?.thinking?.mode).toBe("budget");
|
||||
expect(flash?.thinking?.effortBudgets).toEqual({ minimal: 1000, low: 1000, medium: 4000, high: 10000 });
|
||||
expect(flash?.thinking?.effortRouting?.high).toBe("gemini-3-flash-agent");
|
||||
expect(flash?.requestModelId).toBe("gemini-3.5-flash-extra-low");
|
||||
});
|
||||
|
||||
it("heals a stale alias row alongside the canonical row (merge coexistence)", () => {
|
||||
// The model-manager merge keeps both the bundled exact `gemini-3-flash`
|
||||
// and the discovered canonical `gemini-3.5-flash` (exact-id merge); both
|
||||
// must land on the budget transport and neither is dropped.
|
||||
const stale: ModelSpec<"google-gemini-cli"> = {
|
||||
...memberSpec("gemini-3-flash"),
|
||||
reasoning: true,
|
||||
thinking: { mode: "google-level", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] },
|
||||
};
|
||||
const canonical = collapseEffortVariants(FLASH_TRIPLET(), ANTIGRAVITY_VARIANT_COLLAPSE_TABLE).find(
|
||||
m => m.id === "gemini-3.5-flash",
|
||||
);
|
||||
expect(canonical).toBeDefined();
|
||||
const out = collapseEffortVariants(
|
||||
[stale, canonical as ModelSpec<"google-gemini-cli">],
|
||||
ANTIGRAVITY_VARIANT_COLLAPSE_TABLE,
|
||||
);
|
||||
expect(out.map(m => m.id).sort()).toEqual(["gemini-3-flash", "gemini-3.5-flash"]);
|
||||
expect(out.find(m => m.id === "gemini-3-flash")?.thinking?.mode).toBe("budget");
|
||||
expect(out.find(m => m.id === "gemini-3.5-flash")?.thinking?.mode).toBe("budget");
|
||||
});
|
||||
|
||||
it("refreshes a stale family.id-keyed 3.1-pro snapshot in place to the budget contract", () => {
|
||||
// Pass-through branch: a bundled collapsed `gemini-3.1-pro` on the old level
|
||||
// transport with no live members refreshes from the hand table.
|
||||
const stale: ModelSpec<"google-gemini-cli"> = {
|
||||
...memberSpec("gemini-3.1-pro"),
|
||||
reasoning: true,
|
||||
requestModelId: "gemini-3.1-pro-low",
|
||||
thinking: {
|
||||
mode: "google-level",
|
||||
efforts: [Effort.Low, Effort.High],
|
||||
effortRouting: { off: "gemini-3.1-pro-low", low: "gemini-3.1-pro-low", high: "gemini-pro-agent" },
|
||||
suppressWhenOff: true,
|
||||
},
|
||||
};
|
||||
const out = collapseEffortVariants([stale], ANTIGRAVITY_VARIANT_COLLAPSE_TABLE);
|
||||
const pro = out.find(m => m.id === "gemini-3.1-pro");
|
||||
expect(pro?.thinking?.mode).toBe("budget");
|
||||
expect(pro?.thinking?.effortBudgets).toEqual({ low: 1001, high: 10001 });
|
||||
});
|
||||
});
|
||||
|
||||
describe("stripThinkingVariantToken", () => {
|
||||
@@ -380,8 +477,9 @@ describe("resolveWireModelId", () => {
|
||||
|
||||
expect(model.thinking?.effortRouting).toEqual(collapsed?.thinking?.effortRouting);
|
||||
expect(model.thinking?.suppressWhenOff).toBe(true);
|
||||
expect(resolveWireModelId(model, Effort.High)).toBe("gemini-3.5-flash-low");
|
||||
expect(resolveWireModelId(model, Effort.Minimal)).toBe("gemini-3-flash-agent");
|
||||
expect(resolveWireModelId(model, Effort.High)).toBe("gemini-3-flash-agent");
|
||||
expect(resolveWireModelId(model, Effort.Medium)).toBe("gemini-3.5-flash-low");
|
||||
expect(resolveWireModelId(model, Effort.Minimal)).toBe("gemini-3.5-flash-extra-low");
|
||||
expect(resolveWireModelId(model, undefined)).toBe("gemini-3.5-flash-extra-low");
|
||||
|
||||
// Dropped route (partial family) falls back to requestModelId.
|
||||
@@ -511,8 +609,8 @@ describe("antigravity discovery collapsing", () => {
|
||||
expect(models?.map(m => m.id).sort()).toEqual(["claude-sonnet-4-6", "gemini-2.5-flash", "gemini-3.5-flash"]);
|
||||
const flash = models?.find(m => m.id === "gemini-3.5-flash");
|
||||
expect(flash?.requestModelId).toBe("gemini-3.5-flash-extra-low");
|
||||
expect(flash?.thinking?.effortRouting?.[Effort.High]).toBe("gemini-3.5-flash-low");
|
||||
expect(flash?.thinking?.effortRouting?.[Effort.Minimal]).toBe("gemini-3-flash-agent");
|
||||
expect(flash?.thinking?.effortRouting?.[Effort.High]).toBe("gemini-3-flash-agent");
|
||||
expect(flash?.thinking?.effortRouting?.[Effort.Medium]).toBe("gemini-3.5-flash-low");
|
||||
expect(flash?.thinking?.suppressWhenOff).toBe(true);
|
||||
// The 2.5 pair collapses instead of denylisting the -thinking twin.
|
||||
const flash25 = models?.find(m => m.id === "gemini-2.5-flash");
|
||||
|
||||
Reference in New Issue
Block a user