From 58f319912e8ae5d7aacf067c04b4d4ab0cccacfc Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 13 Aug 2026 23:59:37 +0200 Subject: [PATCH] feat: introduced dynamic version discovery and rate limit parsing for google - Added dynamic Antigravity version discovery with environment variable overrides and fallback endpoint support. - Introduced structured Google RPC `ErrorInfo` rate limit reason parsing and backoff classification. - Added `gemini-3.7-flash` and `deepseek-v4-pro:preview` model configurations alongside updated pricing. - Removed system instruction injection logic and dropped unsigned thinking blocks for Antigravity requests. --- docs/environment-variables.md | 5 +- docs/provider-quirks.md | 6 +- packages/ai/CHANGELOG.md | 10 + packages/ai/src/error/rate-limit.ts | 72 ++++- .../ai/src/providers/google-gemini-cli.ts | 21 -- packages/ai/src/providers/google-shared.ts | 3 + .../src/registry/oauth/google-antigravity.ts | 143 ++++++--- packages/ai/src/usage/gemini.ts | 2 +- .../test/google-gemini-cli-alignment.test.ts | 88 +++-- packages/ai/test/rate-limit-utils.test.ts | 72 +++++ packages/catalog/CHANGELOG.md | 14 + packages/catalog/src/discovery/antigravity.ts | 6 +- packages/catalog/src/models.json | 300 ++++++++++++++---- packages/catalog/src/wire/gemini-headers.ts | 92 ++++-- .../src/web/search/providers/gemini.ts | 13 +- 15 files changed, 655 insertions(+), 192 deletions(-) diff --git a/docs/environment-variables.md b/docs/environment-variables.md index bdfa7cd9d..eb29df889 100644 --- a/docs/environment-variables.md +++ b/docs/environment-variables.md @@ -246,7 +246,10 @@ OAuth host chain: `KIMI_CODE_OAUTH_HOST` → `KIMI_OAUTH_HOST` → `https://auth | Variable | Default / behavior | | --------------------------- | --------------------------------------------------------------- | | `PI_AI_GEMINI_CLI_VERSION` | Overrides Gemini CLI user-agent version tag (`0.46.0` if unset) | -| `PI_AI_ANTIGRAVITY_VERSION` | Overrides Antigravity hub user-agent version (`2.1.4` if unset) | +| `PI_AI_ANTIGRAVITY_VERSION` | Overrides the auto-discovered Antigravity hub user-agent version; when unset and discovery fails, the fallback is `2.8.0` | +| `PI_AI_ANTIGRAVITY_CL` | Overrides Antigravity hub user-agent build changelist (`963137146` if unset) | +| `PI_AI_ANTIGRAVITY_OS` | Overrides Antigravity hub user-agent os_type (pinned `darwin` if unset) | +| `PI_AI_ANTIGRAVITY_ARCH` | Overrides Antigravity hub user-agent arch (pinned `arm64` if unset) | ### GitLab Duo diff --git a/docs/provider-quirks.md b/docs/provider-quirks.md index c3d4722eb..294394a8d 100644 --- a/docs/provider-quirks.md +++ b/docs/provider-quirks.md @@ -246,8 +246,8 @@ Google Cloud Code Assist (CCA) transport wrapper accessing Gemini and Claude mod - **Function Calling Config Mode**: Defaults to `functionCallingConfig: { mode: "VALIDATED" }` for Antigravity in `buildRequest`. Claude models on Antigravity force `VALIDATED` mode even when context contains no declared tools (`isClaudeModel`). Single named tool choice (`options.toolChoice`) sets `mode: "ANY"` with `allowedFunctionNames: [...]`. - **Provider Protocol & Request Envelope**: - **Endpoints**: `google-gemini-cli` defaults to `https://cloudcode-pa.googleapis.com`. `google-antigravity` uses auto-failover across `https://daily-cloudcode-pa.googleapis.com` (primary) and `https://daily-cloudcode-pa.sandbox.googleapis.com` (sandbox), persisting `lastGoodEndpoint` in `AntigravityProviderSessionState`. - - **Headers & User-Agent**: `google-gemini-cli` sends `getGeminiCliHeaders()` (`GeminiCLI/0.46.0/ (platform; arch; terminal)`). `google-antigravity` sends `getAntigravityUserAgent()` (`antigravity/hub/2.1.4 /`). Reasoning Claude models on Antigravity send `anthropic-beta: interleaved-thinking-2025-05-14` (`needsClaudeThinkingBetaHeader`). - - **System Instructions**: Antigravity tags system instructions with `role: "user"`. Claude and Gemini 3 models prepend `ANTIGRAVITY_SYSTEM_INSTRUCTION` ("You are Antigravity, a powerful agentic AI coding assistant...") via `shouldInjectAntigravitySystemInstruction`. + - **Headers & User-Agent**: `google-gemini-cli` sends `getGeminiCliHeaders()` (`GeminiCLI/0.46.0/ (platform; arch; terminal)`). `google-antigravity` sends `getAntigravityUserAgent()` (`antigravity/hub/ (aidev_client; os_type=; arch=; cl=)`); the backend gates newer models (e.g. gemini-3.7-flash) on the client version. Reasoning Claude models on Antigravity send `anthropic-beta: interleaved-thinking-2025-05-14` (`needsClaudeThinkingBetaHeader`). + - **System Instructions**: Antigravity tags system instructions with `role: "user"`. No identity prompt is injected — the backend accepts arbitrary system instructions on all routes (verified against gemini-3.x and Claude wire ids). - **Request Envelope & Session State**: Antigravity wraps requests in `buildAntigravityRequestEnvelope`: `project` (projectId), `requestId` (`agent////`), `userAgent` (`antigravity`), `requestType` (`agent`), and `labels` (`last_step_index`, `model_enum`, `trajectory_id`, `used_claude`, `used_claude_conservative`, `last_execution_id`). State maintains monotonic `stepIndex`, persistent `agentId`, `trajectoryId`, and signed-decimal `sessionId` (`deriveAntigravitySessionId`). - **Wire Profiles**: `getAntigravityModelWireProfile` (`packages/catalog/src/wire/gemini-headers.ts`) maps wire IDs to `maxOutputTokens` and `model_enum`. Claude wire IDs cap `maxOutputTokens` at `64000` (backend rejects >64000 with 400). - **Thinking Configuration & Wire Suppression**: Gemini 2.x models send `thinkingConfig.thinkingBudget`, while Gemini 3 models send `thinkingConfig.thinkingLevel`. When reasoning is disabled for models with `thinking.suppressWhenOff`, `buildRequest` emits explicit wire suppression (`includeThoughts: false` with level/budget). Omitting `thinkingConfig` causes CCA to re-apply server defaults and silently bill thinking tokens. @@ -892,7 +892,7 @@ The Google Antigravity provider (`google-antigravity`) routes requests to Google ### Special casings - **Validated Function Calling Default**: Default tool selection mode in `buildRequest` (`packages/ai/src/providers/google-gemini-cli.ts`) is `VALIDATED` (`functionCallingConfig: { mode: "VALIDATED" }`). Claude models on Antigravity always force `VALIDATED` tool mode even when no tools are declared (`packages/ai/src/providers/google-gemini-cli.ts`). -- **System Instruction & Request Envelope**: `shouldInjectAntigravitySystemInstruction` in `packages/ai/src/providers/google-gemini-cli.ts` prepends `ANTIGRAVITY_SYSTEM_INSTRUCTION` with `role: "user"` for Claude and Gemini 3 models. `buildAntigravityRequestEnvelope` injects structured `requestId` (`agent////`), `userAgent: "antigravity"`, `requestType: "agent"`, `sessionId`, and `labels` (`model_enum`, `trajectory_id`, `last_step_index`, `last_execution_id`, `used_claude*`) using `getAntigravityModelWireProfile`. +- **System Instruction & Request Envelope**: Antigravity tags `systemInstruction` with `role: "user"` and sends the caller's prompts unmodified. `buildAntigravityRequestEnvelope` injects structured `requestId` (`agent////`), `userAgent: "antigravity"`, `requestType: "agent"`, `sessionId`, and `labels` (`model_enum`, `trajectory_id`, `last_step_index`, `last_execution_id`, `used_claude*`) using `getAntigravityModelWireProfile`. - **Endpoint Auto-Failover**: Operates across `ANTIGRAVITY_DAILY_ENDPOINT` (`https://daily-cloudcode-pa.googleapis.com`) and `ANTIGRAVITY_SANDBOX_ENDPOINT` (`https://daily-cloudcode-pa.sandbox.googleapis.com`) with state-tracked fallback in `getAntigravityProviderSessionState` (`packages/ai/src/providers/google-gemini-cli.ts`). ### Auth & usage diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index f810dd67d..eafbc3ccb 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,16 @@ ## [Unreleased] +### Fixed + +- Dropped unsigned thinking blocks from Antigravity Claude requests instead of sending them without a signature, preventing HTTP 400 responses when resuming sessions or switching models. +- Classified Antigravity HTTP 429 responses from structured `google.rpc.ErrorInfo` reasons (`QUOTA_EXHAUSTED`, `RATE_LIMIT_EXCEEDED`, and `INSUFFICIENT_G1_CREDITS_BALANCE`), using retry delays of five minutes or longer to distinguish rotatable quota windows from transient throttling instead of relying only on message regexes. + +### Removed + +- Removed the Antigravity identity-prompt injection (`ANTIGRAVITY_SYSTEM_INSTRUCTION` and `shouldInjectAntigravitySystemInstruction`): Cloud Code Assist accepts arbitrary system instructions on gemini-3.x and Claude routes (verified live), and the injected stub never matched the real client's system prompt anyway. User system prompts are now sent unmodified (still tagged `role: "user"`). + + ## [17.3.0] - 2026-08-13 ### Breaking Changes diff --git a/packages/ai/src/error/rate-limit.ts b/packages/ai/src/error/rate-limit.ts index 65e4dc653..fa1dbf53a 100644 --- a/packages/ai/src/error/rate-limit.ts +++ b/packages/ai/src/error/rate-limit.ts @@ -1,3 +1,5 @@ +import { extractRetryHint } from "@oh-my-pi/pi-utils"; + /** * Rate limit reason classification and backoff calculation utilities. * Ported from opencode-antigravity-auth plugin for consistency. @@ -5,6 +7,7 @@ export type RateLimitReason = | "QUOTA_EXHAUSTED" + | "INSUFFICIENT_G1_CREDITS_BALANCE" | "RATE_LIMIT_EXCEEDED" | "CONCURRENT_LIMIT" | "MODEL_CAPACITY_EXHAUSTED" @@ -67,6 +70,66 @@ const CN_TRANSIENT_CAP_PATTERN = // of rotating through the opaque-429 fallback. const CN_THROTTLE_PATTERN = /速率(?:限制|过快)|频率(?:过高|过快)|过于频繁|稍后[重再]试/; +const GOOGLE_RPC_ERROR_INFO_TYPE = "type.googleapis.com/google.rpc.ErrorInfo"; +const LONG_RATE_LIMIT_DELAY_MS = 5 * 60 * 1000; + +function asRecord(value: unknown): Record | undefined { + return typeof value === "object" && value !== null && !Array.isArray(value) + ? (value as Record) + : undefined; +} + +function parseJsonBody(errorMessage: string): Record | undefined { + const start = errorMessage.indexOf("{"); + const end = errorMessage.lastIndexOf("}"); + if (start < 0 || end < start) return undefined; + try { + const parsed: unknown = JSON.parse(errorMessage.slice(start, end + 1)); + return asRecord(parsed); + } catch { + return undefined; + } +} + +/** + * Classify structured Google RESOURCE_EXHAUSTED bodies before consulting text. + * Cloud Code Assist prefixes the JSON with its HTTP error label, so accept an + * embedded top-level object as well as a raw JSON body. + */ +function parseGoogleRpcRateLimitReason(errorMessage: string): RateLimitReason | undefined { + const body = parseJsonBody(errorMessage); + const error = asRecord(body?.error); + if (typeof error?.status !== "string" || error.status.trim().toUpperCase() !== "RESOURCE_EXHAUSTED") { + return undefined; + } + if (!Array.isArray(error.details)) return undefined; + + for (const value of error.details) { + const detail = asRecord(value); + if (detail?.["@type"] !== GOOGLE_RPC_ERROR_INFO_TYPE || typeof detail.reason !== "string") continue; + const reason = detail.reason.trim().toUpperCase(); + switch (reason) { + case "QUOTA_EXHAUSTED": + return "QUOTA_EXHAUSTED"; + case "INSUFFICIENT_G1_CREDITS_BALANCE": + // Keep Google's specific credit-balance reason available to logs + // and callers while treating it as credential-rotatable below. + return "INSUFFICIENT_G1_CREDITS_BALANCE"; + case "RATE_LIMIT_EXCEEDED": { + const retryDelayMs = extractRetryHint(undefined, errorMessage); + return retryDelayMs !== undefined && retryDelayMs >= LONG_RATE_LIMIT_DELAY_MS + ? "QUOTA_EXHAUSTED" + : "RATE_LIMIT_EXCEEDED"; + } + } + } + return undefined; +} + +function isQuotaExhaustedReason(reason: RateLimitReason): boolean { + return reason === "QUOTA_EXHAUSTED" || reason === "INSUFFICIENT_G1_CREDITS_BALANCE"; +} + /** * Classify a rate-limit error message into a reason category. * Priority order: explicit details in a resource-exhausted error > QUOTA @@ -77,6 +140,8 @@ const CN_THROTTLE_PATTERN = /速率(?:限制|过快)|频率(?:过高|过快)|过 * Explicit details such as "quota exceeded" retain their normal classification. */ export function parseRateLimitReason(errorMessage: string): RateLimitReason { + const structuredReason = parseGoogleRpcRateLimitReason(errorMessage); + if (structuredReason !== undefined) return structuredReason; const lowerWithStatus = errorMessage.toLowerCase(); const lower = lowerWithStatus.replace(RESOURCE_EXHAUSTED_PATTERN, ""); const hasResourceExhaustedStatus = lower !== lowerWithStatus; @@ -162,6 +227,7 @@ export function parseRateLimitReason(errorMessage: string): RateLimitReason { */ export function calculateRateLimitBackoffMs(reason: RateLimitReason): number { switch (reason) { + case "INSUFFICIENT_G1_CREDITS_BALANCE": case "QUOTA_EXHAUSTED": return QUOTA_EXHAUSTED_BACKOFF_MS; case "RATE_LIMIT_EXCEEDED": @@ -215,6 +281,8 @@ export function isUsageLimitStatus(status: number | undefined): boolean { * credentials. */ export function isUsageLimitOutcome(status: number | undefined, message: string | undefined): boolean { + const structuredReason = message ? parseGoogleRpcRateLimitReason(message) : undefined; + if (structuredReason !== undefined) return isQuotaExhaustedReason(structuredReason); // Concurrency caps are shed-and-backoff, not credential-rotatable — but only // for quota-worded 429 / other statuses. HTTP 402 is categorically an // account-billing cap, so a 402 whose body happens to mention concurrency is @@ -235,7 +303,7 @@ export function isUsageLimitOutcome(status: number | undefined, message: string const reason = parseRateLimitReason(message); // For the categorical 402 billing cap a concurrency-worded body is still an // exhausted cap (rotate); for 429 / other only QUOTA_EXHAUSTED rotates. - return reason === "QUOTA_EXHAUSTED" || (isBillingCapStatus && reason === "CONCURRENT_LIMIT"); + return isQuotaExhaustedReason(reason) || (isBillingCapStatus && reason === "CONCURRENT_LIMIT"); } /** @@ -272,6 +340,8 @@ export function isOpaqueStatusBody(message: string): boolean { * {@link isUsageLimitOutcome} uses it for the account-rotation decision. */ export function matchesUsageLimitText(errorMessage: string): boolean { + const structuredReason = parseGoogleRpcRateLimitReason(errorMessage); + if (structuredReason !== undefined) return isQuotaExhaustedReason(structuredReason); return ( USAGE_LIMIT_PATTERN.test(errorMessage) || (CN_QUOTA_EXHAUSTED_PATTERN.test(errorMessage) && !CN_TRANSIENT_CAP_PATTERN.test(errorMessage)) || diff --git a/packages/ai/src/providers/google-gemini-cli.ts b/packages/ai/src/providers/google-gemini-cli.ts index f27538269..9d830517a 100644 --- a/packages/ai/src/providers/google-gemini-cli.ts +++ b/packages/ai/src/providers/google-gemini-cli.ts @@ -8,7 +8,6 @@ import { scheduler } from "node:timers/promises"; import { type } from "@oh-my-pi/omptype"; import { calculateCost } from "@oh-my-pi/pi-catalog/models"; import { - ANTIGRAVITY_SYSTEM_INSTRUCTION, getAntigravityModelWireProfile, getAntigravityUserAgent, getGeminiCliHeaders, @@ -316,13 +315,6 @@ const ANTIGRAVITY_DAILY_ENDPOINT = "https://daily-cloudcode-pa.googleapis.com"; const ANTIGRAVITY_SANDBOX_ENDPOINT = "https://daily-cloudcode-pa.sandbox.googleapis.com"; const ANTIGRAVITY_ENDPOINT_FALLBACKS = [ANTIGRAVITY_DAILY_ENDPOINT, ANTIGRAVITY_SANDBOX_ENDPOINT] as const; -export { - ANTIGRAVITY_SYSTEM_INSTRUCTION, - getAntigravityUserAgent, - getGeminiCliHeaders, - getGeminiCliUserAgent, -} from "@oh-my-pi/pi-catalog/wire/gemini-headers"; - // Retry configuration const MAX_RETRIES = 3; const BASE_DELAY_MS = 1000; @@ -342,11 +334,6 @@ function needsClaudeThinkingBetaHeader(model: Model<"google-gemini-cli">): boole return model.provider === "google-antigravity" && model.id.startsWith("claude-") && model.reasoning; } -function shouldInjectAntigravitySystemInstruction(modelId: string): boolean { - const normalized = modelId.toLowerCase(); - return normalized.includes("claude") || normalized.includes("gemini-3"); -} - const optionalCredentialString = type("unknown").pipe(raw => { const out = type("string")(raw); return out instanceof type.errors ? undefined : out; @@ -1320,14 +1307,6 @@ export function buildRequest( }; } - if (isAntigravity && shouldInjectAntigravitySystemInstruction(model.id)) { - const existingParts = request.systemInstruction?.parts ?? []; - request.systemInstruction = { - role: "user", - parts: [{ text: ANTIGRAVITY_SYSTEM_INSTRUCTION }, ...existingParts], - }; - } - if (context.tools && context.tools.length > 0) { const convertedTools = convertTools(context.tools, model); request.tools = isAntigravity ? normalizeAntigravityTools(convertedTools) : convertedTools; diff --git a/packages/ai/src/providers/google-shared.ts b/packages/ai/src/providers/google-shared.ts index bfce898f7..9af27d847 100644 --- a/packages/ai/src/providers/google-shared.ts +++ b/packages/ai/src/providers/google-shared.ts @@ -233,6 +233,8 @@ export function convertMessages(model: Model, contex const parts: Part[] = []; // Check if message is from same provider and model - only then keep thinking blocks const isSameProviderAndModel = msg.provider === model.provider && msg.model === model.id; + const dropsUnsignedThinking = + model.provider === "google-antigravity" && model.id.toLowerCase().includes("claude"); for (const block of msg.content) { if (block.type === "text") { @@ -247,6 +249,7 @@ export function convertMessages(model: Model, contex // Skip empty thinking blocks if (!block.thinking || block.thinking.trim() === "") continue; const thoughtSignature = resolveThoughtSignature(isSameProviderAndModel, block.thinkingSignature); + if (dropsUnsignedThinking && !thoughtSignature) continue; if (thoughtSignature) { parts.push({ thought: true, diff --git a/packages/ai/src/registry/oauth/google-antigravity.ts b/packages/ai/src/registry/oauth/google-antigravity.ts index 958ac5bcc..bb5317035 100644 --- a/packages/ai/src/registry/oauth/google-antigravity.ts +++ b/packages/ai/src/registry/oauth/google-antigravity.ts @@ -2,7 +2,7 @@ * Antigravity OAuth flow (Gemini 3, Claude, GPT-OSS via Google Cloud) * Uses different OAuth credentials than google-gemini-cli for access to additional models. */ -import { getAntigravityUserAgent } from "@oh-my-pi/pi-catalog/wire/gemini-headers"; +import { getAntigravityUserAgent, getAntigravityVersion } from "@oh-my-pi/pi-catalog/wire/gemini-headers"; import * as AIError from "../../error"; import { oauthFetch, runGoogleOAuthLogin, throwIfLoginCancelled } from "./google-oauth-shared"; import type { OAuthController, OAuthCredentials } from "./types"; @@ -26,10 +26,12 @@ const SCOPES = [ const AUTH_URL = "https://accounts.google.com/o/oauth2/v2/auth"; const TOKEN_URL = "https://oauth2.googleapis.com/token"; const CLOUD_CODE_ENDPOINT = "https://cloudcode-pa.googleapis.com"; -const TIER_LEGACY = "legacy-tier"; +const DAILY_CLOUD_CODE_ENDPOINT = "https://daily-cloudcode-pa.googleapis.com"; +const NODE_API_CLIENT_USER_AGENT = "google-api-nodejs-client/10.3.0"; +const GOOG_API_CLIENT_HEADER = "gl-node/22.21.1"; +const TIER_FREE = "free-tier"; const PROJECT_ONBOARD_MAX_ATTEMPTS = 5; const PROJECT_ONBOARD_INTERVAL_MS = 2000; - interface LoadCodeAssistPayload { cloudaicompanionProject?: string | { id?: string }; currentTier?: { id?: string }; @@ -45,35 +47,65 @@ interface LongRunningOperationResponse { export const ANTIGRAVITY_LOAD_CODE_ASSIST_METADATA = Object.freeze({ ideType: "ANTIGRAVITY", - platform: "PLATFORM_UNSPECIFIED", - pluginType: "GEMINI", }); -function readProjectId(value: string | { id?: string } | undefined): string | undefined { +export interface AntigravityOnboardMetadata { + ide_type: string; + ide_version: string; + ide_name: string; +} + +export function getAntigravityOnboardMetadata(): AntigravityOnboardMetadata { + return { + ide_type: "ANTIGRAVITY", + ide_version: getAntigravityVersion(), + ide_name: "antigravity", + }; +} + +function readProjectId(value: unknown): string | undefined { if (typeof value === "string" && value.length > 0) { - return value; + return value.trim(); } - if (value && typeof value === "object" && typeof value.id === "string" && value.id.length > 0) { - return value.id; + if (value && typeof value === "object" && "id" in value && typeof (value as { id?: unknown }).id === "string") { + const id = (value as { id: string }).id.trim(); + if (id.length > 0) return id; } return undefined; } -function getDefaultTierId(allowedTiers?: Array<{ id?: string; isDefault?: boolean }>): string { - if (!allowedTiers || allowedTiers.length === 0) { - return TIER_LEGACY; +function extractProjectId(payload: unknown): string | undefined { + if (!payload || typeof payload !== "object") return undefined; + const record = payload as Record; + for (const key of ["cloudaicompanionProject", "projectId", "project"]) { + const id = readProjectId(record[key]); + if (id) return id; } - const defaultTier = allowedTiers.find(tier => tier.isDefault && typeof tier.id === "string" && tier.id.length > 0); - if (defaultTier?.id) { - return defaultTier.id; + return undefined; +} + +function getDefaultTierId( + allowedTiers?: Array<{ id?: string; isDefault?: boolean }>, + currentTier?: { id?: string }, +): string { + if (allowedTiers && allowedTiers.length > 0) { + const defaultTier = allowedTiers.find( + tier => tier.isDefault && typeof tier.id === "string" && tier.id.trim().length > 0, + ); + if (defaultTier?.id) { + return defaultTier.id.trim(); + } } - return TIER_LEGACY; + if (currentTier && typeof currentTier.id === "string" && currentTier.id.trim().length > 0) { + return currentTier.id.trim(); + } + return TIER_FREE; } async function onboardProjectWithRetries( endpoint: string, headers: Record, - onboardBody: { tierId: string; metadata: typeof ANTIGRAVITY_LOAD_CODE_ASSIST_METADATA }, + onboardBody: { tier_id: string; metadata: AntigravityOnboardMetadata }, signal: AbortSignal | undefined, onProgress?: (message: string) => void, ): Promise { @@ -104,7 +136,7 @@ async function onboardProjectWithRetries( continue; } - const projectId = readProjectId(operation.response?.cloudaicompanionProject); + const projectId = extractProjectId(operation.response); if (projectId) { return projectId; } @@ -128,42 +160,65 @@ async function discoverProject( }; onProgress?.("Checking for existing project..."); - const endpoint = CLOUD_CODE_ENDPOINT; try { - throwIfLoginCancelled(signal); - const loadResponse = await oauthFetch( - `${endpoint}/v1internal:loadCodeAssist`, - { - method: "POST", - headers, - body: JSON.stringify({ - metadata: ANTIGRAVITY_LOAD_CODE_ASSIST_METADATA, - }), - }, - { provider: "google-antigravity", signal }, - ); + let lastErrorText: string | undefined; + let lastStatus: number | undefined; + let fallbackTierId = TIER_FREE; + let loadedSuccessfully = false; - if (!loadResponse.ok) { - const errorText = await loadResponse.text(); - throw new AIError.OAuthError( - `loadCodeAssist failed: ${loadResponse.status} ${loadResponse.statusText}: ${errorText}`, - { kind: "discovery", status: loadResponse.status }, + for (const endpoint of [DAILY_CLOUD_CODE_ENDPOINT, CLOUD_CODE_ENDPOINT]) { + throwIfLoginCancelled(signal); + const loadResponse = await oauthFetch( + `${endpoint}/v1internal:loadCodeAssist`, + { + method: "POST", + headers, + body: JSON.stringify({ + metadata: ANTIGRAVITY_LOAD_CODE_ASSIST_METADATA, + }), + }, + { provider: "google-antigravity", signal }, ); + + if (!loadResponse.ok) { + lastStatus = loadResponse.status; + lastErrorText = await loadResponse.text(); + continue; + } + + loadedSuccessfully = true; + const loadPayload = (await loadResponse.json()) as LoadCodeAssistPayload; + const existingProject = extractProjectId(loadPayload); + if (existingProject) { + return existingProject; + } + fallbackTierId = getDefaultTierId(loadPayload.allowedTiers, loadPayload.currentTier); } - const loadPayload = (await loadResponse.json()) as LoadCodeAssistPayload; - const existingProject = readProjectId(loadPayload.cloudaicompanionProject); - if (existingProject) { - return existingProject; + if (!loadedSuccessfully && lastStatus !== undefined) { + throw new AIError.OAuthError(`loadCodeAssist failed: ${lastStatus}: ${lastErrorText || "unknown error"}`, { + kind: "discovery", + status: lastStatus, + }); } - const tierId = getDefaultTierId(loadPayload.allowedTiers); onProgress?.("Provisioning project..."); const onboardBody = { - tierId, - metadata: ANTIGRAVITY_LOAD_CODE_ASSIST_METADATA, + tier_id: fallbackTierId, + metadata: getAntigravityOnboardMetadata(), }; - const provisionedProject = await onboardProjectWithRetries(endpoint, headers, onboardBody, signal, onProgress); + const onboardHeaders: Record = { + ...headers, + "User-Agent": `${headers["User-Agent"]} ${NODE_API_CLIENT_USER_AGENT}`, + "X-Goog-Api-Client": GOOG_API_CLIENT_HEADER, + }; + const provisionedProject = await onboardProjectWithRetries( + DAILY_CLOUD_CODE_ENDPOINT, + onboardHeaders, + onboardBody, + signal, + onProgress, + ); return provisionedProject; } catch (error) { if (error instanceof AIError.LoginCancelledError || error instanceof AIError.OAuthError) { diff --git a/packages/ai/src/usage/gemini.ts b/packages/ai/src/usage/gemini.ts index 610c67c6c..ed2701cb9 100644 --- a/packages/ai/src/usage/gemini.ts +++ b/packages/ai/src/usage/gemini.ts @@ -1,4 +1,4 @@ -import { getGeminiCliHeaders } from "../providers/google-gemini-cli"; +import { getGeminiCliHeaders } from "@oh-my-pi/pi-catalog/wire/gemini-headers"; import type { UsageAmount, UsageFetchContext, diff --git a/packages/ai/test/google-gemini-cli-alignment.test.ts b/packages/ai/test/google-gemini-cli-alignment.test.ts index 2ca23ea5b..cf03df3b5 100644 --- a/packages/ai/test/google-gemini-cli-alignment.test.ts +++ b/packages/ai/test/google-gemini-cli-alignment.test.ts @@ -1,7 +1,6 @@ import { describe, expect, it } from "bun:test"; import * as geminiCliProvider from "@oh-my-pi/pi-ai/providers/google-gemini-cli"; import { - ANTIGRAVITY_SYSTEM_INSTRUCTION, buildRequest, parseGeminiCliCredentials, shouldRefreshGeminiCliCredentials, @@ -189,9 +188,70 @@ describe("Google Gemini CLI alignment", () => { expect(payload.request.contents).toEqual([{ role: "user", parts: [{ text: "implement token refresh" }] }]); }); + it("drops only unsigned thinking when replaying Antigravity Claude history", () => { + const signedThinking = "signed reasoning"; + const unsignedThinking = "unsigned reasoning"; + const signature = "c2lnbmVk"; + const createThinkingContext = (model: Model<"google-gemini-cli">): Context => ({ + messages: [ + { + role: "assistant", + content: [ + { type: "thinking", thinking: signedThinking, thinkingSignature: signature }, + { type: "thinking", thinking: unsignedThinking }, + ], + api: model.api, + provider: model.provider, + model: model.id, + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp: 1, + }, + { role: "user", content: "continue", timestamp: 2 }, + ], + }); + const claudeModel = buildModel({ + ...createModel("google-antigravity"), + id: "claude-sonnet-4-6", + name: "Claude Sonnet 4.6", + reasoning: true, + } as ModelSpec<"google-gemini-cli">); + const claudePayload = buildRequest(claudeModel, createThinkingContext(claudeModel), "proj-123", {}, true) as { + request: { + contents: Array<{ + role: string; + parts: Array<{ text?: string; thought?: boolean; thoughtSignature?: string }>; + }>; + }; + }; + const claudeParts = claudePayload.request.contents.find(content => content.role === "model")?.parts; + expect(claudeParts).toEqual([{ thought: true, text: signedThinking, thoughtSignature: signature }]); + + const geminiModel = createModel("google-antigravity"); + const geminiPayload = buildRequest(geminiModel, createThinkingContext(geminiModel), "proj-123", {}, true) as { + request: { + contents: Array<{ + role: string; + parts: Array<{ text?: string; thought?: boolean; thoughtSignature?: string }>; + }>; + }; + }; + const geminiParts = geminiPayload.request.contents.find(content => content.role === "model")?.parts ?? []; + expect(geminiParts).toContainEqual({ thought: true, text: signedThinking, thoughtSignature: signature }); + expect(geminiParts.some(part => part.text?.includes(unsignedThinking))).toBe(true); + }); + it("keeps antigravity metadata in antigravity request payloads", () => { const model = createModel("google-antigravity"); - const payload = buildRequest(model, createContext(), "proj-123", {}, true) as { + const context: Context = { ...createContext(), systemPrompt: ["be terse"] }; + const payload = buildRequest(model, context, "proj-123", {}, true) as { request: { sessionId?: string; labels?: Record; @@ -324,30 +384,6 @@ describe("Google Gemini CLI alignment", () => { expect(parameters).toBeDefined(); expect(JSON.stringify(parameters)).not.toContain('"patternProperties"'); }); - it("injects ANTIGRAVITY_SYSTEM_INSTRUCTION for gemini-3.1-pro-high and gemini-3.1-pro-low", () => { - // Regression test for #1274: shouldInjectAntigravitySystemInstruction checked - // "gemini-3-pro-high" (hyphen) but the deployed model IDs use "gemini-3.1-pro-high" (dot), - // so the injection was silently skipped and the Cloud Code Assist API returned HTTP 400. - for (const modelId of ["gemini-3.1-pro-high", "gemini-3.1-pro-low"] as const) { - const model: Model<"google-gemini-cli"> = buildModel({ - ...createModel("google-antigravity"), - id: modelId, - } as ModelSpec<"google-gemini-cli">); - const context: Context = { - systemPrompt: ["my instructions"], - messages: [{ role: "user", content: "hi", timestamp: Date.now() }], - }; - const payload = buildRequest(model, context, "proj-123", {}, true) as { - request: { systemInstruction?: { role?: string; parts: Array<{ text: string }> } }; - }; - - const parts = payload.request.systemInstruction?.parts ?? []; - // The antigravity identity header must be injected as the first part. - expect(parts[0]?.text).toBe(ANTIGRAVITY_SYSTEM_INSTRUCTION); - // The user-supplied system prompt must appear after the single injected part. - expect(parts.slice(1).some(p => p.text === "my instructions")).toBe(true); - } - }); it("adds anthropic-beta for Antigravity Claude reasoning models without relying on id suffix", async () => { let requestHeaders: Headers | undefined; const fetchMock: FetchImpl = async (_url, init) => { diff --git a/packages/ai/test/rate-limit-utils.test.ts b/packages/ai/test/rate-limit-utils.test.ts index 3d06e6d11..5849c576d 100644 --- a/packages/ai/test/rate-limit-utils.test.ts +++ b/packages/ai/test/rate-limit-utils.test.ts @@ -10,6 +10,30 @@ import { parseRateLimitReason, } from "@oh-my-pi/pi-ai/error/rate-limit"; +function googleRpc429(reason: string, retryDelay?: string, message = "Resource exhausted"): string { + const details: Array> = [ + { + "@type": "type.googleapis.com/google.rpc.ErrorInfo", + reason, + domain: "cloudcode-pa.googleapis.com", + }, + ]; + if (retryDelay) { + details.push({ + "@type": "type.googleapis.com/google.rpc.RetryInfo", + retryDelay, + }); + } + return `Cloud Code Assist API error (429): ${JSON.stringify({ + error: { + code: 429, + message, + status: "RESOURCE_EXHAUSTED", + details, + }, + })}`; +} + describe("parseRateLimitReason", () => { it("classifies Google Quota exceeded as QUOTA_EXHAUSTED", () => { expect( @@ -170,6 +194,54 @@ describe("parseRateLimitReason", () => { ), ).toBe("QUOTA_EXHAUSTED"); }); + + it("uses structured QUOTA_EXHAUSTED before capacity message heuristics", () => { + const body = googleRpc429( + "QUOTA_EXHAUSTED", + "21600s", + "The model has no capacity available; retry another request later.", + ); + expect(parseRateLimitReason(body)).toBe("QUOTA_EXHAUSTED"); + expect(isUsageLimitOutcome(429, body)).toBe(true); + }); + + it("keeps structured RATE_LIMIT_EXCEEDED with a 30s delay transient", () => { + const body = googleRpc429("RATE_LIMIT_EXCEEDED", "30s", "Too many requests"); + expect(parseRateLimitReason(body)).toBe("RATE_LIMIT_EXCEEDED"); + expect(isUsageLimitOutcome(429, body)).toBe(false); + expect(isUsageLimit(Object.assign(new Error(body), { status: 429 }))).toBe(false); + }); + + it("keeps structured RATE_LIMIT_EXCEEDED without a retry delay transient", () => { + const body = googleRpc429("RATE_LIMIT_EXCEEDED", undefined, "Too many requests"); + expect(parseRateLimitReason(body)).toBe("RATE_LIMIT_EXCEEDED"); + expect(isUsageLimitOutcome(429, body)).toBe(false); + }); + + it("treats structured RATE_LIMIT_EXCEEDED with a 6h delay as usage exhaustion", () => { + const body = googleRpc429("RATE_LIMIT_EXCEEDED", "21600s", "Too many requests"); + expect(parseRateLimitReason(body)).toBe("QUOTA_EXHAUSTED"); + expect(isUsageLimitOutcome(429, body)).toBe(true); + }); + + it("treats the five-minute structured rate-limit threshold as usage exhaustion", () => { + const body = googleRpc429("RATE_LIMIT_EXCEEDED", "300s", "Too many requests"); + expect(parseRateLimitReason(body)).toBe("QUOTA_EXHAUSTED"); + expect(isUsageLimitOutcome(429, body)).toBe(true); + }); + + it("preserves structured INSUFFICIENT_G1_CREDITS_BALANCE while rotating credentials", () => { + const body = googleRpc429("INSUFFICIENT_G1_CREDITS_BALANCE", undefined, "Credit balance is unavailable"); + expect(parseRateLimitReason(body)).toBe("INSUFFICIENT_G1_CREDITS_BALANCE"); + expect(isUsageLimitOutcome(429, body)).toBe(true); + expect(isUsageLimit(Object.assign(new Error(body), { status: 429 }))).toBe(true); + }); + + it("falls back to existing text heuristics for non-JSON 429 bodies", () => { + const body = "Cloud Code Assist API error (429): Too many requests"; + expect(parseRateLimitReason(body)).toBe("RATE_LIMIT_EXCEEDED"); + expect(isUsageLimitOutcome(429, body)).toBe(false); + }); }); describe("isUsageLimit", () => { diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 3df6531e8..c3a2f3bfc 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,20 @@ ## [Unreleased] +### Added + +- Added support for the `deepseek-v4-pro:preview` model +- Added support for the `gemini-3.7-flash` model +- Added dynamic Antigravity client-version discovery from the official update manifest (darwin/arm64 channel), so version-gated models appear without a code change; `PI_AI_ANTIGRAVITY_VERSION` remains available as an override. + +### Removed + +- Removed `ANTIGRAVITY_SYSTEM_INSTRUCTION` from `wire/gemini-headers`; the Antigravity transport and web search no longer inject a fake identity prompt. + +### Fixed + +- Fixed Antigravity discovery missing Gemini 3.7 Flash: Cloud Code Assist gates newer models on the client version in the `User-Agent`, and the pinned `antigravity/hub/2.1.4` was too old. The user-agent now matches the captured 2.8.0 client format (`antigravity/hub/2.8.0 (aidev_client; os_type=darwin; arch=arm64; cl=963137146)`); os_type/arch stay pinned to the darwin/arm64 reference client. Overridable via `PI_AI_ANTIGRAVITY_VERSION` / `PI_AI_ANTIGRAVITY_CL` / `PI_AI_ANTIGRAVITY_OS` / `PI_AI_ANTIGRAVITY_ARCH`. + ## [17.3.1] - 2026-08-13 ### Added diff --git a/packages/catalog/src/discovery/antigravity.ts b/packages/catalog/src/discovery/antigravity.ts index 259c678c5..7d6cb68be 100644 --- a/packages/catalog/src/discovery/antigravity.ts +++ b/packages/catalog/src/discovery/antigravity.ts @@ -6,7 +6,7 @@ import { collapseEffortVariants, type VariantCollapseTable, } from "../variant-collapse"; -import { getAntigravityUserAgent } from "../wire/gemini-headers"; +import { ensureAntigravityVersion, getAntigravityUserAgent } from "../wire/gemini-headers"; export const ANTIGRAVITY_PRIMARY_ENDPOINT = "https://daily-cloudcode-pa.googleapis.com"; export const ANTIGRAVITY_SANDBOX_ENDPOINT = "https://daily-cloudcode-pa.sandbox.googleapis.com"; @@ -161,6 +161,10 @@ export interface FetchAntigravityDiscoveryModelsOptions { export async function fetchAntigravityDiscoveryModels( options: FetchAntigravityDiscoveryModelsOptions, ): Promise[] | null> { + if (options.userAgent === undefined) { + await ensureAntigravityVersion(fetch, options.signal); + } + const fetcher = discoveryFetch(options.fetcher); const endpoints = options.endpoint ? [trimTrailingSlashes(options.endpoint)] diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 039c844a3..5dc7d1f9b 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -25988,6 +25988,43 @@ } } }, + "gemini-3.7-flash": { + "id": "gemini-3.7-flash", + "name": "Gemini 3.7 Flash", + "api": "google-gemini-cli", + "provider": "google-antigravity", + "baseUrl": "https://daily-cloudcode-pa.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.75, + "output": 3.75, + "cacheRead": 0.075, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "requestModelId": "gemini-3.7-flash-low", + "thinking": { + "mode": "google-level", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true, + "effortRouting": { + "minimal": "gemini-3.7-flash-low", + "low": "gemini-3.7-flash-low", + "medium": "gemini-3.7-flash-medium", + "high": "gemini-3.7-flash-high" + } + } + }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "GPT OSS 120B", @@ -29128,12 +29165,12 @@ "text" ], "cost": { - "input": 0.079996, - "output": 0.252, - "cacheRead": 0.0252, + "input": 0.072, + "output": 0.144, + "cacheRead": 0.016, "cacheWrite": 0 }, - "contextWindow": 1048576, + "contextWindow": 262144, "maxTokens": 262144, "thinking": { "mode": "effort", @@ -29156,10 +29193,10 @@ "image" ], "cost": { - "input": 1.5, - "output": 7.5, - "cacheRead": 0.15, - "cacheWrite": 0.083333 + "input": 0.375, + "output": 1.875, + "cacheRead": 0.0375, + "cacheWrite": 0.020833 }, "contextWindow": 1048576, "maxTokens": 65536, @@ -29216,13 +29253,13 @@ "image" ], "cost": { - "input": 2.8, - "output": 14, - "cacheRead": 0.29, + "input": 2.4, + "output": 12, + "cacheRead": 0.24, "cacheWrite": 0 }, - "contextWindow": 1048576, - "maxTokens": 1048576, + "contextWindow": 974842, + "maxTokens": 974842, "thinking": { "mode": "effort", "efforts": [ @@ -32370,10 +32407,10 @@ "image" ], "cost": { - "input": 1.5, - "output": 7.5, - "cacheRead": 0.15, - "cacheWrite": 0.083333 + "input": 0.75, + "output": 3.75, + "cacheRead": 0.075, + "cacheWrite": 0.041667 }, "contextWindow": 1048576, "maxTokens": 65536, @@ -32394,19 +32431,29 @@ "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 + "input": 0.75, + "output": 3.75, + "cacheRead": 0.075, + "cacheWrite": 0.041667 }, "contextWindow": 1048576, "maxTokens": 65536, - "supportsComputerUse": false + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } }, "google/gemma-2-27b-it": { "id": "google/gemma-2-27b-it", @@ -41453,9 +41500,9 @@ "text" ], "cost": { - "input": 0.55, - "output": 2.2, - "cacheRead": 0.11, + "input": 0.5, + "output": 2, + "cacheRead": 0.1, "cacheWrite": 0 }, "contextWindow": 202752, @@ -41533,9 +41580,9 @@ "text" ], "cost": { - "input": 0.6, - "output": 2.2, - "cacheRead": 0.11, + "input": 0.4, + "output": 1.75, + "cacheRead": 0.08, "cacheWrite": 0 }, "contextWindow": 202752, @@ -41562,7 +41609,7 @@ "text" ], "cost": { - "input": 0.07, + "input": 0.06, "output": 0.4, "cacheRead": 0.01, "cacheWrite": 0 @@ -41591,8 +41638,8 @@ "text" ], "cost": { - "input": 1, - "output": 3.2, + "input": 0.95, + "output": 2.55, "cacheRead": 0.2, "cacheWrite": 0 }, @@ -41649,7 +41696,7 @@ "text" ], "cost": { - "input": 1.38, + "input": 1.4, "output": 4.4, "cacheRead": 0.26, "cacheWrite": 0 @@ -41683,7 +41730,7 @@ "cacheRead": 0.26, "cacheWrite": 0 }, - "contextWindow": 262144, + "contextWindow": 1048576, "maxTokens": 131072, "thinking": { "mode": "effort", @@ -51050,7 +51097,7 @@ "input": 1.5, "output": 9, "cacheRead": 0.15, - "cacheWrite": 0 + "cacheWrite": 0.083333 }, "contextWindow": 1048576, "maxTokens": 65536, @@ -51110,7 +51157,37 @@ "input": 1.5, "output": 7.5, "cacheRead": 0.15, - "cacheWrite": 0 + "cacheWrite": 0.041667 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, + "google/gemini-3.7-flash": { + "id": "google/gemini-3.7-flash", + "name": "Gemini 3.7 Flash", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.375, + "output": 1.875, + "cacheRead": 0.0375, + "cacheWrite": 0.020833 }, "contextWindow": 1048576, "maxTokens": 65536, @@ -71596,6 +71673,35 @@ ] } }, + "deepseek-v4-pro:preview": { + "id": "deepseek-v4-pro:preview", + "name": "deepseek-v4-pro:preview", + "api": "ollama-chat", + "provider": "ollama-cloud", + "baseUrl": "https://ollama.com", + "reasoning": true, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ] + }, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 524288, + "maxTokens": 65536, + "omitMaxOutputTokens": true, + "supportsComputerUse": false + }, "devstral-2:123b": { "id": "devstral-2:123b", "name": "devstral-2:123b", @@ -76127,6 +76233,36 @@ "requiresEffort": true } }, + "gemini-3.7-flash": { + "id": "gemini-3.7-flash", + "name": "Gemini 3.7 Flash", + "api": "google-generative-ai", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.5, + "output": 7.5, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "google-level", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, "glm-4.6": { "id": "glm-4.6", "name": "GLM-4.6", @@ -77974,13 +78110,13 @@ "text" ], "cost": { - "input": 0.079996, - "output": 0.252, - "cacheRead": 0.0252, + "input": 0.072, + "output": 0.144, + "cacheRead": 0.016, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 131072, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -78007,7 +78143,7 @@ "input": 0.375, "output": 1.875, "cacheRead": 0.0375, - "cacheWrite": 0.0416666666666667 + "cacheWrite": 0.0208333333333333 }, "contextWindow": 1048576, "maxTokens": 65536, @@ -78066,13 +78202,13 @@ "image" ], "cost": { - "input": 2.8, - "output": 14, - "cacheRead": 0.29, + "input": 2.4, + "output": 12, + "cacheRead": 0.24, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 1048576, + "maxTokens": 974842, "thinking": { "mode": "effort", "efforts": [ @@ -81235,10 +81371,10 @@ "image" ], "cost": { - "input": 1.5, - "output": 7.5, - "cacheRead": 0.15, - "cacheWrite": 0.0833333333333333 + "input": 0.75, + "output": 3.75, + "cacheRead": 0.075, + "cacheWrite": 0.0416666666666667 }, "contextWindow": 1048576, "maxTokens": 65536, @@ -81267,10 +81403,10 @@ "image" ], "cost": { - "input": 0.75, - "output": 3.75, - "cacheRead": 0.075, - "cacheWrite": 0.0833333333333333 + "input": 0.375, + "output": 1.875, + "cacheRead": 0.0375, + "cacheWrite": 0.0416666666666667 }, "contextWindow": 1048576, "maxTokens": 65536, @@ -81302,7 +81438,7 @@ "input": 0.375, "output": 1.875, "cacheRead": 0.0375, - "cacheWrite": 0.0416666666666667 + "cacheWrite": 0.0208333333333333 }, "contextWindow": 1048576, "maxTokens": 65536, @@ -81331,10 +81467,10 @@ "image" ], "cost": { - "input": 0.375, - "output": 1.875, - "cacheRead": 0.0375, - "cacheWrite": 0.0416666666666667 + "input": 0.1875, + "output": 0.9375, + "cacheRead": 0.01875, + "cacheWrite": 0.0208333333333333 }, "contextWindow": 1048576, "maxTokens": 65536, @@ -83285,9 +83421,9 @@ "image" ], "cost": { - "input": 0.5795, - "output": 2.44, - "cacheRead": 0.0976, + "input": 0.5605, + "output": 2.36, + "cacheRead": 0.0944, "cacheWrite": 0 }, "contextWindow": 262144, @@ -83915,7 +84051,7 @@ "cacheRead": 0.049999999999999996, "cacheWrite": 0 }, - "contextWindow": 1048576, + "contextWindow": 1000000, "maxTokens": 262144, "thinking": { "mode": "effort", @@ -90620,9 +90756,9 @@ "text" ], "cost": { - "input": 0.49, - "output": 1.54, - "cacheRead": 0.091, + "input": 0.392, + "output": 1.232, + "cacheRead": 0.0728, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -96815,6 +96951,36 @@ }, "supportsComputerUse": false }, + "alibaba/qwen3.8-2.4t-a95b": { + "id": "alibaba/qwen3.8-2.4t-a95b", + "name": "Qwen3.8 2.4T A95B", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 2, + "output": 6, + "cacheRead": 0.25, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 131072, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + }, + "supportsComputerUse": false + }, "alibaba/qwen3.8-max": { "id": "alibaba/qwen3.8-max", "name": "Qwen 3.8 Max", @@ -98529,7 +98695,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 256000, + "contextWindow": 262144, "maxTokens": 131072, "thinking": { "mode": "budget", diff --git a/packages/catalog/src/wire/gemini-headers.ts b/packages/catalog/src/wire/gemini-headers.ts index c56e0955c..33ad43955 100644 --- a/packages/catalog/src/wire/gemini-headers.ts +++ b/packages/catalog/src/wire/gemini-headers.ts @@ -15,30 +15,86 @@ export const getGeminiCliHeaders = (modelId?: string) => ({ "Client-Metadata": "ideType=IDE_UNSPECIFIED,platform=PLATFORM_UNSPECIFIED,pluginType=GEMINI", }); -export const ANTIGRAVITY_SYSTEM_INSTRUCTION = - "You are Antigravity, a powerful agentic AI coding assistant designed by the Google Deepmind team working on Advanced Agentic Coding." + - "You are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question." + - "**Absolute paths only**" + - "**Proactiveness**"; /** * Antigravity / Cloud Code Assist user agent. Lives in its own file so discovery * and usage code can read it without pulling the heavy google-gemini-cli provider * (and its @google/genai → google-auth-library dependency chain) into the startup * parse graph. + * + * Format captured from the real 2.8.0 `antigravity/hub` client: + * `antigravity/hub/2.8.0 (aidev_client; os_type=darwin; arch=arm64; cl=963137146)`. + * The backend gates newer models (e.g. gemini-3.7-flash) on the client version, + * so the version tracks the latest Antigravity release via the update manifest + * (see {@link ensureAntigravityVersion}) with `DEFAULT_ANTIGRAVITY_VERSION` as + * the offline fallback. os_type/arch are pinned to the darwin/arm64 reference + * client the version and manifest are captured from, independent of the host + * platform. Overrides: PI_AI_ANTIGRAVITY_VERSION / _CL / _OS / _ARCH. */ -export let getAntigravityUserAgent = () => { - const DEFAULT_ANTIGRAVITY_VERSION = "2.1.4"; - const version = process.env.PI_AI_ANTIGRAVITY_VERSION || DEFAULT_ANTIGRAVITY_VERSION; - // Map Node.js platform/arch to Antigravity's expected format. - // Verified against Antigravity source: _qn() and wqn() in main.js. - // process.platform: win32→windows, others pass through (darwin, linux) - // process.arch: x64→amd64, ia32→386, others pass through (arm64) - const os = process.platform === "win32" ? "windows" : process.platform; - const arch = process.arch === "x64" ? "amd64" : process.arch === "ia32" ? "386" : process.arch; - const userAgent = `antigravity/hub/${version} ${os}/${arch}`; - getAntigravityUserAgent = () => userAgent; - return userAgent; -}; +export const DEFAULT_ANTIGRAVITY_VERSION = "2.8.0"; + +const ANTIGRAVITY_VERSION_MANIFEST_URL = + "https://antigravity-hub-auto-updater-974169037036.us-central1.run.app/manifest/latest-arm64-mac.yml"; +const ANTIGRAVITY_VERSION_FETCH_TIMEOUT_MS = 5_000; + +let discoveredAntigravityVersion: string | null = null; +let antigravityVersionFetch: Promise | null = null; + +/** Current Antigravity client version: env override → manifest-discovered → pinned fallback. */ +export function getAntigravityVersion(): string { + return process.env.PI_AI_ANTIGRAVITY_VERSION || discoveredAntigravityVersion || DEFAULT_ANTIGRAVITY_VERSION; +} + +/** + * Extracts the client version from an electron-builder update manifest. + * Returns null when no well-formed `version:` line is present. + */ +export function parseAntigravityManifestVersion(yamlText: string): string | null { + for (const line of yamlText.split(/\r?\n/)) { + const match = /^\s*version\s*:\s*(?:"([^"]*)"|'([^']*)'|([^\s#]+))\s*(?:#.*)?$/.exec(line); + if (!match) continue; + const version = (match[1] ?? match[2] ?? match[3] ?? "").trim(); + return /^\d+\.\d+\.\d+$/.test(version) ? version : null; + } + return null; +} + +/** + * Resolves the latest Antigravity release from the official update manifest. + * Success is cached for the process lifetime; failures are silent (the pinned + * fallback stays valid) and clear the in-flight cache so a later call retries. + * Skipped entirely when PI_AI_ANTIGRAVITY_VERSION is set. + */ +export function ensureAntigravityVersion(fetcher: typeof fetch = fetch, signal?: AbortSignal): Promise { + if (process.env.PI_AI_ANTIGRAVITY_VERSION || discoveredAntigravityVersion) return Promise.resolve(); + if (antigravityVersionFetch) return antigravityVersionFetch; + + antigravityVersionFetch = (async () => { + try { + const timeoutSignal = AbortSignal.timeout(ANTIGRAVITY_VERSION_FETCH_TIMEOUT_MS); + const response = await fetcher(ANTIGRAVITY_VERSION_MANIFEST_URL, { + headers: { "Cache-Control": "no-cache", "User-Agent": "electron-builder" }, + signal: signal ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal, + }); + if (response.ok) { + discoveredAntigravityVersion = parseAntigravityManifestVersion(await response.text()); + } + } catch { + // Silent: the pinned fallback remains valid when version discovery fails. + } finally { + if (!discoveredAntigravityVersion) antigravityVersionFetch = null; + } + })(); + return antigravityVersionFetch; +} + +/** Antigravity `User-Agent` header value; rebuilt when the discovered version changes. */ +export function getAntigravityUserAgent(): string { + const version = getAntigravityVersion(); + const cl = process.env.PI_AI_ANTIGRAVITY_CL || "963137146"; + const os = process.env.PI_AI_ANTIGRAVITY_OS || "darwin"; + const arch = process.env.PI_AI_ANTIGRAVITY_ARCH || "arm64"; + return `antigravity/hub/${version} (aidev_client; os_type=${os}; arch=${arch}; cl=${cl})`; +} /** * Per-wire-id Antigravity Cloud Code Assist request constants, captured from the diff --git a/packages/coding-agent/src/web/search/providers/gemini.ts b/packages/coding-agent/src/web/search/providers/gemini.ts index c9ba6d0a3..3691b18d5 100644 --- a/packages/coding-agent/src/web/search/providers/gemini.ts +++ b/packages/coding-agent/src/web/search/providers/gemini.ts @@ -9,11 +9,7 @@ * endpoint. */ import { type AuthStorage, type FetchImpl, type OAuthAccess, withOAuthAccess } from "@oh-my-pi/pi-ai"; -import { - ANTIGRAVITY_SYSTEM_INSTRUCTION, - getAntigravityUserAgent, - getGeminiCliHeaders, -} from "@oh-my-pi/pi-catalog/wire/gemini-headers"; +import { getAntigravityUserAgent, getGeminiCliHeaders } from "@oh-my-pi/pi-catalog/wire/gemini-headers"; import { fetchWithRetry, USER_AGENT } from "@oh-my-pi/pi-utils"; import type { SearchCitation, SearchResponse, SearchSource } from "../../../web/search/types"; @@ -441,10 +437,9 @@ async function callGeminiSearch( }; const normalizedSystemPrompt = systemPrompt?.toWellFormed(); - const systemInstructionParts: Array<{ text: string }> = [ - ...(auth.isAntigravity ? [{ text: ANTIGRAVITY_SYSTEM_INSTRUCTION }] : []), - ...(normalizedSystemPrompt ? [{ text: normalizedSystemPrompt }] : []), - ]; + const systemInstructionParts: Array<{ text: string }> = normalizedSystemPrompt + ? [{ text: normalizedSystemPrompt }] + : []; const requestBody: Record = { project: auth.projectId,