feat: introduced dynamic version discovery and rate limit parsing for google
- Added dynamic Antigravity version discovery with environment variable overrides and fallback endpoint support. - Introduced structured Google RPC `ErrorInfo` rate limit reason parsing and backoff classification. - Added `gemini-3.7-flash` and `deepseek-v4-pro:preview` model configurations alongside updated pricing. - Removed system instruction injection logic and dropped unsigned thinking blocks for Antigravity requests.
This commit is contained in:
@@ -246,7 +246,10 @@ OAuth host chain: `KIMI_CODE_OAUTH_HOST` → `KIMI_OAUTH_HOST` → `https://auth
|
||||
| Variable | Default / behavior |
|
||||
| --------------------------- | --------------------------------------------------------------- |
|
||||
| `PI_AI_GEMINI_CLI_VERSION` | Overrides Gemini CLI user-agent version tag (`0.46.0` if unset) |
|
||||
| `PI_AI_ANTIGRAVITY_VERSION` | Overrides Antigravity hub user-agent version (`2.1.4` if unset) |
|
||||
| `PI_AI_ANTIGRAVITY_VERSION` | Overrides the auto-discovered Antigravity hub user-agent version; when unset and discovery fails, the fallback is `2.8.0` |
|
||||
| `PI_AI_ANTIGRAVITY_CL` | Overrides Antigravity hub user-agent build changelist (`963137146` if unset) |
|
||||
| `PI_AI_ANTIGRAVITY_OS` | Overrides Antigravity hub user-agent os_type (pinned `darwin` if unset) |
|
||||
| `PI_AI_ANTIGRAVITY_ARCH` | Overrides Antigravity hub user-agent arch (pinned `arm64` if unset) |
|
||||
|
||||
### GitLab Duo
|
||||
|
||||
|
||||
@@ -246,8 +246,8 @@ Google Cloud Code Assist (CCA) transport wrapper accessing Gemini and Claude mod
|
||||
- **Function Calling Config Mode**: Defaults to `functionCallingConfig: { mode: "VALIDATED" }` for Antigravity in `buildRequest`. Claude models on Antigravity force `VALIDATED` mode even when context contains no declared tools (`isClaudeModel`). Single named tool choice (`options.toolChoice`) sets `mode: "ANY"` with `allowedFunctionNames: [...]`.
|
||||
- **Provider Protocol & Request Envelope**:
|
||||
- **Endpoints**: `google-gemini-cli` defaults to `https://cloudcode-pa.googleapis.com`. `google-antigravity` uses auto-failover across `https://daily-cloudcode-pa.googleapis.com` (primary) and `https://daily-cloudcode-pa.sandbox.googleapis.com` (sandbox), persisting `lastGoodEndpoint` in `AntigravityProviderSessionState`.
|
||||
- **Headers & User-Agent**: `google-gemini-cli` sends `getGeminiCliHeaders()` (`GeminiCLI/0.46.0/<modelId> (platform; arch; terminal)`). `google-antigravity` sends `getAntigravityUserAgent()` (`antigravity/hub/2.1.4 <os>/<arch>`). Reasoning Claude models on Antigravity send `anthropic-beta: interleaved-thinking-2025-05-14` (`needsClaudeThinkingBetaHeader`).
|
||||
- **System Instructions**: Antigravity tags system instructions with `role: "user"`. Claude and Gemini 3 models prepend `ANTIGRAVITY_SYSTEM_INSTRUCTION` ("You are Antigravity, a powerful agentic AI coding assistant...") via `shouldInjectAntigravitySystemInstruction`.
|
||||
- **Headers & User-Agent**: `google-gemini-cli` sends `getGeminiCliHeaders()` (`GeminiCLI/0.46.0/<modelId> (platform; arch; terminal)`). `google-antigravity` sends `getAntigravityUserAgent()` (`antigravity/hub/<version> (aidev_client; os_type=<os>; arch=<arch>; cl=<cl>)`); the backend gates newer models (e.g. gemini-3.7-flash) on the client version. Reasoning Claude models on Antigravity send `anthropic-beta: interleaved-thinking-2025-05-14` (`needsClaudeThinkingBetaHeader`).
|
||||
- **System Instructions**: Antigravity tags system instructions with `role: "user"`. No identity prompt is injected — the backend accepts arbitrary system instructions on all routes (verified against gemini-3.x and Claude wire ids).
|
||||
- **Request Envelope & Session State**: Antigravity wraps requests in `buildAntigravityRequestEnvelope`: `project` (projectId), `requestId` (`agent/<agentId>/<ts>/<trajectoryId>/<step>`), `userAgent` (`antigravity`), `requestType` (`agent`), and `labels` (`last_step_index`, `model_enum`, `trajectory_id`, `used_claude`, `used_claude_conservative`, `last_execution_id`). State maintains monotonic `stepIndex`, persistent `agentId`, `trajectoryId`, and signed-decimal `sessionId` (`deriveAntigravitySessionId`).
|
||||
- **Wire Profiles**: `getAntigravityModelWireProfile` (`packages/catalog/src/wire/gemini-headers.ts`) maps wire IDs to `maxOutputTokens` and `model_enum`. Claude wire IDs cap `maxOutputTokens` at `64000` (backend rejects >64000 with 400).
|
||||
- **Thinking Configuration & Wire Suppression**: Gemini 2.x models send `thinkingConfig.thinkingBudget`, while Gemini 3 models send `thinkingConfig.thinkingLevel`. When reasoning is disabled for models with `thinking.suppressWhenOff`, `buildRequest` emits explicit wire suppression (`includeThoughts: false` with level/budget). Omitting `thinkingConfig` causes CCA to re-apply server defaults and silently bill thinking tokens.
|
||||
@@ -892,7 +892,7 @@ The Google Antigravity provider (`google-antigravity`) routes requests to Google
|
||||
|
||||
### Special casings
|
||||
- **Validated Function Calling Default**: Default tool selection mode in `buildRequest` (`packages/ai/src/providers/google-gemini-cli.ts`) is `VALIDATED` (`functionCallingConfig: { mode: "VALIDATED" }`). Claude models on Antigravity always force `VALIDATED` tool mode even when no tools are declared (`packages/ai/src/providers/google-gemini-cli.ts`).
|
||||
- **System Instruction & Request Envelope**: `shouldInjectAntigravitySystemInstruction` in `packages/ai/src/providers/google-gemini-cli.ts` prepends `ANTIGRAVITY_SYSTEM_INSTRUCTION` with `role: "user"` for Claude and Gemini 3 models. `buildAntigravityRequestEnvelope` injects structured `requestId` (`agent/<id>/<ts>/<trajectoryId>/<step>`), `userAgent: "antigravity"`, `requestType: "agent"`, `sessionId`, and `labels` (`model_enum`, `trajectory_id`, `last_step_index`, `last_execution_id`, `used_claude*`) using `getAntigravityModelWireProfile`.
|
||||
- **System Instruction & Request Envelope**: Antigravity tags `systemInstruction` with `role: "user"` and sends the caller's prompts unmodified. `buildAntigravityRequestEnvelope` injects structured `requestId` (`agent/<id>/<ts>/<trajectoryId>/<step>`), `userAgent: "antigravity"`, `requestType: "agent"`, `sessionId`, and `labels` (`model_enum`, `trajectory_id`, `last_step_index`, `last_execution_id`, `used_claude*`) using `getAntigravityModelWireProfile`.
|
||||
- **Endpoint Auto-Failover**: Operates across `ANTIGRAVITY_DAILY_ENDPOINT` (`https://daily-cloudcode-pa.googleapis.com`) and `ANTIGRAVITY_SANDBOX_ENDPOINT` (`https://daily-cloudcode-pa.sandbox.googleapis.com`) with state-tracked fallback in `getAntigravityProviderSessionState` (`packages/ai/src/providers/google-gemini-cli.ts`).
|
||||
|
||||
### Auth & usage
|
||||
|
||||
@@ -2,6 +2,16 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Dropped unsigned thinking blocks from Antigravity Claude requests instead of sending them without a signature, preventing HTTP 400 responses when resuming sessions or switching models.
|
||||
- Classified Antigravity HTTP 429 responses from structured `google.rpc.ErrorInfo` reasons (`QUOTA_EXHAUSTED`, `RATE_LIMIT_EXCEEDED`, and `INSUFFICIENT_G1_CREDITS_BALANCE`), using retry delays of five minutes or longer to distinguish rotatable quota windows from transient throttling instead of relying only on message regexes.
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed the Antigravity identity-prompt injection (`ANTIGRAVITY_SYSTEM_INSTRUCTION` and `shouldInjectAntigravitySystemInstruction`): Cloud Code Assist accepts arbitrary system instructions on gemini-3.x and Claude routes (verified live), and the injected stub never matched the real client's system prompt anyway. User system prompts are now sent unmodified (still tagged `role: "user"`).
|
||||
|
||||
|
||||
## [17.3.0] - 2026-08-13
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
import { extractRetryHint } from "@oh-my-pi/pi-utils";
|
||||
|
||||
/**
|
||||
* Rate limit reason classification and backoff calculation utilities.
|
||||
* Ported from opencode-antigravity-auth plugin for consistency.
|
||||
@@ -5,6 +7,7 @@
|
||||
|
||||
export type RateLimitReason =
|
||||
| "QUOTA_EXHAUSTED"
|
||||
| "INSUFFICIENT_G1_CREDITS_BALANCE"
|
||||
| "RATE_LIMIT_EXCEEDED"
|
||||
| "CONCURRENT_LIMIT"
|
||||
| "MODEL_CAPACITY_EXHAUSTED"
|
||||
@@ -67,6 +70,66 @@ const CN_TRANSIENT_CAP_PATTERN =
|
||||
// of rotating through the opaque-429 fallback.
|
||||
const CN_THROTTLE_PATTERN = /速率(?:限制|过快)|频率(?:过高|过快)|过于频繁|稍后[重再]试/;
|
||||
|
||||
const GOOGLE_RPC_ERROR_INFO_TYPE = "type.googleapis.com/google.rpc.ErrorInfo";
|
||||
const LONG_RATE_LIMIT_DELAY_MS = 5 * 60 * 1000;
|
||||
|
||||
function asRecord(value: unknown): Record<string, unknown> | undefined {
|
||||
return typeof value === "object" && value !== null && !Array.isArray(value)
|
||||
? (value as Record<string, unknown>)
|
||||
: undefined;
|
||||
}
|
||||
|
||||
function parseJsonBody(errorMessage: string): Record<string, unknown> | undefined {
|
||||
const start = errorMessage.indexOf("{");
|
||||
const end = errorMessage.lastIndexOf("}");
|
||||
if (start < 0 || end < start) return undefined;
|
||||
try {
|
||||
const parsed: unknown = JSON.parse(errorMessage.slice(start, end + 1));
|
||||
return asRecord(parsed);
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Classify structured Google RESOURCE_EXHAUSTED bodies before consulting text.
|
||||
* Cloud Code Assist prefixes the JSON with its HTTP error label, so accept an
|
||||
* embedded top-level object as well as a raw JSON body.
|
||||
*/
|
||||
function parseGoogleRpcRateLimitReason(errorMessage: string): RateLimitReason | undefined {
|
||||
const body = parseJsonBody(errorMessage);
|
||||
const error = asRecord(body?.error);
|
||||
if (typeof error?.status !== "string" || error.status.trim().toUpperCase() !== "RESOURCE_EXHAUSTED") {
|
||||
return undefined;
|
||||
}
|
||||
if (!Array.isArray(error.details)) return undefined;
|
||||
|
||||
for (const value of error.details) {
|
||||
const detail = asRecord(value);
|
||||
if (detail?.["@type"] !== GOOGLE_RPC_ERROR_INFO_TYPE || typeof detail.reason !== "string") continue;
|
||||
const reason = detail.reason.trim().toUpperCase();
|
||||
switch (reason) {
|
||||
case "QUOTA_EXHAUSTED":
|
||||
return "QUOTA_EXHAUSTED";
|
||||
case "INSUFFICIENT_G1_CREDITS_BALANCE":
|
||||
// Keep Google's specific credit-balance reason available to logs
|
||||
// and callers while treating it as credential-rotatable below.
|
||||
return "INSUFFICIENT_G1_CREDITS_BALANCE";
|
||||
case "RATE_LIMIT_EXCEEDED": {
|
||||
const retryDelayMs = extractRetryHint(undefined, errorMessage);
|
||||
return retryDelayMs !== undefined && retryDelayMs >= LONG_RATE_LIMIT_DELAY_MS
|
||||
? "QUOTA_EXHAUSTED"
|
||||
: "RATE_LIMIT_EXCEEDED";
|
||||
}
|
||||
}
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function isQuotaExhaustedReason(reason: RateLimitReason): boolean {
|
||||
return reason === "QUOTA_EXHAUSTED" || reason === "INSUFFICIENT_G1_CREDITS_BALANCE";
|
||||
}
|
||||
|
||||
/**
|
||||
* Classify a rate-limit error message into a reason category.
|
||||
* Priority order: explicit details in a resource-exhausted error > QUOTA
|
||||
@@ -77,6 +140,8 @@ const CN_THROTTLE_PATTERN = /速率(?:限制|过快)|频率(?:过高|过快)|过
|
||||
* Explicit details such as "quota exceeded" retain their normal classification.
|
||||
*/
|
||||
export function parseRateLimitReason(errorMessage: string): RateLimitReason {
|
||||
const structuredReason = parseGoogleRpcRateLimitReason(errorMessage);
|
||||
if (structuredReason !== undefined) return structuredReason;
|
||||
const lowerWithStatus = errorMessage.toLowerCase();
|
||||
const lower = lowerWithStatus.replace(RESOURCE_EXHAUSTED_PATTERN, "");
|
||||
const hasResourceExhaustedStatus = lower !== lowerWithStatus;
|
||||
@@ -162,6 +227,7 @@ export function parseRateLimitReason(errorMessage: string): RateLimitReason {
|
||||
*/
|
||||
export function calculateRateLimitBackoffMs(reason: RateLimitReason): number {
|
||||
switch (reason) {
|
||||
case "INSUFFICIENT_G1_CREDITS_BALANCE":
|
||||
case "QUOTA_EXHAUSTED":
|
||||
return QUOTA_EXHAUSTED_BACKOFF_MS;
|
||||
case "RATE_LIMIT_EXCEEDED":
|
||||
@@ -215,6 +281,8 @@ export function isUsageLimitStatus(status: number | undefined): boolean {
|
||||
* credentials.
|
||||
*/
|
||||
export function isUsageLimitOutcome(status: number | undefined, message: string | undefined): boolean {
|
||||
const structuredReason = message ? parseGoogleRpcRateLimitReason(message) : undefined;
|
||||
if (structuredReason !== undefined) return isQuotaExhaustedReason(structuredReason);
|
||||
// Concurrency caps are shed-and-backoff, not credential-rotatable — but only
|
||||
// for quota-worded 429 / other statuses. HTTP 402 is categorically an
|
||||
// account-billing cap, so a 402 whose body happens to mention concurrency is
|
||||
@@ -235,7 +303,7 @@ export function isUsageLimitOutcome(status: number | undefined, message: string
|
||||
const reason = parseRateLimitReason(message);
|
||||
// For the categorical 402 billing cap a concurrency-worded body is still an
|
||||
// exhausted cap (rotate); for 429 / other only QUOTA_EXHAUSTED rotates.
|
||||
return reason === "QUOTA_EXHAUSTED" || (isBillingCapStatus && reason === "CONCURRENT_LIMIT");
|
||||
return isQuotaExhaustedReason(reason) || (isBillingCapStatus && reason === "CONCURRENT_LIMIT");
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -272,6 +340,8 @@ export function isOpaqueStatusBody(message: string): boolean {
|
||||
* {@link isUsageLimitOutcome} uses it for the account-rotation decision.
|
||||
*/
|
||||
export function matchesUsageLimitText(errorMessage: string): boolean {
|
||||
const structuredReason = parseGoogleRpcRateLimitReason(errorMessage);
|
||||
if (structuredReason !== undefined) return isQuotaExhaustedReason(structuredReason);
|
||||
return (
|
||||
USAGE_LIMIT_PATTERN.test(errorMessage) ||
|
||||
(CN_QUOTA_EXHAUSTED_PATTERN.test(errorMessage) && !CN_TRANSIENT_CAP_PATTERN.test(errorMessage)) ||
|
||||
|
||||
@@ -8,7 +8,6 @@ import { scheduler } from "node:timers/promises";
|
||||
import { type } from "@oh-my-pi/omptype";
|
||||
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
||||
import {
|
||||
ANTIGRAVITY_SYSTEM_INSTRUCTION,
|
||||
getAntigravityModelWireProfile,
|
||||
getAntigravityUserAgent,
|
||||
getGeminiCliHeaders,
|
||||
@@ -316,13 +315,6 @@ const ANTIGRAVITY_DAILY_ENDPOINT = "https://daily-cloudcode-pa.googleapis.com";
|
||||
const ANTIGRAVITY_SANDBOX_ENDPOINT = "https://daily-cloudcode-pa.sandbox.googleapis.com";
|
||||
const ANTIGRAVITY_ENDPOINT_FALLBACKS = [ANTIGRAVITY_DAILY_ENDPOINT, ANTIGRAVITY_SANDBOX_ENDPOINT] as const;
|
||||
|
||||
export {
|
||||
ANTIGRAVITY_SYSTEM_INSTRUCTION,
|
||||
getAntigravityUserAgent,
|
||||
getGeminiCliHeaders,
|
||||
getGeminiCliUserAgent,
|
||||
} from "@oh-my-pi/pi-catalog/wire/gemini-headers";
|
||||
|
||||
// Retry configuration
|
||||
const MAX_RETRIES = 3;
|
||||
const BASE_DELAY_MS = 1000;
|
||||
@@ -342,11 +334,6 @@ function needsClaudeThinkingBetaHeader(model: Model<"google-gemini-cli">): boole
|
||||
return model.provider === "google-antigravity" && model.id.startsWith("claude-") && model.reasoning;
|
||||
}
|
||||
|
||||
function shouldInjectAntigravitySystemInstruction(modelId: string): boolean {
|
||||
const normalized = modelId.toLowerCase();
|
||||
return normalized.includes("claude") || normalized.includes("gemini-3");
|
||||
}
|
||||
|
||||
const optionalCredentialString = type("unknown").pipe(raw => {
|
||||
const out = type("string")(raw);
|
||||
return out instanceof type.errors ? undefined : out;
|
||||
@@ -1320,14 +1307,6 @@ export function buildRequest(
|
||||
};
|
||||
}
|
||||
|
||||
if (isAntigravity && shouldInjectAntigravitySystemInstruction(model.id)) {
|
||||
const existingParts = request.systemInstruction?.parts ?? [];
|
||||
request.systemInstruction = {
|
||||
role: "user",
|
||||
parts: [{ text: ANTIGRAVITY_SYSTEM_INSTRUCTION }, ...existingParts],
|
||||
};
|
||||
}
|
||||
|
||||
if (context.tools && context.tools.length > 0) {
|
||||
const convertedTools = convertTools(context.tools, model);
|
||||
request.tools = isAntigravity ? normalizeAntigravityTools(convertedTools) : convertedTools;
|
||||
|
||||
@@ -233,6 +233,8 @@ export function convertMessages<T extends GoogleApiType>(model: Model<T>, contex
|
||||
const parts: Part[] = [];
|
||||
// Check if message is from same provider and model - only then keep thinking blocks
|
||||
const isSameProviderAndModel = msg.provider === model.provider && msg.model === model.id;
|
||||
const dropsUnsignedThinking =
|
||||
model.provider === "google-antigravity" && model.id.toLowerCase().includes("claude");
|
||||
|
||||
for (const block of msg.content) {
|
||||
if (block.type === "text") {
|
||||
@@ -247,6 +249,7 @@ export function convertMessages<T extends GoogleApiType>(model: Model<T>, contex
|
||||
// Skip empty thinking blocks
|
||||
if (!block.thinking || block.thinking.trim() === "") continue;
|
||||
const thoughtSignature = resolveThoughtSignature(isSameProviderAndModel, block.thinkingSignature);
|
||||
if (dropsUnsignedThinking && !thoughtSignature) continue;
|
||||
if (thoughtSignature) {
|
||||
parts.push({
|
||||
thought: true,
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
* Antigravity OAuth flow (Gemini 3, Claude, GPT-OSS via Google Cloud)
|
||||
* Uses different OAuth credentials than google-gemini-cli for access to additional models.
|
||||
*/
|
||||
import { getAntigravityUserAgent } from "@oh-my-pi/pi-catalog/wire/gemini-headers";
|
||||
import { getAntigravityUserAgent, getAntigravityVersion } from "@oh-my-pi/pi-catalog/wire/gemini-headers";
|
||||
import * as AIError from "../../error";
|
||||
import { oauthFetch, runGoogleOAuthLogin, throwIfLoginCancelled } from "./google-oauth-shared";
|
||||
import type { OAuthController, OAuthCredentials } from "./types";
|
||||
@@ -26,10 +26,12 @@ const SCOPES = [
|
||||
const AUTH_URL = "https://accounts.google.com/o/oauth2/v2/auth";
|
||||
const TOKEN_URL = "https://oauth2.googleapis.com/token";
|
||||
const CLOUD_CODE_ENDPOINT = "https://cloudcode-pa.googleapis.com";
|
||||
const TIER_LEGACY = "legacy-tier";
|
||||
const DAILY_CLOUD_CODE_ENDPOINT = "https://daily-cloudcode-pa.googleapis.com";
|
||||
const NODE_API_CLIENT_USER_AGENT = "google-api-nodejs-client/10.3.0";
|
||||
const GOOG_API_CLIENT_HEADER = "gl-node/22.21.1";
|
||||
const TIER_FREE = "free-tier";
|
||||
const PROJECT_ONBOARD_MAX_ATTEMPTS = 5;
|
||||
const PROJECT_ONBOARD_INTERVAL_MS = 2000;
|
||||
|
||||
interface LoadCodeAssistPayload {
|
||||
cloudaicompanionProject?: string | { id?: string };
|
||||
currentTier?: { id?: string };
|
||||
@@ -45,35 +47,65 @@ interface LongRunningOperationResponse {
|
||||
|
||||
export const ANTIGRAVITY_LOAD_CODE_ASSIST_METADATA = Object.freeze({
|
||||
ideType: "ANTIGRAVITY",
|
||||
platform: "PLATFORM_UNSPECIFIED",
|
||||
pluginType: "GEMINI",
|
||||
});
|
||||
|
||||
function readProjectId(value: string | { id?: string } | undefined): string | undefined {
|
||||
export interface AntigravityOnboardMetadata {
|
||||
ide_type: string;
|
||||
ide_version: string;
|
||||
ide_name: string;
|
||||
}
|
||||
|
||||
export function getAntigravityOnboardMetadata(): AntigravityOnboardMetadata {
|
||||
return {
|
||||
ide_type: "ANTIGRAVITY",
|
||||
ide_version: getAntigravityVersion(),
|
||||
ide_name: "antigravity",
|
||||
};
|
||||
}
|
||||
|
||||
function readProjectId(value: unknown): string | undefined {
|
||||
if (typeof value === "string" && value.length > 0) {
|
||||
return value;
|
||||
return value.trim();
|
||||
}
|
||||
if (value && typeof value === "object" && typeof value.id === "string" && value.id.length > 0) {
|
||||
return value.id;
|
||||
if (value && typeof value === "object" && "id" in value && typeof (value as { id?: unknown }).id === "string") {
|
||||
const id = (value as { id: string }).id.trim();
|
||||
if (id.length > 0) return id;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function getDefaultTierId(allowedTiers?: Array<{ id?: string; isDefault?: boolean }>): string {
|
||||
if (!allowedTiers || allowedTiers.length === 0) {
|
||||
return TIER_LEGACY;
|
||||
function extractProjectId(payload: unknown): string | undefined {
|
||||
if (!payload || typeof payload !== "object") return undefined;
|
||||
const record = payload as Record<string, unknown>;
|
||||
for (const key of ["cloudaicompanionProject", "projectId", "project"]) {
|
||||
const id = readProjectId(record[key]);
|
||||
if (id) return id;
|
||||
}
|
||||
const defaultTier = allowedTiers.find(tier => tier.isDefault && typeof tier.id === "string" && tier.id.length > 0);
|
||||
if (defaultTier?.id) {
|
||||
return defaultTier.id;
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function getDefaultTierId(
|
||||
allowedTiers?: Array<{ id?: string; isDefault?: boolean }>,
|
||||
currentTier?: { id?: string },
|
||||
): string {
|
||||
if (allowedTiers && allowedTiers.length > 0) {
|
||||
const defaultTier = allowedTiers.find(
|
||||
tier => tier.isDefault && typeof tier.id === "string" && tier.id.trim().length > 0,
|
||||
);
|
||||
if (defaultTier?.id) {
|
||||
return defaultTier.id.trim();
|
||||
}
|
||||
}
|
||||
return TIER_LEGACY;
|
||||
if (currentTier && typeof currentTier.id === "string" && currentTier.id.trim().length > 0) {
|
||||
return currentTier.id.trim();
|
||||
}
|
||||
return TIER_FREE;
|
||||
}
|
||||
|
||||
async function onboardProjectWithRetries(
|
||||
endpoint: string,
|
||||
headers: Record<string, string>,
|
||||
onboardBody: { tierId: string; metadata: typeof ANTIGRAVITY_LOAD_CODE_ASSIST_METADATA },
|
||||
onboardBody: { tier_id: string; metadata: AntigravityOnboardMetadata },
|
||||
signal: AbortSignal | undefined,
|
||||
onProgress?: (message: string) => void,
|
||||
): Promise<string> {
|
||||
@@ -104,7 +136,7 @@ async function onboardProjectWithRetries(
|
||||
continue;
|
||||
}
|
||||
|
||||
const projectId = readProjectId(operation.response?.cloudaicompanionProject);
|
||||
const projectId = extractProjectId(operation.response);
|
||||
if (projectId) {
|
||||
return projectId;
|
||||
}
|
||||
@@ -128,42 +160,65 @@ async function discoverProject(
|
||||
};
|
||||
|
||||
onProgress?.("Checking for existing project...");
|
||||
const endpoint = CLOUD_CODE_ENDPOINT;
|
||||
try {
|
||||
throwIfLoginCancelled(signal);
|
||||
const loadResponse = await oauthFetch(
|
||||
`${endpoint}/v1internal:loadCodeAssist`,
|
||||
{
|
||||
method: "POST",
|
||||
headers,
|
||||
body: JSON.stringify({
|
||||
metadata: ANTIGRAVITY_LOAD_CODE_ASSIST_METADATA,
|
||||
}),
|
||||
},
|
||||
{ provider: "google-antigravity", signal },
|
||||
);
|
||||
let lastErrorText: string | undefined;
|
||||
let lastStatus: number | undefined;
|
||||
let fallbackTierId = TIER_FREE;
|
||||
let loadedSuccessfully = false;
|
||||
|
||||
if (!loadResponse.ok) {
|
||||
const errorText = await loadResponse.text();
|
||||
throw new AIError.OAuthError(
|
||||
`loadCodeAssist failed: ${loadResponse.status} ${loadResponse.statusText}: ${errorText}`,
|
||||
{ kind: "discovery", status: loadResponse.status },
|
||||
for (const endpoint of [DAILY_CLOUD_CODE_ENDPOINT, CLOUD_CODE_ENDPOINT]) {
|
||||
throwIfLoginCancelled(signal);
|
||||
const loadResponse = await oauthFetch(
|
||||
`${endpoint}/v1internal:loadCodeAssist`,
|
||||
{
|
||||
method: "POST",
|
||||
headers,
|
||||
body: JSON.stringify({
|
||||
metadata: ANTIGRAVITY_LOAD_CODE_ASSIST_METADATA,
|
||||
}),
|
||||
},
|
||||
{ provider: "google-antigravity", signal },
|
||||
);
|
||||
|
||||
if (!loadResponse.ok) {
|
||||
lastStatus = loadResponse.status;
|
||||
lastErrorText = await loadResponse.text();
|
||||
continue;
|
||||
}
|
||||
|
||||
loadedSuccessfully = true;
|
||||
const loadPayload = (await loadResponse.json()) as LoadCodeAssistPayload;
|
||||
const existingProject = extractProjectId(loadPayload);
|
||||
if (existingProject) {
|
||||
return existingProject;
|
||||
}
|
||||
fallbackTierId = getDefaultTierId(loadPayload.allowedTiers, loadPayload.currentTier);
|
||||
}
|
||||
|
||||
const loadPayload = (await loadResponse.json()) as LoadCodeAssistPayload;
|
||||
const existingProject = readProjectId(loadPayload.cloudaicompanionProject);
|
||||
if (existingProject) {
|
||||
return existingProject;
|
||||
if (!loadedSuccessfully && lastStatus !== undefined) {
|
||||
throw new AIError.OAuthError(`loadCodeAssist failed: ${lastStatus}: ${lastErrorText || "unknown error"}`, {
|
||||
kind: "discovery",
|
||||
status: lastStatus,
|
||||
});
|
||||
}
|
||||
|
||||
const tierId = getDefaultTierId(loadPayload.allowedTiers);
|
||||
onProgress?.("Provisioning project...");
|
||||
const onboardBody = {
|
||||
tierId,
|
||||
metadata: ANTIGRAVITY_LOAD_CODE_ASSIST_METADATA,
|
||||
tier_id: fallbackTierId,
|
||||
metadata: getAntigravityOnboardMetadata(),
|
||||
};
|
||||
const provisionedProject = await onboardProjectWithRetries(endpoint, headers, onboardBody, signal, onProgress);
|
||||
const onboardHeaders: Record<string, string> = {
|
||||
...headers,
|
||||
"User-Agent": `${headers["User-Agent"]} ${NODE_API_CLIENT_USER_AGENT}`,
|
||||
"X-Goog-Api-Client": GOOG_API_CLIENT_HEADER,
|
||||
};
|
||||
const provisionedProject = await onboardProjectWithRetries(
|
||||
DAILY_CLOUD_CODE_ENDPOINT,
|
||||
onboardHeaders,
|
||||
onboardBody,
|
||||
signal,
|
||||
onProgress,
|
||||
);
|
||||
return provisionedProject;
|
||||
} catch (error) {
|
||||
if (error instanceof AIError.LoginCancelledError || error instanceof AIError.OAuthError) {
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { getGeminiCliHeaders } from "../providers/google-gemini-cli";
|
||||
import { getGeminiCliHeaders } from "@oh-my-pi/pi-catalog/wire/gemini-headers";
|
||||
import type {
|
||||
UsageAmount,
|
||||
UsageFetchContext,
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import * as geminiCliProvider from "@oh-my-pi/pi-ai/providers/google-gemini-cli";
|
||||
import {
|
||||
ANTIGRAVITY_SYSTEM_INSTRUCTION,
|
||||
buildRequest,
|
||||
parseGeminiCliCredentials,
|
||||
shouldRefreshGeminiCliCredentials,
|
||||
@@ -189,9 +188,70 @@ describe("Google Gemini CLI alignment", () => {
|
||||
expect(payload.request.contents).toEqual([{ role: "user", parts: [{ text: "implement token refresh" }] }]);
|
||||
});
|
||||
|
||||
it("drops only unsigned thinking when replaying Antigravity Claude history", () => {
|
||||
const signedThinking = "signed reasoning";
|
||||
const unsignedThinking = "unsigned reasoning";
|
||||
const signature = "c2lnbmVk";
|
||||
const createThinkingContext = (model: Model<"google-gemini-cli">): Context => ({
|
||||
messages: [
|
||||
{
|
||||
role: "assistant",
|
||||
content: [
|
||||
{ type: "thinking", thinking: signedThinking, thinkingSignature: signature },
|
||||
{ type: "thinking", thinking: unsignedThinking },
|
||||
],
|
||||
api: model.api,
|
||||
provider: model.provider,
|
||||
model: model.id,
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "stop",
|
||||
timestamp: 1,
|
||||
},
|
||||
{ role: "user", content: "continue", timestamp: 2 },
|
||||
],
|
||||
});
|
||||
const claudeModel = buildModel({
|
||||
...createModel("google-antigravity"),
|
||||
id: "claude-sonnet-4-6",
|
||||
name: "Claude Sonnet 4.6",
|
||||
reasoning: true,
|
||||
} as ModelSpec<"google-gemini-cli">);
|
||||
const claudePayload = buildRequest(claudeModel, createThinkingContext(claudeModel), "proj-123", {}, true) as {
|
||||
request: {
|
||||
contents: Array<{
|
||||
role: string;
|
||||
parts: Array<{ text?: string; thought?: boolean; thoughtSignature?: string }>;
|
||||
}>;
|
||||
};
|
||||
};
|
||||
const claudeParts = claudePayload.request.contents.find(content => content.role === "model")?.parts;
|
||||
expect(claudeParts).toEqual([{ thought: true, text: signedThinking, thoughtSignature: signature }]);
|
||||
|
||||
const geminiModel = createModel("google-antigravity");
|
||||
const geminiPayload = buildRequest(geminiModel, createThinkingContext(geminiModel), "proj-123", {}, true) as {
|
||||
request: {
|
||||
contents: Array<{
|
||||
role: string;
|
||||
parts: Array<{ text?: string; thought?: boolean; thoughtSignature?: string }>;
|
||||
}>;
|
||||
};
|
||||
};
|
||||
const geminiParts = geminiPayload.request.contents.find(content => content.role === "model")?.parts ?? [];
|
||||
expect(geminiParts).toContainEqual({ thought: true, text: signedThinking, thoughtSignature: signature });
|
||||
expect(geminiParts.some(part => part.text?.includes(unsignedThinking))).toBe(true);
|
||||
});
|
||||
|
||||
it("keeps antigravity metadata in antigravity request payloads", () => {
|
||||
const model = createModel("google-antigravity");
|
||||
const payload = buildRequest(model, createContext(), "proj-123", {}, true) as {
|
||||
const context: Context = { ...createContext(), systemPrompt: ["be terse"] };
|
||||
const payload = buildRequest(model, context, "proj-123", {}, true) as {
|
||||
request: {
|
||||
sessionId?: string;
|
||||
labels?: Record<string, string>;
|
||||
@@ -324,30 +384,6 @@ describe("Google Gemini CLI alignment", () => {
|
||||
expect(parameters).toBeDefined();
|
||||
expect(JSON.stringify(parameters)).not.toContain('"patternProperties"');
|
||||
});
|
||||
it("injects ANTIGRAVITY_SYSTEM_INSTRUCTION for gemini-3.1-pro-high and gemini-3.1-pro-low", () => {
|
||||
// Regression test for #1274: shouldInjectAntigravitySystemInstruction checked
|
||||
// "gemini-3-pro-high" (hyphen) but the deployed model IDs use "gemini-3.1-pro-high" (dot),
|
||||
// so the injection was silently skipped and the Cloud Code Assist API returned HTTP 400.
|
||||
for (const modelId of ["gemini-3.1-pro-high", "gemini-3.1-pro-low"] as const) {
|
||||
const model: Model<"google-gemini-cli"> = buildModel({
|
||||
...createModel("google-antigravity"),
|
||||
id: modelId,
|
||||
} as ModelSpec<"google-gemini-cli">);
|
||||
const context: Context = {
|
||||
systemPrompt: ["my instructions"],
|
||||
messages: [{ role: "user", content: "hi", timestamp: Date.now() }],
|
||||
};
|
||||
const payload = buildRequest(model, context, "proj-123", {}, true) as {
|
||||
request: { systemInstruction?: { role?: string; parts: Array<{ text: string }> } };
|
||||
};
|
||||
|
||||
const parts = payload.request.systemInstruction?.parts ?? [];
|
||||
// The antigravity identity header must be injected as the first part.
|
||||
expect(parts[0]?.text).toBe(ANTIGRAVITY_SYSTEM_INSTRUCTION);
|
||||
// The user-supplied system prompt must appear after the single injected part.
|
||||
expect(parts.slice(1).some(p => p.text === "my instructions")).toBe(true);
|
||||
}
|
||||
});
|
||||
it("adds anthropic-beta for Antigravity Claude reasoning models without relying on id suffix", async () => {
|
||||
let requestHeaders: Headers | undefined;
|
||||
const fetchMock: FetchImpl = async (_url, init) => {
|
||||
|
||||
@@ -10,6 +10,30 @@ import {
|
||||
parseRateLimitReason,
|
||||
} from "@oh-my-pi/pi-ai/error/rate-limit";
|
||||
|
||||
function googleRpc429(reason: string, retryDelay?: string, message = "Resource exhausted"): string {
|
||||
const details: Array<Record<string, string>> = [
|
||||
{
|
||||
"@type": "type.googleapis.com/google.rpc.ErrorInfo",
|
||||
reason,
|
||||
domain: "cloudcode-pa.googleapis.com",
|
||||
},
|
||||
];
|
||||
if (retryDelay) {
|
||||
details.push({
|
||||
"@type": "type.googleapis.com/google.rpc.RetryInfo",
|
||||
retryDelay,
|
||||
});
|
||||
}
|
||||
return `Cloud Code Assist API error (429): ${JSON.stringify({
|
||||
error: {
|
||||
code: 429,
|
||||
message,
|
||||
status: "RESOURCE_EXHAUSTED",
|
||||
details,
|
||||
},
|
||||
})}`;
|
||||
}
|
||||
|
||||
describe("parseRateLimitReason", () => {
|
||||
it("classifies Google Quota exceeded as QUOTA_EXHAUSTED", () => {
|
||||
expect(
|
||||
@@ -170,6 +194,54 @@ describe("parseRateLimitReason", () => {
|
||||
),
|
||||
).toBe("QUOTA_EXHAUSTED");
|
||||
});
|
||||
|
||||
it("uses structured QUOTA_EXHAUSTED before capacity message heuristics", () => {
|
||||
const body = googleRpc429(
|
||||
"QUOTA_EXHAUSTED",
|
||||
"21600s",
|
||||
"The model has no capacity available; retry another request later.",
|
||||
);
|
||||
expect(parseRateLimitReason(body)).toBe("QUOTA_EXHAUSTED");
|
||||
expect(isUsageLimitOutcome(429, body)).toBe(true);
|
||||
});
|
||||
|
||||
it("keeps structured RATE_LIMIT_EXCEEDED with a 30s delay transient", () => {
|
||||
const body = googleRpc429("RATE_LIMIT_EXCEEDED", "30s", "Too many requests");
|
||||
expect(parseRateLimitReason(body)).toBe("RATE_LIMIT_EXCEEDED");
|
||||
expect(isUsageLimitOutcome(429, body)).toBe(false);
|
||||
expect(isUsageLimit(Object.assign(new Error(body), { status: 429 }))).toBe(false);
|
||||
});
|
||||
|
||||
it("keeps structured RATE_LIMIT_EXCEEDED without a retry delay transient", () => {
|
||||
const body = googleRpc429("RATE_LIMIT_EXCEEDED", undefined, "Too many requests");
|
||||
expect(parseRateLimitReason(body)).toBe("RATE_LIMIT_EXCEEDED");
|
||||
expect(isUsageLimitOutcome(429, body)).toBe(false);
|
||||
});
|
||||
|
||||
it("treats structured RATE_LIMIT_EXCEEDED with a 6h delay as usage exhaustion", () => {
|
||||
const body = googleRpc429("RATE_LIMIT_EXCEEDED", "21600s", "Too many requests");
|
||||
expect(parseRateLimitReason(body)).toBe("QUOTA_EXHAUSTED");
|
||||
expect(isUsageLimitOutcome(429, body)).toBe(true);
|
||||
});
|
||||
|
||||
it("treats the five-minute structured rate-limit threshold as usage exhaustion", () => {
|
||||
const body = googleRpc429("RATE_LIMIT_EXCEEDED", "300s", "Too many requests");
|
||||
expect(parseRateLimitReason(body)).toBe("QUOTA_EXHAUSTED");
|
||||
expect(isUsageLimitOutcome(429, body)).toBe(true);
|
||||
});
|
||||
|
||||
it("preserves structured INSUFFICIENT_G1_CREDITS_BALANCE while rotating credentials", () => {
|
||||
const body = googleRpc429("INSUFFICIENT_G1_CREDITS_BALANCE", undefined, "Credit balance is unavailable");
|
||||
expect(parseRateLimitReason(body)).toBe("INSUFFICIENT_G1_CREDITS_BALANCE");
|
||||
expect(isUsageLimitOutcome(429, body)).toBe(true);
|
||||
expect(isUsageLimit(Object.assign(new Error(body), { status: 429 }))).toBe(true);
|
||||
});
|
||||
|
||||
it("falls back to existing text heuristics for non-JSON 429 bodies", () => {
|
||||
const body = "Cloud Code Assist API error (429): Too many requests";
|
||||
expect(parseRateLimitReason(body)).toBe("RATE_LIMIT_EXCEEDED");
|
||||
expect(isUsageLimitOutcome(429, body)).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("isUsageLimit", () => {
|
||||
|
||||
@@ -2,6 +2,20 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added support for the `deepseek-v4-pro:preview` model
|
||||
- Added support for the `gemini-3.7-flash` model
|
||||
- Added dynamic Antigravity client-version discovery from the official update manifest (darwin/arm64 channel), so version-gated models appear without a code change; `PI_AI_ANTIGRAVITY_VERSION` remains available as an override.
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed `ANTIGRAVITY_SYSTEM_INSTRUCTION` from `wire/gemini-headers`; the Antigravity transport and web search no longer inject a fake identity prompt.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Antigravity discovery missing Gemini 3.7 Flash: Cloud Code Assist gates newer models on the client version in the `User-Agent`, and the pinned `antigravity/hub/2.1.4` was too old. The user-agent now matches the captured 2.8.0 client format (`antigravity/hub/2.8.0 (aidev_client; os_type=darwin; arch=arm64; cl=963137146)`); os_type/arch stay pinned to the darwin/arm64 reference client. Overridable via `PI_AI_ANTIGRAVITY_VERSION` / `PI_AI_ANTIGRAVITY_CL` / `PI_AI_ANTIGRAVITY_OS` / `PI_AI_ANTIGRAVITY_ARCH`.
|
||||
|
||||
## [17.3.1] - 2026-08-13
|
||||
|
||||
### Added
|
||||
|
||||
@@ -6,7 +6,7 @@ import {
|
||||
collapseEffortVariants,
|
||||
type VariantCollapseTable,
|
||||
} from "../variant-collapse";
|
||||
import { getAntigravityUserAgent } from "../wire/gemini-headers";
|
||||
import { ensureAntigravityVersion, getAntigravityUserAgent } from "../wire/gemini-headers";
|
||||
|
||||
export const ANTIGRAVITY_PRIMARY_ENDPOINT = "https://daily-cloudcode-pa.googleapis.com";
|
||||
export const ANTIGRAVITY_SANDBOX_ENDPOINT = "https://daily-cloudcode-pa.sandbox.googleapis.com";
|
||||
@@ -161,6 +161,10 @@ export interface FetchAntigravityDiscoveryModelsOptions {
|
||||
export async function fetchAntigravityDiscoveryModels(
|
||||
options: FetchAntigravityDiscoveryModelsOptions,
|
||||
): Promise<ModelSpec<"google-gemini-cli">[] | null> {
|
||||
if (options.userAgent === undefined) {
|
||||
await ensureAntigravityVersion(fetch, options.signal);
|
||||
}
|
||||
|
||||
const fetcher = discoveryFetch(options.fetcher);
|
||||
const endpoints = options.endpoint
|
||||
? [trimTrailingSlashes(options.endpoint)]
|
||||
|
||||
@@ -25988,6 +25988,43 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"gemini-3.7-flash": {
|
||||
"id": "gemini-3.7-flash",
|
||||
"name": "Gemini 3.7 Flash",
|
||||
"api": "google-gemini-cli",
|
||||
"provider": "google-antigravity",
|
||||
"baseUrl": "https://daily-cloudcode-pa.googleapis.com",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.75,
|
||||
"output": 3.75,
|
||||
"cacheRead": 0.075,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 65536,
|
||||
"requestModelId": "gemini-3.7-flash-low",
|
||||
"thinking": {
|
||||
"mode": "google-level",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"requiresEffort": true,
|
||||
"effortRouting": {
|
||||
"minimal": "gemini-3.7-flash-low",
|
||||
"low": "gemini-3.7-flash-low",
|
||||
"medium": "gemini-3.7-flash-medium",
|
||||
"high": "gemini-3.7-flash-high"
|
||||
}
|
||||
}
|
||||
},
|
||||
"gpt-oss-120b": {
|
||||
"id": "gpt-oss-120b",
|
||||
"name": "GPT OSS 120B",
|
||||
@@ -29128,12 +29165,12 @@
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.079996,
|
||||
"output": 0.252,
|
||||
"cacheRead": 0.0252,
|
||||
"input": 0.072,
|
||||
"output": 0.144,
|
||||
"cacheRead": 0.016,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 262144,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
@@ -29156,10 +29193,10 @@
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 1.5,
|
||||
"output": 7.5,
|
||||
"cacheRead": 0.15,
|
||||
"cacheWrite": 0.083333
|
||||
"input": 0.375,
|
||||
"output": 1.875,
|
||||
"cacheRead": 0.0375,
|
||||
"cacheWrite": 0.020833
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 65536,
|
||||
@@ -29216,13 +29253,13 @@
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 2.8,
|
||||
"output": 14,
|
||||
"cacheRead": 0.29,
|
||||
"input": 2.4,
|
||||
"output": 12,
|
||||
"cacheRead": 0.24,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 1048576,
|
||||
"contextWindow": 974842,
|
||||
"maxTokens": 974842,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
@@ -32370,10 +32407,10 @@
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 1.5,
|
||||
"output": 7.5,
|
||||
"cacheRead": 0.15,
|
||||
"cacheWrite": 0.083333
|
||||
"input": 0.75,
|
||||
"output": 3.75,
|
||||
"cacheRead": 0.075,
|
||||
"cacheWrite": 0.041667
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 65536,
|
||||
@@ -32394,19 +32431,29 @@
|
||||
"api": "openai-completions",
|
||||
"provider": "kilo",
|
||||
"baseUrl": "https://api.kilo.ai/api/gateway",
|
||||
"reasoning": false,
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
"input": 0.75,
|
||||
"output": 3.75,
|
||||
"cacheRead": 0.075,
|
||||
"cacheWrite": 0.041667
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 65536,
|
||||
"supportsComputerUse": false
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"requiresEffort": true
|
||||
}
|
||||
},
|
||||
"google/gemma-2-27b-it": {
|
||||
"id": "google/gemma-2-27b-it",
|
||||
@@ -41453,9 +41500,9 @@
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.55,
|
||||
"output": 2.2,
|
||||
"cacheRead": 0.11,
|
||||
"input": 0.5,
|
||||
"output": 2,
|
||||
"cacheRead": 0.1,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 202752,
|
||||
@@ -41533,9 +41580,9 @@
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.6,
|
||||
"output": 2.2,
|
||||
"cacheRead": 0.11,
|
||||
"input": 0.4,
|
||||
"output": 1.75,
|
||||
"cacheRead": 0.08,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 202752,
|
||||
@@ -41562,7 +41609,7 @@
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.07,
|
||||
"input": 0.06,
|
||||
"output": 0.4,
|
||||
"cacheRead": 0.01,
|
||||
"cacheWrite": 0
|
||||
@@ -41591,8 +41638,8 @@
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 1,
|
||||
"output": 3.2,
|
||||
"input": 0.95,
|
||||
"output": 2.55,
|
||||
"cacheRead": 0.2,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
@@ -41649,7 +41696,7 @@
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 1.38,
|
||||
"input": 1.4,
|
||||
"output": 4.4,
|
||||
"cacheRead": 0.26,
|
||||
"cacheWrite": 0
|
||||
@@ -41683,7 +41730,7 @@
|
||||
"cacheRead": 0.26,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 131072,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
@@ -51050,7 +51097,7 @@
|
||||
"input": 1.5,
|
||||
"output": 9,
|
||||
"cacheRead": 0.15,
|
||||
"cacheWrite": 0
|
||||
"cacheWrite": 0.083333
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 65536,
|
||||
@@ -51110,7 +51157,37 @@
|
||||
"input": 1.5,
|
||||
"output": 7.5,
|
||||
"cacheRead": 0.15,
|
||||
"cacheWrite": 0
|
||||
"cacheWrite": 0.041667
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 65536,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"requiresEffort": true
|
||||
}
|
||||
},
|
||||
"google/gemini-3.7-flash": {
|
||||
"id": "google/gemini-3.7-flash",
|
||||
"name": "Gemini 3.7 Flash",
|
||||
"api": "openai-completions",
|
||||
"provider": "nanogpt",
|
||||
"baseUrl": "https://nano-gpt.com/api/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.375,
|
||||
"output": 1.875,
|
||||
"cacheRead": 0.0375,
|
||||
"cacheWrite": 0.020833
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 65536,
|
||||
@@ -71596,6 +71673,35 @@
|
||||
]
|
||||
}
|
||||
},
|
||||
"deepseek-v4-pro:preview": {
|
||||
"id": "deepseek-v4-pro:preview",
|
||||
"name": "deepseek-v4-pro:preview",
|
||||
"api": "ollama-chat",
|
||||
"provider": "ollama-cloud",
|
||||
"baseUrl": "https://ollama.com",
|
||||
"reasoning": true,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"low",
|
||||
"high",
|
||||
"max"
|
||||
]
|
||||
},
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 524288,
|
||||
"maxTokens": 65536,
|
||||
"omitMaxOutputTokens": true,
|
||||
"supportsComputerUse": false
|
||||
},
|
||||
"devstral-2:123b": {
|
||||
"id": "devstral-2:123b",
|
||||
"name": "devstral-2:123b",
|
||||
@@ -76127,6 +76233,36 @@
|
||||
"requiresEffort": true
|
||||
}
|
||||
},
|
||||
"gemini-3.7-flash": {
|
||||
"id": "gemini-3.7-flash",
|
||||
"name": "Gemini 3.7 Flash",
|
||||
"api": "google-generative-ai",
|
||||
"provider": "opencode-zen",
|
||||
"baseUrl": "https://opencode.ai/zen/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 1.5,
|
||||
"output": 7.5,
|
||||
"cacheRead": 0.15,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 65536,
|
||||
"thinking": {
|
||||
"mode": "google-level",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"requiresEffort": true
|
||||
}
|
||||
},
|
||||
"glm-4.6": {
|
||||
"id": "glm-4.6",
|
||||
"name": "GLM-4.6",
|
||||
@@ -77974,13 +78110,13 @@
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.079996,
|
||||
"output": 0.252,
|
||||
"cacheRead": 0.0252,
|
||||
"input": 0.072,
|
||||
"output": 0.144,
|
||||
"cacheRead": 0.016,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 131072,
|
||||
"maxTokens": 262144,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
@@ -78007,7 +78143,7 @@
|
||||
"input": 0.375,
|
||||
"output": 1.875,
|
||||
"cacheRead": 0.0375,
|
||||
"cacheWrite": 0.0416666666666667
|
||||
"cacheWrite": 0.0208333333333333
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 65536,
|
||||
@@ -78066,13 +78202,13 @@
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 2.8,
|
||||
"output": 14,
|
||||
"cacheRead": 0.29,
|
||||
"input": 2.4,
|
||||
"output": 12,
|
||||
"cacheRead": 0.24,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 1048576,
|
||||
"maxTokens": 974842,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
@@ -81235,10 +81371,10 @@
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 1.5,
|
||||
"output": 7.5,
|
||||
"cacheRead": 0.15,
|
||||
"cacheWrite": 0.0833333333333333
|
||||
"input": 0.75,
|
||||
"output": 3.75,
|
||||
"cacheRead": 0.075,
|
||||
"cacheWrite": 0.0416666666666667
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 65536,
|
||||
@@ -81267,10 +81403,10 @@
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.75,
|
||||
"output": 3.75,
|
||||
"cacheRead": 0.075,
|
||||
"cacheWrite": 0.0833333333333333
|
||||
"input": 0.375,
|
||||
"output": 1.875,
|
||||
"cacheRead": 0.0375,
|
||||
"cacheWrite": 0.0416666666666667
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 65536,
|
||||
@@ -81302,7 +81438,7 @@
|
||||
"input": 0.375,
|
||||
"output": 1.875,
|
||||
"cacheRead": 0.0375,
|
||||
"cacheWrite": 0.0416666666666667
|
||||
"cacheWrite": 0.0208333333333333
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 65536,
|
||||
@@ -81331,10 +81467,10 @@
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.375,
|
||||
"output": 1.875,
|
||||
"cacheRead": 0.0375,
|
||||
"cacheWrite": 0.0416666666666667
|
||||
"input": 0.1875,
|
||||
"output": 0.9375,
|
||||
"cacheRead": 0.01875,
|
||||
"cacheWrite": 0.0208333333333333
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 65536,
|
||||
@@ -83285,9 +83421,9 @@
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.5795,
|
||||
"output": 2.44,
|
||||
"cacheRead": 0.0976,
|
||||
"input": 0.5605,
|
||||
"output": 2.36,
|
||||
"cacheRead": 0.0944,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
@@ -83915,7 +84051,7 @@
|
||||
"cacheRead": 0.049999999999999996,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": 262144,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
@@ -90620,9 +90756,9 @@
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.49,
|
||||
"output": 1.54,
|
||||
"cacheRead": 0.091,
|
||||
"input": 0.392,
|
||||
"output": 1.232,
|
||||
"cacheRead": 0.0728,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
@@ -96815,6 +96951,36 @@
|
||||
},
|
||||
"supportsComputerUse": false
|
||||
},
|
||||
"alibaba/qwen3.8-2.4t-a95b": {
|
||||
"id": "alibaba/qwen3.8-2.4t-a95b",
|
||||
"name": "Qwen3.8 2.4T A95B",
|
||||
"api": "anthropic-messages",
|
||||
"provider": "vercel-ai-gateway",
|
||||
"baseUrl": "https://ai-gateway.vercel.sh",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 2,
|
||||
"output": 6,
|
||||
"cacheRead": 0.25,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 131072,
|
||||
"thinking": {
|
||||
"mode": "budget",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
]
|
||||
},
|
||||
"supportsComputerUse": false
|
||||
},
|
||||
"alibaba/qwen3.8-max": {
|
||||
"id": "alibaba/qwen3.8-max",
|
||||
"name": "Qwen 3.8 Max",
|
||||
@@ -98529,7 +98695,7 @@
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 256000,
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 131072,
|
||||
"thinking": {
|
||||
"mode": "budget",
|
||||
|
||||
@@ -15,30 +15,86 @@ export const getGeminiCliHeaders = (modelId?: string) => ({
|
||||
"Client-Metadata": "ideType=IDE_UNSPECIFIED,platform=PLATFORM_UNSPECIFIED,pluginType=GEMINI",
|
||||
});
|
||||
|
||||
export const ANTIGRAVITY_SYSTEM_INSTRUCTION =
|
||||
"You are Antigravity, a powerful agentic AI coding assistant designed by the Google Deepmind team working on Advanced Agentic Coding." +
|
||||
"You are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question." +
|
||||
"**Absolute paths only**" +
|
||||
"**Proactiveness**";
|
||||
/**
|
||||
* Antigravity / Cloud Code Assist user agent. Lives in its own file so discovery
|
||||
* and usage code can read it without pulling the heavy google-gemini-cli provider
|
||||
* (and its @google/genai → google-auth-library dependency chain) into the startup
|
||||
* parse graph.
|
||||
*
|
||||
* Format captured from the real 2.8.0 `antigravity/hub` client:
|
||||
* `antigravity/hub/2.8.0 (aidev_client; os_type=darwin; arch=arm64; cl=963137146)`.
|
||||
* The backend gates newer models (e.g. gemini-3.7-flash) on the client version,
|
||||
* so the version tracks the latest Antigravity release via the update manifest
|
||||
* (see {@link ensureAntigravityVersion}) with `DEFAULT_ANTIGRAVITY_VERSION` as
|
||||
* the offline fallback. os_type/arch are pinned to the darwin/arm64 reference
|
||||
* client the version and manifest are captured from, independent of the host
|
||||
* platform. Overrides: PI_AI_ANTIGRAVITY_VERSION / _CL / _OS / _ARCH.
|
||||
*/
|
||||
export let getAntigravityUserAgent = () => {
|
||||
const DEFAULT_ANTIGRAVITY_VERSION = "2.1.4";
|
||||
const version = process.env.PI_AI_ANTIGRAVITY_VERSION || DEFAULT_ANTIGRAVITY_VERSION;
|
||||
// Map Node.js platform/arch to Antigravity's expected format.
|
||||
// Verified against Antigravity source: _qn() and wqn() in main.js.
|
||||
// process.platform: win32→windows, others pass through (darwin, linux)
|
||||
// process.arch: x64→amd64, ia32→386, others pass through (arm64)
|
||||
const os = process.platform === "win32" ? "windows" : process.platform;
|
||||
const arch = process.arch === "x64" ? "amd64" : process.arch === "ia32" ? "386" : process.arch;
|
||||
const userAgent = `antigravity/hub/${version} ${os}/${arch}`;
|
||||
getAntigravityUserAgent = () => userAgent;
|
||||
return userAgent;
|
||||
};
|
||||
export const DEFAULT_ANTIGRAVITY_VERSION = "2.8.0";
|
||||
|
||||
const ANTIGRAVITY_VERSION_MANIFEST_URL =
|
||||
"https://antigravity-hub-auto-updater-974169037036.us-central1.run.app/manifest/latest-arm64-mac.yml";
|
||||
const ANTIGRAVITY_VERSION_FETCH_TIMEOUT_MS = 5_000;
|
||||
|
||||
let discoveredAntigravityVersion: string | null = null;
|
||||
let antigravityVersionFetch: Promise<void> | null = null;
|
||||
|
||||
/** Current Antigravity client version: env override → manifest-discovered → pinned fallback. */
|
||||
export function getAntigravityVersion(): string {
|
||||
return process.env.PI_AI_ANTIGRAVITY_VERSION || discoveredAntigravityVersion || DEFAULT_ANTIGRAVITY_VERSION;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extracts the client version from an electron-builder update manifest.
|
||||
* Returns null when no well-formed `version:` line is present.
|
||||
*/
|
||||
export function parseAntigravityManifestVersion(yamlText: string): string | null {
|
||||
for (const line of yamlText.split(/\r?\n/)) {
|
||||
const match = /^\s*version\s*:\s*(?:"([^"]*)"|'([^']*)'|([^\s#]+))\s*(?:#.*)?$/.exec(line);
|
||||
if (!match) continue;
|
||||
const version = (match[1] ?? match[2] ?? match[3] ?? "").trim();
|
||||
return /^\d+\.\d+\.\d+$/.test(version) ? version : null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolves the latest Antigravity release from the official update manifest.
|
||||
* Success is cached for the process lifetime; failures are silent (the pinned
|
||||
* fallback stays valid) and clear the in-flight cache so a later call retries.
|
||||
* Skipped entirely when PI_AI_ANTIGRAVITY_VERSION is set.
|
||||
*/
|
||||
export function ensureAntigravityVersion(fetcher: typeof fetch = fetch, signal?: AbortSignal): Promise<void> {
|
||||
if (process.env.PI_AI_ANTIGRAVITY_VERSION || discoveredAntigravityVersion) return Promise.resolve();
|
||||
if (antigravityVersionFetch) return antigravityVersionFetch;
|
||||
|
||||
antigravityVersionFetch = (async () => {
|
||||
try {
|
||||
const timeoutSignal = AbortSignal.timeout(ANTIGRAVITY_VERSION_FETCH_TIMEOUT_MS);
|
||||
const response = await fetcher(ANTIGRAVITY_VERSION_MANIFEST_URL, {
|
||||
headers: { "Cache-Control": "no-cache", "User-Agent": "electron-builder" },
|
||||
signal: signal ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal,
|
||||
});
|
||||
if (response.ok) {
|
||||
discoveredAntigravityVersion = parseAntigravityManifestVersion(await response.text());
|
||||
}
|
||||
} catch {
|
||||
// Silent: the pinned fallback remains valid when version discovery fails.
|
||||
} finally {
|
||||
if (!discoveredAntigravityVersion) antigravityVersionFetch = null;
|
||||
}
|
||||
})();
|
||||
return antigravityVersionFetch;
|
||||
}
|
||||
|
||||
/** Antigravity `User-Agent` header value; rebuilt when the discovered version changes. */
|
||||
export function getAntigravityUserAgent(): string {
|
||||
const version = getAntigravityVersion();
|
||||
const cl = process.env.PI_AI_ANTIGRAVITY_CL || "963137146";
|
||||
const os = process.env.PI_AI_ANTIGRAVITY_OS || "darwin";
|
||||
const arch = process.env.PI_AI_ANTIGRAVITY_ARCH || "arm64";
|
||||
return `antigravity/hub/${version} (aidev_client; os_type=${os}; arch=${arch}; cl=${cl})`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-wire-id Antigravity Cloud Code Assist request constants, captured from the
|
||||
|
||||
@@ -9,11 +9,7 @@
|
||||
* endpoint.
|
||||
*/
|
||||
import { type AuthStorage, type FetchImpl, type OAuthAccess, withOAuthAccess } from "@oh-my-pi/pi-ai";
|
||||
import {
|
||||
ANTIGRAVITY_SYSTEM_INSTRUCTION,
|
||||
getAntigravityUserAgent,
|
||||
getGeminiCliHeaders,
|
||||
} from "@oh-my-pi/pi-catalog/wire/gemini-headers";
|
||||
import { getAntigravityUserAgent, getGeminiCliHeaders } from "@oh-my-pi/pi-catalog/wire/gemini-headers";
|
||||
import { fetchWithRetry, USER_AGENT } from "@oh-my-pi/pi-utils";
|
||||
|
||||
import type { SearchCitation, SearchResponse, SearchSource } from "../../../web/search/types";
|
||||
@@ -441,10 +437,9 @@ async function callGeminiSearch(
|
||||
};
|
||||
|
||||
const normalizedSystemPrompt = systemPrompt?.toWellFormed();
|
||||
const systemInstructionParts: Array<{ text: string }> = [
|
||||
...(auth.isAntigravity ? [{ text: ANTIGRAVITY_SYSTEM_INSTRUCTION }] : []),
|
||||
...(normalizedSystemPrompt ? [{ text: normalizedSystemPrompt }] : []),
|
||||
];
|
||||
const systemInstructionParts: Array<{ text: string }> = normalizedSystemPrompt
|
||||
? [{ text: normalizedSystemPrompt }]
|
||||
: [];
|
||||
|
||||
const requestBody: Record<string, unknown> = {
|
||||
project: auth.projectId,
|
||||
|
||||
Reference in New Issue
Block a user