From 3ea4981eeb5feff29af64ae7d2f4212449c36b1e Mon Sep 17 00:00:00 2001 From: roboomp Date: Wed, 27 May 2026 19:14:26 +0000 Subject: [PATCH] fix(ai): updated google vertex model catalog Replaced Google Vertex project discovery with the models.dev catalog so bundled model selection includes current Vertex MaaS and Gemini entries while pruning retired fallbacks. Fixes #1456 --- packages/ai/CHANGELOG.md | 4 + packages/ai/scripts/generate-models.ts | 21 +- packages/ai/src/models.json | 1024 ++++++++++++++--- packages/ai/src/provider-models/google.ts | 40 +- .../ai/src/provider-models/openai-compat.ts | 19 + packages/ai/src/providers/google-vertex.ts | 6 +- packages/ai/src/stream.ts | 27 +- packages/ai/src/utils/discovery/index.ts | 1 - packages/ai/src/utils/discovery/vertex.ts | 210 ---- .../ai/test/google-vertex-discovery.test.ts | 187 ++- 10 files changed, 1023 insertions(+), 516 deletions(-) delete mode 100644 packages/ai/src/utils/discovery/vertex.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index f3b80e5a9..a0dfae3da 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Google Vertex's bundled model list to use the authoritative models.dev catalog, including MaaS entries such as `deepseek-ai/deepseek-v3.2-maas` and removing retired Gemini 1.5 fallbacks. ([#1456](https://github.com/can1357/oh-my-pi/issues/1456)) + ## [15.5.7] - 2026-05-27 ### Added - `SimpleStreamOptions.openrouterVariant` (`"nitro"`, `"floor"`, `"online"`, `"exacto"`, …) — when set, appends `:` to OpenRouter model IDs at request time, leaving ids that already carry an explicit `:suffix` untouched. Plumbed through `openai-completions` and the pi-native gateway forwarder. diff --git a/packages/ai/scripts/generate-models.ts b/packages/ai/scripts/generate-models.ts index 2052629d1..4bd0a5443 100644 --- a/packages/ai/scripts/generate-models.ts +++ b/packages/ai/scripts/generate-models.ts @@ -354,18 +354,27 @@ async function generateModels() { } } - // Merge previous models.json entries as fallback for any provider/model - // not fetched dynamically. This replaces all hardcoded fallback lists — - // static-only providers (vertex, gemini-cli), auth-gated providers when - // credentials are unavailable, and ad-hoc model additions all persist - // through the existing models.json seed. + const modelsDevAuthoritativeProviders = new Set(); + for (const model of modelsDevModels) { + if (model.provider === "google-vertex") { + modelsDevAuthoritativeProviders.add(model.provider); + } + } + // Merge previous models.json entries as fallback for provider/model pairs not + // fetched dynamically. Providers that models.dev covers authoritatively keep + // the upstream list exactly, so retired entries from the previous snapshot do + // not reappear during regeneration. // Discovery-only providers (local inference servers) — never bundle static models. const discoveryOnlyProviders = new Set(["ollama", "vllm"]); const fetchedKeys = new Set(allModels.map(model => `${model.provider}/${model.id}`)); for (const models of Object.values(prevModelsJson as Record>)) { for (const model of Object.values(models)) { - if (!fetchedKeys.has(`${model.provider}/${model.id}`) && !discoveryOnlyProviders.has(model.provider)) { + if ( + !fetchedKeys.has(`${model.provider}/${model.id}`) && + !discoveryOnlyProviders.has(model.provider) && + !modelsDevAuthoritativeProviders.has(model.provider) + ) { allModels.push(model); } } diff --git a/packages/ai/src/models.json b/packages/ai/src/models.json index b7f90b229..7e1dcab96 100644 --- a/packages/ai/src/models.json +++ b/packages/ai/src/models.json @@ -7880,9 +7880,9 @@ } }, "google-vertex": { - "gemini-1.5-flash": { - "id": "gemini-1.5-flash", - "name": "Gemini 1.5 Flash", + "claude-3-5-haiku@20241022": { + "id": "claude-3-5-haiku@20241022", + "name": "Claude Haiku 3.5", "api": "google-vertex", "provider": "google-vertex", "baseUrl": "https://{location}-aiplatform.googleapis.com", @@ -7892,17 +7892,17 @@ "image" ], "cost": { - "input": 0.075, - "output": 0.3, - "cacheRead": 0.01875, - "cacheWrite": 0 + "input": 0.8, + "output": 4, + "cacheRead": 0.08, + "cacheWrite": 1 }, - "contextWindow": 1000000, + "contextWindow": 200000, "maxTokens": 8192 }, - "gemini-1.5-flash-8b": { - "id": "gemini-1.5-flash-8b", - "name": "Gemini 1.5 Flash-8B", + "claude-3-5-sonnet@20241022": { + "id": "claude-3-5-sonnet@20241022", + "name": "Claude Sonnet 3.5 v2", "api": "google-vertex", "provider": "google-vertex", "baseUrl": "https://{location}-aiplatform.googleapis.com", @@ -7912,33 +7912,311 @@ "image" ], "cost": { - "input": 0.0375, - "output": 0.15, - "cacheRead": 0.01, - "cacheWrite": 0 + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 3.75 }, - "contextWindow": 1000000, + "contextWindow": 200000, "maxTokens": 8192 }, - "gemini-1.5-pro": { - "id": "gemini-1.5-pro", - "name": "Gemini 1.5 Pro", + "claude-3-7-sonnet@20250219": { + "id": "claude-3-7-sonnet@20250219", + "name": "Claude Sonnet 3.7", "api": "google-vertex", "provider": "google-vertex", "baseUrl": "https://{location}-aiplatform.googleapis.com", - "reasoning": false, + "reasoning": true, "input": [ "text", "image" ], "cost": { - "input": 1.25, + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 3.75 + }, + "contextWindow": 200000, + "maxTokens": 64000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "claude-haiku-4-5@20251001": { + "id": "claude-haiku-4-5@20251001", + "name": "Claude Haiku 4.5", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, "output": 5, - "cacheRead": 0.3125, - "cacheWrite": 0 + "cacheRead": 0.1, + "cacheWrite": 1.25 + }, + "contextWindow": 200000, + "maxTokens": 64000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "claude-opus-4-1@20250805": { + "id": "claude-opus-4-1@20250805", + "name": "Claude Opus 4.1", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 15, + "output": 75, + "cacheRead": 1.5, + "cacheWrite": 18.75 + }, + "contextWindow": 200000, + "maxTokens": 32000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "claude-opus-4-5@20251101": { + "id": "claude-opus-4-5@20251101", + "name": "Claude Opus 4.5", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 25, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 200000, + "maxTokens": 64000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "claude-opus-4-6@default": { + "id": "claude-opus-4-6@default", + "name": "Claude Opus 4.6", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 25, + "cacheRead": 0.5, + "cacheWrite": 6.25 }, "contextWindow": 1000000, - "maxTokens": 8192 + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "claude-opus-4-7@default": { + "id": "claude-opus-4-7@default", + "name": "Claude Opus 4.7", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 25, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "claude-opus-4@20250514": { + "id": "claude-opus-4@20250514", + "name": "Claude Opus 4", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 15, + "output": 75, + "cacheRead": 1.5, + "cacheWrite": 18.75 + }, + "contextWindow": 200000, + "maxTokens": 32000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "claude-sonnet-4-5@20250929": { + "id": "claude-sonnet-4-5@20250929", + "name": "Claude Sonnet 4.5", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 3.75 + }, + "contextWindow": 200000, + "maxTokens": 64000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "claude-sonnet-4-6@default": { + "id": "claude-sonnet-4-6@default", + "name": "Claude Sonnet 4.6", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 3.75 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "claude-sonnet-4@20250514": { + "id": "claude-sonnet-4@20250514", + "name": "Claude Sonnet 4", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 3.75 + }, + "contextWindow": 200000, + "maxTokens": 64000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "deepseek-ai/deepseek-v3.1-maas": { + "id": "deepseek-ai/deepseek-v3.1-maas", + "name": "DeepSeek V3.1", + "api": "openai-completions", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/endpoints/openapi", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 1.7, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 163840, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, + "deepseek-ai/deepseek-v3.2-maas": { + "id": "deepseek-ai/deepseek-v3.2-maas", + "name": "DeepSeek V3.2", + "api": "openai-completions", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/endpoints/openapi", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.56, + "output": 1.68, + "cacheRead": 0.056, + "cacheWrite": 0 + }, + "contextWindow": 163840, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "gemini-2.0-flash": { "id": "gemini-2.0-flash", @@ -7954,7 +8232,7 @@ "cost": { "input": 0.15, "output": 0.6, - "cacheRead": 0.0375, + "cacheRead": 0.025, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -7974,7 +8252,7 @@ "cost": { "input": 0.075, "output": 0.3, - "cacheRead": 0.01875, + "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -7994,8 +8272,8 @@ "cost": { "input": 0.3, "output": 2.5, - "cacheRead": 0.03, - "cacheWrite": 0 + "cacheRead": 0.075, + "cacheWrite": 0.383 }, "contextWindow": 1048576, "maxTokens": 65536, @@ -8030,9 +8308,9 @@ "maxLevel": "high" } }, - "gemini-2.5-flash-lite-preview-09-2025": { - "id": "gemini-2.5-flash-lite-preview-09-2025", - "name": "Gemini 2.5 Flash Lite Preview 09-25", + "gemini-2.5-flash-lite-preview-06-17": { + "id": "gemini-2.5-flash-lite-preview-06-17", + "name": "Gemini 2.5 Flash Lite Preview 06-17", "api": "google-vertex", "provider": "google-vertex", "baseUrl": "https://{location}-aiplatform.googleapis.com", @@ -8044,9 +8322,34 @@ "cost": { "input": 0.1, "output": 0.4, - "cacheRead": 0.01, + "cacheRead": 0.025, "cacheWrite": 0 }, + "contextWindow": 65536, + "maxTokens": 65536, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "gemini-2.5-flash-preview-09-2025": { + "id": "gemini-2.5-flash-preview-09-2025", + "name": "Gemini 2.5 Flash Preview 09-25", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 2.5, + "cacheRead": 0.075, + "cacheWrite": 0.383 + }, "contextWindow": 1048576, "maxTokens": 65536, "thinking": { @@ -8122,8 +8425,8 @@ "cacheRead": 0.2, "cacheWrite": 0 }, - "contextWindow": 1000000, - "maxTokens": 64000, + "contextWindow": 1048576, + "maxTokens": 65536, "thinking": { "mode": "google-level", "minLevel": "low", @@ -8134,6 +8437,56 @@ ] } }, + "gemini-3.1-flash-lite": { + "id": "gemini-3.1-flash-lite", + "name": "Gemini 3.1 Flash Lite", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 1.5, + "cacheRead": 0.025, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "google-level", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "gemini-3.1-flash-lite-preview": { + "id": "gemini-3.1-flash-lite-preview", + "name": "Gemini 3.1 Flash Lite Preview", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 1.5, + "cacheRead": 0.025, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "google-level", + "minLevel": "minimal", + "maxLevel": "high" + } + }, "gemini-3.1-pro-preview": { "id": "gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", @@ -8191,6 +8544,264 @@ "high" ] } + }, + "gemini-3.5-flash": { + "id": "gemini-3.5-flash", + "name": "Gemini 3.5 Flash", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.5, + "output": 9, + "cacheRead": 0.15, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "google-level", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "gemini-flash-latest": { + "id": "gemini-flash-latest", + "name": "Gemini Flash Latest", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 2.5, + "cacheRead": 0.075, + "cacheWrite": 0.383 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "gemini-flash-lite-latest": { + "id": "gemini-flash-lite-latest", + "name": "Gemini Flash-Lite Latest", + "api": "google-vertex", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.1, + "output": 0.4, + "cacheRead": 0.025, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "meta/llama-3.3-70b-instruct-maas": { + "id": "meta/llama-3.3-70b-instruct-maas", + "name": "Llama 3.3 70B Instruct", + "api": "openai-completions", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/endpoints/openapi", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.72, + "output": 0.72, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 8192 + }, + "meta/llama-4-maverick-17b-128e-instruct-maas": { + "id": "meta/llama-4-maverick-17b-128e-instruct-maas", + "name": "Llama 4 Maverick 17B 128E Instruct", + "api": "openai-completions", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/endpoints/openapi", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.35, + "output": 1.15, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 524288, + "maxTokens": 8192 + }, + "moonshotai/kimi-k2-thinking-maas": { + "id": "moonshotai/kimi-k2-thinking-maas", + "name": "Kimi K2 Thinking", + "api": "openai-completions", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/endpoints/openapi", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, + "openai/gpt-oss-120b-maas": { + "id": "openai/gpt-oss-120b-maas", + "name": "GPT OSS 120B", + "api": "openai-completions", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/endpoints/openapi", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.09, + "output": 0.36, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, + "openai/gpt-oss-20b-maas": { + "id": "openai/gpt-oss-20b-maas", + "name": "GPT OSS 20B", + "api": "openai-completions", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/endpoints/openapi", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.07, + "output": 0.25, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, + "qwen/qwen3-235b-a22b-instruct-2507-maas": { + "id": "qwen/qwen3-235b-a22b-instruct-2507-maas", + "name": "Qwen3 235B A22B Instruct", + "api": "openai-completions", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/endpoints/openapi", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.22, + "output": 0.88, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "zai-org/glm-4.7-maas": { + "id": "zai-org/glm-4.7-maas", + "name": "GLM-4.7", + "api": "openai-completions", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/endpoints/openapi", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 200000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, + "zai-org/glm-5-maas": { + "id": "zai-org/glm-5-maas", + "name": "GLM-5", + "api": "openai-completions", + "provider": "google-vertex", + "baseUrl": "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/endpoints/openapi", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1, + "output": 3.2, + "cacheRead": 0.1, + "cacheWrite": 0 + }, + "contextWindow": 202752, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } } }, "groq": { @@ -8452,7 +9063,7 @@ "cost": { "input": 1, "output": 3, - "cacheRead": 0, + "cacheRead": 0.5, "cacheWrite": 0 }, "contextWindow": 262144, @@ -8471,7 +9082,7 @@ "cost": { "input": 0.15, "output": 0.6, - "cacheRead": 0, + "cacheRead": 0.075, "cacheWrite": 0 }, "contextWindow": 131072, @@ -8495,7 +9106,7 @@ "cost": { "input": 0.075, "output": 0.3, - "cacheRead": 0, + "cacheRead": 0.0375, "cacheWrite": 0 }, "contextWindow": 131072, @@ -12861,6 +13472,25 @@ "maxLevel": "xhigh" } }, + "moonshotai/kimi-k2.6:free": { + "id": "moonshotai/kimi-k2.6:free", + "name": "MoonshotAI: Kimi K2.6 (free)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "morph-warp-grep-v2": { "id": "morph-warp-grep-v2", "name": "Morph: WarpGrep V2", @@ -21571,13 +22201,14 @@ }, "gemini-2.5-flash-lite-preview-06-17": { "id": "gemini-2.5-flash-lite-preview-06-17", - "name": "gemini-2.5-flash-lite-preview-06-17", + "name": "Gemini 2.5 Flash Lite Preview 06-17", "api": "openai-completions", "provider": "litellm", "baseUrl": "http://localhost:4000/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -21585,8 +22216,13 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 65536, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } }, "gemini-2.5-flash-lite-preview-09-2025": { "id": "gemini-2.5-flash-lite-preview-09-2025", @@ -21685,13 +22321,14 @@ }, "gemini-2.5-flash-preview-09-2025": { "id": "gemini-2.5-flash-preview-09-2025", - "name": "gemini-2.5-flash-preview-09-2025", + "name": "Gemini 2.5 Flash Preview 09-25", "api": "openai-completions", "provider": "litellm", "baseUrl": "http://localhost:4000/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -21699,8 +22336,13 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, - "maxTokens": 8888 + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } }, "gemini-2.5-flash-preview-09-2025-thinking": { "id": "gemini-2.5-flash-preview-09-2025-thinking", @@ -52218,7 +52860,7 @@ "cost": { "input": 0.1, "output": 0.4, - "cacheRead": 0.03, + "cacheRead": 0.025, "cacheWrite": 0 }, "contextWindow": 1047576, @@ -52318,7 +52960,7 @@ "cost": { "input": 0.15, "output": 0.6, - "cacheRead": 0.08, + "cacheRead": 0.075, "cacheWrite": 0 }, "contextWindow": 128000, @@ -52489,7 +53131,7 @@ "cost": { "input": 1.25, "output": 10, - "cacheRead": 0.13, + "cacheRead": 0.125, "cacheWrite": 0 }, "contextWindow": 400000, @@ -53104,7 +53746,7 @@ "cost": { "input": 1.1, "output": 4.4, - "cacheRead": 0.28, + "cacheRead": 0.275, "cacheWrite": 0 }, "contextWindow": 200000, @@ -53933,9 +54575,9 @@ "image" ], "cost": { - "input": 0.4, - "output": 2, - "cacheRead": 0.08, + "input": 0.14, + "output": 0.28, + "cacheRead": 0.0028, "cacheWrite": 0 }, "contextWindow": 1000000, @@ -53957,9 +54599,9 @@ "text" ], "cost": { - "input": 1, - "output": 3, - "cacheRead": 0.2, + "input": 1.74, + "output": 3.48, + "cacheRead": 0.0145, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -54067,6 +54709,30 @@ "minLevel": "minimal", "maxLevel": "high" } + }, + "qwen3.7-max": { + "id": "qwen3.7-max", + "name": "Qwen3.7 Max", + "api": "anthropic-messages", + "provider": "opencode-go", + "baseUrl": "https://opencode.ai/zen/go", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 2.5, + "output": 7.5, + "cacheRead": 0.5, + "cacheWrite": 3.125 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "xhigh" + } } }, "opencode-zen": { @@ -55203,6 +55869,31 @@ "maxLevel": "xhigh" } }, + "mimo-v2.5-free": { + "id": "mimo-v2.5-free", + "name": "MiMo V2.5 Free", + "api": "openai-completions", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "minimax-m2.1": { "id": "minimax-m2.1", "name": "MiniMax M2.1", @@ -58662,6 +59353,31 @@ "maxLevel": "high" } }, + "moonshotai/kimi-k2.6:free": { + "id": "moonshotai/kimi-k2.6:free", + "name": "MoonshotAI: Kimi K2.6 (free)", + "api": "openai-completions", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 8888, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } + }, "nex-agi/deepseek-v3.1-nex-n1": { "id": "nex-agi/deepseek-v3.1-nex-n1", "name": "Nex AGI: DeepSeek V3.1 Nex N1", @@ -59552,11 +60268,11 @@ "cost": { "input": 1.25, "output": 10, - "cacheRead": 0.125, + "cacheRead": 0.13, "cacheWrite": 0 }, "contextWindow": 128000, - "maxTokens": 16384 + "maxTokens": 32000 }, "openai/gpt-5.1-codex": { "id": "openai/gpt-5.1-codex", @@ -59572,7 +60288,7 @@ "cost": { "input": 1.25, "output": 10, - "cacheRead": 0.125, + "cacheRead": 0.13, "cacheWrite": 0 }, "contextWindow": 272000, @@ -59622,11 +60338,11 @@ "cost": { "input": 0.25, "output": 2, - "cacheRead": 0.03, + "cacheRead": 0.024999999999999998, "cacheWrite": 0 }, "contextWindow": 272000, - "maxTokens": 128000, + "maxTokens": 100000, "thinking": { "mode": "effort", "minLevel": "medium", @@ -59676,7 +60392,7 @@ "cacheWrite": 0 }, "contextWindow": 128000, - "maxTokens": 32000 + "maxTokens": 16384 }, "openai/gpt-5.2-codex": { "id": "openai/gpt-5.2-codex", @@ -60539,8 +61255,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 131072, - "maxTokens": 8192, + "contextWindow": 262144, + "maxTokens": 32768, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -61576,7 +62292,7 @@ "input": 0.3, "output": 1.7999999999999998, "cacheRead": 0, - "cacheWrite": 0 + "cacheWrite": 0.375 }, "contextWindow": 1000000, "maxTokens": 65536, @@ -61598,13 +62314,13 @@ "image" ], "cost": { - "input": 0.3, - "output": 2, + "input": 0.29, + "output": 3.1999999999999997, "cacheRead": 0.15, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 262140, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -61623,7 +62339,7 @@ "image" ], "cost": { - "input": 0.15, + "input": 0.14, "output": 1, "cacheRead": 0.049999999999999996, "cacheWrite": 0 @@ -61769,10 +62485,10 @@ "text" ], "cost": { - "input": 2.5, - "output": 7.5, - "cacheRead": 0, - "cacheWrite": 3.125 + "input": 1.25, + "output": 3.75, + "cacheRead": 0.25, + "cacheWrite": 1.5625 }, "contextWindow": 1000000, "maxTokens": 65536, @@ -62595,13 +63311,13 @@ "text" ], "cost": { - "input": 0.13, - "output": 0.85, + "input": 0.125, + "output": 0.84, "cacheRead": 0.024999999999999998, "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 98304, + "maxTokens": 131070, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -69586,9 +70302,9 @@ "image" ], "cost": { - "input": 0.39999999999999997, - "output": 2, - "cacheRead": 0.08, + "input": 0.14, + "output": 0.28, + "cacheRead": 0.0028, "cacheWrite": 0 }, "contextWindow": 1050000, @@ -69610,9 +70326,9 @@ "text" ], "cost": { - "input": 1, - "output": 3, - "cacheRead": 0.19999999999999998, + "input": 0.435, + "output": 0.87, + "cacheRead": 0.0036, "cacheWrite": 0 }, "contextWindow": 1050000, @@ -70574,38 +71290,13 @@ } }, "xai-oauth": { - "grok-build": { - "id": "grok-build", - "name": "Grok Build", + "grok-4.20-0309-non-reasoning": { + "id": "grok-4.20-0309-non-reasoning", + "name": "Grok 4.20 (Non-Reasoning)", "api": "openai-responses", "provider": "xai-oauth", "baseUrl": "https://api.x.ai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 512000, - "maxTokens": 8888, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "supportsReasoningEffort": false - } - }, - "grok-4.3": { - "id": "grok-4.3", - "name": "Grok 4.3", - "api": "openai-responses", - "provider": "xai-oauth", - "baseUrl": "https://api.x.ai/v1", - "reasoning": true, + "reasoning": false, "input": [ "text", "image" @@ -70616,32 +71307,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 1000000, - "maxTokens": 8888, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - } - } - }, - "grok-4.20-multi-agent-0309": { - "id": "grok-4.20-multi-agent-0309", - "name": "Grok 4.20 (Multi-Agent)", - "api": "openai-responses", - "provider": "xai-oauth", - "baseUrl": "https://api.x.ai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, "contextWindow": 2000000, - "maxTokens": 8888, + "maxTokens": 30000, "compat": { "reasoningEffortMap": { "minimal": "low" @@ -70666,24 +71333,28 @@ "cacheWrite": 0 }, "contextWindow": 2000000, - "maxTokens": 8888, + "maxTokens": 30000, "compat": { "reasoningEffortMap": { "minimal": "low" }, "supportsReasoningEffort": false + }, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" } }, - "grok-4.20-0309-non-reasoning": { - "id": "grok-4.20-0309-non-reasoning", - "name": "Grok 4.20 (Non-Reasoning)", + "grok-4.20-multi-agent-0309": { + "id": "grok-4.20-multi-agent-0309", + "name": "Grok 4.20 (Multi-Agent)", "api": "openai-responses", "provider": "xai-oauth", "baseUrl": "https://api.x.ai/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text", - "image" + "text" ], "cost": { "input": 0, @@ -70697,6 +71368,71 @@ "reasoningEffortMap": { "minimal": "low" } + }, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, + "grok-4.3": { + "id": "grok-4.3", + "name": "Grok 4.3", + "api": "openai-responses", + "provider": "xai-oauth", + "baseUrl": "https://api.x.ai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 30000, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + } + }, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, + "grok-build": { + "id": "grok-build", + "name": "Grok Build", + "api": "openai-responses", + "provider": "xai-oauth", + "baseUrl": "https://api.x.ai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 512000, + "maxTokens": 8888, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "supportsReasoningEffort": false + }, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" } } }, diff --git a/packages/ai/src/provider-models/google.ts b/packages/ai/src/provider-models/google.ts index fe40b250c..226a16047 100644 --- a/packages/ai/src/provider-models/google.ts +++ b/packages/ai/src/provider-models/google.ts @@ -2,7 +2,6 @@ import type { ModelManagerOptions } from "../model-manager"; import type { FetchImpl } from "../types"; import { fetchAntigravityDiscoveryModels } from "../utils/discovery/antigravity"; import { fetchGeminiModels } from "../utils/discovery/gemini"; -import { fetchVertexOpenAIModels } from "../utils/discovery/vertex"; export interface GoogleModelManagerConfig { apiKey?: string; @@ -38,43 +37,8 @@ export function googleModelManagerOptions( }; } -export function googleVertexModelManagerOptions(config?: GoogleVertexModelManagerConfig): ModelManagerOptions { - const project = resolveVertexProject(config); - const hasApiKey = (config?.apiKey ?? Bun.env.GOOGLE_CLOUD_API_KEY ?? "").trim().length > 0; - const location = resolveVertexLocation(config); - if (hasApiKey) { - return { providerId: "google-vertex" }; - } - if (project && location) { - return { - providerId: "google-vertex", - staticModels: [], - fetchDynamicModels: () => - fetchVertexOpenAIModels({ - project, - location, - signal: config?.signal, - fetch: config?.fetch, - }), - }; - } - // With neither ADC project+location nor API key auth configured, drop the - // bundled static catalog so stale fallbacks (e.g. `gemini-1.5-*`) cannot leak - // into `/models` alongside an authoritative cached Vertex project catalog on - // the next refresh. - return { providerId: "google-vertex", staticModels: [] }; -} -function resolveVertexProject(config?: GoogleVertexModelManagerConfig): string | undefined { - const project = config?.project ?? Bun.env.GOOGLE_CLOUD_PROJECT ?? Bun.env.GCP_PROJECT ?? Bun.env.GCLOUD_PROJECT; - const trimmed = project?.trim(); - return trimmed ? trimmed : undefined; -} - -function resolveVertexLocation(config?: GoogleVertexModelManagerConfig): string | undefined { - const location = - config?.location ?? Bun.env.GOOGLE_VERTEX_LOCATION ?? Bun.env.GOOGLE_CLOUD_LOCATION ?? Bun.env.VERTEX_LOCATION; - const trimmed = location?.trim(); - return trimmed ? trimmed : undefined; +export function googleVertexModelManagerOptions(_config?: GoogleVertexModelManagerConfig): ModelManagerOptions { + return { providerId: "google-vertex" }; } export function googleAntigravityModelManagerOptions( diff --git a/packages/ai/src/provider-models/openai-compat.ts b/packages/ai/src/provider-models/openai-compat.ts index b23eb3d88..66a245d1e 100644 --- a/packages/ai/src/provider-models/openai-compat.ts +++ b/packages/ai/src/provider-models/openai-compat.ts @@ -2360,6 +2360,17 @@ function anthropicMessagesDescriptor( return simpleModelsDevDescriptor(modelsDevKey, providerId, "anthropic-messages", baseUrl, options); } +const GOOGLE_VERTEX_BASE_URL = "https://{location}-aiplatform.googleapis.com"; +const GOOGLE_VERTEX_OPENAI_BASE_URL = + "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/endpoints/openapi"; + +function resolveGoogleVertexApi(modelId: string, raw: ModelsDevModel): { api: Api; baseUrl: string } { + if (modelId.includes("/") || raw.provider?.npm === "@ai-sdk/openai-compatible") { + return { api: "openai-completions", baseUrl: GOOGLE_VERTEX_OPENAI_BASE_URL }; + } + return { api: "google-vertex", baseUrl: GOOGLE_VERTEX_BASE_URL }; +} + const MODELS_DEV_PROVIDER_DESCRIPTORS_BEDROCK: readonly ModelsDevProviderDescriptor[] = [ // --- Amazon Bedrock --- { @@ -2515,6 +2526,13 @@ const filterActiveToolCallModels = (_id: string, m: ModelsDevModel): boolean => return true; }; +const MODELS_DEV_PROVIDER_DESCRIPTORS_GOOGLE_VERTEX: readonly ModelsDevProviderDescriptor[] = [ + simpleModelsDevDescriptor("google-vertex", "google-vertex", "google-vertex", GOOGLE_VERTEX_BASE_URL, { + filterModel: filterActiveToolCallModels, + resolveApi: resolveGoogleVertexApi, + }), +]; + const MODELS_DEV_PROVIDER_DESCRIPTORS_SPECIALIZED: readonly ModelsDevProviderDescriptor[] = [ // --- Cloudflare AI Gateway --- anthropicMessagesDescriptor( @@ -2592,6 +2610,7 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_SPECIALIZED: readonly ModelsDevProviderDes /** All provider descriptors for models.dev data mapping in generate-models.ts. */ export const MODELS_DEV_PROVIDER_DESCRIPTORS: readonly ModelsDevProviderDescriptor[] = [ ...MODELS_DEV_PROVIDER_DESCRIPTORS_BEDROCK, + ...MODELS_DEV_PROVIDER_DESCRIPTORS_GOOGLE_VERTEX, ...MODELS_DEV_PROVIDER_DESCRIPTORS_CORE, ...MODELS_DEV_PROVIDER_DESCRIPTORS_CODING_PLANS, ...MODELS_DEV_PROVIDER_DESCRIPTORS_SPECIALIZED, diff --git a/packages/ai/src/providers/google-vertex.ts b/packages/ai/src/providers/google-vertex.ts index a8c07f2c4..e17e51e8f 100644 --- a/packages/ai/src/providers/google-vertex.ts +++ b/packages/ai/src/providers/google-vertex.ts @@ -48,7 +48,8 @@ export const streamGoogleVertex: StreamFunction<"google-vertex"> = ( const location = resolveLocation(options); const accessToken = await getVertexAccessToken({ signal: options?.signal, fetch: options?.fetch }); const host = resolveEndpointHost(location); - const url = `https://${host}/${API_VERSION}/projects/${project}/locations/${location}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`; + const publisher = resolvePublisher(model.id); + const url = `https://${host}/${API_VERSION}/projects/${project}/locations/${location}/publishers/${publisher}/models/${model.id}:streamGenerateContent?alt=sse`; return { params, url, @@ -76,6 +77,9 @@ function resolveProject(options?: GoogleVertexOptions): string { return project; } +function resolvePublisher(modelId: string): string { + return modelId.startsWith("claude-") ? "anthropic" : "google"; +} function resolveEndpointHost(location: string): string { return location === "global" ? "aiplatform.googleapis.com" : `${location}-aiplatform.googleapis.com`; } diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 7bbe24fd0..04716ce1d 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -90,11 +90,36 @@ function createVertexOpenAIFetch(options: StreamOptions | undefined): FetchImpl const token = await getVertexAccessToken({ signal: options?.signal, fetch: baseFetch }); const headers = new Headers(init?.headers); headers.set("Authorization", `Bearer ${token}`); - return baseFetch(input, { ...init, headers }); + return baseFetch(resolveVertexOpenAIRequest(input), { ...init, headers }); }; return Object.assign(vertexFetch, baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {}); } +function resolveVertexOpenAIRequest(input: string | URL | Request): string | URL | Request { + const project = $env.GOOGLE_CLOUD_PROJECT || $env.GCP_PROJECT || $env.GCLOUD_PROJECT; + const location = $env.GOOGLE_VERTEX_LOCATION || $env.GOOGLE_CLOUD_LOCATION || $env.VERTEX_LOCATION; + if (!project || !location) return input; + + const rewriteUrl = (url: string): string => { + if (!url.includes("{project}") && !url.includes("{location}")) return url; + const host = location === "global" ? "aiplatform.googleapis.com" : `${location}-aiplatform.googleapis.com`; + return url + .replace("https://{location}-aiplatform.googleapis.com", `https://${host}`) + .replaceAll("{project}", encodeURIComponent(project)) + .replaceAll("{location}", encodeURIComponent(location)); + }; + + if (input instanceof Request) { + const rewrittenUrl = rewriteUrl(input.url); + return rewrittenUrl === input.url ? input : new Request(rewrittenUrl, input); + } + if (input instanceof URL) { + const rewrittenUrl = rewriteUrl(input.toString()); + return rewrittenUrl === input.toString() ? input : new URL(rewrittenUrl); + } + return rewriteUrl(input); +} + type KeyResolver = string | (() => string | undefined); const serviceProviderMap: Record = { diff --git a/packages/ai/src/utils/discovery/index.ts b/packages/ai/src/utils/discovery/index.ts index d6ae7ae90..7af3bebdf 100644 --- a/packages/ai/src/utils/discovery/index.ts +++ b/packages/ai/src/utils/discovery/index.ts @@ -2,4 +2,3 @@ export * from "./antigravity"; export * from "./codex"; export * from "./gemini"; export * from "./openai-compatible"; -export * from "./vertex"; diff --git a/packages/ai/src/utils/discovery/vertex.ts b/packages/ai/src/utils/discovery/vertex.ts deleted file mode 100644 index c0b0e4ba4..000000000 --- a/packages/ai/src/utils/discovery/vertex.ts +++ /dev/null @@ -1,210 +0,0 @@ -import { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS } from "@oh-my-pi/pi-ai"; -import * as z from "zod/v4"; -import { getVertexAccessToken } from "../../providers/google-auth"; -import type { FetchImpl, Model } from "../../types"; - -const API_VERSION = "v1"; -const DEFAULT_PAGE_SIZE = 100; -const DEFAULT_MAX_PAGES = 25; - -const vertexOpenAIModelSchema = z.object({ - id: z.string().optional().catch(undefined), - name: z.string().optional().catch(undefined), - displayName: z.string().optional().catch(undefined), -}); - -const vertexOpenAIModelsResponseSchema = z.object({ - data: z - .array(z.unknown()) - .optional() - .transform(items => { - if (!items) return []; - const parsedItems: VertexOpenAIModelItem[] = []; - for (const item of items) { - const parsed = vertexOpenAIModelSchema.safeParse(item); - if (parsed.success) parsedItems.push(parsed.data); - } - return parsedItems; - }), - nextPageToken: z.string().optional().catch(undefined), -}); - -type VertexOpenAIModelItem = z.infer; - -/** Configuration for Vertex AI OpenAI-compatible model discovery. */ -export interface VertexDiscoveryOptions { - /** Google Cloud project ID hosting the Vertex AI endpoint. */ - project: string; - /** Vertex AI location, for example `global` or `us-central1`. */ - location: string; - /** Optional requested page size for model listing. */ - pageSize?: number; - /** Maximum number of pages to request before stopping pagination. */ - maxPages?: number; - /** Optional abort signal for HTTP requests. */ - signal?: AbortSignal; - /** Optional fetch implementation override for tests. */ - fetch?: FetchImpl; -} - -/** - * Fetches models exposed by Vertex AI's OpenAI-compatible endpoint. - * - * Returns `null` on auth, transport, or protocol failures so callers can fall - * back to cache/static models without surfacing discovery noise at startup. - */ -export async function fetchVertexOpenAIModels( - options: VertexDiscoveryOptions, -): Promise[] | null> { - const project = options.project.trim(); - const location = options.location.trim(); - if (!project || !location) return null; - - const fetchImpl = options.fetch ?? fetch; - const baseUrl = buildVertexOpenAIBaseUrl(project, location); - const pageSize = normalizePositiveInt(options.pageSize, DEFAULT_PAGE_SIZE); - const maxPages = normalizePositiveInt(options.maxPages, DEFAULT_MAX_PAGES); - let accessToken: string; - try { - accessToken = await getVertexAccessToken({ signal: options.signal, fetch: fetchImpl }); - } catch { - return null; - } - - const modelsById = new Map>(); - const seenTokens = new Set(); - let nextPageToken: string | undefined; - for (let page = 0; page < maxPages; page += 1) { - const requestUrl = buildModelsUrl(baseUrl, pageSize, nextPageToken); - let response: Response; - try { - response = await fetchImpl(requestUrl, { - method: "GET", - headers: { Authorization: `Bearer ${accessToken}` }, - signal: options.signal, - }); - } catch { - return null; - } - - if (!response.ok) return null; - - let payload: unknown; - try { - payload = await response.json(); - } catch { - return null; - } - - const parsed = vertexOpenAIModelsResponseSchema.safeParse(payload); - if (!parsed.success) return null; - - for (const item of parsed.data.data) { - const model = normalizeModel(item, baseUrl); - if (model) modelsById.set(model.id, model); - } - - const token = normalizePageToken(parsed.data.nextPageToken); - if (!token || seenTokens.has(token)) break; - seenTokens.add(token); - nextPageToken = token; - } - - return Array.from(modelsById.values()).sort((left, right) => left.id.localeCompare(right.id)); -} - -/** Returns the stable Vertex AI OpenAI-compatible endpoint base URL. */ -export function buildVertexOpenAIBaseUrl(project: string, location: string): string { - const host = location === "global" ? "aiplatform.googleapis.com" : `${location}-aiplatform.googleapis.com`; - return `https://${host}/${API_VERSION}/projects/${project}/locations/${location}/endpoints/openapi`; -} - -function buildModelsUrl(baseUrl: string, pageSize: number, pageToken?: string): URL { - const url = new URL(`${baseUrl}/models`); - url.searchParams.set("pageSize", String(pageSize)); - if (pageToken) url.searchParams.set("pageToken", pageToken); - return url; -} - -function normalizePositiveInt(value: number | undefined, fallback: number): number { - if (typeof value !== "number" || !Number.isFinite(value) || value <= 0) return fallback; - const normalized = Math.floor(value); - return normalized > 0 ? normalized : fallback; -} - -function normalizePageToken(value: unknown): string | undefined { - if (typeof value !== "string") return undefined; - const token = value.trim(); - return token.length > 0 ? token : undefined; -} - -function normalizeModel(item: VertexOpenAIModelItem, baseUrl: string): Model<"openai-completions"> | null { - const id = normalizeModelId(item.id ?? item.name); - if (!id) return null; - return { - id, - name: normalizeModelName(item.displayName, id), - api: "openai-completions", - provider: "google-vertex", - baseUrl, - reasoning: inferReasoning(id), - input: inferInput(id), - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, - contextWindow: UNK_CONTEXT_WINDOW, - maxTokens: UNK_MAX_TOKENS, - }; -} - -function normalizeModelId(value: string | undefined): string | null { - if (!value) return null; - const trimmed = value.trim(); - if (!trimmed) return null; - const marker = "/models/"; - const markerIndex = trimmed.lastIndexOf(marker); - if (markerIndex >= 0) { - const modelId = trimmed.slice(markerIndex + marker.length); - const publisher = extractPublisher(trimmed.slice(0, markerIndex)); - return publisher ? `${publisher}/${modelId}` : modelId; - } - return trimmed; -} - -function extractPublisher(prefix: string): string | undefined { - const marker = "/publishers/"; - const markerIndex = prefix.lastIndexOf(marker); - if (markerIndex < 0) return undefined; - const publisher = prefix.slice(markerIndex + marker.length).trim(); - return publisher.length > 0 ? publisher : undefined; -} - -function normalizeModelName(displayName: string | undefined, id: string): string { - const trimmed = displayName?.trim(); - return trimmed ? trimmed : id; -} - -function inferReasoning(id: string): boolean { - const normalized = id.toLowerCase(); - return ( - normalized.includes("thinking") || - normalized.includes("reasoning") || - normalized.includes("glm-4.5") || - normalized.includes("glm-4.6") || - normalized.includes("glm-4.7") || - normalized.includes("glm-5") || - normalized.includes("gemini-2.5") || - normalized.includes("gemini-3") - ); -} - -function inferInput(id: string): ("text" | "image")[] { - const normalized = id.toLowerCase(); - if ( - normalized.includes("gemini") || - normalized.includes("vision") || - normalized.includes("image") || - normalized.includes("vl") - ) { - return ["text", "image"]; - } - return ["text"]; -} diff --git a/packages/ai/test/google-vertex-discovery.test.ts b/packages/ai/test/google-vertex-discovery.test.ts index f47870fcc..5bd12fba6 100644 --- a/packages/ai/test/google-vertex-discovery.test.ts +++ b/packages/ai/test/google-vertex-discovery.test.ts @@ -1,131 +1,88 @@ -import { afterEach, beforeEach, describe, expect, it } from "bun:test"; -import * as fs from "node:fs/promises"; -import * as os from "node:os"; -import * as path from "node:path"; +import { describe, expect, it } from "bun:test"; import { resolveProviderModels } from "../src/model-manager"; import { googleVertexModelManagerOptions } from "../src/provider-models/google"; -import { __resetVertexTokenCache } from "../src/providers/google-auth"; +import { MODELS_DEV_PROVIDER_DESCRIPTORS, mapModelsDevToModels } from "../src/provider-models/openai-compat"; -const OAUTH_TOKEN_URL = "https://oauth2.googleapis.com/token"; -const METADATA_TOKEN_URL = "http://metadata.google.internal/computeMetadata/v1/instance/service-accounts/default/token"; - -describe("google-vertex model discovery", () => { - let tempDir = ""; - let dbPath = ""; - - beforeEach(async () => { - tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-vertex-models-")); - dbPath = path.join(tempDir, "models.db"); - }); - - afterEach(async () => { - __resetVertexTokenCache(); - if (tempDir) { - await fs.rm(tempDir, { recursive: true, force: true }); - tempDir = ""; - dbPath = ""; - } - }); - - it("uses the Vertex OpenAI-compatible model list as the authoritative project catalog", async () => { - const urls: string[] = []; - const options = googleVertexModelManagerOptions({ - project: "vertex-project", - location: "global", - fetch: async input => { - const url = input instanceof Request ? input.url : input.toString(); - urls.push(url); - if (url === METADATA_TOKEN_URL || url === OAUTH_TOKEN_URL) { - return new Response(JSON.stringify({ access_token: "vertex-token", expires_in: 3600 })); - } - if ( - url.startsWith( - "https://aiplatform.googleapis.com/v1/projects/vertex-project/locations/global/endpoints/openapi/models", - ) - ) { - return new Response( - JSON.stringify({ - data: [ - { id: "zai-org/glm-4.7-maas", displayName: "GLM-4.7" }, - { - name: "projects/vertex-project/locations/global/publishers/anthropic/models/claude-sonnet-4-5", - displayName: "Claude Sonnet 4.5", - }, - ], - }), - ); - } - return new Response("not found", { status: 404 }); +const googleVertexModelsDevPayload = { + "google-vertex": { + models: { + "gemini-3.5-flash": { + name: "Gemini 3.5 Flash", + tool_call: true, + reasoning: true, + modalities: { input: ["text", "image", "pdf"] }, + limit: { context: 1_048_576, output: 65_536 }, + cost: { input: 0.3, output: 2.5, cache_read: 0.03, cache_write: 0.75 }, + provider: { npm: "@ai-sdk/google-vertex" }, }, - }); + "deepseek-ai/deepseek-v3.2-maas": { + name: "DeepSeek V3.2", + tool_call: true, + reasoning: true, + modalities: { input: ["text", "pdf"] }, + limit: { context: 163_840, output: 65_536 }, + provider: { npm: "@ai-sdk/openai-compatible" }, + }, + "claude-sonnet-4@20250514": { + name: "Claude Sonnet 4", + tool_call: true, + reasoning: true, + modalities: { input: ["text", "image", "pdf"] }, + limit: { context: 200_000, output: 64_000 }, + provider: { npm: "@ai-sdk/google-vertex/anthropic" }, + }, + "gemini-embedding-001": { + name: "Gemini Embedding 001", + tool_call: false, + provider: { npm: "@ai-sdk/google-vertex" }, + }, + }, + }, +} satisfies Record; - const result = await resolveProviderModels({ ...options, cacheDbPath: dbPath }, "online"); - - expect(result.stale).toBe(false); - expect(result.models.map(model => model.id)).toEqual(["anthropic/claude-sonnet-4-5", "zai-org/glm-4.7-maas"]); - expect(result.models.every(model => model.provider === "google-vertex")).toBe(true); - expect(result.models.every(model => model.api === "openai-completions")).toBe(true); - expect( - result.models.every( - model => - model.baseUrl === - "https://aiplatform.googleapis.com/v1/projects/vertex-project/locations/global/endpoints/openapi", - ), - ).toBe(true); - expect(result.models.some(model => model.id === "gemini-1.5-pro")).toBe(false); - expect(urls).toContain( - "https://aiplatform.googleapis.com/v1/projects/vertex-project/locations/global/endpoints/openapi/models?pageSize=100", +describe("google-vertex model catalog", () => { + it("maps the models.dev Vertex catalog instead of the project discovery endpoint", () => { + const models = mapModelsDevToModels(googleVertexModelsDevPayload, MODELS_DEV_PROVIDER_DESCRIPTORS).filter( + model => model.provider === "google-vertex", ); + + expect(models.map(model => model.id)).toEqual([ + "gemini-3.5-flash", + "deepseek-ai/deepseek-v3.2-maas", + "claude-sonnet-4@20250514", + ]); + + const gemini = models.find(model => model.id === "gemini-3.5-flash"); + expect(gemini?.api).toBe("google-vertex"); + expect(gemini?.baseUrl).toBe("https://{location}-aiplatform.googleapis.com"); + expect(gemini?.input).toEqual(["text", "image"]); + expect(gemini?.contextWindow).toBe(1_048_576); + + const deepseek = models.find(model => model.id === "deepseek-ai/deepseek-v3.2-maas"); + expect(deepseek?.api).toBe("openai-completions"); + expect(deepseek?.baseUrl).toBe( + "https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/endpoints/openapi", + ); + + const claude = models.find(model => model.id === "claude-sonnet-4@20250514"); + expect(claude?.api).toBe("google-vertex"); + expect(claude?.reasoning).toBe(true); }); - it("keeps the API-key Vertex Gemini catalog when project and location are also configured", async () => { + it("uses the bundled Vertex catalog without ADC project discovery", async () => { const options = googleVertexModelManagerOptions({ - apiKey: "vertex-api-key", project: "vertex-project", location: "global", + fetch: async () => new Response("unexpected", { status: 500 }), }); - const result = await resolveProviderModels({ ...options, cacheDbPath: dbPath }, "offline"); + expect(options.fetchDynamicModels).toBeUndefined(); + expect(options.staticModels).toBeUndefined(); - expect(result.models.some(model => model.id === "gemini-2.5-pro")).toBe(true); - expect(result.models.every(model => model.provider === "google-vertex")).toBe(true); - expect(result.models.every(model => model.api === "google-vertex")).toBe(true); - }); - - it("omits the bundled Vertex Gemini static fallback when neither ADC project nor API key are configured", async () => { - const previousProject = Bun.env.GOOGLE_CLOUD_PROJECT; - const previousGcpProject = Bun.env.GCP_PROJECT; - const previousGcloudProject = Bun.env.GCLOUD_PROJECT; - const previousVertexLocation = Bun.env.GOOGLE_VERTEX_LOCATION; - const previousCloudLocation = Bun.env.GOOGLE_CLOUD_LOCATION; - const previousLocation = Bun.env.VERTEX_LOCATION; - const previousApiKey = Bun.env.GOOGLE_CLOUD_API_KEY; - delete Bun.env.GOOGLE_CLOUD_PROJECT; - delete Bun.env.GCP_PROJECT; - delete Bun.env.GCLOUD_PROJECT; - delete Bun.env.GOOGLE_VERTEX_LOCATION; - delete Bun.env.GOOGLE_CLOUD_LOCATION; - delete Bun.env.VERTEX_LOCATION; - delete Bun.env.GOOGLE_CLOUD_API_KEY; - try { - const options = googleVertexModelManagerOptions(); - const result = await resolveProviderModels({ ...options, cacheDbPath: dbPath }, "offline"); - expect(result.models).toEqual([]); - } finally { - if (previousProject === undefined) delete Bun.env.GOOGLE_CLOUD_PROJECT; - else Bun.env.GOOGLE_CLOUD_PROJECT = previousProject; - if (previousGcpProject === undefined) delete Bun.env.GCP_PROJECT; - else Bun.env.GCP_PROJECT = previousGcpProject; - if (previousGcloudProject === undefined) delete Bun.env.GCLOUD_PROJECT; - else Bun.env.GCLOUD_PROJECT = previousGcloudProject; - if (previousVertexLocation === undefined) delete Bun.env.GOOGLE_VERTEX_LOCATION; - else Bun.env.GOOGLE_VERTEX_LOCATION = previousVertexLocation; - if (previousCloudLocation === undefined) delete Bun.env.GOOGLE_CLOUD_LOCATION; - else Bun.env.GOOGLE_CLOUD_LOCATION = previousCloudLocation; - if (previousLocation === undefined) delete Bun.env.VERTEX_LOCATION; - else Bun.env.VERTEX_LOCATION = previousLocation; - if (previousApiKey === undefined) delete Bun.env.GOOGLE_CLOUD_API_KEY; - else Bun.env.GOOGLE_CLOUD_API_KEY = previousApiKey; - } + const result = await resolveProviderModels(options, "offline"); + expect(result.stale).toBe(false); + expect(result.models.some(model => model.id === "deepseek-ai/deepseek-v3.2-maas")).toBe(true); + expect(result.models.some(model => model.id === "gemini-3.5-flash")).toBe(true); + expect(result.models.some(model => model.id === "gemini-1.5-pro")).toBe(false); }); });