diff --git a/package.json b/package.json index 0ad2ef8cb..81d109ac2 100644 --- a/package.json +++ b/package.json @@ -32,7 +32,8 @@ "prepublishOnly": "bun run check", "publish": "bun run prepublishOnly && npm publish -ws --access public", "publish:dry": "bun run prepublishOnly && npm publish -ws --access public --dry-run", - "release": "bun scripts/release.ts" + "release": "bun scripts/release.ts", + "generate-models": "bun --cwd=packages/ai scripts/generate-models.ts" }, "devDependencies": { "@biomejs/biome": "^2.3.13", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 78480ec11..bfc16ed8a 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -1,8 +1,11 @@ # Changelog ## [Unreleased] - ### Added + +- Added OpenCode Zen provider with API key authentication for accessing multiple AI models +- Added 4 new free models via OpenCode: glm-4.7-free, kimi-k2.5-free, minimax-m2.1-free, trinity-large-preview-free +- Added glm-4.7-flash model via Zai provider - Added Kimi Code provider with OpenAI and Anthropic API format support - Added prompt cache retention support with PI_CACHE_RETENTION env var - Added overflow patterns for Bedrock, MiniMax, Kimi; reclassified 429 as rate limiting @@ -16,10 +19,27 @@ - Added ToolChoice type for controlling tool selection (auto, none, any, required, function) ### Changed + +- Updated Kimi K2.5 cache read pricing from 0.1 to 0.08 +- Updated MiniMax M2 pricing: input 0.6→0.6, output 3→3, cache read 0.1→0.09999999999999999 +- Updated OpenRouter DeepSeek V3.1 pricing and max tokens: input 0.6→0.5, output 3→2.8, maxTokens 262144→4096 +- Updated OpenRouter DeepSeek R1 pricing and max tokens: input 0.06→0.049999999999999996, output 0.24→0.19999999999999998, maxTokens 262144→4096 +- Updated Anthropic Claude 3.5 Sonnet max tokens from 256000 to 65536 on OpenRouter +- Updated Vercel AI Gateway Claude 3.5 Sonnet cache read pricing from 0.125 to 0.13 +- Updated Vercel AI Gateway Claude 3.5 Sonnet New cache read pricing from 0.125 to 0.13 +- Updated Vercel AI Gateway GPT-5.2 cache read pricing from 0.175 to 0.18 and display name to 'GPT 5.2' +- Updated Zai GLM-4.6 cache read pricing from 0.024999999999999998 to 0.03 +- Updated Zai Qwen QwQ max tokens from 66000 to 16384 - Added delta event batching and throttling (50ms, 20 updates/sec max) to AssistantMessageEventStream - Updated MiniMax-M2 pricing: input 1.2→0.6, output 1.2→3, cacheRead 0.6→0.1 +### Removed + +- Removed OpenRouter google/gemini-2.0-flash-exp:free model +- Removed Vercel AI Gateway stealth/sonoma-dusk-alpha and stealth/sonoma-sky-alpha models + ### Fixed + - Fixed rate limit issues with Kimi models by always sending max_tokens - Added handling for sensitive stop reason from Anthropic API safety filters - Added optional chaining for safer JSON schema property access in Anthropic provider diff --git a/packages/ai/src/models.generated.ts b/packages/ai/src/models.generated.ts index 3d4bbe199..7fae097d4 100644 --- a/packages/ai/src/models.generated.ts +++ b/packages/ai/src/models.generated.ts @@ -3102,8 +3102,8 @@ export const MODELS = { reasoning: false, input: ["text"], cost: { - input: 0, - output: 0, + input: 0.4, + output: 2, cacheRead: 0, cacheWrite: 0, }, @@ -4357,6 +4357,23 @@ export const MODELS = { contextWindow: 204800, maxTokens: 131072, } satisfies Model<"openai-completions">, + "glm-4.7-free": { + id: "glm-4.7-free", + name: "GLM-4.7 Free", + api: "openai-completions", + provider: "opencode", + baseUrl: "https://opencode.ai/zen/v1", + reasoning: true, + input: ["text"], + cost: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + }, + contextWindow: 204800, + maxTokens: 131072, + } satisfies Model<"openai-completions">, "gpt-5": { id: "gpt-5", name: "GPT-5", @@ -4555,7 +4572,24 @@ export const MODELS = { cost: { input: 0.6, output: 3, - cacheRead: 0.1, + cacheRead: 0.08, + cacheWrite: 0, + }, + contextWindow: 262144, + maxTokens: 262144, + } satisfies Model<"openai-completions">, + "kimi-k2.5-free": { + id: "kimi-k2.5-free", + name: "Kimi K2.5 Free", + api: "openai-completions", + provider: "opencode", + baseUrl: "https://opencode.ai/zen/v1", + reasoning: true, + input: ["text", "image"], + cost: { + input: 0, + output: 0, + cacheRead: 0, cacheWrite: 0, }, contextWindow: 262144, @@ -4578,6 +4612,23 @@ export const MODELS = { contextWindow: 204800, maxTokens: 131072, } satisfies Model<"openai-completions">, + "minimax-m2.1-free": { + id: "minimax-m2.1-free", + name: "MiniMax M2.1 Free", + api: "anthropic-messages", + provider: "opencode", + baseUrl: "https://opencode.ai/zen", + reasoning: true, + input: ["text"], + cost: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + }, + contextWindow: 204800, + maxTokens: 131072, + } satisfies Model<"anthropic-messages">, "qwen3-coder": { id: "qwen3-coder", name: "Qwen3 Coder", @@ -4595,6 +4646,23 @@ export const MODELS = { contextWindow: 262144, maxTokens: 65536, } satisfies Model<"openai-completions">, + "trinity-large-preview-free": { + id: "trinity-large-preview-free", + name: "Trinity Large Preview", + api: "openai-completions", + provider: "opencode", + baseUrl: "https://opencode.ai/zen/v1", + reasoning: false, + input: ["text"], + cost: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + }, + contextWindow: 131072, + maxTokens: 131072, + } satisfies Model<"openai-completions">, }, "openrouter": { "ai21/jamba-large-1.7": { @@ -5350,23 +5418,6 @@ export const MODELS = { contextWindow: 1048576, maxTokens: 8192, } satisfies Model<"openai-completions">, - "google/gemini-2.0-flash-exp:free": { - id: "google/gemini-2.0-flash-exp:free", - name: "Google: Gemini 2.0 Flash Experimental (free)", - api: "openai-completions", - provider: "openrouter", - baseUrl: "https://openrouter.ai/api/v1", - reasoning: false, - input: ["text", "image"], - cost: { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - }, - contextWindow: 1048576, - maxTokens: 8192, - } satisfies Model<"openai-completions">, "google/gemini-2.0-flash-lite-001": { id: "google/gemini-2.0-flash-lite-001", name: "Google: Gemini 2.0 Flash Lite", @@ -6362,13 +6413,13 @@ export const MODELS = { reasoning: true, input: ["text", "image"], cost: { - input: 0.6, - output: 3, + input: 0.5, + output: 2.8, cacheRead: 0, cacheWrite: 0, }, contextWindow: 262144, - maxTokens: 262144, + maxTokens: 4096, } satisfies Model<"openai-completions">, "nex-agi/deepseek-v3.1-nex-n1": { id: "nex-agi/deepseek-v3.1-nex-n1", @@ -6464,13 +6515,13 @@ export const MODELS = { reasoning: true, input: ["text"], cost: { - input: 0.06, - output: 0.24, + input: 0.049999999999999996, + output: 0.19999999999999998, cacheRead: 0, cacheWrite: 0, }, contextWindow: 262144, - maxTokens: 262144, + maxTokens: 4096, } satisfies Model<"openai-completions">, "nvidia/nemotron-3-nano-30b-a3b:free": { id: "nvidia/nemotron-3-nano-30b-a3b:free", @@ -8699,7 +8750,7 @@ export const MODELS = { cacheWrite: 0, }, contextWindow: 256000, - maxTokens: 256000, + maxTokens: 65536, } satisfies Model<"anthropic-messages">, "anthropic/claude-3-haiku": { id: "anthropic/claude-3-haiku", @@ -9611,9 +9662,9 @@ export const MODELS = { reasoning: true, input: ["text", "image"], cost: { - input: 1.2, - output: 1.2, - cacheRead: 0.6, + input: 0.6, + output: 3, + cacheRead: 0.09999999999999999, cacheWrite: 0, }, contextWindow: 256000, @@ -9732,7 +9783,7 @@ export const MODELS = { cost: { input: 0.09999999999999999, output: 0.39999999999999997, - cacheRead: 0.024999999999999998, + cacheRead: 0.03, cacheWrite: 0, }, contextWindow: 1047576, @@ -9936,7 +9987,7 @@ export const MODELS = { cost: { input: 1.25, output: 10, - cacheRead: 0.125, + cacheRead: 0.13, cacheWrite: 0, }, contextWindow: 128000, @@ -9953,7 +10004,7 @@ export const MODELS = { cost: { input: 1.25, output: 10, - cacheRead: 0.125, + cacheRead: 0.13, cacheWrite: 0, }, contextWindow: 400000, @@ -9961,7 +10012,7 @@ export const MODELS = { } satisfies Model<"anthropic-messages">, "openai/gpt-5.2": { id: "openai/gpt-5.2", - name: "GPT-5.2", + name: "GPT 5.2", api: "anthropic-messages", provider: "vercel-ai-gateway", baseUrl: "https://ai-gateway.vercel.sh", @@ -9970,7 +10021,7 @@ export const MODELS = { cost: { input: 1.75, output: 14, - cacheRead: 0.175, + cacheRead: 0.18, cacheWrite: 0, }, contextWindow: 400000, @@ -10231,40 +10282,6 @@ export const MODELS = { contextWindow: 131072, maxTokens: 131072, } satisfies Model<"anthropic-messages">, - "stealth/sonoma-dusk-alpha": { - id: "stealth/sonoma-dusk-alpha", - name: "Sonoma Dusk Alpha", - api: "anthropic-messages", - provider: "vercel-ai-gateway", - baseUrl: "https://ai-gateway.vercel.sh", - reasoning: false, - input: ["text", "image"], - cost: { - input: 0.19999999999999998, - output: 0.5, - cacheRead: 0.049999999999999996, - cacheWrite: 0, - }, - contextWindow: 2000000, - maxTokens: 131072, - } satisfies Model<"anthropic-messages">, - "stealth/sonoma-sky-alpha": { - id: "stealth/sonoma-sky-alpha", - name: "Sonoma Sky Alpha", - api: "anthropic-messages", - provider: "vercel-ai-gateway", - baseUrl: "https://ai-gateway.vercel.sh", - reasoning: false, - input: ["text", "image"], - cost: { - input: 0.19999999999999998, - output: 0.5, - cacheRead: 0.049999999999999996, - cacheWrite: 0, - }, - contextWindow: 2000000, - maxTokens: 131072, - } satisfies Model<"anthropic-messages">, "vercel/v0-1.0-md": { id: "vercel/v0-1.0-md", name: "v0-1.0-md", @@ -10552,7 +10569,7 @@ export const MODELS = { cacheWrite: 0, }, contextWindow: 65536, - maxTokens: 66000, + maxTokens: 16384, } satisfies Model<"anthropic-messages">, "zai/glm-4.6": { id: "zai/glm-4.6", @@ -11143,5 +11160,23 @@ export const MODELS = { contextWindow: 204800, maxTokens: 131072, } satisfies Model<"openai-completions">, + "glm-4.7-flash": { + id: "glm-4.7-flash", + name: "GLM-4.7-Flash", + api: "openai-completions", + provider: "zai", + baseUrl: "https://api.z.ai/api/coding/paas/v4", + compat: {"supportsDeveloperRole":false,"thinkingFormat":"zai"}, + reasoning: true, + input: ["text"], + cost: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + }, + contextWindow: 200000, + maxTokens: 131072, + } satisfies Model<"openai-completions">, }, } as const; diff --git a/packages/ai/src/utils/oauth/index.ts b/packages/ai/src/utils/oauth/index.ts index 98c64f774..1cf8a936d 100644 --- a/packages/ai/src/utils/oauth/index.ts +++ b/packages/ai/src/utils/oauth/index.ts @@ -48,6 +48,8 @@ export { loginKimi, refreshKimiToken } from "./kimi"; export type { OpenAICodexLoginOptions } from "./openai-codex"; // OpenAI Codex (ChatGPT OAuth) export { loginOpenAICodex, refreshOpenAICodexToken } from "./openai-codex"; +// OpenCode (API key) +export { loginOpenCode } from "./opencode"; export * from "./types"; @@ -93,6 +95,10 @@ export async function refreshOAuthToken( case "cursor": newCredentials = await refreshCursorToken(credentials.refresh); break; + case "opencode": + // API keys don't expire, return as-is + newCredentials = credentials; + break; default: throw new Error(`Unknown OAuth provider: ${provider}`); } @@ -173,5 +179,10 @@ export function getOAuthProviders(): OAuthProviderInfo[] { name: "Cursor (Claude, GPT, etc.)", available: true, }, + { + id: "opencode", + name: "OpenCode Zen", + available: true, + }, ]; } diff --git a/packages/ai/src/utils/oauth/opencode.ts b/packages/ai/src/utils/oauth/opencode.ts new file mode 100644 index 000000000..7a9681bfe --- /dev/null +++ b/packages/ai/src/utils/oauth/opencode.ts @@ -0,0 +1,50 @@ +/** + * OpenCode Zen login flow. + * + * OpenCode Zen is a subscription service that provides access to various AI models + * (GPT-5.x, Claude 4.x, Gemini 3, etc.) through a unified API at opencode.ai/zen. + * + * This is not OAuth - it's a simple API key flow: + * 1. Open browser to https://opencode.ai/auth + * 2. User logs in and copies their API key + * 3. User pastes the API key back into the CLI + */ + +import type { OAuthController } from "./types"; + +const AUTH_URL = "https://opencode.ai/auth"; + +/** + * Login to OpenCode Zen. + * + * Opens browser to auth page, prompts user to paste their API key. + * Returns the API key directly (not OAuthCredentials - this isn't OAuth). + */ +export async function loginOpenCode(options: OAuthController): Promise { + if (!options.onPrompt) { + throw new Error("OpenCode Zen login requires onPrompt callback"); + } + + // Open browser to auth page + options.onAuth?.({ + url: AUTH_URL, + instructions: "Log in and copy your API key", + }); + + // Prompt user to paste their API key + const apiKey = await options.onPrompt({ + message: "Paste your OpenCode Zen API key", + placeholder: "sk-...", + }); + + if (options.signal?.aborted) { + throw new Error("Login cancelled"); + } + + const trimmed = apiKey.trim(); + if (!trimmed) { + throw new Error("API key is required"); + } + + return trimmed; +} diff --git a/packages/ai/src/utils/oauth/types.ts b/packages/ai/src/utils/oauth/types.ts index dcd00ca18..50b63f217 100644 --- a/packages/ai/src/utils/oauth/types.ts +++ b/packages/ai/src/utils/oauth/types.ts @@ -15,6 +15,7 @@ export type OAuthProvider = | "google-antigravity" | "kimi-code" | "openai-codex" + | "opencode" | "cursor"; export type OAuthPrompt = { diff --git a/packages/coding-agent/src/session/auth-storage.ts b/packages/coding-agent/src/session/auth-storage.ts index c859ba4ef..619974e0c 100644 --- a/packages/coding-agent/src/session/auth-storage.ts +++ b/packages/coding-agent/src/session/auth-storage.ts @@ -19,6 +19,7 @@ import { loginGitHubCopilot, loginKimi, loginOpenAICodex, + loginOpenCode, type OAuthController, type OAuthCredentials, type OAuthProvider, @@ -782,6 +783,11 @@ export class AuthStorage { ctrl.onProgress ? () => ctrl.onProgress?.("Waiting for browser authentication...") : undefined, ); break; + case "opencode": { + const apiKey = await loginOpenCode(ctrl); + credentials = { access: apiKey, refresh: apiKey, expires: Number.MAX_SAFE_INTEGER }; + break; + } default: throw new Error(`Unknown OAuth provider: ${provider}`); }