feat(ai): added OpenCode Zen provider and updated AI model configurations

- Added OpenCode Zen provider with API key-based authentication supporting multiple AI models.
- Added four new free models via OpenCode provider: glm-4.7-free, kimi-k2.5-free, minimax-m2.1-free, and trinity-large-preview-free.
- Added glm-4.7-flash model via Zai provider.
- Updated pricing and configuration for multiple models across Kimi, MiniMax, OpenRouter, Anthropic, Vercel AI Gateway, and Zai providers.
- Removed google/gemini-2.0-flash-exp:free from OpenRouter and stealth models from Vercel AI Gateway.
This commit is contained in:
can1357
2026-01-31 02:11:23 +01:00
parent 5280292306
commit 1f6c61ac3a
7 changed files with 196 additions and 72 deletions
+2 -1
View File
@@ -32,7 +32,8 @@
"prepublishOnly": "bun run check",
"publish": "bun run prepublishOnly && npm publish -ws --access public",
"publish:dry": "bun run prepublishOnly && npm publish -ws --access public --dry-run",
"release": "bun scripts/release.ts"
"release": "bun scripts/release.ts",
"generate-models": "bun --cwd=packages/ai scripts/generate-models.ts"
},
"devDependencies": {
"@biomejs/biome": "^2.3.13",
+21 -1
View File
@@ -1,8 +1,11 @@
# Changelog
## [Unreleased]
### Added
- Added OpenCode Zen provider with API key authentication for accessing multiple AI models
- Added 4 new free models via OpenCode: glm-4.7-free, kimi-k2.5-free, minimax-m2.1-free, trinity-large-preview-free
- Added glm-4.7-flash model via Zai provider
- Added Kimi Code provider with OpenAI and Anthropic API format support
- Added prompt cache retention support with PI_CACHE_RETENTION env var
- Added overflow patterns for Bedrock, MiniMax, Kimi; reclassified 429 as rate limiting
@@ -16,10 +19,27 @@
- Added ToolChoice type for controlling tool selection (auto, none, any, required, function)
### Changed
- Updated Kimi K2.5 cache read pricing from 0.1 to 0.08
- Updated MiniMax M2 pricing: input 0.6→0.6, output 3→3, cache read 0.1→0.09999999999999999
- Updated OpenRouter DeepSeek V3.1 pricing and max tokens: input 0.6→0.5, output 3→2.8, maxTokens 262144→4096
- Updated OpenRouter DeepSeek R1 pricing and max tokens: input 0.06→0.049999999999999996, output 0.24→0.19999999999999998, maxTokens 262144→4096
- Updated Anthropic Claude 3.5 Sonnet max tokens from 256000 to 65536 on OpenRouter
- Updated Vercel AI Gateway Claude 3.5 Sonnet cache read pricing from 0.125 to 0.13
- Updated Vercel AI Gateway Claude 3.5 Sonnet New cache read pricing from 0.125 to 0.13
- Updated Vercel AI Gateway GPT-5.2 cache read pricing from 0.175 to 0.18 and display name to 'GPT 5.2'
- Updated Zai GLM-4.6 cache read pricing from 0.024999999999999998 to 0.03
- Updated Zai Qwen QwQ max tokens from 66000 to 16384
- Added delta event batching and throttling (50ms, 20 updates/sec max) to AssistantMessageEventStream
- Updated MiniMax-M2 pricing: input 1.2→0.6, output 1.2→3, cacheRead 0.6→0.1
### Removed
- Removed OpenRouter google/gemini-2.0-flash-exp:free model
- Removed Vercel AI Gateway stealth/sonoma-dusk-alpha and stealth/sonoma-sky-alpha models
### Fixed
- Fixed rate limit issues with Kimi models by always sending max_tokens
- Added handling for sensitive stop reason from Anthropic API safety filters
- Added optional chaining for safer JSON schema property access in Anthropic provider
+105 -70
View File
@@ -3102,8 +3102,8 @@ export const MODELS = {
reasoning: false,
input: ["text"],
cost: {
input: 0,
output: 0,
input: 0.4,
output: 2,
cacheRead: 0,
cacheWrite: 0,
},
@@ -4357,6 +4357,23 @@ export const MODELS = {
contextWindow: 204800,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"glm-4.7-free": {
id: "glm-4.7-free",
name: "GLM-4.7 Free",
api: "openai-completions",
provider: "opencode",
baseUrl: "https://opencode.ai/zen/v1",
reasoning: true,
input: ["text"],
cost: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 204800,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"gpt-5": {
id: "gpt-5",
name: "GPT-5",
@@ -4555,7 +4572,24 @@ export const MODELS = {
cost: {
input: 0.6,
output: 3,
cacheRead: 0.1,
cacheRead: 0.08,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 262144,
} satisfies Model<"openai-completions">,
"kimi-k2.5-free": {
id: "kimi-k2.5-free",
name: "Kimi K2.5 Free",
api: "openai-completions",
provider: "opencode",
baseUrl: "https://opencode.ai/zen/v1",
reasoning: true,
input: ["text", "image"],
cost: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 262144,
@@ -4578,6 +4612,23 @@ export const MODELS = {
contextWindow: 204800,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"minimax-m2.1-free": {
id: "minimax-m2.1-free",
name: "MiniMax M2.1 Free",
api: "anthropic-messages",
provider: "opencode",
baseUrl: "https://opencode.ai/zen",
reasoning: true,
input: ["text"],
cost: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 204800,
maxTokens: 131072,
} satisfies Model<"anthropic-messages">,
"qwen3-coder": {
id: "qwen3-coder",
name: "Qwen3 Coder",
@@ -4595,6 +4646,23 @@ export const MODELS = {
contextWindow: 262144,
maxTokens: 65536,
} satisfies Model<"openai-completions">,
"trinity-large-preview-free": {
id: "trinity-large-preview-free",
name: "Trinity Large Preview",
api: "openai-completions",
provider: "opencode",
baseUrl: "https://opencode.ai/zen/v1",
reasoning: false,
input: ["text"],
cost: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 131072,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
},
"openrouter": {
"ai21/jamba-large-1.7": {
@@ -5350,23 +5418,6 @@ export const MODELS = {
contextWindow: 1048576,
maxTokens: 8192,
} satisfies Model<"openai-completions">,
"google/gemini-2.0-flash-exp:free": {
id: "google/gemini-2.0-flash-exp:free",
name: "Google: Gemini 2.0 Flash Experimental (free)",
api: "openai-completions",
provider: "openrouter",
baseUrl: "https://openrouter.ai/api/v1",
reasoning: false,
input: ["text", "image"],
cost: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 1048576,
maxTokens: 8192,
} satisfies Model<"openai-completions">,
"google/gemini-2.0-flash-lite-001": {
id: "google/gemini-2.0-flash-lite-001",
name: "Google: Gemini 2.0 Flash Lite",
@@ -6362,13 +6413,13 @@ export const MODELS = {
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.6,
output: 3,
input: 0.5,
output: 2.8,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 262144,
maxTokens: 4096,
} satisfies Model<"openai-completions">,
"nex-agi/deepseek-v3.1-nex-n1": {
id: "nex-agi/deepseek-v3.1-nex-n1",
@@ -6464,13 +6515,13 @@ export const MODELS = {
reasoning: true,
input: ["text"],
cost: {
input: 0.06,
output: 0.24,
input: 0.049999999999999996,
output: 0.19999999999999998,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 262144,
maxTokens: 4096,
} satisfies Model<"openai-completions">,
"nvidia/nemotron-3-nano-30b-a3b:free": {
id: "nvidia/nemotron-3-nano-30b-a3b:free",
@@ -8699,7 +8750,7 @@ export const MODELS = {
cacheWrite: 0,
},
contextWindow: 256000,
maxTokens: 256000,
maxTokens: 65536,
} satisfies Model<"anthropic-messages">,
"anthropic/claude-3-haiku": {
id: "anthropic/claude-3-haiku",
@@ -9611,9 +9662,9 @@ export const MODELS = {
reasoning: true,
input: ["text", "image"],
cost: {
input: 1.2,
output: 1.2,
cacheRead: 0.6,
input: 0.6,
output: 3,
cacheRead: 0.09999999999999999,
cacheWrite: 0,
},
contextWindow: 256000,
@@ -9732,7 +9783,7 @@ export const MODELS = {
cost: {
input: 0.09999999999999999,
output: 0.39999999999999997,
cacheRead: 0.024999999999999998,
cacheRead: 0.03,
cacheWrite: 0,
},
contextWindow: 1047576,
@@ -9936,7 +9987,7 @@ export const MODELS = {
cost: {
input: 1.25,
output: 10,
cacheRead: 0.125,
cacheRead: 0.13,
cacheWrite: 0,
},
contextWindow: 128000,
@@ -9953,7 +10004,7 @@ export const MODELS = {
cost: {
input: 1.25,
output: 10,
cacheRead: 0.125,
cacheRead: 0.13,
cacheWrite: 0,
},
contextWindow: 400000,
@@ -9961,7 +10012,7 @@ export const MODELS = {
} satisfies Model<"anthropic-messages">,
"openai/gpt-5.2": {
id: "openai/gpt-5.2",
name: "GPT-5.2",
name: "GPT 5.2",
api: "anthropic-messages",
provider: "vercel-ai-gateway",
baseUrl: "https://ai-gateway.vercel.sh",
@@ -9970,7 +10021,7 @@ export const MODELS = {
cost: {
input: 1.75,
output: 14,
cacheRead: 0.175,
cacheRead: 0.18,
cacheWrite: 0,
},
contextWindow: 400000,
@@ -10231,40 +10282,6 @@ export const MODELS = {
contextWindow: 131072,
maxTokens: 131072,
} satisfies Model<"anthropic-messages">,
"stealth/sonoma-dusk-alpha": {
id: "stealth/sonoma-dusk-alpha",
name: "Sonoma Dusk Alpha",
api: "anthropic-messages",
provider: "vercel-ai-gateway",
baseUrl: "https://ai-gateway.vercel.sh",
reasoning: false,
input: ["text", "image"],
cost: {
input: 0.19999999999999998,
output: 0.5,
cacheRead: 0.049999999999999996,
cacheWrite: 0,
},
contextWindow: 2000000,
maxTokens: 131072,
} satisfies Model<"anthropic-messages">,
"stealth/sonoma-sky-alpha": {
id: "stealth/sonoma-sky-alpha",
name: "Sonoma Sky Alpha",
api: "anthropic-messages",
provider: "vercel-ai-gateway",
baseUrl: "https://ai-gateway.vercel.sh",
reasoning: false,
input: ["text", "image"],
cost: {
input: 0.19999999999999998,
output: 0.5,
cacheRead: 0.049999999999999996,
cacheWrite: 0,
},
contextWindow: 2000000,
maxTokens: 131072,
} satisfies Model<"anthropic-messages">,
"vercel/v0-1.0-md": {
id: "vercel/v0-1.0-md",
name: "v0-1.0-md",
@@ -10552,7 +10569,7 @@ export const MODELS = {
cacheWrite: 0,
},
contextWindow: 65536,
maxTokens: 66000,
maxTokens: 16384,
} satisfies Model<"anthropic-messages">,
"zai/glm-4.6": {
id: "zai/glm-4.6",
@@ -11143,5 +11160,23 @@ export const MODELS = {
contextWindow: 204800,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"glm-4.7-flash": {
id: "glm-4.7-flash",
name: "GLM-4.7-Flash",
api: "openai-completions",
provider: "zai",
baseUrl: "https://api.z.ai/api/coding/paas/v4",
compat: {"supportsDeveloperRole":false,"thinkingFormat":"zai"},
reasoning: true,
input: ["text"],
cost: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 200000,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
},
} as const;
+11
View File
@@ -48,6 +48,8 @@ export { loginKimi, refreshKimiToken } from "./kimi";
export type { OpenAICodexLoginOptions } from "./openai-codex";
// OpenAI Codex (ChatGPT OAuth)
export { loginOpenAICodex, refreshOpenAICodexToken } from "./openai-codex";
// OpenCode (API key)
export { loginOpenCode } from "./opencode";
export * from "./types";
@@ -93,6 +95,10 @@ export async function refreshOAuthToken(
case "cursor":
newCredentials = await refreshCursorToken(credentials.refresh);
break;
case "opencode":
// API keys don't expire, return as-is
newCredentials = credentials;
break;
default:
throw new Error(`Unknown OAuth provider: ${provider}`);
}
@@ -173,5 +179,10 @@ export function getOAuthProviders(): OAuthProviderInfo[] {
name: "Cursor (Claude, GPT, etc.)",
available: true,
},
{
id: "opencode",
name: "OpenCode Zen",
available: true,
},
];
}
+50
View File
@@ -0,0 +1,50 @@
/**
* OpenCode Zen login flow.
*
* OpenCode Zen is a subscription service that provides access to various AI models
* (GPT-5.x, Claude 4.x, Gemini 3, etc.) through a unified API at opencode.ai/zen.
*
* This is not OAuth - it's a simple API key flow:
* 1. Open browser to https://opencode.ai/auth
* 2. User logs in and copies their API key
* 3. User pastes the API key back into the CLI
*/
import type { OAuthController } from "./types";
const AUTH_URL = "https://opencode.ai/auth";
/**
* Login to OpenCode Zen.
*
* Opens browser to auth page, prompts user to paste their API key.
* Returns the API key directly (not OAuthCredentials - this isn't OAuth).
*/
export async function loginOpenCode(options: OAuthController): Promise<string> {
if (!options.onPrompt) {
throw new Error("OpenCode Zen login requires onPrompt callback");
}
// Open browser to auth page
options.onAuth?.({
url: AUTH_URL,
instructions: "Log in and copy your API key",
});
// Prompt user to paste their API key
const apiKey = await options.onPrompt({
message: "Paste your OpenCode Zen API key",
placeholder: "sk-...",
});
if (options.signal?.aborted) {
throw new Error("Login cancelled");
}
const trimmed = apiKey.trim();
if (!trimmed) {
throw new Error("API key is required");
}
return trimmed;
}
+1
View File
@@ -15,6 +15,7 @@ export type OAuthProvider =
| "google-antigravity"
| "kimi-code"
| "openai-codex"
| "opencode"
| "cursor";
export type OAuthPrompt = {
@@ -19,6 +19,7 @@ import {
loginGitHubCopilot,
loginKimi,
loginOpenAICodex,
loginOpenCode,
type OAuthController,
type OAuthCredentials,
type OAuthProvider,
@@ -782,6 +783,11 @@ export class AuthStorage {
ctrl.onProgress ? () => ctrl.onProgress?.("Waiting for browser authentication...") : undefined,
);
break;
case "opencode": {
const apiKey = await loginOpenCode(ctrl);
credentials = { access: apiKey, refresh: apiKey, expires: Number.MAX_SAFE_INTEGER };
break;
}
default:
throw new Error(`Unknown OAuth provider: ${provider}`);
}