From 4c18cc1a1a31b50b4b13593904e0ef8ab346f018 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 3 Jul 2026 05:52:04 +0200 Subject: [PATCH] feat(catalog): integrated baseten provider and updated model definitions - Implement Baseten provider support with authentication and dynamic model discovery. - Register Baseten in the model catalog and provider priority order. - Expand model definitions with new DeepSeek, Kimi, NVIDIA, and Claude variants. - Update model configurations, cost data, and provider-specific metadata. --- packages/ai/CHANGELOG.md | 4 + packages/ai/src/registry/baseten.ts | 22 + packages/ai/src/registry/registry.ts | 2 + packages/catalog/CHANGELOG.md | 14 + packages/catalog/scripts/generate-models.ts | 6 +- packages/catalog/src/hosts.ts | 1 + packages/catalog/src/identity/priority.ts | 1 + packages/catalog/src/models.json | 809 +++++++++++++++--- .../src/provider-models/descriptors.ts | 9 + .../src/provider-models/openai-compat.ts | 97 +++ .../catalog/test/baseten-provider.test.ts | 97 +++ 11 files changed, 959 insertions(+), 103 deletions(-) create mode 100644 packages/ai/src/registry/baseten.ts create mode 100644 packages/catalog/test/baseten-provider.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 2e7c67e01..dcceae02d 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added support for Baseten as an AI provider + ## [16.3.3] - 2026-07-02 ### Added diff --git a/packages/ai/src/registry/baseten.ts b/packages/ai/src/registry/baseten.ts new file mode 100644 index 000000000..5f882cce3 --- /dev/null +++ b/packages/ai/src/registry/baseten.ts @@ -0,0 +1,22 @@ +import { createApiKeyLogin } from "./api-key-login"; +import type { OAuthLoginCallbacks } from "./oauth/types"; +import type { ProviderDefinition } from "./types"; + +export const loginBaseten = createApiKeyLogin({ + providerLabel: "Baseten", + authUrl: "https://app.baseten.co/settings/api_keys", + instructions: "Copy your API key from the Baseten dashboard", + promptMessage: "Paste your Baseten API key", + placeholder: "bt_...", + validation: { + kind: "models-endpoint", + provider: "Baseten", + modelsUrl: "https://inference.baseten.co/v1/models", + }, +}); + +export const basetenProvider = { + id: "baseten", + name: "Baseten", + login: (cb: OAuthLoginCallbacks) => loginBaseten(cb), +} as const satisfies ProviderDefinition; diff --git a/packages/ai/src/registry/registry.ts b/packages/ai/src/registry/registry.ts index 217565e57..e6740996a 100644 --- a/packages/ai/src/registry/registry.ts +++ b/packages/ai/src/registry/registry.ts @@ -4,6 +4,7 @@ import { alibabaCodingPlanProvider } from "./alibaba-coding-plan"; import { amazonBedrockProvider } from "./amazon-bedrock"; import { anthropicProvider } from "./anthropic"; import { azureProvider } from "./azure"; +import { basetenProvider } from "./baseten"; import { cerebrasProvider } from "./cerebras"; import { cloudflareAiGatewayProvider } from "./cloudflare-ai-gateway"; import { coreWeaveProvider } from "./coreweave"; @@ -105,6 +106,7 @@ const ALL = [ deepseekProvider, moonshotProvider, cerebrasProvider, + basetenProvider, fireworksProvider, togetherProvider, nvidiaProvider, diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index e8cab5ff3..672bfcf62 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,20 @@ ## [Unreleased] +### Added + +- Added Baseten as a supported model provider +- Added support for new models from Baseten, including DeepSeek V4 Pro and Kimi series +- Added new Devin agent models: Claude 5 Fable variants +- Added new Github Copilot models: Kimi K2.7 Code and MAI-Code-1-Flash +- Added Poolside Laguna XS 2.1 models via Kilo and OpenRouter providers +- Added support for Claude Fable 5 (Free) via Zenmux provider + +### Changed + +- Updated priority ordering to include Baseten +- Updated pricing and limits for various existing models in the catalog + ## [16.3.3] - 2026-07-02 ### Fixed diff --git a/packages/catalog/scripts/generate-models.ts b/packages/catalog/scripts/generate-models.ts index eee29d44a..09b9548a9 100644 --- a/packages/catalog/scripts/generate-models.ts +++ b/packages/catalog/scripts/generate-models.ts @@ -187,7 +187,11 @@ function applyGlobalModelsDevFallback( const providerScopedKeys = new Set(modelsDevModels.map(model => `${model.provider}/${model.id}`)); const globalReferences = createGlobalModelsDevReferenceMap(modelsDevModels); return models.map(model => { - if (providerScopedKeys.has(`${model.provider}/${model.id}`) || model.provider === "devin") { + if ( + providerScopedKeys.has(`${model.provider}/${model.id}`) || + model.provider === "devin" || + model.provider === "baseten" + ) { return model; } const reference = globalReferences.get(model.id); diff --git a/packages/catalog/src/hosts.ts b/packages/catalog/src/hosts.ts index cfa3bd858..240e59e35 100644 --- a/packages/catalog/src/hosts.ts +++ b/packages/catalog/src/hosts.ts @@ -47,6 +47,7 @@ export const KNOWN_HOSTS = { xai: { providers: ["xai"], urlMarkers: ["api.x.ai"] }, mistral: { providers: ["mistral"], urlMarkers: ["mistral.ai"] }, together: { providers: ["together"], urlMarkers: ["api.together.xyz"] }, + baseten: { providers: ["baseten"], urlMarkers: ["baseten.co"] }, /** URL-only on purpose: the `fireworks`/`firepass` providers route per-model and not every model is Fireworks-shaped. */ fireworks: { urlMarkers: ["fireworks.ai"] }, groq: { providers: ["groq"], urlMarkers: ["api.groq.com"] }, diff --git a/packages/catalog/src/identity/priority.ts b/packages/catalog/src/identity/priority.ts index 6612c9e7b..61eb5136d 100644 --- a/packages/catalog/src/identity/priority.ts +++ b/packages/catalog/src/identity/priority.ts @@ -20,6 +20,7 @@ const DEFAULT_MODEL_PROVIDER_ORDER = [ // High-quality aggregators / hosted inference providers. "fireworks", "cerebras", + "baseten", "openrouter", "aimlapi", "together", diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index be01e1c99..8219476f4 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -10999,6 +10999,7 @@ }, "compat": { "officialEndpoint": true, + "signingEndpoint": true, "disableAdaptiveThinking": false, "supportsEagerToolInputStreaming": true, "supportsLongCacheRetention": true, @@ -11052,6 +11053,7 @@ "maxTokens": 4096, "compat": { "officialEndpoint": true, + "signingEndpoint": true, "disableAdaptiveThinking": false, "supportsEagerToolInputStreaming": true, "supportsLongCacheRetention": true, @@ -11085,6 +11087,7 @@ "maxTokens": 4096, "compat": { "officialEndpoint": true, + "signingEndpoint": true, "disableAdaptiveThinking": false, "supportsEagerToolInputStreaming": true, "supportsLongCacheRetention": true, @@ -12655,6 +12658,335 @@ } } }, + "baseten": { + "deepseek-ai/DeepSeek-V4-Pro": { + "id": "deepseek-ai/DeepSeek-V4-Pro", + "name": "Deepseek V4 Pro", + "api": "openai-completions", + "provider": "baseten", + "baseUrl": "https://inference.baseten.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.74, + "output": 3.48, + "cacheRead": 0.145, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, + "moonshotai/Kimi-K2.5": { + "id": "moonshotai/Kimi-K2.5", + "name": "Kimi K2.5", + "api": "openai-completions", + "provider": "baseten", + "baseUrl": "https://inference.baseten.co/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.6, + "output": 3, + "cacheRead": 0.12, + "cacheWrite": 0 + }, + "contextWindow": 262000, + "maxTokens": 262000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "moonshotai/Kimi-K2.6": { + "id": "moonshotai/Kimi-K2.6", + "name": "Kimi K2.6", + "api": "openai-completions", + "provider": "baseten", + "baseUrl": "https://inference.baseten.co/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.95, + "output": 4, + "cacheRead": 0.16, + "cacheWrite": 0 + }, + "contextWindow": 262000, + "maxTokens": 262000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "moonshotai/Kimi-K2.7-Code": { + "id": "moonshotai/Kimi-K2.7-Code", + "name": "Kimi K2.7 Code", + "api": "openai-completions", + "provider": "baseten", + "baseUrl": "https://inference.baseten.co/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.95, + "output": 4, + "cacheRead": 0.16, + "cacheWrite": 0 + }, + "contextWindow": 262000, + "maxTokens": 262000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "nvidia/Nemotron-120B-A12B": { + "id": "nvidia/Nemotron-120B-A12B", + "name": "Nemotron Super", + "api": "openai-completions", + "provider": "baseten", + "baseUrl": "https://inference.baseten.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 0.75, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 202800, + "maxTokens": 202800, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": { + "id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B", + "name": "Nemotron Ultra", + "api": "openai-completions", + "provider": "baseten", + "baseUrl": "https://inference.baseten.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.4, + "cacheRead": 0.12, + "cacheWrite": 0 + }, + "contextWindow": 202800, + "maxTokens": 202800, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai/gpt-oss-120b": { + "id": "openai/gpt-oss-120b", + "name": "OpenAI GPT 120B", + "api": "openai-completions", + "provider": "baseten", + "baseUrl": "https://inference.baseten.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.09999999999999999, + "output": 0.5, + "cacheRead": 0.09999999999999999, + "cacheWrite": 0 + }, + "contextWindow": 128072, + "maxTokens": 128072, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, + "zai-org/GLM-4.7": { + "id": "zai-org/GLM-4.7", + "name": "GLM 4.7", + "api": "openai-completions", + "provider": "baseten", + "baseUrl": "https://inference.baseten.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.2, + "cacheRead": 0.12, + "cacheWrite": 0 + }, + "contextWindow": 200000, + "maxTokens": 200000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/GLM-5": { + "id": "zai-org/GLM-5", + "name": "GLM 5", + "api": "openai-completions", + "provider": "baseten", + "baseUrl": "https://inference.baseten.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.95, + "output": 3.15, + "cacheRead": 0.19999999999999998, + "cacheWrite": 0 + }, + "contextWindow": 202800, + "maxTokens": 202800, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/GLM-5.1": { + "id": "zai-org/GLM-5.1", + "name": "GLM 5.1", + "api": "openai-completions", + "provider": "baseten", + "baseUrl": "https://inference.baseten.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.3, + "output": 4.300000000000001, + "cacheRead": 0.26, + "cacheWrite": 0 + }, + "contextWindow": 202800, + "maxTokens": 202800, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/GLM-5.2": { + "id": "zai-org/GLM-5.2", + "name": "GLM 5.2", + "api": "openai-completions", + "provider": "baseten", + "baseUrl": "https://inference.baseten.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.4, + "output": 4.4, + "cacheRead": 0.26, + "cacheWrite": 0 + }, + "contextWindow": 202720, + "maxTokens": 202720, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + } + }, "cerebras": { "gemma-4-31b": { "id": "gemma-4-31b", @@ -15785,6 +16117,111 @@ } }, "devin": { + "claude-5-fable-high": { + "id": "claude-5-fable-high", + "name": "Claude Fable 5 High", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "claude-5-fable-low": { + "id": "claude-5-fable-low", + "name": "Claude Fable 5 Low", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "claude-5-fable-max": { + "id": "claude-5-fable-max", + "name": "Claude Fable 5 Max", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "claude-5-fable-medium": { + "id": "claude-5-fable-medium", + "name": "Claude Fable 5 Medium", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "claude-5-fable-xhigh": { + "id": "claude-5-fable-xhigh", + "name": "Claude Fable 5 XHigh", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "Claude Opus 4.6", @@ -18854,6 +19291,81 @@ ] } }, + "kimi-k2.7-code": { + "id": "kimi-k2.7-code", + "name": "Kimi K2.7 Code", + "api": "openai-completions", + "provider": "github-copilot", + "baseUrl": "https://api.githubcopilot.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.95, + "output": 4, + "cacheRead": 0.19, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 32000, + "headers": { + "User-Agent": "opencode/1.3.15", + "X-GitHub-Api-Version": "2026-06-01" + }, + "compat": { + "supportsStore": false, + "supportsDeveloperRole": false, + "supportsReasoningEffort": false + }, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "mai-code-1-flash-picker": { + "id": "mai-code-1-flash-picker", + "name": "MAI-Code-1-Flash", + "api": "openai-completions", + "provider": "github-copilot", + "baseUrl": "https://api.githubcopilot.com", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.75, + "output": 4.5, + "cacheRead": 0.075, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 128000, + "headers": { + "User-Agent": "opencode/1.3.15", + "X-GitHub-Api-Version": "2026-06-01" + }, + "compat": { + "supportsStore": false, + "supportsDeveloperRole": false, + "supportsReasoningEffort": false + }, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "raptor-mini": { "id": "raptor-mini", "name": "Raptor mini", @@ -31341,6 +31853,44 @@ ] } }, + "poolside/laguna-xs-2.1": { + "id": "poolside/laguna-xs-2.1", + "name": "Laguna XS 2.1", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768 + }, + "poolside/laguna-xs-2.1:free": { + "id": "poolside/laguna-xs-2.1:free", + "name": "Laguna XS 2.1 (free)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768 + }, "poolside/laguna-xs.2": { "id": "poolside/laguna-xs.2", "name": "Laguna XS.2", @@ -59027,6 +59577,11 @@ "cacheRead": 0, "cacheWrite": 0 }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, "contextWindow": 272000, "maxTokens": 128000, "preferWebsockets": true, @@ -59408,6 +59963,11 @@ "cacheRead": 0.25, "cacheWrite": 0 }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, "contextWindow": 272000, "maxTokens": 128000, "preferWebsockets": true, @@ -59440,6 +60000,11 @@ "cacheRead": 0.075, "cacheWrite": 0 }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, "contextWindow": 272000, "maxTokens": 128000, "preferWebsockets": true, @@ -59504,6 +60069,11 @@ "cacheRead": 0.5, "cacheWrite": 0 }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, "contextWindow": 272000, "maxTokens": 128000, "preferWebsockets": true, @@ -62458,9 +63028,9 @@ "image" ], "cost": { - "input": 0.55, - "output": 3.1999999999999997, - "cacheRead": 0.11, + "input": 0.66, + "output": 3.41, + "cacheRead": 0.14, "cacheWrite": 0 }, "contextWindow": 262144, @@ -64084,13 +64654,13 @@ "text" ], "cost": { - "input": 0.08900000000000001, + "input": 0.09, "output": 0.18, "cacheRead": 0.018, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 65536, + "maxTokens": 16384, "thinking": { "mode": "effort", "efforts": [ @@ -66146,9 +66716,9 @@ "image" ], "cost": { - "input": 0.55, - "output": 3.1999999999999997, - "cacheRead": 0.11, + "input": 0.66, + "output": 3.41, + "cacheRead": 0.14, "cacheWrite": 0 }, "contextWindow": 262144, @@ -68432,6 +69002,62 @@ ] } }, + "poolside/laguna-xs-2.1": { + "id": "poolside/laguna-xs-2.1", + "name": "Laguna XS 2.1", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.06, + "output": 0.12, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "poolside/laguna-xs-2.1:free": { + "id": "poolside/laguna-xs-2.1:free", + "name": "Laguna XS 2.1 (free)", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "poolside/laguna-xs.2": { "id": "poolside/laguna-xs.2", "name": "Laguna XS.2", @@ -69350,8 +69976,8 @@ "image" ], "cost": { - "input": 0.08, - "output": 0.5, + "input": 0.117, + "output": 0.45499999999999996, "cacheRead": 0, "cacheWrite": 0 }, @@ -71131,13 +71757,13 @@ "text" ], "cost": { - "input": 0.975, - "output": 4.300000000000001, - "cacheRead": 0.182, + "input": 0.966, + "output": 3.036, + "cacheRead": 0.1794, "cacheWrite": 0 }, "contextWindow": 202752, - "maxTokens": 65535, + "maxTokens": 128000, "thinking": { "mode": "effort", "efforts": [ @@ -71381,7 +72007,7 @@ "synthetic": { "hf:MiniMaxAI/MiniMax-M3": { "id": "hf:MiniMaxAI/MiniMax-M3", - "name": "MiniMax-M3", + "name": "MiniMaxAI/MiniMax-M3", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71396,36 +72022,6 @@ "cacheRead": 0.6, "cacheWrite": 0 }, - "contextWindow": 524288, - "maxTokens": 65536, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "hf:moonshotai/Kimi-K2.6": { - "id": "hf:moonshotai/Kimi-K2.6", - "name": "moonshotai/Kimi-K2.6", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0.95, - "output": 4, - "cacheRead": 0.95, - "cacheWrite": 0 - }, "contextWindow": 262144, "maxTokens": 65536, "thinking": { @@ -71441,7 +72037,7 @@ }, "hf:moonshotai/Kimi-K2.7-Code": { "id": "hf:moonshotai/Kimi-K2.7-Code", - "name": "Kimi K2.7 Code", + "name": "moonshotai/Kimi-K2.7-Code", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71471,7 +72067,7 @@ }, "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4": { "id": "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4", - "name": "Nemotron 3 Super 120B A12B", + "name": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71500,7 +72096,7 @@ }, "hf:openai/gpt-oss-120b": { "id": "hf:openai/gpt-oss-120b", - "name": "GPT OSS 120B", + "name": "openai/gpt-oss-120b", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71527,7 +72123,7 @@ }, "hf:Qwen/Qwen3.6-27B": { "id": "hf:Qwen/Qwen3.6-27B", - "name": "Qwen3.6 27B", + "name": "Qwen/Qwen3.6-27B", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71556,7 +72152,7 @@ }, "hf:zai-org/GLM-4.7-Flash": { "id": "hf:zai-org/GLM-4.7-Flash", - "name": "GLM-4.7-Flash", + "name": "zai-org/GLM-4.7-Flash", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71583,38 +72179,9 @@ ] } }, - "hf:zai-org/GLM-5.1": { - "id": "hf:zai-org/GLM-5.1", - "name": "zai-org/GLM-5.1", - "api": "openai-completions", - "provider": "synthetic", - "baseUrl": "https://api.synthetic.new/openai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 1, - "output": 3, - "cacheRead": 1, - "cacheWrite": 0 - }, - "contextWindow": 196608, - "maxTokens": 65536, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high", - "xhigh" - ] - } - }, "hf:zai-org/GLM-5.2": { "id": "hf:zai-org/GLM-5.2", - "name": "GLM-5.2", + "name": "zai-org/GLM-5.2", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -71925,8 +72492,8 @@ "text" ], "cost": { - "input": 0.88, - "output": 0.88, + "input": 1.04, + "output": 1.04, "cacheRead": 0, "cacheWrite": 0 }, @@ -81342,12 +81909,12 @@ "text" ], "cost": { - "input": 1.5, - "output": 4.5, - "cacheRead": 0.3, + "input": 1.4, + "output": 4.4, + "cacheRead": 0.26, "cacheWrite": 0 }, - "contextWindow": 1000000, + "contextWindow": 1040000, "maxTokens": 128000, "thinking": { "mode": "budget", @@ -82388,14 +82955,6 @@ }, "contextWindow": 2000000, "maxTokens": 2000000, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "includeEncryptedReasoning": false, - "filterReasoningHistory": true, - "omitReasoningEffort": false - }, "thinking": { "mode": "effort", "efforts": [ @@ -82408,6 +82967,14 @@ "effortMap": { "minimal": "low" } + }, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "omitReasoningEffort": false } }, "grok-4.3": { @@ -82429,14 +82996,6 @@ }, "contextWindow": 1000000, "maxTokens": 1000000, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "includeEncryptedReasoning": false, - "filterReasoningHistory": true, - "omitReasoningEffort": false - }, "thinking": { "mode": "effort", "efforts": [ @@ -82449,6 +83008,14 @@ "effortMap": { "minimal": "low" } + }, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "omitReasoningEffort": false } }, "grok-build": { @@ -83619,6 +84186,44 @@ "supportsDisplay": true } }, + "anthropic/claude-fable-5-free": { + "id": "anthropic/claude-fable-5-free", + "name": "Claude Fable 5 (Free)", + "api": "anthropic-messages", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/anthropic", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "anthropic-adaptive", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + }, + "supportsDisplay": true + } + }, "anthropic/claude-haiku-4.5": { "id": "anthropic/claude-haiku-4.5", "name": "Claude Haiku 4.5", diff --git a/packages/catalog/src/provider-models/descriptors.ts b/packages/catalog/src/provider-models/descriptors.ts index db94db345..29a80ecf4 100644 --- a/packages/catalog/src/provider-models/descriptors.ts +++ b/packages/catalog/src/provider-models/descriptors.ts @@ -12,6 +12,7 @@ import { aimlApiModelManagerOptions, alibabaCodingPlanModelManagerOptions, anthropicModelManagerOptions, + basetenModelManagerOptions, cerebrasModelManagerOptions, cloudflareAiGatewayModelManagerOptions, coreWeaveModelManagerOptions, @@ -73,6 +74,14 @@ export const CATALOG_PROVIDERS = [ createModelManagerOptions: (config: ModelManagerConfig) => alibabaCodingPlanModelManagerOptions(config), catalogDiscovery: { label: "Alibaba Coding Plan" }, }, + { + id: "baseten", + defaultModel: "moonshotai/Kimi-K2.7-Code", + envVars: ["BASETEN_API_KEY"], + createModelManagerOptions: (config: ModelManagerConfig) => basetenModelManagerOptions(config), + dynamicModelsAuthoritative: true, + catalogDiscovery: { label: "Baseten" }, + }, { id: "amazon-bedrock", defaultModel: "us.anthropic.claude-opus-4-8", diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 513a24f7e..d9bd759bd 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -2551,6 +2551,103 @@ export function veniceModelManagerOptions( }; } +// --------------------------------------------------------------------------- +// 14.5 Baseten +// --------------------------------------------------------------------------- + +export interface BasetenModelManagerConfig { + apiKey?: string; + baseUrl?: string; + fetch?: FetchImpl; +} + +export function basetenModelManagerOptions( + config?: BasetenModelManagerConfig, +): ModelManagerOptions<"openai-completions"> { + const apiKey = config?.apiKey; + const baseUrl = config?.baseUrl ?? "https://inference.baseten.co/v1"; + const references = createBundledReferenceMap<"openai-completions">("baseten"); + return { + providerId: "baseten", + dynamicModelsAuthoritative: true, + ...(apiKey && { + fetchDynamicModels: () => + fetchOpenAICompatibleModels({ + api: "openai-completions", + provider: "baseten", + baseUrl, + apiKey, + mapModel: (entry, defaults) => { + const reference = references.get(defaults.id); + const raw = entry as Record & { + supported_features?: unknown; + input_modalities?: unknown; + pricing?: Record; + }; + const features = Array.isArray(raw.supported_features) ? raw.supported_features : []; + const modalities = Array.isArray(raw.input_modalities) ? raw.input_modalities : []; + + const isBasetenNativeReasoning = + defaults.id === "openai/gpt-oss-120b" || + defaults.id === "deepseek-ai/DeepSeek-V4-Pro" || + defaults.id === "zai-org/GLM-5.2"; + const reasoning = + isBasetenNativeReasoning && + (features.includes("reasoning") || features.includes("reasoning_effort")); + const supportsTools = features.includes("tools") ? undefined : false; + const vision = modalities.includes("image") || (reference?.input.includes("image") ?? false); + + const pricing = raw.pricing ?? {}; + const cost = { + input: toPositiveNumber(pricing.prompt, 0) * 1_000_000, + output: toPositiveNumber(pricing.completion, 0) * 1_000_000, + cacheRead: toPositiveNumber(pricing.input_cache_read, 0) * 1_000_000, + cacheWrite: 0, + }; + + const contextWindow = toPositiveNumber( + raw.context_length, + reference?.contextWindow ?? defaults.contextWindow, + ); + const maxTokens = toPositiveNumber( + raw.max_completion_tokens, + reference?.maxTokens ?? defaults.maxTokens, + ); + + const baseModel = mapWithBundledReference(entry, defaults, reference); + + const isEffortReasoning = defaults.id === "openai/gpt-oss-120b" || defaults.id === "zai-org/GLM-5.2"; + const thinking = isEffortReasoning + ? { + mode: "effort" as const, + efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + effortMap: { + minimal: "high", + low: "high", + medium: "high", + high: "high", + xhigh: "max", + }, + } + : undefined; + + return { + ...baseModel, + reasoning, + input: vision ? ["text", "image"] : ["text"], + cost, + contextWindow, + maxTokens, + ...(thinking ? { thinking } : {}), + ...(supportsTools === false ? { supportsTools } : {}), + }; + }, + fetch: config?.fetch, + }), + }), + }; +} + // --------------------------------------------------------------------------- // 15. Together // --------------------------------------------------------------------------- diff --git a/packages/catalog/test/baseten-provider.test.ts b/packages/catalog/test/baseten-provider.test.ts new file mode 100644 index 000000000..1cf2a603c --- /dev/null +++ b/packages/catalog/test/baseten-provider.test.ts @@ -0,0 +1,97 @@ +import { describe, expect, test } from "bun:test"; +import { basetenModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; +import type { FetchImpl } from "@oh-my-pi/pi-catalog/types"; + +describe("Baseten provider discovery", () => { + test("discovers Baseten models with custom metadata", async () => { + const calls: Array<{ url: string; authorization: string | null }> = []; + const fetchMock: FetchImpl = async (input: string | URL | Request, init?: RequestInit) => { + const headers = new Headers(init?.headers); + calls.push({ + url: String(input), + authorization: headers.get("authorization"), + }); + return new Response( + JSON.stringify({ + data: [ + { + id: "moonshotai/Kimi-K2.7-Code", + object: "model", + name: "Kimi K2.7 Code", + context_length: 262000, + max_completion_tokens: 262000, + supported_features: ["tools", "json_mode", "structured_outputs", "reasoning"], + input_modalities: ["text", "image"], + pricing: { + prompt: "0.00000095", + completion: "0.000004", + input_cache_read: "0.00000016", + }, + }, + { + id: "deepseek-ai/DeepSeek-V4-Pro", + object: "model", + name: "DeepSeek V4 Pro", + context_length: 262144, + max_completion_tokens: 262144, + supported_features: ["tools", "json_mode", "structured_outputs", "reasoning"], + input_modalities: ["text"], + pricing: { + prompt: "0.00000174", + completion: "0.00000348", + input_cache_read: "0.000000145", + }, + }, + ], + }), + { status: 200, headers: { "content-type": "application/json" } }, + ); + }; + + const options = basetenModelManagerOptions({ apiKey: "baseten-test-key", fetch: fetchMock }); + const models = await options.fetchDynamicModels?.(); + + expect(calls).toEqual([ + { + url: "https://inference.baseten.co/v1/models", + authorization: "Bearer baseten-test-key", + }, + ]); + + const kimi = models?.find(model => model.id === "moonshotai/Kimi-K2.7-Code"); + expect(kimi).toBeDefined(); + expect(kimi).toMatchObject({ + provider: "baseten", + api: "openai-completions", + name: "Kimi K2.7 Code", + reasoning: false, + input: ["text", "image"], + contextWindow: 262000, + maxTokens: 262000, + cost: { + input: 0.95, + output: 4, + cacheRead: 0.16, + cacheWrite: 0, + }, + }); + + const deepseek = models?.find(model => model.id === "deepseek-ai/DeepSeek-V4-Pro"); + expect(deepseek).toBeDefined(); + expect(deepseek).toMatchObject({ + provider: "baseten", + api: "openai-completions", + name: "DeepSeek V4 Pro", + reasoning: true, + input: ["text"], + contextWindow: 262144, + maxTokens: 262144, + cost: { + input: 1.74, + output: 3.48, + cacheRead: 0.145, + cacheWrite: 0, + }, + }); + }); +});