diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 21a7f6f64..f18091bfc 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Kimi output caps for Umans AI Coding Plan and Venice so discovery metadata cannot use context-sized token ceilings as request caps. + ## [16.0.1] - 2026-06-15 ### Added diff --git a/packages/catalog/scripts/generate-models.ts b/packages/catalog/scripts/generate-models.ts index 45bd371a4..cf3cfb096 100644 --- a/packages/catalog/scripts/generate-models.ts +++ b/packages/catalog/scripts/generate-models.ts @@ -30,7 +30,9 @@ import { ANTHROPIC_CURATED_FALLBACK_MODELS, buildXaiOAuthStaticSeed, clampFireworksKimiMaxTokens, + clampKimiK27CodeMaxTokens, isFireworksKimiK2ModelId, + isKimiK27CodeModelId, MODELS_DEV_PROVIDER_DESCRIPTORS, mapModelsDevToModels, stripFireworksDeepSeekThinkingToggle, @@ -245,21 +247,22 @@ function applyCodexPricingFallback(models: readonly ModelSpec[]): ModelSpec[] { } /** - * Fireworks-backed Kimi K2.x deployments report `max_completion_tokens: 65536` - * over `/v1/models`, but Kimi's documented output budget on Fireworks is - * lower (#1849). Cap them here so the post-processing pass — which also folds - * in the `prevModelsJson` static fallback used by `firepass` — never lets a - * stale or inflated upstream value through. The resolver applies the same - * cap when discovery runs at runtime; this is the bundle-time safety net. + * Provider discovery sometimes reports context-sized Kimi output ceilings. Keep + * the bundled catalog at the documented/provider-safe caps so request builders + * that always send `max_tokens` do not over-allocate. */ -function applyFireworksKimiMaxTokensCap(models: readonly ModelSpec[]): ModelSpec[] { +function applyKimiMaxTokensCap(models: readonly ModelSpec[]): ModelSpec[] { const FIREWORKS_KIMI_PROVIDERS = new Set(["fireworks", "firepass"]); return models.map(model => { - if (!FIREWORKS_KIMI_PROVIDERS.has(model.provider)) return model; - if (!isFireworksKimiK2ModelId(model.id)) return model; - const capped = clampFireworksKimiMaxTokens(model.id, model.maxTokens); - if (capped === model.maxTokens) return model; - return { ...model, maxTokens: capped }; + if (FIREWORKS_KIMI_PROVIDERS.has(model.provider) && isFireworksKimiK2ModelId(model.id)) { + const capped = clampFireworksKimiMaxTokens(model.id, model.maxTokens); + return capped === model.maxTokens ? model : { ...model, maxTokens: capped }; + } + if (model.provider === "venice" && isKimiK27CodeModelId(model.id)) { + const capped = clampKimiK27CodeMaxTokens(model.id, model.maxTokens); + return capped === model.maxTokens ? model : { ...model, maxTokens: capped }; + } + return model; }); } @@ -418,20 +421,31 @@ async function fetchCodexDiscoveryModels(): Promise isCatalogDescriptor(descriptor) && !DISCOVERY_ONLY_PROVIDERS.has(descriptor.providerId), - ).map(descriptor => fetchProviderModelsFromCatalog(descriptor as CatalogProviderDescriptor)), - ) - ).flat(); + const catalogProviderDescriptors = PROVIDER_DESCRIPTORS.filter( + (descriptor): descriptor is CatalogProviderDescriptor => + isCatalogDescriptor(descriptor) && !DISCOVERY_ONLY_PROVIDERS.has(descriptor.providerId), + ); + const catalogProviderModelBatches = await Promise.all( + catalogProviderDescriptors.map(async descriptor => ({ + descriptor, + models: await fetchProviderModelsFromCatalog(descriptor), + })), + ); + const authoritativeCatalogProviders = new Set( + catalogProviderModelBatches + .filter(batch => batch.descriptor.dynamicModelsAuthoritative === true && batch.models.length > 0) + .map(batch => batch.descriptor.providerId), + ); + const catalogProviderModels = catalogProviderModelBatches.flatMap(batch => batch.models); + const bundledModelsDevModels = modelsDevModels.filter(model => !authoritativeCatalogProviders.has(model.provider)); // getGitLabDuoModels returns built models; project back to spec stage for the bundle. const gitLabDuoModels = getGitLabDuoModels().map(model => toModelSpec(model)); - // Combine models (models.dev has priority) + // Combine models. models.dev has priority unless a provider's successful endpoint + // discovery is authoritative; those endpoint snapshots replace models.dev rows. let allModels = applyGlobalModelsDevFallback( - [...modelsDevModels, ...catalogProviderModels, ...gitLabDuoModels], + [...bundledModelsDevModels, ...catalogProviderModels, ...gitLabDuoModels], modelsDevModels, ); @@ -471,19 +485,16 @@ async function generateModels() { } } - const modelsDevAuthoritativeProviders = new Set(); + const modelsDevSnapshotExcludedProviders = new Set(); for (const model of modelsDevModels) { if (model.provider === "google-vertex") { - modelsDevAuthoritativeProviders.add(model.provider); + modelsDevSnapshotExcludedProviders.add(model.provider); } } - if (catalogProviderModels.some(model => model.provider === "aimlapi")) { - modelsDevAuthoritativeProviders.add("aimlapi"); - } // Merge previous models.json entries as fallback for provider/model pairs not - // fetched dynamically. Providers that models.dev covers authoritatively keep - // the upstream list exactly, so retired entries from the previous snapshot do - // not reappear during regeneration. + // fetched dynamically. Providers covered by authoritative endpoint discovery + // or authoritative models.dev sources keep that upstream list exactly, so + // retired entries from the previous snapshot do not reappear during regeneration. // Discovery-only providers (local inference servers) — never bundle static models. const fetchedKeys = new Set(allModels.map(model => `${model.provider}/${model.id}`)); @@ -495,7 +506,8 @@ async function generateModels() { if ( !fetchedKeys.has(`${model.provider}/${model.id}`) && !DISCOVERY_ONLY_PROVIDERS.has(model.provider) && - !modelsDevAuthoritativeProviders.has(model.provider) + !authoritativeCatalogProviders.has(model.provider) && + !modelsDevSnapshotExcludedProviders.has(model.provider) ) { allModels.push(model); } @@ -505,7 +517,7 @@ async function generateModels() { allModels = applyGlobalModelsDevFallback(allModels, modelsDevModels); allModels = applyPremiumMultiplierOverrides(allModels); allModels = applyCodexPricingFallback(allModels); - allModels = applyFireworksKimiMaxTokensCap(allModels); + allModels = applyKimiMaxTokensCap(allModels); allModels = applyFireworksDeepSeekReasoningShape(allModels); allModels = dropFireworksWireIds(allModels); allModels = dropUnusableZaiContextTierIds(allModels); @@ -534,8 +546,8 @@ async function generateModels() { if (!providers[model.provider]) { providers[model.provider] = {}; } - // Use model ID as key to automatically deduplicate - // Only add if not already present (models.dev takes priority over endpoint discovery) + // Use model ID as key to deduplicate the ordered sources assembled above. + // Earlier sources win. if (!providers[model.provider][model.id]) { providers[model.provider][model.id] = model; } diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index c606f956c..52c3ac3f4 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -60012,8 +60012,8 @@ "text" ], "cost": { - "input": 0.09, - "output": 0.18, + "input": 0.098, + "output": 0.196, "cacheRead": 0.02, "cacheWrite": 0 }, @@ -65288,7 +65288,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, + "maxTokens": 81920, "thinking": { "mode": "effort", "efforts": [ @@ -65311,12 +65311,12 @@ "image" ], "cost": { - "input": 0.39, - "output": 2.34, + "input": 0.385, + "output": 2.4499999999999997, "cacheRead": 0.195, "cacheWrite": 0 }, - "contextWindow": 262144, + "contextWindow": 256000, "maxTokens": 65536, "thinking": { "mode": "effort", @@ -68906,7 +68906,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, + "maxTokens": 32768, "thinking": { "mode": "budget", "efforts": [ @@ -68936,7 +68936,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, + "maxTokens": 32768, "thinking": { "mode": "budget", "efforts": [ @@ -68950,7 +68950,7 @@ }, "umans-glm-5.1": { "id": "umans-glm-5.1", - "name": "GLM 5.1", + "name": "Umans GLM 5.1", "api": "anthropic-messages", "provider": "umans", "baseUrl": "https://api.code.umans.ai", @@ -68965,7 +68965,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 204800, + "contextWindow": 202752, "maxTokens": 131072, "thinking": { "mode": "budget", @@ -68980,7 +68980,7 @@ }, "umans-kimi-k2.6": { "id": "umans-kimi-k2.6", - "name": "Kimi K2.6", + "name": "Umans Kimi K2.6", "api": "anthropic-messages", "provider": "umans", "baseUrl": "https://api.code.umans.ai", @@ -68996,7 +68996,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, + "maxTokens": 32768, "thinking": { "mode": "budget", "efforts": [ @@ -69010,7 +69010,7 @@ }, "umans-kimi-k2.7": { "id": "umans-kimi-k2.7", - "name": "Kimi K2.7 Code", + "name": "Umans Kimi K2.7 Code", "api": "anthropic-messages", "provider": "umans", "baseUrl": "https://api.code.umans.ai", @@ -69026,7 +69026,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, + "maxTokens": 32768, "thinking": { "mode": "budget", "efforts": [ @@ -69041,7 +69041,7 @@ }, "umans-qwen3.6-35b-a3b": { "id": "umans-qwen3.6-35b-a3b", - "name": "Qwen3.6 35B A3B", + "name": "Umans Qwen3.6 35B A3B", "api": "anthropic-messages", "provider": "umans", "baseUrl": "https://api.code.umans.ai", @@ -69057,7 +69057,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, + "maxTokens": 32768, "thinking": { "mode": "budget", "efforts": [ @@ -70539,6 +70539,28 @@ ] } }, + "kimi-k2-7-code": { + "id": "kimi-k2-7-code", + "name": "kimi-k2-7-code", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 32768, + "compat": { + "supportsUsageInStreaming": false + } + }, "kimi-k2-thinking": { "id": "kimi-k2-thinking", "name": "Kimi K2 Thinking", diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 2766d8ae2..a1042a4e1 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -664,7 +664,10 @@ function mapUmansModelInfo( ...(supportsTools === false ? { supportsTools: false } : {}), cost: reference?.cost ?? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: toPositiveNumber(capabilities.context_window, reference?.contextWindow ?? null), - maxTokens: toPositiveNumber(capabilities.max_completion_tokens, reference?.maxTokens ?? null), + maxTokens: toPositiveNumber( + capabilities.recommended_max_tokens, + toPositiveNumber(capabilities.max_completion_tokens, reference?.maxTokens ?? null), + ), }; } @@ -1232,6 +1235,23 @@ export function clampFireworksKimiMaxTokens(modelId: string, candidate: number | return isFireworksKimiK2ModelId(modelId) ? Math.min(candidate, FIREWORKS_KIMI_MAX_TOKENS) : candidate; } +/** + * Kimi K2.7 Code's documented recommended output budget. Some provider + * discovery rows report the context-sized `max_completion_tokens` instead. + */ +export const KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS = 32_768; + +export function isKimiK27CodeModelId(modelId: string): boolean { + return /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code$/i.test(modelId); +} + +export function clampKimiK27CodeMaxTokens(modelId: string, candidate: number): number; +export function clampKimiK27CodeMaxTokens(modelId: string, candidate: number | null): number | null; +export function clampKimiK27CodeMaxTokens(modelId: string, candidate: number | null): number | null { + if (candidate === null) return null; + return isKimiK27CodeModelId(modelId) ? Math.min(candidate, KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS) : candidate; +} + /** * Fireworks DeepSeek V4 accepts effort via `reasoning_effort` but rejects the * DeepSeek-native binary `thinking` toggle when both are present. @@ -2276,6 +2296,7 @@ export function veniceModelManagerOptions( const model = mapWithBundledReference(entry, defaults, reference); return { ...model, + maxTokens: clampKimiK27CodeMaxTokens(defaults.id, model.maxTokens), compat: { ...model.compat, supportsUsageInStreaming: false }, }; }, @@ -3544,7 +3565,12 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_SPECIALIZED: readonly ModelsDevProviderDes // --- Synthetic --- openAiCompletionsDescriptor("synthetic", "synthetic", "https://api.synthetic.new/openai/v1"), // --- Venice AI --- - openAiCompletionsDescriptor("venice", "venice", "https://api.venice.ai/api/v1"), + openAiCompletionsDescriptor("venice", "venice", "https://api.venice.ai/api/v1", { + transformModel: model => { + const maxTokens = clampKimiK27CodeMaxTokens(model.id, model.maxTokens); + return maxTokens === model.maxTokens ? model : { ...model, maxTokens }; + }, + }), // --- Ollama Cloud --- simpleModelsDevDescriptor("ollama-cloud", "ollama-cloud", "ollama-chat", "https://ollama.com"), // --- Xiaomi Token Plan --- diff --git a/packages/catalog/test/umans-provider.test.ts b/packages/catalog/test/umans-provider.test.ts index f36ce803c..8f0557147 100644 --- a/packages/catalog/test/umans-provider.test.ts +++ b/packages/catalog/test/umans-provider.test.ts @@ -33,6 +33,7 @@ describe("umans provider catalog", () => { capabilities: { context_window: 262_144, max_completion_tokens: 262_144, + recommended_max_tokens: 32_768, supports_vision: true, supports_tools: true, reasoning: { supported: true, can_disable: true, default_level: "medium" }, @@ -43,6 +44,7 @@ describe("umans provider catalog", () => { capabilities: { context_window: 262_144, max_completion_tokens: 262_144, + recommended_max_tokens: 32_768, supports_vision: true, supports_tools: true, reasoning: { supported: true, can_disable: false, default_level: "medium" }, @@ -71,13 +73,14 @@ describe("umans provider catalog", () => { reasoning: true, input: ["text", "image"], contextWindow: 262_144, - maxTokens: 262_144, + maxTokens: 32_768, thinking: { defaultLevel: "medium" }, }); const mandatoryReasoningModel = models?.find(item => item.id === "umans-kimi-k2.7"); expect(mandatoryReasoningModel).toMatchObject({ id: "umans-kimi-k2.7", reasoning: true, + maxTokens: 32_768, thinking: { defaultLevel: "medium", requiresEffort: true }, }); }); @@ -137,7 +140,7 @@ describe("umans provider catalog", () => { reasoning: true, input: ["text", "image"], contextWindow: 262_144, - maxTokens: 262_144, + maxTokens: 32_768, }); }); @@ -146,6 +149,7 @@ describe("umans provider catalog", () => { const model = providers.umans?.["umans-kimi-k2.7"]; expect(model).toBeDefined(); + expect(model.maxTokens).toBe(32_768); expect(model.thinking).toMatchObject({ requiresEffort: true, }); diff --git a/packages/catalog/test/venice-provider.test.ts b/packages/catalog/test/venice-provider.test.ts new file mode 100644 index 000000000..f828ff3fa --- /dev/null +++ b/packages/catalog/test/venice-provider.test.ts @@ -0,0 +1,44 @@ +import { describe, expect, it } from "bun:test"; +import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; +import { + KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS, + veniceModelManagerOptions, +} from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; +import type { FetchImpl } from "@oh-my-pi/pi-catalog/types"; + +describe("Venice provider catalog", () => { + it("bundles Kimi K2.7 Code with its recommended output cap", () => { + const model = getBundledModel("venice", "kimi-k2-7-code"); + + expect(model).toBeDefined(); + expect(model.maxTokens).toBe(KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS); + }); + + it("caps Kimi K2.7 Code during runtime discovery", async () => { + const requestedUrls: string[] = []; + const fetchImpl: FetchImpl = async input => { + requestedUrls.push(input instanceof Request ? input.url : String(input)); + return new Response( + JSON.stringify({ + data: [ + { + id: "kimi-k2-7-code", + name: "kimi-k2-7-code", + context_length: 256_000, + max_completion_tokens: 262_144, + }, + ], + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + }; + + const options = veniceModelManagerOptions({ apiKey: "venice-test-key", fetch: fetchImpl }); + const models = await options.fetchDynamicModels?.(); + const model = models?.find(candidate => candidate.id === "kimi-k2-7-code"); + + expect(requestedUrls).toEqual(["https://api.venice.ai/api/v1/models"]); + expect(model).toBeDefined(); + expect(model?.maxTokens).toBe(KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS); + }); +});