diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index ca22febb6..f463a87f4 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Kimi K2.x `maxTokens` on Fireworks and Fire Pass (`fireworks/kimi-k2.5`, `fireworks/kimi-k2.6`, `firepass/kimi-k2.6-turbo`) being inherited from Fireworks `/v1/models` discovery (`max_completion_tokens: 65536`) rather than the published Kimi-on-Fireworks output budget, which let callers (and the openai-completions default-injection safety net) ship a budget the router cannot honor and made runaway reasoning traces more likely. The Fireworks resolver now clamps every Kimi K2.x id (public catalog ids and the canonical `accounts/fireworks/{models,routers}/kimi-k2…` wire form) to 32,768 output tokens, and the generator applies the same cap as a post-processing safety net so the `firepass` static fallback and the bundled `fireworks` entries stay in sync across regens. ([#1849](https://github.com/can1357/oh-my-pi/issues/1849)) + ## [15.9.0] - 2026-06-04 ### Fixed diff --git a/packages/ai/scripts/generate-models.ts b/packages/ai/scripts/generate-models.ts index 843771b29..6431ade6d 100644 --- a/packages/ai/scripts/generate-models.ts +++ b/packages/ai/scripts/generate-models.ts @@ -28,6 +28,8 @@ import { } from "../src/provider-models/descriptors"; import { buildXaiOAuthStaticSeed, + clampFireworksKimiMaxTokens, + isFireworksKimiK2ModelId, MODELS_DEV_PROVIDER_DESCRIPTORS, mapModelsDevToModels, UNK_CONTEXT_WINDOW, @@ -222,6 +224,25 @@ function applyCodexPricingFallback(models: readonly Model[]): Model[] { }); } +/** + * Fireworks-backed Kimi K2.x deployments report `max_completion_tokens: 65536` + * over `/v1/models`, but Kimi's documented output budget on Fireworks is + * lower (#1849). Cap them here so the post-processing pass — which also folds + * in the `prevModelsJson` static fallback used by `firepass` — never lets a + * stale or inflated upstream value through. The resolver applies the same + * cap when discovery runs at runtime; this is the bundle-time safety net. + */ +function applyFireworksKimiMaxTokensCap(models: readonly Model[]): Model[] { + const FIREWORKS_KIMI_PROVIDERS = new Set(["fireworks", "firepass"]); + return models.map(model => { + if (!FIREWORKS_KIMI_PROVIDERS.has(model.provider)) return model; + if (!isFireworksKimiK2ModelId(model.id)) return model; + const capped = clampFireworksKimiMaxTokens(model.id, model.maxTokens); + if (capped === model.maxTokens) return model; + return { ...model, maxTokens: capped }; + }); +} + const ANTIGRAVITY_ENDPOINT = "https://daily-cloudcode-pa.sandbox.googleapis.com"; async function getOAuthAccessFromStorage(provider: OAuthProvider): Promise { @@ -394,6 +415,7 @@ async function generateModels() { allModels = applyGlobalModelsDevFallback(allModels, modelsDevModels); allModels = applyPremiumMultiplierOverrides(allModels); allModels = applyCodexPricingFallback(allModels); + allModels = applyFireworksKimiMaxTokensCap(allModels); applyGeneratedModelPolicies(allModels); linkOpenAIPromotionTargets(allModels); diff --git a/packages/ai/src/models.json b/packages/ai/src/models.json index 2d2809f7b..b4dbd4bf4 100644 --- a/packages/ai/src/models.json +++ b/packages/ai/src/models.json @@ -5325,7 +5325,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 32768, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -5469,7 +5469,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 32768, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -5494,7 +5494,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 32768, "thinking": { "mode": "effort", "minLevel": "minimal", diff --git a/packages/ai/src/provider-models/openai-compat.ts b/packages/ai/src/provider-models/openai-compat.ts index c2b92702e..921a9735f 100644 --- a/packages/ai/src/provider-models/openai-compat.ts +++ b/packages/ai/src/provider-models/openai-compat.ts @@ -925,6 +925,36 @@ const ZHIPU_VISION_PATTERN = /^glm-[45](?:\.\d+)?v(?:-|$)/; // 7.5 Fireworks // --------------------------------------------------------------------------- +/** + * Fireworks-published cap for the Kimi K2 family. Fireworks' `/v1/models` + * envelope generically reports `max_completion_tokens: 65536` for every Kimi + * deployment, but Kimi K2 (instruct / thinking / turbo) on Fireworks is + * documented to ship long reasoning traces that should be bounded — capping + * at 32,768 prevents handing callers a budget the router cannot honor. + * See https://github.com/can1357/oh-my-pi/issues/1849. + */ +export const FIREWORKS_KIMI_MAX_TOKENS = 32_768; + +/** + * Returns true for any Kimi K2.x public model id served by Fireworks-backed + * providers (`fireworks` direct, `firepass` router). Matches both the public + * catalog id (`kimi-k2.5`, `kimi-k2.6`, `kimi-k2.6-turbo`) and the canonical + * Fireworks wire id (`accounts/fireworks/{models,routers}/kimi-k2…`). + */ +export function isFireworksKimiK2ModelId(modelId: string): boolean { + const trimmed = modelId.toLowerCase(); + if (trimmed.startsWith("kimi-k2")) return true; + return /\/kimi-k2(?:p\d+)?(?:[._-]|$)/.test(trimmed); +} + +/** + * Clamp the Kimi K2 family's `maxTokens` to {@link FIREWORKS_KIMI_MAX_TOKENS} + * on Fireworks-backed providers, leaving every other model untouched. + */ +export function clampFireworksKimiMaxTokens(modelId: string, candidate: number): number { + return isFireworksKimiK2ModelId(modelId) ? Math.min(candidate, FIREWORKS_KIMI_MAX_TOKENS) : candidate; +} + export interface FireworksModelManagerConfig { apiKey?: string; baseUrl?: string; @@ -1004,7 +1034,10 @@ export function fireworksModelManagerOptions( name: toFireworksModelName(entry, model.name), input: toBoolean(entry.supports_image_input) === true ? ["text", "image"] : ["text"], contextWindow: toPositiveNumber(entry.context_length, model.contextWindow), - maxTokens: toPositiveNumber(entry.max_completion_tokens, model.maxTokens), + maxTokens: clampFireworksKimiMaxTokens( + publicModelId, + toPositiveNumber(entry.max_completion_tokens, model.maxTokens), + ), }; }, }); diff --git a/packages/ai/test/issue-1849-repro.test.ts b/packages/ai/test/issue-1849-repro.test.ts new file mode 100644 index 000000000..105e4539e --- /dev/null +++ b/packages/ai/test/issue-1849-repro.test.ts @@ -0,0 +1,74 @@ +/** + * Regression for #1849 — Kimi K2.x maxTokens on Fireworks/Fire Pass was + * inherited from `/v1/models` discovery (`max_completion_tokens: 65536`), + * but Kimi K2 on Fireworks is documented to produce runaway reasoning traces + * unless the output budget is bounded. + * + * Two contracts this file defends: + * 1. `clampFireworksKimiMaxTokens` caps any Kimi K2.x id (public or wire) + * to the published 32,768 ceiling and leaves every other model alone. + * 2. The bundled catalog ships the capped value — both for the static + * `firepass/kimi-k2.6-turbo` entry (no dynamic discovery) and for the + * `fireworks/kimi-k2.5` / `fireworks/kimi-k2.6` entries that the + * generator regenerates. + */ +import { describe, expect, it } from "bun:test"; +import { getBundledModel } from "../src/models"; +import { + clampFireworksKimiMaxTokens, + FIREWORKS_KIMI_MAX_TOKENS, + isFireworksKimiK2ModelId, +} from "../src/provider-models/openai-compat"; + +describe("Fireworks Kimi K2 maxTokens cap (#1849)", () => { + it("recognizes Kimi K2.x public and wire ids", () => { + const positives = [ + "kimi-k2.5", + "kimi-k2.6", + "kimi-k2.6-turbo", + "kimi-k2-thinking", + "accounts/fireworks/models/kimi-k2-instruct", + "accounts/fireworks/models/kimi-k2-thinking", + "accounts/fireworks/routers/kimi-k2p6-turbo", + ]; + for (const id of positives) { + expect(isFireworksKimiK2ModelId(id)).toBe(true); + } + const negatives = [ + "kimi-latest", + "kimi-k1.5", + "deepseek-v4-pro", + "glm-5.1", + "accounts/fireworks/models/minimax-m2.7", + ]; + for (const id of negatives) { + expect(isFireworksKimiK2ModelId(id)).toBe(false); + } + }); + + it("clamps Kimi K2.x candidates to the published ceiling and leaves others untouched", () => { + // Inflated upstream value collapses to the cap. + expect(clampFireworksKimiMaxTokens("kimi-k2.6", 65_536)).toBe(FIREWORKS_KIMI_MAX_TOKENS); + expect(clampFireworksKimiMaxTokens("accounts/fireworks/routers/kimi-k2p6-turbo", 131_072)).toBe( + FIREWORKS_KIMI_MAX_TOKENS, + ); + // Already-low candidate stays low — the helper never raises a budget. + expect(clampFireworksKimiMaxTokens("kimi-k2.5", 4_096)).toBe(4_096); + // Non-Kimi ids pass through verbatim. + expect(clampFireworksKimiMaxTokens("deepseek-v4-pro", 65_536)).toBe(65_536); + expect(clampFireworksKimiMaxTokens("glm-5.1", 65_536)).toBe(65_536); + }); + + it("ships the capped maxTokens in the bundled Fireworks/Fire Pass catalog", () => { + const entries: Array<["fireworks" | "firepass", string]> = [ + ["firepass", "kimi-k2.6-turbo"], + ["fireworks", "kimi-k2.5"], + ["fireworks", "kimi-k2.6"], + ]; + for (const [provider, id] of entries) { + const model = getBundledModel(provider, id); + expect(model).toBeDefined(); + expect(model.maxTokens).toBe(FIREWORKS_KIMI_MAX_TOKENS); + } + }); +});