From 99224c87b2e85da8d89682a74ff5e907383970f1 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 4 Jun 2026 11:50:29 +0000 Subject: [PATCH] fix(providers): capped Kimi K2.x maxTokens on Fireworks at documented 32k ceiling MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fireworks /v1/models reports max_completion_tokens: 65536 generically for the Kimi K2 family, but Kimi-on-Fireworks is documented to produce runaway reasoning traces unless the output budget is bounded. The inflated value flowed straight through fireworksModelManagerOptions.mapModel (forwarded verbatim via toPositiveNumber), ended up in the bundled models.json for fireworks/kimi-k2.5, fireworks/kimi-k2.6, and firepass/kimi-k2.6-turbo, and was preserved across regenerations by prevModelsJson — so callers (and the openai-completions default-injection safety net) could ship a budget the router cannot honor. Add a Fireworks-family Kimi cap (FIREWORKS_KIMI_MAX_TOKENS = 32_768) plus isFireworksKimiK2ModelId / clampFireworksKimiMaxTokens helpers that recognize both the public catalog ids (kimi-k2.5, kimi-k2.6, kimi-k2.6-turbo, kimi-k2-thinking) and the canonical wire ids (accounts/fireworks/{models,routers}/kimi-k2…). The Fireworks resolver clamps every discovered Kimi K2.x model at runtime, and a new applyFireworksKimiMaxTokensCap pass in generate-models.ts applies the same ceiling to the fireworks/firepass slice of the assembled catalog so the firepass static fallback and any future regens stay in sync. Fixes #1849 --- packages/ai/CHANGELOG.md | 4 + packages/ai/scripts/generate-models.ts | 22 ++++++ packages/ai/src/models.json | 6 +- .../ai/src/provider-models/openai-compat.ts | 35 ++++++++- packages/ai/test/issue-1849-repro.test.ts | 74 +++++++++++++++++++ 5 files changed, 137 insertions(+), 4 deletions(-) create mode 100644 packages/ai/test/issue-1849-repro.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index ca22febb6..f463a87f4 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Kimi K2.x `maxTokens` on Fireworks and Fire Pass (`fireworks/kimi-k2.5`, `fireworks/kimi-k2.6`, `firepass/kimi-k2.6-turbo`) being inherited from Fireworks `/v1/models` discovery (`max_completion_tokens: 65536`) rather than the published Kimi-on-Fireworks output budget, which let callers (and the openai-completions default-injection safety net) ship a budget the router cannot honor and made runaway reasoning traces more likely. The Fireworks resolver now clamps every Kimi K2.x id (public catalog ids and the canonical `accounts/fireworks/{models,routers}/kimi-k2…` wire form) to 32,768 output tokens, and the generator applies the same cap as a post-processing safety net so the `firepass` static fallback and the bundled `fireworks` entries stay in sync across regens. ([#1849](https://github.com/can1357/oh-my-pi/issues/1849)) + ## [15.9.0] - 2026-06-04 ### Fixed diff --git a/packages/ai/scripts/generate-models.ts b/packages/ai/scripts/generate-models.ts index 843771b29..6431ade6d 100644 --- a/packages/ai/scripts/generate-models.ts +++ b/packages/ai/scripts/generate-models.ts @@ -28,6 +28,8 @@ import { } from "../src/provider-models/descriptors"; import { buildXaiOAuthStaticSeed, + clampFireworksKimiMaxTokens, + isFireworksKimiK2ModelId, MODELS_DEV_PROVIDER_DESCRIPTORS, mapModelsDevToModels, UNK_CONTEXT_WINDOW, @@ -222,6 +224,25 @@ function applyCodexPricingFallback(models: readonly Model[]): Model[] { }); } +/** + * Fireworks-backed Kimi K2.x deployments report `max_completion_tokens: 65536` + * over `/v1/models`, but Kimi's documented output budget on Fireworks is + * lower (#1849). Cap them here so the post-processing pass — which also folds + * in the `prevModelsJson` static fallback used by `firepass` — never lets a + * stale or inflated upstream value through. The resolver applies the same + * cap when discovery runs at runtime; this is the bundle-time safety net. + */ +function applyFireworksKimiMaxTokensCap(models: readonly Model[]): Model[] { + const FIREWORKS_KIMI_PROVIDERS = new Set(["fireworks", "firepass"]); + return models.map(model => { + if (!FIREWORKS_KIMI_PROVIDERS.has(model.provider)) return model; + if (!isFireworksKimiK2ModelId(model.id)) return model; + const capped = clampFireworksKimiMaxTokens(model.id, model.maxTokens); + if (capped === model.maxTokens) return model; + return { ...model, maxTokens: capped }; + }); +} + const ANTIGRAVITY_ENDPOINT = "https://daily-cloudcode-pa.sandbox.googleapis.com"; async function getOAuthAccessFromStorage(provider: OAuthProvider): Promise { @@ -394,6 +415,7 @@ async function generateModels() { allModels = applyGlobalModelsDevFallback(allModels, modelsDevModels); allModels = applyPremiumMultiplierOverrides(allModels); allModels = applyCodexPricingFallback(allModels); + allModels = applyFireworksKimiMaxTokensCap(allModels); applyGeneratedModelPolicies(allModels); linkOpenAIPromotionTargets(allModels); diff --git a/packages/ai/src/models.json b/packages/ai/src/models.json index 2d2809f7b..b4dbd4bf4 100644 --- a/packages/ai/src/models.json +++ b/packages/ai/src/models.json @@ -5325,7 +5325,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 32768, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -5469,7 +5469,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 32768, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -5494,7 +5494,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 32768, "thinking": { "mode": "effort", "minLevel": "minimal", diff --git a/packages/ai/src/provider-models/openai-compat.ts b/packages/ai/src/provider-models/openai-compat.ts index c2b92702e..921a9735f 100644 --- a/packages/ai/src/provider-models/openai-compat.ts +++ b/packages/ai/src/provider-models/openai-compat.ts @@ -925,6 +925,36 @@ const ZHIPU_VISION_PATTERN = /^glm-[45](?:\.\d+)?v(?:-|$)/; // 7.5 Fireworks // --------------------------------------------------------------------------- +/** + * Fireworks-published cap for the Kimi K2 family. Fireworks' `/v1/models` + * envelope generically reports `max_completion_tokens: 65536` for every Kimi + * deployment, but Kimi K2 (instruct / thinking / turbo) on Fireworks is + * documented to ship long reasoning traces that should be bounded — capping + * at 32,768 prevents handing callers a budget the router cannot honor. + * See https://github.com/can1357/oh-my-pi/issues/1849. + */ +export const FIREWORKS_KIMI_MAX_TOKENS = 32_768; + +/** + * Returns true for any Kimi K2.x public model id served by Fireworks-backed + * providers (`fireworks` direct, `firepass` router). Matches both the public + * catalog id (`kimi-k2.5`, `kimi-k2.6`, `kimi-k2.6-turbo`) and the canonical + * Fireworks wire id (`accounts/fireworks/{models,routers}/kimi-k2…`). + */ +export function isFireworksKimiK2ModelId(modelId: string): boolean { + const trimmed = modelId.toLowerCase(); + if (trimmed.startsWith("kimi-k2")) return true; + return /\/kimi-k2(?:p\d+)?(?:[._-]|$)/.test(trimmed); +} + +/** + * Clamp the Kimi K2 family's `maxTokens` to {@link FIREWORKS_KIMI_MAX_TOKENS} + * on Fireworks-backed providers, leaving every other model untouched. + */ +export function clampFireworksKimiMaxTokens(modelId: string, candidate: number): number { + return isFireworksKimiK2ModelId(modelId) ? Math.min(candidate, FIREWORKS_KIMI_MAX_TOKENS) : candidate; +} + export interface FireworksModelManagerConfig { apiKey?: string; baseUrl?: string; @@ -1004,7 +1034,10 @@ export function fireworksModelManagerOptions( name: toFireworksModelName(entry, model.name), input: toBoolean(entry.supports_image_input) === true ? ["text", "image"] : ["text"], contextWindow: toPositiveNumber(entry.context_length, model.contextWindow), - maxTokens: toPositiveNumber(entry.max_completion_tokens, model.maxTokens), + maxTokens: clampFireworksKimiMaxTokens( + publicModelId, + toPositiveNumber(entry.max_completion_tokens, model.maxTokens), + ), }; }, }); diff --git a/packages/ai/test/issue-1849-repro.test.ts b/packages/ai/test/issue-1849-repro.test.ts new file mode 100644 index 000000000..105e4539e --- /dev/null +++ b/packages/ai/test/issue-1849-repro.test.ts @@ -0,0 +1,74 @@ +/** + * Regression for #1849 — Kimi K2.x maxTokens on Fireworks/Fire Pass was + * inherited from `/v1/models` discovery (`max_completion_tokens: 65536`), + * but Kimi K2 on Fireworks is documented to produce runaway reasoning traces + * unless the output budget is bounded. + * + * Two contracts this file defends: + * 1. `clampFireworksKimiMaxTokens` caps any Kimi K2.x id (public or wire) + * to the published 32,768 ceiling and leaves every other model alone. + * 2. The bundled catalog ships the capped value — both for the static + * `firepass/kimi-k2.6-turbo` entry (no dynamic discovery) and for the + * `fireworks/kimi-k2.5` / `fireworks/kimi-k2.6` entries that the + * generator regenerates. + */ +import { describe, expect, it } from "bun:test"; +import { getBundledModel } from "../src/models"; +import { + clampFireworksKimiMaxTokens, + FIREWORKS_KIMI_MAX_TOKENS, + isFireworksKimiK2ModelId, +} from "../src/provider-models/openai-compat"; + +describe("Fireworks Kimi K2 maxTokens cap (#1849)", () => { + it("recognizes Kimi K2.x public and wire ids", () => { + const positives = [ + "kimi-k2.5", + "kimi-k2.6", + "kimi-k2.6-turbo", + "kimi-k2-thinking", + "accounts/fireworks/models/kimi-k2-instruct", + "accounts/fireworks/models/kimi-k2-thinking", + "accounts/fireworks/routers/kimi-k2p6-turbo", + ]; + for (const id of positives) { + expect(isFireworksKimiK2ModelId(id)).toBe(true); + } + const negatives = [ + "kimi-latest", + "kimi-k1.5", + "deepseek-v4-pro", + "glm-5.1", + "accounts/fireworks/models/minimax-m2.7", + ]; + for (const id of negatives) { + expect(isFireworksKimiK2ModelId(id)).toBe(false); + } + }); + + it("clamps Kimi K2.x candidates to the published ceiling and leaves others untouched", () => { + // Inflated upstream value collapses to the cap. + expect(clampFireworksKimiMaxTokens("kimi-k2.6", 65_536)).toBe(FIREWORKS_KIMI_MAX_TOKENS); + expect(clampFireworksKimiMaxTokens("accounts/fireworks/routers/kimi-k2p6-turbo", 131_072)).toBe( + FIREWORKS_KIMI_MAX_TOKENS, + ); + // Already-low candidate stays low — the helper never raises a budget. + expect(clampFireworksKimiMaxTokens("kimi-k2.5", 4_096)).toBe(4_096); + // Non-Kimi ids pass through verbatim. + expect(clampFireworksKimiMaxTokens("deepseek-v4-pro", 65_536)).toBe(65_536); + expect(clampFireworksKimiMaxTokens("glm-5.1", 65_536)).toBe(65_536); + }); + + it("ships the capped maxTokens in the bundled Fireworks/Fire Pass catalog", () => { + const entries: Array<["fireworks" | "firepass", string]> = [ + ["firepass", "kimi-k2.6-turbo"], + ["fireworks", "kimi-k2.5"], + ["fireworks", "kimi-k2.6"], + ]; + for (const [provider, id] of entries) { + const model = getBundledModel(provider, id); + expect(model).toBeDefined(); + expect(model.maxTokens).toBe(FIREWORKS_KIMI_MAX_TOKENS); + } + }); +});