From b09b82946be67f7cde31e155388c7c9718725510 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 26 Jul 2026 14:40:04 +0000 Subject: [PATCH 1/2] fix(catalog): derive kimi-code output caps per family kimi-code discovery mapModel and the bundled catalog hardcoded maxTokens=32000 for every model, truncating k3/k3-256k output at ~4x below their real 131072 ceiling and kimi-for-coding[-highspeed] below 32768. Add kimiCodeMaxTokens() (k3* -> 131072, kimi-for-coding* -> 32768, else fallback), wire it into the discovery mapper and the generator cap policy, and correct the bundled kimi-code entries. Fixes #6711 --- packages/catalog/CHANGELOG.md | 1 + packages/catalog/scripts/generate-models.ts | 7 +++++ packages/catalog/src/models.json | 6 ++-- .../src/provider-models/openai-compat.ts | 30 ++++++++++++++++++- .../catalog/test/kimi-code-provider.test.ts | 23 ++++++++++++++ 5 files changed, 63 insertions(+), 4 deletions(-) diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index d2a155566..4ed9c9d25 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -12,6 +12,7 @@ - Fixed forced `tool_choice` 400s (`tool_choice 'specified' is incompatible with thinking enabled`) on Kimi Code's Anthropic-compatible endpoint for the `kimi-for-coding`, `kimi-for-coding-highspeed`, and `k3` aliases: the Anthropic-surface compat matcher only recognised Moonshot's native `kimi-k2.7-code*` ids, so thinking-locked kimi-code models kept `supportsForcedToolChoice: true` and the forced selector was sent to a host that always thinks. These models now resolve `requiresThinkingEnabled`, keeping thinking on and downgrading forced choices to `auto`. - Retried empty successful provider discovery responses after the short non-authoritative interval instead of caching them for the full catalog TTL ([#6620](https://github.com/can1357/oh-my-pi/issues/6620)). - Fixed GitHub Copilot Claude models with no bundled catalog reference (e.g. a freshly served `claude-opus-5`) discovering with `reasoning: false`/`thinking: null` and no effort dial, and disappearing along with their synthesized `-1m` sibling on offline reads: reference-less Copilot models on the anthropic-messages proxy now derive the adaptive reasoning ladder from the model id, and the cache restores their compile-time `COPILOT_API_HEADERS` by value instead of dropping them as unrestorable ([#6664](https://github.com/can1357/oh-my-pi/issues/6664)). +- Fixed Kimi Code (`kimi-code`) reporting `maxTokens: 32000` for every model — its `/coding/v1/models` discovery mapper and the bundled catalog applied a blanket constant, truncating `k3`/`k3-256k` output at ~4x below their real 131072 ceiling and `kimi-for-coding`/`kimi-for-coding-highspeed` below their 32768 ceiling. Output caps are now derived per family ([#6711](https://github.com/can1357/oh-my-pi/issues/6711)). ## [17.1.3] - 2026-07-24 diff --git a/packages/catalog/scripts/generate-models.ts b/packages/catalog/scripts/generate-models.ts index 6971eab37..48cecef63 100644 --- a/packages/catalog/scripts/generate-models.ts +++ b/packages/catalog/scripts/generate-models.ts @@ -37,6 +37,7 @@ import { clampKimiK27CodeMaxTokens, isFireworksKimiK2ModelId, isKimiK27CodeModelId, + kimiCodeMaxTokens, META_MUSE_STATIC_MODELS, MODELS_DEV_PROVIDER_DESCRIPTORS, mapModelsDevToModels, @@ -304,6 +305,12 @@ function applyKimiMaxTokensCap(models: readonly ModelSpec[]): ModelSpec[] { const capped = clampKimiK27CodeMaxTokens(model.id, model.maxTokens); return capped === model.maxTokens ? model : { ...model, maxTokens: capped }; } + if (model.provider === "kimi-code") { + // Discovery snapshots carried maxTokens=32000 uniformly (#6711); pin the + // documented per-family output ceilings and leave legacy K2 rows as-is. + const capped = kimiCodeMaxTokens(model.id, model.maxTokens); + return capped === model.maxTokens ? model : { ...model, maxTokens: capped }; + } return model; }); } diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index a184329fc..e6a62c0df 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -34829,7 +34829,7 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 32000, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -34867,7 +34867,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32000, + "maxTokens": 32768, "thinking": { "mode": "effort", "efforts": [ @@ -34901,7 +34901,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32000, + "maxTokens": 32768, "thinking": { "mode": "effort", "efforts": [ diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 1b57bd811..ce8222e13 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -2707,6 +2707,34 @@ function mapKimiApiFormat(protocol: unknown): OpenAICompat["kimiApiFormat"] { return undefined; } +/** + * Kimi Code output ceilings by model family. The `/coding/v1/models` discovery + * envelope carries no output-limit field, so the mapper supplies the documented + * per-family caps instead of a blanket constant. Values match models.dev's + * `kimi-for-coding` and `moonshotai` kimi-k3 entries. See #6711. + */ +export const KIMI_CODE_K3_MAX_TOKENS = 131_072; +export const KIMI_CODE_FOR_CODING_MAX_TOKENS = 32_768; + +/** Fallback output cap for Kimi Code families without a documented ceiling (legacy K2 discovery rows). */ +export const KIMI_CODE_DEFAULT_MAX_TOKENS = 32_000; + +/** + * Resolve a Kimi Code model's output ceiling from its id: `k3` / `k3-256k` -> + * 131072, `kimi-for-coding[-highspeed]` -> 32768, everything else -> `fallback`. + */ +export function kimiCodeMaxTokens(modelId: string, fallback?: number): number; +export function kimiCodeMaxTokens(modelId: string, fallback: number | null): number | null; +export function kimiCodeMaxTokens( + modelId: string, + fallback: number | null = KIMI_CODE_DEFAULT_MAX_TOKENS, +): number | null { + const id = modelId.toLowerCase(); + if (id.startsWith("k3")) return KIMI_CODE_K3_MAX_TOKENS; + if (id.startsWith("kimi-for-coding")) return KIMI_CODE_FOR_CODING_MAX_TOKENS; + return fallback; +} + export function kimiCodeModelManagerOptions( config?: KimiCodeModelManagerConfig, ): ModelManagerOptions<"openai-completions"> { @@ -2739,7 +2767,7 @@ export function kimiCodeModelManagerOptions( reasoning, input: entry.supports_image_in === true || id.includes("k2.5") ? ["text", "image"] : ["text"], contextWindow: typeof entry.context_length === "number" ? entry.context_length : 262144, - maxTokens: 32000, + maxTokens: kimiCodeMaxTokens(id), thinking, compat: { thinkingFormat: thinking ? "kimi" : "zai", diff --git a/packages/catalog/test/kimi-code-provider.test.ts b/packages/catalog/test/kimi-code-provider.test.ts index 50ea839e8..3a3534cb2 100644 --- a/packages/catalog/test/kimi-code-provider.test.ts +++ b/packages/catalog/test/kimi-code-provider.test.ts @@ -35,6 +35,7 @@ describe("Kimi Code provider catalog", () => { name: "K3", reasoning: true, contextWindow: 1_048_576, + maxTokens: 131_072, thinking: { mode: "effort", efforts: [Effort.Low, Effort.High, Effort.Max], @@ -67,6 +68,28 @@ describe("Kimi Code provider catalog", () => { expect(legacy?.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]); }); + it("derives per-family output caps instead of a blanket constant (#6711)", async () => { + const models = await discover([ + LIVE_K3, + { ...LIVE_K3, id: "k3-256k", display_name: "K3 256k", context_length: 262_144 }, + { id: "kimi-for-coding", display_name: "K2.7 Code", context_length: 262_144, supports_reasoning: true }, + { + id: "kimi-for-coding-highspeed", + display_name: "K2.7 Code Highspeed", + context_length: 262_144, + supports_reasoning: true, + }, + { id: "kimi-k2", display_name: "Kimi K2", context_length: 262_144 }, + ]); + const maxTokensFor = (id: string) => models.find(model => model.id === id)?.maxTokens; + + expect(maxTokensFor("k3")).toBe(131_072); + expect(maxTokensFor("k3-256k")).toBe(131_072); + expect(maxTokensFor("kimi-for-coding")).toBe(32_768); + expect(maxTokensFor("kimi-for-coding-highspeed")).toBe(32_768); + expect(maxTokensFor("kimi-k2")).toBe(32_000); + }); + it("lets supports_thinking_type override the legacy reasoning flag", async () => { const models = await discover([ { ...LIVE_K3, id: "non-thinking", supports_thinking_type: "no", think_efforts: undefined }, From b008bfc34d810e343a0c37e97f547bce06703d6a Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 26 Jul 2026 14:48:47 +0000 Subject: [PATCH 2/2] fix(catalog): invalidate stale kimi-code cache rows Bump CACHE_SCHEMA_VERSION to 12 so upgrades drop cached kimi-code rows carrying the pre-fix blanket maxTokens=32000. Without this, offline or fresh-cache refreshes never re-run the discovery mapper, and the discovery-only k3-256k (absent from the bundle) is not reconciled by static-fingerprint handling, so the wrong cap persisted until a network refresh. Fixes #6711 --- packages/catalog/CHANGELOG.md | 2 +- packages/catalog/src/model-cache.ts | 8 +++++--- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 4ed9c9d25..dbbdd4c13 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -12,7 +12,7 @@ - Fixed forced `tool_choice` 400s (`tool_choice 'specified' is incompatible with thinking enabled`) on Kimi Code's Anthropic-compatible endpoint for the `kimi-for-coding`, `kimi-for-coding-highspeed`, and `k3` aliases: the Anthropic-surface compat matcher only recognised Moonshot's native `kimi-k2.7-code*` ids, so thinking-locked kimi-code models kept `supportsForcedToolChoice: true` and the forced selector was sent to a host that always thinks. These models now resolve `requiresThinkingEnabled`, keeping thinking on and downgrading forced choices to `auto`. - Retried empty successful provider discovery responses after the short non-authoritative interval instead of caching them for the full catalog TTL ([#6620](https://github.com/can1357/oh-my-pi/issues/6620)). - Fixed GitHub Copilot Claude models with no bundled catalog reference (e.g. a freshly served `claude-opus-5`) discovering with `reasoning: false`/`thinking: null` and no effort dial, and disappearing along with their synthesized `-1m` sibling on offline reads: reference-less Copilot models on the anthropic-messages proxy now derive the adaptive reasoning ladder from the model id, and the cache restores their compile-time `COPILOT_API_HEADERS` by value instead of dropping them as unrestorable ([#6664](https://github.com/can1357/oh-my-pi/issues/6664)). -- Fixed Kimi Code (`kimi-code`) reporting `maxTokens: 32000` for every model — its `/coding/v1/models` discovery mapper and the bundled catalog applied a blanket constant, truncating `k3`/`k3-256k` output at ~4x below their real 131072 ceiling and `kimi-for-coding`/`kimi-for-coding-highspeed` below their 32768 ceiling. Output caps are now derived per family ([#6711](https://github.com/can1357/oh-my-pi/issues/6711)). +- Fixed Kimi Code (`kimi-code`) reporting `maxTokens: 32000` for every model — its `/coding/v1/models` discovery mapper and the bundled catalog applied a blanket constant, truncating `k3`/`k3-256k` output at ~4x below their real 131072 ceiling and `kimi-for-coding`/`kimi-for-coding-highspeed` below their 32768 ceiling. Output caps are now derived per family, and the model cache is invalidated so upgrades drop the stale `maxTokens: 32000` rows (including the discovery-only `k3-256k`) instead of serving them until the next network refresh ([#6711](https://github.com/can1357/oh-my-pi/issues/6711)). ## [17.1.3] - 2026-07-24 diff --git a/packages/catalog/src/model-cache.ts b/packages/catalog/src/model-cache.ts index 871a57406..da10305f9 100644 --- a/packages/catalog/src/model-cache.ts +++ b/packages/catalog/src/model-cache.ts @@ -9,8 +9,10 @@ import type { Api, Model, ModelSpec } from "./types"; // Rows persist ModelSpec JSON (sparse `compat`, never the resolved record); // the model manager rebuilds via `buildModel` on load. Request headers are // intentionally omitted: arbitrary provider-defined header names can carry -// credentials. v11 invalidates rows that may persist derived computer-use -// support without provenance; v10 deletes rows that may contain persisted +// credentials. v12 invalidates Kimi Code rows carrying the blanket +// maxTokens: 32000 that predate per-family output caps (k3/k3-256k -> 131072, +// kimi-for-coding[-highspeed] -> 32768, #6711); v11 invalidates rows that may +// persist derived computer-use // headers and records which model ids lost headers or cannot be rebuilt. // v9 invalidated Kimi Code rows predating live effort and protocol metadata; // v8 invalidated Codex discovery rows predating provider-native V2 compaction @@ -20,7 +22,7 @@ import type { Api, Model, ModelSpec } from "./types"; // retired unknown-limit sentinels (222222/8888); v5 invalidated rows predating // effort-tier variant collapsing (raw `-low`/`-high`/`-thinking` member ids); // v4 dropped the pre-efforts ThinkingConfig shape. -const CACHE_SCHEMA_VERSION = 11; +const CACHE_SCHEMA_VERSION = 12; const HEADER_RESTORE_VERSION = 1; interface CacheRow {