diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index feda2a158..30b6b1c0b 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -14,6 +14,7 @@ - Fixed forced `tool_choice` 400s (`tool_choice 'specified' is incompatible with thinking enabled`) on Kimi Code's Anthropic-compatible endpoint for the `kimi-for-coding`, `kimi-for-coding-highspeed`, and `k3` aliases: the Anthropic-surface compat matcher only recognised Moonshot's native `kimi-k2.7-code*` ids, so thinking-locked kimi-code models kept `supportsForcedToolChoice: true` and the forced selector was sent to a host that always thinks. These models now resolve `requiresThinkingEnabled`, keeping thinking on and downgrading forced choices to `auto`. - Retried empty successful provider discovery responses after the short non-authoritative interval instead of caching them for the full catalog TTL ([#6620](https://github.com/can1357/oh-my-pi/issues/6620)). - Fixed GitHub Copilot Claude models with no bundled catalog reference (e.g. a freshly served `claude-opus-5`) discovering with `reasoning: false`/`thinking: null` and no effort dial, and disappearing along with their synthesized `-1m` sibling on offline reads: reference-less Copilot models on the anthropic-messages proxy now derive the adaptive reasoning ladder from the model id, and the cache restores their compile-time `COPILOT_API_HEADERS` by value instead of dropping them as unrestorable ([#6664](https://github.com/can1357/oh-my-pi/issues/6664)). +- Fixed Kimi Code (`kimi-code`) reporting `maxTokens: 32000` for every model — its `/coding/v1/models` discovery mapper and the bundled catalog applied a blanket constant, truncating `k3`/`k3-256k` output at ~4x below their real 131072 ceiling and `kimi-for-coding`/`kimi-for-coding-highspeed` below their 32768 ceiling. Output caps are now derived per family, and the model cache is invalidated so upgrades drop the stale `maxTokens: 32000` rows (including the discovery-only `k3-256k`) instead of serving them until the next network refresh ([#6711](https://github.com/can1357/oh-my-pi/issues/6711)). ## [17.1.3] - 2026-07-24 diff --git a/packages/catalog/scripts/generate-models.ts b/packages/catalog/scripts/generate-models.ts index 6971eab37..48cecef63 100644 --- a/packages/catalog/scripts/generate-models.ts +++ b/packages/catalog/scripts/generate-models.ts @@ -37,6 +37,7 @@ import { clampKimiK27CodeMaxTokens, isFireworksKimiK2ModelId, isKimiK27CodeModelId, + kimiCodeMaxTokens, META_MUSE_STATIC_MODELS, MODELS_DEV_PROVIDER_DESCRIPTORS, mapModelsDevToModels, @@ -304,6 +305,12 @@ function applyKimiMaxTokensCap(models: readonly ModelSpec[]): ModelSpec[] { const capped = clampKimiK27CodeMaxTokens(model.id, model.maxTokens); return capped === model.maxTokens ? model : { ...model, maxTokens: capped }; } + if (model.provider === "kimi-code") { + // Discovery snapshots carried maxTokens=32000 uniformly (#6711); pin the + // documented per-family output ceilings and leave legacy K2 rows as-is. + const capped = kimiCodeMaxTokens(model.id, model.maxTokens); + return capped === model.maxTokens ? model : { ...model, maxTokens: capped }; + } return model; }); } diff --git a/packages/catalog/src/model-cache.ts b/packages/catalog/src/model-cache.ts index 871a57406..da10305f9 100644 --- a/packages/catalog/src/model-cache.ts +++ b/packages/catalog/src/model-cache.ts @@ -9,8 +9,10 @@ import type { Api, Model, ModelSpec } from "./types"; // Rows persist ModelSpec JSON (sparse `compat`, never the resolved record); // the model manager rebuilds via `buildModel` on load. Request headers are // intentionally omitted: arbitrary provider-defined header names can carry -// credentials. v11 invalidates rows that may persist derived computer-use -// support without provenance; v10 deletes rows that may contain persisted +// credentials. v12 invalidates Kimi Code rows carrying the blanket +// maxTokens: 32000 that predate per-family output caps (k3/k3-256k -> 131072, +// kimi-for-coding[-highspeed] -> 32768, #6711); v11 invalidates rows that may +// persist derived computer-use // headers and records which model ids lost headers or cannot be rebuilt. // v9 invalidated Kimi Code rows predating live effort and protocol metadata; // v8 invalidated Codex discovery rows predating provider-native V2 compaction @@ -20,7 +22,7 @@ import type { Api, Model, ModelSpec } from "./types"; // retired unknown-limit sentinels (222222/8888); v5 invalidated rows predating // effort-tier variant collapsing (raw `-low`/`-high`/`-thinking` member ids); // v4 dropped the pre-efforts ThinkingConfig shape. -const CACHE_SCHEMA_VERSION = 11; +const CACHE_SCHEMA_VERSION = 12; const HEADER_RESTORE_VERSION = 1; interface CacheRow { diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index c3558a915..d89afe2f4 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -35139,7 +35139,7 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 32000, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -35213,7 +35213,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32000, + "maxTokens": 32768, "thinking": { "mode": "effort", "efforts": [ @@ -35248,7 +35248,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32000, + "maxTokens": 32768, "thinking": { "mode": "effort", "efforts": [ diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 1b57bd811..ce8222e13 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -2707,6 +2707,34 @@ function mapKimiApiFormat(protocol: unknown): OpenAICompat["kimiApiFormat"] { return undefined; } +/** + * Kimi Code output ceilings by model family. The `/coding/v1/models` discovery + * envelope carries no output-limit field, so the mapper supplies the documented + * per-family caps instead of a blanket constant. Values match models.dev's + * `kimi-for-coding` and `moonshotai` kimi-k3 entries. See #6711. + */ +export const KIMI_CODE_K3_MAX_TOKENS = 131_072; +export const KIMI_CODE_FOR_CODING_MAX_TOKENS = 32_768; + +/** Fallback output cap for Kimi Code families without a documented ceiling (legacy K2 discovery rows). */ +export const KIMI_CODE_DEFAULT_MAX_TOKENS = 32_000; + +/** + * Resolve a Kimi Code model's output ceiling from its id: `k3` / `k3-256k` -> + * 131072, `kimi-for-coding[-highspeed]` -> 32768, everything else -> `fallback`. + */ +export function kimiCodeMaxTokens(modelId: string, fallback?: number): number; +export function kimiCodeMaxTokens(modelId: string, fallback: number | null): number | null; +export function kimiCodeMaxTokens( + modelId: string, + fallback: number | null = KIMI_CODE_DEFAULT_MAX_TOKENS, +): number | null { + const id = modelId.toLowerCase(); + if (id.startsWith("k3")) return KIMI_CODE_K3_MAX_TOKENS; + if (id.startsWith("kimi-for-coding")) return KIMI_CODE_FOR_CODING_MAX_TOKENS; + return fallback; +} + export function kimiCodeModelManagerOptions( config?: KimiCodeModelManagerConfig, ): ModelManagerOptions<"openai-completions"> { @@ -2739,7 +2767,7 @@ export function kimiCodeModelManagerOptions( reasoning, input: entry.supports_image_in === true || id.includes("k2.5") ? ["text", "image"] : ["text"], contextWindow: typeof entry.context_length === "number" ? entry.context_length : 262144, - maxTokens: 32000, + maxTokens: kimiCodeMaxTokens(id), thinking, compat: { thinkingFormat: thinking ? "kimi" : "zai", diff --git a/packages/catalog/test/kimi-code-provider.test.ts b/packages/catalog/test/kimi-code-provider.test.ts index 50ea839e8..3a3534cb2 100644 --- a/packages/catalog/test/kimi-code-provider.test.ts +++ b/packages/catalog/test/kimi-code-provider.test.ts @@ -35,6 +35,7 @@ describe("Kimi Code provider catalog", () => { name: "K3", reasoning: true, contextWindow: 1_048_576, + maxTokens: 131_072, thinking: { mode: "effort", efforts: [Effort.Low, Effort.High, Effort.Max], @@ -67,6 +68,28 @@ describe("Kimi Code provider catalog", () => { expect(legacy?.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]); }); + it("derives per-family output caps instead of a blanket constant (#6711)", async () => { + const models = await discover([ + LIVE_K3, + { ...LIVE_K3, id: "k3-256k", display_name: "K3 256k", context_length: 262_144 }, + { id: "kimi-for-coding", display_name: "K2.7 Code", context_length: 262_144, supports_reasoning: true }, + { + id: "kimi-for-coding-highspeed", + display_name: "K2.7 Code Highspeed", + context_length: 262_144, + supports_reasoning: true, + }, + { id: "kimi-k2", display_name: "Kimi K2", context_length: 262_144 }, + ]); + const maxTokensFor = (id: string) => models.find(model => model.id === id)?.maxTokens; + + expect(maxTokensFor("k3")).toBe(131_072); + expect(maxTokensFor("k3-256k")).toBe(131_072); + expect(maxTokensFor("kimi-for-coding")).toBe(32_768); + expect(maxTokensFor("kimi-for-coding-highspeed")).toBe(32_768); + expect(maxTokensFor("kimi-k2")).toBe(32_000); + }); + it("lets supports_thinking_type override the legacy reasoning flag", async () => { const models = await discover([ { ...LIVE_K3, id: "non-thinking", supports_thinking_type: "no", think_efforts: undefined },