Merge PR #6713: fix(catalog): derive kimi-code output caps per family (@roboomp)

This commit is contained in:
can1357
2026-07-27 04:58:25 +02:00
6 changed files with 68 additions and 7 deletions
+1
View File
@@ -14,6 +14,7 @@
- Fixed forced `tool_choice` 400s (`tool_choice 'specified' is incompatible with thinking enabled`) on Kimi Code's Anthropic-compatible endpoint for the `kimi-for-coding`, `kimi-for-coding-highspeed`, and `k3` aliases: the Anthropic-surface compat matcher only recognised Moonshot's native `kimi-k2.7-code*` ids, so thinking-locked kimi-code models kept `supportsForcedToolChoice: true` and the forced selector was sent to a host that always thinks. These models now resolve `requiresThinkingEnabled`, keeping thinking on and downgrading forced choices to `auto`.
- Retried empty successful provider discovery responses after the short non-authoritative interval instead of caching them for the full catalog TTL ([#6620](https://github.com/can1357/oh-my-pi/issues/6620)).
- Fixed GitHub Copilot Claude models with no bundled catalog reference (e.g. a freshly served `claude-opus-5`) discovering with `reasoning: false`/`thinking: null` and no effort dial, and disappearing along with their synthesized `-1m` sibling on offline reads: reference-less Copilot models on the anthropic-messages proxy now derive the adaptive reasoning ladder from the model id, and the cache restores their compile-time `COPILOT_API_HEADERS` by value instead of dropping them as unrestorable ([#6664](https://github.com/can1357/oh-my-pi/issues/6664)).
- Fixed Kimi Code (`kimi-code`) reporting `maxTokens: 32000` for every model — its `/coding/v1/models` discovery mapper and the bundled catalog applied a blanket constant, truncating `k3`/`k3-256k` output at ~4x below their real 131072 ceiling and `kimi-for-coding`/`kimi-for-coding-highspeed` below their 32768 ceiling. Output caps are now derived per family, and the model cache is invalidated so upgrades drop the stale `maxTokens: 32000` rows (including the discovery-only `k3-256k`) instead of serving them until the next network refresh ([#6711](https://github.com/can1357/oh-my-pi/issues/6711)).
## [17.1.3] - 2026-07-24
@@ -37,6 +37,7 @@ import {
clampKimiK27CodeMaxTokens,
isFireworksKimiK2ModelId,
isKimiK27CodeModelId,
kimiCodeMaxTokens,
META_MUSE_STATIC_MODELS,
MODELS_DEV_PROVIDER_DESCRIPTORS,
mapModelsDevToModels,
@@ -304,6 +305,12 @@ function applyKimiMaxTokensCap(models: readonly ModelSpec[]): ModelSpec[] {
const capped = clampKimiK27CodeMaxTokens(model.id, model.maxTokens);
return capped === model.maxTokens ? model : { ...model, maxTokens: capped };
}
if (model.provider === "kimi-code") {
// Discovery snapshots carried maxTokens=32000 uniformly (#6711); pin the
// documented per-family output ceilings and leave legacy K2 rows as-is.
const capped = kimiCodeMaxTokens(model.id, model.maxTokens);
return capped === model.maxTokens ? model : { ...model, maxTokens: capped };
}
return model;
});
}
+5 -3
View File
@@ -9,8 +9,10 @@ import type { Api, Model, ModelSpec } from "./types";
// Rows persist ModelSpec JSON (sparse `compat`, never the resolved record);
// the model manager rebuilds via `buildModel` on load. Request headers are
// intentionally omitted: arbitrary provider-defined header names can carry
// credentials. v11 invalidates rows that may persist derived computer-use
// support without provenance; v10 deletes rows that may contain persisted
// credentials. v12 invalidates Kimi Code rows carrying the blanket
// maxTokens: 32000 that predate per-family output caps (k3/k3-256k -> 131072,
// kimi-for-coding[-highspeed] -> 32768, #6711); v11 invalidates rows that may
// persist derived computer-use
// headers and records which model ids lost headers or cannot be rebuilt.
// v9 invalidated Kimi Code rows predating live effort and protocol metadata;
// v8 invalidated Codex discovery rows predating provider-native V2 compaction
@@ -20,7 +22,7 @@ import type { Api, Model, ModelSpec } from "./types";
// retired unknown-limit sentinels (222222/8888); v5 invalidated rows predating
// effort-tier variant collapsing (raw `-low`/`-high`/`-thinking` member ids);
// v4 dropped the pre-efforts ThinkingConfig shape.
const CACHE_SCHEMA_VERSION = 11;
const CACHE_SCHEMA_VERSION = 12;
const HEADER_RESTORE_VERSION = 1;
interface CacheRow {
+3 -3
View File
@@ -35139,7 +35139,7 @@
"cacheWrite": 0
},
"contextWindow": 1048576,
"maxTokens": 32000,
"maxTokens": 131072,
"thinking": {
"mode": "effort",
"efforts": [
@@ -35213,7 +35213,7 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 32000,
"maxTokens": 32768,
"thinking": {
"mode": "effort",
"efforts": [
@@ -35248,7 +35248,7 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 32000,
"maxTokens": 32768,
"thinking": {
"mode": "effort",
"efforts": [
@@ -2707,6 +2707,34 @@ function mapKimiApiFormat(protocol: unknown): OpenAICompat["kimiApiFormat"] {
return undefined;
}
/**
* Kimi Code output ceilings by model family. The `/coding/v1/models` discovery
* envelope carries no output-limit field, so the mapper supplies the documented
* per-family caps instead of a blanket constant. Values match models.dev's
* `kimi-for-coding` and `moonshotai` kimi-k3 entries. See #6711.
*/
export const KIMI_CODE_K3_MAX_TOKENS = 131_072;
export const KIMI_CODE_FOR_CODING_MAX_TOKENS = 32_768;
/** Fallback output cap for Kimi Code families without a documented ceiling (legacy K2 discovery rows). */
export const KIMI_CODE_DEFAULT_MAX_TOKENS = 32_000;
/**
* Resolve a Kimi Code model's output ceiling from its id: `k3` / `k3-256k` ->
* 131072, `kimi-for-coding[-highspeed]` -> 32768, everything else -> `fallback`.
*/
export function kimiCodeMaxTokens(modelId: string, fallback?: number): number;
export function kimiCodeMaxTokens(modelId: string, fallback: number | null): number | null;
export function kimiCodeMaxTokens(
modelId: string,
fallback: number | null = KIMI_CODE_DEFAULT_MAX_TOKENS,
): number | null {
const id = modelId.toLowerCase();
if (id.startsWith("k3")) return KIMI_CODE_K3_MAX_TOKENS;
if (id.startsWith("kimi-for-coding")) return KIMI_CODE_FOR_CODING_MAX_TOKENS;
return fallback;
}
export function kimiCodeModelManagerOptions(
config?: KimiCodeModelManagerConfig,
): ModelManagerOptions<"openai-completions"> {
@@ -2739,7 +2767,7 @@ export function kimiCodeModelManagerOptions(
reasoning,
input: entry.supports_image_in === true || id.includes("k2.5") ? ["text", "image"] : ["text"],
contextWindow: typeof entry.context_length === "number" ? entry.context_length : 262144,
maxTokens: 32000,
maxTokens: kimiCodeMaxTokens(id),
thinking,
compat: {
thinkingFormat: thinking ? "kimi" : "zai",
@@ -35,6 +35,7 @@ describe("Kimi Code provider catalog", () => {
name: "K3",
reasoning: true,
contextWindow: 1_048_576,
maxTokens: 131_072,
thinking: {
mode: "effort",
efforts: [Effort.Low, Effort.High, Effort.Max],
@@ -67,6 +68,28 @@ describe("Kimi Code provider catalog", () => {
expect(legacy?.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]);
});
it("derives per-family output caps instead of a blanket constant (#6711)", async () => {
const models = await discover([
LIVE_K3,
{ ...LIVE_K3, id: "k3-256k", display_name: "K3 256k", context_length: 262_144 },
{ id: "kimi-for-coding", display_name: "K2.7 Code", context_length: 262_144, supports_reasoning: true },
{
id: "kimi-for-coding-highspeed",
display_name: "K2.7 Code Highspeed",
context_length: 262_144,
supports_reasoning: true,
},
{ id: "kimi-k2", display_name: "Kimi K2", context_length: 262_144 },
]);
const maxTokensFor = (id: string) => models.find(model => model.id === id)?.maxTokens;
expect(maxTokensFor("k3")).toBe(131_072);
expect(maxTokensFor("k3-256k")).toBe(131_072);
expect(maxTokensFor("kimi-for-coding")).toBe(32_768);
expect(maxTokensFor("kimi-for-coding-highspeed")).toBe(32_768);
expect(maxTokensFor("kimi-k2")).toBe(32_000);
});
it("lets supports_thinking_type override the legacy reasoning flag", async () => {
const models = await discover([
{ ...LIVE_K3, id: "non-thinking", supports_thinking_type: "no", think_efforts: undefined },