merge PR #5758 via eval/pr-5758: fix(ai): honored kimi-k3 131k output limit
This commit is contained in:
@@ -94,9 +94,9 @@ import {
|
||||
type OpenAIStrictToolsState,
|
||||
parseAzureDeploymentNameMap,
|
||||
resolveOpenAICompatPolicy,
|
||||
resolveOpenAICompletionsOutputClamp,
|
||||
resolveOpenAIOutputTokenParam,
|
||||
resolveOpenAIRequestSetup,
|
||||
resolveZaiReasoningOutputClamp,
|
||||
shouldRetryWithoutStrictTools,
|
||||
} from "./openai-shared";
|
||||
import { transformMessages } from "./transform-messages";
|
||||
@@ -1571,7 +1571,7 @@ function buildParams(
|
||||
omitMaxOutputTokens: model.omitMaxOutputTokens ?? false,
|
||||
isOpenRouterHost: compat.isOpenRouterHost,
|
||||
alwaysSendMaxTokens: compat.alwaysSendMaxTokens,
|
||||
providerOutputClamp: resolveZaiReasoningOutputClamp(model, compat),
|
||||
providerOutputClamp: resolveOpenAICompletionsOutputClamp(model, compat),
|
||||
});
|
||||
if (outputToken) {
|
||||
if (outputToken.field === "max_tokens") {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import type { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import { toFirepassWireModelId, toFireworksWireModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id";
|
||||
import { isGlm52ReasoningEffortModelId } from "@oh-my-pi/pi-catalog/identity";
|
||||
import { isGlm52ReasoningEffortModelId, isKimiK3ModelId } from "@oh-my-pi/pi-catalog/identity";
|
||||
import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
||||
import type {
|
||||
@@ -1025,16 +1025,24 @@ function isZaiReasoningEffortDialect(model: Model<"openai-completions">, compat:
|
||||
}
|
||||
|
||||
/**
|
||||
* Output-token clamp for the Z.AI/GLM-5.2 reasoning dialect: these hosts accept
|
||||
* the full model window on reasoning turns, so clamp to the model cap. Returns
|
||||
* `undefined` for every other model, leaving {@link resolveOpenAIOutputTokenParam}
|
||||
* on its default `OPENAI_MAX_OUTPUT_TOKENS` clamp.
|
||||
* Provider-specific Chat Completions output clamp.
|
||||
*
|
||||
* Most OpenAI-compatible endpoints retain the conservative 64k ceiling from
|
||||
* {@link resolveOpenAIOutputTokenParam}. Z.AI/GLM-5.2 reasoning and native
|
||||
* Moonshot K3 explicitly accept their full advertised model caps, so those
|
||||
* routes clamp to `model.maxTokens` instead.
|
||||
*/
|
||||
export function resolveZaiReasoningOutputClamp(
|
||||
export function resolveOpenAICompletionsOutputClamp(
|
||||
model: Model<"openai-completions">,
|
||||
compat: ResolvedOpenAICompat,
|
||||
): number | undefined {
|
||||
return isZaiReasoningEffortDialect(model, compat) ? (model.maxTokens ?? OPENAI_MAX_OUTPUT_TOKENS) : undefined;
|
||||
if (isZaiReasoningEffortDialect(model, compat)) {
|
||||
return model.maxTokens ?? OPENAI_MAX_OUTPUT_TOKENS;
|
||||
}
|
||||
if (model.provider === "moonshot" && isKimiK3ModelId(model.id)) {
|
||||
return model.maxTokens ?? OPENAI_MAX_OUTPUT_TOKENS;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -11,6 +11,9 @@
|
||||
### Fixed
|
||||
|
||||
- Fixed Umans PAYG models showing as "Free" in `/models` by sourcing the provider's published per-token rates instead of the all-zero coding-plan catalog ([#5733](https://github.com/can1357/oh-my-pi/issues/5733)).
|
||||
### Fixed
|
||||
|
||||
- Fixed native `moonshot/kimi-k3` being labeled "Free" with no capabilities: the discovered id has no bundled/models.dev reference, so it fell through to zero cost, null limits, text-only input, and no reasoning. It now carries Moonshot's official K3 pricing (`$3` input / `$0.30` cache-hit / `$15` output), a 1,048,576-token context window, image input, and reasoning that routes through OpenAI-style `reasoning_effort: "max"` (K3 does not use the K2.x `thinking` block). Native K3 is also exempt from the Kimi forced-tool-choice reasoning suppression (a K2.x-only Moonshot conflict), so plan-mode forced tool turns keep the mandatory `max` effort; its documented 131,072-token output cap is allowed through the Chat Completions request clamp instead of being reduced to the generic 64,000-token ceiling ([#5756](https://github.com/can1357/oh-my-pi/issues/5756)).
|
||||
|
||||
## [17.0.1] - 2026-07-16
|
||||
|
||||
|
||||
@@ -15,6 +15,7 @@ import {
|
||||
isDeepseekModelIdOrName,
|
||||
isGlm52ReasoningEffortModelId,
|
||||
isGrokReasoningEffortCapable,
|
||||
isKimiK3ModelId,
|
||||
isKimiK26ModelId,
|
||||
isKimiModelId,
|
||||
isMimoModelIdOrName,
|
||||
@@ -247,6 +248,11 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
const isKimiModel = isKimiModelId(spec.id);
|
||||
const isMoonshotNative = modelMatchesHost(hostModel, "moonshotNative");
|
||||
const isMoonshotKimi = isKimiModel && isMoonshotNative;
|
||||
// Kimi K3 (native) always reasons via OpenAI-style `reasoning_effort: "max"`
|
||||
// and does NOT accept the K2.x binary `thinking: { type }` block, so it must
|
||||
// stay on the "openai" thinking dialect even though it is a Moonshot-native
|
||||
// Kimi model (#5756).
|
||||
const isMoonshotKimiK3 = isMoonshotKimi && isKimiK3ModelId(spec.id);
|
||||
const requiresEnabledThinking = isMoonshotKimi && matchesKimiK27CodeFamily(spec);
|
||||
const usesMoonshotKimiPreservedThinking = isMoonshotKimi && isKimiK26ModelId(spec.id);
|
||||
const isAnthropicModel =
|
||||
@@ -364,7 +370,10 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
? ALIBABA_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS
|
||||
: isXiaomiMimo
|
||||
? XIAOMI_MIMO_STREAM_IDLE_TIMEOUT_MS
|
||||
: spec.reasoning && (isKimiK26ModelId(spec.id) || (isMoonshotKimi && matchesKimiK27CodeFamily(spec)))
|
||||
: spec.reasoning &&
|
||||
(isKimiK26ModelId(spec.id) ||
|
||||
isMoonshotKimiK3 ||
|
||||
(isMoonshotKimi && matchesKimiK27CodeFamily(spec)))
|
||||
? KIMI_REASONING_STREAM_IDLE_TIMEOUT_MS
|
||||
: spec.reasoning && isDirectDeepseekApi
|
||||
? DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS
|
||||
@@ -385,7 +394,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
? "openrouter"
|
||||
: "raw";
|
||||
const thinkingFormat: ResolvedOpenAISharedCompat["thinkingFormat"] =
|
||||
isZai || isZhipu || isMoonshotKimi || isXiaomiMimo
|
||||
(isMoonshotKimi && !isMoonshotKimiK3) || isZai || isZhipu || isXiaomiMimo
|
||||
? "zai"
|
||||
: isOpenRouter
|
||||
? "openrouter"
|
||||
@@ -426,7 +435,12 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
// every call since the family can otherwise emit very long reasoning traces
|
||||
// before the final answer.
|
||||
alwaysSendMaxTokens: isKimiModel,
|
||||
disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel,
|
||||
// Native Kimi K3 always reasons via `reasoning_effort: "max"` (never the
|
||||
// K2.x binary `thinking` block that #827's forced-tool-choice conflict is
|
||||
// about), so suppressing its effort would strip the mandatory `max` from
|
||||
// normal forced-tool turns (e.g. plan-mode `toolChoice: "required"`) and
|
||||
// leave K3 in an unsupported mode (#5758 review).
|
||||
disableReasoningOnForcedToolChoice: (isKimiModel && !isMoonshotKimiK3) || isAnthropicModel,
|
||||
disableReasoningOnToolChoice: isDeepseekFamily && Boolean(spec.reasoning) && !isOpenRouter,
|
||||
supportsToolChoice: !isDirectDeepseekReasoning,
|
||||
supportsForcedToolChoice: !requiresEnabledThinking,
|
||||
|
||||
@@ -41,6 +41,16 @@ export const isKimiK26ModelId = memo((modelId: string): boolean => {
|
||||
return /(^|\/)kimi-k2(?:\.6|p6)(?:[-:]|$)/i.test(modelId);
|
||||
});
|
||||
|
||||
/**
|
||||
* Kimi K3 in any namespace form (`kimi-k3`, `kimi-k3.1`, `kimi-k3-turbo`,
|
||||
* `moonshotai/kimi-k3`). K3 always reasons and drives thinking via OpenAI-style
|
||||
* `reasoning_effort: "max"`, not the K2.x binary `thinking: { type }` block —
|
||||
* see the moonshot discovery mapper and `buildOpenAICompat`.
|
||||
*/
|
||||
export const isKimiK3ModelId = memo((modelId: string): boolean => {
|
||||
return /(^|\/)kimi-k3(?:\.\d+)?(?:[-.:_]|$)/i.test(modelId);
|
||||
});
|
||||
|
||||
/**
|
||||
* Claude ids in any namespace form: bare (`claude-*`), path-namespaced
|
||||
* (`anthropic/claude.x`), or dot-prefixed (`us.anthropic.claude-…`,
|
||||
|
||||
@@ -8,6 +8,7 @@ import { FIREWORKS_FAST_SUFFIX, toFireworksPublicModelId } from "../fireworks-mo
|
||||
import {
|
||||
isGlmVisionModelId,
|
||||
isGrokReasoningEffortCapable,
|
||||
isKimiK3ModelId,
|
||||
isKimiModelId,
|
||||
isReasoningGlmModelId,
|
||||
} from "../identity/family";
|
||||
@@ -2916,6 +2917,23 @@ export interface MoonshotModelManagerConfig {
|
||||
fetch?: FetchImpl;
|
||||
}
|
||||
|
||||
/**
|
||||
* Moonshot Kimi K3 discovery metadata. K3 is dynamically discovered but absent
|
||||
* from models.dev and the bundled catalog, so `mapWithBundledReference` would
|
||||
* otherwise assign zero cost, null limits, text-only input, and no reasoning —
|
||||
* mislabeling a paid model as "Free" (#5756). Pricing/limits from Moonshot's
|
||||
* official chat-k3 pricing and quickstart guide:
|
||||
* https://platform.kimi.ai/docs/pricing/chat-k3.md
|
||||
* https://platform.kimi.ai/docs/guide/kimi-k3-quickstart
|
||||
* K3 always reasons and supports only `reasoning_effort: "max"` — it does NOT
|
||||
* use the K2.x binary `thinking: { type }` block, so the wire path routes it
|
||||
* through OpenAI-style `reasoning_effort` (see `buildOpenAICompat`).
|
||||
*/
|
||||
const MOONSHOT_KIMI_K3_COST = { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 0 } as const;
|
||||
const MOONSHOT_KIMI_K3_CONTEXT_WINDOW = 1_048_576;
|
||||
const MOONSHOT_KIMI_K3_MAX_TOKENS = 131_072;
|
||||
const MOONSHOT_KIMI_K3_THINKING: ThinkingConfig = { mode: "effort", efforts: [Effort.Max], requiresEffort: true };
|
||||
|
||||
export function moonshotModelManagerOptions(
|
||||
config?: MoonshotModelManagerConfig,
|
||||
): ModelManagerOptions<"openai-completions"> {
|
||||
@@ -2938,6 +2956,25 @@ export function moonshotModelManagerOptions(
|
||||
const reference = references.get(defaults.id);
|
||||
const model = mapWithBundledReference(entry, defaults, reference);
|
||||
const id = model.id.toLowerCase();
|
||||
// Kimi K3 is discovered but has no bundled/models.dev reference, so the
|
||||
// generic dynamic defaults would report it "Free" with no capabilities
|
||||
// (#5756). Stamp the official pricing/limits when the endpoint doesn't
|
||||
// carry them, and mark it reasoning + vision. K3 always reasons via
|
||||
// `reasoning_effort: "max"` and does NOT use the K2.x `thinking` block,
|
||||
// so its thinking config is the single-tier `max` scale — the wire path
|
||||
// routes it through `reasoning_effort` (see `buildOpenAICompat`).
|
||||
if (!reference && isKimiK3ModelId(id)) {
|
||||
const isZeroCost = model.cost.input === 0 && model.cost.output === 0 && model.cost.cacheRead === 0;
|
||||
return {
|
||||
...model,
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: isZeroCost ? { ...MOONSHOT_KIMI_K3_COST } : model.cost,
|
||||
contextWindow: model.contextWindow ?? MOONSHOT_KIMI_K3_CONTEXT_WINDOW,
|
||||
maxTokens: model.maxTokens ?? MOONSHOT_KIMI_K3_MAX_TOKENS,
|
||||
thinking: model.thinking ?? { ...MOONSHOT_KIMI_K3_THINKING },
|
||||
};
|
||||
}
|
||||
// Moonshot's K2.x family (K2.5, K2.6, kimi-k2-thinking, …) is reasoning-capable
|
||||
// and vision-capable on the native API. Without these flags the openai-completions
|
||||
// path skips the z.ai-format `thinking` block, and Moonshot K2.6 stalls on first
|
||||
|
||||
@@ -0,0 +1,159 @@
|
||||
/**
|
||||
* Issue #5756 — `moonshot/kimi-k3 is incorrectly shown as free`
|
||||
*
|
||||
* The native Moonshot `kimi-k3` entry is dynamically discovered but has no
|
||||
* bundled/models.dev reference, so `mapWithBundledReference` produced the
|
||||
* generic dynamic defaults: zero token cost, null limits, text-only input,
|
||||
* and `reasoning: false`. `/models` then labeled the paid model "Free".
|
||||
*
|
||||
* The fix stamps Moonshot's official K3 pricing/limits, marks it reasoning +
|
||||
* vision, and routes reasoning through OpenAI-style `reasoning_effort: "max"`
|
||||
* (K3 does NOT accept the K2.x binary `thinking: { type }` block).
|
||||
*/
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import type { Context } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import { moonshotModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
|
||||
import type { ModelSpec } from "@oh-my-pi/pi-catalog/types";
|
||||
|
||||
function moonshotModelsResponse(): Response {
|
||||
const body = {
|
||||
object: "list",
|
||||
data: [
|
||||
{ id: "kimi-k3", object: "model", owned_by: "moonshot" },
|
||||
{ id: "kimi-k2.6", object: "model", owned_by: "moonshot" },
|
||||
],
|
||||
};
|
||||
return new Response(JSON.stringify(body), {
|
||||
status: 200,
|
||||
headers: { "content-type": "application/json" },
|
||||
});
|
||||
}
|
||||
|
||||
async function discoverKimiK3(): Promise<ModelSpec<"openai-completions">> {
|
||||
const fetchMock = (async (_input: string | URL | Request): Promise<Response> =>
|
||||
moonshotModelsResponse()) as typeof fetch;
|
||||
const models = await moonshotModelManagerOptions({ apiKey: "test-key", fetch: fetchMock }).fetchDynamicModels?.();
|
||||
const k3 = models?.find(m => m.id === "kimi-k3");
|
||||
if (!k3) throw new Error("kimi-k3 not discovered");
|
||||
return k3;
|
||||
}
|
||||
|
||||
function encodeSseChunks(chunks: ReadonlyArray<Record<string, unknown>>): string {
|
||||
return `${chunks.map(c => `data: ${JSON.stringify(c)}\n\n`).join("")}data: [DONE]\n\n`;
|
||||
}
|
||||
|
||||
describe("issue #5756 — moonshot kimi-k3 pricing and wire format", () => {
|
||||
it("discovery mapper stamps K3 pricing, limits, vision, and reasoning", async () => {
|
||||
const k3 = await discoverKimiK3();
|
||||
expect(k3.cost).toEqual({ input: 3, output: 15, cacheRead: 0.3, cacheWrite: 0 });
|
||||
expect(k3.contextWindow).toBe(1_048_576);
|
||||
expect(k3.maxTokens).toBe(131_072);
|
||||
expect(k3.input).toEqual(["text", "image"]);
|
||||
expect(k3.reasoning).toBe(true);
|
||||
expect(k3.thinking).toEqual({ mode: "effort", efforts: [Effort.Max], requiresEffort: true });
|
||||
});
|
||||
|
||||
it("K3 native compat uses the OpenAI reasoning_effort dialect, not the K2 thinking block", async () => {
|
||||
const model = buildModel(await discoverKimiK3());
|
||||
expect(model.compat.thinkingFormat).toBe("openai");
|
||||
expect(model.compat.reasoningDisableMode).toBe("lowest-effort");
|
||||
expect(model.compat.supportsReasoningEffort).toBe(true);
|
||||
});
|
||||
|
||||
it("wire body carries reasoning_effort=max and omits the thinking block", async () => {
|
||||
const model = buildModel(await discoverKimiK3());
|
||||
let body: Record<string, unknown> = {};
|
||||
const fetchMock = (async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
||||
const raw = typeof init?.body === "string" ? init.body : "";
|
||||
body = raw ? (JSON.parse(raw) as Record<string, unknown>) : {};
|
||||
return new Response(
|
||||
encodeSseChunks([
|
||||
{ choices: [{ index: 0, delta: { role: "assistant", content: "hi" }, finish_reason: null }] },
|
||||
{
|
||||
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
|
||||
usage: { prompt_tokens: 1, completion_tokens: 1 },
|
||||
},
|
||||
]),
|
||||
{ status: 200, headers: { "content-type": "text/event-stream" } },
|
||||
);
|
||||
}) as typeof fetch;
|
||||
|
||||
const context: Context = {
|
||||
messages: [{ role: "user", content: "hi", timestamp: Date.now() }],
|
||||
};
|
||||
const stream = streamOpenAICompletions(model, context, {
|
||||
apiKey: "test-key",
|
||||
reasoning: "max",
|
||||
fetch: fetchMock,
|
||||
});
|
||||
for await (const _ of stream) {
|
||||
// drain
|
||||
}
|
||||
|
||||
expect(body.reasoning_effort).toBe("max");
|
||||
expect("thinking" in body).toBe(false);
|
||||
// Moonshot-native Kimi rate-limits on max_tokens, not
|
||||
// max_completion_tokens; K3's default reaches its advertised 131K cap.
|
||||
expect(body.max_tokens).toBe(131_072);
|
||||
expect(body.max_completion_tokens).toBeUndefined();
|
||||
});
|
||||
|
||||
it("keeps reasoning_effort=max on forced-tool-choice turns (mandatory K3 reasoning)", async () => {
|
||||
// K3 always reasons via `reasoning_effort: "max"`. The K2.x Kimi
|
||||
// `disableReasoningOnForcedToolChoice` rule (Moonshot 400s on forced
|
||||
// tool_choice + the binary `thinking` block, #827) must NOT strip K3's
|
||||
// effort, or plan-mode `toolChoice` turns run without the required
|
||||
// reasoning (#5758 review).
|
||||
const model = buildModel(await discoverKimiK3());
|
||||
let body: Record<string, unknown> = {};
|
||||
const fetchMock = (async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
||||
const raw = typeof init?.body === "string" ? init.body : "";
|
||||
body = raw ? (JSON.parse(raw) as Record<string, unknown>) : {};
|
||||
return new Response(
|
||||
encodeSseChunks([
|
||||
{
|
||||
choices: [
|
||||
{
|
||||
index: 0,
|
||||
delta: {
|
||||
role: "assistant",
|
||||
tool_calls: [
|
||||
{ index: 0, id: "c1", type: "function", function: { name: "plan", arguments: "{}" } },
|
||||
],
|
||||
},
|
||||
finish_reason: null,
|
||||
},
|
||||
],
|
||||
},
|
||||
{
|
||||
choices: [{ index: 0, delta: {}, finish_reason: "tool_calls" }],
|
||||
usage: { prompt_tokens: 1, completion_tokens: 1 },
|
||||
},
|
||||
]),
|
||||
{ status: 200, headers: { "content-type": "text/event-stream" } },
|
||||
);
|
||||
}) as typeof fetch;
|
||||
|
||||
const context: Context = {
|
||||
messages: [{ role: "user", content: "hi", timestamp: Date.now() }],
|
||||
tools: [{ name: "plan", description: "plan", parameters: { type: "object", properties: {} } }],
|
||||
};
|
||||
const stream = streamOpenAICompletions(model, context, {
|
||||
apiKey: "test-key",
|
||||
reasoning: "max",
|
||||
maxTokens: 131_072,
|
||||
toolChoice: { type: "tool", name: "plan" },
|
||||
fetch: fetchMock,
|
||||
});
|
||||
for await (const _ of stream) {
|
||||
// drain
|
||||
}
|
||||
|
||||
expect(body.reasoning_effort).toBe("max");
|
||||
expect("thinking" in body).toBe(false);
|
||||
expect(body.max_tokens).toBe(131_072);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user