fix(ai): fixed Kimi regex to match path-prefixed router model IDs

- Changed `^kimi[-.]/i` to `(^|\/)kimi[-.]/i` so Fireworks router IDs like `accounts/fireworks/routers/kimi-k2p6-turbo` correctly trigger Kimi-specific max_tokens behavior.
- Applied the same fix in both the compat detector and buildParams.
- Added a regression test covering the canonical Fire Pass router ID format.
This commit is contained in:
Can Bölük
2026-05-20 17:29:59 +09:00
parent d097aefc65
commit b8584a6d88
3 changed files with 42 additions and 7 deletions
@@ -52,7 +52,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
const isCerebras = provider === "cerebras" || baseUrl.includes("cerebras.ai");
const isZai = provider === "zai" || baseUrl.includes("api.z.ai");
const isKilo = provider === "kilo" || baseUrl.includes("api.kilo.ai");
const isKimiModel = model.id.includes("moonshotai/kimi") || /^kimi[-.]/i.test(model.id);
const isKimiModel = model.id.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(model.id);
const isMoonshotKimi =
isKimiModel &&
(provider === "moonshot" ||
@@ -1038,13 +1038,15 @@ function buildParams(
maybeAddOpenRouterAnthropicCacheControl(model, messages);
const supportsReasoningParams = model.provider !== "github-copilot";
// Kimi (including via OpenRouter and the Fireworks `accounts/fireworks/routers/kimi-*` family)
// calculates TPM rate limits based on max_tokens, not actual output. The official Kimi K2 model
// guidance (https://docs.fireworks.ai/models/kimi-k2) also requires `max_tokens` for every call
// since the family can otherwise emit very long reasoning traces before the final answer.
// Always send max_tokens — match the same Kimi-family regex used by the compat detector.
// Kimi (including via OpenRouter and Fireworks router-form IDs such as
// `accounts/fireworks/routers/kimi-*`) calculates TPM rate limits based on
// max_tokens, not actual output. The official Kimi K2 model guidance
// (https://docs.fireworks.ai/models/kimi-k2) also requires `max_tokens` for
// every call since the family can otherwise emit very long reasoning traces
// before the final answer. Always send max_tokens — match the same
// Kimi-family regex used by the compat detector.
// Note: Direct kimi-code provider is handled by the dedicated Kimi provider in kimi.ts.
const isKimi = model.id.includes("moonshotai/kimi") || /^kimi[-.]/i.test(model.id);
const isKimi = model.id.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(model.id);
const effectiveMaxTokens = options?.maxTokens ?? (isKimi ? model.maxTokens : undefined);
const requestModelId =
+33
View File
@@ -135,4 +135,37 @@ describe("Fire Pass provider", () => {
const parsed = JSON.parse(captured.body ?? "{}") as { max_tokens?: unknown };
expect(parsed.max_tokens).toBe(model.maxTokens);
});
it("applies the Kimi max_tokens default to canonical Fire Pass router ids", async () => {
const bundled = getBundledModel<"openai-completions">("firepass", "kimi-k2.6-turbo");
const model: Model<"openai-completions"> = {
...bundled,
id: "accounts/fireworks/routers/kimi-k2p6-turbo",
};
const captured: { body: string | null } = { body: null };
global.fetch = (async (_input: unknown, init?: RequestInit) => {
captured.body = typeof init?.body === "string" ? init.body : null;
return sseResponse([
{ choices: [{ delta: { content: "ok" }, index: 0 }] },
{ choices: [{ delta: {}, finish_reason: "stop", index: 0 }] },
"[DONE]",
]);
}) as typeof global.fetch;
const context: Context = {
systemPrompt: [],
messages: [{ role: "user", content: "ping", timestamp: Date.now() }],
};
const stream = streamOpenAICompletions(model, context, {
apiKey: "fpk_test",
});
for await (const _event of stream) {
/* drain */
}
expect(captured.body).not.toBeNull();
const parsed = JSON.parse(captured.body ?? "{}") as { max_tokens?: unknown; model?: unknown };
expect(parsed.model).toBe("accounts/fireworks/routers/kimi-k2p6-turbo");
expect(parsed.max_tokens).toBe(model.maxTokens);
});
});