fix(ai): fixed Kimi regex to match path-prefixed router model IDs
- Changed `^kimi[-.]/i` to `(^|\/)kimi[-.]/i` so Fireworks router IDs like `accounts/fireworks/routers/kimi-k2p6-turbo` correctly trigger Kimi-specific max_tokens behavior. - Applied the same fix in both the compat detector and buildParams. - Added a regression test covering the canonical Fire Pass router ID format.
This commit is contained in:
@@ -52,7 +52,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
const isCerebras = provider === "cerebras" || baseUrl.includes("cerebras.ai");
|
||||
const isZai = provider === "zai" || baseUrl.includes("api.z.ai");
|
||||
const isKilo = provider === "kilo" || baseUrl.includes("api.kilo.ai");
|
||||
const isKimiModel = model.id.includes("moonshotai/kimi") || /^kimi[-.]/i.test(model.id);
|
||||
const isKimiModel = model.id.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(model.id);
|
||||
const isMoonshotKimi =
|
||||
isKimiModel &&
|
||||
(provider === "moonshot" ||
|
||||
|
||||
@@ -1038,13 +1038,15 @@ function buildParams(
|
||||
maybeAddOpenRouterAnthropicCacheControl(model, messages);
|
||||
const supportsReasoningParams = model.provider !== "github-copilot";
|
||||
|
||||
// Kimi (including via OpenRouter and the Fireworks `accounts/fireworks/routers/kimi-*` family)
|
||||
// calculates TPM rate limits based on max_tokens, not actual output. The official Kimi K2 model
|
||||
// guidance (https://docs.fireworks.ai/models/kimi-k2) also requires `max_tokens` for every call
|
||||
// since the family can otherwise emit very long reasoning traces before the final answer.
|
||||
// Always send max_tokens — match the same Kimi-family regex used by the compat detector.
|
||||
// Kimi (including via OpenRouter and Fireworks router-form IDs such as
|
||||
// `accounts/fireworks/routers/kimi-*`) calculates TPM rate limits based on
|
||||
// max_tokens, not actual output. The official Kimi K2 model guidance
|
||||
// (https://docs.fireworks.ai/models/kimi-k2) also requires `max_tokens` for
|
||||
// every call since the family can otherwise emit very long reasoning traces
|
||||
// before the final answer. Always send max_tokens — match the same
|
||||
// Kimi-family regex used by the compat detector.
|
||||
// Note: Direct kimi-code provider is handled by the dedicated Kimi provider in kimi.ts.
|
||||
const isKimi = model.id.includes("moonshotai/kimi") || /^kimi[-.]/i.test(model.id);
|
||||
const isKimi = model.id.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(model.id);
|
||||
const effectiveMaxTokens = options?.maxTokens ?? (isKimi ? model.maxTokens : undefined);
|
||||
|
||||
const requestModelId =
|
||||
|
||||
@@ -135,4 +135,37 @@ describe("Fire Pass provider", () => {
|
||||
const parsed = JSON.parse(captured.body ?? "{}") as { max_tokens?: unknown };
|
||||
expect(parsed.max_tokens).toBe(model.maxTokens);
|
||||
});
|
||||
|
||||
it("applies the Kimi max_tokens default to canonical Fire Pass router ids", async () => {
|
||||
const bundled = getBundledModel<"openai-completions">("firepass", "kimi-k2.6-turbo");
|
||||
const model: Model<"openai-completions"> = {
|
||||
...bundled,
|
||||
id: "accounts/fireworks/routers/kimi-k2p6-turbo",
|
||||
};
|
||||
const captured: { body: string | null } = { body: null };
|
||||
global.fetch = (async (_input: unknown, init?: RequestInit) => {
|
||||
captured.body = typeof init?.body === "string" ? init.body : null;
|
||||
return sseResponse([
|
||||
{ choices: [{ delta: { content: "ok" }, index: 0 }] },
|
||||
{ choices: [{ delta: {}, finish_reason: "stop", index: 0 }] },
|
||||
"[DONE]",
|
||||
]);
|
||||
}) as typeof global.fetch;
|
||||
|
||||
const context: Context = {
|
||||
systemPrompt: [],
|
||||
messages: [{ role: "user", content: "ping", timestamp: Date.now() }],
|
||||
};
|
||||
const stream = streamOpenAICompletions(model, context, {
|
||||
apiKey: "fpk_test",
|
||||
});
|
||||
for await (const _event of stream) {
|
||||
/* drain */
|
||||
}
|
||||
|
||||
expect(captured.body).not.toBeNull();
|
||||
const parsed = JSON.parse(captured.body ?? "{}") as { max_tokens?: unknown; model?: unknown };
|
||||
expect(parsed.model).toBe("accounts/fireworks/routers/kimi-k2p6-turbo");
|
||||
expect(parsed.max_tokens).toBe(model.maxTokens);
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user