fix(catalog): widened kimi k2.6 stream watchdog

Widened Kimi K2.6 OpenAI-compatible stream watchdog defaults so long reasoning starts do not hit the generic first-event timeout. Covered Fire Pass public and router ids with a regression test.\n\nFixes #2366
This commit is contained in:
roboomp
2026-06-12 07:17:03 +00:00
parent df49125296
commit bc8d3a5bc7
4 changed files with 25 additions and 8 deletions
@@ -133,6 +133,18 @@ describe("resolveOpenAICompat stream idle timeout", () => {
expect(model.compat.streamIdleTimeoutMs).toBe(300_000);
});
it("widens Kimi K2.6 reasoning streams across OpenAI-compatible hosts", () => {
const bundled = getBundledModel<"openai-completions">("firepass", "kimi-k2.6-turbo");
const canonicalRouter = buildModel({
...bundled,
id: "accounts/fireworks/routers/kimi-k2p6-turbo",
compat: bundled.compatConfig,
} as ModelSpec<"openai-completions">);
expect(bundled.compat.streamIdleTimeoutMs).toBe(300_000);
expect(canonicalRouter.compat.streamIdleTimeoutMs).toBe(300_000);
});
it("leaves non-reasoning DeepSeek-hosted models on the global timeout", () => {
const model = buildModel({
...openAICompletionsModel,
+1
View File
@@ -18,6 +18,7 @@
### Fixed
- Fixed catalog generation to apply effort-tier variant collapsing before provider grouping to ensure collapsed model families are consistently materialized without being impacted by in-loop mutation
- Fixed Kimi K2.6 OpenAI-compatible compat metadata to use a 300s stream watchdog floor, covering Fire Pass router ids as well as public `kimi-k2.6` ids so long reasoning starts do not hit the generic first-event timeout ([#2366](https://github.com/can1357/oh-my-pi/issues/2366)).
## [15.11.4] - 2026-06-12
+10 -6
View File
@@ -25,6 +25,8 @@ const GLM_CODING_PLAN_MODEL_PATTERN = /^glm-5(?:[.-]|$)/i;
const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000;
/** Direct DeepSeek reasoning models stall between thinking and answer phases. */
const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
/** Kimi K2.6 can spend several minutes reasoning before the first visible token. */
const KIMI_K26_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
/**
* OpenCode's gateways (https://opencode.ai/zen|go) gate `reasoning_content`
@@ -178,15 +180,17 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
isCopilotHost ||
isZenmuxHost);
// Stream-watchdog floor: GLM coding-plan SKUs and direct DeepSeek reasoning
// models idle for minutes mid-reasoning; widen the idle timeout so warm-ups
// stop aborting and retrying.
// Stream-watchdog floor: GLM coding-plan SKUs, Kimi K2.6, and direct
// DeepSeek reasoning models can idle for minutes while reasoning; widen the
// idle timeout so warm-ups stop aborting and retrying.
const streamIdleTimeoutMs =
GLM_CODING_PLAN_MODEL_PATTERN.test(spec.id) && (isZai || isZhipu)
? GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS
: spec.reasoning && isDirectDeepseekApi
? DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS
: undefined;
: spec.reasoning && isKimiK26ModelId(spec.id)
? KIMI_K26_REASONING_STREAM_IDLE_TIMEOUT_MS
: spec.reasoning && isDirectDeepseekApi
? DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS
: undefined;
const compat: ResolvedOpenAICompat = {
supportsStore: !isNonStandard,
+2 -2
View File
@@ -14,9 +14,9 @@ export function isKimiModelId(modelId: string): boolean {
return modelId.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(modelId);
}
/** Kimi K2.6 specifically (preserved-thinking transport on Moonshot-native hosts). */
/** Kimi K2.6 specifically, including router ids that spell the version `k2p6`. */
export function isKimiK26ModelId(modelId: string): boolean {
return /(^|\/)kimi-k2\.6(?:[-:]|$)/i.test(modelId);
return /(^|\/)kimi-k2(?:\.6|p6)(?:[-:]|$)/i.test(modelId);
}
/** Claude ids in any namespace form (`claude-*`, `vendor/claude.x`). */