fix(catalog): extend stream idle floor to kimi k2.7 code

Native Kimi K2.7 Code (kimi-k2.7-code / kimi-k2.7-code-highspeed) reasons
for minutes before the first stream event like K2.6, but the
streamIdleTimeoutMs branch in buildOpenAICompat gated only on
isKimiK26ModelId, so K2.7 Code fell through to the 120s default and
aborted on long reasoning turns. Match matchesKimiK27CodeFamily in the
same branch and rename the constant to KIMI_REASONING_STREAM_IDLE_TIMEOUT_MS.

Fixes #4836
This commit is contained in:
roboomp
2026-07-14 16:39:48 +00:00
parent ac2ea80fa3
commit 29c8cae9ba
3 changed files with 29 additions and 4 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Extended the reasoning `streamIdleTimeoutMs` floor (300s) to native Kimi K2.7 Code (`kimi-k2.7-code` / `kimi-k2.7-code-highspeed`), which previously fell through to the 120s default and aborted on long reasoning turns ([#4836](https://github.com/can1357/oh-my-pi/issues/4836)).
## [16.3.11] - 2026-07-06
### Added
+4 -4
View File
@@ -37,8 +37,8 @@ const GLM_CODING_PLAN_MODEL_PATTERN = /(^|\/)glm-5(?:[.-]|$)/i;
const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000;
/** Direct DeepSeek reasoning models stall between thinking and answer phases. */
const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
/** Kimi K2.6 can spend several minutes reasoning before the first visible token. */
const KIMI_K26_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
/** Kimi K2.6 and native K2.7 Code can spend several minutes reasoning before the first visible token. */
const KIMI_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
/**
* Native Kimi K2.7 Code requires `thinking.type: "enabled"` and rejects
* disabled thinking. Match the public id, its Fast variant, and the
@@ -383,8 +383,8 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
? ALIBABA_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS
: isXiaomiMimo
? XIAOMI_MIMO_STREAM_IDLE_TIMEOUT_MS
: spec.reasoning && isKimiK26ModelId(spec.id)
? KIMI_K26_REASONING_STREAM_IDLE_TIMEOUT_MS
: spec.reasoning && (isKimiK26ModelId(spec.id) || matchesKimiK27CodeFamily(spec))
? KIMI_REASONING_STREAM_IDLE_TIMEOUT_MS
: spec.reasoning && isDirectDeepseekApi
? DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS
: isLocalOpenAICompatBackend
+21
View File
@@ -282,6 +282,27 @@ describe("openai-completions wire-quirk compat detection", () => {
expect(buildOpenAICompat(completionsSpec()).reasoningDeltasMayBeCumulative).toBe(false);
});
it("extends the reasoning stream idle floor to Kimi K2.6 and K2.7 Code, not other reasoning models", () => {
const kimiOverrides = {
provider: "moonshot",
baseUrl: "https://api.moonshot.ai/v1",
reasoning: true,
} as const;
expect(buildOpenAICompat(completionsSpec({ ...kimiOverrides, id: "kimi-k2.6" })).streamIdleTimeoutMs).toBe(
300_000,
);
expect(buildOpenAICompat(completionsSpec({ ...kimiOverrides, id: "kimi-k2.7-code" })).streamIdleTimeoutMs).toBe(
300_000,
);
expect(
buildOpenAICompat(completionsSpec({ ...kimiOverrides, id: "kimi-k2.7-code-highspeed" })).streamIdleTimeoutMs,
).toBe(300_000);
// A non-Kimi reasoning model on a generic host keeps the runtime default.
expect(
buildOpenAICompat(completionsSpec({ id: "some-reasoner", reasoning: true })).streamIdleTimeoutMs,
).toBeUndefined();
});
it("maps the remaining provider-keyed wire quirks", () => {
expect(buildOpenAICompat(completionsSpec({ provider: "ollama" })).emptyLengthFinishIsContextError).toBe(true);
expect(buildOpenAICompat(completionsSpec()).emptyLengthFinishIsContextError).toBe(false);