From f5f4f2265b632f7bcccf4987d0b504c018d0d6fc Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 9 Jun 2026 09:16:41 +0000 Subject: [PATCH] fix(ai): widen first-event watchdog for DeepSeek V4 reasoning MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit DeepSeek V4 reasoning models on api.deepseek.com emit no SSE bytes while the model finishes its private chain-of-thought, which routinely takes longer than the generic 100s first-event budget under load. The OpenAI completions stream then aborts with "OpenAI completions stream timed out while waiting for the first event" and silently retries, doubling user-visible latency on almost every chat. getOpenAICompletionsStreamIdleTimeoutFallbackMs now returns a 300s floor for reasoning models when (provider === "deepseek") or (baseUrl includes api.deepseek.com). first-event floors at idle, so the watchdog gains 5 minutes — enough for reasoning warm-ups — without changing steady-state streaming behavior. Mirrors the existing GLM coding-plan widening. Fixes #2177 --- packages/ai/CHANGELOG.md | 4 ++ .../ai/src/providers/openai-completions.ts | 33 +++++++++--- .../openai-completions-progress-chunk.test.ts | 52 +++++++++++++++++++ 3 files changed, 82 insertions(+), 7 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 16e224a18..88f20073a 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Widened the OpenAI-completions first-event watchdog floor from 120s to 300s for DeepSeek V4 reasoning models hosted on the official DeepSeek API. The reasoner emits no SSE bytes until its private chain-of-thought finishes, which routinely takes longer than the generic 100s first-event budget under load — every chat then aborted with `OpenAI completions stream timed out while waiting for the first event` and silently retried. Mirrors the existing GLM coding-plan widening ([#2177](https://github.com/can1357/oh-my-pi/issues/2177)). + ## [15.10.8] - 2026-06-09 ### Added diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index 808077012..4072174c0 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -392,17 +392,36 @@ const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE = const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000; const GLM_CODING_PLAN_MODEL_PATTERN = /^glm-5(?:[.-]|$)/i; -/** Returns the widened OpenAI stream watchdog floor for slow GLM coding-plan reasoning models. */ +// DeepSeek V4 reasoning models on the official api.deepseek.com emit no SSE +// bytes while the model finishes its private chain-of-thought, which routinely +// takes longer than the generic 100s first-event floor under load (issue +// #2177). Mirror the GLM coding-plan widening: a 5-minute idle floor lifts the +// first-event watchdog (it floors at idle) without changing the runtime +// streaming behavior, so reasoning warm-ups stop aborting and retrying. +const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; + +function isDirectDeepseekReasoningModel(model: Model<"openai-completions">): boolean { + if (!model.reasoning) return false; + if (model.provider === "deepseek") return true; + return model.baseUrl.toLowerCase().includes("api.deepseek.com"); +} + +/** Returns the widened OpenAI stream watchdog floor for slow reasoning models hosted on OpenAI-compatible endpoints. */ export function getOpenAICompletionsStreamIdleTimeoutFallbackMs( model: Model<"openai-completions">, ): number | undefined { - if (!GLM_CODING_PLAN_MODEL_PATTERN.test(model.id)) return undefined; - if (model.provider === "zhipu-coding-plan" || model.provider === "zai") - return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS; + if (GLM_CODING_PLAN_MODEL_PATTERN.test(model.id)) { + if (model.provider === "zhipu-coding-plan" || model.provider === "zai") + return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS; - const baseUrl = model.baseUrl.toLowerCase(); - if (baseUrl.includes("open.bigmodel.cn") || baseUrl.includes("api.z.ai")) { - return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS; + const baseUrl = model.baseUrl.toLowerCase(); + if (baseUrl.includes("open.bigmodel.cn") || baseUrl.includes("api.z.ai")) { + return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS; + } + } + + if (isDirectDeepseekReasoningModel(model)) { + return DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS; } return undefined; diff --git a/packages/ai/test/openai-completions-progress-chunk.test.ts b/packages/ai/test/openai-completions-progress-chunk.test.ts index 7e92a8990..933f80717 100644 --- a/packages/ai/test/openai-completions-progress-chunk.test.ts +++ b/packages/ai/test/openai-completions-progress-chunk.test.ts @@ -103,6 +103,58 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => { expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(600_000); }); + it("widens DeepSeek V4 reasoning streams on the official DeepSeek API", () => { + const model = { + ...openAICompletionsModel, + id: "deepseek-v4-pro", + name: "DeepSeek V4 Pro", + provider: "deepseek", + baseUrl: "https://api.deepseek.com", + reasoning: true, + } satisfies Model<"openai-completions">; + + expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(300_000); + }); + + it("widens DeepSeek reasoning streams routed through an aliased OpenAI-compatible provider id", () => { + const model = { + ...openAICompletionsModel, + id: "deepseek-v4-pro", + name: "DeepSeek V4 Pro", + provider: "openai", + baseUrl: "https://api.deepseek.com/v1", + reasoning: true, + } satisfies Model<"openai-completions">; + + expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(300_000); + }); + + it("leaves non-reasoning DeepSeek-hosted models on the global timeout", () => { + const model = { + ...openAICompletionsModel, + id: "deepseek-chat", + name: "DeepSeek Chat", + provider: "deepseek", + baseUrl: "https://api.deepseek.com", + reasoning: false, + } satisfies Model<"openai-completions">; + + expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBeUndefined(); + }); + + it("does not widen DeepSeek V4 reasoning models hosted on third-party OpenAI-compatible proxies", () => { + const model = { + ...openAICompletionsModel, + id: "deepseek-v4-pro", + name: "DeepSeek V4 Pro", + provider: "aimlapi", + baseUrl: "https://api.aimlapi.com/v1", + reasoning: true, + } satisfies Model<"openai-completions">; + + expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBeUndefined(); + }); + it("keeps ordinary OpenAI-compatible models on the global timeout", () => { expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(openAICompletionsModel)).toBeUndefined(); });