Merge remote-tracking branch 'origin/farm/3f9eb0f8/widen-first-event-timeout-deepseek-v4-reasoning'

This commit is contained in:
can1357
2026-06-09 19:19:07 +02:00
3 changed files with 82 additions and 7 deletions
+4
View File
@@ -10,6 +10,10 @@
- Added `antigravityRankingStrategy` and registered it as the default `CredentialRankingStrategy` for `google-antigravity`, so multi-account selection consumes the per-counter Antigravity usage reports (sorted ascending by `remainingFraction` in `fetchAntigravityUsage`) before falling back to round-robin — preventing the exhausted-counter credential from being chosen first when an unblocked sibling has headroom ([#2187](https://github.com/can1357/oh-my-pi/issues/2187)).
### Fixed
- Widened the OpenAI-completions first-event watchdog floor from 120s to 300s for DeepSeek V4 reasoning models hosted on the official DeepSeek API. The reasoner emits no SSE bytes until its private chain-of-thought finishes, which routinely takes longer than the generic 100s first-event budget under load — every chat then aborted with `OpenAI completions stream timed out while waiting for the first event` and silently retried. Mirrors the existing GLM coding-plan widening ([#2177](https://github.com/can1357/oh-my-pi/issues/2177)).
## [15.10.8] - 2026-06-09
### Added
@@ -392,17 +392,36 @@ const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE =
const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000;
const GLM_CODING_PLAN_MODEL_PATTERN = /^glm-5(?:[.-]|$)/i;
/** Returns the widened OpenAI stream watchdog floor for slow GLM coding-plan reasoning models. */
// DeepSeek V4 reasoning models on the official api.deepseek.com emit no SSE
// bytes while the model finishes its private chain-of-thought, which routinely
// takes longer than the generic 100s first-event floor under load (issue
// #2177). Mirror the GLM coding-plan widening: a 5-minute idle floor lifts the
// first-event watchdog (it floors at idle) without changing the runtime
// streaming behavior, so reasoning warm-ups stop aborting and retrying.
const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
function isDirectDeepseekReasoningModel(model: Model<"openai-completions">): boolean {
if (!model.reasoning) return false;
if (model.provider === "deepseek") return true;
return model.baseUrl.toLowerCase().includes("api.deepseek.com");
}
/** Returns the widened OpenAI stream watchdog floor for slow reasoning models hosted on OpenAI-compatible endpoints. */
export function getOpenAICompletionsStreamIdleTimeoutFallbackMs(
model: Model<"openai-completions">,
): number | undefined {
if (!GLM_CODING_PLAN_MODEL_PATTERN.test(model.id)) return undefined;
if (model.provider === "zhipu-coding-plan" || model.provider === "zai")
return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS;
if (GLM_CODING_PLAN_MODEL_PATTERN.test(model.id)) {
if (model.provider === "zhipu-coding-plan" || model.provider === "zai")
return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS;
const baseUrl = model.baseUrl.toLowerCase();
if (baseUrl.includes("open.bigmodel.cn") || baseUrl.includes("api.z.ai")) {
return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS;
const baseUrl = model.baseUrl.toLowerCase();
if (baseUrl.includes("open.bigmodel.cn") || baseUrl.includes("api.z.ai")) {
return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS;
}
}
if (isDirectDeepseekReasoningModel(model)) {
return DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS;
}
return undefined;
@@ -103,6 +103,58 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(600_000);
});
it("widens DeepSeek V4 reasoning streams on the official DeepSeek API", () => {
const model = {
...openAICompletionsModel,
id: "deepseek-v4-pro",
name: "DeepSeek V4 Pro",
provider: "deepseek",
baseUrl: "https://api.deepseek.com",
reasoning: true,
} satisfies Model<"openai-completions">;
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(300_000);
});
it("widens DeepSeek reasoning streams routed through an aliased OpenAI-compatible provider id", () => {
const model = {
...openAICompletionsModel,
id: "deepseek-v4-pro",
name: "DeepSeek V4 Pro",
provider: "openai",
baseUrl: "https://api.deepseek.com/v1",
reasoning: true,
} satisfies Model<"openai-completions">;
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(300_000);
});
it("leaves non-reasoning DeepSeek-hosted models on the global timeout", () => {
const model = {
...openAICompletionsModel,
id: "deepseek-chat",
name: "DeepSeek Chat",
provider: "deepseek",
baseUrl: "https://api.deepseek.com",
reasoning: false,
} satisfies Model<"openai-completions">;
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBeUndefined();
});
it("does not widen DeepSeek V4 reasoning models hosted on third-party OpenAI-compatible proxies", () => {
const model = {
...openAICompletionsModel,
id: "deepseek-v4-pro",
name: "DeepSeek V4 Pro",
provider: "aimlapi",
baseUrl: "https://api.aimlapi.com/v1",
reasoning: true,
} satisfies Model<"openai-completions">;
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBeUndefined();
});
it("keeps ordinary OpenAI-compatible models on the global timeout", () => {
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(openAICompletionsModel)).toBeUndefined();
});