Merge remote-tracking branch 'origin/farm/3f9eb0f8/widen-first-event-timeout-deepseek-v4-reasoning'
This commit is contained in:
@@ -10,6 +10,10 @@
|
||||
|
||||
- Added `antigravityRankingStrategy` and registered it as the default `CredentialRankingStrategy` for `google-antigravity`, so multi-account selection consumes the per-counter Antigravity usage reports (sorted ascending by `remainingFraction` in `fetchAntigravityUsage`) before falling back to round-robin — preventing the exhausted-counter credential from being chosen first when an unblocked sibling has headroom ([#2187](https://github.com/can1357/oh-my-pi/issues/2187)).
|
||||
|
||||
### Fixed
|
||||
|
||||
- Widened the OpenAI-completions first-event watchdog floor from 120s to 300s for DeepSeek V4 reasoning models hosted on the official DeepSeek API. The reasoner emits no SSE bytes until its private chain-of-thought finishes, which routinely takes longer than the generic 100s first-event budget under load — every chat then aborted with `OpenAI completions stream timed out while waiting for the first event` and silently retried. Mirrors the existing GLM coding-plan widening ([#2177](https://github.com/can1357/oh-my-pi/issues/2177)).
|
||||
|
||||
## [15.10.8] - 2026-06-09
|
||||
### Added
|
||||
|
||||
|
||||
@@ -392,17 +392,36 @@ const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE =
|
||||
const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000;
|
||||
const GLM_CODING_PLAN_MODEL_PATTERN = /^glm-5(?:[.-]|$)/i;
|
||||
|
||||
/** Returns the widened OpenAI stream watchdog floor for slow GLM coding-plan reasoning models. */
|
||||
// DeepSeek V4 reasoning models on the official api.deepseek.com emit no SSE
|
||||
// bytes while the model finishes its private chain-of-thought, which routinely
|
||||
// takes longer than the generic 100s first-event floor under load (issue
|
||||
// #2177). Mirror the GLM coding-plan widening: a 5-minute idle floor lifts the
|
||||
// first-event watchdog (it floors at idle) without changing the runtime
|
||||
// streaming behavior, so reasoning warm-ups stop aborting and retrying.
|
||||
const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
|
||||
|
||||
function isDirectDeepseekReasoningModel(model: Model<"openai-completions">): boolean {
|
||||
if (!model.reasoning) return false;
|
||||
if (model.provider === "deepseek") return true;
|
||||
return model.baseUrl.toLowerCase().includes("api.deepseek.com");
|
||||
}
|
||||
|
||||
/** Returns the widened OpenAI stream watchdog floor for slow reasoning models hosted on OpenAI-compatible endpoints. */
|
||||
export function getOpenAICompletionsStreamIdleTimeoutFallbackMs(
|
||||
model: Model<"openai-completions">,
|
||||
): number | undefined {
|
||||
if (!GLM_CODING_PLAN_MODEL_PATTERN.test(model.id)) return undefined;
|
||||
if (model.provider === "zhipu-coding-plan" || model.provider === "zai")
|
||||
return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS;
|
||||
if (GLM_CODING_PLAN_MODEL_PATTERN.test(model.id)) {
|
||||
if (model.provider === "zhipu-coding-plan" || model.provider === "zai")
|
||||
return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS;
|
||||
|
||||
const baseUrl = model.baseUrl.toLowerCase();
|
||||
if (baseUrl.includes("open.bigmodel.cn") || baseUrl.includes("api.z.ai")) {
|
||||
return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS;
|
||||
const baseUrl = model.baseUrl.toLowerCase();
|
||||
if (baseUrl.includes("open.bigmodel.cn") || baseUrl.includes("api.z.ai")) {
|
||||
return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS;
|
||||
}
|
||||
}
|
||||
|
||||
if (isDirectDeepseekReasoningModel(model)) {
|
||||
return DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS;
|
||||
}
|
||||
|
||||
return undefined;
|
||||
|
||||
@@ -103,6 +103,58 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(600_000);
|
||||
});
|
||||
|
||||
it("widens DeepSeek V4 reasoning streams on the official DeepSeek API", () => {
|
||||
const model = {
|
||||
...openAICompletionsModel,
|
||||
id: "deepseek-v4-pro",
|
||||
name: "DeepSeek V4 Pro",
|
||||
provider: "deepseek",
|
||||
baseUrl: "https://api.deepseek.com",
|
||||
reasoning: true,
|
||||
} satisfies Model<"openai-completions">;
|
||||
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(300_000);
|
||||
});
|
||||
|
||||
it("widens DeepSeek reasoning streams routed through an aliased OpenAI-compatible provider id", () => {
|
||||
const model = {
|
||||
...openAICompletionsModel,
|
||||
id: "deepseek-v4-pro",
|
||||
name: "DeepSeek V4 Pro",
|
||||
provider: "openai",
|
||||
baseUrl: "https://api.deepseek.com/v1",
|
||||
reasoning: true,
|
||||
} satisfies Model<"openai-completions">;
|
||||
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(300_000);
|
||||
});
|
||||
|
||||
it("leaves non-reasoning DeepSeek-hosted models on the global timeout", () => {
|
||||
const model = {
|
||||
...openAICompletionsModel,
|
||||
id: "deepseek-chat",
|
||||
name: "DeepSeek Chat",
|
||||
provider: "deepseek",
|
||||
baseUrl: "https://api.deepseek.com",
|
||||
reasoning: false,
|
||||
} satisfies Model<"openai-completions">;
|
||||
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBeUndefined();
|
||||
});
|
||||
|
||||
it("does not widen DeepSeek V4 reasoning models hosted on third-party OpenAI-compatible proxies", () => {
|
||||
const model = {
|
||||
...openAICompletionsModel,
|
||||
id: "deepseek-v4-pro",
|
||||
name: "DeepSeek V4 Pro",
|
||||
provider: "aimlapi",
|
||||
baseUrl: "https://api.aimlapi.com/v1",
|
||||
reasoning: true,
|
||||
} satisfies Model<"openai-completions">;
|
||||
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBeUndefined();
|
||||
});
|
||||
|
||||
it("keeps ordinary OpenAI-compatible models on the global timeout", () => {
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(openAICompletionsModel)).toBeUndefined();
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user