fix(ai): disabled local first-event watchdogs

- Added a per-model first-event watchdog policy and disabled it for local OpenAI-compatible backends while retaining inter-event stall detection.
- Applied the policy to Responses and chat-completions transports with regression coverage for resolver and runtime precedence.

Fixes #6524
This commit is contained in:
roboomp
2026-07-24 15:19:14 +00:00
parent 7ebf141743
commit a421ea8acd
10 changed files with 63 additions and 8 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed OpenAI Responses and chat-completions streams honoring per-model first-event watchdog policy, allowing local llama.cpp-style backends to process arbitrarily large prompts without a premature client cancellation ([#6524](https://github.com/can1357/oh-my-pi/issues/6524)).
## [17.1.2] - 2026-07-24
### Added
@@ -630,7 +630,8 @@ const streamOpenAICompletionsOnce = (
const idleTimeoutFallbackMs = model.compat.streamIdleTimeoutMs;
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(idleTimeoutFallbackMs);
const firstEventTimeoutMs =
options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs);
options?.streamFirstEventTimeoutMs ??
getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, model.compat.streamFirstEventTimeoutMs);
const requestTimeoutMs =
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined;
const { copilotPremiumRequests, baseUrl, headers, query, requestHeaders } = createRequestSetup(
@@ -495,7 +495,8 @@ const streamOpenAIResponsesOnce = (
const idleTimeoutMs =
options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(model.compat.streamIdleTimeoutMs);
const firstEventTimeoutMs =
options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs);
options?.streamFirstEventTimeoutMs ??
getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, model.compat.streamFirstEventTimeoutMs);
const requestTimeoutMs =
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined;
const requestUrl = `${resolvedBaseUrl}/responses`;
+7 -5
View File
@@ -68,11 +68,13 @@ export function getStreamFirstEventTimeoutMs(
* `"0"` disable) wins outright. Otherwise the resolved idle (caller-supplied
* `idleTimeoutMs` — which itself already encompasses per-call
* `streamIdleTimeoutMs` or `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` resolved
* upstream) floors the first-event budget so slow local OpenAI-compatible
* servers are not undercut by a shorter `PI_STREAM_FIRST_EVENT_TIMEOUT_MS`
* or the global default during prompt processing.
* upstream) floors the first-event budget so slow OpenAI-compatible servers
* are not undercut by a shorter `PI_STREAM_FIRST_EVENT_TIMEOUT_MS` or the
* global default during prompt processing. A zero per-provider fallback
* disables the first-event watchdog unless an environment override is set.
*
* Returns `undefined` when an explicit env knob disables the watchdog.
* Returns `undefined` when an explicit env knob or per-provider fallback
* disables the watchdog.
*/
export function getOpenAIStreamFirstEventTimeoutMs(
idleTimeoutMs?: number,
@@ -83,7 +85,7 @@ export function getOpenAIStreamFirstEventTimeoutMs(
return normalizeIdleTimeoutMs(openAIFirstEventRaw, fallbackMs);
}
const base = normalizeIdleTimeoutMs($env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS, fallbackMs);
if (base === undefined) return undefined;
if (base === undefined || base <= 0) return undefined;
if (idleTimeoutMs === undefined || idleTimeoutMs <= 0) return base;
return Math.max(base, idleTimeoutMs);
}
@@ -191,7 +191,7 @@ describe("resolveOpenAICompat stream idle timeout", () => {
expect(openAICompletionsModel.compat.streamIdleTimeoutMs).toBeUndefined();
});
it("widens local OpenAI-compatible stream watchdogs", () => {
it("widens local idle watchdogs and disables first-event deadlines", () => {
const completions = buildModel({
...openAICompletionsModel,
id: "qwen3-local",
@@ -210,7 +210,9 @@ describe("resolveOpenAICompat stream idle timeout", () => {
} as ModelSpec<"openai-responses">);
expect(completions.compat.streamIdleTimeoutMs).toBe(300_000);
expect(completions.compat.streamFirstEventTimeoutMs).toBe(0);
expect(responses.compat.streamIdleTimeoutMs).toBe(300_000);
expect(responses.compat.streamFirstEventTimeoutMs).toBe(0);
});
it("widens custom loopback OpenAI-compatible responses stream watchdogs", () => {
@@ -224,6 +226,7 @@ describe("resolveOpenAICompat stream idle timeout", () => {
} as ModelSpec<"openai-responses">);
expect(model.compat.streamIdleTimeoutMs).toBe(300_000);
expect(model.compat.streamFirstEventTimeoutMs).toBe(0);
});
it("widens Xiaomi MiMo Pro stream watchdog (issue #1770)", () => {
@@ -861,4 +861,29 @@ describe("OpenAI-family first-event timeouts", () => {
expect(result.stopReason).toBe("error");
expect(result.errorMessage).toBe("OpenAI responses stream stalled while waiting for the next event");
});
it("honors streamFirstEventTimeoutMs from model.compat for OpenAI responses streams", async () => {
const customResponsesModel: Model<"openai-responses"> = buildModel({
id: "slow-first-event",
name: "Slow First Event",
api: "openai-responses",
provider: "custom",
baseUrl: "https://example.com/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128000,
maxTokens: 16384,
compat: { streamFirstEventTimeoutMs: 20, streamIdleTimeoutMs: 5 },
});
const fetchMock = createDelayedFetch(30, createOpenAIResponsesSuccessResponse);
const result = await streamOpenAIResponses(customResponsesModel, baseContext(), {
apiKey: "test-key",
fetch: fetchMock,
}).result();
expect(result.stopReason).toBe("error");
expect(result.errorMessage).toBe("OpenAI responses stream timed out while waiting for the first event");
});
});
@@ -123,6 +123,10 @@ describe("getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () =>
expect(getOpenAIStreamFirstEventTimeoutMs(1500, 100_000)).toBeUndefined();
});
it("treats a zero per-provider fallback as a watchdog disable", () => {
expect(getOpenAIStreamFirstEventTimeoutMs(300_000, 0)).toBeUndefined();
});
it("falls back to the generic first-event env when OpenAI env vars are unset", () => {
Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "42";
expect(getOpenAIStreamFirstEventTimeoutMs(undefined, 300_000)).toBe(42);
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Disabled the first-event watchdog for local OpenAI-compatible backends while retaining the 300-second inter-event watchdog, so long llama.cpp prompt prefill is not canceled and retried ([#6524](https://github.com/can1357/oh-my-pi/issues/6524)).
## [17.1.1] - 2026-07-24
### Added
+2
View File
@@ -574,6 +574,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
// to Moonshot verbatim, which 400s on non-MFJS constructs.
toolSchemaFlavor:
isMoonshotNative || isKimiModel ? "moonshot-mfjs" : isLocalOpenAICompatBackend ? "grammar" : undefined,
streamFirstEventTimeoutMs: isLocalServingBackend ? 0 : undefined,
streamIdleTimeoutMs,
stripDeepseekSpecialTokens:
isDeepseekModelIdOrName(spec.id) && (provider === "nvidia" || provider === "deepseek"),
@@ -723,6 +724,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
emptyLengthFinishIsContextError: spec.provider === "ollama",
usesOpenAIToolCallIdLimit: spec.provider === "openai",
promptCacheSessionHeader: spec.provider === "xai-oauth" ? "x-grok-conv-id" : undefined,
streamFirstEventTimeoutMs: isLocalServingBackend ? 0 : spec.compat?.streamFirstEventTimeoutMs,
streamIdleTimeoutMs: isLocalServingBackend
? LOCAL_OPENAI_COMPAT_STREAM_IDLE_TIMEOUT_MS
: spec.compat?.streamIdleTimeoutMs,
+9
View File
@@ -341,6 +341,12 @@ export interface OpenAICompat {
* to opt a host out.
*/
toolSchemaFlavor?: "moonshot-mfjs" | "grammar" | "none";
/**
* Stream-watchdog first-event timeout in ms.
* Set to `0` to allow unbounded prompt processing. Default: auto-detected
* (disabled for local OpenAI-compatible backends).
*/
streamFirstEventTimeoutMs?: number;
/**
* Stream-watchdog idle-timeout floor in ms for slow reasoning hosts.
* Default: auto-detected (GLM coding-plan hosts, direct DeepSeek reasoning).
@@ -565,6 +571,8 @@ export interface ResolvedOpenAISharedCompat {
requiresAssistantContentForToolCalls: boolean;
stripDeepseekSpecialTokens: boolean;
streamMarkupHealingPattern?: OpenAIStreamMarkupHealingPattern;
/** See {@link OpenAICompat.streamFirstEventTimeoutMs}. */
streamFirstEventTimeoutMs?: number;
reasoningDeltasMayBeCumulative: boolean;
emptyLengthFinishIsContextError: boolean;
usesOpenAIToolCallIdLimit: boolean;
@@ -645,6 +653,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
| "extraBody"
| "toolStrictMode"
| "toolSchemaFlavor"
| "streamFirstEventTimeoutMs"
| "streamIdleTimeoutMs"
| "cacheControlFormat"
| "thinkingKeep"