Merge PR #6525: fix(ai): disable local first-event watchdogs (@roboomp)

This commit is contained in:
can1357
2026-07-25 00:59:15 +02:00
10 changed files with 121 additions and 12 deletions
+1
View File
@@ -5,6 +5,7 @@
### Fixed
- Fixed Cursor sessions exposing `ast_edit` (and other staged-preview `xd://` devices) without a reachable resolver: the built-in `write` tool — which carries the `xd://resolve` / `xd://reject` transport that finalizes a staged preview — was filtered out of Cursor's forwarded catalog, so previews could never be resolved and the session aborted after three forced `write` turns. `write` is now re-included in the forwarded catalog whenever pi-agent devices are advertised ([#6536](https://github.com/can1357/oh-my-pi/issues/6536)).
- Fixed OpenAI Responses and chat-completions streams honoring per-model first-event watchdog policy, allowing local llama.cpp-style backends to process arbitrarily large prompts without a premature client cancellation ([#6524](https://github.com/can1357/oh-my-pi/issues/6524)).
## [17.1.2] - 2026-07-24
@@ -630,7 +630,8 @@ const streamOpenAICompletionsOnce = (
const idleTimeoutFallbackMs = model.compat.streamIdleTimeoutMs;
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(idleTimeoutFallbackMs);
const firstEventTimeoutMs =
options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs);
options?.streamFirstEventTimeoutMs ??
getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, model.compat.streamFirstEventTimeoutMs);
const requestTimeoutMs =
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined;
const { copilotPremiumRequests, baseUrl, headers, query, requestHeaders } = createRequestSetup(
@@ -495,7 +495,8 @@ const streamOpenAIResponsesOnce = (
const idleTimeoutMs =
options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(model.compat.streamIdleTimeoutMs);
const firstEventTimeoutMs =
options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs);
options?.streamFirstEventTimeoutMs ??
getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, model.compat.streamFirstEventTimeoutMs);
const requestTimeoutMs =
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined;
const requestUrl = `${resolvedBaseUrl}/responses`;
+8 -6
View File
@@ -68,11 +68,13 @@ export function getStreamFirstEventTimeoutMs(
* `"0"` disable) wins outright. Otherwise the resolved idle (caller-supplied
* `idleTimeoutMs` — which itself already encompasses per-call
* `streamIdleTimeoutMs` or `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` resolved
* upstream) floors the first-event budget so slow local OpenAI-compatible
* servers are not undercut by a shorter `PI_STREAM_FIRST_EVENT_TIMEOUT_MS`
* or the global default during prompt processing.
* upstream) floors the first-event budget so slow OpenAI-compatible servers
* are not undercut by a shorter `PI_STREAM_FIRST_EVENT_TIMEOUT_MS` or the
* global default during prompt processing. A zero per-provider fallback
* disables the first-event watchdog unless an environment override is set.
*
* Returns `undefined` when an explicit env knob disables the watchdog.
* Returns `0` when an explicit env knob or per-provider fallback disables the
* watchdog, preserving the sentinel through iterator timeout resolution.
*/
export function getOpenAIStreamFirstEventTimeoutMs(
idleTimeoutMs?: number,
@@ -80,10 +82,10 @@ export function getOpenAIStreamFirstEventTimeoutMs(
): number | undefined {
const openAIFirstEventRaw = $env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS;
if (openAIFirstEventRaw !== undefined) {
return normalizeIdleTimeoutMs(openAIFirstEventRaw, fallbackMs);
return normalizeIdleTimeoutMs(openAIFirstEventRaw, fallbackMs) ?? 0;
}
const base = normalizeIdleTimeoutMs($env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS, fallbackMs);
if (base === undefined) return undefined;
if (base === undefined || base <= 0) return 0;
if (idleTimeoutMs === undefined || idleTimeoutMs <= 0) return base;
return Math.max(base, idleTimeoutMs);
}
@@ -191,7 +191,7 @@ describe("resolveOpenAICompat stream idle timeout", () => {
expect(openAICompletionsModel.compat.streamIdleTimeoutMs).toBeUndefined();
});
it("widens local OpenAI-compatible stream watchdogs", () => {
it("widens local idle watchdogs and disables first-event deadlines", () => {
const completions = buildModel({
...openAICompletionsModel,
id: "qwen3-local",
@@ -210,7 +210,9 @@ describe("resolveOpenAICompat stream idle timeout", () => {
} as ModelSpec<"openai-responses">);
expect(completions.compat.streamIdleTimeoutMs).toBe(300_000);
expect(completions.compat.streamFirstEventTimeoutMs).toBe(0);
expect(responses.compat.streamIdleTimeoutMs).toBe(300_000);
expect(responses.compat.streamFirstEventTimeoutMs).toBe(0);
});
it("widens custom loopback OpenAI-compatible responses stream watchdogs", () => {
@@ -224,6 +226,7 @@ describe("resolveOpenAICompat stream idle timeout", () => {
} as ModelSpec<"openai-responses">);
expect(model.compat.streamIdleTimeoutMs).toBe(300_000);
expect(model.compat.streamFirstEventTimeoutMs).toBe(0);
});
it("widens Xiaomi MiMo Pro stream watchdog (issue #1770)", () => {
@@ -1,4 +1,4 @@
import { describe, expect, it } from "bun:test";
import { describe, expect, it, vi } from "bun:test";
import { streamAzureOpenAIResponses } from "@oh-my-pi/pi-ai/providers/azure-openai-responses";
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
@@ -861,4 +861,86 @@ describe("OpenAI-family first-event timeouts", () => {
expect(result.stopReason).toBe("error");
expect(result.errorMessage).toBe("OpenAI responses stream stalled while waiting for the next event");
});
it("honors streamFirstEventTimeoutMs from model.compat for OpenAI responses streams", async () => {
const customResponsesModel: Model<"openai-responses"> = buildModel({
id: "slow-first-event",
name: "Slow First Event",
api: "openai-responses",
provider: "custom",
baseUrl: "https://example.com/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128000,
maxTokens: 16384,
compat: { streamFirstEventTimeoutMs: 20, streamIdleTimeoutMs: 5 },
});
const fetchMock = createDelayedFetch(30, createOpenAIResponsesSuccessResponse);
const result = await streamOpenAIResponses(customResponsesModel, baseContext(), {
apiKey: "test-key",
fetch: fetchMock,
}).result();
expect(result.stopReason).toBe("error");
expect(result.errorMessage).toBe("OpenAI responses stream timed out while waiting for the first event");
});
it("keeps a local first-event disable after SSE headers arrive", async () => {
vi.useFakeTimers();
const localResponsesModel: Model<"openai-responses"> = buildModel({
id: "slow-local-prefill",
name: "Slow Local Prefill",
api: "openai-responses",
provider: "llama.cpp",
baseUrl: "http://127.0.0.1:8080/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200000,
maxTokens: 16384,
compat: { streamFirstEventTimeoutMs: 0, streamIdleTimeoutMs: 20 },
});
const responseBody = new Uint8Array(await createOpenAIResponsesSuccessResponse().arrayBuffer());
const bodyRead = Promise.withResolvers<void>();
const releaseBody = Promise.withResolvers<void>();
const fetchMock: FetchImpl = () => {
const body = new ReadableStream<Uint8Array>(
{
async pull(controller) {
bodyRead.resolve();
await releaseBody.promise;
controller.enqueue(responseBody);
controller.close();
},
},
{ highWaterMark: 0 },
);
return Promise.resolve(
new Response(body, {
status: 200,
headers: { "content-type": "text/event-stream" },
}),
);
};
try {
const resultPromise = streamOpenAIResponses(localResponsesModel, baseContext(), {
apiKey: "test-key",
fetch: fetchMock,
}).result();
await bodyRead.promise;
for (let i = 0; i < 10; i++) await Promise.resolve();
expect(vi.getTimerCount()).toBe(0);
releaseBody.resolve();
const result = await resultPromise;
expect(result.stopReason).toBe("stop");
expect(getFirstTextContent(result)).toMatchObject({ type: "text", text: "Hello delayed" });
} finally {
releaseBody.resolve();
vi.useRealTimers();
}
});
});
@@ -120,7 +120,11 @@ describe("getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () =>
it("treats PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS=0 as an explicit watchdog disable", () => {
Bun.env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS = "0";
expect(getOpenAIStreamFirstEventTimeoutMs(1500, 100_000)).toBeUndefined();
expect(getOpenAIStreamFirstEventTimeoutMs(1500, 100_000)).toBe(0);
});
it("treats a zero per-provider fallback as a watchdog disable", () => {
expect(getOpenAIStreamFirstEventTimeoutMs(300_000, 0)).toBe(0);
});
it("falls back to the generic first-event env when OpenAI env vars are unset", () => {
@@ -130,7 +134,7 @@ describe("getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () =>
it("respects PI_STREAM_FIRST_EVENT_TIMEOUT_MS=0 disable when no OpenAI override is set", () => {
Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "0";
expect(getOpenAIStreamFirstEventTimeoutMs(1500, 100_000)).toBeUndefined();
expect(getOpenAIStreamFirstEventTimeoutMs(1500, 100_000)).toBe(0);
});
});
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Disabled the first-event watchdog for local OpenAI-compatible backends while retaining the 300-second inter-event watchdog, so long llama.cpp prompt prefill is not canceled and retried ([#6524](https://github.com/can1357/oh-my-pi/issues/6524)).
## [17.1.1] - 2026-07-24
### Added
+2
View File
@@ -574,6 +574,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
// to Moonshot verbatim, which 400s on non-MFJS constructs.
toolSchemaFlavor:
isMoonshotNative || isKimiModel ? "moonshot-mfjs" : isLocalOpenAICompatBackend ? "grammar" : undefined,
streamFirstEventTimeoutMs: isLocalServingBackend ? 0 : undefined,
streamIdleTimeoutMs,
stripDeepseekSpecialTokens:
isDeepseekModelIdOrName(spec.id) && (provider === "nvidia" || provider === "deepseek"),
@@ -723,6 +724,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
emptyLengthFinishIsContextError: spec.provider === "ollama",
usesOpenAIToolCallIdLimit: spec.provider === "openai",
promptCacheSessionHeader: spec.provider === "xai-oauth" ? "x-grok-conv-id" : undefined,
streamFirstEventTimeoutMs: isLocalServingBackend ? 0 : spec.compat?.streamFirstEventTimeoutMs,
streamIdleTimeoutMs: isLocalServingBackend
? LOCAL_OPENAI_COMPAT_STREAM_IDLE_TIMEOUT_MS
: spec.compat?.streamIdleTimeoutMs,
+9
View File
@@ -341,6 +341,12 @@ export interface OpenAICompat {
* to opt a host out.
*/
toolSchemaFlavor?: "moonshot-mfjs" | "grammar" | "none";
/**
* Stream-watchdog first-event timeout in ms.
* Set to `0` to allow unbounded prompt processing. Default: auto-detected
* (disabled for local OpenAI-compatible backends).
*/
streamFirstEventTimeoutMs?: number;
/**
* Stream-watchdog idle-timeout floor in ms for slow reasoning hosts.
* Default: auto-detected (GLM coding-plan hosts, direct DeepSeek reasoning).
@@ -565,6 +571,8 @@ export interface ResolvedOpenAISharedCompat {
requiresAssistantContentForToolCalls: boolean;
stripDeepseekSpecialTokens: boolean;
streamMarkupHealingPattern?: OpenAIStreamMarkupHealingPattern;
/** See {@link OpenAICompat.streamFirstEventTimeoutMs}. */
streamFirstEventTimeoutMs?: number;
reasoningDeltasMayBeCumulative: boolean;
emptyLengthFinishIsContextError: boolean;
usesOpenAIToolCallIdLimit: boolean;
@@ -645,6 +653,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
| "extraBody"
| "toolStrictMode"
| "toolSchemaFlavor"
| "streamFirstEventTimeoutMs"
| "streamIdleTimeoutMs"
| "cacheControlFormat"
| "thinkingKeep"