fix(ai): disabled local first-event watchdogs
- Added a per-model first-event watchdog policy and disabled it for local OpenAI-compatible backends while retaining inter-event stall detection. - Applied the policy to Responses and chat-completions transports with regression coverage for resolver and runtime precedence. Fixes #6524
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed OpenAI Responses and chat-completions streams honoring per-model first-event watchdog policy, allowing local llama.cpp-style backends to process arbitrarily large prompts without a premature client cancellation ([#6524](https://github.com/can1357/oh-my-pi/issues/6524)).
|
||||
|
||||
## [17.1.2] - 2026-07-24
|
||||
|
||||
### Added
|
||||
|
||||
@@ -630,7 +630,8 @@ const streamOpenAICompletionsOnce = (
|
||||
const idleTimeoutFallbackMs = model.compat.streamIdleTimeoutMs;
|
||||
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(idleTimeoutFallbackMs);
|
||||
const firstEventTimeoutMs =
|
||||
options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs);
|
||||
options?.streamFirstEventTimeoutMs ??
|
||||
getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, model.compat.streamFirstEventTimeoutMs);
|
||||
const requestTimeoutMs =
|
||||
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined;
|
||||
const { copilotPremiumRequests, baseUrl, headers, query, requestHeaders } = createRequestSetup(
|
||||
|
||||
@@ -495,7 +495,8 @@ const streamOpenAIResponsesOnce = (
|
||||
const idleTimeoutMs =
|
||||
options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(model.compat.streamIdleTimeoutMs);
|
||||
const firstEventTimeoutMs =
|
||||
options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs);
|
||||
options?.streamFirstEventTimeoutMs ??
|
||||
getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, model.compat.streamFirstEventTimeoutMs);
|
||||
const requestTimeoutMs =
|
||||
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined;
|
||||
const requestUrl = `${resolvedBaseUrl}/responses`;
|
||||
|
||||
@@ -68,11 +68,13 @@ export function getStreamFirstEventTimeoutMs(
|
||||
* `"0"` disable) wins outright. Otherwise the resolved idle (caller-supplied
|
||||
* `idleTimeoutMs` — which itself already encompasses per-call
|
||||
* `streamIdleTimeoutMs` or `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` resolved
|
||||
* upstream) floors the first-event budget so slow local OpenAI-compatible
|
||||
* servers are not undercut by a shorter `PI_STREAM_FIRST_EVENT_TIMEOUT_MS`
|
||||
* or the global default during prompt processing.
|
||||
* upstream) floors the first-event budget so slow OpenAI-compatible servers
|
||||
* are not undercut by a shorter `PI_STREAM_FIRST_EVENT_TIMEOUT_MS` or the
|
||||
* global default during prompt processing. A zero per-provider fallback
|
||||
* disables the first-event watchdog unless an environment override is set.
|
||||
*
|
||||
* Returns `undefined` when an explicit env knob disables the watchdog.
|
||||
* Returns `undefined` when an explicit env knob or per-provider fallback
|
||||
* disables the watchdog.
|
||||
*/
|
||||
export function getOpenAIStreamFirstEventTimeoutMs(
|
||||
idleTimeoutMs?: number,
|
||||
@@ -83,7 +85,7 @@ export function getOpenAIStreamFirstEventTimeoutMs(
|
||||
return normalizeIdleTimeoutMs(openAIFirstEventRaw, fallbackMs);
|
||||
}
|
||||
const base = normalizeIdleTimeoutMs($env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS, fallbackMs);
|
||||
if (base === undefined) return undefined;
|
||||
if (base === undefined || base <= 0) return undefined;
|
||||
if (idleTimeoutMs === undefined || idleTimeoutMs <= 0) return base;
|
||||
return Math.max(base, idleTimeoutMs);
|
||||
}
|
||||
|
||||
@@ -191,7 +191,7 @@ describe("resolveOpenAICompat stream idle timeout", () => {
|
||||
expect(openAICompletionsModel.compat.streamIdleTimeoutMs).toBeUndefined();
|
||||
});
|
||||
|
||||
it("widens local OpenAI-compatible stream watchdogs", () => {
|
||||
it("widens local idle watchdogs and disables first-event deadlines", () => {
|
||||
const completions = buildModel({
|
||||
...openAICompletionsModel,
|
||||
id: "qwen3-local",
|
||||
@@ -210,7 +210,9 @@ describe("resolveOpenAICompat stream idle timeout", () => {
|
||||
} as ModelSpec<"openai-responses">);
|
||||
|
||||
expect(completions.compat.streamIdleTimeoutMs).toBe(300_000);
|
||||
expect(completions.compat.streamFirstEventTimeoutMs).toBe(0);
|
||||
expect(responses.compat.streamIdleTimeoutMs).toBe(300_000);
|
||||
expect(responses.compat.streamFirstEventTimeoutMs).toBe(0);
|
||||
});
|
||||
|
||||
it("widens custom loopback OpenAI-compatible responses stream watchdogs", () => {
|
||||
@@ -224,6 +226,7 @@ describe("resolveOpenAICompat stream idle timeout", () => {
|
||||
} as ModelSpec<"openai-responses">);
|
||||
|
||||
expect(model.compat.streamIdleTimeoutMs).toBe(300_000);
|
||||
expect(model.compat.streamFirstEventTimeoutMs).toBe(0);
|
||||
});
|
||||
|
||||
it("widens Xiaomi MiMo Pro stream watchdog (issue #1770)", () => {
|
||||
|
||||
@@ -861,4 +861,29 @@ describe("OpenAI-family first-event timeouts", () => {
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorMessage).toBe("OpenAI responses stream stalled while waiting for the next event");
|
||||
});
|
||||
|
||||
it("honors streamFirstEventTimeoutMs from model.compat for OpenAI responses streams", async () => {
|
||||
const customResponsesModel: Model<"openai-responses"> = buildModel({
|
||||
id: "slow-first-event",
|
||||
name: "Slow First Event",
|
||||
api: "openai-responses",
|
||||
provider: "custom",
|
||||
baseUrl: "https://example.com/v1",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128000,
|
||||
maxTokens: 16384,
|
||||
compat: { streamFirstEventTimeoutMs: 20, streamIdleTimeoutMs: 5 },
|
||||
});
|
||||
const fetchMock = createDelayedFetch(30, createOpenAIResponsesSuccessResponse);
|
||||
|
||||
const result = await streamOpenAIResponses(customResponsesModel, baseContext(), {
|
||||
apiKey: "test-key",
|
||||
fetch: fetchMock,
|
||||
}).result();
|
||||
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorMessage).toBe("OpenAI responses stream timed out while waiting for the first event");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -123,6 +123,10 @@ describe("getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () =>
|
||||
expect(getOpenAIStreamFirstEventTimeoutMs(1500, 100_000)).toBeUndefined();
|
||||
});
|
||||
|
||||
it("treats a zero per-provider fallback as a watchdog disable", () => {
|
||||
expect(getOpenAIStreamFirstEventTimeoutMs(300_000, 0)).toBeUndefined();
|
||||
});
|
||||
|
||||
it("falls back to the generic first-event env when OpenAI env vars are unset", () => {
|
||||
Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "42";
|
||||
expect(getOpenAIStreamFirstEventTimeoutMs(undefined, 300_000)).toBe(42);
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Disabled the first-event watchdog for local OpenAI-compatible backends while retaining the 300-second inter-event watchdog, so long llama.cpp prompt prefill is not canceled and retried ([#6524](https://github.com/can1357/oh-my-pi/issues/6524)).
|
||||
|
||||
## [17.1.1] - 2026-07-24
|
||||
|
||||
### Added
|
||||
|
||||
@@ -574,6 +574,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
// to Moonshot verbatim, which 400s on non-MFJS constructs.
|
||||
toolSchemaFlavor:
|
||||
isMoonshotNative || isKimiModel ? "moonshot-mfjs" : isLocalOpenAICompatBackend ? "grammar" : undefined,
|
||||
streamFirstEventTimeoutMs: isLocalServingBackend ? 0 : undefined,
|
||||
streamIdleTimeoutMs,
|
||||
stripDeepseekSpecialTokens:
|
||||
isDeepseekModelIdOrName(spec.id) && (provider === "nvidia" || provider === "deepseek"),
|
||||
@@ -723,6 +724,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
|
||||
emptyLengthFinishIsContextError: spec.provider === "ollama",
|
||||
usesOpenAIToolCallIdLimit: spec.provider === "openai",
|
||||
promptCacheSessionHeader: spec.provider === "xai-oauth" ? "x-grok-conv-id" : undefined,
|
||||
streamFirstEventTimeoutMs: isLocalServingBackend ? 0 : spec.compat?.streamFirstEventTimeoutMs,
|
||||
streamIdleTimeoutMs: isLocalServingBackend
|
||||
? LOCAL_OPENAI_COMPAT_STREAM_IDLE_TIMEOUT_MS
|
||||
: spec.compat?.streamIdleTimeoutMs,
|
||||
|
||||
@@ -341,6 +341,12 @@ export interface OpenAICompat {
|
||||
* to opt a host out.
|
||||
*/
|
||||
toolSchemaFlavor?: "moonshot-mfjs" | "grammar" | "none";
|
||||
/**
|
||||
* Stream-watchdog first-event timeout in ms.
|
||||
* Set to `0` to allow unbounded prompt processing. Default: auto-detected
|
||||
* (disabled for local OpenAI-compatible backends).
|
||||
*/
|
||||
streamFirstEventTimeoutMs?: number;
|
||||
/**
|
||||
* Stream-watchdog idle-timeout floor in ms for slow reasoning hosts.
|
||||
* Default: auto-detected (GLM coding-plan hosts, direct DeepSeek reasoning).
|
||||
@@ -565,6 +571,8 @@ export interface ResolvedOpenAISharedCompat {
|
||||
requiresAssistantContentForToolCalls: boolean;
|
||||
stripDeepseekSpecialTokens: boolean;
|
||||
streamMarkupHealingPattern?: OpenAIStreamMarkupHealingPattern;
|
||||
/** See {@link OpenAICompat.streamFirstEventTimeoutMs}. */
|
||||
streamFirstEventTimeoutMs?: number;
|
||||
reasoningDeltasMayBeCumulative: boolean;
|
||||
emptyLengthFinishIsContextError: boolean;
|
||||
usesOpenAIToolCallIdLimit: boolean;
|
||||
@@ -645,6 +653,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
|
||||
| "extraBody"
|
||||
| "toolStrictMode"
|
||||
| "toolSchemaFlavor"
|
||||
| "streamFirstEventTimeoutMs"
|
||||
| "streamIdleTimeoutMs"
|
||||
| "cacheControlFormat"
|
||||
| "thinkingKeep"
|
||||
|
||||
Reference in New Issue
Block a user