fix(ai): kept openai first-event env honored with per-call idle

Always route OpenAI-family providers through getOpenAIStreamFirstEventTimeoutMs so PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS wins even when callers pass per-call streamIdleTimeoutMs. The OpenAI helper now floors the first-event budget at the caller-resolved idle (which already encompasses per-call streamIdleTimeoutMs or PI_OPENAI_STREAM_IDLE_TIMEOUT_MS upstream), and explicit env disables ("0") on either knob continue to drop the watchdog.

Fixes #1603
This commit is contained in:
roboomp
2026-05-31 19:02:53 +00:00
parent 239bb9858d
commit 062ba5e836
8 changed files with 71 additions and 44 deletions
@@ -24,7 +24,6 @@ import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-ins
import {
getOpenAIStreamFirstEventTimeoutMs,
getOpenAIStreamIdleTimeoutMs,
getStreamFirstEventTimeoutMs,
iterateWithIdleTimeout,
} from "../utils/idle-iterator";
import { sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
@@ -124,10 +123,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
options?.onPayload?.(params);
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
const firstEventTimeoutMs =
options?.streamFirstEventTimeoutMs ??
(options?.streamIdleTimeoutMs === undefined
? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs)
: getStreamFirstEventTimeoutMs(idleTimeoutMs));
options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs);
const requestTimeoutMs =
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined;
rawRequestDump = {
@@ -51,7 +51,6 @@ import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-ins
import {
getOpenAIStreamFirstEventTimeoutMs,
getOpenAIStreamIdleTimeoutMs,
getStreamFirstEventTimeoutMs,
iterateWithIdleTimeout,
} from "../utils/idle-iterator";
import { parseStreamingJson, parseStreamingJsonThrottled } from "../utils/json-parse";
@@ -604,11 +603,7 @@ function createRequestSetup(options: OpenAICodexResponsesOptions | undefined): C
: requestAbortController.signal;
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
const websocketIdleTimeoutMs = options?.streamIdleTimeoutMs ?? getCodexWebSocketIdleTimeoutMs();
const firstEventTimeoutMs =
options?.streamFirstEventTimeoutMs ??
(options?.streamIdleTimeoutMs === undefined
? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs)
: getStreamFirstEventTimeoutMs(idleTimeoutMs));
const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs);
const websocketFirstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getCodexWebSocketFirstEventTimeoutMs();
const wrapCodexSseStream = (
source: AsyncGenerator<Record<string, unknown>>,
@@ -48,7 +48,6 @@ import {
import {
getOpenAIStreamFirstEventTimeoutMs,
getOpenAIStreamIdleTimeoutMs,
getStreamFirstEventTimeoutMs,
iterateWithIdleTimeout,
} from "../utils/idle-iterator";
import { parseStreamingJson, parseStreamingJsonThrottled } from "../utils/json-parse";
@@ -425,10 +424,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
const idleTimeoutFallbackMs = getOpenAICompletionsStreamIdleTimeoutFallbackMs(model);
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(idleTimeoutFallbackMs);
const firstEventTimeoutMs =
options?.streamFirstEventTimeoutMs ??
(options?.streamIdleTimeoutMs === undefined
? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, idleTimeoutFallbackMs)
: getStreamFirstEventTimeoutMs(idleTimeoutMs));
options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs);
const requestTimeoutMs =
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined;
const {
@@ -35,7 +35,6 @@ import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } fr
import {
getOpenAIStreamFirstEventTimeoutMs,
getOpenAIStreamIdleTimeoutMs,
getStreamFirstEventTimeoutMs,
iterateWithIdleTimeout,
} from "../utils/idle-iterator";
import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
@@ -230,10 +229,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
const { params } = buildParams(model, context, options, providerSessionState, baseUrl);
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
const firstEventTimeoutMs =
options?.streamFirstEventTimeoutMs ??
(options?.streamIdleTimeoutMs === undefined
? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs)
: getStreamFirstEventTimeoutMs(idleTimeoutMs));
options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs);
const requestTimeoutMs =
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined;
options?.onPayload?.(params);
+4 -2
View File
@@ -364,8 +364,10 @@ export interface StreamOptions {
* event arrives, `streamIdleTimeoutMs` governs inter-event stalls. Falls
* back to `PI_STREAM_FIRST_EVENT_TIMEOUT_MS` and then to a 100s default.
* OpenAI-family transports additionally honor
* `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` and use
* `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` as the first-event floor.
* `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` as the most-specific override and
* floor the first-event budget at the resolved idle (per-call
* `streamIdleTimeoutMs` or `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS`) so slow local
* OpenAI-compatible servers are not undercut during prompt processing.
*
* Iterator-level honored by: every built-in provider (via the lazy-stream
* forwarder in `register-builtins`). SDK-request honored by:
+17 -11
View File
@@ -61,22 +61,28 @@ export function getStreamFirstEventTimeoutMs(
/**
* Returns the first-event timeout used for OpenAI-family streaming transports.
*
* `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` is the most specific first-event
* override. When it is unset, `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` also widens
* or disables the first-event watchdog so local OpenAI-compatible servers are
* not undercut by the generic first-event setting during slow prompt processing.
* Precedence: explicit `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` (including a
* `"0"` disable) wins outright. Otherwise the resolved idle (caller-supplied
* `idleTimeoutMs` — which itself already encompasses per-call
* `streamIdleTimeoutMs` or `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` resolved
* upstream) floors the first-event budget so slow local OpenAI-compatible
* servers are not undercut by a shorter `PI_STREAM_FIRST_EVENT_TIMEOUT_MS`
* or the global default during prompt processing.
*
* Returns `undefined` when an explicit env knob disables the watchdog.
*/
export function getOpenAIStreamFirstEventTimeoutMs(
idleTimeoutMs?: number,
fallbackMs: number = DEFAULT_STREAM_FIRST_EVENT_TIMEOUT_MS,
): number | undefined {
const fallback = idleTimeoutMs === undefined ? fallbackMs : Math.max(fallbackMs, idleTimeoutMs);
return normalizeIdleTimeoutMs(
$env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS ??
$env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS ??
$env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS,
fallback,
);
const openAIFirstEventRaw = $env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS;
if (openAIFirstEventRaw !== undefined) {
return normalizeIdleTimeoutMs(openAIFirstEventRaw, fallbackMs);
}
const base = normalizeIdleTimeoutMs($env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS, fallbackMs);
if (base === undefined) return undefined;
if (idleTimeoutMs === undefined || idleTimeoutMs <= 0) return base;
return Math.max(base, idleTimeoutMs);
}
export interface IdleTimeoutIteratorOptions {
@@ -352,6 +352,39 @@ describe("OpenAI-family first-event timeouts", () => {
}
});
it("honors PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS even when caller pins streamIdleTimeoutMs", async () => {
const previousOpenAIFirstEventTimeout = Bun.env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS;
const previousGenericFirstEventTimeout = Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS;
const timeoutHeaders: string[] = [];
Bun.env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS = "1500";
Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "20";
global.fetch = createDelayedFetch(30, createOpenAIResponsesSuccessResponse, (input, init) => {
timeoutHeaders.push(getRequestHeader(input, init, "X-Stainless-Timeout") ?? "");
});
try {
const result = await streamOpenAIResponses(openAIResponsesModel, baseContext(), {
apiKey: "test-key",
streamIdleTimeoutMs: 5_000,
}).result();
expect(result.stopReason).toBe("stop");
expect(getFirstTextContent(result)).toMatchObject({ type: "text", text: "Hello delayed" });
expect(timeoutHeaders).toContain("1");
} finally {
if (previousOpenAIFirstEventTimeout === undefined) {
delete Bun.env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS;
} else {
Bun.env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS = previousOpenAIFirstEventTimeout;
}
if (previousGenericFirstEventTimeout === undefined) {
delete Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS;
} else {
Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = previousGenericFirstEventTimeout;
}
}
});
it("times out OpenAI responses streams that only emit no-progress status events", async () => {
global.fetch = ((input: string | URL | Request, init?: RequestInit) =>
Promise.resolve(createNoProgressOpenAIResponsesStream(getRequestSignal(input, init)))) as typeof fetch;
@@ -105,16 +105,20 @@ describe("getStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () => {
});
describe("getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () => {
it("lets the OpenAI idle env widen a lower generic first-event timeout", () => {
it("floors the first-event budget at the caller-resolved idle when the generic env is lower", () => {
Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "20";
Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS = "84";
expect(getOpenAIStreamFirstEventTimeoutMs(84, 300_000)).toBe(84);
expect(getOpenAIStreamFirstEventTimeoutMs(1500, 100_000)).toBe(1500);
});
it("lets the OpenAI first-event env override the OpenAI idle env", () => {
it("honors PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS even when caller pins per-call idle", () => {
Bun.env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS = "42";
Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS = "84";
expect(getOpenAIStreamFirstEventTimeoutMs(84, 300_000)).toBe(42);
Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "20";
expect(getOpenAIStreamFirstEventTimeoutMs(5_000, 100_000)).toBe(42);
});
it("treats PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS=0 as an explicit watchdog disable", () => {
Bun.env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS = "0";
expect(getOpenAIStreamFirstEventTimeoutMs(1500, 100_000)).toBeUndefined();
});
it("falls back to the generic first-event env when OpenAI env vars are unset", () => {
@@ -122,10 +126,9 @@ describe("getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () =>
expect(getOpenAIStreamFirstEventTimeoutMs(undefined, 300_000)).toBe(42);
});
it("treats PI_OPENAI_STREAM_IDLE_TIMEOUT_MS=0 as an OpenAI watchdog disable", () => {
Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "42";
Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS = "0";
expect(getOpenAIStreamFirstEventTimeoutMs(undefined, 300_000)).toBeUndefined();
it("respects PI_STREAM_FIRST_EVENT_TIMEOUT_MS=0 disable when no OpenAI override is set", () => {
Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "0";
expect(getOpenAIStreamFirstEventTimeoutMs(1500, 100_000)).toBeUndefined();
});
});