fix(ai): raised default first-event watchdog for google-gemini-cli streams
- Added optional per-provider fallback parameters to stream timeout helper functions so callers can widen default watchdog values safely. - Threaded provider-specific lazy stream limits into stream creation and set Google Gemini CLI to a 300000ms first-event fallback by default. - Added tests covering fallback defaults, env precedence, and global-default behavior for both idle and first-event timeouts.
This commit is contained in:
@@ -21,6 +21,7 @@
|
||||
### Fixed
|
||||
|
||||
- Fixed expired OAuth handling so provider-level paths no longer attempt direct token refresh calls for expired credentials and instead rely on `AuthStorage` for rotation
|
||||
- Fixed `google-gemini-cli` / `google-antigravity` aborting heavy reasoning runs with "Provider stream timed out while waiting for the first event" before the upstream had a chance to emit its first SSE frame. Cloud Code Assist routinely takes >100s on Gemini 3.x Pro at high thinking levels; the lazy-stream wrapper now floors the first-event watchdog at 5 minutes for these two providers when neither `StreamOptions.streamFirstEventTimeoutMs` nor `PI_STREAM_FIRST_EVENT_TIMEOUT_MS` pins a value. Other providers keep the 100s default. Internally, `getStreamIdleTimeoutMs` and `getStreamFirstEventTimeoutMs` now accept an optional per-provider `fallbackMs` so other slow-first-token providers can opt into the same widening without leaking through to the global default.
|
||||
- Fixed Claude Opus 4.7 on Amazon Bedrock streaming no reasoning output (and appearing to hang on long reasoning runs) because Anthropic silently switched the adaptive-thinking display default to `"omitted"`. The Bedrock provider now sends `thinking.display = "summarized"` by default on Opus 4.7+ adaptive models and on budget-based Claude models, mirroring the existing direct-Anthropic behavior. `BedrockOptions.thinkingDisplay` (`"summarized" | "omitted"`) is exposed for callers that want to opt out, and `hideThinkingSummary` now wires through to the Bedrock case ([#1373](https://github.com/can1357/oh-my-pi/issues/1373)).
|
||||
|
||||
## [15.3.2] - 2026-05-25
|
||||
|
||||
@@ -166,19 +166,47 @@ function hasFinalResult(
|
||||
return typeof (source as { result?: unknown }).result === "function";
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-provider default overrides for the lazy stream watchdogs. These widen the
|
||||
* floor used when neither caller option nor env var pins a value. The env vars
|
||||
* (`PI_STREAM_FIRST_EVENT_TIMEOUT_MS`, `PI_STREAM_IDLE_TIMEOUT_MS`) still take
|
||||
* precedence; `StreamOptions.streamFirstEventTimeoutMs` / `streamIdleTimeoutMs`
|
||||
* still trump everything.
|
||||
*/
|
||||
interface LazyStreamLimits {
|
||||
defaultFirstEventTimeoutMs?: number;
|
||||
defaultIdleTimeoutMs?: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Cloud Code Assist (google-gemini-cli / google-antigravity) routinely takes
|
||||
* longer than the global 100s default to emit its first SSE event when serving
|
||||
* the heavier Gemini 3.x Pro tiers at high thinking levels. Bump the first-event
|
||||
* floor to five minutes so duke et al. stop seeing spurious "stream timed out
|
||||
* while waiting for the first event" aborts on legitimate cold reasoning starts.
|
||||
* The steady-state idle watchdog stays on the global default since the upstream
|
||||
* emits thinking tokens frequently once it gets going.
|
||||
*/
|
||||
const GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS: LazyStreamLimits = {
|
||||
defaultFirstEventTimeoutMs: 300_000,
|
||||
};
|
||||
|
||||
function forwardStream<TApi extends Api>(
|
||||
target: EventStreamImpl,
|
||||
source: AsyncIterable<AssistantMessageEvent>,
|
||||
model: Model<TApi>,
|
||||
options: OptionsForApi<TApi>,
|
||||
abortTracker: AbortSourceTracker,
|
||||
limits?: LazyStreamLimits,
|
||||
): void {
|
||||
(async () => {
|
||||
try {
|
||||
const idleTimeoutMs = options.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs();
|
||||
const idleTimeoutMs = options.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs(limits?.defaultIdleTimeoutMs);
|
||||
const watchedSource = iterateWithIdleTimeout(source, {
|
||||
idleTimeoutMs,
|
||||
firstItemTimeoutMs: options.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
|
||||
firstItemTimeoutMs:
|
||||
options.streamFirstEventTimeoutMs ??
|
||||
getStreamFirstEventTimeoutMs(idleTimeoutMs, limits?.defaultFirstEventTimeoutMs),
|
||||
errorMessage: LAZY_STREAM_IDLE_TIMEOUT_ERROR,
|
||||
firstItemErrorMessage: LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR,
|
||||
onIdle: () => abortTracker.abortLocally(new Error(LAZY_STREAM_IDLE_TIMEOUT_ERROR)),
|
||||
@@ -241,6 +269,7 @@ function createLazyLoadErrorMessage<TApi extends Api>(
|
||||
|
||||
function createLazyStream<TApi extends Api>(
|
||||
loadModule: () => Promise<LazyProviderModule<TApi>>,
|
||||
limits?: LazyStreamLimits,
|
||||
): (model: Model<TApi>, context: Context, options: OptionsForApi<TApi>) => EventStreamImpl {
|
||||
return (model, context, options) => {
|
||||
const outer = new EventStreamImpl();
|
||||
@@ -251,7 +280,7 @@ function createLazyStream<TApi extends Api>(
|
||||
const abortTracker = createAbortSourceTracker(streamOptions.signal);
|
||||
const providerOptions = { ...streamOptions, signal: abortTracker.requestSignal } as OptionsForApi<TApi>;
|
||||
const inner = module.stream(model, context, providerOptions);
|
||||
forwardStream(outer, inner, model, streamOptions, abortTracker);
|
||||
forwardStream(outer, inner, model, streamOptions, abortTracker, limits);
|
||||
})
|
||||
.catch(error => {
|
||||
const message = createLazyLoadErrorMessage(model, error);
|
||||
@@ -369,7 +398,10 @@ function loadBedrockProviderModule(): Promise<LazyProviderModule<"bedrock-conver
|
||||
export const streamAnthropic = createLazyStream(loadAnthropicProviderModule);
|
||||
export const streamAzureOpenAIResponses = createLazyStream(loadAzureOpenAIResponsesProviderModule);
|
||||
export const streamGoogle = createLazyStream(loadGoogleProviderModule);
|
||||
export const streamGoogleGeminiCli = createLazyStream(loadGoogleGeminiCliProviderModule);
|
||||
export const streamGoogleGeminiCli = createLazyStream(
|
||||
loadGoogleGeminiCliProviderModule,
|
||||
GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS,
|
||||
);
|
||||
export const streamGoogleVertex = createLazyStream(loadGoogleVertexProviderModule);
|
||||
export const streamOpenAICodexResponses = createLazyStream(loadOpenAICodexResponsesProviderModule);
|
||||
export const streamOpenAICompletions = createLazyStream(loadOpenAICompletionsProviderModule);
|
||||
|
||||
@@ -16,12 +16,13 @@ function normalizeIdleTimeoutMs(value: string | undefined, fallback: number): nu
|
||||
*
|
||||
* `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` is accepted as a backward-compatible alias.
|
||||
* Set `PI_STREAM_IDLE_TIMEOUT_MS=0` to disable the watchdog.
|
||||
*
|
||||
* Providers that legitimately stream much slower than the global default can pass
|
||||
* `fallbackMs` to widen the floor used when neither env var nor caller option is set.
|
||||
* Caller options still take precedence; env overrides still trump the fallback.
|
||||
*/
|
||||
export function getStreamIdleTimeoutMs(): number | undefined {
|
||||
return normalizeIdleTimeoutMs(
|
||||
$env.PI_STREAM_IDLE_TIMEOUT_MS ?? $env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS,
|
||||
DEFAULT_STREAM_IDLE_TIMEOUT_MS,
|
||||
);
|
||||
export function getStreamIdleTimeoutMs(fallbackMs: number = DEFAULT_STREAM_IDLE_TIMEOUT_MS): number | undefined {
|
||||
return normalizeIdleTimeoutMs($env.PI_STREAM_IDLE_TIMEOUT_MS ?? $env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS, fallbackMs);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -42,12 +43,17 @@ export function getOpenAIStreamIdleTimeoutMs(): number | undefined {
|
||||
* so the default never undershoots the steady-state idle timeout.
|
||||
*
|
||||
* Set `PI_STREAM_FIRST_EVENT_TIMEOUT_MS=0` to disable the watchdog.
|
||||
*
|
||||
* Providers whose first response can legitimately take longer (heavy reasoning,
|
||||
* slow cold-start proxies) can pass `fallbackMs` to widen the floor used when
|
||||
* neither env var nor caller option is set. Caller options still take precedence;
|
||||
* env overrides still trump the fallback.
|
||||
*/
|
||||
export function getStreamFirstEventTimeoutMs(idleTimeoutMs?: number): number | undefined {
|
||||
const fallback =
|
||||
idleTimeoutMs === undefined
|
||||
? DEFAULT_STREAM_FIRST_EVENT_TIMEOUT_MS
|
||||
: Math.max(DEFAULT_STREAM_FIRST_EVENT_TIMEOUT_MS, idleTimeoutMs);
|
||||
export function getStreamFirstEventTimeoutMs(
|
||||
idleTimeoutMs?: number,
|
||||
fallbackMs: number = DEFAULT_STREAM_FIRST_EVENT_TIMEOUT_MS,
|
||||
): number | undefined {
|
||||
const fallback = idleTimeoutMs === undefined ? fallbackMs : Math.max(fallbackMs, idleTimeoutMs);
|
||||
return normalizeIdleTimeoutMs($env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS, fallback);
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
|
||||
import { getStreamFirstEventTimeoutMs, getStreamIdleTimeoutMs } from "../src/utils/idle-iterator";
|
||||
|
||||
/**
|
||||
* Per-provider fallback overrides on the stream-watchdog helpers.
|
||||
*
|
||||
* These are the gear that lets `google-gemini-cli` widen its first-event floor
|
||||
* beyond the 100s global default without forcing every other provider to wait
|
||||
* just as long. Tests pin the precedence contract callers depend on:
|
||||
* caller option > env var > per-provider fallback > base default.
|
||||
*/
|
||||
|
||||
const ENV_KEYS = [
|
||||
"PI_STREAM_IDLE_TIMEOUT_MS",
|
||||
"PI_OPENAI_STREAM_IDLE_TIMEOUT_MS",
|
||||
"PI_STREAM_FIRST_EVENT_TIMEOUT_MS",
|
||||
] as const;
|
||||
|
||||
const originalEnv: Partial<Record<(typeof ENV_KEYS)[number], string | undefined>> = {};
|
||||
|
||||
beforeEach(() => {
|
||||
for (const key of ENV_KEYS) {
|
||||
originalEnv[key] = Bun.env[key];
|
||||
delete Bun.env[key];
|
||||
}
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
for (const key of ENV_KEYS) {
|
||||
const prior = originalEnv[key];
|
||||
if (prior === undefined) {
|
||||
delete Bun.env[key];
|
||||
} else {
|
||||
Bun.env[key] = prior;
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
describe("getStreamIdleTimeoutMs(fallbackMs)", () => {
|
||||
it("returns the per-provider fallback when env vars are unset", () => {
|
||||
expect(getStreamIdleTimeoutMs(300_000)).toBe(300_000);
|
||||
});
|
||||
|
||||
it("lets PI_STREAM_IDLE_TIMEOUT_MS override the per-provider fallback", () => {
|
||||
Bun.env.PI_STREAM_IDLE_TIMEOUT_MS = "42";
|
||||
expect(getStreamIdleTimeoutMs(300_000)).toBe(42);
|
||||
});
|
||||
|
||||
it("treats PI_STREAM_IDLE_TIMEOUT_MS=0 as a watchdog disable", () => {
|
||||
Bun.env.PI_STREAM_IDLE_TIMEOUT_MS = "0";
|
||||
expect(getStreamIdleTimeoutMs(300_000)).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("getStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () => {
|
||||
it("returns the per-provider fallback when env unset and idle timeout is undefined", () => {
|
||||
expect(getStreamFirstEventTimeoutMs(undefined, 300_000)).toBe(300_000);
|
||||
});
|
||||
|
||||
it("floors the first-event timeout at the per-provider fallback even when idle is shorter", () => {
|
||||
expect(getStreamFirstEventTimeoutMs(50_000, 300_000)).toBe(300_000);
|
||||
});
|
||||
|
||||
it("never undershoots the steady-state idle timeout", () => {
|
||||
expect(getStreamFirstEventTimeoutMs(500_000, 300_000)).toBe(500_000);
|
||||
});
|
||||
|
||||
it("lets PI_STREAM_FIRST_EVENT_TIMEOUT_MS override the per-provider fallback", () => {
|
||||
Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "42";
|
||||
expect(getStreamFirstEventTimeoutMs(undefined, 300_000)).toBe(42);
|
||||
});
|
||||
|
||||
it("treats PI_STREAM_FIRST_EVENT_TIMEOUT_MS=0 as a watchdog disable", () => {
|
||||
Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "0";
|
||||
expect(getStreamFirstEventTimeoutMs(undefined, 300_000)).toBeUndefined();
|
||||
});
|
||||
|
||||
it("falls back to the 100s global default when no fallback or env is provided", () => {
|
||||
expect(getStreamFirstEventTimeoutMs()).toBe(100_000);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user