fix(ai): raised default first-event watchdog for google-gemini-cli streams

- Added optional per-provider fallback parameters to stream timeout helper functions so callers can widen default watchdog values safely.
- Threaded provider-specific lazy stream limits into stream creation and set Google Gemini CLI to a 300000ms first-event fallback by default.
- Added tests covering fallback defaults, env precedence, and global-default behavior for both idle and first-event timeouts.
This commit is contained in:
can1357
2026-05-26 05:17:58 +02:00
parent d56c7fcfa8
commit eef35a1cb3
4 changed files with 134 additions and 14 deletions
+1
View File
@@ -21,6 +21,7 @@
### Fixed
- Fixed expired OAuth handling so provider-level paths no longer attempt direct token refresh calls for expired credentials and instead rely on `AuthStorage` for rotation
- Fixed `google-gemini-cli` / `google-antigravity` aborting heavy reasoning runs with "Provider stream timed out while waiting for the first event" before the upstream had a chance to emit its first SSE frame. Cloud Code Assist routinely takes >100s on Gemini 3.x Pro at high thinking levels; the lazy-stream wrapper now floors the first-event watchdog at 5 minutes for these two providers when neither `StreamOptions.streamFirstEventTimeoutMs` nor `PI_STREAM_FIRST_EVENT_TIMEOUT_MS` pins a value. Other providers keep the 100s default. Internally, `getStreamIdleTimeoutMs` and `getStreamFirstEventTimeoutMs` now accept an optional per-provider `fallbackMs` so other slow-first-token providers can opt into the same widening without leaking through to the global default.
- Fixed Claude Opus 4.7 on Amazon Bedrock streaming no reasoning output (and appearing to hang on long reasoning runs) because Anthropic silently switched the adaptive-thinking display default to `"omitted"`. The Bedrock provider now sends `thinking.display = "summarized"` by default on Opus 4.7+ adaptive models and on budget-based Claude models, mirroring the existing direct-Anthropic behavior. `BedrockOptions.thinkingDisplay` (`"summarized" | "omitted"`) is exposed for callers that want to opt out, and `hideThinkingSummary` now wires through to the Bedrock case ([#1373](https://github.com/can1357/oh-my-pi/issues/1373)).
## [15.3.2] - 2026-05-25
+36 -4
View File
@@ -166,19 +166,47 @@ function hasFinalResult(
return typeof (source as { result?: unknown }).result === "function";
}
/**
* Per-provider default overrides for the lazy stream watchdogs. These widen the
* floor used when neither caller option nor env var pins a value. The env vars
* (`PI_STREAM_FIRST_EVENT_TIMEOUT_MS`, `PI_STREAM_IDLE_TIMEOUT_MS`) still take
* precedence; `StreamOptions.streamFirstEventTimeoutMs` / `streamIdleTimeoutMs`
* still trump everything.
*/
interface LazyStreamLimits {
defaultFirstEventTimeoutMs?: number;
defaultIdleTimeoutMs?: number;
}
/**
* Cloud Code Assist (google-gemini-cli / google-antigravity) routinely takes
* longer than the global 100s default to emit its first SSE event when serving
* the heavier Gemini 3.x Pro tiers at high thinking levels. Bump the first-event
* floor to five minutes so duke et al. stop seeing spurious "stream timed out
* while waiting for the first event" aborts on legitimate cold reasoning starts.
* The steady-state idle watchdog stays on the global default since the upstream
* emits thinking tokens frequently once it gets going.
*/
const GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS: LazyStreamLimits = {
defaultFirstEventTimeoutMs: 300_000,
};
function forwardStream<TApi extends Api>(
target: EventStreamImpl,
source: AsyncIterable<AssistantMessageEvent>,
model: Model<TApi>,
options: OptionsForApi<TApi>,
abortTracker: AbortSourceTracker,
limits?: LazyStreamLimits,
): void {
(async () => {
try {
const idleTimeoutMs = options.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs();
const idleTimeoutMs = options.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs(limits?.defaultIdleTimeoutMs);
const watchedSource = iterateWithIdleTimeout(source, {
idleTimeoutMs,
firstItemTimeoutMs: options.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
firstItemTimeoutMs:
options.streamFirstEventTimeoutMs ??
getStreamFirstEventTimeoutMs(idleTimeoutMs, limits?.defaultFirstEventTimeoutMs),
errorMessage: LAZY_STREAM_IDLE_TIMEOUT_ERROR,
firstItemErrorMessage: LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR,
onIdle: () => abortTracker.abortLocally(new Error(LAZY_STREAM_IDLE_TIMEOUT_ERROR)),
@@ -241,6 +269,7 @@ function createLazyLoadErrorMessage<TApi extends Api>(
function createLazyStream<TApi extends Api>(
loadModule: () => Promise<LazyProviderModule<TApi>>,
limits?: LazyStreamLimits,
): (model: Model<TApi>, context: Context, options: OptionsForApi<TApi>) => EventStreamImpl {
return (model, context, options) => {
const outer = new EventStreamImpl();
@@ -251,7 +280,7 @@ function createLazyStream<TApi extends Api>(
const abortTracker = createAbortSourceTracker(streamOptions.signal);
const providerOptions = { ...streamOptions, signal: abortTracker.requestSignal } as OptionsForApi<TApi>;
const inner = module.stream(model, context, providerOptions);
forwardStream(outer, inner, model, streamOptions, abortTracker);
forwardStream(outer, inner, model, streamOptions, abortTracker, limits);
})
.catch(error => {
const message = createLazyLoadErrorMessage(model, error);
@@ -369,7 +398,10 @@ function loadBedrockProviderModule(): Promise<LazyProviderModule<"bedrock-conver
export const streamAnthropic = createLazyStream(loadAnthropicProviderModule);
export const streamAzureOpenAIResponses = createLazyStream(loadAzureOpenAIResponsesProviderModule);
export const streamGoogle = createLazyStream(loadGoogleProviderModule);
export const streamGoogleGeminiCli = createLazyStream(loadGoogleGeminiCliProviderModule);
export const streamGoogleGeminiCli = createLazyStream(
loadGoogleGeminiCliProviderModule,
GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS,
);
export const streamGoogleVertex = createLazyStream(loadGoogleVertexProviderModule);
export const streamOpenAICodexResponses = createLazyStream(loadOpenAICodexResponsesProviderModule);
export const streamOpenAICompletions = createLazyStream(loadOpenAICompletionsProviderModule);
+16 -10
View File
@@ -16,12 +16,13 @@ function normalizeIdleTimeoutMs(value: string | undefined, fallback: number): nu
*
* `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` is accepted as a backward-compatible alias.
* Set `PI_STREAM_IDLE_TIMEOUT_MS=0` to disable the watchdog.
*
* Providers that legitimately stream much slower than the global default can pass
* `fallbackMs` to widen the floor used when neither env var nor caller option is set.
* Caller options still take precedence; env overrides still trump the fallback.
*/
export function getStreamIdleTimeoutMs(): number | undefined {
return normalizeIdleTimeoutMs(
$env.PI_STREAM_IDLE_TIMEOUT_MS ?? $env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS,
DEFAULT_STREAM_IDLE_TIMEOUT_MS,
);
export function getStreamIdleTimeoutMs(fallbackMs: number = DEFAULT_STREAM_IDLE_TIMEOUT_MS): number | undefined {
return normalizeIdleTimeoutMs($env.PI_STREAM_IDLE_TIMEOUT_MS ?? $env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS, fallbackMs);
}
/**
@@ -42,12 +43,17 @@ export function getOpenAIStreamIdleTimeoutMs(): number | undefined {
* so the default never undershoots the steady-state idle timeout.
*
* Set `PI_STREAM_FIRST_EVENT_TIMEOUT_MS=0` to disable the watchdog.
*
* Providers whose first response can legitimately take longer (heavy reasoning,
* slow cold-start proxies) can pass `fallbackMs` to widen the floor used when
* neither env var nor caller option is set. Caller options still take precedence;
* env overrides still trump the fallback.
*/
export function getStreamFirstEventTimeoutMs(idleTimeoutMs?: number): number | undefined {
const fallback =
idleTimeoutMs === undefined
? DEFAULT_STREAM_FIRST_EVENT_TIMEOUT_MS
: Math.max(DEFAULT_STREAM_FIRST_EVENT_TIMEOUT_MS, idleTimeoutMs);
export function getStreamFirstEventTimeoutMs(
idleTimeoutMs?: number,
fallbackMs: number = DEFAULT_STREAM_FIRST_EVENT_TIMEOUT_MS,
): number | undefined {
const fallback = idleTimeoutMs === undefined ? fallbackMs : Math.max(fallbackMs, idleTimeoutMs);
return normalizeIdleTimeoutMs($env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS, fallback);
}
@@ -0,0 +1,81 @@
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import { getStreamFirstEventTimeoutMs, getStreamIdleTimeoutMs } from "../src/utils/idle-iterator";
/**
* Per-provider fallback overrides on the stream-watchdog helpers.
*
* These are the gear that lets `google-gemini-cli` widen its first-event floor
* beyond the 100s global default without forcing every other provider to wait
* just as long. Tests pin the precedence contract callers depend on:
* caller option > env var > per-provider fallback > base default.
*/
const ENV_KEYS = [
"PI_STREAM_IDLE_TIMEOUT_MS",
"PI_OPENAI_STREAM_IDLE_TIMEOUT_MS",
"PI_STREAM_FIRST_EVENT_TIMEOUT_MS",
] as const;
const originalEnv: Partial<Record<(typeof ENV_KEYS)[number], string | undefined>> = {};
beforeEach(() => {
for (const key of ENV_KEYS) {
originalEnv[key] = Bun.env[key];
delete Bun.env[key];
}
});
afterEach(() => {
for (const key of ENV_KEYS) {
const prior = originalEnv[key];
if (prior === undefined) {
delete Bun.env[key];
} else {
Bun.env[key] = prior;
}
}
});
describe("getStreamIdleTimeoutMs(fallbackMs)", () => {
it("returns the per-provider fallback when env vars are unset", () => {
expect(getStreamIdleTimeoutMs(300_000)).toBe(300_000);
});
it("lets PI_STREAM_IDLE_TIMEOUT_MS override the per-provider fallback", () => {
Bun.env.PI_STREAM_IDLE_TIMEOUT_MS = "42";
expect(getStreamIdleTimeoutMs(300_000)).toBe(42);
});
it("treats PI_STREAM_IDLE_TIMEOUT_MS=0 as a watchdog disable", () => {
Bun.env.PI_STREAM_IDLE_TIMEOUT_MS = "0";
expect(getStreamIdleTimeoutMs(300_000)).toBeUndefined();
});
});
describe("getStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () => {
it("returns the per-provider fallback when env unset and idle timeout is undefined", () => {
expect(getStreamFirstEventTimeoutMs(undefined, 300_000)).toBe(300_000);
});
it("floors the first-event timeout at the per-provider fallback even when idle is shorter", () => {
expect(getStreamFirstEventTimeoutMs(50_000, 300_000)).toBe(300_000);
});
it("never undershoots the steady-state idle timeout", () => {
expect(getStreamFirstEventTimeoutMs(500_000, 300_000)).toBe(500_000);
});
it("lets PI_STREAM_FIRST_EVENT_TIMEOUT_MS override the per-provider fallback", () => {
Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "42";
expect(getStreamFirstEventTimeoutMs(undefined, 300_000)).toBe(42);
});
it("treats PI_STREAM_FIRST_EVENT_TIMEOUT_MS=0 as a watchdog disable", () => {
Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "0";
expect(getStreamFirstEventTimeoutMs(undefined, 300_000)).toBeUndefined();
});
it("falls back to the 100s global default when no fallback or env is provided", () => {
expect(getStreamFirstEventTimeoutMs()).toBe(100_000);
});
});