From 661587e110a84d2e73c8409fe91e23f369fa16f0 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 10 Jun 2026 07:34:05 +0200 Subject: [PATCH] feat(coding-agent): raised retry limits to ten with capped exponential backoff - Raised Anthropic provider retries to 10 attempts and used a shared jittered exponential backoff for each retry. - Updated coding-agent retry defaults and session delay calculation to a 500ms base with an 8,000ms jittered cap. - Added tests that verify capped ten-step backoff sequences and recovery after repeated 502 errors. --- docs/non-compaction-retry-policy.md | 20 +++--- packages/ai/CHANGELOG.md | 1 + packages/ai/src/providers/anthropic-client.ts | 4 +- packages/ai/src/providers/anthropic.ts | 10 +-- .../ai/test/anthropic-stream-timeout.test.ts | 46 ++++++++++++-- packages/coding-agent/CHANGELOG.md | 1 + .../src/config/settings-schema.ts | 4 +- .../coding-agent/src/session/agent-session.ts | 11 +++- .../test/agent-session-retry-cap.test.ts | 62 +++++++++++++++++++ 9 files changed, 135 insertions(+), 24 deletions(-) diff --git a/docs/non-compaction-retry-policy.md b/docs/non-compaction-retry-policy.md index e0ce2ea7b..ef84e5678 100644 --- a/docs/non-compaction-retry-policy.md +++ b/docs/non-compaction-retry-policy.md @@ -62,7 +62,7 @@ Flow (`#handleRetryableError`): 3. Increment `#retryAttempt`. 4. Create `#retryPromise` once (first attempt in a chain). 5. If attempt exceeded `retry.maxRetries`, emit final failure event and stop. -6. Compute base delay: `retry.baseDelayMs * 2^(attempt-1)`. +6. Compute capped jittered local delay: `min(retry.baseDelayMs * 2^(attempt-1), 8000ms) * (75–100% jitter)`. 7. For usage-limit errors, parse retry hints and call auth storage (`markUsageLimitReached(...)`); if credential switching succeeds, force delay to `0`. Otherwise wait for whichever comes first — the provider's retry-after/backoff hint, or the earliest moment a temporarily blocked sibling credential frees up (`retryAtMs` + 1s buffer) so the next attempt can pick it up. 8. If no credential switch occurred, suppress the current model selector for cooldown, try configured retry model fallback chains, and force delay to `0` on model switch. 9. If the final delay exceeds `retry.maxDelayMs` and no credential/model switch happened, emit final failure and do not sleep. @@ -87,8 +87,8 @@ Flow (`#handleRetryableError`): Settings: - `retry.enabled` (default `true`) -- `retry.maxRetries` (default `3`) -- `retry.baseDelayMs` (default `2000`) +- `retry.maxRetries` (default `10`) +- `retry.baseDelayMs` (default `500`) - `retry.maxDelayMs` (default `300000`, 5 minutes; `<= 0` disables the fail-fast cap) Attempt numbering: @@ -97,13 +97,17 @@ Attempt numbering: - start events use current attempt (1-based) - max-exceeded end event reports `attempt: this.#retryAttempt - 1` (last attempted retry count) -Backoff sequence with default settings: +Backoff sequence with default settings, before jitter: -- attempt 1: 2000 ms -- attempt 2: 4000 ms -- attempt 3: 8000 ms +- attempt 1: 500 ms +- attempt 2: 1000 ms +- attempt 3: 2000 ms +- attempt 4: 4000 ms +- attempt 5+: 8000 ms -Delay override inputs can come from parsed retry headers (`retry-after-ms`, `retry-after`, `x-ratelimit-reset-ms`, `x-ratelimit-reset`) or usage-limit backoff. Credential/model fallback switches set delay to `0`; otherwise parsed hints can extend the exponential local delay. If the computed delay is greater than `retry.maxDelayMs` and no switch succeeded, retry ends immediately with a final error instead of sleeping. +The actual local sleep is 75–100% of the nominal value, matching Anthropic-style retry jitter so concurrent sessions do not retry in lockstep. + +Delay override inputs can come from parsed retry headers (`retry-after-ms`, `retry-after`, `x-ratelimit-reset-ms`, `x-ratelimit-reset`) or usage-limit backoff. Credential/model fallback switches set delay to `0`; otherwise parsed hints can extend the capped local delay. If the computed delay is greater than `retry.maxDelayMs` and no switch succeeded, retry ends immediately with a final error instead of sleeping. ## Abort mechanics diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index a8001a865..25c5b76b0 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -20,6 +20,7 @@ - Plain provider env-var names moved to the catalog table: registry defs dropped their 48 `envKeys` literals (including the pure `$pickenv` pickers for `huggingface`/`qwen-portal`/`xai-oauth`), `getEnvApiKey` now derives those fallbacks from `CATALOG_PROVIDERS[].envVars`, and `envKeys` remains only for computed resolvers (Anthropic Foundry, Vertex ADC, Bedrock credential chains) and non-catalog providers (`kagi`, `tavily`, `parallel`, `perplexity`) - Protocol handlers are now pure `model.compat` readers — the per-request `resolve*Compat`/`detect*Compat` calls (anthropic ×11, responses ×3, completions wrappers), inline `strictResponsesPairing` host detection, the OpenCode `reasoning_content` mutation block, and all `resolvedBaseUrl` threading are gone. Compat is materialized once at model build time (`@oh-my-pi/pi-catalog` `buildModel`); the OpenCode thinking-mode quirk is a precomputed `compat.whenThinking` pointer swap, and request-time base-URL overrides only feed the HTTP client. Behavior is unchanged (the Anthropic `supportsLongCacheRetention` official-endpoint gate is folded into detection). - Providers now read baked thinking/wire metadata instead of re-parsing model ids per request: the Anthropic handler gates sampling params on `model.compat.supportsSamplingParams` and adaptive `display` on `model.thinking.supportsDisplay` (Bedrock too), adaptive effort tiers come from the baked `thinking.effortMap`, the Google `thinkingLevel` map is static, and effort-dial-less reasoners (`thinking: undefined`, e.g. `xai-oauth/grok-build`) short-circuit `resolveOpenAiReasoningEffort` without the removed `modelOmitsReasoningEffort` predicate. +- Anthropic streaming retries now use a 10-retry budget with the Anthropic-compatible 0.5s exponential backoff capped at 8s with jitter; server `retry-after` hints still win, and retryable pre-content failures such as 502s no longer stop after three tries. ### Fixed diff --git a/packages/ai/src/providers/anthropic-client.ts b/packages/ai/src/providers/anthropic-client.ts index e49c1fed9..aae4045d8 100644 --- a/packages/ai/src/providers/anthropic-client.ts +++ b/packages/ai/src/providers/anthropic-client.ts @@ -140,7 +140,7 @@ export function retryDelayFromHeaders(headers: Headers | undefined): number | un return undefined; } -function defaultRetryDelayMs(attempt: number): number { +export function calculateAnthropicRetryDelayMs(attempt: number): number { const sleepSeconds = Math.min(INITIAL_RETRY_DELAY_S * 2 ** attempt, MAX_RETRY_DELAY_S); const jitter = 1 - Math.random() * 0.25; return sleepSeconds * jitter * 1000; @@ -310,7 +310,7 @@ export class AnthropicMessagesClient implements AnthropicMessagesClientLike { responseHeaders: Headers | undefined, signal: AbortSignal | undefined, ): Promise { - const delayMs = retryDelayFromHeaders(responseHeaders) ?? defaultRetryDelayMs(attempt); + const delayMs = retryDelayFromHeaders(responseHeaders) ?? calculateAnthropicRetryDelayMs(attempt); try { await scheduler.wait(delayMs, { signal }); } catch { diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 2e868d8b2..314ba056f 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -63,6 +63,7 @@ import { type AnthropicFetchOptions, AnthropicMessagesClient, type AnthropicMessagesClientLike, + calculateAnthropicRetryDelayMs, retryDelayFromHeaders, } from "./anthropic-client"; import type { @@ -1370,8 +1371,7 @@ async function* observeDecodedAnthropicSdkEvents( } } -const PROVIDER_MAX_RETRIES = 3; -const PROVIDER_BASE_DELAY_MS = 2000; +const PROVIDER_MAX_RETRIES = 10; /** Transient stream corruption errors where the response was truncated mid-JSON. */ function isTransientStreamParseError(error: unknown): boolean { @@ -1680,8 +1680,8 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = ( const { requestSignal } = activeAbortTracker; // The provider loop owns retries: pin the client's internal retry loop // to zero even when no watchdog timeout is configured (the helper only - // pins it alongside a timeout; the client default of 5 would otherwise - // multiply with PROVIDER_MAX_RETRIES into up to 24 wire attempts). + // pins it alongside a timeout; a client retry budget of 5 would otherwise + // multiply with PROVIDER_MAX_RETRIES into up to 66 wire attempts). const requestOptions = { ...createSdkStreamRequestOptions(requestSignal, requestTimeoutMs), maxRetries: 0 }; const anthropicRequest: unknown = isOAuthToken && client.beta @@ -2136,7 +2136,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = ( throw streamFailure; } providerRetryAttempt++; - const backoffDelayMs = PROVIDER_BASE_DELAY_MS * 2 ** (providerRetryAttempt - 1); + const backoffDelayMs = calculateAnthropicRetryDelayMs(providerRetryAttempt - 1); // Honor the server's retry hint (`retry-after-ms`/`retry-after`) on // 429/529-style failures: retrying sooner than the server asked is a // guaranteed failure that just burns the retry budget. diff --git a/packages/ai/test/anthropic-stream-timeout.test.ts b/packages/ai/test/anthropic-stream-timeout.test.ts index 230118e35..14feb10c5 100644 --- a/packages/ai/test/anthropic-stream-timeout.test.ts +++ b/packages/ai/test/anthropic-stream-timeout.test.ts @@ -204,7 +204,7 @@ describe("anthropic first-event timeout retries", () => { }) as never; }) as unknown as AnthropicMessagesClientLike["messages"]["create"]; const client = { messages: { create } } as AnthropicMessagesClientLike; - const providerRetryWait = vi.fn(async () => {}); + const providerRetryWait = vi.fn(async (_delayMs: number, _signal: AbortSignal | undefined) => {}); const resultPromise = streamAnthropic(model, context, { client, @@ -226,7 +226,13 @@ describe("anthropic first-event timeout retries", () => { ); expect(attempt).toBe(2); - expect(providerRetryWait).toHaveBeenCalledWith(2000, undefined); + expect(providerRetryWait).toHaveBeenCalledTimes(1); + const retryDelayMs = providerRetryWait.mock.calls[0]?.[0]; + if (typeof retryDelayMs !== "number") { + throw new Error("Expected provider retry wait delay"); + } + expect(retryDelayMs).toBeGreaterThanOrEqual(375); + expect(retryDelayMs).toBeLessThanOrEqual(500); expect(requestTimeouts).toEqual([1, 1]); expect(requestMaxRetries).toEqual([0, 0]); expect(result.stopReason).toBe("stop"); @@ -336,10 +342,10 @@ describe("anthropic first-event timeout retries", () => { providerRetryWait, }).result(); - expect(attempt).toBe(4); - expect(providerRetryWait).toHaveBeenCalledTimes(3); - expect(requestTimeouts).toEqual([1, 1, 1, 1]); - expect(requestMaxRetries).toEqual([0, 0, 0, 0]); + expect(attempt).toBe(11); + expect(providerRetryWait).toHaveBeenCalledTimes(10); + expect(requestTimeouts).toEqual(new Array(11).fill(1)); + expect(requestMaxRetries).toEqual(new Array(11).fill(0)); expect(result.stopReason).toBe("error"); expect(result.errorMessage).toBe("Anthropic stream timed out while waiting for the first event"); }); @@ -453,4 +459,32 @@ describe("anthropic provider retry delays", () => { expect(result.stopReason).toBe("stop"); expect(result.content).toEqual([{ type: "text", text: "after backoff" }]); }); + + it("retries 502s ten times with Anthropic-style capped backoff", async () => { + vi.spyOn(Math, "random").mockReturnValue(0); + let attempt = 0; + const create = ((_body: unknown, requestOptions?: { signal?: AbortSignal }) => { + attempt += 1; + if (attempt <= 10) { + return createRejectedAnthropicRequest( + new AnthropicApiError(502, "502 Bad Gateway", new Headers()), + ) as never; + } + return createAnthropicMockStream({ + signal: requestOptions?.signal, + events: createSuccessfulAnthropicEvents("recovered from 502"), + }) as never; + }) as unknown as AnthropicMessagesClientLike["messages"]["create"]; + const client = { messages: { create } } as AnthropicMessagesClientLike; + const providerRetryWait = vi.fn(async (_delayMs: number, _signal: AbortSignal | undefined) => {}); + + const result = await streamAnthropic(model, context, { client, providerRetryWait }).result(); + + expect(attempt).toBe(11); + expect(providerRetryWait.mock.calls.map(call => call[0])).toEqual([ + 500, 1000, 2000, 4000, 8000, 8000, 8000, 8000, 8000, 8000, + ]); + expect(result.stopReason).toBe("stop"); + expect(result.content).toEqual([{ type: "text", text: "recovered from 502" }]); + }); }); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 5c4bc2bea..08aa7cd91 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -46,6 +46,7 @@ - `TranscriptContainer` assembles the transcript incrementally: each block's render is reference-compared and its stripped contribution, separator, and row placement are reused when unchanged, with the persistent row array truncated and re-pushed only from the first divergent block; the leading byte-identical row count is reported to the renderer through pi-tui's new `RenderStablePrefix` seam so off-screen transcript rows are no longer re-rendered, re-prepared, or re-audited every frame. Block components became reference-stable to make this effective: `UserMessageComponent` memoizes its OSC 133 zone wrapping, `WelcomeComponent` and `DynamicBorder` cache their renders, and dashboards copy before padding (render results are `readonly` under the new pi-tui contract) - A live block whose trailing row grows in place as a visible prefix (token streaming into the cursor line) is now commit-safe through its full body instead of being held back by the volatile-tail margin — the growing row itself is the block's last and can never commit while it remains last, so a streaming reply's scrolled-off head reaches native scrollback (tmux pane history) mid-stream - Rewrote the bash tool's coreutils guidance (tool prompt and system prompt) around an explicit litmus: pipelines that compute a new fact (`wc -l`, `sort | uniq -c`, `comm`, `diff`) are legitimate bash, while commands that merely move, page, or trim bytes a dedicated tool can fetch remain banned — output trimming destroys data the `artifact://` capture would have saved. +- Default API auto-retries now use 10 attempts with a 500ms Anthropic-style exponential backoff capped at 8s with jitter, so transient 502/gateway failures get a longer retry budget without multi-minute local sleeps. ### Fixed diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index dbcf512f0..7e7621a65 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -879,7 +879,7 @@ export const SETTINGS_SCHEMA = { "retry.maxRetries": { type: "number", - default: 3, + default: 10, ui: { tab: "model", label: "Retry Attempts", @@ -894,7 +894,7 @@ export const SETTINGS_SCHEMA = { }, }, - "retry.baseDelayMs": { type: "number", default: 2000 }, + "retry.baseDelayMs": { type: "number", default: 500 }, "retry.maxDelayMs": { type: "number", default: 5 * 60 * 1000, diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index f296ca5d5..a86fce63f 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -288,6 +288,15 @@ export type AgentSessionEventListener = (event: AgentSessionEvent) => void; export type AsyncJobSnapshotItem = Pick; const EMPTY_STOP_MAX_RETRIES = 3; +const RETRY_BACKOFF_MAX_DELAY_MS = 8_000; +const RETRY_BACKOFF_JITTER_RATIO = 0.25; + +function calculateRetryBackoffDelayMs(baseDelayMs: number, attempt: number): number { + const cappedDelayMs = Math.min(Math.max(0, baseDelayMs) * 2 ** Math.max(0, attempt - 1), RETRY_BACKOFF_MAX_DELAY_MS); + const jitter = 1 - Math.random() * RETRY_BACKOFF_JITTER_RATIO; + return cappedDelayMs * jitter; +} + /** * Slack added past a sibling credential's block expiry before retrying, so * the next getApiKey lands after the block has actually lapsed. @@ -8313,7 +8322,7 @@ export class AgentSession { const errorMessage = message.errorMessage || "Unknown error"; const parsedRetryAfterMs = this.#parseRetryAfterMsFromError(errorMessage); - let delayMs = retrySettings.baseDelayMs * 2 ** (this.#retryAttempt - 1); + let delayMs = calculateRetryBackoffDelayMs(retrySettings.baseDelayMs, this.#retryAttempt); let switchedCredential = false; let switchedModel = false; // Set when a usage-limit error pinned the wait to credential diff --git a/packages/coding-agent/test/agent-session-retry-cap.test.ts b/packages/coding-agent/test/agent-session-retry-cap.test.ts index 69123439f..0461c0f43 100644 --- a/packages/coding-agent/test/agent-session-retry-cap.test.ts +++ b/packages/coding-agent/test/agent-session-retry-cap.test.ts @@ -479,4 +479,66 @@ describe("AgentSession retry delay cap", () => { expect(last.stopReason).toBe("stop"); expect(last.content).toContainEqual({ type: "text", text: "recovered after generic gateway upstream error" }); }); + + it("defaults 502 auto-retry to ten capped backoff attempts", async () => { + const model = getBundledModel("anthropic", "claude-sonnet-4-5"); + if (!model) { + throw new Error("Expected bundled Anthropic test model to exist"); + } + + const mock = createMockModel(); + let attempts = 0; + const agent = new Agent({ + getApiKey: provider => `${provider}-test-key`, + initialState: { + model, + systemPrompt: ["Test"], + tools: [], + messages: [], + }, + streamFn: (requestedModel, context, options) => { + attempts += 1; + mock.push( + attempts <= 10 + ? { throw: "502 Bad Gateway upstream_error" } + : { content: ["recovered after default 502 retry budget"] }, + ); + return mock.stream(requestedModel, context, options); + }, + }); + + const settings = Settings.isolated({ "compaction.enabled": false }); + settings.setModelRole("default", `${model.provider}/${model.id}`); + + session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings, + modelRegistry, + }); + + vi.spyOn(Math, "random").mockReturnValue(0); + vi.spyOn(scheduler, "wait").mockResolvedValue(undefined); + const retryStartEvents: AutoRetryStartEvent[] = []; + const retryEndEvents: AutoRetryEndEvent[] = []; + session.subscribe(event => { + if (event.type === "auto_retry_start") retryStartEvents.push(event); + if (event.type === "auto_retry_end") retryEndEvents.push(event); + }); + + await session.prompt("Trigger repeated 502s"); + await session.waitForIdle(); + + expect(attempts).toBe(11); + expect(retryStartEvents).toHaveLength(10); + expect(retryStartEvents.map(event => event.maxAttempts)).toEqual(new Array(10).fill(10)); + expect(retryStartEvents.map(event => event.delayMs)).toEqual([ + 500, 1000, 2000, 4000, 8000, 8000, 8000, 8000, 8000, 8000, + ]); + expect(retryEndEvents).toHaveLength(1); + expect(retryEndEvents[0]).toMatchObject({ success: true, attempt: 10 }); + const last = lastAssistant(session); + expect(last.stopReason).toBe("stop"); + expect(last.content).toContainEqual({ type: "text", text: "recovered after default 502 retry budget" }); + }); });