From f3ad5ede6b08d70c4fe7d0ff7e3544bf71b4841d Mon Sep 17 00:00:00 2001 From: Magicien <162632566+lederniermagicien@users.noreply.github.com> Date: Tue, 4 Aug 2026 21:51:35 +0100 Subject: [PATCH] fix(ai): scope the Copilot 8-attempt retry budget to model flaps MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The 3 -> 8 bump in callWithCopilotModelRetry was shared by the generic retryable branch, so a persistent status-less transport blip on Copilot would ramp across 8 attempts (~11.2s of dead time) instead of the 3 it took before, and a repeated Retry-After 429 could stretch the same way on top of the transport's own fetchWithRetry budget. Derive the budget from the failure kind: model-availability 400s keep the 8-attempt reroll, everything else caps at the previous 3. Also read COPILOT_TRANSIENT_MODEL_CODES with Object.hasOwn — `code` is provider-controlled, so a 400 body whose code was `__proto__` or `toString` classified as transient through the prototype chain. --- packages/ai/CHANGELOG.md | 2 +- packages/ai/src/error/flags.ts | 4 +- packages/ai/src/utils/retry.ts | 10 ++++- packages/ai/test/copilot-retry.test.ts | 60 ++++++++++++++++++++++---- 4 files changed, 64 insertions(+), 12 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 892fb1f30..15c9f504e 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed GitHub Copilot requests failing with a raw `HTTP 400 model_not_available_for_integrator` on roughly half of all turns for recently rolled-out models. Copilot's fleet is not uniform — part of it rejects models that `/models` advertises on the same host — and the transient classifier matched only the older `model_not_supported` code at a fixed envelope depth, so these rejections surfaced as terminal errors instead of entering the existing retry path. Model-availability 400s are now recognized at any envelope depth and retried on a flat delay, with the OpenAI-transport retry budget raised from 3 to 8 attempts. +- Fixed GitHub Copilot requests failing with a raw `HTTP 400 model_not_available_for_integrator` on roughly half of all turns for recently rolled-out models. Copilot's fleet is not uniform — part of it rejects models that `/models` advertises on the same host — and the transient classifier matched only the older `model_not_supported` code at a fixed envelope depth, so these rejections surfaced as terminal errors instead of entering the existing retry path. Model-availability 400s are now recognized at any envelope depth and rerolled on a flat delay with a dedicated 8-attempt budget on the OpenAI transports; every other retryable failure keeps its previous backoff and attempt count. ## [17.2.7] - 2026-08-03 diff --git a/packages/ai/src/error/flags.ts b/packages/ai/src/error/flags.ts index 6df73686b..4e348493b 100644 --- a/packages/ai/src/error/flags.ts +++ b/packages/ai/src/error/flags.ts @@ -493,7 +493,9 @@ function providerErrorCode(error: object): string | undefined { export function isCopilotTransientModelError(error: unknown): boolean { if (!error || typeof error !== "object" || status(error) !== 400) return false; const code = providerErrorCode(error); - if (code !== undefined && COPILOT_TRANSIENT_MODEL_CODES[code]) return true; + // `Object.hasOwn`, not a bare index: `code` is provider-controlled, and a + // prototype key (`__proto__`, `toString`, …) would otherwise read truthy. + if (code !== undefined && Object.hasOwn(COPILOT_TRANSIENT_MODEL_CODES, code)) return true; const message: unknown = "message" in error ? error.message : undefined; return typeof message === "string" && COPILOT_MODEL_UNAVAILABLE_PATTERN.test(message); } diff --git a/packages/ai/src/utils/retry.ts b/packages/ai/src/utils/retry.ts index 8d786b94a..fce896bce 100644 --- a/packages/ai/src/utils/retry.ts +++ b/packages/ai/src/utils/retry.ts @@ -15,6 +15,9 @@ export { isCopilotTransientModelError }; // delay keep the residual near 5% at p=0.7 and under 1% at p=0.5, bounded at // ~2.8s of dead time in the pathological case. const COPILOT_MODEL_RETRY_MAX_ATTEMPTS = 8; +// Transport blips and status-bearing failures keep the pre-flap budget: they are +// not coin flips, so a longer ramp only delays surfacing a persistent fault. +const COPILOT_GENERIC_RETRY_MAX_ATTEMPTS = 3; const COPILOT_MODEL_RETRY_BASE_DELAY_MS = 400; /** Longest server-requested backoff we are willing to sit out before giving up. */ const COPILOT_RETRY_AFTER_MAX_WAIT_MS = 30_000; @@ -46,7 +49,12 @@ export async function callWithCopilotModelRetry( if (options.signal?.aborted) throw error; const transientModelError = isCopilotTransientModelError(error); if (!transientModelError && !isRetryableError(error)) throw error; - if (attempt === COPILOT_MODEL_RETRY_MAX_ATTEMPTS - 1) break; + // Budget is per failure kind, counted over attempts already spent: the + // eight-attempt allowance only covers the cheap model-availability reroll. + const maxAttempts = transientModelError + ? COPILOT_MODEL_RETRY_MAX_ATTEMPTS + : COPILOT_GENERIC_RETRY_MAX_ATTEMPTS; + if (attempt >= maxAttempts - 1) break; // Reroll the model flap on a flat delay: a ramp only adds dead time to a // coin flip the next attempt is equally likely to win. Generic retryable // failures (429/5xx/transport) keep the linear backoff below. diff --git a/packages/ai/test/copilot-retry.test.ts b/packages/ai/test/copilot-retry.test.ts index e0f892438..850d26506 100644 --- a/packages/ai/test/copilot-retry.test.ts +++ b/packages/ai/test/copilot-retry.test.ts @@ -12,13 +12,18 @@ type ErrorShape = { code?: string; error?: { code?: string; message?: string } | { error: { code?: string; message?: string } }; message: string; + headers?: Record; }; -function copilotError({ status, code, error, message }: ErrorShape): Error { +function copilotError({ status, code, error, message, headers }: ErrorShape): Error { const err = new Error(message); - (err as unknown as ErrorShape).status = status; - if (code !== undefined) (err as unknown as ErrorShape).code = code; - if (error !== undefined) (err as unknown as ErrorShape).error = error; + // Single sanctioned assertion point: `Error` carries no provider fields, and + // every test reads them back through the real classifier. + const shaped = err as unknown as ErrorShape; + shaped.status = status; + if (code !== undefined) shaped.code = code; + if (error !== undefined) shaped.error = error; + if (headers !== undefined) shaped.headers = headers; return err; } @@ -75,6 +80,13 @@ describe("isCopilotTransientModelError", () => { expect(isCopilotTransientModelError(err)).toBe(false); }); + it("does not match 400 codes that collide with Object.prototype keys", () => { + for (const code of ["__proto__", "constructor", "toString", "hasOwnProperty"]) { + const err = copilotError({ status: 400, code, message: "bad request" }); + expect(isCopilotTransientModelError(err)).toBe(false); + } + }); + it("does not match 401/403/500 regardless of code", () => { for (const status of [401, 403, 500]) { const err = copilotError({ @@ -176,9 +188,7 @@ describe("callWithCopilotModelRetry", () => { async () => { calls += 1; if (calls === 1) { - const err = copilotError({ status: 429, message: "rate limited" }); - (err as unknown as { headers: Record }).headers = { "retry-after": "0.01" }; - throw err; + throw copilotError({ status: 429, message: "rate limited", headers: { "retry-after": "0.01" } }); } return "ok" as const; }, @@ -188,6 +198,21 @@ describe("callWithCopilotModelRetry", () => { expect(calls).toBe(2); }); + it("does not stretch a persistent Retry-After 429 across the flap budget", async () => { + let calls = 0; + const err = copilotError({ status: 429, message: "rate limited", headers: { "retry-after": "0.01" } }); + await expect( + callWithCopilotModelRetry( + async () => { + calls += 1; + throw err; + }, + { provider: "github-copilot", retryBaseDelayMs: 0 }, + ), + ).rejects.toBe(err); + expect(calls).toBe(3); + }); + it("still retries status-less transport blips with the linear backoff", async () => { let calls = 0; const result = await callWithCopilotModelRetry( @@ -206,7 +231,24 @@ describe("callWithCopilotModelRetry", () => { expect(calls).toBe(2); }); - it("keeps the model-flap delay flat while generic retryable failures ramp", async () => { + it("caps persistent generic retryable failures at the pre-flap budget", async () => { + let calls = 0; + const err = new Error( + 'HTTP2StreamReset fetching "https://api.example.com/x". For more information, pass `verbose: true` in the second argument to fetch()', + ); + await expect( + callWithCopilotModelRetry( + async () => { + calls += 1; + throw err; + }, + { provider: "github-copilot", retryBaseDelayMs: 0 }, + ), + ).rejects.toBe(err); + expect(calls).toBe(3); + }); + + it("keeps the flat delay and the larger budget scoped to model flaps", async () => { const flatWaits: number[] = []; const rampWaits: number[] = []; const record = (into: number[]) => { @@ -242,7 +284,7 @@ describe("callWithCopilotModelRetry", () => { vi.restoreAllMocks(); expect(flatWaits).toEqual([100, 100, 100, 100, 100, 100, 100]); - expect(rampWaits).toEqual([100, 200, 300, 400, 500, 600, 700]); + expect(rampWaits).toEqual([100, 200]); }); it("stops retrying when the caller aborts during backoff", async () => {