fix(ai): retry GitHub Copilot transient model_not_supported 400s

GitHub Copilot intermittently returns `HTTP 400 model_not_supported`
for preview models (gpt-5.3-codex, gpt-5.4, gpt-5.4-mini, ...) on OAuth
clients other than VS Code, even when `/models` reports the model as
enabled. Root cause is a per-OAuth-client rollout gap across Copilot's
responses backend; repeating the identical request typically lands on
a backend that has the model. See opencode#13313.

- Add `isCopilotTransientModelError` and `callWithCopilotModelRetry`
  in `utils/retry` (3 attempts, linear backoff, abort-aware, no-op
  for non-Copilot providers).
- Wrap `client.responses.create` in `openai-responses` and the
  initial completions stream in `openai-completions` with the retry.
- Extend Anthropic `isProviderRetryableError` to treat Copilot
  transient model errors as provider-retryable.
- Rename `rewriteCopilotAuthError` to `rewriteCopilotError` and add
  a 400 `model_not_supported` rewrite surfacing actionable guidance
  (retry, switch to a GA model, or run from VS Code) after retries
  exhaust.
- Rename test file accordingly and add a dedicated retry unit test.
This commit is contained in:
Dimitar Ganev
2026-04-16 10:30:16 +03:00
committed by can1357
parent 2921332c44
commit af50d3bfa4
9 changed files with 303 additions and 47 deletions
+8 -1
View File
@@ -2,6 +2,14 @@
## [Unreleased]
### Added
- Added `isCopilotTransientModelError()` and `callWithCopilotModelRetry()` helpers in `utils/retry` that detect GitHub Copilot's intermittent `HTTP 400 model_not_supported` responses for preview models (`gpt-5.3-codex`, `gpt-5.4`, `gpt-5.4-mini`, ...) and retry the request up to three times with backoff. OpenAI Responses, OpenAI Completions, and Anthropic provider paths now participate in this retry when the model is served through Copilot.
### Changed
- Renamed `rewriteCopilotAuthError` to `rewriteCopilotError` and extended it to rewrite `HTTP 400 model_not_supported` after retries are exhausted with guidance about Copilot's OAuth-client-specific rollout gap (see opencode#13313).
## [14.1.3] - 2026-04-17
### Fixed
@@ -9,7 +17,6 @@
- Preserved user-provided `session_id` and `x-client-request-id` headers in OpenAI Responses requests instead of overriding them with automatic session-derived values
- Stopped sending `session_id` and `x-client-request-id` headers for OpenAI Responses requests when `cacheRetention` is set to `none`
- Fixed direct OpenAI Responses requests to send `session_id` and `x-client-request-id` from the same session-derived value as `prompt_cache_key`, improving prompt cache affinity for append-only sessions
## [14.1.1] - 2026-04-14
### Added
+4 -2
View File
@@ -34,10 +34,11 @@ import type {
import { isAnthropicOAuthToken, normalizeToolCallId, resolveCacheRetention } from "../utils";
import { createAbortSourceTracker } from "../utils/abort";
import { AssistantMessageEventStream } from "../utils/event-stream";
import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotAuthError } from "../utils/http-inspector";
import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
import { createWatchdog, getStreamFirstEventTimeoutMs } from "../utils/idle-iterator";
import { parseStreamingJson } from "../utils/json-parse";
import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
import { isCopilotTransientModelError } from "../utils/retry";
import {
buildCopilotDynamicHeaders,
hasCopilotVisionInput,
@@ -627,6 +628,7 @@ function isProviderRetryableStreamEnvelopeError(error: unknown): boolean {
export function isProviderRetryableError(error: unknown): boolean {
if (!(error instanceof Error)) return false;
if (isCopilotTransientModelError(error)) return true;
const msg = error.message.toLowerCase();
return (
/rate.?limit|too many requests|overloaded|service.?unavailable|internal_error|stream error.*received from peer|1302|timed?\s*out while waiting for the first event|timeout waiting for first/i.test(
@@ -994,7 +996,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
const firstEventTimeoutError = activeAbortTracker.getLocalAbortReason();
output.stopReason = activeAbortTracker.wasCallerAbort() ? "aborted" : "error";
output.errorMessage = firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
output.errorMessage = rewriteCopilotAuthError(output.errorMessage, error, model.provider);
output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
output.duration = Date.now() - startTime;
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
stream.push({ type: "error", reason: output.stopReason, error: output });
@@ -35,7 +35,7 @@ import {
type CapturedHttpErrorResponse,
finalizeErrorMessage,
type RawHttpRequestDump,
rewriteCopilotAuthError,
rewriteCopilotError,
} from "../utils/http-inspector";
import {
createWatchdog,
@@ -46,7 +46,7 @@ import {
import { parseStreamingJson } from "../utils/json-parse";
import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
import { getKimiCommonHeaders } from "../utils/oauth/kimi";
import { extractHttpStatusFromError } from "../utils/retry";
import { callWithCopilotModelRetry, extractHttpStatusFromError } from "../utils/retry";
import { adaptSchemaForStrict, NO_STRICT } from "../utils/schema";
import { mapToOpenAICompletionsToolChoice } from "../utils/tool-choice";
import {
@@ -245,7 +245,10 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
};
let openaiStream: AsyncIterable<ChatCompletionChunk>;
try {
openaiStream = await createCompletionsStream();
openaiStream = await callWithCopilotModelRetry(() => createCompletionsStream(), {
provider: model.provider,
signal: requestSignal,
});
} catch (error) {
const capturedErrorResponse = getCapturedErrorResponse();
if (!shouldRetryWithoutStrictTools(error, capturedErrorResponse, appliedToolStrictMode, context.tools)) {
@@ -552,7 +555,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
// Some providers via OpenRouter include extra details here.
const rawMetadata = (error as { error?: { metadata?: { raw?: string } } })?.error?.metadata?.raw;
if (rawMetadata) output.errorMessage += `\n${rawMetadata}`;
output.errorMessage = rewriteCopilotAuthError(output.errorMessage, error, model.provider);
output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
output.duration = Date.now() - startTime;
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
stream.push({ type: "error", reason: output.stopReason, error: output });
@@ -30,7 +30,7 @@ import {
} from "../utils";
import { createAbortSourceTracker } from "../utils/abort";
import { AssistantMessageEventStream } from "../utils/event-stream";
import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotAuthError } from "../utils/http-inspector";
import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
import {
createWatchdog,
getOpenAIStreamIdleTimeoutMs,
@@ -38,6 +38,7 @@ import {
iterateWithIdleTimeout,
} from "../utils/idle-iterator";
import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
import { callWithCopilotModelRetry } from "../utils/retry";
import { adaptSchemaForStrict, NO_STRICT } from "../utils/schema";
import { mapToOpenAIResponsesToolChoice } from "../utils/tool-choice";
import {
@@ -192,7 +193,10 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
url: `${baseUrl ?? "https://api.openai.com/v1"}/responses`,
body: params,
};
const openaiStream = await client.responses.create(params, { signal: requestSignal });
const openaiStream = await callWithCopilotModelRetry(
() => client.responses.create(params, { signal: requestSignal }),
{ provider: model.provider, signal: requestSignal },
);
const firstEventWatchdog = createWatchdog(
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
() => abortTracker.abortLocally(firstEventTimeoutAbortError),
@@ -246,7 +250,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
const firstEventTimeoutError = abortTracker.getLocalAbortReason();
output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
output.errorMessage = firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
output.errorMessage = rewriteCopilotAuthError(output.errorMessage, error, model.provider);
output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
output.duration = Date.now() - startTime;
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
stream.push({ type: "error", reason: output.stopReason, error: output });
+9 -2
View File
@@ -1,6 +1,6 @@
import * as path from "node:path";
import { getLogsDir } from "@oh-my-pi/pi-utils";
import { extractHttpStatusFromError } from "./retry.js";
import { extractHttpStatusFromError, isCopilotTransientModelError } from "./retry.js";
import { formatErrorMessageWithRetryAfter } from "./retry-after.js";
export type RawHttpRequestDump = {
@@ -75,11 +75,15 @@ export function withHttpStatus(error: unknown, status: number): Error {
* Rewrite error message for GitHub Copilot request failures.
* Must run AFTER finalizeErrorMessage since it replaces the message entirely.
*
* 400 `model_not_supported` = Copilot routing rollout gap for our OAuth client.
* A preview model (gpt-5.3-codex, gpt-5.4*, ...) flaps between 200 and
* 400 because only some of Copilot's backends have the model. After the
* in-request retry exhausts, surface guidance rather than the raw error.
* 401 = token invalid/expired → credential removal is safe, prompt re-login.
* 403 = token valid but access denied (plan, model policy, org restriction) →
* do NOT reuse the auth-failed string (which triggers credential removal).
*/
export function rewriteCopilotAuthError(errorMessage: string, error: unknown, provider: string): string {
export function rewriteCopilotError(errorMessage: string, error: unknown, provider: string): string {
if (provider !== "github-copilot") return errorMessage;
const status = extractHttpStatusFromError(error);
if (status === 401) {
@@ -88,6 +92,9 @@ export function rewriteCopilotAuthError(errorMessage: string, error: unknown, pr
if (status === 403) {
return `GitHub Copilot access denied (HTTP 403). Your account may not have access to this model or feature. Check your Copilot plan or model policy settings.`;
}
if (isCopilotTransientModelError(error)) {
return `GitHub Copilot rejected this model (HTTP 400 model_not_supported) after retries. This is a known intermittent rollout gap for preview models on OAuth clients other than VS Code. Try again in a few seconds, switch to a GA model (gpt-5-mini, gpt-5.2), or run this model from VS Code.`;
}
return errorMessage;
}
+63
View File
@@ -1,3 +1,5 @@
import { abortableSleep } from "@oh-my-pi/pi-utils";
type ErrorLike = {
message?: string;
name?: string;
@@ -5,6 +7,8 @@ type ErrorLike = {
statusCode?: number;
response?: { status?: number };
cause?: unknown;
code?: unknown;
error?: { code?: unknown } | null;
};
const TRANSIENT_MESSAGE_PATTERN =
@@ -91,3 +95,62 @@ function extractStatusFromMessage(message: string): number | undefined {
return undefined;
}
/**
* GitHub Copilot intermittently rejects preview models (gpt-5.3-codex,
* gpt-5.4, gpt-5.4-mini, ...) with HTTP 400 `model_not_supported`, even
* though the model is listed as enabled on the user's account via `/models`.
*
* Root cause: Copilot's request-routing backend is rolled out per OAuth
* client. Our OAuth client id is shared with opencode; VS Code uses its own
* client and sees full availability, so the same account may succeed in VS
* Code and flap between 200/400 here. See opencode#13313 and copilot-cli#2597.
*
* Retrying the identical request 2-3 times almost always lands on a backend
* that has the model, so we wrap the initial request with a short retry loop.
*/
export function isCopilotTransientModelError(error: unknown): boolean {
if (extractHttpStatusFromError(error) !== 400) return false;
return extractErrorCode(error) === "model_not_supported";
}
function extractErrorCode(error: unknown): string | undefined {
if (!error || typeof error !== "object") return undefined;
const info = error as ErrorLike;
if (typeof info.code === "string") return info.code;
const nested = info.error;
if (nested && typeof nested === "object" && typeof nested.code === "string") {
return nested.code;
}
return undefined;
}
const COPILOT_MODEL_RETRY_MAX_ATTEMPTS = 3;
const COPILOT_MODEL_RETRY_BASE_DELAY_MS = 400;
/**
* Wrap an initial Copilot request so transient `model_not_supported` 400s are
* retried a small number of times. No-op for non-Copilot providers.
*
* The callback **MUST** create a fresh in-flight request each invocation — a
* once-consumed AsyncIterable cannot be re-iterated.
*/
export async function callWithCopilotModelRetry<T>(
fn: () => Promise<T>,
options: { provider: string; signal?: AbortSignal },
): Promise<T> {
if (options.provider !== "github-copilot") return fn();
let lastError: unknown;
for (let attempt = 0; attempt < COPILOT_MODEL_RETRY_MAX_ATTEMPTS; attempt++) {
try {
return await fn();
} catch (error) {
lastError = error;
if (!isCopilotTransientModelError(error)) throw error;
if (attempt === COPILOT_MODEL_RETRY_MAX_ATTEMPTS - 1) break;
await abortableSleep(COPILOT_MODEL_RETRY_BASE_DELAY_MS * (attempt + 1), options.signal);
}
}
throw lastError;
}
+146
View File
@@ -0,0 +1,146 @@
import { describe, expect, it } from "bun:test";
import { callWithCopilotModelRetry, isCopilotTransientModelError, isRetryableError } from "@oh-my-pi/pi-ai/utils/retry";
type ErrorShape = { status: number; code?: string; error?: { code?: string; message?: string }; message: string };
function copilotError({ status, code, error, message }: ErrorShape): Error {
const err = new Error(message);
(err as unknown as ErrorShape).status = status;
if (code !== undefined) (err as unknown as ErrorShape).code = code;
if (error !== undefined) (err as unknown as ErrorShape).error = error;
return err;
}
describe("isCopilotTransientModelError", () => {
it("matches 400 with top-level code=model_not_supported", () => {
const err = copilotError({
status: 400,
code: "model_not_supported",
message: "400 The requested model is not supported.",
});
expect(isCopilotTransientModelError(err)).toBe(true);
});
it("matches 400 with nested error.code=model_not_supported (OpenAI SDK shape)", () => {
const err = copilotError({
status: 400,
error: { code: "model_not_supported", message: "The requested model is not supported." },
message: "400 The requested model is not supported.",
});
expect(isCopilotTransientModelError(err)).toBe(true);
});
it("does not match other 400 codes", () => {
const err = copilotError({
status: 400,
code: "invalid_request_body",
message: "Unsupported value: 'minimal'",
});
expect(isCopilotTransientModelError(err)).toBe(false);
});
it("does not match 401/403/500 regardless of code", () => {
for (const status of [401, 403, 500]) {
const err = copilotError({
status,
code: "model_not_supported",
message: `${status} error`,
});
expect(isCopilotTransientModelError(err)).toBe(false);
}
});
it("does not match errors without a status", () => {
expect(isCopilotTransientModelError(new Error("oops"))).toBe(false);
expect(isCopilotTransientModelError("not an object")).toBe(false);
expect(isCopilotTransientModelError(null)).toBe(false);
});
});
describe("callWithCopilotModelRetry", () => {
it("is a no-op for non-github-copilot providers", async () => {
let calls = 0;
const err = copilotError({ status: 400, code: "model_not_supported", message: "nope" });
await expect(
callWithCopilotModelRetry(
async () => {
calls += 1;
throw err;
},
{ provider: "openai" },
),
).rejects.toBe(err);
expect(calls).toBe(1);
});
it("retries up to 3 attempts for Copilot transient errors and eventually throws the last error", async () => {
let calls = 0;
const err = copilotError({ status: 400, code: "model_not_supported", message: "transient" });
await expect(
callWithCopilotModelRetry(
async () => {
calls += 1;
throw err;
},
{ provider: "github-copilot" },
),
).rejects.toBe(err);
expect(calls).toBe(3);
});
it("succeeds on the second attempt when the first is transient", async () => {
let calls = 0;
const result = await callWithCopilotModelRetry(
async () => {
calls += 1;
if (calls === 1) {
throw copilotError({ status: 400, code: "model_not_supported", message: "transient" });
}
return "ok" as const;
},
{ provider: "github-copilot" },
);
expect(result).toBe("ok");
expect(calls).toBe(2);
});
it("does not retry non-transient Copilot errors", async () => {
let calls = 0;
const err = copilotError({ status: 401, code: "unauthorized", message: "auth failed" });
await expect(
callWithCopilotModelRetry(
async () => {
calls += 1;
throw err;
},
{ provider: "github-copilot" },
),
).rejects.toBe(err);
expect(calls).toBe(1);
});
it("stops retrying when the caller aborts during backoff", async () => {
const controller = new AbortController();
controller.abort();
let calls = 0;
await expect(
callWithCopilotModelRetry(
async () => {
calls += 1;
throw copilotError({ status: 400, code: "model_not_supported", message: "transient" });
},
{ provider: "github-copilot", signal: controller.signal },
),
).rejects.toBeDefined();
// fn runs once; abortableSleep rejects before a second attempt.
expect(calls).toBe(1);
});
});
describe("isRetryableError does not treat 4xx as retryable", () => {
// Regression guard: the new Copilot carveout must not leak into the generic predicate.
it("returns false for Copilot transient model errors", () => {
const err = copilotError({ status: 400, code: "model_not_supported", message: "x" });
expect(isRetryableError(err)).toBe(false);
});
});
@@ -1,35 +0,0 @@
import { describe, expect, it } from "bun:test";
import { rewriteCopilotAuthError } from "../src/utils/http-inspector";
function errorWithStatus(status: number): Error {
const err = new Error(`${status} Unauthorized`);
(err as any).status = status;
return err;
}
describe("rewriteCopilotAuthError", () => {
it("returns original message for non-copilot providers", () => {
const err = errorWithStatus(401);
expect(rewriteCopilotAuthError("some error", err, "openai")).toBe("some error");
});
it("returns original message for non-401/403 errors", () => {
const err = errorWithStatus(500);
expect(rewriteCopilotAuthError("server error", err, "github-copilot")).toBe("server error");
});
it("rewrites message for 401 with github-copilot provider", () => {
const err = errorWithStatus(401);
const result = rewriteCopilotAuthError("401 Unauthorized: ...", err, "github-copilot");
expect(result).toContain("GitHub Copilot authentication failed (HTTP 401)");
expect(result).toContain("/login github-copilot");
});
it("rewrites 403 with access-denied message (not auth-failed, to avoid credential removal)", () => {
const err = errorWithStatus(403);
const result = rewriteCopilotAuthError("403 Forbidden", err, "github-copilot");
expect(result).toContain("GitHub Copilot access denied (HTTP 403)");
expect(result).not.toContain("GitHub Copilot authentication failed");
expect(result).not.toContain("/login github-copilot");
});
});
@@ -0,0 +1,59 @@
import { describe, expect, it } from "bun:test";
import { rewriteCopilotError } from "../src/utils/http-inspector";
function errorWithStatus(status: number): Error {
const err = new Error(`${status} Unauthorized`);
(err as any).status = status;
return err;
}
describe("rewriteCopilotError", () => {
it("returns original message for non-copilot providers", () => {
const err = errorWithStatus(401);
expect(rewriteCopilotError("some error", err, "openai")).toBe("some error");
});
it("returns original message for non-401/403 errors", () => {
const err = errorWithStatus(500);
expect(rewriteCopilotError("server error", err, "github-copilot")).toBe("server error");
});
it("rewrites message for 401 with github-copilot provider", () => {
const err = errorWithStatus(401);
const result = rewriteCopilotError("401 Unauthorized: ...", err, "github-copilot");
expect(result).toContain("GitHub Copilot authentication failed (HTTP 401)");
expect(result).toContain("/login github-copilot");
});
it("rewrites 403 with access-denied message (not auth-failed, to avoid credential removal)", () => {
const err = errorWithStatus(403);
const result = rewriteCopilotError("403 Forbidden", err, "github-copilot");
expect(result).toContain("GitHub Copilot access denied (HTTP 403)");
expect(result).not.toContain("GitHub Copilot authentication failed");
expect(result).not.toContain("/login github-copilot");
});
it("rewrites 400 model_not_supported with rollout-gap guidance", () => {
const err = new Error("400 The requested model is not supported.");
(err as unknown as { status: number; code: string }).status = 400;
(err as unknown as { status: number; code: string }).code = "model_not_supported";
const result = rewriteCopilotError("original", err, "github-copilot");
expect(result).toContain("HTTP 400 model_not_supported");
expect(result).toContain("rollout gap");
expect(result).not.toContain("authentication failed");
});
it("leaves non-copilot 400 model_not_supported untouched", () => {
const err = new Error("400 model_not_supported");
(err as unknown as { status: number; code: string }).status = 400;
(err as unknown as { status: number; code: string }).code = "model_not_supported";
expect(rewriteCopilotError("orig", err, "openai")).toBe("orig");
});
it("leaves 400 without model_not_supported code untouched", () => {
const err = new Error("400 invalid request");
(err as unknown as { status: number; code: string }).status = 400;
(err as unknown as { status: number; code: string }).code = "invalid_request_body";
expect(rewriteCopilotError("orig", err, "github-copilot")).toBe("orig");
});
});