fix(ai): preserved Copilot integrator entitlement errors
- Classified model_not_available_for_integrator as a permanent entitlement denial instead of transient fleet skew. - Preserved the provider response and Available models list while retaining model_not_supported fleet retries. Fixes #7819
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed GitHub Copilot's permanent `model_not_available_for_integrator` response being retried and replaced with transient fleet-skew guidance, preserving the provider's actionable `Available models` list instead ([#7819](https://github.com/can1357/oh-my-pi/issues/7819)).
|
||||
|
||||
## [17.2.9] - 2026-08-05
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -113,17 +113,15 @@ const STALE_RESPONSE_ITEM_DETAIL_PATTERN = /not[ _]?found|invalid|expired|stale|
|
||||
export const LLAMA_CPP_TOOL_CALL_PARSE_PATTERN =
|
||||
/failed to parse tool call arguments as json|\[json\.exception\.parse_error\.101\]/i;
|
||||
|
||||
// Copilot fleet skew: HTTP 400 rejecting a model that `/models` advertised on
|
||||
// the very same host. Two codes appear in the wild — `model_not_supported`
|
||||
// (per-OAuth-client rollout gap) and `model_not_available_for_integrator`
|
||||
// (replicas whose integrator allowlist predates the model). Both flap
|
||||
// request-to-request, so a retry usually lands on a backend that has the model.
|
||||
// Copilot fleet skew: HTTP 400 `model_not_supported` can reject a model that
|
||||
// `/models` advertised on the same host when the request lands on a stale
|
||||
// replica. `model_not_available_for_integrator` is deliberately excluded:
|
||||
// GitHub also uses it for stable per-integrator entitlement denials and includes
|
||||
// that integrator's actionable `Available models` list in the response.
|
||||
const COPILOT_TRANSIENT_MODEL_CODES: Record<string, true> = {
|
||||
model_not_supported: true,
|
||||
model_not_available_for_integrator: true,
|
||||
};
|
||||
const COPILOT_MODEL_UNAVAILABLE_PATTERN =
|
||||
/model_not_supported|model_not_available_for_integrator|not available for integrator/i;
|
||||
const COPILOT_TRANSIENT_MODEL_PATTERN = /model_not_supported/i;
|
||||
// Anthropic strict-tool grammar too large / schema too complex (400 invalid_request_error).
|
||||
// Feature-gated deployments (Azure Foundry, Baseten, …) reject `strict: true`
|
||||
// tools outright when the hosted model lacks structured outputs, e.g.
|
||||
@@ -377,8 +375,8 @@ function classifyText(errorMessage: string | undefined, errorStatus: number | un
|
||||
kinds |= Flag.StaleResponsesItem;
|
||||
}
|
||||
|
||||
// Copilot fleet-skew model rejection is transient.
|
||||
if (statusClean === 400 && COPILOT_MODEL_UNAVAILABLE_PATTERN.test(cleanMessage)) kinds |= Flag.Transient;
|
||||
// Copilot's `model_not_supported` fleet-skew rejection is transient.
|
||||
if (statusClean === 400 && COPILOT_TRANSIENT_MODEL_PATTERN.test(cleanMessage)) kinds |= Flag.Transient;
|
||||
if (matchesStrictToolsRejection(cleanMessage, statusClean)) kinds |= Flag.Grammar;
|
||||
if (matchesFastModeUnsupported(cleanMessage, statusClean)) kinds |= Flag.FastModeUnsupported;
|
||||
}
|
||||
@@ -513,10 +511,10 @@ function providerErrorCode(error: object): string | undefined {
|
||||
}
|
||||
|
||||
/**
|
||||
* GitHub Copilot 400 rejecting a model its own `/models` catalog advertises —
|
||||
* transient fleet skew, not a malformed request. Reads the structural `code`
|
||||
* through the SDK/body envelopes, then falls back to the stringified body both
|
||||
* SDK families put in `message` (shapes drift; the wire text does not).
|
||||
* GitHub Copilot 400 `model_not_supported` response for a model advertised by
|
||||
* `/models` — transient fleet skew, not a malformed request. Reads the
|
||||
* structural `code` through the SDK/body envelopes, then falls back to the
|
||||
* stringified body both SDK families put in `message`.
|
||||
*/
|
||||
export function isCopilotTransientModelError(error: unknown): boolean {
|
||||
if (!error || typeof error !== "object" || status(error) !== 400) return false;
|
||||
@@ -525,7 +523,7 @@ export function isCopilotTransientModelError(error: unknown): boolean {
|
||||
// prototype key (`__proto__`, `toString`, …) would otherwise read truthy.
|
||||
if (code !== undefined && Object.hasOwn(COPILOT_TRANSIENT_MODEL_CODES, code)) return true;
|
||||
const message: unknown = "message" in error ? error.message : undefined;
|
||||
return typeof message === "string" && COPILOT_MODEL_UNAVAILABLE_PATTERN.test(message);
|
||||
return typeof message === "string" && COPILOT_TRANSIENT_MODEL_PATTERN.test(message);
|
||||
}
|
||||
|
||||
export function classifyMessage(message: {
|
||||
|
||||
@@ -95,11 +95,10 @@ export async function finalizeErrorMessage(
|
||||
* Rewrite error message for GitHub Copilot request failures.
|
||||
* Must run AFTER finalizeErrorMessage since it replaces the message entirely.
|
||||
*
|
||||
* 400 model-unavailable = Copilot fleet skew. A model that `/models` advertises
|
||||
* (claude-sonnet-4.6, claude-opus-4.6, gpt-5.4, gpt-5.3-codex, ...)
|
||||
* flaps between 200 and 400 because only part of Copilot's fleet has it
|
||||
* in the integrator allowlist. After the in-request retry exhausts,
|
||||
* surface guidance rather than the raw error.
|
||||
* 400 `model_not_supported` = Copilot fleet skew. A model that `/models`
|
||||
* advertises can flap between 200 and 400 because only part of
|
||||
* Copilot's fleet has it in the integrator allowlist. After the
|
||||
* in-request retry exhausts, surface guidance rather than the raw error.
|
||||
* 401 = token invalid/expired → credential removal is safe, prompt re-login.
|
||||
* 403 = token valid but access denied (plan, model policy, org restriction) →
|
||||
* do NOT reuse the auth-failed string (which triggers credential removal).
|
||||
|
||||
@@ -23,9 +23,8 @@ const COPILOT_MODEL_RETRY_BASE_DELAY_MS = 400;
|
||||
const COPILOT_RETRY_AFTER_MAX_WAIT_MS = 30_000;
|
||||
|
||||
/**
|
||||
* Wrap an initial Copilot request so transient model-availability 400s
|
||||
* (`model_not_supported`, `model_not_available_for_integrator`) are retried a
|
||||
* small number of times. No-op for non-Copilot providers.
|
||||
* Wrap an initial Copilot request so transient `model_not_supported` 400s are
|
||||
* retried a small number of times. No-op for non-Copilot providers.
|
||||
*
|
||||
* The callback **MUST** create a fresh in-flight request each invocation — a
|
||||
* once-consumed AsyncIterable cannot be re-iterated.
|
||||
|
||||
@@ -106,13 +106,11 @@ describe("isProviderRetryableError", () => {
|
||||
expect(isProviderRetryableError(err)).toBe(false);
|
||||
});
|
||||
|
||||
it("retries Copilot's model_not_available_for_integrator 400 from the Anthropic messages proxy", () => {
|
||||
// Shape thrown by @anthropic-ai/sdk against api.githubcopilot.com/v1/messages:
|
||||
// the parsed body lands on `.error` and is itself `{ error: { code } }`.
|
||||
it("does not retry Copilot's per-integrator entitlement response", () => {
|
||||
const body = {
|
||||
error: {
|
||||
message:
|
||||
'The requested model is not available for integrator "copilot-language-server". Available models: [gpt-4.1 claude-opus-4.7]. Verify the correct Copilot-Integration-Id header is being sent.',
|
||||
'The requested model is not available for integrator "opencode". Available models: [gpt-4.1 claude-opus-4.7].',
|
||||
code: "model_not_available_for_integrator",
|
||||
param: "model",
|
||||
type: "invalid_request_error",
|
||||
@@ -120,7 +118,7 @@ describe("isProviderRetryableError", () => {
|
||||
};
|
||||
const err = new Error(`400 ${JSON.stringify(body)}`);
|
||||
Object.assign(err, { status: 400, error: body });
|
||||
expect(isProviderRetryableError(err, "github-copilot")).toBe(true);
|
||||
expect(isProviderRetryableError(err, "github-copilot")).toBe(false);
|
||||
expect(isProviderRetryableError(err, "anthropic")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -46,29 +46,27 @@ describe("isCopilotTransientModelError", () => {
|
||||
expect(isCopilotTransientModelError(err)).toBe(true);
|
||||
});
|
||||
|
||||
it("matches 400 model_not_available_for_integrator nested two envelopes deep (Anthropic SDK shape)", () => {
|
||||
// api.githubcopilot.com/v1/messages: the SDK stores the parsed body on
|
||||
// `.error`, and that body is itself `{ error: { code } }`.
|
||||
it("does not match per-integrator entitlement errors nested two envelopes deep", () => {
|
||||
const err = copilotError({
|
||||
status: 400,
|
||||
error: {
|
||||
error: {
|
||||
code: "model_not_available_for_integrator",
|
||||
message: 'The requested model is not available for integrator "copilot-language-server".',
|
||||
message: 'The requested model is not available for integrator "opencode".',
|
||||
},
|
||||
},
|
||||
message:
|
||||
'400 {"error":{"message":"The requested model is not available for integrator \\"copilot-language-server\\". Available models: [gpt-4.1 claude-opus-4.7]","code":"model_not_available_for_integrator","param":"model","type":"invalid_request_error"}}',
|
||||
'400 {"error":{"message":"The requested model is not available for integrator \\"opencode\\". Available models: [gpt-4.1 claude-opus-4.7]","code":"model_not_available_for_integrator","param":"model","type":"invalid_request_error"}}',
|
||||
});
|
||||
expect(isCopilotTransientModelError(err)).toBe(true);
|
||||
expect(isCopilotTransientModelError(err)).toBe(false);
|
||||
});
|
||||
|
||||
it("falls back to the stringified body when no envelope exposes a code", () => {
|
||||
it("does not infer fleet skew from per-integrator entitlement text", () => {
|
||||
const err = copilotError({
|
||||
status: 400,
|
||||
message: '400 {"error":{"message":"The requested model is not available for integrator \\"x\\"."}}',
|
||||
message: '400 {"error":{"message":"The requested model is not available for integrator \\"opencode\\"."}}',
|
||||
});
|
||||
expect(isCopilotTransientModelError(err)).toBe(true);
|
||||
expect(isCopilotTransientModelError(err)).toBe(false);
|
||||
});
|
||||
|
||||
it("does not match other 400 codes", () => {
|
||||
@@ -167,6 +165,26 @@ describe("callWithCopilotModelRetry", () => {
|
||||
expect(calls).toBe(1);
|
||||
});
|
||||
|
||||
it("does not retry per-integrator entitlement errors", async () => {
|
||||
let calls = 0;
|
||||
const err = copilotError({
|
||||
status: 400,
|
||||
code: "model_not_available_for_integrator",
|
||||
message:
|
||||
'400 The requested model is not available for integrator "opencode". Available models: [gpt-4.1 claude-opus-4.7]',
|
||||
});
|
||||
await expect(
|
||||
callWithCopilotModelRetry(
|
||||
async () => {
|
||||
calls += 1;
|
||||
throw err;
|
||||
},
|
||||
{ provider: "github-copilot", retryBaseDelayMs: 0 },
|
||||
),
|
||||
).rejects.toBe(err);
|
||||
expect(calls).toBe(1);
|
||||
});
|
||||
|
||||
it("does not blind-retry a 429 that carries no Retry-After guidance", async () => {
|
||||
let calls = 0;
|
||||
const err = copilotError({ status: 429, message: "rate limited" });
|
||||
@@ -263,7 +281,7 @@ describe("callWithCopilotModelRetry", () => {
|
||||
await expect(
|
||||
callWithCopilotModelRetry(
|
||||
async () => {
|
||||
throw copilotError({ status: 400, code: "model_not_available_for_integrator", message: "flap" });
|
||||
throw copilotError({ status: 400, code: "model_not_supported", message: "flap" });
|
||||
},
|
||||
{ provider: "github-copilot", retryBaseDelayMs: 100 },
|
||||
),
|
||||
|
||||
@@ -29,15 +29,13 @@ const testContext: Context = {
|
||||
};
|
||||
|
||||
/**
|
||||
* Verbatim body served by `api.githubcopilot.com/v1/messages` when the request
|
||||
* lands on a fleet replica whose integrator allowlist predates the model, even
|
||||
* though `/models` on the same host advertises it.
|
||||
* Copilot's transient response when a request lands on a fleet replica that
|
||||
* does not yet support a model advertised by `/models`.
|
||||
*/
|
||||
const FLEET_SKEW_BODY = {
|
||||
error: {
|
||||
message:
|
||||
'The requested model is not available for integrator "copilot-language-server". Available models: [gpt-4.1 claude-opus-4.7 claude-sonnet-4.5]. Verify the correct Copilot-Integration-Id header is being sent.',
|
||||
code: "model_not_available_for_integrator",
|
||||
message: "The requested model is not supported by this fleet replica.",
|
||||
code: "model_not_supported",
|
||||
param: "model",
|
||||
type: "invalid_request_error",
|
||||
},
|
||||
|
||||
@@ -1,10 +1,14 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { rewriteCopilotError } from "@oh-my-pi/pi-ai/utils/http-inspector";
|
||||
|
||||
function errorWithStatus(status: number): Error {
|
||||
const err = new Error(`${status} Unauthorized`);
|
||||
(err as any).status = status;
|
||||
return err;
|
||||
function errorWithStatus(
|
||||
status: number,
|
||||
options: { message?: string; code?: string } = {},
|
||||
): Error & { status: number; code?: string } {
|
||||
return Object.assign(new Error(options.message ?? `${status} Unauthorized`), {
|
||||
status,
|
||||
...(options.code === undefined ? {} : { code: options.code }),
|
||||
});
|
||||
}
|
||||
|
||||
describe("rewriteCopilotError", () => {
|
||||
@@ -33,29 +37,37 @@ describe("rewriteCopilotError", () => {
|
||||
expect(result).not.toContain("/login github-copilot");
|
||||
});
|
||||
|
||||
it("rewrites 400 model-unavailable codes with fleet-skew guidance", () => {
|
||||
for (const code of ["model_not_supported", "model_not_available_for_integrator"]) {
|
||||
const err = new Error("400 The requested model is not available.");
|
||||
(err as unknown as { status: number; code: string }).status = 400;
|
||||
(err as unknown as { status: number; code: string }).code = code;
|
||||
const result = rewriteCopilotError("original", err, "github-copilot");
|
||||
expect(result).toContain("HTTP 400");
|
||||
expect(result).toContain("only part of its fleet");
|
||||
expect(result).not.toContain("authentication failed");
|
||||
}
|
||||
it("rewrites 400 model_not_supported with fleet-skew guidance", () => {
|
||||
const err = errorWithStatus(400, {
|
||||
message: "400 The requested model is not supported.",
|
||||
code: "model_not_supported",
|
||||
});
|
||||
const result = rewriteCopilotError("original", err, "github-copilot");
|
||||
expect(result).toContain("HTTP 400");
|
||||
expect(result).toContain("only part of its fleet");
|
||||
expect(result).not.toContain("authentication failed");
|
||||
});
|
||||
|
||||
it("preserves per-integrator entitlement details and available models", () => {
|
||||
const message =
|
||||
'400 The requested model is not available for integrator "opencode". Available models: [gpt-4.1 claude-opus-4.7 gpt-5.5]';
|
||||
const err = errorWithStatus(400, { message, code: "model_not_available_for_integrator" });
|
||||
expect(rewriteCopilotError(message, err, "github-copilot")).toBe(message);
|
||||
});
|
||||
|
||||
it("leaves non-copilot 400 model_not_supported untouched", () => {
|
||||
const err = new Error("400 model_not_supported");
|
||||
(err as unknown as { status: number; code: string }).status = 400;
|
||||
(err as unknown as { status: number; code: string }).code = "model_not_supported";
|
||||
const err = errorWithStatus(400, {
|
||||
message: "400 model_not_supported",
|
||||
code: "model_not_supported",
|
||||
});
|
||||
expect(rewriteCopilotError("orig", err, "openai")).toBe("orig");
|
||||
});
|
||||
|
||||
it("leaves 400 without model_not_supported code untouched", () => {
|
||||
const err = new Error("400 invalid request");
|
||||
(err as unknown as { status: number; code: string }).status = 400;
|
||||
(err as unknown as { status: number; code: string }).code = "invalid_request_body";
|
||||
const err = errorWithStatus(400, {
|
||||
message: "400 invalid request",
|
||||
code: "invalid_request_body",
|
||||
});
|
||||
expect(rewriteCopilotError("orig", err, "github-copilot")).toBe("orig");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -44,6 +44,16 @@ function createUnauthorizedResponse(): Response {
|
||||
});
|
||||
}
|
||||
|
||||
const INTEGRATOR_ENTITLEMENT_BODY = {
|
||||
error: {
|
||||
message:
|
||||
'The requested model is not available for integrator "opencode". Available models: [gpt-4.1 claude-opus-4.7 gpt-5.5]',
|
||||
code: "model_not_available_for_integrator",
|
||||
param: "model",
|
||||
type: "invalid_request_error",
|
||||
},
|
||||
};
|
||||
|
||||
const testToken = "ghu_test_copilot_token";
|
||||
const enterpriseApiKey = JSON.stringify({ token: testToken, enterpriseUrl: "ghe.example.com" });
|
||||
const businessApiKey = JSON.stringify({
|
||||
@@ -86,6 +96,28 @@ describe("GitHub Copilot OpenAI transport base URL", () => {
|
||||
expect(requestedUrls[0]).toBe("https://api.githubcopilot.com/responses");
|
||||
});
|
||||
|
||||
it("surfaces responses API integrator entitlement details without retrying", async () => {
|
||||
const fetchMock = vi.fn(
|
||||
async () =>
|
||||
new Response(JSON.stringify(INTEGRATOR_ENTITLEMENT_BODY), {
|
||||
status: 400,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
const model = getBundledModel("github-copilot", "gpt-5.6-sol") as Model<"openai-responses">;
|
||||
const result = await streamOpenAIResponses(model, testContext, {
|
||||
apiKey: testToken,
|
||||
fetch: fetchMock as unknown as typeof fetch,
|
||||
}).result();
|
||||
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorMessage).toContain('not available for integrator "opencode"');
|
||||
expect(result.errorMessage).toContain("Available models: [gpt-4.1 claude-opus-4.7 gpt-5.5]");
|
||||
expect(result.errorMessage).not.toContain("only part of its fleet");
|
||||
});
|
||||
|
||||
it("omits OpenAI priority service tier while native OpenAI keeps it", async () => {
|
||||
const requestedBodies: Record<string, unknown>[] = [];
|
||||
const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => {
|
||||
|
||||
Reference in New Issue
Block a user