fix(ai): address Anthropic retry cap review
This commit is contained in:
@@ -46,7 +46,7 @@ export interface AnthropicRequestOptions {
|
||||
/**
|
||||
* Maximum delay in milliseconds to wait for a server-directed retry. If the
|
||||
* server's `retry-after` hint exceeds this value, the retry is declined and
|
||||
* the original error is surfaced. `0` disables the cap. Defaults to 60000.
|
||||
* the original error is surfaced. Non-positive values disable the cap. Defaults to 60000.
|
||||
*/
|
||||
maxRetryDelayMs?: number;
|
||||
/** Per-request headers merged after client defaults. */
|
||||
@@ -83,7 +83,7 @@ export interface AnthropicClientOptions {
|
||||
/**
|
||||
* Maximum delay in milliseconds to wait for a server-directed retry. If the
|
||||
* server's `retry-after` hint exceeds this value, the retry is declined and
|
||||
* the original error is surfaced. `0` disables the cap. Defaults to 60000.
|
||||
* the original error is surfaced. Non-positive values disable the cap. Defaults to 60000.
|
||||
*/
|
||||
maxRetryDelayMs?: number;
|
||||
/** Pre-response timeout in milliseconds. Defaults to 10 minutes. */
|
||||
@@ -109,7 +109,7 @@ function shouldRetryResponse(response: Response): boolean {
|
||||
}
|
||||
|
||||
/** Server-suggested delay (`retry-after-ms`, then `retry-after` seconds or HTTP date). */
|
||||
export function retryDelayFromHeaders(headers: Headers | undefined): number | undefined {
|
||||
export function retryDelayFromHeaders(headers: Pick<Headers, "get"> | undefined): number | undefined {
|
||||
if (!headers) return undefined;
|
||||
const retryAfterMs = headers.get("retry-after-ms");
|
||||
if (retryAfterMs) {
|
||||
@@ -254,10 +254,10 @@ export class AnthropicMessagesClient implements AnthropicMessagesClientLike {
|
||||
if (attempt < maxRetries && shouldRetryResponse(response)) {
|
||||
// Bound the server-directed wait: an over-cap `retry-after` declines
|
||||
// the retry and surfaces the original error (status/body/headers
|
||||
// intact) so higher-level recovery can run. `0` disables the cap.
|
||||
// intact) so higher-level recovery can run. A non-positive cap disables enforcement.
|
||||
// Checked before draining the body so `fromResponse` can still read it.
|
||||
const headerDelayMs = retryDelayFromHeaders(response.headers);
|
||||
if (headerDelayMs !== undefined && maxRetryDelayMs !== 0 && headerDelayMs > maxRetryDelayMs) {
|
||||
if (headerDelayMs !== undefined && maxRetryDelayMs > 0 && headerDelayMs > maxRetryDelayMs) {
|
||||
throw await AIError.AnthropicApiError.fromResponse(response, callerSignal);
|
||||
}
|
||||
await response.body?.cancel().catch(() => {});
|
||||
|
||||
@@ -66,7 +66,6 @@ import { createSdkStreamRequestOptions } from "../utils/sdk-stream-timeout";
|
||||
import { notifyRawSseEvent } from "../utils/sse-debug";
|
||||
import { isForcedToolChoice } from "../utils/tool-choice";
|
||||
import {
|
||||
AnthropicApiError,
|
||||
AnthropicConnectionTimeoutError,
|
||||
type AnthropicFetchOptions,
|
||||
AnthropicMessagesClient,
|
||||
@@ -1155,6 +1154,7 @@ export type AnthropicClientOptionsArgs = {
|
||||
thinkingDisplay?: AnthropicThinkingDisplay;
|
||||
disableStrictTools?: boolean;
|
||||
fetch?: FetchImpl;
|
||||
maxRetryDelayMs?: number;
|
||||
claudeCodeSessionId?: string;
|
||||
};
|
||||
|
||||
@@ -1164,6 +1164,7 @@ export type AnthropicClientOptionsResult = {
|
||||
authToken?: string | null;
|
||||
baseURL?: string;
|
||||
maxRetries: number;
|
||||
maxRetryDelayMs?: number;
|
||||
defaultHeaders: Record<string, string>;
|
||||
fetch?: FetchImpl;
|
||||
fetchOptions?: AnthropicFetchOptions;
|
||||
@@ -1574,6 +1575,16 @@ export function isProviderRetryableError(error: unknown, provider?: string): boo
|
||||
});
|
||||
}
|
||||
|
||||
function hasHeaderGetter(value: unknown): value is Pick<Headers, "get"> {
|
||||
return typeof value === "object" && value !== null && "get" in value && typeof value.get === "function";
|
||||
}
|
||||
|
||||
function retryDelayFromErrorHeaders(error: unknown): number | undefined {
|
||||
if (typeof error !== "object" || error === null || !("headers" in error)) return undefined;
|
||||
const { headers } = error as { headers?: unknown };
|
||||
return hasHeaderGetter(headers) ? retryDelayFromHeaders(headers) : undefined;
|
||||
}
|
||||
|
||||
const THINKING_ENVELOPE_OPEN = "<thinking>";
|
||||
const THINKING_ENVELOPE_CLOSE = "</thinking>";
|
||||
|
||||
@@ -1952,6 +1963,7 @@ const streamAnthropicOnce = (
|
||||
thinkingEnabled: options?.thinkingEnabled,
|
||||
thinkingDisplay: options?.thinkingDisplay,
|
||||
fetch: options?.fetch,
|
||||
maxRetryDelayMs: options?.maxRetryDelayMs,
|
||||
claudeCodeSessionId: options?.sessionId ?? extractClaudeMetadataSessionId(options?.metadata?.user_id),
|
||||
disableStrictTools,
|
||||
});
|
||||
@@ -2699,16 +2711,12 @@ const streamAnthropicOnce = (
|
||||
// Honor the server's retry hint (`retry-after-ms`/`retry-after`) on
|
||||
// 429/529-style failures: retrying sooner than the server asked is a
|
||||
// guaranteed failure that just burns the retry budget.
|
||||
const headerDelayMs =
|
||||
streamFailure instanceof Error && streamFailure instanceof AnthropicApiError
|
||||
? retryDelayFromHeaders(streamFailure.headers)
|
||||
: undefined;
|
||||
const headerDelayMs = retryDelayFromErrorHeaders(streamFailure);
|
||||
// Bound the server-directed wait so a multi-hour `retry-after` cannot
|
||||
// park the provider stream before higher-level recovery runs. A cap of
|
||||
// 0 disables the bound; an over-cap hint surfaces the original error
|
||||
// immediately without a second wire attempt or a `providerRetryWait`.
|
||||
// park the provider stream before higher-level recovery runs. A non-positive cap
|
||||
// disables the bound; an over-cap hint surfaces the original error immediately.
|
||||
const maxRetryDelayMs = options?.maxRetryDelayMs ?? 60_000;
|
||||
if (headerDelayMs !== undefined && maxRetryDelayMs !== 0 && headerDelayMs > maxRetryDelayMs) {
|
||||
if (headerDelayMs !== undefined && maxRetryDelayMs > 0 && headerDelayMs > maxRetryDelayMs) {
|
||||
throw streamFailure;
|
||||
}
|
||||
const delayMs = headerDelayMs !== undefined ? Math.max(headerDelayMs, backoffDelayMs) : backoffDelayMs;
|
||||
@@ -2855,6 +2863,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
||||
thinkingEnabled = false,
|
||||
thinkingDisplay,
|
||||
isOAuth,
|
||||
maxRetryDelayMs,
|
||||
claudeCodeSessionId,
|
||||
disableStrictTools: disableStrictToolsOverride,
|
||||
} = args;
|
||||
@@ -2925,6 +2934,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
||||
authToken: copilotApiKey,
|
||||
baseURL: baseUrl,
|
||||
maxRetries: 5,
|
||||
maxRetryDelayMs,
|
||||
defaultHeaders,
|
||||
fetch: cchFetch,
|
||||
fetchOptions,
|
||||
@@ -2972,6 +2982,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
||||
authToken: null,
|
||||
baseURL: baseUrl,
|
||||
maxRetries: 5,
|
||||
maxRetryDelayMs,
|
||||
defaultHeaders,
|
||||
fetch: cchFetch,
|
||||
fetchOptions,
|
||||
@@ -2990,6 +3001,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
||||
authToken: null,
|
||||
baseURL: baseUrl,
|
||||
maxRetries: 5,
|
||||
maxRetryDelayMs,
|
||||
defaultHeaders,
|
||||
fetch: cchFetch,
|
||||
fetchOptions,
|
||||
@@ -3012,6 +3024,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
||||
authToken: oauthToken ? apiKey : undefined,
|
||||
baseURL: baseUrl,
|
||||
maxRetries: 5,
|
||||
maxRetryDelayMs,
|
||||
defaultHeaders,
|
||||
fetch: cchFetch,
|
||||
fetchOptions,
|
||||
|
||||
@@ -28,6 +28,10 @@ const anthropicErrorBody = JSON.stringify({
|
||||
type: "error",
|
||||
error: { type: "invalid_request_error", message: "The compiled grammar is too large." },
|
||||
});
|
||||
const anthropicOverloadedErrorBody = JSON.stringify({
|
||||
type: "error",
|
||||
error: { type: "overloaded_error", message: "Overloaded" },
|
||||
});
|
||||
|
||||
describe("AnthropicMessagesClient error mapping", () => {
|
||||
it("maps non-2xx responses to AnthropicApiError with status and body in message", async () => {
|
||||
@@ -134,6 +138,19 @@ describe("AnthropicMessagesClient retries", () => {
|
||||
expect(error).toBeInstanceOf(AIError.AnthropicApiError);
|
||||
expect(calls.length).toBe(3); // initial attempt + 2 retries
|
||||
});
|
||||
|
||||
it("disables the cap when maxRetryDelayMs is negative", async () => {
|
||||
const { calls, fetch } = createFetchMock([
|
||||
new Response("overloaded", { status: 429, headers: { "retry-after-ms": "1" } }),
|
||||
new Response("{}", { status: 200 }),
|
||||
]);
|
||||
const client = new AnthropicMessagesClient({ apiKey: "sk-test", maxRetries: 5, fetch });
|
||||
|
||||
const response = await client.messages.create(params, { maxRetryDelayMs: -1 }).asResponse();
|
||||
|
||||
expect(response.status).toBe(200);
|
||||
expect(calls.length).toBe(2);
|
||||
});
|
||||
});
|
||||
|
||||
describe("AnthropicMessagesClient timeout and abort", () => {
|
||||
@@ -214,9 +231,27 @@ describe("AnthropicMessagesClient request assembly", () => {
|
||||
});
|
||||
|
||||
describe("AnthropicMessagesClient retry-after cap", () => {
|
||||
it("uses the documented 60-second default cap when callers omit one", async () => {
|
||||
const { calls, fetch } = createFetchMock([
|
||||
new Response(anthropicOverloadedErrorBody, { status: 429, headers: { "retry-after": "120" } }),
|
||||
]);
|
||||
const client = new AnthropicMessagesClient({ apiKey: "sk-test", maxRetries: 5, fetch });
|
||||
|
||||
const error = await client.messages
|
||||
.create(params)
|
||||
.asResponse()
|
||||
.catch(err => err as AIError.AnthropicApiError);
|
||||
|
||||
expect(error).toBeInstanceOf(AIError.AnthropicApiError);
|
||||
expect(error.status).toBe(429);
|
||||
expect(calls.length).toBe(1);
|
||||
});
|
||||
|
||||
it("declines a retry and preserves original status/body/headers when retry-after exceeds maxRetryDelayMs", async () => {
|
||||
const errorHeaders = { "retry-after": "120", "request-id": "req_cap" };
|
||||
const { calls, fetch } = createFetchMock([new Response("overloaded", { status: 429, headers: errorHeaders })]);
|
||||
const { calls, fetch } = createFetchMock([
|
||||
new Response(anthropicOverloadedErrorBody, { status: 429, headers: errorHeaders }),
|
||||
]);
|
||||
const client = new AnthropicMessagesClient({ apiKey: "sk-test", maxRetries: 5, fetch });
|
||||
|
||||
const error = await client.messages
|
||||
@@ -231,20 +266,7 @@ describe("AnthropicMessagesClient retry-after cap", () => {
|
||||
expect(error.headers.get("request-id")).toBe("req_cap");
|
||||
expect(calls.length).toBe(1);
|
||||
});
|
||||
it("declines a retry for a positive server hint when maxRetryDelayMs is negative", async () => {
|
||||
const errorHeaders = { "retry-after": "1", "request-id": "req_neg" };
|
||||
const { calls, fetch } = createFetchMock([new Response("overloaded", { status: 429, headers: errorHeaders })]);
|
||||
const client = new AnthropicMessagesClient({ apiKey: "sk-test", maxRetries: 5, fetch });
|
||||
|
||||
const error = await client.messages
|
||||
.create(params, { maxRetryDelayMs: -1 })
|
||||
.asResponse()
|
||||
.catch(err => err as AIError.AnthropicApiError);
|
||||
|
||||
expect(error).toBeInstanceOf(AIError.AnthropicApiError);
|
||||
expect(error.status).toBe(429);
|
||||
expect(calls.length).toBe(1);
|
||||
});
|
||||
it("cancels and releases an open error-body reader when the caller aborts", async () => {
|
||||
const controller = new AbortController();
|
||||
const encoder = new TextEncoder();
|
||||
@@ -257,7 +279,7 @@ describe("AnthropicMessagesClient retry-after cap", () => {
|
||||
},
|
||||
pull() {
|
||||
readBlocked = true;
|
||||
return new Promise<void>(() => {});
|
||||
return Promise.withResolvers<void>().promise;
|
||||
},
|
||||
cancel() {
|
||||
bodyCancelled = true;
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import * as AIError from "@oh-my-pi/pi-ai/error";
|
||||
import { streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic";
|
||||
import type { AnthropicMessagesClientLike } from "@oh-my-pi/pi-ai/providers/anthropic-client";
|
||||
import type { Context, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { AnthropicMessagesClient, type AnthropicMessagesClientLike } from "@oh-my-pi/pi-ai/providers/anthropic-client";
|
||||
import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { waitForDelayOrAbort } from "./helpers";
|
||||
|
||||
@@ -90,6 +90,36 @@ function createSuccessfulAnthropicEvents(text: string): MockAnthropicEvent[] {
|
||||
];
|
||||
}
|
||||
|
||||
function createAnthropicSseResponse(text: string): Response {
|
||||
const body = createSuccessfulAnthropicEvents(text)
|
||||
.map(event => `event: ${event.type}\ndata: ${JSON.stringify(event)}\n\n`)
|
||||
.join("");
|
||||
return new Response(body, {
|
||||
status: 200,
|
||||
headers: { "content-type": "text/event-stream", "request-id": "req_retry_success" },
|
||||
});
|
||||
}
|
||||
|
||||
function createResponseClient(responses: Response[]): {
|
||||
calls: { count: number };
|
||||
client: AnthropicMessagesClientLike;
|
||||
} {
|
||||
const calls = { count: 0 };
|
||||
const fetch: FetchImpl = async () => {
|
||||
const response = responses[Math.min(calls.count++, responses.length - 1)];
|
||||
if (!response) throw new Error("Expected an Anthropic mock response");
|
||||
return new Response(response.body, {
|
||||
status: response.status,
|
||||
statusText: response.statusText,
|
||||
headers: response.headers,
|
||||
});
|
||||
};
|
||||
return {
|
||||
calls,
|
||||
client: new AnthropicMessagesClient({ apiKey: "sk-test", maxRetries: 0, fetch }),
|
||||
};
|
||||
}
|
||||
|
||||
function createAnthropicMockStream({
|
||||
signal,
|
||||
connectDelayMs = 0,
|
||||
@@ -535,43 +565,31 @@ describe("anthropic provider retry delays", () => {
|
||||
});
|
||||
|
||||
describe("anthropic retry-after cap (maxRetryDelayMs)", () => {
|
||||
it("surfaces the original error without a second attempt when retry-after exceeds the default 60s cap", async () => {
|
||||
let attempt = 0;
|
||||
const create = ((_body: unknown, _requestOptions?: { signal?: AbortSignal }) => {
|
||||
attempt += 1;
|
||||
return createRejectedAnthropicRequest(
|
||||
new AIError.AnthropicApiError(
|
||||
429,
|
||||
'429 {"type":"error","error":{"type":"rate_limit_error","message":"Too many requests"}}',
|
||||
new Headers({ "retry-after": "120" }),
|
||||
),
|
||||
) as never;
|
||||
}) as unknown as AnthropicMessagesClientLike["messages"]["create"];
|
||||
const client = { messages: { create } } as AnthropicMessagesClientLike;
|
||||
it("surfaces the original HTTP error without a second attempt when retry-after exceeds the default 60s cap", async () => {
|
||||
const { calls, client } = createResponseClient([
|
||||
new Response('{"type":"error","error":{"type":"rate_limit_error","message":"Too many requests"}}', {
|
||||
status: 429,
|
||||
headers: { "retry-after": "120" },
|
||||
}),
|
||||
]);
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, { client, providerRetryWait }).result();
|
||||
|
||||
expect(attempt).toBe(1);
|
||||
expect(calls.count).toBe(1);
|
||||
expect(providerRetryWait).not.toHaveBeenCalled();
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorStatus).toBe(429);
|
||||
expect(result.errorMessage).toContain("rate_limit_error");
|
||||
});
|
||||
|
||||
it("surfaces the original error when retry-after exceeds an explicit maxRetryDelayMs cap", async () => {
|
||||
let attempt = 0;
|
||||
const create = ((_body: unknown, _requestOptions?: { signal?: AbortSignal }) => {
|
||||
attempt += 1;
|
||||
return createRejectedAnthropicRequest(
|
||||
new AIError.AnthropicApiError(
|
||||
529,
|
||||
'529 {"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}',
|
||||
new Headers({ "retry-after-ms": "10000" }),
|
||||
),
|
||||
) as never;
|
||||
}) as unknown as AnthropicMessagesClientLike["messages"]["create"];
|
||||
const client = { messages: { create } } as AnthropicMessagesClientLike;
|
||||
it("surfaces the original HTTP error when retry-after exceeds an explicit cap", async () => {
|
||||
const { calls, client } = createResponseClient([
|
||||
new Response('{"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}', {
|
||||
status: 529,
|
||||
headers: { "retry-after-ms": "10000" },
|
||||
}),
|
||||
]);
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, {
|
||||
@@ -580,99 +598,113 @@ describe("anthropic retry-after cap (maxRetryDelayMs)", () => {
|
||||
maxRetryDelayMs: 5_000,
|
||||
}).result();
|
||||
|
||||
expect(calls.count).toBe(1);
|
||||
expect(providerRetryWait).not.toHaveBeenCalled();
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorStatus).toBe(529);
|
||||
});
|
||||
|
||||
it("disables the cap when maxRetryDelayMs is negative", async () => {
|
||||
const { calls, client } = createResponseClient([
|
||||
new Response('{"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}', {
|
||||
status: 529,
|
||||
headers: { "retry-after": "1" },
|
||||
}),
|
||||
createAnthropicSseResponse("after unbounded wait"),
|
||||
]);
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, {
|
||||
client,
|
||||
providerRetryWait,
|
||||
maxRetryDelayMs: -1,
|
||||
}).result();
|
||||
|
||||
expect(calls.count).toBe(2);
|
||||
expect(providerRetryWait).toHaveBeenCalledWith(1_000, undefined);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
});
|
||||
|
||||
it("disables the cap when maxRetryDelayMs is 0 and waits the full server hint", async () => {
|
||||
const { calls, client } = createResponseClient([
|
||||
new Response('{"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}', {
|
||||
status: 529,
|
||||
headers: { "retry-after": "120" },
|
||||
}),
|
||||
createAnthropicSseResponse("after long wait"),
|
||||
]);
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, {
|
||||
client,
|
||||
providerRetryWait,
|
||||
maxRetryDelayMs: 0,
|
||||
}).result();
|
||||
|
||||
expect(calls.count).toBe(2);
|
||||
expect(providerRetryWait).toHaveBeenCalledWith(120_000, undefined);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
});
|
||||
|
||||
it("retries when the HTTP retry-after hint is under the cap", async () => {
|
||||
const { calls, client } = createResponseClient([
|
||||
new Response('{"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}', {
|
||||
status: 529,
|
||||
headers: { "retry-after": "30" },
|
||||
}),
|
||||
createAnthropicSseResponse("after backoff"),
|
||||
]);
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, {
|
||||
client,
|
||||
providerRetryWait,
|
||||
maxRetryDelayMs: 60_000,
|
||||
}).result();
|
||||
|
||||
expect(calls.count).toBe(2);
|
||||
expect(providerRetryWait).toHaveBeenCalledWith(30_000, undefined);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
});
|
||||
|
||||
it("honors retry headers from structurally compatible injected SDK errors", async () => {
|
||||
let attempt = 0;
|
||||
const error = Object.assign(new Error("529 overloaded"), {
|
||||
status: 529,
|
||||
headers: new Headers({ "retry-after-ms": "10000" }),
|
||||
});
|
||||
const create = ((_body: unknown) => {
|
||||
attempt += 1;
|
||||
return createRejectedAnthropicRequest(error) as never;
|
||||
}) as unknown as AnthropicMessagesClientLike["messages"]["create"];
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, {
|
||||
client: { messages: { create } },
|
||||
providerRetryWait,
|
||||
maxRetryDelayMs: 5_000,
|
||||
}).result();
|
||||
|
||||
expect(attempt).toBe(1);
|
||||
expect(providerRetryWait).not.toHaveBeenCalled();
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorStatus).toBe(529);
|
||||
});
|
||||
it("surfaces the original error when maxRetryDelayMs is negative and a server hint is present", async () => {
|
||||
let attempt = 0;
|
||||
const create = (_body: unknown, _requestOptions?: { signal?: AbortSignal }) => {
|
||||
attempt += 1;
|
||||
return createRejectedAnthropicRequest(
|
||||
new AIError.AnthropicApiError(
|
||||
529,
|
||||
'529 {"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}',
|
||||
new Headers({ "retry-after": "1" }),
|
||||
),
|
||||
);
|
||||
|
||||
it("passes maxRetryDelayMs to internally constructed Anthropic clients", async () => {
|
||||
let calls = 0;
|
||||
const fetch: FetchImpl = async () => {
|
||||
calls += 1;
|
||||
return new Response('{"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}', {
|
||||
status: 429,
|
||||
headers: { "retry-after-ms": "10" },
|
||||
});
|
||||
};
|
||||
const client: AnthropicMessagesClientLike = { messages: { create } };
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, {
|
||||
client,
|
||||
providerRetryWait,
|
||||
maxRetryDelayMs: -1,
|
||||
}).result();
|
||||
const result = await streamAnthropic(model, context, { fetch, maxRetryDelayMs: 5 }).result();
|
||||
|
||||
expect(attempt).toBe(1);
|
||||
expect(providerRetryWait).not.toHaveBeenCalled();
|
||||
expect(calls).toBe(1);
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorStatus).toBe(529);
|
||||
});
|
||||
|
||||
it("disables the cap when maxRetryDelayMs is 0 and waits the full server hint", async () => {
|
||||
let attempt = 0;
|
||||
const create = ((_body: unknown, requestOptions?: { signal?: AbortSignal }) => {
|
||||
attempt += 1;
|
||||
if (attempt === 1) {
|
||||
return createRejectedAnthropicRequest(
|
||||
new AIError.AnthropicApiError(
|
||||
529,
|
||||
'529 {"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}',
|
||||
new Headers({ "retry-after": "120" }),
|
||||
),
|
||||
) as never;
|
||||
}
|
||||
return createAnthropicMockStream({
|
||||
signal: requestOptions?.signal,
|
||||
events: createSuccessfulAnthropicEvents("after long wait"),
|
||||
}) as never;
|
||||
}) as unknown as AnthropicMessagesClientLike["messages"]["create"];
|
||||
const client = { messages: { create } } as AnthropicMessagesClientLike;
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, {
|
||||
client,
|
||||
providerRetryWait,
|
||||
maxRetryDelayMs: 0,
|
||||
}).result();
|
||||
|
||||
expect(attempt).toBe(2);
|
||||
expect(providerRetryWait).toHaveBeenCalledWith(120_000, undefined);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
});
|
||||
|
||||
it("keeps current behavior when the server hint is under the cap", async () => {
|
||||
let attempt = 0;
|
||||
const create = ((_body: unknown, requestOptions?: { signal?: AbortSignal }) => {
|
||||
attempt += 1;
|
||||
if (attempt === 1) {
|
||||
return createRejectedAnthropicRequest(
|
||||
new AIError.AnthropicApiError(
|
||||
529,
|
||||
'529 {"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}',
|
||||
new Headers({ "retry-after": "30" }),
|
||||
),
|
||||
) as never;
|
||||
}
|
||||
return createAnthropicMockStream({
|
||||
signal: requestOptions?.signal,
|
||||
events: createSuccessfulAnthropicEvents("after backoff"),
|
||||
}) as never;
|
||||
}) as unknown as AnthropicMessagesClientLike["messages"]["create"];
|
||||
const client = { messages: { create } } as AnthropicMessagesClientLike;
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, {
|
||||
client,
|
||||
providerRetryWait,
|
||||
maxRetryDelayMs: 60_000,
|
||||
}).result();
|
||||
|
||||
expect(attempt).toBe(2);
|
||||
expect(providerRetryWait).toHaveBeenCalledWith(30_000, undefined);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.errorStatus).toBe(429);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -58,6 +58,7 @@ export function createSettingsAwareStreamFn(settings: Settings, base: StreamFn =
|
||||
textVerbosity: streamOptions?.textVerbosity ?? textVerbosity,
|
||||
streamFirstEventTimeoutMs: streamOptions?.streamFirstEventTimeoutMs ?? streamFirstEventTimeoutMs,
|
||||
streamIdleTimeoutMs: streamOptions?.streamIdleTimeoutMs ?? streamIdleTimeoutMs,
|
||||
maxRetryDelayMs: streamOptions?.maxRetryDelayMs ?? settings.get("retry.maxDelayMs"),
|
||||
maxInFlightRequests: validateProviderMaxInFlightRequests(
|
||||
streamOptions?.maxInFlightRequests ?? settings.get("providers.maxInFlightRequests"),
|
||||
),
|
||||
|
||||
@@ -115,6 +115,18 @@ describe("createSettingsAwareStreamFn", () => {
|
||||
expect(calls[1]?.options?.streamIdleTimeoutMs).toBe(10_000);
|
||||
});
|
||||
|
||||
it("forwards retry.maxDelayMs while preserving caller overrides", () => {
|
||||
const settings = Settings.isolated({ "retry.maxDelayMs": 300_000 });
|
||||
const { fn: base, calls } = captureBase();
|
||||
const wrapped = createSettingsAwareStreamFn(settings, base);
|
||||
|
||||
wrapped(stubModel, stubContext, undefined);
|
||||
wrapped(stubModel, stubContext, { maxRetryDelayMs: 5_000 });
|
||||
|
||||
expect(calls[0]?.options?.maxRetryDelayMs).toBe(300_000);
|
||||
expect(calls[1]?.options?.maxRetryDelayMs).toBe(5_000);
|
||||
});
|
||||
|
||||
it("treats the default openrouterVariant as absent so the base call carries no variant", () => {
|
||||
const settings = Settings.isolated({ "providers.openrouterVariant": "default" });
|
||||
const { fn: base, calls } = captureBase();
|
||||
|
||||
Reference in New Issue
Block a user