feat(coding-agent): allowed model fallback after retry budget exhaustion
- Permit model fallback even if the retry budget is exhausted when the current provider is locked by credential rotation or usage limits. - Reset the retry budget when successfully switching to a fallback model to ensure the new model has a full allowance of retries. - Added a regression test to verify that credential rotation failure triggers model fallback.
This commit is contained in:
@@ -19,6 +19,7 @@
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed failure to trigger model fallback when the retry budget is exhausted by credential rotation
|
||||
- Fixed uncontrollable mouse-wheel scrolling in the /models hub: the wheel moved the selection (one step per wheel event, so a single trackpad flick skipped many rows) and wrapped from the bottom back to the top. Wheel scrolling now pans the list viewport only, clamps at the ends, and leaves the selection where it is; keyboard navigation still scrolls the selection into view. Likewise, the wheel over the provider sidebar no longer switches the active scope (or triggers provider refreshes) — it just scrolls the sidebar.
|
||||
- Fixed TPS being inflated several-fold when a provider hides reasoning tokens until late in the stream (e.g. `google/gemini-3.5` vs `google-vertex/gemini-3.5` reporting 648 vs 186 TPS for identical durations): `omp bench`, the per-turn usage row, and the /models perf aggregates now measure tokens/sec over the total request duration instead of the post-TTFT decode window, matching `omp stats`. Stored perf aggregates are purged and re-backfilled from stats history with the corrected math on first launch.
|
||||
- Fixed the model-perf stats.db backfill freezing the TUI (~30s on multi-million-row stats databases) when /models triggered it: the import is now fire-and-forget, walks the newest rows in small chunks with event-loop yields between them, and is bounded to 90 days / 256 newest samples per model — beyond either bound the recency decay would erase the contribution anyway.
|
||||
|
||||
@@ -13993,20 +13993,13 @@ export class AgentSession {
|
||||
this.#retryResolve = resolve;
|
||||
}
|
||||
|
||||
if (this.#retryAttempt > retrySettings.maxRetries) {
|
||||
await this.#persistRetryLifecycleErrorMessage(message);
|
||||
// Max retries exceeded, emit final failure and reset
|
||||
await this.#emitSessionEvent({
|
||||
type: "auto_retry_end",
|
||||
success: false,
|
||||
attempt: this.#retryAttempt - 1,
|
||||
finalError: message.errorMessage,
|
||||
});
|
||||
this.#clearPendingRecoveredRetryErrors();
|
||||
this.#retryAttempt = 0;
|
||||
this.#resolveRetry(); // Resolve so waitForRetry() completes
|
||||
return false;
|
||||
}
|
||||
// All attempts on the current model are spent. Don't fail yet: the
|
||||
// fallback chain below gets one last consult. Credential rotation can
|
||||
// consume the entire budget without the fallback branch ever running
|
||||
// (every rotation sets switchedCredential and skips it), so without
|
||||
// this last resort a provider-wide usage cap never fails over to the
|
||||
// configured chain.
|
||||
const retryBudgetExhausted = this.#retryAttempt > retrySettings.maxRetries;
|
||||
|
||||
const errorMessage = message.errorMessage || "Unknown error";
|
||||
const id = this.#classifyRetryMessage(message);
|
||||
@@ -14025,7 +14018,12 @@ export class AgentSession {
|
||||
this.#resetCurrentResponsesProviderSession("stale replay error");
|
||||
}
|
||||
|
||||
if (this.model && !staleOpenAIResponsesReplayError && AIError.is(id, AIError.Flag.UsageLimit)) {
|
||||
if (
|
||||
!retryBudgetExhausted &&
|
||||
this.model &&
|
||||
!staleOpenAIResponsesReplayError &&
|
||||
AIError.is(id, AIError.Flag.UsageLimit)
|
||||
) {
|
||||
const retryAfterMs = parsedRetryAfterMs ?? calculateRateLimitBackoffMs(parseRateLimitReason(errorMessage));
|
||||
const outcome = await this.#modelRegistry.authStorage.markUsageLimitReached(
|
||||
this.model.provider,
|
||||
@@ -14071,7 +14069,9 @@ export class AgentSession {
|
||||
const allowModelFallback = options?.allowModelFallback !== false;
|
||||
const currentSelector = this.model ? formatRetryFallbackSelector(this.model, this.thinkingLevel) : undefined;
|
||||
if (!staleOpenAIResponsesReplayError && !switchedCredential && currentSelector) {
|
||||
if (allowModelFallback && retrySettings.modelFallback) {
|
||||
// A refusal chain stops at the retry budget: the exhausted-attempt
|
||||
// last resort is for provider failures, not classifier decisions.
|
||||
if (allowModelFallback && retrySettings.modelFallback && !(retryBudgetExhausted && classifierRefusal)) {
|
||||
if (!classifierRefusal) {
|
||||
this.#noteRetryFallbackCooldown(currentSelector, parsedRetryAfterMs, errorMessage);
|
||||
}
|
||||
@@ -14090,6 +14090,26 @@ export class AgentSession {
|
||||
delayMs = parsedRetryAfterMs;
|
||||
}
|
||||
}
|
||||
if (retryBudgetExhausted) {
|
||||
if (!switchedModel) {
|
||||
await this.#persistRetryLifecycleErrorMessage(message);
|
||||
// Max retries exceeded and no fallback model to switch to: emit
|
||||
// final failure and reset.
|
||||
await this.#emitSessionEvent({
|
||||
type: "auto_retry_end",
|
||||
success: false,
|
||||
attempt: this.#retryAttempt - 1,
|
||||
finalError: message.errorMessage,
|
||||
});
|
||||
this.#clearPendingRecoveredRetryErrors();
|
||||
this.#retryAttempt = 0;
|
||||
this.#resolveRetry(); // Resolve so waitForRetry() completes
|
||||
return false;
|
||||
}
|
||||
// The fallback model gets a fresh retry budget — leaving the spent
|
||||
// counter in place would exhaust it again on its first error.
|
||||
this.#retryAttempt = 1;
|
||||
}
|
||||
if (classifierRefusal && !switchedModel) {
|
||||
this.#retryAttempt = 0;
|
||||
this.#resolveRetry();
|
||||
|
||||
@@ -322,6 +322,75 @@ describe("AgentSession retry fallback", () => {
|
||||
]);
|
||||
});
|
||||
|
||||
it("falls back to the chain when credential rotation exhausts the retry budget", async () => {
|
||||
const primaryModel = getBundledModel("anthropic", "claude-sonnet-4-5");
|
||||
const fallbackModel = getBundledModel("openai", "gpt-4o-mini");
|
||||
if (!primaryModel || !fallbackModel) {
|
||||
throw new Error("Expected bundled test models to exist");
|
||||
}
|
||||
|
||||
const requestedModels: string[] = [];
|
||||
const mock = createMockModel();
|
||||
const agent = new Agent({
|
||||
getApiKey: model => `${model.provider}-test-key`,
|
||||
initialState: {
|
||||
model: primaryModel,
|
||||
systemPrompt: ["Test"],
|
||||
tools: [],
|
||||
messages: [],
|
||||
},
|
||||
streamFn: (model, context, options) => {
|
||||
requestedModels.push(`${model.provider}/${model.id}`);
|
||||
if (model.provider === primaryModel.provider && model.id === primaryModel.id) {
|
||||
mock.push({ throw: "429 usage_limit_reached" });
|
||||
} else {
|
||||
mock.push({ content: [`ok:${model.provider}/${model.id}`] });
|
||||
}
|
||||
return mock.stream(model, context, options);
|
||||
},
|
||||
});
|
||||
|
||||
// Rotation always claims a sibling credential is available — the shape
|
||||
// of a multi-account pool where the sibling check passes but every
|
||||
// subsequent request keeps failing on the same capped account.
|
||||
vi.spyOn(modelRegistry.authStorage, "markUsageLimitReached").mockResolvedValue({ switched: true });
|
||||
|
||||
const settings = Settings.isolated({
|
||||
"compaction.enabled": false,
|
||||
"retry.baseDelayMs": 5,
|
||||
"retry.maxRetries": 2,
|
||||
"retry.fallbackChains": {
|
||||
[`${primaryModel.provider}/${primaryModel.id}`]: [`${fallbackModel.provider}/${fallbackModel.id}`],
|
||||
},
|
||||
});
|
||||
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
sessionManager: SessionManager.inMemory(),
|
||||
settings,
|
||||
modelRegistry,
|
||||
});
|
||||
const { retryStartEvents, retryEndEvents } = trackRetryEvents(session);
|
||||
|
||||
await session.prompt("Exhaust rotation, then fail over");
|
||||
await session.waitForIdle();
|
||||
|
||||
// Two rotation retries burn the budget on the primary; the exhausted
|
||||
// attempt consults the chain instead of giving up.
|
||||
expect(requestedModels).toEqual([
|
||||
`${primaryModel.provider}/${primaryModel.id}`,
|
||||
`${primaryModel.provider}/${primaryModel.id}`,
|
||||
`${primaryModel.provider}/${primaryModel.id}`,
|
||||
`${fallbackModel.provider}/${fallbackModel.id}`,
|
||||
]);
|
||||
expect(session.model?.provider).toBe(fallbackModel.provider);
|
||||
expect(session.model?.id).toBe(fallbackModel.id);
|
||||
// The fallback model gets a fresh retry budget (attempt resets to 1).
|
||||
expect(retryStartEvents.map(event => event.attempt)).toEqual([1, 2, 1]);
|
||||
expect(retryEndEvents).toHaveLength(1);
|
||||
expect(retryEndEvents[0]).toMatchObject({ success: true });
|
||||
});
|
||||
|
||||
it("applies a provider-wildcard chain to any model of that provider", async () => {
|
||||
const primaryModel = getBundledModel("anthropic", "claude-opus-4-1");
|
||||
const fallbackModel = getBundledModel("openai", "gpt-4o-mini");
|
||||
|
||||
Reference in New Issue
Block a user