diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index f46ee06ba..60cfda588 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -17,6 +17,7 @@ - Fixed Gemini thought summaries occasionally leaking a raw `` ```thinking `` / `` ``````thinking `` fence delimiter into the reasoning block, so it no longer shows up as fence spam in the thinking display or persisted transcripts ([#8719](https://github.com/can1357/oh-my-pi/issues/8719)). - Fixed the OpenCode Go login prompting for an "OpenCode Zen API key": the shared login flow now names the provider you selected, so connecting OpenCode Go asks for an OpenCode Go key (the `opencode.ai/auth` console is still shared, as documented upstream) ([#8738](https://github.com/can1357/oh-my-pi/issues/8738)). - Fixed Anthropic-compatible endpoints with strict prompt validation (e.g. Z.AI GLM `api.z.ai/api/anthropic`, which rejects the whole request with `400 code 1213 "The prompt parameter was not received normally"`) failing sessions once a tool returned empty output on a vision-capable model: empty successful `tool_result` blocks now encode as `content: ""` instead of `content: []`, which both the official API and strict compatible endpoints accept. +- Fixed `retry.usageReservePct` (Reserve Margin) ignoring Claude Fable/Mythos weekly tier usage until it hit 100%, so a Fable model kept serving turns past the configured reserve; reserve health now honors the mapped tier row while credential-wide hard blocks still require confirmed exhaustion ([#8773](https://github.com/can1357/oh-my-pi/issues/8773)). ## [17.3.5] - 2026-08-16 diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 9504a63df..f314ed245 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -3983,7 +3983,12 @@ export class AuthStorage { } if (!report) return { credentialId: entry.id, credentialType, state: "unknown" }; - const limits = this.#getScopedUsageLimits(strategy, report, rankingContext); + // Reserve health is opt-in and non-destructive: prefer the strategy's + // reserve scoping, which may expose mapped model/tier rows that the + // hard-block scoper withholds until confirmed exhaustion. + const limits = + strategy.scopeLimitsForReserve?.(report, rankingContext) ?? + this.#getScopedUsageLimits(strategy, report, rankingContext); if (limits.length === 0) return { credentialId: entry.id, credentialType, state: "unknown" }; const currentLimits = limits.filter(limit => { diff --git a/packages/ai/src/usage.ts b/packages/ai/src/usage.ts index 2cd7b0f50..1033a11f9 100644 --- a/packages/ai/src/usage.ts +++ b/packages/ai/src/usage.ts @@ -365,6 +365,16 @@ export interface CredentialRankingStrategy { * account-wide quotas can omit this and use all limits. */ scopeLimits?(report: UsageReport, context?: CredentialRankingContext): UsageLimit[]; + /** + * Restrict limits for the opt-in, non-destructive usage-reserve health + * check ({@link AuthStorage.getModelUsageHealth}). Distinct from + * {@link scopeLimits}, which gates credential-wide hard blocks: a provider + * whose model/tier counters are trusted only at confirmed exhaustion for + * hard-blocking can still expose them here so the reserve margin protects + * the mapped quota before it hits the cap. Falls back to {@link scopeLimits} + * when omitted. + */ + scopeLimitsForReserve?(report: UsageReport, context?: CredentialRankingContext): UsageLimit[]; /** * Return a provider-local backoff scope for the requested model. Providers * with backend-specific quotas use this so one exhausted model family does diff --git a/packages/ai/src/usage/claude.ts b/packages/ai/src/usage/claude.ts index 4957f3095..e474fca6b 100644 --- a/packages/ai/src/usage/claude.ts +++ b/packages/ai/src/usage/claude.ts @@ -713,7 +713,9 @@ function getClaudeModelKind(context: CredentialRankingContext | undefined): Clau * Claude model-scoped rows are only relevant to the matching model family. * Credential-wide exhaustion checks stay on shared umbrella windows unless the * request model parses to a concrete Anthropic kind, preventing a Fable cap from - * suppressing unrelated Opus/Sonnet traffic. + * suppressing unrelated Opus/Sonnet traffic. Feeds ranking pressure and the + * opt-in reserve-health scope (`scopeLimitsForReserve`); credential-wide hard + * blocks use {@link scopeClaudeLimitsForModelHardBlock} instead. */ function scopeClaudeLimitsForModel(report: UsageReport, context: CredentialRankingContext | undefined): UsageLimit[] { const kind = getClaudeModelKind(context); @@ -742,7 +744,7 @@ function isConfirmedExhaustedTierRow(limit: UsageLimit, nowMs: number): boolean * weekly caps participate only when {@link isConfirmedExhaustedTierRow} * confirms them, so a confirmed-dead account is skipped up front and a * reactive 429 block extends to the tier reset in markUsageLimitReached, - * while unconfirmed rows remain ranking pressure only via + * while unconfirmed rows remain ranking pressure and opt-in reserve health via * scopeClaudeLimitsForModel. */ function scopeClaudeLimitsForModelHardBlock( @@ -810,6 +812,11 @@ export const claudeRankingStrategy: CredentialRankingStrategy = { return { primary, secondary }; }, scopeLimits: scopeClaudeLimitsForModelHardBlock, + // Reserve health is a non-destructive fallback, not a credential hard + // block, so it trusts the mapped tier row before confirmed exhaustion: a + // Fable/Mythos weekly cap inside the reserve margin should move the turn to + // a healthy candidate rather than serve until 100%. + scopeLimitsForReserve: scopeClaudeLimitsForModel, /** * Fable/Mythos usage-limit errors map to tier-local weekly counters. Scope * reactive backoff blocks for those tiers, mirroring the per-counter diff --git a/packages/ai/test/auth-storage-model-usage-health.test.ts b/packages/ai/test/auth-storage-model-usage-health.test.ts index 84116152f..065194bc2 100644 --- a/packages/ai/test/auth-storage-model-usage-health.test.ts +++ b/packages/ai/test/auth-storage-model-usage-health.test.ts @@ -7,6 +7,7 @@ import { type StoredAuthCredential, } from "@oh-my-pi/pi-ai/auth-storage"; import type { CredentialRankingStrategy, UsageLimit, UsageProvider, UsageReport } from "@oh-my-pi/pi-ai/usage"; +import { claudeRankingStrategy } from "@oh-my-pi/pi-ai/usage/claude"; import { logger } from "@oh-my-pi/pi-utils"; interface CacheEntry { @@ -609,3 +610,108 @@ describe("AuthStorage corrupt persisted block store", () => { expect(errorSpy).toHaveBeenCalledTimes(scenarios.length); }); }); + +describe("AuthStorage Claude tier reserve health", () => { + const storages: AuthStorage[] = []; + afterEach(() => { + for (const storage of storages) storage.close(); + storages.length = 0; + }); + + const WEEK_MS = 7 * 24 * 60 * 60 * 1000; + const HOUR_MS = 60 * 60 * 1000; + + function claudeLimit(opts: { + id: string; + windowId: "5h" | "7d"; + usedFraction: number; + shared?: boolean; + tier?: string; + exhausted?: boolean; + }): UsageLimit { + const resetsAt = Date.now() + 3 * 24 * 60 * 60 * 1000; + return { + id: opts.id, + label: opts.id, + scope: { + provider: "anthropic", + windowId: opts.windowId, + ...(opts.tier !== undefined ? { tier: opts.tier } : {}), + ...(opts.shared !== undefined ? { shared: opts.shared } : {}), + }, + window: { + id: opts.windowId, + label: opts.windowId, + durationMs: opts.windowId === "5h" ? 5 * HOUR_MS : WEEK_MS, + resetsAt, + }, + amount: { usedFraction: opts.usedFraction, unit: "percent" }, + status: opts.exhausted ? "exhausted" : "ok", + }; + } + + async function createClaudeStorage(reports: Record): Promise { + const storage = new AuthStorage(makeStore([oauthRow(1)]), { + usageProviderResolver: provider => (provider === "anthropic" ? makeUsageProvider(reports) : undefined), + rankingStrategyResolver: provider => (provider === "anthropic" ? claudeRankingStrategy : undefined), + configValueResolver: async value => value, + }); + await storage.reload(); + storages.push(storage); + return storage; + } + + function tieredReport(fableUsedFraction: number, exhausted = false): UsageReport { + return report("account-1", [ + claudeLimit({ id: "anthropic:5h", windowId: "5h", usedFraction: 0.1, shared: true }), + claudeLimit({ id: "anthropic:7d", windowId: "7d", usedFraction: 0.72, shared: true }), + claudeLimit({ + id: "anthropic:7d:fable", + windowId: "7d", + usedFraction: fableUsedFraction, + tier: "fable", + exhausted, + }), + ]); + } + + it("reports reserve when the mapped Fable tier row is inside the reserve margin", async () => { + const storage = await createClaudeStorage({ "account-1": tieredReport(0.96) }); + const health = await storage.getModelUsageHealth("anthropic", { + modelId: "claude-fable-5", + reserveFraction: 0.1, + }); + expect(health.state).toBe("reserve"); + expect(health.accounts[0]?.remainingFraction).toBeCloseTo(0.04); + }); + + it("keeps a Fable model healthy while its tier row stays outside the reserve margin", async () => { + const storage = await createClaudeStorage({ "account-1": tieredReport(0.85) }); + const health = await storage.getModelUsageHealth("anthropic", { + modelId: "claude-fable-5", + reserveFraction: 0.1, + }); + expect(health.state).toBe("healthy"); + expect(health.accounts[0]?.remainingFraction).toBeCloseTo(0.15); + }); + + it("does not let a Fable-only tier row pull unrelated Opus traffic into reserve", async () => { + const storage = await createClaudeStorage({ "account-1": tieredReport(0.96) }); + const health = await storage.getModelUsageHealth("anthropic", { + modelId: "claude-opus-4-8", + reserveFraction: 0.1, + }); + expect(health.state).toBe("healthy"); + expect(health.accounts[0]?.remainingFraction).toBeCloseTo(0.28); + }); + + it("still depletes a Fable model on a confirmed 100% tier row", async () => { + const storage = await createClaudeStorage({ "account-1": tieredReport(1, true) }); + const health = await storage.getModelUsageHealth("anthropic", { + modelId: "claude-fable-5", + reserveFraction: 0.1, + }); + expect(health.state).toBe("depleted"); + expect(health.accounts[0]?.resetsAt).toBeGreaterThan(Date.now()); + }); +});