fix(ai): stop header ingests from sliding the usage-report deadline

This commit is contained in:
LunarECL
2026-07-23 22:26:51 +09:00
parent 2e138ee45d
commit 4a9ad17298
2 changed files with 74 additions and 7 deletions
+6 -7
View File
@@ -3203,13 +3203,12 @@ export class AuthStorage {
};
}
// Header-only reports are last-good hints, not completed usage fetches.
// Keep them durable but stale so the next poll probes the provider, while
// preserving any active failure backoff installed by #fetchUsageCached.
const expiresAt =
merged.metadata?.source === "ratelimit-headers"
? Math.max(priorEntry?.expiresAt ?? now - 1, now - 1)
: now + USAGE_REPORT_TTL_MS;
// Header ingestion merges values but never extends a cache entry's lifetime.
// Preserve the existing expiry (including active failure cooldowns) so full
// reports refetch on their original 5-minute schedule and full-payload-only
// rows such as extra usage stay current; headers only refresh window rows
// between fetches. A newly minted header-only report is durable but stale.
const expiresAt = Math.max(priorEntry?.expiresAt ?? now - 1, now - 1);
this.#usageCache.set(cacheKey, { value: merged, expiresAt });
this.#usageHeaderIngestAt.set(cacheKey, now);
return true;
@@ -492,6 +492,74 @@ describe("AuthStorage usage cache: header ingestion", () => {
expect(calls).toBe(2);
});
it("does not let header ingestion slide the full-report refresh deadline", async () => {
const start = Date.now();
const now = vi.spyOn(Date, "now").mockReturnValue(start);
vi.spyOn(Math, "random").mockReturnValue(0.5);
const makeFullReport = (extraUsed: number): UsageReport => {
const baseReport = makeTieredReport("a@example.com");
return {
...baseReport,
limits: [
...baseReport.limits,
{
id: "anthropic:extra",
label: "Claude Extra Usage",
scope: { provider: "anthropic", windowId: "extra" },
amount: {
used: extraUsed,
limit: 100,
usedFraction: extraUsed / 100,
unit: "usd",
},
status: "ok",
},
],
raw: { extra_usage: { used: extraUsed * 100, limit: 10_000 } },
};
};
const firstFullReport = makeFullReport(12.34);
const secondFullReport = makeFullReport(56.78);
const fetchSpy = vi
.spyOn(claudeUsage.claudeUsageProvider, "fetchUsage")
.mockResolvedValueOnce(firstFullReport)
.mockResolvedValue(secondFullReport);
const initialReport = requireAnthropicReport(await storage.fetchUsageReports());
expect(fetchSpy).toHaveBeenCalledTimes(1);
expect(requireLimit(initialReport, "anthropic:extra").amount.used).toBe(12.34);
expect(await storage.getApiKey("anthropic", "sliding-session")).toBe("oat-1");
now.mockReturnValue(start + 60_000);
expect(
storage.ingestUsageHeaders("anthropic", usageHeaders("0.05", "0.6"), {
sessionId: "sliding-session",
}),
).toBe(true);
now.mockReturnValue(start + 120_000);
expect(
storage.ingestUsageHeaders("anthropic", usageHeaders("0.06", "0.61"), {
sessionId: "sliding-session",
}),
).toBe(true);
now.mockReturnValue(start + 240_000);
expect(
storage.ingestUsageHeaders("anthropic", usageHeaders("0.07", "0.62"), {
sessionId: "sliding-session",
}),
).toBe(true);
now.mockReturnValue(start + 299_999);
const beforeDeadline = requireAnthropicReport(await storage.fetchUsageReports());
expect(fetchSpy).toHaveBeenCalledTimes(1);
expect(requireLimit(beforeDeadline, "anthropic:extra").amount.used).toBe(12.34);
now.mockReturnValue(start + 376_000);
const refreshed = requireAnthropicReport(await storage.fetchUsageReports());
expect(fetchSpy).toHaveBeenCalledTimes(2);
expect(requireLimit(refreshed, "anthropic:extra").amount.used).toBe(56.78);
});
it("merges header umbrella windows onto the last real report and preserves tier limits", async () => {
const realReport = makeTieredReport("a@example.com");
let calls = 0;