diff --git a/docs/provider-quirks.md b/docs/provider-quirks.md index c2bc3c1db..8b368f3d8 100644 --- a/docs/provider-quirks.md +++ b/docs/provider-quirks.md @@ -1456,7 +1456,7 @@ Umans AI Coding Plan is a proxy service for AI coding models, operating via the ### Auth & usage - **Auth**: Uses `UMANS_AI_CODING_PLAN_API_KEY` environment variable or `/login umans` key prompt (`packages/ai/src/registry/umans.ts`, `packages/ai/src/registry/registry.ts`). Key validation executes a lightweight Anthropic messages call (`max_tokens: 1`) to `https://api.code.umans.ai/v1/messages`. - **Usage endpoint**: Fetches quota and rate limit status from `GET /v1/usage` (`packages/ai/src/usage/umans.ts`) using `Authorization: Bearer `. -- **Limits surfaced**: Returns a rolling 5-hour request split into a model-weighted soft cap (`umans:requests:soft`, the "effective requests" contract) and a raw burst ceiling (`umans:requests:hard`, `hard_cap`), plus an instantaneous session concurrency limit (`umans:concurrency`). The soft cap only ever warns — `exhausted` is reserved for the burst ceiling, where throttling actually starts. Legacy payloads without weighted counters fall back to a single raw `umans:requests` row. Also surfaces low-priority status notes when rate-limit bursts occur. +- **Limits surfaced**: Returns a rolling 5-hour request split into a model-weighted soft cap (`umans:requests:soft`, the "effective requests" contract) and a raw burst ceiling (`umans:requests:hard`, `hard_cap`), plus an instantaneous session concurrency limit (`umans:concurrency`). The soft cap only ever warns — `exhausted` is reserved for the burst ceiling, where throttling actually starts. Payloads without a reported burst ceiling (`hard_cap`) collapse to a single weighted `umans:requests` row that can exhaust at the effective-request limit, so request exhaustion is never unreportable; legacy payloads without weighted counters fall back to a single raw `umans:requests` row. In both single-row shapes the weighted counter (when present) stays authoritative — raw burst traffic above the limit never fabricates an exhausted state. Also surfaces low-priority status notes when rate-limit bursts occur. ### Catalog model handling - **Descriptor & discovery**: Registered as `umans` with default model `umans-coder` (`packages/catalog/src/provider-models/descriptors.ts`). Dynamic discovery fetches model details from `GET /v1/models/info` (`packages/catalog/src/provider-models/openai-compat.ts`). diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 1a2d0c917..b0e806ac1 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed Umans usage reporting computing utilization from raw request counts instead of the model-weighted "effective requests", falsely reporting the quota as exhausted while the account still had weighted headroom. The request limit is now split into a weighted soft-cap row (warns, never exhausts) and a raw burst-ceiling row (exhausted only where throttling actually starts), and the rolling 5h window's absolute `resets_at` is surfaced as a countdown ([#7858](https://github.com/can1357/oh-my-pi/issues/7858)). +- Fixed Umans usage reporting computing utilization from raw request counts instead of the model-weighted "effective requests", falsely reporting the quota as exhausted while the account still had weighted headroom. The request limit is now split into a weighted soft-cap row (warns, never exhausts) and a raw burst-ceiling row (exhausted only where throttling actually starts), and the rolling 5h window's absolute `resets_at` is surfaced as a countdown ([#7858](https://github.com/can1357/oh-my-pi/issues/7858)). Payloads without a reported burst ceiling collapse to a single weighted `umans:requests` row that can exhaust, so request exhaustion is never unreportable. ## [17.3.4] - 2026-08-14 diff --git a/packages/ai/src/usage/umans.ts b/packages/ai/src/usage/umans.ts index 934638880..61a1e64ee 100644 --- a/packages/ai/src/usage/umans.ts +++ b/packages/ai/src/usage/umans.ts @@ -119,10 +119,20 @@ function buildRequestsLimits(payload: UmansUsagePayload, provider: string): Usag ...(resetsAt !== undefined ? { resetsAt, resetLabel: "tick" } : {}), }; - // Payloads without weighted counters predate the soft/hard split; keep the - // legacy single row keyed off raw counts. - if (weightedUsed === undefined) { - const amount = buildAmount({ used: rawUsed, limit, remaining: rawRemaining, unit: "requests" }); + // Single row: either payloads without weighted counters (legacy) or payloads + // that report weighted usage but no burst ceiling (`hard_cap`). Without a + // burst ceiling there is no hard row to defer exhaustion to, so the + // authoritative counter — weighted when available, else raw — drives the + // single row and CAN exhaust at the limit. Raw burst traffic above the + // limit still never drives exhaustion on its own: weighted headroom stays + // decisive (https://github.com/can1357/oh-my-pi/issues/7858). + if (weightedUsed === undefined || hardCap === undefined) { + const amount = buildAmount({ + used: weightedUsed ?? rawUsed, + limit, + remaining: weightedUsed !== undefined ? weightedRemaining : rawRemaining, + unit: "requests", + }); return [ { id: "umans:requests", diff --git a/packages/ai/test/umans-usage.test.ts b/packages/ai/test/umans-usage.test.ts index 8dd8b9a4d..061604f33 100644 --- a/packages/ai/test/umans-usage.test.ts +++ b/packages/ai/test/umans-usage.test.ts @@ -174,6 +174,90 @@ describe("umans usage provider", () => { expect(report?.limits.some(l => l.status === "exhausted")).toBe(false); }); + it("collapses to a single weighted requests row that can exhaust when no burst ceiling is reported", async () => { + // Weighted counters present but `hard_cap` absent: without a burst + // ceiling there is no hard row to defer exhaustion to, so the weighted + // effective-request budget is the operative ceiling — the single row + // must be able to report `exhausted` or a spent account could never + // trigger the usage-aware fallback. + const report = await umansUsageProvider.fetchUsage( + { + provider: "umans", + credential: { type: "api_key", apiKey: "sk-test" }, + }, + { + fetch: fakeFetch( + umansPayload({ + limits: { + requests: { limit: 200, window_seconds: 18000 }, + concurrency: { limit: 4, hard_cap: 8, burst_pct: 1.0 }, + }, + usage: { + requests_in_window: 400, + remaining_requests: 0, + weighted_in_window: 200, + weighted_remaining_requests: 0, + concurrent_sessions: 0, + tokens_in: 0, + tokens_out: 0, + priority: { low: false }, + }, + }), + ), + }, + ); + const requests = report?.limits.find(l => l.id === "umans:requests"); + expect(requests).toBeDefined(); + // No soft/hard split without a reported burst ceiling. + expect(report?.limits.some(l => l.id.startsWith("umans:requests:"))).toBe(false); + // Weighted effective requests are authoritative: raw 400 overshoots the + // 200 limit, but it is the weighted 200/200 that reports exhausted. + expect(requests?.amount.used).toBe(200); + expect(requests?.amount.limit).toBe(200); + expect(requests?.amount.usedFraction).toBe(1); + expect(requests?.status).toBe("exhausted"); + }); + + it("keeps weighted headroom decisive when no burst ceiling is reported", async () => { + // Same #7858 shape (raw usage over the soft limit, weighted headroom + // remaining) but with no `hard_cap` in the payload: the weighted counter + // must still decide, so raw burst traffic cannot fabricate an exhausted + // state even when there is no hard row to buffer it. + const report = await umansUsageProvider.fetchUsage( + { + provider: "umans", + credential: { type: "api_key", apiKey: "sk-test" }, + }, + { + fetch: fakeFetch( + umansPayload({ + limits: { + requests: { limit: 200, window_seconds: 18000 }, + concurrency: { limit: 4, hard_cap: 8, burst_pct: 1.0 }, + }, + usage: { + requests_in_window: 300, + remaining_requests: 0, + weighted_in_window: 100, + weighted_remaining_requests: 100, + concurrent_sessions: 0, + tokens_in: 0, + tokens_out: 0, + priority: { low: false }, + }, + }), + ), + }, + ); + const requests = report?.limits.find(l => l.id === "umans:requests"); + expect(requests).toBeDefined(); + expect(requests?.amount.used).toBe(100); + expect(requests?.amount.remaining).toBe(100); + expect(requests?.amount.usedFraction).toBeCloseTo(0.5, 5); + expect(requests?.status).toBe("ok"); + expect(report?.limits.some(l => l.status === "exhausted")).toBe(false); + }); + it("reserves exhausted for the burst ceiling and warns at the soft cap", async () => { const report = await umansUsageProvider.fetchUsage( {