From f02679bdf58ba39d113137e564cc52408f2e7bfe Mon Sep 17 00:00:00 2001 From: MertSoylu Date: Fri, 14 Aug 2026 12:10:18 +0300 Subject: [PATCH] fix(ai): report Umans usage from weighted effective requests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Split the request limit into a weighted soft-cap row and a raw burst-ceiling row so healthy accounts no longer read as exhausted; surface the rolling window's resets_at as a countdown. Closes #7858. Generated with Codebuff 🤖 Co-Authored-By: Codebuff --- docs/provider-quirks.md | 2 +- packages/ai/CHANGELOG.md | 1 + packages/ai/src/usage/umans.ts | 103 +++++++++++++++---- packages/ai/test/umans-usage.test.ts | 147 +++++++++++++++++++++++++-- 4 files changed, 225 insertions(+), 28 deletions(-) diff --git a/docs/provider-quirks.md b/docs/provider-quirks.md index 294394a8d..c2bc3c1db 100644 --- a/docs/provider-quirks.md +++ b/docs/provider-quirks.md @@ -1456,7 +1456,7 @@ Umans AI Coding Plan is a proxy service for AI coding models, operating via the ### Auth & usage - **Auth**: Uses `UMANS_AI_CODING_PLAN_API_KEY` environment variable or `/login umans` key prompt (`packages/ai/src/registry/umans.ts`, `packages/ai/src/registry/registry.ts`). Key validation executes a lightweight Anthropic messages call (`max_tokens: 1`) to `https://api.code.umans.ai/v1/messages`. - **Usage endpoint**: Fetches quota and rate limit status from `GET /v1/usage` (`packages/ai/src/usage/umans.ts`) using `Authorization: Bearer `. -- **Limits surfaced**: Returns a rolling 5-hour request limit (`umans:requests`) and an instantaneous session concurrency limit (`umans:concurrency`). Also surfaces low-priority status notes when rate-limit bursts occur. +- **Limits surfaced**: Returns a rolling 5-hour request split into a model-weighted soft cap (`umans:requests:soft`, the "effective requests" contract) and a raw burst ceiling (`umans:requests:hard`, `hard_cap`), plus an instantaneous session concurrency limit (`umans:concurrency`). The soft cap only ever warns — `exhausted` is reserved for the burst ceiling, where throttling actually starts. Legacy payloads without weighted counters fall back to a single raw `umans:requests` row. Also surfaces low-priority status notes when rate-limit bursts occur. ### Catalog model handling - **Descriptor & discovery**: Registered as `umans` with default model `umans-coder` (`packages/catalog/src/provider-models/descriptors.ts`). Dynamic discovery fetches model details from `GET /v1/models/info` (`packages/catalog/src/provider-models/openai-compat.ts`). diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 50c78e065..304a4ae10 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -5,6 +5,7 @@ ### Fixed - Fixed `omp usage invalidate` to discard stale OAuth and API-key usage snapshots, then force a cache-bypassing, per-provider serialized refresh so upgraded subscriptions do not silently retain pre-change quota data. +- Fixed Umans usage reporting computing utilization from raw request counts instead of the model-weighted "effective requests", falsely reporting the quota as exhausted while the account still had weighted headroom. The request limit is now split into a weighted soft-cap row (warns, never exhausts) and a raw burst-ceiling row (exhausted only where throttling actually starts), and the rolling 5h window's absolute `resets_at` is surfaced as a countdown ([#7858](https://github.com/can1357/oh-my-pi/issues/7858)). ## [17.3.3] - 2026-08-14 diff --git a/packages/ai/src/usage/umans.ts b/packages/ai/src/usage/umans.ts index a171731c8..934638880 100644 --- a/packages/ai/src/usage/umans.ts +++ b/packages/ai/src/usage/umans.ts @@ -23,9 +23,14 @@ interface UmansUsagePayload { requests?: { limit?: number; hard_cap?: number | null; window_seconds?: number }; concurrency?: { limit?: number; hard_cap?: number | null }; }; + /** Rolling 5h window metadata; `resets_at` anchors the status-line countdown. */ + window?: { started_at?: string; resets_at?: string; remaining_minutes?: number }; usage?: { requests_in_window?: number; remaining_requests?: number; + /** Model-weighted "effective requests" (umans-flash counts 0.5). */ + weighted_in_window?: number; + weighted_remaining_requests?: number; concurrent_sessions?: number; tokens_in?: number; tokens_out?: number; @@ -55,6 +60,18 @@ function resolveStatus(usedFraction: number | undefined): UsageStatus | undefine return "ok"; } +/** + * Soft-cap status never reaches `exhausted`: hitting the effective-request + * limit only means burst headroom is being consumed — Umans throttles (429) + * only near the burst ceiling, which the hard row tracks. `exhausted` must + * stay off this row or the usage-aware fallback demotes a healthy account. + */ +function softCapStatus(usedFraction: number | undefined): UsageStatus | undefined { + if (usedFraction === undefined) return undefined; + if (usedFraction >= 0.9) return "warning"; + return "ok"; +} + function buildAmount(args: { used: number | undefined; limit: number | undefined; @@ -75,30 +92,78 @@ function buildAmount(args: { }; } -function buildRequestsLimit(payload: UmansUsagePayload, provider: string): UsageLimit | null { +function buildRequestsLimits(payload: UmansUsagePayload, provider: string): UsageLimit[] { const limit = toFiniteNumber(payload.limits?.requests?.limit); + const hardCap = toFiniteNumber(payload.limits?.requests?.hard_cap); const windowSeconds = toFiniteNumber(payload.limits?.requests?.window_seconds); - const used = toFiniteNumber(payload.usage?.requests_in_window); - const remaining = toFiniteNumber(payload.usage?.remaining_requests); - if (limit === undefined && used === undefined) return null; - const amount = buildAmount({ used, limit, remaining, unit: "requests" }); - // Rolling window: each request ages out `window_seconds` after it fired, so - // there is no single reset timestamp. Surface the window size + label only. - // `window.id` is `"5h"` to match the status-line usage segment's window-id - // contract (it only recognizes `"5h"`/`"7d"`); `label` stays human-readable. + const rawUsed = toFiniteNumber(payload.usage?.requests_in_window); + const rawRemaining = toFiniteNumber(payload.usage?.remaining_requests); + const weightedUsed = toFiniteNumber(payload.usage?.weighted_in_window); + const weightedRemaining = toFiniteNumber(payload.usage?.weighted_remaining_requests); + if (limit === undefined && rawUsed === undefined && weightedUsed === undefined) return []; + + // The 5h window is rolling (FIFO: each request ages out five hours after it + // fired), but the payload still reports an absolute `resets_at` for the + // current window epoch — surface it as an incremental countdown (`tick`) + // rather than a hard reset. `window.id` is `"5h"` to match the status-line + // usage segment's window-id contract (it only recognizes `"5h"`/`"7d"`). + let resetsAt: number | undefined; + if (payload.window?.resets_at) { + const parsed = Date.parse(payload.window.resets_at); + resetsAt = Number.isNaN(parsed) ? undefined : parsed; + } const window: UsageWindow = { id: "5h", label: "rolling 5h", durationMs: windowSeconds ? windowSeconds * 1000 : 5 * HOUR_MS, + ...(resetsAt !== undefined ? { resetsAt, resetLabel: "tick" } : {}), }; - return { - id: "umans:requests", - label: "Requests (rolling 5h)", - scope: { provider, windowId: window.id, shared: true }, - window, - amount, - status: resolveStatus(amount.usedFraction), - }; + + // Payloads without weighted counters predate the soft/hard split; keep the + // legacy single row keyed off raw counts. + if (weightedUsed === undefined) { + const amount = buildAmount({ used: rawUsed, limit, remaining: rawRemaining, unit: "requests" }); + return [ + { + id: "umans:requests", + label: "Requests (rolling 5h)", + scope: { provider, windowId: window.id, shared: true }, + window, + amount, + status: resolveStatus(amount.usedFraction), + }, + ]; + } + + // Umans weights requests by model ("effective requests": umans-flash counts + // 0.5), so the weighted counters are the authoritative utilization against + // the soft `limit`; the raw counters include burst/superseded traffic and + // read as exhausted mid-window while the account still has weighted + // headroom (https://github.com/can1357/oh-my-pi/issues/7858). Soft cap hits + // warn; only the burst ceiling (`hard_cap`, raw counts) can exhaust. + const softAmount = buildAmount({ used: weightedUsed, limit, remaining: weightedRemaining, unit: "requests" }); + const limits: UsageLimit[] = [ + { + id: "umans:requests:soft", + label: "Requests (soft cap)", + scope: { provider, windowId: window.id, shared: true }, + window, + amount: softAmount, + status: softCapStatus(softAmount.usedFraction), + }, + ]; + if (hardCap !== undefined && rawUsed !== undefined) { + const hardAmount = buildAmount({ used: rawUsed, limit: hardCap, remaining: undefined, unit: "requests" }); + limits.push({ + id: "umans:requests:hard", + label: "Requests (burst ceiling)", + scope: { provider, windowId: window.id, shared: true }, + window, + amount: hardAmount, + status: resolveStatus(hardAmount.usedFraction), + }); + } + return limits; } function buildConcurrencyLimit(payload: UmansUsagePayload, provider: string): UsageLimit | null { @@ -157,9 +222,7 @@ async function fetchUmansUsage(params: UsageFetchParams, ctx: UsageFetchContext) return null; } - const limits: UsageLimit[] = []; - const requests = buildRequestsLimit(payload, params.provider); - if (requests) limits.push(requests); + const limits: UsageLimit[] = [...buildRequestsLimits(payload, params.provider)]; const concurrency = buildConcurrencyLimit(payload, params.provider); if (concurrency) limits.push(concurrency); if (limits.length === 0) return null; diff --git a/packages/ai/test/umans-usage.test.ts b/packages/ai/test/umans-usage.test.ts index e75b1738d..8dd8b9a4d 100644 --- a/packages/ai/test/umans-usage.test.ts +++ b/packages/ai/test/umans-usage.test.ts @@ -4,6 +4,8 @@ import { umansUsageProvider } from "../src/usage/umans"; const DEFAULT_BASE_URL = "https://api.code.umans.ai"; +const RESETS_AT = "2026-08-06T21:52:21.202174+00:00"; + function umansPayload(overrides: Record = {}): Record { return { plan: { display_name: "Code Max" }, @@ -11,9 +13,16 @@ function umansPayload(overrides: Record = {}): Record { - it("parses the rolling 5h request window into a UsageLimit with used/remaining/fraction", async () => { + it("splits requests into soft-cap (weighted) and burst-ceiling (raw) limits", async () => { const report = await umansUsageProvider.fetchUsage( { provider: "umans", @@ -60,20 +68,145 @@ describe("umans usage provider", () => { { fetch: fakeFetch(umansPayload()) }, ); expect(report).not.toBeNull(); + const soft = report?.limits.find(l => l.id === "umans:requests:soft"); + expect(soft).toBeDefined(); + // Weighted "effective requests" are authoritative against the soft cap: + // 96 effective used of 200. + expect(soft?.amount.used).toBe(96); + expect(soft?.amount.limit).toBe(200); + expect(soft?.amount.remaining).toBe(104); + expect(soft?.amount.usedFraction).toBeCloseTo(0.48, 5); + expect(soft?.amount.remainingFraction).toBeCloseTo(0.52, 5); + expect(soft?.amount.unit).toBe("requests"); + expect(soft?.status).toBe("ok"); + // The rolling 5h window still exposes its absolute `resets_at` as an + // incremental countdown for the status line. + expect(soft?.window?.resetsAt).toBe(Date.parse(RESETS_AT)); + expect(soft?.window?.resetLabel).toBe("tick"); + expect(soft?.window?.durationMs).toBe(18000_000); + expect(soft?.window?.label).toBe("rolling 5h"); + // Raw counts against the burst ceiling are a separate row. + const hard = report?.limits.find(l => l.id === "umans:requests:hard"); + expect(hard).toBeDefined(); + expect(hard?.amount.used).toBe(48); + expect(hard?.amount.limit).toBe(400); + expect(hard?.amount.usedFraction).toBeCloseTo(0.12, 5); + expect(hard?.status).toBe("ok"); + }); + + it("falls back to the legacy raw row when weighted fields are absent", async () => { + const report = await umansUsageProvider.fetchUsage( + { + provider: "umans", + credential: { type: "api_key", apiKey: "sk-test" }, + }, + { + fetch: fakeFetch( + umansPayload({ + window: undefined, + usage: { + requests_in_window: 48, + remaining_requests: 152, + concurrent_sessions: 1, + tokens_in: 0, + tokens_out: 0, + priority: { low: false }, + }, + }), + ), + }, + ); const requests = report?.limits.find(l => l.id === "umans:requests"); expect(requests).toBeDefined(); + expect(report?.limits.some(l => l.id.startsWith("umans:requests:"))).toBe(false); expect(requests?.amount.used).toBe(48); - expect(requests?.amount.limit).toBe(200); expect(requests?.amount.remaining).toBe(152); expect(requests?.amount.usedFraction).toBeCloseTo(0.24, 5); - expect(requests?.amount.remainingFraction).toBeCloseTo(0.76, 5); - expect(requests?.amount.unit).toBe("requests"); - // Rolling window: no fabricated reset timestamp. expect(requests?.window?.resetsAt).toBeUndefined(); - expect(requests?.window?.durationMs).toBe(18000_000); expect(requests?.window?.label).toBe("rolling 5h"); }); + it("does not report exhausted when raw requests exceed the soft cap but weighted headroom remains (#7858)", async () => { + // Real payload from https://github.com/can1357/oh-my-pi/issues/7858: + // raw 838 exceeds the 500 soft cap (previously clamped to 1.0 → false + // exhausted), while weighted "effective requests" are 207/500 with 293 + // remaining — the account continues normally. Raw traffic only reaches + // the burst ceiling (1000) before throttling applies. + const report = await umansUsageProvider.fetchUsage( + { + provider: "umans", + credential: { type: "api_key", apiKey: "sk-test" }, + }, + { + fetch: fakeFetch( + umansPayload({ + limits: { + requests: { limit: 500, hard_cap: 1000, burst_pct: 1.0, window_seconds: 18000 }, + concurrency: { limit: 4, hard_cap: 8, burst_pct: 1.0 }, + }, + usage: { + requests_in_window: 838, + remaining_requests: 0, + weighted_in_window: 207, + weighted_remaining_requests: 293, + concurrent_sessions: 0, + tokens_in: 3_557_477, + tokens_out: 723_550, + priority: { low: false, boxed_until: null, reason: null }, + }, + }), + ), + }, + ); + const soft = report?.limits.find(l => l.id === "umans:requests:soft"); + expect(soft).toBeDefined(); + expect(soft?.amount.used).toBe(207); + expect(soft?.amount.remaining).toBe(293); + expect(soft?.amount.usedFraction).toBeCloseTo(0.414, 3); + expect(soft?.status).toBe("ok"); + expect(soft?.window?.resetsAt).toBe(Date.parse(RESETS_AT)); + const hard = report?.limits.find(l => l.id === "umans:requests:hard"); + expect(hard).toBeDefined(); + expect(hard?.amount.used).toBe(838); + expect(hard?.amount.limit).toBe(1000); + expect(hard?.amount.usedFraction).toBeCloseTo(0.838, 3); + expect(hard?.status).toBe("ok"); + expect(report?.limits.some(l => l.status === "exhausted")).toBe(false); + }); + + it("reserves exhausted for the burst ceiling and warns at the soft cap", async () => { + const report = await umansUsageProvider.fetchUsage( + { + provider: "umans", + credential: { type: "api_key", apiKey: "sk-test" }, + }, + { + fetch: fakeFetch( + umansPayload({ + usage: { + requests_in_window: 1000, + remaining_requests: 0, + weighted_in_window: 500, + weighted_remaining_requests: 0, + concurrent_sessions: 0, + tokens_in: 0, + tokens_out: 0, + priority: { low: false }, + }, + }), + ), + }, + ); + const soft = report?.limits.find(l => l.id === "umans:requests:soft"); + expect(soft?.amount.usedFraction).toBe(1); + // Soft cap hit = burst headroom in use; warn, never exhaust. + expect(soft?.status).toBe("warning"); + const hard = report?.limits.find(l => l.id === "umans:requests:hard"); + expect(hard?.amount.usedFraction).toBe(1); + // Only the raw burst ceiling can exhaust (that's where 429s start). + expect(hard?.status).toBe("exhausted"); + }); + it("emits a concurrency limit from limits.concurrency", async () => { const report = await umansUsageProvider.fetchUsage( {