Merge PR #8537: fix(ai): report Umans usage from weighted effective requests (@MertSoylu)

This commit is contained in:
can1357
2026-08-16 02:03:15 +02:00
4 changed files with 319 additions and 28 deletions
+1 -1
View File
@@ -1456,7 +1456,7 @@ Umans AI Coding Plan is a proxy service for AI coding models, operating via the
### Auth & usage
- **Auth**: Uses `UMANS_AI_CODING_PLAN_API_KEY` environment variable or `/login umans` key prompt (`packages/ai/src/registry/umans.ts`, `packages/ai/src/registry/registry.ts`). Key validation executes a lightweight Anthropic messages call (`max_tokens: 1`) to `https://api.code.umans.ai/v1/messages`.
- **Usage endpoint**: Fetches quota and rate limit status from `GET /v1/usage` (`packages/ai/src/usage/umans.ts`) using `Authorization: Bearer <key>`.
- **Limits surfaced**: Returns a rolling 5-hour request limit (`umans:requests`) and an instantaneous session concurrency limit (`umans:concurrency`). Also surfaces low-priority status notes when rate-limit bursts occur.
- **Limits surfaced**: Returns a rolling 5-hour request split into a model-weighted soft cap (`umans:requests:soft`, the "effective requests" contract) and a raw burst ceiling (`umans:requests:hard`, `hard_cap`), plus an instantaneous session concurrency limit (`umans:concurrency`). The soft cap only ever warns — `exhausted` is reserved for the burst ceiling, where throttling actually starts. Payloads without a reported burst ceiling (`hard_cap`) collapse to a single weighted `umans:requests` row that can exhaust at the effective-request limit, so request exhaustion is never unreportable; legacy payloads without weighted counters fall back to a single raw `umans:requests` row. In both single-row shapes the weighted counter (when present) stays authoritative — raw burst traffic above the limit never fabricates an exhausted state. Also surfaces low-priority status notes when rate-limit bursts occur.
### Catalog model handling
- **Descriptor & discovery**: Registered as `umans` with default model `umans-coder` (`packages/catalog/src/provider-models/descriptors.ts`). Dynamic discovery fetches model details from `GET /v1/models/info` (`packages/catalog/src/provider-models/openai-compat.ts`).
+1
View File
@@ -7,6 +7,7 @@
- Stopped treating `XAI_API_KEY` as SuperGrok (`xai-oauth`) sign-in for availability, so paid-key-only setups default to `xai/grok-4.5` instead of the zero-cost SuperGrok catalog path. Explicit `xai-oauth/…` selectors still accept the paid key via the existing env fallback.
- Omitted unsupported `reasoning.summary` on paid xAI Responses requests (`xai/grok-4.5`), matching SuperGrok, so a thinking level no longer serializes `summary: "auto"`.
- Omitted presence/frequency penalties on all first-party xAI Responses models, including non-reasoning ids such as `xai/grok-2`.
- Fixed Umans usage reporting computing utilization from raw request counts instead of the model-weighted "effective requests", falsely reporting the quota as exhausted while the account still had weighted headroom. The request limit is now split into a weighted soft-cap row (warns, never exhausts) and a raw burst-ceiling row (exhausted only where throttling actually starts), and the rolling 5h window's absolute `resets_at` is surfaced as a countdown ([#7858](https://github.com/can1357/oh-my-pi/issues/7858)). Payloads without a reported burst ceiling collapse to a single weighted `umans:requests` row that can exhaust, so request exhaustion is never unreportable.
## [17.3.4] - 2026-08-14
+87 -14
View File
@@ -23,9 +23,14 @@ interface UmansUsagePayload {
requests?: { limit?: number; hard_cap?: number | null; window_seconds?: number };
concurrency?: { limit?: number; hard_cap?: number | null };
};
/** Rolling 5h window metadata; `resets_at` anchors the status-line countdown. */
window?: { started_at?: string; resets_at?: string; remaining_minutes?: number };
usage?: {
requests_in_window?: number;
remaining_requests?: number;
/** Model-weighted "effective requests" (umans-flash counts 0.5). */
weighted_in_window?: number;
weighted_remaining_requests?: number;
concurrent_sessions?: number;
tokens_in?: number;
tokens_out?: number;
@@ -55,6 +60,18 @@ function resolveStatus(usedFraction: number | undefined): UsageStatus | undefine
return "ok";
}
/**
* Soft-cap status never reaches `exhausted`: hitting the effective-request
* limit only means burst headroom is being consumed — Umans throttles (429)
* only near the burst ceiling, which the hard row tracks. `exhausted` must
* stay off this row or the usage-aware fallback demotes a healthy account.
*/
function softCapStatus(usedFraction: number | undefined): UsageStatus | undefined {
if (usedFraction === undefined) return undefined;
if (usedFraction >= 0.9) return "warning";
return "ok";
}
function buildAmount(args: {
used: number | undefined;
limit: number | undefined;
@@ -75,30 +92,88 @@ function buildAmount(args: {
};
}
function buildRequestsLimit(payload: UmansUsagePayload, provider: string): UsageLimit | null {
function buildRequestsLimits(payload: UmansUsagePayload, provider: string): UsageLimit[] {
const limit = toFiniteNumber(payload.limits?.requests?.limit);
const hardCap = toFiniteNumber(payload.limits?.requests?.hard_cap);
const windowSeconds = toFiniteNumber(payload.limits?.requests?.window_seconds);
const used = toFiniteNumber(payload.usage?.requests_in_window);
const remaining = toFiniteNumber(payload.usage?.remaining_requests);
if (limit === undefined && used === undefined) return null;
const amount = buildAmount({ used, limit, remaining, unit: "requests" });
// Rolling window: each request ages out `window_seconds` after it fired, so
// there is no single reset timestamp. Surface the window size + label only.
// `window.id` is `"5h"` to match the status-line usage segment's window-id
// contract (it only recognizes `"5h"`/`"7d"`); `label` stays human-readable.
const rawUsed = toFiniteNumber(payload.usage?.requests_in_window);
const rawRemaining = toFiniteNumber(payload.usage?.remaining_requests);
const weightedUsed = toFiniteNumber(payload.usage?.weighted_in_window);
const weightedRemaining = toFiniteNumber(payload.usage?.weighted_remaining_requests);
if (limit === undefined && rawUsed === undefined && weightedUsed === undefined) return [];
// The 5h window is rolling (FIFO: each request ages out five hours after it
// fired), but the payload still reports an absolute `resets_at` for the
// current window epoch — surface it as an incremental countdown (`tick`)
// rather than a hard reset. `window.id` is `"5h"` to match the status-line
// usage segment's window-id contract (it only recognizes `"5h"`/`"7d"`).
let resetsAt: number | undefined;
if (payload.window?.resets_at) {
const parsed = Date.parse(payload.window.resets_at);
resetsAt = Number.isNaN(parsed) ? undefined : parsed;
}
const window: UsageWindow = {
id: "5h",
label: "rolling 5h",
durationMs: windowSeconds ? windowSeconds * 1000 : 5 * HOUR_MS,
...(resetsAt !== undefined ? { resetsAt, resetLabel: "tick" } : {}),
};
return {
// Single row: either payloads without weighted counters (legacy) or payloads
// that report weighted usage but no burst ceiling (`hard_cap`). Without a
// burst ceiling there is no hard row to defer exhaustion to, so the
// authoritative counter — weighted when available, else raw — drives the
// single row and CAN exhaust at the limit. Raw burst traffic above the
// limit still never drives exhaustion on its own: weighted headroom stays
// decisive (https://github.com/can1357/oh-my-pi/issues/7858).
if (weightedUsed === undefined || hardCap === undefined) {
const amount = buildAmount({
used: weightedUsed ?? rawUsed,
limit,
remaining: weightedUsed !== undefined ? weightedRemaining : rawRemaining,
unit: "requests",
});
return [
{
id: "umans:requests",
label: "Requests (rolling 5h)",
scope: { provider, windowId: window.id, shared: true },
window,
amount,
status: resolveStatus(amount.usedFraction),
};
},
];
}
// Umans weights requests by model ("effective requests": umans-flash counts
// 0.5), so the weighted counters are the authoritative utilization against
// the soft `limit`; the raw counters include burst/superseded traffic and
// read as exhausted mid-window while the account still has weighted
// headroom (https://github.com/can1357/oh-my-pi/issues/7858). Soft cap hits
// warn; only the burst ceiling (`hard_cap`, raw counts) can exhaust.
const softAmount = buildAmount({ used: weightedUsed, limit, remaining: weightedRemaining, unit: "requests" });
const limits: UsageLimit[] = [
{
id: "umans:requests:soft",
label: "Requests (soft cap)",
scope: { provider, windowId: window.id, shared: true },
window,
amount: softAmount,
status: softCapStatus(softAmount.usedFraction),
},
];
if (hardCap !== undefined && rawUsed !== undefined) {
const hardAmount = buildAmount({ used: rawUsed, limit: hardCap, remaining: undefined, unit: "requests" });
limits.push({
id: "umans:requests:hard",
label: "Requests (burst ceiling)",
scope: { provider, windowId: window.id, shared: true },
window,
amount: hardAmount,
status: resolveStatus(hardAmount.usedFraction),
});
}
return limits;
}
function buildConcurrencyLimit(payload: UmansUsagePayload, provider: string): UsageLimit | null {
@@ -157,9 +232,7 @@ async function fetchUmansUsage(params: UsageFetchParams, ctx: UsageFetchContext)
return null;
}
const limits: UsageLimit[] = [];
const requests = buildRequestsLimit(payload, params.provider);
if (requests) limits.push(requests);
const limits: UsageLimit[] = [...buildRequestsLimits(payload, params.provider)];
const concurrency = buildConcurrencyLimit(payload, params.provider);
if (concurrency) limits.push(concurrency);
if (limits.length === 0) return null;
+224 -7
View File
@@ -4,6 +4,8 @@ import { umansUsageProvider } from "../src/usage/umans";
const DEFAULT_BASE_URL = "https://api.code.umans.ai";
const RESETS_AT = "2026-08-06T21:52:21.202174+00:00";
function umansPayload(overrides: Record<string, unknown> = {}): Record<string, unknown> {
return {
plan: { display_name: "Code Max" },
@@ -11,9 +13,16 @@ function umansPayload(overrides: Record<string, unknown> = {}): Record<string, u
requests: { limit: 200, hard_cap: 400, burst_pct: 1.0, window_seconds: 18000 },
concurrency: { limit: 4, hard_cap: 8, burst_pct: 1.0 },
},
window: {
started_at: "2026-08-06T16:52:21.202174+00:00",
resets_at: RESETS_AT,
remaining_minutes: 9,
},
usage: {
requests_in_window: 48,
remaining_requests: 152,
weighted_in_window: 96,
weighted_remaining_requests: 104,
concurrent_sessions: 1,
tokens_in: 1_200_000,
tokens_out: 340_000,
@@ -49,9 +58,8 @@ function fetchRecorder(
};
return fn as unknown as typeof fetch;
}
describe("umans usage provider", () => {
it("parses the rolling 5h request window into a UsageLimit with used/remaining/fraction", async () => {
it("splits requests into soft-cap (weighted) and burst-ceiling (raw) limits", async () => {
const report = await umansUsageProvider.fetchUsage(
{
provider: "umans",
@@ -60,20 +68,229 @@ describe("umans usage provider", () => {
{ fetch: fakeFetch(umansPayload()) },
);
expect(report).not.toBeNull();
const soft = report?.limits.find(l => l.id === "umans:requests:soft");
expect(soft).toBeDefined();
// Weighted "effective requests" are authoritative against the soft cap:
// 96 effective used of 200.
expect(soft?.amount.used).toBe(96);
expect(soft?.amount.limit).toBe(200);
expect(soft?.amount.remaining).toBe(104);
expect(soft?.amount.usedFraction).toBeCloseTo(0.48, 5);
expect(soft?.amount.remainingFraction).toBeCloseTo(0.52, 5);
expect(soft?.amount.unit).toBe("requests");
expect(soft?.status).toBe("ok");
// The rolling 5h window still exposes its absolute `resets_at` as an
// incremental countdown for the status line.
expect(soft?.window?.resetsAt).toBe(Date.parse(RESETS_AT));
expect(soft?.window?.resetLabel).toBe("tick");
expect(soft?.window?.durationMs).toBe(18000_000);
expect(soft?.window?.label).toBe("rolling 5h");
// Raw counts against the burst ceiling are a separate row.
const hard = report?.limits.find(l => l.id === "umans:requests:hard");
expect(hard).toBeDefined();
expect(hard?.amount.used).toBe(48);
expect(hard?.amount.limit).toBe(400);
expect(hard?.amount.usedFraction).toBeCloseTo(0.12, 5);
expect(hard?.status).toBe("ok");
});
it("falls back to the legacy raw row when weighted fields are absent", async () => {
const report = await umansUsageProvider.fetchUsage(
{
provider: "umans",
credential: { type: "api_key", apiKey: "sk-test" },
},
{
fetch: fakeFetch(
umansPayload({
window: undefined,
usage: {
requests_in_window: 48,
remaining_requests: 152,
concurrent_sessions: 1,
tokens_in: 0,
tokens_out: 0,
priority: { low: false },
},
}),
),
},
);
const requests = report?.limits.find(l => l.id === "umans:requests");
expect(requests).toBeDefined();
expect(report?.limits.some(l => l.id.startsWith("umans:requests:"))).toBe(false);
expect(requests?.amount.used).toBe(48);
expect(requests?.amount.limit).toBe(200);
expect(requests?.amount.remaining).toBe(152);
expect(requests?.amount.usedFraction).toBeCloseTo(0.24, 5);
expect(requests?.amount.remainingFraction).toBeCloseTo(0.76, 5);
expect(requests?.amount.unit).toBe("requests");
// Rolling window: no fabricated reset timestamp.
expect(requests?.window?.resetsAt).toBeUndefined();
expect(requests?.window?.durationMs).toBe(18000_000);
expect(requests?.window?.label).toBe("rolling 5h");
});
it("does not report exhausted when raw requests exceed the soft cap but weighted headroom remains (#7858)", async () => {
// Real payload from https://github.com/can1357/oh-my-pi/issues/7858:
// raw 838 exceeds the 500 soft cap (previously clamped to 1.0 → false
// exhausted), while weighted "effective requests" are 207/500 with 293
// remaining — the account continues normally. Raw traffic only reaches
// the burst ceiling (1000) before throttling applies.
const report = await umansUsageProvider.fetchUsage(
{
provider: "umans",
credential: { type: "api_key", apiKey: "sk-test" },
},
{
fetch: fakeFetch(
umansPayload({
limits: {
requests: { limit: 500, hard_cap: 1000, burst_pct: 1.0, window_seconds: 18000 },
concurrency: { limit: 4, hard_cap: 8, burst_pct: 1.0 },
},
usage: {
requests_in_window: 838,
remaining_requests: 0,
weighted_in_window: 207,
weighted_remaining_requests: 293,
concurrent_sessions: 0,
tokens_in: 3_557_477,
tokens_out: 723_550,
priority: { low: false, boxed_until: null, reason: null },
},
}),
),
},
);
const soft = report?.limits.find(l => l.id === "umans:requests:soft");
expect(soft).toBeDefined();
expect(soft?.amount.used).toBe(207);
expect(soft?.amount.remaining).toBe(293);
expect(soft?.amount.usedFraction).toBeCloseTo(0.414, 3);
expect(soft?.status).toBe("ok");
expect(soft?.window?.resetsAt).toBe(Date.parse(RESETS_AT));
const hard = report?.limits.find(l => l.id === "umans:requests:hard");
expect(hard).toBeDefined();
expect(hard?.amount.used).toBe(838);
expect(hard?.amount.limit).toBe(1000);
expect(hard?.amount.usedFraction).toBeCloseTo(0.838, 3);
expect(hard?.status).toBe("ok");
expect(report?.limits.some(l => l.status === "exhausted")).toBe(false);
});
it("collapses to a single weighted requests row that can exhaust when no burst ceiling is reported", async () => {
// Weighted counters present but `hard_cap` absent: without a burst
// ceiling there is no hard row to defer exhaustion to, so the weighted
// effective-request budget is the operative ceiling — the single row
// must be able to report `exhausted` or a spent account could never
// trigger the usage-aware fallback.
const report = await umansUsageProvider.fetchUsage(
{
provider: "umans",
credential: { type: "api_key", apiKey: "sk-test" },
},
{
fetch: fakeFetch(
umansPayload({
limits: {
requests: { limit: 200, window_seconds: 18000 },
concurrency: { limit: 4, hard_cap: 8, burst_pct: 1.0 },
},
usage: {
requests_in_window: 400,
remaining_requests: 0,
weighted_in_window: 200,
weighted_remaining_requests: 0,
concurrent_sessions: 0,
tokens_in: 0,
tokens_out: 0,
priority: { low: false },
},
}),
),
},
);
const requests = report?.limits.find(l => l.id === "umans:requests");
expect(requests).toBeDefined();
// No soft/hard split without a reported burst ceiling.
expect(report?.limits.some(l => l.id.startsWith("umans:requests:"))).toBe(false);
// Weighted effective requests are authoritative: raw 400 overshoots the
// 200 limit, but it is the weighted 200/200 that reports exhausted.
expect(requests?.amount.used).toBe(200);
expect(requests?.amount.limit).toBe(200);
expect(requests?.amount.usedFraction).toBe(1);
expect(requests?.status).toBe("exhausted");
});
it("keeps weighted headroom decisive when no burst ceiling is reported", async () => {
// Same #7858 shape (raw usage over the soft limit, weighted headroom
// remaining) but with no `hard_cap` in the payload: the weighted counter
// must still decide, so raw burst traffic cannot fabricate an exhausted
// state even when there is no hard row to buffer it.
const report = await umansUsageProvider.fetchUsage(
{
provider: "umans",
credential: { type: "api_key", apiKey: "sk-test" },
},
{
fetch: fakeFetch(
umansPayload({
limits: {
requests: { limit: 200, window_seconds: 18000 },
concurrency: { limit: 4, hard_cap: 8, burst_pct: 1.0 },
},
usage: {
requests_in_window: 300,
remaining_requests: 0,
weighted_in_window: 100,
weighted_remaining_requests: 100,
concurrent_sessions: 0,
tokens_in: 0,
tokens_out: 0,
priority: { low: false },
},
}),
),
},
);
const requests = report?.limits.find(l => l.id === "umans:requests");
expect(requests).toBeDefined();
expect(requests?.amount.used).toBe(100);
expect(requests?.amount.remaining).toBe(100);
expect(requests?.amount.usedFraction).toBeCloseTo(0.5, 5);
expect(requests?.status).toBe("ok");
expect(report?.limits.some(l => l.status === "exhausted")).toBe(false);
});
it("reserves exhausted for the burst ceiling and warns at the soft cap", async () => {
const report = await umansUsageProvider.fetchUsage(
{
provider: "umans",
credential: { type: "api_key", apiKey: "sk-test" },
},
{
fetch: fakeFetch(
umansPayload({
usage: {
requests_in_window: 1000,
remaining_requests: 0,
weighted_in_window: 500,
weighted_remaining_requests: 0,
concurrent_sessions: 0,
tokens_in: 0,
tokens_out: 0,
priority: { low: false },
},
}),
),
},
);
const soft = report?.limits.find(l => l.id === "umans:requests:soft");
expect(soft?.amount.usedFraction).toBe(1);
// Soft cap hit = burst headroom in use; warn, never exhaust.
expect(soft?.status).toBe("warning");
const hard = report?.limits.find(l => l.id === "umans:requests:hard");
expect(hard?.amount.usedFraction).toBe(1);
// Only the raw burst ceiling can exhaust (that's where 429s start).
expect(hard?.status).toBe("exhausted");
});
it("emits a concurrency limit from limits.concurrency", async () => {
const report = await umansUsageProvider.fetchUsage(
{