fix(ai): fixed OpenRouter cache write attribution in usage parsing
- Updated parseChunkUsage to subtract prompt_tokens_details.cache_write_tokens from prompt token input so OpenRouter write tokens are not misclassified as billable input. - Set cacheWrite and total token counts to include cache-write usage, while preserving cache-read behavior from existing cached_tokens handling. - Added OpenRouter attribution tests verifying cacheWrite and cacheRead totals for write-heavy and cache-warm prompts.
This commit is contained in:
@@ -970,9 +970,17 @@ export function parseChunkUsage(
|
||||
getOptionalNumberProperty(rawUsage, "cached_tokens") ??
|
||||
(promptTokenDetails ? getOptionalNumberProperty(promptTokenDetails, "cached_tokens") : undefined) ??
|
||||
0;
|
||||
// OpenRouter exposes cache writes via `prompt_tokens_details.cache_write_tokens`
|
||||
// and INCLUDES them in `prompt_tokens`. Without subtracting, cache-write tokens
|
||||
// leak into `input` (e.g. GLM/Anthropic via OpenRouter on a fresh cache).
|
||||
// Ref: https://openrouter.ai/docs/guides/best-practices/prompt-caching
|
||||
const cacheWriteTokens = promptTokenDetails
|
||||
? (getOptionalNumberProperty(promptTokenDetails, "cache_write_tokens") ?? 0)
|
||||
: 0;
|
||||
const reasoningTokens =
|
||||
(completionTokenDetails ? getOptionalNumberProperty(completionTokenDetails, "reasoning_tokens") : undefined) ?? 0;
|
||||
const input = (getOptionalNumberProperty(rawUsage, "prompt_tokens") ?? 0) - cachedTokens;
|
||||
const promptTokens = getOptionalNumberProperty(rawUsage, "prompt_tokens") ?? 0;
|
||||
const input = Math.max(0, promptTokens - cachedTokens - cacheWriteTokens);
|
||||
// Per OpenAI's CompletionUsage spec, `reasoning_tokens` is a subset of
|
||||
// `completion_tokens` (which is the total billed output). Adding them would
|
||||
// double-count.
|
||||
@@ -981,8 +989,8 @@ export function parseChunkUsage(
|
||||
input,
|
||||
output: outputTokens,
|
||||
cacheRead: cachedTokens,
|
||||
cacheWrite: 0,
|
||||
totalTokens: input + outputTokens + cachedTokens,
|
||||
cacheWrite: cacheWriteTokens,
|
||||
totalTokens: input + outputTokens + cachedTokens + cacheWriteTokens,
|
||||
...(reasoningTokens > 0 ? { reasoningTokens } : {}),
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
...(copilotPremiumRequests !== undefined ? { premiumRequests: copilotPremiumRequests } : {}),
|
||||
|
||||
@@ -55,6 +55,44 @@ describe("openai-completions parseChunkUsage", () => {
|
||||
expect(usage.reasoningTokens).toBeUndefined();
|
||||
expect(usage.output).toBe(25);
|
||||
});
|
||||
|
||||
it("attributes OpenRouter cache_write_tokens to cacheWrite, not input", () => {
|
||||
// OpenRouter (https://openrouter.ai/docs/guides/best-practices/prompt-caching)
|
||||
// reports cache writes via prompt_tokens_details.cache_write_tokens and
|
||||
// INCLUDES them in prompt_tokens. Naively subtracting only cached_tokens
|
||||
// leaves cache-write tokens stuck in `input`.
|
||||
const usage = parseChunkUsage(
|
||||
{
|
||||
prompt_tokens: 6_000,
|
||||
completion_tokens: 250,
|
||||
prompt_tokens_details: { cached_tokens: 0, cache_write_tokens: 5_500 },
|
||||
},
|
||||
OPENAI_MODEL,
|
||||
undefined,
|
||||
);
|
||||
|
||||
expect(usage.input).toBe(500);
|
||||
expect(usage.cacheWrite).toBe(5_500);
|
||||
expect(usage.cacheRead).toBe(0);
|
||||
expect(usage.totalTokens).toBe(6_250);
|
||||
});
|
||||
|
||||
it("attributes OpenRouter cache_read_tokens correctly when cache is warm", () => {
|
||||
const usage = parseChunkUsage(
|
||||
{
|
||||
prompt_tokens: 6_000,
|
||||
completion_tokens: 250,
|
||||
prompt_tokens_details: { cached_tokens: 5_800, cache_write_tokens: 0 },
|
||||
},
|
||||
OPENAI_MODEL,
|
||||
undefined,
|
||||
);
|
||||
|
||||
expect(usage.input).toBe(200);
|
||||
expect(usage.cacheRead).toBe(5_800);
|
||||
expect(usage.cacheWrite).toBe(0);
|
||||
expect(usage.totalTokens).toBe(6_250);
|
||||
});
|
||||
});
|
||||
|
||||
describe("anthropic applyAnthropicUsageExtras", () => {
|
||||
|
||||
@@ -46,7 +46,10 @@ interface BenchmarkClient {
|
||||
onEvent(listener: (event: { type: string; [key: string]: unknown }) => void): () => void;
|
||||
prompt(text: string): Promise<void>;
|
||||
followUp(text: string): Promise<void>;
|
||||
getSessionStats(): Promise<{ tokens: { input: number; output: number; total: number }; assistantMessages: number }>;
|
||||
getSessionStats(): Promise<{
|
||||
tokens: { input: number; output: number; cacheRead: number; cacheWrite: number; total: number };
|
||||
assistantMessages: number;
|
||||
}>;
|
||||
getLastAssistantText(): Promise<string | null>;
|
||||
getMessages(): Promise<AgentMessage[]>;
|
||||
getState(): Promise<ConversationDumpSessionState>;
|
||||
@@ -1918,22 +1921,32 @@ function estimateTokens(text: string): number {
|
||||
return Math.ceil(text.length / 4);
|
||||
}
|
||||
|
||||
function diffTokenStats(
|
||||
before: { tokens: { input: number; output: number; total: number }; assistantMessages: number },
|
||||
after: { tokens: { input: number; output: number; total: number }; assistantMessages: number },
|
||||
systemPromptTokens: number,
|
||||
): TokenStats {
|
||||
// The system prompt (and tool definitions) live in cacheRead/cacheWrite, not in `input`.
|
||||
// `input` already excludes the cached system prompt; only `total` (which sums cache too)
|
||||
// needs the overhead subtracted, once per LLM call.
|
||||
function diffTokenStats(before: SessionTokenStats, after: SessionTokenStats, systemPromptTokens: number): TokenStats {
|
||||
// `input` here is the total prompt tokens delivered to the model on the wire,
|
||||
// summed across all four buckets the providers expose: non-cached input,
|
||||
// cacheRead, cacheWrite. Summing makes the metric comparable across providers
|
||||
// with different caching behavior — Anthropic with a hot cache reports its
|
||||
// prompt entirely under cacheRead/cacheWrite while non-caching providers put
|
||||
// the same content under `input`.
|
||||
//
|
||||
// The system prompt and tool definitions are constant per-call overhead. We
|
||||
// subtract `calls * systemPromptTokens` once per assistant turn so the
|
||||
// reported figure reflects task-driven prompt cost rather than fixed boilerplate.
|
||||
const calls = Math.max(0, after.assistantMessages - before.assistantMessages);
|
||||
const overhead = calls * systemPromptTokens;
|
||||
const input = Math.max(0, after.tokens.input - before.tokens.input);
|
||||
const beforePrompt = before.tokens.input + before.tokens.cacheRead + before.tokens.cacheWrite;
|
||||
const afterPrompt = after.tokens.input + after.tokens.cacheRead + after.tokens.cacheWrite;
|
||||
const input = Math.max(0, afterPrompt - beforePrompt - overhead);
|
||||
const output = Math.max(0, after.tokens.output - before.tokens.output);
|
||||
const total = Math.max(0, after.tokens.total - before.tokens.total - overhead);
|
||||
const total = input + output;
|
||||
return { input, output, total };
|
||||
}
|
||||
|
||||
type SessionTokenStats = {
|
||||
tokens: { input: number; output: number; cacheRead: number; cacheWrite: number };
|
||||
assistantMessages: number;
|
||||
};
|
||||
|
||||
function isTransportFailure(r: TaskRunResult): boolean {
|
||||
if (r.success) return false;
|
||||
const err = r.error ?? "";
|
||||
|
||||
Reference in New Issue
Block a user