From 084488b6809e1dc0dd6f48846b54b85a69a7f21f Mon Sep 17 00:00:00 2001 From: Miroslav Drbal Date: Wed, 13 May 2026 19:26:38 +0200 Subject: [PATCH] fix(coding-agent): exclude cacheRead from token display, add per-subagent cost Token counter (token_total status-line segment, subagent progress tree, session-observer stats line) previously included cacheRead in its cumulative sum. With Anthropic prompt caching, cacheRead per turn equals the full cached context, so summing across N turns gives N*context_size -- a session with a 1M context and 5 turns showed ~5M tokens despite no compaction occurring. Fix: display shows input + output + cacheWrite per turn. cacheWrite is kept because each byte is written once; cacheRead re-reads the same context every turn. Dedicated cache_read/cache_write status-line segments still show cache activity; billing cost is unaffected. Also adds per-subagent cost display (dollar amount, statusLineCost color) accumulated incrementally from message_end events. Hidden when cost is zero (subscription/OAuth providers). Brings token and cost display in line with what Claude Code shows per-agent. --- packages/coding-agent/CHANGELOG.md | 6 +++++ .../components/session-observer-overlay.ts | 5 +++- .../modes/components/status-line/segments.ts | 7 ++++-- packages/coding-agent/src/task/executor.ts | 24 +++++++++++++------ packages/coding-agent/src/task/index.ts | 2 ++ packages/coding-agent/src/task/render.ts | 6 +++++ packages/coding-agent/src/task/types.ts | 4 ++++ 7 files changed, 44 insertions(+), 10 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 21e3ba775..e4826b71b 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -8,6 +8,8 @@ - Added per-line column cap shared across streaming tool outputs (`bash`, `ssh`, `python`, `js eval`) and the `read` tool. Lines wider than `tools.outputMaxColumns` bytes (default **768**) are ellipsis-truncated at write time and remaining bytes up to the next `\n` are dropped — bounded memory even on multi-MB single-line outputs (e.g. `cat /dev/urandom`). The cap lives on `OutputSink` as the new `maxColumns` option, persists state across chunk boundaries so split-mid-line writes still respect the budget, and exposes `columnDroppedBytes` / `columnTruncatedLines` on `OutputSummary`. Middle-elision byte math subtracts column drops so the "elided from middle" count stays honest. `read` reuses the same setting but trims its already-collected lines via `truncateLine`. Skipped when the read selector is `:raw`. The artifact file (`artifact://`) keeps the full uncapped stream. Set `tools.outputMaxColumns = 0` to disable. - Added Bun HTTP/2 fetch opt-in. Dev scripts (`bun run dev`, `bun run stats`) now pass `bun --experimental-http2-fetch` so every `fetch()` advertises `h2` in the TLS ALPN list and falls back to HTTP/1.1 when the server doesn't select it. Multiplexing collapses parallel requests to the same origin onto one TLS connection. For the installed `omp` binary, export `BUN_FEATURE_FLAG_EXPERIMENTAL_HTTP2_CLIENT=1` in your shell to enable the same behavior (the flag has to be set before Bun starts; `process.env` from inside JS is too late). Requires Bun **1.3.14**. +- Added per-subagent cost display (`$X.XX` in the task progress tree and the session-observer stats line). Cost is accumulated incrementally from `message_end` events and shown only when non-zero, using the `statusLineCost` theme color. Providers that do not report per-turn cost data (e.g. subscription/OAuth usage) continue to show nothing. + ### Changed - Raised the image downscaling default JPEG quality from 75 to 80 in `resizeImage` output generation @@ -19,6 +21,10 @@ - Changed search truncation metadata/renderer output from match/result-based limits to file-based limits (`fileLimitReached`, `perFileLimitReached`) and updated truncation labels accordingly - Lowered `read.defaultLimit` default from `500` to `300` lines, and split the per-range context padding into asymmetric `RANGE_LEADING_CONTEXT_LINES = 1` / `RANGE_TRAILING_CONTEXT_LINES = 3` (was symmetric `RANGE_CONTEXT_LINES = 3`). Replay analysis over post-summarizer sessions (`scripts/session-stats/optimize_read_config.py`) showed that bare-path reads are over-provisioned at the median (file p50 = 220 lines) and that most follow-up reads are disjoint hops rather than adjacent extensions — so a smaller default plus narrower leading context reclaims tokens without measurably changing first-cover rate. Trailing context stays at 3 lines to keep anchor-stale recovery on narrow reads. Explicit `read.defaultLimit` overrides in settings are honoured unchanged. +### Fixed + +- Fixed token display for sessions and subagents inflating far beyond the context window. `token_total` status-line segment and the subagent overlay token counter now show `input + output + cacheWrite` instead of `input + output + cacheRead + cacheWrite`. With prompt caching, `cacheRead` per turn equals the full cached context — summing it across all turns produces a cumulative total that is N×context_size (e.g. a 5-turn session with a 1 M-token context reported ~5 M tokens). Cache activity is still visible via the dedicated `cache_read`/`cache_write` status-line segments; billing cost is unaffected. + ## [15.0.0] - 2026-05-13 ### Breaking Changes diff --git a/packages/coding-agent/src/modes/components/session-observer-overlay.ts b/packages/coding-agent/src/modes/components/session-observer-overlay.ts index cf37498d2..689ab5b10 100644 --- a/packages/coding-agent/src/modes/components/session-observer-overlay.ts +++ b/packages/coding-agent/src/modes/components/session-observer-overlay.ts @@ -267,7 +267,10 @@ export class SessionObserverOverlayComponent extends Container { if (progress.toolCount > 0) stats.push(`${formatNumber(progress.toolCount)} tools`); if (progress.tokens > 0) stats.push(`${formatNumber(progress.tokens)} tokens`); if (progress.durationMs > 0) stats.push(formatDuration(progress.durationMs)); - return stats.length > 0 ? theme.fg("dim", stats.join(theme.sep.dot)) : ""; + const parts: string[] = []; + if (stats.length > 0) parts.push(theme.fg("dim", stats.join(theme.sep.dot))); + if (progress.cost > 0) parts.push(theme.fg("statusLineCost", `$${progress.cost.toFixed(2)}`)); + return parts.join(theme.sep.dot); } #buildTranscriptLines(messageEntries: SessionMessageEntry[], lines: string[]): void { diff --git a/packages/coding-agent/src/modes/components/status-line/segments.ts b/packages/coding-agent/src/modes/components/status-line/segments.ts index 5c5160f5b..078dd5eec 100644 --- a/packages/coding-agent/src/modes/components/status-line/segments.ts +++ b/packages/coding-agent/src/modes/components/status-line/segments.ts @@ -216,8 +216,11 @@ const tokenOutSegment: StatusLineSegment = { const tokenTotalSegment: StatusLineSegment = { id: "token_total", render(ctx) { - const { input, output, cacheRead, cacheWrite } = ctx.usageStats; - const total = input + output + cacheRead + cacheWrite; + // Excludes cacheRead: that field re-reads the full cached context every + // turn, making the cumulative sum N×context_size. The dedicated cache_read + // segment handles cache monitoring; the cost segment handles billing. + const { input, output, cacheWrite } = ctx.usageStats; + const total = input + output + cacheWrite; if (!total) return { content: "", visible: false }; const content = withIcon(theme.icon.tokens, formatNumber(total)); diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index ca0233bcd..c27aff586 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -379,21 +379,29 @@ function firstNumberField(record: Record, keys: string[]): numb } /** - * Normalize usage objects from different event formats. + * Tokens for progress display: input + output + cacheWrite per turn. + * + * Deliberately excludes cacheRead. With prompt caching, cacheRead in each turn + * equals the full cached context (potentially hundreds of KB), so summing it + * across all turns produces a cumulative total that is N×context_size — far + * larger than the context window and misleading as a "work done" metric. + * cacheWrite is kept because each byte is written once, not repeated per turn. + * The cost segment handles billing; dedicated cache_read/cache_write segments + * handle cache-specific monitoring. */ function getUsageTokens(usage: unknown): number { if (!usage || typeof usage !== "object") return 0; const record = usage as Record; - const totalTokens = firstNumberField(record, ["totalTokens", "total_tokens"]); - if (totalTokens !== undefined && totalTokens > 0) return totalTokens; - const input = firstNumberField(record, ["input", "input_tokens", "inputTokens"]) ?? 0; const output = firstNumberField(record, ["output", "output_tokens", "outputTokens"]) ?? 0; - const cacheRead = firstNumberField(record, ["cacheRead", "cache_read", "cacheReadTokens"]) ?? 0; const cacheWrite = firstNumberField(record, ["cacheWrite", "cache_write", "cacheWriteTokens"]) ?? 0; - - return input + output + cacheRead + cacheWrite; + const computed = input + output + cacheWrite; + if (computed > 0) return computed; + // Fallback for providers that only surface a pre-summed total without individual + // field breakdown. This total includes cacheRead, but returning it is still better + // than silently showing 0 for those providers. + return firstNumberField(record, ["totalTokens", "total_tokens"]) ?? 0; } /** @@ -497,6 +505,7 @@ export async function runSubprocess(options: ExecutorOptions): Promise { recentOutput: [], toolCount: 0, tokens: 0, + cost: 0, durationMs: 0, }); } @@ -831,6 +832,7 @@ export class TaskTool implements AgentTool { recentOutput: [], toolCount: 0, tokens: 0, + cost: 0, durationMs: 0, modelOverride, description: taskItem.description, diff --git a/packages/coding-agent/src/task/render.ts b/packages/coding-agent/src/task/render.ts index 6c946af04..331daea5c 100644 --- a/packages/coding-agent/src/task/render.ts +++ b/packages/coding-agent/src/task/render.ts @@ -532,6 +532,9 @@ function renderAgentProgress( if (progress.tokens > 0) { statusLine += `${theme.sep.dot}${theme.fg("dim", `${formatNumber(progress.tokens)} tokens`)}`; } + if (progress.cost > 0) { + statusLine += `${theme.sep.dot}${theme.fg("statusLineCost", `$${progress.cost.toFixed(2)}`)}`; + } } else if (progress.status === "completed") { if (progress.toolCount > 0) { statusLine += `${theme.sep.dot}${theme.fg("dim", `${progress.toolCount} tools`)}`; @@ -539,6 +542,9 @@ function renderAgentProgress( if (progress.tokens > 0) { statusLine += `${theme.sep.dot}${theme.fg("dim", `${formatNumber(progress.tokens)} tokens`)}`; } + if (progress.cost > 0) { + statusLine += `${theme.sep.dot}${theme.fg("statusLineCost", `$${progress.cost.toFixed(2)}`)}`; + } } lines.push(statusLine); diff --git a/packages/coding-agent/src/task/types.ts b/packages/coding-agent/src/task/types.ts index 21c58bc38..8deb22b0e 100644 --- a/packages/coding-agent/src/task/types.ts +++ b/packages/coding-agent/src/task/types.ts @@ -217,7 +217,10 @@ export interface AgentProgress { recentTools: Array<{ tool: string; args: string; endMs: number }>; recentOutput: string[]; toolCount: number; + /** Cumulative input + output + cacheWrite tokens across all turns. Excludes cacheRead (re-reads cached context every turn, making cumulative sum misleading). */ tokens: number; + /** Cumulative billing cost in USD, accumulated incrementally from message_end events. */ + cost: number; durationMs: number; modelOverride?: string | string[]; /** Data extracted by registered subprocess tool handlers (keyed by tool name) */ @@ -239,6 +242,7 @@ export interface SingleResult { stderr: string; truncated: boolean; durationMs: number; + /** Cumulative input + output + cacheWrite tokens across all turns. Excludes cacheRead (re-reads cached context every turn, making cumulative sum misleading). */ tokens: number; modelOverride?: string | string[]; error?: string;