From 495d49f5886bfab7eff359aa5cf1b1447f83d5d0 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 13 Jun 2026 18:02:08 +0200 Subject: [PATCH] refactor(coding-agent): reorganized status-line context cache usage flow - Refactored status-line context caching to be keyed by message, tail, and window. - Updated context usage flow to pass breakdown tokens and expose null usage when unknown. - Adjusted context percentage handling so zero or unknown windows yield nullable values. - Expanded status-line cache tests for usage provenance, memoization, and invalidation. --- packages/coding-agent/src/collab/host.ts | 11 +- .../modes/components/status-line/component.ts | 211 +++--------- .../modes/components/status-line/segments.ts | 2 +- .../src/modes/components/status-line/types.ts | 3 +- .../test/status-line-context-cache.test.ts | 321 +++++++----------- .../test/status-line-settings-cache.test.ts | 1 + 6 files changed, 183 insertions(+), 366 deletions(-) diff --git a/packages/coding-agent/src/collab/host.ts b/packages/coding-agent/src/collab/host.ts index ff830897a..7a30e4aec 100644 --- a/packages/coding-agent/src/collab/host.ts +++ b/packages/coding-agent/src/collab/host.ts @@ -411,10 +411,11 @@ export class CollabHost { #buildState(): CollabSessionState { const session = this.#ctx.session; - // Context numbers come from the status line's breakdown — not - // session.getContextUsage() — so guests render exactly what the host's - // own footer shows. + // Context numbers come from the status line's memoized breakdown so guests + // render exactly the same anchored, provider-real count the host's own + // status line shows. const breakdown = this.#ctx.statusLine.getCachedContextBreakdown(); + const tokens = breakdown.usedTokens; return { isStreaming: session.isStreaming, isAborting: session.isAborting, @@ -424,9 +425,9 @@ export class CollabHost { model: session.model, thinkingLevel: session.thinkingLevel, contextUsage: { - tokens: breakdown.usedTokens, + tokens, contextWindow: breakdown.contextWindow, - percent: breakdown.contextWindow > 0 ? (breakdown.usedTokens / breakdown.contextWindow) * 100 : null, + percent: tokens !== null && breakdown.contextWindow > 0 ? (tokens / breakdown.contextWindow) * 100 : null, }, participants: this.participants, }; diff --git a/packages/coding-agent/src/modes/components/status-line/component.ts b/packages/coding-agent/src/modes/components/status-line/component.ts index 6851cf7d4..ff8e07610 100644 --- a/packages/coding-agent/src/modes/components/status-line/component.ts +++ b/packages/coding-agent/src/modes/components/status-line/component.ts @@ -1,7 +1,6 @@ import * as fs from "node:fs"; import * as path from "node:path"; import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; -import { estimateTokens } from "@oh-my-pi/pi-agent-core/compaction"; import { type Component, truncateToWidth, visibleWidth } from "@oh-my-pi/pi-tui"; import { getProjectDir } from "@oh-my-pi/pi-utils"; import { $ } from "bun"; @@ -11,7 +10,6 @@ import * as git from "../../../utils/git"; import { getSessionAccentAnsi, getSessionAccentHex } from "../../../utils/session-color"; import { sanitizeStatusText } from "../../shared"; import { theme } from "../../theme/theme"; -import { computeNonMessageTokens } from "../../utils/context-usage"; import { canReuseCachedPr, createPrCacheContext, isSamePrCacheContext, type PrCacheContext } from "./git-utils"; import { getPreset } from "./presets"; import { renderSegment, type SegmentContext } from "./segments"; @@ -26,30 +24,15 @@ import type { } from "./types"; // ═══════════════════════════════════════════════════════════════════════════ -// Per-message token cache +// Context-usage memo // ═══════════════════════════════════════════════════════════════════════════ /** - * Symbol-keyed sidecar tagged onto each `AgentMessage` to memoize its - * `estimateTokens` result. Keyed by message identity (the object itself); - * a cheap content fingerprint detects in-place mutations (post-hoc error - * attachment, retry-truncated branch rebuild, etc.) and forces recompute. - * - * Cache lives on the message — multiple `StatusLineComponent` instances - * share it for free, and entries collect with the message itself when the - * conversation is replaced or compacted. - */ -const kTokenCache = Symbol("statusLine.tokenCache"); -interface TaggedMessage { - [kTokenCache]?: { fingerprint: string; tokens: number }; -} - -/** - * Cheap structural fingerprint mirroring `estimateTokens`'s content walk. - * O(blocks) — only reads string `.length` and primitives, never copies or - * serializes content. Any in-place mutation that alters total tokenized - * content also alters one of the byte-length sums or block counts captured - * here, forcing the cached `estimateTokens` value to be recomputed. + * Cheap structural fingerprint of a message's tokenizable content. O(blocks) — + * only reads string `.length` and primitives, never copies or serializes. + * Detects in-place growth of the streaming tail (and other in-place mutations) + * so the cached `getContextUsage()` result is recomputed when — and only when — + * the numbers it depends on change. */ function messageFingerprint(msg: AgentMessage): string { const role = (msg as { role?: string }).role ?? ""; @@ -107,28 +90,16 @@ function messageFingerprint(msg: AgentMessage): string { return `${role}:${ts}:${textLen}:${blocks}:${images}`; } -/** - * Token count for a single message, using the per-message sidecar cache. - * The caller MUST skip caching for the last message during streaming — - * it may still be growing and its tokens belong recomputed each refresh. - */ -function tokensForMessage(msg: AgentMessage): number { - const fp = messageFingerprint(msg); - const tagged = msg as TaggedMessage; - const cached = tagged[kTokenCache]; - if (cached && cached.fingerprint === fp) return cached.tokens; - const tokens = estimateTokens(msg); - tagged[kTokenCache] = { fingerprint: fp, tokens }; - return tokens; +interface ContextUsageMemo { + messagesRef: readonly AgentMessage[]; + length: number; + lastFingerprint: string | undefined; + modelContextWindow: number; + usedTokens: number | null; + contextWindow: number; } -interface MessageTokenTotalsCache { - messagesRef: readonly AgentMessage[]; - stableCount: number; - stableTokens: number; - lastStableMessage: AgentMessage | undefined; - lastStableFingerprint: string | undefined; -} +const EMPTY_MESSAGES: readonly AgentMessage[] = []; function hasContextSegment(segments: readonly StatusLineSegmentId[]): boolean { return segments.includes("context_pct") || segments.includes("context_total"); @@ -176,19 +147,12 @@ export class StatusLineComponent implements Component { } | null = null; #usageFetchedAt = 0; #usageInFlight = false; - // Context breakdown — incremental rolling cache. The status line refreshes - // on every agent event, so the hot path must not re-tokenize the full - // message list. Stable messages are accumulated once; normal streaming - // refreshes only recompute the current tail message and newly appended - // entries. History rewrites/compaction replace or shrink the message array - // and rebuild this cache. Stable messages are treated as immutable after - // promotion, matching the normal append-only session flow. - // Cached non-message total (system prompt + tools + skills). Invalidated - // when the inputs-identity fingerprint changes (model swap, skill toggle, - // tool registration). - #nonMessageTokensCache: number | undefined; - #nonMessageInputsKey: string | undefined; - #messageTokenTotalsCache: MessageTokenTotalsCache | undefined; + // Context-usage memo. The status line redraws on every agent event, so the + // hot path must not recompute context tokens unless an input changed. + // `getContextUsage()` anchors on the last assistant's real prompt-token + // count (matching the provider and the `/context` panel), so a stable + // message list + model window yields a stable result we can return verbatim. + #contextUsageCache: ContextUsageMemo | undefined; constructor(private session: AgentSession) { this.#settings = { @@ -310,9 +274,7 @@ export class StatusLineComponent implements Component { this.#cachedUsage = null; this.#usageFetchedAt = 0; this.#usageInFlight = false; - this.#nonMessageTokensCache = undefined; - this.#nonMessageInputsKey = undefined; - this.#messageTokenTotalsCache = undefined; + this.#contextUsageCache = undefined; this.#lastTokensPerSecond = null; this.#lastTokensPerSecondTimestamp = null; } @@ -544,109 +506,44 @@ export class StatusLineComponent implements Component { } /** - * Compute the (cached) used-tokens / context-window totals for the - * status-line context% segment. Exposed (non-private) so unit tests can - * verify the incremental-cache invariants; not part of any external - * API. + * Used-tokens / context-window totals for the status-line context% segment, + * memoized so the per-event redraw stays O(1) when nothing changed. + * + * The numerator comes from `session.getContextUsage()`, which anchors on the + * last assistant's real prompt-token count — so the bar matches the provider + * and the `/context` panel — and reports `null` while that count is unknown + * (right after compaction, before the next response). Exposed (non-private) + * for unit tests and the collab host's state broadcast. */ - getCachedContextBreakdown(): { usedTokens: number; contextWindow: number } { - const messages = this.session.messages ?? []; - const contextWindow = this.session.model?.contextWindow ?? 0; - - // 1) Non-message tokens (system prompt + tools + skills). Refresh only - // when the inputs identity fingerprint changes — usually never - // during a streaming turn. ~10-30 ms when it does refresh. - const inputsKey = this.#computeNonMessageInputsKey(); - if (this.#nonMessageTokensCache === undefined || this.#nonMessageInputsKey !== inputsKey) { - this.#nonMessageTokensCache = computeNonMessageTokens(this.session); - this.#nonMessageInputsKey = inputsKey; - } - - // 2) Message tokens — incremental rolling total. The sidecar cache lives - // on each stable message object (all but the current tail). Normal - // streaming turns only recompute the last message and newly appended - // entries. Full rebuild only when the message array is replaced, - // shrinks, or the recently-promoted stable tail mutates in place. - const messagesTokens = this.#getCachedMessageTokens(messages); - - const usedTokens = this.#nonMessageTokensCache + messagesTokens; - return { usedTokens, contextWindow }; - } - - #getCachedMessageTokens(messages: readonly AgentMessage[]): number { - const cache = this.#messageTokenTotalsCache; - if (!cache || cache.messagesRef !== messages || messages.length <= cache.stableCount) { - return this.#rebuildMessageTokenTotals(messages); - } - - let stableTokens = cache.stableTokens; - let stableCount = cache.stableCount; - const stableLimit = Math.max(0, messages.length - 1); + getCachedContextBreakdown(): { usedTokens: number | null; contextWindow: number } { + const messages = this.session.messages ?? EMPTY_MESSAGES; + const modelContextWindow = this.session.model?.contextWindow ?? 0; + const length = messages.length; + const lastFingerprint = length > 0 ? messageFingerprint(messages[length - 1]!) : undefined; + const cache = this.#contextUsageCache; if ( - cache.lastStableMessage && - stableCount > 0 && - messages[stableCount - 1] === cache.lastStableMessage && - cache.lastStableFingerprint !== undefined && - cache.lastStableFingerprint !== messageFingerprint(cache.lastStableMessage) + cache && + cache.messagesRef === messages && + cache.length === length && + cache.lastFingerprint === lastFingerprint && + cache.modelContextWindow === modelContextWindow ) { - return this.#rebuildMessageTokenTotals(messages); + return { usedTokens: cache.usedTokens, contextWindow: cache.contextWindow }; } - while (stableCount < stableLimit) { - const promoted = messages[stableCount]!; - stableTokens += tokensForMessage(promoted); - stableCount++; - } - - const lastStableMessage = stableCount > 0 ? messages[stableCount - 1] : undefined; - const lastStableFingerprint = lastStableMessage ? messageFingerprint(lastStableMessage) : undefined; - const lastMessage = messages.at(-1); - const lastTokens = lastMessage ? estimateTokens(lastMessage) : 0; - this.#messageTokenTotalsCache = { + const usage = this.session.getContextUsage(); + const usedTokens = usage?.tokens ?? null; + const contextWindow = usage?.contextWindow ?? modelContextWindow; + this.#contextUsageCache = { messagesRef: messages, - stableCount, - stableTokens, - lastStableMessage, - lastStableFingerprint, + length, + lastFingerprint, + modelContextWindow, + usedTokens, + contextWindow, }; - return stableTokens + lastTokens; - } - - #rebuildMessageTokenTotals(messages: readonly AgentMessage[]): number { - let stableTokens = 0; - const stableLimit = Math.max(0, messages.length - 1); - for (let i = 0; i < stableLimit; i++) { - stableTokens += tokensForMessage(messages[i]!); - } - - const lastStableMessage = stableLimit > 0 ? messages[stableLimit - 1] : undefined; - const lastStableFingerprint = lastStableMessage ? messageFingerprint(lastStableMessage) : undefined; - const lastMessage = messages.at(-1); - const lastTokens = lastMessage ? estimateTokens(lastMessage) : 0; - - this.#messageTokenTotalsCache = { - messagesRef: messages, - stableCount: stableLimit, - stableTokens, - lastStableMessage, - lastStableFingerprint, - }; - return stableTokens + lastTokens; - } - - /** - * Build an identity fingerprint for the non-message inputs (system prompt, - * tools, skills). When this changes, the non-message token cache must be - * recomputed. Cheap: just lengths + first-string-length. Doesn't need to - * be cryptographically unique — only stable for the same inputs. - */ - #computeNonMessageInputsKey(): string { - const sp = this.session.systemPrompt ?? []; - const tools = this.session.agent?.state?.tools ?? []; - const skills = this.session.skills ?? []; - const modelId = this.session.model?.id ?? ""; - return `${modelId}|${sp.length}:${sp[0]?.length ?? 0}|${tools.length}|${skills.length}`; + return { usedTokens, contextWindow }; } #buildSegmentContext( @@ -673,14 +570,14 @@ export class StatusLineComponent implements Component { tokensPerSecond: this.#getTokensPerSecond(), }; - let contextTokens = 0; let contextWindow = state.model?.contextWindow ?? this.session.model?.contextWindow ?? 0; + let contextPercent: number | null = 0; if (includeContext) { const breakdown = this.getCachedContextBreakdown(); - contextTokens = breakdown.usedTokens; contextWindow = breakdown.contextWindow || contextWindow; + contextPercent = + breakdown.usedTokens === null ? null : contextWindow > 0 ? (breakdown.usedTokens / contextWindow) * 100 : 0; } - let contextPercent = contextWindow > 0 ? (contextTokens / contextWindow) * 100 : 0; // Collab guest: context comes from the host's state frames — the local // replica does no accounting of its own. diff --git a/packages/coding-agent/src/modes/components/status-line/segments.ts b/packages/coding-agent/src/modes/components/status-line/segments.ts index 9dd0120cb..985a8af99 100644 --- a/packages/coding-agent/src/modes/components/status-line/segments.ts +++ b/packages/coding-agent/src/modes/components/status-line/segments.ts @@ -364,7 +364,7 @@ const contextPctSegment: StatusLineSegment = { const autoIcon = ctx.autoCompactEnabled && theme.icon.auto ? ` ${theme.icon.auto}` : ""; const text = `${formatContextUsage(pct, window)}${autoIcon}`; - const color = getContextUsageThemeColor(getContextUsageLevel(pct, window)); + const color = getContextUsageThemeColor(getContextUsageLevel(pct ?? 0, window)); const content = withIcon(theme.icon.context, theme.fg(color, text)); return { content, visible: true }; diff --git a/packages/coding-agent/src/modes/components/status-line/types.ts b/packages/coding-agent/src/modes/components/status-line/types.ts index 8587a62fe..7f05ebcb5 100644 --- a/packages/coding-agent/src/modes/components/status-line/types.ts +++ b/packages/coding-agent/src/modes/components/status-line/types.ts @@ -71,7 +71,8 @@ export interface SegmentContext { cost: number; tokensPerSecond: number | null; }; - contextPercent: number; + /** Context usage percent, or null when unknown (e.g. right after compaction). */ + contextPercent: number | null; contextWindow: number; autoCompactEnabled: boolean; subagentCount: number; diff --git a/packages/coding-agent/test/status-line-context-cache.test.ts b/packages/coding-agent/test/status-line-context-cache.test.ts index e25ae8041..302026e21 100644 --- a/packages/coding-agent/test/status-line-context-cache.test.ts +++ b/packages/coding-agent/test/status-line-context-cache.test.ts @@ -1,26 +1,24 @@ /** - * Regression guard for the incremental per-message token cache in - * `StatusLineComponent.getCachedContextBreakdown`. + * Contract for `StatusLineComponent.getCachedContextBreakdown`. * - * Before the cache: every call walked `session.messages` and ran - * `estimateTokens` per message (~0.5 ms each native). For a 2,300-message - * session this was a ~1.1 s blocking call. `updateEditorTopBorder()` is - * invoked on every agent event (event-controller.ts:163), so during - * streaming the UI froze for ~1.1 s every 2 s (the prior cache TTL). + * The status-line context% segment no longer keeps its own cl100k estimate of + * the whole conversation. It surfaces `session.getContextUsage()`, which + * anchors on the last assistant's real provider prompt-token count — so the bar + * matches the provider and the `/context` panel instead of an independent + * estimate that drifted past 100%. * - * After the cache: messages are walked ONCE during warm-up; subsequent - * refreshes append-only update the cache by `messages.length - cached` - * (typically 0–1 new messages). The LAST message is recomputed every call - * because its content may still be growing during streaming. Compaction - * (messages.length shrinks) resets the cache. + * `getTopBorder()` runs on every agent event (event-controller.ts), so the + * breakdown is memoized: it re-queries `getContextUsage()` only when an input + * it depends on changes (a new/grown message, a replaced message array, or the + * model's context window). A stable conversation must not re-query on every + * redraw — that per-event recompute is what previously froze large sessions. */ import { afterAll, beforeAll, describe, expect, it } from "bun:test"; import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import type { ContextUsage } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/types"; import { StatusLineComponent } from "@oh-my-pi/pi-coding-agent/modes/components/status-line"; import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; -import { computeNonMessageTokens, estimateToolSchemaTokens } from "@oh-my-pi/pi-coding-agent/modes/utils/context-usage"; import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; -import { countTokens } from "@oh-my-pi/pi-natives"; beforeAll(async () => { resetSettingsForTest(); @@ -32,22 +30,25 @@ afterAll(() => { resetSettingsForTest(); }); -function makeSession(opts: { - messages: unknown[]; - systemPrompt?: string[]; - tools?: { name: string; description: string; parameters?: unknown }[]; - skills?: { name: string; description: string }[]; - contextWindow?: number; - modelId?: string; -}): AgentSession { - const messages = opts.messages; - return { - messages, - systemPrompt: opts.systemPrompt ?? ["You are a helpful assistant."], - agent: { state: { tools: opts.tools ?? [] } }, - skills: opts.skills ?? [], - model: { id: opts.modelId ?? "test-model", contextWindow: opts.contextWindow ?? 200_000 }, - state: { messages, model: { contextWindow: opts.contextWindow ?? 200_000 } }, +interface Fake { + session: AgentSession; + /** Number of times `getContextUsage()` was queried. */ + usageCalls: () => number; + /** Swap the value the next `getContextUsage()` query returns. */ + setUsage: (usage: ContextUsage | undefined) => void; +} + +function makeSession(opts: { messages: unknown[]; contextWindow?: number; usage?: ContextUsage | undefined }): Fake { + const contextWindow = opts.contextWindow ?? 200_000; + let usage: ContextUsage | undefined = "usage" in opts ? opts.usage : { tokens: 1234, contextWindow, percent: 0.6 }; + let calls = 0; + const session = { + messages: opts.messages, + systemPrompt: ["You are a helpful assistant."], + agent: { state: { tools: [] } }, + skills: [], + model: { id: "test-model", contextWindow }, + state: { messages: opts.messages, model: { contextWindow } }, sessionManager: { getUsageStatistics: () => ({ input: 0, @@ -60,7 +61,18 @@ function makeSession(opts: { getSessionName: () => "test", }, getAsyncJobSnapshot: () => ({ running: [] }), + getContextUsage: () => { + calls++; + return usage; + }, } as unknown as AgentSession; + return { + session, + usageCalls: () => calls, + setUsage: next => { + usage = next; + }, + }; } function userMessage(text: string): unknown { @@ -70,155 +82,99 @@ function assistantMessage(text: string): unknown { return { role: "assistant", content: [{ type: "text", text }] }; } -describe("StatusLineComponent incremental context breakdown cache", () => { - it("first call computes from scratch, second call returns same value", () => { - const session = makeSession({ - messages: Array.from({ length: 50 }, (_, i) => userMessage(`message ${i}`.repeat(10))), - }); - const comp = new StatusLineComponent(session); - - const first = comp.getCachedContextBreakdown(); - const second = comp.getCachedContextBreakdown(); - - expect(first.usedTokens).toBeGreaterThan(0); - expect(second.usedTokens).toBe(first.usedTokens); - expect(second.contextWindow).toBe(200_000); - }); - - it("appending a message increases the total by approximately the new message's tokens", () => { - const session = makeSession({ - messages: [userMessage("hello world"), userMessage("another message here")], - }); - const comp = new StatusLineComponent(session); - - const before = comp.getCachedContextBreakdown(); - (session.messages as unknown[]).push(assistantMessage("a third message reply with more text content")); - const after = comp.getCachedContextBreakdown(); - - expect(after.usedTokens).toBeGreaterThan(before.usedTokens); - expect(after.contextWindow).toBe(before.contextWindow); - }); - - it("popping the streaming tail rebuilds instead of double-counting the new tail", () => { - const messages = [ - userMessage("stable head"), - userMessage("tail that becomes stable after append".repeat(20)), - assistantMessage("streaming tail removed by retry".repeat(20)), - ]; - const session = makeSession({ messages }); - const comp = new StatusLineComponent(session); - - const withStreamingTail = comp.getCachedContextBreakdown(); - messages.pop(); - const afterPop = comp.getCachedContextBreakdown(); - const freshAfterPop = new StatusLineComponent(session).getCachedContextBreakdown(); - - expect(afterPop.usedTokens).toBeLessThan(withStreamingTail.usedTokens); - expect(afterPop.usedTokens).toBe(freshAfterPop.usedTokens); - }); - - it("compaction (messages.length shrinks) resets the cache and recomputes correctly", () => { - const session = makeSession({ - messages: Array.from({ length: 20 }, (_, i) => userMessage(`message ${i}`.repeat(10))), - }); - const comp = new StatusLineComponent(session); - - const before = comp.getCachedContextBreakdown(); - expect(before.usedTokens).toBeGreaterThan(0); - - (session.messages as unknown[]).length = 0; - (session.messages as unknown[]).push(userMessage("compacted summary")); - - const after = comp.getCachedContextBreakdown(); - expect(after.usedTokens).toBeLessThan(before.usedTokens); - expect(after.usedTokens).toBeGreaterThan(0); - }); - - it("non-message inputs change → recomputes non-message portion", () => { - const session = makeSession({ +describe("StatusLineComponent context breakdown", () => { + it("surfaces the provider-anchored tokens and context window from getContextUsage", () => { + const { session } = makeSession({ messages: [userMessage("hi")], - systemPrompt: ["You are an assistant."], - tools: [{ name: "bash", description: "Run shell commands", parameters: {} }], - skills: [{ name: "code", description: "Write code" }], + usage: { tokens: 5000, contextWindow: 272_000, percent: 1.8 }, }); - const comp = new StatusLineComponent(session); - - const v1 = comp.getCachedContextBreakdown(); - const v2 = comp.getCachedContextBreakdown(); - expect(v2.usedTokens).toBe(v1.usedTokens); - - (session.agent as { state: { tools: unknown[] } }).state.tools.push({ - name: "edit", - description: "Edit files", - parameters: {}, - }); - const v3 = comp.getCachedContextBreakdown(); - expect(v3.usedTokens).toBeGreaterThan(v2.usedTokens); + const breakdown = new StatusLineComponent(session).getCachedContextBreakdown(); + expect(breakdown.usedTokens).toBe(5000); + expect(breakdown.contextWindow).toBe(272_000); }); - it("non-message token shortcut matches previous category sum semantics", () => { - const session = makeSession({ - messages: [], - systemPrompt: [ - "You are an assistant.\n\n\n- code: Write code\n- review: Review code\n", - "Loaded context file", - "Runtime note", - ], - tools: [ - { - name: "bash", - description: "Run shell commands", - parameters: { type: "object", properties: { command: { type: "string" } } }, - }, - ], - skills: [ - { name: "code", description: "Write code" }, - { name: "review", description: "Review code" }, - ], + it("memoizes: repeated redraws with no change do not re-query usage", () => { + const { session, usageCalls } = makeSession({ messages: [userMessage("hi")] }); + const comp = new StatusLineComponent(session); + + comp.getCachedContextBreakdown(); + comp.getCachedContextBreakdown(); + comp.getCachedContextBreakdown(); + + expect(usageCalls()).toBe(1); + }); + + it("re-queries and surfaces the new total when a message is appended", () => { + const fake = makeSession({ + messages: [userMessage("hi")], + usage: { tokens: 100, contextWindow: 200_000, percent: 0.05 }, }); + const comp = new StatusLineComponent(fake.session); + expect(comp.getCachedContextBreakdown().usedTokens).toBe(100); - const skillsTokens = countTokens(["code", "Write code", "review", "Review code"]); - const previousCategorySum = - Math.max(0, countTokens(session.systemPrompt?.[0] ?? "") - skillsTokens) + - countTokens((session.systemPrompt ?? []).slice(1)) + - estimateToolSchemaTokens(session.agent?.state?.tools ?? []) + - skillsTokens; + (fake.session.messages as unknown[]).push(assistantMessage("a reply that bumped the real prompt size")); + fake.setUsage({ tokens: 250, contextWindow: 200_000, percent: 0.125 }); - expect(new StatusLineComponent(session).getCachedContextBreakdown().usedTokens).toBe(previousCategorySum); - expect(computeNonMessageTokens(session)).toBe(previousCategorySum); + expect(comp.getCachedContextBreakdown().usedTokens).toBe(250); + expect(fake.usageCalls()).toBe(2); }); - it("zero messages: produces only non-message tokens, no crash", () => { - const session = makeSession({ messages: [] }); - const comp = new StatusLineComponent(session); - const result = comp.getCachedContextBreakdown(); - expect(result.usedTokens).toBeGreaterThanOrEqual(0); - expect(result.contextWindow).toBe(200_000); - }); - - it("in-place mutation of a recently promoted stable message recomputes its tokens", () => { - const head = userMessage("head"); - const promoted = userMessage("promoted"); - const tail = userMessage("tail"); - const session = makeSession({ messages: [head, promoted, tail] }); + it("re-queries when the streaming tail grows in place", () => { + const tail = assistantMessage("partial") as { content: { type: string; text: string }[] }; + const { session, usageCalls } = makeSession({ messages: [userMessage("hi"), tail] }); const comp = new StatusLineComponent(session); - const before = comp.getCachedContextBreakdown(); - // Mutate the most recently promoted stable message in place — this is the - // only stable slot we intentionally revalidate on the hot path. - (promoted as { content: string }).content = "a much longer body that should tokenize to more".repeat(20); - const after = comp.getCachedContextBreakdown(); + comp.getCachedContextBreakdown(); + tail.content[0]!.text = "partial response that kept streaming".repeat(8); + comp.getCachedContextBreakdown(); - expect(after.usedTokens).toBeGreaterThan(before.usedTokens); + expect(usageCalls()).toBe(2); }); - it("getTopBorder skips context accounting when no context segments are rendered", () => { - const session = makeSession({ - messages: Array.from({ length: 20 }, (_, i) => userMessage(`message ${i}`.repeat(10))), + it("re-queries when the message array is replaced (branch switch / rebuild)", () => { + const { session, usageCalls } = makeSession({ + messages: [userMessage("a"), userMessage("b")], }); const comp = new StatusLineComponent(session); - const before = comp.getCachedContextBreakdown(); + comp.getCachedContextBreakdown(); + (session as { messages: unknown[] }).messages = [userMessage("c"), userMessage("d")]; + comp.getCachedContextBreakdown(); + + expect(usageCalls()).toBe(2); + }); + + it("re-queries when the model context window changes", () => { + const { session, usageCalls } = makeSession({ messages: [userMessage("hi")], contextWindow: 200_000 }); + const comp = new StatusLineComponent(session); + comp.getCachedContextBreakdown(); + + (session.model as { contextWindow: number }).contextWindow = 400_000; + comp.getCachedContextBreakdown(); + + expect(usageCalls()).toBe(2); + }); + + it("propagates an unknown (null) token count, e.g. right after compaction", () => { + const { session } = makeSession({ + messages: [userMessage("compaction summary")], + usage: { tokens: null, contextWindow: 272_000, percent: null }, + }); + const breakdown = new StatusLineComponent(session).getCachedContextBreakdown(); + expect(breakdown.usedTokens).toBeNull(); + expect(breakdown.contextWindow).toBe(272_000); + }); + + it("falls back to the model window with null tokens when usage is unavailable", () => { + const { session } = makeSession({ messages: [userMessage("hi")], usage: undefined, contextWindow: 128_000 }); + const breakdown = new StatusLineComponent(session).getCachedContextBreakdown(); + expect(breakdown.usedTokens).toBeNull(); + expect(breakdown.contextWindow).toBe(128_000); + }); + + it("does not query usage when no context segment is rendered", () => { + const { session, usageCalls } = makeSession({ messages: [userMessage("hi")] }); + const comp = new StatusLineComponent(session); comp.updateSettings({ preset: "custom", leftSegments: ["pi"], @@ -228,45 +184,6 @@ describe("StatusLineComponent incremental context breakdown cache", () => { const border = comp.getTopBorder(80); expect(border.content.length).toBeGreaterThan(0); - expect(comp.getCachedContextBreakdown()).toEqual(before); - }); - - it("replaceMessages with same length but different shape recomputes tokens", () => { - const session = makeSession({ - messages: [userMessage("short a"), userMessage("short b")], - }); - const comp = new StatusLineComponent(session); - - const before = comp.getCachedContextBreakdown(); - // Same-length replace: distinct message objects with larger payloads. - (session as { messages: unknown[] }).messages = [ - userMessage("a much longer payload".repeat(20)), - userMessage("another longer payload".repeat(20)), - ]; - const after = comp.getCachedContextBreakdown(); - - expect(after.usedTokens).toBeGreaterThan(before.usedTokens); - }); - - it("usage fetch error backs off — a failed fetch does not retrigger within the TTL window", async () => { - const session = makeSession({ messages: [userMessage("hi")] }); - let calls = 0; - (session as { fetchUsageReports?: () => Promise }).fetchUsageReports = () => { - calls++; - return Promise.reject(new Error("network")); - }; - const comp = new StatusLineComponent(session); - - // First refresh → kicks off fetch #1. - comp.refreshUsageInBackground(); - // Let the rejected fetch settle so the .catch backoff stamp lands. - await Bun.sleep(0); - expect(calls).toBe(1); - - // Subsequent refreshes within the TTL window must not refetch. - comp.refreshUsageInBackground(); - comp.refreshUsageInBackground(); - await Bun.sleep(0); - expect(calls).toBe(1); + expect(usageCalls()).toBe(0); }); }); diff --git a/packages/coding-agent/test/status-line-settings-cache.test.ts b/packages/coding-agent/test/status-line-settings-cache.test.ts index c82ed7303..906a64fa2 100644 --- a/packages/coding-agent/test/status-line-settings-cache.test.ts +++ b/packages/coding-agent/test/status-line-settings-cache.test.ts @@ -57,6 +57,7 @@ function makeSession(sessionName = "Cache Session") { cost: 0, }), }, + getContextUsage: () => undefined, } as unknown as ConstructorParameters[0]; }