diff --git a/packages/coding-agent/src/collab/host.ts b/packages/coding-agent/src/collab/host.ts
index ff830897a..7a30e4aec 100644
--- a/packages/coding-agent/src/collab/host.ts
+++ b/packages/coding-agent/src/collab/host.ts
@@ -411,10 +411,11 @@ export class CollabHost {
#buildState(): CollabSessionState {
const session = this.#ctx.session;
- // Context numbers come from the status line's breakdown — not
- // session.getContextUsage() — so guests render exactly what the host's
- // own footer shows.
+ // Context numbers come from the status line's memoized breakdown so guests
+ // render exactly the same anchored, provider-real count the host's own
+ // status line shows.
const breakdown = this.#ctx.statusLine.getCachedContextBreakdown();
+ const tokens = breakdown.usedTokens;
return {
isStreaming: session.isStreaming,
isAborting: session.isAborting,
@@ -424,9 +425,9 @@ export class CollabHost {
model: session.model,
thinkingLevel: session.thinkingLevel,
contextUsage: {
- tokens: breakdown.usedTokens,
+ tokens,
contextWindow: breakdown.contextWindow,
- percent: breakdown.contextWindow > 0 ? (breakdown.usedTokens / breakdown.contextWindow) * 100 : null,
+ percent: tokens !== null && breakdown.contextWindow > 0 ? (tokens / breakdown.contextWindow) * 100 : null,
},
participants: this.participants,
};
diff --git a/packages/coding-agent/src/modes/components/status-line/component.ts b/packages/coding-agent/src/modes/components/status-line/component.ts
index 6851cf7d4..ff8e07610 100644
--- a/packages/coding-agent/src/modes/components/status-line/component.ts
+++ b/packages/coding-agent/src/modes/components/status-line/component.ts
@@ -1,7 +1,6 @@
import * as fs from "node:fs";
import * as path from "node:path";
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
-import { estimateTokens } from "@oh-my-pi/pi-agent-core/compaction";
import { type Component, truncateToWidth, visibleWidth } from "@oh-my-pi/pi-tui";
import { getProjectDir } from "@oh-my-pi/pi-utils";
import { $ } from "bun";
@@ -11,7 +10,6 @@ import * as git from "../../../utils/git";
import { getSessionAccentAnsi, getSessionAccentHex } from "../../../utils/session-color";
import { sanitizeStatusText } from "../../shared";
import { theme } from "../../theme/theme";
-import { computeNonMessageTokens } from "../../utils/context-usage";
import { canReuseCachedPr, createPrCacheContext, isSamePrCacheContext, type PrCacheContext } from "./git-utils";
import { getPreset } from "./presets";
import { renderSegment, type SegmentContext } from "./segments";
@@ -26,30 +24,15 @@ import type {
} from "./types";
// ═══════════════════════════════════════════════════════════════════════════
-// Per-message token cache
+// Context-usage memo
// ═══════════════════════════════════════════════════════════════════════════
/**
- * Symbol-keyed sidecar tagged onto each `AgentMessage` to memoize its
- * `estimateTokens` result. Keyed by message identity (the object itself);
- * a cheap content fingerprint detects in-place mutations (post-hoc error
- * attachment, retry-truncated branch rebuild, etc.) and forces recompute.
- *
- * Cache lives on the message — multiple `StatusLineComponent` instances
- * share it for free, and entries collect with the message itself when the
- * conversation is replaced or compacted.
- */
-const kTokenCache = Symbol("statusLine.tokenCache");
-interface TaggedMessage {
- [kTokenCache]?: { fingerprint: string; tokens: number };
-}
-
-/**
- * Cheap structural fingerprint mirroring `estimateTokens`'s content walk.
- * O(blocks) — only reads string `.length` and primitives, never copies or
- * serializes content. Any in-place mutation that alters total tokenized
- * content also alters one of the byte-length sums or block counts captured
- * here, forcing the cached `estimateTokens` value to be recomputed.
+ * Cheap structural fingerprint of a message's tokenizable content. O(blocks) —
+ * only reads string `.length` and primitives, never copies or serializes.
+ * Detects in-place growth of the streaming tail (and other in-place mutations)
+ * so the cached `getContextUsage()` result is recomputed when — and only when —
+ * the numbers it depends on change.
*/
function messageFingerprint(msg: AgentMessage): string {
const role = (msg as { role?: string }).role ?? "";
@@ -107,28 +90,16 @@ function messageFingerprint(msg: AgentMessage): string {
return `${role}:${ts}:${textLen}:${blocks}:${images}`;
}
-/**
- * Token count for a single message, using the per-message sidecar cache.
- * The caller MUST skip caching for the last message during streaming —
- * it may still be growing and its tokens belong recomputed each refresh.
- */
-function tokensForMessage(msg: AgentMessage): number {
- const fp = messageFingerprint(msg);
- const tagged = msg as TaggedMessage;
- const cached = tagged[kTokenCache];
- if (cached && cached.fingerprint === fp) return cached.tokens;
- const tokens = estimateTokens(msg);
- tagged[kTokenCache] = { fingerprint: fp, tokens };
- return tokens;
+interface ContextUsageMemo {
+ messagesRef: readonly AgentMessage[];
+ length: number;
+ lastFingerprint: string | undefined;
+ modelContextWindow: number;
+ usedTokens: number | null;
+ contextWindow: number;
}
-interface MessageTokenTotalsCache {
- messagesRef: readonly AgentMessage[];
- stableCount: number;
- stableTokens: number;
- lastStableMessage: AgentMessage | undefined;
- lastStableFingerprint: string | undefined;
-}
+const EMPTY_MESSAGES: readonly AgentMessage[] = [];
function hasContextSegment(segments: readonly StatusLineSegmentId[]): boolean {
return segments.includes("context_pct") || segments.includes("context_total");
@@ -176,19 +147,12 @@ export class StatusLineComponent implements Component {
} | null = null;
#usageFetchedAt = 0;
#usageInFlight = false;
- // Context breakdown — incremental rolling cache. The status line refreshes
- // on every agent event, so the hot path must not re-tokenize the full
- // message list. Stable messages are accumulated once; normal streaming
- // refreshes only recompute the current tail message and newly appended
- // entries. History rewrites/compaction replace or shrink the message array
- // and rebuild this cache. Stable messages are treated as immutable after
- // promotion, matching the normal append-only session flow.
- // Cached non-message total (system prompt + tools + skills). Invalidated
- // when the inputs-identity fingerprint changes (model swap, skill toggle,
- // tool registration).
- #nonMessageTokensCache: number | undefined;
- #nonMessageInputsKey: string | undefined;
- #messageTokenTotalsCache: MessageTokenTotalsCache | undefined;
+ // Context-usage memo. The status line redraws on every agent event, so the
+ // hot path must not recompute context tokens unless an input changed.
+ // `getContextUsage()` anchors on the last assistant's real prompt-token
+ // count (matching the provider and the `/context` panel), so a stable
+ // message list + model window yields a stable result we can return verbatim.
+ #contextUsageCache: ContextUsageMemo | undefined;
constructor(private session: AgentSession) {
this.#settings = {
@@ -310,9 +274,7 @@ export class StatusLineComponent implements Component {
this.#cachedUsage = null;
this.#usageFetchedAt = 0;
this.#usageInFlight = false;
- this.#nonMessageTokensCache = undefined;
- this.#nonMessageInputsKey = undefined;
- this.#messageTokenTotalsCache = undefined;
+ this.#contextUsageCache = undefined;
this.#lastTokensPerSecond = null;
this.#lastTokensPerSecondTimestamp = null;
}
@@ -544,109 +506,44 @@ export class StatusLineComponent implements Component {
}
/**
- * Compute the (cached) used-tokens / context-window totals for the
- * status-line context% segment. Exposed (non-private) so unit tests can
- * verify the incremental-cache invariants; not part of any external
- * API.
+ * Used-tokens / context-window totals for the status-line context% segment,
+ * memoized so the per-event redraw stays O(1) when nothing changed.
+ *
+ * The numerator comes from `session.getContextUsage()`, which anchors on the
+ * last assistant's real prompt-token count — so the bar matches the provider
+ * and the `/context` panel — and reports `null` while that count is unknown
+ * (right after compaction, before the next response). Exposed (non-private)
+ * for unit tests and the collab host's state broadcast.
*/
- getCachedContextBreakdown(): { usedTokens: number; contextWindow: number } {
- const messages = this.session.messages ?? [];
- const contextWindow = this.session.model?.contextWindow ?? 0;
-
- // 1) Non-message tokens (system prompt + tools + skills). Refresh only
- // when the inputs identity fingerprint changes — usually never
- // during a streaming turn. ~10-30 ms when it does refresh.
- const inputsKey = this.#computeNonMessageInputsKey();
- if (this.#nonMessageTokensCache === undefined || this.#nonMessageInputsKey !== inputsKey) {
- this.#nonMessageTokensCache = computeNonMessageTokens(this.session);
- this.#nonMessageInputsKey = inputsKey;
- }
-
- // 2) Message tokens — incremental rolling total. The sidecar cache lives
- // on each stable message object (all but the current tail). Normal
- // streaming turns only recompute the last message and newly appended
- // entries. Full rebuild only when the message array is replaced,
- // shrinks, or the recently-promoted stable tail mutates in place.
- const messagesTokens = this.#getCachedMessageTokens(messages);
-
- const usedTokens = this.#nonMessageTokensCache + messagesTokens;
- return { usedTokens, contextWindow };
- }
-
- #getCachedMessageTokens(messages: readonly AgentMessage[]): number {
- const cache = this.#messageTokenTotalsCache;
- if (!cache || cache.messagesRef !== messages || messages.length <= cache.stableCount) {
- return this.#rebuildMessageTokenTotals(messages);
- }
-
- let stableTokens = cache.stableTokens;
- let stableCount = cache.stableCount;
- const stableLimit = Math.max(0, messages.length - 1);
+ getCachedContextBreakdown(): { usedTokens: number | null; contextWindow: number } {
+ const messages = this.session.messages ?? EMPTY_MESSAGES;
+ const modelContextWindow = this.session.model?.contextWindow ?? 0;
+ const length = messages.length;
+ const lastFingerprint = length > 0 ? messageFingerprint(messages[length - 1]!) : undefined;
+ const cache = this.#contextUsageCache;
if (
- cache.lastStableMessage &&
- stableCount > 0 &&
- messages[stableCount - 1] === cache.lastStableMessage &&
- cache.lastStableFingerprint !== undefined &&
- cache.lastStableFingerprint !== messageFingerprint(cache.lastStableMessage)
+ cache &&
+ cache.messagesRef === messages &&
+ cache.length === length &&
+ cache.lastFingerprint === lastFingerprint &&
+ cache.modelContextWindow === modelContextWindow
) {
- return this.#rebuildMessageTokenTotals(messages);
+ return { usedTokens: cache.usedTokens, contextWindow: cache.contextWindow };
}
- while (stableCount < stableLimit) {
- const promoted = messages[stableCount]!;
- stableTokens += tokensForMessage(promoted);
- stableCount++;
- }
-
- const lastStableMessage = stableCount > 0 ? messages[stableCount - 1] : undefined;
- const lastStableFingerprint = lastStableMessage ? messageFingerprint(lastStableMessage) : undefined;
- const lastMessage = messages.at(-1);
- const lastTokens = lastMessage ? estimateTokens(lastMessage) : 0;
- this.#messageTokenTotalsCache = {
+ const usage = this.session.getContextUsage();
+ const usedTokens = usage?.tokens ?? null;
+ const contextWindow = usage?.contextWindow ?? modelContextWindow;
+ this.#contextUsageCache = {
messagesRef: messages,
- stableCount,
- stableTokens,
- lastStableMessage,
- lastStableFingerprint,
+ length,
+ lastFingerprint,
+ modelContextWindow,
+ usedTokens,
+ contextWindow,
};
- return stableTokens + lastTokens;
- }
-
- #rebuildMessageTokenTotals(messages: readonly AgentMessage[]): number {
- let stableTokens = 0;
- const stableLimit = Math.max(0, messages.length - 1);
- for (let i = 0; i < stableLimit; i++) {
- stableTokens += tokensForMessage(messages[i]!);
- }
-
- const lastStableMessage = stableLimit > 0 ? messages[stableLimit - 1] : undefined;
- const lastStableFingerprint = lastStableMessage ? messageFingerprint(lastStableMessage) : undefined;
- const lastMessage = messages.at(-1);
- const lastTokens = lastMessage ? estimateTokens(lastMessage) : 0;
-
- this.#messageTokenTotalsCache = {
- messagesRef: messages,
- stableCount: stableLimit,
- stableTokens,
- lastStableMessage,
- lastStableFingerprint,
- };
- return stableTokens + lastTokens;
- }
-
- /**
- * Build an identity fingerprint for the non-message inputs (system prompt,
- * tools, skills). When this changes, the non-message token cache must be
- * recomputed. Cheap: just lengths + first-string-length. Doesn't need to
- * be cryptographically unique — only stable for the same inputs.
- */
- #computeNonMessageInputsKey(): string {
- const sp = this.session.systemPrompt ?? [];
- const tools = this.session.agent?.state?.tools ?? [];
- const skills = this.session.skills ?? [];
- const modelId = this.session.model?.id ?? "";
- return `${modelId}|${sp.length}:${sp[0]?.length ?? 0}|${tools.length}|${skills.length}`;
+ return { usedTokens, contextWindow };
}
#buildSegmentContext(
@@ -673,14 +570,14 @@ export class StatusLineComponent implements Component {
tokensPerSecond: this.#getTokensPerSecond(),
};
- let contextTokens = 0;
let contextWindow = state.model?.contextWindow ?? this.session.model?.contextWindow ?? 0;
+ let contextPercent: number | null = 0;
if (includeContext) {
const breakdown = this.getCachedContextBreakdown();
- contextTokens = breakdown.usedTokens;
contextWindow = breakdown.contextWindow || contextWindow;
+ contextPercent =
+ breakdown.usedTokens === null ? null : contextWindow > 0 ? (breakdown.usedTokens / contextWindow) * 100 : 0;
}
- let contextPercent = contextWindow > 0 ? (contextTokens / contextWindow) * 100 : 0;
// Collab guest: context comes from the host's state frames — the local
// replica does no accounting of its own.
diff --git a/packages/coding-agent/src/modes/components/status-line/segments.ts b/packages/coding-agent/src/modes/components/status-line/segments.ts
index 9dd0120cb..985a8af99 100644
--- a/packages/coding-agent/src/modes/components/status-line/segments.ts
+++ b/packages/coding-agent/src/modes/components/status-line/segments.ts
@@ -364,7 +364,7 @@ const contextPctSegment: StatusLineSegment = {
const autoIcon = ctx.autoCompactEnabled && theme.icon.auto ? ` ${theme.icon.auto}` : "";
const text = `${formatContextUsage(pct, window)}${autoIcon}`;
- const color = getContextUsageThemeColor(getContextUsageLevel(pct, window));
+ const color = getContextUsageThemeColor(getContextUsageLevel(pct ?? 0, window));
const content = withIcon(theme.icon.context, theme.fg(color, text));
return { content, visible: true };
diff --git a/packages/coding-agent/src/modes/components/status-line/types.ts b/packages/coding-agent/src/modes/components/status-line/types.ts
index 8587a62fe..7f05ebcb5 100644
--- a/packages/coding-agent/src/modes/components/status-line/types.ts
+++ b/packages/coding-agent/src/modes/components/status-line/types.ts
@@ -71,7 +71,8 @@ export interface SegmentContext {
cost: number;
tokensPerSecond: number | null;
};
- contextPercent: number;
+ /** Context usage percent, or null when unknown (e.g. right after compaction). */
+ contextPercent: number | null;
contextWindow: number;
autoCompactEnabled: boolean;
subagentCount: number;
diff --git a/packages/coding-agent/test/status-line-context-cache.test.ts b/packages/coding-agent/test/status-line-context-cache.test.ts
index e25ae8041..302026e21 100644
--- a/packages/coding-agent/test/status-line-context-cache.test.ts
+++ b/packages/coding-agent/test/status-line-context-cache.test.ts
@@ -1,26 +1,24 @@
/**
- * Regression guard for the incremental per-message token cache in
- * `StatusLineComponent.getCachedContextBreakdown`.
+ * Contract for `StatusLineComponent.getCachedContextBreakdown`.
*
- * Before the cache: every call walked `session.messages` and ran
- * `estimateTokens` per message (~0.5 ms each native). For a 2,300-message
- * session this was a ~1.1 s blocking call. `updateEditorTopBorder()` is
- * invoked on every agent event (event-controller.ts:163), so during
- * streaming the UI froze for ~1.1 s every 2 s (the prior cache TTL).
+ * The status-line context% segment no longer keeps its own cl100k estimate of
+ * the whole conversation. It surfaces `session.getContextUsage()`, which
+ * anchors on the last assistant's real provider prompt-token count — so the bar
+ * matches the provider and the `/context` panel instead of an independent
+ * estimate that drifted past 100%.
*
- * After the cache: messages are walked ONCE during warm-up; subsequent
- * refreshes append-only update the cache by `messages.length - cached`
- * (typically 0–1 new messages). The LAST message is recomputed every call
- * because its content may still be growing during streaming. Compaction
- * (messages.length shrinks) resets the cache.
+ * `getTopBorder()` runs on every agent event (event-controller.ts), so the
+ * breakdown is memoized: it re-queries `getContextUsage()` only when an input
+ * it depends on changes (a new/grown message, a replaced message array, or the
+ * model's context window). A stable conversation must not re-query on every
+ * redraw — that per-event recompute is what previously froze large sessions.
*/
import { afterAll, beforeAll, describe, expect, it } from "bun:test";
import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
+import type { ContextUsage } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/types";
import { StatusLineComponent } from "@oh-my-pi/pi-coding-agent/modes/components/status-line";
import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
-import { computeNonMessageTokens, estimateToolSchemaTokens } from "@oh-my-pi/pi-coding-agent/modes/utils/context-usage";
import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
-import { countTokens } from "@oh-my-pi/pi-natives";
beforeAll(async () => {
resetSettingsForTest();
@@ -32,22 +30,25 @@ afterAll(() => {
resetSettingsForTest();
});
-function makeSession(opts: {
- messages: unknown[];
- systemPrompt?: string[];
- tools?: { name: string; description: string; parameters?: unknown }[];
- skills?: { name: string; description: string }[];
- contextWindow?: number;
- modelId?: string;
-}): AgentSession {
- const messages = opts.messages;
- return {
- messages,
- systemPrompt: opts.systemPrompt ?? ["You are a helpful assistant."],
- agent: { state: { tools: opts.tools ?? [] } },
- skills: opts.skills ?? [],
- model: { id: opts.modelId ?? "test-model", contextWindow: opts.contextWindow ?? 200_000 },
- state: { messages, model: { contextWindow: opts.contextWindow ?? 200_000 } },
+interface Fake {
+ session: AgentSession;
+ /** Number of times `getContextUsage()` was queried. */
+ usageCalls: () => number;
+ /** Swap the value the next `getContextUsage()` query returns. */
+ setUsage: (usage: ContextUsage | undefined) => void;
+}
+
+function makeSession(opts: { messages: unknown[]; contextWindow?: number; usage?: ContextUsage | undefined }): Fake {
+ const contextWindow = opts.contextWindow ?? 200_000;
+ let usage: ContextUsage | undefined = "usage" in opts ? opts.usage : { tokens: 1234, contextWindow, percent: 0.6 };
+ let calls = 0;
+ const session = {
+ messages: opts.messages,
+ systemPrompt: ["You are a helpful assistant."],
+ agent: { state: { tools: [] } },
+ skills: [],
+ model: { id: "test-model", contextWindow },
+ state: { messages: opts.messages, model: { contextWindow } },
sessionManager: {
getUsageStatistics: () => ({
input: 0,
@@ -60,7 +61,18 @@ function makeSession(opts: {
getSessionName: () => "test",
},
getAsyncJobSnapshot: () => ({ running: [] }),
+ getContextUsage: () => {
+ calls++;
+ return usage;
+ },
} as unknown as AgentSession;
+ return {
+ session,
+ usageCalls: () => calls,
+ setUsage: next => {
+ usage = next;
+ },
+ };
}
function userMessage(text: string): unknown {
@@ -70,155 +82,99 @@ function assistantMessage(text: string): unknown {
return { role: "assistant", content: [{ type: "text", text }] };
}
-describe("StatusLineComponent incremental context breakdown cache", () => {
- it("first call computes from scratch, second call returns same value", () => {
- const session = makeSession({
- messages: Array.from({ length: 50 }, (_, i) => userMessage(`message ${i}`.repeat(10))),
- });
- const comp = new StatusLineComponent(session);
-
- const first = comp.getCachedContextBreakdown();
- const second = comp.getCachedContextBreakdown();
-
- expect(first.usedTokens).toBeGreaterThan(0);
- expect(second.usedTokens).toBe(first.usedTokens);
- expect(second.contextWindow).toBe(200_000);
- });
-
- it("appending a message increases the total by approximately the new message's tokens", () => {
- const session = makeSession({
- messages: [userMessage("hello world"), userMessage("another message here")],
- });
- const comp = new StatusLineComponent(session);
-
- const before = comp.getCachedContextBreakdown();
- (session.messages as unknown[]).push(assistantMessage("a third message reply with more text content"));
- const after = comp.getCachedContextBreakdown();
-
- expect(after.usedTokens).toBeGreaterThan(before.usedTokens);
- expect(after.contextWindow).toBe(before.contextWindow);
- });
-
- it("popping the streaming tail rebuilds instead of double-counting the new tail", () => {
- const messages = [
- userMessage("stable head"),
- userMessage("tail that becomes stable after append".repeat(20)),
- assistantMessage("streaming tail removed by retry".repeat(20)),
- ];
- const session = makeSession({ messages });
- const comp = new StatusLineComponent(session);
-
- const withStreamingTail = comp.getCachedContextBreakdown();
- messages.pop();
- const afterPop = comp.getCachedContextBreakdown();
- const freshAfterPop = new StatusLineComponent(session).getCachedContextBreakdown();
-
- expect(afterPop.usedTokens).toBeLessThan(withStreamingTail.usedTokens);
- expect(afterPop.usedTokens).toBe(freshAfterPop.usedTokens);
- });
-
- it("compaction (messages.length shrinks) resets the cache and recomputes correctly", () => {
- const session = makeSession({
- messages: Array.from({ length: 20 }, (_, i) => userMessage(`message ${i}`.repeat(10))),
- });
- const comp = new StatusLineComponent(session);
-
- const before = comp.getCachedContextBreakdown();
- expect(before.usedTokens).toBeGreaterThan(0);
-
- (session.messages as unknown[]).length = 0;
- (session.messages as unknown[]).push(userMessage("compacted summary"));
-
- const after = comp.getCachedContextBreakdown();
- expect(after.usedTokens).toBeLessThan(before.usedTokens);
- expect(after.usedTokens).toBeGreaterThan(0);
- });
-
- it("non-message inputs change → recomputes non-message portion", () => {
- const session = makeSession({
+describe("StatusLineComponent context breakdown", () => {
+ it("surfaces the provider-anchored tokens and context window from getContextUsage", () => {
+ const { session } = makeSession({
messages: [userMessage("hi")],
- systemPrompt: ["You are an assistant."],
- tools: [{ name: "bash", description: "Run shell commands", parameters: {} }],
- skills: [{ name: "code", description: "Write code" }],
+ usage: { tokens: 5000, contextWindow: 272_000, percent: 1.8 },
});
- const comp = new StatusLineComponent(session);
-
- const v1 = comp.getCachedContextBreakdown();
- const v2 = comp.getCachedContextBreakdown();
- expect(v2.usedTokens).toBe(v1.usedTokens);
-
- (session.agent as { state: { tools: unknown[] } }).state.tools.push({
- name: "edit",
- description: "Edit files",
- parameters: {},
- });
- const v3 = comp.getCachedContextBreakdown();
- expect(v3.usedTokens).toBeGreaterThan(v2.usedTokens);
+ const breakdown = new StatusLineComponent(session).getCachedContextBreakdown();
+ expect(breakdown.usedTokens).toBe(5000);
+ expect(breakdown.contextWindow).toBe(272_000);
});
- it("non-message token shortcut matches previous category sum semantics", () => {
- const session = makeSession({
- messages: [],
- systemPrompt: [
- "You are an assistant.\n\n\n- code: Write code\n- review: Review code\n",
- "Loaded context file",
- "Runtime note",
- ],
- tools: [
- {
- name: "bash",
- description: "Run shell commands",
- parameters: { type: "object", properties: { command: { type: "string" } } },
- },
- ],
- skills: [
- { name: "code", description: "Write code" },
- { name: "review", description: "Review code" },
- ],
+ it("memoizes: repeated redraws with no change do not re-query usage", () => {
+ const { session, usageCalls } = makeSession({ messages: [userMessage("hi")] });
+ const comp = new StatusLineComponent(session);
+
+ comp.getCachedContextBreakdown();
+ comp.getCachedContextBreakdown();
+ comp.getCachedContextBreakdown();
+
+ expect(usageCalls()).toBe(1);
+ });
+
+ it("re-queries and surfaces the new total when a message is appended", () => {
+ const fake = makeSession({
+ messages: [userMessage("hi")],
+ usage: { tokens: 100, contextWindow: 200_000, percent: 0.05 },
});
+ const comp = new StatusLineComponent(fake.session);
+ expect(comp.getCachedContextBreakdown().usedTokens).toBe(100);
- const skillsTokens = countTokens(["code", "Write code", "review", "Review code"]);
- const previousCategorySum =
- Math.max(0, countTokens(session.systemPrompt?.[0] ?? "") - skillsTokens) +
- countTokens((session.systemPrompt ?? []).slice(1)) +
- estimateToolSchemaTokens(session.agent?.state?.tools ?? []) +
- skillsTokens;
+ (fake.session.messages as unknown[]).push(assistantMessage("a reply that bumped the real prompt size"));
+ fake.setUsage({ tokens: 250, contextWindow: 200_000, percent: 0.125 });
- expect(new StatusLineComponent(session).getCachedContextBreakdown().usedTokens).toBe(previousCategorySum);
- expect(computeNonMessageTokens(session)).toBe(previousCategorySum);
+ expect(comp.getCachedContextBreakdown().usedTokens).toBe(250);
+ expect(fake.usageCalls()).toBe(2);
});
- it("zero messages: produces only non-message tokens, no crash", () => {
- const session = makeSession({ messages: [] });
- const comp = new StatusLineComponent(session);
- const result = comp.getCachedContextBreakdown();
- expect(result.usedTokens).toBeGreaterThanOrEqual(0);
- expect(result.contextWindow).toBe(200_000);
- });
-
- it("in-place mutation of a recently promoted stable message recomputes its tokens", () => {
- const head = userMessage("head");
- const promoted = userMessage("promoted");
- const tail = userMessage("tail");
- const session = makeSession({ messages: [head, promoted, tail] });
+ it("re-queries when the streaming tail grows in place", () => {
+ const tail = assistantMessage("partial") as { content: { type: string; text: string }[] };
+ const { session, usageCalls } = makeSession({ messages: [userMessage("hi"), tail] });
const comp = new StatusLineComponent(session);
- const before = comp.getCachedContextBreakdown();
- // Mutate the most recently promoted stable message in place — this is the
- // only stable slot we intentionally revalidate on the hot path.
- (promoted as { content: string }).content = "a much longer body that should tokenize to more".repeat(20);
- const after = comp.getCachedContextBreakdown();
+ comp.getCachedContextBreakdown();
+ tail.content[0]!.text = "partial response that kept streaming".repeat(8);
+ comp.getCachedContextBreakdown();
- expect(after.usedTokens).toBeGreaterThan(before.usedTokens);
+ expect(usageCalls()).toBe(2);
});
- it("getTopBorder skips context accounting when no context segments are rendered", () => {
- const session = makeSession({
- messages: Array.from({ length: 20 }, (_, i) => userMessage(`message ${i}`.repeat(10))),
+ it("re-queries when the message array is replaced (branch switch / rebuild)", () => {
+ const { session, usageCalls } = makeSession({
+ messages: [userMessage("a"), userMessage("b")],
});
const comp = new StatusLineComponent(session);
- const before = comp.getCachedContextBreakdown();
+ comp.getCachedContextBreakdown();
+ (session as { messages: unknown[] }).messages = [userMessage("c"), userMessage("d")];
+ comp.getCachedContextBreakdown();
+
+ expect(usageCalls()).toBe(2);
+ });
+
+ it("re-queries when the model context window changes", () => {
+ const { session, usageCalls } = makeSession({ messages: [userMessage("hi")], contextWindow: 200_000 });
+ const comp = new StatusLineComponent(session);
+ comp.getCachedContextBreakdown();
+
+ (session.model as { contextWindow: number }).contextWindow = 400_000;
+ comp.getCachedContextBreakdown();
+
+ expect(usageCalls()).toBe(2);
+ });
+
+ it("propagates an unknown (null) token count, e.g. right after compaction", () => {
+ const { session } = makeSession({
+ messages: [userMessage("compaction summary")],
+ usage: { tokens: null, contextWindow: 272_000, percent: null },
+ });
+ const breakdown = new StatusLineComponent(session).getCachedContextBreakdown();
+ expect(breakdown.usedTokens).toBeNull();
+ expect(breakdown.contextWindow).toBe(272_000);
+ });
+
+ it("falls back to the model window with null tokens when usage is unavailable", () => {
+ const { session } = makeSession({ messages: [userMessage("hi")], usage: undefined, contextWindow: 128_000 });
+ const breakdown = new StatusLineComponent(session).getCachedContextBreakdown();
+ expect(breakdown.usedTokens).toBeNull();
+ expect(breakdown.contextWindow).toBe(128_000);
+ });
+
+ it("does not query usage when no context segment is rendered", () => {
+ const { session, usageCalls } = makeSession({ messages: [userMessage("hi")] });
+ const comp = new StatusLineComponent(session);
comp.updateSettings({
preset: "custom",
leftSegments: ["pi"],
@@ -228,45 +184,6 @@ describe("StatusLineComponent incremental context breakdown cache", () => {
const border = comp.getTopBorder(80);
expect(border.content.length).toBeGreaterThan(0);
- expect(comp.getCachedContextBreakdown()).toEqual(before);
- });
-
- it("replaceMessages with same length but different shape recomputes tokens", () => {
- const session = makeSession({
- messages: [userMessage("short a"), userMessage("short b")],
- });
- const comp = new StatusLineComponent(session);
-
- const before = comp.getCachedContextBreakdown();
- // Same-length replace: distinct message objects with larger payloads.
- (session as { messages: unknown[] }).messages = [
- userMessage("a much longer payload".repeat(20)),
- userMessage("another longer payload".repeat(20)),
- ];
- const after = comp.getCachedContextBreakdown();
-
- expect(after.usedTokens).toBeGreaterThan(before.usedTokens);
- });
-
- it("usage fetch error backs off — a failed fetch does not retrigger within the TTL window", async () => {
- const session = makeSession({ messages: [userMessage("hi")] });
- let calls = 0;
- (session as { fetchUsageReports?: () => Promise }).fetchUsageReports = () => {
- calls++;
- return Promise.reject(new Error("network"));
- };
- const comp = new StatusLineComponent(session);
-
- // First refresh → kicks off fetch #1.
- comp.refreshUsageInBackground();
- // Let the rejected fetch settle so the .catch backoff stamp lands.
- await Bun.sleep(0);
- expect(calls).toBe(1);
-
- // Subsequent refreshes within the TTL window must not refetch.
- comp.refreshUsageInBackground();
- comp.refreshUsageInBackground();
- await Bun.sleep(0);
- expect(calls).toBe(1);
+ expect(usageCalls()).toBe(0);
});
});
diff --git a/packages/coding-agent/test/status-line-settings-cache.test.ts b/packages/coding-agent/test/status-line-settings-cache.test.ts
index c82ed7303..906a64fa2 100644
--- a/packages/coding-agent/test/status-line-settings-cache.test.ts
+++ b/packages/coding-agent/test/status-line-settings-cache.test.ts
@@ -57,6 +57,7 @@ function makeSession(sessionName = "Cache Session") {
cost: 0,
}),
},
+ getContextUsage: () => undefined,
} as unknown as ConstructorParameters[0];
}