refactor(coding-agent): reorganized status-line context cache usage flow

- Refactored status-line context caching to be keyed by message, tail, and window.
- Updated context usage flow to pass breakdown tokens and expose null usage when unknown.
- Adjusted context percentage handling so zero or unknown windows yield nullable values.
- Expanded status-line cache tests for usage provenance, memoization, and invalidation.
This commit is contained in:
can1357
2026-06-13 18:02:08 +02:00
parent b0ff8d5f83
commit 495d49f588
6 changed files with 183 additions and 366 deletions
+6 -5
View File
@@ -411,10 +411,11 @@ export class CollabHost {
#buildState(): CollabSessionState {
const session = this.#ctx.session;
// Context numbers come from the status line's breakdown — not
// session.getContextUsage() — so guests render exactly what the host's
// own footer shows.
// Context numbers come from the status line's memoized breakdown so guests
// render exactly the same anchored, provider-real count the host's own
// status line shows.
const breakdown = this.#ctx.statusLine.getCachedContextBreakdown();
const tokens = breakdown.usedTokens;
return {
isStreaming: session.isStreaming,
isAborting: session.isAborting,
@@ -424,9 +425,9 @@ export class CollabHost {
model: session.model,
thinkingLevel: session.thinkingLevel,
contextUsage: {
tokens: breakdown.usedTokens,
tokens,
contextWindow: breakdown.contextWindow,
percent: breakdown.contextWindow > 0 ? (breakdown.usedTokens / breakdown.contextWindow) * 100 : null,
percent: tokens !== null && breakdown.contextWindow > 0 ? (tokens / breakdown.contextWindow) * 100 : null,
},
participants: this.participants,
};
@@ -1,7 +1,6 @@
import * as fs from "node:fs";
import * as path from "node:path";
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
import { estimateTokens } from "@oh-my-pi/pi-agent-core/compaction";
import { type Component, truncateToWidth, visibleWidth } from "@oh-my-pi/pi-tui";
import { getProjectDir } from "@oh-my-pi/pi-utils";
import { $ } from "bun";
@@ -11,7 +10,6 @@ import * as git from "../../../utils/git";
import { getSessionAccentAnsi, getSessionAccentHex } from "../../../utils/session-color";
import { sanitizeStatusText } from "../../shared";
import { theme } from "../../theme/theme";
import { computeNonMessageTokens } from "../../utils/context-usage";
import { canReuseCachedPr, createPrCacheContext, isSamePrCacheContext, type PrCacheContext } from "./git-utils";
import { getPreset } from "./presets";
import { renderSegment, type SegmentContext } from "./segments";
@@ -26,30 +24,15 @@ import type {
} from "./types";
// ═══════════════════════════════════════════════════════════════════════════
// Per-message token cache
// Context-usage memo
// ═══════════════════════════════════════════════════════════════════════════
/**
* Symbol-keyed sidecar tagged onto each `AgentMessage` to memoize its
* `estimateTokens` result. Keyed by message identity (the object itself);
* a cheap content fingerprint detects in-place mutations (post-hoc error
* attachment, retry-truncated branch rebuild, etc.) and forces recompute.
*
* Cache lives on the message — multiple `StatusLineComponent` instances
* share it for free, and entries collect with the message itself when the
* conversation is replaced or compacted.
*/
const kTokenCache = Symbol("statusLine.tokenCache");
interface TaggedMessage {
[kTokenCache]?: { fingerprint: string; tokens: number };
}
/**
* Cheap structural fingerprint mirroring `estimateTokens`'s content walk.
* O(blocks) — only reads string `.length` and primitives, never copies or
* serializes content. Any in-place mutation that alters total tokenized
* content also alters one of the byte-length sums or block counts captured
* here, forcing the cached `estimateTokens` value to be recomputed.
* Cheap structural fingerprint of a message's tokenizable content. O(blocks) —
* only reads string `.length` and primitives, never copies or serializes.
* Detects in-place growth of the streaming tail (and other in-place mutations)
* so the cached `getContextUsage()` result is recomputed when — and only when —
* the numbers it depends on change.
*/
function messageFingerprint(msg: AgentMessage): string {
const role = (msg as { role?: string }).role ?? "";
@@ -107,28 +90,16 @@ function messageFingerprint(msg: AgentMessage): string {
return `${role}:${ts}:${textLen}:${blocks}:${images}`;
}
/**
* Token count for a single message, using the per-message sidecar cache.
* The caller MUST skip caching for the last message during streaming —
* it may still be growing and its tokens belong recomputed each refresh.
*/
function tokensForMessage(msg: AgentMessage): number {
const fp = messageFingerprint(msg);
const tagged = msg as TaggedMessage;
const cached = tagged[kTokenCache];
if (cached && cached.fingerprint === fp) return cached.tokens;
const tokens = estimateTokens(msg);
tagged[kTokenCache] = { fingerprint: fp, tokens };
return tokens;
interface ContextUsageMemo {
messagesRef: readonly AgentMessage[];
length: number;
lastFingerprint: string | undefined;
modelContextWindow: number;
usedTokens: number | null;
contextWindow: number;
}
interface MessageTokenTotalsCache {
messagesRef: readonly AgentMessage[];
stableCount: number;
stableTokens: number;
lastStableMessage: AgentMessage | undefined;
lastStableFingerprint: string | undefined;
}
const EMPTY_MESSAGES: readonly AgentMessage[] = [];
function hasContextSegment(segments: readonly StatusLineSegmentId[]): boolean {
return segments.includes("context_pct") || segments.includes("context_total");
@@ -176,19 +147,12 @@ export class StatusLineComponent implements Component {
} | null = null;
#usageFetchedAt = 0;
#usageInFlight = false;
// Context breakdown — incremental rolling cache. The status line refreshes
// on every agent event, so the hot path must not re-tokenize the full
// message list. Stable messages are accumulated once; normal streaming
// refreshes only recompute the current tail message and newly appended
// entries. History rewrites/compaction replace or shrink the message array
// and rebuild this cache. Stable messages are treated as immutable after
// promotion, matching the normal append-only session flow.
// Cached non-message total (system prompt + tools + skills). Invalidated
// when the inputs-identity fingerprint changes (model swap, skill toggle,
// tool registration).
#nonMessageTokensCache: number | undefined;
#nonMessageInputsKey: string | undefined;
#messageTokenTotalsCache: MessageTokenTotalsCache | undefined;
// Context-usage memo. The status line redraws on every agent event, so the
// hot path must not recompute context tokens unless an input changed.
// `getContextUsage()` anchors on the last assistant's real prompt-token
// count (matching the provider and the `/context` panel), so a stable
// message list + model window yields a stable result we can return verbatim.
#contextUsageCache: ContextUsageMemo | undefined;
constructor(private session: AgentSession) {
this.#settings = {
@@ -310,9 +274,7 @@ export class StatusLineComponent implements Component {
this.#cachedUsage = null;
this.#usageFetchedAt = 0;
this.#usageInFlight = false;
this.#nonMessageTokensCache = undefined;
this.#nonMessageInputsKey = undefined;
this.#messageTokenTotalsCache = undefined;
this.#contextUsageCache = undefined;
this.#lastTokensPerSecond = null;
this.#lastTokensPerSecondTimestamp = null;
}
@@ -544,109 +506,44 @@ export class StatusLineComponent implements Component {
}
/**
* Compute the (cached) used-tokens / context-window totals for the
* status-line context% segment. Exposed (non-private) so unit tests can
* verify the incremental-cache invariants; not part of any external
* API.
* Used-tokens / context-window totals for the status-line context% segment,
* memoized so the per-event redraw stays O(1) when nothing changed.
*
* The numerator comes from `session.getContextUsage()`, which anchors on the
* last assistant's real prompt-token count — so the bar matches the provider
* and the `/context` panel — and reports `null` while that count is unknown
* (right after compaction, before the next response). Exposed (non-private)
* for unit tests and the collab host's state broadcast.
*/
getCachedContextBreakdown(): { usedTokens: number; contextWindow: number } {
const messages = this.session.messages ?? [];
const contextWindow = this.session.model?.contextWindow ?? 0;
// 1) Non-message tokens (system prompt + tools + skills). Refresh only
// when the inputs identity fingerprint changes — usually never
// during a streaming turn. ~10-30 ms when it does refresh.
const inputsKey = this.#computeNonMessageInputsKey();
if (this.#nonMessageTokensCache === undefined || this.#nonMessageInputsKey !== inputsKey) {
this.#nonMessageTokensCache = computeNonMessageTokens(this.session);
this.#nonMessageInputsKey = inputsKey;
}
// 2) Message tokens — incremental rolling total. The sidecar cache lives
// on each stable message object (all but the current tail). Normal
// streaming turns only recompute the last message and newly appended
// entries. Full rebuild only when the message array is replaced,
// shrinks, or the recently-promoted stable tail mutates in place.
const messagesTokens = this.#getCachedMessageTokens(messages);
const usedTokens = this.#nonMessageTokensCache + messagesTokens;
return { usedTokens, contextWindow };
}
#getCachedMessageTokens(messages: readonly AgentMessage[]): number {
const cache = this.#messageTokenTotalsCache;
if (!cache || cache.messagesRef !== messages || messages.length <= cache.stableCount) {
return this.#rebuildMessageTokenTotals(messages);
}
let stableTokens = cache.stableTokens;
let stableCount = cache.stableCount;
const stableLimit = Math.max(0, messages.length - 1);
getCachedContextBreakdown(): { usedTokens: number | null; contextWindow: number } {
const messages = this.session.messages ?? EMPTY_MESSAGES;
const modelContextWindow = this.session.model?.contextWindow ?? 0;
const length = messages.length;
const lastFingerprint = length > 0 ? messageFingerprint(messages[length - 1]!) : undefined;
const cache = this.#contextUsageCache;
if (
cache.lastStableMessage &&
stableCount > 0 &&
messages[stableCount - 1] === cache.lastStableMessage &&
cache.lastStableFingerprint !== undefined &&
cache.lastStableFingerprint !== messageFingerprint(cache.lastStableMessage)
cache &&
cache.messagesRef === messages &&
cache.length === length &&
cache.lastFingerprint === lastFingerprint &&
cache.modelContextWindow === modelContextWindow
) {
return this.#rebuildMessageTokenTotals(messages);
return { usedTokens: cache.usedTokens, contextWindow: cache.contextWindow };
}
while (stableCount < stableLimit) {
const promoted = messages[stableCount]!;
stableTokens += tokensForMessage(promoted);
stableCount++;
}
const lastStableMessage = stableCount > 0 ? messages[stableCount - 1] : undefined;
const lastStableFingerprint = lastStableMessage ? messageFingerprint(lastStableMessage) : undefined;
const lastMessage = messages.at(-1);
const lastTokens = lastMessage ? estimateTokens(lastMessage) : 0;
this.#messageTokenTotalsCache = {
const usage = this.session.getContextUsage();
const usedTokens = usage?.tokens ?? null;
const contextWindow = usage?.contextWindow ?? modelContextWindow;
this.#contextUsageCache = {
messagesRef: messages,
stableCount,
stableTokens,
lastStableMessage,
lastStableFingerprint,
length,
lastFingerprint,
modelContextWindow,
usedTokens,
contextWindow,
};
return stableTokens + lastTokens;
}
#rebuildMessageTokenTotals(messages: readonly AgentMessage[]): number {
let stableTokens = 0;
const stableLimit = Math.max(0, messages.length - 1);
for (let i = 0; i < stableLimit; i++) {
stableTokens += tokensForMessage(messages[i]!);
}
const lastStableMessage = stableLimit > 0 ? messages[stableLimit - 1] : undefined;
const lastStableFingerprint = lastStableMessage ? messageFingerprint(lastStableMessage) : undefined;
const lastMessage = messages.at(-1);
const lastTokens = lastMessage ? estimateTokens(lastMessage) : 0;
this.#messageTokenTotalsCache = {
messagesRef: messages,
stableCount: stableLimit,
stableTokens,
lastStableMessage,
lastStableFingerprint,
};
return stableTokens + lastTokens;
}
/**
* Build an identity fingerprint for the non-message inputs (system prompt,
* tools, skills). When this changes, the non-message token cache must be
* recomputed. Cheap: just lengths + first-string-length. Doesn't need to
* be cryptographically unique — only stable for the same inputs.
*/
#computeNonMessageInputsKey(): string {
const sp = this.session.systemPrompt ?? [];
const tools = this.session.agent?.state?.tools ?? [];
const skills = this.session.skills ?? [];
const modelId = this.session.model?.id ?? "";
return `${modelId}|${sp.length}:${sp[0]?.length ?? 0}|${tools.length}|${skills.length}`;
return { usedTokens, contextWindow };
}
#buildSegmentContext(
@@ -673,14 +570,14 @@ export class StatusLineComponent implements Component {
tokensPerSecond: this.#getTokensPerSecond(),
};
let contextTokens = 0;
let contextWindow = state.model?.contextWindow ?? this.session.model?.contextWindow ?? 0;
let contextPercent: number | null = 0;
if (includeContext) {
const breakdown = this.getCachedContextBreakdown();
contextTokens = breakdown.usedTokens;
contextWindow = breakdown.contextWindow || contextWindow;
contextPercent =
breakdown.usedTokens === null ? null : contextWindow > 0 ? (breakdown.usedTokens / contextWindow) * 100 : 0;
}
let contextPercent = contextWindow > 0 ? (contextTokens / contextWindow) * 100 : 0;
// Collab guest: context comes from the host's state frames — the local
// replica does no accounting of its own.
@@ -364,7 +364,7 @@ const contextPctSegment: StatusLineSegment = {
const autoIcon = ctx.autoCompactEnabled && theme.icon.auto ? ` ${theme.icon.auto}` : "";
const text = `${formatContextUsage(pct, window)}${autoIcon}`;
const color = getContextUsageThemeColor(getContextUsageLevel(pct, window));
const color = getContextUsageThemeColor(getContextUsageLevel(pct ?? 0, window));
const content = withIcon(theme.icon.context, theme.fg(color, text));
return { content, visible: true };
@@ -71,7 +71,8 @@ export interface SegmentContext {
cost: number;
tokensPerSecond: number | null;
};
contextPercent: number;
/** Context usage percent, or null when unknown (e.g. right after compaction). */
contextPercent: number | null;
contextWindow: number;
autoCompactEnabled: boolean;
subagentCount: number;
@@ -1,26 +1,24 @@
/**
* Regression guard for the incremental per-message token cache in
* `StatusLineComponent.getCachedContextBreakdown`.
* Contract for `StatusLineComponent.getCachedContextBreakdown`.
*
* Before the cache: every call walked `session.messages` and ran
* `estimateTokens` per message (~0.5 ms each native). For a 2,300-message
* session this was a ~1.1 s blocking call. `updateEditorTopBorder()` is
* invoked on every agent event (event-controller.ts:163), so during
* streaming the UI froze for ~1.1 s every 2 s (the prior cache TTL).
* The status-line context% segment no longer keeps its own cl100k estimate of
* the whole conversation. It surfaces `session.getContextUsage()`, which
* anchors on the last assistant's real provider prompt-token count — so the bar
* matches the provider and the `/context` panel instead of an independent
* estimate that drifted past 100%.
*
* After the cache: messages are walked ONCE during warm-up; subsequent
* refreshes append-only update the cache by `messages.length - cached`
* (typically 0–1 new messages). The LAST message is recomputed every call
* because its content may still be growing during streaming. Compaction
* (messages.length shrinks) resets the cache.
* `getTopBorder()` runs on every agent event (event-controller.ts), so the
* breakdown is memoized: it re-queries `getContextUsage()` only when an input
* it depends on changes (a new/grown message, a replaced message array, or the
* model's context window). A stable conversation must not re-query on every
* redraw — that per-event recompute is what previously froze large sessions.
*/
import { afterAll, beforeAll, describe, expect, it } from "bun:test";
import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import type { ContextUsage } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/types";
import { StatusLineComponent } from "@oh-my-pi/pi-coding-agent/modes/components/status-line";
import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
import { computeNonMessageTokens, estimateToolSchemaTokens } from "@oh-my-pi/pi-coding-agent/modes/utils/context-usage";
import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { countTokens } from "@oh-my-pi/pi-natives";
beforeAll(async () => {
resetSettingsForTest();
@@ -32,22 +30,25 @@ afterAll(() => {
resetSettingsForTest();
});
function makeSession(opts: {
messages: unknown[];
systemPrompt?: string[];
tools?: { name: string; description: string; parameters?: unknown }[];
skills?: { name: string; description: string }[];
contextWindow?: number;
modelId?: string;
}): AgentSession {
const messages = opts.messages;
return {
messages,
systemPrompt: opts.systemPrompt ?? ["You are a helpful assistant."],
agent: { state: { tools: opts.tools ?? [] } },
skills: opts.skills ?? [],
model: { id: opts.modelId ?? "test-model", contextWindow: opts.contextWindow ?? 200_000 },
state: { messages, model: { contextWindow: opts.contextWindow ?? 200_000 } },
interface Fake {
session: AgentSession;
/** Number of times `getContextUsage()` was queried. */
usageCalls: () => number;
/** Swap the value the next `getContextUsage()` query returns. */
setUsage: (usage: ContextUsage | undefined) => void;
}
function makeSession(opts: { messages: unknown[]; contextWindow?: number; usage?: ContextUsage | undefined }): Fake {
const contextWindow = opts.contextWindow ?? 200_000;
let usage: ContextUsage | undefined = "usage" in opts ? opts.usage : { tokens: 1234, contextWindow, percent: 0.6 };
let calls = 0;
const session = {
messages: opts.messages,
systemPrompt: ["You are a helpful assistant."],
agent: { state: { tools: [] } },
skills: [],
model: { id: "test-model", contextWindow },
state: { messages: opts.messages, model: { contextWindow } },
sessionManager: {
getUsageStatistics: () => ({
input: 0,
@@ -60,7 +61,18 @@ function makeSession(opts: {
getSessionName: () => "test",
},
getAsyncJobSnapshot: () => ({ running: [] }),
getContextUsage: () => {
calls++;
return usage;
},
} as unknown as AgentSession;
return {
session,
usageCalls: () => calls,
setUsage: next => {
usage = next;
},
};
}
function userMessage(text: string): unknown {
@@ -70,155 +82,99 @@ function assistantMessage(text: string): unknown {
return { role: "assistant", content: [{ type: "text", text }] };
}
describe("StatusLineComponent incremental context breakdown cache", () => {
it("first call computes from scratch, second call returns same value", () => {
const session = makeSession({
messages: Array.from({ length: 50 }, (_, i) => userMessage(`message ${i}`.repeat(10))),
});
const comp = new StatusLineComponent(session);
const first = comp.getCachedContextBreakdown();
const second = comp.getCachedContextBreakdown();
expect(first.usedTokens).toBeGreaterThan(0);
expect(second.usedTokens).toBe(first.usedTokens);
expect(second.contextWindow).toBe(200_000);
});
it("appending a message increases the total by approximately the new message's tokens", () => {
const session = makeSession({
messages: [userMessage("hello world"), userMessage("another message here")],
});
const comp = new StatusLineComponent(session);
const before = comp.getCachedContextBreakdown();
(session.messages as unknown[]).push(assistantMessage("a third message reply with more text content"));
const after = comp.getCachedContextBreakdown();
expect(after.usedTokens).toBeGreaterThan(before.usedTokens);
expect(after.contextWindow).toBe(before.contextWindow);
});
it("popping the streaming tail rebuilds instead of double-counting the new tail", () => {
const messages = [
userMessage("stable head"),
userMessage("tail that becomes stable after append".repeat(20)),
assistantMessage("streaming tail removed by retry".repeat(20)),
];
const session = makeSession({ messages });
const comp = new StatusLineComponent(session);
const withStreamingTail = comp.getCachedContextBreakdown();
messages.pop();
const afterPop = comp.getCachedContextBreakdown();
const freshAfterPop = new StatusLineComponent(session).getCachedContextBreakdown();
expect(afterPop.usedTokens).toBeLessThan(withStreamingTail.usedTokens);
expect(afterPop.usedTokens).toBe(freshAfterPop.usedTokens);
});
it("compaction (messages.length shrinks) resets the cache and recomputes correctly", () => {
const session = makeSession({
messages: Array.from({ length: 20 }, (_, i) => userMessage(`message ${i}`.repeat(10))),
});
const comp = new StatusLineComponent(session);
const before = comp.getCachedContextBreakdown();
expect(before.usedTokens).toBeGreaterThan(0);
(session.messages as unknown[]).length = 0;
(session.messages as unknown[]).push(userMessage("compacted summary"));
const after = comp.getCachedContextBreakdown();
expect(after.usedTokens).toBeLessThan(before.usedTokens);
expect(after.usedTokens).toBeGreaterThan(0);
});
it("non-message inputs change → recomputes non-message portion", () => {
const session = makeSession({
describe("StatusLineComponent context breakdown", () => {
it("surfaces the provider-anchored tokens and context window from getContextUsage", () => {
const { session } = makeSession({
messages: [userMessage("hi")],
systemPrompt: ["You are an assistant."],
tools: [{ name: "bash", description: "Run shell commands", parameters: {} }],
skills: [{ name: "code", description: "Write code" }],
usage: { tokens: 5000, contextWindow: 272_000, percent: 1.8 },
});
const comp = new StatusLineComponent(session);
const v1 = comp.getCachedContextBreakdown();
const v2 = comp.getCachedContextBreakdown();
expect(v2.usedTokens).toBe(v1.usedTokens);
(session.agent as { state: { tools: unknown[] } }).state.tools.push({
name: "edit",
description: "Edit files",
parameters: {},
});
const v3 = comp.getCachedContextBreakdown();
expect(v3.usedTokens).toBeGreaterThan(v2.usedTokens);
const breakdown = new StatusLineComponent(session).getCachedContextBreakdown();
expect(breakdown.usedTokens).toBe(5000);
expect(breakdown.contextWindow).toBe(272_000);
});
it("non-message token shortcut matches previous category sum semantics", () => {
const session = makeSession({
messages: [],
systemPrompt: [
"You are an assistant.\n\n<skills>\n- code: Write code\n- review: Review code\n</skills>",
"Loaded context file",
"Runtime note",
],
tools: [
{
name: "bash",
description: "Run shell commands",
parameters: { type: "object", properties: { command: { type: "string" } } },
},
],
skills: [
{ name: "code", description: "Write code" },
{ name: "review", description: "Review code" },
],
it("memoizes: repeated redraws with no change do not re-query usage", () => {
const { session, usageCalls } = makeSession({ messages: [userMessage("hi")] });
const comp = new StatusLineComponent(session);
comp.getCachedContextBreakdown();
comp.getCachedContextBreakdown();
comp.getCachedContextBreakdown();
expect(usageCalls()).toBe(1);
});
it("re-queries and surfaces the new total when a message is appended", () => {
const fake = makeSession({
messages: [userMessage("hi")],
usage: { tokens: 100, contextWindow: 200_000, percent: 0.05 },
});
const comp = new StatusLineComponent(fake.session);
expect(comp.getCachedContextBreakdown().usedTokens).toBe(100);
const skillsTokens = countTokens(["code", "Write code", "review", "Review code"]);
const previousCategorySum =
Math.max(0, countTokens(session.systemPrompt?.[0] ?? "") - skillsTokens) +
countTokens((session.systemPrompt ?? []).slice(1)) +
estimateToolSchemaTokens(session.agent?.state?.tools ?? []) +
skillsTokens;
(fake.session.messages as unknown[]).push(assistantMessage("a reply that bumped the real prompt size"));
fake.setUsage({ tokens: 250, contextWindow: 200_000, percent: 0.125 });
expect(new StatusLineComponent(session).getCachedContextBreakdown().usedTokens).toBe(previousCategorySum);
expect(computeNonMessageTokens(session)).toBe(previousCategorySum);
expect(comp.getCachedContextBreakdown().usedTokens).toBe(250);
expect(fake.usageCalls()).toBe(2);
});
it("zero messages: produces only non-message tokens, no crash", () => {
const session = makeSession({ messages: [] });
const comp = new StatusLineComponent(session);
const result = comp.getCachedContextBreakdown();
expect(result.usedTokens).toBeGreaterThanOrEqual(0);
expect(result.contextWindow).toBe(200_000);
});
it("in-place mutation of a recently promoted stable message recomputes its tokens", () => {
const head = userMessage("head");
const promoted = userMessage("promoted");
const tail = userMessage("tail");
const session = makeSession({ messages: [head, promoted, tail] });
it("re-queries when the streaming tail grows in place", () => {
const tail = assistantMessage("partial") as { content: { type: string; text: string }[] };
const { session, usageCalls } = makeSession({ messages: [userMessage("hi"), tail] });
const comp = new StatusLineComponent(session);
const before = comp.getCachedContextBreakdown();
// Mutate the most recently promoted stable message in place — this is the
// only stable slot we intentionally revalidate on the hot path.
(promoted as { content: string }).content = "a much longer body that should tokenize to more".repeat(20);
const after = comp.getCachedContextBreakdown();
comp.getCachedContextBreakdown();
tail.content[0]!.text = "partial response that kept streaming".repeat(8);
comp.getCachedContextBreakdown();
expect(after.usedTokens).toBeGreaterThan(before.usedTokens);
expect(usageCalls()).toBe(2);
});
it("getTopBorder skips context accounting when no context segments are rendered", () => {
const session = makeSession({
messages: Array.from({ length: 20 }, (_, i) => userMessage(`message ${i}`.repeat(10))),
it("re-queries when the message array is replaced (branch switch / rebuild)", () => {
const { session, usageCalls } = makeSession({
messages: [userMessage("a"), userMessage("b")],
});
const comp = new StatusLineComponent(session);
const before = comp.getCachedContextBreakdown();
comp.getCachedContextBreakdown();
(session as { messages: unknown[] }).messages = [userMessage("c"), userMessage("d")];
comp.getCachedContextBreakdown();
expect(usageCalls()).toBe(2);
});
it("re-queries when the model context window changes", () => {
const { session, usageCalls } = makeSession({ messages: [userMessage("hi")], contextWindow: 200_000 });
const comp = new StatusLineComponent(session);
comp.getCachedContextBreakdown();
(session.model as { contextWindow: number }).contextWindow = 400_000;
comp.getCachedContextBreakdown();
expect(usageCalls()).toBe(2);
});
it("propagates an unknown (null) token count, e.g. right after compaction", () => {
const { session } = makeSession({
messages: [userMessage("compaction summary")],
usage: { tokens: null, contextWindow: 272_000, percent: null },
});
const breakdown = new StatusLineComponent(session).getCachedContextBreakdown();
expect(breakdown.usedTokens).toBeNull();
expect(breakdown.contextWindow).toBe(272_000);
});
it("falls back to the model window with null tokens when usage is unavailable", () => {
const { session } = makeSession({ messages: [userMessage("hi")], usage: undefined, contextWindow: 128_000 });
const breakdown = new StatusLineComponent(session).getCachedContextBreakdown();
expect(breakdown.usedTokens).toBeNull();
expect(breakdown.contextWindow).toBe(128_000);
});
it("does not query usage when no context segment is rendered", () => {
const { session, usageCalls } = makeSession({ messages: [userMessage("hi")] });
const comp = new StatusLineComponent(session);
comp.updateSettings({
preset: "custom",
leftSegments: ["pi"],
@@ -228,45 +184,6 @@ describe("StatusLineComponent incremental context breakdown cache", () => {
const border = comp.getTopBorder(80);
expect(border.content.length).toBeGreaterThan(0);
expect(comp.getCachedContextBreakdown()).toEqual(before);
});
it("replaceMessages with same length but different shape recomputes tokens", () => {
const session = makeSession({
messages: [userMessage("short a"), userMessage("short b")],
});
const comp = new StatusLineComponent(session);
const before = comp.getCachedContextBreakdown();
// Same-length replace: distinct message objects with larger payloads.
(session as { messages: unknown[] }).messages = [
userMessage("a much longer payload".repeat(20)),
userMessage("another longer payload".repeat(20)),
];
const after = comp.getCachedContextBreakdown();
expect(after.usedTokens).toBeGreaterThan(before.usedTokens);
});
it("usage fetch error backs off — a failed fetch does not retrigger within the TTL window", async () => {
const session = makeSession({ messages: [userMessage("hi")] });
let calls = 0;
(session as { fetchUsageReports?: () => Promise<unknown> }).fetchUsageReports = () => {
calls++;
return Promise.reject(new Error("network"));
};
const comp = new StatusLineComponent(session);
// First refresh → kicks off fetch #1.
comp.refreshUsageInBackground();
// Let the rejected fetch settle so the .catch backoff stamp lands.
await Bun.sleep(0);
expect(calls).toBe(1);
// Subsequent refreshes within the TTL window must not refetch.
comp.refreshUsageInBackground();
comp.refreshUsageInBackground();
await Bun.sleep(0);
expect(calls).toBe(1);
expect(usageCalls()).toBe(0);
});
});
@@ -57,6 +57,7 @@ function makeSession(sessionName = "Cache Session") {
cost: 0,
}),
},
getContextUsage: () => undefined,
} as unknown as ConstructorParameters<typeof StatusLineComponent>[0];
}