refactor(coding-agent): reorganized status-line context cache usage flow
- Refactored status-line context caching to be keyed by message, tail, and window. - Updated context usage flow to pass breakdown tokens and expose null usage when unknown. - Adjusted context percentage handling so zero or unknown windows yield nullable values. - Expanded status-line cache tests for usage provenance, memoization, and invalidation.
This commit is contained in:
@@ -411,10 +411,11 @@ export class CollabHost {
|
||||
|
||||
#buildState(): CollabSessionState {
|
||||
const session = this.#ctx.session;
|
||||
// Context numbers come from the status line's breakdown — not
|
||||
// session.getContextUsage() — so guests render exactly what the host's
|
||||
// own footer shows.
|
||||
// Context numbers come from the status line's memoized breakdown so guests
|
||||
// render exactly the same anchored, provider-real count the host's own
|
||||
// status line shows.
|
||||
const breakdown = this.#ctx.statusLine.getCachedContextBreakdown();
|
||||
const tokens = breakdown.usedTokens;
|
||||
return {
|
||||
isStreaming: session.isStreaming,
|
||||
isAborting: session.isAborting,
|
||||
@@ -424,9 +425,9 @@ export class CollabHost {
|
||||
model: session.model,
|
||||
thinkingLevel: session.thinkingLevel,
|
||||
contextUsage: {
|
||||
tokens: breakdown.usedTokens,
|
||||
tokens,
|
||||
contextWindow: breakdown.contextWindow,
|
||||
percent: breakdown.contextWindow > 0 ? (breakdown.usedTokens / breakdown.contextWindow) * 100 : null,
|
||||
percent: tokens !== null && breakdown.contextWindow > 0 ? (tokens / breakdown.contextWindow) * 100 : null,
|
||||
},
|
||||
participants: this.participants,
|
||||
};
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
|
||||
import { estimateTokens } from "@oh-my-pi/pi-agent-core/compaction";
|
||||
import { type Component, truncateToWidth, visibleWidth } from "@oh-my-pi/pi-tui";
|
||||
import { getProjectDir } from "@oh-my-pi/pi-utils";
|
||||
import { $ } from "bun";
|
||||
@@ -11,7 +10,6 @@ import * as git from "../../../utils/git";
|
||||
import { getSessionAccentAnsi, getSessionAccentHex } from "../../../utils/session-color";
|
||||
import { sanitizeStatusText } from "../../shared";
|
||||
import { theme } from "../../theme/theme";
|
||||
import { computeNonMessageTokens } from "../../utils/context-usage";
|
||||
import { canReuseCachedPr, createPrCacheContext, isSamePrCacheContext, type PrCacheContext } from "./git-utils";
|
||||
import { getPreset } from "./presets";
|
||||
import { renderSegment, type SegmentContext } from "./segments";
|
||||
@@ -26,30 +24,15 @@ import type {
|
||||
} from "./types";
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════════════
|
||||
// Per-message token cache
|
||||
// Context-usage memo
|
||||
// ═══════════════════════════════════════════════════════════════════════════
|
||||
|
||||
/**
|
||||
* Symbol-keyed sidecar tagged onto each `AgentMessage` to memoize its
|
||||
* `estimateTokens` result. Keyed by message identity (the object itself);
|
||||
* a cheap content fingerprint detects in-place mutations (post-hoc error
|
||||
* attachment, retry-truncated branch rebuild, etc.) and forces recompute.
|
||||
*
|
||||
* Cache lives on the message — multiple `StatusLineComponent` instances
|
||||
* share it for free, and entries collect with the message itself when the
|
||||
* conversation is replaced or compacted.
|
||||
*/
|
||||
const kTokenCache = Symbol("statusLine.tokenCache");
|
||||
interface TaggedMessage {
|
||||
[kTokenCache]?: { fingerprint: string; tokens: number };
|
||||
}
|
||||
|
||||
/**
|
||||
* Cheap structural fingerprint mirroring `estimateTokens`'s content walk.
|
||||
* O(blocks) — only reads string `.length` and primitives, never copies or
|
||||
* serializes content. Any in-place mutation that alters total tokenized
|
||||
* content also alters one of the byte-length sums or block counts captured
|
||||
* here, forcing the cached `estimateTokens` value to be recomputed.
|
||||
* Cheap structural fingerprint of a message's tokenizable content. O(blocks) —
|
||||
* only reads string `.length` and primitives, never copies or serializes.
|
||||
* Detects in-place growth of the streaming tail (and other in-place mutations)
|
||||
* so the cached `getContextUsage()` result is recomputed when — and only when —
|
||||
* the numbers it depends on change.
|
||||
*/
|
||||
function messageFingerprint(msg: AgentMessage): string {
|
||||
const role = (msg as { role?: string }).role ?? "";
|
||||
@@ -107,28 +90,16 @@ function messageFingerprint(msg: AgentMessage): string {
|
||||
return `${role}:${ts}:${textLen}:${blocks}:${images}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Token count for a single message, using the per-message sidecar cache.
|
||||
* The caller MUST skip caching for the last message during streaming —
|
||||
* it may still be growing and its tokens belong recomputed each refresh.
|
||||
*/
|
||||
function tokensForMessage(msg: AgentMessage): number {
|
||||
const fp = messageFingerprint(msg);
|
||||
const tagged = msg as TaggedMessage;
|
||||
const cached = tagged[kTokenCache];
|
||||
if (cached && cached.fingerprint === fp) return cached.tokens;
|
||||
const tokens = estimateTokens(msg);
|
||||
tagged[kTokenCache] = { fingerprint: fp, tokens };
|
||||
return tokens;
|
||||
interface ContextUsageMemo {
|
||||
messagesRef: readonly AgentMessage[];
|
||||
length: number;
|
||||
lastFingerprint: string | undefined;
|
||||
modelContextWindow: number;
|
||||
usedTokens: number | null;
|
||||
contextWindow: number;
|
||||
}
|
||||
|
||||
interface MessageTokenTotalsCache {
|
||||
messagesRef: readonly AgentMessage[];
|
||||
stableCount: number;
|
||||
stableTokens: number;
|
||||
lastStableMessage: AgentMessage | undefined;
|
||||
lastStableFingerprint: string | undefined;
|
||||
}
|
||||
const EMPTY_MESSAGES: readonly AgentMessage[] = [];
|
||||
|
||||
function hasContextSegment(segments: readonly StatusLineSegmentId[]): boolean {
|
||||
return segments.includes("context_pct") || segments.includes("context_total");
|
||||
@@ -176,19 +147,12 @@ export class StatusLineComponent implements Component {
|
||||
} | null = null;
|
||||
#usageFetchedAt = 0;
|
||||
#usageInFlight = false;
|
||||
// Context breakdown — incremental rolling cache. The status line refreshes
|
||||
// on every agent event, so the hot path must not re-tokenize the full
|
||||
// message list. Stable messages are accumulated once; normal streaming
|
||||
// refreshes only recompute the current tail message and newly appended
|
||||
// entries. History rewrites/compaction replace or shrink the message array
|
||||
// and rebuild this cache. Stable messages are treated as immutable after
|
||||
// promotion, matching the normal append-only session flow.
|
||||
// Cached non-message total (system prompt + tools + skills). Invalidated
|
||||
// when the inputs-identity fingerprint changes (model swap, skill toggle,
|
||||
// tool registration).
|
||||
#nonMessageTokensCache: number | undefined;
|
||||
#nonMessageInputsKey: string | undefined;
|
||||
#messageTokenTotalsCache: MessageTokenTotalsCache | undefined;
|
||||
// Context-usage memo. The status line redraws on every agent event, so the
|
||||
// hot path must not recompute context tokens unless an input changed.
|
||||
// `getContextUsage()` anchors on the last assistant's real prompt-token
|
||||
// count (matching the provider and the `/context` panel), so a stable
|
||||
// message list + model window yields a stable result we can return verbatim.
|
||||
#contextUsageCache: ContextUsageMemo | undefined;
|
||||
|
||||
constructor(private session: AgentSession) {
|
||||
this.#settings = {
|
||||
@@ -310,9 +274,7 @@ export class StatusLineComponent implements Component {
|
||||
this.#cachedUsage = null;
|
||||
this.#usageFetchedAt = 0;
|
||||
this.#usageInFlight = false;
|
||||
this.#nonMessageTokensCache = undefined;
|
||||
this.#nonMessageInputsKey = undefined;
|
||||
this.#messageTokenTotalsCache = undefined;
|
||||
this.#contextUsageCache = undefined;
|
||||
this.#lastTokensPerSecond = null;
|
||||
this.#lastTokensPerSecondTimestamp = null;
|
||||
}
|
||||
@@ -544,109 +506,44 @@ export class StatusLineComponent implements Component {
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute the (cached) used-tokens / context-window totals for the
|
||||
* status-line context% segment. Exposed (non-private) so unit tests can
|
||||
* verify the incremental-cache invariants; not part of any external
|
||||
* API.
|
||||
* Used-tokens / context-window totals for the status-line context% segment,
|
||||
* memoized so the per-event redraw stays O(1) when nothing changed.
|
||||
*
|
||||
* The numerator comes from `session.getContextUsage()`, which anchors on the
|
||||
* last assistant's real prompt-token count — so the bar matches the provider
|
||||
* and the `/context` panel — and reports `null` while that count is unknown
|
||||
* (right after compaction, before the next response). Exposed (non-private)
|
||||
* for unit tests and the collab host's state broadcast.
|
||||
*/
|
||||
getCachedContextBreakdown(): { usedTokens: number; contextWindow: number } {
|
||||
const messages = this.session.messages ?? [];
|
||||
const contextWindow = this.session.model?.contextWindow ?? 0;
|
||||
|
||||
// 1) Non-message tokens (system prompt + tools + skills). Refresh only
|
||||
// when the inputs identity fingerprint changes — usually never
|
||||
// during a streaming turn. ~10-30 ms when it does refresh.
|
||||
const inputsKey = this.#computeNonMessageInputsKey();
|
||||
if (this.#nonMessageTokensCache === undefined || this.#nonMessageInputsKey !== inputsKey) {
|
||||
this.#nonMessageTokensCache = computeNonMessageTokens(this.session);
|
||||
this.#nonMessageInputsKey = inputsKey;
|
||||
}
|
||||
|
||||
// 2) Message tokens — incremental rolling total. The sidecar cache lives
|
||||
// on each stable message object (all but the current tail). Normal
|
||||
// streaming turns only recompute the last message and newly appended
|
||||
// entries. Full rebuild only when the message array is replaced,
|
||||
// shrinks, or the recently-promoted stable tail mutates in place.
|
||||
const messagesTokens = this.#getCachedMessageTokens(messages);
|
||||
|
||||
const usedTokens = this.#nonMessageTokensCache + messagesTokens;
|
||||
return { usedTokens, contextWindow };
|
||||
}
|
||||
|
||||
#getCachedMessageTokens(messages: readonly AgentMessage[]): number {
|
||||
const cache = this.#messageTokenTotalsCache;
|
||||
if (!cache || cache.messagesRef !== messages || messages.length <= cache.stableCount) {
|
||||
return this.#rebuildMessageTokenTotals(messages);
|
||||
}
|
||||
|
||||
let stableTokens = cache.stableTokens;
|
||||
let stableCount = cache.stableCount;
|
||||
const stableLimit = Math.max(0, messages.length - 1);
|
||||
getCachedContextBreakdown(): { usedTokens: number | null; contextWindow: number } {
|
||||
const messages = this.session.messages ?? EMPTY_MESSAGES;
|
||||
const modelContextWindow = this.session.model?.contextWindow ?? 0;
|
||||
const length = messages.length;
|
||||
const lastFingerprint = length > 0 ? messageFingerprint(messages[length - 1]!) : undefined;
|
||||
|
||||
const cache = this.#contextUsageCache;
|
||||
if (
|
||||
cache.lastStableMessage &&
|
||||
stableCount > 0 &&
|
||||
messages[stableCount - 1] === cache.lastStableMessage &&
|
||||
cache.lastStableFingerprint !== undefined &&
|
||||
cache.lastStableFingerprint !== messageFingerprint(cache.lastStableMessage)
|
||||
cache &&
|
||||
cache.messagesRef === messages &&
|
||||
cache.length === length &&
|
||||
cache.lastFingerprint === lastFingerprint &&
|
||||
cache.modelContextWindow === modelContextWindow
|
||||
) {
|
||||
return this.#rebuildMessageTokenTotals(messages);
|
||||
return { usedTokens: cache.usedTokens, contextWindow: cache.contextWindow };
|
||||
}
|
||||
|
||||
while (stableCount < stableLimit) {
|
||||
const promoted = messages[stableCount]!;
|
||||
stableTokens += tokensForMessage(promoted);
|
||||
stableCount++;
|
||||
}
|
||||
|
||||
const lastStableMessage = stableCount > 0 ? messages[stableCount - 1] : undefined;
|
||||
const lastStableFingerprint = lastStableMessage ? messageFingerprint(lastStableMessage) : undefined;
|
||||
const lastMessage = messages.at(-1);
|
||||
const lastTokens = lastMessage ? estimateTokens(lastMessage) : 0;
|
||||
this.#messageTokenTotalsCache = {
|
||||
const usage = this.session.getContextUsage();
|
||||
const usedTokens = usage?.tokens ?? null;
|
||||
const contextWindow = usage?.contextWindow ?? modelContextWindow;
|
||||
this.#contextUsageCache = {
|
||||
messagesRef: messages,
|
||||
stableCount,
|
||||
stableTokens,
|
||||
lastStableMessage,
|
||||
lastStableFingerprint,
|
||||
length,
|
||||
lastFingerprint,
|
||||
modelContextWindow,
|
||||
usedTokens,
|
||||
contextWindow,
|
||||
};
|
||||
return stableTokens + lastTokens;
|
||||
}
|
||||
|
||||
#rebuildMessageTokenTotals(messages: readonly AgentMessage[]): number {
|
||||
let stableTokens = 0;
|
||||
const stableLimit = Math.max(0, messages.length - 1);
|
||||
for (let i = 0; i < stableLimit; i++) {
|
||||
stableTokens += tokensForMessage(messages[i]!);
|
||||
}
|
||||
|
||||
const lastStableMessage = stableLimit > 0 ? messages[stableLimit - 1] : undefined;
|
||||
const lastStableFingerprint = lastStableMessage ? messageFingerprint(lastStableMessage) : undefined;
|
||||
const lastMessage = messages.at(-1);
|
||||
const lastTokens = lastMessage ? estimateTokens(lastMessage) : 0;
|
||||
|
||||
this.#messageTokenTotalsCache = {
|
||||
messagesRef: messages,
|
||||
stableCount: stableLimit,
|
||||
stableTokens,
|
||||
lastStableMessage,
|
||||
lastStableFingerprint,
|
||||
};
|
||||
return stableTokens + lastTokens;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build an identity fingerprint for the non-message inputs (system prompt,
|
||||
* tools, skills). When this changes, the non-message token cache must be
|
||||
* recomputed. Cheap: just lengths + first-string-length. Doesn't need to
|
||||
* be cryptographically unique — only stable for the same inputs.
|
||||
*/
|
||||
#computeNonMessageInputsKey(): string {
|
||||
const sp = this.session.systemPrompt ?? [];
|
||||
const tools = this.session.agent?.state?.tools ?? [];
|
||||
const skills = this.session.skills ?? [];
|
||||
const modelId = this.session.model?.id ?? "";
|
||||
return `${modelId}|${sp.length}:${sp[0]?.length ?? 0}|${tools.length}|${skills.length}`;
|
||||
return { usedTokens, contextWindow };
|
||||
}
|
||||
|
||||
#buildSegmentContext(
|
||||
@@ -673,14 +570,14 @@ export class StatusLineComponent implements Component {
|
||||
tokensPerSecond: this.#getTokensPerSecond(),
|
||||
};
|
||||
|
||||
let contextTokens = 0;
|
||||
let contextWindow = state.model?.contextWindow ?? this.session.model?.contextWindow ?? 0;
|
||||
let contextPercent: number | null = 0;
|
||||
if (includeContext) {
|
||||
const breakdown = this.getCachedContextBreakdown();
|
||||
contextTokens = breakdown.usedTokens;
|
||||
contextWindow = breakdown.contextWindow || contextWindow;
|
||||
contextPercent =
|
||||
breakdown.usedTokens === null ? null : contextWindow > 0 ? (breakdown.usedTokens / contextWindow) * 100 : 0;
|
||||
}
|
||||
let contextPercent = contextWindow > 0 ? (contextTokens / contextWindow) * 100 : 0;
|
||||
|
||||
// Collab guest: context comes from the host's state frames — the local
|
||||
// replica does no accounting of its own.
|
||||
|
||||
@@ -364,7 +364,7 @@ const contextPctSegment: StatusLineSegment = {
|
||||
const autoIcon = ctx.autoCompactEnabled && theme.icon.auto ? ` ${theme.icon.auto}` : "";
|
||||
const text = `${formatContextUsage(pct, window)}${autoIcon}`;
|
||||
|
||||
const color = getContextUsageThemeColor(getContextUsageLevel(pct, window));
|
||||
const color = getContextUsageThemeColor(getContextUsageLevel(pct ?? 0, window));
|
||||
const content = withIcon(theme.icon.context, theme.fg(color, text));
|
||||
|
||||
return { content, visible: true };
|
||||
|
||||
@@ -71,7 +71,8 @@ export interface SegmentContext {
|
||||
cost: number;
|
||||
tokensPerSecond: number | null;
|
||||
};
|
||||
contextPercent: number;
|
||||
/** Context usage percent, or null when unknown (e.g. right after compaction). */
|
||||
contextPercent: number | null;
|
||||
contextWindow: number;
|
||||
autoCompactEnabled: boolean;
|
||||
subagentCount: number;
|
||||
|
||||
@@ -1,26 +1,24 @@
|
||||
/**
|
||||
* Regression guard for the incremental per-message token cache in
|
||||
* `StatusLineComponent.getCachedContextBreakdown`.
|
||||
* Contract for `StatusLineComponent.getCachedContextBreakdown`.
|
||||
*
|
||||
* Before the cache: every call walked `session.messages` and ran
|
||||
* `estimateTokens` per message (~0.5 ms each native). For a 2,300-message
|
||||
* session this was a ~1.1 s blocking call. `updateEditorTopBorder()` is
|
||||
* invoked on every agent event (event-controller.ts:163), so during
|
||||
* streaming the UI froze for ~1.1 s every 2 s (the prior cache TTL).
|
||||
* The status-line context% segment no longer keeps its own cl100k estimate of
|
||||
* the whole conversation. It surfaces `session.getContextUsage()`, which
|
||||
* anchors on the last assistant's real provider prompt-token count — so the bar
|
||||
* matches the provider and the `/context` panel instead of an independent
|
||||
* estimate that drifted past 100%.
|
||||
*
|
||||
* After the cache: messages are walked ONCE during warm-up; subsequent
|
||||
* refreshes append-only update the cache by `messages.length - cached`
|
||||
* (typically 0–1 new messages). The LAST message is recomputed every call
|
||||
* because its content may still be growing during streaming. Compaction
|
||||
* (messages.length shrinks) resets the cache.
|
||||
* `getTopBorder()` runs on every agent event (event-controller.ts), so the
|
||||
* breakdown is memoized: it re-queries `getContextUsage()` only when an input
|
||||
* it depends on changes (a new/grown message, a replaced message array, or the
|
||||
* model's context window). A stable conversation must not re-query on every
|
||||
* redraw — that per-event recompute is what previously froze large sessions.
|
||||
*/
|
||||
import { afterAll, beforeAll, describe, expect, it } from "bun:test";
|
||||
import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import type { ContextUsage } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/types";
|
||||
import { StatusLineComponent } from "@oh-my-pi/pi-coding-agent/modes/components/status-line";
|
||||
import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
|
||||
import { computeNonMessageTokens, estimateToolSchemaTokens } from "@oh-my-pi/pi-coding-agent/modes/utils/context-usage";
|
||||
import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
|
||||
import { countTokens } from "@oh-my-pi/pi-natives";
|
||||
|
||||
beforeAll(async () => {
|
||||
resetSettingsForTest();
|
||||
@@ -32,22 +30,25 @@ afterAll(() => {
|
||||
resetSettingsForTest();
|
||||
});
|
||||
|
||||
function makeSession(opts: {
|
||||
messages: unknown[];
|
||||
systemPrompt?: string[];
|
||||
tools?: { name: string; description: string; parameters?: unknown }[];
|
||||
skills?: { name: string; description: string }[];
|
||||
contextWindow?: number;
|
||||
modelId?: string;
|
||||
}): AgentSession {
|
||||
const messages = opts.messages;
|
||||
return {
|
||||
messages,
|
||||
systemPrompt: opts.systemPrompt ?? ["You are a helpful assistant."],
|
||||
agent: { state: { tools: opts.tools ?? [] } },
|
||||
skills: opts.skills ?? [],
|
||||
model: { id: opts.modelId ?? "test-model", contextWindow: opts.contextWindow ?? 200_000 },
|
||||
state: { messages, model: { contextWindow: opts.contextWindow ?? 200_000 } },
|
||||
interface Fake {
|
||||
session: AgentSession;
|
||||
/** Number of times `getContextUsage()` was queried. */
|
||||
usageCalls: () => number;
|
||||
/** Swap the value the next `getContextUsage()` query returns. */
|
||||
setUsage: (usage: ContextUsage | undefined) => void;
|
||||
}
|
||||
|
||||
function makeSession(opts: { messages: unknown[]; contextWindow?: number; usage?: ContextUsage | undefined }): Fake {
|
||||
const contextWindow = opts.contextWindow ?? 200_000;
|
||||
let usage: ContextUsage | undefined = "usage" in opts ? opts.usage : { tokens: 1234, contextWindow, percent: 0.6 };
|
||||
let calls = 0;
|
||||
const session = {
|
||||
messages: opts.messages,
|
||||
systemPrompt: ["You are a helpful assistant."],
|
||||
agent: { state: { tools: [] } },
|
||||
skills: [],
|
||||
model: { id: "test-model", contextWindow },
|
||||
state: { messages: opts.messages, model: { contextWindow } },
|
||||
sessionManager: {
|
||||
getUsageStatistics: () => ({
|
||||
input: 0,
|
||||
@@ -60,7 +61,18 @@ function makeSession(opts: {
|
||||
getSessionName: () => "test",
|
||||
},
|
||||
getAsyncJobSnapshot: () => ({ running: [] }),
|
||||
getContextUsage: () => {
|
||||
calls++;
|
||||
return usage;
|
||||
},
|
||||
} as unknown as AgentSession;
|
||||
return {
|
||||
session,
|
||||
usageCalls: () => calls,
|
||||
setUsage: next => {
|
||||
usage = next;
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
function userMessage(text: string): unknown {
|
||||
@@ -70,155 +82,99 @@ function assistantMessage(text: string): unknown {
|
||||
return { role: "assistant", content: [{ type: "text", text }] };
|
||||
}
|
||||
|
||||
describe("StatusLineComponent incremental context breakdown cache", () => {
|
||||
it("first call computes from scratch, second call returns same value", () => {
|
||||
const session = makeSession({
|
||||
messages: Array.from({ length: 50 }, (_, i) => userMessage(`message ${i}`.repeat(10))),
|
||||
});
|
||||
const comp = new StatusLineComponent(session);
|
||||
|
||||
const first = comp.getCachedContextBreakdown();
|
||||
const second = comp.getCachedContextBreakdown();
|
||||
|
||||
expect(first.usedTokens).toBeGreaterThan(0);
|
||||
expect(second.usedTokens).toBe(first.usedTokens);
|
||||
expect(second.contextWindow).toBe(200_000);
|
||||
});
|
||||
|
||||
it("appending a message increases the total by approximately the new message's tokens", () => {
|
||||
const session = makeSession({
|
||||
messages: [userMessage("hello world"), userMessage("another message here")],
|
||||
});
|
||||
const comp = new StatusLineComponent(session);
|
||||
|
||||
const before = comp.getCachedContextBreakdown();
|
||||
(session.messages as unknown[]).push(assistantMessage("a third message reply with more text content"));
|
||||
const after = comp.getCachedContextBreakdown();
|
||||
|
||||
expect(after.usedTokens).toBeGreaterThan(before.usedTokens);
|
||||
expect(after.contextWindow).toBe(before.contextWindow);
|
||||
});
|
||||
|
||||
it("popping the streaming tail rebuilds instead of double-counting the new tail", () => {
|
||||
const messages = [
|
||||
userMessage("stable head"),
|
||||
userMessage("tail that becomes stable after append".repeat(20)),
|
||||
assistantMessage("streaming tail removed by retry".repeat(20)),
|
||||
];
|
||||
const session = makeSession({ messages });
|
||||
const comp = new StatusLineComponent(session);
|
||||
|
||||
const withStreamingTail = comp.getCachedContextBreakdown();
|
||||
messages.pop();
|
||||
const afterPop = comp.getCachedContextBreakdown();
|
||||
const freshAfterPop = new StatusLineComponent(session).getCachedContextBreakdown();
|
||||
|
||||
expect(afterPop.usedTokens).toBeLessThan(withStreamingTail.usedTokens);
|
||||
expect(afterPop.usedTokens).toBe(freshAfterPop.usedTokens);
|
||||
});
|
||||
|
||||
it("compaction (messages.length shrinks) resets the cache and recomputes correctly", () => {
|
||||
const session = makeSession({
|
||||
messages: Array.from({ length: 20 }, (_, i) => userMessage(`message ${i}`.repeat(10))),
|
||||
});
|
||||
const comp = new StatusLineComponent(session);
|
||||
|
||||
const before = comp.getCachedContextBreakdown();
|
||||
expect(before.usedTokens).toBeGreaterThan(0);
|
||||
|
||||
(session.messages as unknown[]).length = 0;
|
||||
(session.messages as unknown[]).push(userMessage("compacted summary"));
|
||||
|
||||
const after = comp.getCachedContextBreakdown();
|
||||
expect(after.usedTokens).toBeLessThan(before.usedTokens);
|
||||
expect(after.usedTokens).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("non-message inputs change → recomputes non-message portion", () => {
|
||||
const session = makeSession({
|
||||
describe("StatusLineComponent context breakdown", () => {
|
||||
it("surfaces the provider-anchored tokens and context window from getContextUsage", () => {
|
||||
const { session } = makeSession({
|
||||
messages: [userMessage("hi")],
|
||||
systemPrompt: ["You are an assistant."],
|
||||
tools: [{ name: "bash", description: "Run shell commands", parameters: {} }],
|
||||
skills: [{ name: "code", description: "Write code" }],
|
||||
usage: { tokens: 5000, contextWindow: 272_000, percent: 1.8 },
|
||||
});
|
||||
const comp = new StatusLineComponent(session);
|
||||
|
||||
const v1 = comp.getCachedContextBreakdown();
|
||||
const v2 = comp.getCachedContextBreakdown();
|
||||
expect(v2.usedTokens).toBe(v1.usedTokens);
|
||||
|
||||
(session.agent as { state: { tools: unknown[] } }).state.tools.push({
|
||||
name: "edit",
|
||||
description: "Edit files",
|
||||
parameters: {},
|
||||
});
|
||||
const v3 = comp.getCachedContextBreakdown();
|
||||
expect(v3.usedTokens).toBeGreaterThan(v2.usedTokens);
|
||||
const breakdown = new StatusLineComponent(session).getCachedContextBreakdown();
|
||||
expect(breakdown.usedTokens).toBe(5000);
|
||||
expect(breakdown.contextWindow).toBe(272_000);
|
||||
});
|
||||
|
||||
it("non-message token shortcut matches previous category sum semantics", () => {
|
||||
const session = makeSession({
|
||||
messages: [],
|
||||
systemPrompt: [
|
||||
"You are an assistant.\n\n<skills>\n- code: Write code\n- review: Review code\n</skills>",
|
||||
"Loaded context file",
|
||||
"Runtime note",
|
||||
],
|
||||
tools: [
|
||||
{
|
||||
name: "bash",
|
||||
description: "Run shell commands",
|
||||
parameters: { type: "object", properties: { command: { type: "string" } } },
|
||||
},
|
||||
],
|
||||
skills: [
|
||||
{ name: "code", description: "Write code" },
|
||||
{ name: "review", description: "Review code" },
|
||||
],
|
||||
it("memoizes: repeated redraws with no change do not re-query usage", () => {
|
||||
const { session, usageCalls } = makeSession({ messages: [userMessage("hi")] });
|
||||
const comp = new StatusLineComponent(session);
|
||||
|
||||
comp.getCachedContextBreakdown();
|
||||
comp.getCachedContextBreakdown();
|
||||
comp.getCachedContextBreakdown();
|
||||
|
||||
expect(usageCalls()).toBe(1);
|
||||
});
|
||||
|
||||
it("re-queries and surfaces the new total when a message is appended", () => {
|
||||
const fake = makeSession({
|
||||
messages: [userMessage("hi")],
|
||||
usage: { tokens: 100, contextWindow: 200_000, percent: 0.05 },
|
||||
});
|
||||
const comp = new StatusLineComponent(fake.session);
|
||||
expect(comp.getCachedContextBreakdown().usedTokens).toBe(100);
|
||||
|
||||
const skillsTokens = countTokens(["code", "Write code", "review", "Review code"]);
|
||||
const previousCategorySum =
|
||||
Math.max(0, countTokens(session.systemPrompt?.[0] ?? "") - skillsTokens) +
|
||||
countTokens((session.systemPrompt ?? []).slice(1)) +
|
||||
estimateToolSchemaTokens(session.agent?.state?.tools ?? []) +
|
||||
skillsTokens;
|
||||
(fake.session.messages as unknown[]).push(assistantMessage("a reply that bumped the real prompt size"));
|
||||
fake.setUsage({ tokens: 250, contextWindow: 200_000, percent: 0.125 });
|
||||
|
||||
expect(new StatusLineComponent(session).getCachedContextBreakdown().usedTokens).toBe(previousCategorySum);
|
||||
expect(computeNonMessageTokens(session)).toBe(previousCategorySum);
|
||||
expect(comp.getCachedContextBreakdown().usedTokens).toBe(250);
|
||||
expect(fake.usageCalls()).toBe(2);
|
||||
});
|
||||
|
||||
it("zero messages: produces only non-message tokens, no crash", () => {
|
||||
const session = makeSession({ messages: [] });
|
||||
const comp = new StatusLineComponent(session);
|
||||
const result = comp.getCachedContextBreakdown();
|
||||
expect(result.usedTokens).toBeGreaterThanOrEqual(0);
|
||||
expect(result.contextWindow).toBe(200_000);
|
||||
});
|
||||
|
||||
it("in-place mutation of a recently promoted stable message recomputes its tokens", () => {
|
||||
const head = userMessage("head");
|
||||
const promoted = userMessage("promoted");
|
||||
const tail = userMessage("tail");
|
||||
const session = makeSession({ messages: [head, promoted, tail] });
|
||||
it("re-queries when the streaming tail grows in place", () => {
|
||||
const tail = assistantMessage("partial") as { content: { type: string; text: string }[] };
|
||||
const { session, usageCalls } = makeSession({ messages: [userMessage("hi"), tail] });
|
||||
const comp = new StatusLineComponent(session);
|
||||
|
||||
const before = comp.getCachedContextBreakdown();
|
||||
// Mutate the most recently promoted stable message in place — this is the
|
||||
// only stable slot we intentionally revalidate on the hot path.
|
||||
(promoted as { content: string }).content = "a much longer body that should tokenize to more".repeat(20);
|
||||
const after = comp.getCachedContextBreakdown();
|
||||
comp.getCachedContextBreakdown();
|
||||
tail.content[0]!.text = "partial response that kept streaming".repeat(8);
|
||||
comp.getCachedContextBreakdown();
|
||||
|
||||
expect(after.usedTokens).toBeGreaterThan(before.usedTokens);
|
||||
expect(usageCalls()).toBe(2);
|
||||
});
|
||||
|
||||
it("getTopBorder skips context accounting when no context segments are rendered", () => {
|
||||
const session = makeSession({
|
||||
messages: Array.from({ length: 20 }, (_, i) => userMessage(`message ${i}`.repeat(10))),
|
||||
it("re-queries when the message array is replaced (branch switch / rebuild)", () => {
|
||||
const { session, usageCalls } = makeSession({
|
||||
messages: [userMessage("a"), userMessage("b")],
|
||||
});
|
||||
const comp = new StatusLineComponent(session);
|
||||
const before = comp.getCachedContextBreakdown();
|
||||
comp.getCachedContextBreakdown();
|
||||
|
||||
(session as { messages: unknown[] }).messages = [userMessage("c"), userMessage("d")];
|
||||
comp.getCachedContextBreakdown();
|
||||
|
||||
expect(usageCalls()).toBe(2);
|
||||
});
|
||||
|
||||
it("re-queries when the model context window changes", () => {
|
||||
const { session, usageCalls } = makeSession({ messages: [userMessage("hi")], contextWindow: 200_000 });
|
||||
const comp = new StatusLineComponent(session);
|
||||
comp.getCachedContextBreakdown();
|
||||
|
||||
(session.model as { contextWindow: number }).contextWindow = 400_000;
|
||||
comp.getCachedContextBreakdown();
|
||||
|
||||
expect(usageCalls()).toBe(2);
|
||||
});
|
||||
|
||||
it("propagates an unknown (null) token count, e.g. right after compaction", () => {
|
||||
const { session } = makeSession({
|
||||
messages: [userMessage("compaction summary")],
|
||||
usage: { tokens: null, contextWindow: 272_000, percent: null },
|
||||
});
|
||||
const breakdown = new StatusLineComponent(session).getCachedContextBreakdown();
|
||||
expect(breakdown.usedTokens).toBeNull();
|
||||
expect(breakdown.contextWindow).toBe(272_000);
|
||||
});
|
||||
|
||||
it("falls back to the model window with null tokens when usage is unavailable", () => {
|
||||
const { session } = makeSession({ messages: [userMessage("hi")], usage: undefined, contextWindow: 128_000 });
|
||||
const breakdown = new StatusLineComponent(session).getCachedContextBreakdown();
|
||||
expect(breakdown.usedTokens).toBeNull();
|
||||
expect(breakdown.contextWindow).toBe(128_000);
|
||||
});
|
||||
|
||||
it("does not query usage when no context segment is rendered", () => {
|
||||
const { session, usageCalls } = makeSession({ messages: [userMessage("hi")] });
|
||||
const comp = new StatusLineComponent(session);
|
||||
comp.updateSettings({
|
||||
preset: "custom",
|
||||
leftSegments: ["pi"],
|
||||
@@ -228,45 +184,6 @@ describe("StatusLineComponent incremental context breakdown cache", () => {
|
||||
|
||||
const border = comp.getTopBorder(80);
|
||||
expect(border.content.length).toBeGreaterThan(0);
|
||||
expect(comp.getCachedContextBreakdown()).toEqual(before);
|
||||
});
|
||||
|
||||
it("replaceMessages with same length but different shape recomputes tokens", () => {
|
||||
const session = makeSession({
|
||||
messages: [userMessage("short a"), userMessage("short b")],
|
||||
});
|
||||
const comp = new StatusLineComponent(session);
|
||||
|
||||
const before = comp.getCachedContextBreakdown();
|
||||
// Same-length replace: distinct message objects with larger payloads.
|
||||
(session as { messages: unknown[] }).messages = [
|
||||
userMessage("a much longer payload".repeat(20)),
|
||||
userMessage("another longer payload".repeat(20)),
|
||||
];
|
||||
const after = comp.getCachedContextBreakdown();
|
||||
|
||||
expect(after.usedTokens).toBeGreaterThan(before.usedTokens);
|
||||
});
|
||||
|
||||
it("usage fetch error backs off — a failed fetch does not retrigger within the TTL window", async () => {
|
||||
const session = makeSession({ messages: [userMessage("hi")] });
|
||||
let calls = 0;
|
||||
(session as { fetchUsageReports?: () => Promise<unknown> }).fetchUsageReports = () => {
|
||||
calls++;
|
||||
return Promise.reject(new Error("network"));
|
||||
};
|
||||
const comp = new StatusLineComponent(session);
|
||||
|
||||
// First refresh → kicks off fetch #1.
|
||||
comp.refreshUsageInBackground();
|
||||
// Let the rejected fetch settle so the .catch backoff stamp lands.
|
||||
await Bun.sleep(0);
|
||||
expect(calls).toBe(1);
|
||||
|
||||
// Subsequent refreshes within the TTL window must not refetch.
|
||||
comp.refreshUsageInBackground();
|
||||
comp.refreshUsageInBackground();
|
||||
await Bun.sleep(0);
|
||||
expect(calls).toBe(1);
|
||||
expect(usageCalls()).toBe(0);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -57,6 +57,7 @@ function makeSession(sessionName = "Cache Session") {
|
||||
cost: 0,
|
||||
}),
|
||||
},
|
||||
getContextUsage: () => undefined,
|
||||
} as unknown as ConstructorParameters<typeof StatusLineComponent>[0];
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user