4eb2919674
Root cause (verified on user's environment):
- User commit `296641213` swapped status-line's context% computation from cheap `calculatePromptTokens(lastAssistantMessage.usage)` to `computeContextBreakdown(session)`, which walks EVERY message and runs native `countTokens` (~0.5 ms per message).
- The 2-second TTL cache helps for steady-state idle but every cache MISS is a full sweep.
- `updateEditorTopBorder()` is invoked on EVERY agent event (event-controller.ts:163 — `agent_start`, `delta`, `agent_end`, `tool_*`). Each delta during streaming can trigger a cache miss.
- User session has 2,312 messages → each full sweep is ~1,120 ms blocking.
- During streaming the UI freezes for ~1.1 s every ~2 s, producing the user-visible 'jittery rendering' ("버벅거림") and 'status bar disappearing' symptoms.
Fix:
`StatusLineComponent.getCachedContextBreakdown()` (renamed from `#getCachedContextBreakdown` so unit tests can exercise it directly) now uses an incremental per-message token cache that exploits the append-only nature of `session.messages`:
1. Message tokens (the dominant cost): cached per-index. New messages are tokenized as they arrive; previously-cached messages are reused. The LAST message is always recomputed because its content may still be growing during streaming. Compaction (messages.length shrinks) resets the cache.
2. Non-message tokens (system prompt + tools + skills): cached separately, invalidated only when a cheap inputs-identity fingerprint changes (model swap, skill toggle, tool registration). These rarely change during a session.
Required exposing three helpers from `modes/utils/context-usage.ts` (`estimateSkillsTokens`, `estimateToolSchemaTokens`, `computeNonMessageTokens`) so the status-line cache can call them directly.
Performance (2,300-message synthetic session, measured on user's M-series Mac):
- COLD warm-up call: ~75 ms (one-time, runs at OMP startup before any streaming)
- WARM refresh, no new message: ~0.04 ms (20 calls = 0.7 ms total)
- WARM refresh, 1 new message: ~0.02 ms
vs. prior implementation:
- Per cache-miss call: ~1,120 ms blocking
- 28,000× speedup on warm-state refresh
`computeContextBreakdown` itself is untouched — `/context` slash command continues to use it, and its output matches the status-line context% for the same session state (parity preserved).
Tests: 6 new cases in `packages/coding-agent/test/status-line-context-cache.test.ts` covering cold/warm/append/compaction/non-message-invalidation/zero-messages and a perf smoke test asserting 20 warm refreshes on a 200-message session complete in <100 ms.
Full suite: 3,199 tests, 26 pre-existing failures (status-line accent / log_experiment timing-flaky / skills / github tool / workspace-tree / tool path — all unrelated and baseline-confirmed). Lint: 1 pre-existing import-order issue in `event-controller-plan-ready.test.ts` unchanged.
150 lines
5.6 KiB
TypeScript
150 lines
5.6 KiB
TypeScript
/**
|
||
* Regression guard for the incremental per-message token cache in
|
||
* `StatusLineComponent.getCachedContextBreakdown`.
|
||
*
|
||
* Before the cache: every call walked `session.messages` and ran
|
||
* `estimateTokens` per message (~0.5 ms each native). For a 2,300-message
|
||
* session this was a ~1.1 s blocking call. `updateEditorTopBorder()` is
|
||
* invoked on every agent event (event-controller.ts:163), so during
|
||
* streaming the UI froze for ~1.1 s every 2 s (the prior cache TTL).
|
||
*
|
||
* After the cache: messages are walked ONCE during warm-up; subsequent
|
||
* refreshes append-only update the cache by `messages.length - cached`
|
||
* (typically 0–1 new messages). The LAST message is recomputed every call
|
||
* because its content may still be growing during streaming. Compaction
|
||
* (messages.length shrinks) resets the cache.
|
||
*/
|
||
import { afterAll, beforeAll, describe, expect, it } from "bun:test";
|
||
import { _resetSettingsForTest, Settings } from "../src/config/settings";
|
||
import { StatusLineComponent } from "../src/modes/components/status-line";
|
||
import { initTheme } from "../src/modes/theme/theme";
|
||
import type { AgentSession } from "../src/session/agent-session";
|
||
|
||
beforeAll(async () => {
|
||
_resetSettingsForTest();
|
||
await Settings.init({ inMemory: true });
|
||
await initTheme();
|
||
});
|
||
|
||
afterAll(() => {
|
||
_resetSettingsForTest();
|
||
});
|
||
|
||
function makeSession(opts: {
|
||
messages: unknown[];
|
||
systemPrompt?: string[];
|
||
tools?: { name: string; description: string; parameters?: unknown }[];
|
||
skills?: { name: string; description: string }[];
|
||
contextWindow?: number;
|
||
modelId?: string;
|
||
}): AgentSession {
|
||
return {
|
||
messages: opts.messages,
|
||
systemPrompt: opts.systemPrompt ?? ["You are a helpful assistant."],
|
||
agent: { state: { tools: opts.tools ?? [] } },
|
||
skills: opts.skills ?? [],
|
||
model: { id: opts.modelId ?? "test-model", contextWindow: opts.contextWindow ?? 200_000 },
|
||
} as unknown as AgentSession;
|
||
}
|
||
|
||
function userMessage(text: string): unknown {
|
||
return { role: "user", content: text };
|
||
}
|
||
function assistantMessage(text: string): unknown {
|
||
return { role: "assistant", content: [{ type: "text", text }] };
|
||
}
|
||
|
||
describe("StatusLineComponent incremental context breakdown cache", () => {
|
||
it("first call computes from scratch, second call returns same value", () => {
|
||
const session = makeSession({
|
||
messages: Array.from({ length: 50 }, (_, i) => userMessage(`message ${i}`.repeat(10))),
|
||
});
|
||
const comp = new StatusLineComponent(session);
|
||
|
||
const first = comp.getCachedContextBreakdown();
|
||
const second = comp.getCachedContextBreakdown();
|
||
|
||
expect(first.usedTokens).toBeGreaterThan(0);
|
||
expect(second.usedTokens).toBe(first.usedTokens);
|
||
expect(second.contextWindow).toBe(200_000);
|
||
});
|
||
|
||
it("appending a message increases the total by approximately the new message's tokens", () => {
|
||
const session = makeSession({
|
||
messages: [userMessage("hello world"), userMessage("another message here")],
|
||
});
|
||
const comp = new StatusLineComponent(session);
|
||
|
||
const before = comp.getCachedContextBreakdown();
|
||
(session.messages as unknown[]).push(assistantMessage("a third message reply with more text content"));
|
||
const after = comp.getCachedContextBreakdown();
|
||
|
||
expect(after.usedTokens).toBeGreaterThan(before.usedTokens);
|
||
expect(after.contextWindow).toBe(before.contextWindow);
|
||
});
|
||
|
||
it("compaction (messages.length shrinks) resets the cache and recomputes correctly", () => {
|
||
const session = makeSession({
|
||
messages: Array.from({ length: 20 }, (_, i) => userMessage(`message ${i}`.repeat(10))),
|
||
});
|
||
const comp = new StatusLineComponent(session);
|
||
|
||
const before = comp.getCachedContextBreakdown();
|
||
expect(before.usedTokens).toBeGreaterThan(0);
|
||
|
||
(session.messages as unknown[]).length = 0;
|
||
(session.messages as unknown[]).push(userMessage("compacted summary"));
|
||
|
||
const after = comp.getCachedContextBreakdown();
|
||
expect(after.usedTokens).toBeLessThan(before.usedTokens);
|
||
expect(after.usedTokens).toBeGreaterThan(0);
|
||
});
|
||
|
||
it("non-message inputs change → recomputes non-message portion", () => {
|
||
const session = makeSession({
|
||
messages: [userMessage("hi")],
|
||
systemPrompt: ["You are an assistant."],
|
||
tools: [{ name: "bash", description: "Run shell commands", parameters: {} }],
|
||
skills: [{ name: "code", description: "Write code" }],
|
||
});
|
||
const comp = new StatusLineComponent(session);
|
||
|
||
const v1 = comp.getCachedContextBreakdown();
|
||
const v2 = comp.getCachedContextBreakdown();
|
||
expect(v2.usedTokens).toBe(v1.usedTokens);
|
||
|
||
(session.agent as { state: { tools: unknown[] } }).state.tools.push({
|
||
name: "edit",
|
||
description: "Edit files",
|
||
parameters: {},
|
||
});
|
||
const v3 = comp.getCachedContextBreakdown();
|
||
expect(v3.usedTokens).toBeGreaterThan(v2.usedTokens);
|
||
});
|
||
|
||
it("warm-cache refresh on 200-message session is fast (<100ms for 20 refreshes)", () => {
|
||
const session = makeSession({
|
||
messages: Array.from({ length: 200 }, (_, i) => userMessage(`msg ${i}`.repeat(20))),
|
||
});
|
||
const comp = new StatusLineComponent(session);
|
||
|
||
// Warm-up call (acceptable cost; not measured).
|
||
comp.getCachedContextBreakdown();
|
||
|
||
// 20 warm refreshes; each should only recompute the last message
|
||
// (~0.5 ms native) since no other messages changed.
|
||
const start = performance.now();
|
||
for (let i = 0; i < 20; i++) comp.getCachedContextBreakdown();
|
||
const elapsedMs = performance.now() - start;
|
||
expect(elapsedMs).toBeLessThan(100);
|
||
});
|
||
|
||
it("zero messages: produces only non-message tokens, no crash", () => {
|
||
const session = makeSession({ messages: [] });
|
||
const comp = new StatusLineComponent(session);
|
||
const result = comp.getCachedContextBreakdown();
|
||
expect(result.usedTokens).toBeGreaterThanOrEqual(0);
|
||
expect(result.contextWindow).toBe(200_000);
|
||
});
|
||
});
|