perf(coding-agent): memoize incremental grapheme slicing in streaming reveal

Each 30fps streaming-reveal tick re-segmented the whole revealed prefix via sliceGraphemes (per-tick slice cost grew with the prefix). BlockUnitCounter already memoized per-block grapheme counts; apply the same idea to slicing: cache the slice boundary per block index and re-segment only the per-step delta from the boundary cluster, so per-update cost grows with the delta instead of the rendered prefix.

buildDisplayMessage gains an optional sliceOf seam (default = sliceGraphemes, so existing callers are unchanged); the controller wires its BlockUnitCounter.slice through it via a #build helper that also de-duplicates six previously-repeated buildDisplayMessage call sites.

Only an exact (text, units) hit skips segmentation; the incremental guard (text === cached.text || text.startsWith(cached.text)) && units >= cached.units re-segments from the boundary cluster, so an append that extends the final cluster (e.g. a -> a combining acute, ZWJ family merge) is never stale.

Benchmark (bench/streaming-throughput.bench.ts, 30520-grapheme message, 61 ticks/episode): 23.78ms -> 2.27ms per episode (~10x). Regression tests vs a pure Intl.Segmenter reference cover fixed-text growing units, append growth, boundary-cluster extension, multi-block indices, shrink/regrow, full replace, and a 400-step seeded fuzz.
This commit is contained in:
oldschoola
2026-06-29 17:47:22 -07:00
parent 0ba736f5bc
commit 718c7cea2f
5 changed files with 326 additions and 66 deletions
+22 -13
View File
@@ -7,7 +7,7 @@ import { AssistantMessageComponent } from "../src/modes/components/assistant-mes
import { TranscriptContainer } from "../src/modes/components/transcript-container";
import { Settings } from "../src/config/settings";
import { getEditorTheme } from "../src/modes/theme/theme";
import { buildDisplayMessage, nextStep, visibleUnits } from "../src/modes/controllers/streaming-reveal";
import { BlockUnitCounter, buildDisplayMessage, nextStep, visibleUnits } from "../src/modes/controllers/streaming-reveal";
import * as os from "node:os";
import * as path from "node:path";
import * as fs from "node:fs";
@@ -55,11 +55,12 @@ bench("WelcomeComponent.render", () => {
// ── A2: streaming reveal + editor render baselines ──────────────────────────
//
// Diagnostic series, not a fixed-iteration micro-op. `streamingReveal` proves
// or refutes the O(N^2) reveal hypothesis: per-step cost (visibleUnits +
// buildDisplayMessage, the work every stream delta/30fps tick does) is sampled
// at growing revealed lengths. Rising per-step ms => O(N) per tick => O(N^2)
// over the message. Flat per-step => already linear.
// Diagnostic series, not a fixed-iteration micro-op. The full-reveal loops
// mirror the controller: a per-episode BlockUnitCounter feeds countOf + sliceOf
// (memoized, O(delta)/tick). `streamingReveal` (C1) instead measures the DEFAULT
// pure-sliceGraphemes path at a fixed revealed length — the un-memoized cost the
// counter avoids. Representative controller-path throughput lives in
// bench/streaming-throughput.bench.ts.
function makeMarkdownCorpus(targetGraphemes: number): string {
const para =
@@ -106,7 +107,7 @@ const REVEAL_CORPUS = makeMarkdownCorpus(6000);
const REVEAL_CHECKPOINTS = [1000, 2000, 3000, 4000, 5000, 6000];
const STEP_REPS = 40;
console.log("\nstreamingReveal (isolated C1: visibleUnits + buildDisplayMessage per delta):");
console.log("\nstreamingReveal (C1: default pure-slice path, fixed revealed length, no memoization):");
for (const n of REVEAL_CHECKPOINTS) {
const msg = makeTextMessage(REVEAL_CORPUS.slice(0, n));
const revealed = Math.floor(n * 0.9);
@@ -117,13 +118,18 @@ for (const n of REVEAL_CHECKPOINTS) {
console.log(` len=${n}: ${ms.toFixed(4)}ms/step`);
}
// Real streaming cost: text GROWS every tick, so Markdown's text-keyed cache
// misses each step (the actual interactive path). Total ms to fully reveal an
// N-grapheme message in nextStep increments — the number C1+C2 reduce.
console.log("\nstreamingRevealFull (C1+C2: full incremental reveal, growing text => cache-miss/tick):");
// Controller path: a per-episode BlockUnitCounter memoizes count + slice, so
// buildDisplayMessage is O(delta)/tick. The Markdown render (component.render)
// still re-lexes the growing text each step here (no { transient: true }), so
// total ms is dominated by the render, not the slice. Total ms to fully reveal
// an N-grapheme message in nextStep increments.
console.log("\nstreamingRevealFull (controller-path counter + Markdown render, growing text):");
try {
for (const n of REVEAL_CHECKPOINTS) {
const full = makeTextMessage(REVEAL_CORPUS.slice(0, n));
const counter = new BlockUnitCounter();
const countOf = (index: number, text: string): number => counter.count(index, text);
const sliceOf = (index: number, text: string, units: number): string => counter.slice(index, text, units);
const total = visibleUnits(full, false);
const component = new AssistantMessageComponent();
const start = Bun.nanoseconds();
@@ -131,7 +137,7 @@ try {
let steps = 0;
while (revealed < total) {
revealed = Math.min(total, revealed + nextStep(total - revealed));
component.updateContent(buildDisplayMessage(full, revealed, false));
component.updateContent(buildDisplayMessage(full, revealed, false, countOf, sliceOf));
component.render(WIDTH);
steps++;
}
@@ -153,6 +159,9 @@ try {
const thinking = makeMarkdownCorpus(2500);
for (const n of [2000, 4000, 6000]) {
const full = makeThinkingPlusText(thinking, REVEAL_CORPUS.slice(0, n));
const counter = new BlockUnitCounter();
const countOf = (index: number, text: string): number => counter.count(index, text);
const sliceOf = (index: number, text: string, units: number): string => counter.slice(index, text, units);
const total = visibleUnits(full, false);
const component = new AssistantMessageComponent();
const start = Bun.nanoseconds();
@@ -160,7 +169,7 @@ try {
let steps = 0;
while (revealed < total) {
revealed = Math.min(total, revealed + nextStep(total - revealed));
component.updateContent(buildDisplayMessage(full, revealed, false));
component.updateContent(buildDisplayMessage(full, revealed, false, countOf, sliceOf));
component.render(WIDTH);
steps++;
}
@@ -0,0 +1,125 @@
/**
* Streaming-reveal per-tick compute benchmark.
*
* Mirrors `StreamingRevealController`'s per-tick work: re-count the visible
* units of the target (memoized `BlockUnitCounter.count`) and rebuild the
* display message (`buildDisplayMessage`), which slices each text block to the
* revealed prefix. One `BlockUnitCounter` is created per episode and shared by
* `countOf` + `sliceOf`, exactly as the controller holds one `#unitCounter` per
* streaming episode.
*
* The Markdown render is intentionally excluded: every controller tick passes
* `{ transient: true }` to `updateContent`, which disables the L2 cache and code
* highlighting, so the dominant per-tick cost is the slice of the growing prefix
* (re-segmented from offset 0 by the baseline `sliceGraphemes`). This isolates
* exactly that path.
*
* Metric (lower is better): total wall-clock to fully reveal one representative
* large assistant message through the controller's `nextStep` progression,
* averaged over episodes. `reveal_ms_per_step` is the same work divided by the
* number of reveal ticks.
*
* Run: bun run packages/coding-agent/bench/streaming-throughput.bench.ts
*/
import type { AssistantMessage } from "@oh-my-pi/pi-ai";
import { BlockUnitCounter, buildDisplayMessage, nextStep } from "../src/modes/controllers/streaming-reveal";
const HIDE_THINKING = false;
const PROSE_ONLY = true;
const WARMUP_EPISODES = 6;
const MEASURE_EPISODES = 40;
/** Prose + code + accented text + multibyte grapheme clusters (ZWJ families,
* flag sequences, skin-tone modifiers, CJK), representative of LLM output that
* makes Intl.Segmenter do real per-cluster work. */
const CHUNK = `Here is an overview of the rendering pipeline changes.
The streaming reveal controller now advances the revealed prefix each tick. Consider the helper:
\`\`\`ts
function sliceToUnits(text: string, units: number): string {
\tlet end = 0;
\tfor (const { index, segment } of segmenter.segment(text)) {
\t\tif (--units < 0) break;
\t\tend = index + segment.length;
\t}
\treturn text.slice(0, end);
}
\`\`\`
This handles café, naïve résumés, and emoji clusters like the 👨‍👩‍👧‍👦 family, the 🏳️‍🌈 flag, and the 👩🏽 skin-tone modifier. CJK text such as 日本語のテスト also segments correctly, and the heart ❤️ beats steadily. Each grapheme cluster is one user-perceived character: "👨‍👩‍👧‍👦" is a single unit, not seven code points. The decomposed sequence e followed by a combining acute accent (e + \\u0301) is likewise a single cluster, distinct from the precomposed form.
The adaptive step is \`nextStep = max(3, ceil(backlog / 8))\`, so the tick count stays roughly constant while the per-tick slice cost grows with the rendered prefix. Incremental slicing lowers that per-update cost from the prefix length to the per-step delta.
`;
function makeMessage(textBlocks: string[]): AssistantMessage {
return {
role: "assistant",
content: textBlocks.map(text => ({ type: "text" as const, text })),
api: "anthropic-messages",
provider: "anthropic",
model: "mock",
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "stop",
timestamp: 0,
};
}
/** Total visible graphemes of the target's text blocks via the counter (mirrors
* the controller's `#visibleUnits` for text-only blocks). */
function textUnits(target: AssistantMessage, counter: BlockUnitCounter): number {
let total = 0;
for (let i = 0; i < target.content.length; i++) {
const block = target.content[i];
if (block?.type === "text") total += counter.count(i, block.text);
}
return total;
}
/** Drive one full reveal episode: a fresh counter shared by countOf + sliceOf,
* an initial render at revealed = 0 (mirrors `begin`), then the `nextStep`
* catch-up loop (mirrors `#tick`). Returns the number of reveal ticks. */
function revealEpisode(target: AssistantMessage): number {
const counter = new BlockUnitCounter();
const countOf = (index: number, text: string): number => counter.count(index, text);
const sliceOf = (index: number, text: string, units: number): string => counter.slice(index, text, units);
buildDisplayMessage(target, 0, HIDE_THINKING, PROSE_ONLY, countOf, sliceOf);
let revealed = 0;
let ticks = 0;
for (;;) {
const total = textUnits(target, counter);
if (revealed >= total) break;
revealed = Math.min(total, revealed + nextStep(total - revealed));
buildDisplayMessage(target, revealed, HIDE_THINKING, PROSE_ONLY, countOf, sliceOf);
ticks += 1;
}
return ticks;
}
// Two text blocks of differing lengths exercise multi-block counter indexing.
const target = makeMessage([CHUNK.repeat(16), CHUNK.repeat(12)]);
const sizingCounter = new BlockUnitCounter();
const graphemes = textUnits(target, sizingCounter);
for (let episode = 0; episode < WARMUP_EPISODES; episode++) revealEpisode(target);
let totalTicks = 0;
const start = performance.now();
for (let episode = 0; episode < MEASURE_EPISODES; episode++) totalTicks += revealEpisode(target);
const elapsedMs = performance.now() - start;
const msPerEpisode = elapsedMs / MEASURE_EPISODES;
const msPerStep = elapsedMs / totalTicks;
console.log(`METRIC reveal_ms_per_episode=${msPerEpisode.toFixed(4)}`);
console.log(`METRIC reveal_ms_per_step=${msPerStep.toFixed(5)}`);
console.log(`ASI graphemes=${graphemes} episodes=${MEASURE_EPISODES} ticks_per_episode=${(totalTicks / MEASURE_EPISODES).toFixed(2)} warmup=${WARMUP_EPISODES}`);
console.log(`(reveal: ${graphemes} graphemes, ${(totalTicks / MEASURE_EPISODES).toFixed(1)} ticks/episode, ${msPerEpisode.toFixed(3)} ms/episode, ${msPerStep.toFixed(4)} ms/step)`);