perf(coding-agent): memoize incremental grapheme slicing in streaming reveal
Each 30fps streaming-reveal tick re-segmented the whole revealed prefix via sliceGraphemes (per-tick slice cost grew with the prefix). BlockUnitCounter already memoized per-block grapheme counts; apply the same idea to slicing: cache the slice boundary per block index and re-segment only the per-step delta from the boundary cluster, so per-update cost grows with the delta instead of the rendered prefix. buildDisplayMessage gains an optional sliceOf seam (default = sliceGraphemes, so existing callers are unchanged); the controller wires its BlockUnitCounter.slice through it via a #build helper that also de-duplicates six previously-repeated buildDisplayMessage call sites. Only an exact (text, units) hit skips segmentation; the incremental guard (text === cached.text || text.startsWith(cached.text)) && units >= cached.units re-segments from the boundary cluster, so an append that extends the final cluster (e.g. a -> a combining acute, ZWJ family merge) is never stale. Benchmark (bench/streaming-throughput.bench.ts, 30520-grapheme message, 61 ticks/episode): 23.78ms -> 2.27ms per episode (~10x). Regression tests vs a pure Intl.Segmenter reference cover fixed-text growing units, append growth, boundary-cluster extension, multi-block indices, shrink/regrow, full replace, and a 400-step seeded fuzz.
This commit is contained in:
@@ -7,7 +7,7 @@ import { AssistantMessageComponent } from "../src/modes/components/assistant-mes
|
||||
import { TranscriptContainer } from "../src/modes/components/transcript-container";
|
||||
import { Settings } from "../src/config/settings";
|
||||
import { getEditorTheme } from "../src/modes/theme/theme";
|
||||
import { buildDisplayMessage, nextStep, visibleUnits } from "../src/modes/controllers/streaming-reveal";
|
||||
import { BlockUnitCounter, buildDisplayMessage, nextStep, visibleUnits } from "../src/modes/controllers/streaming-reveal";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import * as fs from "node:fs";
|
||||
@@ -55,11 +55,12 @@ bench("WelcomeComponent.render", () => {
|
||||
|
||||
// ── A2: streaming reveal + editor render baselines ──────────────────────────
|
||||
//
|
||||
// Diagnostic series, not a fixed-iteration micro-op. `streamingReveal` proves
|
||||
// or refutes the O(N^2) reveal hypothesis: per-step cost (visibleUnits +
|
||||
// buildDisplayMessage, the work every stream delta/30fps tick does) is sampled
|
||||
// at growing revealed lengths. Rising per-step ms => O(N) per tick => O(N^2)
|
||||
// over the message. Flat per-step => already linear.
|
||||
// Diagnostic series, not a fixed-iteration micro-op. The full-reveal loops
|
||||
// mirror the controller: a per-episode BlockUnitCounter feeds countOf + sliceOf
|
||||
// (memoized, O(delta)/tick). `streamingReveal` (C1) instead measures the DEFAULT
|
||||
// pure-sliceGraphemes path at a fixed revealed length — the un-memoized cost the
|
||||
// counter avoids. Representative controller-path throughput lives in
|
||||
// bench/streaming-throughput.bench.ts.
|
||||
|
||||
function makeMarkdownCorpus(targetGraphemes: number): string {
|
||||
const para =
|
||||
@@ -106,7 +107,7 @@ const REVEAL_CORPUS = makeMarkdownCorpus(6000);
|
||||
const REVEAL_CHECKPOINTS = [1000, 2000, 3000, 4000, 5000, 6000];
|
||||
const STEP_REPS = 40;
|
||||
|
||||
console.log("\nstreamingReveal (isolated C1: visibleUnits + buildDisplayMessage per delta):");
|
||||
console.log("\nstreamingReveal (C1: default pure-slice path, fixed revealed length, no memoization):");
|
||||
for (const n of REVEAL_CHECKPOINTS) {
|
||||
const msg = makeTextMessage(REVEAL_CORPUS.slice(0, n));
|
||||
const revealed = Math.floor(n * 0.9);
|
||||
@@ -117,13 +118,18 @@ for (const n of REVEAL_CHECKPOINTS) {
|
||||
console.log(` len=${n}: ${ms.toFixed(4)}ms/step`);
|
||||
}
|
||||
|
||||
// Real streaming cost: text GROWS every tick, so Markdown's text-keyed cache
|
||||
// misses each step (the actual interactive path). Total ms to fully reveal an
|
||||
// N-grapheme message in nextStep increments — the number C1+C2 reduce.
|
||||
console.log("\nstreamingRevealFull (C1+C2: full incremental reveal, growing text => cache-miss/tick):");
|
||||
// Controller path: a per-episode BlockUnitCounter memoizes count + slice, so
|
||||
// buildDisplayMessage is O(delta)/tick. The Markdown render (component.render)
|
||||
// still re-lexes the growing text each step here (no { transient: true }), so
|
||||
// total ms is dominated by the render, not the slice. Total ms to fully reveal
|
||||
// an N-grapheme message in nextStep increments.
|
||||
console.log("\nstreamingRevealFull (controller-path counter + Markdown render, growing text):");
|
||||
try {
|
||||
for (const n of REVEAL_CHECKPOINTS) {
|
||||
const full = makeTextMessage(REVEAL_CORPUS.slice(0, n));
|
||||
const counter = new BlockUnitCounter();
|
||||
const countOf = (index: number, text: string): number => counter.count(index, text);
|
||||
const sliceOf = (index: number, text: string, units: number): string => counter.slice(index, text, units);
|
||||
const total = visibleUnits(full, false);
|
||||
const component = new AssistantMessageComponent();
|
||||
const start = Bun.nanoseconds();
|
||||
@@ -131,7 +137,7 @@ try {
|
||||
let steps = 0;
|
||||
while (revealed < total) {
|
||||
revealed = Math.min(total, revealed + nextStep(total - revealed));
|
||||
component.updateContent(buildDisplayMessage(full, revealed, false));
|
||||
component.updateContent(buildDisplayMessage(full, revealed, false, countOf, sliceOf));
|
||||
component.render(WIDTH);
|
||||
steps++;
|
||||
}
|
||||
@@ -153,6 +159,9 @@ try {
|
||||
const thinking = makeMarkdownCorpus(2500);
|
||||
for (const n of [2000, 4000, 6000]) {
|
||||
const full = makeThinkingPlusText(thinking, REVEAL_CORPUS.slice(0, n));
|
||||
const counter = new BlockUnitCounter();
|
||||
const countOf = (index: number, text: string): number => counter.count(index, text);
|
||||
const sliceOf = (index: number, text: string, units: number): string => counter.slice(index, text, units);
|
||||
const total = visibleUnits(full, false);
|
||||
const component = new AssistantMessageComponent();
|
||||
const start = Bun.nanoseconds();
|
||||
@@ -160,7 +169,7 @@ try {
|
||||
let steps = 0;
|
||||
while (revealed < total) {
|
||||
revealed = Math.min(total, revealed + nextStep(total - revealed));
|
||||
component.updateContent(buildDisplayMessage(full, revealed, false));
|
||||
component.updateContent(buildDisplayMessage(full, revealed, false, countOf, sliceOf));
|
||||
component.render(WIDTH);
|
||||
steps++;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
/**
|
||||
* Streaming-reveal per-tick compute benchmark.
|
||||
*
|
||||
* Mirrors `StreamingRevealController`'s per-tick work: re-count the visible
|
||||
* units of the target (memoized `BlockUnitCounter.count`) and rebuild the
|
||||
* display message (`buildDisplayMessage`), which slices each text block to the
|
||||
* revealed prefix. One `BlockUnitCounter` is created per episode and shared by
|
||||
* `countOf` + `sliceOf`, exactly as the controller holds one `#unitCounter` per
|
||||
* streaming episode.
|
||||
*
|
||||
* The Markdown render is intentionally excluded: every controller tick passes
|
||||
* `{ transient: true }` to `updateContent`, which disables the L2 cache and code
|
||||
* highlighting, so the dominant per-tick cost is the slice of the growing prefix
|
||||
* (re-segmented from offset 0 by the baseline `sliceGraphemes`). This isolates
|
||||
* exactly that path.
|
||||
*
|
||||
* Metric (lower is better): total wall-clock to fully reveal one representative
|
||||
* large assistant message through the controller's `nextStep` progression,
|
||||
* averaged over episodes. `reveal_ms_per_step` is the same work divided by the
|
||||
* number of reveal ticks.
|
||||
*
|
||||
* Run: bun run packages/coding-agent/bench/streaming-throughput.bench.ts
|
||||
*/
|
||||
import type { AssistantMessage } from "@oh-my-pi/pi-ai";
|
||||
import { BlockUnitCounter, buildDisplayMessage, nextStep } from "../src/modes/controllers/streaming-reveal";
|
||||
|
||||
const HIDE_THINKING = false;
|
||||
const PROSE_ONLY = true;
|
||||
const WARMUP_EPISODES = 6;
|
||||
const MEASURE_EPISODES = 40;
|
||||
|
||||
/** Prose + code + accented text + multibyte grapheme clusters (ZWJ families,
|
||||
* flag sequences, skin-tone modifiers, CJK), representative of LLM output that
|
||||
* makes Intl.Segmenter do real per-cluster work. */
|
||||
const CHUNK = `Here is an overview of the rendering pipeline changes.
|
||||
|
||||
The streaming reveal controller now advances the revealed prefix each tick. Consider the helper:
|
||||
|
||||
\`\`\`ts
|
||||
function sliceToUnits(text: string, units: number): string {
|
||||
\tlet end = 0;
|
||||
\tfor (const { index, segment } of segmenter.segment(text)) {
|
||||
\t\tif (--units < 0) break;
|
||||
\t\tend = index + segment.length;
|
||||
\t}
|
||||
\treturn text.slice(0, end);
|
||||
}
|
||||
\`\`\`
|
||||
|
||||
This handles café, naïve résumés, and emoji clusters like the 👨👩👧👦 family, the 🏳️🌈 flag, and the 👩🏽 skin-tone modifier. CJK text such as 日本語のテスト also segments correctly, and the heart ❤️ beats steadily. Each grapheme cluster is one user-perceived character: "👨👩👧👦" is a single unit, not seven code points. The decomposed sequence e followed by a combining acute accent (e + \\u0301) is likewise a single cluster, distinct from the precomposed form.
|
||||
|
||||
The adaptive step is \`nextStep = max(3, ceil(backlog / 8))\`, so the tick count stays roughly constant while the per-tick slice cost grows with the rendered prefix. Incremental slicing lowers that per-update cost from the prefix length to the per-step delta.
|
||||
|
||||
`;
|
||||
|
||||
function makeMessage(textBlocks: string[]): AssistantMessage {
|
||||
return {
|
||||
role: "assistant",
|
||||
content: textBlocks.map(text => ({ type: "text" as const, text })),
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
model: "mock",
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "stop",
|
||||
timestamp: 0,
|
||||
};
|
||||
}
|
||||
|
||||
/** Total visible graphemes of the target's text blocks via the counter (mirrors
|
||||
* the controller's `#visibleUnits` for text-only blocks). */
|
||||
function textUnits(target: AssistantMessage, counter: BlockUnitCounter): number {
|
||||
let total = 0;
|
||||
for (let i = 0; i < target.content.length; i++) {
|
||||
const block = target.content[i];
|
||||
if (block?.type === "text") total += counter.count(i, block.text);
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
/** Drive one full reveal episode: a fresh counter shared by countOf + sliceOf,
|
||||
* an initial render at revealed = 0 (mirrors `begin`), then the `nextStep`
|
||||
* catch-up loop (mirrors `#tick`). Returns the number of reveal ticks. */
|
||||
function revealEpisode(target: AssistantMessage): number {
|
||||
const counter = new BlockUnitCounter();
|
||||
const countOf = (index: number, text: string): number => counter.count(index, text);
|
||||
const sliceOf = (index: number, text: string, units: number): string => counter.slice(index, text, units);
|
||||
buildDisplayMessage(target, 0, HIDE_THINKING, PROSE_ONLY, countOf, sliceOf);
|
||||
let revealed = 0;
|
||||
let ticks = 0;
|
||||
for (;;) {
|
||||
const total = textUnits(target, counter);
|
||||
if (revealed >= total) break;
|
||||
revealed = Math.min(total, revealed + nextStep(total - revealed));
|
||||
buildDisplayMessage(target, revealed, HIDE_THINKING, PROSE_ONLY, countOf, sliceOf);
|
||||
ticks += 1;
|
||||
}
|
||||
return ticks;
|
||||
}
|
||||
|
||||
// Two text blocks of differing lengths exercise multi-block counter indexing.
|
||||
const target = makeMessage([CHUNK.repeat(16), CHUNK.repeat(12)]);
|
||||
const sizingCounter = new BlockUnitCounter();
|
||||
const graphemes = textUnits(target, sizingCounter);
|
||||
|
||||
for (let episode = 0; episode < WARMUP_EPISODES; episode++) revealEpisode(target);
|
||||
|
||||
let totalTicks = 0;
|
||||
const start = performance.now();
|
||||
for (let episode = 0; episode < MEASURE_EPISODES; episode++) totalTicks += revealEpisode(target);
|
||||
const elapsedMs = performance.now() - start;
|
||||
|
||||
const msPerEpisode = elapsedMs / MEASURE_EPISODES;
|
||||
const msPerStep = elapsedMs / totalTicks;
|
||||
|
||||
console.log(`METRIC reveal_ms_per_episode=${msPerEpisode.toFixed(4)}`);
|
||||
console.log(`METRIC reveal_ms_per_step=${msPerStep.toFixed(5)}`);
|
||||
console.log(`ASI graphemes=${graphemes} episodes=${MEASURE_EPISODES} ticks_per_episode=${(totalTicks / MEASURE_EPISODES).toFixed(2)} warmup=${WARMUP_EPISODES}`);
|
||||
console.log(`(reveal: ${graphemes} graphemes, ${(totalTicks / MEASURE_EPISODES).toFixed(1)} ticks/episode, ${msPerEpisode.toFixed(3)} ms/episode, ${msPerStep.toFixed(4)} ms/step)`);
|
||||
Reference in New Issue
Block a user