From a1ba50b4da4b1e7ffeac43f62bd4a08f7871632f Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 31 May 2026 01:12:49 +0200 Subject: [PATCH] feat(benchmark): added median, p1, and p99 token distribution stats - Added `percentile` and `summarizeTokenDistribution` helpers to runner. - Extended `BenchmarkSummary` with `medianTokensPerTask`, `p1TokensPerTask`, and `p99TokensPerTask`. - Updated live progress output and markdown report table to show distribution columns. - Added unit tests covering percentile interpolation and summary fields. --- .../typescript-edit-benchmark/src/index.ts | 22 +++++++-- .../typescript-edit-benchmark/src/report.ts | 14 +++--- .../typescript-edit-benchmark/src/runner.ts | 46 +++++++++++++++++++ .../test/runner.test.ts | 35 ++++++++++++++ 4 files changed, 107 insertions(+), 10 deletions(-) diff --git a/packages/typescript-edit-benchmark/src/index.ts b/packages/typescript-edit-benchmark/src/index.ts index f2a2052ca..1688b6fd8 100755 --- a/packages/typescript-edit-benchmark/src/index.ts +++ b/packages/typescript-edit-benchmark/src/index.ts @@ -20,6 +20,7 @@ import { type BenchmarkConfig, type BenchmarkResult, buildBenchmarkResult, + percentile, type ProgressEvent, runBenchmark, } from "./runner"; @@ -519,6 +520,9 @@ async function main(): Promise { console.log( ` Total tokens (best): ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`, ); + console.log( + ` Tokens/task (best total): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total}`, + ); if (result.summary.ghostRuns > 0) { console.log(` Ghost runs (0/0/0): ${result.summary.ghostRuns}`); } @@ -556,6 +560,9 @@ class LiveProgress { #totalEditSuccesses = 0; #totalToolInputChars = 0; #indentScores: number[] = []; + #inputTokens: number[] = []; + #outputTokens: number[] = []; + #totalTokens: number[] = []; #lastLineLength = 0; constructor(totalRuns: number, runsPerTask: number) { @@ -581,6 +588,9 @@ class LiveProgress { } this.#totalInput += event.result.tokens.input; this.#totalOutput += event.result.tokens.output; + this.#inputTokens.push(event.result.tokens.input); + this.#outputTokens.push(event.result.tokens.output); + this.#totalTokens.push(event.result.tokens.total); this.#totalDuration += event.result.duration; this.#totalReads += event.result.toolCalls.read; this.#totalEdits += event.result.toolCalls.edit; @@ -665,9 +675,15 @@ class LiveProgress { console.log(` Avg indent score: ${avgIndent.toFixed(2)}`); console.log(` Tool calls: read=${this.#totalReads} edit=${this.#totalEdits} write=${this.#totalWrites}`); console.log(` Tool input chars: ${this.#totalToolInputChars.toLocaleString()}`); - console.log( - ` Avg tokens/task: ${Math.round(this.#totalInput / denom)} in / ${Math.round(this.#totalOutput / denom)} out`, - ); + const fmtTokens = (samples: number[]): string => { + if (samples.length === 0) return "mean=0 median=0 p1=0 p99=0"; + const sorted = [...samples].sort((a, b) => a - b); + const mean = Math.round(sorted.reduce((a, b) => a + b, 0) / sorted.length); + return `mean=${mean} median=${Math.round(percentile(sorted, 50))} p1=${Math.round(percentile(sorted, 1))} p99=${Math.round(percentile(sorted, 99))}`; + }; + console.log(` Tokens/task in: ${fmtTokens(this.#inputTokens)}`); + console.log(` Tokens/task out: ${fmtTokens(this.#outputTokens)}`); + console.log(` Tokens/task tot: ${fmtTokens(this.#totalTokens)}`); console.log(` Avg time/task: ${Math.round(this.#totalDuration / denom)}ms`); } diff --git a/packages/typescript-edit-benchmark/src/report.ts b/packages/typescript-edit-benchmark/src/report.ts index 9e0de939e..4147cad76 100644 --- a/packages/typescript-edit-benchmark/src/report.ts +++ b/packages/typescript-edit-benchmark/src/report.ts @@ -174,21 +174,21 @@ export function generateReport(result: BenchmarkResult): string { lines.push(""); lines.push("### Tokens & Time"); lines.push(""); - lines.push("| Metric | Total (best) | Avg/Task |"); - lines.push("|--------|--------------|----------|"); + lines.push("| Metric | Total (best) | Avg/Task | Median | P1 | P99 |"); + lines.push("|--------|--------------|----------|--------|----|----|"); lines.push( - `| Input Tokens | ${formatNumber(summary.totalTokens.input)} | ${formatNumber(summary.avgTokensPerTask.input)} |`, + `| Input Tokens | ${formatNumber(summary.totalTokens.input)} | ${formatNumber(summary.avgTokensPerTask.input)} | ${formatNumber(summary.medianTokensPerTask.input)} | ${formatNumber(summary.p1TokensPerTask.input)} | ${formatNumber(summary.p99TokensPerTask.input)} |`, ); lines.push( - `| Output Tokens | ${formatNumber(summary.totalTokens.output)} | ${formatNumber(summary.avgTokensPerTask.output)} |`, + `| Output Tokens | ${formatNumber(summary.totalTokens.output)} | ${formatNumber(summary.avgTokensPerTask.output)} | ${formatNumber(summary.medianTokensPerTask.output)} | ${formatNumber(summary.p1TokensPerTask.output)} | ${formatNumber(summary.p99TokensPerTask.output)} |`, ); lines.push( - `| Total Tokens | ${formatNumber(summary.totalTokens.total)} | ${formatNumber(summary.avgTokensPerTask.total)} |`, + `| Total Tokens | ${formatNumber(summary.totalTokens.total)} | ${formatNumber(summary.avgTokensPerTask.total)} | ${formatNumber(summary.medianTokensPerTask.total)} | ${formatNumber(summary.p1TokensPerTask.total)} | ${formatNumber(summary.p99TokensPerTask.total)} |`, ); lines.push( - `| Duration | ${formatDuration(summary.totalDuration)} | ${formatDuration(summary.avgDurationPerTask)} |`, + `| Duration | ${formatDuration(summary.totalDuration)} | ${formatDuration(summary.avgDurationPerTask)} | — | — | — |`, ); - lines.push(`| **Avg Indent Score** | — | **${formatScore(summary.avgIndentScore)}** |`); + lines.push(`| **Avg Indent Score** | — | **${formatScore(summary.avgIndentScore)}** | — | — | — |`); lines.push(""); if (summary.hashlineEditSubtypes) { diff --git a/packages/typescript-edit-benchmark/src/runner.ts b/packages/typescript-edit-benchmark/src/runner.ts index 68c6a753f..857259d9b 100644 --- a/packages/typescript-edit-benchmark/src/runner.ts +++ b/packages/typescript-edit-benchmark/src/runner.ts @@ -873,6 +873,12 @@ export interface BenchmarkSummary { totalTokens: TokenStats; /** Average tokens per task (sum of best runs / number of tasks). */ avgTokensPerTask: TokenStats; + /** Median tokens across best runs (per-task distribution). */ + medianTokensPerTask: TokenStats; + /** 1st-percentile tokens across best runs (per-task distribution). */ + p1TokensPerTask: TokenStats; + /** 99th-percentile tokens across best runs (per-task distribution). */ + p99TokensPerTask: TokenStats; /** Duration summed over best runs. */ totalDuration: number; /** Average duration of the best run per task. */ @@ -1775,6 +1781,42 @@ async function runConcurrentBenchmarkRun( } } +/** + * Linear-interpolated percentile (NumPy "linear" / type-7) over an ascending-sorted + * sample. `p` is a percentage in [0, 100]. Returns 0 for an empty sample. + */ +export function percentile(sortedAscending: readonly number[], p: number): number { + const n = sortedAscending.length; + if (n === 0) return 0; + if (n === 1) return sortedAscending[0]!; + const rank = (p / 100) * (n - 1); + const lo = Math.floor(rank); + const loVal = sortedAscending[lo]!; + const hi = Math.ceil(rank); + if (lo === hi) return loVal; + return loVal + (sortedAscending[hi]! - loVal) * (rank - lo); +} + +/** Median / 1st / 99th percentile token stats over a set of runs (one sample per run). */ +export interface TokenDistribution { + median: TokenStats; + p1: TokenStats; + p99: TokenStats; +} + +/** Compute the per-run token distribution (median, p1, p99) across the given runs. */ +export function summarizeTokenDistribution(runs: readonly TaskRunResult[]): TokenDistribution { + const input = runs.map(r => r.tokens.input).sort((a, b) => a - b); + const output = runs.map(r => r.tokens.output).sort((a, b) => a - b); + const total = runs.map(r => r.tokens.total).sort((a, b) => a - b); + const at = (p: number): TokenStats => ({ + input: Math.round(percentile(input, p)), + output: Math.round(percentile(output, p)), + total: Math.round(percentile(total, p)), + }); + return { median: at(50), p1: at(1), p99: at(99) }; +} + export function buildBenchmarkResult(params: { tasks: EditTask[]; config: BenchmarkConfig; @@ -1835,6 +1877,7 @@ export function buildBenchmarkResult(params: { output: bestRuns.reduce((sum, r) => sum + r.tokens.output, 0), total: bestRuns.reduce((sum, r) => sum + r.tokens.total, 0), }; + const tokenDistribution = summarizeTokenDistribution(bestRuns); const totalDuration = bestRuns.reduce((sum, r) => sum + r.duration, 0); const totalToolCalls: ToolCallStats = { read: bestRuns.reduce((sum, r) => sum + r.toolCalls.read, 0), @@ -1878,6 +1921,9 @@ export function buildBenchmarkResult(params: { output: Math.round(totalTokens.output / taskDenom), total: Math.round(totalTokens.total / taskDenom), }, + medianTokensPerTask: tokenDistribution.median, + p1TokensPerTask: tokenDistribution.p1, + p99TokensPerTask: tokenDistribution.p99, totalDuration, avgDurationPerTask: Math.round(totalDuration / taskDenom), avgIndentScore, diff --git a/packages/typescript-edit-benchmark/test/runner.test.ts b/packages/typescript-edit-benchmark/test/runner.test.ts index 3acce9812..4af14210f 100644 --- a/packages/typescript-edit-benchmark/test/runner.test.ts +++ b/packages/typescript-edit-benchmark/test/runner.test.ts @@ -271,6 +271,41 @@ describe("buildBenchmarkResult", () => { expect(taskResult.tokens.total).toBe(60); expect(result.summary.ghostRuns).toBe(1); }); + + it("reports median, p1, and p99 token stats across best runs", () => { + // Five tasks, each a single successful best run with a distinct token cost. + const totals = [110, 220, 330, 440, 550]; + const tasks = totals.map((_, i) => createTask(`t${i}`)); + const resultsByTask = new Map( + totals.map((total, i) => [ + tasks[i]!.id, + [createRun(0, true, { tokens: { input: (i + 1) * 100, output: (i + 1) * 10, total } })], + ]), + ); + + const result = buildBenchmarkResult({ + tasks, + config: { + provider: "anthropic", + model: "claude", + runsPerTask: 1, + timeout: 1000, + taskConcurrency: 1, + }, + resultsByTask, + startTime: "2026-04-28T00:00:00.000Z", + endTime: "2026-04-28T00:00:01.000Z", + }); + + const { summary } = result; + // Mean is unchanged by the new fields: total sum 1650 / 5 tasks = 330. + expect(summary.avgTokensPerTask.total).toBe(330); + // Median = the middle sample (linear interpolation at rank 2 of [110..550]). + expect(summary.medianTokensPerTask).toEqual({ input: 300, output: 30, total: 330 }); + // p1/p99 interpolate near the extremes (ranks 0.04 and 3.96 over 5 samples). + expect(summary.p1TokensPerTask).toEqual({ input: 104, output: 10, total: 114 }); + expect(summary.p99TokensPerTask).toEqual({ input: 496, output: 50, total: 546 }); + }); }); describe("writeConversationDump", () => {