feat(benchmark): added median, p1, and p99 token distribution stats
- Added `percentile` and `summarizeTokenDistribution` helpers to runner. - Extended `BenchmarkSummary` with `medianTokensPerTask`, `p1TokensPerTask`, and `p99TokensPerTask`. - Updated live progress output and markdown report table to show distribution columns. - Added unit tests covering percentile interpolation and summary fields.
This commit is contained in:
@@ -20,6 +20,7 @@ import {
|
||||
type BenchmarkConfig,
|
||||
type BenchmarkResult,
|
||||
buildBenchmarkResult,
|
||||
percentile,
|
||||
type ProgressEvent,
|
||||
runBenchmark,
|
||||
} from "./runner";
|
||||
@@ -519,6 +520,9 @@ async function main(): Promise<void> {
|
||||
console.log(
|
||||
` Total tokens (best): ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`,
|
||||
);
|
||||
console.log(
|
||||
` Tokens/task (best total): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total}`,
|
||||
);
|
||||
if (result.summary.ghostRuns > 0) {
|
||||
console.log(` Ghost runs (0/0/0): ${result.summary.ghostRuns}`);
|
||||
}
|
||||
@@ -556,6 +560,9 @@ class LiveProgress {
|
||||
#totalEditSuccesses = 0;
|
||||
#totalToolInputChars = 0;
|
||||
#indentScores: number[] = [];
|
||||
#inputTokens: number[] = [];
|
||||
#outputTokens: number[] = [];
|
||||
#totalTokens: number[] = [];
|
||||
#lastLineLength = 0;
|
||||
|
||||
constructor(totalRuns: number, runsPerTask: number) {
|
||||
@@ -581,6 +588,9 @@ class LiveProgress {
|
||||
}
|
||||
this.#totalInput += event.result.tokens.input;
|
||||
this.#totalOutput += event.result.tokens.output;
|
||||
this.#inputTokens.push(event.result.tokens.input);
|
||||
this.#outputTokens.push(event.result.tokens.output);
|
||||
this.#totalTokens.push(event.result.tokens.total);
|
||||
this.#totalDuration += event.result.duration;
|
||||
this.#totalReads += event.result.toolCalls.read;
|
||||
this.#totalEdits += event.result.toolCalls.edit;
|
||||
@@ -665,9 +675,15 @@ class LiveProgress {
|
||||
console.log(` Avg indent score: ${avgIndent.toFixed(2)}`);
|
||||
console.log(` Tool calls: read=${this.#totalReads} edit=${this.#totalEdits} write=${this.#totalWrites}`);
|
||||
console.log(` Tool input chars: ${this.#totalToolInputChars.toLocaleString()}`);
|
||||
console.log(
|
||||
` Avg tokens/task: ${Math.round(this.#totalInput / denom)} in / ${Math.round(this.#totalOutput / denom)} out`,
|
||||
);
|
||||
const fmtTokens = (samples: number[]): string => {
|
||||
if (samples.length === 0) return "mean=0 median=0 p1=0 p99=0";
|
||||
const sorted = [...samples].sort((a, b) => a - b);
|
||||
const mean = Math.round(sorted.reduce((a, b) => a + b, 0) / sorted.length);
|
||||
return `mean=${mean} median=${Math.round(percentile(sorted, 50))} p1=${Math.round(percentile(sorted, 1))} p99=${Math.round(percentile(sorted, 99))}`;
|
||||
};
|
||||
console.log(` Tokens/task in: ${fmtTokens(this.#inputTokens)}`);
|
||||
console.log(` Tokens/task out: ${fmtTokens(this.#outputTokens)}`);
|
||||
console.log(` Tokens/task tot: ${fmtTokens(this.#totalTokens)}`);
|
||||
console.log(` Avg time/task: ${Math.round(this.#totalDuration / denom)}ms`);
|
||||
}
|
||||
|
||||
|
||||
@@ -174,21 +174,21 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
lines.push("");
|
||||
lines.push("### Tokens & Time");
|
||||
lines.push("");
|
||||
lines.push("| Metric | Total (best) | Avg/Task |");
|
||||
lines.push("|--------|--------------|----------|");
|
||||
lines.push("| Metric | Total (best) | Avg/Task | Median | P1 | P99 |");
|
||||
lines.push("|--------|--------------|----------|--------|----|----|");
|
||||
lines.push(
|
||||
`| Input Tokens | ${formatNumber(summary.totalTokens.input)} | ${formatNumber(summary.avgTokensPerTask.input)} |`,
|
||||
`| Input Tokens | ${formatNumber(summary.totalTokens.input)} | ${formatNumber(summary.avgTokensPerTask.input)} | ${formatNumber(summary.medianTokensPerTask.input)} | ${formatNumber(summary.p1TokensPerTask.input)} | ${formatNumber(summary.p99TokensPerTask.input)} |`,
|
||||
);
|
||||
lines.push(
|
||||
`| Output Tokens | ${formatNumber(summary.totalTokens.output)} | ${formatNumber(summary.avgTokensPerTask.output)} |`,
|
||||
`| Output Tokens | ${formatNumber(summary.totalTokens.output)} | ${formatNumber(summary.avgTokensPerTask.output)} | ${formatNumber(summary.medianTokensPerTask.output)} | ${formatNumber(summary.p1TokensPerTask.output)} | ${formatNumber(summary.p99TokensPerTask.output)} |`,
|
||||
);
|
||||
lines.push(
|
||||
`| Total Tokens | ${formatNumber(summary.totalTokens.total)} | ${formatNumber(summary.avgTokensPerTask.total)} |`,
|
||||
`| Total Tokens | ${formatNumber(summary.totalTokens.total)} | ${formatNumber(summary.avgTokensPerTask.total)} | ${formatNumber(summary.medianTokensPerTask.total)} | ${formatNumber(summary.p1TokensPerTask.total)} | ${formatNumber(summary.p99TokensPerTask.total)} |`,
|
||||
);
|
||||
lines.push(
|
||||
`| Duration | ${formatDuration(summary.totalDuration)} | ${formatDuration(summary.avgDurationPerTask)} |`,
|
||||
`| Duration | ${formatDuration(summary.totalDuration)} | ${formatDuration(summary.avgDurationPerTask)} | — | — | — |`,
|
||||
);
|
||||
lines.push(`| **Avg Indent Score** | — | **${formatScore(summary.avgIndentScore)}** |`);
|
||||
lines.push(`| **Avg Indent Score** | — | **${formatScore(summary.avgIndentScore)}** | — | — | — |`);
|
||||
lines.push("");
|
||||
|
||||
if (summary.hashlineEditSubtypes) {
|
||||
|
||||
@@ -873,6 +873,12 @@ export interface BenchmarkSummary {
|
||||
totalTokens: TokenStats;
|
||||
/** Average tokens per task (sum of best runs / number of tasks). */
|
||||
avgTokensPerTask: TokenStats;
|
||||
/** Median tokens across best runs (per-task distribution). */
|
||||
medianTokensPerTask: TokenStats;
|
||||
/** 1st-percentile tokens across best runs (per-task distribution). */
|
||||
p1TokensPerTask: TokenStats;
|
||||
/** 99th-percentile tokens across best runs (per-task distribution). */
|
||||
p99TokensPerTask: TokenStats;
|
||||
/** Duration summed over best runs. */
|
||||
totalDuration: number;
|
||||
/** Average duration of the best run per task. */
|
||||
@@ -1775,6 +1781,42 @@ async function runConcurrentBenchmarkRun(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Linear-interpolated percentile (NumPy "linear" / type-7) over an ascending-sorted
|
||||
* sample. `p` is a percentage in [0, 100]. Returns 0 for an empty sample.
|
||||
*/
|
||||
export function percentile(sortedAscending: readonly number[], p: number): number {
|
||||
const n = sortedAscending.length;
|
||||
if (n === 0) return 0;
|
||||
if (n === 1) return sortedAscending[0]!;
|
||||
const rank = (p / 100) * (n - 1);
|
||||
const lo = Math.floor(rank);
|
||||
const loVal = sortedAscending[lo]!;
|
||||
const hi = Math.ceil(rank);
|
||||
if (lo === hi) return loVal;
|
||||
return loVal + (sortedAscending[hi]! - loVal) * (rank - lo);
|
||||
}
|
||||
|
||||
/** Median / 1st / 99th percentile token stats over a set of runs (one sample per run). */
|
||||
export interface TokenDistribution {
|
||||
median: TokenStats;
|
||||
p1: TokenStats;
|
||||
p99: TokenStats;
|
||||
}
|
||||
|
||||
/** Compute the per-run token distribution (median, p1, p99) across the given runs. */
|
||||
export function summarizeTokenDistribution(runs: readonly TaskRunResult[]): TokenDistribution {
|
||||
const input = runs.map(r => r.tokens.input).sort((a, b) => a - b);
|
||||
const output = runs.map(r => r.tokens.output).sort((a, b) => a - b);
|
||||
const total = runs.map(r => r.tokens.total).sort((a, b) => a - b);
|
||||
const at = (p: number): TokenStats => ({
|
||||
input: Math.round(percentile(input, p)),
|
||||
output: Math.round(percentile(output, p)),
|
||||
total: Math.round(percentile(total, p)),
|
||||
});
|
||||
return { median: at(50), p1: at(1), p99: at(99) };
|
||||
}
|
||||
|
||||
export function buildBenchmarkResult(params: {
|
||||
tasks: EditTask[];
|
||||
config: BenchmarkConfig;
|
||||
@@ -1835,6 +1877,7 @@ export function buildBenchmarkResult(params: {
|
||||
output: bestRuns.reduce((sum, r) => sum + r.tokens.output, 0),
|
||||
total: bestRuns.reduce((sum, r) => sum + r.tokens.total, 0),
|
||||
};
|
||||
const tokenDistribution = summarizeTokenDistribution(bestRuns);
|
||||
const totalDuration = bestRuns.reduce((sum, r) => sum + r.duration, 0);
|
||||
const totalToolCalls: ToolCallStats = {
|
||||
read: bestRuns.reduce((sum, r) => sum + r.toolCalls.read, 0),
|
||||
@@ -1878,6 +1921,9 @@ export function buildBenchmarkResult(params: {
|
||||
output: Math.round(totalTokens.output / taskDenom),
|
||||
total: Math.round(totalTokens.total / taskDenom),
|
||||
},
|
||||
medianTokensPerTask: tokenDistribution.median,
|
||||
p1TokensPerTask: tokenDistribution.p1,
|
||||
p99TokensPerTask: tokenDistribution.p99,
|
||||
totalDuration,
|
||||
avgDurationPerTask: Math.round(totalDuration / taskDenom),
|
||||
avgIndentScore,
|
||||
|
||||
@@ -271,6 +271,41 @@ describe("buildBenchmarkResult", () => {
|
||||
expect(taskResult.tokens.total).toBe(60);
|
||||
expect(result.summary.ghostRuns).toBe(1);
|
||||
});
|
||||
|
||||
it("reports median, p1, and p99 token stats across best runs", () => {
|
||||
// Five tasks, each a single successful best run with a distinct token cost.
|
||||
const totals = [110, 220, 330, 440, 550];
|
||||
const tasks = totals.map((_, i) => createTask(`t${i}`));
|
||||
const resultsByTask = new Map(
|
||||
totals.map((total, i) => [
|
||||
tasks[i]!.id,
|
||||
[createRun(0, true, { tokens: { input: (i + 1) * 100, output: (i + 1) * 10, total } })],
|
||||
]),
|
||||
);
|
||||
|
||||
const result = buildBenchmarkResult({
|
||||
tasks,
|
||||
config: {
|
||||
provider: "anthropic",
|
||||
model: "claude",
|
||||
runsPerTask: 1,
|
||||
timeout: 1000,
|
||||
taskConcurrency: 1,
|
||||
},
|
||||
resultsByTask,
|
||||
startTime: "2026-04-28T00:00:00.000Z",
|
||||
endTime: "2026-04-28T00:00:01.000Z",
|
||||
});
|
||||
|
||||
const { summary } = result;
|
||||
// Mean is unchanged by the new fields: total sum 1650 / 5 tasks = 330.
|
||||
expect(summary.avgTokensPerTask.total).toBe(330);
|
||||
// Median = the middle sample (linear interpolation at rank 2 of [110..550]).
|
||||
expect(summary.medianTokensPerTask).toEqual({ input: 300, output: 30, total: 330 });
|
||||
// p1/p99 interpolate near the extremes (ranks 0.04 and 3.96 over 5 samples).
|
||||
expect(summary.p1TokensPerTask).toEqual({ input: 104, output: 10, total: 114 });
|
||||
expect(summary.p99TokensPerTask).toEqual({ input: 496, output: 50, total: 546 });
|
||||
});
|
||||
});
|
||||
|
||||
describe("writeConversationDump", () => {
|
||||
|
||||
Reference in New Issue
Block a user