From 18f3386aed0500d69ee6078acf19fb7ec9df6f9c Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 28 Jun 2026 07:21:31 +0200 Subject: [PATCH] feat(typescript-edit-benchmark): tracked reasoning tokens - Added reasoning token counts to benchmark results and reporting. - Updated session statistics and task summaries to include reasoning metrics. - Updated report generation to display reasoning token breakdown. --- .../typescript-edit-benchmark/src/index.ts | 4 +-- .../typescript-edit-benchmark/src/report.ts | 6 ++++ .../typescript-edit-benchmark/src/runner.ts | 28 ++++++++++++---- .../test/runner.test.ts | 32 +++++++++---------- 4 files changed, 46 insertions(+), 24 deletions(-) diff --git a/packages/typescript-edit-benchmark/src/index.ts b/packages/typescript-edit-benchmark/src/index.ts index 32db2bd2a..fcb0035da 100755 --- a/packages/typescript-edit-benchmark/src/index.ts +++ b/packages/typescript-edit-benchmark/src/index.ts @@ -521,13 +521,13 @@ async function main(): Promise { ` Total tokens (best, overall): ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`, ); console.log( - ` Tokens/task (best, overall): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total}`, + ` Tokens/task (best, overall): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total} reasoning=${result.summary.avgTokensPerTask.reasoning}`, ); console.log( ` Total tokens (one-shot successes): ${result.summary.totalOneShotSuccessTokens.input} in / ${result.summary.totalOneShotSuccessTokens.output} out`, ); console.log( - ` Tokens/task (one-shot successes): mean=${result.summary.avgOneShotSuccessTokensPerTask.total} median=${result.summary.medianOneShotSuccessTokensPerTask.total} p1=${result.summary.p1OneShotSuccessTokensPerTask.total} p99=${result.summary.p99OneShotSuccessTokensPerTask.total}`, + ` Tokens/task (one-shot successes): mean=${result.summary.avgOneShotSuccessTokensPerTask.total} median=${result.summary.medianOneShotSuccessTokensPerTask.total} p1=${result.summary.p1OneShotSuccessTokensPerTask.total} p99=${result.summary.p99OneShotSuccessTokensPerTask.total} reasoning=${result.summary.avgOneShotSuccessTokensPerTask.reasoning}`, ); if (result.summary.ghostRuns > 0) { console.log(` Ghost runs (0/0/0): ${result.summary.ghostRuns}`); diff --git a/packages/typescript-edit-benchmark/src/report.ts b/packages/typescript-edit-benchmark/src/report.ts index 08974a46a..bf0274cd3 100644 --- a/packages/typescript-edit-benchmark/src/report.ts +++ b/packages/typescript-edit-benchmark/src/report.ts @@ -182,6 +182,9 @@ export function generateReport(result: BenchmarkResult): string { lines.push( `| Output Tokens | ${formatNumber(summary.totalTokens.output)} | ${formatNumber(summary.avgTokensPerTask.output)} | ${formatNumber(summary.medianTokensPerTask.output)} | ${formatNumber(summary.p1TokensPerTask.output)} | ${formatNumber(summary.p99TokensPerTask.output)} |`, ); + lines.push( + `| Reasoning Tokens | ${formatNumber(summary.totalTokens.reasoning)} | ${formatNumber(summary.avgTokensPerTask.reasoning)} | ${formatNumber(summary.medianTokensPerTask.reasoning)} | ${formatNumber(summary.p1TokensPerTask.reasoning)} | ${formatNumber(summary.p99TokensPerTask.reasoning)} |`, + ); lines.push( `| Total Tokens | ${formatNumber(summary.totalTokens.total)} | ${formatNumber(summary.avgTokensPerTask.total)} | ${formatNumber(summary.medianTokensPerTask.total)} | ${formatNumber(summary.p1TokensPerTask.total)} | ${formatNumber(summary.p99TokensPerTask.total)} |`, ); @@ -200,6 +203,9 @@ export function generateReport(result: BenchmarkResult): string { lines.push( `| Output Tokens | ${formatNumber(summary.totalOneShotSuccessTokens.output)} | ${formatNumber(summary.avgOneShotSuccessTokensPerTask.output)} | ${formatNumber(summary.medianOneShotSuccessTokensPerTask.output)} | ${formatNumber(summary.p1OneShotSuccessTokensPerTask.output)} | ${formatNumber(summary.p99OneShotSuccessTokensPerTask.output)} |`, ); + lines.push( + `| Reasoning Tokens | ${formatNumber(summary.totalOneShotSuccessTokens.reasoning)} | ${formatNumber(summary.avgOneShotSuccessTokensPerTask.reasoning)} | ${formatNumber(summary.medianOneShotSuccessTokensPerTask.reasoning)} | ${formatNumber(summary.p1OneShotSuccessTokensPerTask.reasoning)} | ${formatNumber(summary.p99OneShotSuccessTokensPerTask.reasoning)} |`, + ); lines.push( `| Total Tokens | ${formatNumber(summary.totalOneShotSuccessTokens.total)} | ${formatNumber(summary.avgOneShotSuccessTokensPerTask.total)} | ${formatNumber(summary.medianOneShotSuccessTokensPerTask.total)} | ${formatNumber(summary.p1OneShotSuccessTokensPerTask.total)} | ${formatNumber(summary.p99OneShotSuccessTokensPerTask.total)} |`, ); diff --git a/packages/typescript-edit-benchmark/src/runner.ts b/packages/typescript-edit-benchmark/src/runner.ts index 902d453cd..e401c5c97 100644 --- a/packages/typescript-edit-benchmark/src/runner.ts +++ b/packages/typescript-edit-benchmark/src/runner.ts @@ -48,7 +48,14 @@ interface BenchmarkClient { prompt(text: string): Promise; followUp(text: string): Promise; getSessionStats(): Promise<{ - tokens: { input: number; output: number; cacheRead: number; cacheWrite: number; total: number }; + tokens: { + input: number; + output: number; + reasoning: number; + cacheRead: number; + cacheWrite: number; + total: number; + }; assistantMessages: number; }>; getLastAssistantText(): Promise; @@ -766,6 +773,7 @@ function buildBenchmarkRpcArgs(config: BenchmarkConfig, multiFile: boolean, prov export interface TokenStats { input: number; output: number; + reasoning: number; total: number; } @@ -1002,7 +1010,7 @@ async function runSingleTask( let indentScore: number | undefined; let formattedEquivalent: boolean | undefined; let diffStats: { linesChanged: number; charsChanged: number } | undefined; - let tokens: TokenStats = { input: 0, output: 0, total: 0 }; + let tokens: TokenStats = { input: 0, output: 0, reasoning: 0, total: 0 }; let agentResponse: string | undefined; let diff: string | undefined; const editFailures: EditFailure[] = []; @@ -1171,6 +1179,7 @@ async function runSingleTask( tokens = { input: tokens.input + attemptTokens.input, output: tokens.output + attemptTokens.output, + reasoning: tokens.reasoning + attemptTokens.reasoning, total: tokens.total + attemptTokens.total, }; await logEvent({ type: "stats", before: statsBefore, after: statsAfter, attempt: attempt + 1 }); @@ -1713,12 +1722,13 @@ function diffTokenStats(before: SessionTokenStats, after: SessionTokenStats, sys const afterPrompt = after.tokens.input + after.tokens.cacheRead + after.tokens.cacheWrite; const input = Math.max(0, afterPrompt - beforePrompt - overhead); const output = Math.max(0, after.tokens.output - before.tokens.output); + const reasoning = Math.max(0, after.tokens.reasoning - before.tokens.reasoning); const total = input + output; - return { input, output, total }; + return { input, output, reasoning, total }; } type SessionTokenStats = { - tokens: { input: number; output: number; cacheRead: number; cacheWrite: number }; + tokens: { input: number; output: number; reasoning: number; cacheRead: number; cacheWrite: number }; assistantMessages: number; }; @@ -1781,7 +1791,7 @@ function summarizeTaskRuns(task: EditTask, runs: TaskRunResult[]): TaskResult { const bestIdx = pickBestRunIndex(orderedRuns); const best = bestIdx === -1 ? undefined : orderedRuns[bestIdx]!; - const tokens: TokenStats = best ? { ...best.tokens } : { input: 0, output: 0, total: 0 }; + const tokens: TokenStats = best ? { ...best.tokens } : { input: 0, output: 0, reasoning: 0, total: 0 }; const duration = best?.duration ?? 0; const indentScore = typeof best?.indentScore === "number" ? best.indentScore : 0; const toolCalls: ToolCallStats = best ? { ...best.toolCalls } : { ...EMPTY_TOOL_CALL_STATS }; @@ -1812,7 +1822,7 @@ function buildFailureResult(item: TaskRunItem, error: string): TaskRunResult { patchApplied: false, verificationPassed: false, error, - tokens: { input: 0, output: 0, total: 0 }, + tokens: { input: 0, output: 0, reasoning: 0, total: 0 }, duration: 0, toolCalls: { read: 0, @@ -1879,10 +1889,12 @@ export interface TokenDistribution { export function summarizeTokenDistribution(runs: readonly TaskRunResult[]): TokenDistribution { const input = runs.map(r => r.tokens.input).sort((a, b) => a - b); const output = runs.map(r => r.tokens.output).sort((a, b) => a - b); + const reasoning = runs.map(r => r.tokens.reasoning).sort((a, b) => a - b); const total = runs.map(r => r.tokens.total).sort((a, b) => a - b); const at = (p: number): TokenStats => ({ input: Math.round(percentile(input, p)), output: Math.round(percentile(output, p)), + reasoning: Math.round(percentile(reasoning, p)), total: Math.round(percentile(total, p)), }); return { median: at(50), p1: at(1), p99: at(99) }; @@ -1946,6 +1958,7 @@ export function buildBenchmarkResult(params: { const totalTokens: TokenStats = { input: bestRuns.reduce((sum, r) => sum + r.tokens.input, 0), output: bestRuns.reduce((sum, r) => sum + r.tokens.output, 0), + reasoning: bestRuns.reduce((sum, r) => sum + r.tokens.reasoning, 0), total: bestRuns.reduce((sum, r) => sum + r.tokens.total, 0), }; const tokenDistribution = summarizeTokenDistribution(bestRuns); @@ -1986,6 +1999,7 @@ export function buildBenchmarkResult(params: { const totalOneShotSuccessTokens: TokenStats = { input: oneShotSuccessRuns.reduce((sum, r) => sum + r.tokens.input, 0), output: oneShotSuccessRuns.reduce((sum, r) => sum + r.tokens.output, 0), + reasoning: oneShotSuccessRuns.reduce((sum, r) => sum + r.tokens.reasoning, 0), total: oneShotSuccessRuns.reduce((sum, r) => sum + r.tokens.total, 0), }; const oneShotTokenDistribution = summarizeTokenDistribution(oneShotSuccessRuns); @@ -1997,6 +2011,7 @@ export function buildBenchmarkResult(params: { avgOneShotSuccessTokensPerTask: { input: Math.round(totalOneShotSuccessTokens.input / oneShotDenom), output: Math.round(totalOneShotSuccessTokens.output / oneShotDenom), + reasoning: Math.round(totalOneShotSuccessTokens.reasoning / oneShotDenom), total: Math.round(totalOneShotSuccessTokens.total / oneShotDenom), }, medianOneShotSuccessTokensPerTask: oneShotTokenDistribution.median, @@ -2013,6 +2028,7 @@ export function buildBenchmarkResult(params: { avgTokensPerTask: { input: Math.round(totalTokens.input / taskDenom), output: Math.round(totalTokens.output / taskDenom), + reasoning: Math.round(totalTokens.reasoning / taskDenom), total: Math.round(totalTokens.total / taskDenom), }, medianTokensPerTask: tokenDistribution.median, diff --git a/packages/typescript-edit-benchmark/test/runner.test.ts b/packages/typescript-edit-benchmark/test/runner.test.ts index 967966e4e..2ff9434ad 100644 --- a/packages/typescript-edit-benchmark/test/runner.test.ts +++ b/packages/typescript-edit-benchmark/test/runner.test.ts @@ -45,7 +45,7 @@ function createRun(runIndex: number, success: boolean, overrides: Partial { it("picks the successful run with the lowest tokens as the task best", () => { const task = createTask("best"); - const losing = createRun(0, false, { tokens: { input: 5, output: 5, total: 10 } }); - const winning = createRun(1, true, { tokens: { input: 100, output: 50, total: 150 } }); - const expensive = createRun(2, true, { tokens: { input: 500, output: 250, total: 750 } }); + const losing = createRun(0, false, { tokens: { input: 5, output: 5, reasoning: 0, total: 10 } }); + const winning = createRun(1, true, { tokens: { input: 100, output: 50, reasoning: 0, total: 150 } }); + const expensive = createRun(2, true, { tokens: { input: 500, output: 250, reasoning: 0, total: 750 } }); const result = buildBenchmarkResult({ tasks: [task], config: { @@ -216,8 +216,8 @@ describe("buildBenchmarkResult", () => { it("falls back to the cheapest failure when no run succeeded", () => { const task = createTask("none"); - const expensiveFail = createRun(0, false, { tokens: { input: 200, output: 100, total: 300 } }); - const cheapFail = createRun(1, false, { tokens: { input: 20, output: 10, total: 30 } }); + const expensiveFail = createRun(0, false, { tokens: { input: 200, output: 100, reasoning: 0, total: 300 } }); + const cheapFail = createRun(1, false, { tokens: { input: 20, output: 10, reasoning: 0, total: 30 } }); const result = buildBenchmarkResult({ tasks: [task], config: { @@ -243,7 +243,7 @@ describe("buildBenchmarkResult", () => { it("ignores ghost runs when picking the best non-successful run", () => { const task = createTask("ghost"); const ghostRun = createRun(0, false, { - tokens: { input: 0, output: 0, total: 0 }, + tokens: { input: 0, output: 0, reasoning: 0, total: 0 }, toolCalls: { read: 0, edit: 0, @@ -255,7 +255,7 @@ describe("buildBenchmarkResult", () => { totalInputChars: 0, }, }); - const realFailure = createRun(1, false, { tokens: { input: 40, output: 20, total: 60 } }); + const realFailure = createRun(1, false, { tokens: { input: 40, output: 20, reasoning: 0, total: 60 } }); const result = buildBenchmarkResult({ tasks: [task], config: { @@ -283,7 +283,7 @@ describe("buildBenchmarkResult", () => { const resultsByTask = new Map( totals.map((total, i) => [ tasks[i]!.id, - [createRun(0, true, { tokens: { input: (i + 1) * 100, output: (i + 1) * 10, total } })], + [createRun(0, true, { tokens: { input: (i + 1) * 100, output: (i + 1) * 10, reasoning: 0, total } })], ]), ); @@ -305,10 +305,10 @@ describe("buildBenchmarkResult", () => { // Mean is unchanged by the new fields: total sum 1650 / 5 tasks = 330. expect(summary.avgTokensPerTask.total).toBe(330); // Median = the middle sample (linear interpolation at rank 2 of [110..550]). - expect(summary.medianTokensPerTask).toEqual({ input: 300, output: 30, total: 330 }); + expect(summary.medianTokensPerTask).toEqual({ input: 300, output: 30, reasoning: 0, total: 330 }); // p1/p99 interpolate near the extremes (ranks 0.04 and 3.96 over 5 samples). - expect(summary.p1TokensPerTask).toEqual({ input: 104, output: 10, total: 114 }); - expect(summary.p99TokensPerTask).toEqual({ input: 496, output: 50, total: 546 }); + expect(summary.p1TokensPerTask).toEqual({ input: 104, output: 10, reasoning: 0, total: 114 }); + expect(summary.p99TokensPerTask).toEqual({ input: 496, output: 50, reasoning: 0, total: 546 }); }); it("separates token stats for successfully one-shot tasks vs overall", () => { @@ -317,15 +317,15 @@ describe("buildBenchmarkResult", () => { // Task 3: Failed on run 0 (200 tokens). const tasks = [createTask("t1"), createTask("t2"), createTask("t3")]; const resultsByTask = new Map([ - ["t1", [createRun(0, true, { tokens: { input: 80, output: 20, total: 100 } })]], + ["t1", [createRun(0, true, { tokens: { input: 80, output: 20, reasoning: 0, total: 100 } })]], [ "t2", [ - createRun(0, false, { tokens: { input: 120, output: 30, total: 150 } }), - createRun(1, true, { tokens: { input: 40, output: 10, total: 50 } }), + createRun(0, false, { tokens: { input: 120, output: 30, reasoning: 0, total: 150 } }), + createRun(1, true, { tokens: { input: 40, output: 10, reasoning: 0, total: 50 } }), ], ], - ["t3", [createRun(0, false, { tokens: { input: 160, output: 40, total: 200 } })]], + ["t3", [createRun(0, false, { tokens: { input: 160, output: 40, reasoning: 0, total: 200 } })]], ]); const result = buildBenchmarkResult({