feat(typescript-edit-benchmark): tracked reasoning tokens
- Added reasoning token counts to benchmark results and reporting. - Updated session statistics and task summaries to include reasoning metrics. - Updated report generation to display reasoning token breakdown.
This commit is contained in:
@@ -521,13 +521,13 @@ async function main(): Promise<void> {
|
||||
` Total tokens (best, overall): ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`,
|
||||
);
|
||||
console.log(
|
||||
` Tokens/task (best, overall): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total}`,
|
||||
` Tokens/task (best, overall): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total} reasoning=${result.summary.avgTokensPerTask.reasoning}`,
|
||||
);
|
||||
console.log(
|
||||
` Total tokens (one-shot successes): ${result.summary.totalOneShotSuccessTokens.input} in / ${result.summary.totalOneShotSuccessTokens.output} out`,
|
||||
);
|
||||
console.log(
|
||||
` Tokens/task (one-shot successes): mean=${result.summary.avgOneShotSuccessTokensPerTask.total} median=${result.summary.medianOneShotSuccessTokensPerTask.total} p1=${result.summary.p1OneShotSuccessTokensPerTask.total} p99=${result.summary.p99OneShotSuccessTokensPerTask.total}`,
|
||||
` Tokens/task (one-shot successes): mean=${result.summary.avgOneShotSuccessTokensPerTask.total} median=${result.summary.medianOneShotSuccessTokensPerTask.total} p1=${result.summary.p1OneShotSuccessTokensPerTask.total} p99=${result.summary.p99OneShotSuccessTokensPerTask.total} reasoning=${result.summary.avgOneShotSuccessTokensPerTask.reasoning}`,
|
||||
);
|
||||
if (result.summary.ghostRuns > 0) {
|
||||
console.log(` Ghost runs (0/0/0): ${result.summary.ghostRuns}`);
|
||||
|
||||
@@ -182,6 +182,9 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
lines.push(
|
||||
`| Output Tokens | ${formatNumber(summary.totalTokens.output)} | ${formatNumber(summary.avgTokensPerTask.output)} | ${formatNumber(summary.medianTokensPerTask.output)} | ${formatNumber(summary.p1TokensPerTask.output)} | ${formatNumber(summary.p99TokensPerTask.output)} |`,
|
||||
);
|
||||
lines.push(
|
||||
`| Reasoning Tokens | ${formatNumber(summary.totalTokens.reasoning)} | ${formatNumber(summary.avgTokensPerTask.reasoning)} | ${formatNumber(summary.medianTokensPerTask.reasoning)} | ${formatNumber(summary.p1TokensPerTask.reasoning)} | ${formatNumber(summary.p99TokensPerTask.reasoning)} |`,
|
||||
);
|
||||
lines.push(
|
||||
`| Total Tokens | ${formatNumber(summary.totalTokens.total)} | ${formatNumber(summary.avgTokensPerTask.total)} | ${formatNumber(summary.medianTokensPerTask.total)} | ${formatNumber(summary.p1TokensPerTask.total)} | ${formatNumber(summary.p99TokensPerTask.total)} |`,
|
||||
);
|
||||
@@ -200,6 +203,9 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
lines.push(
|
||||
`| Output Tokens | ${formatNumber(summary.totalOneShotSuccessTokens.output)} | ${formatNumber(summary.avgOneShotSuccessTokensPerTask.output)} | ${formatNumber(summary.medianOneShotSuccessTokensPerTask.output)} | ${formatNumber(summary.p1OneShotSuccessTokensPerTask.output)} | ${formatNumber(summary.p99OneShotSuccessTokensPerTask.output)} |`,
|
||||
);
|
||||
lines.push(
|
||||
`| Reasoning Tokens | ${formatNumber(summary.totalOneShotSuccessTokens.reasoning)} | ${formatNumber(summary.avgOneShotSuccessTokensPerTask.reasoning)} | ${formatNumber(summary.medianOneShotSuccessTokensPerTask.reasoning)} | ${formatNumber(summary.p1OneShotSuccessTokensPerTask.reasoning)} | ${formatNumber(summary.p99OneShotSuccessTokensPerTask.reasoning)} |`,
|
||||
);
|
||||
lines.push(
|
||||
`| Total Tokens | ${formatNumber(summary.totalOneShotSuccessTokens.total)} | ${formatNumber(summary.avgOneShotSuccessTokensPerTask.total)} | ${formatNumber(summary.medianOneShotSuccessTokensPerTask.total)} | ${formatNumber(summary.p1OneShotSuccessTokensPerTask.total)} | ${formatNumber(summary.p99OneShotSuccessTokensPerTask.total)} |`,
|
||||
);
|
||||
|
||||
@@ -48,7 +48,14 @@ interface BenchmarkClient {
|
||||
prompt(text: string): Promise<void>;
|
||||
followUp(text: string): Promise<void>;
|
||||
getSessionStats(): Promise<{
|
||||
tokens: { input: number; output: number; cacheRead: number; cacheWrite: number; total: number };
|
||||
tokens: {
|
||||
input: number;
|
||||
output: number;
|
||||
reasoning: number;
|
||||
cacheRead: number;
|
||||
cacheWrite: number;
|
||||
total: number;
|
||||
};
|
||||
assistantMessages: number;
|
||||
}>;
|
||||
getLastAssistantText(): Promise<string | null>;
|
||||
@@ -766,6 +773,7 @@ function buildBenchmarkRpcArgs(config: BenchmarkConfig, multiFile: boolean, prov
|
||||
export interface TokenStats {
|
||||
input: number;
|
||||
output: number;
|
||||
reasoning: number;
|
||||
total: number;
|
||||
}
|
||||
|
||||
@@ -1002,7 +1010,7 @@ async function runSingleTask(
|
||||
let indentScore: number | undefined;
|
||||
let formattedEquivalent: boolean | undefined;
|
||||
let diffStats: { linesChanged: number; charsChanged: number } | undefined;
|
||||
let tokens: TokenStats = { input: 0, output: 0, total: 0 };
|
||||
let tokens: TokenStats = { input: 0, output: 0, reasoning: 0, total: 0 };
|
||||
let agentResponse: string | undefined;
|
||||
let diff: string | undefined;
|
||||
const editFailures: EditFailure[] = [];
|
||||
@@ -1171,6 +1179,7 @@ async function runSingleTask(
|
||||
tokens = {
|
||||
input: tokens.input + attemptTokens.input,
|
||||
output: tokens.output + attemptTokens.output,
|
||||
reasoning: tokens.reasoning + attemptTokens.reasoning,
|
||||
total: tokens.total + attemptTokens.total,
|
||||
};
|
||||
await logEvent({ type: "stats", before: statsBefore, after: statsAfter, attempt: attempt + 1 });
|
||||
@@ -1713,12 +1722,13 @@ function diffTokenStats(before: SessionTokenStats, after: SessionTokenStats, sys
|
||||
const afterPrompt = after.tokens.input + after.tokens.cacheRead + after.tokens.cacheWrite;
|
||||
const input = Math.max(0, afterPrompt - beforePrompt - overhead);
|
||||
const output = Math.max(0, after.tokens.output - before.tokens.output);
|
||||
const reasoning = Math.max(0, after.tokens.reasoning - before.tokens.reasoning);
|
||||
const total = input + output;
|
||||
return { input, output, total };
|
||||
return { input, output, reasoning, total };
|
||||
}
|
||||
|
||||
type SessionTokenStats = {
|
||||
tokens: { input: number; output: number; cacheRead: number; cacheWrite: number };
|
||||
tokens: { input: number; output: number; reasoning: number; cacheRead: number; cacheWrite: number };
|
||||
assistantMessages: number;
|
||||
};
|
||||
|
||||
@@ -1781,7 +1791,7 @@ function summarizeTaskRuns(task: EditTask, runs: TaskRunResult[]): TaskResult {
|
||||
const bestIdx = pickBestRunIndex(orderedRuns);
|
||||
const best = bestIdx === -1 ? undefined : orderedRuns[bestIdx]!;
|
||||
|
||||
const tokens: TokenStats = best ? { ...best.tokens } : { input: 0, output: 0, total: 0 };
|
||||
const tokens: TokenStats = best ? { ...best.tokens } : { input: 0, output: 0, reasoning: 0, total: 0 };
|
||||
const duration = best?.duration ?? 0;
|
||||
const indentScore = typeof best?.indentScore === "number" ? best.indentScore : 0;
|
||||
const toolCalls: ToolCallStats = best ? { ...best.toolCalls } : { ...EMPTY_TOOL_CALL_STATS };
|
||||
@@ -1812,7 +1822,7 @@ function buildFailureResult(item: TaskRunItem, error: string): TaskRunResult {
|
||||
patchApplied: false,
|
||||
verificationPassed: false,
|
||||
error,
|
||||
tokens: { input: 0, output: 0, total: 0 },
|
||||
tokens: { input: 0, output: 0, reasoning: 0, total: 0 },
|
||||
duration: 0,
|
||||
toolCalls: {
|
||||
read: 0,
|
||||
@@ -1879,10 +1889,12 @@ export interface TokenDistribution {
|
||||
export function summarizeTokenDistribution(runs: readonly TaskRunResult[]): TokenDistribution {
|
||||
const input = runs.map(r => r.tokens.input).sort((a, b) => a - b);
|
||||
const output = runs.map(r => r.tokens.output).sort((a, b) => a - b);
|
||||
const reasoning = runs.map(r => r.tokens.reasoning).sort((a, b) => a - b);
|
||||
const total = runs.map(r => r.tokens.total).sort((a, b) => a - b);
|
||||
const at = (p: number): TokenStats => ({
|
||||
input: Math.round(percentile(input, p)),
|
||||
output: Math.round(percentile(output, p)),
|
||||
reasoning: Math.round(percentile(reasoning, p)),
|
||||
total: Math.round(percentile(total, p)),
|
||||
});
|
||||
return { median: at(50), p1: at(1), p99: at(99) };
|
||||
@@ -1946,6 +1958,7 @@ export function buildBenchmarkResult(params: {
|
||||
const totalTokens: TokenStats = {
|
||||
input: bestRuns.reduce((sum, r) => sum + r.tokens.input, 0),
|
||||
output: bestRuns.reduce((sum, r) => sum + r.tokens.output, 0),
|
||||
reasoning: bestRuns.reduce((sum, r) => sum + r.tokens.reasoning, 0),
|
||||
total: bestRuns.reduce((sum, r) => sum + r.tokens.total, 0),
|
||||
};
|
||||
const tokenDistribution = summarizeTokenDistribution(bestRuns);
|
||||
@@ -1986,6 +1999,7 @@ export function buildBenchmarkResult(params: {
|
||||
const totalOneShotSuccessTokens: TokenStats = {
|
||||
input: oneShotSuccessRuns.reduce((sum, r) => sum + r.tokens.input, 0),
|
||||
output: oneShotSuccessRuns.reduce((sum, r) => sum + r.tokens.output, 0),
|
||||
reasoning: oneShotSuccessRuns.reduce((sum, r) => sum + r.tokens.reasoning, 0),
|
||||
total: oneShotSuccessRuns.reduce((sum, r) => sum + r.tokens.total, 0),
|
||||
};
|
||||
const oneShotTokenDistribution = summarizeTokenDistribution(oneShotSuccessRuns);
|
||||
@@ -1997,6 +2011,7 @@ export function buildBenchmarkResult(params: {
|
||||
avgOneShotSuccessTokensPerTask: {
|
||||
input: Math.round(totalOneShotSuccessTokens.input / oneShotDenom),
|
||||
output: Math.round(totalOneShotSuccessTokens.output / oneShotDenom),
|
||||
reasoning: Math.round(totalOneShotSuccessTokens.reasoning / oneShotDenom),
|
||||
total: Math.round(totalOneShotSuccessTokens.total / oneShotDenom),
|
||||
},
|
||||
medianOneShotSuccessTokensPerTask: oneShotTokenDistribution.median,
|
||||
@@ -2013,6 +2028,7 @@ export function buildBenchmarkResult(params: {
|
||||
avgTokensPerTask: {
|
||||
input: Math.round(totalTokens.input / taskDenom),
|
||||
output: Math.round(totalTokens.output / taskDenom),
|
||||
reasoning: Math.round(totalTokens.reasoning / taskDenom),
|
||||
total: Math.round(totalTokens.total / taskDenom),
|
||||
},
|
||||
medianTokensPerTask: tokenDistribution.median,
|
||||
|
||||
@@ -45,7 +45,7 @@ function createRun(runIndex: number, success: boolean, overrides: Partial<TaskRu
|
||||
success,
|
||||
patchApplied: success,
|
||||
verificationPassed: success,
|
||||
tokens: { input: 12, output: 8, total: 20 },
|
||||
tokens: { input: 12, output: 8, reasoning: 0, total: 20 },
|
||||
duration: 100,
|
||||
toolCalls: {
|
||||
read: 1,
|
||||
@@ -185,9 +185,9 @@ describe("buildBenchmarkResult", () => {
|
||||
|
||||
it("picks the successful run with the lowest tokens as the task best", () => {
|
||||
const task = createTask("best");
|
||||
const losing = createRun(0, false, { tokens: { input: 5, output: 5, total: 10 } });
|
||||
const winning = createRun(1, true, { tokens: { input: 100, output: 50, total: 150 } });
|
||||
const expensive = createRun(2, true, { tokens: { input: 500, output: 250, total: 750 } });
|
||||
const losing = createRun(0, false, { tokens: { input: 5, output: 5, reasoning: 0, total: 10 } });
|
||||
const winning = createRun(1, true, { tokens: { input: 100, output: 50, reasoning: 0, total: 150 } });
|
||||
const expensive = createRun(2, true, { tokens: { input: 500, output: 250, reasoning: 0, total: 750 } });
|
||||
const result = buildBenchmarkResult({
|
||||
tasks: [task],
|
||||
config: {
|
||||
@@ -216,8 +216,8 @@ describe("buildBenchmarkResult", () => {
|
||||
|
||||
it("falls back to the cheapest failure when no run succeeded", () => {
|
||||
const task = createTask("none");
|
||||
const expensiveFail = createRun(0, false, { tokens: { input: 200, output: 100, total: 300 } });
|
||||
const cheapFail = createRun(1, false, { tokens: { input: 20, output: 10, total: 30 } });
|
||||
const expensiveFail = createRun(0, false, { tokens: { input: 200, output: 100, reasoning: 0, total: 300 } });
|
||||
const cheapFail = createRun(1, false, { tokens: { input: 20, output: 10, reasoning: 0, total: 30 } });
|
||||
const result = buildBenchmarkResult({
|
||||
tasks: [task],
|
||||
config: {
|
||||
@@ -243,7 +243,7 @@ describe("buildBenchmarkResult", () => {
|
||||
it("ignores ghost runs when picking the best non-successful run", () => {
|
||||
const task = createTask("ghost");
|
||||
const ghostRun = createRun(0, false, {
|
||||
tokens: { input: 0, output: 0, total: 0 },
|
||||
tokens: { input: 0, output: 0, reasoning: 0, total: 0 },
|
||||
toolCalls: {
|
||||
read: 0,
|
||||
edit: 0,
|
||||
@@ -255,7 +255,7 @@ describe("buildBenchmarkResult", () => {
|
||||
totalInputChars: 0,
|
||||
},
|
||||
});
|
||||
const realFailure = createRun(1, false, { tokens: { input: 40, output: 20, total: 60 } });
|
||||
const realFailure = createRun(1, false, { tokens: { input: 40, output: 20, reasoning: 0, total: 60 } });
|
||||
const result = buildBenchmarkResult({
|
||||
tasks: [task],
|
||||
config: {
|
||||
@@ -283,7 +283,7 @@ describe("buildBenchmarkResult", () => {
|
||||
const resultsByTask = new Map(
|
||||
totals.map((total, i) => [
|
||||
tasks[i]!.id,
|
||||
[createRun(0, true, { tokens: { input: (i + 1) * 100, output: (i + 1) * 10, total } })],
|
||||
[createRun(0, true, { tokens: { input: (i + 1) * 100, output: (i + 1) * 10, reasoning: 0, total } })],
|
||||
]),
|
||||
);
|
||||
|
||||
@@ -305,10 +305,10 @@ describe("buildBenchmarkResult", () => {
|
||||
// Mean is unchanged by the new fields: total sum 1650 / 5 tasks = 330.
|
||||
expect(summary.avgTokensPerTask.total).toBe(330);
|
||||
// Median = the middle sample (linear interpolation at rank 2 of [110..550]).
|
||||
expect(summary.medianTokensPerTask).toEqual({ input: 300, output: 30, total: 330 });
|
||||
expect(summary.medianTokensPerTask).toEqual({ input: 300, output: 30, reasoning: 0, total: 330 });
|
||||
// p1/p99 interpolate near the extremes (ranks 0.04 and 3.96 over 5 samples).
|
||||
expect(summary.p1TokensPerTask).toEqual({ input: 104, output: 10, total: 114 });
|
||||
expect(summary.p99TokensPerTask).toEqual({ input: 496, output: 50, total: 546 });
|
||||
expect(summary.p1TokensPerTask).toEqual({ input: 104, output: 10, reasoning: 0, total: 114 });
|
||||
expect(summary.p99TokensPerTask).toEqual({ input: 496, output: 50, reasoning: 0, total: 546 });
|
||||
});
|
||||
|
||||
it("separates token stats for successfully one-shot tasks vs overall", () => {
|
||||
@@ -317,15 +317,15 @@ describe("buildBenchmarkResult", () => {
|
||||
// Task 3: Failed on run 0 (200 tokens).
|
||||
const tasks = [createTask("t1"), createTask("t2"), createTask("t3")];
|
||||
const resultsByTask = new Map([
|
||||
["t1", [createRun(0, true, { tokens: { input: 80, output: 20, total: 100 } })]],
|
||||
["t1", [createRun(0, true, { tokens: { input: 80, output: 20, reasoning: 0, total: 100 } })]],
|
||||
[
|
||||
"t2",
|
||||
[
|
||||
createRun(0, false, { tokens: { input: 120, output: 30, total: 150 } }),
|
||||
createRun(1, true, { tokens: { input: 40, output: 10, total: 50 } }),
|
||||
createRun(0, false, { tokens: { input: 120, output: 30, reasoning: 0, total: 150 } }),
|
||||
createRun(1, true, { tokens: { input: 40, output: 10, reasoning: 0, total: 50 } }),
|
||||
],
|
||||
],
|
||||
["t3", [createRun(0, false, { tokens: { input: 160, output: 40, total: 200 } })]],
|
||||
["t3", [createRun(0, false, { tokens: { input: 160, output: 40, reasoning: 0, total: 200 } })]],
|
||||
]);
|
||||
|
||||
const result = buildBenchmarkResult({
|
||||
|
||||
Reference in New Issue
Block a user