refactor: restructured hashline to use file-level hash validation with colon separators
- Replaced per-line hash anchors with file-level hash validation in hashline format, changing anchor syntax from LINE+HASH to bare LINE numbers. - Simplified hashline line separator from pipe (|) to colon (:) and replaced replace operator (->) with colon, added delete operator (!) for explicit line deletion. - Implemented file-read snapshot caching with multi-snapshot ring buffer per path and file-hash-based recovery to detect and recover from stale edits. - Refactored hashline grammar, parser, and execution to support file-level hash binding, anchor-scoped validation, and structural bracket warnings for delete operations. - Updated documentation and test fixtures to reflect new hashline syntax with file hashes, colon separators, and delete operator throughout.
This commit is contained in:
@@ -513,8 +513,12 @@ async function main(): Promise<void> {
|
||||
|
||||
console.log("");
|
||||
console.log("Benchmark complete!");
|
||||
console.log(` Success rate: ${(result.summary.overallSuccessRate * 100).toFixed(1)}%`);
|
||||
console.log(` Total tokens: ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`);
|
||||
console.log(
|
||||
` Task success rate (best of ${config.runsPerTask}): ${(result.summary.taskSuccessRate * 100).toFixed(1)}% (${result.summary.successfulTasks}/${result.summary.totalTasks})`,
|
||||
);
|
||||
console.log(
|
||||
` Total tokens (best): ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`,
|
||||
);
|
||||
if (result.summary.ghostRuns > 0) {
|
||||
console.log(` Ghost runs (0/0/0): ${result.summary.ghostRuns}`);
|
||||
}
|
||||
|
||||
@@ -5,22 +5,28 @@
|
||||
import { formatDuration, formatPercent, truncate } from "@oh-my-pi/pi-utils";
|
||||
import { type BenchmarkResult, EDIT_FAILURE_CATEGORIES, type TaskResult } from "./runner";
|
||||
|
||||
function getStatusEmoji(successRate: number, runsPerTask: number): string {
|
||||
const passing = Math.round(successRate * runsPerTask);
|
||||
if (passing === runsPerTask) return "✅";
|
||||
if (passing === 0) return "❌";
|
||||
return "⚠️";
|
||||
function formatBestStatus(task: TaskResult, runsPerTask: number): { status: string; label: string } {
|
||||
const completed = task.runs.filter(run => !isCompletedGhost(run)).length;
|
||||
const succeeded = task.runs.filter(run => run.success).length;
|
||||
if (task.success) {
|
||||
// best-of-N pass; flag flakiness when not every run succeeded.
|
||||
const flaky = completed > 0 && succeeded < completed;
|
||||
const status = flaky ? "⚠️" : "✅";
|
||||
const label = `PASS (${succeeded}/${completed || runsPerTask})`;
|
||||
return { status, label };
|
||||
}
|
||||
return { status: "❌", label: `FAIL (0/${completed || runsPerTask})` };
|
||||
}
|
||||
|
||||
function isCompletedGhost(run: TaskResult["runs"][number]): boolean {
|
||||
if (run.success) return false;
|
||||
return run.tokens.total === 0 && run.toolCalls.read === 0 && run.toolCalls.edit === 0 && run.toolCalls.write === 0;
|
||||
}
|
||||
|
||||
function formatNumber(n: number): string {
|
||||
return n.toLocaleString();
|
||||
}
|
||||
|
||||
function formatPassRate(successRate: number, runsPerTask: number): string {
|
||||
const passing = Math.round(successRate * runsPerTask);
|
||||
return `${passing}/${runsPerTask}`;
|
||||
}
|
||||
|
||||
function formatRate(numerator: number, denominator: number): string {
|
||||
if (denominator === 0) return "—";
|
||||
const percent = (numerator / denominator) * 100;
|
||||
@@ -82,7 +88,6 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
);
|
||||
const verifiedRuns = nonGhostRuns.filter(run => run.verificationPassed).length;
|
||||
const editToolRuns = nonGhostRuns.filter(run => run.patchApplied).length;
|
||||
const successRuns = nonGhostRuns.filter(run => run.success).length;
|
||||
const totalEditAttempts = nonGhostRuns.reduce((sum, run) => sum + run.toolCalls.edit, 0);
|
||||
const totalEditFailures = nonGhostRuns.reduce((sum, run) => sum + run.toolCalls.editFailures, 0);
|
||||
|
||||
@@ -115,17 +120,21 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
|
||||
lines.push("## Summary");
|
||||
lines.push("");
|
||||
lines.push(
|
||||
"Primary metrics (tokens, duration, tool calls) are aggregated over the **best run** of each task. Diagnostic counts (ghost runs, timeouts, retries, failure categories) span every executed run.",
|
||||
);
|
||||
lines.push("");
|
||||
lines.push("| Metric | Value |");
|
||||
lines.push("|--------|-------|");
|
||||
lines.push(`| Total Tasks | ${summary.totalTasks} |`);
|
||||
lines.push(`| Total Runs | ${summary.totalRuns} |`);
|
||||
lines.push(`| Successful Runs | ${summary.successfulRuns} |`);
|
||||
lines.push(`| **Task Success Rate** | **${formatRate(successRuns, summary.totalRuns)}** |`);
|
||||
lines.push(`| **Task Success Rate** | **${formatRate(summary.successfulTasks, summary.totalTasks)}** |`);
|
||||
if (config.editVariant === "hashline") {
|
||||
lines.push(
|
||||
`| **Autocorrect-Free Success Rate** | **${formatRate(summary.autocorrectFreeSuccessfulRuns, summary.totalRuns)}** |`,
|
||||
`| **Autocorrect-Free Success Rate** | **${formatRate(summary.autocorrectFreeSuccessfulTasks, summary.totalTasks)}** |`,
|
||||
);
|
||||
lines.push(`| Autocorrected Runs | ${formatRate(summary.autocorrectedRuns, summary.totalRuns)} |`);
|
||||
lines.push(`| Autocorrected Best Runs | ${formatRate(summary.autocorrectedBestRuns, summary.totalTasks)} |`);
|
||||
lines.push(`| Edit Autocorrect Rate | ${formatPercent(summary.editAutocorrectRate)} |`);
|
||||
}
|
||||
lines.push(`| Verified Rate | ${formatRate(verifiedRuns, summary.totalRuns)} |`);
|
||||
@@ -149,34 +158,36 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
if (config.editVariant === "patch" || config.editVariant === "hashline") {
|
||||
lines.push(`| Patch Failure Rate | ${formatRate(totalEditFailures, totalEditAttempts)} |`);
|
||||
}
|
||||
lines.push(`| Tasks All Passing | ${summary.tasksWithAllPassing} |`);
|
||||
lines.push(`| Tasks Flaky/Failing | ${summary.tasksWithAnyFailing} |`);
|
||||
lines.push(`| Tasks All Passing | ${summary.consistentlyPassingTasks} |`);
|
||||
lines.push(`| Tasks Flaky/Failing | ${summary.totalTasks - summary.consistentlyPassingTasks} |`);
|
||||
lines.push("");
|
||||
lines.push("### Tool Calls");
|
||||
lines.push("");
|
||||
lines.push("| Tool | Total | Avg/Run |");
|
||||
lines.push("|------|-------|---------|");
|
||||
lines.push(`| Read | ${summary.totalToolCalls.read} | ${summary.avgToolCallsPerRun.read.toFixed(1)} |`);
|
||||
lines.push(`| Edit | ${summary.totalToolCalls.edit} | ${summary.avgToolCallsPerRun.edit.toFixed(1)} |`);
|
||||
lines.push(`| Write | ${summary.totalToolCalls.write} | ${summary.avgToolCallsPerRun.write.toFixed(1)} |`);
|
||||
lines.push("| Tool | Total (best) | Avg/Task |");
|
||||
lines.push("|------|--------------|----------|");
|
||||
lines.push(`| Read | ${summary.totalToolCalls.read} | ${summary.avgToolCallsPerTask.read.toFixed(1)} |`);
|
||||
lines.push(`| Edit | ${summary.totalToolCalls.edit} | ${summary.avgToolCallsPerTask.edit.toFixed(1)} |`);
|
||||
lines.push(`| Write | ${summary.totalToolCalls.write} | ${summary.avgToolCallsPerTask.write.toFixed(1)} |`);
|
||||
lines.push(
|
||||
`| **Tool Input Chars** | ${formatNumber(summary.totalToolCalls.totalInputChars)} | ${formatNumber(Math.round(summary.avgToolCallsPerRun.totalInputChars))} |`,
|
||||
`| **Tool Input Chars** | ${formatNumber(summary.totalToolCalls.totalInputChars)} | ${formatNumber(Math.round(summary.avgToolCallsPerTask.totalInputChars))} |`,
|
||||
);
|
||||
lines.push("");
|
||||
lines.push("### Tokens & Time");
|
||||
lines.push("");
|
||||
lines.push("| Metric | Total | Avg/Run |");
|
||||
lines.push("|--------|-------|---------|");
|
||||
lines.push("| Metric | Total (best) | Avg/Task |");
|
||||
lines.push("|--------|--------------|----------|");
|
||||
lines.push(
|
||||
`| Input Tokens | ${formatNumber(summary.totalTokens.input)} | ${formatNumber(summary.avgTokensPerRun.input)} |`,
|
||||
`| Input Tokens | ${formatNumber(summary.totalTokens.input)} | ${formatNumber(summary.avgTokensPerTask.input)} |`,
|
||||
);
|
||||
lines.push(
|
||||
`| Output Tokens | ${formatNumber(summary.totalTokens.output)} | ${formatNumber(summary.avgTokensPerRun.output)} |`,
|
||||
`| Output Tokens | ${formatNumber(summary.totalTokens.output)} | ${formatNumber(summary.avgTokensPerTask.output)} |`,
|
||||
);
|
||||
lines.push(
|
||||
`| Total Tokens | ${formatNumber(summary.totalTokens.total)} | ${formatNumber(summary.avgTokensPerRun.total)} |`,
|
||||
`| Total Tokens | ${formatNumber(summary.totalTokens.total)} | ${formatNumber(summary.avgTokensPerTask.total)} |`,
|
||||
);
|
||||
lines.push(
|
||||
`| Duration | ${formatDuration(summary.totalDuration)} | ${formatDuration(summary.avgDurationPerTask)} |`,
|
||||
);
|
||||
lines.push(`| Duration | ${formatDuration(summary.totalDuration)} | ${formatDuration(summary.avgDurationPerRun)} |`);
|
||||
lines.push(`| **Avg Indent Score** | — | **${formatScore(summary.avgIndentScore)}** |`);
|
||||
lines.push("");
|
||||
|
||||
@@ -222,12 +233,11 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
lines.push("|------|------|---------|----------|-------|-----------------|------|--------|");
|
||||
|
||||
for (const task of tasks) {
|
||||
const status = getStatusEmoji(task.successRate, runsPerTask);
|
||||
const passRate = formatPassRate(task.successRate, runsPerTask);
|
||||
const { status, label } = formatBestStatus(task, runsPerTask);
|
||||
const editHitRate = formatPercent(task.editSuccessRate);
|
||||
const toolCalls = `${task.avgToolCalls.read.toFixed(0)}/${task.avgToolCalls.edit.toFixed(0)}/${task.avgToolCalls.write.toFixed(0)}`;
|
||||
const toolCalls = `${task.toolCalls.read.toFixed(0)}/${task.toolCalls.edit.toFixed(0)}/${task.toolCalls.write.toFixed(0)}`;
|
||||
lines.push(
|
||||
`| ${escapeMarkdown(task.name)} | ${escapeMarkdown(formatFiles(task.files))} | ${passRate} ${status} | ${editHitRate} | ${toolCalls} | ${formatNumber(task.avgTokens.input)}/${formatNumber(task.avgTokens.output)} | ${formatDuration(task.avgDuration)} | ${formatScore(task.avgIndentScore)} |`,
|
||||
`| ${escapeMarkdown(task.name)} | ${escapeMarkdown(formatFiles(task.files))} | ${label} ${status} | ${editHitRate} | ${toolCalls} | ${formatNumber(task.tokens.input)}/${formatNumber(task.tokens.output)} | ${formatDuration(task.duration)} | ${formatScore(task.indentScore)} |`,
|
||||
);
|
||||
}
|
||||
lines.push("");
|
||||
@@ -279,30 +289,38 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
}
|
||||
}
|
||||
|
||||
const flakyTasks = tasks.filter(t => t.successRate > 0 && t.successRate < 1);
|
||||
const flakyTasks = tasks.filter(task => {
|
||||
if (!task.success) return false;
|
||||
const nonGhost = task.runs.filter(run => !isCompletedGhost(run));
|
||||
return nonGhost.length > 0 && nonGhost.some(run => !run.success);
|
||||
});
|
||||
if (flakyTasks.length > 0) {
|
||||
lines.push("## Flaky Tasks (partial passing)");
|
||||
lines.push("## Flaky Tasks (best passed; some runs failed)");
|
||||
lines.push("");
|
||||
|
||||
for (const task of flakyTasks) {
|
||||
const passing = Math.round(task.successRate * runsPerTask);
|
||||
lines.push(`### ${task.name} (${formatFiles(task.files)}) — ${passing}/${runsPerTask}`);
|
||||
const nonGhost = task.runs.filter(run => !isCompletedGhost(run));
|
||||
const passing = nonGhost.filter(run => run.success).length;
|
||||
const denom = nonGhost.length || runsPerTask;
|
||||
const bestNote = task.bestRunIndex >= 0 ? ` (best: run ${task.bestRunIndex + 1})` : "";
|
||||
lines.push(`### ${task.name} (${formatFiles(task.files)}) — ${passing}/${denom}${bestNote}`);
|
||||
lines.push("");
|
||||
lines.push("| Run | Status | Error | Tokens (in/out) | Time |");
|
||||
lines.push("|-----|--------|-------|-----------------|------|");
|
||||
|
||||
for (const run of task.runs) {
|
||||
const marker = run.runIndex === task.bestRunIndex ? " ★" : "";
|
||||
const status = run.success ? "✅" : "❌";
|
||||
const error = run.error ? truncate(escapeMarkdown(run.error), 50) : "—";
|
||||
lines.push(
|
||||
`| ${run.runIndex + 1} | ${status} | ${error} | ${formatNumber(run.tokens.input)} / ${formatNumber(run.tokens.output)} | ${formatDuration(run.duration)} |`,
|
||||
`| ${run.runIndex + 1}${marker} | ${status} | ${error} | ${formatNumber(run.tokens.input)} / ${formatNumber(run.tokens.output)} | ${formatDuration(run.duration)} |`,
|
||||
);
|
||||
}
|
||||
lines.push("");
|
||||
}
|
||||
}
|
||||
|
||||
const failedTasks = tasks.filter(t => t.successRate === 0);
|
||||
const failedTasks = tasks.filter(task => !task.success);
|
||||
if (failedTasks.length > 0) {
|
||||
lines.push("## Failed Tasks (0% passing)");
|
||||
lines.push("");
|
||||
|
||||
@@ -9,7 +9,7 @@ import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import type { AgentMessage, ResolvedThinkingLevel, ThinkingLevel } from "@oh-my-pi/pi-agent-core";
|
||||
import type { Model } from "@oh-my-pi/pi-ai";
|
||||
import { computeLineHash, formatSessionDumpText, RpcClient } from "@oh-my-pi/pi-coding-agent";
|
||||
import { computeFileHash, formatSessionDumpText, RpcClient } from "@oh-my-pi/pi-coding-agent";
|
||||
import { prompt } from "@oh-my-pi/pi-utils";
|
||||
import { diffLines } from "diff";
|
||||
import { formatDirectory } from "./formatter";
|
||||
@@ -294,27 +294,30 @@ function buildMutationPreviewAgainstOriginal(original: string, current: string):
|
||||
|
||||
const changes = diffLines(original, current);
|
||||
const preview: string[] = [];
|
||||
let lineNum = 1;
|
||||
let origLineNum = 1;
|
||||
let newLineNum = 1;
|
||||
|
||||
// Hashline diff-preview format: `-LINE:TEXT` for removed (pre-edit line
|
||||
// number), `+LINE:TEXT` for added (post-edit line number). No per-line hash.
|
||||
for (const change of changes) {
|
||||
const lines = splitLines(change.value);
|
||||
if (!change.added && !change.removed) {
|
||||
lineNum += lines.length;
|
||||
origLineNum += lines.length;
|
||||
newLineNum += lines.length;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (change.removed) {
|
||||
for (const line of lines) {
|
||||
const hash = computeLineHash(lineNum, line);
|
||||
preview.push(`${lineNum}#${hash}|-${line}`);
|
||||
lineNum += 1;
|
||||
preview.push(`-${origLineNum}:${line}`);
|
||||
origLineNum += 1;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
for (const line of lines) {
|
||||
const hash = computeLineHash(lineNum, line);
|
||||
preview.push(`${lineNum}#${hash}|+${line}`);
|
||||
preview.push(`+${newLineNum}:${line}`);
|
||||
newLineNum += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -524,69 +527,56 @@ async function evaluateMutationIntent(
|
||||
};
|
||||
}
|
||||
|
||||
type GuidedHashlineEdit =
|
||||
| { set: { ref: string; body: string[] } }
|
||||
| { set_range: { beg: string; end: string; body: string[] } }
|
||||
| { insert: { after: string; body: string[] } };
|
||||
|
||||
function buildGuidedHashlineEdits(actual: string, expected: string): GuidedHashlineEdit[] {
|
||||
/**
|
||||
* Build a textual hashline patch (with `¶path#hash` section header) that
|
||||
* transforms `actual` into `expected`. Returns null when no changes are
|
||||
* needed or the diff isn't expressible as straight insert/replace/delete ops.
|
||||
*/
|
||||
function buildGuidedHashlinePatch(file: string, actual: string, expected: string): string | null {
|
||||
const changes = diffLines(actual, expected);
|
||||
const actualLines = actual.split("\n");
|
||||
// File-trailing newline produces a phantom empty last entry that is not a
|
||||
// real line; the hashline grammar's line numbers count real lines only.
|
||||
const fileLineCount =
|
||||
actualLines.length > 0 && actualLines[actualLines.length - 1] === ""
|
||||
? actualLines.length - 1
|
||||
: actualLines.length;
|
||||
|
||||
const ops: string[] = [];
|
||||
let line = 1;
|
||||
let pendingStart = 1;
|
||||
let pendingRemoved: string[] = [];
|
||||
let pendingRemoved = 0;
|
||||
let pendingAdded: string[] = [];
|
||||
const edits: GuidedHashlineEdit[] = [];
|
||||
|
||||
const formatPayload = (body: string[]): string => (body.length === 0 ? "" : `\n${body.join("\n")}`);
|
||||
|
||||
const flush = () => {
|
||||
if (pendingRemoved.length === 0 && pendingAdded.length === 0) {
|
||||
return;
|
||||
}
|
||||
if (pendingRemoved === 0 && pendingAdded.length === 0) return;
|
||||
|
||||
if (pendingRemoved.length === 0) {
|
||||
const insertLine = pendingStart;
|
||||
if (pendingRemoved === 0) {
|
||||
// Pure insertion at `pendingStart` (line numbers are 1-indexed and
|
||||
// refer to the pre-edit file).
|
||||
if (pendingAdded.length === 0) return;
|
||||
if (insertLine === 1) {
|
||||
const firstLine = actualLines[0] ?? "";
|
||||
const firstRef = `1#${computeLineHash(1, firstLine)}`;
|
||||
edits.push({
|
||||
set: { ref: firstRef, body: [...pendingAdded, firstLine] },
|
||||
});
|
||||
} else if (insertLine <= actualLines.length) {
|
||||
const afterLine = actualLines[insertLine - 2] ?? "";
|
||||
const afterRef = `${insertLine - 1}#${computeLineHash(insertLine - 1, afterLine)}`;
|
||||
edits.push({
|
||||
insert: { after: afterRef, body: [...pendingAdded] },
|
||||
});
|
||||
} else if (insertLine === actualLines.length + 1 && actualLines.length > 0) {
|
||||
const afterLine = actualLines[actualLines.length - 1] ?? "";
|
||||
const afterRef = `${actualLines.length}#${computeLineHash(actualLines.length, afterLine)}`;
|
||||
edits.push({
|
||||
insert: { after: afterRef, body: [...pendingAdded] },
|
||||
});
|
||||
if (pendingStart <= 1) {
|
||||
ops.push(`BOF↓${formatPayload(pendingAdded)}`);
|
||||
} else if (pendingStart > fileLineCount) {
|
||||
ops.push(`EOF↓${formatPayload(pendingAdded)}`);
|
||||
} else {
|
||||
// Insert above `pendingStart` so the new content lands at that line.
|
||||
ops.push(`${pendingStart}↑${formatPayload(pendingAdded)}`);
|
||||
}
|
||||
} else {
|
||||
const startLine = pendingStart;
|
||||
const endLine = pendingStart + pendingRemoved.length - 1;
|
||||
const startContent = actualLines[startLine - 1] ?? "";
|
||||
const startRef = `${startLine}#${computeLineHash(startLine, startContent)}`;
|
||||
if (startLine === endLine) {
|
||||
edits.push({ set: { ref: startRef, body: [...pendingAdded] } });
|
||||
const endLine = pendingStart + pendingRemoved - 1;
|
||||
const anchor = startLine === endLine ? `${startLine}` : `${startLine}-${endLine}`;
|
||||
if (pendingAdded.length === 0) {
|
||||
ops.push(`${anchor}!`);
|
||||
} else {
|
||||
const endContent = actualLines[endLine - 1] ?? "";
|
||||
const endRef = `${endLine}#${computeLineHash(endLine, endContent)}`;
|
||||
edits.push({
|
||||
set_range: {
|
||||
beg: startRef,
|
||||
end: endRef,
|
||||
body: [...pendingAdded],
|
||||
},
|
||||
});
|
||||
ops.push(`${anchor}:${formatPayload(pendingAdded)}`);
|
||||
}
|
||||
}
|
||||
|
||||
pendingRemoved = [];
|
||||
pendingRemoved = 0;
|
||||
pendingAdded = [];
|
||||
};
|
||||
|
||||
@@ -595,13 +585,14 @@ function buildGuidedHashlineEdits(actual: string, expected: string): GuidedHashl
|
||||
if (!change.added && !change.removed) {
|
||||
flush();
|
||||
line += lines.length;
|
||||
pendingStart = line;
|
||||
continue;
|
||||
}
|
||||
if (pendingRemoved.length === 0 && pendingAdded.length === 0) {
|
||||
if (pendingRemoved === 0 && pendingAdded.length === 0) {
|
||||
pendingStart = line;
|
||||
}
|
||||
if (change.removed) {
|
||||
pendingRemoved.push(...lines);
|
||||
pendingRemoved += lines.length;
|
||||
line += lines.length;
|
||||
}
|
||||
if (change.added) {
|
||||
@@ -610,7 +601,9 @@ function buildGuidedHashlineEdits(actual: string, expected: string): GuidedHashl
|
||||
}
|
||||
flush();
|
||||
|
||||
return edits;
|
||||
if (ops.length === 0) return null;
|
||||
const header = `¶${file}#${computeFileHash(actual)}`;
|
||||
return `${header}\n${ops.join("\n")}`;
|
||||
}
|
||||
|
||||
async function buildGuidedContext(
|
||||
@@ -635,11 +628,13 @@ async function buildGuidedContext(
|
||||
.catch(() => null);
|
||||
if (actual === null || expected === null) return null;
|
||||
|
||||
const edits = buildGuidedHashlineEdits(actual, expected);
|
||||
if (edits.length === 0) return null;
|
||||
if (edits.length > 25) return null;
|
||||
const patch = buildGuidedHashlinePatch(file, actual, expected);
|
||||
if (patch === null) return null;
|
||||
// Rough complexity guard: too many ops or too long → skip guidance.
|
||||
const opCount = patch.split("\n").filter(l => /[↑↓→]/.test(l)).length;
|
||||
if (opCount === 0 || opCount > 25) return null;
|
||||
|
||||
const args = { path: file, edits };
|
||||
const args = { path: file, input: patch };
|
||||
const argsText = JSON.stringify(args, null, 2);
|
||||
if (argsText.length > 20_000) return null;
|
||||
const metaParts: string[] = [];
|
||||
@@ -836,46 +831,78 @@ export interface TaskResult {
|
||||
name: string;
|
||||
files: string[];
|
||||
runs: TaskRunResult[];
|
||||
successRate: number;
|
||||
avgTokens: TokenStats;
|
||||
avgDuration: number;
|
||||
avgIndentScore: number;
|
||||
avgToolCalls: ToolCallStats;
|
||||
/** Index into `runs` (ordered by runIndex) of the selected best run; -1 if no runs completed. */
|
||||
bestRunIndex: number;
|
||||
/** True when the selected best run succeeded. */
|
||||
success: boolean;
|
||||
/** Token usage of the best run. */
|
||||
tokens: TokenStats;
|
||||
/** Duration (ms) of the best run. */
|
||||
duration: number;
|
||||
/** Indent score of the best run, or 0 if unscored. */
|
||||
indentScore: number;
|
||||
/** Tool call stats of the best run. */
|
||||
toolCalls: ToolCallStats;
|
||||
/** Edit-tool success rate of the best run (defaults to 1 when no edit attempts). */
|
||||
editSuccessRate: number;
|
||||
autocorrectFreeSuccessRate: number;
|
||||
/** True if the best run succeeded with zero autocorrects. */
|
||||
autocorrectFreeSuccess: boolean;
|
||||
/** Fraction of completed (non-ghost) runs that succeeded — flakiness indicator. */
|
||||
flakeSuccessRate: number;
|
||||
}
|
||||
|
||||
export interface BenchmarkSummary {
|
||||
totalTasks: number;
|
||||
/** Total completed runs across all tasks (excludes ghost runs). */
|
||||
totalRuns: number;
|
||||
/** Successful runs across every executed run (any of N). Diagnostic. */
|
||||
successfulRuns: number;
|
||||
overallSuccessRate: number;
|
||||
tasksWithAllPassing: number;
|
||||
tasksWithAnyFailing: number;
|
||||
/** Tasks whose best run succeeded (best-of-N). Primary headline metric. */
|
||||
successfulTasks: number;
|
||||
/** successfulTasks / totalTasks. */
|
||||
taskSuccessRate: number;
|
||||
/** Tasks where best succeeded but at least one of N failed (flakiness). */
|
||||
flakyTasks: number;
|
||||
/** Tasks where every executed non-ghost run succeeded. */
|
||||
consistentlyPassingTasks: number;
|
||||
/** Tokens summed over the best run of each task. */
|
||||
totalTokens: TokenStats;
|
||||
avgTokensPerRun: TokenStats;
|
||||
/** Average tokens per task (sum of best runs / number of tasks). */
|
||||
avgTokensPerTask: TokenStats;
|
||||
/** Duration summed over best runs. */
|
||||
totalDuration: number;
|
||||
avgDurationPerRun: number;
|
||||
/** Average duration of the best run per task. */
|
||||
avgDurationPerTask: number;
|
||||
/** Average indent score over best runs (only counts runs with a score). */
|
||||
avgIndentScore: number;
|
||||
/** Tool calls summed over best runs. */
|
||||
totalToolCalls: ToolCallStats;
|
||||
avgToolCallsPerRun: ToolCallStats;
|
||||
/** Average tool calls per task (sum of best runs / number of tasks). */
|
||||
avgToolCallsPerTask: ToolCallStats;
|
||||
/** Edit-tool success rate aggregated across best runs. */
|
||||
editSuccessRate: number;
|
||||
autocorrectFreeSuccessfulRuns: number;
|
||||
/** Tasks where the best run succeeded without any autocorrects. */
|
||||
autocorrectFreeSuccessfulTasks: number;
|
||||
/** autocorrectFreeSuccessfulTasks / totalTasks. */
|
||||
autocorrectFreeSuccessRate: number;
|
||||
autocorrectedRuns: number;
|
||||
/** Best runs with any autocorrects. */
|
||||
autocorrectedBestRuns: number;
|
||||
/** Autocorrect rate across best-run edit successes. */
|
||||
editAutocorrectRate: number;
|
||||
/** Diagnostic: runs (across all N) that timed out. */
|
||||
timeoutRuns: number;
|
||||
/** Total retry counts across all runs */
|
||||
/** Diagnostic: total retry counts across all runs. */
|
||||
totalTimeoutRetries: number;
|
||||
totalZeroToolRetries: number;
|
||||
totalProviderFailureRetries: number;
|
||||
/** Runs where the 0/0/0 ghost signature was detected (0 tokens, 0 tool calls) */
|
||||
/** Diagnostic: ghost runs (0 tokens, 0 tool calls) across all N. */
|
||||
ghostRuns: number;
|
||||
/** Runs excluded because provider/transport stalls exhausted retries (subset of ghostRuns when error matches). */
|
||||
/** Diagnostic: runs excluded because provider/transport stalls exhausted retries. */
|
||||
transportFailureRuns: number;
|
||||
mutationIntentMatchRate?: number;
|
||||
/** Edit failure categories across all runs. */
|
||||
editFailureCategories: Record<EditFailureCategory, number>;
|
||||
/** Hashline edit subtype totals — only when editVariant is hashline */
|
||||
/** Hashline edit subtype totals across all runs — only when editVariant is hashline. */
|
||||
hashlineEditSubtypes?: Record<string, number>;
|
||||
}
|
||||
|
||||
@@ -1629,70 +1656,71 @@ function isGhostRun(r: TaskRunResult): boolean {
|
||||
return noProgress || isTransportFailure(r);
|
||||
}
|
||||
|
||||
const EMPTY_TOOL_CALL_STATS: ToolCallStats = {
|
||||
read: 0,
|
||||
edit: 0,
|
||||
write: 0,
|
||||
editSuccesses: 0,
|
||||
editFailures: 0,
|
||||
editWarnings: 0,
|
||||
editAutocorrects: 0,
|
||||
totalInputChars: 0,
|
||||
};
|
||||
|
||||
/**
|
||||
* Strict ordering used to pick the "best" run for a task:
|
||||
* 1. Successful runs win over failed runs.
|
||||
* 2. Then prefer non-ghost runs (real work over 0/0/0 stalls).
|
||||
* 3. Then prefer the run with lower total token usage.
|
||||
* 4. Then prefer the earlier runIndex for stability.
|
||||
*/
|
||||
function isBetterRun(a: TaskRunResult, b: TaskRunResult): boolean {
|
||||
if (a.success !== b.success) return a.success;
|
||||
const aGhost = isGhostRun(a);
|
||||
const bGhost = isGhostRun(b);
|
||||
if (aGhost !== bGhost) return !aGhost;
|
||||
if (a.tokens.total !== b.tokens.total) return a.tokens.total < b.tokens.total;
|
||||
return a.runIndex < b.runIndex;
|
||||
}
|
||||
|
||||
function pickBestRunIndex(orderedRuns: TaskRunResult[]): number {
|
||||
if (orderedRuns.length === 0) return -1;
|
||||
let bestIdx = 0;
|
||||
for (let i = 1; i < orderedRuns.length; i++) {
|
||||
if (isBetterRun(orderedRuns[i]!, orderedRuns[bestIdx]!)) bestIdx = i;
|
||||
}
|
||||
return bestIdx;
|
||||
}
|
||||
|
||||
function summarizeTaskRuns(task: EditTask, runs: TaskRunResult[]): TaskResult {
|
||||
const orderedRuns = runs.slice().sort((a, b) => a.runIndex - b.runIndex);
|
||||
const nonGhostRuns = orderedRuns.filter(r => !isGhostRun(r));
|
||||
const effective = nonGhostRuns.length;
|
||||
const successfulRuns = orderedRuns.filter(r => r.success).length;
|
||||
const successRate = effective > 0 ? successfulRuns / effective : 0;
|
||||
const successfulNonGhost = nonGhostRuns.filter(r => r.success).length;
|
||||
const flakeSuccessRate = nonGhostRuns.length > 0 ? successfulNonGhost / nonGhostRuns.length : 0;
|
||||
const bestIdx = pickBestRunIndex(orderedRuns);
|
||||
const best = bestIdx === -1 ? undefined : orderedRuns[bestIdx]!;
|
||||
|
||||
const avgTokens: TokenStats =
|
||||
effective > 0
|
||||
? {
|
||||
input: Math.round(nonGhostRuns.reduce((sum, r) => sum + r.tokens.input, 0) / effective),
|
||||
output: Math.round(nonGhostRuns.reduce((sum, r) => sum + r.tokens.output, 0) / effective),
|
||||
total: Math.round(nonGhostRuns.reduce((sum, r) => sum + r.tokens.total, 0) / effective),
|
||||
}
|
||||
: { input: 0, output: 0, total: 0 };
|
||||
|
||||
const avgDuration = effective > 0 ? Math.round(nonGhostRuns.reduce((sum, r) => sum + r.duration, 0) / effective) : 0;
|
||||
const indentScores = orderedRuns
|
||||
.map(run => run.indentScore)
|
||||
.filter((score): score is number => typeof score === "number");
|
||||
const avgIndentScore =
|
||||
indentScores.length > 0 ? indentScores.reduce((sum, score) => sum + score, 0) / indentScores.length : 0;
|
||||
|
||||
const avgToolCalls: ToolCallStats =
|
||||
effective > 0
|
||||
? {
|
||||
read: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.read, 0) / effective,
|
||||
edit: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.edit, 0) / effective,
|
||||
write: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.write, 0) / effective,
|
||||
editSuccesses: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.editSuccesses, 0) / effective,
|
||||
editFailures: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.editFailures, 0) / effective,
|
||||
editWarnings: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.editWarnings, 0) / effective,
|
||||
editAutocorrects: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.editAutocorrects, 0) / effective,
|
||||
totalInputChars: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.totalInputChars, 0) / effective,
|
||||
}
|
||||
: {
|
||||
read: 0,
|
||||
edit: 0,
|
||||
write: 0,
|
||||
editSuccesses: 0,
|
||||
editFailures: 0,
|
||||
editWarnings: 0,
|
||||
editAutocorrects: 0,
|
||||
totalInputChars: 0,
|
||||
};
|
||||
|
||||
const totalEditAttempts = nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.edit, 0);
|
||||
const totalEditSuccesses = nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.editSuccesses, 0);
|
||||
const editSuccessRate = totalEditAttempts > 0 ? totalEditSuccesses / totalEditAttempts : 1;
|
||||
const autocorrectFreeSuccesses = nonGhostRuns.filter(run => run.success && run.editAutocorrectCount === 0).length;
|
||||
const autocorrectFreeSuccessRate = effective > 0 ? autocorrectFreeSuccesses / effective : 0;
|
||||
const tokens: TokenStats = best ? { ...best.tokens } : { input: 0, output: 0, total: 0 };
|
||||
const duration = best?.duration ?? 0;
|
||||
const indentScore = typeof best?.indentScore === "number" ? best.indentScore : 0;
|
||||
const toolCalls: ToolCallStats = best ? { ...best.toolCalls } : { ...EMPTY_TOOL_CALL_STATS };
|
||||
const editSuccessRate = toolCalls.edit > 0 ? toolCalls.editSuccesses / toolCalls.edit : 1;
|
||||
const autocorrectFreeSuccess = Boolean(best?.success) && (best?.editAutocorrectCount ?? 0) === 0;
|
||||
|
||||
return {
|
||||
id: task.id,
|
||||
name: task.name,
|
||||
files: task.files,
|
||||
runs: orderedRuns,
|
||||
successRate,
|
||||
avgTokens,
|
||||
avgDuration,
|
||||
avgIndentScore,
|
||||
avgToolCalls,
|
||||
bestRunIndex: best?.runIndex ?? -1,
|
||||
success: Boolean(best?.success),
|
||||
tokens,
|
||||
duration,
|
||||
indentScore,
|
||||
toolCalls,
|
||||
editSuccessRate,
|
||||
autocorrectFreeSuccessRate,
|
||||
autocorrectFreeSuccess,
|
||||
flakeSuccessRate,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -1754,45 +1782,14 @@ export function buildBenchmarkResult(params: {
|
||||
|
||||
const endTime = params.endTime ?? new Date().toISOString();
|
||||
|
||||
// Diagnostic aggregates run over *every* executed run (across all N) so the
|
||||
// report still surfaces ghost/timeout/retry signals.
|
||||
const allRuns = taskResults.flatMap(t => t.runs);
|
||||
const totalRuns = allRuns.length;
|
||||
const ghostRuns = allRuns.filter(r => isGhostRun(r)).length;
|
||||
const transportFailureRuns = allRuns.filter(r => isTransportFailure(r)).length;
|
||||
const effectiveRuns = totalRuns - ghostRuns;
|
||||
const nonGhostRuns = allRuns.filter(r => !isGhostRun(r));
|
||||
const totalRuns = nonGhostRuns.length;
|
||||
const successfulRuns = allRuns.filter(r => r.success).length;
|
||||
|
||||
const totalTokens: TokenStats = {
|
||||
input: nonGhostRuns.reduce((sum, r) => sum + r.tokens.input, 0),
|
||||
output: nonGhostRuns.reduce((sum, r) => sum + r.tokens.output, 0),
|
||||
total: nonGhostRuns.reduce((sum, r) => sum + r.tokens.total, 0),
|
||||
};
|
||||
|
||||
const totalDuration = nonGhostRuns.reduce((sum, r) => sum + r.duration, 0);
|
||||
const indentScores = nonGhostRuns
|
||||
.map(run => run.indentScore)
|
||||
.filter((score): score is number => typeof score === "number");
|
||||
const avgIndentScore =
|
||||
indentScores.length > 0 ? indentScores.reduce((sum, score) => sum + score, 0) / indentScores.length : 0;
|
||||
|
||||
const totalToolCalls: ToolCallStats = {
|
||||
read: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.read, 0),
|
||||
edit: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.edit, 0),
|
||||
write: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.write, 0),
|
||||
editSuccesses: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.editSuccesses, 0),
|
||||
editFailures: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.editFailures, 0),
|
||||
editWarnings: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.editWarnings, 0),
|
||||
editAutocorrects: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.editAutocorrects, 0),
|
||||
totalInputChars: nonGhostRuns.reduce((sum, r) => sum + r.toolCalls.totalInputChars, 0),
|
||||
};
|
||||
|
||||
const editSuccessRate = totalToolCalls.edit > 0 ? totalToolCalls.editSuccesses / totalToolCalls.edit : 1;
|
||||
const autocorrectFreeSuccessfulRuns = nonGhostRuns.filter(
|
||||
run => run.success && run.editAutocorrectCount === 0,
|
||||
).length;
|
||||
const autocorrectedRuns = nonGhostRuns.filter(run => run.editAutocorrectCount > 0).length;
|
||||
const editAutocorrectRate =
|
||||
totalToolCalls.editSuccesses > 0 ? totalToolCalls.editAutocorrects / totalToolCalls.editSuccesses : 0;
|
||||
const timeoutRuns = nonGhostRuns.filter(
|
||||
r => r.error?.includes("Timeout") || r.error?.includes("Timeout exhausted"),
|
||||
).length;
|
||||
@@ -1802,13 +1799,7 @@ export function buildBenchmarkResult(params: {
|
||||
(sum, r) => sum + (r.retryStats?.providerFailureRetries ?? 0),
|
||||
0,
|
||||
);
|
||||
const runsWithMutationIntent = nonGhostRuns.filter(r => typeof r.mutationIntentMatched === "boolean");
|
||||
const mutationIntentMatchRate =
|
||||
runsWithMutationIntent.length > 0
|
||||
? runsWithMutationIntent.filter(r => r.mutationIntentMatched).length / runsWithMutationIntent.length
|
||||
: undefined;
|
||||
const editFailureCategories = countEditFailureCategories(nonGhostRuns);
|
||||
|
||||
const hashlineEditSubtypes: Record<string, number> | undefined =
|
||||
params.config.editVariant === "hashline"
|
||||
? Object.fromEntries(
|
||||
@@ -1816,38 +1807,91 @@ export function buildBenchmarkResult(params: {
|
||||
)
|
||||
: undefined;
|
||||
|
||||
const denom = effectiveRuns || 1;
|
||||
// Primary aggregates run over the *best* run of each completed task.
|
||||
const bestRuns: TaskRunResult[] = [];
|
||||
for (const task of taskResults) {
|
||||
if (task.bestRunIndex < 0) continue;
|
||||
const best = task.runs.find(r => r.runIndex === task.bestRunIndex);
|
||||
if (best) bestRuns.push(best);
|
||||
}
|
||||
const tasksWithBestRun = bestRuns.length;
|
||||
const totalTasks = params.tasks.length;
|
||||
const denom = totalTasks || 1;
|
||||
|
||||
const successfulTasks = taskResults.filter(t => t.success).length;
|
||||
const consistentlyPassingTasks = taskResults.filter(
|
||||
t => t.success && t.runs.filter(r => !isGhostRun(r)).every(r => r.success),
|
||||
).length;
|
||||
const flakyTasks = taskResults.filter(
|
||||
t => t.success && t.runs.filter(r => !isGhostRun(r)).some(r => !r.success),
|
||||
).length;
|
||||
|
||||
const totalTokens: TokenStats = {
|
||||
input: bestRuns.reduce((sum, r) => sum + r.tokens.input, 0),
|
||||
output: bestRuns.reduce((sum, r) => sum + r.tokens.output, 0),
|
||||
total: bestRuns.reduce((sum, r) => sum + r.tokens.total, 0),
|
||||
};
|
||||
const totalDuration = bestRuns.reduce((sum, r) => sum + r.duration, 0);
|
||||
const totalToolCalls: ToolCallStats = {
|
||||
read: bestRuns.reduce((sum, r) => sum + r.toolCalls.read, 0),
|
||||
edit: bestRuns.reduce((sum, r) => sum + r.toolCalls.edit, 0),
|
||||
write: bestRuns.reduce((sum, r) => sum + r.toolCalls.write, 0),
|
||||
editSuccesses: bestRuns.reduce((sum, r) => sum + r.toolCalls.editSuccesses, 0),
|
||||
editFailures: bestRuns.reduce((sum, r) => sum + r.toolCalls.editFailures, 0),
|
||||
editWarnings: bestRuns.reduce((sum, r) => sum + r.toolCalls.editWarnings, 0),
|
||||
editAutocorrects: bestRuns.reduce((sum, r) => sum + r.toolCalls.editAutocorrects, 0),
|
||||
totalInputChars: bestRuns.reduce((sum, r) => sum + r.toolCalls.totalInputChars, 0),
|
||||
};
|
||||
const bestIndentScores = bestRuns
|
||||
.map(r => r.indentScore)
|
||||
.filter((score): score is number => typeof score === "number");
|
||||
const avgIndentScore =
|
||||
bestIndentScores.length > 0 ? bestIndentScores.reduce((sum, s) => sum + s, 0) / bestIndentScores.length : 0;
|
||||
|
||||
const editSuccessRate = totalToolCalls.edit > 0 ? totalToolCalls.editSuccesses / totalToolCalls.edit : 1;
|
||||
const autocorrectFreeSuccessfulTasks = bestRuns.filter(r => r.success && r.editAutocorrectCount === 0).length;
|
||||
const autocorrectedBestRuns = bestRuns.filter(r => r.editAutocorrectCount > 0).length;
|
||||
const editAutocorrectRate =
|
||||
totalToolCalls.editSuccesses > 0 ? totalToolCalls.editAutocorrects / totalToolCalls.editSuccesses : 0;
|
||||
const bestWithMutationIntent = bestRuns.filter(r => typeof r.mutationIntentMatched === "boolean");
|
||||
const mutationIntentMatchRate =
|
||||
bestWithMutationIntent.length > 0
|
||||
? bestWithMutationIntent.filter(r => r.mutationIntentMatched).length / bestWithMutationIntent.length
|
||||
: undefined;
|
||||
|
||||
const taskDenom = tasksWithBestRun || 1;
|
||||
const summary: BenchmarkSummary = {
|
||||
totalTasks: params.tasks.length,
|
||||
totalRuns: effectiveRuns,
|
||||
totalTasks,
|
||||
totalRuns,
|
||||
successfulRuns,
|
||||
overallSuccessRate: successfulRuns / denom,
|
||||
tasksWithAllPassing: taskResults.filter(t => t.successRate === 1).length,
|
||||
tasksWithAnyFailing: taskResults.filter(t => t.successRate < 1).length,
|
||||
successfulTasks,
|
||||
taskSuccessRate: successfulTasks / denom,
|
||||
flakyTasks,
|
||||
consistentlyPassingTasks,
|
||||
totalTokens,
|
||||
avgTokensPerRun: {
|
||||
input: Math.round(totalTokens.input / denom),
|
||||
output: Math.round(totalTokens.output / denom),
|
||||
total: Math.round(totalTokens.total / denom),
|
||||
avgTokensPerTask: {
|
||||
input: Math.round(totalTokens.input / taskDenom),
|
||||
output: Math.round(totalTokens.output / taskDenom),
|
||||
total: Math.round(totalTokens.total / taskDenom),
|
||||
},
|
||||
totalDuration,
|
||||
avgDurationPerRun: Math.round(totalDuration / denom),
|
||||
avgDurationPerTask: Math.round(totalDuration / taskDenom),
|
||||
avgIndentScore,
|
||||
totalToolCalls,
|
||||
avgToolCallsPerRun: {
|
||||
read: totalToolCalls.read / denom,
|
||||
edit: totalToolCalls.edit / denom,
|
||||
write: totalToolCalls.write / denom,
|
||||
editSuccesses: totalToolCalls.editSuccesses / denom,
|
||||
editFailures: totalToolCalls.editFailures / denom,
|
||||
editWarnings: totalToolCalls.editWarnings / denom,
|
||||
editAutocorrects: totalToolCalls.editAutocorrects / denom,
|
||||
totalInputChars: totalToolCalls.totalInputChars / denom,
|
||||
avgToolCallsPerTask: {
|
||||
read: totalToolCalls.read / taskDenom,
|
||||
edit: totalToolCalls.edit / taskDenom,
|
||||
write: totalToolCalls.write / taskDenom,
|
||||
editSuccesses: totalToolCalls.editSuccesses / taskDenom,
|
||||
editFailures: totalToolCalls.editFailures / taskDenom,
|
||||
editWarnings: totalToolCalls.editWarnings / taskDenom,
|
||||
editAutocorrects: totalToolCalls.editAutocorrects / taskDenom,
|
||||
totalInputChars: totalToolCalls.totalInputChars / taskDenom,
|
||||
},
|
||||
editSuccessRate,
|
||||
autocorrectFreeSuccessfulRuns,
|
||||
autocorrectFreeSuccessRate: autocorrectFreeSuccessfulRuns / denom,
|
||||
autocorrectedRuns,
|
||||
autocorrectFreeSuccessfulTasks,
|
||||
autocorrectFreeSuccessRate: autocorrectFreeSuccessfulTasks / denom,
|
||||
autocorrectedBestRuns,
|
||||
editAutocorrectRate,
|
||||
timeoutRuns,
|
||||
totalTimeoutRetries,
|
||||
@@ -1888,29 +1932,43 @@ export async function runBenchmark(
|
||||
: undefined;
|
||||
|
||||
try {
|
||||
const runItems: TaskRunItem[] = tasks.flatMap(task =>
|
||||
Array.from({ length: config.runsPerTask }, (_, runIndex) => ({ task, runIndex })),
|
||||
);
|
||||
|
||||
const pending = shuffle(runItems);
|
||||
const runsPerTask = Math.max(1, Math.floor(config.runsPerTask));
|
||||
const taskQueue = shuffle(tasks.slice());
|
||||
const resultsByTask = new Map<string, TaskRunResult[]>();
|
||||
const concurrency = Math.max(1, Math.floor(config.taskConcurrency));
|
||||
const running: Promise<void>[] = [];
|
||||
|
||||
const runNext = async (): Promise<void> => {
|
||||
const nextItem = pending.shift();
|
||||
if (!nextItem) return;
|
||||
const { task, result } = await runConcurrentBenchmarkRun(nextItem, config, onProgress, shared);
|
||||
const recordResult = (task: EditTask, result: TaskRunResult) => {
|
||||
const list = resultsByTask.get(task.id) ?? [];
|
||||
list.push(result);
|
||||
resultsByTask.set(task.id, list);
|
||||
onResultSnapshot?.(buildBenchmarkResult({ tasks, config, resultsByTask, startTime }));
|
||||
await runNext();
|
||||
};
|
||||
|
||||
const slots = Math.min(concurrency, pending.length);
|
||||
// Each worker takes one task at a time and launches all N runs for that
|
||||
// task concurrently. The best run is chosen later via summarizeTaskRuns;
|
||||
// taskConcurrency caps the number of in-flight tasks (not runs).
|
||||
const runTaskAllRuns = async (task: EditTask): Promise<void> => {
|
||||
const items: TaskRunItem[] = Array.from({ length: runsPerTask }, (_, runIndex) => ({ task, runIndex }));
|
||||
await Promise.all(
|
||||
items.map(async item => {
|
||||
const { result } = await runConcurrentBenchmarkRun(item, config, onProgress, shared);
|
||||
recordResult(task, result);
|
||||
}),
|
||||
);
|
||||
};
|
||||
|
||||
const worker = async (): Promise<void> => {
|
||||
while (true) {
|
||||
const task = taskQueue.shift();
|
||||
if (!task) return;
|
||||
await runTaskAllRuns(task);
|
||||
}
|
||||
};
|
||||
|
||||
const slots = Math.min(concurrency, taskQueue.length);
|
||||
const running: Promise<void>[] = [];
|
||||
for (let i = 0; i < slots; i++) {
|
||||
running.push(runNext());
|
||||
running.push(worker());
|
||||
}
|
||||
|
||||
await Promise.all(running);
|
||||
|
||||
@@ -35,7 +35,7 @@ function createTask(id: string): EditTask {
|
||||
};
|
||||
}
|
||||
|
||||
function createRun(runIndex: number, success: boolean): TaskRunResult {
|
||||
function createRun(runIndex: number, success: boolean, overrides: Partial<TaskRunResult> = {}): TaskRunResult {
|
||||
return {
|
||||
runIndex,
|
||||
success,
|
||||
@@ -56,6 +56,7 @@ function createRun(runIndex: number, success: boolean): TaskRunResult {
|
||||
editFailures: [],
|
||||
editWarnings: [],
|
||||
editAutocorrectCount: 0,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -177,6 +178,99 @@ describe("buildBenchmarkResult", () => {
|
||||
expect(report).toContain("| range-continuation | 1 | 100.0% |");
|
||||
expect(report).toContain("- Category: range-continuation");
|
||||
});
|
||||
|
||||
it("picks the successful run with the lowest tokens as the task best", () => {
|
||||
const task = createTask("best");
|
||||
const losing = createRun(0, false, { tokens: { input: 5, output: 5, total: 10 } });
|
||||
const winning = createRun(1, true, { tokens: { input: 100, output: 50, total: 150 } });
|
||||
const expensive = createRun(2, true, { tokens: { input: 500, output: 250, total: 750 } });
|
||||
const result = buildBenchmarkResult({
|
||||
tasks: [task],
|
||||
config: {
|
||||
provider: "anthropic",
|
||||
model: "claude",
|
||||
runsPerTask: 3,
|
||||
timeout: 1000,
|
||||
taskConcurrency: 1,
|
||||
},
|
||||
resultsByTask: new Map([[task.id, [losing, winning, expensive]]]),
|
||||
startTime: "2026-04-28T00:00:00.000Z",
|
||||
endTime: "2026-04-28T00:00:01.000Z",
|
||||
});
|
||||
|
||||
const taskResult = result.tasks[0]!;
|
||||
expect(taskResult.success).toBe(true);
|
||||
expect(taskResult.bestRunIndex).toBe(1);
|
||||
expect(taskResult.tokens.total).toBe(150);
|
||||
expect(result.summary.successfulTasks).toBe(1);
|
||||
expect(result.summary.successfulRuns).toBe(2);
|
||||
expect(result.summary.totalTokens.total).toBe(150);
|
||||
expect(result.summary.taskSuccessRate).toBe(1);
|
||||
expect(result.summary.flakyTasks).toBe(1);
|
||||
expect(result.summary.consistentlyPassingTasks).toBe(0);
|
||||
});
|
||||
|
||||
it("falls back to the cheapest failure when no run succeeded", () => {
|
||||
const task = createTask("none");
|
||||
const expensiveFail = createRun(0, false, { tokens: { input: 200, output: 100, total: 300 } });
|
||||
const cheapFail = createRun(1, false, { tokens: { input: 20, output: 10, total: 30 } });
|
||||
const result = buildBenchmarkResult({
|
||||
tasks: [task],
|
||||
config: {
|
||||
provider: "anthropic",
|
||||
model: "claude",
|
||||
runsPerTask: 2,
|
||||
timeout: 1000,
|
||||
taskConcurrency: 1,
|
||||
},
|
||||
resultsByTask: new Map([[task.id, [expensiveFail, cheapFail]]]),
|
||||
startTime: "2026-04-28T00:00:00.000Z",
|
||||
endTime: "2026-04-28T00:00:01.000Z",
|
||||
});
|
||||
|
||||
const taskResult = result.tasks[0]!;
|
||||
expect(taskResult.success).toBe(false);
|
||||
expect(taskResult.bestRunIndex).toBe(1);
|
||||
expect(taskResult.tokens.total).toBe(30);
|
||||
expect(result.summary.successfulTasks).toBe(0);
|
||||
expect(result.summary.taskSuccessRate).toBe(0);
|
||||
});
|
||||
|
||||
it("ignores ghost runs when picking the best non-successful run", () => {
|
||||
const task = createTask("ghost");
|
||||
const ghostRun = createRun(0, false, {
|
||||
tokens: { input: 0, output: 0, total: 0 },
|
||||
toolCalls: {
|
||||
read: 0,
|
||||
edit: 0,
|
||||
write: 0,
|
||||
editSuccesses: 0,
|
||||
editFailures: 0,
|
||||
editWarnings: 0,
|
||||
editAutocorrects: 0,
|
||||
totalInputChars: 0,
|
||||
},
|
||||
});
|
||||
const realFailure = createRun(1, false, { tokens: { input: 40, output: 20, total: 60 } });
|
||||
const result = buildBenchmarkResult({
|
||||
tasks: [task],
|
||||
config: {
|
||||
provider: "anthropic",
|
||||
model: "claude",
|
||||
runsPerTask: 2,
|
||||
timeout: 1000,
|
||||
taskConcurrency: 1,
|
||||
},
|
||||
resultsByTask: new Map([[task.id, [ghostRun, realFailure]]]),
|
||||
startTime: "2026-04-28T00:00:00.000Z",
|
||||
endTime: "2026-04-28T00:00:01.000Z",
|
||||
});
|
||||
|
||||
const taskResult = result.tasks[0]!;
|
||||
expect(taskResult.bestRunIndex).toBe(1);
|
||||
expect(taskResult.tokens.total).toBe(60);
|
||||
expect(result.summary.ghostRuns).toBe(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("writeConversationDump", () => {
|
||||
|
||||
Reference in New Issue
Block a user