fix(coding-agent/edit): changed default edit separator to '~' for better reliability
- Switched HL_EDIT_SEP initialization to read from process.env, trimming the value and defaulting to "~" when missing or invalid. - Expanded the separator documentation with benchmark data and reliability notes explaining why "~" was selected as the fallback over other candidates.
This commit is contained in:
@@ -736,12 +736,41 @@ const regexEscape = (str: string): string => str.replace(/[.*+?^${}()|[\]\\]/g,
|
||||
* their appended/prepended text.
|
||||
*
|
||||
* Override at runtime with the `PI_HL_SEP` env var (e.g.
|
||||
* `PI_HL_SEP=">"`, `PI_HL_SEP="\\"`). The value is read once at
|
||||
* module load; the edit grammar, prompt helper, and edit parser derive from it.
|
||||
* `PI_HL_SEP=">"`, `PI_HL_SEP="\\"`). The value is read once at module load;
|
||||
* the edit grammar, prompt helper, and edit parser derive from it.
|
||||
*
|
||||
* Default is `~`, chosen empirically. Benchmark across 8 candidate separators
|
||||
* x 3 models (glm-4.7:nitro, gpt-5.4-nano, claude-sonnet-4-6), 24-48 runs per
|
||||
* cell, hashline variant, 12 sampled tasks per run:
|
||||
*
|
||||
* sep | task ✓ | edit ✓ | patch fail | tok/run
|
||||
* ----|--------|--------|-----------------|--------
|
||||
* + | 70.8% | 78.0% | 27/125 (21.6%) | 32,127
|
||||
* ÷ | 70.7% | 90.6% | 22/211 (10.4%) | 31,666
|
||||
* ~ | 69.4% | 94.9% | 6/107 ( 5.6%) | 30,529 <-- default
|
||||
* > | 69.2% | 91.5% | 21/219 ( 9.6%) | 30,777
|
||||
* : | 66.7% | 86.4% | 20/126 (15.9%) | 33,900
|
||||
* | | 65.9% | 86.9% | 20/127 (15.7%) | 34,589
|
||||
* \ | 65.5% | 89.8% | 16/124 (12.9%) | 36,010
|
||||
* % | 63.9% | 92.8% | 11/125 ( 8.8%) | 36,530
|
||||
*
|
||||
* `~` wins because:
|
||||
* - highest edit-tool success rate (94.9%) of any tested separator
|
||||
* - lowest patch-failure rate (5.6%) — model rarely emits a malformed payload
|
||||
* - cheapest in tokens alongside `>` (no retry overhead from format collisions)
|
||||
* - no line-leading role in any mainstream language, markdown, diff, regex,
|
||||
* or shell, so payload lines are unambiguous to both the parser and models
|
||||
* - task-success is statistically tied with `>` and `÷` (within run-to-run
|
||||
* noise), so the edit-reliability win is free
|
||||
*
|
||||
* `+` and `÷` lead on raw task-success but at the cost of ~2-4x more patch
|
||||
* failures (the model retries until it lands a valid edit). `:`, `|`, `\`
|
||||
* collide with line-leading syntax (label/object-key, body separator, escape)
|
||||
* and degrade both edit reliability and intent-match.
|
||||
*/
|
||||
export const HL_EDIT_SEP = (() => {
|
||||
const sep = Bun.env.PI_HL_SEP;
|
||||
return sep?.length === 1 ? sep : ">";
|
||||
const sep = process.env.PI_HL_SEP?.trim();
|
||||
return sep?.length === 1 ? sep : "~";
|
||||
})();
|
||||
|
||||
/** Regex-escaped form of {@link HL_EDIT_SEP}, safe for regexes. */
|
||||
|
||||
@@ -0,0 +1,191 @@
|
||||
#!/usr/bin/env bun
|
||||
/**
|
||||
* Run `bun run bench:edit --edit-variant hashline` across the cartesian product
|
||||
* of `PI_HL_SEP` separator values and models, with a fixed concurrency cap.
|
||||
*
|
||||
* Each invocation writes its markdown report and a captured stdout/stderr log
|
||||
* into `runs/hashline-sep-<timestamp>/`.
|
||||
*
|
||||
* Usage:
|
||||
* bun scripts/bench-edit-hashline-sep.ts
|
||||
*/
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as path from "node:path";
|
||||
|
||||
const SEPARATORS = ["~", "%", "÷", ">", ":"] as const;
|
||||
|
||||
const MODELS = [
|
||||
"openrouter/z-ai/glm-4.7:nitro",
|
||||
"openai/gpt-5.4-nano",
|
||||
"p-anthropic/claude-sonnet-4-6",
|
||||
] as const;
|
||||
|
||||
const CONCURRENCY = 3;
|
||||
const MAX_TASKS = "12";
|
||||
const VARIANT = "hashline";
|
||||
|
||||
const SEP_SLUGS: Record<string, string> = {
|
||||
"~": "tilde",
|
||||
"%": "pct",
|
||||
"÷": "div",
|
||||
">": "gt",
|
||||
":": "colon",
|
||||
};
|
||||
|
||||
function slugifyModel(model: string): string {
|
||||
return model.replace(/[^a-zA-Z0-9._-]+/g, "_");
|
||||
}
|
||||
|
||||
interface Job {
|
||||
sep: string;
|
||||
sepSlug: string;
|
||||
model: string;
|
||||
output: string;
|
||||
logPath: string;
|
||||
tag: string;
|
||||
}
|
||||
|
||||
interface JobResult {
|
||||
job: Job;
|
||||
exitCode: number | null;
|
||||
durationMs: number;
|
||||
}
|
||||
|
||||
const repoRoot = path.resolve(import.meta.dir, "..");
|
||||
const stamp = new Date().toISOString().replace(/[:.]/g, "-").replace(/Z$/, "Z");
|
||||
const runDir = path.join(repoRoot, "runs", `hashline-sep-${stamp}`);
|
||||
await fs.mkdir(runDir, { recursive: true });
|
||||
|
||||
const jobs: Job[] = [];
|
||||
for (const sep of SEPARATORS) {
|
||||
for (const model of MODELS) {
|
||||
const sepSlug = SEP_SLUGS[sep] ?? `sep_${sep.charCodeAt(0).toString(16)}`;
|
||||
const slug = `${sepSlug}__${slugifyModel(model)}`;
|
||||
jobs.push({
|
||||
sep,
|
||||
sepSlug,
|
||||
model,
|
||||
output: path.join(runDir, `${slug}.md`),
|
||||
logPath: path.join(runDir, `${slug}.log`),
|
||||
tag: `[${sepSlug} ${model}]`,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
console.log(`Total runs: ${jobs.length} concurrency: ${CONCURRENCY}`);
|
||||
console.log(`Output dir: ${runDir}\n`);
|
||||
|
||||
const results: JobResult[] = [];
|
||||
let cursor = 0;
|
||||
let finished = 0;
|
||||
|
||||
async function pipeStream(
|
||||
stream: ReadableStream<Uint8Array>,
|
||||
sink: Bun.FileSink,
|
||||
tag: string,
|
||||
): Promise<void> {
|
||||
const reader = stream.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let buf = "";
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
const text = decoder.decode(value, { stream: true });
|
||||
sink.write(text);
|
||||
buf += text;
|
||||
let nl = buf.indexOf("\n");
|
||||
while (nl !== -1) {
|
||||
const line = buf.slice(0, nl);
|
||||
buf = buf.slice(nl + 1);
|
||||
console.log(`${tag} ${line}`);
|
||||
nl = buf.indexOf("\n");
|
||||
}
|
||||
}
|
||||
const tail = decoder.decode();
|
||||
if (tail) {
|
||||
sink.write(tail);
|
||||
buf += tail;
|
||||
}
|
||||
if (buf) console.log(`${tag} ${buf}`);
|
||||
}
|
||||
|
||||
async function runJob(job: Job): Promise<JobResult> {
|
||||
const started = Date.now();
|
||||
const sink = Bun.file(job.logPath).writer();
|
||||
sink.write(`# cmd: PI_HL_SEP=${JSON.stringify(job.sep)} bun run bench:edit \\\n`);
|
||||
sink.write(
|
||||
`# --edit-variant ${VARIANT} --model ${job.model} --max-tasks ${MAX_TASKS} --output ${job.output}\n\n`,
|
||||
);
|
||||
|
||||
const proc = Bun.spawn({
|
||||
cmd: [
|
||||
"bun",
|
||||
"run",
|
||||
"bench:edit",
|
||||
"--edit-variant",
|
||||
VARIANT,
|
||||
"--model",
|
||||
job.model,
|
||||
"--max-tasks",
|
||||
MAX_TASKS,
|
||||
"--output",
|
||||
job.output,
|
||||
],
|
||||
cwd: repoRoot,
|
||||
env: { ...process.env, PI_HL_SEP: job.sep },
|
||||
stdout: "pipe",
|
||||
stderr: "pipe",
|
||||
});
|
||||
|
||||
await Promise.all([
|
||||
pipeStream(proc.stdout as ReadableStream<Uint8Array>, sink, job.tag),
|
||||
pipeStream(proc.stderr as ReadableStream<Uint8Array>, sink, job.tag),
|
||||
]);
|
||||
const exitCode = await proc.exited;
|
||||
await sink.end();
|
||||
return { job, exitCode, durationMs: Date.now() - started };
|
||||
}
|
||||
|
||||
async function worker(workerId: number): Promise<void> {
|
||||
while (true) {
|
||||
const idx = cursor++;
|
||||
if (idx >= jobs.length) return;
|
||||
const job = jobs[idx];
|
||||
console.log(`${job.tag} starting (worker ${workerId}, ${idx + 1}/${jobs.length})`);
|
||||
const result = await runJob(job);
|
||||
results.push(result);
|
||||
finished++;
|
||||
const status = result.exitCode === 0 ? "ok" : `FAIL exit=${result.exitCode}`;
|
||||
console.log(
|
||||
`${job.tag} ${status} in ${(result.durationMs / 1000).toFixed(1)}s [${finished}/${jobs.length}]`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const wallStart = Date.now();
|
||||
await Promise.all(Array.from({ length: CONCURRENCY }, (_, i) => worker(i + 1)));
|
||||
const wallMs = Date.now() - wallStart;
|
||||
|
||||
console.log("\n=== Summary ===");
|
||||
results.sort(
|
||||
(a, b) =>
|
||||
SEPARATORS.indexOf(a.job.sep as (typeof SEPARATORS)[number]) -
|
||||
SEPARATORS.indexOf(b.job.sep as (typeof SEPARATORS)[number]) ||
|
||||
MODELS.indexOf(a.job.model as (typeof MODELS)[number]) -
|
||||
MODELS.indexOf(b.job.model as (typeof MODELS)[number]),
|
||||
);
|
||||
for (const r of results) {
|
||||
const status = r.exitCode === 0 ? "ok " : `FAIL`;
|
||||
console.log(
|
||||
`${status} ${r.job.sepSlug.padEnd(7)} ${r.job.model.padEnd(40)} ${(r.durationMs / 1000)
|
||||
.toFixed(1)
|
||||
.padStart(6)}s ${path.relative(repoRoot, r.job.output)}`,
|
||||
);
|
||||
}
|
||||
console.log(`\nWall time: ${(wallMs / 1000).toFixed(1)}s`);
|
||||
|
||||
const failures = results.filter(r => r.exitCode !== 0);
|
||||
if (failures.length > 0) {
|
||||
console.log(`${failures.length} job(s) failed`);
|
||||
process.exit(1);
|
||||
}
|
||||
@@ -0,0 +1,463 @@
|
||||
#!/usr/bin/env bun
|
||||
/**
|
||||
* Summarize benchmark runs across (separator × model) by parsing the markdown
|
||||
* reports produced by `bun run bench:edit`.
|
||||
*
|
||||
* Usage:
|
||||
* bun scripts/eval-bench-runs.ts runs/hashline-sep-2026-05-03T06-17-44-702Z
|
||||
* bun scripts/eval-bench-runs.ts <dir> --format md > eval.md
|
||||
* bun scripts/eval-bench-runs.ts <dir> --format csv > eval.csv
|
||||
*/
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as path from "node:path";
|
||||
|
||||
interface ReportRow {
|
||||
file: string;
|
||||
sepSlug: string;
|
||||
model: string;
|
||||
totalTasks: number;
|
||||
totalRuns: number;
|
||||
successfulRuns: number;
|
||||
taskSuccessPct: number;
|
||||
verifiedPct: number;
|
||||
editToolUsagePct: number;
|
||||
editSuccessPct: number;
|
||||
patchFailurePct: number;
|
||||
patchFailures: number;
|
||||
patchAttempts: number;
|
||||
mutationIntentPct: number;
|
||||
autocorrectFreePct: number;
|
||||
tasksAllPassing: number;
|
||||
tasksFlakyFailing: number;
|
||||
timeoutRuns: number;
|
||||
inputTokensTotal: number;
|
||||
outputTokensTotal: number;
|
||||
totalTokens: number;
|
||||
inputTokensAvg: number;
|
||||
outputTokensAvg: number;
|
||||
totalTokensAvg: number;
|
||||
durationTotal: string;
|
||||
durationAvg: string;
|
||||
avgIndentScore: number | null;
|
||||
readTotal: number;
|
||||
editTotal: number;
|
||||
writeTotal: number;
|
||||
}
|
||||
|
||||
const SEPARATOR_DISPLAY: Record<string, string> = {
|
||||
gt: ">",
|
||||
plus: "+",
|
||||
div: "÷",
|
||||
pipe: "|",
|
||||
bslash: "\\",
|
||||
tilde: "~",
|
||||
pct: "%",
|
||||
colon: ":",
|
||||
};
|
||||
|
||||
const args = process.argv.slice(2);
|
||||
const dirs: string[] = [];
|
||||
let format: "table" | "md" | "csv" | "json" = "table";
|
||||
let sortBy: "sep" | "model" | "task" | "edit" | "tokens" = "sep";
|
||||
let aggregate = false;
|
||||
|
||||
for (let i = 0; i < args.length; i++) {
|
||||
const a = args[i];
|
||||
if (a === "--format") {
|
||||
format = args[++i] as typeof format;
|
||||
} else if (a === "--sort") {
|
||||
sortBy = args[++i] as typeof sortBy;
|
||||
} else if (a === "--aggregate") {
|
||||
aggregate = true;
|
||||
} else if (!a.startsWith("--")) {
|
||||
dirs.push(a);
|
||||
}
|
||||
}
|
||||
|
||||
if (dirs.length === 0) {
|
||||
console.error(
|
||||
"usage: bun scripts/eval-bench-runs.ts <runs-dir> [<runs-dir>...] [--aggregate] [--format table|md|csv|json] [--sort sep|model|task|edit|tokens]",
|
||||
);
|
||||
process.exit(2);
|
||||
}
|
||||
|
||||
const resolvedDirs = dirs.map(d => path.resolve(d));
|
||||
|
||||
function parseNumber(text: string): number {
|
||||
return Number.parseFloat(text.replace(/,/g, ""));
|
||||
}
|
||||
|
||||
function getCell(text: string, label: string): string | null {
|
||||
const re = new RegExp(`^\\|\\s*\\*?\\*?${escapeRegex(label)}\\*?\\*?\\s*\\|\\s*\\*?\\*?(.+?)\\*?\\*?\\s*\\|\\s*$`, "m");
|
||||
const m = text.match(re);
|
||||
return m ? m[1].trim() : null;
|
||||
}
|
||||
|
||||
function getRow(text: string, label: string): string[] | null {
|
||||
const re = new RegExp(`^\\|\\s*\\*?\\*?${escapeRegex(label)}\\*?\\*?\\s*\\|(.+)\\|\\s*$`, "m");
|
||||
const m = text.match(re);
|
||||
if (!m) return null;
|
||||
return m[1].split("|").map(s => s.trim().replace(/^\*\*|\*\*$/g, ""));
|
||||
}
|
||||
|
||||
function escapeRegex(s: string): string {
|
||||
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
||||
}
|
||||
|
||||
function parsePercent(value: string | null): number {
|
||||
if (!value) return Number.NaN;
|
||||
const m = value.match(/([0-9]+(?:\.[0-9]+)?)\s*%/);
|
||||
return m ? Number.parseFloat(m[1]) : Number.NaN;
|
||||
}
|
||||
|
||||
function parseRatePair(value: string | null): { numerator: number; denominator: number; pct: number } {
|
||||
if (!value) return { numerator: 0, denominator: 0, pct: Number.NaN };
|
||||
const m = value.match(/([0-9.]+)\s*%\s*\(\s*([0-9,]+)\s*\/\s*([0-9,]+)\s*\)/);
|
||||
if (!m) {
|
||||
return { numerator: 0, denominator: 0, pct: parsePercent(value) };
|
||||
}
|
||||
return {
|
||||
pct: Number.parseFloat(m[1]),
|
||||
numerator: parseNumber(m[2]),
|
||||
denominator: parseNumber(m[3]),
|
||||
};
|
||||
}
|
||||
|
||||
function parseFraction(value: string | null): { num: number; denom: number } {
|
||||
if (!value) return { num: 0, denom: 0 };
|
||||
const m = value.match(/([0-9,]+)\s*\/\s*([0-9,]+)/);
|
||||
return m ? { num: parseNumber(m[1]), denom: parseNumber(m[2]) } : { num: 0, denom: 0 };
|
||||
}
|
||||
|
||||
async function parseReport(file: string): Promise<ReportRow> {
|
||||
const text = await Bun.file(file).text();
|
||||
|
||||
const base = path.basename(file, ".md");
|
||||
const [sepSlug, ...modelParts] = base.split("__");
|
||||
const model = modelParts.join("__").replace(/_/g, "/");
|
||||
|
||||
const totalTasks = Number.parseInt(getCell(text, "Total Tasks") ?? "0", 10);
|
||||
const totalRuns = Number.parseInt(getCell(text, "Total Runs") ?? "0", 10);
|
||||
const successfulRuns = Number.parseInt(getCell(text, "Successful Runs") ?? "0", 10);
|
||||
const taskSuccessPct = parsePercent(getCell(text, "Task Success Rate"));
|
||||
const verifiedPct = parsePercent(getCell(text, "Verified Rate"));
|
||||
const editTool = parseRatePair(getCell(text, "Edit Tool Usage Rate"));
|
||||
const editSuccessPct = parsePercent(getCell(text, "Edit Success Rate"));
|
||||
const patchFailure = parseRatePair(getCell(text, "Patch Failure Rate"));
|
||||
const mutationIntentPct = parsePercent(getCell(text, "Mutation Intent Match Rate"));
|
||||
const autocorrectFreePct = parsePercent(getCell(text, "Autocorrect-Free Success Rate"));
|
||||
const tasksAllPassing = Number.parseInt(getCell(text, "Tasks All Passing") ?? "0", 10);
|
||||
const tasksFlakyFailing = Number.parseInt(getCell(text, "Tasks Flaky/Failing") ?? "0", 10);
|
||||
const timeoutRuns = Number.parseInt(getCell(text, "Timeout Runs") ?? "0", 10);
|
||||
|
||||
const inRow = getRow(text, "Input Tokens") ?? ["0", "0"];
|
||||
const outRow = getRow(text, "Output Tokens") ?? ["0", "0"];
|
||||
const totalRow = getRow(text, "Total Tokens") ?? ["0", "0"];
|
||||
const durationRow = getRow(text, "Duration") ?? ["", ""];
|
||||
const indentRow = getRow(text, "Avg Indent Score") ?? ["—", "—"];
|
||||
|
||||
const readRow = getRow(text, "Read") ?? ["0", "0"];
|
||||
const editRow = getRow(text, "Edit") ?? ["0", "0"];
|
||||
const writeRow = getRow(text, "Write") ?? ["0", "0"];
|
||||
|
||||
const indentValue = indentRow[1]?.replace(/[*\s—-]/g, "");
|
||||
|
||||
return {
|
||||
file,
|
||||
sepSlug,
|
||||
model,
|
||||
totalTasks,
|
||||
totalRuns,
|
||||
successfulRuns,
|
||||
taskSuccessPct,
|
||||
verifiedPct,
|
||||
editToolUsagePct: editTool.pct,
|
||||
editSuccessPct,
|
||||
patchFailurePct: patchFailure.pct,
|
||||
patchFailures: patchFailure.numerator,
|
||||
patchAttempts: patchFailure.denominator,
|
||||
mutationIntentPct,
|
||||
autocorrectFreePct,
|
||||
tasksAllPassing,
|
||||
tasksFlakyFailing,
|
||||
timeoutRuns,
|
||||
inputTokensTotal: parseNumber(inRow[0]),
|
||||
outputTokensTotal: parseNumber(outRow[0]),
|
||||
totalTokens: parseNumber(totalRow[0]),
|
||||
inputTokensAvg: parseNumber(inRow[1]),
|
||||
outputTokensAvg: parseNumber(outRow[1]),
|
||||
totalTokensAvg: parseNumber(totalRow[1]),
|
||||
durationTotal: durationRow[0] ?? "",
|
||||
durationAvg: durationRow[1] ?? "",
|
||||
avgIndentScore: indentValue && indentValue !== "" ? Number.parseFloat(indentValue) : null,
|
||||
readTotal: parseNumber(readRow[0]),
|
||||
editTotal: parseNumber(editRow[0]),
|
||||
writeTotal: parseNumber(writeRow[0]),
|
||||
};
|
||||
}
|
||||
|
||||
function fmtPct(value: number): string {
|
||||
return Number.isFinite(value) ? `${value.toFixed(1)}%` : "—";
|
||||
}
|
||||
|
||||
function fmtNum(value: number): string {
|
||||
return Number.isFinite(value) ? value.toLocaleString() : "—";
|
||||
}
|
||||
|
||||
function shortModel(model: string): string {
|
||||
const cleaned = model.replace(/^p-anthropic\//, "anthropic/");
|
||||
const segs = cleaned.split("/");
|
||||
return segs[segs.length - 1].replace(/:nitro/, "");
|
||||
}
|
||||
|
||||
function sortRows(rows: ReportRow[], by: typeof sortBy): ReportRow[] {
|
||||
const sepOrder = ["gt", "plus", "div", "pipe", "bslash", "tilde", "pct", "colon"];
|
||||
const modelOrder = (m: string) => {
|
||||
if (m.includes("glm")) return 0;
|
||||
if (m.includes("gpt")) return 1;
|
||||
if (m.includes("claude")) return 2;
|
||||
return 3;
|
||||
};
|
||||
const cmp: Record<typeof sortBy, (a: ReportRow, b: ReportRow) => number> = {
|
||||
sep: (a, b) =>
|
||||
sepOrder.indexOf(a.sepSlug) - sepOrder.indexOf(b.sepSlug) ||
|
||||
modelOrder(a.model) - modelOrder(b.model),
|
||||
model: (a, b) =>
|
||||
modelOrder(a.model) - modelOrder(b.model) ||
|
||||
sepOrder.indexOf(a.sepSlug) - sepOrder.indexOf(b.sepSlug),
|
||||
task: (a, b) => b.taskSuccessPct - a.taskSuccessPct,
|
||||
edit: (a, b) => b.editSuccessPct - a.editSuccessPct,
|
||||
tokens: (a, b) => a.totalTokensAvg - b.totalTokensAvg,
|
||||
};
|
||||
return [...rows].sort(cmp[by]);
|
||||
}
|
||||
|
||||
const entries = (
|
||||
await Promise.all(
|
||||
resolvedDirs.map(async d =>
|
||||
(await fs.readdir(d, { withFileTypes: true }))
|
||||
.filter(e => e.isFile() && e.name.endsWith(".md"))
|
||||
.map(e => path.join(d, e.name)),
|
||||
),
|
||||
)
|
||||
).flat();
|
||||
|
||||
const rawRows = await Promise.all(entries.map(parseReport));
|
||||
const rows = aggregate ? mergeRows(rawRows) : rawRows;
|
||||
const sorted = sortRows(rows, sortBy);
|
||||
|
||||
function mergeRows(input: ReportRow[]): ReportRow[] {
|
||||
const groups = new Map<string, ReportRow[]>();
|
||||
for (const r of input) {
|
||||
const key = `${r.sepSlug}::${r.model}`;
|
||||
const list = groups.get(key);
|
||||
if (list) list.push(r);
|
||||
else groups.set(key, [r]);
|
||||
}
|
||||
const out: ReportRow[] = [];
|
||||
for (const [, list] of groups) {
|
||||
if (list.length === 1) {
|
||||
out.push(list[0]);
|
||||
continue;
|
||||
}
|
||||
const totalRuns = sumField(list, r => r.totalRuns);
|
||||
const successfulRuns = sumField(list, r => r.successfulRuns);
|
||||
const patchFailures = sumField(list, r => r.patchFailures);
|
||||
const patchAttempts = sumField(list, r => r.patchAttempts);
|
||||
const totalTokens = sumField(list, r => r.totalTokens);
|
||||
const inputTokensTotal = sumField(list, r => r.inputTokensTotal);
|
||||
const outputTokensTotal = sumField(list, r => r.outputTokensTotal);
|
||||
const editTotal = sumField(list, r => r.editTotal);
|
||||
const readTotal = sumField(list, r => r.readTotal);
|
||||
const writeTotal = sumField(list, r => r.writeTotal);
|
||||
const ratio = (n: number, d: number) => (d === 0 ? Number.NaN : (n / d) * 100);
|
||||
const indentVals = list.map(r => r.avgIndentScore).filter((v): v is number => v !== null);
|
||||
const indent = indentVals.length === 0 ? null : indentVals.reduce((a, b) => a + b, 0) / indentVals.length;
|
||||
out.push({
|
||||
file: list.map(r => r.file).join(","),
|
||||
sepSlug: list[0].sepSlug,
|
||||
model: list[0].model,
|
||||
totalTasks: sumField(list, r => r.totalTasks),
|
||||
totalRuns,
|
||||
successfulRuns,
|
||||
taskSuccessPct: ratio(successfulRuns, totalRuns),
|
||||
verifiedPct: ratio(successfulRuns, totalRuns),
|
||||
editToolUsagePct: ratio(sumField(list, r => Math.round((r.editToolUsagePct / 100) * r.totalRuns)), totalRuns),
|
||||
editSuccessPct: ratio(patchAttempts - patchFailures, patchAttempts),
|
||||
patchFailurePct: ratio(patchFailures, patchAttempts),
|
||||
patchFailures,
|
||||
patchAttempts,
|
||||
mutationIntentPct: list.reduce((a, r) => a + (Number.isFinite(r.mutationIntentPct) ? r.mutationIntentPct : 0), 0) / list.length,
|
||||
autocorrectFreePct: list.reduce((a, r) => a + (Number.isFinite(r.autocorrectFreePct) ? r.autocorrectFreePct : 0), 0) / list.length,
|
||||
tasksAllPassing: sumField(list, r => r.tasksAllPassing),
|
||||
tasksFlakyFailing: sumField(list, r => r.tasksFlakyFailing),
|
||||
timeoutRuns: sumField(list, r => r.timeoutRuns),
|
||||
inputTokensTotal,
|
||||
outputTokensTotal,
|
||||
totalTokens,
|
||||
inputTokensAvg: totalRuns === 0 ? 0 : Math.round(inputTokensTotal / totalRuns),
|
||||
outputTokensAvg: totalRuns === 0 ? 0 : Math.round(outputTokensTotal / totalRuns),
|
||||
totalTokensAvg: totalRuns === 0 ? 0 : Math.round(totalTokens / totalRuns),
|
||||
durationTotal: list.map(r => r.durationTotal).join(" + "),
|
||||
durationAvg: list.map(r => r.durationAvg).join(" / "),
|
||||
avgIndentScore: indent,
|
||||
readTotal,
|
||||
editTotal,
|
||||
writeTotal,
|
||||
});
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function sumField(list: ReportRow[], pick: (r: ReportRow) => number): number {
|
||||
return list.reduce((a, r) => a + (Number.isFinite(pick(r)) ? pick(r) : 0), 0);
|
||||
}
|
||||
|
||||
if (format === "json") {
|
||||
console.log(JSON.stringify(sorted, null, 2));
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
if (format === "csv") {
|
||||
const cols: Array<keyof ReportRow> = [
|
||||
"sepSlug",
|
||||
"model",
|
||||
"totalRuns",
|
||||
"successfulRuns",
|
||||
"taskSuccessPct",
|
||||
"editToolUsagePct",
|
||||
"editSuccessPct",
|
||||
"patchFailurePct",
|
||||
"patchFailures",
|
||||
"patchAttempts",
|
||||
"mutationIntentPct",
|
||||
"avgIndentScore",
|
||||
"inputTokensTotal",
|
||||
"outputTokensTotal",
|
||||
"totalTokens",
|
||||
"totalTokensAvg",
|
||||
"durationTotal",
|
||||
"durationAvg",
|
||||
"editTotal",
|
||||
"readTotal",
|
||||
];
|
||||
console.log(cols.join(","));
|
||||
for (const r of sorted) {
|
||||
console.log(cols.map(c => JSON.stringify(r[c] ?? "")).join(","));
|
||||
}
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
const headers = [
|
||||
"sep",
|
||||
"model",
|
||||
"task ✓",
|
||||
"edit ✓",
|
||||
"patch fail",
|
||||
"intent",
|
||||
"in tok/run",
|
||||
"out tok/run",
|
||||
"tok/run",
|
||||
"avg time",
|
||||
"indent",
|
||||
];
|
||||
|
||||
const data: string[][] = sorted.map(r => [
|
||||
SEPARATOR_DISPLAY[r.sepSlug] ?? r.sepSlug,
|
||||
shortModel(r.model),
|
||||
`${fmtPct(r.taskSuccessPct)} (${r.successfulRuns}/${r.totalRuns})`,
|
||||
fmtPct(r.editSuccessPct),
|
||||
`${fmtPct(r.patchFailurePct)} (${r.patchFailures}/${r.patchAttempts})`,
|
||||
fmtPct(r.mutationIntentPct),
|
||||
fmtNum(r.inputTokensAvg),
|
||||
fmtNum(r.outputTokensAvg),
|
||||
fmtNum(r.totalTokensAvg),
|
||||
r.durationAvg,
|
||||
r.avgIndentScore !== null ? r.avgIndentScore.toFixed(2) : "—",
|
||||
]);
|
||||
|
||||
if (format === "md") {
|
||||
const align = headers.map(() => "---");
|
||||
const out: string[] = [];
|
||||
out.push(`| ${headers.join(" | ")} |`);
|
||||
out.push(`|${align.map(a => `${a}`).join("|")}|`);
|
||||
for (const row of data) out.push(`| ${row.join(" | ")} |`);
|
||||
console.log(out.join("\n"));
|
||||
console.log();
|
||||
groupAggregates(sorted, "md");
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
// Plain text aligned table
|
||||
const widths = headers.map((h, i) => Math.max(h.length, ...data.map(r => r[i].length)));
|
||||
const fmtRow = (row: string[]) => row.map((c, i) => c.padEnd(widths[i])).join(" ");
|
||||
console.log(fmtRow(headers));
|
||||
console.log(widths.map(w => "-".repeat(w)).join(" "));
|
||||
for (const row of data) console.log(fmtRow(row));
|
||||
|
||||
groupAggregates(sorted, "table");
|
||||
|
||||
function groupAggregates(rows: ReportRow[], fmt: "md" | "table"): void {
|
||||
const bySep = new Map<string, ReportRow[]>();
|
||||
const byModel = new Map<string, ReportRow[]>();
|
||||
for (const r of rows) {
|
||||
(bySep.get(r.sepSlug) ?? bySep.set(r.sepSlug, []).get(r.sepSlug)!).push(r);
|
||||
(byModel.get(r.model) ?? byModel.set(r.model, []).get(r.model)!).push(r);
|
||||
}
|
||||
|
||||
const sepHeaders = ["sep", "task ✓ (avg)", "edit ✓ (avg)", "patch fail (sum)", "tok/run (avg)"];
|
||||
const sepData: string[][] = [];
|
||||
for (const [slug, list] of bySep) {
|
||||
sepData.push([
|
||||
SEPARATOR_DISPLAY[slug] ?? slug,
|
||||
fmtPct(avg(list, r => r.taskSuccessPct)),
|
||||
fmtPct(avg(list, r => r.editSuccessPct)),
|
||||
`${sum(list, r => r.patchFailures)}/${sum(list, r => r.patchAttempts)}`,
|
||||
fmtNum(Math.round(avg(list, r => r.totalTokensAvg))),
|
||||
]);
|
||||
}
|
||||
|
||||
const modelHeaders = ["model", "task ✓ (avg)", "edit ✓ (avg)", "patch fail (sum)", "tok/run (avg)"];
|
||||
const modelData: string[][] = [];
|
||||
for (const [model, list] of byModel) {
|
||||
modelData.push([
|
||||
shortModel(model),
|
||||
fmtPct(avg(list, r => r.taskSuccessPct)),
|
||||
fmtPct(avg(list, r => r.editSuccessPct)),
|
||||
`${sum(list, r => r.patchFailures)}/${sum(list, r => r.patchAttempts)}`,
|
||||
fmtNum(Math.round(avg(list, r => r.totalTokensAvg))),
|
||||
]);
|
||||
}
|
||||
|
||||
if (fmt === "md") {
|
||||
console.log("### Per separator (avg across models)\n");
|
||||
printMd(sepHeaders, sepData);
|
||||
console.log("\n### Per model (avg across separators)\n");
|
||||
printMd(modelHeaders, modelData);
|
||||
} else {
|
||||
console.log("\nPer separator (avg across models):");
|
||||
printTable(sepHeaders, sepData);
|
||||
console.log("\nPer model (avg across separators):");
|
||||
printTable(modelHeaders, modelData);
|
||||
}
|
||||
}
|
||||
|
||||
function avg(list: ReportRow[], pick: (r: ReportRow) => number): number {
|
||||
const vals = list.map(pick).filter(Number.isFinite);
|
||||
return vals.length === 0 ? Number.NaN : vals.reduce((a, b) => a + b, 0) / vals.length;
|
||||
}
|
||||
function sum(list: ReportRow[], pick: (r: ReportRow) => number): number {
|
||||
return list.reduce((a, r) => a + (Number.isFinite(pick(r)) ? pick(r) : 0), 0);
|
||||
}
|
||||
function printMd(headers: string[], data: string[][]): void {
|
||||
console.log(`| ${headers.join(" | ")} |`);
|
||||
console.log(`|${headers.map(() => "---").join("|")}|`);
|
||||
for (const r of data) console.log(`| ${r.join(" | ")} |`);
|
||||
}
|
||||
function printTable(headers: string[], data: string[][]): void {
|
||||
const widths = headers.map((h, i) => Math.max(h.length, ...data.map(r => r[i].length)));
|
||||
const row = (r: string[]) => r.map((c, i) => c.padEnd(widths[i])).join(" ");
|
||||
console.log(row(headers));
|
||||
console.log(widths.map(w => "-".repeat(w)).join(" "));
|
||||
for (const r of data) console.log(row(r));
|
||||
}
|
||||
Reference in New Issue
Block a user