feat(metaharness/adapters): implemented interactive cli runner and reporting

- Added comprehensive CLI arguments for thinking levels, timeouts, concurrency, and report formatting.
- Implemented real-time interactive terminal progress rendering and runtime statistics summary.
- Enhanced runner logic with tokenizer-based hashline operation detection and aggregated reporting.
- Added bench:edit script to package configuration and updated default report output paths.
This commit is contained in:
can1357
2026-07-30 06:14:45 +02:00
parent a6001c04a3
commit 4e5f480f14
5 changed files with 650 additions and 78 deletions
+309 -43
View File
@@ -1,97 +1,363 @@
#!/usr/bin/env bun
/** Manager-owned executable adapter for the TypeScript edit benchmark. */
/**
* Executable adapter for the TypeScript edit benchmark.
*
* Two audiences share this entry point:
* - The metaharness manager spawns it headless (`--model … --output result.json`)
* and consumes only the continuously-rewritten report file.
* - Humans run it directly and get the interactive experience: a config
* banner, a live progress bar with per-run failure diffs (stderr, TTY-aware),
* a runtime-stats summary, and a markdown or JSON report.
*/
import * as fs from "node:fs/promises";
import * as path from "node:path";
import { parseArgs } from "node:util";
import { TempDir } from "@oh-my-pi/pi-utils";
import { loadTasksFromDir } from "@oh-my-pi/typescript-edit-benchmark/tasks";
import { generateJsonReport } from "./report";
import { type BenchmarkConfig, runBenchmark } from "./runner";
import { type ResolvedThinkingLevel, ThinkingLevel } from "@oh-my-pi/pi-agent-core";
import { THINKING_EFFORTS } from "@oh-my-pi/pi-ai";
import { postmortem, TempDir } from "@oh-my-pi/pi-utils";
import { loadTasksFromDir, validateFixturesFromDir } from "@oh-my-pi/typescript-edit-benchmark/tasks";
import { LiveProgress } from "./live-progress";
import { generateJsonReport, generateReport } from "./report";
import { type BenchmarkConfig, type BenchmarkResult, buildBenchmarkResult, runBenchmark } from "./runner";
const EDIT_PACKAGE = path.resolve(import.meta.dir, "..", "..", "..", "typescript-edit-benchmark");
const RUNS_DIR = path.resolve(import.meta.dir, "..", "..", "..", "..", "runs");
async function extractFixtures(): Promise<{ dir: string; temp: TempDir }> {
type ReportFormat = "markdown" | "json";
function log(line = ""): void {
process.stderr.write(`${line}\n`);
}
function fail(message: string): never {
throw new Error(message);
}
function parseThinkingLevel(value: string): ResolvedThinkingLevel {
if ([ThinkingLevel.Off, ...THINKING_EFFORTS].includes(value as ResolvedThinkingLevel)) {
return value as ResolvedThinkingLevel;
}
return fail(`Invalid thinking level: ${value}. Valid levels: ${[ThinkingLevel.Off, ...THINKING_EFFORTS].join(", ")}`);
}
function parsePositiveInt(raw: string | undefined, flag: string, fallback: number): number {
if (raw === undefined) return fallback;
const parsed = Number.parseInt(raw, 10);
if (Number.isNaN(parsed) || parsed < 1) fail(`Invalid ${flag} value: ${raw}. Must be a positive integer.`);
return parsed;
}
function parseFuzzy(raw: string | undefined): boolean | "auto" | undefined {
if (raw === undefined) return undefined;
if (raw === "auto") return "auto";
if (raw === "true" || raw === "1") return true;
if (raw === "false" || raw === "0") return false;
return fail(`Invalid --edit-fuzzy: ${raw}. Must be true, false, 1, 0, or auto.`);
}
function parseFuzzyThreshold(raw: string | undefined): number | "auto" | undefined {
if (raw === undefined) return undefined;
if (raw === "auto") return "auto";
const parsed = Number.parseFloat(raw);
if (Number.isNaN(parsed) || parsed < 0 || parsed > 1) {
fail(`Invalid --edit-fuzzy-threshold: ${raw}. Must be 0-1 or auto.`);
}
return parsed;
}
function resolveFormat(explicit: string | undefined, outputPath: string): ReportFormat {
if (explicit === "markdown" || explicit === "json") return explicit;
if (explicit !== undefined) fail(`Invalid --format: ${explicit}. Must be markdown or json.`);
return outputPath.endsWith(".md") || outputPath.endsWith(".markdown") ? "markdown" : "json";
}
function generateReportFilename(config: BenchmarkConfig): string {
const modelName = config.model
.split("/")
.pop()!
.replace(/[^a-zA-Z0-9-]/g, "_");
const variant = config.editVariant ?? "auto";
const timestamp = new Date().toISOString().replace(/:/g, "-").replace(/\..+$/, "Z");
return path.join(RUNS_DIR, `${modelName}_${variant}_${timestamp}.md`);
}
/** Sibling `.dump` directory for an output path, timestamped on collision. */
async function resolveConversationDumpDir(outputPath: string): Promise<string> {
const parsed = path.parse(outputPath);
const preferred = path.join(parsed.dir, `${parsed.name}.dump`);
try {
await fs.stat(preferred);
} catch {
return preferred;
}
const timestamp = new Date().toISOString().replace(/[:.]/g, "-");
return path.join(parsed.dir, `${parsed.name}.${timestamp}.dump`);
}
interface ResolvedFixtures {
dir: string;
cleanup?: () => Promise<void>;
}
/** Resolve `--fixtures` (directory or tarball; default: the built-in tarball). */
async function resolveFixtures(fixturesArg: string | undefined): Promise<ResolvedFixtures> {
const source = fixturesArg ?? path.join(EDIT_PACKAGE, "fixtures.tar.gz");
if (!source.endsWith(".tar.gz") && !source.endsWith(".tgz")) {
return { dir: source };
}
const temp = await TempDir.create("@metaharness-edit-fixtures-");
const archive = new Bun.Archive(await Bun.file(path.join(EDIT_PACKAGE, "fixtures.tar.gz")).arrayBuffer());
const archive = new Bun.Archive(await Bun.file(source).arrayBuffer());
for (const [filePath, file] of await archive.files()) {
await Bun.write(path.join(temp.path(), filePath), file);
}
const entries = await fs.readdir(temp.path(), { withFileTypes: true });
const directories = entries.filter(entry => entry.isDirectory());
const files = entries.filter(entry => entry.isFile());
return { dir: directories.length === 1 && files.length === 0 ? path.join(temp.path(), directories[0]!.name) : temp.path(), temp };
const dir = directories.length === 1 && files.length === 0 ? path.join(temp.path(), directories[0]!.name) : temp.path();
return { dir, cleanup: () => temp.remove() };
}
/** Execute an edit benchmark and continuously materialize its normalized source artifact. */
function printUsage(): void {
log(`
Edit Benchmark - Evaluate patch application success rates
Usage:
bun adapters/edit/cli.ts --model <provider/model> [options]
Options:
--model <id> Provider/model ID, e.g. anthropic/claude-sonnet-4 (required)
--provider <id> Override provider (auto-detected from model prefix)
--thinking <level> Thinking level: off, minimal, low, medium, high, xhigh, max
--runs <n> Runs per task (default: 1)
--timeout <ms> Timeout per run in ms (default: 120000)
--connection-timeout <ms> Timeout for first event before fast-retry (default: 30000)
--max-turns <n> Max turns per attempt before failing (default: 30)
--task-concurrency <n> Max tasks to run in parallel (default: 32)
--tasks <ids> Comma-separated task IDs to run (default: sampled)
--max-tasks <n> Max tasks to sample evenly (default: 80, 0 = all)
--fixtures <path> Fixtures directory or .tar.gz archive (default: built-in)
--edit-variant <v> Edit variant, e.g. hashline, replace, apply_patch (default: auto)
--edit-fuzzy <bool> Fuzzy matching: true, false, auto
--edit-fuzzy-threshold <n> Fuzzy threshold 0-1 or auto
--guided Include an authoritative suggested edit payload
--max-attempts <n> Max prompt attempts per run (default: 1)
--no-early-stop-on-match Don't short-circuit when output matches expected
--output <file> Report file (default: runs/<model>_<variant>_<ts>.md)
--format <fmt> Report format: markdown, json (default: by extension)
--check-fixtures Validate fixtures and exit
--quiet Suppress the live progress view
--list Print task ids as JSON and exit
--help Show this help message
Examples:
# Full run against the default sample of 80 tasks
bun adapters/edit/cli.ts --model anthropic/claude-sonnet-4
# Compare edit variants
bun adapters/edit/cli.ts --model openai/gpt-5 --edit-variant hashline --output hashline.md
bun adapters/edit/cli.ts --model openai/gpt-5 --edit-variant apply_patch --output apply_patch.md
# Specific tasks, more runs
bun adapters/edit/cli.ts --model anthropic/claude-sonnet-4 --tasks logic-flip-strict-equality-001 --runs 5
`);
}
/** Execute an edit benchmark and continuously materialize its report artifact. */
export async function main(argv = process.argv.slice(2)): Promise<void> {
const { values } = parseArgs({
args: argv,
options: {
model: { type: "string" },
output: { type: "string" },
"max-tasks": { type: "string", default: "80" },
tasks: { type: "string" },
"task-concurrency": { type: "string", default: "32" },
provider: { type: "string" },
thinking: { type: "string" },
runs: { type: "string", default: "1" },
timeout: { type: "string", default: "120000" },
"connection-timeout": { type: "string", default: "30000" },
"max-turns": { type: "string", default: "30" },
"task-concurrency": { type: "string", default: "32" },
tasks: { type: "string" },
"max-tasks": { type: "string", default: "80" },
fixtures: { type: "string" },
"edit-variant": { type: "string" },
"edit-fuzzy": { type: "string" },
"edit-fuzzy-threshold": { type: "string" },
guided: { type: "boolean", default: false },
"max-attempts": { type: "string", default: "1" },
"no-op-retry-limit": { type: "string", default: "2" },
"max-timeout-retries": { type: "string", default: "3" },
"max-provider-retries": { type: "string", default: "3" },
"mutation-scope-window": { type: "string", default: "20" },
"no-early-stop-on-match": { type: "boolean", default: false },
output: { type: "string" },
format: { type: "string" },
"check-fixtures": { type: "boolean", default: false },
quiet: { type: "boolean", default: false },
list: { type: "boolean", default: false },
help: { type: "boolean", default: false },
},
strict: true,
});
const fixtures = await extractFixtures();
if (values.help) {
printUsage();
return;
}
const fixtures = await resolveFixtures(values.fixtures);
try {
if (values["check-fixtures"]) {
const issues = await validateFixturesFromDir(fixtures.dir);
if (issues.length === 0) {
log("Fixtures OK");
return;
}
log("Fixture validation failed:");
for (const issue of issues) log(` - ${issue.taskId}: ${issue.message}`);
process.exitCode = 1;
return;
}
let tasks = await loadTasksFromDir(fixtures.dir);
if (values.list) {
process.stdout.write(`${JSON.stringify(tasks.map(task => ({ id: task.id, name: task.name })))}\n`);
return;
}
if (!values.model || !values.output) throw new Error("edit adapter requires --model and --output");
if (!values.model) fail("edit adapter requires --model (see --help)");
if (values.tasks) {
const selected = new Set(values.tasks.split(",").map(value => value.trim()));
tasks = tasks.filter(task => selected.has(task.id));
if (tasks.length !== selected.size) throw new Error("one or more edit task ids were not found");
if (tasks.length !== selected.size) fail("one or more edit task ids were not found (see --list)");
} else {
const limit = Number(values["max-tasks"]);
if (limit > 0 && tasks.length > limit) {
// Deterministic even sampling across the id-sorted list keeps
// mutation-category coverage representative.
const sorted = tasks.slice().sort((a, b) => a.id.localeCompare(b.id));
const step = sorted.length / limit;
tasks = Array.from({ length: limit }, (_, index) => sorted[Math.floor(index * step)]!);
}
}
const slash = values.model.indexOf("/");
const model = values.model;
const slash = model.indexOf("/");
const config: BenchmarkConfig = {
provider: slash === -1 ? "anthropic" : values.model.slice(0, slash),
model: values.model,
runsPerTask: Number(values.runs),
timeout: 120_000,
connectionTimeout: 30_000,
maxTurns: 30,
taskConcurrency: Number(values["task-concurrency"]),
guided: false,
maxAttempts: 1,
noOpRetryLimit: 2,
maxTimeoutRetries: 3,
maxProviderFailureRetries: 3,
mutationScopeWindow: 20,
conversationDumpDir: path.join(path.dirname(values.output), "result.dump"),
provider: values.provider ?? (slash === -1 ? "anthropic" : model.slice(0, slash)),
model,
...(values.thinking === undefined ? {} : { thinkingLevel: parseThinkingLevel(values.thinking) }),
runsPerTask: parsePositiveInt(values.runs, "--runs", 1),
timeout: parsePositiveInt(values.timeout, "--timeout", 120_000),
connectionTimeout: parsePositiveInt(values["connection-timeout"], "--connection-timeout", 30_000),
maxTurns: parsePositiveInt(values["max-turns"], "--max-turns", 30),
taskConcurrency: parsePositiveInt(values["task-concurrency"], "--task-concurrency", 32),
guided: values.guided,
maxAttempts: parsePositiveInt(values["max-attempts"], "--max-attempts", 1),
noOpRetryLimit: parsePositiveInt(values["no-op-retry-limit"], "--no-op-retry-limit", 2),
maxTimeoutRetries: parsePositiveInt(values["max-timeout-retries"], "--max-timeout-retries", 3),
maxProviderFailureRetries: parsePositiveInt(values["max-provider-retries"], "--max-provider-retries", 3),
mutationScopeWindow: parsePositiveInt(values["mutation-scope-window"], "--mutation-scope-window", 20),
inProcess: true,
earlyStopOnMatch: true,
earlyStopOnMatch: !values["no-early-stop-on-match"],
};
let writes = Promise.resolve();
const result = await runBenchmark(tasks, config, undefined, snapshot => {
writes = writes.then(async () => {
await Bun.write(values.output!, generateJsonReport(snapshot));
});
const editVariant = values["edit-variant"];
if (editVariant) config.editVariant = editVariant;
const editFuzzy = parseFuzzy(values["edit-fuzzy"]);
if (editFuzzy !== undefined) config.editFuzzy = editFuzzy;
const editFuzzyThreshold = parseFuzzyThreshold(values["edit-fuzzy-threshold"]);
if (editFuzzyThreshold !== undefined) config.editFuzzyThreshold = editFuzzyThreshold;
let outputPath = values.output;
if (outputPath === undefined) {
await fs.mkdir(RUNS_DIR, { recursive: true });
outputPath = generateReportFilename(config);
}
const format = resolveFormat(values.format, outputPath);
config.conversationDumpDir = await resolveConversationDumpDir(outputPath);
log("Edit Benchmark");
log("==============");
log(`Provider: ${config.provider}`);
log(`Model: ${config.model}`);
if (config.thinkingLevel) log(`Thinking: ${config.thinkingLevel}`);
log(`Runs per task: ${config.runsPerTask}`);
log(`Timeout: ${config.timeout}ms`);
log(`Task concurrency: ${config.taskConcurrency}`);
log(`Guided mode: ${config.guided ? "enabled" : "disabled"}`);
if (config.editVariant) log(`Edit variant: ${config.editVariant}`);
if (config.editFuzzy !== undefined) log(`Edit fuzzy: ${config.editFuzzy}`);
if (config.editFuzzyThreshold !== undefined) log(`Edit fuzzy threshold: ${config.editFuzzyThreshold}`);
log(`Tasks: ${tasks.length}`);
log(`Report: ${outputPath}`);
log(`Conversation dumps: ${config.conversationDumpDir}`);
log();
const renderReport = (result: BenchmarkResult): string =>
format === "json" ? generateJsonReport(result) : generateReport(result);
const progress = values.quiet ? undefined : new LiveProgress(tasks.length * config.runsPerTask, config.runsPerTask);
let latestResult = buildBenchmarkResult({
tasks,
config,
resultsByTask: new Map(),
startTime: new Date().toISOString(),
});
let writes = Promise.resolve();
const queueReportWrite = (result: BenchmarkResult) => {
writes = writes.then(async () => {
await Bun.write(outputPath, renderReport(result));
});
};
// A killed/interrupted run still leaves a usable partial report behind.
const unregisterCleanup = postmortem.register("edit-benchmark-report", async reason => {
if (reason === postmortem.Reason.EXIT) return;
progress?.finish();
log("Benchmark interrupted; writing partial report...");
await Bun.write(outputPath, renderReport(latestResult));
await fixtures.cleanup?.();
});
const result = await runBenchmark(
tasks,
config,
event => progress?.handleEvent(event),
snapshot => {
latestResult = snapshot;
queueReportWrite(snapshot);
},
);
latestResult = result;
progress?.finish();
queueReportWrite(result);
await writes;
await Bun.write(values.output, generateJsonReport(result));
unregisterCleanup();
log();
log("Benchmark complete!");
log(
` Task success rate (best of ${config.runsPerTask}): ${(result.summary.taskSuccessRate * 100).toFixed(1)}% (${result.summary.successfulTasks}/${result.summary.totalTasks})`,
);
log(` Total tokens (best): ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`);
log(
` Tokens/task (best): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total} reasoning=${result.summary.avgTokensPerTask.reasoning}`,
);
if (result.summary.ghostRuns > 0) log(` Ghost runs (0/0/0): ${result.summary.ghostRuns}`);
if (result.summary.timeoutRuns > 0) log(` Timeout runs: ${result.summary.timeoutRuns}`);
log(`Report written to: ${outputPath}`);
} finally {
await fixtures.temp.remove();
await fixtures.cleanup?.();
}
}
if (import.meta.main) {
main().catch(error => {
process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`);
process.exitCode = 1;
});
main()
.then(async () => {
// In-process benchmark runs can leave provider keep-alive sockets and
// background AgentSession timers alive after the report is written.
// Treat the final report as the CLI boundary.
await postmortem.quit(Number(process.exitCode ?? 0));
})
.catch(async error => {
process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`);
await postmortem.quit(1);
});
}
@@ -0,0 +1,241 @@
/**
* Live terminal renderer for edit benchmark runs.
*
* Thin view over {@link runBenchmark}'s `onProgress` stream: a single
* in-place progress line (bar, pass rates, token/latency averages, in-flight
* count) on a TTY, plain per-run lines otherwise, plus inline colored diffs
* for failed runs and a runtime-stats block at the end. Renders to stderr so
* stdout stays free for machine output (`--list` JSON), and the report file
* remains the only artifact the manager consumes.
*/
import { percentile, type ProgressEvent } from "./runner";
const ANSI = {
reset: "\x1b[0m",
bold: "\x1b[1m",
dim: "\x1b[2m",
red: "\x1b[31m",
green: "\x1b[32m",
yellow: "\x1b[33m",
cyan: "\x1b[36m",
} as const;
function rateColor(percent: number): string {
if (percent >= 80) return ANSI.green;
if (percent >= 50) return ANSI.yellow;
return ANSI.red;
}
export class LiveProgress {
readonly #totalRuns: number;
readonly #runsPerTask: number;
readonly #isTty: boolean;
readonly #colors: boolean;
#started = 0;
#completed = 0;
#success = 0;
#totalInput = 0;
#totalOutput = 0;
#totalDuration = 0;
#totalReads = 0;
#totalEdits = 0;
#totalWrites = 0;
#totalEditSuccesses = 0;
#totalToolInputChars = 0;
#indentScores: number[] = [];
#inputTokens: number[] = [];
#outputTokens: number[] = [];
#totalTokens: number[] = [];
#oneShotSuccessTokens: number[] = [];
#lastLineLength = 0;
constructor(totalRuns: number, runsPerTask: number) {
this.#totalRuns = totalRuns;
this.#runsPerTask = runsPerTask;
this.#isTty = Boolean(process.stderr.isTTY);
this.#colors = this.#isTty && !process.env.NO_COLOR;
}
#paint(code: string, text: string): string {
return this.#colors ? `${code}${text}${ANSI.reset}` : text;
}
#log(line: string): void {
process.stderr.write(`${line}\n`);
}
handleEvent(event: ProgressEvent): void {
if (event.status === "started") {
this.#started += 1;
if (!this.#isTty) {
this.#log(` [${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} started...`);
}
this.#renderLine();
return;
}
this.#completed += 1;
if (event.result) {
if (event.result.success) {
this.#success += 1;
}
if (event.result.success && event.runIndex === 0) {
this.#oneShotSuccessTokens.push(event.result.tokens.total);
}
this.#totalInput += event.result.tokens.input;
this.#totalOutput += event.result.tokens.output;
this.#inputTokens.push(event.result.tokens.input);
this.#outputTokens.push(event.result.tokens.output);
this.#totalTokens.push(event.result.tokens.total);
this.#totalDuration += event.result.duration;
this.#totalReads += event.result.toolCalls.read;
this.#totalEdits += event.result.toolCalls.edit;
this.#totalWrites += event.result.toolCalls.write;
this.#totalEditSuccesses += event.result.toolCalls.editSuccesses;
this.#totalToolInputChars += event.result.toolCalls.totalInputChars;
if (typeof event.result.indentScore === "number") {
this.#indentScores.push(event.result.indentScore);
}
}
const result = event.result;
if (result && !result.success && result.error) {
this.#flushLine();
const header = this.#paint(
ANSI.red,
`[${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} failed:`,
);
this.#log(` ${header} ${result.error}`);
if (result.diff) {
const changeLines = result.diff
.split("\n")
.filter(line => /^[-+@]/.test(line) && !/^(---|\+\+\+)/.test(line));
const maxLines = 40;
for (const line of changeLines.slice(0, maxLines)) {
let color: string | undefined;
if (line.startsWith("@@")) color = ANSI.cyan;
else if (line.startsWith("-")) color = ANSI.red;
else if (line.startsWith("+")) color = ANSI.green;
this.#log(` ${color ? this.#paint(color, line) : line}`);
}
if (changeLines.length > maxLines) {
this.#log(this.#paint(ANSI.dim, ` ... (${changeLines.length - maxLines} more change lines)`));
}
}
}
if (result?.editFailures && result.editFailures.length > 0) {
this.#flushLine();
for (const [i, failure] of result.editFailures.entries()) {
const args = (failure.args ?? {}) as Record<string, unknown>;
const target =
typeof args.path === "string" ? args.path : typeof args.file === "string" ? args.file : undefined;
const op = typeof args.operation === "string" ? args.operation : undefined;
const oneLine = failure.error.replace(/\s+/g, " ").trim();
const clipped = oneLine.length > 240 ? `${oneLine.slice(0, 237)}...` : oneLine;
const tag = this.#paint(ANSI.yellow, `[${event.taskId}] schema #${i + 1}`);
const metaParts = [op, target].filter((value): value is string => Boolean(value));
const meta = metaParts.length > 0 ? this.#paint(ANSI.dim, metaParts.join(" ")) : "";
this.#log(` ${tag}${meta ? ` ${meta}` : ""} ${clipped}`);
if (failure.rawBlock) {
const rawLine = failure.rawBlock.replace(/\s+/g, " ").trim();
const clippedRaw = rawLine.length > 240 ? `${rawLine.slice(0, 237)}...` : rawLine;
this.#log(` ${this.#paint(ANSI.dim, "raw")} ${clippedRaw}`);
}
}
}
if (!this.#isTty) {
const status = event.result?.success ? "completed" : "failed";
this.#log(` [${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} ${status}`);
}
this.#renderLine();
}
finish(): void {
this.#flushLine();
this.#printSummary();
}
#printSummary(): void {
const n = this.#completed;
const denom = n || 1;
const successRate = (this.#success / denom) * 100;
const editSuccessRate = this.#totalEdits > 0 ? (this.#totalEditSuccesses / this.#totalEdits) * 100 : 100;
const avgIndent =
this.#indentScores.length > 0 ? this.#indentScores.reduce((a, b) => a + b, 0) / this.#indentScores.length : 0;
this.#log("");
this.#log(this.#paint(ANSI.bold, "Runtime Stats:"));
this.#log(
` Task success: ${this.#paint(rateColor(successRate), `${successRate.toFixed(1)}% (${this.#success}/${n})`)}`,
);
this.#log(
` Edit success: ${this.#paint(rateColor(editSuccessRate), `${editSuccessRate.toFixed(1)}% (${this.#totalEditSuccesses}/${this.#totalEdits})`)}`,
);
this.#log(` Avg indent score: ${avgIndent.toFixed(2)}`);
this.#log(` Tool calls: read=${this.#totalReads} edit=${this.#totalEdits} write=${this.#totalWrites}`);
this.#log(` Tool input chars: ${this.#totalToolInputChars.toLocaleString()}`);
const fmtTokens = (samples: number[]): string => {
if (samples.length === 0) return "mean=0 median=0 p1=0 p99=0";
const sorted = [...samples].sort((a, b) => a - b);
const mean = Math.round(sorted.reduce((a, b) => a + b, 0) / sorted.length);
return `mean=${mean} median=${Math.round(percentile(sorted, 50))} p1=${Math.round(percentile(sorted, 1))} p99=${Math.round(percentile(sorted, 99))}`;
};
this.#log(` Tokens/task in: ${fmtTokens(this.#inputTokens)}`);
this.#log(` Tokens/task out: ${fmtTokens(this.#outputTokens)}`);
this.#log(` Tokens/task tot: ${fmtTokens(this.#totalTokens)}`);
this.#log(` Tokens/task (one-shot successes): ${fmtTokens(this.#oneShotSuccessTokens)}`);
this.#log(` Avg time/task: ${Math.round(this.#totalDuration / denom)}ms`);
}
#renderLine(): void {
if (!this.#isTty) {
return;
}
const successRate = this.#completed > 0 ? (this.#success / this.#completed) * 100 : 0;
const editRate = this.#totalEdits > 0 ? (this.#totalEditSuccesses / this.#totalEdits) * 100 : 100;
const avgInput = this.#completed > 0 ? Math.round(this.#totalInput / this.#completed) : 0;
const avgOutput = this.#completed > 0 ? Math.round(this.#totalOutput / this.#completed) : 0;
const avgDuration = this.#completed > 0 ? Math.round(this.#totalDuration / this.#completed) : 0;
const inFlight = this.#started - this.#completed;
const bar = this.#renderBar(this.#completed, this.#totalRuns, 20);
const progress = this.#paint(ANSI.bold, `${this.#completed}/${this.#totalRuns}`);
const taskCol = `task=${this.#paint(rateColor(successRate), `${successRate.toFixed(0)}%`)}`;
const editCol = `edit=${this.#paint(rateColor(editRate), `${editRate.toFixed(0)}%`)}`;
const tokCol = this.#paint(ANSI.dim, `tok=${avgInput}/${avgOutput}`);
const durCol = this.#paint(ANSI.dim, `${avgDuration}ms`);
const rewCol = this.#paint(ANSI.dim, `r/e/w=${this.#totalReads}/${this.#totalEdits}/${this.#totalWrites}`);
const flyCol = `fly=${this.#paint(ANSI.cyan, String(inFlight))}`;
const line = ` ${bar} ${progress} ${taskCol} ${editCol} ${tokCol} ${durCol} ${rewCol} ${flyCol}`;
this.#writeLine(line);
}
#renderBar(done: number, total: number, width: number): string {
const ratio = total === 0 ? 0 : done / total;
const filled = Math.round(ratio * width);
const empty = Math.max(0, width - filled);
const filledPart = this.#paint(ANSI.green, "#".repeat(filled));
const emptyPart = this.#paint(ANSI.dim, "-".repeat(empty));
return `[${filledPart}${emptyPart}]`;
}
#writeLine(line: string): void {
const lineWidth = Bun.stringWidth(line);
const pad = this.#lastLineLength > lineWidth ? " ".repeat(this.#lastLineLength - lineWidth) : "";
process.stderr.write(`\r${line}${pad}`);
this.#lastLineLength = lineWidth;
}
#flushLine(): void {
if (!this.#isTty) {
return;
}
if (this.#lastLineLength > 0) {
process.stderr.write(`\r${" ".repeat(this.#lastLineLength)}\r`);
this.#lastLineLength = 0;
}
}
}
+9 -7
View File
@@ -212,17 +212,19 @@ export function generateReport(result: BenchmarkResult): string {
lines.push("");
if (summary.hashlineEditSubtypes) {
const order = ["set", "set_range", "insert"] as const;
const total = order.reduce((sum, key) => sum + (summary.hashlineEditSubtypes?.[key] ?? 0), 0);
const rows = Object.entries(summary.hashlineEditSubtypes)
.filter(([, count]) => count > 0)
.sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]));
const total = rows.reduce((sum, [, count]) => sum + count, 0);
if (total > 0) {
lines.push("### Hashline Edit Subtypes");
lines.push("### Hashline Op Breakdown");
lines.push("");
lines.push("All attempted edit calls across all runs (retries and failed calls included).");
lines.push("");
lines.push("| Operation | Count | % |");
lines.push("|-----------|-------|---|");
for (const key of order) {
const count = summary.hashlineEditSubtypes[key] ?? 0;
const pct = formatPercent(count / total);
lines.push(`| ${key} | ${count} | ${pct} |`);
for (const [op, count] of rows) {
lines.push(`| \`${op}\` | ${count} | ${formatPercent(count / total)} |`);
}
lines.push(`| **Total** | **${total}** | 100% |`);
lines.push("");
+90 -28
View File
@@ -7,7 +7,12 @@
/// <reference types="./bun-imports.d.ts" />
import * as fs from "node:fs";
import * as path from "node:path";
import { formatHashlineHeader, InMemorySnapshotStore } from "@oh-my-pi/hashline";
import {
type BlockTarget as HashlineBlockTarget,
formatHashlineHeader,
InMemorySnapshotStore,
Tokenizer as HashlineTokenizer,
} from "@oh-my-pi/hashline";
import type { AgentMessage, ResolvedThinkingLevel, ThinkingLevel } from "@oh-my-pi/pi-agent-core";
import type { Model, ToolExample } from "@oh-my-pi/pi-ai";
import { formatSessionDumpText, RpcClient } from "@oh-my-pi/pi-coding-agent";
@@ -262,7 +267,6 @@ function countEditFailureCategories(runs: TaskRunResult[]): Record<EditFailureCa
return counts;
}
const HL_SUBTYPES = ["set", "set_range", "insert"] as const;
const BENCHMARK_TOOL_NAMES = ["read", "edit", "write", "apply_patch"] as const;
const EDIT_TOOL_NAMES = ["edit", "apply_patch"] as const;
@@ -274,21 +278,71 @@ function isMutationTool(toolName: unknown): boolean {
return isEditTool(toolName) || toolName === "write";
}
function countHashlineEditSubtypes(args: unknown): Record<string, number> {
const counts: Record<string, number> = Object.fromEntries(HL_SUBTYPES.map(k => [k, 0]));
if (!args || typeof args !== "object") return counts;
const edits = (args as { edits?: unknown[] }).edits;
if (!Array.isArray(edits)) return counts;
for (const edit of edits) {
if (!edit || typeof edit !== "object") continue;
for (const key of HL_SUBTYPES) {
if (key in edit) {
counts[key]++;
break;
// Pure classification — single shared tokenizer is safe.
const HASHLINE_OP_TOKENIZER = new HashlineTokenizer();
/** Display label for a hashline op header, e.g. `SWAP.BLK` or `PASTE.POST`. */
function hashlineOpLabel(target: HashlineBlockTarget): string {
switch (target.kind) {
case "replace":
return "SWAP";
case "block":
return "SWAP.BLK";
case "delete":
return "DEL";
case "delete_block":
return "DEL.BLK";
case "insert_before":
return "INS.PRE";
case "insert_after":
return "INS.POST";
case "bof":
return "INS.HEAD";
case "eof":
return "INS.TAIL";
case "insert_after_block":
return "INS.BLK.POST";
case "copy":
return target.cut ? "CUT" : "COPY";
case "copy_block":
return target.cut ? "CUT.BLK" : "COPY.BLK";
case "paste":
switch (target.cursor.kind) {
case "before_anchor":
return "PASTE.PRE";
case "after_anchor":
return "PASTE.POST";
case "bof":
return "PASTE.HEAD";
case "eof":
return "PASTE.TAIL";
}
}
break;
case "paste_after_block":
return "PASTE.BLK.POST";
case "rem":
return "REM";
case "move":
return "MV";
}
return counts;
}
/**
* Count hashline op headers (`SWAP`, `DEL.BLK`, `CUT`, `PASTE.POST`, …) in an
* edit call's patch input. Returns `null` when the args carry no hashline
* patch — non-hashline edit variants and malformed calls contribute nothing.
*/
function countHashlineOps(args: unknown): Record<string, number> | null {
if (!args || typeof args !== "object" || !("input" in args)) return null;
const input = args.input;
if (typeof input !== "string" || input.length === 0) return null;
const counts: Record<string, number> = {};
for (const token of HASHLINE_OP_TOKENIZER.tokenizeAll(input)) {
if (token.kind !== "op-block") continue;
const label = hashlineOpLabel(token.target);
counts[label] = (counts[label] ?? 0) + 1;
}
return Object.keys(counts).length > 0 ? counts : null;
}
async function collectOriginalFileContents(cwd: string, files: string[]): Promise<Map<string, string>> {
@@ -829,7 +883,7 @@ export interface TaskRunResult {
editFailures: EditFailure[];
editWarnings: string[];
editAutocorrectCount: number;
/** Hashline edit subtype counts (replaceLine, replaceLines, etc.) — only when editVariant is hashline */
/** Hashline op counts (`SWAP`, `DEL.BLK`, `CUT`, …) — present when edit calls carried hashline patches. */
hashlineEditSubtypes?: Record<string, number>;
mutationIntentMatched?: boolean;
mutationIntentReason?: string;
@@ -945,7 +999,7 @@ export interface BenchmarkSummary {
mutationIntentMatchRate?: number;
/** Edit failure categories across all runs. */
editFailureCategories: Record<EditFailureCategory, number>;
/** Hashline edit subtype totals across all runs — only when editVariant is hashline. */
/** Hashline op totals across all runs — present when any run's edit calls carried hashline patches. */
hashlineEditSubtypes?: Record<string, number>;
}
@@ -1037,7 +1091,7 @@ async function runSingleTask(
editAutocorrects: 0,
totalInputChars: 0,
};
const hashlineSubtypes: Record<string, number> = Object.fromEntries(HL_SUBTYPES.map(k => [k, 0]));
const hashlineSubtypes: Record<string, number> = {};
const logFile = path.join(TMP, `run-${task.id}-${runIndex}.jsonl`);
const logEvent = async (event: unknown) => {
@@ -1271,10 +1325,10 @@ async function runSingleTask(
const pendingEdit = pendingEdits.get(e.toolCallId) ?? { args: null };
const args = pendingEdit.args;
pendingEdits.delete(e.toolCallId);
if (config.editVariant === "hashline" && args) {
const counts = countHashlineEditSubtypes(args);
for (const key of HL_SUBTYPES) {
hashlineSubtypes[key] += counts[key];
const hashlineOpCounts = countHashlineOps(args);
if (hashlineOpCounts) {
for (const key in hashlineOpCounts) {
hashlineSubtypes[key] = (hashlineSubtypes[key] ?? 0) + hashlineOpCounts[key];
}
}
if (e.isError) {
@@ -1422,7 +1476,7 @@ async function runSingleTask(
editFailures,
editWarnings,
editAutocorrectCount,
hashlineEditSubtypes: config.editVariant === "hashline" ? hashlineSubtypes : undefined,
hashlineEditSubtypes: Object.keys(hashlineSubtypes).length > 0 ? hashlineSubtypes : undefined,
mutationIntentMatched: mutationIntentValidation?.matched,
mutationIntentReason: mutationIntentValidation?.reason,
timeoutTelemetry,
@@ -1936,12 +1990,20 @@ export function buildBenchmarkResult(params: {
0,
);
const editFailureCategories = countEditFailureCategories(nonGhostRuns);
// Op counts aggregate every attempted edit call across ALL runs (retries and
// failed calls included) — deliberately: the mix shows what the model reaches
// for, not just what landed. Best-run-only would hide exactly the flailing
// (failed SWAPs before a successful retry) this table exists to expose.
const hashlineEditSubtypeTotals: Record<string, number> = {};
for (const run of allRuns) {
const runOps = run.hashlineEditSubtypes;
if (!runOps) continue;
for (const key in runOps) {
hashlineEditSubtypeTotals[key] = (hashlineEditSubtypeTotals[key] ?? 0) + runOps[key];
}
}
const hashlineEditSubtypes: Record<string, number> | undefined =
params.config.editVariant === "hashline"
? Object.fromEntries(
HL_SUBTYPES.map(key => [key, allRuns.reduce((sum, r) => sum + (r.hashlineEditSubtypes?.[key] ?? 0), 0)]),
)
: undefined;
Object.keys(hashlineEditSubtypeTotals).length > 0 ? hashlineEditSubtypeTotals : undefined;
// Primary aggregates run over the *best* run of each completed task.
const bestRuns: TaskRunResult[] = [];
+1
View File
@@ -20,6 +20,7 @@
"check:types": "tsgo -p tsconfig.json --noEmit && tsgo -p adapters/edit/tsconfig.json --noEmit && tsgo -p scripts/tsconfig.json --noEmit",
"lint": "biome lint .",
"serve": "bun run src/server.ts",
"bench:edit": "bun adapters/edit/cli.ts",
"dev": "bun --hot src/server.ts",
"test": "bun test"
},