From 4e5f480f142c6a34ab373fec81a20cb1033cafad Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 30 Jul 2026 06:14:45 +0200 Subject: [PATCH] feat(metaharness/adapters): implemented interactive cli runner and reporting - Added comprehensive CLI arguments for thinking levels, timeouts, concurrency, and report formatting. - Implemented real-time interactive terminal progress rendering and runtime statistics summary. - Enhanced runner logic with tokenizer-based hashline operation detection and aggregated reporting. - Added bench:edit script to package configuration and updated default report output paths. --- packages/metaharness/adapters/edit/cli.ts | 352 +++++++++++++++--- .../adapters/edit/live-progress.ts | 241 ++++++++++++ packages/metaharness/adapters/edit/report.ts | 16 +- packages/metaharness/adapters/edit/runner.ts | 118 ++++-- packages/metaharness/package.json | 1 + 5 files changed, 650 insertions(+), 78 deletions(-) mode change 100644 => 100755 packages/metaharness/adapters/edit/cli.ts create mode 100644 packages/metaharness/adapters/edit/live-progress.ts diff --git a/packages/metaharness/adapters/edit/cli.ts b/packages/metaharness/adapters/edit/cli.ts old mode 100644 new mode 100755 index eff9e91de..cb3fa3bc5 --- a/packages/metaharness/adapters/edit/cli.ts +++ b/packages/metaharness/adapters/edit/cli.ts @@ -1,97 +1,363 @@ #!/usr/bin/env bun -/** Manager-owned executable adapter for the TypeScript edit benchmark. */ +/** + * Executable adapter for the TypeScript edit benchmark. + * + * Two audiences share this entry point: + * - The metaharness manager spawns it headless (`--model … --output result.json`) + * and consumes only the continuously-rewritten report file. + * - Humans run it directly and get the interactive experience: a config + * banner, a live progress bar with per-run failure diffs (stderr, TTY-aware), + * a runtime-stats summary, and a markdown or JSON report. + */ import * as fs from "node:fs/promises"; import * as path from "node:path"; import { parseArgs } from "node:util"; -import { TempDir } from "@oh-my-pi/pi-utils"; -import { loadTasksFromDir } from "@oh-my-pi/typescript-edit-benchmark/tasks"; -import { generateJsonReport } from "./report"; -import { type BenchmarkConfig, runBenchmark } from "./runner"; +import { type ResolvedThinkingLevel, ThinkingLevel } from "@oh-my-pi/pi-agent-core"; +import { THINKING_EFFORTS } from "@oh-my-pi/pi-ai"; +import { postmortem, TempDir } from "@oh-my-pi/pi-utils"; +import { loadTasksFromDir, validateFixturesFromDir } from "@oh-my-pi/typescript-edit-benchmark/tasks"; +import { LiveProgress } from "./live-progress"; +import { generateJsonReport, generateReport } from "./report"; +import { type BenchmarkConfig, type BenchmarkResult, buildBenchmarkResult, runBenchmark } from "./runner"; const EDIT_PACKAGE = path.resolve(import.meta.dir, "..", "..", "..", "typescript-edit-benchmark"); +const RUNS_DIR = path.resolve(import.meta.dir, "..", "..", "..", "..", "runs"); -async function extractFixtures(): Promise<{ dir: string; temp: TempDir }> { +type ReportFormat = "markdown" | "json"; + +function log(line = ""): void { + process.stderr.write(`${line}\n`); +} + +function fail(message: string): never { + throw new Error(message); +} + +function parseThinkingLevel(value: string): ResolvedThinkingLevel { + if ([ThinkingLevel.Off, ...THINKING_EFFORTS].includes(value as ResolvedThinkingLevel)) { + return value as ResolvedThinkingLevel; + } + return fail(`Invalid thinking level: ${value}. Valid levels: ${[ThinkingLevel.Off, ...THINKING_EFFORTS].join(", ")}`); +} + +function parsePositiveInt(raw: string | undefined, flag: string, fallback: number): number { + if (raw === undefined) return fallback; + const parsed = Number.parseInt(raw, 10); + if (Number.isNaN(parsed) || parsed < 1) fail(`Invalid ${flag} value: ${raw}. Must be a positive integer.`); + return parsed; +} + +function parseFuzzy(raw: string | undefined): boolean | "auto" | undefined { + if (raw === undefined) return undefined; + if (raw === "auto") return "auto"; + if (raw === "true" || raw === "1") return true; + if (raw === "false" || raw === "0") return false; + return fail(`Invalid --edit-fuzzy: ${raw}. Must be true, false, 1, 0, or auto.`); +} + +function parseFuzzyThreshold(raw: string | undefined): number | "auto" | undefined { + if (raw === undefined) return undefined; + if (raw === "auto") return "auto"; + const parsed = Number.parseFloat(raw); + if (Number.isNaN(parsed) || parsed < 0 || parsed > 1) { + fail(`Invalid --edit-fuzzy-threshold: ${raw}. Must be 0-1 or auto.`); + } + return parsed; +} + +function resolveFormat(explicit: string | undefined, outputPath: string): ReportFormat { + if (explicit === "markdown" || explicit === "json") return explicit; + if (explicit !== undefined) fail(`Invalid --format: ${explicit}. Must be markdown or json.`); + return outputPath.endsWith(".md") || outputPath.endsWith(".markdown") ? "markdown" : "json"; +} + +function generateReportFilename(config: BenchmarkConfig): string { + const modelName = config.model + .split("/") + .pop()! + .replace(/[^a-zA-Z0-9-]/g, "_"); + const variant = config.editVariant ?? "auto"; + const timestamp = new Date().toISOString().replace(/:/g, "-").replace(/\..+$/, "Z"); + return path.join(RUNS_DIR, `${modelName}_${variant}_${timestamp}.md`); +} + +/** Sibling `.dump` directory for an output path, timestamped on collision. */ +async function resolveConversationDumpDir(outputPath: string): Promise { + const parsed = path.parse(outputPath); + const preferred = path.join(parsed.dir, `${parsed.name}.dump`); + try { + await fs.stat(preferred); + } catch { + return preferred; + } + const timestamp = new Date().toISOString().replace(/[:.]/g, "-"); + return path.join(parsed.dir, `${parsed.name}.${timestamp}.dump`); +} + +interface ResolvedFixtures { + dir: string; + cleanup?: () => Promise; +} + +/** Resolve `--fixtures` (directory or tarball; default: the built-in tarball). */ +async function resolveFixtures(fixturesArg: string | undefined): Promise { + const source = fixturesArg ?? path.join(EDIT_PACKAGE, "fixtures.tar.gz"); + if (!source.endsWith(".tar.gz") && !source.endsWith(".tgz")) { + return { dir: source }; + } const temp = await TempDir.create("@metaharness-edit-fixtures-"); - const archive = new Bun.Archive(await Bun.file(path.join(EDIT_PACKAGE, "fixtures.tar.gz")).arrayBuffer()); + const archive = new Bun.Archive(await Bun.file(source).arrayBuffer()); for (const [filePath, file] of await archive.files()) { await Bun.write(path.join(temp.path(), filePath), file); } const entries = await fs.readdir(temp.path(), { withFileTypes: true }); const directories = entries.filter(entry => entry.isDirectory()); const files = entries.filter(entry => entry.isFile()); - return { dir: directories.length === 1 && files.length === 0 ? path.join(temp.path(), directories[0]!.name) : temp.path(), temp }; + const dir = directories.length === 1 && files.length === 0 ? path.join(temp.path(), directories[0]!.name) : temp.path(); + return { dir, cleanup: () => temp.remove() }; } -/** Execute an edit benchmark and continuously materialize its normalized source artifact. */ +function printUsage(): void { + log(` +Edit Benchmark - Evaluate patch application success rates + +Usage: + bun adapters/edit/cli.ts --model [options] + +Options: + --model Provider/model ID, e.g. anthropic/claude-sonnet-4 (required) + --provider Override provider (auto-detected from model prefix) + --thinking Thinking level: off, minimal, low, medium, high, xhigh, max + --runs Runs per task (default: 1) + --timeout Timeout per run in ms (default: 120000) + --connection-timeout Timeout for first event before fast-retry (default: 30000) + --max-turns Max turns per attempt before failing (default: 30) + --task-concurrency Max tasks to run in parallel (default: 32) + --tasks Comma-separated task IDs to run (default: sampled) + --max-tasks Max tasks to sample evenly (default: 80, 0 = all) + --fixtures Fixtures directory or .tar.gz archive (default: built-in) + --edit-variant Edit variant, e.g. hashline, replace, apply_patch (default: auto) + --edit-fuzzy Fuzzy matching: true, false, auto + --edit-fuzzy-threshold Fuzzy threshold 0-1 or auto + --guided Include an authoritative suggested edit payload + --max-attempts Max prompt attempts per run (default: 1) + --no-early-stop-on-match Don't short-circuit when output matches expected + --output Report file (default: runs/__.md) + --format Report format: markdown, json (default: by extension) + --check-fixtures Validate fixtures and exit + --quiet Suppress the live progress view + --list Print task ids as JSON and exit + --help Show this help message + +Examples: + # Full run against the default sample of 80 tasks + bun adapters/edit/cli.ts --model anthropic/claude-sonnet-4 + + # Compare edit variants + bun adapters/edit/cli.ts --model openai/gpt-5 --edit-variant hashline --output hashline.md + bun adapters/edit/cli.ts --model openai/gpt-5 --edit-variant apply_patch --output apply_patch.md + + # Specific tasks, more runs + bun adapters/edit/cli.ts --model anthropic/claude-sonnet-4 --tasks logic-flip-strict-equality-001 --runs 5 +`); +} + +/** Execute an edit benchmark and continuously materialize its report artifact. */ export async function main(argv = process.argv.slice(2)): Promise { const { values } = parseArgs({ args: argv, options: { model: { type: "string" }, - output: { type: "string" }, - "max-tasks": { type: "string", default: "80" }, - tasks: { type: "string" }, - "task-concurrency": { type: "string", default: "32" }, + provider: { type: "string" }, + thinking: { type: "string" }, runs: { type: "string", default: "1" }, + timeout: { type: "string", default: "120000" }, + "connection-timeout": { type: "string", default: "30000" }, + "max-turns": { type: "string", default: "30" }, + "task-concurrency": { type: "string", default: "32" }, + tasks: { type: "string" }, + "max-tasks": { type: "string", default: "80" }, + fixtures: { type: "string" }, + "edit-variant": { type: "string" }, + "edit-fuzzy": { type: "string" }, + "edit-fuzzy-threshold": { type: "string" }, + guided: { type: "boolean", default: false }, + "max-attempts": { type: "string", default: "1" }, + "no-op-retry-limit": { type: "string", default: "2" }, + "max-timeout-retries": { type: "string", default: "3" }, + "max-provider-retries": { type: "string", default: "3" }, + "mutation-scope-window": { type: "string", default: "20" }, + "no-early-stop-on-match": { type: "boolean", default: false }, + output: { type: "string" }, + format: { type: "string" }, + "check-fixtures": { type: "boolean", default: false }, + quiet: { type: "boolean", default: false }, list: { type: "boolean", default: false }, + help: { type: "boolean", default: false }, }, strict: true, }); - const fixtures = await extractFixtures(); + + if (values.help) { + printUsage(); + return; + } + + const fixtures = await resolveFixtures(values.fixtures); try { + if (values["check-fixtures"]) { + const issues = await validateFixturesFromDir(fixtures.dir); + if (issues.length === 0) { + log("Fixtures OK"); + return; + } + log("Fixture validation failed:"); + for (const issue of issues) log(` - ${issue.taskId}: ${issue.message}`); + process.exitCode = 1; + return; + } + let tasks = await loadTasksFromDir(fixtures.dir); if (values.list) { process.stdout.write(`${JSON.stringify(tasks.map(task => ({ id: task.id, name: task.name })))}\n`); return; } - if (!values.model || !values.output) throw new Error("edit adapter requires --model and --output"); + if (!values.model) fail("edit adapter requires --model (see --help)"); + if (values.tasks) { const selected = new Set(values.tasks.split(",").map(value => value.trim())); tasks = tasks.filter(task => selected.has(task.id)); - if (tasks.length !== selected.size) throw new Error("one or more edit task ids were not found"); + if (tasks.length !== selected.size) fail("one or more edit task ids were not found (see --list)"); } else { const limit = Number(values["max-tasks"]); if (limit > 0 && tasks.length > limit) { + // Deterministic even sampling across the id-sorted list keeps + // mutation-category coverage representative. const sorted = tasks.slice().sort((a, b) => a.id.localeCompare(b.id)); const step = sorted.length / limit; tasks = Array.from({ length: limit }, (_, index) => sorted[Math.floor(index * step)]!); } } - const slash = values.model.indexOf("/"); + + const model = values.model; + const slash = model.indexOf("/"); const config: BenchmarkConfig = { - provider: slash === -1 ? "anthropic" : values.model.slice(0, slash), - model: values.model, - runsPerTask: Number(values.runs), - timeout: 120_000, - connectionTimeout: 30_000, - maxTurns: 30, - taskConcurrency: Number(values["task-concurrency"]), - guided: false, - maxAttempts: 1, - noOpRetryLimit: 2, - maxTimeoutRetries: 3, - maxProviderFailureRetries: 3, - mutationScopeWindow: 20, - conversationDumpDir: path.join(path.dirname(values.output), "result.dump"), + provider: values.provider ?? (slash === -1 ? "anthropic" : model.slice(0, slash)), + model, + ...(values.thinking === undefined ? {} : { thinkingLevel: parseThinkingLevel(values.thinking) }), + runsPerTask: parsePositiveInt(values.runs, "--runs", 1), + timeout: parsePositiveInt(values.timeout, "--timeout", 120_000), + connectionTimeout: parsePositiveInt(values["connection-timeout"], "--connection-timeout", 30_000), + maxTurns: parsePositiveInt(values["max-turns"], "--max-turns", 30), + taskConcurrency: parsePositiveInt(values["task-concurrency"], "--task-concurrency", 32), + guided: values.guided, + maxAttempts: parsePositiveInt(values["max-attempts"], "--max-attempts", 1), + noOpRetryLimit: parsePositiveInt(values["no-op-retry-limit"], "--no-op-retry-limit", 2), + maxTimeoutRetries: parsePositiveInt(values["max-timeout-retries"], "--max-timeout-retries", 3), + maxProviderFailureRetries: parsePositiveInt(values["max-provider-retries"], "--max-provider-retries", 3), + mutationScopeWindow: parsePositiveInt(values["mutation-scope-window"], "--mutation-scope-window", 20), inProcess: true, - earlyStopOnMatch: true, + earlyStopOnMatch: !values["no-early-stop-on-match"], }; - let writes = Promise.resolve(); - const result = await runBenchmark(tasks, config, undefined, snapshot => { - writes = writes.then(async () => { - await Bun.write(values.output!, generateJsonReport(snapshot)); - }); + const editVariant = values["edit-variant"]; + if (editVariant) config.editVariant = editVariant; + const editFuzzy = parseFuzzy(values["edit-fuzzy"]); + if (editFuzzy !== undefined) config.editFuzzy = editFuzzy; + const editFuzzyThreshold = parseFuzzyThreshold(values["edit-fuzzy-threshold"]); + if (editFuzzyThreshold !== undefined) config.editFuzzyThreshold = editFuzzyThreshold; + + let outputPath = values.output; + if (outputPath === undefined) { + await fs.mkdir(RUNS_DIR, { recursive: true }); + outputPath = generateReportFilename(config); + } + const format = resolveFormat(values.format, outputPath); + config.conversationDumpDir = await resolveConversationDumpDir(outputPath); + + log("Edit Benchmark"); + log("=============="); + log(`Provider: ${config.provider}`); + log(`Model: ${config.model}`); + if (config.thinkingLevel) log(`Thinking: ${config.thinkingLevel}`); + log(`Runs per task: ${config.runsPerTask}`); + log(`Timeout: ${config.timeout}ms`); + log(`Task concurrency: ${config.taskConcurrency}`); + log(`Guided mode: ${config.guided ? "enabled" : "disabled"}`); + if (config.editVariant) log(`Edit variant: ${config.editVariant}`); + if (config.editFuzzy !== undefined) log(`Edit fuzzy: ${config.editFuzzy}`); + if (config.editFuzzyThreshold !== undefined) log(`Edit fuzzy threshold: ${config.editFuzzyThreshold}`); + log(`Tasks: ${tasks.length}`); + log(`Report: ${outputPath}`); + log(`Conversation dumps: ${config.conversationDumpDir}`); + log(); + + const renderReport = (result: BenchmarkResult): string => + format === "json" ? generateJsonReport(result) : generateReport(result); + + const progress = values.quiet ? undefined : new LiveProgress(tasks.length * config.runsPerTask, config.runsPerTask); + let latestResult = buildBenchmarkResult({ + tasks, + config, + resultsByTask: new Map(), + startTime: new Date().toISOString(), }); + let writes = Promise.resolve(); + const queueReportWrite = (result: BenchmarkResult) => { + writes = writes.then(async () => { + await Bun.write(outputPath, renderReport(result)); + }); + }; + // A killed/interrupted run still leaves a usable partial report behind. + const unregisterCleanup = postmortem.register("edit-benchmark-report", async reason => { + if (reason === postmortem.Reason.EXIT) return; + progress?.finish(); + log("Benchmark interrupted; writing partial report..."); + await Bun.write(outputPath, renderReport(latestResult)); + await fixtures.cleanup?.(); + }); + + const result = await runBenchmark( + tasks, + config, + event => progress?.handleEvent(event), + snapshot => { + latestResult = snapshot; + queueReportWrite(snapshot); + }, + ); + latestResult = result; + progress?.finish(); + queueReportWrite(result); await writes; - await Bun.write(values.output, generateJsonReport(result)); + unregisterCleanup(); + + log(); + log("Benchmark complete!"); + log( + ` Task success rate (best of ${config.runsPerTask}): ${(result.summary.taskSuccessRate * 100).toFixed(1)}% (${result.summary.successfulTasks}/${result.summary.totalTasks})`, + ); + log(` Total tokens (best): ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`); + log( + ` Tokens/task (best): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total} reasoning=${result.summary.avgTokensPerTask.reasoning}`, + ); + if (result.summary.ghostRuns > 0) log(` Ghost runs (0/0/0): ${result.summary.ghostRuns}`); + if (result.summary.timeoutRuns > 0) log(` Timeout runs: ${result.summary.timeoutRuns}`); + log(`Report written to: ${outputPath}`); } finally { - await fixtures.temp.remove(); + await fixtures.cleanup?.(); } } if (import.meta.main) { - main().catch(error => { - process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`); - process.exitCode = 1; - }); + main() + .then(async () => { + // In-process benchmark runs can leave provider keep-alive sockets and + // background AgentSession timers alive after the report is written. + // Treat the final report as the CLI boundary. + await postmortem.quit(Number(process.exitCode ?? 0)); + }) + .catch(async error => { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`); + await postmortem.quit(1); + }); } diff --git a/packages/metaharness/adapters/edit/live-progress.ts b/packages/metaharness/adapters/edit/live-progress.ts new file mode 100644 index 000000000..efdc65bc7 --- /dev/null +++ b/packages/metaharness/adapters/edit/live-progress.ts @@ -0,0 +1,241 @@ +/** + * Live terminal renderer for edit benchmark runs. + * + * Thin view over {@link runBenchmark}'s `onProgress` stream: a single + * in-place progress line (bar, pass rates, token/latency averages, in-flight + * count) on a TTY, plain per-run lines otherwise, plus inline colored diffs + * for failed runs and a runtime-stats block at the end. Renders to stderr so + * stdout stays free for machine output (`--list` JSON), and the report file + * remains the only artifact the manager consumes. + */ +import { percentile, type ProgressEvent } from "./runner"; + +const ANSI = { + reset: "\x1b[0m", + bold: "\x1b[1m", + dim: "\x1b[2m", + red: "\x1b[31m", + green: "\x1b[32m", + yellow: "\x1b[33m", + cyan: "\x1b[36m", +} as const; + +function rateColor(percent: number): string { + if (percent >= 80) return ANSI.green; + if (percent >= 50) return ANSI.yellow; + return ANSI.red; +} + +export class LiveProgress { + readonly #totalRuns: number; + readonly #runsPerTask: number; + readonly #isTty: boolean; + readonly #colors: boolean; + #started = 0; + #completed = 0; + #success = 0; + #totalInput = 0; + #totalOutput = 0; + #totalDuration = 0; + #totalReads = 0; + #totalEdits = 0; + #totalWrites = 0; + #totalEditSuccesses = 0; + #totalToolInputChars = 0; + #indentScores: number[] = []; + #inputTokens: number[] = []; + #outputTokens: number[] = []; + #totalTokens: number[] = []; + #oneShotSuccessTokens: number[] = []; + #lastLineLength = 0; + + constructor(totalRuns: number, runsPerTask: number) { + this.#totalRuns = totalRuns; + this.#runsPerTask = runsPerTask; + this.#isTty = Boolean(process.stderr.isTTY); + this.#colors = this.#isTty && !process.env.NO_COLOR; + } + + #paint(code: string, text: string): string { + return this.#colors ? `${code}${text}${ANSI.reset}` : text; + } + + #log(line: string): void { + process.stderr.write(`${line}\n`); + } + + handleEvent(event: ProgressEvent): void { + if (event.status === "started") { + this.#started += 1; + if (!this.#isTty) { + this.#log(` [${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} started...`); + } + this.#renderLine(); + return; + } + + this.#completed += 1; + if (event.result) { + if (event.result.success) { + this.#success += 1; + } + if (event.result.success && event.runIndex === 0) { + this.#oneShotSuccessTokens.push(event.result.tokens.total); + } + this.#totalInput += event.result.tokens.input; + this.#totalOutput += event.result.tokens.output; + this.#inputTokens.push(event.result.tokens.input); + this.#outputTokens.push(event.result.tokens.output); + this.#totalTokens.push(event.result.tokens.total); + this.#totalDuration += event.result.duration; + this.#totalReads += event.result.toolCalls.read; + this.#totalEdits += event.result.toolCalls.edit; + this.#totalWrites += event.result.toolCalls.write; + this.#totalEditSuccesses += event.result.toolCalls.editSuccesses; + this.#totalToolInputChars += event.result.toolCalls.totalInputChars; + if (typeof event.result.indentScore === "number") { + this.#indentScores.push(event.result.indentScore); + } + } + + const result = event.result; + if (result && !result.success && result.error) { + this.#flushLine(); + const header = this.#paint( + ANSI.red, + `[${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} failed:`, + ); + this.#log(` ${header} ${result.error}`); + if (result.diff) { + const changeLines = result.diff + .split("\n") + .filter(line => /^[-+@]/.test(line) && !/^(---|\+\+\+)/.test(line)); + const maxLines = 40; + for (const line of changeLines.slice(0, maxLines)) { + let color: string | undefined; + if (line.startsWith("@@")) color = ANSI.cyan; + else if (line.startsWith("-")) color = ANSI.red; + else if (line.startsWith("+")) color = ANSI.green; + this.#log(` ${color ? this.#paint(color, line) : line}`); + } + if (changeLines.length > maxLines) { + this.#log(this.#paint(ANSI.dim, ` ... (${changeLines.length - maxLines} more change lines)`)); + } + } + } + + if (result?.editFailures && result.editFailures.length > 0) { + this.#flushLine(); + for (const [i, failure] of result.editFailures.entries()) { + const args = (failure.args ?? {}) as Record; + const target = + typeof args.path === "string" ? args.path : typeof args.file === "string" ? args.file : undefined; + const op = typeof args.operation === "string" ? args.operation : undefined; + const oneLine = failure.error.replace(/\s+/g, " ").trim(); + const clipped = oneLine.length > 240 ? `${oneLine.slice(0, 237)}...` : oneLine; + const tag = this.#paint(ANSI.yellow, `[${event.taskId}] schema #${i + 1}`); + const metaParts = [op, target].filter((value): value is string => Boolean(value)); + const meta = metaParts.length > 0 ? this.#paint(ANSI.dim, metaParts.join(" ")) : ""; + this.#log(` ${tag}${meta ? ` ${meta}` : ""} ${clipped}`); + if (failure.rawBlock) { + const rawLine = failure.rawBlock.replace(/\s+/g, " ").trim(); + const clippedRaw = rawLine.length > 240 ? `${rawLine.slice(0, 237)}...` : rawLine; + this.#log(` ${this.#paint(ANSI.dim, "raw")} ${clippedRaw}`); + } + } + } + + if (!this.#isTty) { + const status = event.result?.success ? "completed" : "failed"; + this.#log(` [${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} ${status}`); + } + + this.#renderLine(); + } + + finish(): void { + this.#flushLine(); + this.#printSummary(); + } + + #printSummary(): void { + const n = this.#completed; + const denom = n || 1; + + const successRate = (this.#success / denom) * 100; + const editSuccessRate = this.#totalEdits > 0 ? (this.#totalEditSuccesses / this.#totalEdits) * 100 : 100; + const avgIndent = + this.#indentScores.length > 0 ? this.#indentScores.reduce((a, b) => a + b, 0) / this.#indentScores.length : 0; + + this.#log(""); + this.#log(this.#paint(ANSI.bold, "Runtime Stats:")); + this.#log( + ` Task success: ${this.#paint(rateColor(successRate), `${successRate.toFixed(1)}% (${this.#success}/${n})`)}`, + ); + this.#log( + ` Edit success: ${this.#paint(rateColor(editSuccessRate), `${editSuccessRate.toFixed(1)}% (${this.#totalEditSuccesses}/${this.#totalEdits})`)}`, + ); + this.#log(` Avg indent score: ${avgIndent.toFixed(2)}`); + this.#log(` Tool calls: read=${this.#totalReads} edit=${this.#totalEdits} write=${this.#totalWrites}`); + this.#log(` Tool input chars: ${this.#totalToolInputChars.toLocaleString()}`); + const fmtTokens = (samples: number[]): string => { + if (samples.length === 0) return "mean=0 median=0 p1=0 p99=0"; + const sorted = [...samples].sort((a, b) => a - b); + const mean = Math.round(sorted.reduce((a, b) => a + b, 0) / sorted.length); + return `mean=${mean} median=${Math.round(percentile(sorted, 50))} p1=${Math.round(percentile(sorted, 1))} p99=${Math.round(percentile(sorted, 99))}`; + }; + this.#log(` Tokens/task in: ${fmtTokens(this.#inputTokens)}`); + this.#log(` Tokens/task out: ${fmtTokens(this.#outputTokens)}`); + this.#log(` Tokens/task tot: ${fmtTokens(this.#totalTokens)}`); + this.#log(` Tokens/task (one-shot successes): ${fmtTokens(this.#oneShotSuccessTokens)}`); + this.#log(` Avg time/task: ${Math.round(this.#totalDuration / denom)}ms`); + } + + #renderLine(): void { + if (!this.#isTty) { + return; + } + const successRate = this.#completed > 0 ? (this.#success / this.#completed) * 100 : 0; + const editRate = this.#totalEdits > 0 ? (this.#totalEditSuccesses / this.#totalEdits) * 100 : 100; + const avgInput = this.#completed > 0 ? Math.round(this.#totalInput / this.#completed) : 0; + const avgOutput = this.#completed > 0 ? Math.round(this.#totalOutput / this.#completed) : 0; + const avgDuration = this.#completed > 0 ? Math.round(this.#totalDuration / this.#completed) : 0; + const inFlight = this.#started - this.#completed; + const bar = this.#renderBar(this.#completed, this.#totalRuns, 20); + const progress = this.#paint(ANSI.bold, `${this.#completed}/${this.#totalRuns}`); + const taskCol = `task=${this.#paint(rateColor(successRate), `${successRate.toFixed(0)}%`)}`; + const editCol = `edit=${this.#paint(rateColor(editRate), `${editRate.toFixed(0)}%`)}`; + const tokCol = this.#paint(ANSI.dim, `tok=${avgInput}/${avgOutput}`); + const durCol = this.#paint(ANSI.dim, `${avgDuration}ms`); + const rewCol = this.#paint(ANSI.dim, `r/e/w=${this.#totalReads}/${this.#totalEdits}/${this.#totalWrites}`); + const flyCol = `fly=${this.#paint(ANSI.cyan, String(inFlight))}`; + const line = ` ${bar} ${progress} ${taskCol} ${editCol} ${tokCol} ${durCol} ${rewCol} ${flyCol}`; + this.#writeLine(line); + } + + #renderBar(done: number, total: number, width: number): string { + const ratio = total === 0 ? 0 : done / total; + const filled = Math.round(ratio * width); + const empty = Math.max(0, width - filled); + const filledPart = this.#paint(ANSI.green, "#".repeat(filled)); + const emptyPart = this.#paint(ANSI.dim, "-".repeat(empty)); + return `[${filledPart}${emptyPart}]`; + } + + #writeLine(line: string): void { + const lineWidth = Bun.stringWidth(line); + const pad = this.#lastLineLength > lineWidth ? " ".repeat(this.#lastLineLength - lineWidth) : ""; + process.stderr.write(`\r${line}${pad}`); + this.#lastLineLength = lineWidth; + } + + #flushLine(): void { + if (!this.#isTty) { + return; + } + if (this.#lastLineLength > 0) { + process.stderr.write(`\r${" ".repeat(this.#lastLineLength)}\r`); + this.#lastLineLength = 0; + } + } +} diff --git a/packages/metaharness/adapters/edit/report.ts b/packages/metaharness/adapters/edit/report.ts index bf0274cd3..5943587bf 100644 --- a/packages/metaharness/adapters/edit/report.ts +++ b/packages/metaharness/adapters/edit/report.ts @@ -212,17 +212,19 @@ export function generateReport(result: BenchmarkResult): string { lines.push(""); if (summary.hashlineEditSubtypes) { - const order = ["set", "set_range", "insert"] as const; - const total = order.reduce((sum, key) => sum + (summary.hashlineEditSubtypes?.[key] ?? 0), 0); + const rows = Object.entries(summary.hashlineEditSubtypes) + .filter(([, count]) => count > 0) + .sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])); + const total = rows.reduce((sum, [, count]) => sum + count, 0); if (total > 0) { - lines.push("### Hashline Edit Subtypes"); + lines.push("### Hashline Op Breakdown"); + lines.push(""); + lines.push("All attempted edit calls across all runs (retries and failed calls included)."); lines.push(""); lines.push("| Operation | Count | % |"); lines.push("|-----------|-------|---|"); - for (const key of order) { - const count = summary.hashlineEditSubtypes[key] ?? 0; - const pct = formatPercent(count / total); - lines.push(`| ${key} | ${count} | ${pct} |`); + for (const [op, count] of rows) { + lines.push(`| \`${op}\` | ${count} | ${formatPercent(count / total)} |`); } lines.push(`| **Total** | **${total}** | 100% |`); lines.push(""); diff --git a/packages/metaharness/adapters/edit/runner.ts b/packages/metaharness/adapters/edit/runner.ts index 47568e6af..397c55caf 100644 --- a/packages/metaharness/adapters/edit/runner.ts +++ b/packages/metaharness/adapters/edit/runner.ts @@ -7,7 +7,12 @@ /// import * as fs from "node:fs"; import * as path from "node:path"; -import { formatHashlineHeader, InMemorySnapshotStore } from "@oh-my-pi/hashline"; +import { + type BlockTarget as HashlineBlockTarget, + formatHashlineHeader, + InMemorySnapshotStore, + Tokenizer as HashlineTokenizer, +} from "@oh-my-pi/hashline"; import type { AgentMessage, ResolvedThinkingLevel, ThinkingLevel } from "@oh-my-pi/pi-agent-core"; import type { Model, ToolExample } from "@oh-my-pi/pi-ai"; import { formatSessionDumpText, RpcClient } from "@oh-my-pi/pi-coding-agent"; @@ -262,7 +267,6 @@ function countEditFailureCategories(runs: TaskRunResult[]): Record { - const counts: Record = Object.fromEntries(HL_SUBTYPES.map(k => [k, 0])); - if (!args || typeof args !== "object") return counts; - const edits = (args as { edits?: unknown[] }).edits; - if (!Array.isArray(edits)) return counts; - for (const edit of edits) { - if (!edit || typeof edit !== "object") continue; - for (const key of HL_SUBTYPES) { - if (key in edit) { - counts[key]++; - break; +// Pure classification — single shared tokenizer is safe. +const HASHLINE_OP_TOKENIZER = new HashlineTokenizer(); + +/** Display label for a hashline op header, e.g. `SWAP.BLK` or `PASTE.POST`. */ +function hashlineOpLabel(target: HashlineBlockTarget): string { + switch (target.kind) { + case "replace": + return "SWAP"; + case "block": + return "SWAP.BLK"; + case "delete": + return "DEL"; + case "delete_block": + return "DEL.BLK"; + case "insert_before": + return "INS.PRE"; + case "insert_after": + return "INS.POST"; + case "bof": + return "INS.HEAD"; + case "eof": + return "INS.TAIL"; + case "insert_after_block": + return "INS.BLK.POST"; + case "copy": + return target.cut ? "CUT" : "COPY"; + case "copy_block": + return target.cut ? "CUT.BLK" : "COPY.BLK"; + case "paste": + switch (target.cursor.kind) { + case "before_anchor": + return "PASTE.PRE"; + case "after_anchor": + return "PASTE.POST"; + case "bof": + return "PASTE.HEAD"; + case "eof": + return "PASTE.TAIL"; } - } + break; + case "paste_after_block": + return "PASTE.BLK.POST"; + case "rem": + return "REM"; + case "move": + return "MV"; } - return counts; +} + +/** + * Count hashline op headers (`SWAP`, `DEL.BLK`, `CUT`, `PASTE.POST`, …) in an + * edit call's patch input. Returns `null` when the args carry no hashline + * patch — non-hashline edit variants and malformed calls contribute nothing. + */ +function countHashlineOps(args: unknown): Record | null { + if (!args || typeof args !== "object" || !("input" in args)) return null; + const input = args.input; + if (typeof input !== "string" || input.length === 0) return null; + const counts: Record = {}; + for (const token of HASHLINE_OP_TOKENIZER.tokenizeAll(input)) { + if (token.kind !== "op-block") continue; + const label = hashlineOpLabel(token.target); + counts[label] = (counts[label] ?? 0) + 1; + } + return Object.keys(counts).length > 0 ? counts : null; } async function collectOriginalFileContents(cwd: string, files: string[]): Promise> { @@ -829,7 +883,7 @@ export interface TaskRunResult { editFailures: EditFailure[]; editWarnings: string[]; editAutocorrectCount: number; - /** Hashline edit subtype counts (replaceLine, replaceLines, etc.) — only when editVariant is hashline */ + /** Hashline op counts (`SWAP`, `DEL.BLK`, `CUT`, …) — present when edit calls carried hashline patches. */ hashlineEditSubtypes?: Record; mutationIntentMatched?: boolean; mutationIntentReason?: string; @@ -945,7 +999,7 @@ export interface BenchmarkSummary { mutationIntentMatchRate?: number; /** Edit failure categories across all runs. */ editFailureCategories: Record; - /** Hashline edit subtype totals across all runs — only when editVariant is hashline. */ + /** Hashline op totals across all runs — present when any run's edit calls carried hashline patches. */ hashlineEditSubtypes?: Record; } @@ -1037,7 +1091,7 @@ async function runSingleTask( editAutocorrects: 0, totalInputChars: 0, }; - const hashlineSubtypes: Record = Object.fromEntries(HL_SUBTYPES.map(k => [k, 0])); + const hashlineSubtypes: Record = {}; const logFile = path.join(TMP, `run-${task.id}-${runIndex}.jsonl`); const logEvent = async (event: unknown) => { @@ -1271,10 +1325,10 @@ async function runSingleTask( const pendingEdit = pendingEdits.get(e.toolCallId) ?? { args: null }; const args = pendingEdit.args; pendingEdits.delete(e.toolCallId); - if (config.editVariant === "hashline" && args) { - const counts = countHashlineEditSubtypes(args); - for (const key of HL_SUBTYPES) { - hashlineSubtypes[key] += counts[key]; + const hashlineOpCounts = countHashlineOps(args); + if (hashlineOpCounts) { + for (const key in hashlineOpCounts) { + hashlineSubtypes[key] = (hashlineSubtypes[key] ?? 0) + hashlineOpCounts[key]; } } if (e.isError) { @@ -1422,7 +1476,7 @@ async function runSingleTask( editFailures, editWarnings, editAutocorrectCount, - hashlineEditSubtypes: config.editVariant === "hashline" ? hashlineSubtypes : undefined, + hashlineEditSubtypes: Object.keys(hashlineSubtypes).length > 0 ? hashlineSubtypes : undefined, mutationIntentMatched: mutationIntentValidation?.matched, mutationIntentReason: mutationIntentValidation?.reason, timeoutTelemetry, @@ -1936,12 +1990,20 @@ export function buildBenchmarkResult(params: { 0, ); const editFailureCategories = countEditFailureCategories(nonGhostRuns); + // Op counts aggregate every attempted edit call across ALL runs (retries and + // failed calls included) — deliberately: the mix shows what the model reaches + // for, not just what landed. Best-run-only would hide exactly the flailing + // (failed SWAPs before a successful retry) this table exists to expose. + const hashlineEditSubtypeTotals: Record = {}; + for (const run of allRuns) { + const runOps = run.hashlineEditSubtypes; + if (!runOps) continue; + for (const key in runOps) { + hashlineEditSubtypeTotals[key] = (hashlineEditSubtypeTotals[key] ?? 0) + runOps[key]; + } + } const hashlineEditSubtypes: Record | undefined = - params.config.editVariant === "hashline" - ? Object.fromEntries( - HL_SUBTYPES.map(key => [key, allRuns.reduce((sum, r) => sum + (r.hashlineEditSubtypes?.[key] ?? 0), 0)]), - ) - : undefined; + Object.keys(hashlineEditSubtypeTotals).length > 0 ? hashlineEditSubtypeTotals : undefined; // Primary aggregates run over the *best* run of each completed task. const bestRuns: TaskRunResult[] = []; diff --git a/packages/metaharness/package.json b/packages/metaharness/package.json index c41e98cb0..cbbedf4fc 100644 --- a/packages/metaharness/package.json +++ b/packages/metaharness/package.json @@ -20,6 +20,7 @@ "check:types": "tsgo -p tsconfig.json --noEmit && tsgo -p adapters/edit/tsconfig.json --noEmit && tsgo -p scripts/tsconfig.json --noEmit", "lint": "biome lint .", "serve": "bun run src/server.ts", + "bench:edit": "bun adapters/edit/cli.ts", "dev": "bun --hot src/server.ts", "test": "bun test" },