feat(metaharness/adapters): implemented interactive cli runner and reporting
- Added comprehensive CLI arguments for thinking levels, timeouts, concurrency, and report formatting. - Implemented real-time interactive terminal progress rendering and runtime statistics summary. - Enhanced runner logic with tokenizer-based hashline operation detection and aggregated reporting. - Added bench:edit script to package configuration and updated default report output paths.
This commit is contained in:
Regular → Executable
+309
-43
@@ -1,97 +1,363 @@
|
||||
#!/usr/bin/env bun
|
||||
/** Manager-owned executable adapter for the TypeScript edit benchmark. */
|
||||
/**
|
||||
* Executable adapter for the TypeScript edit benchmark.
|
||||
*
|
||||
* Two audiences share this entry point:
|
||||
* - The metaharness manager spawns it headless (`--model … --output result.json`)
|
||||
* and consumes only the continuously-rewritten report file.
|
||||
* - Humans run it directly and get the interactive experience: a config
|
||||
* banner, a live progress bar with per-run failure diffs (stderr, TTY-aware),
|
||||
* a runtime-stats summary, and a markdown or JSON report.
|
||||
*/
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as path from "node:path";
|
||||
import { parseArgs } from "node:util";
|
||||
import { TempDir } from "@oh-my-pi/pi-utils";
|
||||
import { loadTasksFromDir } from "@oh-my-pi/typescript-edit-benchmark/tasks";
|
||||
import { generateJsonReport } from "./report";
|
||||
import { type BenchmarkConfig, runBenchmark } from "./runner";
|
||||
import { type ResolvedThinkingLevel, ThinkingLevel } from "@oh-my-pi/pi-agent-core";
|
||||
import { THINKING_EFFORTS } from "@oh-my-pi/pi-ai";
|
||||
import { postmortem, TempDir } from "@oh-my-pi/pi-utils";
|
||||
import { loadTasksFromDir, validateFixturesFromDir } from "@oh-my-pi/typescript-edit-benchmark/tasks";
|
||||
import { LiveProgress } from "./live-progress";
|
||||
import { generateJsonReport, generateReport } from "./report";
|
||||
import { type BenchmarkConfig, type BenchmarkResult, buildBenchmarkResult, runBenchmark } from "./runner";
|
||||
|
||||
const EDIT_PACKAGE = path.resolve(import.meta.dir, "..", "..", "..", "typescript-edit-benchmark");
|
||||
const RUNS_DIR = path.resolve(import.meta.dir, "..", "..", "..", "..", "runs");
|
||||
|
||||
async function extractFixtures(): Promise<{ dir: string; temp: TempDir }> {
|
||||
type ReportFormat = "markdown" | "json";
|
||||
|
||||
function log(line = ""): void {
|
||||
process.stderr.write(`${line}\n`);
|
||||
}
|
||||
|
||||
function fail(message: string): never {
|
||||
throw new Error(message);
|
||||
}
|
||||
|
||||
function parseThinkingLevel(value: string): ResolvedThinkingLevel {
|
||||
if ([ThinkingLevel.Off, ...THINKING_EFFORTS].includes(value as ResolvedThinkingLevel)) {
|
||||
return value as ResolvedThinkingLevel;
|
||||
}
|
||||
return fail(`Invalid thinking level: ${value}. Valid levels: ${[ThinkingLevel.Off, ...THINKING_EFFORTS].join(", ")}`);
|
||||
}
|
||||
|
||||
function parsePositiveInt(raw: string | undefined, flag: string, fallback: number): number {
|
||||
if (raw === undefined) return fallback;
|
||||
const parsed = Number.parseInt(raw, 10);
|
||||
if (Number.isNaN(parsed) || parsed < 1) fail(`Invalid ${flag} value: ${raw}. Must be a positive integer.`);
|
||||
return parsed;
|
||||
}
|
||||
|
||||
function parseFuzzy(raw: string | undefined): boolean | "auto" | undefined {
|
||||
if (raw === undefined) return undefined;
|
||||
if (raw === "auto") return "auto";
|
||||
if (raw === "true" || raw === "1") return true;
|
||||
if (raw === "false" || raw === "0") return false;
|
||||
return fail(`Invalid --edit-fuzzy: ${raw}. Must be true, false, 1, 0, or auto.`);
|
||||
}
|
||||
|
||||
function parseFuzzyThreshold(raw: string | undefined): number | "auto" | undefined {
|
||||
if (raw === undefined) return undefined;
|
||||
if (raw === "auto") return "auto";
|
||||
const parsed = Number.parseFloat(raw);
|
||||
if (Number.isNaN(parsed) || parsed < 0 || parsed > 1) {
|
||||
fail(`Invalid --edit-fuzzy-threshold: ${raw}. Must be 0-1 or auto.`);
|
||||
}
|
||||
return parsed;
|
||||
}
|
||||
|
||||
function resolveFormat(explicit: string | undefined, outputPath: string): ReportFormat {
|
||||
if (explicit === "markdown" || explicit === "json") return explicit;
|
||||
if (explicit !== undefined) fail(`Invalid --format: ${explicit}. Must be markdown or json.`);
|
||||
return outputPath.endsWith(".md") || outputPath.endsWith(".markdown") ? "markdown" : "json";
|
||||
}
|
||||
|
||||
function generateReportFilename(config: BenchmarkConfig): string {
|
||||
const modelName = config.model
|
||||
.split("/")
|
||||
.pop()!
|
||||
.replace(/[^a-zA-Z0-9-]/g, "_");
|
||||
const variant = config.editVariant ?? "auto";
|
||||
const timestamp = new Date().toISOString().replace(/:/g, "-").replace(/\..+$/, "Z");
|
||||
return path.join(RUNS_DIR, `${modelName}_${variant}_${timestamp}.md`);
|
||||
}
|
||||
|
||||
/** Sibling `.dump` directory for an output path, timestamped on collision. */
|
||||
async function resolveConversationDumpDir(outputPath: string): Promise<string> {
|
||||
const parsed = path.parse(outputPath);
|
||||
const preferred = path.join(parsed.dir, `${parsed.name}.dump`);
|
||||
try {
|
||||
await fs.stat(preferred);
|
||||
} catch {
|
||||
return preferred;
|
||||
}
|
||||
const timestamp = new Date().toISOString().replace(/[:.]/g, "-");
|
||||
return path.join(parsed.dir, `${parsed.name}.${timestamp}.dump`);
|
||||
}
|
||||
|
||||
interface ResolvedFixtures {
|
||||
dir: string;
|
||||
cleanup?: () => Promise<void>;
|
||||
}
|
||||
|
||||
/** Resolve `--fixtures` (directory or tarball; default: the built-in tarball). */
|
||||
async function resolveFixtures(fixturesArg: string | undefined): Promise<ResolvedFixtures> {
|
||||
const source = fixturesArg ?? path.join(EDIT_PACKAGE, "fixtures.tar.gz");
|
||||
if (!source.endsWith(".tar.gz") && !source.endsWith(".tgz")) {
|
||||
return { dir: source };
|
||||
}
|
||||
const temp = await TempDir.create("@metaharness-edit-fixtures-");
|
||||
const archive = new Bun.Archive(await Bun.file(path.join(EDIT_PACKAGE, "fixtures.tar.gz")).arrayBuffer());
|
||||
const archive = new Bun.Archive(await Bun.file(source).arrayBuffer());
|
||||
for (const [filePath, file] of await archive.files()) {
|
||||
await Bun.write(path.join(temp.path(), filePath), file);
|
||||
}
|
||||
const entries = await fs.readdir(temp.path(), { withFileTypes: true });
|
||||
const directories = entries.filter(entry => entry.isDirectory());
|
||||
const files = entries.filter(entry => entry.isFile());
|
||||
return { dir: directories.length === 1 && files.length === 0 ? path.join(temp.path(), directories[0]!.name) : temp.path(), temp };
|
||||
const dir = directories.length === 1 && files.length === 0 ? path.join(temp.path(), directories[0]!.name) : temp.path();
|
||||
return { dir, cleanup: () => temp.remove() };
|
||||
}
|
||||
|
||||
/** Execute an edit benchmark and continuously materialize its normalized source artifact. */
|
||||
function printUsage(): void {
|
||||
log(`
|
||||
Edit Benchmark - Evaluate patch application success rates
|
||||
|
||||
Usage:
|
||||
bun adapters/edit/cli.ts --model <provider/model> [options]
|
||||
|
||||
Options:
|
||||
--model <id> Provider/model ID, e.g. anthropic/claude-sonnet-4 (required)
|
||||
--provider <id> Override provider (auto-detected from model prefix)
|
||||
--thinking <level> Thinking level: off, minimal, low, medium, high, xhigh, max
|
||||
--runs <n> Runs per task (default: 1)
|
||||
--timeout <ms> Timeout per run in ms (default: 120000)
|
||||
--connection-timeout <ms> Timeout for first event before fast-retry (default: 30000)
|
||||
--max-turns <n> Max turns per attempt before failing (default: 30)
|
||||
--task-concurrency <n> Max tasks to run in parallel (default: 32)
|
||||
--tasks <ids> Comma-separated task IDs to run (default: sampled)
|
||||
--max-tasks <n> Max tasks to sample evenly (default: 80, 0 = all)
|
||||
--fixtures <path> Fixtures directory or .tar.gz archive (default: built-in)
|
||||
--edit-variant <v> Edit variant, e.g. hashline, replace, apply_patch (default: auto)
|
||||
--edit-fuzzy <bool> Fuzzy matching: true, false, auto
|
||||
--edit-fuzzy-threshold <n> Fuzzy threshold 0-1 or auto
|
||||
--guided Include an authoritative suggested edit payload
|
||||
--max-attempts <n> Max prompt attempts per run (default: 1)
|
||||
--no-early-stop-on-match Don't short-circuit when output matches expected
|
||||
--output <file> Report file (default: runs/<model>_<variant>_<ts>.md)
|
||||
--format <fmt> Report format: markdown, json (default: by extension)
|
||||
--check-fixtures Validate fixtures and exit
|
||||
--quiet Suppress the live progress view
|
||||
--list Print task ids as JSON and exit
|
||||
--help Show this help message
|
||||
|
||||
Examples:
|
||||
# Full run against the default sample of 80 tasks
|
||||
bun adapters/edit/cli.ts --model anthropic/claude-sonnet-4
|
||||
|
||||
# Compare edit variants
|
||||
bun adapters/edit/cli.ts --model openai/gpt-5 --edit-variant hashline --output hashline.md
|
||||
bun adapters/edit/cli.ts --model openai/gpt-5 --edit-variant apply_patch --output apply_patch.md
|
||||
|
||||
# Specific tasks, more runs
|
||||
bun adapters/edit/cli.ts --model anthropic/claude-sonnet-4 --tasks logic-flip-strict-equality-001 --runs 5
|
||||
`);
|
||||
}
|
||||
|
||||
/** Execute an edit benchmark and continuously materialize its report artifact. */
|
||||
export async function main(argv = process.argv.slice(2)): Promise<void> {
|
||||
const { values } = parseArgs({
|
||||
args: argv,
|
||||
options: {
|
||||
model: { type: "string" },
|
||||
output: { type: "string" },
|
||||
"max-tasks": { type: "string", default: "80" },
|
||||
tasks: { type: "string" },
|
||||
"task-concurrency": { type: "string", default: "32" },
|
||||
provider: { type: "string" },
|
||||
thinking: { type: "string" },
|
||||
runs: { type: "string", default: "1" },
|
||||
timeout: { type: "string", default: "120000" },
|
||||
"connection-timeout": { type: "string", default: "30000" },
|
||||
"max-turns": { type: "string", default: "30" },
|
||||
"task-concurrency": { type: "string", default: "32" },
|
||||
tasks: { type: "string" },
|
||||
"max-tasks": { type: "string", default: "80" },
|
||||
fixtures: { type: "string" },
|
||||
"edit-variant": { type: "string" },
|
||||
"edit-fuzzy": { type: "string" },
|
||||
"edit-fuzzy-threshold": { type: "string" },
|
||||
guided: { type: "boolean", default: false },
|
||||
"max-attempts": { type: "string", default: "1" },
|
||||
"no-op-retry-limit": { type: "string", default: "2" },
|
||||
"max-timeout-retries": { type: "string", default: "3" },
|
||||
"max-provider-retries": { type: "string", default: "3" },
|
||||
"mutation-scope-window": { type: "string", default: "20" },
|
||||
"no-early-stop-on-match": { type: "boolean", default: false },
|
||||
output: { type: "string" },
|
||||
format: { type: "string" },
|
||||
"check-fixtures": { type: "boolean", default: false },
|
||||
quiet: { type: "boolean", default: false },
|
||||
list: { type: "boolean", default: false },
|
||||
help: { type: "boolean", default: false },
|
||||
},
|
||||
strict: true,
|
||||
});
|
||||
const fixtures = await extractFixtures();
|
||||
|
||||
if (values.help) {
|
||||
printUsage();
|
||||
return;
|
||||
}
|
||||
|
||||
const fixtures = await resolveFixtures(values.fixtures);
|
||||
try {
|
||||
if (values["check-fixtures"]) {
|
||||
const issues = await validateFixturesFromDir(fixtures.dir);
|
||||
if (issues.length === 0) {
|
||||
log("Fixtures OK");
|
||||
return;
|
||||
}
|
||||
log("Fixture validation failed:");
|
||||
for (const issue of issues) log(` - ${issue.taskId}: ${issue.message}`);
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
|
||||
let tasks = await loadTasksFromDir(fixtures.dir);
|
||||
if (values.list) {
|
||||
process.stdout.write(`${JSON.stringify(tasks.map(task => ({ id: task.id, name: task.name })))}\n`);
|
||||
return;
|
||||
}
|
||||
if (!values.model || !values.output) throw new Error("edit adapter requires --model and --output");
|
||||
if (!values.model) fail("edit adapter requires --model (see --help)");
|
||||
|
||||
if (values.tasks) {
|
||||
const selected = new Set(values.tasks.split(",").map(value => value.trim()));
|
||||
tasks = tasks.filter(task => selected.has(task.id));
|
||||
if (tasks.length !== selected.size) throw new Error("one or more edit task ids were not found");
|
||||
if (tasks.length !== selected.size) fail("one or more edit task ids were not found (see --list)");
|
||||
} else {
|
||||
const limit = Number(values["max-tasks"]);
|
||||
if (limit > 0 && tasks.length > limit) {
|
||||
// Deterministic even sampling across the id-sorted list keeps
|
||||
// mutation-category coverage representative.
|
||||
const sorted = tasks.slice().sort((a, b) => a.id.localeCompare(b.id));
|
||||
const step = sorted.length / limit;
|
||||
tasks = Array.from({ length: limit }, (_, index) => sorted[Math.floor(index * step)]!);
|
||||
}
|
||||
}
|
||||
const slash = values.model.indexOf("/");
|
||||
|
||||
const model = values.model;
|
||||
const slash = model.indexOf("/");
|
||||
const config: BenchmarkConfig = {
|
||||
provider: slash === -1 ? "anthropic" : values.model.slice(0, slash),
|
||||
model: values.model,
|
||||
runsPerTask: Number(values.runs),
|
||||
timeout: 120_000,
|
||||
connectionTimeout: 30_000,
|
||||
maxTurns: 30,
|
||||
taskConcurrency: Number(values["task-concurrency"]),
|
||||
guided: false,
|
||||
maxAttempts: 1,
|
||||
noOpRetryLimit: 2,
|
||||
maxTimeoutRetries: 3,
|
||||
maxProviderFailureRetries: 3,
|
||||
mutationScopeWindow: 20,
|
||||
conversationDumpDir: path.join(path.dirname(values.output), "result.dump"),
|
||||
provider: values.provider ?? (slash === -1 ? "anthropic" : model.slice(0, slash)),
|
||||
model,
|
||||
...(values.thinking === undefined ? {} : { thinkingLevel: parseThinkingLevel(values.thinking) }),
|
||||
runsPerTask: parsePositiveInt(values.runs, "--runs", 1),
|
||||
timeout: parsePositiveInt(values.timeout, "--timeout", 120_000),
|
||||
connectionTimeout: parsePositiveInt(values["connection-timeout"], "--connection-timeout", 30_000),
|
||||
maxTurns: parsePositiveInt(values["max-turns"], "--max-turns", 30),
|
||||
taskConcurrency: parsePositiveInt(values["task-concurrency"], "--task-concurrency", 32),
|
||||
guided: values.guided,
|
||||
maxAttempts: parsePositiveInt(values["max-attempts"], "--max-attempts", 1),
|
||||
noOpRetryLimit: parsePositiveInt(values["no-op-retry-limit"], "--no-op-retry-limit", 2),
|
||||
maxTimeoutRetries: parsePositiveInt(values["max-timeout-retries"], "--max-timeout-retries", 3),
|
||||
maxProviderFailureRetries: parsePositiveInt(values["max-provider-retries"], "--max-provider-retries", 3),
|
||||
mutationScopeWindow: parsePositiveInt(values["mutation-scope-window"], "--mutation-scope-window", 20),
|
||||
inProcess: true,
|
||||
earlyStopOnMatch: true,
|
||||
earlyStopOnMatch: !values["no-early-stop-on-match"],
|
||||
};
|
||||
let writes = Promise.resolve();
|
||||
const result = await runBenchmark(tasks, config, undefined, snapshot => {
|
||||
writes = writes.then(async () => {
|
||||
await Bun.write(values.output!, generateJsonReport(snapshot));
|
||||
});
|
||||
const editVariant = values["edit-variant"];
|
||||
if (editVariant) config.editVariant = editVariant;
|
||||
const editFuzzy = parseFuzzy(values["edit-fuzzy"]);
|
||||
if (editFuzzy !== undefined) config.editFuzzy = editFuzzy;
|
||||
const editFuzzyThreshold = parseFuzzyThreshold(values["edit-fuzzy-threshold"]);
|
||||
if (editFuzzyThreshold !== undefined) config.editFuzzyThreshold = editFuzzyThreshold;
|
||||
|
||||
let outputPath = values.output;
|
||||
if (outputPath === undefined) {
|
||||
await fs.mkdir(RUNS_DIR, { recursive: true });
|
||||
outputPath = generateReportFilename(config);
|
||||
}
|
||||
const format = resolveFormat(values.format, outputPath);
|
||||
config.conversationDumpDir = await resolveConversationDumpDir(outputPath);
|
||||
|
||||
log("Edit Benchmark");
|
||||
log("==============");
|
||||
log(`Provider: ${config.provider}`);
|
||||
log(`Model: ${config.model}`);
|
||||
if (config.thinkingLevel) log(`Thinking: ${config.thinkingLevel}`);
|
||||
log(`Runs per task: ${config.runsPerTask}`);
|
||||
log(`Timeout: ${config.timeout}ms`);
|
||||
log(`Task concurrency: ${config.taskConcurrency}`);
|
||||
log(`Guided mode: ${config.guided ? "enabled" : "disabled"}`);
|
||||
if (config.editVariant) log(`Edit variant: ${config.editVariant}`);
|
||||
if (config.editFuzzy !== undefined) log(`Edit fuzzy: ${config.editFuzzy}`);
|
||||
if (config.editFuzzyThreshold !== undefined) log(`Edit fuzzy threshold: ${config.editFuzzyThreshold}`);
|
||||
log(`Tasks: ${tasks.length}`);
|
||||
log(`Report: ${outputPath}`);
|
||||
log(`Conversation dumps: ${config.conversationDumpDir}`);
|
||||
log();
|
||||
|
||||
const renderReport = (result: BenchmarkResult): string =>
|
||||
format === "json" ? generateJsonReport(result) : generateReport(result);
|
||||
|
||||
const progress = values.quiet ? undefined : new LiveProgress(tasks.length * config.runsPerTask, config.runsPerTask);
|
||||
let latestResult = buildBenchmarkResult({
|
||||
tasks,
|
||||
config,
|
||||
resultsByTask: new Map(),
|
||||
startTime: new Date().toISOString(),
|
||||
});
|
||||
let writes = Promise.resolve();
|
||||
const queueReportWrite = (result: BenchmarkResult) => {
|
||||
writes = writes.then(async () => {
|
||||
await Bun.write(outputPath, renderReport(result));
|
||||
});
|
||||
};
|
||||
// A killed/interrupted run still leaves a usable partial report behind.
|
||||
const unregisterCleanup = postmortem.register("edit-benchmark-report", async reason => {
|
||||
if (reason === postmortem.Reason.EXIT) return;
|
||||
progress?.finish();
|
||||
log("Benchmark interrupted; writing partial report...");
|
||||
await Bun.write(outputPath, renderReport(latestResult));
|
||||
await fixtures.cleanup?.();
|
||||
});
|
||||
|
||||
const result = await runBenchmark(
|
||||
tasks,
|
||||
config,
|
||||
event => progress?.handleEvent(event),
|
||||
snapshot => {
|
||||
latestResult = snapshot;
|
||||
queueReportWrite(snapshot);
|
||||
},
|
||||
);
|
||||
latestResult = result;
|
||||
progress?.finish();
|
||||
queueReportWrite(result);
|
||||
await writes;
|
||||
await Bun.write(values.output, generateJsonReport(result));
|
||||
unregisterCleanup();
|
||||
|
||||
log();
|
||||
log("Benchmark complete!");
|
||||
log(
|
||||
` Task success rate (best of ${config.runsPerTask}): ${(result.summary.taskSuccessRate * 100).toFixed(1)}% (${result.summary.successfulTasks}/${result.summary.totalTasks})`,
|
||||
);
|
||||
log(` Total tokens (best): ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`);
|
||||
log(
|
||||
` Tokens/task (best): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total} reasoning=${result.summary.avgTokensPerTask.reasoning}`,
|
||||
);
|
||||
if (result.summary.ghostRuns > 0) log(` Ghost runs (0/0/0): ${result.summary.ghostRuns}`);
|
||||
if (result.summary.timeoutRuns > 0) log(` Timeout runs: ${result.summary.timeoutRuns}`);
|
||||
log(`Report written to: ${outputPath}`);
|
||||
} finally {
|
||||
await fixtures.temp.remove();
|
||||
await fixtures.cleanup?.();
|
||||
}
|
||||
}
|
||||
|
||||
if (import.meta.main) {
|
||||
main().catch(error => {
|
||||
process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`);
|
||||
process.exitCode = 1;
|
||||
});
|
||||
main()
|
||||
.then(async () => {
|
||||
// In-process benchmark runs can leave provider keep-alive sockets and
|
||||
// background AgentSession timers alive after the report is written.
|
||||
// Treat the final report as the CLI boundary.
|
||||
await postmortem.quit(Number(process.exitCode ?? 0));
|
||||
})
|
||||
.catch(async error => {
|
||||
process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`);
|
||||
await postmortem.quit(1);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -0,0 +1,241 @@
|
||||
/**
|
||||
* Live terminal renderer for edit benchmark runs.
|
||||
*
|
||||
* Thin view over {@link runBenchmark}'s `onProgress` stream: a single
|
||||
* in-place progress line (bar, pass rates, token/latency averages, in-flight
|
||||
* count) on a TTY, plain per-run lines otherwise, plus inline colored diffs
|
||||
* for failed runs and a runtime-stats block at the end. Renders to stderr so
|
||||
* stdout stays free for machine output (`--list` JSON), and the report file
|
||||
* remains the only artifact the manager consumes.
|
||||
*/
|
||||
import { percentile, type ProgressEvent } from "./runner";
|
||||
|
||||
const ANSI = {
|
||||
reset: "\x1b[0m",
|
||||
bold: "\x1b[1m",
|
||||
dim: "\x1b[2m",
|
||||
red: "\x1b[31m",
|
||||
green: "\x1b[32m",
|
||||
yellow: "\x1b[33m",
|
||||
cyan: "\x1b[36m",
|
||||
} as const;
|
||||
|
||||
function rateColor(percent: number): string {
|
||||
if (percent >= 80) return ANSI.green;
|
||||
if (percent >= 50) return ANSI.yellow;
|
||||
return ANSI.red;
|
||||
}
|
||||
|
||||
export class LiveProgress {
|
||||
readonly #totalRuns: number;
|
||||
readonly #runsPerTask: number;
|
||||
readonly #isTty: boolean;
|
||||
readonly #colors: boolean;
|
||||
#started = 0;
|
||||
#completed = 0;
|
||||
#success = 0;
|
||||
#totalInput = 0;
|
||||
#totalOutput = 0;
|
||||
#totalDuration = 0;
|
||||
#totalReads = 0;
|
||||
#totalEdits = 0;
|
||||
#totalWrites = 0;
|
||||
#totalEditSuccesses = 0;
|
||||
#totalToolInputChars = 0;
|
||||
#indentScores: number[] = [];
|
||||
#inputTokens: number[] = [];
|
||||
#outputTokens: number[] = [];
|
||||
#totalTokens: number[] = [];
|
||||
#oneShotSuccessTokens: number[] = [];
|
||||
#lastLineLength = 0;
|
||||
|
||||
constructor(totalRuns: number, runsPerTask: number) {
|
||||
this.#totalRuns = totalRuns;
|
||||
this.#runsPerTask = runsPerTask;
|
||||
this.#isTty = Boolean(process.stderr.isTTY);
|
||||
this.#colors = this.#isTty && !process.env.NO_COLOR;
|
||||
}
|
||||
|
||||
#paint(code: string, text: string): string {
|
||||
return this.#colors ? `${code}${text}${ANSI.reset}` : text;
|
||||
}
|
||||
|
||||
#log(line: string): void {
|
||||
process.stderr.write(`${line}\n`);
|
||||
}
|
||||
|
||||
handleEvent(event: ProgressEvent): void {
|
||||
if (event.status === "started") {
|
||||
this.#started += 1;
|
||||
if (!this.#isTty) {
|
||||
this.#log(` [${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} started...`);
|
||||
}
|
||||
this.#renderLine();
|
||||
return;
|
||||
}
|
||||
|
||||
this.#completed += 1;
|
||||
if (event.result) {
|
||||
if (event.result.success) {
|
||||
this.#success += 1;
|
||||
}
|
||||
if (event.result.success && event.runIndex === 0) {
|
||||
this.#oneShotSuccessTokens.push(event.result.tokens.total);
|
||||
}
|
||||
this.#totalInput += event.result.tokens.input;
|
||||
this.#totalOutput += event.result.tokens.output;
|
||||
this.#inputTokens.push(event.result.tokens.input);
|
||||
this.#outputTokens.push(event.result.tokens.output);
|
||||
this.#totalTokens.push(event.result.tokens.total);
|
||||
this.#totalDuration += event.result.duration;
|
||||
this.#totalReads += event.result.toolCalls.read;
|
||||
this.#totalEdits += event.result.toolCalls.edit;
|
||||
this.#totalWrites += event.result.toolCalls.write;
|
||||
this.#totalEditSuccesses += event.result.toolCalls.editSuccesses;
|
||||
this.#totalToolInputChars += event.result.toolCalls.totalInputChars;
|
||||
if (typeof event.result.indentScore === "number") {
|
||||
this.#indentScores.push(event.result.indentScore);
|
||||
}
|
||||
}
|
||||
|
||||
const result = event.result;
|
||||
if (result && !result.success && result.error) {
|
||||
this.#flushLine();
|
||||
const header = this.#paint(
|
||||
ANSI.red,
|
||||
`[${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} failed:`,
|
||||
);
|
||||
this.#log(` ${header} ${result.error}`);
|
||||
if (result.diff) {
|
||||
const changeLines = result.diff
|
||||
.split("\n")
|
||||
.filter(line => /^[-+@]/.test(line) && !/^(---|\+\+\+)/.test(line));
|
||||
const maxLines = 40;
|
||||
for (const line of changeLines.slice(0, maxLines)) {
|
||||
let color: string | undefined;
|
||||
if (line.startsWith("@@")) color = ANSI.cyan;
|
||||
else if (line.startsWith("-")) color = ANSI.red;
|
||||
else if (line.startsWith("+")) color = ANSI.green;
|
||||
this.#log(` ${color ? this.#paint(color, line) : line}`);
|
||||
}
|
||||
if (changeLines.length > maxLines) {
|
||||
this.#log(this.#paint(ANSI.dim, ` ... (${changeLines.length - maxLines} more change lines)`));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (result?.editFailures && result.editFailures.length > 0) {
|
||||
this.#flushLine();
|
||||
for (const [i, failure] of result.editFailures.entries()) {
|
||||
const args = (failure.args ?? {}) as Record<string, unknown>;
|
||||
const target =
|
||||
typeof args.path === "string" ? args.path : typeof args.file === "string" ? args.file : undefined;
|
||||
const op = typeof args.operation === "string" ? args.operation : undefined;
|
||||
const oneLine = failure.error.replace(/\s+/g, " ").trim();
|
||||
const clipped = oneLine.length > 240 ? `${oneLine.slice(0, 237)}...` : oneLine;
|
||||
const tag = this.#paint(ANSI.yellow, `[${event.taskId}] schema #${i + 1}`);
|
||||
const metaParts = [op, target].filter((value): value is string => Boolean(value));
|
||||
const meta = metaParts.length > 0 ? this.#paint(ANSI.dim, metaParts.join(" ")) : "";
|
||||
this.#log(` ${tag}${meta ? ` ${meta}` : ""} ${clipped}`);
|
||||
if (failure.rawBlock) {
|
||||
const rawLine = failure.rawBlock.replace(/\s+/g, " ").trim();
|
||||
const clippedRaw = rawLine.length > 240 ? `${rawLine.slice(0, 237)}...` : rawLine;
|
||||
this.#log(` ${this.#paint(ANSI.dim, "raw")} ${clippedRaw}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!this.#isTty) {
|
||||
const status = event.result?.success ? "completed" : "failed";
|
||||
this.#log(` [${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} ${status}`);
|
||||
}
|
||||
|
||||
this.#renderLine();
|
||||
}
|
||||
|
||||
finish(): void {
|
||||
this.#flushLine();
|
||||
this.#printSummary();
|
||||
}
|
||||
|
||||
#printSummary(): void {
|
||||
const n = this.#completed;
|
||||
const denom = n || 1;
|
||||
|
||||
const successRate = (this.#success / denom) * 100;
|
||||
const editSuccessRate = this.#totalEdits > 0 ? (this.#totalEditSuccesses / this.#totalEdits) * 100 : 100;
|
||||
const avgIndent =
|
||||
this.#indentScores.length > 0 ? this.#indentScores.reduce((a, b) => a + b, 0) / this.#indentScores.length : 0;
|
||||
|
||||
this.#log("");
|
||||
this.#log(this.#paint(ANSI.bold, "Runtime Stats:"));
|
||||
this.#log(
|
||||
` Task success: ${this.#paint(rateColor(successRate), `${successRate.toFixed(1)}% (${this.#success}/${n})`)}`,
|
||||
);
|
||||
this.#log(
|
||||
` Edit success: ${this.#paint(rateColor(editSuccessRate), `${editSuccessRate.toFixed(1)}% (${this.#totalEditSuccesses}/${this.#totalEdits})`)}`,
|
||||
);
|
||||
this.#log(` Avg indent score: ${avgIndent.toFixed(2)}`);
|
||||
this.#log(` Tool calls: read=${this.#totalReads} edit=${this.#totalEdits} write=${this.#totalWrites}`);
|
||||
this.#log(` Tool input chars: ${this.#totalToolInputChars.toLocaleString()}`);
|
||||
const fmtTokens = (samples: number[]): string => {
|
||||
if (samples.length === 0) return "mean=0 median=0 p1=0 p99=0";
|
||||
const sorted = [...samples].sort((a, b) => a - b);
|
||||
const mean = Math.round(sorted.reduce((a, b) => a + b, 0) / sorted.length);
|
||||
return `mean=${mean} median=${Math.round(percentile(sorted, 50))} p1=${Math.round(percentile(sorted, 1))} p99=${Math.round(percentile(sorted, 99))}`;
|
||||
};
|
||||
this.#log(` Tokens/task in: ${fmtTokens(this.#inputTokens)}`);
|
||||
this.#log(` Tokens/task out: ${fmtTokens(this.#outputTokens)}`);
|
||||
this.#log(` Tokens/task tot: ${fmtTokens(this.#totalTokens)}`);
|
||||
this.#log(` Tokens/task (one-shot successes): ${fmtTokens(this.#oneShotSuccessTokens)}`);
|
||||
this.#log(` Avg time/task: ${Math.round(this.#totalDuration / denom)}ms`);
|
||||
}
|
||||
|
||||
#renderLine(): void {
|
||||
if (!this.#isTty) {
|
||||
return;
|
||||
}
|
||||
const successRate = this.#completed > 0 ? (this.#success / this.#completed) * 100 : 0;
|
||||
const editRate = this.#totalEdits > 0 ? (this.#totalEditSuccesses / this.#totalEdits) * 100 : 100;
|
||||
const avgInput = this.#completed > 0 ? Math.round(this.#totalInput / this.#completed) : 0;
|
||||
const avgOutput = this.#completed > 0 ? Math.round(this.#totalOutput / this.#completed) : 0;
|
||||
const avgDuration = this.#completed > 0 ? Math.round(this.#totalDuration / this.#completed) : 0;
|
||||
const inFlight = this.#started - this.#completed;
|
||||
const bar = this.#renderBar(this.#completed, this.#totalRuns, 20);
|
||||
const progress = this.#paint(ANSI.bold, `${this.#completed}/${this.#totalRuns}`);
|
||||
const taskCol = `task=${this.#paint(rateColor(successRate), `${successRate.toFixed(0)}%`)}`;
|
||||
const editCol = `edit=${this.#paint(rateColor(editRate), `${editRate.toFixed(0)}%`)}`;
|
||||
const tokCol = this.#paint(ANSI.dim, `tok=${avgInput}/${avgOutput}`);
|
||||
const durCol = this.#paint(ANSI.dim, `${avgDuration}ms`);
|
||||
const rewCol = this.#paint(ANSI.dim, `r/e/w=${this.#totalReads}/${this.#totalEdits}/${this.#totalWrites}`);
|
||||
const flyCol = `fly=${this.#paint(ANSI.cyan, String(inFlight))}`;
|
||||
const line = ` ${bar} ${progress} ${taskCol} ${editCol} ${tokCol} ${durCol} ${rewCol} ${flyCol}`;
|
||||
this.#writeLine(line);
|
||||
}
|
||||
|
||||
#renderBar(done: number, total: number, width: number): string {
|
||||
const ratio = total === 0 ? 0 : done / total;
|
||||
const filled = Math.round(ratio * width);
|
||||
const empty = Math.max(0, width - filled);
|
||||
const filledPart = this.#paint(ANSI.green, "#".repeat(filled));
|
||||
const emptyPart = this.#paint(ANSI.dim, "-".repeat(empty));
|
||||
return `[${filledPart}${emptyPart}]`;
|
||||
}
|
||||
|
||||
#writeLine(line: string): void {
|
||||
const lineWidth = Bun.stringWidth(line);
|
||||
const pad = this.#lastLineLength > lineWidth ? " ".repeat(this.#lastLineLength - lineWidth) : "";
|
||||
process.stderr.write(`\r${line}${pad}`);
|
||||
this.#lastLineLength = lineWidth;
|
||||
}
|
||||
|
||||
#flushLine(): void {
|
||||
if (!this.#isTty) {
|
||||
return;
|
||||
}
|
||||
if (this.#lastLineLength > 0) {
|
||||
process.stderr.write(`\r${" ".repeat(this.#lastLineLength)}\r`);
|
||||
this.#lastLineLength = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -212,17 +212,19 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
lines.push("");
|
||||
|
||||
if (summary.hashlineEditSubtypes) {
|
||||
const order = ["set", "set_range", "insert"] as const;
|
||||
const total = order.reduce((sum, key) => sum + (summary.hashlineEditSubtypes?.[key] ?? 0), 0);
|
||||
const rows = Object.entries(summary.hashlineEditSubtypes)
|
||||
.filter(([, count]) => count > 0)
|
||||
.sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]));
|
||||
const total = rows.reduce((sum, [, count]) => sum + count, 0);
|
||||
if (total > 0) {
|
||||
lines.push("### Hashline Edit Subtypes");
|
||||
lines.push("### Hashline Op Breakdown");
|
||||
lines.push("");
|
||||
lines.push("All attempted edit calls across all runs (retries and failed calls included).");
|
||||
lines.push("");
|
||||
lines.push("| Operation | Count | % |");
|
||||
lines.push("|-----------|-------|---|");
|
||||
for (const key of order) {
|
||||
const count = summary.hashlineEditSubtypes[key] ?? 0;
|
||||
const pct = formatPercent(count / total);
|
||||
lines.push(`| ${key} | ${count} | ${pct} |`);
|
||||
for (const [op, count] of rows) {
|
||||
lines.push(`| \`${op}\` | ${count} | ${formatPercent(count / total)} |`);
|
||||
}
|
||||
lines.push(`| **Total** | **${total}** | 100% |`);
|
||||
lines.push("");
|
||||
|
||||
@@ -7,7 +7,12 @@
|
||||
/// <reference types="./bun-imports.d.ts" />
|
||||
import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import { formatHashlineHeader, InMemorySnapshotStore } from "@oh-my-pi/hashline";
|
||||
import {
|
||||
type BlockTarget as HashlineBlockTarget,
|
||||
formatHashlineHeader,
|
||||
InMemorySnapshotStore,
|
||||
Tokenizer as HashlineTokenizer,
|
||||
} from "@oh-my-pi/hashline";
|
||||
import type { AgentMessage, ResolvedThinkingLevel, ThinkingLevel } from "@oh-my-pi/pi-agent-core";
|
||||
import type { Model, ToolExample } from "@oh-my-pi/pi-ai";
|
||||
import { formatSessionDumpText, RpcClient } from "@oh-my-pi/pi-coding-agent";
|
||||
@@ -262,7 +267,6 @@ function countEditFailureCategories(runs: TaskRunResult[]): Record<EditFailureCa
|
||||
return counts;
|
||||
}
|
||||
|
||||
const HL_SUBTYPES = ["set", "set_range", "insert"] as const;
|
||||
const BENCHMARK_TOOL_NAMES = ["read", "edit", "write", "apply_patch"] as const;
|
||||
const EDIT_TOOL_NAMES = ["edit", "apply_patch"] as const;
|
||||
|
||||
@@ -274,21 +278,71 @@ function isMutationTool(toolName: unknown): boolean {
|
||||
return isEditTool(toolName) || toolName === "write";
|
||||
}
|
||||
|
||||
function countHashlineEditSubtypes(args: unknown): Record<string, number> {
|
||||
const counts: Record<string, number> = Object.fromEntries(HL_SUBTYPES.map(k => [k, 0]));
|
||||
if (!args || typeof args !== "object") return counts;
|
||||
const edits = (args as { edits?: unknown[] }).edits;
|
||||
if (!Array.isArray(edits)) return counts;
|
||||
for (const edit of edits) {
|
||||
if (!edit || typeof edit !== "object") continue;
|
||||
for (const key of HL_SUBTYPES) {
|
||||
if (key in edit) {
|
||||
counts[key]++;
|
||||
break;
|
||||
// Pure classification — single shared tokenizer is safe.
|
||||
const HASHLINE_OP_TOKENIZER = new HashlineTokenizer();
|
||||
|
||||
/** Display label for a hashline op header, e.g. `SWAP.BLK` or `PASTE.POST`. */
|
||||
function hashlineOpLabel(target: HashlineBlockTarget): string {
|
||||
switch (target.kind) {
|
||||
case "replace":
|
||||
return "SWAP";
|
||||
case "block":
|
||||
return "SWAP.BLK";
|
||||
case "delete":
|
||||
return "DEL";
|
||||
case "delete_block":
|
||||
return "DEL.BLK";
|
||||
case "insert_before":
|
||||
return "INS.PRE";
|
||||
case "insert_after":
|
||||
return "INS.POST";
|
||||
case "bof":
|
||||
return "INS.HEAD";
|
||||
case "eof":
|
||||
return "INS.TAIL";
|
||||
case "insert_after_block":
|
||||
return "INS.BLK.POST";
|
||||
case "copy":
|
||||
return target.cut ? "CUT" : "COPY";
|
||||
case "copy_block":
|
||||
return target.cut ? "CUT.BLK" : "COPY.BLK";
|
||||
case "paste":
|
||||
switch (target.cursor.kind) {
|
||||
case "before_anchor":
|
||||
return "PASTE.PRE";
|
||||
case "after_anchor":
|
||||
return "PASTE.POST";
|
||||
case "bof":
|
||||
return "PASTE.HEAD";
|
||||
case "eof":
|
||||
return "PASTE.TAIL";
|
||||
}
|
||||
}
|
||||
break;
|
||||
case "paste_after_block":
|
||||
return "PASTE.BLK.POST";
|
||||
case "rem":
|
||||
return "REM";
|
||||
case "move":
|
||||
return "MV";
|
||||
}
|
||||
return counts;
|
||||
}
|
||||
|
||||
/**
|
||||
* Count hashline op headers (`SWAP`, `DEL.BLK`, `CUT`, `PASTE.POST`, …) in an
|
||||
* edit call's patch input. Returns `null` when the args carry no hashline
|
||||
* patch — non-hashline edit variants and malformed calls contribute nothing.
|
||||
*/
|
||||
function countHashlineOps(args: unknown): Record<string, number> | null {
|
||||
if (!args || typeof args !== "object" || !("input" in args)) return null;
|
||||
const input = args.input;
|
||||
if (typeof input !== "string" || input.length === 0) return null;
|
||||
const counts: Record<string, number> = {};
|
||||
for (const token of HASHLINE_OP_TOKENIZER.tokenizeAll(input)) {
|
||||
if (token.kind !== "op-block") continue;
|
||||
const label = hashlineOpLabel(token.target);
|
||||
counts[label] = (counts[label] ?? 0) + 1;
|
||||
}
|
||||
return Object.keys(counts).length > 0 ? counts : null;
|
||||
}
|
||||
|
||||
async function collectOriginalFileContents(cwd: string, files: string[]): Promise<Map<string, string>> {
|
||||
@@ -829,7 +883,7 @@ export interface TaskRunResult {
|
||||
editFailures: EditFailure[];
|
||||
editWarnings: string[];
|
||||
editAutocorrectCount: number;
|
||||
/** Hashline edit subtype counts (replaceLine, replaceLines, etc.) — only when editVariant is hashline */
|
||||
/** Hashline op counts (`SWAP`, `DEL.BLK`, `CUT`, …) — present when edit calls carried hashline patches. */
|
||||
hashlineEditSubtypes?: Record<string, number>;
|
||||
mutationIntentMatched?: boolean;
|
||||
mutationIntentReason?: string;
|
||||
@@ -945,7 +999,7 @@ export interface BenchmarkSummary {
|
||||
mutationIntentMatchRate?: number;
|
||||
/** Edit failure categories across all runs. */
|
||||
editFailureCategories: Record<EditFailureCategory, number>;
|
||||
/** Hashline edit subtype totals across all runs — only when editVariant is hashline. */
|
||||
/** Hashline op totals across all runs — present when any run's edit calls carried hashline patches. */
|
||||
hashlineEditSubtypes?: Record<string, number>;
|
||||
}
|
||||
|
||||
@@ -1037,7 +1091,7 @@ async function runSingleTask(
|
||||
editAutocorrects: 0,
|
||||
totalInputChars: 0,
|
||||
};
|
||||
const hashlineSubtypes: Record<string, number> = Object.fromEntries(HL_SUBTYPES.map(k => [k, 0]));
|
||||
const hashlineSubtypes: Record<string, number> = {};
|
||||
|
||||
const logFile = path.join(TMP, `run-${task.id}-${runIndex}.jsonl`);
|
||||
const logEvent = async (event: unknown) => {
|
||||
@@ -1271,10 +1325,10 @@ async function runSingleTask(
|
||||
const pendingEdit = pendingEdits.get(e.toolCallId) ?? { args: null };
|
||||
const args = pendingEdit.args;
|
||||
pendingEdits.delete(e.toolCallId);
|
||||
if (config.editVariant === "hashline" && args) {
|
||||
const counts = countHashlineEditSubtypes(args);
|
||||
for (const key of HL_SUBTYPES) {
|
||||
hashlineSubtypes[key] += counts[key];
|
||||
const hashlineOpCounts = countHashlineOps(args);
|
||||
if (hashlineOpCounts) {
|
||||
for (const key in hashlineOpCounts) {
|
||||
hashlineSubtypes[key] = (hashlineSubtypes[key] ?? 0) + hashlineOpCounts[key];
|
||||
}
|
||||
}
|
||||
if (e.isError) {
|
||||
@@ -1422,7 +1476,7 @@ async function runSingleTask(
|
||||
editFailures,
|
||||
editWarnings,
|
||||
editAutocorrectCount,
|
||||
hashlineEditSubtypes: config.editVariant === "hashline" ? hashlineSubtypes : undefined,
|
||||
hashlineEditSubtypes: Object.keys(hashlineSubtypes).length > 0 ? hashlineSubtypes : undefined,
|
||||
mutationIntentMatched: mutationIntentValidation?.matched,
|
||||
mutationIntentReason: mutationIntentValidation?.reason,
|
||||
timeoutTelemetry,
|
||||
@@ -1936,12 +1990,20 @@ export function buildBenchmarkResult(params: {
|
||||
0,
|
||||
);
|
||||
const editFailureCategories = countEditFailureCategories(nonGhostRuns);
|
||||
// Op counts aggregate every attempted edit call across ALL runs (retries and
|
||||
// failed calls included) — deliberately: the mix shows what the model reaches
|
||||
// for, not just what landed. Best-run-only would hide exactly the flailing
|
||||
// (failed SWAPs before a successful retry) this table exists to expose.
|
||||
const hashlineEditSubtypeTotals: Record<string, number> = {};
|
||||
for (const run of allRuns) {
|
||||
const runOps = run.hashlineEditSubtypes;
|
||||
if (!runOps) continue;
|
||||
for (const key in runOps) {
|
||||
hashlineEditSubtypeTotals[key] = (hashlineEditSubtypeTotals[key] ?? 0) + runOps[key];
|
||||
}
|
||||
}
|
||||
const hashlineEditSubtypes: Record<string, number> | undefined =
|
||||
params.config.editVariant === "hashline"
|
||||
? Object.fromEntries(
|
||||
HL_SUBTYPES.map(key => [key, allRuns.reduce((sum, r) => sum + (r.hashlineEditSubtypes?.[key] ?? 0), 0)]),
|
||||
)
|
||||
: undefined;
|
||||
Object.keys(hashlineEditSubtypeTotals).length > 0 ? hashlineEditSubtypeTotals : undefined;
|
||||
|
||||
// Primary aggregates run over the *best* run of each completed task.
|
||||
const bestRuns: TaskRunResult[] = [];
|
||||
|
||||
@@ -20,6 +20,7 @@
|
||||
"check:types": "tsgo -p tsconfig.json --noEmit && tsgo -p adapters/edit/tsconfig.json --noEmit && tsgo -p scripts/tsconfig.json --noEmit",
|
||||
"lint": "biome lint .",
|
||||
"serve": "bun run src/server.ts",
|
||||
"bench:edit": "bun adapters/edit/cli.ts",
|
||||
"dev": "bun --hot src/server.ts",
|
||||
"test": "bun test"
|
||||
},
|
||||
|
||||
Reference in New Issue
Block a user