feat(harbor-manager): unified benchmark normalization and reporting

- Implemented a unified benchmark normalization layer to handle metrics and artifacts across harbor, edit, and snapcompact benchmarks.
- Integrated the TypeScript edit benchmark directly into the manager, migrating logic and deprecating the standalone package.
- Updated the API and database schema to support standardized benchmark configurations, metrics, and trace-based reporting.
- Enhanced the UI to visualize comparative performance metrics, including pass rate, cost, and latency deltas for benchmark runs.
This commit is contained in:
can1357
2026-07-13 03:39:05 +02:00
parent ac16253613
commit 0856055dfe
25 changed files with 1010 additions and 992 deletions
+8 -4
View File
@@ -140,12 +140,19 @@
"name": "@oh-my-pi/harbor-manager",
"version": "0.0.1",
"bin": {
"harbor-manager": "src/runner.ts",
"harbor-manager": "src/server.ts",
},
"dependencies": {
"@oh-my-pi/hashline": "catalog:",
"@oh-my-pi/pi-agent-core": "catalog:",
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-coding-agent": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
"@oh-my-pi/typescript-edit-benchmark": "workspace:*",
"clsx": "^2.1.1",
"d3-scale": "^4.0.2",
"d3-shape": "^3.2.0",
"diff": "catalog:",
"motion": "^12.15.0",
"react": "^19.1.0",
"react-dom": "^19.1.0",
@@ -276,9 +283,6 @@
"packages/typescript-edit-benchmark": {
"name": "@oh-my-pi/typescript-edit-benchmark",
"version": "0.0.1",
"bin": {
"typescript-edit-benchmark": "src/index.ts",
},
"dependencies": {
"@babel/generator": "catalog:",
"@babel/parser": "catalog:",
-1
View File
@@ -149,7 +149,6 @@
"ci:release:publish": "bun scripts/ci-release-publish.ts",
"ci:release:publish-native-leaf": "bun scripts/ci-release-publish.ts --native-leaf",
"bench:gen-fixtures": "bun --cwd=packages/typescript-edit-benchmark run src/generate.ts --typescript-dir /tmp/typescript-source --count-per-type 8",
"bench:edit": "bun --cwd=packages/typescript-edit-benchmark run start",
"stats:sync": "python3 scripts/session-stats/sync.py",
"stats:tools": "python3 scripts/session-stats/analyze.py tools",
"stats:edits": "python3 scripts/session-stats/analyze.py edits",
+22 -26
View File
@@ -1,22 +1,16 @@
# @oh-my-pi/harbor-manager
Manage [Harbor](https://github.com/laude-institute/harbor) benchmark runs against
the **local `omp` build**: a CLI runner with a live terminal dashboard, a
SQLite-backed run store, a REST/SSE API, and a web dashboard with live
run → trial → transcript drill-down.
Works with any Harbor dataset (`terminal-bench@2.0` by default,
`swe-bench/swe-bench-verified`, …).
One manager for repository benchmarks. Harbor, TypeScript edit, and SnapCompact
runs use the same experiment → run → trace model, SQLite store, REST/SSE API,
and dashboard. Benchmark-native artifacts remain on disk; adapters normalize
their live progress, scores, token usage, costs, and traces.
```bash
# CLI run (owns the terminal, writes a markdown report)
bun src/runner.ts --model anthropic/claude-sonnet-4-6 --tasks 20 --concurrency 4
# Web manager: dashboard + API on :4700
bun run serve # bun src/server.ts [--port 4700] [--jobs-dir <path>]
# Dashboard + API on :4700; launch every benchmark from the same “new run” form
bun run serve --port 4700
```
## How runs execute
## How Harbor runs execute
1. **Local omp, not npm.** By default the runner bind-mounts the repo
read-only into each task container (`--install source`) and runs omp
@@ -35,34 +29,36 @@ bun run serve # bun src/server.ts [--port 4700] [--jobs-dir <path>]
## Server
- `GET /` — dashboard: run list (SSE-live), trial grid, transcript viewer
that tails live sessions.
- `GET /api/runs` — run rows (rollups: pass/fail/error/spend/tokens).
- `POST /api/runs` — launch. Body:
- `GET /` — experiments, runs, normalized traces, and a launch form for every benchmark.
- `GET /api/experiments` — experiment summaries across all benchmark types.
- `GET /api/runs` — uniform run rows with benchmark, score, progress, spend, and tokens.
- `POST /api/runs` — launch through a benchmark adapter. Body:
```json
{
"benchmark": "edit",
"model": "anthropic/claude-opus-4-8",
"dataset": "swe-bench/swe-bench-verified",
"tasks": 20,
"concurrency": 4,
"timeoutMultiplier": 2,
"include": ["swe-bench/astropy__astropy-14995"],
"slide": { "model": "google/gemini-3.5-flash", "onAction": true, "plan": true }
"attempts": 2,
"jobName": "edit-baseline",
"role": "baseline",
"goal": "compare edit strategies"
}
```
`slide.turns` and `slide.onAction` are mutually exclusive triggers.
- `GET /api/runs/:name` — `{ run, trials }` (syncs from disk on read).
`benchmark` is `harbor`, `edit`, or `snapcompact`. Harbor uses `dataset`,
`include`, `timeoutMultiplier`, and `slide`; edit uses `include` as task IDs;
SnapCompact uses `conditions` and treats `tasks` as the passage limit.
- `GET /api/runs/:name` — `{ run, traces }` (syncs native artifacts on read).
- `DELETE /api/runs/:name` — cancel a manager-launched run.
- `GET /api/runs/:name/trials/:trial/transcript?tail=N[&raw=1]` — compact
(or raw JSONL) view of the trial's live session log.
- `GET /api/runs/:name/traces/:trace[?raw=1]` — normalized or native trace.
- `GET /api/events` — SSE stream of run-list snapshots (sent on change).
State lives in `<jobs-dir>/_manager/harbor-manager.sqlite`; the filesystem
stays the source of truth and historical CLI runs are auto-discovered.
## Runner options (excerpt)
## Harbor runner options (excerpt)
| Option | Default | Notes |
|---|---|---|
@@ -0,0 +1,4 @@
declare module "*.tar.gz" {
const content: string;
export default content;
}
@@ -0,0 +1,97 @@
#!/usr/bin/env bun
/** Manager-owned executable adapter for the TypeScript edit benchmark. */
import * as fs from "node:fs/promises";
import * as path from "node:path";
import { parseArgs } from "node:util";
import { TempDir } from "@oh-my-pi/pi-utils";
import { loadTasksFromDir } from "@oh-my-pi/typescript-edit-benchmark/tasks";
import { generateJsonReport } from "./report";
import { type BenchmarkConfig, runBenchmark } from "./runner";
const EDIT_PACKAGE = path.resolve(import.meta.dir, "..", "..", "..", "typescript-edit-benchmark");
async function extractFixtures(): Promise<{ dir: string; temp: TempDir }> {
const temp = await TempDir.create("@harbor-edit-fixtures-");
const archive = new Bun.Archive(await Bun.file(path.join(EDIT_PACKAGE, "fixtures.tar.gz")).arrayBuffer());
for (const [filePath, file] of await archive.files()) {
await Bun.write(path.join(temp.path(), filePath), file);
}
const entries = await fs.readdir(temp.path(), { withFileTypes: true });
const directories = entries.filter(entry => entry.isDirectory());
const files = entries.filter(entry => entry.isFile());
return { dir: directories.length === 1 && files.length === 0 ? path.join(temp.path(), directories[0]!.name) : temp.path(), temp };
}
/** Execute an edit benchmark and continuously materialize its normalized source artifact. */
export async function main(argv = process.argv.slice(2)): Promise<void> {
const { values } = parseArgs({
args: argv,
options: {
model: { type: "string" },
output: { type: "string" },
"max-tasks": { type: "string", default: "80" },
tasks: { type: "string" },
"task-concurrency": { type: "string", default: "32" },
runs: { type: "string", default: "1" },
list: { type: "boolean", default: false },
},
strict: true,
});
const fixtures = await extractFixtures();
try {
let tasks = await loadTasksFromDir(fixtures.dir);
if (values.list) {
process.stdout.write(`${JSON.stringify(tasks.map(task => ({ id: task.id, name: task.name })))}\n`);
return;
}
if (!values.model || !values.output) throw new Error("edit adapter requires --model and --output");
if (values.tasks) {
const selected = new Set(values.tasks.split(",").map(value => value.trim()));
tasks = tasks.filter(task => selected.has(task.id));
if (tasks.length !== selected.size) throw new Error("one or more edit task ids were not found");
} else {
const limit = Number(values["max-tasks"]);
if (limit > 0 && tasks.length > limit) {
const sorted = tasks.slice().sort((a, b) => a.id.localeCompare(b.id));
const step = sorted.length / limit;
tasks = Array.from({ length: limit }, (_, index) => sorted[Math.floor(index * step)]!);
}
}
const slash = values.model.indexOf("/");
const config: BenchmarkConfig = {
provider: slash === -1 ? "anthropic" : values.model.slice(0, slash),
model: values.model,
runsPerTask: Number(values.runs),
timeout: 120_000,
connectionTimeout: 30_000,
maxTurns: 30,
taskConcurrency: Number(values["task-concurrency"]),
guided: false,
maxAttempts: 1,
noOpRetryLimit: 2,
maxTimeoutRetries: 3,
maxProviderFailureRetries: 3,
mutationScopeWindow: 20,
conversationDumpDir: path.join(path.dirname(values.output), "result.dump"),
inProcess: true,
earlyStopOnMatch: true,
};
let writes = Promise.resolve();
const result = await runBenchmark(tasks, config, undefined, snapshot => {
writes = writes.then(async () => {
await Bun.write(values.output!, generateJsonReport(snapshot));
});
});
await writes;
await Bun.write(values.output, generateJsonReport(result));
} finally {
await fixtures.temp.remove();
}
}
if (import.meta.main) {
main().catch(error => {
process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`);
process.exitCode = 1;
});
}
@@ -4,12 +4,8 @@ import * as path from "node:path";
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
import { formatSessionDumpText, SessionManager } from "@oh-my-pi/pi-coding-agent";
import { TempDir } from "@oh-my-pi/pi-utils";
import { generateReport } from "@oh-my-pi/typescript-edit-benchmark/report";
import {
buildBenchmarkResult,
type TaskRunResult,
writeConversationDump,
} from "@oh-my-pi/typescript-edit-benchmark/runner";
import { generateReport } from "./report";
import { buildBenchmarkResult, type TaskRunResult, writeConversationDump } from "./runner";
import type { EditTask } from "@oh-my-pi/typescript-edit-benchmark/tasks";
const tempDirs: TempDir[] = [];
@@ -13,15 +13,22 @@ import type { Model, ToolExample } from "@oh-my-pi/pi-ai";
import { formatSessionDumpText, RpcClient } from "@oh-my-pi/pi-coding-agent";
import { prompt } from "@oh-my-pi/pi-utils";
import { diffLines } from "diff";
import { formatDirectory } from "./formatter";
import { discoverSharedInfra, InProcessClient, type SharedInfra } from "./in-process-client";
import { formatDirectory } from "@oh-my-pi/typescript-edit-benchmark/formatter";
import {
discoverSharedInfra,
InProcessClient,
type SharedInfra,
} from "@oh-my-pi/typescript-edit-benchmark/in-process-client";
import benchmarkRetryPrompt from "./prompts/benchmark-retry.md" with { type: "text" };
import benchmarkSystemPrompt from "./prompts/benchmark-system.md" with { type: "text" };
import benchmarkTaskPrompt from "./prompts/benchmark-task.md" with { type: "text" };
import type { EditTask } from "./tasks";
import { verifyExpectedFileSubset, verifyExpectedFiles } from "./verify";
import type { EditTask } from "@oh-my-pi/typescript-edit-benchmark/tasks";
import {
verifyExpectedFileSubset,
verifyExpectedFiles,
} from "@oh-my-pi/typescript-edit-benchmark/verify";
const REPO_ROOT = path.resolve(import.meta.dir, "..", "..", "..");
const REPO_ROOT = path.resolve(import.meta.dir, "..", "..", "..", "..");
const RUNS_DIR = path.join(REPO_ROOT, "runs");
const TMP = path.join(RUNS_DIR, `rb-${Math.random().toString(36).slice(2, 10)}`);
const CLI_PATH = Bun.fileURLToPath(import.meta.resolve("@oh-my-pi/pi-coding-agent/cli"));
@@ -0,0 +1,4 @@
{
"extends": "../../../tsconfig.workspace.json",
"include": ["."]
}
+11 -4
View File
@@ -3,7 +3,7 @@
"private": true,
"name": "@oh-my-pi/harbor-manager",
"version": "0.0.1",
"description": "Manage Harbor benchmark runs against the local omp build: CLI runner, SQLite-backed run store, REST/SSE API, and a live web dashboard",
"description": "Unified benchmark runners plus Harbor run storage, REST/SSE APIs, and a live web dashboard",
"homepage": "https://omp.sh",
"author": "Can Boluk",
"license": "MIT",
@@ -13,20 +13,27 @@
"directory": "packages/harbor-manager"
},
"bin": {
"harbor-manager": "src/runner.ts"
"harbor-manager": "src/server.ts"
},
"scripts": {
"check": "biome check . && bun run check:types",
"check:types": "tsgo -p tsconfig.json --noEmit",
"check:types": "tsgo -p tsconfig.json --noEmit && tsgo -p adapters/edit/tsconfig.json --noEmit",
"lint": "biome lint .",
"start": "bun run src/runner.ts",
"start": "bun run src/server.ts",
"serve": "bun run src/server.ts",
"test": "bun test"
},
"dependencies": {
"@oh-my-pi/hashline": "catalog:",
"@oh-my-pi/pi-agent-core": "catalog:",
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-coding-agent": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
"@oh-my-pi/typescript-edit-benchmark": "workspace:*",
"clsx": "^2.1.1",
"d3-scale": "^4.0.2",
"d3-shape": "^3.2.0",
"diff": "catalog:",
"motion": "^12.15.0",
"react": "^19.1.0",
"react-dom": "^19.1.0",
@@ -36,6 +36,7 @@ from concurrent.futures import ThreadPoolExecutor
from pathlib import Path
HERE = Path(__file__).resolve().parent
RESEARCH = HERE.parents[2] / "snapcompact" / "research"
def find_agent_prompts() -> Path:
for parent in HERE.parents:
@@ -48,16 +49,16 @@ def find_agent_prompts() -> Path:
raise FileNotFoundError("Could not find agent compaction prompts")
sys.path.insert(0, str(HERE))
sys.path.insert(0, str(RESEARCH))
import squad # noqa: E402
from anthropic_api import complete, image_block, load_api_key # noqa: E402
from bdf import VARIANTS, FontCfg, capacity, render # noqa: E402
AGENT_PROMPTS = find_agent_prompts()
CACHE = HERE / ".cache"
CACHE = RESEARCH / ".cache"
QA_CACHE = CACHE / "qa"
RESULTS = HERE / "results"
RESULTS = RESEARCH / "results"
FONTS = {
"8x13": FontCfg("8x13", "8x13", 8, 13),
@@ -100,7 +101,7 @@ def sha8(*parts: str) -> str:
def load_prompt(name: str) -> str:
return (HERE / "prompts" / name).read_text()
return (RESEARCH / "prompts" / name).read_text()
def agent_prompt(name: str) -> str:
@@ -276,6 +277,7 @@ def main() -> None:
ap.add_argument("--price-in", type=float, default=10.0, help="$ per 1M input tokens")
ap.add_argument("--price-out", type=float, default=50.0, help="$ per 1M output tokens")
ap.add_argument("--env", default="~/.env")
ap.add_argument("--output-dir", help="write records.jsonl and summary.json directly to this directory")
args = ap.parse_args()
CACHE.mkdir(exist_ok=True)
@@ -288,7 +290,11 @@ def main() -> None:
f"-e{args.effort}" if args.effort else "",
]
)
run_dir = RESULTS / f"{args.model}-seed{args.seed}-qpc{args.qpc}-{scope}{tag}"
run_dir = (
Path(args.output_dir).expanduser().resolve()
if args.output_dir
else RESULTS / f"{args.model}-seed{args.seed}-qpc{args.qpc}-{scope}{tag}"
)
run_dir.mkdir(parents=True, exist_ok=True)
paras = squad.load_paragraphs(CACHE)
@@ -312,18 +318,19 @@ def main() -> None:
ctx_args = {"args": args, "flow": flow, "paras": paras, "offsets": offsets, "api_key": api_key}
records: list[dict] = []
done = 0
with (run_dir / "records.jsonl").open("w") as records_file:
with ThreadPoolExecutor(args.workers) as pool:
futures = [pool.submit(run_chunk, cond, start, end, ctx_args) for cond, start, end in tasks]
for fut in futures:
records.extend(fut.result())
chunk_records = fut.result()
records.extend(chunk_records)
for record in chunk_records:
records_file.write(json.dumps(record) + "\n")
records_file.flush()
done += 1
if done % 20 == 0:
print(f" {done}/{len(tasks)} chunks", flush=True)
with (run_dir / "records.jsonl").open("w") as fh:
for r in records:
fh.write(json.dumps(r) + "\n")
rows = [
aggregate(cond["name"], [r for r in records if r["cond"] == cond["name"]], args.price_in, args.price_out)
for cond in conditions
@@ -0,0 +1,85 @@
import { afterEach, describe, expect, it } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { BENCHMARK_DEFINITIONS, readBenchmarkSnapshot } from "./benchmarks";
const cleanups: string[] = [];
function jobDir(): string {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), "harbor-benchmark-"));
cleanups.push(dir);
return dir;
}
afterEach(() => {
for (const dir of cleanups.splice(0)) fs.rmSync(dir, { recursive: true, force: true });
});
describe("benchmark adapters", () => {
it("normalizes edit attempts, traces, tokens, and declared metrics", () => {
const dir = jobDir();
fs.writeFileSync(
path.join(dir, "result.json"),
JSON.stringify({
tasks: [
{
id: "rename-symbol",
name: "Rename symbol",
runs: [
{
runIndex: 0,
success: true,
duration: 1200,
tokens: { input: 100, output: 20, reasoning: 5 },
},
],
},
],
summary: {
totalRuns: 1,
successfulRuns: 1,
taskSuccessRate: 1,
editSuccessRate: 0.75,
totalTokens: { input: 100, output: 20 },
},
}),
);
const snapshot = readBenchmarkSnapshot("edit", dir);
expect(snapshot.metrics).toEqual({ task_success_rate: 1, edit_success_rate: 0.75 });
expect(snapshot.traces[0]).toMatchObject({
name: "rename-symbol__1",
status: "pass",
tracePath: path.join("result.dump", "rename-symbol", "run-1.md"),
});
expect([snapshot.tokIn, snapshot.tokOut]).toEqual([100, 20]);
});
it("normalizes SnapCompact records and weighted quality metrics", () => {
const dir = jobDir();
fs.writeFileSync(
path.join(dir, "records.jsonl"),
`${JSON.stringify({ cond: "text", chunk: 0, pos_rel: 0.2, q: "q", answer: "a", golds: ["a"], em: 1, f1: 1 })}\n`,
);
fs.writeFileSync(
path.join(dir, "summary.json"),
JSON.stringify({
rows: [
{ n: 2, f1: 0.75, em: 0.5, cost_usd: 1.25, tokens_in: 200, tokens_out: 30, cache_w: 10, cache_r: 20 },
],
}),
);
const snapshot = readBenchmarkSnapshot("snapcompact", dir);
expect(snapshot.metrics).toEqual({ f1: 0.75, exact_match: 0.5 });
expect(snapshot.traces[0]).toMatchObject({ status: "pass", reward: 1, tracePath: "record:1" });
expect(snapshot.costUsd).toBe(1.25);
expect(snapshot.tokCache).toBe(30);
});
it("publishes metric definitions for every managed benchmark", () => {
expect(BENCHMARK_DEFINITIONS.map(definition => definition.kind)).toEqual(["harbor", "edit", "snapcompact"]);
expect(BENCHMARK_DEFINITIONS.every(definition => definition.metrics.length > 0)).toBe(true);
});
});
+273
View File
@@ -0,0 +1,273 @@
/** Benchmark adapters normalize native artifacts into manager runs and traces. */
import * as fs from "node:fs";
import * as path from "node:path";
import { aggregate, readJobResult, readTrials } from "./runner";
import type { BenchmarkKind } from "./store";
/** Describes a benchmark metric so storage and UI do not hard-code benchmark semantics. */
export interface MetricDefinition {
key: string;
label: string;
format: "percent" | "number" | "usd";
higherIsBetter: boolean;
}
/** Adapter metadata exposed to launch clients and the dashboard. */
export interface BenchmarkDefinition {
kind: BenchmarkKind;
label: string;
metrics: MetricDefinition[];
}
/** Built-in benchmark adapters and their native score definitions. */
export const BENCHMARK_DEFINITIONS: BenchmarkDefinition[] = [
{
kind: "harbor",
label: "Harbor",
metrics: [{ key: "success_rate", label: "Success rate", format: "percent", higherIsBetter: true }],
},
{
kind: "edit",
label: "TypeScript edit",
metrics: [
{ key: "task_success_rate", label: "Task success", format: "percent", higherIsBetter: true },
{ key: "edit_success_rate", label: "Edit success", format: "percent", higherIsBetter: true },
],
},
{
kind: "snapcompact",
label: "SnapCompact",
metrics: [
{ key: "f1", label: "F1", format: "percent", higherIsBetter: true },
{ key: "exact_match", label: "Exact match", format: "percent", higherIsBetter: true },
],
},
];
/** A normalized trace emitted by any benchmark adapter. */
export interface BenchmarkTrace {
name: string;
task: string;
status: "pass" | "fail" | "error" | "running";
reward: number | null;
costUsd: number;
durationMs: number;
detail: string;
tracePath: string | null;
}
/** Uniform aggregate and traces read from benchmark-native artifacts. */
export interface BenchmarkSnapshot {
traces: BenchmarkTrace[];
total: number;
done: number;
pass: number;
fail: number;
error: number;
running: number;
costUsd: number;
tokIn: number;
tokOut: number;
tokCache: number;
score: number | null;
metrics: Record<string, number | null>;
}
interface EditRun {
runIndex: number;
success: boolean;
error?: string;
duration: number;
tokens: { input: number; output: number; reasoning: number };
toolCalls?: { read: number; edit: number; write: number };
}
interface EditTask {
id: string;
name: string;
runs: EditRun[];
}
interface EditResult {
tasks: EditTask[];
summary: {
totalRuns: number;
successfulRuns: number;
taskSuccessRate: number;
editSuccessRate: number;
totalTokens: { input: number; output: number };
};
}
interface SnapRecord {
cond: string;
chunk: number;
pos_rel: number;
q: string;
answer: string;
golds: string[];
em: number;
f1: number;
}
interface SnapSummaryRow {
n: number;
f1: number;
em: number;
cost_usd: number;
tokens_in: number;
tokens_out: number;
cache_w: number;
cache_r: number;
}
interface SnapSummary {
rows: SnapSummaryRow[];
}
function emptySnapshot(): BenchmarkSnapshot {
return {
traces: [],
total: 0,
done: 0,
pass: 0,
fail: 0,
error: 0,
running: 0,
costUsd: 0,
tokIn: 0,
tokOut: 0,
tokCache: 0,
score: null,
metrics: {},
};
}
function readEditSnapshot(jobDir: string): BenchmarkSnapshot {
const file = path.join(jobDir, "result.json");
if (!fs.existsSync(file)) return emptySnapshot();
const result: EditResult = JSON.parse(fs.readFileSync(file, "utf8"));
const traces: BenchmarkTrace[] = [];
let tokIn = 0;
let tokOut = 0;
for (const task of result.tasks) {
for (const run of task.runs) {
tokIn += run.tokens.input;
tokOut += run.tokens.output;
const runNumber = run.runIndex + 1;
traces.push({
name: `${task.id}__${runNumber}`,
task: task.id,
status: run.success ? "pass" : run.error ? "error" : "fail",
reward: run.success ? 1 : 0,
costUsd: 0,
durationMs: run.duration,
detail: JSON.stringify({ name: task.name, error: run.error ?? null, tools: run.toolCalls ?? null }),
tracePath: path.join("result.dump", task.id.replace(/[^a-zA-Z0-9._-]/g, "_"), `run-${runNumber}.md`),
});
}
}
const pass = traces.filter(trace => trace.status === "pass").length;
const error = traces.filter(trace => trace.status === "error").length;
return {
traces,
total: result.summary.totalRuns,
done: traces.length,
pass,
fail: traces.length - pass,
error,
running: Math.max(0, result.summary.totalRuns - traces.length),
costUsd: 0,
tokIn,
tokOut,
tokCache: 0,
score: result.summary.taskSuccessRate,
metrics: {
task_success_rate: result.summary.taskSuccessRate,
edit_success_rate: result.summary.editSuccessRate,
},
};
}
function readSnapcompactSnapshot(jobDir: string): BenchmarkSnapshot {
const recordsFile = path.join(jobDir, "records.jsonl");
const summaryFile = path.join(jobDir, "summary.json");
if (!fs.existsSync(recordsFile)) return emptySnapshot();
const records = fs
.readFileSync(recordsFile, "utf8")
.split("\n")
.filter(Boolean)
.map(line => JSON.parse(line) as SnapRecord);
const traces = records.map(
(record, index): BenchmarkTrace => ({
name: `${record.cond}__${record.chunk}__${index + 1}`,
task: `${record.cond}:${record.chunk}`,
status: record.f1 > 0 ? "pass" : "fail",
reward: record.f1,
costUsd: 0,
durationMs: 0,
detail: JSON.stringify({ question: record.q, answer: record.answer, golds: record.golds, em: record.em }),
tracePath: `record:${index + 1}`,
}),
);
let rows: SnapSummaryRow[] = [];
if (fs.existsSync(summaryFile)) {
const summary: SnapSummary = JSON.parse(fs.readFileSync(summaryFile, "utf8"));
rows = summary.rows;
}
const samples = rows.reduce((sum, row) => sum + row.n, 0);
const weightedF1 = rows.reduce((sum, row) => sum + row.f1 * row.n, 0);
const weightedEm = rows.reduce((sum, row) => sum + row.em * row.n, 0);
const pass = traces.filter(trace => trace.status === "pass").length;
return {
traces,
total: traces.length,
done: traces.length,
pass,
fail: traces.length - pass,
error: 0,
running: 0,
costUsd: rows.reduce((sum, row) => sum + row.cost_usd, 0),
tokIn: rows.reduce((sum, row) => sum + row.tokens_in, 0),
tokOut: rows.reduce((sum, row) => sum + row.tokens_out, 0),
tokCache: rows.reduce((sum, row) => sum + row.cache_w + row.cache_r, 0),
score: samples > 0 ? weightedF1 / samples : null,
metrics: {
f1: samples > 0 ? weightedF1 / samples : null,
exact_match: samples > 0 ? weightedEm / samples : null,
},
};
}
/** Read and normalize the latest artifacts for a benchmark run. */
export function readBenchmarkSnapshot(benchmark: BenchmarkKind, jobDir: string): BenchmarkSnapshot {
if (benchmark === "edit") return readEditSnapshot(jobDir);
if (benchmark === "snapcompact") return readSnapcompactSnapshot(jobDir);
const trials = readTrials(jobDir);
const job = readJobResult(jobDir);
const totals = aggregate(trials, job, job?.nTotal ?? trials.length);
return {
traces: trials.map(trial => ({
name: trial.name,
task: trial.name.replace(/__[^_]+$/, ""),
status: trial.status,
reward: trial.reward,
costUsd: trial.costUsd,
durationMs: trial.durationMs,
detail: trial.detail,
tracePath: path.join(trial.name, "agent", "omp.txt"),
})),
total: totals.total,
done: totals.done,
pass: totals.pass,
fail: totals.fail,
error: totals.error,
running: totals.running,
costUsd: totals.costUsd,
tokIn: totals.tokIn,
tokOut: totals.tokOut,
tokCache: totals.tokCache,
score: totals.done > 0 ? totals.pass / totals.done : null,
metrics: { success_rate: totals.done > 0 ? totals.pass / totals.done : null },
};
}
@@ -1,6 +1,6 @@
import { describe, expect, it } from "bun:test";
import { armOf, experimentOf, summarizeArm } from "./experiments";
import type { RunRow, TrialRow } from "./store";
import type { RunRow, TraceRow } from "./store";
/**
* Contracts under test:
@@ -11,11 +11,13 @@ import type { RunRow, TrialRow } from "./store";
function runRow(overrides: Partial<RunRow>): RunRow {
return {
benchmark: "harbor",
jobName: "exp-arm",
dataset: "d",
agent: "omp",
models: "anthropic/claude-opus-4-8",
slide: null,
config: {},
role: "",
note: "",
status: "running",
@@ -33,11 +35,13 @@ function runRow(overrides: Partial<RunRow>): RunRow {
tokIn: 0,
tokOut: 0,
tokCache: 0,
score: null,
metrics: {},
...overrides,
};
}
function trialRow(overrides: Partial<TrialRow>): TrialRow {
function traceRow(overrides: Partial<TraceRow>): TraceRow {
return {
jobName: "exp-arm",
name: "task__x",
@@ -48,6 +52,7 @@ function trialRow(overrides: Partial<TrialRow>): TrialRow {
durationMs: 60_000,
detail: "",
updatedAt: Date.now(),
tracePath: null,
...overrides,
};
}
@@ -75,8 +80,8 @@ describe("summarizeArm", () => {
costUsd: 5,
}),
[
trialRow({ status: "pass", durationMs: 120_000 }),
trialRow({ name: "b__x", task: "b", status: "fail", reward: 0, durationMs: 240_000 }),
traceRow({ status: "pass", durationMs: 120_000 }),
traceRow({ name: "b__x", task: "b", status: "fail", reward: 0, durationMs: 240_000 }),
],
);
expect(running.arm).toBe("n8");
@@ -94,7 +99,7 @@ describe("summarizeArm", () => {
const finished = summarizeArm(
runRow({ jobName: "sb2-opus", status: "complete", nTotal: 20, done: 20, pass: 15, costUsd: 30 }),
[trialRow({})],
[traceRow({})],
);
expect(finished.projected).toBeNull();
expect(finished.costPerTask).toBeCloseTo(1.5, 5);
@@ -108,6 +113,6 @@ describe("summarizeArm", () => {
}),
[],
);
expect(arm.config).toBe("anthropic/claude-opus-4-8 → google/gemini-3.5-flash on first edit/write +plan");
expect(arm.config).toBe("harbor · anthropic/claude-opus-4-8 → google/gemini-3.5-flash on first edit/write +plan");
});
});
+5 -5
View File
@@ -3,7 +3,7 @@
* `sb2-gemini` → experiment `sb2`) so comparable arms can be charted together,
* with linear projections for arms still in flight.
*/
import type { RunRow, RunStore, TrialRow } from "./store";
import type { RunRow, RunStore, TraceRow } from "./store";
/** Linear extrapolation of a running arm to its full task count. */
export interface ArmProjection {
@@ -79,8 +79,8 @@ function slideLabel(slideJson: string | null): string {
}
}
export function summarizeArm(run: RunRow, trials: TrialRow[]): ArmSummary {
const decided = trials.filter(t => t.status === "pass" || t.status === "fail" || t.status === "error");
export function summarizeArm(run: RunRow, traces: TraceRow[]): ArmSummary {
const decided = traces.filter(t => t.status === "pass" || t.status === "fail" || t.status === "error");
const durations = decided.filter(t => t.durationMs > 0).map(t => t.durationMs);
const meanTrialMs = durations.length > 0 ? durations.reduce((a, b) => a + b, 0) / durations.length : null;
const passPct = decided.length > 0 ? (100 * run.pass) / decided.length : null;
@@ -102,7 +102,7 @@ export function summarizeArm(run: RunRow, trials: TrialRow[]): ArmSummary {
return {
run,
arm: armOf(run.jobName),
config: `${run.models}${slideLabel(run.slide)}`,
config: `${run.benchmark} · ${run.models}${slideLabel(run.slide)}`,
passPct,
costPerTask,
meanTrialMs,
@@ -150,7 +150,7 @@ export function experimentDetail(store: RunStore, id: string): ExperimentDetail
const matrix: ExperimentDetail["matrix"] = {};
const tasks = new Set<string>();
for (const run of runs) {
const trials = store.listTrials(run.jobName);
const trials = store.listTraces(run.jobName);
arms.push(summarizeArm(run, trials));
const cells: Record<string, { status: string; reward: number | null }> = {};
for (const t of trials) {
+87 -10
View File
@@ -104,13 +104,13 @@ describe("RunStore", () => {
expect(run?.running).toBe(1);
expect(run?.costUsd).toBeCloseTo(0.7, 5);
const trials = store.listTrials("job-a");
expect(trials.map(t => [t.task, t.status])).toEqual([
const traces = store.listTraces("job-a");
expect(traces.map(t => [t.task, t.status])).toEqual([
["alpha", "pass"],
["beta", "error"],
["gamma", "running"],
]);
expect(trials[1].detail).toBe("AgentTimeoutError");
expect(traces[1].detail).toBe("AgentTimeoutError");
// re-discover is idempotent
expect(store.discover()).toBe(0);
@@ -162,6 +162,7 @@ describe("RunStore", () => {
const store = new RunStore(jobsDir);
cleanups.push(() => store.close());
store.registerLaunch({
benchmark: "harbor",
jobName: "job-b",
dataset: "test-dataset@1.0",
agent: "omp",
@@ -175,7 +176,7 @@ describe("RunStore", () => {
});
describe("ManagerServer API", () => {
it("serves runs, trials, transcripts, and validates launches", async () => {
it("serves uniform runs, traces, and rejects invalid launches", async () => {
const jobsDir = makeJobsDir();
writeFixtureJob(jobsDir, "job-api");
const manager = new ManagerServer(jobsDir);
@@ -190,15 +191,15 @@ describe("ManagerServer API", () => {
const detailRes = await fetch(`${base}/api/runs/job-api`);
expect(detailRes.status).toBe(200);
const detail = (await detailRes.json()) as { run: { pass: number }; trials: Array<{ status: string }> };
const detail = (await detailRes.json()) as { run: { pass: number }; traces: Array<{ status: string }> };
expect(detail.run.pass).toBe(1);
expect(detail.trials).toHaveLength(3);
expect(detail.traces).toHaveLength(3);
const tr = await fetch(`${base}/api/runs/job-api/trials/alpha__abc/transcript?tail=10`);
const tr = await fetch(`${base}/api/runs/job-api/traces/alpha__abc?tail=10`);
expect(tr.status).toBe(200);
const transcript = (await tr.json()) as { entries: Array<{ kind: string; tools?: string[] }> };
expect(transcript.entries.map(e => e.kind)).toEqual(["assistant", "toolResult"]);
expect(transcript.entries[0].tools).toEqual(["read"]);
const trace = (await tr.json()) as { entries: Array<{ kind: string; tools?: string[] }> };
expect(trace.entries.map(e => e.kind)).toEqual(["assistant", "toolResult"]);
expect(trace.entries[0].tools).toEqual(["read"]);
const missing = await fetch(`${base}/api/runs/nope`);
expect(missing.status).toBe(404);
@@ -215,4 +216,80 @@ describe("ManagerServer API", () => {
};
expect(cancelUnknown.cancelled).toBe(false);
});
it("serves edit and SnapCompact metrics and native traces through one API", async () => {
const jobsDir = makeJobsDir();
const manager = new ManagerServer(jobsDir);
for (const benchmark of ["edit", "snapcompact"] as const) {
const jobName = `${benchmark}-arm`;
manager.store.registerLaunch({
benchmark,
jobName,
dataset: benchmark === "edit" ? "typescript-edit" : "squad-dev",
agent: benchmark,
models: ["test/model"],
pid: process.pid,
});
manager.store.markExit(jobName, 0);
}
const editDir = path.join(jobsDir, "edit-arm");
fs.writeFileSync(
path.join(editDir, "result.json"),
JSON.stringify({
tasks: [
{
id: "rename",
name: "Rename",
runs: [{ runIndex: 0, success: true, duration: 10, tokens: { input: 8, output: 2, reasoning: 0 } }],
},
],
summary: {
totalRuns: 1,
successfulRuns: 1,
taskSuccessRate: 1,
editSuccessRate: 1,
totalTokens: { input: 8, output: 2 },
},
}),
);
fs.mkdirSync(path.join(editDir, "result.dump", "rename"), { recursive: true });
fs.writeFileSync(path.join(editDir, "result.dump", "rename", "run-1.md"), "# conversation\n\nassistant answer");
const snapDir = path.join(jobsDir, "snapcompact-arm");
fs.writeFileSync(
path.join(snapDir, "records.jsonl"),
`${JSON.stringify({ cond: "text", chunk: 0, pos_rel: 0, q: "question", answer: "answer", golds: ["gold"], em: 0, f1: 0.5 })}\n`,
);
fs.writeFileSync(
path.join(snapDir, "summary.json"),
JSON.stringify({
rows: [{ n: 1, f1: 0.5, em: 0, cost_usd: 0.1, tokens_in: 10, tokens_out: 2, cache_w: 0, cache_r: 0 }],
}),
);
manager.store.syncAll();
const server = manager.start(0);
cleanups.push(() => {
void manager.stop();
});
const base = `http://localhost:${server.port}`;
const edit = (await (await fetch(`${base}/api/runs/edit-arm`)).json()) as {
run: { benchmark: string; metrics: Record<string, number> };
traces: Array<{ name: string }>;
};
expect(edit.run).toMatchObject({ benchmark: "edit", metrics: { task_success_rate: 1, edit_success_rate: 1 } });
const editTrace = (await (
await fetch(`${base}/api/runs/edit-arm/traces/${encodeURIComponent(edit.traces[0].name)}`)
).json()) as { entries: Array<{ kind: string; text: string }> };
expect(editTrace.entries).toEqual([{ kind: "conversation", text: "# conversation\n\nassistant answer" }]);
const snap = (await (await fetch(`${base}/api/runs/snapcompact-arm`)).json()) as {
run: { benchmark: string; metrics: Record<string, number> };
traces: Array<{ name: string }>;
};
expect(snap.run).toMatchObject({ benchmark: "snapcompact", metrics: { f1: 0.5, exact_match: 0 } });
const snapTrace = (await (
await fetch(`${base}/api/runs/snapcompact-arm/traces/${encodeURIComponent(snap.traces[0].name)}`)
).json()) as { entries: Array<{ kind: string }> };
expect(snapTrace.entries.map(entry => entry.kind)).toEqual(["question", "answer", "reference"]);
});
});
+4 -4
View File
@@ -11,9 +11,9 @@
* process renders a live dashboard (progress / success% / spend / tokens / ETA)
* by polling each trial's `result.json`. On completion it writes a markdown report.
*
* bun src/runner.ts --model anthropic/claude-sonnet-4-6 --tasks 20 --concurrency 4
* bun src/runner.ts --agent oracle --tasks 2 # cheap pipeline smoke
* bun src/runner.ts --help
* harbor-manager harbor --model anthropic/claude-sonnet-4-6 --tasks 20 --concurrency 4
* harbor-manager harbor --agent oracle --tasks 2 # cheap pipeline smoke
* harbor-manager harbor --help
*/
import { spawnSync } from "node:child_process";
import * as fs from "node:fs";
@@ -108,7 +108,7 @@ function defaultConfig(): Config {
const HELP = `harbor-manager runner (local omp)
Usage: bun src/runner.ts [options] [-- <extra harbor args>]
Usage: harbor-manager harbor [options] [-- <extra harbor args>]
Commands:
cleanup Force-remove ALL leftover Harbor containers + networks, then exit
+113 -46
View File
@@ -6,18 +6,20 @@
* bun src/server.ts [--port 4700] [--jobs-dir <path>]
*
* API:
* GET /api/experiments → experiment summaries across all benchmarks
* GET /api/runs → RunRow[]
* POST /api/runs → launch a run (JSON body, see LaunchRequest)
* GET /api/runs/:name → { run, trials }
* DELETE /api/runs/:name → cancel a manager-launched run
* GET /api/runs/:name/trials/:trial/transcript?tail=N[&raw=1]
* POST /api/runs → launch any benchmark
* GET /api/runs/:name → { run, traces }
* DELETE /api/runs/:name → cancel a managed run
* GET /api/runs/:name/traces/:trace → normalized trace
* GET /api/events → SSE: run-list snapshots on change
*/
import * as fs from "node:fs";
import * as path from "node:path";
import type { Server, Subprocess } from "bun";
import { BENCHMARK_DEFINITIONS } from "./benchmarks";
import { buildExperiments, experimentDetail, experimentOf } from "./experiments";
import { type RunRole, RunStore } from "./store";
import { type BenchmarkKind, type RunRole, RunStore } from "./store";
/** PUT /api/experiments/:id body — goal and per-run role/note metadata. */
export interface ExperimentMetaUpdate {
@@ -33,6 +35,8 @@ const DEFAULT_JOBS_DIR = path.join(REPO_ROOT, "runs", "harbor");
/** POST /api/runs body. Mirrors the runner CLI surface we actually use. */
export interface LaunchRequest {
/** Benchmark adapter to execute. */
benchmark?: BenchmarkKind;
model: string;
dataset?: string;
/** Task count for a dataset sample, or omit when `include` is given. */
@@ -40,12 +44,14 @@ export interface LaunchRequest {
/** Explicit task names (passed as repeated --include). */
include?: string[];
concurrency?: number;
/** SnapCompact conditions; ignored by other benchmarks. */
conditions?: string[];
timeoutMultiplier?: number;
attempts?: number;
agent?: string;
jobName?: string;
webSearch?: boolean;
slide?: { model: string; turns?: number; onAction?: boolean; plan?: boolean };
slide?: { model: string; turns?: number; onAction?: boolean; plan?: boolean; checklist?: boolean };
/** Role of this run inside its experiment (baseline vs treatment). */
role?: RunRole;
/** One-line description of what this arm tests. */
@@ -180,6 +186,9 @@ export class ManagerServer {
});
}
if (p === "/api/events") return this.#sseResponse();
if (p === "/api/benchmarks" && request.method === "GET") {
return Response.json(BENCHMARK_DEFINITIONS);
}
if (p === "/api/experiments" && request.method === "GET") {
return Response.json(buildExperiments(this.#store));
}
@@ -207,15 +216,15 @@ export class ManagerServer {
if (request.method === "DELETE") return Response.json(this.cancel(jobName));
const run = this.#store.syncRun(jobName);
if (!run) return Response.json({ error: "run not found" }, { status: 404 });
return Response.json({ run, trials: this.#store.listTrials(jobName) });
return Response.json({ run, traces: this.#store.listTraces(jobName) });
}
const trialMatch = p.match(/^\/api\/runs\/([^/]+)\/trials\/([^/]+)\/transcript$/);
if (trialMatch) {
const jobName = decodeURIComponent(trialMatch[1]);
const trial = decodeURIComponent(trialMatch[2]);
const traceMatch = p.match(/^\/api\/runs\/([^/]+)\/traces\/([^/]+)$/);
if (traceMatch) {
const jobName = decodeURIComponent(traceMatch[1]);
const trace = decodeURIComponent(traceMatch[2]);
const tail = Number(url.searchParams.get("tail") ?? "120");
const raw = url.searchParams.get("raw") === "1";
return this.#transcript(jobName, trial, tail, raw);
return this.#trace(jobName, trace, tail, raw);
}
return Response.json({ error: "not found" }, { status: 404 });
} catch (err) {
@@ -248,23 +257,60 @@ export class ManagerServer {
});
}
/** Spawn the CLI runner for `request` and register the run. */
/** Launch any supported benchmark and register it in the uniform run store. */
launch(request: LaunchRequest): { jobName: string; pid: number } {
if (!request.model) throw new Error("model is required");
const dataset = request.dataset ?? "terminal-bench@2.0";
const benchmark = request.benchmark ?? "harbor";
if (benchmark !== "harbor" && benchmark !== "edit" && benchmark !== "snapcompact") {
throw new Error(`unsupported benchmark: ${benchmark}`);
}
const dataset =
request.dataset ??
(benchmark === "harbor" ? "terminal-bench@2.0" : benchmark === "edit" ? "typescript-edit" : "squad-dev");
const stamp = new Date().toISOString().replace(/[:.]/g, "-").slice(0, 19);
const modelSlug = request.model.replace(/[^a-zA-Z0-9]+/g, "-");
const jobName = request.jobName ?? `${modelSlug}-${stamp}`;
if (this.#children.has(jobName) || this.#store.getRun(jobName)?.status === "running") {
throw new Error(`run ${jobName} is already running`);
}
const jobDir = path.join(this.jobsDir, jobName);
fs.mkdirSync(jobDir, { recursive: true });
const argv = ["bun", "src/runner.ts", "--model", request.model, "-d", dataset, "--job-name", jobName];
let argv: string[];
let cwd: string;
if (benchmark === "edit") {
cwd = PKG_DIR;
argv = ["bun", "adapters/edit/cli.ts", "--model", request.model, "--output", path.join(jobDir, "result.json")];
if (request.tasks !== undefined) argv.push("--max-tasks", String(request.tasks));
if (request.include?.length) argv.push("--tasks", request.include.join(","));
if (request.concurrency !== undefined) argv.push("--task-concurrency", String(request.concurrency));
if (request.attempts !== undefined) argv.push("--runs", String(request.attempts));
} else if (benchmark === "snapcompact") {
cwd = PKG_DIR;
argv = ["uv", "run", "src/adapters/snapcompact.py", "--model", request.model, "--output-dir", jobDir];
if (request.tasks !== undefined) argv.push("--limit-paras", String(request.tasks));
if (request.concurrency !== undefined) argv.push("--workers", String(request.concurrency));
if (request.conditions?.length) argv.push("--conditions", request.conditions.join(","));
} else {
cwd = PKG_DIR;
argv = [
"bun",
"src/runner.ts",
"--model",
request.model,
"-d",
dataset,
"--job-name",
jobName,
"--jobs-dir",
this.jobsDir,
];
if (request.agent) argv.push("--agent", request.agent);
if (request.tasks !== undefined) argv.push("--tasks", String(request.tasks));
if (request.concurrency !== undefined) argv.push("--concurrency", String(request.concurrency));
if (request.attempts !== undefined) argv.push("--attempts", String(request.attempts));
if (request.timeoutMultiplier !== undefined) argv.push("--timeout-multiplier", String(request.timeoutMultiplier));
if (request.timeoutMultiplier !== undefined)
argv.push("--timeout-multiplier", String(request.timeoutMultiplier));
if (request.webSearch) argv.push("--web-search");
for (const task of request.include ?? []) argv.push("--include", task);
if (request.slide) {
@@ -274,26 +320,24 @@ export class ManagerServer {
argv.push("--agent-arg", "--reasoning-slide-turns", "--agent-arg", String(request.slide.turns));
} else throw new Error("slide requires turns or onAction");
if (request.slide.plan) argv.push("--agent-arg", "--reasoning-slide-plan");
// The runner only auto-routes gateway auth for the primary model's
// provider; declare the slide model's provider explicitly so its
// requests reach the gateway too.
if (request.slide.checklist) argv.push("--agent-arg", "--reasoning-slide-checklist");
const slideProvider = request.slide.model.split("/", 1)[0];
if (slideProvider) argv.push("--providers", slideProvider);
}
// Default to source mode (repo bind-mount, no rebuild); prebuilt binaries only on request.
if (request.prebuiltBinaries) {
for (const name of ["omp-linux-arm64", "omp-linux-x64"]) {
const binary = path.join(REPO_ROOT, "packages", "coding-agent", "dist", name);
if (fs.existsSync(binary)) argv.push("--binary", binary);
}
}
}
argv.push(...(request.extraArgs ?? []));
const logDir = path.join(this.jobsDir, "_manager", "logs");
fs.mkdirSync(logDir, { recursive: true });
const logFile = fs.openSync(path.join(logDir, `${jobName}.log`), "w");
const proc = Bun.spawn(argv, {
cwd: PKG_DIR,
cwd,
stdout: logFile,
stderr: logFile,
env: { ...process.env },
@@ -309,11 +353,13 @@ export class ManagerServer {
this.#tick();
});
this.#store.registerLaunch({
benchmark,
jobName,
dataset,
agent: request.agent ?? "omp",
models: [request.model],
slide: request.slide,
config: { ...request },
pid: proc.pid,
role: request.role,
note: request.note,
@@ -354,12 +400,44 @@ export class ManagerServer {
return { jobName, cancelled: false };
}
/** Compact transcript view of a trial's omp.txt session JSONL. */
#transcript(jobName: string, trial: string, tail: number, raw: boolean): Response {
const file = path.join(this.jobsDir, jobName, trial, "agent", "omp.txt");
if (!fs.existsSync(file)) return Response.json({ error: "transcript not found" }, { status: 404 });
const lines = fs.readFileSync(file, "utf8").split("\n").filter(Boolean);
/** Return a normalized trace regardless of the benchmark's native artifact format. */
#trace(jobName: string, traceName: string, tail: number, raw: boolean): Response {
const trace = this.#store.listTraces(jobName).find(item => item.name === traceName);
if (!trace?.tracePath) return Response.json({ error: "trace not found" }, { status: 404 });
const jobDir = path.join(this.jobsDir, jobName);
const n = Number.isSafeInteger(tail) && tail > 0 ? Math.min(tail, 2000) : 120;
if (trace.tracePath.startsWith("record:")) {
const lineNumber = Number(trace.tracePath.slice("record:".length));
const line = fs.readFileSync(path.join(jobDir, "records.jsonl"), "utf8").split("\n")[lineNumber - 1];
if (!line) return Response.json({ error: "trace not found" }, { status: 404 });
if (raw) return new Response(line, { headers: { "content-type": "application/json" } });
const record = JSON.parse(line) as Record<string, unknown>;
return Response.json({
jobName,
trace: traceName,
entries: [
{ kind: "question", text: String(record.q ?? "") },
{ kind: "answer", model: this.#store.getRun(jobName)?.models ?? "", text: String(record.answer ?? "") },
{ kind: "reference", text: JSON.stringify(record.golds ?? []) },
],
totalEvents: 3,
});
}
const file = path.resolve(jobDir, trace.tracePath);
if (!file.startsWith(`${path.resolve(jobDir)}${path.sep}`) || !fs.existsSync(file)) {
return Response.json({ error: "trace not found" }, { status: 404 });
}
const text = fs.readFileSync(file, "utf8");
if (!file.endsWith(".txt")) {
if (raw) return new Response(text, { headers: { "content-type": "text/plain; charset=utf-8" } });
return Response.json({
jobName,
trace: traceName,
entries: [{ kind: "conversation", text }],
totalEvents: 1,
});
}
const lines = text.split("\n").filter(Boolean);
if (raw) {
return new Response(lines.slice(-n).join("\n"), {
headers: { "content-type": "application/x-ndjson" },
@@ -373,41 +451,30 @@ export class ManagerServer {
} catch {
continue;
}
const type = event.type;
if (type === "message_end") {
if (event.type === "message_end") {
const message = event.message as Record<string, unknown> | undefined;
if (!message) continue;
const role = message.role;
if (role === "assistant") {
const content = Array.isArray(message.content)
? (message.content as Array<Record<string, unknown>>)
: [];
const text = content
const content = Array.isArray(message.content) ? (message.content as Array<Record<string, unknown>>) : [];
const body = content
.filter(block => block.type === "text")
.map(block => String(block.text ?? ""))
.join("\n");
if (message.role === "assistant") {
const tools = content.filter(block => block.type === "toolCall").map(block => String(block.name ?? "?"));
entries.push({ kind: "assistant", model: message.model ?? "", text, tools });
} else if (role === "toolResult") {
const content = Array.isArray(message.content)
? (message.content as Array<Record<string, unknown>>)
: [];
const text = content
.filter(block => block.type === "text")
.map(block => String(block.text ?? ""))
.join("\n");
entries.push({ kind: "assistant", model: message.model ?? "", text: body, tools });
} else if (message.role === "toolResult") {
entries.push({
kind: "toolResult",
tool: message.toolName ?? "?",
isError: message.isError === true,
text: text.length > 1600 ? `${text.slice(0, 1600)}…` : text,
text: body.length > 1600 ? `${body.slice(0, 1600)}…` : body,
});
}
} else if (type === "notice") {
} else if (event.type === "notice") {
entries.push({ kind: "notice", text: event.message ?? "" });
}
}
return Response.json({ jobName, trial, entries: entries.slice(-n), totalEvents: lines.length });
return Response.json({ jobName, trace: traceName, entries: entries.slice(-n), totalEvents: lines.length });
}
}
+94 -45
View File
@@ -10,19 +10,26 @@
import { Database } from "bun:sqlite";
import * as fs from "node:fs";
import * as path from "node:path";
import { aggregate, readJobResult, readTrials } from "./runner";
import { readBenchmarkSnapshot } from "./benchmarks";
import { readJobResult } from "./runner";
export type RunStatus = "running" | "complete" | "failed" | "cancelled";
/** Benchmark implementation that produced a run. */
export type BenchmarkKind = "harbor" | "edit" | "snapcompact";
/** How a run relates to its experiment's question. */
export type RunRole = "baseline" | "variant" | "";
export interface RunRow {
benchmark: BenchmarkKind;
jobName: string;
dataset: string;
agent: string;
models: string;
slide: string | null;
/** Benchmark-specific launch configuration. */
config: Record<string, unknown>;
/** Role inside the experiment (baseline vs treatment); "" when unspecified. */
role: RunRole;
/** One-line description of what this arm tests (e.g. "slide→flash after 8 turns"). */
@@ -42,9 +49,13 @@ export interface RunRow {
tokIn: number;
tokOut: number;
tokCache: number;
/** Benchmark-native aggregate score, when the benchmark exposes one. */
score: number | null;
/** Values keyed by the adapter's metric definitions. */
metrics: Record<string, number | null>;
}
export interface TrialRow {
export interface TraceRow {
jobName: string;
name: string;
task: string;
@@ -54,9 +65,12 @@ export interface TrialRow {
durationMs: number;
detail: string;
updatedAt: number;
/** Adapter-owned locator used by the uniform trace endpoint. */
tracePath: string | null;
}
export interface LaunchRecord {
benchmark: BenchmarkKind;
jobName: string;
dataset: string;
agent: string;
@@ -65,17 +79,20 @@ export interface LaunchRecord {
pid: number;
role?: RunRole;
note?: string;
config?: Record<string, unknown>;
}
const SCHEMA = `
CREATE TABLE IF NOT EXISTS runs (
job_name TEXT PRIMARY KEY,
benchmark TEXT NOT NULL DEFAULT 'harbor',
dataset TEXT NOT NULL DEFAULT '',
agent TEXT NOT NULL DEFAULT 'omp',
models TEXT NOT NULL DEFAULT '',
slide TEXT,
role TEXT NOT NULL DEFAULT '',
note TEXT NOT NULL DEFAULT '',
config_json TEXT NOT NULL DEFAULT '{}',
status TEXT NOT NULL DEFAULT 'running',
pid INTEGER,
exit_code INTEGER,
@@ -90,6 +107,8 @@ CREATE TABLE IF NOT EXISTS runs (
cost_usd REAL NOT NULL DEFAULT 0,
tok_in INTEGER NOT NULL DEFAULT 0,
tok_out INTEGER NOT NULL DEFAULT 0,
score REAL,
metrics_json TEXT NOT NULL DEFAULT '{}',
tok_cache INTEGER NOT NULL DEFAULT 0
);
CREATE TABLE IF NOT EXISTS trials (
@@ -101,6 +120,7 @@ CREATE TABLE IF NOT EXISTS trials (
cost_usd REAL NOT NULL DEFAULT 0,
duration_ms INTEGER NOT NULL DEFAULT 0,
detail TEXT NOT NULL DEFAULT '',
trace_path TEXT,
updated_at INTEGER NOT NULL,
PRIMARY KEY (job_name, name)
);
@@ -125,12 +145,25 @@ export class RunStore {
this.#db = new Database(dbPath ?? path.join(jobsDir, "_manager", "harbor-manager.sqlite"));
this.#db.run("PRAGMA journal_mode = WAL");
this.#db.run(SCHEMA);
// Migration for stores created before run roles/notes existed.
const columns = new Set(
const runColumns = new Set(
(this.#db.query("PRAGMA table_info(runs)").all() as Array<{ name: string }>).map(c => c.name),
);
if (!columns.has("role")) this.#db.run("ALTER TABLE runs ADD COLUMN role TEXT NOT NULL DEFAULT ''");
if (!columns.has("note")) this.#db.run("ALTER TABLE runs ADD COLUMN note TEXT NOT NULL DEFAULT ''");
if (!runColumns.has("role")) this.#db.run("ALTER TABLE runs ADD COLUMN role TEXT NOT NULL DEFAULT ''");
if (!runColumns.has("note")) this.#db.run("ALTER TABLE runs ADD COLUMN note TEXT NOT NULL DEFAULT ''");
if (!runColumns.has("benchmark")) {
this.#db.run("ALTER TABLE runs ADD COLUMN benchmark TEXT NOT NULL DEFAULT 'harbor'");
}
if (!runColumns.has("config_json")) {
this.#db.run("ALTER TABLE runs ADD COLUMN config_json TEXT NOT NULL DEFAULT '{}'");
}
if (!runColumns.has("score")) this.#db.run("ALTER TABLE runs ADD COLUMN score REAL");
if (!runColumns.has("metrics_json")) {
this.#db.run("ALTER TABLE runs ADD COLUMN metrics_json TEXT NOT NULL DEFAULT '{}'");
}
const traceColumns = new Set(
(this.#db.query("PRAGMA table_info(trials)").all() as Array<{ name: string }>).map(c => c.name),
);
if (!traceColumns.has("trace_path")) this.#db.run("ALTER TABLE trials ADD COLUMN trace_path TEXT");
}
close(): void {
@@ -139,26 +172,34 @@ export class RunStore {
/** Register a run this manager just launched (pid-owning). */
registerLaunch(launch: LaunchRecord): void {
this.#db.query("DELETE FROM trials WHERE job_name = ?").run(launch.jobName);
this.#db
.query(
`INSERT INTO runs (job_name, dataset, agent, models, slide, role, note, status, pid, created_at)
VALUES (?, ?, ?, ?, ?, ?, ?, 'running', ?, ?)
`INSERT INTO runs
(job_name, benchmark, dataset, agent, models, slide, role, note, config_json, status, pid, created_at)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'running', ?, ?)
ON CONFLICT(job_name) DO UPDATE SET
pid = excluded.pid, status = 'running',
benchmark = excluded.benchmark, pid = excluded.pid, status = 'running',
config_json = excluded.config_json,
role = CASE WHEN excluded.role != '' THEN excluded.role ELSE runs.role END,
note = CASE WHEN excluded.note != '' THEN excluded.note ELSE runs.note END`,
)
.run(
launch.jobName,
launch.benchmark,
launch.dataset,
launch.agent,
launch.models.join(","),
launch.slide ? JSON.stringify(launch.slide) : null,
launch.role ?? "",
launch.note ?? "",
JSON.stringify(launch.config ?? {}),
launch.pid,
Date.now(),
);
const jobDir = path.join(this.jobsDir, launch.jobName);
fs.mkdirSync(jobDir, { recursive: true });
fs.writeFileSync(path.join(jobDir, "manager.json"), JSON.stringify(launch, null, 2));
}
/** Upsert the experiment's stated goal. */
@@ -230,65 +271,68 @@ export class RunStore {
syncRun(jobName: string): RunRow | null {
const jobDir = path.join(this.jobsDir, jobName);
if (!fs.existsSync(jobDir)) return this.getRun(jobName);
const trials = readTrials(jobDir);
const job = readJobResult(jobDir);
const totals = aggregate(trials, job, job?.nTotal ?? trials.length);
const row = this.getRun(jobName);
if (!row) return null;
const snapshot = readBenchmarkSnapshot(row.benchmark, jobDir);
const now = Date.now();
const upsert = this.#db.query(
`INSERT INTO trials (job_name, name, task, status, reward, cost_usd, duration_ms, detail, updated_at)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
`INSERT INTO trials
(job_name, name, task, status, reward, cost_usd, duration_ms, detail, trace_path, updated_at)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
ON CONFLICT(job_name, name) DO UPDATE SET
status = excluded.status, reward = excluded.reward, cost_usd = excluded.cost_usd,
duration_ms = excluded.duration_ms, detail = excluded.detail, updated_at = excluded.updated_at`,
duration_ms = excluded.duration_ms, detail = excluded.detail,
trace_path = excluded.trace_path, updated_at = excluded.updated_at`,
);
const tx = this.#db.transaction(() => {
for (const t of trials) {
for (const trace of snapshot.traces) {
upsert.run(
jobName,
t.name,
t.name.replace(/__[^_]+$/, ""),
t.status,
t.reward,
t.costUsd,
t.durationMs,
t.detail,
trace.name,
trace.task,
trace.status,
trace.reward,
trace.costUsd,
trace.durationMs,
trace.detail,
trace.tracePath,
now,
);
}
this.#db
.query(
`UPDATE runs SET n_total = ?, done = ?, pass = ?, fail = ?, error = ?, running = ?,
cost_usd = ?, tok_in = ?, tok_out = ?, tok_cache = ? WHERE job_name = ?`,
cost_usd = ?, tok_in = ?, tok_out = ?, tok_cache = ?, score = ?, metrics_json = ?
WHERE job_name = ?`,
)
.run(
totals.total,
totals.done,
totals.pass,
totals.fail,
totals.error,
totals.running,
totals.costUsd,
totals.tokIn,
totals.tokOut,
totals.tokCache,
snapshot.total,
snapshot.done,
snapshot.pass,
snapshot.fail,
snapshot.error,
snapshot.running,
snapshot.costUsd,
snapshot.tokIn,
snapshot.tokOut,
snapshot.tokCache,
snapshot.score,
JSON.stringify(snapshot.metrics),
jobName,
);
// Foreign runs (no owning pid and never finalized by markExit) infer
// their lifecycle from Harbor's job-level result: `finished_at` is
// terminal; a missing terminal marker with a fresh job dir means still
// running; stale for >30 min means the harness died mid-run.
const row = this.getRun(jobName);
if (row && row.pid === null && row.finishedAt === null && row.status !== "cancelled") {
const job2 = readJobResult(jobDir);
// Historical Harbor runs have no owning process. Infer their terminal
// state from result metadata or directory freshness.
if (row.benchmark === "harbor" && row.pid === null && row.finishedAt === null && row.status !== "cancelled") {
const result = readJobResult(jobDir);
let status: RunStatus;
let finishedAt: number | null = null;
if (job2?.finishedAt != null) {
if (result?.finishedAt != null) {
status = "complete";
finishedAt = job2.finishedAt;
finishedAt = result.finishedAt;
} else if (jobDirFresh(jobDir)) {
status = "running";
} else {
status = totals.done > 0 && totals.done >= totals.total ? "complete" : "failed";
status = snapshot.done > 0 && snapshot.done >= snapshot.total ? "complete" : "failed";
finishedAt = jobDirMtime(jobDir);
}
if (status !== row.status) {
@@ -343,7 +387,7 @@ export class RunStore {
return rows.map(rowToRun);
}
listTrials(jobName: string): TrialRow[] {
listTraces(jobName: string): TraceRow[] {
const rows = this.#db.query("SELECT * FROM trials WHERE job_name = ? ORDER BY name").all(jobName) as Array<
Record<string, unknown>
>;
@@ -357,17 +401,20 @@ export class RunStore {
durationMs: Number(r.duration_ms),
detail: String(r.detail),
updatedAt: Number(r.updated_at),
tracePath: r.trace_path === null ? null : String(r.trace_path),
}));
}
}
function rowToRun(r: Record<string, unknown>): RunRow {
return {
benchmark: String(r.benchmark ?? "harbor") as BenchmarkKind,
jobName: String(r.job_name),
dataset: String(r.dataset),
agent: String(r.agent),
models: String(r.models),
slide: r.slide === null ? null : String(r.slide),
config: JSON.parse(String(r.config_json ?? "{}")),
role: String(r.role ?? "") as RunRole,
note: String(r.note ?? ""),
status: String(r.status) as RunStatus,
@@ -385,6 +432,8 @@ function rowToRun(r: Record<string, unknown>): RunRow {
tokIn: Number(r.tok_in),
tokOut: Number(r.tok_out),
tokCache: Number(r.tok_cache),
score: r.score === null ? null : Number(r.score),
metrics: JSON.parse(String(r.metrics_json ?? "{}")),
};
}
+131 -23
View File
@@ -6,7 +6,7 @@
* #/exp/<id> experiment detail — arm table, dithered comparison charts
* (projected values for in-flight arms, dimmed), task matrix
* #/runs flat run list (legacy view)
* #/runs/<name> run detail — trial grid + live transcript tail
* #/runs/<name> run detail — normalized trace grid + live trace viewer
*/
import { useCallback, useEffect, useRef, useState } from "react";
import { createRoot } from "react-dom/client";
@@ -14,11 +14,13 @@ import { createRoot } from "react-dom/client";
// ── api types (mirrors server modules) ──────────────────────────────────────
interface RunRow {
benchmark: "harbor" | "edit" | "snapcompact";
jobName: string;
dataset: string;
agent: string;
models: string;
slide: string | null;
config: Record<string, unknown>;
role: "baseline" | "variant" | "";
note: string;
status: "running" | "complete" | "failed" | "cancelled";
@@ -32,9 +34,11 @@ interface RunRow {
error: number;
running: number;
costUsd: number;
score: number | null;
metrics: Record<string, number | null>;
}
interface TrialRow {
interface TraceRow {
name: string;
task: string;
status: string;
@@ -193,7 +197,7 @@ function ExperimentsIndex() {
<a
key={exp.id}
href={`#/exp/${encodeURIComponent(exp.id)}`}
className="flex items-center gap-6 rounded-lg border border-zinc-800 bg-zinc-900/60 px-5 py-4 hover:border-zinc-600"
className="flex min-w-0 items-center gap-6 overflow-hidden rounded-lg border border-zinc-800 bg-zinc-900/60 px-5 py-4 hover:border-zinc-600"
>
<div className="w-40 shrink-0">
<div className="font-semibold">{exp.id}</div>
@@ -375,6 +379,53 @@ const CELL_CLASS: Record<string, string> = {
running: "bg-sky-500 animate-pulse",
};
/**
* The comparison anchor for an experiment: the completed baseline arm with the
* highest pass rate (the "ceiling" a reasoning slide tries to preserve). Ties
* break toward the cheaper arm. Returns null when no baseline has finished data.
*/
function pickReferenceArm(arms: ArmSummary[]): ArmSummary | null {
let ref: ArmSummary | null = null;
for (const a of arms) {
if (a.run.role !== "baseline" || a.passPct === null) continue;
if (
ref === null ||
a.passPct > (ref.passPct ?? -1) ||
(a.passPct === ref.passPct && (a.costPerTask ?? Infinity) < (ref.costPerTask ?? Infinity))
) {
ref = a;
}
}
return ref;
}
/**
* Signed, colour-coded offset of a metric from the reference arm. `points`
* shows absolute percentage-point difference (pass rate); `relative` shows a
* percentage change (cost, time). `higherBetter` decides which direction is green.
*/
function Delta({
value,
reference,
mode,
higherBetter,
}: {
value: number | null;
reference: number | null;
mode: "points" | "relative";
higherBetter: boolean;
}) {
if (value === null || reference === null) return null;
const raw =
mode === "points" ? value - reference : reference === 0 ? Number.NaN : ((value - reference) / reference) * 100;
if (!Number.isFinite(raw) || Math.abs(raw) < 0.5) {
return <span className="ml-1 text-[10px] text-zinc-600">≈</span>;
}
const good = higherBetter ? raw > 0 : raw < 0;
const body = `${raw > 0 ? "+" : "−"}${Math.abs(raw).toFixed(0)}${mode === "relative" ? "%" : ""}`;
return <span className={`ml-1 text-[10px] ${good ? "text-emerald-500" : "text-red-400"}`}>({body})</span>;
}
function ExperimentPage({ id }: { id: string }) {
const detail = usePolled<ExperimentDetail>(`/api/experiments/${encodeURIComponent(id)}`, 3000);
if (!detail) return <div className="p-10 text-zinc-500">loading…</div>;
@@ -394,12 +445,19 @@ function ExperimentPage({ id }: { id: string }) {
a => (a.meanTrialMs === null ? null : a.meanTrialMs / 60000),
p => p.meanTrialMs / 60000,
);
const ref = pickReferenceArm(arms);
return (
<div className="mx-auto max-w-7xl p-6">
<div className="mb-1 flex items-baseline gap-4">
<h2 className="text-lg font-semibold">{id}</h2>
<span className="text-xs text-zinc-500">
{arms.length} arms · {tasks.length} tasks
{ref && (
<>
{" "}
· Δ vs <span className="text-zinc-400">{ref.arm}</span>
</>
)}
</span>
</div>
{goal && <p className="mb-4 max-w-4xl text-sm text-zinc-400">{goal}</p>}
@@ -430,6 +488,14 @@ function ExperimentPage({ id }: { id: string }) {
{arm.run.role}
</span>
)}
{ref?.arm === arm.arm && (
<span
className="ml-1 text-[10px] text-zinc-500"
title="reference arm (highest-pass baseline); deltas are measured against it"
>
ref
</span>
)}
</td>
<td
className="max-w-md truncate pr-4 text-xs text-zinc-400"
@@ -447,12 +513,33 @@ function ExperimentPage({ id }: { id: string }) {
<td className="pr-4">
{arm.passPct !== null ? `${arm.passPct.toFixed(0)}%` : "—"}
{arm.projected && <span className="text-zinc-500"> →{arm.projected.passPct.toFixed(0)}%</span>}
{ref && ref.arm !== arm.arm && (
<Delta value={arm.passPct} reference={ref.passPct} mode="points" higherBetter />
)}
</td>
<td className="pr-4">
{arm.costPerTask !== null ? fmtUsd(arm.costPerTask) : "—"}
{arm.projected && <span className="text-zinc-500"> Σ{fmtUsd(arm.projected.totalCostUsd)}</span>}
{ref && ref.arm !== arm.arm && (
<Delta
value={arm.costPerTask}
reference={ref.costPerTask}
mode="relative"
higherBetter={false}
/>
)}
</td>
<td className="pr-4">
{arm.meanTrialMs !== null ? fmtMin(arm.meanTrialMs) : "—"}
{ref && ref.arm !== arm.arm && (
<Delta
value={arm.meanTrialMs}
reference={ref.meanTrialMs}
mode="relative"
higherBetter={false}
/>
)}
</td>
<td className="pr-4">{arm.meanTrialMs !== null ? fmtMin(arm.meanTrialMs) : "—"}</td>
<td>
<a
className="text-xs text-zinc-500 underline hover:text-zinc-300"
@@ -534,23 +621,23 @@ function useRunsSse(): RunRow[] | null {
function RunsPage({ selected }: { selected: string | null }) {
const runs = useRunsSse();
const detail = usePolled<{ run: RunRow; trials: TrialRow[] }>(
const detail = usePolled<{ run: RunRow; traces: TraceRow[] }>(
selected ? `/api/runs/${encodeURIComponent(selected)}` : null,
2500,
);
const [trial, setTrial] = useState<string | null>(null);
const transcript = usePolled<{ entries: TranscriptEntry[] }>(
selected && trial
? `/api/runs/${encodeURIComponent(selected)}/trials/${encodeURIComponent(trial)}/transcript?tail=60`
const [trace, setTrace] = useState<string | null>(null);
const traceData = usePolled<{ entries: TranscriptEntry[] }>(
selected && trace
? `/api/runs/${encodeURIComponent(selected)}/traces/${encodeURIComponent(trace)}?tail=60`
: null,
2500,
);
const transcriptRef = useRef<HTMLDivElement | null>(null);
const traceRef = useRef<HTMLDivElement | null>(null);
useEffect(() => {
if (!transcript) return;
const el = transcriptRef.current;
if (!traceData) return;
const el = traceRef.current;
if (el) el.scrollTop = el.scrollHeight;
}, [transcript]);
}, [traceData]);
const cancel = useCallback(async (name: string) => {
if (confirm(`stop ${name}?`)) await fetch(`/api/runs/${encodeURIComponent(name)}`, { method: "DELETE" });
}, []);
@@ -578,6 +665,7 @@ function RunsPage({ selected }: { selected: string | null }) {
>
<td className="px-3 py-1.5" title={r.models}>
{r.jobName}
<div className="text-[10px] uppercase tracking-wide text-zinc-600">{r.benchmark}</div>
{(r.note || r.role) && (
<div className="text-[11px] text-zinc-500">
{r.role && (
@@ -622,34 +710,42 @@ function RunsPage({ selected }: { selected: string | null }) {
<div className="border-b border-zinc-800 px-4 py-2 text-sm">
<span className="font-semibold">{detail.run.jobName}</span> <Chip label={detail.run.status} />{" "}
<span className="text-xs text-zinc-500">
{detail.run.dataset} · {detail.run.models}
{detail.run.benchmark} · {detail.run.dataset} · {detail.run.models}
{detail.run.score !== null ? ` · score ${(100 * detail.run.score).toFixed(1)}%` : ""}
{detail.run.slide ? ` → ${detail.run.slide}` : ""}
</span>
<div className="mt-1 flex gap-3 text-xs text-zinc-400">
{Object.entries(detail.run.metrics).map(([key, value]) => (
<span key={key}>
{key.replaceAll("_", " ")}: {value === null ? "—" : `${(100 * value).toFixed(1)}%`}
</span>
))}
</div>
</div>
<div className="min-h-0 flex-1 overflow-auto">
<table className="w-full text-sm">
<tbody>
{detail.trials.map(t => (
{detail.traces.map(t => (
<tr
key={t.name}
onClick={() => setTrial(t.name)}
className={`cursor-pointer border-t border-zinc-800/60 hover:bg-zinc-900 ${t.name === trial ? "bg-zinc-900" : ""}`}
onClick={() => setTrace(t.name)}
className={`cursor-pointer border-t border-zinc-800/60 hover:bg-zinc-900 ${t.name === trace ? "bg-zinc-900" : ""}`}
>
<td className="px-4 py-1">{t.task}</td>
<td>
<Chip label={t.status} />
</td>
<td>{t.reward === null ? "—" : t.reward.toFixed(3)}</td>
<td>{fmtUsd(t.costUsd)}</td>
<td>{t.durationMs ? fmtMin(t.durationMs) : "—"}</td>
<td className="text-xs text-zinc-500">{t.detail}</td>
</tr>
))}
</tbody>
</table>
</div>
{trial && (
<div ref={transcriptRef} className="h-2/5 overflow-auto border-t border-zinc-800 bg-zinc-950/60">
{(transcript?.entries ?? []).map((e, i) => (
{trace && (
<div ref={traceRef} className="h-2/5 overflow-auto border-t border-zinc-800 bg-zinc-950/60">
{(traceData?.entries ?? []).map((e, i) => (
// biome-ignore lint/suspicious/noArrayIndexKey: tail window, entries have no ids
<div key={i} className="border-b border-zinc-900 px-4 py-2">
<div className="text-xs text-zinc-500">
@@ -687,7 +783,7 @@ function LaunchForm({ onDone }: { onDone: () => void }) {
async (ev: React.FormEvent<HTMLFormElement>) => {
ev.preventDefault();
const f = new FormData(ev.currentTarget);
const body: Record<string, unknown> = { model: f.get("model") };
const body: Record<string, unknown> = { benchmark: f.get("benchmark"), model: f.get("model") };
if (f.get("jobName")) body.jobName = f.get("jobName");
if (f.get("dataset")) body.dataset = f.get("dataset");
if (f.get("tasks")) body.tasks = Number(f.get("tasks"));
@@ -699,6 +795,12 @@ function LaunchForm({ onDone }: { onDone: () => void }) {
.map(s => s.trim())
.filter(Boolean);
}
if (f.get("conditions")) {
body.conditions = String(f.get("conditions"))
.split(",")
.map(s => s.trim())
.filter(Boolean);
}
if (f.get("goal")) body.goal = f.get("goal");
if (f.get("role")) body.role = f.get("role");
if (f.get("note")) body.note = f.get("note");
@@ -724,10 +826,15 @@ function LaunchForm({ onDone }: { onDone: () => void }) {
const input = "rounded border border-zinc-700 bg-zinc-950 px-2 py-1 text-sm";
return (
<form onSubmit={submit} className="grid grid-cols-4 gap-2 border-b border-zinc-800 bg-zinc-900/70 p-4 text-sm">
<select name="benchmark" className={input}>
<option value="harbor">Harbor</option>
<option value="edit">TypeScript edit</option>
<option value="snapcompact">SnapCompact</option>
</select>
<input name="model" placeholder="model (required)" required className={input} />
<input name="dataset" placeholder="dataset (terminal-bench@2.0)" className={input} />
<input name="jobName" placeholder="job name (exp-arm)" className={input} />
<input name="tasks" type="number" placeholder="tasks" className={input} />
<input name="tasks" type="number" placeholder="task/passages limit" className={input} />
<input name="concurrency" type="number" placeholder="concurrency" className={input} />
<input name="timeoutMultiplier" type="number" step="0.5" placeholder="timeout ×" className={input} />
<input name="slideModel" placeholder="slide model" className={input} />
@@ -741,6 +848,7 @@ function LaunchForm({ onDone }: { onDone: () => void }) {
<input type="checkbox" name="slidePlan" /> plan nudge
</label>
<input name="include" placeholder="include tasks, comma-sep" className={`${input} col-span-2`} />
<input name="conditions" placeholder="SnapCompact conditions, comma-sep" className={`${input} col-span-2`} />
<input
name="goal"
placeholder="experiment goal (what question does this answer?)"
@@ -12,21 +12,13 @@
"url": "git+https://github.com/can1357/oh-my-pi.git",
"directory": "packages/typescript-edit-benchmark"
},
"main": "./src/index.ts",
"exports": {
".": {
"types": "./src/index.ts",
"import": "./src/index.ts"
},
"./*": {
"types": "./src/*.ts",
"import": "./src/*.ts"
},
"./*.js": "./src/*.ts"
},
"bin": {
"typescript-edit-benchmark": "src/index.ts"
},
"scripts": {
"check": "biome check . && bun run check:types",
"check:types": "tsgo -p tsconfig.json --noEmit",
@@ -34,8 +26,7 @@
"test": "bun test --parallel",
"fix": "biome check --write --unsafe .",
"fmt": "biome format --write .",
"generate": "bun run src/generate.ts --typescript-dir /tmp/pi-mono-source --count-per-type 4",
"start": "bun run src/index.ts"
"generate": "bun run src/generate.ts --typescript-dir /tmp/pi-mono-source --count-per-type 4"
},
"dependencies": {
"@babel/generator": "catalog:",
@@ -1,758 +0,0 @@
#!/usr/bin/env bun
/**
* Edit benchmark CLI entry point.
*
* Usage:
* bun run bench:edit --model anthropic/claude-sonnet-4-5
* bun run bench:edit --tasks core-memory-recall,operations-division
* bun run bench:edit --runs 5 --output report.md
* bun run bench:edit --fixtures fixtures.tar.gz
*/
import * as fs from "node:fs";
import * as path from "node:path";
import { parseArgs } from "node:util";
import { type ResolvedThinkingLevel, ThinkingLevel } from "@oh-my-pi/pi-agent-core";
import { Effort, THINKING_EFFORTS } from "@oh-my-pi/pi-ai";
import { padding, visibleWidth } from "@oh-my-pi/pi-tui";
import { postmortem, TempDir } from "@oh-my-pi/pi-utils";
import { generateJsonReport, generateReport } from "./report";
import {
type BenchmarkConfig,
type BenchmarkResult,
buildBenchmarkResult,
type ProgressEvent,
percentile,
runBenchmark,
} from "./runner";
import { type EditTask, loadTasksFromDir, validateFixturesFromDir } from "./tasks";
const COLOR_ENABLED = Boolean(process.stdout.isTTY) && !process.env.NO_COLOR;
const ANSI = {
reset: "\x1b[0m",
bold: "\x1b[1m",
dim: "\x1b[2m",
red: "\x1b[31m",
green: "\x1b[32m",
yellow: "\x1b[33m",
blue: "\x1b[34m",
magenta: "\x1b[35m",
cyan: "\x1b[36m",
} as const;
const RUNS_DIR = path.resolve(import.meta.dir, "..", "..", "..", "runs");
fs.mkdirSync(RUNS_DIR, { recursive: true });
function paint(code: string, text: string): string {
return COLOR_ENABLED ? `${code}${text}${ANSI.reset}` : text;
}
function rateColor(percent: number): string {
if (percent >= 80) return ANSI.green;
if (percent >= 50) return ANSI.yellow;
return ANSI.red;
}
function parseThinkingLevel(value: string | null | undefined): ResolvedThinkingLevel | undefined {
return value !== undefined &&
value !== null &&
[ThinkingLevel.Off, ...THINKING_EFFORTS].includes(value as ResolvedThinkingLevel)
? (value as ResolvedThinkingLevel)
: undefined;
}
function generateReportFilename(config: BenchmarkConfig, format: "markdown" | "json"): string {
const modelName = config.model
.split("/")
.pop()!
.replace(/[^a-zA-Z0-9-]/g, "_");
const variant = config.editVariant ?? "replace";
const timestamp = new Date().toISOString().replace(/:/g, "-").replace(/\..+$/, "").replace(/Z$/, "Z");
const ext = format === "json" ? "json" : "md";
return path.join(RUNS_DIR, `${modelName}_${variant}_${timestamp}.${ext}`);
}
async function resolveConversationDumpDir(outputPath: string): Promise<string> {
const parsed = path.parse(outputPath);
const preferredPath = path.join(parsed.dir, `${parsed.name}.dump`);
try {
await fs.promises.stat(preferredPath);
} catch (error) {
if ((error as NodeJS.ErrnoException).code === "ENOENT") {
return preferredPath;
}
throw error;
}
const timestamp = new Date().toISOString().replace(/[:.]/g, "-");
return path.join(parsed.dir, `${parsed.name}.${timestamp}.dump`);
}
async function conversationDumpStatus(dumpDir: string): Promise<string> {
try {
const stat = await fs.promises.stat(dumpDir);
if (stat.isDirectory()) {
return `Conversation dumps written to: ${dumpDir}`;
}
return `Conversation dump path is not a directory: ${dumpDir}`;
} catch (error) {
if ((error as NodeJS.ErrnoException).code === "ENOENT") {
return `No conversation dumps written: ${dumpDir}`;
}
throw error;
}
}
function printUsage(tasks?: EditTask[]): void {
const taskList = tasks
? tasks.map(t => ` ${t.id.padEnd(30)} ${t.name}`).join("\n")
: " (use --list to see available tasks)";
console.log(`
Edit Benchmark - Evaluate patch application success rates
Usage:
bun run bench:edit [options]
Options:
--model <id> Provider/model ID, e.g. anthropic/claude-sonnet-4-20250514 (default)
--provider <id> Override provider (auto-detected from model prefix if omitted)
--thinking <level> Thinking level: off, minimal, low, medium, high, xhigh, max
--runs <n> Runs per task (default: 1)
--timeout <ms> Timeout per run in ms (default: 120000)
--connection-timeout <ms> Timeout for first event before fast-retry (default: 30000)
--task-concurrency <n> Max tasks to run in parallel (default: 16)
--tasks <ids> Comma-separated task IDs to run (default: all)
--max-tasks <n> Max tasks to sample (default: 80, 0 = all)
--fixtures <path> Fixtures directory or .tar.gz archive (default: built-in)
--edit-variant <v> Edit variant: any string (e.g. replace, patch, hashline, vim, atom, apply_patch), or auto (default: auto)
--edit-fuzzy <bool> Fuzzy matching: true, false, auto (default: auto)
--edit-fuzzy-threshold <n> Fuzzy threshold 0-1 or auto (default: auto)
--auto-format Auto-format output files after verify (debug only)
--guided Include an authoritative suggested edit payload (default: false)
--no-guided Disable guided mode
--max-attempts <n> Max prompt attempts per run (default: 1)
--no-op-retry-limit <n> Stop after repeated preventable no-op failures (default: 2)
--mutation-scope-window <n> Allowed line-distance from mutation target for hashline refs (default: 20)
--max-turns <n> Max turn_start events per attempt before failing (default: 30)
--output <file> Output file (default: run_<model>_<variant>_<fuzzy>_<threshold>_<timestamp>.md)
--format <fmt> Output format: markdown, json (default: markdown)
--check-fixtures Validate fixtures and exit
--require-edit-tool-call Require edit tool usage for success (default: false)
--require-read-tool-call Require read tool usage for success (default: false)
--no-edit-required Remove "must edit" prompt requirement (default: false)
--no-early-stop-on-match Don't short-circuit the run when output matches expected (default: false)
--list List available tasks and exit
--help Show this help message
Available Tasks:
${taskList}
Examples:
# Run full benchmark with default model
bun run bench:edit
# Run specific tasks
bun run bench:edit --tasks core-memory-recall,operations-division
# Compare different models
bun run bench:edit --model claude-sonnet-4-20250514 --output sonnet.md
bun run bench:edit --model claude-opus-4-5-20251101 --output opus.md
# Run with extended thinking
bun run bench:edit --thinking high --runs 5
# Run from a fixtures archive
bun run bench:edit --fixtures edit-fixtures.tar.gz
`);
}
async function resolveExtractedDir(tempDir: string): Promise<string> {
const entries = await fs.promises.readdir(tempDir, { withFileTypes: true });
const dirs = entries.filter(entry => entry.isDirectory());
const files = entries.filter(entry => entry.isFile());
if (dirs.length === 1 && files.length === 0) {
return path.join(tempDir, dirs[0]!.name);
}
return tempDir;
}
async function extractTarGz(archivePath: string): Promise<{ dir: string; cleanupDir: string }> {
const tempDirObj = await TempDir.create("@reach-benchmark-fixtures-");
const tempDir = tempDirObj.path();
try {
const bytes = await Bun.file(archivePath).arrayBuffer();
const archive = new Bun.Archive(bytes);
const files = await archive.files();
for (const [filePath, file] of files) {
const destPath = path.join(tempDir, filePath);
await Bun.write(destPath, file);
}
} catch (error) {
await tempDirObj.remove();
const message = error instanceof Error ? error.message : String(error);
throw new Error(`Failed to extract archive: ${message}`, { cause: error });
}
return { dir: await resolveExtractedDir(tempDir), cleanupDir: tempDir };
}
async function resolveFixtures(fixturesArg?: string): Promise<{ tasks: EditTask[]; cleanup?: () => Promise<void> }> {
fixturesArg ??= path.join(import.meta.dir, "../fixtures.tar.gz");
if (fixturesArg.endsWith(".tar.gz") || fixturesArg.endsWith(".tgz")) {
const extracted = await extractTarGz(fixturesArg);
return {
tasks: await loadTasksFromDir(extracted.dir),
cleanup: () => fs.promises.rm(extracted.cleanupDir, { recursive: true, force: true }),
};
}
return { tasks: await loadTasksFromDir(fixturesArg) };
}
async function main(): Promise<void> {
const { values } = parseArgs({
options: {
provider: { type: "string" },
model: { type: "string", default: "anthropic/claude-sonnet-4-20250514" },
thinking: { type: "string", default: "low" },
runs: { type: "string", default: "2" },
timeout: { type: "string", default: "120000" },
"connection-timeout": { type: "string", default: "30000" },
"max-turns": { type: "string", default: "30" },
"task-concurrency": { type: "string", default: "32" },
tasks: { type: "string" },
fixtures: { type: "string" },
output: { type: "string" },
format: { type: "string", default: "markdown" },
"check-fixtures": { type: "boolean", default: false },
"auto-format": { type: "boolean", default: false },
guided: { type: "boolean", default: false },
"no-guided": { type: "boolean", default: false },
"max-attempts": { type: "string", default: "1" },
"no-op-retry-limit": { type: "string", default: "2" },
"max-timeout-retries": { type: "string", default: "3" },
"max-provider-retries": { type: "string", default: "3" },
"mutation-scope-window": { type: "string", default: "20" },
"require-edit-tool-call": { type: "boolean", default: false },
"require-read-tool-call": { type: "boolean", default: false },
"no-edit-required": { type: "boolean", default: false },
"edit-variant": { type: "string" },
"edit-fuzzy": { type: "string" },
"edit-fuzzy-threshold": { type: "string" },
"no-in-process": { type: "boolean", default: false },
"no-early-stop-on-match": { type: "boolean", default: false },
"max-tasks": { type: "string", default: "80" },
list: { type: "boolean", default: false },
help: { type: "boolean", default: false },
},
allowPositionals: true,
});
// Extract provider for display/config purposes only.
// The full model string (e.g. "openrouter/google/gemini-2.5-flash-lite") is passed
// as --model to the CLI, which handles resolution via parseModelPattern.
const model = values.model!;
const slashIndex = model.indexOf("/");
const provider = values.provider ?? (slashIndex !== -1 ? model.slice(0, slashIndex) : "anthropic");
if (values.help) {
printUsage();
process.exit(0);
}
if (values["check-fixtures"] && values.fixtures) {
const issues = await validateFixturesFromDir(values.fixtures);
if (issues.length === 0) {
console.log("Fixtures OK");
process.exit(0);
}
console.error("Fixture validation failed:");
for (const issue of issues) {
console.error(` - ${issue.taskId}: ${issue.message}`);
}
process.exit(1);
}
const { tasks: allTasks, cleanup } = await resolveFixtures(values.fixtures);
if (values.list) {
console.log("Available Tasks:\n");
for (const task of allTasks) {
console.log(` ${task.id}`);
console.log(` Name: ${task.name}`);
console.log(` Files: ${task.files.join(", ")}`);
console.log("");
}
process.exit(0);
}
let thinkingLevel: ResolvedThinkingLevel = Effort.Low;
if (values.thinking) {
const level = parseThinkingLevel(values.thinking);
if (!level) {
console.error(`Invalid thinking level: ${values.thinking}`);
console.error(`Valid levels: ${[ThinkingLevel.Off, ...THINKING_EFFORTS].join(", ")}`);
process.exit(1);
}
thinkingLevel = level;
}
const runsPerTask = parseInt(values.runs!, 10);
if (Number.isNaN(runsPerTask) || runsPerTask < 1) {
console.error(`Invalid runs value: ${values.runs}`);
process.exit(1);
}
const timeout = parseInt(values.timeout!, 10);
if (Number.isNaN(timeout) || timeout < 1000) {
console.error(`Invalid timeout value: ${values.timeout}`);
process.exit(1);
}
const maxTurns = parseInt(values["max-turns"]!, 10);
if (Number.isNaN(maxTurns) || maxTurns < 1) {
console.error(`Invalid max-turns value: ${values["max-turns"]}. Must be >= 1.`);
process.exit(1);
}
const taskConcurrency = parseInt(values["task-concurrency"]!, 10);
if (Number.isNaN(taskConcurrency) || taskConcurrency < 1) {
console.error(`Invalid task concurrency value: ${values["task-concurrency"]}`);
process.exit(1);
}
const maxAttempts = parseInt(values["max-attempts"] ?? "2", 10);
if (Number.isNaN(maxAttempts) || maxAttempts < 1 || maxAttempts > 5) {
console.error(`Invalid max-attempts value: ${values["max-attempts"]}. Must be 1-5.`);
process.exit(1);
}
const noOpRetryLimit = parseInt(values["no-op-retry-limit"] ?? "2", 10);
const maxTimeoutRetries = parseInt(values["max-timeout-retries"] ?? "3", 10);
const maxProviderRetries = parseInt(values["max-provider-retries"] ?? "3", 10);
const mutationScopeWindow = parseInt(values["mutation-scope-window"] ?? "20", 10);
const connectionTimeout = parseInt(values["connection-timeout"] ?? "30000", 10);
let tasksToRun = allTasks;
if (values.tasks) {
const taskIds = values.tasks.split(",").map(s => s.trim());
tasksToRun = [];
for (const id of taskIds) {
const task = allTasks.find(t => t.id === id);
if (!task) {
console.error(`Unknown task ID: ${id}`);
console.error(`Available tasks: ${allTasks.map(t => t.id).join(", ")}`);
process.exit(1);
}
tasksToRun.push(task);
}
}
// Apply --max-tasks sampling (deterministic by sorting on id)
const maxTasks = parseInt(values["max-tasks"] ?? "80", 10);
if (maxTasks > 0 && tasksToRun.length > maxTasks && !values.tasks) {
// Evenly sample across mutation categories for representative coverage
const sorted = tasksToRun.slice().sort((a, b) => a.id.localeCompare(b.id));
const step = sorted.length / maxTasks;
tasksToRun = Array.from({ length: maxTasks }, (_, i) => sorted[Math.floor(i * step)]!);
}
const rawEditVariant = values["edit-variant"] as string | undefined;
const editVariant = rawEditVariant === "" ? undefined : rawEditVariant;
let editFuzzy: boolean | "auto" | undefined;
if (values["edit-fuzzy"] !== undefined) {
if (values["edit-fuzzy"] === "auto") {
editFuzzy = "auto";
} else if (values["edit-fuzzy"] === "true" || values["edit-fuzzy"] === "1") {
editFuzzy = true;
} else if (values["edit-fuzzy"] === "false" || values["edit-fuzzy"] === "0") {
editFuzzy = false;
} else {
console.error(`Invalid edit-fuzzy: ${values["edit-fuzzy"]}. Must be true, false, 1, 0, or auto.`);
process.exit(1);
}
}
let editFuzzyThreshold: number | "auto" | undefined;
if (values["edit-fuzzy-threshold"] !== undefined) {
if (values["edit-fuzzy-threshold"] === "auto") {
editFuzzyThreshold = "auto";
} else {
const parsed = parseFloat(values["edit-fuzzy-threshold"]);
if (Number.isNaN(parsed) || parsed < 0 || parsed > 1) {
console.error(`Invalid edit-fuzzy-threshold: ${values["edit-fuzzy-threshold"]}. Must be 0-1 or auto.`);
process.exit(1);
}
editFuzzyThreshold = parsed;
}
}
const guided = values["no-guided"] ? false : values.guided;
const formatType = values.format === "json" ? "json" : "markdown";
const config: BenchmarkConfig = {
provider,
model,
thinkingLevel,
runsPerTask,
timeout,
maxTurns,
taskConcurrency,
autoFormat: values["auto-format"],
guided,
maxAttempts,
requireEditToolCall: values["require-edit-tool-call"],
requireReadToolCall: values["require-read-tool-call"],
noEditRequired: values["no-edit-required"],
editVariant,
editFuzzy,
editFuzzyThreshold,
noOpRetryLimit,
maxTimeoutRetries,
maxProviderFailureRetries: maxProviderRetries,
mutationScopeWindow,
connectionTimeout,
inProcess: !values["no-in-process"],
earlyStopOnMatch: !values["no-early-stop-on-match"],
};
const outputPath = values.output ?? generateReportFilename(config, formatType);
config.conversationDumpDir = await resolveConversationDumpDir(outputPath);
console.log("Edit Benchmark");
console.log("==============");
console.log(`Provider: ${config.provider}`);
console.log(`Model: ${config.model}`);
if (config.thinkingLevel) {
console.log(`Thinking: ${config.thinkingLevel}`);
}
console.log(`Runs per task: ${config.runsPerTask}`);
console.log(`Timeout: ${config.timeout}ms`);
console.log(`Task concurrency: ${config.taskConcurrency}`);
if (config.autoFormat) {
console.log("Auto-format: enabled");
}
console.log(`Guided mode: ${config.guided ? "enabled" : "disabled"}`);
console.log(`Max attempts: ${config.maxAttempts}`);
if (config.maxTurns !== undefined) {
console.log(`Max turns per attempt: ${config.maxTurns}`);
}
if (config.requireEditToolCall) {
console.log("Require edit tool call: yes");
}
if (config.requireReadToolCall) {
console.log("Require read tool call: yes");
}
if (config.noEditRequired) {
console.log("No-edit-required baseline: yes");
}
if (config.editVariant) {
console.log(`Edit variant: ${config.editVariant}`);
}
if (config.editFuzzy !== undefined) {
console.log(`Edit fuzzy: ${config.editFuzzy}`);
}
if (config.editFuzzyThreshold !== undefined) {
console.log(`Edit fuzzy threshold: ${config.editFuzzyThreshold}`);
}
console.log(`Tasks: ${tasksToRun.length}`);
console.log(`Conversation dumps: ${config.conversationDumpDir}`);
console.log("");
const progress = new LiveProgress(tasksToRun.length * config.runsPerTask, config.runsPerTask);
let latestResult = buildBenchmarkResult({
tasks: tasksToRun,
config,
resultsByTask: new Map(),
startTime: new Date().toISOString(),
});
let progressFinished = false;
let reportWritePromise: Promise<void> | undefined;
const finishProgress = () => {
if (progressFinished) return;
progress.finish();
progressFinished = true;
};
const writeReport = async (result: BenchmarkResult, interrupted: boolean) => {
if (reportWritePromise) return reportWritePromise;
reportWritePromise = (async () => {
if (interrupted) {
console.log("");
console.log("Benchmark interrupted; writing partial report...");
}
const report = formatType === "json" ? generateJsonReport(result) : generateReport(result);
await Bun.write(outputPath, report);
console.log(`Report written to: ${outputPath}`);
if (config.conversationDumpDir) {
console.log(await conversationDumpStatus(config.conversationDumpDir));
}
})();
return reportWritePromise;
};
const unregisterReportCleanup = postmortem.register("typescript-edit-benchmark-report", async reason => {
if (reason === postmortem.Reason.EXIT) return;
finishProgress();
await writeReport(latestResult, true);
if (cleanup) {
await cleanup();
}
});
const result = await runBenchmark(
tasksToRun,
config,
event => {
progress.handleEvent(event);
},
snapshot => {
latestResult = snapshot;
},
);
latestResult = result;
finishProgress();
console.log("");
console.log("Benchmark complete!");
console.log(
` Task success rate (best of ${config.runsPerTask}): ${(result.summary.taskSuccessRate * 100).toFixed(1)}% (${result.summary.successfulTasks}/${result.summary.totalTasks})`,
);
console.log(
` Total tokens (best, overall): ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`,
);
console.log(
` Tokens/task (best, overall): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total} reasoning=${result.summary.avgTokensPerTask.reasoning}`,
);
console.log(
` Total tokens (one-shot successes): ${result.summary.totalOneShotSuccessTokens.input} in / ${result.summary.totalOneShotSuccessTokens.output} out`,
);
console.log(
` Tokens/task (one-shot successes): mean=${result.summary.avgOneShotSuccessTokensPerTask.total} median=${result.summary.medianOneShotSuccessTokensPerTask.total} p1=${result.summary.p1OneShotSuccessTokensPerTask.total} p99=${result.summary.p99OneShotSuccessTokensPerTask.total} reasoning=${result.summary.avgOneShotSuccessTokensPerTask.reasoning}`,
);
if (result.summary.ghostRuns > 0) {
console.log(` Ghost runs (0/0/0): ${result.summary.ghostRuns}`);
}
if (result.summary.timeoutRuns > 0) {
console.log(` Timeout runs: ${result.summary.timeoutRuns}`);
}
console.log("");
await writeReport(result, false);
unregisterReportCleanup();
if (cleanup) {
await cleanup();
}
// In-process benchmark runs can leave provider keep-alive sockets and
// background AgentSession timers alive after the report is written. Treat the
// final report as the CLI boundary so the command returns to the shell.
await postmortem.quit(0);
}
class LiveProgress {
readonly #totalRuns: number;
readonly #runsPerTask: number;
readonly #isTty: boolean;
#started = 0;
#completed = 0;
#success = 0;
#totalInput = 0;
#totalOutput = 0;
#totalDuration = 0;
#totalReads = 0;
#totalEdits = 0;
#totalWrites = 0;
#totalEditSuccesses = 0;
#totalToolInputChars = 0;
#indentScores: number[] = [];
#inputTokens: number[] = [];
#outputTokens: number[] = [];
#totalTokens: number[] = [];
#oneShotSuccessTokens: number[] = [];
#lastLineLength = 0;
constructor(totalRuns: number, runsPerTask: number) {
this.#totalRuns = totalRuns;
this.#runsPerTask = runsPerTask;
this.#isTty = Boolean(process.stdout.isTTY);
}
handleEvent(event: ProgressEvent): void {
if (event.status === "started") {
this.#started += 1;
if (!this.#isTty) {
console.log(` [${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} started...`);
}
this.#renderLine();
return;
}
this.#completed += 1;
if (event.result) {
if (event.result.success) {
this.#success += 1;
}
if (event.result.success && event.runIndex === 0) {
this.#oneShotSuccessTokens.push(event.result.tokens.total);
}
this.#totalInput += event.result.tokens.input;
this.#totalOutput += event.result.tokens.output;
this.#inputTokens.push(event.result.tokens.input);
this.#outputTokens.push(event.result.tokens.output);
this.#totalTokens.push(event.result.tokens.total);
this.#totalDuration += event.result.duration;
this.#totalReads += event.result.toolCalls.read;
this.#totalEdits += event.result.toolCalls.edit;
this.#totalWrites += event.result.toolCalls.write;
this.#totalEditSuccesses += event.result.toolCalls.editSuccesses;
this.#totalToolInputChars += event.result.toolCalls.totalInputChars;
if (typeof event.result.indentScore === "number") {
this.#indentScores.push(event.result.indentScore);
}
}
const result = event.result;
if (result && !result.success && result.error) {
this.#flushLine();
const header = paint(ANSI.red, `[${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} failed:`);
console.log(` ${header} ${result.error}`);
if (result.diff) {
const changeLines = result.diff
.split("\n")
.filter(line => /^[-+@]/.test(line) && !/^(---|\+\+\+)/.test(line));
const maxLines = 40;
const shown = changeLines.slice(0, maxLines);
for (const line of shown) {
let color: string | undefined;
if (line.startsWith("@@")) color = ANSI.cyan;
else if (line.startsWith("-")) color = ANSI.red;
else if (line.startsWith("+")) color = ANSI.green;
console.log(` ${color ? paint(color, line) : line}`);
}
if (changeLines.length > maxLines) {
console.log(paint(ANSI.dim, ` ... (${changeLines.length - maxLines} more change lines)`));
}
}
}
if (result?.editFailures && result.editFailures.length > 0) {
this.#flushLine();
for (const [i, failure] of result.editFailures.entries()) {
const args = (failure.args ?? {}) as Record<string, unknown>;
const target =
typeof args.path === "string" ? args.path : typeof args.file === "string" ? args.file : undefined;
const op = typeof args.operation === "string" ? args.operation : undefined;
const oneLine = failure.error.replace(/\s+/g, " ").trim();
const clipped = oneLine.length > 240 ? `${oneLine.slice(0, 237)}...` : oneLine;
const tag = paint(ANSI.yellow, `[${event.taskId}] schema #${i + 1}`);
const metaParts = [op, target].filter((v): v is string => Boolean(v));
const meta = metaParts.length > 0 ? paint(ANSI.dim, metaParts.join(" ")) : "";
console.log(` ${tag}${meta ? ` ${meta}` : ""} ${clipped}`);
if (failure.rawBlock) {
const rawLine = failure.rawBlock.replace(/\s+/g, " ").trim();
const clippedRaw = rawLine.length > 240 ? `${rawLine.slice(0, 237)}...` : rawLine;
console.log(` ${paint(ANSI.dim, "raw")} ${clippedRaw}`);
}
}
}
if (!this.#isTty) {
const status = event.result?.success ? "completed" : "failed";
console.log(` [${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} ${status}`);
}
this.#renderLine();
}
finish(): void {
this.#flushLine();
this.#printSummary();
}
#printSummary(): void {
const n = this.#completed;
const denom = n || 1;
const successRate = (this.#success / denom) * 100;
const editSuccessRate = this.#totalEdits > 0 ? (this.#totalEditSuccesses / this.#totalEdits) * 100 : 100;
const avgIndent =
this.#indentScores.length > 0 ? this.#indentScores.reduce((a, b) => a + b, 0) / this.#indentScores.length : 0;
console.log("");
console.log(paint(ANSI.bold, "Runtime Stats:"));
console.log(
` Task success: ${paint(rateColor(successRate), `${successRate.toFixed(1)}% (${this.#success}/${n})`)}`,
);
console.log(
` Edit success: ${paint(rateColor(editSuccessRate), `${editSuccessRate.toFixed(1)}% (${this.#totalEditSuccesses}/${this.#totalEdits})`)}`,
);
console.log(` Avg indent score: ${avgIndent.toFixed(2)}`);
console.log(` Tool calls: read=${this.#totalReads} edit=${this.#totalEdits} write=${this.#totalWrites}`);
console.log(` Tool input chars: ${this.#totalToolInputChars.toLocaleString()}`);
const fmtTokens = (samples: number[]): string => {
if (samples.length === 0) return "mean=0 median=0 p1=0 p99=0";
const sorted = [...samples].sort((a, b) => a - b);
const mean = Math.round(sorted.reduce((a, b) => a + b, 0) / sorted.length);
return `mean=${mean} median=${Math.round(percentile(sorted, 50))} p1=${Math.round(percentile(sorted, 1))} p99=${Math.round(percentile(sorted, 99))}`;
};
console.log(` Tokens/task in: ${fmtTokens(this.#inputTokens)}`);
console.log(` Tokens/task out: ${fmtTokens(this.#outputTokens)}`);
console.log(` Tokens/task tot: ${fmtTokens(this.#totalTokens)}`);
console.log(` Tokens/task (one-shot successes): ${fmtTokens(this.#oneShotSuccessTokens)}`);
console.log(` Avg time/task: ${Math.round(this.#totalDuration / denom)}ms`);
}
#renderLine(): void {
if (!this.#isTty) {
return;
}
const successRate = this.#completed > 0 ? (this.#success / this.#completed) * 100 : 0;
const editRate = this.#totalEdits > 0 ? (this.#totalEditSuccesses / this.#totalEdits) * 100 : 100;
const avgInput = this.#completed > 0 ? Math.round(this.#totalInput / this.#completed) : 0;
const avgOutput = this.#completed > 0 ? Math.round(this.#totalOutput / this.#completed) : 0;
const avgDuration = this.#completed > 0 ? Math.round(this.#totalDuration / this.#completed) : 0;
const inFlight = this.#started - this.#completed;
const bar = this.#renderBar(this.#completed, this.#totalRuns, 20);
const progress = paint(ANSI.bold, `${this.#completed}/${this.#totalRuns}`);
const taskCol = `task=${paint(rateColor(successRate), `${successRate.toFixed(0)}%`)}`;
const editCol = `edit=${paint(rateColor(editRate), `${editRate.toFixed(0)}%`)}`;
const tokCol = paint(ANSI.dim, `tok=${avgInput}/${avgOutput}`);
const durCol = paint(ANSI.dim, `${avgDuration}ms`);
const rewCol = paint(ANSI.dim, `r/e/w=${this.#totalReads}/${this.#totalEdits}/${this.#totalWrites}`);
const flyCol = `fly=${paint(ANSI.cyan, String(inFlight))}`;
const line = ` ${bar} ${progress} ${taskCol} ${editCol} ${tokCol} ${durCol} ${rewCol} ${flyCol}`;
this.#writeLine(line);
}
#renderBar(done: number, total: number, width: number): string {
const ratio = total === 0 ? 0 : done / total;
const filled = Math.round(ratio * width);
const empty = Math.max(0, width - filled);
const filledPart = paint(ANSI.green, "#".repeat(filled));
const emptyPart = paint(ANSI.dim, "-".repeat(empty));
return `[${filledPart}${emptyPart}]`;
}
#writeLine(line: string): void {
const lineWidth = visibleWidth(line);
const pad = this.#lastLineLength > lineWidth ? padding(this.#lastLineLength - lineWidth) : "";
process.stdout.write(`\r${line}${pad}`);
this.#lastLineLength = lineWidth;
}
#flushLine(): void {
if (!this.#isTty) {
return;
}
if (this.#lastLineLength > 0) {
process.stdout.write(`\r${padding(this.#lastLineLength)}\r`);
this.#lastLineLength = 0;
}
}
}
main().catch(async err => {
console.error("Benchmark failed:", err);
await postmortem.quit(1);
});