From 0856055dfed52d7aa95b076152525b01a992b839 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 13 Jul 2026 03:39:05 +0200 Subject: [PATCH] feat(harbor-manager): unified benchmark normalization and reporting - Implemented a unified benchmark normalization layer to handle metrics and artifacts across harbor, edit, and snapcompact benchmarks. - Integrated the TypeScript edit benchmark directly into the manager, migrating logic and deprecating the standalone package. - Updated the API and database schema to support standardized benchmark configurations, metrics, and trace-based reporting. - Enhanced the UI to visualize comparative performance metrics, including pass rate, cost, and latency deltas for benchmark runs. --- bun.lock | 12 +- package.json | 1 - packages/harbor-manager/README.md | 48 +- .../adapters/edit/bun-imports.d.ts | 4 + packages/harbor-manager/adapters/edit/cli.ts | 97 +++ .../adapters/edit}/prompts/benchmark-retry.md | 0 .../edit}/prompts/benchmark-system.md | 0 .../adapters/edit}/prompts/benchmark-task.md | 0 .../adapters/edit}/report.ts | 0 .../adapters/edit}/runner.test.ts | 8 +- .../adapters/edit}/runner.ts | 17 +- .../adapters/edit/tsconfig.json | 4 + packages/harbor-manager/package.json | 15 +- .../src/adapters/snapcompact.py} | 39 +- .../harbor-manager/src/benchmarks.test.ts | 85 ++ packages/harbor-manager/src/benchmarks.ts | 273 +++++++ .../harbor-manager/src/experiments.test.ts | 17 +- packages/harbor-manager/src/experiments.ts | 10 +- packages/harbor-manager/src/manager.test.ts | 97 ++- packages/harbor-manager/src/runner.ts | 8 +- packages/harbor-manager/src/server.ts | 205 +++-- packages/harbor-manager/src/store.ts | 139 ++-- packages/harbor-manager/src/web/app.tsx | 154 +++- .../typescript-edit-benchmark/package.json | 11 +- .../typescript-edit-benchmark/src/index.ts | 758 ------------------ 25 files changed, 1010 insertions(+), 992 deletions(-) create mode 100644 packages/harbor-manager/adapters/edit/bun-imports.d.ts create mode 100644 packages/harbor-manager/adapters/edit/cli.ts rename packages/{typescript-edit-benchmark/src => harbor-manager/adapters/edit}/prompts/benchmark-retry.md (100%) rename packages/{typescript-edit-benchmark/src => harbor-manager/adapters/edit}/prompts/benchmark-system.md (100%) rename packages/{typescript-edit-benchmark/src => harbor-manager/adapters/edit}/prompts/benchmark-task.md (100%) rename packages/{typescript-edit-benchmark/src => harbor-manager/adapters/edit}/report.ts (100%) rename packages/{typescript-edit-benchmark/test => harbor-manager/adapters/edit}/runner.test.ts (98%) rename packages/{typescript-edit-benchmark/src => harbor-manager/adapters/edit}/runner.ts (99%) create mode 100644 packages/harbor-manager/adapters/edit/tsconfig.json rename packages/{snapcompact/research/run.py => harbor-manager/src/adapters/snapcompact.py} (93%) create mode 100644 packages/harbor-manager/src/benchmarks.test.ts create mode 100644 packages/harbor-manager/src/benchmarks.ts delete mode 100755 packages/typescript-edit-benchmark/src/index.ts diff --git a/bun.lock b/bun.lock index 80270a3e9..b9ec794d5 100644 --- a/bun.lock +++ b/bun.lock @@ -140,12 +140,19 @@ "name": "@oh-my-pi/harbor-manager", "version": "0.0.1", "bin": { - "harbor-manager": "src/runner.ts", + "harbor-manager": "src/server.ts", }, "dependencies": { + "@oh-my-pi/hashline": "catalog:", + "@oh-my-pi/pi-agent-core": "catalog:", + "@oh-my-pi/pi-ai": "catalog:", + "@oh-my-pi/pi-coding-agent": "catalog:", + "@oh-my-pi/pi-utils": "catalog:", + "@oh-my-pi/typescript-edit-benchmark": "workspace:*", "clsx": "^2.1.1", "d3-scale": "^4.0.2", "d3-shape": "^3.2.0", + "diff": "catalog:", "motion": "^12.15.0", "react": "^19.1.0", "react-dom": "^19.1.0", @@ -276,9 +283,6 @@ "packages/typescript-edit-benchmark": { "name": "@oh-my-pi/typescript-edit-benchmark", "version": "0.0.1", - "bin": { - "typescript-edit-benchmark": "src/index.ts", - }, "dependencies": { "@babel/generator": "catalog:", "@babel/parser": "catalog:", diff --git a/package.json b/package.json index 5865a13bc..91b40ea2a 100644 --- a/package.json +++ b/package.json @@ -149,7 +149,6 @@ "ci:release:publish": "bun scripts/ci-release-publish.ts", "ci:release:publish-native-leaf": "bun scripts/ci-release-publish.ts --native-leaf", "bench:gen-fixtures": "bun --cwd=packages/typescript-edit-benchmark run src/generate.ts --typescript-dir /tmp/typescript-source --count-per-type 8", - "bench:edit": "bun --cwd=packages/typescript-edit-benchmark run start", "stats:sync": "python3 scripts/session-stats/sync.py", "stats:tools": "python3 scripts/session-stats/analyze.py tools", "stats:edits": "python3 scripts/session-stats/analyze.py edits", diff --git a/packages/harbor-manager/README.md b/packages/harbor-manager/README.md index 107a06389..f289964ea 100644 --- a/packages/harbor-manager/README.md +++ b/packages/harbor-manager/README.md @@ -1,22 +1,16 @@ # @oh-my-pi/harbor-manager -Manage [Harbor](https://github.com/laude-institute/harbor) benchmark runs against -the **local `omp` build**: a CLI runner with a live terminal dashboard, a -SQLite-backed run store, a REST/SSE API, and a web dashboard with live -run → trial → transcript drill-down. - -Works with any Harbor dataset (`terminal-bench@2.0` by default, -`swe-bench/swe-bench-verified`, …). +One manager for repository benchmarks. Harbor, TypeScript edit, and SnapCompact +runs use the same experiment → run → trace model, SQLite store, REST/SSE API, +and dashboard. Benchmark-native artifacts remain on disk; adapters normalize +their live progress, scores, token usage, costs, and traces. ```bash -# CLI run (owns the terminal, writes a markdown report) -bun src/runner.ts --model anthropic/claude-sonnet-4-6 --tasks 20 --concurrency 4 - -# Web manager: dashboard + API on :4700 -bun run serve # bun src/server.ts [--port 4700] [--jobs-dir ] +# Dashboard + API on :4700; launch every benchmark from the same “new run” form +bun run serve --port 4700 ``` -## How runs execute +## How Harbor runs execute 1. **Local omp, not npm.** By default the runner bind-mounts the repo read-only into each task container (`--install source`) and runs omp @@ -35,34 +29,36 @@ bun run serve # bun src/server.ts [--port 4700] [--jobs-dir ] ## Server -- `GET /` — dashboard: run list (SSE-live), trial grid, transcript viewer - that tails live sessions. -- `GET /api/runs` — run rows (rollups: pass/fail/error/spend/tokens). -- `POST /api/runs` — launch. Body: +- `GET /` — experiments, runs, normalized traces, and a launch form for every benchmark. +- `GET /api/experiments` — experiment summaries across all benchmark types. +- `GET /api/runs` — uniform run rows with benchmark, score, progress, spend, and tokens. +- `POST /api/runs` — launch through a benchmark adapter. Body: ```json { + "benchmark": "edit", "model": "anthropic/claude-opus-4-8", - "dataset": "swe-bench/swe-bench-verified", "tasks": 20, "concurrency": 4, - "timeoutMultiplier": 2, - "include": ["swe-bench/astropy__astropy-14995"], - "slide": { "model": "google/gemini-3.5-flash", "onAction": true, "plan": true } + "attempts": 2, + "jobName": "edit-baseline", + "role": "baseline", + "goal": "compare edit strategies" } ``` - `slide.turns` and `slide.onAction` are mutually exclusive triggers. -- `GET /api/runs/:name` — `{ run, trials }` (syncs from disk on read). + `benchmark` is `harbor`, `edit`, or `snapcompact`. Harbor uses `dataset`, + `include`, `timeoutMultiplier`, and `slide`; edit uses `include` as task IDs; + SnapCompact uses `conditions` and treats `tasks` as the passage limit. +- `GET /api/runs/:name` — `{ run, traces }` (syncs native artifacts on read). - `DELETE /api/runs/:name` — cancel a manager-launched run. -- `GET /api/runs/:name/trials/:trial/transcript?tail=N[&raw=1]` — compact - (or raw JSONL) view of the trial's live session log. +- `GET /api/runs/:name/traces/:trace[?raw=1]` — normalized or native trace. - `GET /api/events` — SSE stream of run-list snapshots (sent on change). State lives in `/_manager/harbor-manager.sqlite`; the filesystem stays the source of truth and historical CLI runs are auto-discovered. -## Runner options (excerpt) +## Harbor runner options (excerpt) | Option | Default | Notes | |---|---|---| diff --git a/packages/harbor-manager/adapters/edit/bun-imports.d.ts b/packages/harbor-manager/adapters/edit/bun-imports.d.ts new file mode 100644 index 000000000..7c90c3d17 --- /dev/null +++ b/packages/harbor-manager/adapters/edit/bun-imports.d.ts @@ -0,0 +1,4 @@ +declare module "*.tar.gz" { + const content: string; + export default content; +} diff --git a/packages/harbor-manager/adapters/edit/cli.ts b/packages/harbor-manager/adapters/edit/cli.ts new file mode 100644 index 000000000..dbbe8ce04 --- /dev/null +++ b/packages/harbor-manager/adapters/edit/cli.ts @@ -0,0 +1,97 @@ +#!/usr/bin/env bun +/** Manager-owned executable adapter for the TypeScript edit benchmark. */ +import * as fs from "node:fs/promises"; +import * as path from "node:path"; +import { parseArgs } from "node:util"; +import { TempDir } from "@oh-my-pi/pi-utils"; +import { loadTasksFromDir } from "@oh-my-pi/typescript-edit-benchmark/tasks"; +import { generateJsonReport } from "./report"; +import { type BenchmarkConfig, runBenchmark } from "./runner"; + +const EDIT_PACKAGE = path.resolve(import.meta.dir, "..", "..", "..", "typescript-edit-benchmark"); + +async function extractFixtures(): Promise<{ dir: string; temp: TempDir }> { + const temp = await TempDir.create("@harbor-edit-fixtures-"); + const archive = new Bun.Archive(await Bun.file(path.join(EDIT_PACKAGE, "fixtures.tar.gz")).arrayBuffer()); + for (const [filePath, file] of await archive.files()) { + await Bun.write(path.join(temp.path(), filePath), file); + } + const entries = await fs.readdir(temp.path(), { withFileTypes: true }); + const directories = entries.filter(entry => entry.isDirectory()); + const files = entries.filter(entry => entry.isFile()); + return { dir: directories.length === 1 && files.length === 0 ? path.join(temp.path(), directories[0]!.name) : temp.path(), temp }; +} + +/** Execute an edit benchmark and continuously materialize its normalized source artifact. */ +export async function main(argv = process.argv.slice(2)): Promise { + const { values } = parseArgs({ + args: argv, + options: { + model: { type: "string" }, + output: { type: "string" }, + "max-tasks": { type: "string", default: "80" }, + tasks: { type: "string" }, + "task-concurrency": { type: "string", default: "32" }, + runs: { type: "string", default: "1" }, + list: { type: "boolean", default: false }, + }, + strict: true, + }); + const fixtures = await extractFixtures(); + try { + let tasks = await loadTasksFromDir(fixtures.dir); + if (values.list) { + process.stdout.write(`${JSON.stringify(tasks.map(task => ({ id: task.id, name: task.name })))}\n`); + return; + } + if (!values.model || !values.output) throw new Error("edit adapter requires --model and --output"); + if (values.tasks) { + const selected = new Set(values.tasks.split(",").map(value => value.trim())); + tasks = tasks.filter(task => selected.has(task.id)); + if (tasks.length !== selected.size) throw new Error("one or more edit task ids were not found"); + } else { + const limit = Number(values["max-tasks"]); + if (limit > 0 && tasks.length > limit) { + const sorted = tasks.slice().sort((a, b) => a.id.localeCompare(b.id)); + const step = sorted.length / limit; + tasks = Array.from({ length: limit }, (_, index) => sorted[Math.floor(index * step)]!); + } + } + const slash = values.model.indexOf("/"); + const config: BenchmarkConfig = { + provider: slash === -1 ? "anthropic" : values.model.slice(0, slash), + model: values.model, + runsPerTask: Number(values.runs), + timeout: 120_000, + connectionTimeout: 30_000, + maxTurns: 30, + taskConcurrency: Number(values["task-concurrency"]), + guided: false, + maxAttempts: 1, + noOpRetryLimit: 2, + maxTimeoutRetries: 3, + maxProviderFailureRetries: 3, + mutationScopeWindow: 20, + conversationDumpDir: path.join(path.dirname(values.output), "result.dump"), + inProcess: true, + earlyStopOnMatch: true, + }; + let writes = Promise.resolve(); + const result = await runBenchmark(tasks, config, undefined, snapshot => { + writes = writes.then(async () => { + await Bun.write(values.output!, generateJsonReport(snapshot)); + }); + }); + await writes; + await Bun.write(values.output, generateJsonReport(result)); + } finally { + await fixtures.temp.remove(); + } +} + +if (import.meta.main) { + main().catch(error => { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`); + process.exitCode = 1; + }); +} diff --git a/packages/typescript-edit-benchmark/src/prompts/benchmark-retry.md b/packages/harbor-manager/adapters/edit/prompts/benchmark-retry.md similarity index 100% rename from packages/typescript-edit-benchmark/src/prompts/benchmark-retry.md rename to packages/harbor-manager/adapters/edit/prompts/benchmark-retry.md diff --git a/packages/typescript-edit-benchmark/src/prompts/benchmark-system.md b/packages/harbor-manager/adapters/edit/prompts/benchmark-system.md similarity index 100% rename from packages/typescript-edit-benchmark/src/prompts/benchmark-system.md rename to packages/harbor-manager/adapters/edit/prompts/benchmark-system.md diff --git a/packages/typescript-edit-benchmark/src/prompts/benchmark-task.md b/packages/harbor-manager/adapters/edit/prompts/benchmark-task.md similarity index 100% rename from packages/typescript-edit-benchmark/src/prompts/benchmark-task.md rename to packages/harbor-manager/adapters/edit/prompts/benchmark-task.md diff --git a/packages/typescript-edit-benchmark/src/report.ts b/packages/harbor-manager/adapters/edit/report.ts similarity index 100% rename from packages/typescript-edit-benchmark/src/report.ts rename to packages/harbor-manager/adapters/edit/report.ts diff --git a/packages/typescript-edit-benchmark/test/runner.test.ts b/packages/harbor-manager/adapters/edit/runner.test.ts similarity index 98% rename from packages/typescript-edit-benchmark/test/runner.test.ts rename to packages/harbor-manager/adapters/edit/runner.test.ts index 2ff9434ad..b30cccd84 100644 --- a/packages/typescript-edit-benchmark/test/runner.test.ts +++ b/packages/harbor-manager/adapters/edit/runner.test.ts @@ -4,12 +4,8 @@ import * as path from "node:path"; import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; import { formatSessionDumpText, SessionManager } from "@oh-my-pi/pi-coding-agent"; import { TempDir } from "@oh-my-pi/pi-utils"; -import { generateReport } from "@oh-my-pi/typescript-edit-benchmark/report"; -import { - buildBenchmarkResult, - type TaskRunResult, - writeConversationDump, -} from "@oh-my-pi/typescript-edit-benchmark/runner"; +import { generateReport } from "./report"; +import { buildBenchmarkResult, type TaskRunResult, writeConversationDump } from "./runner"; import type { EditTask } from "@oh-my-pi/typescript-edit-benchmark/tasks"; const tempDirs: TempDir[] = []; diff --git a/packages/typescript-edit-benchmark/src/runner.ts b/packages/harbor-manager/adapters/edit/runner.ts similarity index 99% rename from packages/typescript-edit-benchmark/src/runner.ts rename to packages/harbor-manager/adapters/edit/runner.ts index e401c5c97..47568e6af 100644 --- a/packages/typescript-edit-benchmark/src/runner.ts +++ b/packages/harbor-manager/adapters/edit/runner.ts @@ -13,15 +13,22 @@ import type { Model, ToolExample } from "@oh-my-pi/pi-ai"; import { formatSessionDumpText, RpcClient } from "@oh-my-pi/pi-coding-agent"; import { prompt } from "@oh-my-pi/pi-utils"; import { diffLines } from "diff"; -import { formatDirectory } from "./formatter"; -import { discoverSharedInfra, InProcessClient, type SharedInfra } from "./in-process-client"; +import { formatDirectory } from "@oh-my-pi/typescript-edit-benchmark/formatter"; +import { + discoverSharedInfra, + InProcessClient, + type SharedInfra, +} from "@oh-my-pi/typescript-edit-benchmark/in-process-client"; import benchmarkRetryPrompt from "./prompts/benchmark-retry.md" with { type: "text" }; import benchmarkSystemPrompt from "./prompts/benchmark-system.md" with { type: "text" }; import benchmarkTaskPrompt from "./prompts/benchmark-task.md" with { type: "text" }; -import type { EditTask } from "./tasks"; -import { verifyExpectedFileSubset, verifyExpectedFiles } from "./verify"; +import type { EditTask } from "@oh-my-pi/typescript-edit-benchmark/tasks"; +import { + verifyExpectedFileSubset, + verifyExpectedFiles, +} from "@oh-my-pi/typescript-edit-benchmark/verify"; -const REPO_ROOT = path.resolve(import.meta.dir, "..", "..", ".."); +const REPO_ROOT = path.resolve(import.meta.dir, "..", "..", "..", ".."); const RUNS_DIR = path.join(REPO_ROOT, "runs"); const TMP = path.join(RUNS_DIR, `rb-${Math.random().toString(36).slice(2, 10)}`); const CLI_PATH = Bun.fileURLToPath(import.meta.resolve("@oh-my-pi/pi-coding-agent/cli")); diff --git a/packages/harbor-manager/adapters/edit/tsconfig.json b/packages/harbor-manager/adapters/edit/tsconfig.json new file mode 100644 index 000000000..936fe3f3c --- /dev/null +++ b/packages/harbor-manager/adapters/edit/tsconfig.json @@ -0,0 +1,4 @@ +{ + "extends": "../../../tsconfig.workspace.json", + "include": ["."] +} diff --git a/packages/harbor-manager/package.json b/packages/harbor-manager/package.json index 42146e3f4..c54a1be65 100644 --- a/packages/harbor-manager/package.json +++ b/packages/harbor-manager/package.json @@ -3,7 +3,7 @@ "private": true, "name": "@oh-my-pi/harbor-manager", "version": "0.0.1", - "description": "Manage Harbor benchmark runs against the local omp build: CLI runner, SQLite-backed run store, REST/SSE API, and a live web dashboard", + "description": "Unified benchmark runners plus Harbor run storage, REST/SSE APIs, and a live web dashboard", "homepage": "https://omp.sh", "author": "Can Boluk", "license": "MIT", @@ -13,20 +13,27 @@ "directory": "packages/harbor-manager" }, "bin": { - "harbor-manager": "src/runner.ts" + "harbor-manager": "src/server.ts" }, "scripts": { "check": "biome check . && bun run check:types", - "check:types": "tsgo -p tsconfig.json --noEmit", + "check:types": "tsgo -p tsconfig.json --noEmit && tsgo -p adapters/edit/tsconfig.json --noEmit", "lint": "biome lint .", - "start": "bun run src/runner.ts", + "start": "bun run src/server.ts", "serve": "bun run src/server.ts", "test": "bun test" }, "dependencies": { + "@oh-my-pi/hashline": "catalog:", + "@oh-my-pi/pi-agent-core": "catalog:", + "@oh-my-pi/pi-ai": "catalog:", + "@oh-my-pi/pi-coding-agent": "catalog:", + "@oh-my-pi/pi-utils": "catalog:", + "@oh-my-pi/typescript-edit-benchmark": "workspace:*", "clsx": "^2.1.1", "d3-scale": "^4.0.2", "d3-shape": "^3.2.0", + "diff": "catalog:", "motion": "^12.15.0", "react": "^19.1.0", "react-dom": "^19.1.0", diff --git a/packages/snapcompact/research/run.py b/packages/harbor-manager/src/adapters/snapcompact.py similarity index 93% rename from packages/snapcompact/research/run.py rename to packages/harbor-manager/src/adapters/snapcompact.py index efc570d75..b45daf69b 100644 --- a/packages/snapcompact/research/run.py +++ b/packages/harbor-manager/src/adapters/snapcompact.py @@ -36,6 +36,7 @@ from concurrent.futures import ThreadPoolExecutor from pathlib import Path HERE = Path(__file__).resolve().parent +RESEARCH = HERE.parents[2] / "snapcompact" / "research" def find_agent_prompts() -> Path: for parent in HERE.parents: @@ -48,16 +49,16 @@ def find_agent_prompts() -> Path: raise FileNotFoundError("Could not find agent compaction prompts") -sys.path.insert(0, str(HERE)) +sys.path.insert(0, str(RESEARCH)) import squad # noqa: E402 from anthropic_api import complete, image_block, load_api_key # noqa: E402 from bdf import VARIANTS, FontCfg, capacity, render # noqa: E402 AGENT_PROMPTS = find_agent_prompts() -CACHE = HERE / ".cache" +CACHE = RESEARCH / ".cache" QA_CACHE = CACHE / "qa" -RESULTS = HERE / "results" +RESULTS = RESEARCH / "results" FONTS = { "8x13": FontCfg("8x13", "8x13", 8, 13), @@ -100,7 +101,7 @@ def sha8(*parts: str) -> str: def load_prompt(name: str) -> str: - return (HERE / "prompts" / name).read_text() + return (RESEARCH / "prompts" / name).read_text() def agent_prompt(name: str) -> str: @@ -276,6 +277,7 @@ def main() -> None: ap.add_argument("--price-in", type=float, default=10.0, help="$ per 1M input tokens") ap.add_argument("--price-out", type=float, default=50.0, help="$ per 1M output tokens") ap.add_argument("--env", default="~/.env") + ap.add_argument("--output-dir", help="write records.jsonl and summary.json directly to this directory") args = ap.parse_args() CACHE.mkdir(exist_ok=True) @@ -288,7 +290,11 @@ def main() -> None: f"-e{args.effort}" if args.effort else "", ] ) - run_dir = RESULTS / f"{args.model}-seed{args.seed}-qpc{args.qpc}-{scope}{tag}" + run_dir = ( + Path(args.output_dir).expanduser().resolve() + if args.output_dir + else RESULTS / f"{args.model}-seed{args.seed}-qpc{args.qpc}-{scope}{tag}" + ) run_dir.mkdir(parents=True, exist_ok=True) paras = squad.load_paragraphs(CACHE) @@ -312,17 +318,18 @@ def main() -> None: ctx_args = {"args": args, "flow": flow, "paras": paras, "offsets": offsets, "api_key": api_key} records: list[dict] = [] done = 0 - with ThreadPoolExecutor(args.workers) as pool: - futures = [pool.submit(run_chunk, cond, start, end, ctx_args) for cond, start, end in tasks] - for fut in futures: - records.extend(fut.result()) - done += 1 - if done % 20 == 0: - print(f" {done}/{len(tasks)} chunks", flush=True) - - with (run_dir / "records.jsonl").open("w") as fh: - for r in records: - fh.write(json.dumps(r) + "\n") + with (run_dir / "records.jsonl").open("w") as records_file: + with ThreadPoolExecutor(args.workers) as pool: + futures = [pool.submit(run_chunk, cond, start, end, ctx_args) for cond, start, end in tasks] + for fut in futures: + chunk_records = fut.result() + records.extend(chunk_records) + for record in chunk_records: + records_file.write(json.dumps(record) + "\n") + records_file.flush() + done += 1 + if done % 20 == 0: + print(f" {done}/{len(tasks)} chunks", flush=True) rows = [ aggregate(cond["name"], [r for r in records if r["cond"] == cond["name"]], args.price_in, args.price_out) diff --git a/packages/harbor-manager/src/benchmarks.test.ts b/packages/harbor-manager/src/benchmarks.test.ts new file mode 100644 index 000000000..d7a00018a --- /dev/null +++ b/packages/harbor-manager/src/benchmarks.test.ts @@ -0,0 +1,85 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { BENCHMARK_DEFINITIONS, readBenchmarkSnapshot } from "./benchmarks"; + +const cleanups: string[] = []; + +function jobDir(): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), "harbor-benchmark-")); + cleanups.push(dir); + return dir; +} + +afterEach(() => { + for (const dir of cleanups.splice(0)) fs.rmSync(dir, { recursive: true, force: true }); +}); + +describe("benchmark adapters", () => { + it("normalizes edit attempts, traces, tokens, and declared metrics", () => { + const dir = jobDir(); + fs.writeFileSync( + path.join(dir, "result.json"), + JSON.stringify({ + tasks: [ + { + id: "rename-symbol", + name: "Rename symbol", + runs: [ + { + runIndex: 0, + success: true, + duration: 1200, + tokens: { input: 100, output: 20, reasoning: 5 }, + }, + ], + }, + ], + summary: { + totalRuns: 1, + successfulRuns: 1, + taskSuccessRate: 1, + editSuccessRate: 0.75, + totalTokens: { input: 100, output: 20 }, + }, + }), + ); + + const snapshot = readBenchmarkSnapshot("edit", dir); + expect(snapshot.metrics).toEqual({ task_success_rate: 1, edit_success_rate: 0.75 }); + expect(snapshot.traces[0]).toMatchObject({ + name: "rename-symbol__1", + status: "pass", + tracePath: path.join("result.dump", "rename-symbol", "run-1.md"), + }); + expect([snapshot.tokIn, snapshot.tokOut]).toEqual([100, 20]); + }); + + it("normalizes SnapCompact records and weighted quality metrics", () => { + const dir = jobDir(); + fs.writeFileSync( + path.join(dir, "records.jsonl"), + `${JSON.stringify({ cond: "text", chunk: 0, pos_rel: 0.2, q: "q", answer: "a", golds: ["a"], em: 1, f1: 1 })}\n`, + ); + fs.writeFileSync( + path.join(dir, "summary.json"), + JSON.stringify({ + rows: [ + { n: 2, f1: 0.75, em: 0.5, cost_usd: 1.25, tokens_in: 200, tokens_out: 30, cache_w: 10, cache_r: 20 }, + ], + }), + ); + + const snapshot = readBenchmarkSnapshot("snapcompact", dir); + expect(snapshot.metrics).toEqual({ f1: 0.75, exact_match: 0.5 }); + expect(snapshot.traces[0]).toMatchObject({ status: "pass", reward: 1, tracePath: "record:1" }); + expect(snapshot.costUsd).toBe(1.25); + expect(snapshot.tokCache).toBe(30); + }); + + it("publishes metric definitions for every managed benchmark", () => { + expect(BENCHMARK_DEFINITIONS.map(definition => definition.kind)).toEqual(["harbor", "edit", "snapcompact"]); + expect(BENCHMARK_DEFINITIONS.every(definition => definition.metrics.length > 0)).toBe(true); + }); +}); diff --git a/packages/harbor-manager/src/benchmarks.ts b/packages/harbor-manager/src/benchmarks.ts new file mode 100644 index 000000000..93523ddfa --- /dev/null +++ b/packages/harbor-manager/src/benchmarks.ts @@ -0,0 +1,273 @@ +/** Benchmark adapters normalize native artifacts into manager runs and traces. */ +import * as fs from "node:fs"; +import * as path from "node:path"; +import { aggregate, readJobResult, readTrials } from "./runner"; +import type { BenchmarkKind } from "./store"; + +/** Describes a benchmark metric so storage and UI do not hard-code benchmark semantics. */ +export interface MetricDefinition { + key: string; + label: string; + format: "percent" | "number" | "usd"; + higherIsBetter: boolean; +} + +/** Adapter metadata exposed to launch clients and the dashboard. */ +export interface BenchmarkDefinition { + kind: BenchmarkKind; + label: string; + metrics: MetricDefinition[]; +} + +/** Built-in benchmark adapters and their native score definitions. */ +export const BENCHMARK_DEFINITIONS: BenchmarkDefinition[] = [ + { + kind: "harbor", + label: "Harbor", + metrics: [{ key: "success_rate", label: "Success rate", format: "percent", higherIsBetter: true }], + }, + { + kind: "edit", + label: "TypeScript edit", + metrics: [ + { key: "task_success_rate", label: "Task success", format: "percent", higherIsBetter: true }, + { key: "edit_success_rate", label: "Edit success", format: "percent", higherIsBetter: true }, + ], + }, + { + kind: "snapcompact", + label: "SnapCompact", + metrics: [ + { key: "f1", label: "F1", format: "percent", higherIsBetter: true }, + { key: "exact_match", label: "Exact match", format: "percent", higherIsBetter: true }, + ], + }, +]; + +/** A normalized trace emitted by any benchmark adapter. */ +export interface BenchmarkTrace { + name: string; + task: string; + status: "pass" | "fail" | "error" | "running"; + reward: number | null; + costUsd: number; + durationMs: number; + detail: string; + tracePath: string | null; +} + +/** Uniform aggregate and traces read from benchmark-native artifacts. */ +export interface BenchmarkSnapshot { + traces: BenchmarkTrace[]; + total: number; + done: number; + pass: number; + fail: number; + error: number; + running: number; + costUsd: number; + tokIn: number; + tokOut: number; + tokCache: number; + score: number | null; + metrics: Record; +} + +interface EditRun { + runIndex: number; + success: boolean; + error?: string; + duration: number; + tokens: { input: number; output: number; reasoning: number }; + toolCalls?: { read: number; edit: number; write: number }; +} + +interface EditTask { + id: string; + name: string; + runs: EditRun[]; +} + +interface EditResult { + tasks: EditTask[]; + summary: { + totalRuns: number; + successfulRuns: number; + taskSuccessRate: number; + editSuccessRate: number; + totalTokens: { input: number; output: number }; + }; +} + +interface SnapRecord { + cond: string; + chunk: number; + pos_rel: number; + q: string; + answer: string; + golds: string[]; + em: number; + f1: number; +} + +interface SnapSummaryRow { + n: number; + f1: number; + em: number; + cost_usd: number; + tokens_in: number; + tokens_out: number; + cache_w: number; + cache_r: number; +} + +interface SnapSummary { + rows: SnapSummaryRow[]; +} + +function emptySnapshot(): BenchmarkSnapshot { + return { + traces: [], + total: 0, + done: 0, + pass: 0, + fail: 0, + error: 0, + running: 0, + costUsd: 0, + tokIn: 0, + tokOut: 0, + tokCache: 0, + score: null, + metrics: {}, + }; +} + +function readEditSnapshot(jobDir: string): BenchmarkSnapshot { + const file = path.join(jobDir, "result.json"); + if (!fs.existsSync(file)) return emptySnapshot(); + const result: EditResult = JSON.parse(fs.readFileSync(file, "utf8")); + const traces: BenchmarkTrace[] = []; + let tokIn = 0; + let tokOut = 0; + for (const task of result.tasks) { + for (const run of task.runs) { + tokIn += run.tokens.input; + tokOut += run.tokens.output; + const runNumber = run.runIndex + 1; + traces.push({ + name: `${task.id}__${runNumber}`, + task: task.id, + status: run.success ? "pass" : run.error ? "error" : "fail", + reward: run.success ? 1 : 0, + costUsd: 0, + durationMs: run.duration, + detail: JSON.stringify({ name: task.name, error: run.error ?? null, tools: run.toolCalls ?? null }), + tracePath: path.join("result.dump", task.id.replace(/[^a-zA-Z0-9._-]/g, "_"), `run-${runNumber}.md`), + }); + } + } + const pass = traces.filter(trace => trace.status === "pass").length; + const error = traces.filter(trace => trace.status === "error").length; + return { + traces, + total: result.summary.totalRuns, + done: traces.length, + pass, + fail: traces.length - pass, + error, + running: Math.max(0, result.summary.totalRuns - traces.length), + costUsd: 0, + tokIn, + tokOut, + tokCache: 0, + score: result.summary.taskSuccessRate, + metrics: { + task_success_rate: result.summary.taskSuccessRate, + edit_success_rate: result.summary.editSuccessRate, + }, + }; +} + +function readSnapcompactSnapshot(jobDir: string): BenchmarkSnapshot { + const recordsFile = path.join(jobDir, "records.jsonl"); + const summaryFile = path.join(jobDir, "summary.json"); + if (!fs.existsSync(recordsFile)) return emptySnapshot(); + const records = fs + .readFileSync(recordsFile, "utf8") + .split("\n") + .filter(Boolean) + .map(line => JSON.parse(line) as SnapRecord); + const traces = records.map( + (record, index): BenchmarkTrace => ({ + name: `${record.cond}__${record.chunk}__${index + 1}`, + task: `${record.cond}:${record.chunk}`, + status: record.f1 > 0 ? "pass" : "fail", + reward: record.f1, + costUsd: 0, + durationMs: 0, + detail: JSON.stringify({ question: record.q, answer: record.answer, golds: record.golds, em: record.em }), + tracePath: `record:${index + 1}`, + }), + ); + let rows: SnapSummaryRow[] = []; + if (fs.existsSync(summaryFile)) { + const summary: SnapSummary = JSON.parse(fs.readFileSync(summaryFile, "utf8")); + rows = summary.rows; + } + const samples = rows.reduce((sum, row) => sum + row.n, 0); + const weightedF1 = rows.reduce((sum, row) => sum + row.f1 * row.n, 0); + const weightedEm = rows.reduce((sum, row) => sum + row.em * row.n, 0); + const pass = traces.filter(trace => trace.status === "pass").length; + return { + traces, + total: traces.length, + done: traces.length, + pass, + fail: traces.length - pass, + error: 0, + running: 0, + costUsd: rows.reduce((sum, row) => sum + row.cost_usd, 0), + tokIn: rows.reduce((sum, row) => sum + row.tokens_in, 0), + tokOut: rows.reduce((sum, row) => sum + row.tokens_out, 0), + tokCache: rows.reduce((sum, row) => sum + row.cache_w + row.cache_r, 0), + score: samples > 0 ? weightedF1 / samples : null, + metrics: { + f1: samples > 0 ? weightedF1 / samples : null, + exact_match: samples > 0 ? weightedEm / samples : null, + }, + }; +} + +/** Read and normalize the latest artifacts for a benchmark run. */ +export function readBenchmarkSnapshot(benchmark: BenchmarkKind, jobDir: string): BenchmarkSnapshot { + if (benchmark === "edit") return readEditSnapshot(jobDir); + if (benchmark === "snapcompact") return readSnapcompactSnapshot(jobDir); + const trials = readTrials(jobDir); + const job = readJobResult(jobDir); + const totals = aggregate(trials, job, job?.nTotal ?? trials.length); + return { + traces: trials.map(trial => ({ + name: trial.name, + task: trial.name.replace(/__[^_]+$/, ""), + status: trial.status, + reward: trial.reward, + costUsd: trial.costUsd, + durationMs: trial.durationMs, + detail: trial.detail, + tracePath: path.join(trial.name, "agent", "omp.txt"), + })), + total: totals.total, + done: totals.done, + pass: totals.pass, + fail: totals.fail, + error: totals.error, + running: totals.running, + costUsd: totals.costUsd, + tokIn: totals.tokIn, + tokOut: totals.tokOut, + tokCache: totals.tokCache, + score: totals.done > 0 ? totals.pass / totals.done : null, + metrics: { success_rate: totals.done > 0 ? totals.pass / totals.done : null }, + }; +} diff --git a/packages/harbor-manager/src/experiments.test.ts b/packages/harbor-manager/src/experiments.test.ts index d659168ca..454df656a 100644 --- a/packages/harbor-manager/src/experiments.test.ts +++ b/packages/harbor-manager/src/experiments.test.ts @@ -1,6 +1,6 @@ import { describe, expect, it } from "bun:test"; import { armOf, experimentOf, summarizeArm } from "./experiments"; -import type { RunRow, TrialRow } from "./store"; +import type { RunRow, TraceRow } from "./store"; /** * Contracts under test: @@ -11,11 +11,13 @@ import type { RunRow, TrialRow } from "./store"; function runRow(overrides: Partial): RunRow { return { + benchmark: "harbor", jobName: "exp-arm", dataset: "d", agent: "omp", models: "anthropic/claude-opus-4-8", slide: null, + config: {}, role: "", note: "", status: "running", @@ -33,11 +35,13 @@ function runRow(overrides: Partial): RunRow { tokIn: 0, tokOut: 0, tokCache: 0, + score: null, + metrics: {}, ...overrides, }; } -function trialRow(overrides: Partial): TrialRow { +function traceRow(overrides: Partial): TraceRow { return { jobName: "exp-arm", name: "task__x", @@ -48,6 +52,7 @@ function trialRow(overrides: Partial): TrialRow { durationMs: 60_000, detail: "", updatedAt: Date.now(), + tracePath: null, ...overrides, }; } @@ -75,8 +80,8 @@ describe("summarizeArm", () => { costUsd: 5, }), [ - trialRow({ status: "pass", durationMs: 120_000 }), - trialRow({ name: "b__x", task: "b", status: "fail", reward: 0, durationMs: 240_000 }), + traceRow({ status: "pass", durationMs: 120_000 }), + traceRow({ name: "b__x", task: "b", status: "fail", reward: 0, durationMs: 240_000 }), ], ); expect(running.arm).toBe("n8"); @@ -94,7 +99,7 @@ describe("summarizeArm", () => { const finished = summarizeArm( runRow({ jobName: "sb2-opus", status: "complete", nTotal: 20, done: 20, pass: 15, costUsd: 30 }), - [trialRow({})], + [traceRow({})], ); expect(finished.projected).toBeNull(); expect(finished.costPerTask).toBeCloseTo(1.5, 5); @@ -108,6 +113,6 @@ describe("summarizeArm", () => { }), [], ); - expect(arm.config).toBe("anthropic/claude-opus-4-8 → google/gemini-3.5-flash on first edit/write +plan"); + expect(arm.config).toBe("harbor · anthropic/claude-opus-4-8 → google/gemini-3.5-flash on first edit/write +plan"); }); }); diff --git a/packages/harbor-manager/src/experiments.ts b/packages/harbor-manager/src/experiments.ts index 277128d54..829d0a021 100644 --- a/packages/harbor-manager/src/experiments.ts +++ b/packages/harbor-manager/src/experiments.ts @@ -3,7 +3,7 @@ * `sb2-gemini` → experiment `sb2`) so comparable arms can be charted together, * with linear projections for arms still in flight. */ -import type { RunRow, RunStore, TrialRow } from "./store"; +import type { RunRow, RunStore, TraceRow } from "./store"; /** Linear extrapolation of a running arm to its full task count. */ export interface ArmProjection { @@ -79,8 +79,8 @@ function slideLabel(slideJson: string | null): string { } } -export function summarizeArm(run: RunRow, trials: TrialRow[]): ArmSummary { - const decided = trials.filter(t => t.status === "pass" || t.status === "fail" || t.status === "error"); +export function summarizeArm(run: RunRow, traces: TraceRow[]): ArmSummary { + const decided = traces.filter(t => t.status === "pass" || t.status === "fail" || t.status === "error"); const durations = decided.filter(t => t.durationMs > 0).map(t => t.durationMs); const meanTrialMs = durations.length > 0 ? durations.reduce((a, b) => a + b, 0) / durations.length : null; const passPct = decided.length > 0 ? (100 * run.pass) / decided.length : null; @@ -102,7 +102,7 @@ export function summarizeArm(run: RunRow, trials: TrialRow[]): ArmSummary { return { run, arm: armOf(run.jobName), - config: `${run.models}${slideLabel(run.slide)}`, + config: `${run.benchmark} · ${run.models}${slideLabel(run.slide)}`, passPct, costPerTask, meanTrialMs, @@ -150,7 +150,7 @@ export function experimentDetail(store: RunStore, id: string): ExperimentDetail const matrix: ExperimentDetail["matrix"] = {}; const tasks = new Set(); for (const run of runs) { - const trials = store.listTrials(run.jobName); + const trials = store.listTraces(run.jobName); arms.push(summarizeArm(run, trials)); const cells: Record = {}; for (const t of trials) { diff --git a/packages/harbor-manager/src/manager.test.ts b/packages/harbor-manager/src/manager.test.ts index 26974c862..3d26fb274 100644 --- a/packages/harbor-manager/src/manager.test.ts +++ b/packages/harbor-manager/src/manager.test.ts @@ -104,13 +104,13 @@ describe("RunStore", () => { expect(run?.running).toBe(1); expect(run?.costUsd).toBeCloseTo(0.7, 5); - const trials = store.listTrials("job-a"); - expect(trials.map(t => [t.task, t.status])).toEqual([ + const traces = store.listTraces("job-a"); + expect(traces.map(t => [t.task, t.status])).toEqual([ ["alpha", "pass"], ["beta", "error"], ["gamma", "running"], ]); - expect(trials[1].detail).toBe("AgentTimeoutError"); + expect(traces[1].detail).toBe("AgentTimeoutError"); // re-discover is idempotent expect(store.discover()).toBe(0); @@ -162,6 +162,7 @@ describe("RunStore", () => { const store = new RunStore(jobsDir); cleanups.push(() => store.close()); store.registerLaunch({ + benchmark: "harbor", jobName: "job-b", dataset: "test-dataset@1.0", agent: "omp", @@ -175,7 +176,7 @@ describe("RunStore", () => { }); describe("ManagerServer API", () => { - it("serves runs, trials, transcripts, and validates launches", async () => { + it("serves uniform runs, traces, and rejects invalid launches", async () => { const jobsDir = makeJobsDir(); writeFixtureJob(jobsDir, "job-api"); const manager = new ManagerServer(jobsDir); @@ -190,15 +191,15 @@ describe("ManagerServer API", () => { const detailRes = await fetch(`${base}/api/runs/job-api`); expect(detailRes.status).toBe(200); - const detail = (await detailRes.json()) as { run: { pass: number }; trials: Array<{ status: string }> }; + const detail = (await detailRes.json()) as { run: { pass: number }; traces: Array<{ status: string }> }; expect(detail.run.pass).toBe(1); - expect(detail.trials).toHaveLength(3); + expect(detail.traces).toHaveLength(3); - const tr = await fetch(`${base}/api/runs/job-api/trials/alpha__abc/transcript?tail=10`); + const tr = await fetch(`${base}/api/runs/job-api/traces/alpha__abc?tail=10`); expect(tr.status).toBe(200); - const transcript = (await tr.json()) as { entries: Array<{ kind: string; tools?: string[] }> }; - expect(transcript.entries.map(e => e.kind)).toEqual(["assistant", "toolResult"]); - expect(transcript.entries[0].tools).toEqual(["read"]); + const trace = (await tr.json()) as { entries: Array<{ kind: string; tools?: string[] }> }; + expect(trace.entries.map(e => e.kind)).toEqual(["assistant", "toolResult"]); + expect(trace.entries[0].tools).toEqual(["read"]); const missing = await fetch(`${base}/api/runs/nope`); expect(missing.status).toBe(404); @@ -215,4 +216,80 @@ describe("ManagerServer API", () => { }; expect(cancelUnknown.cancelled).toBe(false); }); + + it("serves edit and SnapCompact metrics and native traces through one API", async () => { + const jobsDir = makeJobsDir(); + const manager = new ManagerServer(jobsDir); + for (const benchmark of ["edit", "snapcompact"] as const) { + const jobName = `${benchmark}-arm`; + manager.store.registerLaunch({ + benchmark, + jobName, + dataset: benchmark === "edit" ? "typescript-edit" : "squad-dev", + agent: benchmark, + models: ["test/model"], + pid: process.pid, + }); + manager.store.markExit(jobName, 0); + } + const editDir = path.join(jobsDir, "edit-arm"); + fs.writeFileSync( + path.join(editDir, "result.json"), + JSON.stringify({ + tasks: [ + { + id: "rename", + name: "Rename", + runs: [{ runIndex: 0, success: true, duration: 10, tokens: { input: 8, output: 2, reasoning: 0 } }], + }, + ], + summary: { + totalRuns: 1, + successfulRuns: 1, + taskSuccessRate: 1, + editSuccessRate: 1, + totalTokens: { input: 8, output: 2 }, + }, + }), + ); + fs.mkdirSync(path.join(editDir, "result.dump", "rename"), { recursive: true }); + fs.writeFileSync(path.join(editDir, "result.dump", "rename", "run-1.md"), "# conversation\n\nassistant answer"); + const snapDir = path.join(jobsDir, "snapcompact-arm"); + fs.writeFileSync( + path.join(snapDir, "records.jsonl"), + `${JSON.stringify({ cond: "text", chunk: 0, pos_rel: 0, q: "question", answer: "answer", golds: ["gold"], em: 0, f1: 0.5 })}\n`, + ); + fs.writeFileSync( + path.join(snapDir, "summary.json"), + JSON.stringify({ + rows: [{ n: 1, f1: 0.5, em: 0, cost_usd: 0.1, tokens_in: 10, tokens_out: 2, cache_w: 0, cache_r: 0 }], + }), + ); + manager.store.syncAll(); + const server = manager.start(0); + cleanups.push(() => { + void manager.stop(); + }); + const base = `http://localhost:${server.port}`; + + const edit = (await (await fetch(`${base}/api/runs/edit-arm`)).json()) as { + run: { benchmark: string; metrics: Record }; + traces: Array<{ name: string }>; + }; + expect(edit.run).toMatchObject({ benchmark: "edit", metrics: { task_success_rate: 1, edit_success_rate: 1 } }); + const editTrace = (await ( + await fetch(`${base}/api/runs/edit-arm/traces/${encodeURIComponent(edit.traces[0].name)}`) + ).json()) as { entries: Array<{ kind: string; text: string }> }; + expect(editTrace.entries).toEqual([{ kind: "conversation", text: "# conversation\n\nassistant answer" }]); + + const snap = (await (await fetch(`${base}/api/runs/snapcompact-arm`)).json()) as { + run: { benchmark: string; metrics: Record }; + traces: Array<{ name: string }>; + }; + expect(snap.run).toMatchObject({ benchmark: "snapcompact", metrics: { f1: 0.5, exact_match: 0 } }); + const snapTrace = (await ( + await fetch(`${base}/api/runs/snapcompact-arm/traces/${encodeURIComponent(snap.traces[0].name)}`) + ).json()) as { entries: Array<{ kind: string }> }; + expect(snapTrace.entries.map(entry => entry.kind)).toEqual(["question", "answer", "reference"]); + }); }); diff --git a/packages/harbor-manager/src/runner.ts b/packages/harbor-manager/src/runner.ts index a716fd39e..b1bd1472c 100755 --- a/packages/harbor-manager/src/runner.ts +++ b/packages/harbor-manager/src/runner.ts @@ -11,9 +11,9 @@ * process renders a live dashboard (progress / success% / spend / tokens / ETA) * by polling each trial's `result.json`. On completion it writes a markdown report. * - * bun src/runner.ts --model anthropic/claude-sonnet-4-6 --tasks 20 --concurrency 4 - * bun src/runner.ts --agent oracle --tasks 2 # cheap pipeline smoke - * bun src/runner.ts --help + * harbor-manager harbor --model anthropic/claude-sonnet-4-6 --tasks 20 --concurrency 4 + * harbor-manager harbor --agent oracle --tasks 2 # cheap pipeline smoke + * harbor-manager harbor --help */ import { spawnSync } from "node:child_process"; import * as fs from "node:fs"; @@ -108,7 +108,7 @@ function defaultConfig(): Config { const HELP = `harbor-manager runner (local omp) -Usage: bun src/runner.ts [options] [-- ] +Usage: harbor-manager harbor [options] [-- ] Commands: cleanup Force-remove ALL leftover Harbor containers + networks, then exit diff --git a/packages/harbor-manager/src/server.ts b/packages/harbor-manager/src/server.ts index 7068d555c..a96353d9a 100755 --- a/packages/harbor-manager/src/server.ts +++ b/packages/harbor-manager/src/server.ts @@ -6,18 +6,20 @@ * bun src/server.ts [--port 4700] [--jobs-dir ] * * API: + * GET /api/experiments → experiment summaries across all benchmarks * GET /api/runs → RunRow[] - * POST /api/runs → launch a run (JSON body, see LaunchRequest) - * GET /api/runs/:name → { run, trials } - * DELETE /api/runs/:name → cancel a manager-launched run - * GET /api/runs/:name/trials/:trial/transcript?tail=N[&raw=1] + * POST /api/runs → launch any benchmark + * GET /api/runs/:name → { run, traces } + * DELETE /api/runs/:name → cancel a managed run + * GET /api/runs/:name/traces/:trace → normalized trace * GET /api/events → SSE: run-list snapshots on change */ import * as fs from "node:fs"; import * as path from "node:path"; import type { Server, Subprocess } from "bun"; +import { BENCHMARK_DEFINITIONS } from "./benchmarks"; import { buildExperiments, experimentDetail, experimentOf } from "./experiments"; -import { type RunRole, RunStore } from "./store"; +import { type BenchmarkKind, type RunRole, RunStore } from "./store"; /** PUT /api/experiments/:id body — goal and per-run role/note metadata. */ export interface ExperimentMetaUpdate { @@ -33,6 +35,8 @@ const DEFAULT_JOBS_DIR = path.join(REPO_ROOT, "runs", "harbor"); /** POST /api/runs body. Mirrors the runner CLI surface we actually use. */ export interface LaunchRequest { + /** Benchmark adapter to execute. */ + benchmark?: BenchmarkKind; model: string; dataset?: string; /** Task count for a dataset sample, or omit when `include` is given. */ @@ -40,12 +44,14 @@ export interface LaunchRequest { /** Explicit task names (passed as repeated --include). */ include?: string[]; concurrency?: number; + /** SnapCompact conditions; ignored by other benchmarks. */ + conditions?: string[]; timeoutMultiplier?: number; attempts?: number; agent?: string; jobName?: string; webSearch?: boolean; - slide?: { model: string; turns?: number; onAction?: boolean; plan?: boolean }; + slide?: { model: string; turns?: number; onAction?: boolean; plan?: boolean; checklist?: boolean }; /** Role of this run inside its experiment (baseline vs treatment). */ role?: RunRole; /** One-line description of what this arm tests. */ @@ -180,6 +186,9 @@ export class ManagerServer { }); } if (p === "/api/events") return this.#sseResponse(); + if (p === "/api/benchmarks" && request.method === "GET") { + return Response.json(BENCHMARK_DEFINITIONS); + } if (p === "/api/experiments" && request.method === "GET") { return Response.json(buildExperiments(this.#store)); } @@ -207,15 +216,15 @@ export class ManagerServer { if (request.method === "DELETE") return Response.json(this.cancel(jobName)); const run = this.#store.syncRun(jobName); if (!run) return Response.json({ error: "run not found" }, { status: 404 }); - return Response.json({ run, trials: this.#store.listTrials(jobName) }); + return Response.json({ run, traces: this.#store.listTraces(jobName) }); } - const trialMatch = p.match(/^\/api\/runs\/([^/]+)\/trials\/([^/]+)\/transcript$/); - if (trialMatch) { - const jobName = decodeURIComponent(trialMatch[1]); - const trial = decodeURIComponent(trialMatch[2]); + const traceMatch = p.match(/^\/api\/runs\/([^/]+)\/traces\/([^/]+)$/); + if (traceMatch) { + const jobName = decodeURIComponent(traceMatch[1]); + const trace = decodeURIComponent(traceMatch[2]); const tail = Number(url.searchParams.get("tail") ?? "120"); const raw = url.searchParams.get("raw") === "1"; - return this.#transcript(jobName, trial, tail, raw); + return this.#trace(jobName, trace, tail, raw); } return Response.json({ error: "not found" }, { status: 404 }); } catch (err) { @@ -248,43 +257,78 @@ export class ManagerServer { }); } - /** Spawn the CLI runner for `request` and register the run. */ + /** Launch any supported benchmark and register it in the uniform run store. */ launch(request: LaunchRequest): { jobName: string; pid: number } { if (!request.model) throw new Error("model is required"); - const dataset = request.dataset ?? "terminal-bench@2.0"; + const benchmark = request.benchmark ?? "harbor"; + if (benchmark !== "harbor" && benchmark !== "edit" && benchmark !== "snapcompact") { + throw new Error(`unsupported benchmark: ${benchmark}`); + } + const dataset = + request.dataset ?? + (benchmark === "harbor" ? "terminal-bench@2.0" : benchmark === "edit" ? "typescript-edit" : "squad-dev"); const stamp = new Date().toISOString().replace(/[:.]/g, "-").slice(0, 19); const modelSlug = request.model.replace(/[^a-zA-Z0-9]+/g, "-"); const jobName = request.jobName ?? `${modelSlug}-${stamp}`; if (this.#children.has(jobName) || this.#store.getRun(jobName)?.status === "running") { throw new Error(`run ${jobName} is already running`); } + const jobDir = path.join(this.jobsDir, jobName); + fs.mkdirSync(jobDir, { recursive: true }); - const argv = ["bun", "src/runner.ts", "--model", request.model, "-d", dataset, "--job-name", jobName]; - if (request.agent) argv.push("--agent", request.agent); - if (request.tasks !== undefined) argv.push("--tasks", String(request.tasks)); - if (request.concurrency !== undefined) argv.push("--concurrency", String(request.concurrency)); - if (request.attempts !== undefined) argv.push("--attempts", String(request.attempts)); - if (request.timeoutMultiplier !== undefined) argv.push("--timeout-multiplier", String(request.timeoutMultiplier)); - if (request.webSearch) argv.push("--web-search"); - for (const task of request.include ?? []) argv.push("--include", task); - if (request.slide) { - argv.push("--agent-arg", "--reasoning-slide-model", "--agent-arg", request.slide.model); - if (request.slide.onAction) argv.push("--agent-arg", "--reasoning-slide-on-action"); - else if (request.slide.turns !== undefined) { - argv.push("--agent-arg", "--reasoning-slide-turns", "--agent-arg", String(request.slide.turns)); - } else throw new Error("slide requires turns or onAction"); - if (request.slide.plan) argv.push("--agent-arg", "--reasoning-slide-plan"); - // The runner only auto-routes gateway auth for the primary model's - // provider; declare the slide model's provider explicitly so its - // requests reach the gateway too. - const slideProvider = request.slide.model.split("/", 1)[0]; - if (slideProvider) argv.push("--providers", slideProvider); - } - // Default to source mode (repo bind-mount, no rebuild); prebuilt binaries only on request. - if (request.prebuiltBinaries) { - for (const name of ["omp-linux-arm64", "omp-linux-x64"]) { - const binary = path.join(REPO_ROOT, "packages", "coding-agent", "dist", name); - if (fs.existsSync(binary)) argv.push("--binary", binary); + let argv: string[]; + let cwd: string; + if (benchmark === "edit") { + cwd = PKG_DIR; + argv = ["bun", "adapters/edit/cli.ts", "--model", request.model, "--output", path.join(jobDir, "result.json")]; + if (request.tasks !== undefined) argv.push("--max-tasks", String(request.tasks)); + if (request.include?.length) argv.push("--tasks", request.include.join(",")); + if (request.concurrency !== undefined) argv.push("--task-concurrency", String(request.concurrency)); + if (request.attempts !== undefined) argv.push("--runs", String(request.attempts)); + } else if (benchmark === "snapcompact") { + cwd = PKG_DIR; + argv = ["uv", "run", "src/adapters/snapcompact.py", "--model", request.model, "--output-dir", jobDir]; + if (request.tasks !== undefined) argv.push("--limit-paras", String(request.tasks)); + if (request.concurrency !== undefined) argv.push("--workers", String(request.concurrency)); + if (request.conditions?.length) argv.push("--conditions", request.conditions.join(",")); + } else { + cwd = PKG_DIR; + argv = [ + "bun", + "src/runner.ts", + "--model", + request.model, + "-d", + dataset, + "--job-name", + jobName, + "--jobs-dir", + this.jobsDir, + ]; + if (request.agent) argv.push("--agent", request.agent); + if (request.tasks !== undefined) argv.push("--tasks", String(request.tasks)); + if (request.concurrency !== undefined) argv.push("--concurrency", String(request.concurrency)); + if (request.attempts !== undefined) argv.push("--attempts", String(request.attempts)); + if (request.timeoutMultiplier !== undefined) + argv.push("--timeout-multiplier", String(request.timeoutMultiplier)); + if (request.webSearch) argv.push("--web-search"); + for (const task of request.include ?? []) argv.push("--include", task); + if (request.slide) { + argv.push("--agent-arg", "--reasoning-slide-model", "--agent-arg", request.slide.model); + if (request.slide.onAction) argv.push("--agent-arg", "--reasoning-slide-on-action"); + else if (request.slide.turns !== undefined) { + argv.push("--agent-arg", "--reasoning-slide-turns", "--agent-arg", String(request.slide.turns)); + } else throw new Error("slide requires turns or onAction"); + if (request.slide.plan) argv.push("--agent-arg", "--reasoning-slide-plan"); + if (request.slide.checklist) argv.push("--agent-arg", "--reasoning-slide-checklist"); + const slideProvider = request.slide.model.split("/", 1)[0]; + if (slideProvider) argv.push("--providers", slideProvider); + } + if (request.prebuiltBinaries) { + for (const name of ["omp-linux-arm64", "omp-linux-x64"]) { + const binary = path.join(REPO_ROOT, "packages", "coding-agent", "dist", name); + if (fs.existsSync(binary)) argv.push("--binary", binary); + } } } argv.push(...(request.extraArgs ?? [])); @@ -293,7 +337,7 @@ export class ManagerServer { fs.mkdirSync(logDir, { recursive: true }); const logFile = fs.openSync(path.join(logDir, `${jobName}.log`), "w"); const proc = Bun.spawn(argv, { - cwd: PKG_DIR, + cwd, stdout: logFile, stderr: logFile, env: { ...process.env }, @@ -309,11 +353,13 @@ export class ManagerServer { this.#tick(); }); this.#store.registerLaunch({ + benchmark, jobName, dataset, agent: request.agent ?? "omp", models: [request.model], slide: request.slide, + config: { ...request }, pid: proc.pid, role: request.role, note: request.note, @@ -354,12 +400,44 @@ export class ManagerServer { return { jobName, cancelled: false }; } - /** Compact transcript view of a trial's omp.txt session JSONL. */ - #transcript(jobName: string, trial: string, tail: number, raw: boolean): Response { - const file = path.join(this.jobsDir, jobName, trial, "agent", "omp.txt"); - if (!fs.existsSync(file)) return Response.json({ error: "transcript not found" }, { status: 404 }); - const lines = fs.readFileSync(file, "utf8").split("\n").filter(Boolean); + /** Return a normalized trace regardless of the benchmark's native artifact format. */ + #trace(jobName: string, traceName: string, tail: number, raw: boolean): Response { + const trace = this.#store.listTraces(jobName).find(item => item.name === traceName); + if (!trace?.tracePath) return Response.json({ error: "trace not found" }, { status: 404 }); + const jobDir = path.join(this.jobsDir, jobName); const n = Number.isSafeInteger(tail) && tail > 0 ? Math.min(tail, 2000) : 120; + if (trace.tracePath.startsWith("record:")) { + const lineNumber = Number(trace.tracePath.slice("record:".length)); + const line = fs.readFileSync(path.join(jobDir, "records.jsonl"), "utf8").split("\n")[lineNumber - 1]; + if (!line) return Response.json({ error: "trace not found" }, { status: 404 }); + if (raw) return new Response(line, { headers: { "content-type": "application/json" } }); + const record = JSON.parse(line) as Record; + return Response.json({ + jobName, + trace: traceName, + entries: [ + { kind: "question", text: String(record.q ?? "") }, + { kind: "answer", model: this.#store.getRun(jobName)?.models ?? "", text: String(record.answer ?? "") }, + { kind: "reference", text: JSON.stringify(record.golds ?? []) }, + ], + totalEvents: 3, + }); + } + const file = path.resolve(jobDir, trace.tracePath); + if (!file.startsWith(`${path.resolve(jobDir)}${path.sep}`) || !fs.existsSync(file)) { + return Response.json({ error: "trace not found" }, { status: 404 }); + } + const text = fs.readFileSync(file, "utf8"); + if (!file.endsWith(".txt")) { + if (raw) return new Response(text, { headers: { "content-type": "text/plain; charset=utf-8" } }); + return Response.json({ + jobName, + trace: traceName, + entries: [{ kind: "conversation", text }], + totalEvents: 1, + }); + } + const lines = text.split("\n").filter(Boolean); if (raw) { return new Response(lines.slice(-n).join("\n"), { headers: { "content-type": "application/x-ndjson" }, @@ -373,41 +451,30 @@ export class ManagerServer { } catch { continue; } - const type = event.type; - if (type === "message_end") { + if (event.type === "message_end") { const message = event.message as Record | undefined; if (!message) continue; - const role = message.role; - if (role === "assistant") { - const content = Array.isArray(message.content) - ? (message.content as Array>) - : []; - const text = content - .filter(block => block.type === "text") - .map(block => String(block.text ?? "")) - .join("\n"); + const content = Array.isArray(message.content) ? (message.content as Array>) : []; + const body = content + .filter(block => block.type === "text") + .map(block => String(block.text ?? "")) + .join("\n"); + if (message.role === "assistant") { const tools = content.filter(block => block.type === "toolCall").map(block => String(block.name ?? "?")); - entries.push({ kind: "assistant", model: message.model ?? "", text, tools }); - } else if (role === "toolResult") { - const content = Array.isArray(message.content) - ? (message.content as Array>) - : []; - const text = content - .filter(block => block.type === "text") - .map(block => String(block.text ?? "")) - .join("\n"); + entries.push({ kind: "assistant", model: message.model ?? "", text: body, tools }); + } else if (message.role === "toolResult") { entries.push({ kind: "toolResult", tool: message.toolName ?? "?", isError: message.isError === true, - text: text.length > 1600 ? `${text.slice(0, 1600)}…` : text, + text: body.length > 1600 ? `${body.slice(0, 1600)}…` : body, }); } - } else if (type === "notice") { + } else if (event.type === "notice") { entries.push({ kind: "notice", text: event.message ?? "" }); } } - return Response.json({ jobName, trial, entries: entries.slice(-n), totalEvents: lines.length }); + return Response.json({ jobName, trace: traceName, entries: entries.slice(-n), totalEvents: lines.length }); } } diff --git a/packages/harbor-manager/src/store.ts b/packages/harbor-manager/src/store.ts index 152a56d63..848cd6415 100644 --- a/packages/harbor-manager/src/store.ts +++ b/packages/harbor-manager/src/store.ts @@ -10,19 +10,26 @@ import { Database } from "bun:sqlite"; import * as fs from "node:fs"; import * as path from "node:path"; -import { aggregate, readJobResult, readTrials } from "./runner"; +import { readBenchmarkSnapshot } from "./benchmarks"; +import { readJobResult } from "./runner"; export type RunStatus = "running" | "complete" | "failed" | "cancelled"; +/** Benchmark implementation that produced a run. */ +export type BenchmarkKind = "harbor" | "edit" | "snapcompact"; + /** How a run relates to its experiment's question. */ export type RunRole = "baseline" | "variant" | ""; export interface RunRow { + benchmark: BenchmarkKind; jobName: string; dataset: string; agent: string; models: string; slide: string | null; + /** Benchmark-specific launch configuration. */ + config: Record; /** Role inside the experiment (baseline vs treatment); "" when unspecified. */ role: RunRole; /** One-line description of what this arm tests (e.g. "slide→flash after 8 turns"). */ @@ -42,9 +49,13 @@ export interface RunRow { tokIn: number; tokOut: number; tokCache: number; + /** Benchmark-native aggregate score, when the benchmark exposes one. */ + score: number | null; + /** Values keyed by the adapter's metric definitions. */ + metrics: Record; } -export interface TrialRow { +export interface TraceRow { jobName: string; name: string; task: string; @@ -54,9 +65,12 @@ export interface TrialRow { durationMs: number; detail: string; updatedAt: number; + /** Adapter-owned locator used by the uniform trace endpoint. */ + tracePath: string | null; } export interface LaunchRecord { + benchmark: BenchmarkKind; jobName: string; dataset: string; agent: string; @@ -65,17 +79,20 @@ export interface LaunchRecord { pid: number; role?: RunRole; note?: string; + config?: Record; } const SCHEMA = ` CREATE TABLE IF NOT EXISTS runs ( job_name TEXT PRIMARY KEY, + benchmark TEXT NOT NULL DEFAULT 'harbor', dataset TEXT NOT NULL DEFAULT '', agent TEXT NOT NULL DEFAULT 'omp', models TEXT NOT NULL DEFAULT '', slide TEXT, role TEXT NOT NULL DEFAULT '', note TEXT NOT NULL DEFAULT '', + config_json TEXT NOT NULL DEFAULT '{}', status TEXT NOT NULL DEFAULT 'running', pid INTEGER, exit_code INTEGER, @@ -90,6 +107,8 @@ CREATE TABLE IF NOT EXISTS runs ( cost_usd REAL NOT NULL DEFAULT 0, tok_in INTEGER NOT NULL DEFAULT 0, tok_out INTEGER NOT NULL DEFAULT 0, + score REAL, + metrics_json TEXT NOT NULL DEFAULT '{}', tok_cache INTEGER NOT NULL DEFAULT 0 ); CREATE TABLE IF NOT EXISTS trials ( @@ -101,6 +120,7 @@ CREATE TABLE IF NOT EXISTS trials ( cost_usd REAL NOT NULL DEFAULT 0, duration_ms INTEGER NOT NULL DEFAULT 0, detail TEXT NOT NULL DEFAULT '', + trace_path TEXT, updated_at INTEGER NOT NULL, PRIMARY KEY (job_name, name) ); @@ -125,12 +145,25 @@ export class RunStore { this.#db = new Database(dbPath ?? path.join(jobsDir, "_manager", "harbor-manager.sqlite")); this.#db.run("PRAGMA journal_mode = WAL"); this.#db.run(SCHEMA); - // Migration for stores created before run roles/notes existed. - const columns = new Set( + const runColumns = new Set( (this.#db.query("PRAGMA table_info(runs)").all() as Array<{ name: string }>).map(c => c.name), ); - if (!columns.has("role")) this.#db.run("ALTER TABLE runs ADD COLUMN role TEXT NOT NULL DEFAULT ''"); - if (!columns.has("note")) this.#db.run("ALTER TABLE runs ADD COLUMN note TEXT NOT NULL DEFAULT ''"); + if (!runColumns.has("role")) this.#db.run("ALTER TABLE runs ADD COLUMN role TEXT NOT NULL DEFAULT ''"); + if (!runColumns.has("note")) this.#db.run("ALTER TABLE runs ADD COLUMN note TEXT NOT NULL DEFAULT ''"); + if (!runColumns.has("benchmark")) { + this.#db.run("ALTER TABLE runs ADD COLUMN benchmark TEXT NOT NULL DEFAULT 'harbor'"); + } + if (!runColumns.has("config_json")) { + this.#db.run("ALTER TABLE runs ADD COLUMN config_json TEXT NOT NULL DEFAULT '{}'"); + } + if (!runColumns.has("score")) this.#db.run("ALTER TABLE runs ADD COLUMN score REAL"); + if (!runColumns.has("metrics_json")) { + this.#db.run("ALTER TABLE runs ADD COLUMN metrics_json TEXT NOT NULL DEFAULT '{}'"); + } + const traceColumns = new Set( + (this.#db.query("PRAGMA table_info(trials)").all() as Array<{ name: string }>).map(c => c.name), + ); + if (!traceColumns.has("trace_path")) this.#db.run("ALTER TABLE trials ADD COLUMN trace_path TEXT"); } close(): void { @@ -139,26 +172,34 @@ export class RunStore { /** Register a run this manager just launched (pid-owning). */ registerLaunch(launch: LaunchRecord): void { + this.#db.query("DELETE FROM trials WHERE job_name = ?").run(launch.jobName); this.#db .query( - `INSERT INTO runs (job_name, dataset, agent, models, slide, role, note, status, pid, created_at) - VALUES (?, ?, ?, ?, ?, ?, ?, 'running', ?, ?) + `INSERT INTO runs + (job_name, benchmark, dataset, agent, models, slide, role, note, config_json, status, pid, created_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'running', ?, ?) ON CONFLICT(job_name) DO UPDATE SET - pid = excluded.pid, status = 'running', + benchmark = excluded.benchmark, pid = excluded.pid, status = 'running', + config_json = excluded.config_json, role = CASE WHEN excluded.role != '' THEN excluded.role ELSE runs.role END, note = CASE WHEN excluded.note != '' THEN excluded.note ELSE runs.note END`, ) .run( launch.jobName, + launch.benchmark, launch.dataset, launch.agent, launch.models.join(","), launch.slide ? JSON.stringify(launch.slide) : null, launch.role ?? "", launch.note ?? "", + JSON.stringify(launch.config ?? {}), launch.pid, Date.now(), ); + const jobDir = path.join(this.jobsDir, launch.jobName); + fs.mkdirSync(jobDir, { recursive: true }); + fs.writeFileSync(path.join(jobDir, "manager.json"), JSON.stringify(launch, null, 2)); } /** Upsert the experiment's stated goal. */ @@ -230,65 +271,68 @@ export class RunStore { syncRun(jobName: string): RunRow | null { const jobDir = path.join(this.jobsDir, jobName); if (!fs.existsSync(jobDir)) return this.getRun(jobName); - const trials = readTrials(jobDir); - const job = readJobResult(jobDir); - const totals = aggregate(trials, job, job?.nTotal ?? trials.length); + const row = this.getRun(jobName); + if (!row) return null; + const snapshot = readBenchmarkSnapshot(row.benchmark, jobDir); const now = Date.now(); const upsert = this.#db.query( - `INSERT INTO trials (job_name, name, task, status, reward, cost_usd, duration_ms, detail, updated_at) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + `INSERT INTO trials + (job_name, name, task, status, reward, cost_usd, duration_ms, detail, trace_path, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?) ON CONFLICT(job_name, name) DO UPDATE SET status = excluded.status, reward = excluded.reward, cost_usd = excluded.cost_usd, - duration_ms = excluded.duration_ms, detail = excluded.detail, updated_at = excluded.updated_at`, + duration_ms = excluded.duration_ms, detail = excluded.detail, + trace_path = excluded.trace_path, updated_at = excluded.updated_at`, ); const tx = this.#db.transaction(() => { - for (const t of trials) { + for (const trace of snapshot.traces) { upsert.run( jobName, - t.name, - t.name.replace(/__[^_]+$/, ""), - t.status, - t.reward, - t.costUsd, - t.durationMs, - t.detail, + trace.name, + trace.task, + trace.status, + trace.reward, + trace.costUsd, + trace.durationMs, + trace.detail, + trace.tracePath, now, ); } this.#db .query( `UPDATE runs SET n_total = ?, done = ?, pass = ?, fail = ?, error = ?, running = ?, - cost_usd = ?, tok_in = ?, tok_out = ?, tok_cache = ? WHERE job_name = ?`, + cost_usd = ?, tok_in = ?, tok_out = ?, tok_cache = ?, score = ?, metrics_json = ? + WHERE job_name = ?`, ) .run( - totals.total, - totals.done, - totals.pass, - totals.fail, - totals.error, - totals.running, - totals.costUsd, - totals.tokIn, - totals.tokOut, - totals.tokCache, + snapshot.total, + snapshot.done, + snapshot.pass, + snapshot.fail, + snapshot.error, + snapshot.running, + snapshot.costUsd, + snapshot.tokIn, + snapshot.tokOut, + snapshot.tokCache, + snapshot.score, + JSON.stringify(snapshot.metrics), jobName, ); - // Foreign runs (no owning pid and never finalized by markExit) infer - // their lifecycle from Harbor's job-level result: `finished_at` is - // terminal; a missing terminal marker with a fresh job dir means still - // running; stale for >30 min means the harness died mid-run. - const row = this.getRun(jobName); - if (row && row.pid === null && row.finishedAt === null && row.status !== "cancelled") { - const job2 = readJobResult(jobDir); + // Historical Harbor runs have no owning process. Infer their terminal + // state from result metadata or directory freshness. + if (row.benchmark === "harbor" && row.pid === null && row.finishedAt === null && row.status !== "cancelled") { + const result = readJobResult(jobDir); let status: RunStatus; let finishedAt: number | null = null; - if (job2?.finishedAt != null) { + if (result?.finishedAt != null) { status = "complete"; - finishedAt = job2.finishedAt; + finishedAt = result.finishedAt; } else if (jobDirFresh(jobDir)) { status = "running"; } else { - status = totals.done > 0 && totals.done >= totals.total ? "complete" : "failed"; + status = snapshot.done > 0 && snapshot.done >= snapshot.total ? "complete" : "failed"; finishedAt = jobDirMtime(jobDir); } if (status !== row.status) { @@ -343,7 +387,7 @@ export class RunStore { return rows.map(rowToRun); } - listTrials(jobName: string): TrialRow[] { + listTraces(jobName: string): TraceRow[] { const rows = this.#db.query("SELECT * FROM trials WHERE job_name = ? ORDER BY name").all(jobName) as Array< Record >; @@ -357,17 +401,20 @@ export class RunStore { durationMs: Number(r.duration_ms), detail: String(r.detail), updatedAt: Number(r.updated_at), + tracePath: r.trace_path === null ? null : String(r.trace_path), })); } } function rowToRun(r: Record): RunRow { return { + benchmark: String(r.benchmark ?? "harbor") as BenchmarkKind, jobName: String(r.job_name), dataset: String(r.dataset), agent: String(r.agent), models: String(r.models), slide: r.slide === null ? null : String(r.slide), + config: JSON.parse(String(r.config_json ?? "{}")), role: String(r.role ?? "") as RunRole, note: String(r.note ?? ""), status: String(r.status) as RunStatus, @@ -385,6 +432,8 @@ function rowToRun(r: Record): RunRow { tokIn: Number(r.tok_in), tokOut: Number(r.tok_out), tokCache: Number(r.tok_cache), + score: r.score === null ? null : Number(r.score), + metrics: JSON.parse(String(r.metrics_json ?? "{}")), }; } diff --git a/packages/harbor-manager/src/web/app.tsx b/packages/harbor-manager/src/web/app.tsx index f1aed9356..37515c981 100644 --- a/packages/harbor-manager/src/web/app.tsx +++ b/packages/harbor-manager/src/web/app.tsx @@ -6,7 +6,7 @@ * #/exp/ experiment detail — arm table, dithered comparison charts * (projected values for in-flight arms, dimmed), task matrix * #/runs flat run list (legacy view) - * #/runs/ run detail — trial grid + live transcript tail + * #/runs/ run detail — normalized trace grid + live trace viewer */ import { useCallback, useEffect, useRef, useState } from "react"; import { createRoot } from "react-dom/client"; @@ -14,11 +14,13 @@ import { createRoot } from "react-dom/client"; // ── api types (mirrors server modules) ────────────────────────────────────── interface RunRow { + benchmark: "harbor" | "edit" | "snapcompact"; jobName: string; dataset: string; agent: string; models: string; slide: string | null; + config: Record; role: "baseline" | "variant" | ""; note: string; status: "running" | "complete" | "failed" | "cancelled"; @@ -32,9 +34,11 @@ interface RunRow { error: number; running: number; costUsd: number; + score: number | null; + metrics: Record; } -interface TrialRow { +interface TraceRow { name: string; task: string; status: string; @@ -193,7 +197,7 @@ function ExperimentsIndex() {
{exp.id}
@@ -375,6 +379,53 @@ const CELL_CLASS: Record = { running: "bg-sky-500 animate-pulse", }; +/** + * The comparison anchor for an experiment: the completed baseline arm with the + * highest pass rate (the "ceiling" a reasoning slide tries to preserve). Ties + * break toward the cheaper arm. Returns null when no baseline has finished data. + */ +function pickReferenceArm(arms: ArmSummary[]): ArmSummary | null { + let ref: ArmSummary | null = null; + for (const a of arms) { + if (a.run.role !== "baseline" || a.passPct === null) continue; + if ( + ref === null || + a.passPct > (ref.passPct ?? -1) || + (a.passPct === ref.passPct && (a.costPerTask ?? Infinity) < (ref.costPerTask ?? Infinity)) + ) { + ref = a; + } + } + return ref; +} + +/** + * Signed, colour-coded offset of a metric from the reference arm. `points` + * shows absolute percentage-point difference (pass rate); `relative` shows a + * percentage change (cost, time). `higherBetter` decides which direction is green. + */ +function Delta({ + value, + reference, + mode, + higherBetter, +}: { + value: number | null; + reference: number | null; + mode: "points" | "relative"; + higherBetter: boolean; +}) { + if (value === null || reference === null) return null; + const raw = + mode === "points" ? value - reference : reference === 0 ? Number.NaN : ((value - reference) / reference) * 100; + if (!Number.isFinite(raw) || Math.abs(raw) < 0.5) { + return ≈; + } + const good = higherBetter ? raw > 0 : raw < 0; + const body = `${raw > 0 ? "+" : "−"}${Math.abs(raw).toFixed(0)}${mode === "relative" ? "%" : ""}`; + return ({body}); +} + function ExperimentPage({ id }: { id: string }) { const detail = usePolled(`/api/experiments/${encodeURIComponent(id)}`, 3000); if (!detail) return
loading…
; @@ -394,12 +445,19 @@ function ExperimentPage({ id }: { id: string }) { a => (a.meanTrialMs === null ? null : a.meanTrialMs / 60000), p => p.meanTrialMs / 60000, ); + const ref = pickReferenceArm(arms); return (

{id}

{arms.length} arms · {tasks.length} tasks + {ref && ( + <> + {" "} + · Δ vs {ref.arm} + + )}
{goal &&

{goal}

} @@ -430,6 +488,14 @@ function ExperimentPage({ id }: { id: string }) { {arm.run.role} )} + {ref?.arm === arm.arm && ( + + ref + + )} {arm.passPct !== null ? `${arm.passPct.toFixed(0)}%` : "—"} {arm.projected && →{arm.projected.passPct.toFixed(0)}%} + {ref && ref.arm !== arm.arm && ( + + )} {arm.costPerTask !== null ? fmtUsd(arm.costPerTask) : "—"} {arm.projected && Σ{fmtUsd(arm.projected.totalCostUsd)}} + {ref && ref.arm !== arm.arm && ( + + )} + + + {arm.meanTrialMs !== null ? fmtMin(arm.meanTrialMs) : "—"} + {ref && ref.arm !== arm.arm && ( + + )} - {arm.meanTrialMs !== null ? fmtMin(arm.meanTrialMs) : "—"}
( + const detail = usePolled<{ run: RunRow; traces: TraceRow[] }>( selected ? `/api/runs/${encodeURIComponent(selected)}` : null, 2500, ); - const [trial, setTrial] = useState(null); - const transcript = usePolled<{ entries: TranscriptEntry[] }>( - selected && trial - ? `/api/runs/${encodeURIComponent(selected)}/trials/${encodeURIComponent(trial)}/transcript?tail=60` + const [trace, setTrace] = useState(null); + const traceData = usePolled<{ entries: TranscriptEntry[] }>( + selected && trace + ? `/api/runs/${encodeURIComponent(selected)}/traces/${encodeURIComponent(trace)}?tail=60` : null, 2500, ); - const transcriptRef = useRef(null); + const traceRef = useRef(null); useEffect(() => { - if (!transcript) return; - const el = transcriptRef.current; + if (!traceData) return; + const el = traceRef.current; if (el) el.scrollTop = el.scrollHeight; - }, [transcript]); + }, [traceData]); const cancel = useCallback(async (name: string) => { if (confirm(`stop ${name}?`)) await fetch(`/api/runs/${encodeURIComponent(name)}`, { method: "DELETE" }); }, []); @@ -578,6 +665,7 @@ function RunsPage({ selected }: { selected: string | null }) { > {r.jobName} +
{r.benchmark}
{(r.note || r.role) && (
{r.role && ( @@ -622,34 +710,42 @@ function RunsPage({ selected }: { selected: string | null }) {
{detail.run.jobName} {" "} - {detail.run.dataset} · {detail.run.models} + {detail.run.benchmark} · {detail.run.dataset} · {detail.run.models} + {detail.run.score !== null ? ` · score ${(100 * detail.run.score).toFixed(1)}%` : ""} {detail.run.slide ? ` → ${detail.run.slide}` : ""} +
+ {Object.entries(detail.run.metrics).map(([key, value]) => ( + + {key.replaceAll("_", " ")}: {value === null ? "—" : `${(100 * value).toFixed(1)}%`} + + ))} +
- {detail.trials.map(t => ( + {detail.traces.map(t => ( setTrial(t.name)} - className={`cursor-pointer border-t border-zinc-800/60 hover:bg-zinc-900 ${t.name === trial ? "bg-zinc-900" : ""}`} + onClick={() => setTrace(t.name)} + className={`cursor-pointer border-t border-zinc-800/60 hover:bg-zinc-900 ${t.name === trace ? "bg-zinc-900" : ""}`} > + - ))}
{t.task} {t.reward === null ? "—" : t.reward.toFixed(3)} {fmtUsd(t.costUsd)} {t.durationMs ? fmtMin(t.durationMs) : "—"}{t.detail}
- {trial && ( -
- {(transcript?.entries ?? []).map((e, i) => ( + {trace && ( +
+ {(traceData?.entries ?? []).map((e, i) => ( // biome-ignore lint/suspicious/noArrayIndexKey: tail window, entries have no ids
@@ -687,7 +783,7 @@ function LaunchForm({ onDone }: { onDone: () => void }) { async (ev: React.FormEvent) => { ev.preventDefault(); const f = new FormData(ev.currentTarget); - const body: Record = { model: f.get("model") }; + const body: Record = { benchmark: f.get("benchmark"), model: f.get("model") }; if (f.get("jobName")) body.jobName = f.get("jobName"); if (f.get("dataset")) body.dataset = f.get("dataset"); if (f.get("tasks")) body.tasks = Number(f.get("tasks")); @@ -699,6 +795,12 @@ function LaunchForm({ onDone }: { onDone: () => void }) { .map(s => s.trim()) .filter(Boolean); } + if (f.get("conditions")) { + body.conditions = String(f.get("conditions")) + .split(",") + .map(s => s.trim()) + .filter(Boolean); + } if (f.get("goal")) body.goal = f.get("goal"); if (f.get("role")) body.role = f.get("role"); if (f.get("note")) body.note = f.get("note"); @@ -724,10 +826,15 @@ function LaunchForm({ onDone }: { onDone: () => void }) { const input = "rounded border border-zinc-700 bg-zinc-950 px-2 py-1 text-sm"; return (
+ - + @@ -741,6 +848,7 @@ function LaunchForm({ onDone }: { onDone: () => void }) { plan nudge + = 80) return ANSI.green; - if (percent >= 50) return ANSI.yellow; - return ANSI.red; -} - -function parseThinkingLevel(value: string | null | undefined): ResolvedThinkingLevel | undefined { - return value !== undefined && - value !== null && - [ThinkingLevel.Off, ...THINKING_EFFORTS].includes(value as ResolvedThinkingLevel) - ? (value as ResolvedThinkingLevel) - : undefined; -} - -function generateReportFilename(config: BenchmarkConfig, format: "markdown" | "json"): string { - const modelName = config.model - .split("/") - .pop()! - .replace(/[^a-zA-Z0-9-]/g, "_"); - const variant = config.editVariant ?? "replace"; - const timestamp = new Date().toISOString().replace(/:/g, "-").replace(/\..+$/, "").replace(/Z$/, "Z"); - const ext = format === "json" ? "json" : "md"; - return path.join(RUNS_DIR, `${modelName}_${variant}_${timestamp}.${ext}`); -} - -async function resolveConversationDumpDir(outputPath: string): Promise { - const parsed = path.parse(outputPath); - const preferredPath = path.join(parsed.dir, `${parsed.name}.dump`); - try { - await fs.promises.stat(preferredPath); - } catch (error) { - if ((error as NodeJS.ErrnoException).code === "ENOENT") { - return preferredPath; - } - throw error; - } - const timestamp = new Date().toISOString().replace(/[:.]/g, "-"); - return path.join(parsed.dir, `${parsed.name}.${timestamp}.dump`); -} - -async function conversationDumpStatus(dumpDir: string): Promise { - try { - const stat = await fs.promises.stat(dumpDir); - if (stat.isDirectory()) { - return `Conversation dumps written to: ${dumpDir}`; - } - return `Conversation dump path is not a directory: ${dumpDir}`; - } catch (error) { - if ((error as NodeJS.ErrnoException).code === "ENOENT") { - return `No conversation dumps written: ${dumpDir}`; - } - throw error; - } -} - -function printUsage(tasks?: EditTask[]): void { - const taskList = tasks - ? tasks.map(t => ` ${t.id.padEnd(30)} ${t.name}`).join("\n") - : " (use --list to see available tasks)"; - console.log(` -Edit Benchmark - Evaluate patch application success rates - -Usage: - bun run bench:edit [options] - -Options: - --model Provider/model ID, e.g. anthropic/claude-sonnet-4-20250514 (default) - --provider Override provider (auto-detected from model prefix if omitted) - --thinking Thinking level: off, minimal, low, medium, high, xhigh, max - --runs Runs per task (default: 1) - --timeout Timeout per run in ms (default: 120000) - --connection-timeout Timeout for first event before fast-retry (default: 30000) - --task-concurrency Max tasks to run in parallel (default: 16) - --tasks Comma-separated task IDs to run (default: all) - --max-tasks Max tasks to sample (default: 80, 0 = all) - --fixtures Fixtures directory or .tar.gz archive (default: built-in) - --edit-variant Edit variant: any string (e.g. replace, patch, hashline, vim, atom, apply_patch), or auto (default: auto) - --edit-fuzzy Fuzzy matching: true, false, auto (default: auto) - --edit-fuzzy-threshold Fuzzy threshold 0-1 or auto (default: auto) - --auto-format Auto-format output files after verify (debug only) - --guided Include an authoritative suggested edit payload (default: false) - --no-guided Disable guided mode - --max-attempts Max prompt attempts per run (default: 1) - --no-op-retry-limit Stop after repeated preventable no-op failures (default: 2) - --mutation-scope-window Allowed line-distance from mutation target for hashline refs (default: 20) - --max-turns Max turn_start events per attempt before failing (default: 30) - --output Output file (default: run_____.md) - --format Output format: markdown, json (default: markdown) - --check-fixtures Validate fixtures and exit - --require-edit-tool-call Require edit tool usage for success (default: false) - --require-read-tool-call Require read tool usage for success (default: false) - --no-edit-required Remove "must edit" prompt requirement (default: false) - --no-early-stop-on-match Don't short-circuit the run when output matches expected (default: false) - --list List available tasks and exit - --help Show this help message - -Available Tasks: -${taskList} - -Examples: - # Run full benchmark with default model - bun run bench:edit - - # Run specific tasks - bun run bench:edit --tasks core-memory-recall,operations-division - - # Compare different models - bun run bench:edit --model claude-sonnet-4-20250514 --output sonnet.md - bun run bench:edit --model claude-opus-4-5-20251101 --output opus.md - - # Run with extended thinking - bun run bench:edit --thinking high --runs 5 - - # Run from a fixtures archive - bun run bench:edit --fixtures edit-fixtures.tar.gz -`); -} - -async function resolveExtractedDir(tempDir: string): Promise { - const entries = await fs.promises.readdir(tempDir, { withFileTypes: true }); - const dirs = entries.filter(entry => entry.isDirectory()); - const files = entries.filter(entry => entry.isFile()); - if (dirs.length === 1 && files.length === 0) { - return path.join(tempDir, dirs[0]!.name); - } - return tempDir; -} - -async function extractTarGz(archivePath: string): Promise<{ dir: string; cleanupDir: string }> { - const tempDirObj = await TempDir.create("@reach-benchmark-fixtures-"); - const tempDir = tempDirObj.path(); - try { - const bytes = await Bun.file(archivePath).arrayBuffer(); - const archive = new Bun.Archive(bytes); - const files = await archive.files(); - - for (const [filePath, file] of files) { - const destPath = path.join(tempDir, filePath); - await Bun.write(destPath, file); - } - } catch (error) { - await tempDirObj.remove(); - const message = error instanceof Error ? error.message : String(error); - throw new Error(`Failed to extract archive: ${message}`, { cause: error }); - } - - return { dir: await resolveExtractedDir(tempDir), cleanupDir: tempDir }; -} - -async function resolveFixtures(fixturesArg?: string): Promise<{ tasks: EditTask[]; cleanup?: () => Promise }> { - fixturesArg ??= path.join(import.meta.dir, "../fixtures.tar.gz"); - - if (fixturesArg.endsWith(".tar.gz") || fixturesArg.endsWith(".tgz")) { - const extracted = await extractTarGz(fixturesArg); - return { - tasks: await loadTasksFromDir(extracted.dir), - cleanup: () => fs.promises.rm(extracted.cleanupDir, { recursive: true, force: true }), - }; - } - - return { tasks: await loadTasksFromDir(fixturesArg) }; -} - -async function main(): Promise { - const { values } = parseArgs({ - options: { - provider: { type: "string" }, - model: { type: "string", default: "anthropic/claude-sonnet-4-20250514" }, - thinking: { type: "string", default: "low" }, - runs: { type: "string", default: "2" }, - timeout: { type: "string", default: "120000" }, - "connection-timeout": { type: "string", default: "30000" }, - "max-turns": { type: "string", default: "30" }, - "task-concurrency": { type: "string", default: "32" }, - tasks: { type: "string" }, - fixtures: { type: "string" }, - output: { type: "string" }, - format: { type: "string", default: "markdown" }, - "check-fixtures": { type: "boolean", default: false }, - "auto-format": { type: "boolean", default: false }, - guided: { type: "boolean", default: false }, - "no-guided": { type: "boolean", default: false }, - "max-attempts": { type: "string", default: "1" }, - "no-op-retry-limit": { type: "string", default: "2" }, - "max-timeout-retries": { type: "string", default: "3" }, - "max-provider-retries": { type: "string", default: "3" }, - "mutation-scope-window": { type: "string", default: "20" }, - "require-edit-tool-call": { type: "boolean", default: false }, - "require-read-tool-call": { type: "boolean", default: false }, - "no-edit-required": { type: "boolean", default: false }, - "edit-variant": { type: "string" }, - "edit-fuzzy": { type: "string" }, - "edit-fuzzy-threshold": { type: "string" }, - "no-in-process": { type: "boolean", default: false }, - "no-early-stop-on-match": { type: "boolean", default: false }, - "max-tasks": { type: "string", default: "80" }, - list: { type: "boolean", default: false }, - help: { type: "boolean", default: false }, - }, - allowPositionals: true, - }); - - // Extract provider for display/config purposes only. - // The full model string (e.g. "openrouter/google/gemini-2.5-flash-lite") is passed - // as --model to the CLI, which handles resolution via parseModelPattern. - const model = values.model!; - const slashIndex = model.indexOf("/"); - const provider = values.provider ?? (slashIndex !== -1 ? model.slice(0, slashIndex) : "anthropic"); - - if (values.help) { - printUsage(); - process.exit(0); - } - - if (values["check-fixtures"] && values.fixtures) { - const issues = await validateFixturesFromDir(values.fixtures); - if (issues.length === 0) { - console.log("Fixtures OK"); - process.exit(0); - } - console.error("Fixture validation failed:"); - for (const issue of issues) { - console.error(` - ${issue.taskId}: ${issue.message}`); - } - process.exit(1); - } - - const { tasks: allTasks, cleanup } = await resolveFixtures(values.fixtures); - - if (values.list) { - console.log("Available Tasks:\n"); - for (const task of allTasks) { - console.log(` ${task.id}`); - console.log(` Name: ${task.name}`); - console.log(` Files: ${task.files.join(", ")}`); - console.log(""); - } - process.exit(0); - } - - let thinkingLevel: ResolvedThinkingLevel = Effort.Low; - if (values.thinking) { - const level = parseThinkingLevel(values.thinking); - if (!level) { - console.error(`Invalid thinking level: ${values.thinking}`); - console.error(`Valid levels: ${[ThinkingLevel.Off, ...THINKING_EFFORTS].join(", ")}`); - process.exit(1); - } - thinkingLevel = level; - } - - const runsPerTask = parseInt(values.runs!, 10); - if (Number.isNaN(runsPerTask) || runsPerTask < 1) { - console.error(`Invalid runs value: ${values.runs}`); - process.exit(1); - } - - const timeout = parseInt(values.timeout!, 10); - if (Number.isNaN(timeout) || timeout < 1000) { - console.error(`Invalid timeout value: ${values.timeout}`); - process.exit(1); - } - - const maxTurns = parseInt(values["max-turns"]!, 10); - if (Number.isNaN(maxTurns) || maxTurns < 1) { - console.error(`Invalid max-turns value: ${values["max-turns"]}. Must be >= 1.`); - process.exit(1); - } - - const taskConcurrency = parseInt(values["task-concurrency"]!, 10); - if (Number.isNaN(taskConcurrency) || taskConcurrency < 1) { - console.error(`Invalid task concurrency value: ${values["task-concurrency"]}`); - process.exit(1); - } - - const maxAttempts = parseInt(values["max-attempts"] ?? "2", 10); - if (Number.isNaN(maxAttempts) || maxAttempts < 1 || maxAttempts > 5) { - console.error(`Invalid max-attempts value: ${values["max-attempts"]}. Must be 1-5.`); - process.exit(1); - } - - const noOpRetryLimit = parseInt(values["no-op-retry-limit"] ?? "2", 10); - const maxTimeoutRetries = parseInt(values["max-timeout-retries"] ?? "3", 10); - const maxProviderRetries = parseInt(values["max-provider-retries"] ?? "3", 10); - const mutationScopeWindow = parseInt(values["mutation-scope-window"] ?? "20", 10); - const connectionTimeout = parseInt(values["connection-timeout"] ?? "30000", 10); - - let tasksToRun = allTasks; - if (values.tasks) { - const taskIds = values.tasks.split(",").map(s => s.trim()); - tasksToRun = []; - for (const id of taskIds) { - const task = allTasks.find(t => t.id === id); - if (!task) { - console.error(`Unknown task ID: ${id}`); - console.error(`Available tasks: ${allTasks.map(t => t.id).join(", ")}`); - process.exit(1); - } - tasksToRun.push(task); - } - } - - // Apply --max-tasks sampling (deterministic by sorting on id) - const maxTasks = parseInt(values["max-tasks"] ?? "80", 10); - if (maxTasks > 0 && tasksToRun.length > maxTasks && !values.tasks) { - // Evenly sample across mutation categories for representative coverage - const sorted = tasksToRun.slice().sort((a, b) => a.id.localeCompare(b.id)); - const step = sorted.length / maxTasks; - tasksToRun = Array.from({ length: maxTasks }, (_, i) => sorted[Math.floor(i * step)]!); - } - - const rawEditVariant = values["edit-variant"] as string | undefined; - const editVariant = rawEditVariant === "" ? undefined : rawEditVariant; - - let editFuzzy: boolean | "auto" | undefined; - if (values["edit-fuzzy"] !== undefined) { - if (values["edit-fuzzy"] === "auto") { - editFuzzy = "auto"; - } else if (values["edit-fuzzy"] === "true" || values["edit-fuzzy"] === "1") { - editFuzzy = true; - } else if (values["edit-fuzzy"] === "false" || values["edit-fuzzy"] === "0") { - editFuzzy = false; - } else { - console.error(`Invalid edit-fuzzy: ${values["edit-fuzzy"]}. Must be true, false, 1, 0, or auto.`); - process.exit(1); - } - } - - let editFuzzyThreshold: number | "auto" | undefined; - if (values["edit-fuzzy-threshold"] !== undefined) { - if (values["edit-fuzzy-threshold"] === "auto") { - editFuzzyThreshold = "auto"; - } else { - const parsed = parseFloat(values["edit-fuzzy-threshold"]); - if (Number.isNaN(parsed) || parsed < 0 || parsed > 1) { - console.error(`Invalid edit-fuzzy-threshold: ${values["edit-fuzzy-threshold"]}. Must be 0-1 or auto.`); - process.exit(1); - } - editFuzzyThreshold = parsed; - } - } - - const guided = values["no-guided"] ? false : values.guided; - - const formatType = values.format === "json" ? "json" : "markdown"; - const config: BenchmarkConfig = { - provider, - model, - thinkingLevel, - runsPerTask, - timeout, - maxTurns, - taskConcurrency, - autoFormat: values["auto-format"], - guided, - maxAttempts, - requireEditToolCall: values["require-edit-tool-call"], - requireReadToolCall: values["require-read-tool-call"], - noEditRequired: values["no-edit-required"], - editVariant, - editFuzzy, - editFuzzyThreshold, - noOpRetryLimit, - maxTimeoutRetries, - maxProviderFailureRetries: maxProviderRetries, - mutationScopeWindow, - connectionTimeout, - inProcess: !values["no-in-process"], - earlyStopOnMatch: !values["no-early-stop-on-match"], - }; - const outputPath = values.output ?? generateReportFilename(config, formatType); - config.conversationDumpDir = await resolveConversationDumpDir(outputPath); - - console.log("Edit Benchmark"); - console.log("=============="); - console.log(`Provider: ${config.provider}`); - console.log(`Model: ${config.model}`); - if (config.thinkingLevel) { - console.log(`Thinking: ${config.thinkingLevel}`); - } - console.log(`Runs per task: ${config.runsPerTask}`); - console.log(`Timeout: ${config.timeout}ms`); - console.log(`Task concurrency: ${config.taskConcurrency}`); - if (config.autoFormat) { - console.log("Auto-format: enabled"); - } - console.log(`Guided mode: ${config.guided ? "enabled" : "disabled"}`); - console.log(`Max attempts: ${config.maxAttempts}`); - if (config.maxTurns !== undefined) { - console.log(`Max turns per attempt: ${config.maxTurns}`); - } - if (config.requireEditToolCall) { - console.log("Require edit tool call: yes"); - } - if (config.requireReadToolCall) { - console.log("Require read tool call: yes"); - } - if (config.noEditRequired) { - console.log("No-edit-required baseline: yes"); - } - if (config.editVariant) { - console.log(`Edit variant: ${config.editVariant}`); - } - if (config.editFuzzy !== undefined) { - console.log(`Edit fuzzy: ${config.editFuzzy}`); - } - if (config.editFuzzyThreshold !== undefined) { - console.log(`Edit fuzzy threshold: ${config.editFuzzyThreshold}`); - } - console.log(`Tasks: ${tasksToRun.length}`); - console.log(`Conversation dumps: ${config.conversationDumpDir}`); - console.log(""); - - const progress = new LiveProgress(tasksToRun.length * config.runsPerTask, config.runsPerTask); - let latestResult = buildBenchmarkResult({ - tasks: tasksToRun, - config, - resultsByTask: new Map(), - startTime: new Date().toISOString(), - }); - let progressFinished = false; - let reportWritePromise: Promise | undefined; - const finishProgress = () => { - if (progressFinished) return; - progress.finish(); - progressFinished = true; - }; - const writeReport = async (result: BenchmarkResult, interrupted: boolean) => { - if (reportWritePromise) return reportWritePromise; - reportWritePromise = (async () => { - if (interrupted) { - console.log(""); - console.log("Benchmark interrupted; writing partial report..."); - } - const report = formatType === "json" ? generateJsonReport(result) : generateReport(result); - await Bun.write(outputPath, report); - console.log(`Report written to: ${outputPath}`); - if (config.conversationDumpDir) { - console.log(await conversationDumpStatus(config.conversationDumpDir)); - } - })(); - return reportWritePromise; - }; - const unregisterReportCleanup = postmortem.register("typescript-edit-benchmark-report", async reason => { - if (reason === postmortem.Reason.EXIT) return; - finishProgress(); - await writeReport(latestResult, true); - if (cleanup) { - await cleanup(); - } - }); - const result = await runBenchmark( - tasksToRun, - config, - event => { - progress.handleEvent(event); - }, - snapshot => { - latestResult = snapshot; - }, - ); - latestResult = result; - finishProgress(); - - console.log(""); - console.log("Benchmark complete!"); - console.log( - ` Task success rate (best of ${config.runsPerTask}): ${(result.summary.taskSuccessRate * 100).toFixed(1)}% (${result.summary.successfulTasks}/${result.summary.totalTasks})`, - ); - console.log( - ` Total tokens (best, overall): ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`, - ); - console.log( - ` Tokens/task (best, overall): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total} reasoning=${result.summary.avgTokensPerTask.reasoning}`, - ); - console.log( - ` Total tokens (one-shot successes): ${result.summary.totalOneShotSuccessTokens.input} in / ${result.summary.totalOneShotSuccessTokens.output} out`, - ); - console.log( - ` Tokens/task (one-shot successes): mean=${result.summary.avgOneShotSuccessTokensPerTask.total} median=${result.summary.medianOneShotSuccessTokensPerTask.total} p1=${result.summary.p1OneShotSuccessTokensPerTask.total} p99=${result.summary.p99OneShotSuccessTokensPerTask.total} reasoning=${result.summary.avgOneShotSuccessTokensPerTask.reasoning}`, - ); - if (result.summary.ghostRuns > 0) { - console.log(` Ghost runs (0/0/0): ${result.summary.ghostRuns}`); - } - if (result.summary.timeoutRuns > 0) { - console.log(` Timeout runs: ${result.summary.timeoutRuns}`); - } - console.log(""); - - await writeReport(result, false); - unregisterReportCleanup(); - - if (cleanup) { - await cleanup(); - } - - // In-process benchmark runs can leave provider keep-alive sockets and - // background AgentSession timers alive after the report is written. Treat the - // final report as the CLI boundary so the command returns to the shell. - await postmortem.quit(0); -} - -class LiveProgress { - readonly #totalRuns: number; - readonly #runsPerTask: number; - readonly #isTty: boolean; - #started = 0; - #completed = 0; - #success = 0; - #totalInput = 0; - #totalOutput = 0; - #totalDuration = 0; - #totalReads = 0; - #totalEdits = 0; - #totalWrites = 0; - #totalEditSuccesses = 0; - #totalToolInputChars = 0; - #indentScores: number[] = []; - #inputTokens: number[] = []; - #outputTokens: number[] = []; - #totalTokens: number[] = []; - #oneShotSuccessTokens: number[] = []; - #lastLineLength = 0; - - constructor(totalRuns: number, runsPerTask: number) { - this.#totalRuns = totalRuns; - this.#runsPerTask = runsPerTask; - this.#isTty = Boolean(process.stdout.isTTY); - } - - handleEvent(event: ProgressEvent): void { - if (event.status === "started") { - this.#started += 1; - if (!this.#isTty) { - console.log(` [${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} started...`); - } - this.#renderLine(); - return; - } - - this.#completed += 1; - if (event.result) { - if (event.result.success) { - this.#success += 1; - } - if (event.result.success && event.runIndex === 0) { - this.#oneShotSuccessTokens.push(event.result.tokens.total); - } - this.#totalInput += event.result.tokens.input; - this.#totalOutput += event.result.tokens.output; - this.#inputTokens.push(event.result.tokens.input); - this.#outputTokens.push(event.result.tokens.output); - this.#totalTokens.push(event.result.tokens.total); - this.#totalDuration += event.result.duration; - this.#totalReads += event.result.toolCalls.read; - this.#totalEdits += event.result.toolCalls.edit; - this.#totalWrites += event.result.toolCalls.write; - this.#totalEditSuccesses += event.result.toolCalls.editSuccesses; - this.#totalToolInputChars += event.result.toolCalls.totalInputChars; - if (typeof event.result.indentScore === "number") { - this.#indentScores.push(event.result.indentScore); - } - } - - const result = event.result; - if (result && !result.success && result.error) { - this.#flushLine(); - const header = paint(ANSI.red, `[${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} failed:`); - console.log(` ${header} ${result.error}`); - if (result.diff) { - const changeLines = result.diff - .split("\n") - .filter(line => /^[-+@]/.test(line) && !/^(---|\+\+\+)/.test(line)); - const maxLines = 40; - const shown = changeLines.slice(0, maxLines); - for (const line of shown) { - let color: string | undefined; - if (line.startsWith("@@")) color = ANSI.cyan; - else if (line.startsWith("-")) color = ANSI.red; - else if (line.startsWith("+")) color = ANSI.green; - console.log(` ${color ? paint(color, line) : line}`); - } - if (changeLines.length > maxLines) { - console.log(paint(ANSI.dim, ` ... (${changeLines.length - maxLines} more change lines)`)); - } - } - } - - if (result?.editFailures && result.editFailures.length > 0) { - this.#flushLine(); - for (const [i, failure] of result.editFailures.entries()) { - const args = (failure.args ?? {}) as Record; - const target = - typeof args.path === "string" ? args.path : typeof args.file === "string" ? args.file : undefined; - const op = typeof args.operation === "string" ? args.operation : undefined; - const oneLine = failure.error.replace(/\s+/g, " ").trim(); - const clipped = oneLine.length > 240 ? `${oneLine.slice(0, 237)}...` : oneLine; - const tag = paint(ANSI.yellow, `[${event.taskId}] schema #${i + 1}`); - const metaParts = [op, target].filter((v): v is string => Boolean(v)); - const meta = metaParts.length > 0 ? paint(ANSI.dim, metaParts.join(" ")) : ""; - console.log(` ${tag}${meta ? ` ${meta}` : ""} ${clipped}`); - if (failure.rawBlock) { - const rawLine = failure.rawBlock.replace(/\s+/g, " ").trim(); - const clippedRaw = rawLine.length > 240 ? `${rawLine.slice(0, 237)}...` : rawLine; - console.log(` ${paint(ANSI.dim, "raw")} ${clippedRaw}`); - } - } - } - - if (!this.#isTty) { - const status = event.result?.success ? "completed" : "failed"; - console.log(` [${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} ${status}`); - } - - this.#renderLine(); - } - - finish(): void { - this.#flushLine(); - this.#printSummary(); - } - - #printSummary(): void { - const n = this.#completed; - const denom = n || 1; - - const successRate = (this.#success / denom) * 100; - const editSuccessRate = this.#totalEdits > 0 ? (this.#totalEditSuccesses / this.#totalEdits) * 100 : 100; - const avgIndent = - this.#indentScores.length > 0 ? this.#indentScores.reduce((a, b) => a + b, 0) / this.#indentScores.length : 0; - - console.log(""); - console.log(paint(ANSI.bold, "Runtime Stats:")); - console.log( - ` Task success: ${paint(rateColor(successRate), `${successRate.toFixed(1)}% (${this.#success}/${n})`)}`, - ); - console.log( - ` Edit success: ${paint(rateColor(editSuccessRate), `${editSuccessRate.toFixed(1)}% (${this.#totalEditSuccesses}/${this.#totalEdits})`)}`, - ); - console.log(` Avg indent score: ${avgIndent.toFixed(2)}`); - console.log(` Tool calls: read=${this.#totalReads} edit=${this.#totalEdits} write=${this.#totalWrites}`); - console.log(` Tool input chars: ${this.#totalToolInputChars.toLocaleString()}`); - const fmtTokens = (samples: number[]): string => { - if (samples.length === 0) return "mean=0 median=0 p1=0 p99=0"; - const sorted = [...samples].sort((a, b) => a - b); - const mean = Math.round(sorted.reduce((a, b) => a + b, 0) / sorted.length); - return `mean=${mean} median=${Math.round(percentile(sorted, 50))} p1=${Math.round(percentile(sorted, 1))} p99=${Math.round(percentile(sorted, 99))}`; - }; - console.log(` Tokens/task in: ${fmtTokens(this.#inputTokens)}`); - console.log(` Tokens/task out: ${fmtTokens(this.#outputTokens)}`); - console.log(` Tokens/task tot: ${fmtTokens(this.#totalTokens)}`); - console.log(` Tokens/task (one-shot successes): ${fmtTokens(this.#oneShotSuccessTokens)}`); - console.log(` Avg time/task: ${Math.round(this.#totalDuration / denom)}ms`); - } - - #renderLine(): void { - if (!this.#isTty) { - return; - } - const successRate = this.#completed > 0 ? (this.#success / this.#completed) * 100 : 0; - const editRate = this.#totalEdits > 0 ? (this.#totalEditSuccesses / this.#totalEdits) * 100 : 100; - const avgInput = this.#completed > 0 ? Math.round(this.#totalInput / this.#completed) : 0; - const avgOutput = this.#completed > 0 ? Math.round(this.#totalOutput / this.#completed) : 0; - const avgDuration = this.#completed > 0 ? Math.round(this.#totalDuration / this.#completed) : 0; - const inFlight = this.#started - this.#completed; - const bar = this.#renderBar(this.#completed, this.#totalRuns, 20); - const progress = paint(ANSI.bold, `${this.#completed}/${this.#totalRuns}`); - const taskCol = `task=${paint(rateColor(successRate), `${successRate.toFixed(0)}%`)}`; - const editCol = `edit=${paint(rateColor(editRate), `${editRate.toFixed(0)}%`)}`; - const tokCol = paint(ANSI.dim, `tok=${avgInput}/${avgOutput}`); - const durCol = paint(ANSI.dim, `${avgDuration}ms`); - const rewCol = paint(ANSI.dim, `r/e/w=${this.#totalReads}/${this.#totalEdits}/${this.#totalWrites}`); - const flyCol = `fly=${paint(ANSI.cyan, String(inFlight))}`; - const line = ` ${bar} ${progress} ${taskCol} ${editCol} ${tokCol} ${durCol} ${rewCol} ${flyCol}`; - this.#writeLine(line); - } - - #renderBar(done: number, total: number, width: number): string { - const ratio = total === 0 ? 0 : done / total; - const filled = Math.round(ratio * width); - const empty = Math.max(0, width - filled); - const filledPart = paint(ANSI.green, "#".repeat(filled)); - const emptyPart = paint(ANSI.dim, "-".repeat(empty)); - return `[${filledPart}${emptyPart}]`; - } - - #writeLine(line: string): void { - const lineWidth = visibleWidth(line); - const pad = this.#lastLineLength > lineWidth ? padding(this.#lastLineLength - lineWidth) : ""; - process.stdout.write(`\r${line}${pad}`); - this.#lastLineLength = lineWidth; - } - - #flushLine(): void { - if (!this.#isTty) { - return; - } - if (this.#lastLineLength > 0) { - process.stdout.write(`\r${padding(this.#lastLineLength)}\r`); - this.#lastLineLength = 0; - } - } -} - -main().catch(async err => { - console.error("Benchmark failed:", err); - await postmortem.quit(1); -});