feat(harbor-manager): unified benchmark normalization and reporting
- Implemented a unified benchmark normalization layer to handle metrics and artifacts across harbor, edit, and snapcompact benchmarks. - Integrated the TypeScript edit benchmark directly into the manager, migrating logic and deprecating the standalone package. - Updated the API and database schema to support standardized benchmark configurations, metrics, and trace-based reporting. - Enhanced the UI to visualize comparative performance metrics, including pass rate, cost, and latency deltas for benchmark runs.
This commit is contained in:
@@ -140,12 +140,19 @@
|
||||
"name": "@oh-my-pi/harbor-manager",
|
||||
"version": "0.0.1",
|
||||
"bin": {
|
||||
"harbor-manager": "src/runner.ts",
|
||||
"harbor-manager": "src/server.ts",
|
||||
},
|
||||
"dependencies": {
|
||||
"@oh-my-pi/hashline": "catalog:",
|
||||
"@oh-my-pi/pi-agent-core": "catalog:",
|
||||
"@oh-my-pi/pi-ai": "catalog:",
|
||||
"@oh-my-pi/pi-coding-agent": "catalog:",
|
||||
"@oh-my-pi/pi-utils": "catalog:",
|
||||
"@oh-my-pi/typescript-edit-benchmark": "workspace:*",
|
||||
"clsx": "^2.1.1",
|
||||
"d3-scale": "^4.0.2",
|
||||
"d3-shape": "^3.2.0",
|
||||
"diff": "catalog:",
|
||||
"motion": "^12.15.0",
|
||||
"react": "^19.1.0",
|
||||
"react-dom": "^19.1.0",
|
||||
@@ -276,9 +283,6 @@
|
||||
"packages/typescript-edit-benchmark": {
|
||||
"name": "@oh-my-pi/typescript-edit-benchmark",
|
||||
"version": "0.0.1",
|
||||
"bin": {
|
||||
"typescript-edit-benchmark": "src/index.ts",
|
||||
},
|
||||
"dependencies": {
|
||||
"@babel/generator": "catalog:",
|
||||
"@babel/parser": "catalog:",
|
||||
|
||||
@@ -149,7 +149,6 @@
|
||||
"ci:release:publish": "bun scripts/ci-release-publish.ts",
|
||||
"ci:release:publish-native-leaf": "bun scripts/ci-release-publish.ts --native-leaf",
|
||||
"bench:gen-fixtures": "bun --cwd=packages/typescript-edit-benchmark run src/generate.ts --typescript-dir /tmp/typescript-source --count-per-type 8",
|
||||
"bench:edit": "bun --cwd=packages/typescript-edit-benchmark run start",
|
||||
"stats:sync": "python3 scripts/session-stats/sync.py",
|
||||
"stats:tools": "python3 scripts/session-stats/analyze.py tools",
|
||||
"stats:edits": "python3 scripts/session-stats/analyze.py edits",
|
||||
|
||||
@@ -1,22 +1,16 @@
|
||||
# @oh-my-pi/harbor-manager
|
||||
|
||||
Manage [Harbor](https://github.com/laude-institute/harbor) benchmark runs against
|
||||
the **local `omp` build**: a CLI runner with a live terminal dashboard, a
|
||||
SQLite-backed run store, a REST/SSE API, and a web dashboard with live
|
||||
run → trial → transcript drill-down.
|
||||
|
||||
Works with any Harbor dataset (`terminal-bench@2.0` by default,
|
||||
`swe-bench/swe-bench-verified`, …).
|
||||
One manager for repository benchmarks. Harbor, TypeScript edit, and SnapCompact
|
||||
runs use the same experiment → run → trace model, SQLite store, REST/SSE API,
|
||||
and dashboard. Benchmark-native artifacts remain on disk; adapters normalize
|
||||
their live progress, scores, token usage, costs, and traces.
|
||||
|
||||
```bash
|
||||
# CLI run (owns the terminal, writes a markdown report)
|
||||
bun src/runner.ts --model anthropic/claude-sonnet-4-6 --tasks 20 --concurrency 4
|
||||
|
||||
# Web manager: dashboard + API on :4700
|
||||
bun run serve # bun src/server.ts [--port 4700] [--jobs-dir <path>]
|
||||
# Dashboard + API on :4700; launch every benchmark from the same “new run” form
|
||||
bun run serve --port 4700
|
||||
```
|
||||
|
||||
## How runs execute
|
||||
## How Harbor runs execute
|
||||
|
||||
1. **Local omp, not npm.** By default the runner bind-mounts the repo
|
||||
read-only into each task container (`--install source`) and runs omp
|
||||
@@ -35,34 +29,36 @@ bun run serve # bun src/server.ts [--port 4700] [--jobs-dir <path>]
|
||||
|
||||
## Server
|
||||
|
||||
- `GET /` — dashboard: run list (SSE-live), trial grid, transcript viewer
|
||||
that tails live sessions.
|
||||
- `GET /api/runs` — run rows (rollups: pass/fail/error/spend/tokens).
|
||||
- `POST /api/runs` — launch. Body:
|
||||
- `GET /` — experiments, runs, normalized traces, and a launch form for every benchmark.
|
||||
- `GET /api/experiments` — experiment summaries across all benchmark types.
|
||||
- `GET /api/runs` — uniform run rows with benchmark, score, progress, spend, and tokens.
|
||||
- `POST /api/runs` — launch through a benchmark adapter. Body:
|
||||
|
||||
```json
|
||||
{
|
||||
"benchmark": "edit",
|
||||
"model": "anthropic/claude-opus-4-8",
|
||||
"dataset": "swe-bench/swe-bench-verified",
|
||||
"tasks": 20,
|
||||
"concurrency": 4,
|
||||
"timeoutMultiplier": 2,
|
||||
"include": ["swe-bench/astropy__astropy-14995"],
|
||||
"slide": { "model": "google/gemini-3.5-flash", "onAction": true, "plan": true }
|
||||
"attempts": 2,
|
||||
"jobName": "edit-baseline",
|
||||
"role": "baseline",
|
||||
"goal": "compare edit strategies"
|
||||
}
|
||||
```
|
||||
|
||||
`slide.turns` and `slide.onAction` are mutually exclusive triggers.
|
||||
- `GET /api/runs/:name` — `{ run, trials }` (syncs from disk on read).
|
||||
`benchmark` is `harbor`, `edit`, or `snapcompact`. Harbor uses `dataset`,
|
||||
`include`, `timeoutMultiplier`, and `slide`; edit uses `include` as task IDs;
|
||||
SnapCompact uses `conditions` and treats `tasks` as the passage limit.
|
||||
- `GET /api/runs/:name` — `{ run, traces }` (syncs native artifacts on read).
|
||||
- `DELETE /api/runs/:name` — cancel a manager-launched run.
|
||||
- `GET /api/runs/:name/trials/:trial/transcript?tail=N[&raw=1]` — compact
|
||||
(or raw JSONL) view of the trial's live session log.
|
||||
- `GET /api/runs/:name/traces/:trace[?raw=1]` — normalized or native trace.
|
||||
- `GET /api/events` — SSE stream of run-list snapshots (sent on change).
|
||||
|
||||
State lives in `<jobs-dir>/_manager/harbor-manager.sqlite`; the filesystem
|
||||
stays the source of truth and historical CLI runs are auto-discovered.
|
||||
|
||||
## Runner options (excerpt)
|
||||
## Harbor runner options (excerpt)
|
||||
|
||||
| Option | Default | Notes |
|
||||
|---|---|---|
|
||||
|
||||
@@ -0,0 +1,4 @@
|
||||
declare module "*.tar.gz" {
|
||||
const content: string;
|
||||
export default content;
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
#!/usr/bin/env bun
|
||||
/** Manager-owned executable adapter for the TypeScript edit benchmark. */
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as path from "node:path";
|
||||
import { parseArgs } from "node:util";
|
||||
import { TempDir } from "@oh-my-pi/pi-utils";
|
||||
import { loadTasksFromDir } from "@oh-my-pi/typescript-edit-benchmark/tasks";
|
||||
import { generateJsonReport } from "./report";
|
||||
import { type BenchmarkConfig, runBenchmark } from "./runner";
|
||||
|
||||
const EDIT_PACKAGE = path.resolve(import.meta.dir, "..", "..", "..", "typescript-edit-benchmark");
|
||||
|
||||
async function extractFixtures(): Promise<{ dir: string; temp: TempDir }> {
|
||||
const temp = await TempDir.create("@harbor-edit-fixtures-");
|
||||
const archive = new Bun.Archive(await Bun.file(path.join(EDIT_PACKAGE, "fixtures.tar.gz")).arrayBuffer());
|
||||
for (const [filePath, file] of await archive.files()) {
|
||||
await Bun.write(path.join(temp.path(), filePath), file);
|
||||
}
|
||||
const entries = await fs.readdir(temp.path(), { withFileTypes: true });
|
||||
const directories = entries.filter(entry => entry.isDirectory());
|
||||
const files = entries.filter(entry => entry.isFile());
|
||||
return { dir: directories.length === 1 && files.length === 0 ? path.join(temp.path(), directories[0]!.name) : temp.path(), temp };
|
||||
}
|
||||
|
||||
/** Execute an edit benchmark and continuously materialize its normalized source artifact. */
|
||||
export async function main(argv = process.argv.slice(2)): Promise<void> {
|
||||
const { values } = parseArgs({
|
||||
args: argv,
|
||||
options: {
|
||||
model: { type: "string" },
|
||||
output: { type: "string" },
|
||||
"max-tasks": { type: "string", default: "80" },
|
||||
tasks: { type: "string" },
|
||||
"task-concurrency": { type: "string", default: "32" },
|
||||
runs: { type: "string", default: "1" },
|
||||
list: { type: "boolean", default: false },
|
||||
},
|
||||
strict: true,
|
||||
});
|
||||
const fixtures = await extractFixtures();
|
||||
try {
|
||||
let tasks = await loadTasksFromDir(fixtures.dir);
|
||||
if (values.list) {
|
||||
process.stdout.write(`${JSON.stringify(tasks.map(task => ({ id: task.id, name: task.name })))}\n`);
|
||||
return;
|
||||
}
|
||||
if (!values.model || !values.output) throw new Error("edit adapter requires --model and --output");
|
||||
if (values.tasks) {
|
||||
const selected = new Set(values.tasks.split(",").map(value => value.trim()));
|
||||
tasks = tasks.filter(task => selected.has(task.id));
|
||||
if (tasks.length !== selected.size) throw new Error("one or more edit task ids were not found");
|
||||
} else {
|
||||
const limit = Number(values["max-tasks"]);
|
||||
if (limit > 0 && tasks.length > limit) {
|
||||
const sorted = tasks.slice().sort((a, b) => a.id.localeCompare(b.id));
|
||||
const step = sorted.length / limit;
|
||||
tasks = Array.from({ length: limit }, (_, index) => sorted[Math.floor(index * step)]!);
|
||||
}
|
||||
}
|
||||
const slash = values.model.indexOf("/");
|
||||
const config: BenchmarkConfig = {
|
||||
provider: slash === -1 ? "anthropic" : values.model.slice(0, slash),
|
||||
model: values.model,
|
||||
runsPerTask: Number(values.runs),
|
||||
timeout: 120_000,
|
||||
connectionTimeout: 30_000,
|
||||
maxTurns: 30,
|
||||
taskConcurrency: Number(values["task-concurrency"]),
|
||||
guided: false,
|
||||
maxAttempts: 1,
|
||||
noOpRetryLimit: 2,
|
||||
maxTimeoutRetries: 3,
|
||||
maxProviderFailureRetries: 3,
|
||||
mutationScopeWindow: 20,
|
||||
conversationDumpDir: path.join(path.dirname(values.output), "result.dump"),
|
||||
inProcess: true,
|
||||
earlyStopOnMatch: true,
|
||||
};
|
||||
let writes = Promise.resolve();
|
||||
const result = await runBenchmark(tasks, config, undefined, snapshot => {
|
||||
writes = writes.then(async () => {
|
||||
await Bun.write(values.output!, generateJsonReport(snapshot));
|
||||
});
|
||||
});
|
||||
await writes;
|
||||
await Bun.write(values.output, generateJsonReport(result));
|
||||
} finally {
|
||||
await fixtures.temp.remove();
|
||||
}
|
||||
}
|
||||
|
||||
if (import.meta.main) {
|
||||
main().catch(error => {
|
||||
process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`);
|
||||
process.exitCode = 1;
|
||||
});
|
||||
}
|
||||
+2
-6
@@ -4,12 +4,8 @@ import * as path from "node:path";
|
||||
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
|
||||
import { formatSessionDumpText, SessionManager } from "@oh-my-pi/pi-coding-agent";
|
||||
import { TempDir } from "@oh-my-pi/pi-utils";
|
||||
import { generateReport } from "@oh-my-pi/typescript-edit-benchmark/report";
|
||||
import {
|
||||
buildBenchmarkResult,
|
||||
type TaskRunResult,
|
||||
writeConversationDump,
|
||||
} from "@oh-my-pi/typescript-edit-benchmark/runner";
|
||||
import { generateReport } from "./report";
|
||||
import { buildBenchmarkResult, type TaskRunResult, writeConversationDump } from "./runner";
|
||||
import type { EditTask } from "@oh-my-pi/typescript-edit-benchmark/tasks";
|
||||
|
||||
const tempDirs: TempDir[] = [];
|
||||
+12
-5
@@ -13,15 +13,22 @@ import type { Model, ToolExample } from "@oh-my-pi/pi-ai";
|
||||
import { formatSessionDumpText, RpcClient } from "@oh-my-pi/pi-coding-agent";
|
||||
import { prompt } from "@oh-my-pi/pi-utils";
|
||||
import { diffLines } from "diff";
|
||||
import { formatDirectory } from "./formatter";
|
||||
import { discoverSharedInfra, InProcessClient, type SharedInfra } from "./in-process-client";
|
||||
import { formatDirectory } from "@oh-my-pi/typescript-edit-benchmark/formatter";
|
||||
import {
|
||||
discoverSharedInfra,
|
||||
InProcessClient,
|
||||
type SharedInfra,
|
||||
} from "@oh-my-pi/typescript-edit-benchmark/in-process-client";
|
||||
import benchmarkRetryPrompt from "./prompts/benchmark-retry.md" with { type: "text" };
|
||||
import benchmarkSystemPrompt from "./prompts/benchmark-system.md" with { type: "text" };
|
||||
import benchmarkTaskPrompt from "./prompts/benchmark-task.md" with { type: "text" };
|
||||
import type { EditTask } from "./tasks";
|
||||
import { verifyExpectedFileSubset, verifyExpectedFiles } from "./verify";
|
||||
import type { EditTask } from "@oh-my-pi/typescript-edit-benchmark/tasks";
|
||||
import {
|
||||
verifyExpectedFileSubset,
|
||||
verifyExpectedFiles,
|
||||
} from "@oh-my-pi/typescript-edit-benchmark/verify";
|
||||
|
||||
const REPO_ROOT = path.resolve(import.meta.dir, "..", "..", "..");
|
||||
const REPO_ROOT = path.resolve(import.meta.dir, "..", "..", "..", "..");
|
||||
const RUNS_DIR = path.join(REPO_ROOT, "runs");
|
||||
const TMP = path.join(RUNS_DIR, `rb-${Math.random().toString(36).slice(2, 10)}`);
|
||||
const CLI_PATH = Bun.fileURLToPath(import.meta.resolve("@oh-my-pi/pi-coding-agent/cli"));
|
||||
@@ -0,0 +1,4 @@
|
||||
{
|
||||
"extends": "../../../tsconfig.workspace.json",
|
||||
"include": ["."]
|
||||
}
|
||||
@@ -3,7 +3,7 @@
|
||||
"private": true,
|
||||
"name": "@oh-my-pi/harbor-manager",
|
||||
"version": "0.0.1",
|
||||
"description": "Manage Harbor benchmark runs against the local omp build: CLI runner, SQLite-backed run store, REST/SSE API, and a live web dashboard",
|
||||
"description": "Unified benchmark runners plus Harbor run storage, REST/SSE APIs, and a live web dashboard",
|
||||
"homepage": "https://omp.sh",
|
||||
"author": "Can Boluk",
|
||||
"license": "MIT",
|
||||
@@ -13,20 +13,27 @@
|
||||
"directory": "packages/harbor-manager"
|
||||
},
|
||||
"bin": {
|
||||
"harbor-manager": "src/runner.ts"
|
||||
"harbor-manager": "src/server.ts"
|
||||
},
|
||||
"scripts": {
|
||||
"check": "biome check . && bun run check:types",
|
||||
"check:types": "tsgo -p tsconfig.json --noEmit",
|
||||
"check:types": "tsgo -p tsconfig.json --noEmit && tsgo -p adapters/edit/tsconfig.json --noEmit",
|
||||
"lint": "biome lint .",
|
||||
"start": "bun run src/runner.ts",
|
||||
"start": "bun run src/server.ts",
|
||||
"serve": "bun run src/server.ts",
|
||||
"test": "bun test"
|
||||
},
|
||||
"dependencies": {
|
||||
"@oh-my-pi/hashline": "catalog:",
|
||||
"@oh-my-pi/pi-agent-core": "catalog:",
|
||||
"@oh-my-pi/pi-ai": "catalog:",
|
||||
"@oh-my-pi/pi-coding-agent": "catalog:",
|
||||
"@oh-my-pi/pi-utils": "catalog:",
|
||||
"@oh-my-pi/typescript-edit-benchmark": "workspace:*",
|
||||
"clsx": "^2.1.1",
|
||||
"d3-scale": "^4.0.2",
|
||||
"d3-shape": "^3.2.0",
|
||||
"diff": "catalog:",
|
||||
"motion": "^12.15.0",
|
||||
"react": "^19.1.0",
|
||||
"react-dom": "^19.1.0",
|
||||
|
||||
+23
-16
@@ -36,6 +36,7 @@ from concurrent.futures import ThreadPoolExecutor
|
||||
from pathlib import Path
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
RESEARCH = HERE.parents[2] / "snapcompact" / "research"
|
||||
|
||||
def find_agent_prompts() -> Path:
|
||||
for parent in HERE.parents:
|
||||
@@ -48,16 +49,16 @@ def find_agent_prompts() -> Path:
|
||||
raise FileNotFoundError("Could not find agent compaction prompts")
|
||||
|
||||
|
||||
sys.path.insert(0, str(HERE))
|
||||
sys.path.insert(0, str(RESEARCH))
|
||||
|
||||
import squad # noqa: E402
|
||||
from anthropic_api import complete, image_block, load_api_key # noqa: E402
|
||||
from bdf import VARIANTS, FontCfg, capacity, render # noqa: E402
|
||||
|
||||
AGENT_PROMPTS = find_agent_prompts()
|
||||
CACHE = HERE / ".cache"
|
||||
CACHE = RESEARCH / ".cache"
|
||||
QA_CACHE = CACHE / "qa"
|
||||
RESULTS = HERE / "results"
|
||||
RESULTS = RESEARCH / "results"
|
||||
|
||||
FONTS = {
|
||||
"8x13": FontCfg("8x13", "8x13", 8, 13),
|
||||
@@ -100,7 +101,7 @@ def sha8(*parts: str) -> str:
|
||||
|
||||
|
||||
def load_prompt(name: str) -> str:
|
||||
return (HERE / "prompts" / name).read_text()
|
||||
return (RESEARCH / "prompts" / name).read_text()
|
||||
|
||||
|
||||
def agent_prompt(name: str) -> str:
|
||||
@@ -276,6 +277,7 @@ def main() -> None:
|
||||
ap.add_argument("--price-in", type=float, default=10.0, help="$ per 1M input tokens")
|
||||
ap.add_argument("--price-out", type=float, default=50.0, help="$ per 1M output tokens")
|
||||
ap.add_argument("--env", default="~/.env")
|
||||
ap.add_argument("--output-dir", help="write records.jsonl and summary.json directly to this directory")
|
||||
args = ap.parse_args()
|
||||
|
||||
CACHE.mkdir(exist_ok=True)
|
||||
@@ -288,7 +290,11 @@ def main() -> None:
|
||||
f"-e{args.effort}" if args.effort else "",
|
||||
]
|
||||
)
|
||||
run_dir = RESULTS / f"{args.model}-seed{args.seed}-qpc{args.qpc}-{scope}{tag}"
|
||||
run_dir = (
|
||||
Path(args.output_dir).expanduser().resolve()
|
||||
if args.output_dir
|
||||
else RESULTS / f"{args.model}-seed{args.seed}-qpc{args.qpc}-{scope}{tag}"
|
||||
)
|
||||
run_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
paras = squad.load_paragraphs(CACHE)
|
||||
@@ -312,17 +318,18 @@ def main() -> None:
|
||||
ctx_args = {"args": args, "flow": flow, "paras": paras, "offsets": offsets, "api_key": api_key}
|
||||
records: list[dict] = []
|
||||
done = 0
|
||||
with ThreadPoolExecutor(args.workers) as pool:
|
||||
futures = [pool.submit(run_chunk, cond, start, end, ctx_args) for cond, start, end in tasks]
|
||||
for fut in futures:
|
||||
records.extend(fut.result())
|
||||
done += 1
|
||||
if done % 20 == 0:
|
||||
print(f" {done}/{len(tasks)} chunks", flush=True)
|
||||
|
||||
with (run_dir / "records.jsonl").open("w") as fh:
|
||||
for r in records:
|
||||
fh.write(json.dumps(r) + "\n")
|
||||
with (run_dir / "records.jsonl").open("w") as records_file:
|
||||
with ThreadPoolExecutor(args.workers) as pool:
|
||||
futures = [pool.submit(run_chunk, cond, start, end, ctx_args) for cond, start, end in tasks]
|
||||
for fut in futures:
|
||||
chunk_records = fut.result()
|
||||
records.extend(chunk_records)
|
||||
for record in chunk_records:
|
||||
records_file.write(json.dumps(record) + "\n")
|
||||
records_file.flush()
|
||||
done += 1
|
||||
if done % 20 == 0:
|
||||
print(f" {done}/{len(tasks)} chunks", flush=True)
|
||||
|
||||
rows = [
|
||||
aggregate(cond["name"], [r for r in records if r["cond"] == cond["name"]], args.price_in, args.price_out)
|
||||
@@ -0,0 +1,85 @@
|
||||
import { afterEach, describe, expect, it } from "bun:test";
|
||||
import * as fs from "node:fs";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { BENCHMARK_DEFINITIONS, readBenchmarkSnapshot } from "./benchmarks";
|
||||
|
||||
const cleanups: string[] = [];
|
||||
|
||||
function jobDir(): string {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), "harbor-benchmark-"));
|
||||
cleanups.push(dir);
|
||||
return dir;
|
||||
}
|
||||
|
||||
afterEach(() => {
|
||||
for (const dir of cleanups.splice(0)) fs.rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
describe("benchmark adapters", () => {
|
||||
it("normalizes edit attempts, traces, tokens, and declared metrics", () => {
|
||||
const dir = jobDir();
|
||||
fs.writeFileSync(
|
||||
path.join(dir, "result.json"),
|
||||
JSON.stringify({
|
||||
tasks: [
|
||||
{
|
||||
id: "rename-symbol",
|
||||
name: "Rename symbol",
|
||||
runs: [
|
||||
{
|
||||
runIndex: 0,
|
||||
success: true,
|
||||
duration: 1200,
|
||||
tokens: { input: 100, output: 20, reasoning: 5 },
|
||||
},
|
||||
],
|
||||
},
|
||||
],
|
||||
summary: {
|
||||
totalRuns: 1,
|
||||
successfulRuns: 1,
|
||||
taskSuccessRate: 1,
|
||||
editSuccessRate: 0.75,
|
||||
totalTokens: { input: 100, output: 20 },
|
||||
},
|
||||
}),
|
||||
);
|
||||
|
||||
const snapshot = readBenchmarkSnapshot("edit", dir);
|
||||
expect(snapshot.metrics).toEqual({ task_success_rate: 1, edit_success_rate: 0.75 });
|
||||
expect(snapshot.traces[0]).toMatchObject({
|
||||
name: "rename-symbol__1",
|
||||
status: "pass",
|
||||
tracePath: path.join("result.dump", "rename-symbol", "run-1.md"),
|
||||
});
|
||||
expect([snapshot.tokIn, snapshot.tokOut]).toEqual([100, 20]);
|
||||
});
|
||||
|
||||
it("normalizes SnapCompact records and weighted quality metrics", () => {
|
||||
const dir = jobDir();
|
||||
fs.writeFileSync(
|
||||
path.join(dir, "records.jsonl"),
|
||||
`${JSON.stringify({ cond: "text", chunk: 0, pos_rel: 0.2, q: "q", answer: "a", golds: ["a"], em: 1, f1: 1 })}\n`,
|
||||
);
|
||||
fs.writeFileSync(
|
||||
path.join(dir, "summary.json"),
|
||||
JSON.stringify({
|
||||
rows: [
|
||||
{ n: 2, f1: 0.75, em: 0.5, cost_usd: 1.25, tokens_in: 200, tokens_out: 30, cache_w: 10, cache_r: 20 },
|
||||
],
|
||||
}),
|
||||
);
|
||||
|
||||
const snapshot = readBenchmarkSnapshot("snapcompact", dir);
|
||||
expect(snapshot.metrics).toEqual({ f1: 0.75, exact_match: 0.5 });
|
||||
expect(snapshot.traces[0]).toMatchObject({ status: "pass", reward: 1, tracePath: "record:1" });
|
||||
expect(snapshot.costUsd).toBe(1.25);
|
||||
expect(snapshot.tokCache).toBe(30);
|
||||
});
|
||||
|
||||
it("publishes metric definitions for every managed benchmark", () => {
|
||||
expect(BENCHMARK_DEFINITIONS.map(definition => definition.kind)).toEqual(["harbor", "edit", "snapcompact"]);
|
||||
expect(BENCHMARK_DEFINITIONS.every(definition => definition.metrics.length > 0)).toBe(true);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,273 @@
|
||||
/** Benchmark adapters normalize native artifacts into manager runs and traces. */
|
||||
import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import { aggregate, readJobResult, readTrials } from "./runner";
|
||||
import type { BenchmarkKind } from "./store";
|
||||
|
||||
/** Describes a benchmark metric so storage and UI do not hard-code benchmark semantics. */
|
||||
export interface MetricDefinition {
|
||||
key: string;
|
||||
label: string;
|
||||
format: "percent" | "number" | "usd";
|
||||
higherIsBetter: boolean;
|
||||
}
|
||||
|
||||
/** Adapter metadata exposed to launch clients and the dashboard. */
|
||||
export interface BenchmarkDefinition {
|
||||
kind: BenchmarkKind;
|
||||
label: string;
|
||||
metrics: MetricDefinition[];
|
||||
}
|
||||
|
||||
/** Built-in benchmark adapters and their native score definitions. */
|
||||
export const BENCHMARK_DEFINITIONS: BenchmarkDefinition[] = [
|
||||
{
|
||||
kind: "harbor",
|
||||
label: "Harbor",
|
||||
metrics: [{ key: "success_rate", label: "Success rate", format: "percent", higherIsBetter: true }],
|
||||
},
|
||||
{
|
||||
kind: "edit",
|
||||
label: "TypeScript edit",
|
||||
metrics: [
|
||||
{ key: "task_success_rate", label: "Task success", format: "percent", higherIsBetter: true },
|
||||
{ key: "edit_success_rate", label: "Edit success", format: "percent", higherIsBetter: true },
|
||||
],
|
||||
},
|
||||
{
|
||||
kind: "snapcompact",
|
||||
label: "SnapCompact",
|
||||
metrics: [
|
||||
{ key: "f1", label: "F1", format: "percent", higherIsBetter: true },
|
||||
{ key: "exact_match", label: "Exact match", format: "percent", higherIsBetter: true },
|
||||
],
|
||||
},
|
||||
];
|
||||
|
||||
/** A normalized trace emitted by any benchmark adapter. */
|
||||
export interface BenchmarkTrace {
|
||||
name: string;
|
||||
task: string;
|
||||
status: "pass" | "fail" | "error" | "running";
|
||||
reward: number | null;
|
||||
costUsd: number;
|
||||
durationMs: number;
|
||||
detail: string;
|
||||
tracePath: string | null;
|
||||
}
|
||||
|
||||
/** Uniform aggregate and traces read from benchmark-native artifacts. */
|
||||
export interface BenchmarkSnapshot {
|
||||
traces: BenchmarkTrace[];
|
||||
total: number;
|
||||
done: number;
|
||||
pass: number;
|
||||
fail: number;
|
||||
error: number;
|
||||
running: number;
|
||||
costUsd: number;
|
||||
tokIn: number;
|
||||
tokOut: number;
|
||||
tokCache: number;
|
||||
score: number | null;
|
||||
metrics: Record<string, number | null>;
|
||||
}
|
||||
|
||||
interface EditRun {
|
||||
runIndex: number;
|
||||
success: boolean;
|
||||
error?: string;
|
||||
duration: number;
|
||||
tokens: { input: number; output: number; reasoning: number };
|
||||
toolCalls?: { read: number; edit: number; write: number };
|
||||
}
|
||||
|
||||
interface EditTask {
|
||||
id: string;
|
||||
name: string;
|
||||
runs: EditRun[];
|
||||
}
|
||||
|
||||
interface EditResult {
|
||||
tasks: EditTask[];
|
||||
summary: {
|
||||
totalRuns: number;
|
||||
successfulRuns: number;
|
||||
taskSuccessRate: number;
|
||||
editSuccessRate: number;
|
||||
totalTokens: { input: number; output: number };
|
||||
};
|
||||
}
|
||||
|
||||
interface SnapRecord {
|
||||
cond: string;
|
||||
chunk: number;
|
||||
pos_rel: number;
|
||||
q: string;
|
||||
answer: string;
|
||||
golds: string[];
|
||||
em: number;
|
||||
f1: number;
|
||||
}
|
||||
|
||||
interface SnapSummaryRow {
|
||||
n: number;
|
||||
f1: number;
|
||||
em: number;
|
||||
cost_usd: number;
|
||||
tokens_in: number;
|
||||
tokens_out: number;
|
||||
cache_w: number;
|
||||
cache_r: number;
|
||||
}
|
||||
|
||||
interface SnapSummary {
|
||||
rows: SnapSummaryRow[];
|
||||
}
|
||||
|
||||
function emptySnapshot(): BenchmarkSnapshot {
|
||||
return {
|
||||
traces: [],
|
||||
total: 0,
|
||||
done: 0,
|
||||
pass: 0,
|
||||
fail: 0,
|
||||
error: 0,
|
||||
running: 0,
|
||||
costUsd: 0,
|
||||
tokIn: 0,
|
||||
tokOut: 0,
|
||||
tokCache: 0,
|
||||
score: null,
|
||||
metrics: {},
|
||||
};
|
||||
}
|
||||
|
||||
function readEditSnapshot(jobDir: string): BenchmarkSnapshot {
|
||||
const file = path.join(jobDir, "result.json");
|
||||
if (!fs.existsSync(file)) return emptySnapshot();
|
||||
const result: EditResult = JSON.parse(fs.readFileSync(file, "utf8"));
|
||||
const traces: BenchmarkTrace[] = [];
|
||||
let tokIn = 0;
|
||||
let tokOut = 0;
|
||||
for (const task of result.tasks) {
|
||||
for (const run of task.runs) {
|
||||
tokIn += run.tokens.input;
|
||||
tokOut += run.tokens.output;
|
||||
const runNumber = run.runIndex + 1;
|
||||
traces.push({
|
||||
name: `${task.id}__${runNumber}`,
|
||||
task: task.id,
|
||||
status: run.success ? "pass" : run.error ? "error" : "fail",
|
||||
reward: run.success ? 1 : 0,
|
||||
costUsd: 0,
|
||||
durationMs: run.duration,
|
||||
detail: JSON.stringify({ name: task.name, error: run.error ?? null, tools: run.toolCalls ?? null }),
|
||||
tracePath: path.join("result.dump", task.id.replace(/[^a-zA-Z0-9._-]/g, "_"), `run-${runNumber}.md`),
|
||||
});
|
||||
}
|
||||
}
|
||||
const pass = traces.filter(trace => trace.status === "pass").length;
|
||||
const error = traces.filter(trace => trace.status === "error").length;
|
||||
return {
|
||||
traces,
|
||||
total: result.summary.totalRuns,
|
||||
done: traces.length,
|
||||
pass,
|
||||
fail: traces.length - pass,
|
||||
error,
|
||||
running: Math.max(0, result.summary.totalRuns - traces.length),
|
||||
costUsd: 0,
|
||||
tokIn,
|
||||
tokOut,
|
||||
tokCache: 0,
|
||||
score: result.summary.taskSuccessRate,
|
||||
metrics: {
|
||||
task_success_rate: result.summary.taskSuccessRate,
|
||||
edit_success_rate: result.summary.editSuccessRate,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
function readSnapcompactSnapshot(jobDir: string): BenchmarkSnapshot {
|
||||
const recordsFile = path.join(jobDir, "records.jsonl");
|
||||
const summaryFile = path.join(jobDir, "summary.json");
|
||||
if (!fs.existsSync(recordsFile)) return emptySnapshot();
|
||||
const records = fs
|
||||
.readFileSync(recordsFile, "utf8")
|
||||
.split("\n")
|
||||
.filter(Boolean)
|
||||
.map(line => JSON.parse(line) as SnapRecord);
|
||||
const traces = records.map(
|
||||
(record, index): BenchmarkTrace => ({
|
||||
name: `${record.cond}__${record.chunk}__${index + 1}`,
|
||||
task: `${record.cond}:${record.chunk}`,
|
||||
status: record.f1 > 0 ? "pass" : "fail",
|
||||
reward: record.f1,
|
||||
costUsd: 0,
|
||||
durationMs: 0,
|
||||
detail: JSON.stringify({ question: record.q, answer: record.answer, golds: record.golds, em: record.em }),
|
||||
tracePath: `record:${index + 1}`,
|
||||
}),
|
||||
);
|
||||
let rows: SnapSummaryRow[] = [];
|
||||
if (fs.existsSync(summaryFile)) {
|
||||
const summary: SnapSummary = JSON.parse(fs.readFileSync(summaryFile, "utf8"));
|
||||
rows = summary.rows;
|
||||
}
|
||||
const samples = rows.reduce((sum, row) => sum + row.n, 0);
|
||||
const weightedF1 = rows.reduce((sum, row) => sum + row.f1 * row.n, 0);
|
||||
const weightedEm = rows.reduce((sum, row) => sum + row.em * row.n, 0);
|
||||
const pass = traces.filter(trace => trace.status === "pass").length;
|
||||
return {
|
||||
traces,
|
||||
total: traces.length,
|
||||
done: traces.length,
|
||||
pass,
|
||||
fail: traces.length - pass,
|
||||
error: 0,
|
||||
running: 0,
|
||||
costUsd: rows.reduce((sum, row) => sum + row.cost_usd, 0),
|
||||
tokIn: rows.reduce((sum, row) => sum + row.tokens_in, 0),
|
||||
tokOut: rows.reduce((sum, row) => sum + row.tokens_out, 0),
|
||||
tokCache: rows.reduce((sum, row) => sum + row.cache_w + row.cache_r, 0),
|
||||
score: samples > 0 ? weightedF1 / samples : null,
|
||||
metrics: {
|
||||
f1: samples > 0 ? weightedF1 / samples : null,
|
||||
exact_match: samples > 0 ? weightedEm / samples : null,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/** Read and normalize the latest artifacts for a benchmark run. */
|
||||
export function readBenchmarkSnapshot(benchmark: BenchmarkKind, jobDir: string): BenchmarkSnapshot {
|
||||
if (benchmark === "edit") return readEditSnapshot(jobDir);
|
||||
if (benchmark === "snapcompact") return readSnapcompactSnapshot(jobDir);
|
||||
const trials = readTrials(jobDir);
|
||||
const job = readJobResult(jobDir);
|
||||
const totals = aggregate(trials, job, job?.nTotal ?? trials.length);
|
||||
return {
|
||||
traces: trials.map(trial => ({
|
||||
name: trial.name,
|
||||
task: trial.name.replace(/__[^_]+$/, ""),
|
||||
status: trial.status,
|
||||
reward: trial.reward,
|
||||
costUsd: trial.costUsd,
|
||||
durationMs: trial.durationMs,
|
||||
detail: trial.detail,
|
||||
tracePath: path.join(trial.name, "agent", "omp.txt"),
|
||||
})),
|
||||
total: totals.total,
|
||||
done: totals.done,
|
||||
pass: totals.pass,
|
||||
fail: totals.fail,
|
||||
error: totals.error,
|
||||
running: totals.running,
|
||||
costUsd: totals.costUsd,
|
||||
tokIn: totals.tokIn,
|
||||
tokOut: totals.tokOut,
|
||||
tokCache: totals.tokCache,
|
||||
score: totals.done > 0 ? totals.pass / totals.done : null,
|
||||
metrics: { success_rate: totals.done > 0 ? totals.pass / totals.done : null },
|
||||
};
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { armOf, experimentOf, summarizeArm } from "./experiments";
|
||||
import type { RunRow, TrialRow } from "./store";
|
||||
import type { RunRow, TraceRow } from "./store";
|
||||
|
||||
/**
|
||||
* Contracts under test:
|
||||
@@ -11,11 +11,13 @@ import type { RunRow, TrialRow } from "./store";
|
||||
|
||||
function runRow(overrides: Partial<RunRow>): RunRow {
|
||||
return {
|
||||
benchmark: "harbor",
|
||||
jobName: "exp-arm",
|
||||
dataset: "d",
|
||||
agent: "omp",
|
||||
models: "anthropic/claude-opus-4-8",
|
||||
slide: null,
|
||||
config: {},
|
||||
role: "",
|
||||
note: "",
|
||||
status: "running",
|
||||
@@ -33,11 +35,13 @@ function runRow(overrides: Partial<RunRow>): RunRow {
|
||||
tokIn: 0,
|
||||
tokOut: 0,
|
||||
tokCache: 0,
|
||||
score: null,
|
||||
metrics: {},
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
function trialRow(overrides: Partial<TrialRow>): TrialRow {
|
||||
function traceRow(overrides: Partial<TraceRow>): TraceRow {
|
||||
return {
|
||||
jobName: "exp-arm",
|
||||
name: "task__x",
|
||||
@@ -48,6 +52,7 @@ function trialRow(overrides: Partial<TrialRow>): TrialRow {
|
||||
durationMs: 60_000,
|
||||
detail: "",
|
||||
updatedAt: Date.now(),
|
||||
tracePath: null,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
@@ -75,8 +80,8 @@ describe("summarizeArm", () => {
|
||||
costUsd: 5,
|
||||
}),
|
||||
[
|
||||
trialRow({ status: "pass", durationMs: 120_000 }),
|
||||
trialRow({ name: "b__x", task: "b", status: "fail", reward: 0, durationMs: 240_000 }),
|
||||
traceRow({ status: "pass", durationMs: 120_000 }),
|
||||
traceRow({ name: "b__x", task: "b", status: "fail", reward: 0, durationMs: 240_000 }),
|
||||
],
|
||||
);
|
||||
expect(running.arm).toBe("n8");
|
||||
@@ -94,7 +99,7 @@ describe("summarizeArm", () => {
|
||||
|
||||
const finished = summarizeArm(
|
||||
runRow({ jobName: "sb2-opus", status: "complete", nTotal: 20, done: 20, pass: 15, costUsd: 30 }),
|
||||
[trialRow({})],
|
||||
[traceRow({})],
|
||||
);
|
||||
expect(finished.projected).toBeNull();
|
||||
expect(finished.costPerTask).toBeCloseTo(1.5, 5);
|
||||
@@ -108,6 +113,6 @@ describe("summarizeArm", () => {
|
||||
}),
|
||||
[],
|
||||
);
|
||||
expect(arm.config).toBe("anthropic/claude-opus-4-8 → google/gemini-3.5-flash on first edit/write +plan");
|
||||
expect(arm.config).toBe("harbor · anthropic/claude-opus-4-8 → google/gemini-3.5-flash on first edit/write +plan");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
* `sb2-gemini` → experiment `sb2`) so comparable arms can be charted together,
|
||||
* with linear projections for arms still in flight.
|
||||
*/
|
||||
import type { RunRow, RunStore, TrialRow } from "./store";
|
||||
import type { RunRow, RunStore, TraceRow } from "./store";
|
||||
|
||||
/** Linear extrapolation of a running arm to its full task count. */
|
||||
export interface ArmProjection {
|
||||
@@ -79,8 +79,8 @@ function slideLabel(slideJson: string | null): string {
|
||||
}
|
||||
}
|
||||
|
||||
export function summarizeArm(run: RunRow, trials: TrialRow[]): ArmSummary {
|
||||
const decided = trials.filter(t => t.status === "pass" || t.status === "fail" || t.status === "error");
|
||||
export function summarizeArm(run: RunRow, traces: TraceRow[]): ArmSummary {
|
||||
const decided = traces.filter(t => t.status === "pass" || t.status === "fail" || t.status === "error");
|
||||
const durations = decided.filter(t => t.durationMs > 0).map(t => t.durationMs);
|
||||
const meanTrialMs = durations.length > 0 ? durations.reduce((a, b) => a + b, 0) / durations.length : null;
|
||||
const passPct = decided.length > 0 ? (100 * run.pass) / decided.length : null;
|
||||
@@ -102,7 +102,7 @@ export function summarizeArm(run: RunRow, trials: TrialRow[]): ArmSummary {
|
||||
return {
|
||||
run,
|
||||
arm: armOf(run.jobName),
|
||||
config: `${run.models}${slideLabel(run.slide)}`,
|
||||
config: `${run.benchmark} · ${run.models}${slideLabel(run.slide)}`,
|
||||
passPct,
|
||||
costPerTask,
|
||||
meanTrialMs,
|
||||
@@ -150,7 +150,7 @@ export function experimentDetail(store: RunStore, id: string): ExperimentDetail
|
||||
const matrix: ExperimentDetail["matrix"] = {};
|
||||
const tasks = new Set<string>();
|
||||
for (const run of runs) {
|
||||
const trials = store.listTrials(run.jobName);
|
||||
const trials = store.listTraces(run.jobName);
|
||||
arms.push(summarizeArm(run, trials));
|
||||
const cells: Record<string, { status: string; reward: number | null }> = {};
|
||||
for (const t of trials) {
|
||||
|
||||
@@ -104,13 +104,13 @@ describe("RunStore", () => {
|
||||
expect(run?.running).toBe(1);
|
||||
expect(run?.costUsd).toBeCloseTo(0.7, 5);
|
||||
|
||||
const trials = store.listTrials("job-a");
|
||||
expect(trials.map(t => [t.task, t.status])).toEqual([
|
||||
const traces = store.listTraces("job-a");
|
||||
expect(traces.map(t => [t.task, t.status])).toEqual([
|
||||
["alpha", "pass"],
|
||||
["beta", "error"],
|
||||
["gamma", "running"],
|
||||
]);
|
||||
expect(trials[1].detail).toBe("AgentTimeoutError");
|
||||
expect(traces[1].detail).toBe("AgentTimeoutError");
|
||||
|
||||
// re-discover is idempotent
|
||||
expect(store.discover()).toBe(0);
|
||||
@@ -162,6 +162,7 @@ describe("RunStore", () => {
|
||||
const store = new RunStore(jobsDir);
|
||||
cleanups.push(() => store.close());
|
||||
store.registerLaunch({
|
||||
benchmark: "harbor",
|
||||
jobName: "job-b",
|
||||
dataset: "test-dataset@1.0",
|
||||
agent: "omp",
|
||||
@@ -175,7 +176,7 @@ describe("RunStore", () => {
|
||||
});
|
||||
|
||||
describe("ManagerServer API", () => {
|
||||
it("serves runs, trials, transcripts, and validates launches", async () => {
|
||||
it("serves uniform runs, traces, and rejects invalid launches", async () => {
|
||||
const jobsDir = makeJobsDir();
|
||||
writeFixtureJob(jobsDir, "job-api");
|
||||
const manager = new ManagerServer(jobsDir);
|
||||
@@ -190,15 +191,15 @@ describe("ManagerServer API", () => {
|
||||
|
||||
const detailRes = await fetch(`${base}/api/runs/job-api`);
|
||||
expect(detailRes.status).toBe(200);
|
||||
const detail = (await detailRes.json()) as { run: { pass: number }; trials: Array<{ status: string }> };
|
||||
const detail = (await detailRes.json()) as { run: { pass: number }; traces: Array<{ status: string }> };
|
||||
expect(detail.run.pass).toBe(1);
|
||||
expect(detail.trials).toHaveLength(3);
|
||||
expect(detail.traces).toHaveLength(3);
|
||||
|
||||
const tr = await fetch(`${base}/api/runs/job-api/trials/alpha__abc/transcript?tail=10`);
|
||||
const tr = await fetch(`${base}/api/runs/job-api/traces/alpha__abc?tail=10`);
|
||||
expect(tr.status).toBe(200);
|
||||
const transcript = (await tr.json()) as { entries: Array<{ kind: string; tools?: string[] }> };
|
||||
expect(transcript.entries.map(e => e.kind)).toEqual(["assistant", "toolResult"]);
|
||||
expect(transcript.entries[0].tools).toEqual(["read"]);
|
||||
const trace = (await tr.json()) as { entries: Array<{ kind: string; tools?: string[] }> };
|
||||
expect(trace.entries.map(e => e.kind)).toEqual(["assistant", "toolResult"]);
|
||||
expect(trace.entries[0].tools).toEqual(["read"]);
|
||||
|
||||
const missing = await fetch(`${base}/api/runs/nope`);
|
||||
expect(missing.status).toBe(404);
|
||||
@@ -215,4 +216,80 @@ describe("ManagerServer API", () => {
|
||||
};
|
||||
expect(cancelUnknown.cancelled).toBe(false);
|
||||
});
|
||||
|
||||
it("serves edit and SnapCompact metrics and native traces through one API", async () => {
|
||||
const jobsDir = makeJobsDir();
|
||||
const manager = new ManagerServer(jobsDir);
|
||||
for (const benchmark of ["edit", "snapcompact"] as const) {
|
||||
const jobName = `${benchmark}-arm`;
|
||||
manager.store.registerLaunch({
|
||||
benchmark,
|
||||
jobName,
|
||||
dataset: benchmark === "edit" ? "typescript-edit" : "squad-dev",
|
||||
agent: benchmark,
|
||||
models: ["test/model"],
|
||||
pid: process.pid,
|
||||
});
|
||||
manager.store.markExit(jobName, 0);
|
||||
}
|
||||
const editDir = path.join(jobsDir, "edit-arm");
|
||||
fs.writeFileSync(
|
||||
path.join(editDir, "result.json"),
|
||||
JSON.stringify({
|
||||
tasks: [
|
||||
{
|
||||
id: "rename",
|
||||
name: "Rename",
|
||||
runs: [{ runIndex: 0, success: true, duration: 10, tokens: { input: 8, output: 2, reasoning: 0 } }],
|
||||
},
|
||||
],
|
||||
summary: {
|
||||
totalRuns: 1,
|
||||
successfulRuns: 1,
|
||||
taskSuccessRate: 1,
|
||||
editSuccessRate: 1,
|
||||
totalTokens: { input: 8, output: 2 },
|
||||
},
|
||||
}),
|
||||
);
|
||||
fs.mkdirSync(path.join(editDir, "result.dump", "rename"), { recursive: true });
|
||||
fs.writeFileSync(path.join(editDir, "result.dump", "rename", "run-1.md"), "# conversation\n\nassistant answer");
|
||||
const snapDir = path.join(jobsDir, "snapcompact-arm");
|
||||
fs.writeFileSync(
|
||||
path.join(snapDir, "records.jsonl"),
|
||||
`${JSON.stringify({ cond: "text", chunk: 0, pos_rel: 0, q: "question", answer: "answer", golds: ["gold"], em: 0, f1: 0.5 })}\n`,
|
||||
);
|
||||
fs.writeFileSync(
|
||||
path.join(snapDir, "summary.json"),
|
||||
JSON.stringify({
|
||||
rows: [{ n: 1, f1: 0.5, em: 0, cost_usd: 0.1, tokens_in: 10, tokens_out: 2, cache_w: 0, cache_r: 0 }],
|
||||
}),
|
||||
);
|
||||
manager.store.syncAll();
|
||||
const server = manager.start(0);
|
||||
cleanups.push(() => {
|
||||
void manager.stop();
|
||||
});
|
||||
const base = `http://localhost:${server.port}`;
|
||||
|
||||
const edit = (await (await fetch(`${base}/api/runs/edit-arm`)).json()) as {
|
||||
run: { benchmark: string; metrics: Record<string, number> };
|
||||
traces: Array<{ name: string }>;
|
||||
};
|
||||
expect(edit.run).toMatchObject({ benchmark: "edit", metrics: { task_success_rate: 1, edit_success_rate: 1 } });
|
||||
const editTrace = (await (
|
||||
await fetch(`${base}/api/runs/edit-arm/traces/${encodeURIComponent(edit.traces[0].name)}`)
|
||||
).json()) as { entries: Array<{ kind: string; text: string }> };
|
||||
expect(editTrace.entries).toEqual([{ kind: "conversation", text: "# conversation\n\nassistant answer" }]);
|
||||
|
||||
const snap = (await (await fetch(`${base}/api/runs/snapcompact-arm`)).json()) as {
|
||||
run: { benchmark: string; metrics: Record<string, number> };
|
||||
traces: Array<{ name: string }>;
|
||||
};
|
||||
expect(snap.run).toMatchObject({ benchmark: "snapcompact", metrics: { f1: 0.5, exact_match: 0 } });
|
||||
const snapTrace = (await (
|
||||
await fetch(`${base}/api/runs/snapcompact-arm/traces/${encodeURIComponent(snap.traces[0].name)}`)
|
||||
).json()) as { entries: Array<{ kind: string }> };
|
||||
expect(snapTrace.entries.map(entry => entry.kind)).toEqual(["question", "answer", "reference"]);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -11,9 +11,9 @@
|
||||
* process renders a live dashboard (progress / success% / spend / tokens / ETA)
|
||||
* by polling each trial's `result.json`. On completion it writes a markdown report.
|
||||
*
|
||||
* bun src/runner.ts --model anthropic/claude-sonnet-4-6 --tasks 20 --concurrency 4
|
||||
* bun src/runner.ts --agent oracle --tasks 2 # cheap pipeline smoke
|
||||
* bun src/runner.ts --help
|
||||
* harbor-manager harbor --model anthropic/claude-sonnet-4-6 --tasks 20 --concurrency 4
|
||||
* harbor-manager harbor --agent oracle --tasks 2 # cheap pipeline smoke
|
||||
* harbor-manager harbor --help
|
||||
*/
|
||||
import { spawnSync } from "node:child_process";
|
||||
import * as fs from "node:fs";
|
||||
@@ -108,7 +108,7 @@ function defaultConfig(): Config {
|
||||
|
||||
const HELP = `harbor-manager runner (local omp)
|
||||
|
||||
Usage: bun src/runner.ts [options] [-- <extra harbor args>]
|
||||
Usage: harbor-manager harbor [options] [-- <extra harbor args>]
|
||||
|
||||
Commands:
|
||||
cleanup Force-remove ALL leftover Harbor containers + networks, then exit
|
||||
|
||||
@@ -6,18 +6,20 @@
|
||||
* bun src/server.ts [--port 4700] [--jobs-dir <path>]
|
||||
*
|
||||
* API:
|
||||
* GET /api/experiments → experiment summaries across all benchmarks
|
||||
* GET /api/runs → RunRow[]
|
||||
* POST /api/runs → launch a run (JSON body, see LaunchRequest)
|
||||
* GET /api/runs/:name → { run, trials }
|
||||
* DELETE /api/runs/:name → cancel a manager-launched run
|
||||
* GET /api/runs/:name/trials/:trial/transcript?tail=N[&raw=1]
|
||||
* POST /api/runs → launch any benchmark
|
||||
* GET /api/runs/:name → { run, traces }
|
||||
* DELETE /api/runs/:name → cancel a managed run
|
||||
* GET /api/runs/:name/traces/:trace → normalized trace
|
||||
* GET /api/events → SSE: run-list snapshots on change
|
||||
*/
|
||||
import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import type { Server, Subprocess } from "bun";
|
||||
import { BENCHMARK_DEFINITIONS } from "./benchmarks";
|
||||
import { buildExperiments, experimentDetail, experimentOf } from "./experiments";
|
||||
import { type RunRole, RunStore } from "./store";
|
||||
import { type BenchmarkKind, type RunRole, RunStore } from "./store";
|
||||
|
||||
/** PUT /api/experiments/:id body — goal and per-run role/note metadata. */
|
||||
export interface ExperimentMetaUpdate {
|
||||
@@ -33,6 +35,8 @@ const DEFAULT_JOBS_DIR = path.join(REPO_ROOT, "runs", "harbor");
|
||||
|
||||
/** POST /api/runs body. Mirrors the runner CLI surface we actually use. */
|
||||
export interface LaunchRequest {
|
||||
/** Benchmark adapter to execute. */
|
||||
benchmark?: BenchmarkKind;
|
||||
model: string;
|
||||
dataset?: string;
|
||||
/** Task count for a dataset sample, or omit when `include` is given. */
|
||||
@@ -40,12 +44,14 @@ export interface LaunchRequest {
|
||||
/** Explicit task names (passed as repeated --include). */
|
||||
include?: string[];
|
||||
concurrency?: number;
|
||||
/** SnapCompact conditions; ignored by other benchmarks. */
|
||||
conditions?: string[];
|
||||
timeoutMultiplier?: number;
|
||||
attempts?: number;
|
||||
agent?: string;
|
||||
jobName?: string;
|
||||
webSearch?: boolean;
|
||||
slide?: { model: string; turns?: number; onAction?: boolean; plan?: boolean };
|
||||
slide?: { model: string; turns?: number; onAction?: boolean; plan?: boolean; checklist?: boolean };
|
||||
/** Role of this run inside its experiment (baseline vs treatment). */
|
||||
role?: RunRole;
|
||||
/** One-line description of what this arm tests. */
|
||||
@@ -180,6 +186,9 @@ export class ManagerServer {
|
||||
});
|
||||
}
|
||||
if (p === "/api/events") return this.#sseResponse();
|
||||
if (p === "/api/benchmarks" && request.method === "GET") {
|
||||
return Response.json(BENCHMARK_DEFINITIONS);
|
||||
}
|
||||
if (p === "/api/experiments" && request.method === "GET") {
|
||||
return Response.json(buildExperiments(this.#store));
|
||||
}
|
||||
@@ -207,15 +216,15 @@ export class ManagerServer {
|
||||
if (request.method === "DELETE") return Response.json(this.cancel(jobName));
|
||||
const run = this.#store.syncRun(jobName);
|
||||
if (!run) return Response.json({ error: "run not found" }, { status: 404 });
|
||||
return Response.json({ run, trials: this.#store.listTrials(jobName) });
|
||||
return Response.json({ run, traces: this.#store.listTraces(jobName) });
|
||||
}
|
||||
const trialMatch = p.match(/^\/api\/runs\/([^/]+)\/trials\/([^/]+)\/transcript$/);
|
||||
if (trialMatch) {
|
||||
const jobName = decodeURIComponent(trialMatch[1]);
|
||||
const trial = decodeURIComponent(trialMatch[2]);
|
||||
const traceMatch = p.match(/^\/api\/runs\/([^/]+)\/traces\/([^/]+)$/);
|
||||
if (traceMatch) {
|
||||
const jobName = decodeURIComponent(traceMatch[1]);
|
||||
const trace = decodeURIComponent(traceMatch[2]);
|
||||
const tail = Number(url.searchParams.get("tail") ?? "120");
|
||||
const raw = url.searchParams.get("raw") === "1";
|
||||
return this.#transcript(jobName, trial, tail, raw);
|
||||
return this.#trace(jobName, trace, tail, raw);
|
||||
}
|
||||
return Response.json({ error: "not found" }, { status: 404 });
|
||||
} catch (err) {
|
||||
@@ -248,43 +257,78 @@ export class ManagerServer {
|
||||
});
|
||||
}
|
||||
|
||||
/** Spawn the CLI runner for `request` and register the run. */
|
||||
/** Launch any supported benchmark and register it in the uniform run store. */
|
||||
launch(request: LaunchRequest): { jobName: string; pid: number } {
|
||||
if (!request.model) throw new Error("model is required");
|
||||
const dataset = request.dataset ?? "terminal-bench@2.0";
|
||||
const benchmark = request.benchmark ?? "harbor";
|
||||
if (benchmark !== "harbor" && benchmark !== "edit" && benchmark !== "snapcompact") {
|
||||
throw new Error(`unsupported benchmark: ${benchmark}`);
|
||||
}
|
||||
const dataset =
|
||||
request.dataset ??
|
||||
(benchmark === "harbor" ? "terminal-bench@2.0" : benchmark === "edit" ? "typescript-edit" : "squad-dev");
|
||||
const stamp = new Date().toISOString().replace(/[:.]/g, "-").slice(0, 19);
|
||||
const modelSlug = request.model.replace(/[^a-zA-Z0-9]+/g, "-");
|
||||
const jobName = request.jobName ?? `${modelSlug}-${stamp}`;
|
||||
if (this.#children.has(jobName) || this.#store.getRun(jobName)?.status === "running") {
|
||||
throw new Error(`run ${jobName} is already running`);
|
||||
}
|
||||
const jobDir = path.join(this.jobsDir, jobName);
|
||||
fs.mkdirSync(jobDir, { recursive: true });
|
||||
|
||||
const argv = ["bun", "src/runner.ts", "--model", request.model, "-d", dataset, "--job-name", jobName];
|
||||
if (request.agent) argv.push("--agent", request.agent);
|
||||
if (request.tasks !== undefined) argv.push("--tasks", String(request.tasks));
|
||||
if (request.concurrency !== undefined) argv.push("--concurrency", String(request.concurrency));
|
||||
if (request.attempts !== undefined) argv.push("--attempts", String(request.attempts));
|
||||
if (request.timeoutMultiplier !== undefined) argv.push("--timeout-multiplier", String(request.timeoutMultiplier));
|
||||
if (request.webSearch) argv.push("--web-search");
|
||||
for (const task of request.include ?? []) argv.push("--include", task);
|
||||
if (request.slide) {
|
||||
argv.push("--agent-arg", "--reasoning-slide-model", "--agent-arg", request.slide.model);
|
||||
if (request.slide.onAction) argv.push("--agent-arg", "--reasoning-slide-on-action");
|
||||
else if (request.slide.turns !== undefined) {
|
||||
argv.push("--agent-arg", "--reasoning-slide-turns", "--agent-arg", String(request.slide.turns));
|
||||
} else throw new Error("slide requires turns or onAction");
|
||||
if (request.slide.plan) argv.push("--agent-arg", "--reasoning-slide-plan");
|
||||
// The runner only auto-routes gateway auth for the primary model's
|
||||
// provider; declare the slide model's provider explicitly so its
|
||||
// requests reach the gateway too.
|
||||
const slideProvider = request.slide.model.split("/", 1)[0];
|
||||
if (slideProvider) argv.push("--providers", slideProvider);
|
||||
}
|
||||
// Default to source mode (repo bind-mount, no rebuild); prebuilt binaries only on request.
|
||||
if (request.prebuiltBinaries) {
|
||||
for (const name of ["omp-linux-arm64", "omp-linux-x64"]) {
|
||||
const binary = path.join(REPO_ROOT, "packages", "coding-agent", "dist", name);
|
||||
if (fs.existsSync(binary)) argv.push("--binary", binary);
|
||||
let argv: string[];
|
||||
let cwd: string;
|
||||
if (benchmark === "edit") {
|
||||
cwd = PKG_DIR;
|
||||
argv = ["bun", "adapters/edit/cli.ts", "--model", request.model, "--output", path.join(jobDir, "result.json")];
|
||||
if (request.tasks !== undefined) argv.push("--max-tasks", String(request.tasks));
|
||||
if (request.include?.length) argv.push("--tasks", request.include.join(","));
|
||||
if (request.concurrency !== undefined) argv.push("--task-concurrency", String(request.concurrency));
|
||||
if (request.attempts !== undefined) argv.push("--runs", String(request.attempts));
|
||||
} else if (benchmark === "snapcompact") {
|
||||
cwd = PKG_DIR;
|
||||
argv = ["uv", "run", "src/adapters/snapcompact.py", "--model", request.model, "--output-dir", jobDir];
|
||||
if (request.tasks !== undefined) argv.push("--limit-paras", String(request.tasks));
|
||||
if (request.concurrency !== undefined) argv.push("--workers", String(request.concurrency));
|
||||
if (request.conditions?.length) argv.push("--conditions", request.conditions.join(","));
|
||||
} else {
|
||||
cwd = PKG_DIR;
|
||||
argv = [
|
||||
"bun",
|
||||
"src/runner.ts",
|
||||
"--model",
|
||||
request.model,
|
||||
"-d",
|
||||
dataset,
|
||||
"--job-name",
|
||||
jobName,
|
||||
"--jobs-dir",
|
||||
this.jobsDir,
|
||||
];
|
||||
if (request.agent) argv.push("--agent", request.agent);
|
||||
if (request.tasks !== undefined) argv.push("--tasks", String(request.tasks));
|
||||
if (request.concurrency !== undefined) argv.push("--concurrency", String(request.concurrency));
|
||||
if (request.attempts !== undefined) argv.push("--attempts", String(request.attempts));
|
||||
if (request.timeoutMultiplier !== undefined)
|
||||
argv.push("--timeout-multiplier", String(request.timeoutMultiplier));
|
||||
if (request.webSearch) argv.push("--web-search");
|
||||
for (const task of request.include ?? []) argv.push("--include", task);
|
||||
if (request.slide) {
|
||||
argv.push("--agent-arg", "--reasoning-slide-model", "--agent-arg", request.slide.model);
|
||||
if (request.slide.onAction) argv.push("--agent-arg", "--reasoning-slide-on-action");
|
||||
else if (request.slide.turns !== undefined) {
|
||||
argv.push("--agent-arg", "--reasoning-slide-turns", "--agent-arg", String(request.slide.turns));
|
||||
} else throw new Error("slide requires turns or onAction");
|
||||
if (request.slide.plan) argv.push("--agent-arg", "--reasoning-slide-plan");
|
||||
if (request.slide.checklist) argv.push("--agent-arg", "--reasoning-slide-checklist");
|
||||
const slideProvider = request.slide.model.split("/", 1)[0];
|
||||
if (slideProvider) argv.push("--providers", slideProvider);
|
||||
}
|
||||
if (request.prebuiltBinaries) {
|
||||
for (const name of ["omp-linux-arm64", "omp-linux-x64"]) {
|
||||
const binary = path.join(REPO_ROOT, "packages", "coding-agent", "dist", name);
|
||||
if (fs.existsSync(binary)) argv.push("--binary", binary);
|
||||
}
|
||||
}
|
||||
}
|
||||
argv.push(...(request.extraArgs ?? []));
|
||||
@@ -293,7 +337,7 @@ export class ManagerServer {
|
||||
fs.mkdirSync(logDir, { recursive: true });
|
||||
const logFile = fs.openSync(path.join(logDir, `${jobName}.log`), "w");
|
||||
const proc = Bun.spawn(argv, {
|
||||
cwd: PKG_DIR,
|
||||
cwd,
|
||||
stdout: logFile,
|
||||
stderr: logFile,
|
||||
env: { ...process.env },
|
||||
@@ -309,11 +353,13 @@ export class ManagerServer {
|
||||
this.#tick();
|
||||
});
|
||||
this.#store.registerLaunch({
|
||||
benchmark,
|
||||
jobName,
|
||||
dataset,
|
||||
agent: request.agent ?? "omp",
|
||||
models: [request.model],
|
||||
slide: request.slide,
|
||||
config: { ...request },
|
||||
pid: proc.pid,
|
||||
role: request.role,
|
||||
note: request.note,
|
||||
@@ -354,12 +400,44 @@ export class ManagerServer {
|
||||
return { jobName, cancelled: false };
|
||||
}
|
||||
|
||||
/** Compact transcript view of a trial's omp.txt session JSONL. */
|
||||
#transcript(jobName: string, trial: string, tail: number, raw: boolean): Response {
|
||||
const file = path.join(this.jobsDir, jobName, trial, "agent", "omp.txt");
|
||||
if (!fs.existsSync(file)) return Response.json({ error: "transcript not found" }, { status: 404 });
|
||||
const lines = fs.readFileSync(file, "utf8").split("\n").filter(Boolean);
|
||||
/** Return a normalized trace regardless of the benchmark's native artifact format. */
|
||||
#trace(jobName: string, traceName: string, tail: number, raw: boolean): Response {
|
||||
const trace = this.#store.listTraces(jobName).find(item => item.name === traceName);
|
||||
if (!trace?.tracePath) return Response.json({ error: "trace not found" }, { status: 404 });
|
||||
const jobDir = path.join(this.jobsDir, jobName);
|
||||
const n = Number.isSafeInteger(tail) && tail > 0 ? Math.min(tail, 2000) : 120;
|
||||
if (trace.tracePath.startsWith("record:")) {
|
||||
const lineNumber = Number(trace.tracePath.slice("record:".length));
|
||||
const line = fs.readFileSync(path.join(jobDir, "records.jsonl"), "utf8").split("\n")[lineNumber - 1];
|
||||
if (!line) return Response.json({ error: "trace not found" }, { status: 404 });
|
||||
if (raw) return new Response(line, { headers: { "content-type": "application/json" } });
|
||||
const record = JSON.parse(line) as Record<string, unknown>;
|
||||
return Response.json({
|
||||
jobName,
|
||||
trace: traceName,
|
||||
entries: [
|
||||
{ kind: "question", text: String(record.q ?? "") },
|
||||
{ kind: "answer", model: this.#store.getRun(jobName)?.models ?? "", text: String(record.answer ?? "") },
|
||||
{ kind: "reference", text: JSON.stringify(record.golds ?? []) },
|
||||
],
|
||||
totalEvents: 3,
|
||||
});
|
||||
}
|
||||
const file = path.resolve(jobDir, trace.tracePath);
|
||||
if (!file.startsWith(`${path.resolve(jobDir)}${path.sep}`) || !fs.existsSync(file)) {
|
||||
return Response.json({ error: "trace not found" }, { status: 404 });
|
||||
}
|
||||
const text = fs.readFileSync(file, "utf8");
|
||||
if (!file.endsWith(".txt")) {
|
||||
if (raw) return new Response(text, { headers: { "content-type": "text/plain; charset=utf-8" } });
|
||||
return Response.json({
|
||||
jobName,
|
||||
trace: traceName,
|
||||
entries: [{ kind: "conversation", text }],
|
||||
totalEvents: 1,
|
||||
});
|
||||
}
|
||||
const lines = text.split("\n").filter(Boolean);
|
||||
if (raw) {
|
||||
return new Response(lines.slice(-n).join("\n"), {
|
||||
headers: { "content-type": "application/x-ndjson" },
|
||||
@@ -373,41 +451,30 @@ export class ManagerServer {
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
const type = event.type;
|
||||
if (type === "message_end") {
|
||||
if (event.type === "message_end") {
|
||||
const message = event.message as Record<string, unknown> | undefined;
|
||||
if (!message) continue;
|
||||
const role = message.role;
|
||||
if (role === "assistant") {
|
||||
const content = Array.isArray(message.content)
|
||||
? (message.content as Array<Record<string, unknown>>)
|
||||
: [];
|
||||
const text = content
|
||||
.filter(block => block.type === "text")
|
||||
.map(block => String(block.text ?? ""))
|
||||
.join("\n");
|
||||
const content = Array.isArray(message.content) ? (message.content as Array<Record<string, unknown>>) : [];
|
||||
const body = content
|
||||
.filter(block => block.type === "text")
|
||||
.map(block => String(block.text ?? ""))
|
||||
.join("\n");
|
||||
if (message.role === "assistant") {
|
||||
const tools = content.filter(block => block.type === "toolCall").map(block => String(block.name ?? "?"));
|
||||
entries.push({ kind: "assistant", model: message.model ?? "", text, tools });
|
||||
} else if (role === "toolResult") {
|
||||
const content = Array.isArray(message.content)
|
||||
? (message.content as Array<Record<string, unknown>>)
|
||||
: [];
|
||||
const text = content
|
||||
.filter(block => block.type === "text")
|
||||
.map(block => String(block.text ?? ""))
|
||||
.join("\n");
|
||||
entries.push({ kind: "assistant", model: message.model ?? "", text: body, tools });
|
||||
} else if (message.role === "toolResult") {
|
||||
entries.push({
|
||||
kind: "toolResult",
|
||||
tool: message.toolName ?? "?",
|
||||
isError: message.isError === true,
|
||||
text: text.length > 1600 ? `${text.slice(0, 1600)}…` : text,
|
||||
text: body.length > 1600 ? `${body.slice(0, 1600)}…` : body,
|
||||
});
|
||||
}
|
||||
} else if (type === "notice") {
|
||||
} else if (event.type === "notice") {
|
||||
entries.push({ kind: "notice", text: event.message ?? "" });
|
||||
}
|
||||
}
|
||||
return Response.json({ jobName, trial, entries: entries.slice(-n), totalEvents: lines.length });
|
||||
return Response.json({ jobName, trace: traceName, entries: entries.slice(-n), totalEvents: lines.length });
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -10,19 +10,26 @@
|
||||
import { Database } from "bun:sqlite";
|
||||
import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import { aggregate, readJobResult, readTrials } from "./runner";
|
||||
import { readBenchmarkSnapshot } from "./benchmarks";
|
||||
import { readJobResult } from "./runner";
|
||||
|
||||
export type RunStatus = "running" | "complete" | "failed" | "cancelled";
|
||||
|
||||
/** Benchmark implementation that produced a run. */
|
||||
export type BenchmarkKind = "harbor" | "edit" | "snapcompact";
|
||||
|
||||
/** How a run relates to its experiment's question. */
|
||||
export type RunRole = "baseline" | "variant" | "";
|
||||
|
||||
export interface RunRow {
|
||||
benchmark: BenchmarkKind;
|
||||
jobName: string;
|
||||
dataset: string;
|
||||
agent: string;
|
||||
models: string;
|
||||
slide: string | null;
|
||||
/** Benchmark-specific launch configuration. */
|
||||
config: Record<string, unknown>;
|
||||
/** Role inside the experiment (baseline vs treatment); "" when unspecified. */
|
||||
role: RunRole;
|
||||
/** One-line description of what this arm tests (e.g. "slide→flash after 8 turns"). */
|
||||
@@ -42,9 +49,13 @@ export interface RunRow {
|
||||
tokIn: number;
|
||||
tokOut: number;
|
||||
tokCache: number;
|
||||
/** Benchmark-native aggregate score, when the benchmark exposes one. */
|
||||
score: number | null;
|
||||
/** Values keyed by the adapter's metric definitions. */
|
||||
metrics: Record<string, number | null>;
|
||||
}
|
||||
|
||||
export interface TrialRow {
|
||||
export interface TraceRow {
|
||||
jobName: string;
|
||||
name: string;
|
||||
task: string;
|
||||
@@ -54,9 +65,12 @@ export interface TrialRow {
|
||||
durationMs: number;
|
||||
detail: string;
|
||||
updatedAt: number;
|
||||
/** Adapter-owned locator used by the uniform trace endpoint. */
|
||||
tracePath: string | null;
|
||||
}
|
||||
|
||||
export interface LaunchRecord {
|
||||
benchmark: BenchmarkKind;
|
||||
jobName: string;
|
||||
dataset: string;
|
||||
agent: string;
|
||||
@@ -65,17 +79,20 @@ export interface LaunchRecord {
|
||||
pid: number;
|
||||
role?: RunRole;
|
||||
note?: string;
|
||||
config?: Record<string, unknown>;
|
||||
}
|
||||
|
||||
const SCHEMA = `
|
||||
CREATE TABLE IF NOT EXISTS runs (
|
||||
job_name TEXT PRIMARY KEY,
|
||||
benchmark TEXT NOT NULL DEFAULT 'harbor',
|
||||
dataset TEXT NOT NULL DEFAULT '',
|
||||
agent TEXT NOT NULL DEFAULT 'omp',
|
||||
models TEXT NOT NULL DEFAULT '',
|
||||
slide TEXT,
|
||||
role TEXT NOT NULL DEFAULT '',
|
||||
note TEXT NOT NULL DEFAULT '',
|
||||
config_json TEXT NOT NULL DEFAULT '{}',
|
||||
status TEXT NOT NULL DEFAULT 'running',
|
||||
pid INTEGER,
|
||||
exit_code INTEGER,
|
||||
@@ -90,6 +107,8 @@ CREATE TABLE IF NOT EXISTS runs (
|
||||
cost_usd REAL NOT NULL DEFAULT 0,
|
||||
tok_in INTEGER NOT NULL DEFAULT 0,
|
||||
tok_out INTEGER NOT NULL DEFAULT 0,
|
||||
score REAL,
|
||||
metrics_json TEXT NOT NULL DEFAULT '{}',
|
||||
tok_cache INTEGER NOT NULL DEFAULT 0
|
||||
);
|
||||
CREATE TABLE IF NOT EXISTS trials (
|
||||
@@ -101,6 +120,7 @@ CREATE TABLE IF NOT EXISTS trials (
|
||||
cost_usd REAL NOT NULL DEFAULT 0,
|
||||
duration_ms INTEGER NOT NULL DEFAULT 0,
|
||||
detail TEXT NOT NULL DEFAULT '',
|
||||
trace_path TEXT,
|
||||
updated_at INTEGER NOT NULL,
|
||||
PRIMARY KEY (job_name, name)
|
||||
);
|
||||
@@ -125,12 +145,25 @@ export class RunStore {
|
||||
this.#db = new Database(dbPath ?? path.join(jobsDir, "_manager", "harbor-manager.sqlite"));
|
||||
this.#db.run("PRAGMA journal_mode = WAL");
|
||||
this.#db.run(SCHEMA);
|
||||
// Migration for stores created before run roles/notes existed.
|
||||
const columns = new Set(
|
||||
const runColumns = new Set(
|
||||
(this.#db.query("PRAGMA table_info(runs)").all() as Array<{ name: string }>).map(c => c.name),
|
||||
);
|
||||
if (!columns.has("role")) this.#db.run("ALTER TABLE runs ADD COLUMN role TEXT NOT NULL DEFAULT ''");
|
||||
if (!columns.has("note")) this.#db.run("ALTER TABLE runs ADD COLUMN note TEXT NOT NULL DEFAULT ''");
|
||||
if (!runColumns.has("role")) this.#db.run("ALTER TABLE runs ADD COLUMN role TEXT NOT NULL DEFAULT ''");
|
||||
if (!runColumns.has("note")) this.#db.run("ALTER TABLE runs ADD COLUMN note TEXT NOT NULL DEFAULT ''");
|
||||
if (!runColumns.has("benchmark")) {
|
||||
this.#db.run("ALTER TABLE runs ADD COLUMN benchmark TEXT NOT NULL DEFAULT 'harbor'");
|
||||
}
|
||||
if (!runColumns.has("config_json")) {
|
||||
this.#db.run("ALTER TABLE runs ADD COLUMN config_json TEXT NOT NULL DEFAULT '{}'");
|
||||
}
|
||||
if (!runColumns.has("score")) this.#db.run("ALTER TABLE runs ADD COLUMN score REAL");
|
||||
if (!runColumns.has("metrics_json")) {
|
||||
this.#db.run("ALTER TABLE runs ADD COLUMN metrics_json TEXT NOT NULL DEFAULT '{}'");
|
||||
}
|
||||
const traceColumns = new Set(
|
||||
(this.#db.query("PRAGMA table_info(trials)").all() as Array<{ name: string }>).map(c => c.name),
|
||||
);
|
||||
if (!traceColumns.has("trace_path")) this.#db.run("ALTER TABLE trials ADD COLUMN trace_path TEXT");
|
||||
}
|
||||
|
||||
close(): void {
|
||||
@@ -139,26 +172,34 @@ export class RunStore {
|
||||
|
||||
/** Register a run this manager just launched (pid-owning). */
|
||||
registerLaunch(launch: LaunchRecord): void {
|
||||
this.#db.query("DELETE FROM trials WHERE job_name = ?").run(launch.jobName);
|
||||
this.#db
|
||||
.query(
|
||||
`INSERT INTO runs (job_name, dataset, agent, models, slide, role, note, status, pid, created_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, 'running', ?, ?)
|
||||
`INSERT INTO runs
|
||||
(job_name, benchmark, dataset, agent, models, slide, role, note, config_json, status, pid, created_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'running', ?, ?)
|
||||
ON CONFLICT(job_name) DO UPDATE SET
|
||||
pid = excluded.pid, status = 'running',
|
||||
benchmark = excluded.benchmark, pid = excluded.pid, status = 'running',
|
||||
config_json = excluded.config_json,
|
||||
role = CASE WHEN excluded.role != '' THEN excluded.role ELSE runs.role END,
|
||||
note = CASE WHEN excluded.note != '' THEN excluded.note ELSE runs.note END`,
|
||||
)
|
||||
.run(
|
||||
launch.jobName,
|
||||
launch.benchmark,
|
||||
launch.dataset,
|
||||
launch.agent,
|
||||
launch.models.join(","),
|
||||
launch.slide ? JSON.stringify(launch.slide) : null,
|
||||
launch.role ?? "",
|
||||
launch.note ?? "",
|
||||
JSON.stringify(launch.config ?? {}),
|
||||
launch.pid,
|
||||
Date.now(),
|
||||
);
|
||||
const jobDir = path.join(this.jobsDir, launch.jobName);
|
||||
fs.mkdirSync(jobDir, { recursive: true });
|
||||
fs.writeFileSync(path.join(jobDir, "manager.json"), JSON.stringify(launch, null, 2));
|
||||
}
|
||||
|
||||
/** Upsert the experiment's stated goal. */
|
||||
@@ -230,65 +271,68 @@ export class RunStore {
|
||||
syncRun(jobName: string): RunRow | null {
|
||||
const jobDir = path.join(this.jobsDir, jobName);
|
||||
if (!fs.existsSync(jobDir)) return this.getRun(jobName);
|
||||
const trials = readTrials(jobDir);
|
||||
const job = readJobResult(jobDir);
|
||||
const totals = aggregate(trials, job, job?.nTotal ?? trials.length);
|
||||
const row = this.getRun(jobName);
|
||||
if (!row) return null;
|
||||
const snapshot = readBenchmarkSnapshot(row.benchmark, jobDir);
|
||||
const now = Date.now();
|
||||
const upsert = this.#db.query(
|
||||
`INSERT INTO trials (job_name, name, task, status, reward, cost_usd, duration_ms, detail, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
`INSERT INTO trials
|
||||
(job_name, name, task, status, reward, cost_usd, duration_ms, detail, trace_path, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT(job_name, name) DO UPDATE SET
|
||||
status = excluded.status, reward = excluded.reward, cost_usd = excluded.cost_usd,
|
||||
duration_ms = excluded.duration_ms, detail = excluded.detail, updated_at = excluded.updated_at`,
|
||||
duration_ms = excluded.duration_ms, detail = excluded.detail,
|
||||
trace_path = excluded.trace_path, updated_at = excluded.updated_at`,
|
||||
);
|
||||
const tx = this.#db.transaction(() => {
|
||||
for (const t of trials) {
|
||||
for (const trace of snapshot.traces) {
|
||||
upsert.run(
|
||||
jobName,
|
||||
t.name,
|
||||
t.name.replace(/__[^_]+$/, ""),
|
||||
t.status,
|
||||
t.reward,
|
||||
t.costUsd,
|
||||
t.durationMs,
|
||||
t.detail,
|
||||
trace.name,
|
||||
trace.task,
|
||||
trace.status,
|
||||
trace.reward,
|
||||
trace.costUsd,
|
||||
trace.durationMs,
|
||||
trace.detail,
|
||||
trace.tracePath,
|
||||
now,
|
||||
);
|
||||
}
|
||||
this.#db
|
||||
.query(
|
||||
`UPDATE runs SET n_total = ?, done = ?, pass = ?, fail = ?, error = ?, running = ?,
|
||||
cost_usd = ?, tok_in = ?, tok_out = ?, tok_cache = ? WHERE job_name = ?`,
|
||||
cost_usd = ?, tok_in = ?, tok_out = ?, tok_cache = ?, score = ?, metrics_json = ?
|
||||
WHERE job_name = ?`,
|
||||
)
|
||||
.run(
|
||||
totals.total,
|
||||
totals.done,
|
||||
totals.pass,
|
||||
totals.fail,
|
||||
totals.error,
|
||||
totals.running,
|
||||
totals.costUsd,
|
||||
totals.tokIn,
|
||||
totals.tokOut,
|
||||
totals.tokCache,
|
||||
snapshot.total,
|
||||
snapshot.done,
|
||||
snapshot.pass,
|
||||
snapshot.fail,
|
||||
snapshot.error,
|
||||
snapshot.running,
|
||||
snapshot.costUsd,
|
||||
snapshot.tokIn,
|
||||
snapshot.tokOut,
|
||||
snapshot.tokCache,
|
||||
snapshot.score,
|
||||
JSON.stringify(snapshot.metrics),
|
||||
jobName,
|
||||
);
|
||||
// Foreign runs (no owning pid and never finalized by markExit) infer
|
||||
// their lifecycle from Harbor's job-level result: `finished_at` is
|
||||
// terminal; a missing terminal marker with a fresh job dir means still
|
||||
// running; stale for >30 min means the harness died mid-run.
|
||||
const row = this.getRun(jobName);
|
||||
if (row && row.pid === null && row.finishedAt === null && row.status !== "cancelled") {
|
||||
const job2 = readJobResult(jobDir);
|
||||
// Historical Harbor runs have no owning process. Infer their terminal
|
||||
// state from result metadata or directory freshness.
|
||||
if (row.benchmark === "harbor" && row.pid === null && row.finishedAt === null && row.status !== "cancelled") {
|
||||
const result = readJobResult(jobDir);
|
||||
let status: RunStatus;
|
||||
let finishedAt: number | null = null;
|
||||
if (job2?.finishedAt != null) {
|
||||
if (result?.finishedAt != null) {
|
||||
status = "complete";
|
||||
finishedAt = job2.finishedAt;
|
||||
finishedAt = result.finishedAt;
|
||||
} else if (jobDirFresh(jobDir)) {
|
||||
status = "running";
|
||||
} else {
|
||||
status = totals.done > 0 && totals.done >= totals.total ? "complete" : "failed";
|
||||
status = snapshot.done > 0 && snapshot.done >= snapshot.total ? "complete" : "failed";
|
||||
finishedAt = jobDirMtime(jobDir);
|
||||
}
|
||||
if (status !== row.status) {
|
||||
@@ -343,7 +387,7 @@ export class RunStore {
|
||||
return rows.map(rowToRun);
|
||||
}
|
||||
|
||||
listTrials(jobName: string): TrialRow[] {
|
||||
listTraces(jobName: string): TraceRow[] {
|
||||
const rows = this.#db.query("SELECT * FROM trials WHERE job_name = ? ORDER BY name").all(jobName) as Array<
|
||||
Record<string, unknown>
|
||||
>;
|
||||
@@ -357,17 +401,20 @@ export class RunStore {
|
||||
durationMs: Number(r.duration_ms),
|
||||
detail: String(r.detail),
|
||||
updatedAt: Number(r.updated_at),
|
||||
tracePath: r.trace_path === null ? null : String(r.trace_path),
|
||||
}));
|
||||
}
|
||||
}
|
||||
|
||||
function rowToRun(r: Record<string, unknown>): RunRow {
|
||||
return {
|
||||
benchmark: String(r.benchmark ?? "harbor") as BenchmarkKind,
|
||||
jobName: String(r.job_name),
|
||||
dataset: String(r.dataset),
|
||||
agent: String(r.agent),
|
||||
models: String(r.models),
|
||||
slide: r.slide === null ? null : String(r.slide),
|
||||
config: JSON.parse(String(r.config_json ?? "{}")),
|
||||
role: String(r.role ?? "") as RunRole,
|
||||
note: String(r.note ?? ""),
|
||||
status: String(r.status) as RunStatus,
|
||||
@@ -385,6 +432,8 @@ function rowToRun(r: Record<string, unknown>): RunRow {
|
||||
tokIn: Number(r.tok_in),
|
||||
tokOut: Number(r.tok_out),
|
||||
tokCache: Number(r.tok_cache),
|
||||
score: r.score === null ? null : Number(r.score),
|
||||
metrics: JSON.parse(String(r.metrics_json ?? "{}")),
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
* #/exp/<id> experiment detail — arm table, dithered comparison charts
|
||||
* (projected values for in-flight arms, dimmed), task matrix
|
||||
* #/runs flat run list (legacy view)
|
||||
* #/runs/<name> run detail — trial grid + live transcript tail
|
||||
* #/runs/<name> run detail — normalized trace grid + live trace viewer
|
||||
*/
|
||||
import { useCallback, useEffect, useRef, useState } from "react";
|
||||
import { createRoot } from "react-dom/client";
|
||||
@@ -14,11 +14,13 @@ import { createRoot } from "react-dom/client";
|
||||
// ── api types (mirrors server modules) ──────────────────────────────────────
|
||||
|
||||
interface RunRow {
|
||||
benchmark: "harbor" | "edit" | "snapcompact";
|
||||
jobName: string;
|
||||
dataset: string;
|
||||
agent: string;
|
||||
models: string;
|
||||
slide: string | null;
|
||||
config: Record<string, unknown>;
|
||||
role: "baseline" | "variant" | "";
|
||||
note: string;
|
||||
status: "running" | "complete" | "failed" | "cancelled";
|
||||
@@ -32,9 +34,11 @@ interface RunRow {
|
||||
error: number;
|
||||
running: number;
|
||||
costUsd: number;
|
||||
score: number | null;
|
||||
metrics: Record<string, number | null>;
|
||||
}
|
||||
|
||||
interface TrialRow {
|
||||
interface TraceRow {
|
||||
name: string;
|
||||
task: string;
|
||||
status: string;
|
||||
@@ -193,7 +197,7 @@ function ExperimentsIndex() {
|
||||
<a
|
||||
key={exp.id}
|
||||
href={`#/exp/${encodeURIComponent(exp.id)}`}
|
||||
className="flex items-center gap-6 rounded-lg border border-zinc-800 bg-zinc-900/60 px-5 py-4 hover:border-zinc-600"
|
||||
className="flex min-w-0 items-center gap-6 overflow-hidden rounded-lg border border-zinc-800 bg-zinc-900/60 px-5 py-4 hover:border-zinc-600"
|
||||
>
|
||||
<div className="w-40 shrink-0">
|
||||
<div className="font-semibold">{exp.id}</div>
|
||||
@@ -375,6 +379,53 @@ const CELL_CLASS: Record<string, string> = {
|
||||
running: "bg-sky-500 animate-pulse",
|
||||
};
|
||||
|
||||
/**
|
||||
* The comparison anchor for an experiment: the completed baseline arm with the
|
||||
* highest pass rate (the "ceiling" a reasoning slide tries to preserve). Ties
|
||||
* break toward the cheaper arm. Returns null when no baseline has finished data.
|
||||
*/
|
||||
function pickReferenceArm(arms: ArmSummary[]): ArmSummary | null {
|
||||
let ref: ArmSummary | null = null;
|
||||
for (const a of arms) {
|
||||
if (a.run.role !== "baseline" || a.passPct === null) continue;
|
||||
if (
|
||||
ref === null ||
|
||||
a.passPct > (ref.passPct ?? -1) ||
|
||||
(a.passPct === ref.passPct && (a.costPerTask ?? Infinity) < (ref.costPerTask ?? Infinity))
|
||||
) {
|
||||
ref = a;
|
||||
}
|
||||
}
|
||||
return ref;
|
||||
}
|
||||
|
||||
/**
|
||||
* Signed, colour-coded offset of a metric from the reference arm. `points`
|
||||
* shows absolute percentage-point difference (pass rate); `relative` shows a
|
||||
* percentage change (cost, time). `higherBetter` decides which direction is green.
|
||||
*/
|
||||
function Delta({
|
||||
value,
|
||||
reference,
|
||||
mode,
|
||||
higherBetter,
|
||||
}: {
|
||||
value: number | null;
|
||||
reference: number | null;
|
||||
mode: "points" | "relative";
|
||||
higherBetter: boolean;
|
||||
}) {
|
||||
if (value === null || reference === null) return null;
|
||||
const raw =
|
||||
mode === "points" ? value - reference : reference === 0 ? Number.NaN : ((value - reference) / reference) * 100;
|
||||
if (!Number.isFinite(raw) || Math.abs(raw) < 0.5) {
|
||||
return <span className="ml-1 text-[10px] text-zinc-600">≈</span>;
|
||||
}
|
||||
const good = higherBetter ? raw > 0 : raw < 0;
|
||||
const body = `${raw > 0 ? "+" : "−"}${Math.abs(raw).toFixed(0)}${mode === "relative" ? "%" : ""}`;
|
||||
return <span className={`ml-1 text-[10px] ${good ? "text-emerald-500" : "text-red-400"}`}>({body})</span>;
|
||||
}
|
||||
|
||||
function ExperimentPage({ id }: { id: string }) {
|
||||
const detail = usePolled<ExperimentDetail>(`/api/experiments/${encodeURIComponent(id)}`, 3000);
|
||||
if (!detail) return <div className="p-10 text-zinc-500">loading…</div>;
|
||||
@@ -394,12 +445,19 @@ function ExperimentPage({ id }: { id: string }) {
|
||||
a => (a.meanTrialMs === null ? null : a.meanTrialMs / 60000),
|
||||
p => p.meanTrialMs / 60000,
|
||||
);
|
||||
const ref = pickReferenceArm(arms);
|
||||
return (
|
||||
<div className="mx-auto max-w-7xl p-6">
|
||||
<div className="mb-1 flex items-baseline gap-4">
|
||||
<h2 className="text-lg font-semibold">{id}</h2>
|
||||
<span className="text-xs text-zinc-500">
|
||||
{arms.length} arms · {tasks.length} tasks
|
||||
{ref && (
|
||||
<>
|
||||
{" "}
|
||||
· Δ vs <span className="text-zinc-400">{ref.arm}</span>
|
||||
</>
|
||||
)}
|
||||
</span>
|
||||
</div>
|
||||
{goal && <p className="mb-4 max-w-4xl text-sm text-zinc-400">{goal}</p>}
|
||||
@@ -430,6 +488,14 @@ function ExperimentPage({ id }: { id: string }) {
|
||||
{arm.run.role}
|
||||
</span>
|
||||
)}
|
||||
{ref?.arm === arm.arm && (
|
||||
<span
|
||||
className="ml-1 text-[10px] text-zinc-500"
|
||||
title="reference arm (highest-pass baseline); deltas are measured against it"
|
||||
>
|
||||
ref
|
||||
</span>
|
||||
)}
|
||||
</td>
|
||||
<td
|
||||
className="max-w-md truncate pr-4 text-xs text-zinc-400"
|
||||
@@ -447,12 +513,33 @@ function ExperimentPage({ id }: { id: string }) {
|
||||
<td className="pr-4">
|
||||
{arm.passPct !== null ? `${arm.passPct.toFixed(0)}%` : "—"}
|
||||
{arm.projected && <span className="text-zinc-500"> →{arm.projected.passPct.toFixed(0)}%</span>}
|
||||
{ref && ref.arm !== arm.arm && (
|
||||
<Delta value={arm.passPct} reference={ref.passPct} mode="points" higherBetter />
|
||||
)}
|
||||
</td>
|
||||
<td className="pr-4">
|
||||
{arm.costPerTask !== null ? fmtUsd(arm.costPerTask) : "—"}
|
||||
{arm.projected && <span className="text-zinc-500"> Σ{fmtUsd(arm.projected.totalCostUsd)}</span>}
|
||||
{ref && ref.arm !== arm.arm && (
|
||||
<Delta
|
||||
value={arm.costPerTask}
|
||||
reference={ref.costPerTask}
|
||||
mode="relative"
|
||||
higherBetter={false}
|
||||
/>
|
||||
)}
|
||||
</td>
|
||||
<td className="pr-4">
|
||||
{arm.meanTrialMs !== null ? fmtMin(arm.meanTrialMs) : "—"}
|
||||
{ref && ref.arm !== arm.arm && (
|
||||
<Delta
|
||||
value={arm.meanTrialMs}
|
||||
reference={ref.meanTrialMs}
|
||||
mode="relative"
|
||||
higherBetter={false}
|
||||
/>
|
||||
)}
|
||||
</td>
|
||||
<td className="pr-4">{arm.meanTrialMs !== null ? fmtMin(arm.meanTrialMs) : "—"}</td>
|
||||
<td>
|
||||
<a
|
||||
className="text-xs text-zinc-500 underline hover:text-zinc-300"
|
||||
@@ -534,23 +621,23 @@ function useRunsSse(): RunRow[] | null {
|
||||
|
||||
function RunsPage({ selected }: { selected: string | null }) {
|
||||
const runs = useRunsSse();
|
||||
const detail = usePolled<{ run: RunRow; trials: TrialRow[] }>(
|
||||
const detail = usePolled<{ run: RunRow; traces: TraceRow[] }>(
|
||||
selected ? `/api/runs/${encodeURIComponent(selected)}` : null,
|
||||
2500,
|
||||
);
|
||||
const [trial, setTrial] = useState<string | null>(null);
|
||||
const transcript = usePolled<{ entries: TranscriptEntry[] }>(
|
||||
selected && trial
|
||||
? `/api/runs/${encodeURIComponent(selected)}/trials/${encodeURIComponent(trial)}/transcript?tail=60`
|
||||
const [trace, setTrace] = useState<string | null>(null);
|
||||
const traceData = usePolled<{ entries: TranscriptEntry[] }>(
|
||||
selected && trace
|
||||
? `/api/runs/${encodeURIComponent(selected)}/traces/${encodeURIComponent(trace)}?tail=60`
|
||||
: null,
|
||||
2500,
|
||||
);
|
||||
const transcriptRef = useRef<HTMLDivElement | null>(null);
|
||||
const traceRef = useRef<HTMLDivElement | null>(null);
|
||||
useEffect(() => {
|
||||
if (!transcript) return;
|
||||
const el = transcriptRef.current;
|
||||
if (!traceData) return;
|
||||
const el = traceRef.current;
|
||||
if (el) el.scrollTop = el.scrollHeight;
|
||||
}, [transcript]);
|
||||
}, [traceData]);
|
||||
const cancel = useCallback(async (name: string) => {
|
||||
if (confirm(`stop ${name}?`)) await fetch(`/api/runs/${encodeURIComponent(name)}`, { method: "DELETE" });
|
||||
}, []);
|
||||
@@ -578,6 +665,7 @@ function RunsPage({ selected }: { selected: string | null }) {
|
||||
>
|
||||
<td className="px-3 py-1.5" title={r.models}>
|
||||
{r.jobName}
|
||||
<div className="text-[10px] uppercase tracking-wide text-zinc-600">{r.benchmark}</div>
|
||||
{(r.note || r.role) && (
|
||||
<div className="text-[11px] text-zinc-500">
|
||||
{r.role && (
|
||||
@@ -622,34 +710,42 @@ function RunsPage({ selected }: { selected: string | null }) {
|
||||
<div className="border-b border-zinc-800 px-4 py-2 text-sm">
|
||||
<span className="font-semibold">{detail.run.jobName}</span> <Chip label={detail.run.status} />{" "}
|
||||
<span className="text-xs text-zinc-500">
|
||||
{detail.run.dataset} · {detail.run.models}
|
||||
{detail.run.benchmark} · {detail.run.dataset} · {detail.run.models}
|
||||
{detail.run.score !== null ? ` · score ${(100 * detail.run.score).toFixed(1)}%` : ""}
|
||||
{detail.run.slide ? ` → ${detail.run.slide}` : ""}
|
||||
</span>
|
||||
<div className="mt-1 flex gap-3 text-xs text-zinc-400">
|
||||
{Object.entries(detail.run.metrics).map(([key, value]) => (
|
||||
<span key={key}>
|
||||
{key.replaceAll("_", " ")}: {value === null ? "—" : `${(100 * value).toFixed(1)}%`}
|
||||
</span>
|
||||
))}
|
||||
</div>
|
||||
</div>
|
||||
<div className="min-h-0 flex-1 overflow-auto">
|
||||
<table className="w-full text-sm">
|
||||
<tbody>
|
||||
{detail.trials.map(t => (
|
||||
{detail.traces.map(t => (
|
||||
<tr
|
||||
key={t.name}
|
||||
onClick={() => setTrial(t.name)}
|
||||
className={`cursor-pointer border-t border-zinc-800/60 hover:bg-zinc-900 ${t.name === trial ? "bg-zinc-900" : ""}`}
|
||||
onClick={() => setTrace(t.name)}
|
||||
className={`cursor-pointer border-t border-zinc-800/60 hover:bg-zinc-900 ${t.name === trace ? "bg-zinc-900" : ""}`}
|
||||
>
|
||||
<td className="px-4 py-1">{t.task}</td>
|
||||
<td>
|
||||
<Chip label={t.status} />
|
||||
</td>
|
||||
<td>{t.reward === null ? "—" : t.reward.toFixed(3)}</td>
|
||||
<td>{fmtUsd(t.costUsd)}</td>
|
||||
<td>{t.durationMs ? fmtMin(t.durationMs) : "—"}</td>
|
||||
<td className="text-xs text-zinc-500">{t.detail}</td>
|
||||
</tr>
|
||||
))}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
{trial && (
|
||||
<div ref={transcriptRef} className="h-2/5 overflow-auto border-t border-zinc-800 bg-zinc-950/60">
|
||||
{(transcript?.entries ?? []).map((e, i) => (
|
||||
{trace && (
|
||||
<div ref={traceRef} className="h-2/5 overflow-auto border-t border-zinc-800 bg-zinc-950/60">
|
||||
{(traceData?.entries ?? []).map((e, i) => (
|
||||
// biome-ignore lint/suspicious/noArrayIndexKey: tail window, entries have no ids
|
||||
<div key={i} className="border-b border-zinc-900 px-4 py-2">
|
||||
<div className="text-xs text-zinc-500">
|
||||
@@ -687,7 +783,7 @@ function LaunchForm({ onDone }: { onDone: () => void }) {
|
||||
async (ev: React.FormEvent<HTMLFormElement>) => {
|
||||
ev.preventDefault();
|
||||
const f = new FormData(ev.currentTarget);
|
||||
const body: Record<string, unknown> = { model: f.get("model") };
|
||||
const body: Record<string, unknown> = { benchmark: f.get("benchmark"), model: f.get("model") };
|
||||
if (f.get("jobName")) body.jobName = f.get("jobName");
|
||||
if (f.get("dataset")) body.dataset = f.get("dataset");
|
||||
if (f.get("tasks")) body.tasks = Number(f.get("tasks"));
|
||||
@@ -699,6 +795,12 @@ function LaunchForm({ onDone }: { onDone: () => void }) {
|
||||
.map(s => s.trim())
|
||||
.filter(Boolean);
|
||||
}
|
||||
if (f.get("conditions")) {
|
||||
body.conditions = String(f.get("conditions"))
|
||||
.split(",")
|
||||
.map(s => s.trim())
|
||||
.filter(Boolean);
|
||||
}
|
||||
if (f.get("goal")) body.goal = f.get("goal");
|
||||
if (f.get("role")) body.role = f.get("role");
|
||||
if (f.get("note")) body.note = f.get("note");
|
||||
@@ -724,10 +826,15 @@ function LaunchForm({ onDone }: { onDone: () => void }) {
|
||||
const input = "rounded border border-zinc-700 bg-zinc-950 px-2 py-1 text-sm";
|
||||
return (
|
||||
<form onSubmit={submit} className="grid grid-cols-4 gap-2 border-b border-zinc-800 bg-zinc-900/70 p-4 text-sm">
|
||||
<select name="benchmark" className={input}>
|
||||
<option value="harbor">Harbor</option>
|
||||
<option value="edit">TypeScript edit</option>
|
||||
<option value="snapcompact">SnapCompact</option>
|
||||
</select>
|
||||
<input name="model" placeholder="model (required)" required className={input} />
|
||||
<input name="dataset" placeholder="dataset (terminal-bench@2.0)" className={input} />
|
||||
<input name="jobName" placeholder="job name (exp-arm)" className={input} />
|
||||
<input name="tasks" type="number" placeholder="tasks" className={input} />
|
||||
<input name="tasks" type="number" placeholder="task/passages limit" className={input} />
|
||||
<input name="concurrency" type="number" placeholder="concurrency" className={input} />
|
||||
<input name="timeoutMultiplier" type="number" step="0.5" placeholder="timeout ×" className={input} />
|
||||
<input name="slideModel" placeholder="slide model" className={input} />
|
||||
@@ -741,6 +848,7 @@ function LaunchForm({ onDone }: { onDone: () => void }) {
|
||||
<input type="checkbox" name="slidePlan" /> plan nudge
|
||||
</label>
|
||||
<input name="include" placeholder="include tasks, comma-sep" className={`${input} col-span-2`} />
|
||||
<input name="conditions" placeholder="SnapCompact conditions, comma-sep" className={`${input} col-span-2`} />
|
||||
<input
|
||||
name="goal"
|
||||
placeholder="experiment goal (what question does this answer?)"
|
||||
|
||||
@@ -12,21 +12,13 @@
|
||||
"url": "git+https://github.com/can1357/oh-my-pi.git",
|
||||
"directory": "packages/typescript-edit-benchmark"
|
||||
},
|
||||
"main": "./src/index.ts",
|
||||
"exports": {
|
||||
".": {
|
||||
"types": "./src/index.ts",
|
||||
"import": "./src/index.ts"
|
||||
},
|
||||
"./*": {
|
||||
"types": "./src/*.ts",
|
||||
"import": "./src/*.ts"
|
||||
},
|
||||
"./*.js": "./src/*.ts"
|
||||
},
|
||||
"bin": {
|
||||
"typescript-edit-benchmark": "src/index.ts"
|
||||
},
|
||||
"scripts": {
|
||||
"check": "biome check . && bun run check:types",
|
||||
"check:types": "tsgo -p tsconfig.json --noEmit",
|
||||
@@ -34,8 +26,7 @@
|
||||
"test": "bun test --parallel",
|
||||
"fix": "biome check --write --unsafe .",
|
||||
"fmt": "biome format --write .",
|
||||
"generate": "bun run src/generate.ts --typescript-dir /tmp/pi-mono-source --count-per-type 4",
|
||||
"start": "bun run src/index.ts"
|
||||
"generate": "bun run src/generate.ts --typescript-dir /tmp/pi-mono-source --count-per-type 4"
|
||||
},
|
||||
"dependencies": {
|
||||
"@babel/generator": "catalog:",
|
||||
|
||||
@@ -1,758 +0,0 @@
|
||||
#!/usr/bin/env bun
|
||||
/**
|
||||
* Edit benchmark CLI entry point.
|
||||
*
|
||||
* Usage:
|
||||
* bun run bench:edit --model anthropic/claude-sonnet-4-5
|
||||
* bun run bench:edit --tasks core-memory-recall,operations-division
|
||||
* bun run bench:edit --runs 5 --output report.md
|
||||
* bun run bench:edit --fixtures fixtures.tar.gz
|
||||
*/
|
||||
import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import { parseArgs } from "node:util";
|
||||
import { type ResolvedThinkingLevel, ThinkingLevel } from "@oh-my-pi/pi-agent-core";
|
||||
import { Effort, THINKING_EFFORTS } from "@oh-my-pi/pi-ai";
|
||||
import { padding, visibleWidth } from "@oh-my-pi/pi-tui";
|
||||
import { postmortem, TempDir } from "@oh-my-pi/pi-utils";
|
||||
import { generateJsonReport, generateReport } from "./report";
|
||||
import {
|
||||
type BenchmarkConfig,
|
||||
type BenchmarkResult,
|
||||
buildBenchmarkResult,
|
||||
type ProgressEvent,
|
||||
percentile,
|
||||
runBenchmark,
|
||||
} from "./runner";
|
||||
import { type EditTask, loadTasksFromDir, validateFixturesFromDir } from "./tasks";
|
||||
|
||||
const COLOR_ENABLED = Boolean(process.stdout.isTTY) && !process.env.NO_COLOR;
|
||||
|
||||
const ANSI = {
|
||||
reset: "\x1b[0m",
|
||||
bold: "\x1b[1m",
|
||||
dim: "\x1b[2m",
|
||||
red: "\x1b[31m",
|
||||
green: "\x1b[32m",
|
||||
yellow: "\x1b[33m",
|
||||
blue: "\x1b[34m",
|
||||
magenta: "\x1b[35m",
|
||||
cyan: "\x1b[36m",
|
||||
} as const;
|
||||
|
||||
const RUNS_DIR = path.resolve(import.meta.dir, "..", "..", "..", "runs");
|
||||
|
||||
fs.mkdirSync(RUNS_DIR, { recursive: true });
|
||||
|
||||
function paint(code: string, text: string): string {
|
||||
return COLOR_ENABLED ? `${code}${text}${ANSI.reset}` : text;
|
||||
}
|
||||
|
||||
function rateColor(percent: number): string {
|
||||
if (percent >= 80) return ANSI.green;
|
||||
if (percent >= 50) return ANSI.yellow;
|
||||
return ANSI.red;
|
||||
}
|
||||
|
||||
function parseThinkingLevel(value: string | null | undefined): ResolvedThinkingLevel | undefined {
|
||||
return value !== undefined &&
|
||||
value !== null &&
|
||||
[ThinkingLevel.Off, ...THINKING_EFFORTS].includes(value as ResolvedThinkingLevel)
|
||||
? (value as ResolvedThinkingLevel)
|
||||
: undefined;
|
||||
}
|
||||
|
||||
function generateReportFilename(config: BenchmarkConfig, format: "markdown" | "json"): string {
|
||||
const modelName = config.model
|
||||
.split("/")
|
||||
.pop()!
|
||||
.replace(/[^a-zA-Z0-9-]/g, "_");
|
||||
const variant = config.editVariant ?? "replace";
|
||||
const timestamp = new Date().toISOString().replace(/:/g, "-").replace(/\..+$/, "").replace(/Z$/, "Z");
|
||||
const ext = format === "json" ? "json" : "md";
|
||||
return path.join(RUNS_DIR, `${modelName}_${variant}_${timestamp}.${ext}`);
|
||||
}
|
||||
|
||||
async function resolveConversationDumpDir(outputPath: string): Promise<string> {
|
||||
const parsed = path.parse(outputPath);
|
||||
const preferredPath = path.join(parsed.dir, `${parsed.name}.dump`);
|
||||
try {
|
||||
await fs.promises.stat(preferredPath);
|
||||
} catch (error) {
|
||||
if ((error as NodeJS.ErrnoException).code === "ENOENT") {
|
||||
return preferredPath;
|
||||
}
|
||||
throw error;
|
||||
}
|
||||
const timestamp = new Date().toISOString().replace(/[:.]/g, "-");
|
||||
return path.join(parsed.dir, `${parsed.name}.${timestamp}.dump`);
|
||||
}
|
||||
|
||||
async function conversationDumpStatus(dumpDir: string): Promise<string> {
|
||||
try {
|
||||
const stat = await fs.promises.stat(dumpDir);
|
||||
if (stat.isDirectory()) {
|
||||
return `Conversation dumps written to: ${dumpDir}`;
|
||||
}
|
||||
return `Conversation dump path is not a directory: ${dumpDir}`;
|
||||
} catch (error) {
|
||||
if ((error as NodeJS.ErrnoException).code === "ENOENT") {
|
||||
return `No conversation dumps written: ${dumpDir}`;
|
||||
}
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
function printUsage(tasks?: EditTask[]): void {
|
||||
const taskList = tasks
|
||||
? tasks.map(t => ` ${t.id.padEnd(30)} ${t.name}`).join("\n")
|
||||
: " (use --list to see available tasks)";
|
||||
console.log(`
|
||||
Edit Benchmark - Evaluate patch application success rates
|
||||
|
||||
Usage:
|
||||
bun run bench:edit [options]
|
||||
|
||||
Options:
|
||||
--model <id> Provider/model ID, e.g. anthropic/claude-sonnet-4-20250514 (default)
|
||||
--provider <id> Override provider (auto-detected from model prefix if omitted)
|
||||
--thinking <level> Thinking level: off, minimal, low, medium, high, xhigh, max
|
||||
--runs <n> Runs per task (default: 1)
|
||||
--timeout <ms> Timeout per run in ms (default: 120000)
|
||||
--connection-timeout <ms> Timeout for first event before fast-retry (default: 30000)
|
||||
--task-concurrency <n> Max tasks to run in parallel (default: 16)
|
||||
--tasks <ids> Comma-separated task IDs to run (default: all)
|
||||
--max-tasks <n> Max tasks to sample (default: 80, 0 = all)
|
||||
--fixtures <path> Fixtures directory or .tar.gz archive (default: built-in)
|
||||
--edit-variant <v> Edit variant: any string (e.g. replace, patch, hashline, vim, atom, apply_patch), or auto (default: auto)
|
||||
--edit-fuzzy <bool> Fuzzy matching: true, false, auto (default: auto)
|
||||
--edit-fuzzy-threshold <n> Fuzzy threshold 0-1 or auto (default: auto)
|
||||
--auto-format Auto-format output files after verify (debug only)
|
||||
--guided Include an authoritative suggested edit payload (default: false)
|
||||
--no-guided Disable guided mode
|
||||
--max-attempts <n> Max prompt attempts per run (default: 1)
|
||||
--no-op-retry-limit <n> Stop after repeated preventable no-op failures (default: 2)
|
||||
--mutation-scope-window <n> Allowed line-distance from mutation target for hashline refs (default: 20)
|
||||
--max-turns <n> Max turn_start events per attempt before failing (default: 30)
|
||||
--output <file> Output file (default: run_<model>_<variant>_<fuzzy>_<threshold>_<timestamp>.md)
|
||||
--format <fmt> Output format: markdown, json (default: markdown)
|
||||
--check-fixtures Validate fixtures and exit
|
||||
--require-edit-tool-call Require edit tool usage for success (default: false)
|
||||
--require-read-tool-call Require read tool usage for success (default: false)
|
||||
--no-edit-required Remove "must edit" prompt requirement (default: false)
|
||||
--no-early-stop-on-match Don't short-circuit the run when output matches expected (default: false)
|
||||
--list List available tasks and exit
|
||||
--help Show this help message
|
||||
|
||||
Available Tasks:
|
||||
${taskList}
|
||||
|
||||
Examples:
|
||||
# Run full benchmark with default model
|
||||
bun run bench:edit
|
||||
|
||||
# Run specific tasks
|
||||
bun run bench:edit --tasks core-memory-recall,operations-division
|
||||
|
||||
# Compare different models
|
||||
bun run bench:edit --model claude-sonnet-4-20250514 --output sonnet.md
|
||||
bun run bench:edit --model claude-opus-4-5-20251101 --output opus.md
|
||||
|
||||
# Run with extended thinking
|
||||
bun run bench:edit --thinking high --runs 5
|
||||
|
||||
# Run from a fixtures archive
|
||||
bun run bench:edit --fixtures edit-fixtures.tar.gz
|
||||
`);
|
||||
}
|
||||
|
||||
async function resolveExtractedDir(tempDir: string): Promise<string> {
|
||||
const entries = await fs.promises.readdir(tempDir, { withFileTypes: true });
|
||||
const dirs = entries.filter(entry => entry.isDirectory());
|
||||
const files = entries.filter(entry => entry.isFile());
|
||||
if (dirs.length === 1 && files.length === 0) {
|
||||
return path.join(tempDir, dirs[0]!.name);
|
||||
}
|
||||
return tempDir;
|
||||
}
|
||||
|
||||
async function extractTarGz(archivePath: string): Promise<{ dir: string; cleanupDir: string }> {
|
||||
const tempDirObj = await TempDir.create("@reach-benchmark-fixtures-");
|
||||
const tempDir = tempDirObj.path();
|
||||
try {
|
||||
const bytes = await Bun.file(archivePath).arrayBuffer();
|
||||
const archive = new Bun.Archive(bytes);
|
||||
const files = await archive.files();
|
||||
|
||||
for (const [filePath, file] of files) {
|
||||
const destPath = path.join(tempDir, filePath);
|
||||
await Bun.write(destPath, file);
|
||||
}
|
||||
} catch (error) {
|
||||
await tempDirObj.remove();
|
||||
const message = error instanceof Error ? error.message : String(error);
|
||||
throw new Error(`Failed to extract archive: ${message}`, { cause: error });
|
||||
}
|
||||
|
||||
return { dir: await resolveExtractedDir(tempDir), cleanupDir: tempDir };
|
||||
}
|
||||
|
||||
async function resolveFixtures(fixturesArg?: string): Promise<{ tasks: EditTask[]; cleanup?: () => Promise<void> }> {
|
||||
fixturesArg ??= path.join(import.meta.dir, "../fixtures.tar.gz");
|
||||
|
||||
if (fixturesArg.endsWith(".tar.gz") || fixturesArg.endsWith(".tgz")) {
|
||||
const extracted = await extractTarGz(fixturesArg);
|
||||
return {
|
||||
tasks: await loadTasksFromDir(extracted.dir),
|
||||
cleanup: () => fs.promises.rm(extracted.cleanupDir, { recursive: true, force: true }),
|
||||
};
|
||||
}
|
||||
|
||||
return { tasks: await loadTasksFromDir(fixturesArg) };
|
||||
}
|
||||
|
||||
async function main(): Promise<void> {
|
||||
const { values } = parseArgs({
|
||||
options: {
|
||||
provider: { type: "string" },
|
||||
model: { type: "string", default: "anthropic/claude-sonnet-4-20250514" },
|
||||
thinking: { type: "string", default: "low" },
|
||||
runs: { type: "string", default: "2" },
|
||||
timeout: { type: "string", default: "120000" },
|
||||
"connection-timeout": { type: "string", default: "30000" },
|
||||
"max-turns": { type: "string", default: "30" },
|
||||
"task-concurrency": { type: "string", default: "32" },
|
||||
tasks: { type: "string" },
|
||||
fixtures: { type: "string" },
|
||||
output: { type: "string" },
|
||||
format: { type: "string", default: "markdown" },
|
||||
"check-fixtures": { type: "boolean", default: false },
|
||||
"auto-format": { type: "boolean", default: false },
|
||||
guided: { type: "boolean", default: false },
|
||||
"no-guided": { type: "boolean", default: false },
|
||||
"max-attempts": { type: "string", default: "1" },
|
||||
"no-op-retry-limit": { type: "string", default: "2" },
|
||||
"max-timeout-retries": { type: "string", default: "3" },
|
||||
"max-provider-retries": { type: "string", default: "3" },
|
||||
"mutation-scope-window": { type: "string", default: "20" },
|
||||
"require-edit-tool-call": { type: "boolean", default: false },
|
||||
"require-read-tool-call": { type: "boolean", default: false },
|
||||
"no-edit-required": { type: "boolean", default: false },
|
||||
"edit-variant": { type: "string" },
|
||||
"edit-fuzzy": { type: "string" },
|
||||
"edit-fuzzy-threshold": { type: "string" },
|
||||
"no-in-process": { type: "boolean", default: false },
|
||||
"no-early-stop-on-match": { type: "boolean", default: false },
|
||||
"max-tasks": { type: "string", default: "80" },
|
||||
list: { type: "boolean", default: false },
|
||||
help: { type: "boolean", default: false },
|
||||
},
|
||||
allowPositionals: true,
|
||||
});
|
||||
|
||||
// Extract provider for display/config purposes only.
|
||||
// The full model string (e.g. "openrouter/google/gemini-2.5-flash-lite") is passed
|
||||
// as --model to the CLI, which handles resolution via parseModelPattern.
|
||||
const model = values.model!;
|
||||
const slashIndex = model.indexOf("/");
|
||||
const provider = values.provider ?? (slashIndex !== -1 ? model.slice(0, slashIndex) : "anthropic");
|
||||
|
||||
if (values.help) {
|
||||
printUsage();
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
if (values["check-fixtures"] && values.fixtures) {
|
||||
const issues = await validateFixturesFromDir(values.fixtures);
|
||||
if (issues.length === 0) {
|
||||
console.log("Fixtures OK");
|
||||
process.exit(0);
|
||||
}
|
||||
console.error("Fixture validation failed:");
|
||||
for (const issue of issues) {
|
||||
console.error(` - ${issue.taskId}: ${issue.message}`);
|
||||
}
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const { tasks: allTasks, cleanup } = await resolveFixtures(values.fixtures);
|
||||
|
||||
if (values.list) {
|
||||
console.log("Available Tasks:\n");
|
||||
for (const task of allTasks) {
|
||||
console.log(` ${task.id}`);
|
||||
console.log(` Name: ${task.name}`);
|
||||
console.log(` Files: ${task.files.join(", ")}`);
|
||||
console.log("");
|
||||
}
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
let thinkingLevel: ResolvedThinkingLevel = Effort.Low;
|
||||
if (values.thinking) {
|
||||
const level = parseThinkingLevel(values.thinking);
|
||||
if (!level) {
|
||||
console.error(`Invalid thinking level: ${values.thinking}`);
|
||||
console.error(`Valid levels: ${[ThinkingLevel.Off, ...THINKING_EFFORTS].join(", ")}`);
|
||||
process.exit(1);
|
||||
}
|
||||
thinkingLevel = level;
|
||||
}
|
||||
|
||||
const runsPerTask = parseInt(values.runs!, 10);
|
||||
if (Number.isNaN(runsPerTask) || runsPerTask < 1) {
|
||||
console.error(`Invalid runs value: ${values.runs}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const timeout = parseInt(values.timeout!, 10);
|
||||
if (Number.isNaN(timeout) || timeout < 1000) {
|
||||
console.error(`Invalid timeout value: ${values.timeout}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const maxTurns = parseInt(values["max-turns"]!, 10);
|
||||
if (Number.isNaN(maxTurns) || maxTurns < 1) {
|
||||
console.error(`Invalid max-turns value: ${values["max-turns"]}. Must be >= 1.`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const taskConcurrency = parseInt(values["task-concurrency"]!, 10);
|
||||
if (Number.isNaN(taskConcurrency) || taskConcurrency < 1) {
|
||||
console.error(`Invalid task concurrency value: ${values["task-concurrency"]}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const maxAttempts = parseInt(values["max-attempts"] ?? "2", 10);
|
||||
if (Number.isNaN(maxAttempts) || maxAttempts < 1 || maxAttempts > 5) {
|
||||
console.error(`Invalid max-attempts value: ${values["max-attempts"]}. Must be 1-5.`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const noOpRetryLimit = parseInt(values["no-op-retry-limit"] ?? "2", 10);
|
||||
const maxTimeoutRetries = parseInt(values["max-timeout-retries"] ?? "3", 10);
|
||||
const maxProviderRetries = parseInt(values["max-provider-retries"] ?? "3", 10);
|
||||
const mutationScopeWindow = parseInt(values["mutation-scope-window"] ?? "20", 10);
|
||||
const connectionTimeout = parseInt(values["connection-timeout"] ?? "30000", 10);
|
||||
|
||||
let tasksToRun = allTasks;
|
||||
if (values.tasks) {
|
||||
const taskIds = values.tasks.split(",").map(s => s.trim());
|
||||
tasksToRun = [];
|
||||
for (const id of taskIds) {
|
||||
const task = allTasks.find(t => t.id === id);
|
||||
if (!task) {
|
||||
console.error(`Unknown task ID: ${id}`);
|
||||
console.error(`Available tasks: ${allTasks.map(t => t.id).join(", ")}`);
|
||||
process.exit(1);
|
||||
}
|
||||
tasksToRun.push(task);
|
||||
}
|
||||
}
|
||||
|
||||
// Apply --max-tasks sampling (deterministic by sorting on id)
|
||||
const maxTasks = parseInt(values["max-tasks"] ?? "80", 10);
|
||||
if (maxTasks > 0 && tasksToRun.length > maxTasks && !values.tasks) {
|
||||
// Evenly sample across mutation categories for representative coverage
|
||||
const sorted = tasksToRun.slice().sort((a, b) => a.id.localeCompare(b.id));
|
||||
const step = sorted.length / maxTasks;
|
||||
tasksToRun = Array.from({ length: maxTasks }, (_, i) => sorted[Math.floor(i * step)]!);
|
||||
}
|
||||
|
||||
const rawEditVariant = values["edit-variant"] as string | undefined;
|
||||
const editVariant = rawEditVariant === "" ? undefined : rawEditVariant;
|
||||
|
||||
let editFuzzy: boolean | "auto" | undefined;
|
||||
if (values["edit-fuzzy"] !== undefined) {
|
||||
if (values["edit-fuzzy"] === "auto") {
|
||||
editFuzzy = "auto";
|
||||
} else if (values["edit-fuzzy"] === "true" || values["edit-fuzzy"] === "1") {
|
||||
editFuzzy = true;
|
||||
} else if (values["edit-fuzzy"] === "false" || values["edit-fuzzy"] === "0") {
|
||||
editFuzzy = false;
|
||||
} else {
|
||||
console.error(`Invalid edit-fuzzy: ${values["edit-fuzzy"]}. Must be true, false, 1, 0, or auto.`);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
let editFuzzyThreshold: number | "auto" | undefined;
|
||||
if (values["edit-fuzzy-threshold"] !== undefined) {
|
||||
if (values["edit-fuzzy-threshold"] === "auto") {
|
||||
editFuzzyThreshold = "auto";
|
||||
} else {
|
||||
const parsed = parseFloat(values["edit-fuzzy-threshold"]);
|
||||
if (Number.isNaN(parsed) || parsed < 0 || parsed > 1) {
|
||||
console.error(`Invalid edit-fuzzy-threshold: ${values["edit-fuzzy-threshold"]}. Must be 0-1 or auto.`);
|
||||
process.exit(1);
|
||||
}
|
||||
editFuzzyThreshold = parsed;
|
||||
}
|
||||
}
|
||||
|
||||
const guided = values["no-guided"] ? false : values.guided;
|
||||
|
||||
const formatType = values.format === "json" ? "json" : "markdown";
|
||||
const config: BenchmarkConfig = {
|
||||
provider,
|
||||
model,
|
||||
thinkingLevel,
|
||||
runsPerTask,
|
||||
timeout,
|
||||
maxTurns,
|
||||
taskConcurrency,
|
||||
autoFormat: values["auto-format"],
|
||||
guided,
|
||||
maxAttempts,
|
||||
requireEditToolCall: values["require-edit-tool-call"],
|
||||
requireReadToolCall: values["require-read-tool-call"],
|
||||
noEditRequired: values["no-edit-required"],
|
||||
editVariant,
|
||||
editFuzzy,
|
||||
editFuzzyThreshold,
|
||||
noOpRetryLimit,
|
||||
maxTimeoutRetries,
|
||||
maxProviderFailureRetries: maxProviderRetries,
|
||||
mutationScopeWindow,
|
||||
connectionTimeout,
|
||||
inProcess: !values["no-in-process"],
|
||||
earlyStopOnMatch: !values["no-early-stop-on-match"],
|
||||
};
|
||||
const outputPath = values.output ?? generateReportFilename(config, formatType);
|
||||
config.conversationDumpDir = await resolveConversationDumpDir(outputPath);
|
||||
|
||||
console.log("Edit Benchmark");
|
||||
console.log("==============");
|
||||
console.log(`Provider: ${config.provider}`);
|
||||
console.log(`Model: ${config.model}`);
|
||||
if (config.thinkingLevel) {
|
||||
console.log(`Thinking: ${config.thinkingLevel}`);
|
||||
}
|
||||
console.log(`Runs per task: ${config.runsPerTask}`);
|
||||
console.log(`Timeout: ${config.timeout}ms`);
|
||||
console.log(`Task concurrency: ${config.taskConcurrency}`);
|
||||
if (config.autoFormat) {
|
||||
console.log("Auto-format: enabled");
|
||||
}
|
||||
console.log(`Guided mode: ${config.guided ? "enabled" : "disabled"}`);
|
||||
console.log(`Max attempts: ${config.maxAttempts}`);
|
||||
if (config.maxTurns !== undefined) {
|
||||
console.log(`Max turns per attempt: ${config.maxTurns}`);
|
||||
}
|
||||
if (config.requireEditToolCall) {
|
||||
console.log("Require edit tool call: yes");
|
||||
}
|
||||
if (config.requireReadToolCall) {
|
||||
console.log("Require read tool call: yes");
|
||||
}
|
||||
if (config.noEditRequired) {
|
||||
console.log("No-edit-required baseline: yes");
|
||||
}
|
||||
if (config.editVariant) {
|
||||
console.log(`Edit variant: ${config.editVariant}`);
|
||||
}
|
||||
if (config.editFuzzy !== undefined) {
|
||||
console.log(`Edit fuzzy: ${config.editFuzzy}`);
|
||||
}
|
||||
if (config.editFuzzyThreshold !== undefined) {
|
||||
console.log(`Edit fuzzy threshold: ${config.editFuzzyThreshold}`);
|
||||
}
|
||||
console.log(`Tasks: ${tasksToRun.length}`);
|
||||
console.log(`Conversation dumps: ${config.conversationDumpDir}`);
|
||||
console.log("");
|
||||
|
||||
const progress = new LiveProgress(tasksToRun.length * config.runsPerTask, config.runsPerTask);
|
||||
let latestResult = buildBenchmarkResult({
|
||||
tasks: tasksToRun,
|
||||
config,
|
||||
resultsByTask: new Map(),
|
||||
startTime: new Date().toISOString(),
|
||||
});
|
||||
let progressFinished = false;
|
||||
let reportWritePromise: Promise<void> | undefined;
|
||||
const finishProgress = () => {
|
||||
if (progressFinished) return;
|
||||
progress.finish();
|
||||
progressFinished = true;
|
||||
};
|
||||
const writeReport = async (result: BenchmarkResult, interrupted: boolean) => {
|
||||
if (reportWritePromise) return reportWritePromise;
|
||||
reportWritePromise = (async () => {
|
||||
if (interrupted) {
|
||||
console.log("");
|
||||
console.log("Benchmark interrupted; writing partial report...");
|
||||
}
|
||||
const report = formatType === "json" ? generateJsonReport(result) : generateReport(result);
|
||||
await Bun.write(outputPath, report);
|
||||
console.log(`Report written to: ${outputPath}`);
|
||||
if (config.conversationDumpDir) {
|
||||
console.log(await conversationDumpStatus(config.conversationDumpDir));
|
||||
}
|
||||
})();
|
||||
return reportWritePromise;
|
||||
};
|
||||
const unregisterReportCleanup = postmortem.register("typescript-edit-benchmark-report", async reason => {
|
||||
if (reason === postmortem.Reason.EXIT) return;
|
||||
finishProgress();
|
||||
await writeReport(latestResult, true);
|
||||
if (cleanup) {
|
||||
await cleanup();
|
||||
}
|
||||
});
|
||||
const result = await runBenchmark(
|
||||
tasksToRun,
|
||||
config,
|
||||
event => {
|
||||
progress.handleEvent(event);
|
||||
},
|
||||
snapshot => {
|
||||
latestResult = snapshot;
|
||||
},
|
||||
);
|
||||
latestResult = result;
|
||||
finishProgress();
|
||||
|
||||
console.log("");
|
||||
console.log("Benchmark complete!");
|
||||
console.log(
|
||||
` Task success rate (best of ${config.runsPerTask}): ${(result.summary.taskSuccessRate * 100).toFixed(1)}% (${result.summary.successfulTasks}/${result.summary.totalTasks})`,
|
||||
);
|
||||
console.log(
|
||||
` Total tokens (best, overall): ${result.summary.totalTokens.input} in / ${result.summary.totalTokens.output} out`,
|
||||
);
|
||||
console.log(
|
||||
` Tokens/task (best, overall): mean=${result.summary.avgTokensPerTask.total} median=${result.summary.medianTokensPerTask.total} p1=${result.summary.p1TokensPerTask.total} p99=${result.summary.p99TokensPerTask.total} reasoning=${result.summary.avgTokensPerTask.reasoning}`,
|
||||
);
|
||||
console.log(
|
||||
` Total tokens (one-shot successes): ${result.summary.totalOneShotSuccessTokens.input} in / ${result.summary.totalOneShotSuccessTokens.output} out`,
|
||||
);
|
||||
console.log(
|
||||
` Tokens/task (one-shot successes): mean=${result.summary.avgOneShotSuccessTokensPerTask.total} median=${result.summary.medianOneShotSuccessTokensPerTask.total} p1=${result.summary.p1OneShotSuccessTokensPerTask.total} p99=${result.summary.p99OneShotSuccessTokensPerTask.total} reasoning=${result.summary.avgOneShotSuccessTokensPerTask.reasoning}`,
|
||||
);
|
||||
if (result.summary.ghostRuns > 0) {
|
||||
console.log(` Ghost runs (0/0/0): ${result.summary.ghostRuns}`);
|
||||
}
|
||||
if (result.summary.timeoutRuns > 0) {
|
||||
console.log(` Timeout runs: ${result.summary.timeoutRuns}`);
|
||||
}
|
||||
console.log("");
|
||||
|
||||
await writeReport(result, false);
|
||||
unregisterReportCleanup();
|
||||
|
||||
if (cleanup) {
|
||||
await cleanup();
|
||||
}
|
||||
|
||||
// In-process benchmark runs can leave provider keep-alive sockets and
|
||||
// background AgentSession timers alive after the report is written. Treat the
|
||||
// final report as the CLI boundary so the command returns to the shell.
|
||||
await postmortem.quit(0);
|
||||
}
|
||||
|
||||
class LiveProgress {
|
||||
readonly #totalRuns: number;
|
||||
readonly #runsPerTask: number;
|
||||
readonly #isTty: boolean;
|
||||
#started = 0;
|
||||
#completed = 0;
|
||||
#success = 0;
|
||||
#totalInput = 0;
|
||||
#totalOutput = 0;
|
||||
#totalDuration = 0;
|
||||
#totalReads = 0;
|
||||
#totalEdits = 0;
|
||||
#totalWrites = 0;
|
||||
#totalEditSuccesses = 0;
|
||||
#totalToolInputChars = 0;
|
||||
#indentScores: number[] = [];
|
||||
#inputTokens: number[] = [];
|
||||
#outputTokens: number[] = [];
|
||||
#totalTokens: number[] = [];
|
||||
#oneShotSuccessTokens: number[] = [];
|
||||
#lastLineLength = 0;
|
||||
|
||||
constructor(totalRuns: number, runsPerTask: number) {
|
||||
this.#totalRuns = totalRuns;
|
||||
this.#runsPerTask = runsPerTask;
|
||||
this.#isTty = Boolean(process.stdout.isTTY);
|
||||
}
|
||||
|
||||
handleEvent(event: ProgressEvent): void {
|
||||
if (event.status === "started") {
|
||||
this.#started += 1;
|
||||
if (!this.#isTty) {
|
||||
console.log(` [${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} started...`);
|
||||
}
|
||||
this.#renderLine();
|
||||
return;
|
||||
}
|
||||
|
||||
this.#completed += 1;
|
||||
if (event.result) {
|
||||
if (event.result.success) {
|
||||
this.#success += 1;
|
||||
}
|
||||
if (event.result.success && event.runIndex === 0) {
|
||||
this.#oneShotSuccessTokens.push(event.result.tokens.total);
|
||||
}
|
||||
this.#totalInput += event.result.tokens.input;
|
||||
this.#totalOutput += event.result.tokens.output;
|
||||
this.#inputTokens.push(event.result.tokens.input);
|
||||
this.#outputTokens.push(event.result.tokens.output);
|
||||
this.#totalTokens.push(event.result.tokens.total);
|
||||
this.#totalDuration += event.result.duration;
|
||||
this.#totalReads += event.result.toolCalls.read;
|
||||
this.#totalEdits += event.result.toolCalls.edit;
|
||||
this.#totalWrites += event.result.toolCalls.write;
|
||||
this.#totalEditSuccesses += event.result.toolCalls.editSuccesses;
|
||||
this.#totalToolInputChars += event.result.toolCalls.totalInputChars;
|
||||
if (typeof event.result.indentScore === "number") {
|
||||
this.#indentScores.push(event.result.indentScore);
|
||||
}
|
||||
}
|
||||
|
||||
const result = event.result;
|
||||
if (result && !result.success && result.error) {
|
||||
this.#flushLine();
|
||||
const header = paint(ANSI.red, `[${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} failed:`);
|
||||
console.log(` ${header} ${result.error}`);
|
||||
if (result.diff) {
|
||||
const changeLines = result.diff
|
||||
.split("\n")
|
||||
.filter(line => /^[-+@]/.test(line) && !/^(---|\+\+\+)/.test(line));
|
||||
const maxLines = 40;
|
||||
const shown = changeLines.slice(0, maxLines);
|
||||
for (const line of shown) {
|
||||
let color: string | undefined;
|
||||
if (line.startsWith("@@")) color = ANSI.cyan;
|
||||
else if (line.startsWith("-")) color = ANSI.red;
|
||||
else if (line.startsWith("+")) color = ANSI.green;
|
||||
console.log(` ${color ? paint(color, line) : line}`);
|
||||
}
|
||||
if (changeLines.length > maxLines) {
|
||||
console.log(paint(ANSI.dim, ` ... (${changeLines.length - maxLines} more change lines)`));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (result?.editFailures && result.editFailures.length > 0) {
|
||||
this.#flushLine();
|
||||
for (const [i, failure] of result.editFailures.entries()) {
|
||||
const args = (failure.args ?? {}) as Record<string, unknown>;
|
||||
const target =
|
||||
typeof args.path === "string" ? args.path : typeof args.file === "string" ? args.file : undefined;
|
||||
const op = typeof args.operation === "string" ? args.operation : undefined;
|
||||
const oneLine = failure.error.replace(/\s+/g, " ").trim();
|
||||
const clipped = oneLine.length > 240 ? `${oneLine.slice(0, 237)}...` : oneLine;
|
||||
const tag = paint(ANSI.yellow, `[${event.taskId}] schema #${i + 1}`);
|
||||
const metaParts = [op, target].filter((v): v is string => Boolean(v));
|
||||
const meta = metaParts.length > 0 ? paint(ANSI.dim, metaParts.join(" ")) : "";
|
||||
console.log(` ${tag}${meta ? ` ${meta}` : ""} ${clipped}`);
|
||||
if (failure.rawBlock) {
|
||||
const rawLine = failure.rawBlock.replace(/\s+/g, " ").trim();
|
||||
const clippedRaw = rawLine.length > 240 ? `${rawLine.slice(0, 237)}...` : rawLine;
|
||||
console.log(` ${paint(ANSI.dim, "raw")} ${clippedRaw}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!this.#isTty) {
|
||||
const status = event.result?.success ? "completed" : "failed";
|
||||
console.log(` [${event.taskId}] Run ${event.runIndex + 1}/${this.#runsPerTask} ${status}`);
|
||||
}
|
||||
|
||||
this.#renderLine();
|
||||
}
|
||||
|
||||
finish(): void {
|
||||
this.#flushLine();
|
||||
this.#printSummary();
|
||||
}
|
||||
|
||||
#printSummary(): void {
|
||||
const n = this.#completed;
|
||||
const denom = n || 1;
|
||||
|
||||
const successRate = (this.#success / denom) * 100;
|
||||
const editSuccessRate = this.#totalEdits > 0 ? (this.#totalEditSuccesses / this.#totalEdits) * 100 : 100;
|
||||
const avgIndent =
|
||||
this.#indentScores.length > 0 ? this.#indentScores.reduce((a, b) => a + b, 0) / this.#indentScores.length : 0;
|
||||
|
||||
console.log("");
|
||||
console.log(paint(ANSI.bold, "Runtime Stats:"));
|
||||
console.log(
|
||||
` Task success: ${paint(rateColor(successRate), `${successRate.toFixed(1)}% (${this.#success}/${n})`)}`,
|
||||
);
|
||||
console.log(
|
||||
` Edit success: ${paint(rateColor(editSuccessRate), `${editSuccessRate.toFixed(1)}% (${this.#totalEditSuccesses}/${this.#totalEdits})`)}`,
|
||||
);
|
||||
console.log(` Avg indent score: ${avgIndent.toFixed(2)}`);
|
||||
console.log(` Tool calls: read=${this.#totalReads} edit=${this.#totalEdits} write=${this.#totalWrites}`);
|
||||
console.log(` Tool input chars: ${this.#totalToolInputChars.toLocaleString()}`);
|
||||
const fmtTokens = (samples: number[]): string => {
|
||||
if (samples.length === 0) return "mean=0 median=0 p1=0 p99=0";
|
||||
const sorted = [...samples].sort((a, b) => a - b);
|
||||
const mean = Math.round(sorted.reduce((a, b) => a + b, 0) / sorted.length);
|
||||
return `mean=${mean} median=${Math.round(percentile(sorted, 50))} p1=${Math.round(percentile(sorted, 1))} p99=${Math.round(percentile(sorted, 99))}`;
|
||||
};
|
||||
console.log(` Tokens/task in: ${fmtTokens(this.#inputTokens)}`);
|
||||
console.log(` Tokens/task out: ${fmtTokens(this.#outputTokens)}`);
|
||||
console.log(` Tokens/task tot: ${fmtTokens(this.#totalTokens)}`);
|
||||
console.log(` Tokens/task (one-shot successes): ${fmtTokens(this.#oneShotSuccessTokens)}`);
|
||||
console.log(` Avg time/task: ${Math.round(this.#totalDuration / denom)}ms`);
|
||||
}
|
||||
|
||||
#renderLine(): void {
|
||||
if (!this.#isTty) {
|
||||
return;
|
||||
}
|
||||
const successRate = this.#completed > 0 ? (this.#success / this.#completed) * 100 : 0;
|
||||
const editRate = this.#totalEdits > 0 ? (this.#totalEditSuccesses / this.#totalEdits) * 100 : 100;
|
||||
const avgInput = this.#completed > 0 ? Math.round(this.#totalInput / this.#completed) : 0;
|
||||
const avgOutput = this.#completed > 0 ? Math.round(this.#totalOutput / this.#completed) : 0;
|
||||
const avgDuration = this.#completed > 0 ? Math.round(this.#totalDuration / this.#completed) : 0;
|
||||
const inFlight = this.#started - this.#completed;
|
||||
const bar = this.#renderBar(this.#completed, this.#totalRuns, 20);
|
||||
const progress = paint(ANSI.bold, `${this.#completed}/${this.#totalRuns}`);
|
||||
const taskCol = `task=${paint(rateColor(successRate), `${successRate.toFixed(0)}%`)}`;
|
||||
const editCol = `edit=${paint(rateColor(editRate), `${editRate.toFixed(0)}%`)}`;
|
||||
const tokCol = paint(ANSI.dim, `tok=${avgInput}/${avgOutput}`);
|
||||
const durCol = paint(ANSI.dim, `${avgDuration}ms`);
|
||||
const rewCol = paint(ANSI.dim, `r/e/w=${this.#totalReads}/${this.#totalEdits}/${this.#totalWrites}`);
|
||||
const flyCol = `fly=${paint(ANSI.cyan, String(inFlight))}`;
|
||||
const line = ` ${bar} ${progress} ${taskCol} ${editCol} ${tokCol} ${durCol} ${rewCol} ${flyCol}`;
|
||||
this.#writeLine(line);
|
||||
}
|
||||
|
||||
#renderBar(done: number, total: number, width: number): string {
|
||||
const ratio = total === 0 ? 0 : done / total;
|
||||
const filled = Math.round(ratio * width);
|
||||
const empty = Math.max(0, width - filled);
|
||||
const filledPart = paint(ANSI.green, "#".repeat(filled));
|
||||
const emptyPart = paint(ANSI.dim, "-".repeat(empty));
|
||||
return `[${filledPart}${emptyPart}]`;
|
||||
}
|
||||
|
||||
#writeLine(line: string): void {
|
||||
const lineWidth = visibleWidth(line);
|
||||
const pad = this.#lastLineLength > lineWidth ? padding(this.#lastLineLength - lineWidth) : "";
|
||||
process.stdout.write(`\r${line}${pad}`);
|
||||
this.#lastLineLength = lineWidth;
|
||||
}
|
||||
|
||||
#flushLine(): void {
|
||||
if (!this.#isTty) {
|
||||
return;
|
||||
}
|
||||
if (this.#lastLineLength > 0) {
|
||||
process.stdout.write(`\r${padding(this.#lastLineLength)}\r`);
|
||||
this.#lastLineLength = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
main().catch(async err => {
|
||||
console.error("Benchmark failed:", err);
|
||||
await postmortem.quit(1);
|
||||
});
|
||||
Reference in New Issue
Block a user