From ce97b4520a415ab972c762650e22aae95b961e20 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 20 Jun 2026 21:26:48 +0200 Subject: [PATCH] feat(terminal-bench): implemented terminal-based benchmarking runner - Introduced a new CLI tool to orchestrate Harbor-based task execution and local benchmarking. - Implemented a live terminal dashboard for real-time monitoring of trial progress, metrics, and cost aggregation. - Added automated Docker management for Harbor task cleanup and host-to-container gateway authentication. - Enabled support for custom agents, model selection, and advisor configurations through a pluggable architecture. --- packages/terminal-bench/README.md | 131 +++ packages/terminal-bench/agent/omp_local.py | 484 +++++++++ packages/terminal-bench/package.json | 30 + packages/terminal-bench/src/runner.ts | 1140 ++++++++++++++++++++ packages/terminal-bench/tsconfig.json | 6 + 5 files changed, 1791 insertions(+) create mode 100644 packages/terminal-bench/README.md create mode 100644 packages/terminal-bench/agent/omp_local.py create mode 100644 packages/terminal-bench/package.json create mode 100755 packages/terminal-bench/src/runner.ts create mode 100644 packages/terminal-bench/tsconfig.json diff --git a/packages/terminal-bench/README.md b/packages/terminal-bench/README.md new file mode 100644 index 000000000..1c5786942 --- /dev/null +++ b/packages/terminal-bench/README.md @@ -0,0 +1,131 @@ +# @oh-my-pi/terminal-bench + +Run [harbor-framework/terminal-bench-2](https://github.com/harbor-framework/terminal-bench-2) +against the **local `omp` build** with a live progress / spend / success dashboard. + +It drives [Harbor](https://github.com/laude-institute/harbor) (the official TB-2 +harness) under the hood and renders its own dashboard by polling each trial's +`result.json`. + +``` +bun src/runner.ts --model anthropic/claude-sonnet-4-6 --tasks 20 --concurrency 4 +``` + +``` +terminal-bench-2 · omp · anthropic/claude-sonnet-4-6 · conc=4 k=1 +████████████░░░░░░░░ 12/20 elapsed 14:32 eta ~6:10 +pass 9 (75%) fail 2 err 1 run 4 pend 4 +spend $1.84 in 1.2M out 84k cache 3.1M +────────────────────────────────────────────────────── + ✓ fix-git r1.00 $0.12 1:40 + ✗ regex-chess r0.00 $0.31 4:10 + ! qemu-startup — $0.04 0:30 TimeoutError + ⠙ path-tracing · $0.05 2:01 +────────────────────────────────────────────────────── +harbor: ... +``` + +## How it works + +1. **Local omp, not npm.** Harbor's built-in `pi` agent installs the upstream + `@mariozechner/pi-coding-agent`. This runner instead packs the working tree — + `bun pm pack` in `packages/coding-agent`, which bundles every workspace TS + package into `dist/cli.js` — and a custom Harbor agent + ([`agent/omp_local.py`](./agent/omp_local.py)) uploads that tarball into each + task container, installs Bun, `bun install`s the bundle's external deps + the + matching `@oh-my-pi/pi-natives-linux-` prebuilt, and runs + `bun .../dist/cli.js --print --mode json --no-session --auto-approve`. +2. **Auth via the host gateway — no keys in containers.** A generated + `~/.omp/agent/models.yml` routes the model providers' `baseUrl` at the host + pm2 `omp-auth-gateway` (`http://host.docker.internal:4000`, `transport: + pi-native`). The gateway resolves credentials host-side; containers only ever + see a dummy `apiKey`. Cost/tokens are parsed from omp's `message_end` events. +3. **Live dashboard.** Harbor's own output is redirected to `harbor.log`; this + process owns the terminal and polls `/result.json` (authoritative + totals) + `//result.json` (per-task reward/cost/tokens). + +## Prerequisites + +- **Docker** running (Docker Desktop on macOS — provides `host.docker.internal`). +- **Harbor**: `uv tool install harbor` (provides `harbor` on `PATH`). +- **Auth gateway** running on the host (pm2 `omp-auth-gateway`, default + `127.0.0.1:4000`, started `--no-auth` on loopback). Check: + `curl -s 127.0.0.1:4000/healthz`. + +## Usage + +``` +bun src/runner.ts [options] [-- ] +``` + +| Option | Default | Notes | +|---|---|---| +| `-m, --model ` | `anthropic/claude-sonnet-4-6` | Repeatable | +| `-l, --tasks ` | `20` | Max tasks | +| `-n, --concurrency ` | `4` | Concurrent trials | +| `-k, --attempts ` | `1` | Attempts per task (pass@k) | +| `-i/-x, --include/--exclude ` | — | Task filters (repeatable) | +| `--thinking ` | — | `off…xhigh` | +| `--advisor-model

` | — | Second model reviewing the primary; spend summed in | +| `--agent ` | `omp` | `oracle`/`nop`/any harbor agent (bypasses omp) | +| `--install ` | `local` | `published` = npm `@oh-my-pi/pi-coding-agent` | +| `--tarball ` / `--no-build` | — | Reuse a prebuilt omp tarball | +| `--gateway-url ` | `http://host.docker.internal:4000` | | +| `--no-gateway` | off | Pass host provider keys into containers instead | +| `--allow-host ` | — | `harbor --allow-agent-host` (allowlist tasks) | +| `-o, --jobs-dir ` | `/runs/tb2` | | +| `--timeout-multiplier ` | — | Scale task timeouts | +| `--dry-run` | off | Print the harbor command + models.yml and exit | + +### Examples + +```bash +# Default 20-task run +bun src/runner.ts -m anthropic/claude-sonnet-4-6 -l 20 -n 4 + +# Compare models (separate runs / job dirs) +bun src/runner.ts -m openai-codex/gpt-5.1-codex -l 20 --thinking high + +# Primary model + advisor model (combined spend reported) +bun src/runner.ts -m anthropic/claude-sonnet-4-6 --advisor-model anthropic/claude-haiku-4-5 -l 20 + +# Cheap pipeline smoke (no model spend — uses task reference solutions) +bun src/runner.ts --agent oracle -l 2 + +# Inspect exactly what will run +bun src/runner.ts --dry-run -l 20 +``` + +## Output + +Per run, under `--jobs-dir`: + +- `/` — Harbor's trial dirs (`result.json` per trial). +- `_bench//report.md` — markdown summary table. +- `_bench//harbor.log` — full Harbor output. +- `_bench//models.yml` — generated gateway routing (dummy key). + +## Caveats + +- **Network policy.** On Harbor's local Docker backend only **public** + agent-phase tasks can reach the host gateway. `no_network` tasks cut off + `host.docker.internal`; allowlist tasks aren't supported by the Docker backend + (use `--allow-host` only when you know the task allows it). Such tasks surface + as agent errors in the dashboard. +- **Architecture.** The native prebuilt is selected from the container's + `uname -m` (linux-arm64 / linux-x64). Alpine/musl base images are unsupported + (no musl prebuilt) and fail fast in `install()`. +- **`web_search` is off by default.** It can't authenticate through the gateway + (it uses dedicated search-provider creds), so it's disabled in the container + config to avoid 401s and false negatives. Re-enable with `--web-search` only + if the container can reach a search provider. The report shows `web_search=on/off`. +- **Advisor spend accuracy vs speed.** `--advisor-model` runs a second model + (separate spend, summed into the reported total + an `(advisor $…)` breakdown). + Its turns are flushed to `__advisor.jsonl` only when the primary stays caught + up, so the default `--advisor-sync 1` waits per turn for accurate spend. + `--advisor-sync off` is faster but can drop end-of-run advisor backlog, + undercounting advisor cost. +- **`--install local` reflects local TS changes** across the monorepo (they're + inlined into `dist/cli.js`), but **not** uncommitted changes to the Rust + natives or other externalized deps (mupdf, puppeteer) — those resolve from npm + at the bundle's version. diff --git a/packages/terminal-bench/agent/omp_local.py b/packages/terminal-bench/agent/omp_local.py new file mode 100644 index 000000000..90bf864bc --- /dev/null +++ b/packages/terminal-bench/agent/omp_local.py @@ -0,0 +1,484 @@ +"""Harbor agent that runs the LOCAL oh-my-pi (`omp`) build inside task containers. + +Unlike Harbor's built-in `pi` agent (which `npm i -g @mariozechner/pi-coding-agent`), +this installs the working tree at `/work/pi`: + + * the runner packs `packages/coding-agent` with `bun pm pack` (bundles every + workspace TS package into `dist/cli.js`) and hands us the tarball path, + * we upload it, install Bun, `bun install` the bundle's external deps + the + platform native addon, and run `bun .../dist/cli.js`. + +Auth never enters the container: a generated `~/.omp/agent/models.yml` routes the +configured providers' `baseUrl` at the host's pm2 auth-gateway (default +`http://host.docker.internal:4000`, `transport: pi-native`), so the gateway +resolves credentials host-side. No provider API keys are passed in. + +All knobs come from environment variables the runner sets on the `harbor` process +(see `OMP_TB_*` below); the agent reads them from `os.environ` directly. + +Selected via `harbor run --agent-import-path omp_local:OmpLocal` with the +directory of this file on `PYTHONPATH`. +""" + +from __future__ import annotations + +import asyncio +import json +import os +import shlex +from dataclasses import dataclass +from pathlib import Path +from typing import override + +from harbor.agents.installed.base import BaseInstalledAgent, with_prompt_template +from harbor.environments.base import BaseEnvironment +from harbor.models.agent.context import AgentContext + + +def _patch_harbor_cleanup_cancellation() -> None: + """Keep Harbor's Docker cleanup alive when trial cancellation interrupts it.""" + from harbor.trial.trial import Trial + + if getattr(Trial, "_omp_cleanup_cancellation_patch", False): + return + + async def _stop_agent_environment(self) -> None: + if self._is_agent_environment_stopped: + return + + stop_task = asyncio.create_task( + self.agent_environment.stop(delete=self.config.environment.delete) + ) + cancellation_logged = False + while not stop_task.done(): + try: + await asyncio.shield(stop_task) + except asyncio.CancelledError: + if not cancellation_logged: + self.logger.debug( + f"Cleanup cancellation delayed for {self.config.trial_name}; " + "waiting for agent environment stop to finish" + ) + cancellation_logged = True + except Exception: + break + try: + await stop_task + self._is_agent_environment_stopped = True + except asyncio.CancelledError: + self._is_agent_environment_stopped = True + self.logger.debug( + f"Agent environment stop was cancelled for {self.config.trial_name}" + ) + except Exception as exc: + self._is_agent_environment_stopped = True + self.logger.debug( + "Warning: Agent environment cleanup failed for " + f"{self.config.trial_name}: {exc}" + ) + self._record_exception(exc) + + Trial._stop_agent_environment = _stop_agent_environment + Trial._omp_cleanup_cancellation_patch = True + + +_patch_harbor_cleanup_cancellation() + +# Container-side staging paths (absolute; never depend on $HOME at write time). +_TARBALL_DST = "/tmp/omp-local.tgz" +_MODELS_DST = "/tmp/omp-models.yml" +_CONFIG_DST = "/tmp/omp-config.yml" +_OUTPUT_FILENAME = "omp.txt" +_ADVISOR_FILENAME = "advisor.jsonl" + +# Provider → host env vars used in --no-gateway (direct-auth) mode only. +_PROVIDER_KEYS: dict[str, list[str]] = { + "amazon-bedrock": ["AWS_ACCESS_KEY_ID", "AWS_SECRET_ACCESS_KEY", "AWS_REGION"], + "anthropic": ["ANTHROPIC_API_KEY", "ANTHROPIC_OAUTH_TOKEN"], + "github-copilot": ["GITHUB_TOKEN"], + "google": [ + "GEMINI_API_KEY", + "GOOGLE_GENERATIVE_AI_API_KEY", + "GOOGLE_API_KEY", + "GOOGLE_APPLICATION_CREDENTIALS", + "GOOGLE_CLOUD_PROJECT", + "GOOGLE_CLOUD_LOCATION", + "GOOGLE_GENAI_USE_VERTEXAI", + ], + "groq": ["GROQ_API_KEY"], + "huggingface": ["HF_TOKEN"], + "mistral": ["MISTRAL_API_KEY"], + "openai": ["OPENAI_API_KEY"], + "openrouter": ["OPENROUTER_API_KEY"], + "xai": ["XAI_API_KEY"], +} + + +def _env(name: str, default: str = "") -> str: + value = os.environ.get(name) + return value if value is not None and value != "" else default + + +def _truthy(value: str) -> bool: + return value.strip().lower() in ("1", "true", "yes", "on") + +def _loads(line: str) -> dict | None: + line = line.strip() + if not line.startswith("{"): + return None + try: + value = json.loads(line) + except (json.JSONDecodeError, ValueError): + return None + return value if isinstance(value, dict) else None + + +@dataclass +class _Usage: + """Running sum of token/cost usage across assistant turns.""" + + in_tok: int = 0 + out_tok: int = 0 + cache_read: int = 0 + cache_write: int = 0 + cost: float = 0.0 + + def add(self, usage: object) -> None: + if not isinstance(usage, dict): + return + self.in_tok += int(usage.get("input", 0) or 0) + self.out_tok += int(usage.get("output", 0) or 0) + self.cache_read += int(usage.get("cacheRead", 0) or 0) + self.cache_write += int(usage.get("cacheWrite", 0) or 0) + cost = usage.get("cost") + if isinstance(cost, dict): + self.cost += float(cost.get("total", 0.0) or 0.0) + + def empty(self) -> bool: + return self.in_tok == 0 and self.out_tok == 0 and self.cost == 0.0 + + +class OmpLocal(BaseInstalledAgent): + # No declarative CLI flags: the run command is built by hand so model/thinking + # routing stays in one place. + CLI_FLAGS = [] # type: ignore[assignment] + ENV_VARS = [] # type: ignore[assignment] + + def __init__(self, *args, **kwargs) -> None: # noqa: D401 - thin wrapper + super().__init__(*args, **kwargs) + self._install_mode = _env("OMP_TB_INSTALL", "local") + self._tarball = _env("OMP_TB_TARBALL") + self._pkg_version = _env("OMP_TB_VERSION", "latest") + self._models_yaml_path = _env("OMP_TB_MODELS_YAML") + self._gateway_url = _env("OMP_TB_GATEWAY_URL", "http://host.docker.internal:4000") + self._gateway_token = _env("OMP_TB_GATEWAY_TOKEN", "no-auth-dummy") + self._gateway_providers = [ + p.strip() + for p in _env("OMP_TB_GATEWAY_PROVIDERS", "anthropic,openai-codex").split(",") + if p.strip() + ] + self._thinking = _env("OMP_TB_THINKING") + self._auto_approve = _truthy(_env("OMP_TB_AUTO_APPROVE", "1")) + self._extra_args = _env("OMP_TB_EXTRA_ARGS") + self._bun_version = _env("OMP_TB_BUN_VERSION", "1.3.14") + self._gateway_on = _env("OMP_TB_GATEWAY", "1") != "0" + # Optional second model reviewing the primary (separate spend, summed in). + self._advisor_model = _env("OMP_TB_ADVISOR_MODEL") + self._advisor_sync = _env("OMP_TB_ADVISOR_SYNC", "1") + # web_search auth can't route through the gateway (dedicated provider creds); + # off by default so search-using tasks don't false-negative on 401s. + self._web_search = _truthy(_env("OMP_TB_WEB_SEARCH", "0")) + # Resolved during install(); reused by version + run commands. + self._home = "/root" + self._bun = "/root/.bun/bin/bun" + self._cli = "/root/.omp-bench/app/dist/cli.js" + + @staticmethod + @override + def name() -> str: + return "omp" + + @override + def version(self) -> str | None: + return self._version + + @override + def get_version_command(self) -> str | None: + return self._wrap(f"{shlex.quote(self._bun)} {shlex.quote(self._cli)} --version") + + @override + def parse_version(self, stdout: str) -> str: + return stdout.strip().splitlines()[-1].strip() if stdout.strip() else "local" + + # ------------------------------------------------------------------ install + + def _wrap(self, command: str) -> str: + """Prefix a command with the Bun runtime on PATH. + + omp spawns Bun worker subprocesses at runtime, so `bun` must resolve on + PATH during `run()` too — not just for the entrypoint. + """ + return ( + f'export BUN_INSTALL={shlex.quote(self._home + "/.bun")}; ' + f'export PATH="{self._home}/.bun/bin:$PATH"; ' + f"{command}" + ) + + @override + async def install(self, environment: BaseEnvironment) -> None: + # 1) System deps (root). curl+unzip for the Bun installer; ca-certs for TLS. + await self.exec_as_root( + environment, + command=( + "set -e; " + "if command -v apt-get >/dev/null 2>&1; then " + " apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y curl unzip ca-certificates tar; " + "elif command -v apk >/dev/null 2>&1; then " + " echo 'ERROR: Alpine/musl base image; @oh-my-pi/pi-natives ships no musl prebuilt' >&2; exit 3; " + "elif command -v dnf >/dev/null 2>&1; then dnf install -y curl unzip tar; " + "elif command -v yum >/dev/null 2>&1; then yum install -y curl unzip tar; " + "fi" + ), + ) + + # Resolve the agent user's HOME (root vs non-root tasks differ). + home = (await self.exec_as_agent(environment, command='printf %s "$HOME"')).stdout + self._home = (home or "/root").strip() or "/root" + + # 2) Bun (agent user). + await self.exec_as_agent( + environment, + command=( + "set -e; " + f"export BUN_INSTALL={shlex.quote(self._home + '/.bun')}; " + f'curl -fsSL https://bun.sh/install | bash -s "bun-v{self._bun_version}"; ' + f'{shlex.quote(self._home + "/.bun/bin/bun")} --version' + ), + ) + self._bun = f"{self._home}/.bun/bin/bun" + + if self._install_mode == "published": + self._cli = await self._install_published(environment) + else: + self._cli = await self._install_local(environment) + + # 3) Auth + model config under $HOME/.omp/agent. + if self._gateway_on: + # Gateway routing — no provider keys ever enter the container. + await self._write_models_yaml(environment) + await self._write_config(environment) + + async def _install_local(self, environment: BaseEnvironment) -> str: + if not self._tarball: + raise RuntimeError("OMP_TB_INSTALL=local requires OMP_TB_TARBALL (host tarball path)") + await environment.upload_file(self._tarball, _TARBALL_DST) + app = f"{self._home}/.omp-bench/app" + await self.exec_as_agent( + environment, + command=self._wrap( + "set -e; " + f"mkdir -p {shlex.quote(app)}; " + f"tar xzf {_TARBALL_DST} -C {shlex.quote(app)} --strip-components=1; " + f"cd {shlex.quote(app)}; " + # Bundle inlines workspace TS; only externalized deps are needed. + # Skip heavy optionals (transformers/sherpa) but add the native addon. + "bun install --production --omit=optional; " + 'arch=$(uname -m); ' + 'case "$arch" in aarch64|arm64) na=arm64 ;; x86_64|amd64) na=x64 ;; ' + '*) echo "unsupported arch $arch" >&2; exit 4 ;; esac; ' + # Native leaf MUST match the bundle version exactly (loader/API skew + # otherwise). Read it straight from the packed package.json. + 'ver=$(bun -e "process.stdout.write(require(\\"./package.json\\").version)"); ' + 'echo "pinning native @oh-my-pi/pi-natives-linux-$na@$ver"; ' + 'bun add --production "@oh-my-pi/pi-natives-linux-$na@$ver"' + ), + timeout_sec=900, + ) + return f"{app}/dist/cli.js" + + async def _install_published(self, environment: BaseEnvironment) -> str: + app = f"{self._home}/.omp-bench/app" + spec = f"@oh-my-pi/pi-coding-agent@{self._pkg_version}" + await self.exec_as_agent( + environment, + command=self._wrap( + "set -e; " + f"mkdir -p {shlex.quote(app)}; cd {shlex.quote(app)}; " + 'printf "{}" > package.json; ' + f"bun add {shlex.quote(spec)}" + ), + timeout_sec=900, + ) + return f"{app}/node_modules/@oh-my-pi/pi-coding-agent/dist/cli.js" + + async def _write_models_yaml(self, environment: BaseEnvironment) -> None: + if self._models_yaml_path and os.path.isfile(self._models_yaml_path): + await environment.upload_file(self._models_yaml_path, _MODELS_DST) + staged = _MODELS_DST + else: + content = self._generate_models_yaml() + staged = _MODELS_DST + heredoc = f"cat > {_MODELS_DST} <<'OMP_MODELS_EOF'\n{content}\nOMP_MODELS_EOF" + await self.exec_as_agent(environment, command=heredoc) + await self.exec_as_agent( + environment, + command=( + f'mkdir -p "$HOME/.omp/agent"; ' + f'cp {shlex.quote(staged)} "$HOME/.omp/agent/models.yml"' + ), + ) + + def _generate_models_yaml(self) -> str: + lines = ["# Generated by terminal-bench runner — routes auth via host gateway.", "providers:"] + for provider in self._gateway_providers: + lines += [ + f" {provider}:", + f" baseUrl: {self._gateway_url}", + " auth: oauth", + " transport: pi-native", + f" apiKey: {self._gateway_token}", + ] + return "\n".join(lines) + + async def _write_config(self, environment: BaseEnvironment) -> None: + """Write $HOME/.omp/agent/config.yml: web_search toggle + optional advisor. + + The advisor is a separate model with its own spend; its turns are written + to /__advisor.jsonl (requires a persisted session, see run()). + web_search can't authenticate through the gateway, so it's off by default. + """ + lines = [ + "# Generated by terminal-bench runner.", + "web_search:", + f" enabled: {'true' if self._web_search else 'false'}", + ] + if self._advisor_model: + lines += [ + "modelRoles:", + f" advisor: {self._advisor_model}", + "advisor:", + " enabled: true", + f' syncBacklog: "{self._advisor_sync}"', + ] + content = "\n".join(lines) + heredoc = f"cat > {_CONFIG_DST} <<'OMP_CONFIG_EOF'\n{content}\nOMP_CONFIG_EOF" + await self.exec_as_agent(environment, command=heredoc) + await self.exec_as_agent( + environment, + command=( + f'mkdir -p "$HOME/.omp/agent"; ' + f'cp {shlex.quote(_CONFIG_DST)} "$HOME/.omp/agent/config.yml"' + ), + ) + + def _collect_provider_keys(self, provider: str) -> dict[str, str]: + """Host env vars for the primary + advisor providers (direct-auth mode).""" + providers = {provider} + if self._advisor_model and "/" in self._advisor_model: + providers.add(self._advisor_model.split("/", 1)[0]) + env: dict[str, str] = {} + for prov in providers: + for key in _PROVIDER_KEYS.get(prov, []): + value = os.environ.get(key) + if value: + env[key] = value + return env + + # ---------------------------------------------------------------------- run + + @with_prompt_template + @override + async def run( + self, + instruction: str, + environment: BaseEnvironment, + context: AgentContext, + ) -> None: + if not self.model_name or "/" not in self.model_name: + raise ValueError("model must be 'provider/model' (e.g. anthropic/claude-sonnet-4-6)") + provider, model = self.model_name.split("/", 1) + + parts = [ + shlex.quote(self._bun), + shlex.quote(self._cli), + "--print", + "--mode json", + f"--provider {shlex.quote(provider)}", + f"--model {shlex.quote(model)}", + ] + # The advisor records its (separately-billed) turns to /__advisor.jsonl, + # which only exists with a persisted session — so keep sessions on for advisor runs. + if not self._advisor_model: + parts.append("--no-session") + if self._auto_approve: + parts.append("--auto-approve") + if self._thinking: + parts.append(f"--thinking {shlex.quote(self._thinking)}") + if self._extra_args: + parts.append(self._extra_args) + parts.append(shlex.quote(instruction)) + # No pipes/stdbuf (absent in minimal images): redirect raw JSONL to the + # mounted agent log dir; populate_context_post_run parses it on the host. + run = " ".join(parts) + f" > /logs/agent/{_OUTPUT_FILENAME} 2>&1" + if self._advisor_model: + # Preserve omp's exit code, then collect advisor spend into the mounted dir. + run += ( + "; rc=$?; " + f'find "$HOME/.omp/agent/sessions" -name __advisor.jsonl -exec cat {{}} + ' + f"> /logs/agent/{_ADVISOR_FILENAME} 2>/dev/null || true; exit $rc" + ) + # Direct-auth (no-gateway) mode: forward only the selected providers' keys + # via the exec env (never argv), mirroring harbor's built-in Pi.run(). + run_env: dict[str, str] | None = None + if not self._gateway_on: + run_env = self._collect_provider_keys(provider) + await self.exec_as_agent(environment, command=self._wrap(run), env=run_env) + + @override + def populate_context_post_run(self, context: AgentContext) -> None: + main = _Usage() + self._sum_main(self.logs_dir / _OUTPUT_FILENAME, main) + advisor = _Usage() + if self._advisor_model: + self._sum_advisor(self.logs_dir / _ADVISOR_FILENAME, advisor) + if main.empty() and advisor.empty(): + return + total_cost = main.cost + advisor.cost + context.n_input_tokens = main.in_tok + main.cache_read + advisor.in_tok + advisor.cache_read + context.n_output_tokens = main.out_tok + advisor.out_tok + context.n_cache_tokens = main.cache_read + advisor.cache_read + context.cost_usd = total_cost if total_cost > 0 else None + context.metadata = { + **(context.metadata or {}), + "cache_write_tokens": main.cache_write + advisor.cache_write, + "main_cost_usd": main.cost, + "advisor_cost_usd": advisor.cost, + } + + def _sum_main(self, path: Path, acc: "_Usage") -> None: + """Sum assistant `message_end` usage from omp's stdout JSONL.""" + if not path.exists(): + return + for line in path.read_text(errors="replace").splitlines(): + event = _loads(line) + if not event or event.get("type") != "message_end": + continue + message = event.get("message") + if isinstance(message, dict) and message.get("role") == "assistant": + acc.add(message.get("usage")) + + def _sum_advisor(self, path: Path, acc: "_Usage") -> None: + """Sum assistant-turn usage from concatenated __advisor.jsonl session entries.""" + if not path.exists(): + return + for line in path.read_text(errors="replace").splitlines(): + entry = _loads(line) + if not entry: + continue + # Session-tree entries are flat: {role: "assistant", usage: {...}}. + if entry.get("role") == "assistant": + acc.add(entry.get("usage")) + else: + message = entry.get("message") + if isinstance(message, dict) and message.get("role") == "assistant": + acc.add(message.get("usage")) diff --git a/packages/terminal-bench/package.json b/packages/terminal-bench/package.json new file mode 100644 index 000000000..80f2d9c4a --- /dev/null +++ b/packages/terminal-bench/package.json @@ -0,0 +1,30 @@ +{ + "type": "module", + "private": true, + "name": "@oh-my-pi/terminal-bench", + "version": "0.0.1", + "description": "Run harbor-framework/terminal-bench-2 against the local omp build with a live progress/spend/success dashboard", + "homepage": "https://omp.sh", + "author": "Can Boluk", + "license": "MIT", + "repository": { + "type": "git", + "url": "git+https://github.com/can1357/oh-my-pi.git", + "directory": "packages/terminal-bench" + }, + "bin": { + "tb2": "src/runner.ts" + }, + "scripts": { + "check": "biome check . && bun run check:types", + "check:types": "tsgo -p tsconfig.json --noEmit", + "lint": "biome lint .", + "start": "bun run src/runner.ts" + }, + "devDependencies": { + "@types/bun": "catalog:" + }, + "engines": { + "bun": ">=1.3.14" + } +} diff --git a/packages/terminal-bench/src/runner.ts b/packages/terminal-bench/src/runner.ts new file mode 100755 index 000000000..3de0a787c --- /dev/null +++ b/packages/terminal-bench/src/runner.ts @@ -0,0 +1,1140 @@ +#!/usr/bin/env bun +/** + * terminal-bench-2 runner for the local `omp` build. + * + * Orchestrates Harbor (`harbor run`) against the harbor-framework/terminal-bench-2 + * dataset using a custom agent (`agent/omp_local.py`) that installs the working + * tree at /work/pi and routes all model auth through the host pm2 auth-gateway + * (no provider keys ever enter the task containers). + * + * It owns the terminal: Harbor's own output is redirected to a log file and this + * process renders a live dashboard (progress / success% / spend / tokens / ETA) + * by polling each trial's `result.json`. On completion it writes a markdown report. + * + * bun src/runner.ts --model anthropic/claude-sonnet-4-6 --tasks 20 --concurrency 4 + * bun src/runner.ts --agent oracle --tasks 2 # cheap pipeline smoke + * bun src/runner.ts --help + */ +import { spawnSync } from "node:child_process"; +import * as fs from "node:fs"; +import * as path from "node:path"; + +// ────────────────────────────────────────────────────────────────────── config + +const REPO_ROOT = path.resolve(import.meta.dir, "..", "..", ".."); +const PKG_DIR = path.resolve(import.meta.dir, ".."); +const AGENT_DIR = path.join(PKG_DIR, "agent"); +const CODING_AGENT_DIR = path.join(REPO_ROOT, "packages", "coding-agent"); +const AGENT_IMPORT_PATH = "omp_local:OmpLocal"; + +interface Config { + models: string[]; + dataset: string; + tasks: number; + concurrency: number; + attempts: number; + include: string[]; + exclude: string[]; + thinking: string | null; + advisorModel: string | null; + advisorSync: string; + agent: string; + install: "local" | "published"; + version: string | null; + tarball: string | null; + build: boolean; + jobsDir: string; + jobName: string | null; + gatewayUrl: string; + gatewayToken: string; + providers: string[]; + gateway: boolean; + webSearch: boolean; + allowHosts: string[]; + timeoutMultiplier: number | null; + yes: boolean; + dryRun: boolean; + cleanup: boolean; + cleanupForce: boolean; + hostNetwork: boolean; + passthrough: string[]; +} + +function defaultConfig(): Config { + return { + models: [], + dataset: "terminal-bench@2.0", + tasks: 20, + concurrency: 4, + attempts: 1, + include: [], + exclude: [], + thinking: null, + advisorModel: null, + advisorSync: "1", + agent: "omp", + install: "local", + version: null, + tarball: null, + build: true, + jobsDir: path.join(REPO_ROOT, "runs", "tb2"), + jobName: null, + gatewayUrl: "http://host.docker.internal:4000", + gatewayToken: "no-auth", + providers: [], + gateway: true, + webSearch: false, + allowHosts: [], + timeoutMultiplier: null, + yes: true, + dryRun: false, + cleanup: false, + cleanupForce: false, + hostNetwork: false, + passthrough: [], + }; +} + +const HELP = `terminal-bench-2 runner (local omp) + +Usage: bun src/runner.ts [options] [-- ] + +Model / agent: + -m, --model Model (repeatable). Default anthropic/claude-sonnet-4-6 + --agent omp (default) | oracle | nop | any harbor agent + --install omp source. local = pack /work/pi (default) + --version omp version for published install (default: latest) + --thinking off|minimal|low|medium|high|xhigh + --advisor-model

Second model reviewing the primary (spend summed in) + --advisor-sync Advisor catch-up backlog (default 1 = accurate spend; off = faster) + --tarball Reuse a prebuilt omp tarball (implies --no-build) + --no-build Skip packing; reuse newest tarball in bench dir + +Dataset / scale: + -l, --tasks Max tasks (default 20) + -n, --concurrency Concurrent trials (default 4) + -k, --attempts Attempts per task (default 1) + -i, --include Include task name (repeatable) + -x, --exclude Exclude task name (repeatable) + -d, --dataset Default terminal-bench@2.0 + +Gateway (auth, no keys in container): + --gateway-url Default http://host.docker.internal:4000 + --gateway-token Default "no-auth" (gateway runs --no-auth) + --providers Providers to route (default: model provider + anthropic,openai-codex) + --no-gateway Pass host provider API keys into containers instead + --web-search Enable omp web_search (off by default; can't auth via gateway) + --allow-host harbor --allow-agent-host (repeatable) + +Output / control: + -o, --jobs-dir Default /runs/tb2 + --job-name Default tb2-- + --dry-run Print the harbor command + models.yml and exit + --cleanup Clean up stale and exited Harbor Docker resources safely before starting + --cleanup-force Force-stop and remove ALL previous Harbor Docker containers and networks + --host-network Run Docker task containers using host networking (experimental) + -h, --help This help +`; + +// ───────────────────────────────────────────────────────────────── arg parsing + +function parseArgs(argv: string[]): Config { + const cfg = defaultConfig(); + for (let i = 0; i < argv.length; i++) { + let arg = argv[i]; + if (arg === "--") { + cfg.passthrough.push(...argv.slice(i + 1)); + break; + } + let inlineValue: string | null = null; + const eq = arg.startsWith("--") ? arg.indexOf("=") : -1; + if (eq !== -1) { + inlineValue = arg.slice(eq + 1); + arg = arg.slice(0, eq); + } + const take = (flag: string): string => { + if (inlineValue !== null) return inlineValue; + const v = argv[i + 1]; + if (v === undefined) throw new Error(`missing value for ${flag}`); + i++; + return v; + }; + switch (arg) { + case "-m": + case "--model": + cfg.models.push(take(arg)); + break; + case "--agent": + cfg.agent = take(arg); + break; + case "--install": { + const v = take(arg); + if (v !== "local" && v !== "published") throw new Error("--install must be local|published"); + cfg.install = v; + break; + } + case "--version": + cfg.version = take(arg); + break; + case "--thinking": + cfg.thinking = take(arg); + break; + case "--advisor-model": + cfg.advisorModel = take(arg); + break; + case "--advisor-sync": + cfg.advisorSync = take(arg); + break; + case "--tarball": + cfg.tarball = path.resolve(take(arg)); + cfg.build = false; + break; + case "--no-build": + cfg.build = false; + break; + case "-l": + case "--tasks": + case "--n-tasks": + cfg.tasks = Number(take(arg)); + break; + case "-n": + case "--concurrency": + case "--n-concurrent": + cfg.concurrency = Number(take(arg)); + break; + case "-k": + case "--attempts": + case "--n-attempts": + cfg.attempts = Number(take(arg)); + break; + case "-i": + case "--include": + cfg.include.push(take(arg)); + break; + case "-x": + case "--exclude": + cfg.exclude.push(take(arg)); + break; + case "-d": + case "--dataset": + cfg.dataset = take(arg); + break; + case "--gateway-url": + cfg.gatewayUrl = take(arg); + break; + case "--gateway-token": + cfg.gatewayToken = take(arg); + break; + case "--providers": + cfg.providers.push( + ...take(arg) + .split(",") + .map(s => s.trim()) + .filter(Boolean), + ); + break; + case "--no-gateway": + cfg.gateway = false; + break; + case "--web-search": + cfg.webSearch = true; + break; + case "--allow-host": + cfg.allowHosts.push(take(arg)); + break; + case "-o": + case "--jobs-dir": + cfg.jobsDir = path.resolve(take(arg)); + break; + case "--job-name": + cfg.jobName = take(arg); + break; + case "--timeout-multiplier": + cfg.timeoutMultiplier = Number(take(arg)); + break; + case "--dry-run": + cfg.dryRun = true; + break; + case "--cleanup": + cfg.cleanup = true; + break; + case "--cleanup-force": + cfg.cleanupForce = true; + break; + case "--host-network": + cfg.hostNetwork = true; + break; + case "-y": + case "--yes": + cfg.yes = true; + break; + case "-h": + case "--help": + process.stdout.write(HELP); + process.exit(0); + break; + default: + throw new Error(`unknown flag: ${arg} (see --help)`); + } + } + if (cfg.models.length === 0) cfg.models = ["anthropic/claude-sonnet-4-6"]; + return cfg; +} + +// ──────────────────────────────────────────────────────────────────── helpers + +const isTTY = Boolean(process.stdout.isTTY); +const useColor = isTTY && !process.env.NO_COLOR; +const ESC = "\x1b["; +function c(code: string, s: string): string { + return useColor ? `${ESC}${code}m${s}${ESC}0m` : s; +} +const dim = (s: string): string => c("2", s); +const bold = (s: string): string => c("1", s); +const green = (s: string): string => c("32", s); +const red = (s: string): string => c("31", s); +const yellow = (s: string): string => c("33", s); +const cyan = (s: string): string => c("36", s); +const gray = (s: string): string => c("90", s); + +function fmtUsd(n: number): string { + if (n >= 100) return `$${n.toFixed(0)}`; + if (n >= 1) return `$${n.toFixed(2)}`; + return `$${n.toFixed(3)}`; +} +function fmtNum(n: number): string { + if (n >= 1e6) return `${(n / 1e6).toFixed(2)}M`; + if (n >= 1e3) return `${(n / 1e3).toFixed(1)}k`; + return `${n}`; +} +function fmtDur(ms: number): string { + if (!Number.isFinite(ms) || ms < 0) return "—"; + const s = Math.floor(ms / 1000); + const h = Math.floor(s / 3600); + const m = Math.floor((s % 3600) / 60); + const sec = s % 60; + if (h > 0) return `${h}:${String(m).padStart(2, "0")}:${String(sec).padStart(2, "0")}`; + return `${m}:${String(sec).padStart(2, "0")}`; +} +function bar(frac: number, width: number): string { + const f = Math.max(0, Math.min(1, frac)); + const filled = Math.round(f * width); + return "█".repeat(filled) + "░".repeat(width - filled); +} +function pad(s: string, w: number): string { + return s.length >= w ? s.slice(0, w) : s + " ".repeat(w - s.length); +} + +// ───────────────────────────────────────────────────────────── result parsing + +type TrialStatus = "pass" | "fail" | "error" | "running"; + +interface Trial { + name: string; + status: TrialStatus; + reward: number | null; + costUsd: number; + advisorCostUsd: number; + tokIn: number; + tokOut: number; + tokCache: number; + durationMs: number; + detail: string; +} + +interface AgentCtxLike { + n_input_tokens?: unknown; + n_cache_tokens?: unknown; + n_output_tokens?: unknown; + cost_usd?: unknown; + metadata?: unknown; +} + +function num(v: unknown): number { + return typeof v === "number" && Number.isFinite(v) ? v : 0; +} + +function resolveReward(rewards: Record | null): number | null { + if (!rewards) return null; + const vals = Object.values(rewards).filter(v => typeof v === "number"); + if (vals.length === 0) return null; + if (typeof rewards.reward === "number") return rewards.reward; + return Math.max(...vals); +} + +function readJson(file: string): unknown { + try { + return JSON.parse(fs.readFileSync(file, "utf8")); + } catch { + return null; + } +} + +/** Parse one trial directory into a Trial, or null if it isn't a trial dir yet. */ +function parseTrial(dir: string, name: string): Trial | null { + const resultPath = path.join(dir, "result.json"); + if (!fs.existsSync(resultPath)) { + // running: dir exists, no result yet. Use dir mtime as start proxy. + let started = Date.now(); + try { + started = fs.statSync(dir).mtimeMs; + } catch { + /* ignore */ + } + + // Try to parse realtime cost from the live agent omp.txt log if it exists + let costUsd = 0; + let tokIn = 0; + let tokOut = 0; + let tokCache = 0; + const ompLogPath = path.join(dir, "agent", "omp.txt"); + if (fs.existsSync(ompLogPath)) { + try { + const content = fs.readFileSync(ompLogPath, "utf8"); + for (const line of content.split("\n")) { + const trimmed = line.trim(); + if (!trimmed) continue; + try { + const event = JSON.parse(trimmed); + if (event && event.type === "message_end") { + const message = event.message; + if (message && typeof message === "object" && message.role === "assistant") { + const usage = message.usage; + if (usage && typeof usage === "object") { + tokIn += num(usage.input) + num(usage.cacheRead); + tokOut += num(usage.output); + tokCache += num(usage.cacheRead); + const cost = usage.cost; + if (cost && typeof cost === "object") { + costUsd += num(cost.total); + } + } + } + } + } catch { + /* Ignore malformed lines from incomplete writes */ + } + } + } catch { + /* ignore */ + } + } + + return { + name, + status: "running", + reward: null, + costUsd, + advisorCostUsd: 0, + tokIn, + tokOut, + tokCache, + durationMs: Date.now() - started, + detail: "", + }; + } + const raw = readJson(resultPath); + if (!raw || typeof raw !== "object") return null; + const r = raw as Record; + + // token/cost: prefer top-level agent_result, fall back to step_results[].agent_result + const ctxs: AgentCtxLike[] = []; + if (r.agent_result && typeof r.agent_result === "object") ctxs.push(r.agent_result as AgentCtxLike); + if (Array.isArray(r.step_results)) { + for (const st of r.step_results) { + if (st && typeof st === "object") { + const ar = (st as Record).agent_result; + if (ar && typeof ar === "object") ctxs.push(ar as AgentCtxLike); + } + } + } + let costUsd = 0, + advisorCostUsd = 0, + tokIn = 0, + tokOut = 0, + tokCache = 0; + for (const ctx of ctxs) { + costUsd += num(ctx.cost_usd); + tokIn += num(ctx.n_input_tokens); + tokOut += num(ctx.n_output_tokens); + tokCache += num(ctx.n_cache_tokens); + if (ctx.metadata && typeof ctx.metadata === "object") { + advisorCostUsd += num((ctx.metadata as Record).advisor_cost_usd); + } + } + + // rewards: top-level verifier_result, else step_results last verifier + let rewards: Record | null = null; + const collectRewards = (vr: unknown): void => { + if (vr && typeof vr === "object") { + const rw = (vr as Record).rewards; + if (rw && typeof rw === "object") rewards = rw as Record; + } + }; + collectRewards(r.verifier_result); + if (!rewards && Array.isArray(r.step_results)) { + for (const st of r.step_results) { + if (st && typeof st === "object") collectRewards((st as Record).verifier_result); + } + } + const reward = resolveReward(rewards); + + // exception + const exc = + r.exception_info && typeof r.exception_info === "object" ? (r.exception_info as Record) : null; + + // duration + let durationMs = 0; + const start = typeof r.started_at === "string" ? Date.parse(r.started_at) : NaN; + const end = typeof r.finished_at === "string" ? Date.parse(r.finished_at) : NaN; + if (Number.isFinite(start) && Number.isFinite(end)) durationMs = end - start; + + let status: TrialStatus; + let detail = ""; + if (exc) { + status = "error"; + detail = typeof exc.exception_type === "string" ? exc.exception_type : "error"; + } else if (reward !== null && reward >= 1 - 1e-9) { + status = "pass"; + } else { + status = "fail"; + } + return { name, status, reward, costUsd, advisorCostUsd, tokIn, tokOut, tokCache, durationMs, detail }; +} + +function readTrials(jobDir: string): Trial[] { + let entries: fs.Dirent[] = []; + try { + entries = fs.readdirSync(jobDir, { withFileTypes: true }); + } catch { + return []; + } + const trials: Trial[] = []; + for (const e of entries) { + if (!e.isDirectory()) continue; + const t = parseTrial(path.join(jobDir, e.name), e.name); + if (t) trials.push(t); + } + return trials; +} + +/** Authoritative job-level totals from /result.json (written incrementally). */ +interface JobInfo { + nTotal: number; + running: number | null; + pending: number | null; +} + +function readJobResult(jobDir: string): JobInfo | null { + const raw = readJson(path.join(jobDir, "result.json")); + if (!raw || typeof raw !== "object") return null; + const r = raw as Record; + const nTotal = typeof r.n_total_trials === "number" ? r.n_total_trials : 0; + let running: number | null = null; + let pending: number | null = null; + if (r.stats && typeof r.stats === "object") { + const s = r.stats as Record; + if (typeof s.n_running_trials === "number") running = s.n_running_trials; + if (typeof s.n_pending_trials === "number") pending = s.n_pending_trials; + } + return nTotal > 0 ? { nTotal, running, pending } : null; +} + +// ──────────────────────────────────────────────────────────────────── totals + +interface Totals { + total: number; + done: number; + pass: number; + fail: number; + error: number; + running: number; + pending: number; + costUsd: number; + advisorCostUsd: number; + tokIn: number; + tokOut: number; + tokCache: number; +} + +function aggregate(trials: Trial[], job: JobInfo | null, fallbackExpected: number): Totals { + const t: Totals = { + total: fallbackExpected, + done: 0, + pass: 0, + fail: 0, + error: 0, + running: 0, + pending: 0, + costUsd: 0, + advisorCostUsd: 0, + tokIn: 0, + tokOut: 0, + tokCache: 0, + }; + for (const tr of trials) { + t.costUsd += tr.costUsd; + t.advisorCostUsd += tr.advisorCostUsd; + t.tokIn += tr.tokIn; + t.tokOut += tr.tokOut; + t.tokCache += tr.tokCache; + if (tr.status === "running") { + t.running++; + continue; + } + t.done++; + if (tr.status === "pass") t.pass++; + else if (tr.status === "error") t.error++; + else t.fail++; + } + // Prefer harbor's authoritative job-level totals; fall back to disk scan. + t.total = job ? job.nTotal : Math.max(fallbackExpected, trials.length); + if (job && job.running !== null) t.running = job.running; + t.pending = Math.max(0, t.total - t.done - t.running); + return t; +} + +// ──────────────────────────────────────────────────────────────── dashboard IO + +const SPINNER = ["⠋", "⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇", "⠏"]; + +function statusIcon(s: TrialStatus, tick: number): string { + switch (s) { + case "pass": + return green("✓"); + case "fail": + return red("✗"); + case "error": + return yellow("!"); + case "running": + return cyan(SPINNER[tick % SPINNER.length]); + } +} + +function tailFile(file: string, maxLines: number): string[] { + try { + const buf = fs.readFileSync(file, "utf8"); + const lines = buf.split("\n").filter(l => l.trim().length > 0); + return lines.slice(-maxLines); + } catch { + return []; + } +} + +interface RenderState { + cfg: Config; + jobDir: string; + logPath: string; + startMs: number; + expected: number; + tick: number; +} + +function render(st: RenderState): void { + const trials = readTrials(st.jobDir); + const tot = aggregate(trials, readJobResult(st.jobDir), st.expected); + const elapsed = Date.now() - st.startMs; + const rate = tot.done > 0 ? elapsed / tot.done : 0; + const eta = rate > 0 && tot.done < tot.total ? rate * (tot.total - tot.done) : 0; + const successPct = tot.done > 0 ? (tot.pass / tot.done) * 100 : 0; + + const rows: string[] = []; + const advisorTag = st.cfg.advisorModel ? `${dim(" + advisor ")}${st.cfg.advisorModel}` : ""; + const header = `${bold("terminal-bench-2")} ${dim("·")} ${cyan(st.cfg.agent)} ${dim("·")} ${st.cfg.models.join(",")}${advisorTag} ${dim(`· conc=${st.cfg.concurrency} k=${st.cfg.attempts}`)}`; + rows.push(header); + const width = 28; + rows.push( + `${bar(tot.total > 0 ? tot.done / tot.total : 0, width)} ${bold(`${tot.done}/${tot.total}`)} ${dim("elapsed")} ${fmtDur(elapsed)} ${dim("eta")} ${eta > 0 ? `~${fmtDur(eta)}` : "—"}`, + ); + rows.push( + `${green(`pass ${tot.pass}`)} ${dim(`(${successPct.toFixed(0)}%)`)} ${red(`fail ${tot.fail}`)} ${yellow(`err ${tot.error}`)} ${cyan(`run ${tot.running}`)} ${gray(`pend ${tot.pending}`)}`, + ); + const advisorSpend = tot.advisorCostUsd > 0 ? dim(` (advisor ${fmtUsd(tot.advisorCostUsd)})`) : ""; + rows.push( + `${bold("spend")} ${fmtUsd(tot.costUsd)}${advisorSpend} ${dim("in")} ${fmtNum(tot.tokIn)} ${dim("out")} ${fmtNum(tot.tokOut)} ${dim("cache")} ${fmtNum(tot.tokCache)}`, + ); + rows.push(dim("─".repeat(54))); + + // table: running first, then errors/fails, then passes; recent first within + const order: Record = { running: 0, error: 1, fail: 2, pass: 3 }; + const sorted = [...trials].sort((a, b) => order[a.status] - order[b.status] || a.name.localeCompare(b.name)); + const maxRows = isTTY ? Math.max(6, (process.stdout.rows ?? 40) - rows.length - 4) : sorted.length; + for (const tr of sorted.slice(0, maxRows)) { + const rw = tr.reward !== null ? `r${tr.reward.toFixed(2)}` : tr.status === "running" ? "·" : "—"; + const right = `${pad(rw, 6)} ${pad(fmtUsd(tr.costUsd), 7)} ${pad(fmtDur(tr.durationMs), 7)}`; + const detail = tr.detail ? ` ${yellow(tr.detail)}` : ""; + rows.push(` ${statusIcon(tr.status, st.tick)} ${pad(tr.name, 28)} ${dim(right)}${detail}`); + } + if (sorted.length > maxRows) rows.push(dim(` … ${sorted.length - maxRows} more`)); + rows.push(dim("─".repeat(54))); + const lastLog = tailFile(st.logPath, 1)[0] ?? ""; + rows.push(gray(`harbor: ${lastLog.slice(0, 70)}`)); + + if (isTTY) { + // home + clear to end of screen, then write frame + let out = `${ESC}H${ESC}J`; + out += rows.join(`${ESC}K\n`); + process.stdout.write(out); + } else { + process.stdout.write( + `[tb2] ${tot.done}/${tot.total} pass=${tot.pass}(${successPct.toFixed(0)}%) fail=${tot.fail} err=${tot.error} run=${tot.running} spend=${fmtUsd(tot.costUsd)} elapsed=${fmtDur(elapsed)}\n`, + ); + } +} + +// ────────────────────────────────────────────────────────────────────── report + +function writeReport(st: RenderState, benchDir: string, exitCode: number): string { + const trials = readTrials(st.jobDir).sort((a, b) => a.name.localeCompare(b.name)); + const tot = aggregate(trials, readJobResult(st.jobDir), st.expected); + const successPct = tot.done > 0 ? (tot.pass / tot.done) * 100 : 0; + const lines: string[] = []; + const isOmp = st.cfg.agent === "omp"; + const modelLine = + isOmp && st.cfg.advisorModel + ? `${st.cfg.models.join(", ")} + advisor ${st.cfg.advisorModel}` + : st.cfg.models.join(", "); + lines.push(`# terminal-bench-2 — ${st.cfg.agent} — ${modelLine}`); + lines.push(""); + lines.push(`- dataset: \`${st.cfg.dataset}\``); + lines.push(`- tasks: ${st.cfg.tasks} · attempts: ${st.cfg.attempts} · concurrency: ${st.cfg.concurrency}`); + if (isOmp) { + lines.push( + `- install: ${st.cfg.install} · auth: ${st.cfg.gateway ? "host gateway (no keys in container)" : "direct provider keys"}`, + ); + lines.push(`- tools: web_search=${st.cfg.webSearch ? "on" : "off"}`); + if (st.cfg.advisorModel) lines.push(`- advisor: ${st.cfg.advisorModel}`); + } + lines.push(`- elapsed: ${fmtDur(Date.now() - st.startMs)} · harbor exit: ${exitCode}`); + lines.push(""); + const advisorSpend = tot.advisorCostUsd > 0 ? ` (advisor ${fmtUsd(tot.advisorCostUsd)})` : ""; + lines.push( + `**${tot.pass}/${tot.done} passed (${successPct.toFixed(1)}%)** · fail ${tot.fail} · error ${tot.error} · spend ${fmtUsd(tot.costUsd)}${advisorSpend}`, + ); + lines.push(`tokens: in ${fmtNum(tot.tokIn)} · out ${fmtNum(tot.tokOut)} · cache ${fmtNum(tot.tokCache)}`); + lines.push(""); + lines.push("| task | result | reward | cost | duration | detail |"); + lines.push("|---|---|---|---|---|---|"); + for (const t of trials) { + const res = + t.status === "pass" + ? "✅ pass" + : t.status === "fail" + ? "❌ fail" + : t.status === "error" + ? "⚠️ error" + : "⏳ running"; + lines.push( + `| ${t.name} | ${res} | ${t.reward !== null ? t.reward.toFixed(2) : "—"} | ${fmtUsd(t.costUsd)} | ${fmtDur(t.durationMs)} | ${t.detail} |`, + ); + } + lines.push(""); + const reportPath = path.join(benchDir, "report.md"); + fs.writeFileSync(reportPath, lines.join("\n")); + return reportPath; +} + +// ──────────────────────────────────────────────────────────────────── setup + +function which(bin: string): string | null { + const r = spawnSync("bash", ["-lc", `command -v ${bin}`], { encoding: "utf8" }); + const out = r.stdout?.trim(); + return r.status === 0 && out ? out : null; +} + +function readPkgVersion(): string { + const raw = readJson(path.join(CODING_AGENT_DIR, "package.json")); + if (raw && typeof raw === "object") { + const v = (raw as Record).version; + if (typeof v === "string") return v; + } + return "latest"; +} + +function buildTarball(benchDir: string): string { + process.stdout.write(dim("packing local omp (bun pm pack)…\n")); + const r = spawnSync("bun", ["pm", "pack", "--destination", benchDir], { + cwd: CODING_AGENT_DIR, + encoding: "utf8", + stdio: ["ignore", "pipe", "pipe"], + }); + if (r.status !== 0) { + process.stderr.write((r.stdout ?? "") + (r.stderr ?? "")); + throw new Error("bun pm pack failed"); + } + const tgz = fs + .readdirSync(benchDir) + .filter(f => f.endsWith(".tgz")) + .map(f => ({ f, m: fs.statSync(path.join(benchDir, f)).mtimeMs })) + .sort((a, b) => b.m - a.m)[0]; + if (!tgz) throw new Error("no .tgz produced by bun pm pack"); + return path.join(benchDir, tgz.f); +} + +function newestTarball(benchDir: string): string | null { + try { + const tgz = fs + .readdirSync(benchDir) + .filter(f => f.endsWith(".tgz")) + .map(f => ({ f, m: fs.statSync(path.join(benchDir, f)).mtimeMs })) + .sort((a, b) => b.m - a.m)[0]; + return tgz ? path.join(benchDir, tgz.f) : null; + } catch { + return null; + } +} + +function deriveProviders(cfg: Config): string[] { + const set = new Set(cfg.providers); + for (const m of cfg.models) { + const slash = m.indexOf("/"); + if (slash > 0) set.add(m.slice(0, slash)); + } + if (cfg.advisorModel) { + const slash = cfg.advisorModel.indexOf("/"); + if (slash > 0) set.add(cfg.advisorModel.slice(0, slash)); + } + if (set.size === 0) { + set.add("anthropic"); + set.add("openai-codex"); + } + return [...set]; +} + +function writeModelsYaml(benchDir: string, cfg: Config): string { + const providers = deriveProviders(cfg); + const lines = ["# Generated by terminal-bench runner — auth via host pm2 gateway.", "providers:"]; + for (const p of providers) { + lines.push(` ${p}:`); + lines.push(` baseUrl: ${cfg.gatewayUrl}`); + lines.push(" auth: oauth"); + lines.push(" transport: pi-native"); + lines.push(` apiKey: ${cfg.gatewayToken}`); + } + const file = path.join(benchDir, "models.yml"); + fs.writeFileSync(file, `${lines.join("\n")}\n`); + return file; +} + +function gatewayHealthOk(url: string): boolean { + const hostUrl = url.replace("host.docker.internal", "127.0.0.1").replace(/\/+$/, ""); + const r = spawnSync("curl", ["-s", "--max-time", "4", `${hostUrl}/healthz`], { encoding: "utf8" }); + return r.status === 0 && (r.stdout ?? "").includes('"ok":true'); +} + +function buildHarborArgs( + cfg: Config, + jobName: string, + modelsYaml: string, + tarball: string | null, + hostNetworkOverlayPath: string | null, +): string[] { + const a: string[] = ["run", "-d", cfg.dataset, "-o", cfg.jobsDir, "--job-name", jobName]; + a.push("-n", String(cfg.concurrency), "-k", String(cfg.attempts), "-l", String(cfg.tasks)); + for (const m of cfg.models) a.push("-m", m); + for (const inc of cfg.include) a.push("-i", inc); + for (const exc of cfg.exclude) a.push("-x", exc); + for (const h of cfg.allowHosts) a.push("--allow-agent-host", h); + if (cfg.timeoutMultiplier !== null) a.push("--timeout-multiplier", String(cfg.timeoutMultiplier)); + if (cfg.yes) a.push("-y"); + if (hostNetworkOverlayPath) { + a.push("--extra-docker-compose", hostNetworkOverlayPath); + } + + if (cfg.agent === "omp") { + // Config + secrets travel via env (OMP_TB_*); the agent reads os.environ. + a.push("--agent-import-path", AGENT_IMPORT_PATH); + void modelsYaml; + void tarball; + } else { + a.push("-a", cfg.agent); + } + a.push(...cfg.passthrough); + return a; +} + +function buildHarborEnv( + cfg: Config, + modelsYaml: string, + tarball: string | null, + version: string, +): Record { + const env: Record = { ...(process.env as Record) }; + if (cfg.agent !== "omp") return env; + const prepend = (k: string, v: string): void => { + env[k] = env[k] ? `${v}:${env[k]}` : v; + }; + prepend("PYTHONPATH", AGENT_DIR); + env.OMP_TB_INSTALL = cfg.install; + env.OMP_TB_VERSION = cfg.version ?? version; + if (tarball) env.OMP_TB_TARBALL = tarball; + if (cfg.thinking) env.OMP_TB_THINKING = cfg.thinking; + if (cfg.advisorModel) { + env.OMP_TB_ADVISOR_MODEL = cfg.advisorModel; + env.OMP_TB_ADVISOR_SYNC = cfg.advisorSync; + } + if (cfg.webSearch) env.OMP_TB_WEB_SEARCH = "1"; + env.OMP_TB_GATEWAY = cfg.gateway ? "1" : "0"; + if (cfg.gateway) { + env.OMP_TB_MODELS_YAML = modelsYaml; + env.OMP_TB_GATEWAY_URL = cfg.gatewayUrl; + env.OMP_TB_GATEWAY_TOKEN = cfg.gatewayToken; + env.OMP_TB_GATEWAY_PROVIDERS = deriveProviders(cfg).join(","); + } + return env; +} + +// ──────────────────────────────────────────────────────────────────────── main + +async function main(): Promise { + const cfg = parseArgs(process.argv.slice(2)); + + if (!which("harbor")) { + throw new Error("harbor not found on PATH. Install with: uv tool install harbor"); + } + if (cfg.agent === "omp" && !which("docker")) { + throw new Error("docker not found on PATH (required to run task containers)."); + } + + const stamp = new Date().toISOString().replace(/[:.]/g, "-").slice(0, 19); + const modelSlug = cfg.models[0].replace(/[^a-zA-Z0-9]+/g, "-"); + const jobName = cfg.jobName ?? `tb2-${modelSlug}-${stamp}`; + const jobDir = path.join(cfg.jobsDir, jobName); + const benchDir = path.join(cfg.jobsDir, "_bench", jobName); + fs.mkdirSync(benchDir, { recursive: true }); + + const version = readPkgVersion(); + + // tarball (local install only) + let tarball: string | null = cfg.tarball; + if (cfg.agent === "omp" && cfg.install === "local") { + if (tarball) { + process.stdout.write(dim(`using tarball ${tarball}\n`)); + } else if (cfg.build) { + tarball = buildTarball(path.join(cfg.jobsDir, "_bench")); + } else { + tarball = newestTarball(path.join(cfg.jobsDir, "_bench")); + if (!tarball) throw new Error("--no-build but no tarball found; pass --tarball or drop --no-build"); + } + } + + // models.yml (gateway) + let modelsYaml = ""; + if (cfg.agent === "omp" && cfg.gateway) { + modelsYaml = writeModelsYaml(benchDir, cfg); + if (!gatewayHealthOk(cfg.gatewayUrl)) { + process.stderr.write( + yellow( + `warning: gateway ${cfg.gatewayUrl} health check failed (continuing). Is the pm2 'omp-auth-gateway' running?\n`, + ), + ); + } + } + let hostNetworkOverlayPath: string | null = null; + if (cfg.hostNetwork) { + hostNetworkOverlayPath = path.join(benchDir, "host-network-overlay.yaml"); + const content = `services: + main: + network_mode: "host" +`; + fs.writeFileSync(hostNetworkOverlayPath, content); + } + + const harborArgs = buildHarborArgs(cfg, jobName, modelsYaml, tarball, hostNetworkOverlayPath); + const harborEnv = buildHarborEnv(cfg, modelsYaml, tarball, version); + const logPath = path.join(benchDir, "harbor.log"); + if (cfg.dryRun) { + process.stdout.write(bold("\nharbor command:\n")); + process.stdout.write(`harbor ${harborArgs.join(" ")}\n\n`); + if (modelsYaml) { + process.stdout.write(bold("models.yml:\n")); + process.stdout.write(`${fs.readFileSync(modelsYaml, "utf8")}\n`); + } + process.stdout.write(bold("omp env:\n")); + for (const k in harborEnv) { + if (k.startsWith("OMP_TB_") || k === "PYTHONPATH") process.stdout.write(` ${k}=${harborEnv[k]}\n`); + } + process.stdout.write(`\njob dir: ${jobDir}\nbench dir: ${benchDir}\n`); + return; + } + + // Handle targeted cleanup if requested via --cleanup or --cleanup-force + if ((cfg.cleanup || cfg.cleanupForce) && which("docker")) { + try { + process.stdout.write(dim("Running safe harbor-targeted Docker cleanup...\n")); + + // 1. Identify container task instances whose working_dir is within '.cache/harbor/tasks' + const containerArgs = ["ps", "-a"]; + if (!cfg.cleanupForce) { + containerArgs.push("--filter", "status=exited"); + containerArgs.push("--filter", "status=created"); + containerArgs.push("--filter", "status=dead"); + } + containerArgs.push( + "--format", + '{{.ID}}\t{{.Label "com.docker.compose.project.working_dir"}}\t{{.Label "com.docker.compose.project"}}', + ); + + const inspectCommand = spawnSync("docker", containerArgs, { encoding: "utf8" }); + + if (inspectCommand.status === 0 && inspectCommand.stdout) { + const lines = inspectCommand.stdout.trim().split("\n"); + const containerIdsToRemove: string[] = []; + + for (const line of lines) { + const parts = line.split("\t"); + if (parts.length >= 3) { + const [id, workingDir] = parts; + if (workingDir?.includes(".cache/harbor/tasks")) { + containerIdsToRemove.push(id); + } + } + } + + if (containerIdsToRemove.length > 0) { + if (cfg.cleanupForce) { + process.stdout.write( + dim(`Stopping and force-removing ${containerIdsToRemove.length} stale task containers...\n`), + ); + spawnSync("docker", ["rm", "-f", ...containerIdsToRemove]); + } else { + process.stdout.write( + dim(`Removing ${containerIdsToRemove.length} exited/stale task containers...\n`), + ); + spawnSync("docker", ["rm", ...containerIdsToRemove]); + } + } + } + + // Check which projects still have running containers before deleting companion or orphan networks. + const activeInspect = spawnSync( + "docker", + ["ps", "-a", "--filter", "status=running", "--format", '{{.Label "com.docker.compose.project"}}'], + { encoding: "utf8" }, + ); + + const activeProjects = new Set(); + if (activeInspect.status === 0 && activeInspect.stdout) { + for (const line of activeInspect.stdout.trim().split("\n")) { + if (line.trim()) { + activeProjects.add(line.trim()); + } + } + } + + // 2. Identify corresponding compose networks with matching labels to clean up. + // Specifically, find unattached Compose networks whose name matches typical Harbor trial patterns + // and which currently have no running containers. This will safely clear orphan networks leftover from prior crashed runs. + const netInspect = spawnSync("docker", ["network", "ls", "--format", "{{.ID}}\t{{.Labels}}"], { + encoding: "utf8", + }); + + if (netInspect.status === 0 && netInspect.stdout) { + const netLines = netInspect.stdout.trim().split("\n"); + const netIdsToRemove: string[] = []; + for (const netLine of netLines) { + const parts = netLine.split("\t"); + if (parts.length >= 2) { + const [netId, labels] = parts; + // extract com.docker.compose.project label + const projMatch = labels.match(/com\.docker\.compose\.project=([^,]+)/); + if (projMatch) { + const project = projMatch[1]; + const isHarborProject = /^[a-z0-9_.-]+__[a-zA-Z0-9]{7}$/.test(project); + if (isHarborProject && !activeProjects.has(project)) { + netIdsToRemove.push(netId); + } + } + } + } + + if (netIdsToRemove.length > 0) { + process.stdout.write(dim(`Removing ${netIdsToRemove.length} stale trial Docker networks...\n`)); + for (const netId of netIdsToRemove) { + spawnSync("docker", ["network", "rm", netId]); + } + } + } + process.stdout.write("Docker cleanup completed successfully.\n"); + } catch (err: unknown) { + process.stdout.write( + `\nwarning: failed to run docker cleanup: ${err instanceof Error ? err.message : String(err)}\n`, + ); + } + } + + process.stdout.write(dim(`launching harbor → ${logPath}\n`)); + const logFd = fs.openSync(logPath, "a"); + const proc = Bun.spawn(["harbor", ...harborArgs], { + env: harborEnv, + stdout: logFd, + stderr: logFd, + stdin: "ignore", + }); + + const expected = Math.max(1, cfg.tasks * cfg.attempts * cfg.models.length); + const st: RenderState = { cfg, jobDir, logPath, startMs: Date.now(), expected, tick: 0 }; + + if (isTTY) process.stdout.write(`${ESC}?1049h${ESC}?25l`); // alt screen, hide cursor + let exitCode = 0; + let finished = false; + proc.exited.then((code: number) => { + exitCode = code; + finished = true; + }); + + const onSig = (): void => { + try { + proc.kill("SIGINT"); + } catch { + /* ignore */ + } + }; + process.on("SIGINT", onSig); + process.on("SIGTERM", onSig); + + try { + while (!finished) { + render(st); + st.tick++; + await Bun.sleep(isTTY ? 700 : 10000); + } + render(st); // final frame + } finally { + if (isTTY) process.stdout.write(`${ESC}?25h${ESC}?1049l`); // restore cursor + screen + try { + fs.closeSync(logFd); + } catch { + /* ignore */ + } + process.off("SIGINT", onSig); + process.off("SIGTERM", onSig); + } + + // final summary (printed to the normal screen) + const trials = readTrials(jobDir); + const tot = aggregate(trials, readJobResult(jobDir), expected); + const successPct = tot.done > 0 ? (tot.pass / tot.done) * 100 : 0; + const reportPath = writeReport(st, benchDir, exitCode); + process.stdout.write("\n"); + process.stdout.write( + `${bold("terminal-bench-2 complete")} — ${green(`${tot.pass}/${tot.done} passed (${successPct.toFixed(1)}%)`)}\n`, + ); + process.stdout.write( + `fail ${tot.fail} · error ${tot.error} · spend ${fmtUsd(tot.costUsd)} · elapsed ${fmtDur(Date.now() - st.startMs)}\n`, + ); + process.stdout.write( + `tokens: in ${fmtNum(tot.tokIn)} · out ${fmtNum(tot.tokOut)} · cache ${fmtNum(tot.tokCache)}\n`, + ); + process.stdout.write(`${dim("report:")} ${reportPath}\n`); + process.stdout.write(`${dim("logs: ")} ${logPath}\n`); + process.stdout.write(`${dim("trials:")} ${jobDir}\n`); + if (exitCode !== 0) process.stdout.write(yellow(`harbor exited ${exitCode}; see harbor.log\n`)); + process.exit(exitCode); +} + +main().catch((err: unknown) => { + if (isTTY) process.stdout.write(`${ESC}?25h${ESC}?1049l`); + process.stderr.write(red(`\nerror: ${err instanceof Error ? err.message : String(err)}\n`)); + process.exit(1); +}); diff --git a/packages/terminal-bench/tsconfig.json b/packages/terminal-bench/tsconfig.json new file mode 100644 index 000000000..d9a6e62ad --- /dev/null +++ b/packages/terminal-bench/tsconfig.json @@ -0,0 +1,6 @@ +{ + "extends": "../tsconfig.workspace.json", + "include": [ + "src" + ] +}